From b481724eec764091ada89241458e2cf0b57c60a9 Mon Sep 17 00:00:00 2001 From: ryerraguntla Date: Tue, 25 Aug 2026 05:12:03 +0530 Subject: [PATCH 001/182] Templated code initial version --- Cargo.lock | 36 ++ Cargo.toml | 2 + core/connectors/BLOG_POST.md | 104 ++++ .../connectors/sink_template.toml | 47 ++ .../connectors/source_template.toml | 46 ++ core/connectors/sinks/README.md | 1 + .../connectors/sinks/sink_template/Cargo.toml | 58 ++ core/connectors/sinks/sink_template/README.md | 70 +++ .../sinks/sink_template/config.toml | 48 ++ .../connectors/sinks/sink_template/src/lib.rs | 504 +++++++++++++++ core/connectors/sources/README.md | 1 + .../sources/source_template/Cargo.toml | 60 ++ .../sources/source_template/README.md | 64 ++ .../sources/source_template/config.toml | 47 ++ .../sources/source_template/src/lib.rs | 586 ++++++++++++++++++ 15 files changed, 1674 insertions(+) create mode 100644 core/connectors/BLOG_POST.md create mode 100644 core/connectors/runtime/example_config/connectors/sink_template.toml create mode 100644 core/connectors/runtime/example_config/connectors/source_template.toml create mode 100644 core/connectors/sinks/sink_template/Cargo.toml create mode 100644 core/connectors/sinks/sink_template/README.md create mode 100644 core/connectors/sinks/sink_template/config.toml create mode 100644 core/connectors/sinks/sink_template/src/lib.rs create mode 100644 core/connectors/sources/source_template/Cargo.toml create mode 100644 core/connectors/sources/source_template/README.md create mode 100644 core/connectors/sources/source_template/config.toml create mode 100644 core/connectors/sources/source_template/src/lib.rs diff --git a/Cargo.lock b/Cargo.lock index bc9698629a..fa3a3efb13 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -7226,6 +7226,42 @@ dependencies = [ "tracing", ] +[[package]] +name = "iggy_connector_template_sink" +version = "0.1.0" +dependencies = [ + "async-trait", + "dashmap", + "iggy_common", + "iggy_connector_sdk", + "reqwest 0.13.4", + "reqwest-middleware", + "secrecy", + "serde", + "serde_json", + "tokio", + "tracing", +] + +[[package]] +name = "iggy_connector_template_source" +version = "0.1.0" +dependencies = [ + "async-trait", + "dashmap", + "humantime", + "iggy_common", + "iggy_connector_sdk", + "reqwest 0.13.4", + "reqwest-middleware", + "rmp-serde", + "secrecy", + "serde", + "serde_json", + "tokio", + "tracing", +] + [[package]] name = "iggy_examples" version = "0.0.6" diff --git a/Cargo.toml b/Cargo.toml index 073e46a65d..7ef6257a2c 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -45,12 +45,14 @@ members = [ "core/connectors/sinks/postgres_sink", "core/connectors/sinks/quickwit_sink", "core/connectors/sinks/s3_sink", + "core/connectors/sinks/sink_template", "core/connectors/sinks/stdout_sink", "core/connectors/sinks/surrealdb_sink", "core/connectors/sources/elasticsearch_source", "core/connectors/sources/influxdb_source", "core/connectors/sources/postgres_source", "core/connectors/sources/random_source", + "core/connectors/sources/source_template", "core/consensus", "core/cpu_allocation", "core/harness_derive", diff --git a/core/connectors/BLOG_POST.md b/core/connectors/BLOG_POST.md new file mode 100644 index 0000000000..fe8b45ea1c --- /dev/null +++ b/core/connectors/BLOG_POST.md @@ -0,0 +1,104 @@ +# Announcing sink and source connector templates + +*Draft for the Apache Iggy project blog. Replace this header line with +the final publish date and author byline before posting.* + +Apache Iggy's connectors subsystem has grown fast. In the past few +months alone, contributors have shipped or proposed sink and source +connectors for Postgres, MongoDB, Elasticsearch, Iceberg, Delta Lake, +S3, InfluxDB, Doris, ClickHouse, SurrealDB, Meilisearch, OpenSearch, +Redshift, and more — each one a plugin that moves real data between +Apache Iggy and an external system, often in production. That growth +is great news for the project. It also means new contributors keep +re-solving the same non-backend-specific problems from scratch before +their PR can even get to the interesting part: talking to their actual +system. + +## What we found + +Looking back across recent connector PR reviews, the same handful of +issues came up again and again, and none of them had anything to do +with the destination or source system being integrated: + +- **Credentials typed as plain `String`.** Connection strings and API + keys landing in `Debug`/log output because the field wasn't + `secrecy::SecretString`. +- **Cursor commits that don't survive a failed delivery.** Since + [#3855](https://github.com/apache/iggy/pull/3855), sources use a + formal ACK/NACK handshake — `poll()` stages candidate state, and + only `on_batch_result()` commits it — but that shape has to be + learned and wired up correctly every time. +- **Errors that don't distinguish retry-worthy from permanent.** + Network hiccups and "this payload will never be accepted" ending up + in the same catch-all error variant. +- **Config knobs with drifting names.** `retry_max_delay` here, + `max_retry_delay` there, `request_timeout` somewhere else, for the + same concept. +- **Missing canonical tests.** State restore/round-trip and ACK/NACK + commit/discard behavior left untested because nobody had a reference + test suite to copy. + +None of this is specific to any one backend. It's framework plumbing +that every sink and every source needs, and until now every author +either copied the closest existing plugin and stripped it down, or +started from a blank `lib.rs` and rediscovered each of these the hard +way, one review round at a time. + +## The templates + +[`core/connectors/sinks/sink_template`](sinks/sink_template) and +[`core/connectors/sources/source_template`](sources/source_template) +are compiling, tested crates you copy and fill in — not prose +describing a pattern, but the pattern itself, already wired up and +passing `cargo test`: + +- Config parsing with `#[serde(deny_unknown_fields)]`, so a typo'd TOML + key fails loudly instead of silently doing nothing. +- `SecretString` on every credential-shaped field + (`connection_string`, `auth_token`), via + `iggy_common::serde_secret::serialize_secret`. +- A retry-wrapped HTTP client plus a startup connectivity probe with + its own backoff. +- A `CircuitBreaker` that's actually consulted before each call and + updated once per `consume()`/`poll()`, not per chunk. +- Sink: batching by a configurable size, and a `last_err` pattern that + never swallows a failed batch into `Ok(())`. +- Source: the full #3855 ACK/NACK contract — `poll()` stages a + candidate cursor, `on_batch_result()` commits it on `Ack` or + discards it on `Nack`, so a dropped batch gets re-polled instead of + silently lost. +- The canonical test suites: six tests for the sink, eight for the + source (four state tests — restore, no-state, invalid-state, + round-trip — plus the two ACK/NACK tests, plus config validation and + the circuit-breaker short-circuit path). + +What's left is marked `TODO(Developer)` in each crate's `src/lib.rs`: +one spot for a sink (`push_batch()`), two for a source +(`build_raw_client()` if you're not talking HTTP, and +`fetch_records()`). Everything else — the parts that used to eat a +review round — is already done. + +## Using one + +Copy the crate, rename the package and the directory, add it to the +workspace `members` list, fill in the `TODO(Developer)` spots, and +update `config.toml` for your system. Each crate's own `README.md` +walks through the exact steps. Both templates already build, `clippy +--all-targets -- -D warnings` clean, and pass their tests as committed +— the only thing that should break when you fill in the TODOs is the +`Err(Error::InitError("not implemented yet"))` stub they start from. + +## Why this matters beyond Apache Iggy's connectors + +The pattern generalizes past this one subsystem: any plugin system +with a real framework contract — secrets, retries, staged +commit/rollback, canonical tests — benefits more from a working, +compiling example than from a checklist alone. A checklist tells you +what to verify; a template gives you the thing already verified, so +your own diff is just the part only you can write. + +--- + +*Feedback and discussion: see the project's +[GitHub Discussions](https://github.com/apache/iggy/discussions) or +[Discord](https://discord.gg/apache-iggy).* diff --git a/core/connectors/runtime/example_config/connectors/sink_template.toml b/core/connectors/runtime/example_config/connectors/sink_template.toml new file mode 100644 index 0000000000..90c3937f95 --- /dev/null +++ b/core/connectors/runtime/example_config/connectors/sink_template.toml @@ -0,0 +1,47 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +type = "sink" +key = "template" +enabled = true +version = 0 +name = "Template sink" +path = "target/release/libiggy_connector_template_sink" +plugin_config_format = "toml" +verbose = false + +[[streams]] +stream = "example_stream" +topics = ["example_topic"] +schema = "json" +batch_length = 100 +poll_interval = "5ms" +consumer_group = "template_sink_connector" + +[plugin_config] +connection_string = "https://api.example.com" +target = "events" +health_check_path = "/health" +batch_size = 100 +timeout = "30s" +max_retries = 3 +retry_delay = "500ms" +retry_max_delay = "5s" +max_open_retries = 10 +open_retry_max_delay = "60s" +circuit_breaker_threshold = 5 +circuit_breaker_cool_down = "30s" diff --git a/core/connectors/runtime/example_config/connectors/source_template.toml b/core/connectors/runtime/example_config/connectors/source_template.toml new file mode 100644 index 0000000000..c53439a0ae --- /dev/null +++ b/core/connectors/runtime/example_config/connectors/source_template.toml @@ -0,0 +1,46 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +type = "source" +key = "template" +enabled = true +version = 0 +name = "Template source" +path = "target/release/libiggy_connector_template_source" +plugin_config_format = "toml" +verbose = false + +[[streams]] +stream = "example_stream" +topic = "example_topic" +schema = "json" +batch_length = 100 +linger_time = "5ms" + +[plugin_config] +connection_string = "https://api.example.com" +health_check_path = "/health" +batch_size = 100 +poll_interval = "1s" +timeout = "30s" +max_retries = 3 +retry_delay = "500ms" +retry_max_delay = "5s" +max_open_retries = 10 +open_retry_max_delay = "60s" +circuit_breaker_threshold = 5 +circuit_breaker_cool_down = "30s" diff --git a/core/connectors/sinks/README.md b/core/connectors/sinks/README.md index e23e1ace9c..990cbe59f8 100644 --- a/core/connectors/sinks/README.md +++ b/core/connectors/sinks/README.md @@ -16,6 +16,7 @@ Sink connectors are responsible for writing data from Iggy streams to external s | **postgres_sink** | Stores messages in PostgreSQL database tables with configurable schemas | | **quickwit_sink** | Indexes messages in Quickwit search engine for log analytics | | **s3_sink** | Writes messages to Amazon S3 and S3-compatible stores (MinIO, R2, B2, DO Spaces) | +| **sink_template** | Fill-in-the-blank starting point for a new sink; framework/security plumbing done, one `TODO(Developer)` spot left | | **stdout_sink** | Prints messages to standard output (useful for debugging and development) | | **surrealdb_sink** | Writes messages into SurrealDB with deterministic record IDs for idempotent replay | diff --git a/core/connectors/sinks/sink_template/Cargo.toml b/core/connectors/sinks/sink_template/Cargo.toml new file mode 100644 index 0000000000..8ab161bd1e --- /dev/null +++ b/core/connectors/sinks/sink_template/Cargo.toml @@ -0,0 +1,58 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. +# +# TEMPLATE — rename the package (and this directory) to +# `iggy_connector__sink` before publishing, and update the +# `[[sinks]]` entry you add to the workspace root Cargo.toml accordingly. + +[package] +name = "iggy_connector_template_sink" +version = "0.1.0" +description = "Template for an Apache Iggy sink connector — copy this crate and fill in the TODO sections." +edition = "2024" +license = "Apache-2.0" +keywords = ["iggy", "messaging", "streaming"] +categories = ["command-line-utilities", "database", "network-programming"] +homepage = "https://iggy.apache.org" +documentation = "https://iggy.apache.org/docs" +repository = "https://github.com/apache/iggy" +publish = false + +[package.metadata.cargo-machete] +# dashmap is used only inside the `sink_connector!` macro expansion, so a +# naive unused-dependency scan won't see the usage — keep it ignored rather +# than removing it, or the plugin will fail to compile. +ignored = ["dashmap"] + +[lib] +crate-type = ["cdylib", "lib"] + +[dependencies] +async-trait = { workspace = true } +dashmap = { workspace = true } +iggy_common = { workspace = true } +iggy_connector_sdk = { workspace = true } +reqwest = { workspace = true } +reqwest-middleware = { workspace = true } +secrecy = { workspace = true } +serde = { workspace = true } +serde_json = { workspace = true } +tokio = { workspace = true } +tracing = { workspace = true } + +[dev-dependencies] +tokio = { workspace = true, features = ["test-util"] } diff --git a/core/connectors/sinks/sink_template/README.md b/core/connectors/sinks/sink_template/README.md new file mode 100644 index 0000000000..16547f5d14 --- /dev/null +++ b/core/connectors/sinks/sink_template/README.md @@ -0,0 +1,70 @@ +# Template sink connector + +Starting point for a new Apache Iggy **sink** connector. Everything except +pushing data to your actual destination is already implemented and follows +the project's required resilience/security patterns — see the module-level +doc comment at the top of `src/lib.rs` for the full rationale, and the +`iggy-connector-review` skill / the "Building Connectors That Pass Review" +blog post for the checklist this template is built against. + +## What's already done for you + +- Config parsing with `#[serde(deny_unknown_fields)]` so a typo in a TOML + file fails loudly instead of silently doing nothing. +- Config validation in `open()` (not `new()`, which has no way to return an + error) — including validating `target` (the destination table/index/ + collection name) against an allowlist pattern *before* it can ever reach a + query, path, or URL. +- `connection_string` and the optional `auth_token` field both typed as + `SecretString`, since either can carry credentials. +- A retry-wrapped HTTP client (`iggy_connector_sdk::retry::build_retry_client`) + and a startup connectivity probe with its own backoff + (`check_connectivity_with_retry`). +- A `CircuitBreaker` that's actually consulted before each `consume()` call + and updated once per call based on the outcome — not just constructed and + forgotten, and not reset mid-batch by a partial success. +- Batching: `consume()` chunks the incoming messages by a configurable + `batch_size` instead of sending everything in one unbounded request. +- The `sink_connector!` FFI macro invocation and a `Cargo.toml` with the + right `crate-type`, workspace-pinned dependencies, and license header. +- Tests for config/identifier validation and the circuit-breaker short-circuit + path. + +## What you need to fill in + +Search for `TODO(Developer)` in `src/lib.rs` — there is exactly one spot: + +**`push_batch()`** — build the request/write that actually sends one chunk +of messages to your destination, using `self.config.connection_string` (and +`self.config.target`, already validated by the time this runs) via +`self.client` (already retry-wrapped). Distinguish permanent failures (bad +schema, a destination that will reject this payload shape no matter how many +times you retry) from transient ones (network error, 5xx, timeout) by +returning `Error::PermanentHttpError` for the former — see the doc comment on +that variant for why the distinction matters to the circuit breaker. + +If your destination isn't HTTP, also revisit **`build_raw_client()`**: swap +the `reqwest::Client` for your driver's connection/pool setup (see +`core/connectors/sinks/postgres_sink` or `core/connectors/sinks/s3_sink` for +non-HTTP examples), store it on `TemplateSink` in place of the HTTP-specific +`client` field, and adjust or remove the `check_connectivity_with_retry` call +in `open()` in favor of whatever connectivity check your driver offers. + +## Using it + +1. Copy this directory, rename it and the package in `Cargo.toml` + (`iggy_connector__sink`), and add it to the `members` list in + the workspace root `Cargo.toml`. +2. Fill in the `TODO(Developer)` section(s). +3. Update `config.toml` with your real `connection_string` and `target`, and + any settings specific to your system; delete `auth_token` if you don't + need it, or add fields of your own the same way (see + `TemplateSinkConfig`). +4. `cargo build --release -p iggy_connector__sink`, point a + runtime connector config file's `path` at the built `.so`/`.dylib`/`.dll`, + and run the connector runtime — see `core/connectors/README.md` in this + repo for the full runtime quick-start. +5. Before opening a PR: `cargo test`, `cargo clippy --all-targets`, + `cargo fmt --check`, and re-read the connector-review checklist once more + with fresh eyes — most review round-trips come from one of the items in + that list, not from the connector-specific logic in `push_batch()`. diff --git a/core/connectors/sinks/sink_template/config.toml b/core/connectors/sinks/sink_template/config.toml new file mode 100644 index 0000000000..8156530c9d --- /dev/null +++ b/core/connectors/sinks/sink_template/config.toml @@ -0,0 +1,48 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +type = "sink" +key = "template" +enabled = true +version = 0 +name = "Template sink" +path = "../../target/release/libiggy_connector_template_sink" +plugin_config_format = "toml" +verbose = false + +[[streams]] +stream = "example_stream" +topics = ["example_topic"] +schema = "json" +batch_length = 100 +poll_interval = "5ms" +consumer_group = "template_sink_connector" + +[plugin_config] +connection_string = "https://api.example.com" +target = "events" +health_check_path = "/health" +batch_size = 100 +timeout = "30s" +max_retries = 3 +retry_delay = "500ms" +retry_max_delay = "5s" +max_open_retries = 10 +open_retry_max_delay = "60s" +circuit_breaker_threshold = 5 +circuit_breaker_cool_down = "30s" +# auth_token = "replace-me" # uncomment if your destination needs bearer/API-key auth diff --git a/core/connectors/sinks/sink_template/src/lib.rs b/core/connectors/sinks/sink_template/src/lib.rs new file mode 100644 index 0000000000..20a28aa9d6 --- /dev/null +++ b/core/connectors/sinks/sink_template/src/lib.rs @@ -0,0 +1,504 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Template Apache Iggy **sink** connector. +//! +//! Everything in this file is already wired up and follows the patterns the +//! connector-review checklist expects: config validation happens in +//! `open()`, secrets are `SecretString`, the destination identifier is +//! validated before it's ever interpolated into a request, outbound calls +//! go through a retry-wrapped client plus a circuit breaker, and messages +//! are chunked by a configurable batch size instead of shipped as one +//! unbounded request. +//! +//! There is exactly **one** place you need to touch, marked `TODO(Developer)`: +//! `TemplateSink::push_batch()` — build the request/write that actually +//! pushes one chunk of messages to your destination, using +//! `self.config.connection_string` (and `self.config.target`, if your +//! destination has a table/index/collection-shaped name). +//! +//! This template assumes an HTTP-ish destination and uses `reqwest` wrapped +//! by the SDK's retry middleware, because that's what +//! `iggy_connector_sdk::retry` is built for and it covers the common case. +//! If your destination talks something else (a database, a queue, object +//! storage), swap the client type in `connect()`/`push_batch()` for your +//! driver of choice and lean on its own retry/pooling behavior — keep the +//! surrounding shape (validation in `open()`, circuit breaker, batching, +//! identifier validation) unchanged. See `core/connectors/sinks/postgres_sink` +//! or `core/connectors/sinks/s3_sink` in this repo for non-HTTP examples of +//! that same shape. + +use async_trait::async_trait; +use iggy_connector_sdk::retry::{ + CircuitBreaker, ConnectivityConfig, build_retry_client, check_connectivity_with_retry, + parse_duration, +}; +use iggy_connector_sdk::{ + ConsumedMessage, Error, MessagesMetadata, Sink, TopicMetadata, sink_connector, +}; +use reqwest::Url; +use reqwest_middleware::ClientWithMiddleware; +use secrecy::{ExposeSecret, SecretString}; +use serde::{Deserialize, Serialize}; +use std::sync::Arc; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::time::Duration; +use tokio::sync::Mutex; +use tracing::{error, info, warn}; + +sink_connector!(TemplateSink); + +const CONNECTOR_NAME: &str = "Template sink"; + +const DEFAULT_BATCH_SIZE: usize = 100; +const DEFAULT_TIMEOUT: &str = "30s"; +const DEFAULT_MAX_RETRIES: u32 = 3; +const DEFAULT_RETRY_DELAY: &str = "500ms"; +const DEFAULT_RETRY_MAX_DELAY: &str = "5s"; +const DEFAULT_MAX_OPEN_RETRIES: u32 = 10; +const DEFAULT_OPEN_RETRY_MAX_DELAY: &str = "60s"; +const DEFAULT_CIRCUIT_BREAKER_THRESHOLD: u32 = 5; +const DEFAULT_CIRCUIT_BREAKER_COOL_DOWN: &str = "30s"; + +// ── Configuration ─────────────────────────────────────────────────────────── +// +// Every tunable except `connection_string` and `target` is optional with a +// sane default, and unknown keys are rejected outright so a typo in a TOML +// file fails at load time instead of silently doing nothing. + +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct TemplateSinkConfig { + /// TODO(Developer): document the exact shape this connector expects, e.g. + /// "https://api.example.com" or "postgres://user:pass@host:5432/db". + /// `SecretString` because DSNs commonly embed credentials — never plain + /// `String` for this field, see `PostgresSinkConfig::connection_string` + /// in `sinks/postgres_sink` for the same pattern. + #[serde(serialize_with = "iggy_common::serde_secret::serialize_secret")] + pub connection_string: SecretString, + + /// The destination table/index/collection/bucket name. Kept as its own + /// field (rather than folded into `connection_string`) specifically so + /// it can be validated in `open()` before ever being interpolated into + /// a query, path, or URL — see `validate_identifier` below. Delete this + /// field if your destination has no such dynamic identifier. + pub target: String, + + /// Example of a secret-shaped setting. `SecretString` keeps it out of + /// `Debug`/log output; delete this field if `connection_string` already + /// carries all required auth. Read it with `.expose_secret()` (from the + /// `secrecy::ExposeSecret` trait) at the one place you actually need the + /// plaintext — e.g. when building an auth header in `connect()`. + #[serde( + default, + serialize_with = "iggy_common::serde_secret::serialize_optional_secret" + )] + pub auth_token: Option, + + /// Optional path (e.g. "/health") probed with retry during `open()` + /// before the connector is considered ready. Leave unset if your target + /// has no health endpoint — the probe is skipped, not failed, in that case. + pub health_check_path: Option, + + pub batch_size: Option, + pub timeout: Option, + pub max_retries: Option, + pub retry_delay: Option, + pub retry_max_delay: Option, + pub max_open_retries: Option, + pub open_retry_max_delay: Option, + pub circuit_breaker_threshold: Option, + pub circuit_breaker_cool_down: Option, +} + +/// Rejects anything that isn't a plain alphanumeric/underscore identifier. +/// Adjust the allowed character set to whatever your destination's naming +/// rules actually are, but always validate *something* before a +/// config-or-message-derived name is interpolated into a query, path, or +/// URL — see `doris_sink::validate_identifier` / `surrealdb_sink::validate_identifier` +/// in this repo for the same pattern applied to a real destination. +fn validate_identifier(field: &str, value: &str) -> Result<(), Error> { + if value.is_empty() || !value.chars().all(|c| c.is_ascii_alphanumeric() || c == '_') { + return Err(Error::InvalidConfigValue(format!( + "{field} must be non-empty and contain only ASCII alphanumeric characters and \ + underscores, got: {value:?}" + ))); + } + Ok(()) +} + +// ── Internal state ────────────────────────────────────────────────────────── + +#[derive(Debug, Default)] +struct State { + invocations_count: u64, + messages_written: u64, + messages_failed: u64, +} + +#[derive(Debug)] +pub struct TemplateSink { + id: u32, + config: TemplateSinkConfig, + client: Option, + circuit_breaker: Arc, + batch_size_limit: usize, + retry_delay: Duration, + state: Mutex, + records_written_total: AtomicU64, +} + +impl TemplateSink { + pub fn new(id: u32, config: TemplateSinkConfig) -> Self { + let retry_delay = parse_duration(config.retry_delay.as_deref(), DEFAULT_RETRY_DELAY); + let circuit_breaker = Arc::new(CircuitBreaker::new( + config + .circuit_breaker_threshold + .unwrap_or(DEFAULT_CIRCUIT_BREAKER_THRESHOLD), + parse_duration( + config.circuit_breaker_cool_down.as_deref(), + DEFAULT_CIRCUIT_BREAKER_COOL_DOWN, + ), + )); + let batch_size_limit = config.batch_size.unwrap_or(DEFAULT_BATCH_SIZE).max(1); + + Self { + id, + config, + client: None, + circuit_breaker, + batch_size_limit, + retry_delay, + state: Mutex::new(State::default()), + records_written_total: AtomicU64::new(0), + } + } + + /// TODO(Developer): build your actual client/connection here using + /// `self.config.connection_string` (and `self.config.auth_token`, if + /// your destination needs it). This template builds a plain + /// `reqwest::Client` to hand to `build_retry_client` — if you're not + /// talking HTTP, replace this with e.g. a database connection pool or + /// your driver's equivalent, store it on `self` (add a field, since + /// `client` here is HTTP-specific), and skip the + /// `check_connectivity_with_retry` call below in favor of whatever your + /// driver offers (a ping, a test query). + fn build_raw_client(&self) -> Result { + let timeout = parse_duration(self.config.timeout.as_deref(), DEFAULT_TIMEOUT); + reqwest::Client::builder() + .timeout(timeout) + .build() + .map_err(|e| Error::Connection(format!("failed to build HTTP client: {e}"))) + } + + /// TODO(Developer): push one chunk of already-batched messages to your + /// destination using `self.config.connection_string` and + /// `self.config.target`, via `self.client` (already retry-wrapped). + /// Distinguish permanent failures (bad schema, destination rejects the + /// payload shape — will not succeed on retry) from transient ones + /// (network error, 5xx, timeout — should retry) by returning + /// `Error::PermanentHttpError` for the former; see the doc comment on + /// that variant in `iggy_connector_sdk::Error` for why the distinction + /// matters to the circuit breaker. + async fn push_batch( + &self, + client: &ClientWithMiddleware, + batch: &[ConsumedMessage], + ) -> Result<(), Error> { + let _ = (client, batch); // remove once implemented + Err(Error::InitError( + "TemplateSink::push_batch is not implemented yet — see the TODO(Developer) comment in \ + template_sink/src/lib.rs" + .to_string(), + )) + } +} + +// ── Sink trait ──────────────────────────────────────────────────────────────── + +#[async_trait] +impl Sink for TemplateSink { + async fn open(&mut self) -> Result<(), Error> { + // Structural validation happens here, not in `new()`, because only + // `open()` can return an error — `new()` is a plain factory function + // with nowhere to send a "this config is invalid" result. + if self + .config + .connection_string + .expose_secret() + .trim() + .is_empty() + { + return Err(Error::InvalidConfigValue( + "connection_string must not be empty".to_string(), + )); + } + validate_identifier("target", &self.config.target)?; + + info!( + "Opening {CONNECTOR_NAME} connector with ID: {}, target: {}, batch_size: {}", + self.id, self.config.target, self.batch_size_limit + ); + + let raw_client = self.build_raw_client()?; + + if let Some(health_path) = &self.config.health_check_path { + let base = Url::parse(self.config.connection_string.expose_secret()).map_err(|e| { + Error::InvalidConfigValue(format!("connection_string is not a valid URL: {e}")) + })?; + let health_url = base.join(health_path).map_err(|e| { + Error::InvalidConfigValue(format!("invalid health_check_path: {e}")) + })?; + check_connectivity_with_retry( + &raw_client, + health_url, + CONNECTOR_NAME, + self.id, + &ConnectivityConfig { + max_open_retries: self + .config + .max_open_retries + .unwrap_or(DEFAULT_MAX_OPEN_RETRIES), + open_retry_max_delay: parse_duration( + self.config.open_retry_max_delay.as_deref(), + DEFAULT_OPEN_RETRY_MAX_DELAY, + ), + retry_delay: self.retry_delay, + }, + ) + .await?; + } else { + warn!( + "{CONNECTOR_NAME} connector with ID: {} has no health_check_path configured — \ + skipping the startup connectivity probe. Consider adding one.", + self.id + ); + } + + self.client = Some(build_retry_client( + raw_client, + self.config + .max_retries + .unwrap_or(DEFAULT_MAX_RETRIES) + .max(1), + self.retry_delay, + parse_duration( + self.config.retry_max_delay.as_deref(), + DEFAULT_RETRY_MAX_DELAY, + ), + CONNECTOR_NAME, + )); + + info!( + "{CONNECTOR_NAME} connector with ID: {} opened successfully", + self.id + ); + Ok(()) + } + + async fn consume( + &self, + topic_metadata: &TopicMetadata, + messages_metadata: MessagesMetadata, + messages: Vec, + ) -> Result<(), Error> { + let mut state = self.state.lock().await; + state.invocations_count += 1; + let invocation = state.invocations_count; + drop(state); + + info!( + "{CONNECTOR_NAME} with ID: {} received: {} messages, schema: {}, stream: {}, topic: {}, \ + partition: {}, offset: {}, invocation: {}", + self.id, + messages.len(), + messages_metadata.schema, + topic_metadata.stream, + topic_metadata.topic, + messages_metadata.partition_id, + messages_metadata.current_offset, + invocation + ); + + if self.circuit_breaker.is_open().await { + warn!( + "{CONNECTOR_NAME} connector with ID: {} — circuit breaker OPEN, refusing {} messages", + self.id, + messages.len() + ); + return Err(Error::CannotStoreData( + "Circuit breaker is open".to_string(), + )); + } + + let client = self.client.as_ref().ok_or_else(|| { + Error::Connection("client not initialized -- was open() called?".into()) + })?; + + let mut first_error: Option = None; + let mut written = 0u64; + let mut failed = 0u64; + + for batch in messages.chunks(self.batch_size_limit) { + match self.push_batch(client, batch).await { + Ok(()) => written += batch.len() as u64, + Err(err) => { + failed += batch.len() as u64; + error!( + "{CONNECTOR_NAME} connector with ID: {} failed a batch of {}: {err}", + self.id, + batch.len() + ); + if first_error.is_none() { + first_error = Some(err); + } + } + } + } + + // Record the circuit breaker outcome once per `consume()` call, not + // once per chunk — recording success partway through would reset + // the failure counter mid-consume and prevent the breaker from + // opening on a batch with sustained, mixed-success chunks. + match &first_error { + None => self.circuit_breaker.record_success(), + Some(e) if !matches!(e, Error::PermanentHttpError(_)) => { + self.circuit_breaker.record_failure().await; + } + Some(_) => {} + } + + let mut state = self.state.lock().await; + state.messages_written += written; + state.messages_failed += failed; + drop(state); + self.records_written_total + .fetch_add(written, Ordering::Relaxed); + + match first_error { + None => Ok(()), + Some(err) => Err(err), + } + } + + async fn close(&mut self) -> Result<(), Error> { + let state = self.state.lock().await; + info!( + "{CONNECTOR_NAME} connector with ID: {} closing. Stats: {} invocations, {} messages written, \ + {} messages failed", + self.id, state.invocations_count, state.messages_written, state.messages_failed + ); + drop(state); + self.client = None; + info!("{CONNECTOR_NAME} connector with ID: {} is closed.", self.id); + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn test_config() -> TemplateSinkConfig { + TemplateSinkConfig { + connection_string: SecretString::from("https://api.example.com"), + target: "events".to_string(), + auth_token: None, + health_check_path: None, + batch_size: Some(10), + timeout: Some("5s".to_string()), + max_retries: Some(2), + retry_delay: Some("10ms".to_string()), + retry_max_delay: Some("100ms".to_string()), + max_open_retries: Some(2), + open_retry_max_delay: Some("100ms".to_string()), + circuit_breaker_threshold: Some(3), + circuit_breaker_cool_down: Some("50ms".to_string()), + } + } + + #[tokio::test] + async fn open_rejects_empty_connection_string() { + let mut config = test_config(); + config.connection_string = SecretString::from(" "); + let mut sink = TemplateSink::new(1, config); + assert!(matches!( + sink.open().await, + Err(Error::InvalidConfigValue(_)) + )); + } + + #[tokio::test] + async fn open_rejects_invalid_target_identifier() { + let mut config = test_config(); + config.target = "events; DROP TABLE users;--".to_string(); + let mut sink = TemplateSink::new(1, config); + assert!(matches!( + sink.open().await, + Err(Error::InvalidConfigValue(_)) + )); + } + + #[test] + fn validate_identifier_accepts_plain_names() { + assert!(validate_identifier("target", "events_v2").is_ok()); + } + + #[test] + fn validate_identifier_rejects_empty() { + assert!(validate_identifier("target", "").is_err()); + } + + #[test] + fn validate_identifier_rejects_path_and_query_characters() { + for bad in [ + "../etc/passwd", + "events?x=1", + "events/../secrets", + "events;drop", + ] { + assert!( + validate_identifier("target", bad).is_err(), + "expected {bad:?} to be rejected" + ); + } + } + + #[tokio::test] + async fn consume_short_circuits_when_breaker_is_open() { + let sink = TemplateSink::new(1, test_config()); + sink.circuit_breaker.record_failure().await; + sink.circuit_breaker.record_failure().await; + sink.circuit_breaker.record_failure().await; + assert!(sink.circuit_breaker.is_open().await); + + let topic_metadata = TopicMetadata { + stream: "s".to_string(), + topic: "t".to_string(), + }; + let messages_metadata = MessagesMetadata { + partition_id: 1, + current_offset: 0, + schema: iggy_connector_sdk::Schema::Json, + }; + + let result = sink + .consume(&topic_metadata, messages_metadata, Vec::new()) + .await; + assert!(matches!(result, Err(Error::CannotStoreData(_)))); + } +} diff --git a/core/connectors/sources/README.md b/core/connectors/sources/README.md index a795735f43..640cd0ad5c 100644 --- a/core/connectors/sources/README.md +++ b/core/connectors/sources/README.md @@ -12,6 +12,7 @@ Source connectors are responsible for ingesting data from external sources into | **influxdb_source** | Polls InfluxDB with cursor-based timestamp tracking; supports V2 (Flux, annotated CSV) and V3 (SQL, JSONL) | | **postgres_source** | Reads rows from PostgreSQL tables with multiple strategies: delete after read, mark as processed, or timestamp tracking | | **random_source** | Generates random test messages (useful for testing and development) | +| **source_template** | Fill-in-the-blank starting point for a new source; framework/security plumbing done, two `TODO(Developer)` spots left | The source is represented by the single `Source` trait, which defines the basic interface for all source connectors. It provides methods for initializing the source, reading data from it, and closing the source. diff --git a/core/connectors/sources/source_template/Cargo.toml b/core/connectors/sources/source_template/Cargo.toml new file mode 100644 index 0000000000..a54d7cbccc --- /dev/null +++ b/core/connectors/sources/source_template/Cargo.toml @@ -0,0 +1,60 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. +# +# TEMPLATE — rename the package (and this directory) to +# `iggy_connector__source` before publishing, and update the +# `[[sources]]` entry you add to the workspace root Cargo.toml accordingly. + +[package] +name = "iggy_connector_template_source" +version = "0.1.0" +description = "Template for an Apache Iggy source connector — copy this crate and fill in the TODO sections." +edition = "2024" +license = "Apache-2.0" +keywords = ["iggy", "messaging", "streaming"] +categories = ["command-line-utilities", "database", "network-programming"] +homepage = "https://iggy.apache.org" +documentation = "https://iggy.apache.org/docs" +repository = "https://github.com/apache/iggy" +publish = false + +[package.metadata.cargo-machete] +# dashmap is used only inside the `source_connector!` macro expansion, so a +# naive unused-dependency scan won't see the usage — keep it ignored rather +# than removing it, or the plugin will fail to compile. +ignored = ["dashmap"] + +[lib] +crate-type = ["cdylib", "lib"] + +[dependencies] +async-trait = { workspace = true } +dashmap = { workspace = true } +humantime = { workspace = true } +iggy_common = { workspace = true } +iggy_connector_sdk = { workspace = true } +reqwest = { workspace = true } +reqwest-middleware = { workspace = true } +rmp-serde = { workspace = true } +secrecy = { workspace = true } +serde = { workspace = true } +serde_json = { workspace = true } +tokio = { workspace = true } +tracing = { workspace = true } + +[dev-dependencies] +tokio = { workspace = true, features = ["test-util"] } diff --git a/core/connectors/sources/source_template/README.md b/core/connectors/sources/source_template/README.md new file mode 100644 index 0000000000..f4a39d3bac --- /dev/null +++ b/core/connectors/sources/source_template/README.md @@ -0,0 +1,64 @@ +# Template source connector + +Starting point for a new Apache Iggy **source** connector. Everything except +talking to your actual external system is already implemented and follows +the project's required resilience/security patterns — see the module-level +doc comment at the top of `src/lib.rs` for the full rationale, and the +`iggy-connector-review` skill / the "Building Connectors That Pass Review" +blog post for the checklist this template is built against. + +## What's already done for you + +- Config parsing with `#[serde(deny_unknown_fields)]` so a typo in a TOML + file fails loudly instead of silently doing nothing. +- Config validation in `open()` (not `new()`, which has no way to return an + error). +- `connection_string` and the optional `auth_token` field both typed as + `SecretString`, since either can carry credentials. +- A retry-wrapped HTTP client (`iggy_connector_sdk::retry::build_retry_client`) + and a startup connectivity probe with its own backoff + (`check_connectivity_with_retry`). +- A `CircuitBreaker` that's actually consulted before polling and updated + after every attempt — not just constructed and forgotten. +- Cursor staging: `poll()` never commits its progress directly; it stages a + candidate and `on_batch_result()` commits it only on `Ack`, discarding it + on `Nack` so a failed delivery gets re-polled instead of silently lost. +- The `source_connector!` FFI macro invocation and a `Cargo.toml` with the + right `crate-type`, workspace-pinned dependencies, and license header. +- Tests for config validation and the Ack/Nack state-commit behavior. + +## What you need to fill in + +Search for `TODO(Developer)` in `src/lib.rs` — there are exactly two spots: + +1. **`build_raw_client()`** — if your source isn't HTTP, replace the + `reqwest::Client` construction with your driver's connection/pool setup + (see `core/connectors/sources/postgres_source` for a real non-HTTP + example), store it on `TemplateSource` (you'll need to add a field — + `client: Option` here is HTTP-specific), and adjust + or remove the `check_connectivity_with_retry` call in `open()` in favor + of whatever connectivity check your driver offers. +2. **`fetch_records()`** — fetch up to `self.batch_size` new records from + your system, ordered after `cursor` (`None` = start from the beginning, + or from "now" — whichever fits your source). Map each result to a + `FetchedRecord { cursor_value, payload }`, using something monotonically + increasing as `cursor_value` (a timestamp, an ID, a page token) — that's + what lets the cursor-staging logic advance correctly. + +## Using it + +1. Copy this directory, rename it and the package in `Cargo.toml` + (`iggy_connector__source`), and add it to the `members` list in + the workspace root `Cargo.toml`. +2. Fill in the two `TODO(Developer)` sections. +3. Update `config.toml` with your real `connection_string` and any + settings specific to your system; delete `auth_token` if you don't need + it, or add fields of your own the same way (see `TemplateSourceConfig`). +4. `cargo build --release -p iggy_connector__source`, point a + runtime connector config file's `path` at the built `.so`/`.dylib`/`.dll`, + and run the connector runtime — see `core/connectors/README.md` in this + repo for the full runtime quick-start. +5. Before opening a PR: `cargo test`, `cargo clippy --all-targets`, + `cargo fmt --check`, and re-read the connector-review checklist once more + with fresh eyes — most review round-trips come from one of the items in + that list, not from the connector-specific logic in `fetch_records()`. diff --git a/core/connectors/sources/source_template/config.toml b/core/connectors/sources/source_template/config.toml new file mode 100644 index 0000000000..53ac14874e --- /dev/null +++ b/core/connectors/sources/source_template/config.toml @@ -0,0 +1,47 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +type = "source" +key = "template" +enabled = true +version = 0 +name = "Template source" +path = "../../target/release/libiggy_connector_template_source" +plugin_config_format = "toml" +verbose = false + +[[streams]] +stream = "example_stream" +topic = "example_topic" +schema = "json" +batch_length = 100 +linger_time = "5ms" + +[plugin_config] +connection_string = "https://api.example.com" +health_check_path = "/health" +batch_size = 100 +poll_interval = "1s" +timeout = "30s" +max_retries = 3 +retry_delay = "500ms" +retry_max_delay = "5s" +max_open_retries = 10 +open_retry_max_delay = "60s" +circuit_breaker_threshold = 5 +circuit_breaker_cool_down = "30s" +# auth_token = "replace-me" # uncomment if your source needs bearer/API-key auth diff --git a/core/connectors/sources/source_template/src/lib.rs b/core/connectors/sources/source_template/src/lib.rs new file mode 100644 index 0000000000..24f52b44c1 --- /dev/null +++ b/core/connectors/sources/source_template/src/lib.rs @@ -0,0 +1,586 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Template Apache Iggy **source** connector. +//! +//! Everything in this file is already wired up and follows the patterns the +//! connector-review checklist expects: config validation happens in +//! `open()`, secrets are `SecretString`, outbound calls go through a +//! retry-wrapped client plus a circuit breaker, and the read cursor is +//! staged in `poll()` and only committed in `on_batch_result()` on an ACK so +//! a dropped/nacked batch can be re-polled instead of silently lost. +//! +//! There are exactly **two** places you need to touch, each marked +//! `TODO(Developer)`: +//! 1. `TemplateSource::connect()` — build your actual client/connection +//! from `config.connection_string` (and `config.auth_token`, if used). +//! 2. `TemplateSource::fetch_records()` — fetch up to `batch_size` new +//! records from your external system, starting after `cursor`. +//! +//! This template assumes an HTTP-ish source and uses `reqwest` wrapped by +//! the SDK's retry middleware, because that's what `iggy_connector_sdk::retry` +//! is built for and it covers the common case. If your source talks to +//! something else (a database, a queue, a filesystem), swap the client type +//! in `connect()`/`fetch_records()` for your driver of choice and lean on +//! its own retry/pooling behavior — keep the surrounding shape (validation +//! in `open()`, circuit breaker, cursor staging, batching) unchanged. See +//! `core/connectors/sources/postgres_source` in this repo for a real +//! non-HTTP example of that same shape. + +use async_trait::async_trait; +use iggy_connector_sdk::retry::{ + CircuitBreaker, ConnectivityConfig, build_retry_client, check_connectivity_with_retry, + parse_duration, +}; +use iggy_connector_sdk::{ + ConnectorState, Error, ProducedMessage, ProducedMessages, Schema, Source, + source::SourceBatchResult, source_connector, +}; +use reqwest::Url; +use reqwest_middleware::ClientWithMiddleware; +use secrecy::{ExposeSecret, SecretString}; +use serde::{Deserialize, Serialize}; +use std::sync::Arc; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::time::Duration; +use tokio::sync::Mutex; +use tracing::{error, info, warn}; + +source_connector!(TemplateSource); + +const CONNECTOR_NAME: &str = "Template source"; + +const DEFAULT_POLL_INTERVAL: &str = "1s"; +const DEFAULT_BATCH_SIZE: u32 = 100; +const DEFAULT_TIMEOUT: &str = "30s"; +const DEFAULT_MAX_RETRIES: u32 = 3; +const DEFAULT_RETRY_DELAY: &str = "500ms"; +const DEFAULT_RETRY_MAX_DELAY: &str = "5s"; +const DEFAULT_MAX_OPEN_RETRIES: u32 = 10; +const DEFAULT_OPEN_RETRY_MAX_DELAY: &str = "60s"; +const DEFAULT_CIRCUIT_BREAKER_THRESHOLD: u32 = 5; +const DEFAULT_CIRCUIT_BREAKER_COOL_DOWN: &str = "30s"; + +// ── Configuration ─────────────────────────────────────────────────────────── +// +// Every tunable except `connection_string` is optional with a sane default, +// and unknown keys are rejected outright so a typo in a TOML file fails at +// load time instead of silently doing nothing. + +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct TemplateSourceConfig { + /// TODO(Developer): document the exact shape this connector expects, e.g. + /// "https://api.example.com" or "postgres://user:pass@host:5432/db". + /// `SecretString` because DSNs commonly embed credentials — never plain + /// `String` for this field, see `PostgresSinkConfig::connection_string` + /// in `sinks/postgres_sink` for the same pattern. + #[serde(serialize_with = "iggy_common::serde_secret::serialize_secret")] + pub connection_string: SecretString, + + /// Example of a secret-shaped setting. `SecretString` keeps it out of + /// `Debug`/log output; delete this field if `connection_string` already + /// carries all required auth. Read it with `.expose_secret()` (from the + /// `secrecy::ExposeSecret` trait) at the one place you actually need the + /// plaintext — e.g. when building an auth header in `connect()`. + #[serde( + default, + serialize_with = "iggy_common::serde_secret::serialize_optional_secret" + )] + pub auth_token: Option, + + /// Optional path (e.g. "/health") probed with retry during `open()` + /// before the connector is considered ready. Leave unset if your target + /// has no health endpoint — the probe is skipped, not failed, in that case. + pub health_check_path: Option, + + pub batch_size: Option, + pub poll_interval: Option, + pub timeout: Option, + pub max_retries: Option, + pub retry_delay: Option, + pub retry_max_delay: Option, + pub max_open_retries: Option, + pub open_retry_max_delay: Option, + pub circuit_breaker_threshold: Option, + pub circuit_breaker_cool_down: Option, +} + +// ── Internal state ────────────────────────────────────────────────────────── + +/// Read cursor. Kept as a plain `Option` so it fits whatever ordering +/// field your source uses (a timestamp, an auto-increment ID, an opaque +/// pagination token, ...) — stringify it however makes sense for your data. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +struct State { + cursor: Option, +} + +/// One record as fetched from the external system, before it's turned into +/// an Iggy message. Replace the `payload` type with whatever your +/// `fetch_records()` actually produces. +struct FetchedRecord { + /// The value `cursor` should advance to once this record's batch is + /// acknowledged — typically this record's timestamp/ID/token. + cursor_value: String, + payload: serde_json::Value, +} + +#[derive(Debug)] +pub struct TemplateSource { + id: u32, + config: TemplateSourceConfig, + client: Option, + circuit_breaker: Arc, + batch_size: u32, + poll_interval: Duration, + retry_delay: Duration, + state: Mutex, + pending_state: Mutex>, + records_produced: AtomicU64, +} + +impl TemplateSource { + pub fn new(id: u32, config: TemplateSourceConfig, state: Option) -> Self { + let poll_interval = *humantime::Duration::from_str_lossy( + config + .poll_interval + .as_deref() + .unwrap_or(DEFAULT_POLL_INTERVAL), + ); + let retry_delay = parse_duration(config.retry_delay.as_deref(), DEFAULT_RETRY_DELAY); + let circuit_breaker = Arc::new(CircuitBreaker::new( + config + .circuit_breaker_threshold + .unwrap_or(DEFAULT_CIRCUIT_BREAKER_THRESHOLD), + parse_duration( + config.circuit_breaker_cool_down.as_deref(), + DEFAULT_CIRCUIT_BREAKER_COOL_DOWN, + ), + )); + let batch_size = config.batch_size.unwrap_or(DEFAULT_BATCH_SIZE).max(1); + + let restored_state = state + .and_then(|s| s.deserialize::(CONNECTOR_NAME, id)) + .inspect(|s| { + info!( + "Restored state for {CONNECTOR_NAME} connector with ID: {id}. Cursor: {:?}", + s.cursor + ); + }); + + Self { + id, + config, + client: None, + circuit_breaker, + batch_size, + poll_interval, + retry_delay, + state: Mutex::new(restored_state.unwrap_or_default()), + pending_state: Mutex::new(None), + records_produced: AtomicU64::new(0), + } + } + + fn serialize_state(&self, state: &State) -> Option { + ConnectorState::serialize(state, CONNECTOR_NAME, self.id) + } + + /// TODO(Developer): build your actual client/connection here using + /// `self.config.connection_string` (and `self.config.auth_token`, if + /// your system needs it). This template builds a plain `reqwest::Client` + /// to hand to `build_retry_client` — if you're not talking HTTP, replace + /// this with e.g. a `sqlx::PgPool::connect(...)` or your driver's + /// equivalent, store it on `self` (add a field, since `client` here is + /// HTTP-specific), and skip the `check_connectivity_with_retry` call + /// below in favor of whatever your driver offers (a ping, a test query). + fn build_raw_client(&self) -> Result { + let timeout = parse_duration(self.config.timeout.as_deref(), DEFAULT_TIMEOUT); + reqwest::Client::builder() + .timeout(timeout) + .build() + .map_err(|e| Error::Connection(format!("failed to build HTTP client: {e}"))) + } + + /// TODO(Developer): fetch up to `self.batch_size` new records from your + /// external system, ordered after `cursor` (`None` means "from the + /// beginning" or "from now" — whichever is right for your source). + /// Use `self.config.connection_string` as the base address and + /// `self.client` (already retry-wrapped) to make the request. Map each + /// result row/document/event to a `FetchedRecord`, using something that + /// monotonically increases (a timestamp, an ID, a page token) as + /// `cursor_value` so the state-staging logic in `poll()` below can + /// advance the cursor correctly. + async fn fetch_records( + &self, + client: &ClientWithMiddleware, + cursor: Option<&str>, + ) -> Result, Error> { + let _ = (client, cursor); // remove once implemented + Err(Error::InitError( + "TemplateSource::fetch_records is not implemented yet — see the TODO(Developer) comment \ + in template_source/src/lib.rs" + .to_string(), + )) + } +} + +// ── Source trait ──────────────────────────────────────────────────────────── + +#[async_trait] +impl Source for TemplateSource { + async fn open(&mut self) -> Result<(), Error> { + // Structural validation happens here, not in `new()`, because only + // `open()` can return an error — `new()` is a plain factory function + // with nowhere to send a "this config is invalid" result. + if self + .config + .connection_string + .expose_secret() + .trim() + .is_empty() + { + return Err(Error::InvalidConfigValue( + "connection_string must not be empty".to_string(), + )); + } + + info!( + "Opening {CONNECTOR_NAME} connector with ID: {}, batch_size: {}, poll_interval: {:?}", + self.id, self.batch_size, self.poll_interval + ); + + let raw_client = self.build_raw_client()?; + + if let Some(health_path) = &self.config.health_check_path { + let base = Url::parse(self.config.connection_string.expose_secret()).map_err(|e| { + Error::InvalidConfigValue(format!("connection_string is not a valid URL: {e}")) + })?; + let health_url = base.join(health_path).map_err(|e| { + Error::InvalidConfigValue(format!("invalid health_check_path: {e}")) + })?; + check_connectivity_with_retry( + &raw_client, + health_url, + CONNECTOR_NAME, + self.id, + &ConnectivityConfig { + max_open_retries: self + .config + .max_open_retries + .unwrap_or(DEFAULT_MAX_OPEN_RETRIES), + open_retry_max_delay: parse_duration( + self.config.open_retry_max_delay.as_deref(), + DEFAULT_OPEN_RETRY_MAX_DELAY, + ), + retry_delay: self.retry_delay, + }, + ) + .await?; + } else { + warn!( + "{CONNECTOR_NAME} connector with ID: {} has no health_check_path configured — \ + skipping the startup connectivity probe. Consider adding one.", + self.id + ); + } + + self.client = Some(build_retry_client( + raw_client, + self.config + .max_retries + .unwrap_or(DEFAULT_MAX_RETRIES) + .max(1), + self.retry_delay, + parse_duration( + self.config.retry_max_delay.as_deref(), + DEFAULT_RETRY_MAX_DELAY, + ), + CONNECTOR_NAME, + )); + + info!( + "{CONNECTOR_NAME} connector with ID: {} opened successfully", + self.id + ); + Ok(()) + } + + async fn poll(&self) -> Result { + // If the breaker is open, sleep for the normal poll interval and + // return an empty (not an error) result. Returning `Err` here would + // make the runtime retry `poll()` again immediately with no delay — + // see `handle_messages` in the SDK's source container — so an empty + // ACK-free result is how a source waits out a known-bad window + // without busy-looping or counting against the NACK budget. + if self.circuit_breaker.is_open().await { + warn!( + "{CONNECTOR_NAME} connector with ID: {} — circuit breaker OPEN, skipping poll", + self.id + ); + tokio::time::sleep(self.poll_interval).await; + return Ok(ProducedMessages { + schema: Schema::Json, + messages: Vec::new(), + state: None, + }); + } + tokio::time::sleep(self.poll_interval).await; + + let client = self.client.as_ref().ok_or_else(|| { + Error::Connection("client not initialized -- was open() called?".into()) + })?; + let cursor = self.state.lock().await.cursor.clone(); + + let records = match self.fetch_records(client, cursor.as_deref()).await { + Ok(records) => { + self.circuit_breaker.record_success(); + records + } + Err(err) => { + if !matches!(err, Error::PermanentHttpError(_)) { + self.circuit_breaker.record_failure().await; + } + return Err(err); + } + }; + + if records.is_empty() { + return Ok(ProducedMessages { + schema: Schema::Json, + messages: Vec::new(), + state: None, + }); + } + + let mut messages = Vec::with_capacity(records.len()); + let mut new_cursor = cursor; + for record in records { + new_cursor = Some(record.cursor_value); + let Ok(payload) = serde_json::to_vec(&record.payload) else { + error!( + "Failed to serialize a record fetched by {CONNECTOR_NAME} connector with ID: {}", + self.id + ); + continue; + }; + messages.push(ProducedMessage { + id: None, + headers: None, + checksum: None, + timestamp: None, + origin_timestamp: None, + payload, + }); + } + + let candidate_state = State { cursor: new_cursor }; + let persisted_state = self.serialize_state(&candidate_state).ok_or_else(|| { + Error::Serialization(format!( + "failed to serialize state for {CONNECTOR_NAME} connector with ID: {}", + self.id + )) + })?; + *self.pending_state.lock().await = Some(candidate_state); + + self.records_produced + .fetch_add(messages.len() as u64, Ordering::Relaxed); + + Ok(ProducedMessages { + schema: Schema::Json, + messages, + state: Some(persisted_state), + }) + } + + /// The staged cursor from `poll()` is only committed here, and only on + /// an ACK. A NACK (delivery failed, batch timed out, runtime is + /// shutting down) discards the candidate so the same range is re-polled + /// next time — the cursor never moves past data that wasn't confirmed + /// delivered. If `fetch_records()` ever needs to perform a destructive + /// read against the source (delete-after-read, mark-as-processed), + /// stage that side effect the same way and only apply it here on `Ack`. + async fn on_batch_result(&self, result: SourceBatchResult) -> Result<(), Error> { + let candidate_state = self.pending_state.lock().await.take(); + if result == SourceBatchResult::Ack + && let Some(candidate_state) = candidate_state + { + *self.state.lock().await = candidate_state; + } + Ok(()) + } + + async fn close(&mut self) -> Result<(), Error> { + let state = self.state.lock().await; + info!( + "{CONNECTOR_NAME} connector with ID: {} closed. Cursor: {:?}, total records produced: {}", + self.id, + state.cursor, + self.records_produced.load(Ordering::Relaxed) + ); + drop(state); + self.client = None; + Ok(()) + } +} + +// Small local shim so `new()` doesn't need to pull in `humantime::Duration`'s +// `FromStr` (which returns `Result`) just to apply a default — mirrors the +// fallback-with-warning behavior of `iggy_connector_sdk::retry::parse_duration` +// for the one duration field (`poll_interval`) that isn't itself optional in +// spirit (there's always a poll interval, just maybe the default one). +trait DurationExt { + fn from_str_lossy(s: &str) -> humantime::Duration; +} +impl DurationExt for humantime::Duration { + fn from_str_lossy(s: &str) -> humantime::Duration { + use std::str::FromStr; + humantime::Duration::from_str(s).unwrap_or_else(|_| { + warn!("Invalid poll_interval {s:?}, falling back to {DEFAULT_POLL_INTERVAL}"); + humantime::Duration::from_str(DEFAULT_POLL_INTERVAL) + .expect("DEFAULT_POLL_INTERVAL must itself be a valid duration literal") + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn test_config() -> TemplateSourceConfig { + TemplateSourceConfig { + connection_string: SecretString::from("https://api.example.com"), + auth_token: None, + health_check_path: None, + batch_size: Some(50), + poll_interval: Some("50ms".to_string()), + timeout: Some("5s".to_string()), + max_retries: Some(2), + retry_delay: Some("10ms".to_string()), + retry_max_delay: Some("100ms".to_string()), + max_open_retries: Some(2), + open_retry_max_delay: Some("100ms".to_string()), + circuit_breaker_threshold: Some(3), + circuit_breaker_cool_down: Some("50ms".to_string()), + } + } + + #[tokio::test] + async fn open_rejects_empty_connection_string() { + let mut config = test_config(); + config.connection_string = SecretString::from(" "); + let mut source = TemplateSource::new(1, config, None); + let result = source.open().await; + assert!(matches!(result, Err(Error::InvalidConfigValue(_)))); + } + + #[tokio::test] + async fn given_no_state_should_start_with_no_cursor() { + let source = TemplateSource::new(1, test_config(), None); + assert_eq!(source.state.lock().await.cursor, None); + } + + #[tokio::test] + async fn given_persisted_state_should_restore_cursor() { + let state = State { + cursor: Some("2024-01-01T00:00:00Z".to_string()), + }; + let serialized = rmp_serde::to_vec(&state).expect("failed to serialize state"); + let source = TemplateSource::new(1, test_config(), Some(ConnectorState(serialized))); + assert_eq!( + source.state.lock().await.cursor, + Some("2024-01-01T00:00:00Z".to_string()) + ); + } + + #[tokio::test] + async fn given_invalid_persisted_state_should_start_fresh() { + let source = TemplateSource::new( + 1, + test_config(), + Some(ConnectorState(b"not valid msgpack".to_vec())), + ); + assert_eq!(source.state.lock().await.cursor, None); + } + + #[test] + fn state_should_be_serializable_and_deserializable() { + let original = State { + cursor: Some("2024-01-01T00:00:00Z".to_string()), + }; + + let serialized = rmp_serde::to_vec(&original).expect("failed to serialize state"); + let deserialized: State = + rmp_serde::from_slice(&serialized).expect("failed to deserialize state"); + + assert_eq!(original.cursor, deserialized.cursor); + } + + #[tokio::test] + async fn given_ack_should_commit_staged_cursor() { + let source = TemplateSource::new(1, test_config(), None); + *source.pending_state.lock().await = Some(State { + cursor: Some("next-cursor".to_string()), + }); + + source + .on_batch_result(SourceBatchResult::Ack) + .await + .expect("ACK should be applied"); + + assert_eq!( + source.state.lock().await.cursor, + Some("next-cursor".to_string()) + ); + assert!(source.pending_state.lock().await.is_none()); + } + + #[tokio::test] + async fn given_nack_should_discard_staged_cursor() { + let source = TemplateSource::new(1, test_config(), None); + *source.pending_state.lock().await = Some(State { + cursor: Some("next-cursor".to_string()), + }); + + source + .on_batch_result(SourceBatchResult::Nack) + .await + .expect("NACK should be applied"); + + // The committed cursor is unchanged (still None) — the candidate is + // simply discarded so the same range is polled again. + assert_eq!(source.state.lock().await.cursor, None); + assert!(source.pending_state.lock().await.is_none()); + } + + #[tokio::test] + async fn poll_returns_empty_without_error_when_circuit_is_open() { + let source = TemplateSource::new(1, test_config(), None); + source.circuit_breaker.record_failure().await; + source.circuit_breaker.record_failure().await; + source.circuit_breaker.record_failure().await; + assert!(source.circuit_breaker.is_open().await); + + let result = source + .poll() + .await + .expect("open breaker should not error poll()"); + assert!(result.messages.is_empty()); + assert!(result.state.is_none()); + } +} From 8f097c8525038ca01003e82ba3e62ecedb5cb18c Mon Sep 17 00:00:00 2001 From: ryerraguntla Date: Tue, 25 Aug 2026 09:16:11 +0530 Subject: [PATCH 002/182] Normalising the naming convention --- .claude/skills/connector-pr-review/SKILL.md | 248 +++++++++ .claude/skills/connector-runtime/SKILL.md | 4 +- .claude/skills/connector-sink/SKILL.md | 3 +- .claude/skills/connector-sink/TEMPLATE.md | 329 ++++++++++-- .claude/skills/connector-source/SKILL.md | 104 +++- .claude/skills/connector-source/TEMPLATE.md | 487 +++++++++++++++--- .claude/skills/connectors-overview/SKILL.md | 70 ++- core/connectors/BLOG_POST.md | 4 +- core/connectors/sinks/README.md | 2 +- core/connectors/sinks/sink_template/README.md | 4 +- .../connectors/sinks/sink_template/src/lib.rs | 10 +- core/connectors/sources/README.md | 2 +- .../sources/source_template/README.md | 4 +- .../sources/source_template/src/lib.rs | 10 +- 14 files changed, 1078 insertions(+), 203 deletions(-) create mode 100644 .claude/skills/connector-pr-review/SKILL.md diff --git a/.claude/skills/connector-pr-review/SKILL.md b/.claude/skills/connector-pr-review/SKILL.md new file mode 100644 index 0000000000..c386099c79 --- /dev/null +++ b/.claude/skills/connector-pr-review/SKILL.md @@ -0,0 +1,248 @@ +--- +name: connector-pr-review +description: Review checklist for Apache Iggy connector sink/source PRs. Load when reviewing a connectors plugin PR, when authoring a new sink/source and wanting to pre-flight against common review blockers, or when diagnosing why a connectors PR is stuck in review. Encodes recurring review patterns mined from real apache/iggy connector PRs. NOT for runtime/SDK internals (use connector-runtime / connector-sdk). +--- + +# Connector PR review checklist + +> Universal rules live in [connectors-overview](../connectors-overview/SKILL.md). +> Authoring skills: [connector-sink](../connector-sink/SKILL.md), +> [connector-source](../connector-source/SKILL.md), +> [connector-testing](../connector-testing/SKILL.md). +> Fill-in-the-blank kits: those skills' `TEMPLATE.md` files. + +Use this skill to **catch the issues that repeatedly burn review cycles** +before asking for a human re-review. Cite symbols/paths, not stale line numbers. + +## Contents + +- [How to use](#how-to-use) +- [Blockers (must fix before merge)](#blockers-must-fix-before-merge) +- [High-frequency convention nits](#high-frequency-convention-nits) +- [Delivery semantics (document honestly)](#delivery-semantics-document-honestly) +- [PR / CI hygiene](#pr--ci-hygiene) +- [Pre-flight author checklist](#pre-flight-author-checklist) +- [Evidence base](#evidence-base) + +## How to use + +1. Load this skill for any PR under `core/connectors/sinks/` or `core/connectors/sources/`. +2. Walk **Blockers** first. Any hit is CHANGES_REQUESTED. +3. Then **Convention nits** and **Delivery semantics**. +4. End with **PR / CI hygiene** (cheap passes that still delay first review). +5. Prefer "copy the closest exemplar" over inventing new knobs. + +## Blockers (must fix before merge) + +### B1. Secrets + +- [ ] Every credential field is `secrecy::SecretString` with + `#[serde(serialize_with = "iggy_common::serde_secret::serialize_secret")]`. +- [ ] No connection string / API key / token in `info!`/`debug!`/`error!` / + `format!` into SQL / file-state metadata. +- [ ] Source state must not persist URL userinfo or bearer tokens. + +Plain `String` for a credential is a review-blocker. Pattern: +`PostgresSinkConfig::connection_string` in `sinks/postgres_sink`. + +### B2. Swallowing `Err` while offsets advance (sinks) + +- [ ] `consume()` must **not** catch a batch failure and return `Ok(())`. +- [ ] Prefer `last_err` pattern: process all batches, return the last transient + error (see `connector-sink` Hard rules / TEMPLATE). +- [ ] README must not claim "no data loss" / strong idempotency unless the + backend + runtime path actually enforce it. + +Today the runtime can commit consumer offsets even when plugin errors are +poorly surfaced (#2927 / #2928 class issues). Authors must be honest about +loss windows instead of overselling. + +### B3. Source cursor / side-effects staged for `on_batch_result`, not committed in `poll()` + +Since #3855, this is an SDK-enforced contract, not just a convention: +`poll()` returns *candidate* state; the runtime sends the batch, saves +that state on success, then calls `on_batch_result(Ack | Nack)`. Only +`on_batch_result` may commit. + +- [ ] `poll()` stages cursor changes and destructive work (delete-after-read / + mark-processed / ACK upstream) - it does not mutate committed state or + touch upstream rows directly. Committing in `poll()` leaves nothing to + roll back on a Nack, defeating the point of the handshake. +- [ ] `on_batch_result(SourceBatchResult::Ack)` commits the staged work; + `on_batch_result(SourceBatchResult::Nack)` discards it so the batch is + redelivered against unchanged committed state. A source with no staged + work (e.g. a pure generator) may rely on the SDK's default no-op impl. +- [ ] Returning `Err` from `on_batch_result` is deliberate - it stops the SDK + from polling further rather than risk silently advancing past a failed + rollback. Don't swallow a rollback failure into `Ok(())`. +- [ ] Always return `ConnectorState` in every `ProducedMessages` that made + progress; return `state: None` for an empty poll that made none (avoids + an unnecessary write and can't persist state left over from a failed + batch). +- [ ] `poll()` sleeps **first**, then fetches (never sleep after holding a batch). + +### B4. Transient vs permanent errors + +- [ ] Infra/auth/schema-gone failures map to `Error::PermanentHttpError` / + `Error::InitError` / `Error::SchemaMismatch` — not `InvalidRecord`. +- [ ] Retryable network/5xx/SQLSTATE map to transient variants + (`HttpRequestFailed`, `Connection`, `CannotStoreData`). +- [ ] Do **not** classify retryability by substring-matching `err.to_string()`. +- [ ] `max_retries` means **total attempts** (default 3). README must match code. +- [ ] Cap retry budget so a dead backend cannot delay shutdown unboundedly. + +### B5. Idempotency claims must be real + +- [ ] Stable `ProducedMessage.id` / sink dedup key from natural IDs + (table+PK, document `_id`, `stream:topic:partition:message_id`) — + never random UUIDs per emit. +- [ ] If the backend PK / unique index is informational only (e.g. Redshift), + do not advertise idempotency in README. +- [ ] External workflow IDs (Airflow `dag_run_id`, etc.) must be deterministic + across retries. + +### B6. Secrets / license policy for new SDKs + +- [ ] New backend crates pass `scripts/ci/third-party-licenses.sh` (no BUSL / + incompatible licenses pulled into the tree). +- [ ] Prefer workspace deps; avoid vendoring a license-hostile SDK just to wrap HTTP. + +### B7. Tests that must exist + +#### Sources + +- [ ] Four canonical state tests (restore / no-state / invalid-state / + round-trip), plus two ACK/NACK tests (`given_ack_when_batch_is_staged_ + should_commit_candidate_state`, `given_nack_when_batch_is_staged_ + should_keep_committed_state`) if `on_batch_result` is overridden - six + total. Copy `sources/random_source/src/lib.rs::tests`. A source relying + on the SDK's default no-op `on_batch_result` (no staged work) may skip + the ACK/NACK pair. + +#### Any external backend plugin + +- [ ] At least one real-infra integration test under + `core/integration/tests/connectors//` with `#[iggy_harness]` + + `testcontainers-modules` (or `wiremock` for pure HTTP). +- [ ] No false-green mocks that diverge from real backend semantics + (Decimal, COPY, PK enforcement, etc.). + +### B8. Config validation timing + +- [ ] Structural validation + unknown enum rejection in `new()` / + `open()` — not on first `poll()`/`consume()` after sleep. +- [ ] Connectivity check in `open()`; fail with `Error::InitError`. +- [ ] Invalid restored state: start fresh + `warn!`, but do **not** silently + re-emit an entire index without calling that out in README. +- [ ] Config flag combos that no-op should `warn!` or `Err`, not silently ignore. + +## High-frequency convention nits + +These are "cheap" but burn full review rounds when missed. + +### C1. Copy the closest exemplar + +- [ ] File layout, log labels, error mapping, and test structure match the + nearest sibling (`postgres_*`, `http_sink`, `elasticsearch_*`, …). +- [ ] Do not invent new names for existing knobs. + +### C2. Config knob name canon + +| Concept | Canonical field | Notes | +| ------- | --------------- | ----- | +| Request timeout | `timeout` | Not `request_timeout` | +| Retry attempts | `max_retries` | Total attempts, default 3 | +| Base backoff | `retry_delay` | humantime `Option` | +| Backoff ceiling | `max_retry_delay` | Not `retry_max_delay` | +| Poll cadence (sources) | `poll_interval` | humantime; sleep first | +| Plugin verbosity | `verbose_logging` | Mirror runtime `verbose` | +| Credentials | `connection_string` / `api_key` / … | Always `SecretString` | + +- [ ] Durations: `Option` + `humantime::Duration` in `new()`; fall back + with `warn!`, never panic. Workspace `humantime` — not a pinned + `humantime-serde`. +- [ ] New fields: `Option` + `#[serde(default)]` where needed. +- [ ] Prefer `#[serde(deny_unknown_fields)]` on plugin config so typo’d knobs + fail loud. + +### C3. Crate / path / docs checklist + +- [ ] `[lib] crate-type = ["cdylib", "lib"]`. +- [ ] Example TOML plugin path uses `../../target/release/lib…` like siblings. +- [ ] Row added to `sinks/README.md` or `sources/README.md`. +- [ ] Sample under `runtime/example_config/connectors/`. +- [ ] README defaults **byte-equal** to consts in code (diff them). +- [ ] No links to non-existent docs. + +### C4. Hot path + +- [ ] No `payload.clone().try_to_bytes()` — use `try_to_bytes(&self)`. +- [ ] `Vec::with_capacity(n)` for per-batch buffers. +- [ ] No `std::sync::Mutex` across `.await`; use `tokio::sync::Mutex`. +- [ ] `&self` on `consume` / `poll` (interior mutability only). +- [ ] No `tokio::spawn` inside plugin code. +- [ ] No `unwrap()`/`expect()` on external I/O outside tests. +- [ ] No eager `format!` around tracing args. + +### C5. Containers / fixtures + +- [ ] Testcontainers named `iggy-test-*` via `fixtures::unique_container_name` + (or fixed `iggy-test-` for reuse fixtures). +- [ ] Custom Docker networks cleaned up. + +## Delivery semantics (document honestly) + +Every new connector README must answer in one short paragraph: + +1. **What happens on transient failure?** (retry N times, then Err) +2. **What happens on permanent failure?** (drop/skip vs fail batch) +3. **What is the duplication window?** (at-least-once because state saves + after Iggy send; or at-most-once if upstream ACK precedes send — say so) +4. **What is the dedup key?** (or "none — duplicates possible") + +If the answer is hand-wavy, the PR is not ready. + +## PR / CI hygiene + +- [ ] Conventional commit: `feat(connectors): …` / `fix(connectors): …`. +- [ ] PR template filled (motivation, linked issue). +- [ ] `cargo fmt --all` + `cargo sort --no-format --workspace` + + `cargo clippy -p --all-targets -- -D warnings` + + `cargo test -p ` green locally. +- [ ] Minimal `Cargo.lock` delta — no unrelated dependency churn. +- [ ] Do not modify unrelated Java/Python/foreign SDK trees in a connectors PR. +- [ ] Mark ready for review only after the above; stale-bot closes waiting PRs. + +## Pre-flight author checklist + +Paste into the PR description (or run mentally before `/ready`): + +```text +[ ] SecretString on all credentials; no secret logs/state +[ ] consume/poll never returns Ok(()) after a failed batch that should retry +[ ] Transient vs permanent errors mapped (no Display substring matching) +[ ] Stable message / dedup IDs (no random UUID per emit) +[ ] README delivery semantics paragraph present and honest +[ ] README defaults match code consts +[ ] deny_unknown_fields on plugin config +[ ] Canonical knob names (timeout, max_retries, retry_delay, poll_interval) +[ ] Sources: 4 state tests (+2 ACK/NACK if on_batch_result overridden); sleep-first poll; state staged in poll, committed only in on_batch_result +[ ] External backend: real-infra integration test (not a lying mock) +[ ] example_config + sinks/sources README row +[ ] fmt / sort --no-format / clippy -D warnings / unit tests green +[ ] Cargo.lock churn limited to this crate's deps +``` + +## Evidence base + +Recurring comments mined from connector PRs including (non-exhaustive): +SurrealDB sink (#3453), Meilisearch sink/source (#3497/#3498), OpenSearch +source (#3515), Quickwit convention (#3523), MySQL source (#3568), JDBC +source (#3588), Doris retry (#3574), Redshift sink (#3654), Airflow trigger +(#3716), Fluss sink (#3782). Highest-density themes: delivery/offset +semantics, idempotency IDs, transient/permanent mapping, config-name drift, +secrets, README/code drift, false-green tests, CI/lockfile hygiene. + +--- + +Discussion / help: see [AGENTS.md](../../../AGENTS.md#discussion-and-support). diff --git a/.claude/skills/connector-runtime/SKILL.md b/.claude/skills/connector-runtime/SKILL.md index cee028180d..ce194c08e6 100644 --- a/.claude/skills/connector-runtime/SKILL.md +++ b/.claude/skills/connector-runtime/SKILL.md @@ -211,7 +211,7 @@ Fatal errors propagate to `main` and exit. Per-connector / per-message errors ar All families labeled by `connector_key` + `connector_type` (histogram adds `stage`): -- **Counters**: `iggy_connector_messages_{produced,sent,consumed,processed,filtered}_total` and `iggy_connector_errors_total`. These are the *rendered* names; each is registered without the `_total`, which the OpenMetrics encoder appends. +- **Counters**: `iggy_connector_messages_{produced,sent,consumed,processed,filtered,errors}_total`. - `messages_filtered_total` - intentional drops via transform `Ok(None)`. - `errors_total` - unexpected drops (decode/encode/build failure, missing field, ...) + batch-level failures. - **Histograms**: `iggy_connector_stage_duration_seconds{stage}` (snake_case stage labels - `prepare`, `ffi`, `decode`, `iggy_send`, `state_save`, `total`). Buckets `STAGE_BUCKETS_SECONDS`. Always populated regardless of any flag. Scraped at `/metrics` when `[http.metrics] enabled = true`. @@ -221,7 +221,7 @@ All families labeled by `connector_key` + `connector_type` (histogram adds `stag When adding a metric: -- Add family to `Metrics` struct + `init`, register with name + help text. Never end a `Counter` family's registered name in `_total`: the encoder appends it and the series renders `_total_total`. Gauges get no suffix, so a gauge name may end in `_total` literally. +- Add family to `Metrics` struct + `init`, register with name + help text. - New label sets define `EncodeLabelSet` struct + label enum (hand-impl `EncodeLabelValue` for snake_case values - the derive emits PascalCase). - Histograms: pass `fn() -> Histogram` to `Family::new_with_constructor`. - Add unit tests under `mod tests` with `given_*_when_*_should_*` BDD names. diff --git a/.claude/skills/connector-sink/SKILL.md b/.claude/skills/connector-sink/SKILL.md index 80533036b6..9d40b08323 100644 --- a/.claude/skills/connector-sink/SKILL.md +++ b/.claude/skills/connector-sink/SKILL.md @@ -34,7 +34,8 @@ for getting them to the external system reliably and efficiently. ## Quick reference -- Skeleton: [TEMPLATE.md](TEMPLATE.md) (load on demand when authoring). +- Skeleton: [TEMPLATE.md](TEMPLATE.md) (fill-in-the-blank kit — implement only `TODO(ConnectorDeveloper)`). +- PR pre-flight: [connector-pr-review](../connector-pr-review/SKILL.md). - Exemplars: `stdout_sink` (minimal), `postgres_sink` (DB + transient detection), `http_sink` (validation, batch modes, retry middleware), `mongodb_sink` (atomic counters), `elasticsearch_sink` / `iceberg_sink` (backend-specific idioms). ## Hard rules diff --git a/.claude/skills/connector-sink/TEMPLATE.md b/.claude/skills/connector-sink/TEMPLATE.md index 32a9f97270..19a6956309 100644 --- a/.claude/skills/connector-sink/TEMPLATE.md +++ b/.claude/skills/connector-sink/TEMPLATE.md @@ -1,100 +1,194 @@ -# Sink plugin skeleton +# Sink plugin fill-in-the-blank kit -Boilerplate for a new `core/connectors/sinks/_sink/`. Adapt the -`MySink` / `Client` types to the backend driver you're integrating. +Copy this kit into `core/connectors/sinks/_sink/`. The scaffolding +covers config, secrets, retry, error classification, batching, logging, +and unit-test shape. **You only implement the marked `TODO(ConnectorDeveloper)` +sections:** build a client from the connection string, and push one +batch. + +Also read [SKILL.md](SKILL.md) and pre-flight with +[connector-pr-review](../connector-pr-review/SKILL.md) before `/ready`. + +## Files to create + +```text +core/connectors/sinks/_sink/ +├── Cargo.toml +├── README.md +├── config.toml +└── src/lib.rs +``` + +Add a workspace member, a row in `sinks/README.md`, and a sample under +`runtime/example_config/connectors/`. + +--- ## Cargo.toml ```toml -# Apache 2.0 header (copy verbatim from any existing sink Cargo.toml) +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + [package] name = "iggy_connector__sink" -version = "0.4.1-edge.1" # match the version most other sinks use +version = "0.4.1-edge.1" edition = "2024" license = "Apache-2.0" publish = false -# ...keywords, description, repository, homepage all identical to existing sinks +description = "Apache Iggy sink connector" +repository = "https://github.com/apache/iggy" +homepage = "https://iggy.apache.org" [package.metadata.cargo-machete] -ignored = ["dashmap", "once_cell"] # used by sink_connector! macro +ignored = ["dashmap", "once_cell"] [lib] -crate-type = ["cdylib", "lib"] # cdylib = runtime-loadable; lib = unit tests +crate-type = ["cdylib", "lib"] [dependencies] -async-trait = { workspace = true } -dashmap = { workspace = true } +async-trait = { workspace = true } +dashmap = { workspace = true } +humantime = { workspace = true } +iggy_common = { workspace = true } iggy_connector_sdk = { workspace = true } -once_cell = { workspace = true } -serde = { workspace = true } -tokio = { workspace = true } -tracing = { workspace = true } -# + your client crate (reqwest, sqlx, mongodb, ...) +once_cell = { workspace = true } +secrecy = { workspace = true } +serde = { workspace = true } +tokio = { workspace = true } +tracing = { workspace = true } +# TODO(ConnectorDeveloper): add your client crate as a workspace dependency +``` + +Run `cargo sort --no-format --workspace` after edits. Keep `Cargo.lock` +churn limited to this crate's deps. + +--- + +## config.toml (example) + +```toml +# Plugin path matches sibling sinks (relative to connectors runtime cwd). +path = "../../target/release/libiggy_connector__sink" + +[[sinks]] +key = "" +enabled = true +# path is also set via IGGY_CONNECTORS_SINK__PATH in integration tests + +[sinks..plugin_config] +# Never commit real secrets. Use env overrides in tests/ops. +connection_string = "scheme://user:pass@host:port/db" +batch_size = 100 +max_retries = 3 +retry_delay = "500ms" +verbose_logging = false ``` -Run `cargo sort --no-format --workspace` after edits. +--- ## src/lib.rs -Code reads top to bottom. Public types first, then `impl`, then private -helpers. +Replace `` / `` and implement only the `TODO(ConnectorDeveloper)` blocks. ```rust -/* Licensed to the Apache Software Foundation (ASF) under one - * or more contributor license agreements... */ // full Apache header +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. use async_trait::async_trait; +use humantime::Duration as HumanDuration; use iggy_connector_sdk::{ ConsumedMessage, Error, MessagesMetadata, Sink, TopicMetadata, sink_connector, }; +use secrecy::{ExposeSecret, SecretString}; use serde::{Deserialize, Serialize}; use std::str::FromStr; use std::time::Duration; use tokio::sync::Mutex; use tracing::{debug, error, info, warn}; -sink_connector!(MySink); // generates FFI symbols + version export +sink_connector!(NameSink); -const CONNECTOR_NAME: &str = "My sink"; +const CONNECTOR_NAME: &str = "Name sink"; +const DEFAULT_BATCH_SIZE: u32 = 100; +const DEFAULT_MAX_RETRIES: u32 = 3; // total attempts +const DEFAULT_RETRY_DELAY: &str = "500ms"; + +/// Backend client. Replace with the real driver type. +struct BackendClient { + // TODO(ConnectorDeveloper): fields +} #[derive(Debug, Serialize, Deserialize)] -pub struct MySinkConfig { - pub endpoint: String, +#[serde(deny_unknown_fields)] +pub struct NameSinkConfig { + #[serde(serialize_with = "iggy_common::serde_secret::serialize_secret")] + pub connection_string: SecretString, pub batch_size: Option, pub max_retries: Option, - pub retry_delay: Option, // humantime, e.g. "500ms" - pub verbose_logging: Option, // mirror runtime's `verbose` flag - // Every runtime-tunable field is Option; defaults applied in new() + pub retry_delay: Option, + pub verbose_logging: Option, + // TODO(ConnectorDeveloper): optional non-secret knobs (table, index, batch_mode, ...) } #[derive(Debug)] -pub struct MySink { +pub struct NameSink { id: u32, - config: MySinkConfig, + config: NameSinkConfig, batch_size: usize, max_retries: u32, retry_delay: Duration, verbose: bool, - client: Option, + client: Option, state: Mutex, } -#[derive(Debug)] +#[derive(Debug, Default)] struct State { messages_processed: u64, errors: u64, } -impl MySink { - pub fn new(id: u32, config: MySinkConfig) -> Self { - let batch_size = config.batch_size.unwrap_or(100) as usize; - let max_retries = config.max_retries.unwrap_or(3); +impl NameSink { + pub fn new(id: u32, config: NameSinkConfig) -> Self { + let batch_size = config.batch_size.unwrap_or(DEFAULT_BATCH_SIZE) as usize; + let max_retries = config.max_retries.unwrap_or(DEFAULT_MAX_RETRIES); let retry_delay = config .retry_delay .as_deref() - .and_then(|raw| humantime::Duration::from_str(raw).ok().map(|d| *d)) + .and_then(|raw| HumanDuration::from_str(raw).ok().map(|d| *d)) .unwrap_or_else(|| { - warn!("Invalid retry_delay for {CONNECTOR_NAME} ID: {id}, defaulting to 500ms"); + warn!( + "Invalid retry_delay for {CONNECTOR_NAME} ID: {id}, defaulting to {DEFAULT_RETRY_DELAY}" + ); Duration::from_millis(500) }); let verbose = config.verbose_logging.unwrap_or(false); @@ -106,24 +200,49 @@ impl MySink { retry_delay, verbose, client: None, - state: Mutex::new(State { messages_processed: 0, errors: 0 }), + state: Mutex::new(State::default()), + } + } + + async fn send_batch_with_retry( + &self, + client: &BackendClient, + topic_metadata: &TopicMetadata, + batch: &[ConsumedMessage], + ) -> Result<(), Error> { + let mut attempt = 0u32; + loop { + attempt += 1; + match push_batch(client, topic_metadata, batch).await { + Ok(()) => return Ok(()), + Err(error) if is_permanent(&error) => return Err(error), + Err(error) if attempt >= self.max_retries => return Err(error), + Err(error) => { + warn!( + "{CONNECTOR_NAME} ID: {} retry {attempt}/{}: {error}", + self.id, self.max_retries + ); + tokio::time::sleep(self.retry_delay.saturating_mul(attempt)).await; + } + } } } } #[async_trait] -impl Sink for MySink { +impl Sink for NameSink { async fn open(&mut self) -> Result<(), Error> { + // Structural validation belongs here / in new(), not in consume(). let client = build_client(&self.config) .await .map_err(|e| Error::InitError(format!("client build failed: {e}")))?; - client.ping() + ping(&client) .await .map_err(|e| Error::InitError(format!("connectivity check failed: {e}")))?; self.client = Some(client); info!( - "Opened {CONNECTOR_NAME} connector ID: {}, endpoint: {}", - self.id, self.config.endpoint + "Opened {CONNECTOR_NAME} connector ID: {}, endpoint: ", + self.id ); Ok(()) } @@ -140,28 +259,44 @@ impl Sink for MySink { if self.verbose { info!( - "{CONNECTOR_NAME} ID: {} consuming {} messages from stream: {}, topic: {}, offset: {}, current_offset: {}", - self.id, messages.len(), topic_metadata.stream, topic_metadata.topic, - messages_metadata.partition_id, messages_metadata.current_offset + "{CONNECTOR_NAME} ID: {} consuming {} messages from stream: {}, topic: {}, partition_id: {}, current_offset: {}", + self.id, + messages.len(), + topic_metadata.stream, + topic_metadata.topic, + messages_metadata.partition_id, + messages_metadata.current_offset ); } else { debug!( "{CONNECTOR_NAME} ID: {} consuming {} messages", - self.id, messages.len() + self.id, + messages.len() ); } + // Never swallow a failed batch as Ok(()) — offsets may still advance. let mut last_err: Option = None; for batch in messages.chunks(self.batch_size) { - match self.send_batch(client, batch).await { - Ok(()) => { /* counter */ } + match self + .send_batch_with_retry(client, topic_metadata, batch) + .await + { + Ok(()) => { + let mut state = self.state.lock().await; + state.messages_processed += batch.len() as u64; + } Err(Error::PermanentHttpError(message)) => { error!( "{CONNECTOR_NAME} ID: {} dropping batch (permanent): {message}", self.id ); + let mut state = self.state.lock().await; + state.errors += 1; } Err(error) => { + let mut state = self.state.lock().await; + state.errors += 1; last_err = Some(error); } } @@ -173,9 +308,8 @@ impl Sink for MySink { } async fn close(&mut self) -> Result<(), Error> { - // sqlx pools have `.close().await`; reqwest/mongodb/elasticsearch just drop. if let Some(client) = self.client.take() { - let _ = client; // or `client.close().await;` for sqlx + close_client(client).await; } let state = self.state.lock().await; info!( @@ -186,5 +320,98 @@ impl Sink for MySink { } } -async fn build_client(config: &MySinkConfig) -> Result { /* ... */ } +// ─── Backend surface: implement these ─────────────────────────────────────── + +/// TODO(ConnectorDeveloper): parse `config.connection_string.expose_secret()` and build the client. +async fn build_client(config: &NameSinkConfig) -> Result { + let _secret = config.connection_string.expose_secret(); + Err("TODO(ConnectorDeveloper): build_client".into()) +} + +/// TODO(ConnectorDeveloper): cheap connectivity probe used from open(). +async fn ping(_client: &BackendClient) -> Result<(), String> { + Ok(()) +} + +/// TODO(ConnectorDeveloper): push one batch. Prefer a stable dedup key from +/// `stream:topic:partition:message_id` (or backend natural key). +/// Use `message.payload.try_to_bytes()` — do not clone Payload::Json. +async fn push_batch( + _client: &BackendClient, + _topic_metadata: &TopicMetadata, + _batch: &[ConsumedMessage], +) -> Result<(), Error> { + Err(Error::InitError("TODO(ConnectorDeveloper): push_batch".into())) +} + +/// Map driver errors. Never classify via `err.to_string()` substrings. +fn is_permanent(error: &Error) -> bool { + matches!( + error, + Error::PermanentHttpError(_) | Error::SchemaMismatch(_) | Error::InvalidRecordValue(_) + ) +} + +async fn close_client(_client: BackendClient) { + // sqlx: pool.close().await; most HTTP clients: drop +} + +#[cfg(test)] +mod tests { + use super::*; + + fn test_config() -> NameSinkConfig { + NameSinkConfig { + connection_string: SecretString::from("scheme://localhost/db"), + batch_size: Some(10), + max_retries: Some(2), + retry_delay: Some("10ms".into()), + verbose_logging: Some(false), + } + } + + #[test] + fn given_defaults_should_apply_consts() { + let sink = NameSink::new( + 1, + NameSinkConfig { + connection_string: SecretString::from("scheme://localhost/db"), + batch_size: None, + max_retries: None, + retry_delay: None, + verbose_logging: None, + }, + ); + assert_eq!(sink.batch_size, DEFAULT_BATCH_SIZE as usize); + assert_eq!(sink.max_retries, DEFAULT_MAX_RETRIES); + } + + #[test] + fn given_invalid_retry_delay_should_fall_back_to_default() { + let mut config = test_config(); + config.retry_delay = Some("not-a-duration".into()); + let sink = NameSink::new(1, config); + assert_eq!(sink.retry_delay, Duration::from_millis(500)); + } +} ``` + +--- + +## README.md (required paragraphs) + +Your README must include a **Delivery semantics** section answering: + +1. Transient failure behavior (retry N times, then `Err`) +2. Permanent failure behavior (drop/skip vs fail) +3. Duplication window (usually at-least-once) +4. Dedup key (or "none — duplicates possible") + +Diff README defaults against the `DEFAULT_*` consts before opening the PR. + +--- + +## Before `/ready` + +Run the pre-flight checklist in +[connector-pr-review](../connector-pr-review/SKILL.md#pre-flight-author-checklist). diff --git a/.claude/skills/connector-source/SKILL.md b/.claude/skills/connector-source/SKILL.md index 5627b1812d..36037b1062 100644 --- a/.claude/skills/connector-source/SKILL.md +++ b/.claude/skills/connector-source/SKILL.md @@ -9,8 +9,13 @@ A **source** is a Rust `cdylib` that implements `iggy_connector_sdk::Source` and exposes FFI symbols via the `source_connector!` macro. The runtime calls `poll()` in a loop, applies transforms, encodes via the configured `Schema`, sends to -Apache Iggy, and persists the returned `ConnectorState` after every -successful send. +Apache Iggy, and persists the state `poll()` returned - but only after +the send succeeds. Only one batch is ever in flight: the runtime does +not call `poll()` again until it has reported `Ack` or `Nack` for the +current one via `on_batch_result()` (source batch acknowledgment, see PR #3855 +for the full contract). See +[State persistence](#state-persistence-stage-in-poll-commit-in-on_batch_result) +below. > **Universal connector rules** (SecretString, benchmark, verbose flag, drop accounting, filter contract, exemplar patterns) live in > [connectors-overview](../connectors-overview/SKILL.md). This skill @@ -27,14 +32,15 @@ successful send. ## STOP and ask the user before -- Changing the SDK trait surface (`Source::open` / `poll` / `close`) - that's an SDK change. +- Changing the SDK trait surface (`Source::open` / `poll` / `on_batch_result` / `close`) - that's an SDK change, and `poll`/`on_batch_result` are also an FFI change (`iggy_source_handle_v2`, `iggy_source_batch_result` - breaks every pre-built plugin `.so`). - Adding a long-running side task in the plugin - the runtime owns lifecycle. orphans survive `close()`. - Persisting unbounded state - `State` is rewritten every batch. - Adding a source that requires authoritative offsets external to Apache Iggy without coordinating retention. ## Quick reference -- Skeleton: [TEMPLATE.md](TEMPLATE.md) (load on demand). +- Skeleton: [TEMPLATE.md](TEMPLATE.md) (fill-in-the-blank kit — implement only `TODO(ConnectorDeveloper)`). +- PR pre-flight: [connector-pr-review](../connector-pr-review/SKILL.md). - Exemplars: `random_source` (minimal + canonical state tests), `postgres_source` (cursor / delete-after-read / processed-column modes, restart-survives-state tests), `elasticsearch_source` (scroll cursor), `influxdb_source` (time-series scan). ## Hard rules @@ -57,12 +63,61 @@ let persisted = { // brief write }; ``` -### State persistence +### State persistence: stage in `poll()`, commit in `on_batch_result()` + +Source connectors use a one-in-flight-batch ACK/NACK contract (#3855) +between the plugin and the runtime: + +1. `poll()` returns messages and *candidate* state without committing + cursor changes or destructive operations (deletes, mark-processed). +2. The runtime sends the batch to Apache Iggy and waits for the + producer result. +3. After a successful send, the runtime persists the candidate state + to `{state_path}/source_{key}.state`. +4. The runtime calls `on_batch_result(SourceBatchResult::Ack)`. A send + or state-save failure calls `on_batch_result(SourceBatchResult::Nack)` + instead. +5. `on_batch_result()` commits or discards the plugin's staged work + before the next `poll()` starts. The SDK allows only one batch in + flight - it will not call `poll()` again until `on_batch_result()` + for the current batch has returned. + +Canonical pattern (`sources/random_source/src/lib.rs`): + +```rust +pending_state: Mutex>, // staged, not yet committed + +async fn poll(&self) -> Result { + // ... fetch ... + let candidate_state = State { cursor: next_cursor }; + *self.pending_state.lock().await = Some(candidate_state.clone()); + Ok(ProducedMessages { + schema: Schema::Json, + messages, + state: Some(ConnectorState::serialize(&candidate_state, NAME, self.id)?), + }) +} + +async fn on_batch_result(&self, result: SourceBatchResult) -> Result<(), Error> { + let candidate_state = self.pending_state.lock().await.take(); + if result == SourceBatchResult::Ack + && let Some(candidate_state) = candidate_state + { + *self.state.lock().await = candidate_state; + } + // Nack: drop candidate_state, committed self.state is untouched - + // the same range is polled again. + Ok(()) +} +``` - `ConnectorState` is `Vec` via MessagePack (`rmp_serde`). Use `ConnectorState::serialize(&state, NAME, id)` + `ConnectorState::deserialize::(NAME, id)`. Both return `Option` and log on failure (non-fatal). -- Runtime saves to `{state_path}/source_{key}.state` only after a successful Iggy send. Between `poll()` returning and the runtime persisting the save, a crash leaves the same cursor for the next poll - downstream must tolerate at-least-once. -- **Always return state in every `ProducedMessages`**, including empty polls. Empty results still need to advance watermarks (timestamp sources) or affirm "nothing new." +- **The default `on_batch_result` is a no-op.** Only override it - and only then does staging via `pending_state` matter - if `poll()` advances a cursor or performs destructive work (delete-after-read, mark-processed). A source with no staged work (e.g. a pure generator) can rely on the default. +- Returning `Err` from `on_batch_result` **stops the SDK from polling further** - a failed rollback must not be allowed to silently advance to the next batch. +- **Always return state in every `ProducedMessages`**, including empty polls that made progress. Return `state: None` for an empty poll that made *no* progress - this avoids an unnecessary state write and cannot persist state left over from a failed batch. - Keep `State` small - rewritten every batch. No unbounded vecs. +- NACK handling must discard staged cursor changes and staged delete/mark operations so polling redelivers the batch. The SDK retries NACKed batches with capped exponential backoff and stops the source after repeated consecutive NACKs. +- Crash recovery is at-least-once at every point except after the plugin has processed the ACK (see the SDK README's crash-point table, `core/connectors/sdk/README.md#source-delivery-acknowledgment`, for the full breakdown). ### Sleep first @@ -86,18 +141,20 @@ Match `ProducedMessages.schema` to the bytes in `messages[i].payload`: ### Concurrency - Runtime spawns ONE `poll()` task per source. No concurrent `poll()`. +- Only one batch is ever in flight: the SDK does not call `poll()` again until `on_batch_result()` has returned for the previous batch (up to a 30s result timeout, after which the SDK treats it as a Nack). - Don't spawn your own long-running Tokio tasks - runtime owns lifecycle. ### Errors -| Scenario | Variant | -| ------------------------------------------- | ------------------------------------------------- | -| Bad config in `new()`/`open()` | `Error::InitError` | -| Cannot reach external system at startup | `Error::InitError` or `Error::Connection` | -| Transient fetch failure (retry-worthy) | `Error::Connection` or `Error::HttpRequestFailed` | -| Permanent fetch failure (auth, schema gone) | `Error::PermanentHttpError` | -| Row failed to serialize | `Error::Serialization(...)` | -| State serialization failed | log + skip (non-fatal) | +| Scenario | Variant | +| --------------------------------------------------- | ------------------------------------------------- | +| Bad config in `new()`/`open()` | `Error::InitError` | +| Cannot reach external system at startup | `Error::InitError` or `Error::Connection` | +| Transient fetch failure (retry-worthy) | `Error::Connection` or `Error::HttpRequestFailed` | +| Permanent fetch failure (auth, schema gone) | `Error::PermanentHttpError` | +| Row failed to serialize | `Error::Serialization(...)` | +| State serialization failed | log + skip (non-fatal) | +| `on_batch_result()` failed to roll back staged work | `Err` - stops the SDK from polling further | Returning `Err` from `poll()` is only logged by the SDK's FFI bridge (`sdk/src/source.rs::handle_messages`) - the loop continues, the next @@ -127,15 +184,20 @@ Iggy consumer-loop labels use literal API names (`offset=`, `current_offset=`). 1. `async fn poll(&mut self)` - won't compile. Use `&self` + `Mutex`. 2. Holding `state.lock()` across the fetch I/O - blocks `close()`, causes shutdown timeouts. 3. Forgetting to sleep - 100% CPU on idle source. -4. Returning state only on success - state should advance on empty polls too. -5. Unbounded data in `State` - rewritten every batch. keep O(constant). -6. `std::sync::Mutex` - blocks the executor. Use `tokio::sync::Mutex`. -7. Not setting `ProducedMessage.id` when a stable ID exists - loses idempotency. -8. Spawning side tasks - the runtime owns the scheduler. +4. Returning `state: None` for an empty poll that *did* make progress (e.g. advanced a watermark) - only a no-progress empty poll should return `None`. +5. Committing a cursor or destructive work (delete/mark-processed) directly in `poll()` instead of staging it and applying it in `on_batch_result()` on `Ack` - a Nack (send or state-save failure) has nothing to discard, and the batch is redelivered against already-mutated state. +6. Unbounded data in `State` - rewritten every batch. keep O(constant). +7. `std::sync::Mutex` - blocks the executor. Use `tokio::sync::Mutex`. +8. Not setting `ProducedMessage.id` when a stable ID exists - loses idempotency. +9. Spawning side tasks - the runtime owns the scheduler. ## Tests -Mandatory four canonical source state tests (see [connector-testing](../connector-testing/SKILL.md) for the full pattern). Copy from `sources/random_source/src/lib.rs::tests`. Plus config defaults, payload building, schema selection. +Mandatory six canonical source tests (see [connector-testing](../connector-testing/SKILL.md) for the full pattern): the four +state tests (restore / no-state / invalid-state / round-trip) plus `given_ack_when_batch_is_staged_should_commit_candidate_state` +and `given_nack_when_batch_is_staged_should_keep_committed_state`. Copy from `sources/random_source/src/lib.rs::tests`. Plus +config defaults, payload building, schema selection. A source relying on the default no-op `on_batch_result` (no staged work) +may skip the ack/nack pair. Integration tests under `core/integration/tests/connectors//` for any source backed by external infra. Use `#[iggy_harness]` + a `TestFixture` backed by `testcontainers-modules`. Reference: `core/integration/tests/connectors/postgres/postgres_source.rs` (multi-mode tests) + `restart.rs` (state survives restart). diff --git a/.claude/skills/connector-source/TEMPLATE.md b/.claude/skills/connector-source/TEMPLATE.md index 7a27953654..c5db2183b7 100644 --- a/.claude/skills/connector-source/TEMPLATE.md +++ b/.claude/skills/connector-source/TEMPLATE.md @@ -1,25 +1,95 @@ -# Source plugin skeleton +# Source plugin fill-in-the-blank kit -Boilerplate for a new `core/connectors/sources/_source/`. Adapt -the `MySource` / `Client` / row types to the backend driver. +Copy this kit into `core/connectors/sources/_source/`. The +scaffolding covers config, secrets, sleep-first poll, lock discipline, +staged/committed state ser·de (ACK/NACK batch acknowledgment, #3855), +retry classification, logging, and the six canonical state tests. +**You only implement the marked `TODO(ConnectorDeveloper)` sections:** build a +client from the connection string, and fetch the next batch (advancing +a cursor). -`Cargo.toml` is identical to a sink's (see -[connector-sink/TEMPLATE.md](../connector-sink/TEMPLATE.md)) - only -the crate name suffix changes (`iggy_connector__source`) and the -upstream client dep. +Also read [SKILL.md](SKILL.md) and pre-flight with +[connector-pr-review](../connector-pr-review/SKILL.md) before `/ready`. + +## Files to create + +```text +core/connectors/sources/_source/ +├── Cargo.toml +├── README.md +├── config.toml +└── src/lib.rs +``` + +Add a workspace member, a row in `sources/README.md`, and a sample under +`runtime/example_config/connectors/`. + +--- + +## Cargo.toml + +Same shape as the sink kit (`cdylib` + `lib`, workspace deps, Apache +header). Only the package name suffix changes: + +```toml +name = "iggy_connector__source" +# ... identical metadata / machete ignored / crate-type ... +# TODO(ConnectorDeveloper): add your client crate as a workspace dependency +``` + +--- + +## config.toml (example) + +```toml +path = "../../target/release/libiggy_connector__source" + +[[sources]] +key = "" +enabled = true + +[sources..plugin_config] +connection_string = "scheme://user:pass@host:port/db" +poll_interval = "5s" +batch_size = 100 +max_retries = 3 +retry_delay = "500ms" +verbose_logging = false +``` + +Defaults in this file must match `DEFAULT_*` consts in code. + +--- ## src/lib.rs -Code reads top to bottom. Public types first, `impl Source` after, -helpers below. +Replace `` / `` and implement only the `TODO(ConnectorDeveloper)` blocks. ```rust -/* Apache 2.0 header */ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. use async_trait::async_trait; +use humantime::Duration as HumanDuration; use iggy_connector_sdk::{ - ConnectorState, Error, ProducedMessage, ProducedMessages, Schema, Source, source_connector, + ConnectorState, Error, ProducedMessage, ProducedMessages, Schema, Source, + source::SourceBatchResult, source_connector, }; +use secrecy::{ExposeSecret, SecretString}; use serde::{Deserialize, Serialize}; use std::str::FromStr; use std::time::Duration; @@ -27,135 +97,224 @@ use tokio::sync::Mutex; use tokio::time::sleep; use tracing::{debug, error, info, warn}; -source_connector!(MySource); +source_connector!(NameSource); -const CONNECTOR_NAME: &str = "My source"; +const CONNECTOR_NAME: &str = "Name source"; +const DEFAULT_POLL_INTERVAL: &str = "5s"; +const DEFAULT_BATCH_SIZE: u32 = 100; +const DEFAULT_MAX_RETRIES: u32 = 3; // total attempts +const DEFAULT_RETRY_DELAY: &str = "500ms"; + +struct BackendClient { + // TODO(ConnectorDeveloper): fields +} #[derive(Debug, Serialize, Deserialize)] -pub struct MySourceConfig { - pub endpoint: String, - pub poll_interval: Option, // humantime, "10s" default +#[serde(deny_unknown_fields)] +pub struct NameSourceConfig { + #[serde(serialize_with = "iggy_common::serde_secret::serialize_secret")] + pub connection_string: SecretString, + pub poll_interval: Option, pub batch_size: Option, pub max_retries: Option, pub retry_delay: Option, pub verbose_logging: Option, + // TODO(ConnectorDeveloper): optional non-secret knobs (query, table, index, ...) } -#[derive(Debug, Serialize, Deserialize)] +#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)] struct State { - cursor: Option, // WAL LSN, scroll id, timestamp, ... - last_offset: u64, + /// Opaque backend cursor (LSN, scroll id, timestamp, PK, ...). Keep O(1). + cursor: Option, messages_produced: u64, } #[derive(Debug)] -pub struct MySource { +pub struct NameSource { id: u32, - config: MySourceConfig, + config: NameSourceConfig, poll_interval: Duration, + batch_size: usize, + max_retries: u32, + retry_delay: Duration, verbose: bool, - client: Option, + client: Option, state: Mutex, + /// Staged in `poll()`, committed to `state` on `Ack`, discarded on `Nack` + /// (source batch acknowledgment, #3855). Only one batch is ever staged + /// at a time - the SDK enforces one in-flight batch. + pending_state: Mutex>, +} + +struct FetchedBatch { + messages: Vec, + /// Next cursor, staged via `pending_state` and committed only after + /// `on_batch_result(SourceBatchResult::Ack)`. + next_cursor: Option, } -impl MySource { - pub fn new(id: u32, config: MySourceConfig, state: Option) -> Self { - let raw_interval = config.poll_interval.clone().unwrap_or_else(|| "10s".into()); - let poll_interval = humantime::Duration::from_str(&raw_interval) +impl NameSource { + pub fn new(id: u32, config: NameSourceConfig, state: Option) -> Self { + let raw_interval = config + .poll_interval + .clone() + .unwrap_or_else(|| DEFAULT_POLL_INTERVAL.into()); + let poll_interval = HumanDuration::from_str(&raw_interval) .map(|d| *d) .unwrap_or_else(|_| { - warn!("Invalid poll_interval for {CONNECTOR_NAME} ID: {id}, defaulting to 10s"); - Duration::from_secs(10) + warn!( + "Invalid poll_interval for {CONNECTOR_NAME} ID: {id}, defaulting to {DEFAULT_POLL_INTERVAL}" + ); + Duration::from_secs(5) + }); + let batch_size = config.batch_size.unwrap_or(DEFAULT_BATCH_SIZE) as usize; + let max_retries = config.max_retries.unwrap_or(DEFAULT_MAX_RETRIES); + let retry_delay = config + .retry_delay + .as_deref() + .and_then(|raw| HumanDuration::from_str(raw).ok().map(|d| *d)) + .unwrap_or_else(|| { + warn!( + "Invalid retry_delay for {CONNECTOR_NAME} ID: {id}, defaulting to {DEFAULT_RETRY_DELAY}" + ); + Duration::from_millis(500) }); - let verbose = config.verbose_logging.unwrap_or(false); let restored = state .and_then(|s| s.deserialize::(CONNECTOR_NAME, id)) - .inspect(|s| info!( - "Restored state for {CONNECTOR_NAME} ID: {id}, last_offset: {}, cursor: {:?}", - s.last_offset, s.cursor - )); + .inspect(|s| { + info!( + "Restored state for {CONNECTOR_NAME} ID: {id}, cursor: {:?}, messages_produced: {}", + s.cursor, s.messages_produced + ); + }); Self { id, config, poll_interval, + batch_size, + max_retries, + retry_delay, verbose, client: None, state: Mutex::new(restored.unwrap_or(State { cursor: None, - last_offset: 0, messages_produced: 0, })), + pending_state: Mutex::new(None), + } + } + + async fn fetch_with_retry( + &self, + client: &BackendClient, + cursor: Option, + ) -> Result { + let mut attempt = 0u32; + loop { + attempt += 1; + match fetch_batch(client, cursor.as_deref(), self.batch_size).await { + Ok(batch) => return Ok(batch), + Err(error) if is_permanent(&error) => return Err(error), + Err(error) if attempt >= self.max_retries => return Err(error), + Err(error) => { + warn!( + "{CONNECTOR_NAME} ID: {} retry {attempt}/{}: {error}", + self.id, self.max_retries + ); + sleep(self.retry_delay.saturating_mul(attempt)).await; + } + } } } } #[async_trait] -impl Source for MySource { +impl Source for NameSource { async fn open(&mut self) -> Result<(), Error> { + // Validate query/cursor shape here — not on first poll after sleep. let client = build_client(&self.config) .await .map_err(|e| Error::InitError(format!("client build failed: {e}")))?; + ping(&client) + .await + .map_err(|e| Error::InitError(format!("connectivity check failed: {e}")))?; self.client = Some(client); info!( - "Opened {CONNECTOR_NAME} connector ID: {}, endpoint: {}", - self.id, self.config.endpoint + "Opened {CONNECTOR_NAME} connector ID: {}, endpoint: ", + self.id ); Ok(()) } async fn poll(&self) -> Result { - sleep(self.poll_interval).await; // sleep first - backpressure - - let cursor = { self.state.lock().await.cursor.clone() }; // brief read - - let fetched = self.fetch_since(cursor.as_deref()).await?; // no lock held - - let mut messages = Vec::with_capacity(fetched.len()); - let mut next_cursor = None; - for row in fetched { - let payload = simd_json::to_vec(&row).map_err(|e| - Error::Serialization(format!("row serialize: {e}")) - )?; - messages.push(ProducedMessage { - id: Some(row.id as u128), - checksum: None, - timestamp: None, - origin_timestamp: Some(row.created_at_ns), - headers: None, - payload, - }); - next_cursor = Some(row.cursor_value); - } + // Sleep FIRST or an idle source spins the CPU. + sleep(self.poll_interval).await; + + let Some(client) = self.client.as_ref() else { + return Err(Error::InitError("client not initialized".into())); + }; + + // Brief lock read → drop → I/O. Never hold across upstream await. + let cursor = { self.state.lock().await.cursor.clone() }; + + let fetched = self.fetch_with_retry(client, cursor).await?; if self.verbose { info!( - "{CONNECTOR_NAME} ID: {} produced {} messages, next cursor: {:?}", - self.id, messages.len(), next_cursor + "{CONNECTOR_NAME} ID: {} polled {} messages, next_cursor: {:?}", + self.id, + fetched.messages.len(), + fetched.next_cursor + ); + } else { + debug!( + "{CONNECTOR_NAME} ID: {} polled {} messages", + self.id, + fetched.messages.len() ); } - let persisted = { // brief write - let mut state = self.state.lock().await; - state.messages_produced += messages.len() as u64; - if let Some(c) = next_cursor { - state.cursor = Some(c); - } - ConnectorState::serialize(&*state, CONNECTOR_NAME, self.id) + // Stage the candidate state - do NOT commit it to `self.state` here, and do + // NOT delete/mark upstream rows here either. The runtime sends the batch and + // saves this staged state; `on_batch_result()` below commits it (Ack) or + // discards it (Nack) before the next poll() runs. Committing directly in + // poll() would leave nothing to roll back on a Nack. + let messages_produced = self.state.lock().await.messages_produced; + let candidate_state = State { + cursor: fetched.next_cursor, + messages_produced: messages_produced + fetched.messages.len() as u64, }; + let state_bytes = ConnectorState::serialize(&candidate_state, CONNECTOR_NAME, self.id); + *self.pending_state.lock().await = Some(candidate_state); Ok(ProducedMessages { - schema: Schema::Json, - messages, - state: persisted, + schema: Schema::Json, // TODO(ConnectorDeveloper): match actual payload bytes + messages: fetched.messages, + state: state_bytes, }) } + // Commits or discards the batch staged in poll() above. The default + // trait impl is a no-op - override is mandatory whenever poll() stages + // a cursor or destructive work, which this template always does. + async fn on_batch_result(&self, result: SourceBatchResult) -> Result<(), Error> { + let candidate_state = self.pending_state.lock().await.take(); + if result == SourceBatchResult::Ack + && let Some(candidate_state) = candidate_state + { + *self.state.lock().await = candidate_state; + } + // Nack: candidate_state is dropped, self.state (committed) is untouched, + // so the next poll() re-fetches from the same committed cursor. + Ok(()) + } + async fn close(&mut self) -> Result<(), Error> { if let Some(client) = self.client.take() { - let _ = client; // or `client.close().await;` for sqlx pools + close_client(client).await; } let state = self.state.lock().await; info!( @@ -166,5 +325,187 @@ impl Source for MySource { } } -async fn build_client(config: &MySourceConfig) -> Result { /* ... */ } +// ─── Backend surface: implement these ─────────────────────────────────────── + +/// TODO(ConnectorDeveloper): parse `config.connection_string.expose_secret()` and build the client. +async fn build_client(config: &NameSourceConfig) -> Result { + let _secret = config.connection_string.expose_secret(); + Err("TODO(ConnectorDeveloper): build_client".into()) +} + +/// TODO(ConnectorDeveloper): cheap connectivity probe used from open(). +async fn ping(_client: &BackendClient) -> Result<(), String> { + Ok(()) +} + +/// TODO(ConnectorDeveloper): fetch up to `limit` rows after `cursor`. +/// Set `ProducedMessage.id` from a stable natural key (never random UUID). +/// Set `origin_timestamp` when the backend has event time (nanoseconds). +async fn fetch_batch( + _client: &BackendClient, + _cursor: Option<&str>, + _limit: usize, +) -> Result { + Err(Error::InitError("TODO(ConnectorDeveloper): fetch_batch".into())) +} + +fn is_permanent(error: &Error) -> bool { + matches!( + error, + Error::PermanentHttpError(_) | Error::SchemaMismatch(_) | Error::InvalidConfigValue(_) + ) +} + +async fn close_client(_client: BackendClient) {} + +#[cfg(test)] +mod tests { + use super::*; + + fn test_config() -> NameSourceConfig { + NameSourceConfig { + connection_string: SecretString::from("scheme://localhost/db"), + poll_interval: Some("100ms".into()), + batch_size: Some(10), + max_retries: Some(2), + retry_delay: Some("10ms".into()), + verbose_logging: Some(false), + } + } + + #[test] + fn given_persisted_state_should_restore_cursor() { + let state = State { + cursor: Some("cursor-1".into()), + messages_produced: 7, + }; + let bytes = rmp_serde::to_vec(&state).expect("serialize"); + let source = NameSource::new(1, test_config(), Some(ConnectorState(bytes))); + let runtime = tokio::runtime::Runtime::new().unwrap(); + runtime.block_on(async { + let restored = source.state.lock().await; + assert_eq!(restored.cursor.as_deref(), Some("cursor-1")); + assert_eq!(restored.messages_produced, 7); + }); + } + + #[test] + fn given_no_state_should_start_fresh() { + let source = NameSource::new(1, test_config(), None); + let runtime = tokio::runtime::Runtime::new().unwrap(); + runtime.block_on(async { + let restored = source.state.lock().await; + assert!(restored.cursor.is_none()); + assert_eq!(restored.messages_produced, 0); + }); + } + + #[test] + fn given_invalid_state_should_start_fresh() { + let invalid = ConnectorState(b"not valid msgpack".to_vec()); + let source = NameSource::new(1, test_config(), Some(invalid)); + let runtime = tokio::runtime::Runtime::new().unwrap(); + runtime.block_on(async { + let restored = source.state.lock().await; + assert!(restored.cursor.is_none()); + assert_eq!(restored.messages_produced, 0); + }); + } + + #[test] + fn state_should_be_serializable_and_deserializable() { + let original = State { + cursor: Some("c".into()), + messages_produced: 3, + }; + let bytes = rmp_serde::to_vec(&original).unwrap(); + let restored: State = rmp_serde::from_slice(&bytes).unwrap(); + assert_eq!(original, restored); + } + + #[test] + fn given_defaults_should_apply_consts() { + let source = NameSource::new( + 1, + NameSourceConfig { + connection_string: SecretString::from("scheme://localhost/db"), + poll_interval: None, + batch_size: None, + max_retries: None, + retry_delay: None, + verbose_logging: None, + }, + None, + ); + assert_eq!(source.batch_size, DEFAULT_BATCH_SIZE as usize); + assert_eq!(source.max_retries, DEFAULT_MAX_RETRIES); + assert_eq!(source.poll_interval, Duration::from_secs(5)); + } + + // Stages `pending_state` directly rather than going through poll() - poll()'s + // TODO(ConnectorDeveloper) fetch is unimplemented in this template, but on_batch_result's + // commit/discard contract is independently testable and must stay covered once + // the backend is filled in. + + #[test] + fn given_ack_when_batch_is_staged_should_commit_candidate_state() { + let source = NameSource::new(1, test_config(), None); + let runtime = tokio::runtime::Runtime::new().unwrap(); + runtime.block_on(async { + let candidate = State { + cursor: Some("cursor-2".into()), + messages_produced: 5, + }; + *source.pending_state.lock().await = Some(candidate.clone()); + + source + .on_batch_result(SourceBatchResult::Ack) + .await + .expect("ack should be applied"); + + assert_eq!(*source.state.lock().await, candidate); + assert!(source.pending_state.lock().await.is_none()); + }); + } + + #[test] + fn given_nack_when_batch_is_staged_should_keep_committed_state() { + let source = NameSource::new(1, test_config(), None); + let runtime = tokio::runtime::Runtime::new().unwrap(); + runtime.block_on(async { + let committed_before = source.state.lock().await.clone(); + *source.pending_state.lock().await = Some(State { + cursor: Some("cursor-2".into()), + messages_produced: 5, + }); + + source + .on_batch_result(SourceBatchResult::Nack) + .await + .expect("nack should be applied"); + + assert_eq!(*source.state.lock().await, committed_before); + assert!(source.pending_state.lock().await.is_none()); + }); + } +} ``` + +--- + +## README.md (required paragraphs) + +Include a **Delivery semantics** section: + +1. Transient fetch failure → retry N times, then `Err` (loop continues; see `connector-source`) +2. Cursor commits only on `on_batch_result(Ack)` - a Nack (send or state-save failure) discards the staged cursor and redelivers the batch +3. Whether destructive work (delete/mark-processed) is staged in `poll()` and applied only in `on_batch_result()` on Ack (standard - no loss window to document), or must happen earlier for some architectural reason (if so, document the loss window explicitly) +4. Dedup key for `ProducedMessage.id` (or "none") + +--- + +## Before `/ready` + +Run the pre-flight checklist in +[connector-pr-review](../connector-pr-review/SKILL.md#pre-flight-author-checklist). +Mandatory: the six canonical state tests above must stay green. diff --git a/.claude/skills/connectors-overview/SKILL.md b/.claude/skills/connectors-overview/SKILL.md index 230ad5354b..d119dcf409 100644 --- a/.claude/skills/connectors-overview/SKILL.md +++ b/.claude/skills/connectors-overview/SKILL.md @@ -25,7 +25,7 @@ Repo-wide rules (Apache headers, fmt/sort/clippy order, idiomatic Rust traits, i ## STOP and ask the user before -- Bumping `iggy_connector_sdk` MAJOR version or changing any FFI signature in `sdk/src/{sink,source}.rs` - breaks every pre-built plugin `.so`. +- Bumping `iggy_connector_sdk` MAJOR version or changing any FFI signature in `sdk/src/{sink,source}.rs` - breaks every pre-built plugin `.so`. Source FFI is `iggy_source_handle_v2` (batch ID carried to the runtime callback) + `iggy_source_batch_result` (plugin-exported ACK/NACK, #3855). - Changing the runtime's wire conventions (postcard FFI payload structs, default consumer group naming, plugin path resolution). - Modifying `runtime/src/state.rs` save protocol (atomic rename + fsync ordering) - corruption risk. - Renaming or repurposing a `Schema` variant - decoders/encoders pinned to wire bytes. @@ -42,27 +42,31 @@ Both compile as `cdylib` shared libraries (`.so`/`.dylib`/`.dll`) loaded by the ```text ┌─ optional transforms ─┐ External ──poll──▶ SOURCE ──FFI──▶ RUNTIME ──encode──▶ Apache Iggy stream - system plugin ▲ ▲ - │ │ - state save (msgpack) │ - │ - ┌─ optional transforms ─┐ │ + system plugin ▲ ▲ + │ │ + Ack/Nack (on_batch_result) │ + │ state save (msgpack) + └────────────────────┘ + ┌─ optional transforms ─┐ Apache Iggy stream ──decode──▶ RUNTIME ──FFI──▶ SINK ──write──▶ External plugin system ``` +Source is a request/response FFI, not fire-and-forget: after the runtime sends the batch and saves the state `poll()` staged, it reports `SourceBatchResult::Ack` or `::Nack` back to the plugin via `iggy_source_batch_result` (#3855) before calling `poll()` again. See [connector-source](../connector-source/SKILL.md#state-persistence-stage-in-poll-commit-in-on_batch_result) for the full handshake. + Headers set on the source side ride through transforms (which may modify, drop, or pass them) and arrive at the sink with `BTreeMap` preserved deterministically. ## Which skill to load -| Task | Skill | -| --------------------------------------------------- | --------------------- | -| Write a new sink plugin | `connector-sink` | -| Write a new source plugin | `connector-source` | -| Add schema / decoder / encoder / SDK trait surface | `connector-sdk` | -| Change runtime internals (FFI, manager, state, ...) | `connector-runtime` | -| Add a transform (field-level or format conversion) | `connector-transform` | -| Write unit / integration tests for any of the above | `connector-testing` | +| Task | Skill | +| ---------------------------------------------------- | --------------------- | +| Write a new sink plugin | `connector-sink` | +| Write a new source plugin | `connector-source` | +| Add schema / decoder / encoder / SDK trait surface | `connector-sdk` | +| Change runtime internals (FFI, manager, state, ...) | `connector-runtime` | +| Add a transform (field-level or format conversion) | `connector-transform` | +| Write unit / integration tests for any of the above | `connector-testing` | +| Review a sink/source PR / pre-flight before `/ready` | `connector-pr-review` | ## Stick to conventions @@ -96,24 +100,14 @@ The connectors codebase is intentionally repetitive across plugins. Cross-plugin ### Secrets -Any credential-bearing field (connection strings, API keys, bearer tokens, AWS keys) must be `SecretString` from the `secrecy` crate. Plain `String` for a credential is a review-blocker: `SecretString` redacts on `Debug`, so it is what keeps a credential out of a log line that formats the whole config. - -**`serde_secret::serialize_secret` EXPOSES the secret. It does not redact.** It calls `expose_secret()` and writes the plaintext. `SecretString` deliberately has no `Serialize` impl, and that absence is the protection - so adding `serialize_with` is what *unblocks* the derive and turns a compile-time guarantee into plaintext output. Use it only where the plaintext is the point: a wire payload, a persisted config, an API response that exposes credentials by design. - -So the default for a plugin config struct is **derive `Deserialize`, but not `Serialize`**. `Deserialize` is required: the SDK glue deserializes the config into the plugin's own struct (`sdk/src/{sink,source}.rs` call `serde_json::from_str::` under a `DeserializeOwned` bound). - -What never happens is the return trip. The runtime holds plugin configuration as a `serde_json::Value` - parsed from TOML, posted as JSON to the control API, or injected by env var - and hands that across the FFI, so nothing re-serializes the plugin's struct. Leaving `Serialize` off makes that compiler-enforced instead of convention-enforced (`sources/http_source/src/lib.rs::HttpSourceConfig` does this, and comments the omission so nobody adds it back). - -Pattern: +Any credential-bearing field (connection strings, API keys, bearer tokens, AWS keys) must be `SecretString` from the `secrecy` crate, with the workspace serde wrapper applied so `Debug` and serialization both redact. Runtime exposes plugin configs over the `/stats` HTTP surface via serialization - plain `String` leaks the secret to anyone who can hit the endpoint. Plain `String` for a credential is a review-blocker. Pattern (from `sinks/postgres_sink/src/lib.rs::PostgresSinkConfig`): ```rust use secrecy::{ExposeSecret, SecretString}; -// `Deserialize` only. Nothing re-serializes a plugin config, and leaving -// `Serialize` off is what makes the credential unserializable rather than -// merely un-serialized. -#[derive(Debug, Clone, Deserialize)] +#[derive(Debug, Clone, Serialize, Deserialize)] pub struct MyConfig { + #[serde(serialize_with = "iggy_common::serde_secret::serialize_secret")] pub connection_string: SecretString, } @@ -123,13 +117,7 @@ let pool = PgPoolOptions::new() .await?; ``` -If a config struct genuinely needs `Serialize`, `serde_secret::serialize_redacted` (and `serialize_optional_redacted`) write `[REDACTED]` in place of the value. Reach for `serialize_secret` only when the caller must get the real thing back. The sinks and sources listed below predate that helper and use the exposing one; the annotation is inert today, but it is not the protection it looks like. - -Note that none of this protects the credential from the runtime's own control API, which returns plugin configuration verbatim - see #3802. Plugin-side annotations are inert there because the runtime never routes through them. - -Plugin-side uses of the exposing helpers: `sinks/{postgres,mongodb,elasticsearch,influxdb,s3,surrealdb}_sink`, `sources/{postgres,elasticsearch,influxdb}_source`. - -That list is plugin-side only, not an inventory of every caller in the tree, and the others are not all mistakes: `runtime/src/api/config.rs` puts `serialize_secret` on `HttpConfig::api_key` (inert for the same reason), and several `core/common` wire-payload types (login, create-user, change-password, PAT) use these helpers by design, because there the credential *is* the payload. +In-tree uses: `sinks/{postgres,mongodb,elasticsearch,influxdb,delta}_sink`, `sources/{postgres,elasticsearch,influxdb}_source`. ### Errors @@ -181,7 +169,7 @@ JSON log format, parser unit tests). | Real-infra sink + integration tests | `sinks/postgres_sink/` + `integration/tests/connectors/postgres/postgres_sink.rs` | | Feature-rich sink config (validation, batch modes, retry) | `sinks/http_sink/` | | Atomic counters on hot path | `sinks/mongodb_sink/` | -| Simplest source (4 canonical state tests) | `sources/random_source/` | +| Simplest source (6 canonical state + ACK/NACK tests) | `sources/random_source/` | | Real-infra source | `sources/postgres_source/` + `integration/tests/connectors/postgres/postgres_source.rs` | Read the relevant exemplar end-to-end before writing or modifying a connector. @@ -209,7 +197,7 @@ Each implemented in at least one in-tree plugin or runtime path. | `flume::unbounded()` channel | `runtime/src/source.rs::spawn_source_handler` / `source_forwarding_loop` | MPSC handoff from SDK async task to runtime loop | | `tokio::sync::watch::channel(())` | `sdk/src/{sink,source}.rs`, `runtime/src/sink.rs`, `runtime/src/manager/*` | One-shot shutdown broadcast | | `dashmap::DashMap` | `runtime/src/manager/sink.rs`, `source.rs::SOURCE_SENDERS`, SDK `INSTANCES` | Lock-free concurrent keyed access | -| `secrecy::SecretString` + `iggy_common::serde_secret::serialize_secret` | `sinks/postgres_sink::PostgresSinkConfig::connection_string` | `Debug` redacts; `serialize_secret` EXPOSES | +| `secrecy::SecretString` + `iggy_common::serde_secret::serialize_secret` | `sinks/postgres_sink::PostgresSinkConfig::connection_string` | Auto-redact on Debug/Display + serialization | ## Drop accounting @@ -222,13 +210,21 @@ Each implemented in at least one in-tree plugin or runtime path. - `&mut self` on `Sink::consume` or `Source::poll` impls - won't compile, flag any creative workaround. - `std::sync::Mutex` held across `.await` - swap for `tokio::sync::Mutex`. - Missing `[lib] crate-type = ["cdylib", "lib"]` in plugin `Cargo.toml`. -- Source plugin without the four canonical state tests (see `connector-testing`). +- Source plugin without the four canonical state tests, or without ACK/NACK tests when `on_batch_result` is overridden (see `connector-testing`). +- Source `poll()` committing a cursor or destructive work directly instead of staging it for `on_batch_result` to commit on `Ack` / discard on `Nack`. - New silent message drop without a metric increment. - Wrapping `format!()` around args passed to `error!`/`warn!`/`info!`/`debug!` - eager `format!` allocates even when level filters the line out. Pass args directly: `error!("foo: {x}")` or `error!(error = %x, "foo")`. - Logging a connection string, API key, or token. - Plain `String` for a credential field - use `SecretString`. - `tokio::spawn` inside plugin code - runtime owns lifecycle. - `std::time::SystemTime::now()` in transforms - non-deterministic, breaks tests. +- Returning `Ok(())` from `consume` after a failed batch (offsets can still advance). +- Random UUIDs as message IDs / dedup keys. +- Classifying retryability via `err.to_string()` substring matches. +- README defaults that disagree with code consts. +- Invented config knob names (`request_timeout`, `retry_max_delay`) instead of the canon in `connector-pr-review`. + +For the full PR review checklist (blockers, delivery-semantics paragraph, pre-flight paste), load [connector-pr-review](../connector-pr-review/SKILL.md). ## File map diff --git a/core/connectors/BLOG_POST.md b/core/connectors/BLOG_POST.md index fe8b45ea1c..02db07fa96 100644 --- a/core/connectors/BLOG_POST.md +++ b/core/connectors/BLOG_POST.md @@ -72,7 +72,7 @@ passing `cargo test`: round-trip — plus the two ACK/NACK tests, plus config validation and the circuit-breaker short-circuit path). -What's left is marked `TODO(Developer)` in each crate's `src/lib.rs`: +What's left is marked `TODO(ConnectorDeveloper)` in each crate's `src/lib.rs`: one spot for a sink (`push_batch()`), two for a source (`build_raw_client()` if you're not talking HTTP, and `fetch_records()`). Everything else — the parts that used to eat a @@ -81,7 +81,7 @@ review round — is already done. ## Using one Copy the crate, rename the package and the directory, add it to the -workspace `members` list, fill in the `TODO(Developer)` spots, and +workspace `members` list, fill in the `TODO(ConnectorDeveloper)` spots, and update `config.toml` for your system. Each crate's own `README.md` walks through the exact steps. Both templates already build, `clippy --all-targets -- -D warnings` clean, and pass their tests as committed diff --git a/core/connectors/sinks/README.md b/core/connectors/sinks/README.md index 990cbe59f8..c0d786f332 100644 --- a/core/connectors/sinks/README.md +++ b/core/connectors/sinks/README.md @@ -16,7 +16,7 @@ Sink connectors are responsible for writing data from Iggy streams to external s | **postgres_sink** | Stores messages in PostgreSQL database tables with configurable schemas | | **quickwit_sink** | Indexes messages in Quickwit search engine for log analytics | | **s3_sink** | Writes messages to Amazon S3 and S3-compatible stores (MinIO, R2, B2, DO Spaces) | -| **sink_template** | Fill-in-the-blank starting point for a new sink; framework/security plumbing done, one `TODO(Developer)` spot left | +| **sink_template** | Fill-in-the-blank starting point for a new sink; framework/security plumbing done, one `TODO(ConnectorDeveloper)` spot left | | **stdout_sink** | Prints messages to standard output (useful for debugging and development) | | **surrealdb_sink** | Writes messages into SurrealDB with deterministic record IDs for idempotent replay | diff --git a/core/connectors/sinks/sink_template/README.md b/core/connectors/sinks/sink_template/README.md index 16547f5d14..002409da6b 100644 --- a/core/connectors/sinks/sink_template/README.md +++ b/core/connectors/sinks/sink_template/README.md @@ -32,7 +32,7 @@ blog post for the checklist this template is built against. ## What you need to fill in -Search for `TODO(Developer)` in `src/lib.rs` — there is exactly one spot: +Search for `TODO(ConnectorDeveloper)` in `src/lib.rs` — there is exactly one spot: **`push_batch()`** — build the request/write that actually sends one chunk of messages to your destination, using `self.config.connection_string` (and @@ -55,7 +55,7 @@ in `open()` in favor of whatever connectivity check your driver offers. 1. Copy this directory, rename it and the package in `Cargo.toml` (`iggy_connector__sink`), and add it to the `members` list in the workspace root `Cargo.toml`. -2. Fill in the `TODO(Developer)` section(s). +2. Fill in the `TODO(ConnectorDeveloper)` section(s). 3. Update `config.toml` with your real `connection_string` and `target`, and any settings specific to your system; delete `auth_token` if you don't need it, or add fields of your own the same way (see diff --git a/core/connectors/sinks/sink_template/src/lib.rs b/core/connectors/sinks/sink_template/src/lib.rs index 20a28aa9d6..152121c5c7 100644 --- a/core/connectors/sinks/sink_template/src/lib.rs +++ b/core/connectors/sinks/sink_template/src/lib.rs @@ -25,7 +25,7 @@ //! are chunked by a configurable batch size instead of shipped as one //! unbounded request. //! -//! There is exactly **one** place you need to touch, marked `TODO(Developer)`: +//! There is exactly **one** place you need to touch, marked `TODO(ConnectorDeveloper)`: //! `TemplateSink::push_batch()` — build the request/write that actually //! pushes one chunk of messages to your destination, using //! `self.config.connection_string` (and `self.config.target`, if your @@ -83,7 +83,7 @@ const DEFAULT_CIRCUIT_BREAKER_COOL_DOWN: &str = "30s"; #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(deny_unknown_fields)] pub struct TemplateSinkConfig { - /// TODO(Developer): document the exact shape this connector expects, e.g. + /// TODO(ConnectorDeveloper): document the exact shape this connector expects, e.g. /// "https://api.example.com" or "postgres://user:pass@host:5432/db". /// `SecretString` because DSNs commonly embed credentials — never plain /// `String` for this field, see `PostgresSinkConfig::connection_string` @@ -188,7 +188,7 @@ impl TemplateSink { } } - /// TODO(Developer): build your actual client/connection here using + /// TODO(ConnectorDeveloper): build your actual client/connection here using /// `self.config.connection_string` (and `self.config.auth_token`, if /// your destination needs it). This template builds a plain /// `reqwest::Client` to hand to `build_retry_client` — if you're not @@ -205,7 +205,7 @@ impl TemplateSink { .map_err(|e| Error::Connection(format!("failed to build HTTP client: {e}"))) } - /// TODO(Developer): push one chunk of already-batched messages to your + /// TODO(ConnectorDeveloper): push one chunk of already-batched messages to your /// destination using `self.config.connection_string` and /// `self.config.target`, via `self.client` (already retry-wrapped). /// Distinguish permanent failures (bad schema, destination rejects the @@ -221,7 +221,7 @@ impl TemplateSink { ) -> Result<(), Error> { let _ = (client, batch); // remove once implemented Err(Error::InitError( - "TemplateSink::push_batch is not implemented yet — see the TODO(Developer) comment in \ + "TemplateSink::push_batch is not implemented yet — see the TODO(ConnectorDeveloper) comment in \ template_sink/src/lib.rs" .to_string(), )) diff --git a/core/connectors/sources/README.md b/core/connectors/sources/README.md index 640cd0ad5c..f79b253513 100644 --- a/core/connectors/sources/README.md +++ b/core/connectors/sources/README.md @@ -12,7 +12,7 @@ Source connectors are responsible for ingesting data from external sources into | **influxdb_source** | Polls InfluxDB with cursor-based timestamp tracking; supports V2 (Flux, annotated CSV) and V3 (SQL, JSONL) | | **postgres_source** | Reads rows from PostgreSQL tables with multiple strategies: delete after read, mark as processed, or timestamp tracking | | **random_source** | Generates random test messages (useful for testing and development) | -| **source_template** | Fill-in-the-blank starting point for a new source; framework/security plumbing done, two `TODO(Developer)` spots left | +| **source_template** | Fill-in-the-blank starting point for a new source; framework/security plumbing done, two `TODO(ConnectorDeveloper)` spots left | The source is represented by the single `Source` trait, which defines the basic interface for all source connectors. It provides methods for initializing the source, reading data from it, and closing the source. diff --git a/core/connectors/sources/source_template/README.md b/core/connectors/sources/source_template/README.md index f4a39d3bac..415e2c210a 100644 --- a/core/connectors/sources/source_template/README.md +++ b/core/connectors/sources/source_template/README.md @@ -29,7 +29,7 @@ blog post for the checklist this template is built against. ## What you need to fill in -Search for `TODO(Developer)` in `src/lib.rs` — there are exactly two spots: +Search for `TODO(ConnectorDeveloper)` in `src/lib.rs` — there are exactly two spots: 1. **`build_raw_client()`** — if your source isn't HTTP, replace the `reqwest::Client` construction with your driver's connection/pool setup @@ -50,7 +50,7 @@ Search for `TODO(Developer)` in `src/lib.rs` — there are exactly two spots: 1. Copy this directory, rename it and the package in `Cargo.toml` (`iggy_connector__source`), and add it to the `members` list in the workspace root `Cargo.toml`. -2. Fill in the two `TODO(Developer)` sections. +2. Fill in the two `TODO(ConnectorDeveloper)` sections. 3. Update `config.toml` with your real `connection_string` and any settings specific to your system; delete `auth_token` if you don't need it, or add fields of your own the same way (see `TemplateSourceConfig`). diff --git a/core/connectors/sources/source_template/src/lib.rs b/core/connectors/sources/source_template/src/lib.rs index 24f52b44c1..57a2c576b0 100644 --- a/core/connectors/sources/source_template/src/lib.rs +++ b/core/connectors/sources/source_template/src/lib.rs @@ -25,7 +25,7 @@ //! a dropped/nacked batch can be re-polled instead of silently lost. //! //! There are exactly **two** places you need to touch, each marked -//! `TODO(Developer)`: +//! `TODO(ConnectorDeveloper)`: //! 1. `TemplateSource::connect()` — build your actual client/connection //! from `config.connection_string` (and `config.auth_token`, if used). //! 2. `TemplateSource::fetch_records()` — fetch up to `batch_size` new @@ -84,7 +84,7 @@ const DEFAULT_CIRCUIT_BREAKER_COOL_DOWN: &str = "30s"; #[derive(Debug, Clone, Serialize, Deserialize)] #[serde(deny_unknown_fields)] pub struct TemplateSourceConfig { - /// TODO(Developer): document the exact shape this connector expects, e.g. + /// TODO(ConnectorDeveloper): document the exact shape this connector expects, e.g. /// "https://api.example.com" or "postgres://user:pass@host:5432/db". /// `SecretString` because DSNs commonly embed credentials — never plain /// `String` for this field, see `PostgresSinkConfig::connection_string` @@ -201,7 +201,7 @@ impl TemplateSource { ConnectorState::serialize(state, CONNECTOR_NAME, self.id) } - /// TODO(Developer): build your actual client/connection here using + /// TODO(ConnectorDeveloper): build your actual client/connection here using /// `self.config.connection_string` (and `self.config.auth_token`, if /// your system needs it). This template builds a plain `reqwest::Client` /// to hand to `build_retry_client` — if you're not talking HTTP, replace @@ -217,7 +217,7 @@ impl TemplateSource { .map_err(|e| Error::Connection(format!("failed to build HTTP client: {e}"))) } - /// TODO(Developer): fetch up to `self.batch_size` new records from your + /// TODO(ConnectorDeveloper): fetch up to `self.batch_size` new records from your /// external system, ordered after `cursor` (`None` means "from the /// beginning" or "from now" — whichever is right for your source). /// Use `self.config.connection_string` as the base address and @@ -233,7 +233,7 @@ impl TemplateSource { ) -> Result, Error> { let _ = (client, cursor); // remove once implemented Err(Error::InitError( - "TemplateSource::fetch_records is not implemented yet — see the TODO(Developer) comment \ + "TemplateSource::fetch_records is not implemented yet — see the TODO(ConnectorDeveloper) comment \ in template_source/src/lib.rs" .to_string(), )) From 4cf2ba22d803a47563607938060ed60b63d6b3d9 Mon Sep 17 00:00:00 2001 From: ryerraguntla Date: Tue, 25 Aug 2026 13:10:36 -0400 Subject: [PATCH 003/182] Potential fix for pull request finding Co-authored-by: Copilot Autofix powered by AI <175728472+Copilot@users.noreply.github.com> --- core/connectors/sinks/sink_template/src/lib.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/core/connectors/sinks/sink_template/src/lib.rs b/core/connectors/sinks/sink_template/src/lib.rs index 152121c5c7..7b62d6cddb 100644 --- a/core/connectors/sinks/sink_template/src/lib.rs +++ b/core/connectors/sinks/sink_template/src/lib.rs @@ -118,7 +118,7 @@ pub struct TemplateSinkConfig { pub timeout: Option, pub max_retries: Option, pub retry_delay: Option, - pub retry_max_delay: Option, + pub max_retry_delay: Option, pub max_open_retries: Option, pub open_retry_max_delay: Option, pub circuit_breaker_threshold: Option, From ff0365a0853e2f6413bdc03e096897f686339703 Mon Sep 17 00:00:00 2001 From: ryerraguntla Date: Tue, 25 Aug 2026 22:56:03 +0530 Subject: [PATCH 004/182] Changed the license headers --- .../connectors/sinks/sink_template/Cargo.toml | 40 +++++++++---------- .../sources/source_template/Cargo.toml | 40 +++++++++---------- 2 files changed, 40 insertions(+), 40 deletions(-) diff --git a/core/connectors/sinks/sink_template/Cargo.toml b/core/connectors/sinks/sink_template/Cargo.toml index 8ab161bd1e..b4ab09edc3 100644 --- a/core/connectors/sinks/sink_template/Cargo.toml +++ b/core/connectors/sinks/sink_template/Cargo.toml @@ -1,23 +1,23 @@ -# Licensed to the Apache Software Foundation (ASF) under one -# or more contributor license agreements. See the NOTICE file -# distributed with this work for additional information -# regarding copyright ownership. The ASF licenses this file -# to you under the Apache License, Version 2.0 (the -# "License"); you may not use this file except in compliance -# with the License. You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, -# software distributed under the License is distributed on an -# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY -# KIND, either express or implied. See the License for the -# specific language governing permissions and limitations -# under the License. -# -# TEMPLATE — rename the package (and this directory) to -# `iggy_connector__sink` before publishing, and update the -# `[[sinks]]` entry you add to the workspace root Cargo.toml accordingly. +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. +// +//! TEMPLATE — rename the package (and this directory) to +//! `iggy_connector__sink` before publishing, and update the +//! `[[sinks]]` entry you add to the workspace root Cargo.toml accordingly. [package] name = "iggy_connector_template_sink" diff --git a/core/connectors/sources/source_template/Cargo.toml b/core/connectors/sources/source_template/Cargo.toml index a54d7cbccc..392504dacf 100644 --- a/core/connectors/sources/source_template/Cargo.toml +++ b/core/connectors/sources/source_template/Cargo.toml @@ -1,23 +1,23 @@ -# Licensed to the Apache Software Foundation (ASF) under one -# or more contributor license agreements. See the NOTICE file -# distributed with this work for additional information -# regarding copyright ownership. The ASF licenses this file -# to you under the Apache License, Version 2.0 (the -# "License"); you may not use this file except in compliance -# with the License. You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, -# software distributed under the License is distributed on an -# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY -# KIND, either express or implied. See the License for the -# specific language governing permissions and limitations -# under the License. -# -# TEMPLATE — rename the package (and this directory) to -# `iggy_connector__source` before publishing, and update the -# `[[sources]]` entry you add to the workspace root Cargo.toml accordingly. +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. +// +//! TEMPLATE — rename the package (and this directory) to +//! `iggy_connector__source` before publishing, and update the +//! `[[sources]]` entry you add to the workspace root Cargo.toml accordingly. [package] name = "iggy_connector_template_source" From 5fe30bc6009556304e0196cde7a56c1631813f84 Mon Sep 17 00:00:00 2001 From: ryerraguntla Date: Sat, 29 Aug 2026 10:42:21 +0530 Subject: [PATCH 005/182] Fixing the review comments --- .claude/skills/connector-pr-review/SKILL.md | 2 +- .claude/skills/connector-sink/TEMPLATE.md | 6 + .claude/skills/connector-source/TEMPLATE.md | 6 + .claude/skills/connectors-overview/SKILL.md | 2 +- .../connectors/sink_template.toml | 1 + .../connectors/source_template.toml | 1 + .../connectors/sinks/sink_template/Cargo.toml | 40 ++-- core/connectors/sinks/sink_template/README.md | 30 ++- .../sinks/sink_template/config.toml | 1 + .../connectors/sinks/sink_template/src/lib.rs | 197 ++++++++++++++---- .../sources/source_template/Cargo.toml | 40 ++-- .../sources/source_template/README.md | 26 ++- .../sources/source_template/config.toml | 1 + .../sources/source_template/src/lib.rs | 76 ++++++- 14 files changed, 309 insertions(+), 120 deletions(-) diff --git a/.claude/skills/connector-pr-review/SKILL.md b/.claude/skills/connector-pr-review/SKILL.md index c386099c79..01f2c38915 100644 --- a/.claude/skills/connector-pr-review/SKILL.md +++ b/.claude/skills/connector-pr-review/SKILL.md @@ -153,7 +153,7 @@ These are "cheap" but burn full review rounds when missed. | Request timeout | `timeout` | Not `request_timeout` | | Retry attempts | `max_retries` | Total attempts, default 3 | | Base backoff | `retry_delay` | humantime `Option` | -| Backoff ceiling | `max_retry_delay` | Not `retry_max_delay` | +| Backoff ceiling | `retry_max_delay` | Matches SDK's `ConnectivityConfig::open_retry_max_delay` | | Poll cadence (sources) | `poll_interval` | humantime; sleep first | | Plugin verbosity | `verbose_logging` | Mirror runtime `verbose` | | Credentials | `connection_string` / `api_key` / … | Always `SecretString` | diff --git a/.claude/skills/connector-sink/TEMPLATE.md b/.claude/skills/connector-sink/TEMPLATE.md index 19a6956309..cb4bee4d5c 100644 --- a/.claude/skills/connector-sink/TEMPLATE.md +++ b/.claude/skills/connector-sink/TEMPLATE.md @@ -9,6 +9,12 @@ batch. Also read [SKILL.md](SKILL.md) and pre-flight with [connector-pr-review](../connector-pr-review/SKILL.md) before `/ready`. +Prefer starting from a compiling crate over copying this prose kit: +`core/connectors/sinks/sink_template/` implements the same shape as real, +tested code you can `cargo build`/`cargo test` immediately, with the same +`TODO(ConnectorDeveloper)` markers. Use this kit instead only when copying +a whole crate is more scaffolding than you need. + ## Files to create ```text diff --git a/.claude/skills/connector-source/TEMPLATE.md b/.claude/skills/connector-source/TEMPLATE.md index c5db2183b7..306cba25f2 100644 --- a/.claude/skills/connector-source/TEMPLATE.md +++ b/.claude/skills/connector-source/TEMPLATE.md @@ -11,6 +11,12 @@ a cursor). Also read [SKILL.md](SKILL.md) and pre-flight with [connector-pr-review](../connector-pr-review/SKILL.md) before `/ready`. +Prefer starting from a compiling crate over copying this prose kit: +`core/connectors/sources/source_template/` implements the same shape as +real, tested code you can `cargo build`/`cargo test` immediately, with the +same `TODO(ConnectorDeveloper)` markers. Use this kit instead only when +copying a whole crate is more scaffolding than you need. + ## Files to create ```text diff --git a/.claude/skills/connectors-overview/SKILL.md b/.claude/skills/connectors-overview/SKILL.md index d119dcf409..4f6f5e169e 100644 --- a/.claude/skills/connectors-overview/SKILL.md +++ b/.claude/skills/connectors-overview/SKILL.md @@ -222,7 +222,7 @@ Each implemented in at least one in-tree plugin or runtime path. - Random UUIDs as message IDs / dedup keys. - Classifying retryability via `err.to_string()` substring matches. - README defaults that disagree with code consts. -- Invented config knob names (`request_timeout`, `retry_max_delay`) instead of the canon in `connector-pr-review`. +- Invented config knob names (e.g. `request_timeout` instead of `timeout`) instead of the canon in `connector-pr-review`. For the full PR review checklist (blockers, delivery-semantics paragraph, pre-flight paste), load [connector-pr-review](../connector-pr-review/SKILL.md). diff --git a/core/connectors/runtime/example_config/connectors/sink_template.toml b/core/connectors/runtime/example_config/connectors/sink_template.toml index 90c3937f95..ee9a7bd3de 100644 --- a/core/connectors/runtime/example_config/connectors/sink_template.toml +++ b/core/connectors/runtime/example_config/connectors/sink_template.toml @@ -45,3 +45,4 @@ max_open_retries = 10 open_retry_max_delay = "60s" circuit_breaker_threshold = 5 circuit_breaker_cool_down = "30s" +verbose_logging = false diff --git a/core/connectors/runtime/example_config/connectors/source_template.toml b/core/connectors/runtime/example_config/connectors/source_template.toml index c53439a0ae..02dbe0eab8 100644 --- a/core/connectors/runtime/example_config/connectors/source_template.toml +++ b/core/connectors/runtime/example_config/connectors/source_template.toml @@ -44,3 +44,4 @@ max_open_retries = 10 open_retry_max_delay = "60s" circuit_breaker_threshold = 5 circuit_breaker_cool_down = "30s" +verbose_logging = false diff --git a/core/connectors/sinks/sink_template/Cargo.toml b/core/connectors/sinks/sink_template/Cargo.toml index b4ab09edc3..6ee035f63b 100644 --- a/core/connectors/sinks/sink_template/Cargo.toml +++ b/core/connectors/sinks/sink_template/Cargo.toml @@ -1,23 +1,23 @@ -// Licensed to the Apache Software Foundation (ASF) under one -// or more contributor license agreements. See the NOTICE file -// distributed with this work for additional information -// regarding copyright ownership. The ASF licenses this file -// to you under the Apache License, Version 2.0 (the -// "License"); you may not use this file except in compliance -// with the License. You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, -// software distributed under the License is distributed on an -// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY -// KIND, either express or implied. See the License for the -// specific language governing permissions and limitations -// under the License. -// -//! TEMPLATE — rename the package (and this directory) to -//! `iggy_connector__sink` before publishing, and update the -//! `[[sinks]]` entry you add to the workspace root Cargo.toml accordingly. +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# TEMPLATE — rename the package (and this directory) to +# `iggy_connector__sink` before publishing, and update the +# `[[sinks]]` entry you add to the workspace root Cargo.toml accordingly. [package] name = "iggy_connector_template_sink" diff --git a/core/connectors/sinks/sink_template/README.md b/core/connectors/sinks/sink_template/README.md index 002409da6b..86da056d81 100644 --- a/core/connectors/sinks/sink_template/README.md +++ b/core/connectors/sinks/sink_template/README.md @@ -27,21 +27,29 @@ blog post for the checklist this template is built against. `batch_size` instead of sending everything in one unbounded request. - The `sink_connector!` FFI macro invocation and a `Cargo.toml` with the right `crate-type`, workspace-pinned dependencies, and license header. -- Tests for config/identifier validation and the circuit-breaker short-circuit - path. +- `verbose_logging: Option` upgrading the per-batch log line from + `debug!` to `info!`, mirroring the runtime's own `verbose` flag. +- Tests for config/identifier validation, the circuit-breaker short-circuit + path, the `verbose_logging` flag, and `consume()`'s batch loop end to end. ## What you need to fill in -Search for `TODO(ConnectorDeveloper)` in `src/lib.rs` — there is exactly one spot: +Search for `TODO(ConnectorDeveloper)` in `src/lib.rs` — there is exactly one +spot that requires code, plus two more that are conditional/documentation: -**`push_batch()`** — build the request/write that actually sends one chunk -of messages to your destination, using `self.config.connection_string` (and -`self.config.target`, already validated by the time this runs) via -`self.client` (already retry-wrapped). Distinguish permanent failures (bad -schema, a destination that will reject this payload shape no matter how many -times you retry) from transient ones (network error, 5xx, timeout) by -returning `Error::PermanentHttpError` for the former — see the doc comment on -that variant for why the distinction matters to the circuit breaker. +**`push_batch()`** (required) — build the request/write that actually sends +one chunk of messages to your destination, using +`self.config.connection_string` (and `self.config.target`, already validated +by the time this runs) via `self.client` (already retry-wrapped). Distinguish +permanent failures (bad schema, a destination that will reject this payload +shape no matter how many times you retry) from transient ones (network +error, 5xx, timeout) by returning `Error::PermanentHttpError` for the +former — `consume()` drops and counts a permanent failure instead of +propagating it, so a single bad message can't take the whole connector down; +any other error stops `consume()` and is returned as-is. + +**`connection_string`'s doc comment** (documentation only) — describe the +exact shape your connector expects instead of the generic example. If your destination isn't HTTP, also revisit **`build_raw_client()`**: swap the `reqwest::Client` for your driver's connection/pool setup (see diff --git a/core/connectors/sinks/sink_template/config.toml b/core/connectors/sinks/sink_template/config.toml index 8156530c9d..86c91de246 100644 --- a/core/connectors/sinks/sink_template/config.toml +++ b/core/connectors/sinks/sink_template/config.toml @@ -45,4 +45,5 @@ max_open_retries = 10 open_retry_max_delay = "60s" circuit_breaker_threshold = 5 circuit_breaker_cool_down = "30s" +verbose_logging = false # auth_token = "replace-me" # uncomment if your destination needs bearer/API-key auth diff --git a/core/connectors/sinks/sink_template/src/lib.rs b/core/connectors/sinks/sink_template/src/lib.rs index 7b62d6cddb..f1ad735068 100644 --- a/core/connectors/sinks/sink_template/src/lib.rs +++ b/core/connectors/sinks/sink_template/src/lib.rs @@ -25,18 +25,23 @@ //! are chunked by a configurable batch size instead of shipped as one //! unbounded request. //! -//! There is exactly **one** place you need to touch, marked `TODO(ConnectorDeveloper)`: -//! `TemplateSink::push_batch()` — build the request/write that actually -//! pushes one chunk of messages to your destination, using -//! `self.config.connection_string` (and `self.config.target`, if your -//! destination has a table/index/collection-shaped name). +//! There is exactly **one** place you need to write code, marked +//! `TODO(ConnectorDeveloper)`: `TemplateSink::push_batch()` — build the +//! request/write that actually pushes one chunk of messages to your +//! destination, using `self.config.connection_string` (and +//! `self.config.target`, if your destination has a table/index/ +//! collection-shaped name). A second `TODO(ConnectorDeveloper)` on the +//! `connection_string` field's doc comment just asks you to describe its +//! expected shape — not code, but worth personalizing too. `grep` for the +//! marker and you'll find both, plus one more on `build_raw_client()` that +//! only applies if your destination isn't HTTP (see below). //! //! This template assumes an HTTP-ish destination and uses `reqwest` wrapped //! by the SDK's retry middleware, because that's what //! `iggy_connector_sdk::retry` is built for and it covers the common case. //! If your destination talks something else (a database, a queue, object -//! storage), swap the client type in `connect()`/`push_batch()` for your -//! driver of choice and lean on its own retry/pooling behavior — keep the +//! storage), swap the client type in `build_raw_client()`/`push_batch()` for +//! your driver of choice and lean on its own retry/pooling behavior — keep the //! surrounding shape (validation in `open()`, circuit breaker, batching, //! identifier validation) unchanged. See `core/connectors/sinks/postgres_sink` //! or `core/connectors/sinks/s3_sink` in this repo for non-HTTP examples of @@ -58,7 +63,7 @@ use std::sync::Arc; use std::sync::atomic::{AtomicU64, Ordering}; use std::time::Duration; use tokio::sync::Mutex; -use tracing::{error, info, warn}; +use tracing::{debug, error, info, warn}; sink_connector!(TemplateSink); @@ -102,7 +107,7 @@ pub struct TemplateSinkConfig { /// `Debug`/log output; delete this field if `connection_string` already /// carries all required auth. Read it with `.expose_secret()` (from the /// `secrecy::ExposeSecret` trait) at the one place you actually need the - /// plaintext — e.g. when building an auth header in `connect()`. + /// plaintext — e.g. when building an auth header in `build_raw_client()`. #[serde( default, serialize_with = "iggy_common::serde_secret::serialize_optional_secret" @@ -118,11 +123,16 @@ pub struct TemplateSinkConfig { pub timeout: Option, pub max_retries: Option, pub retry_delay: Option, - pub max_retry_delay: Option, + pub retry_max_delay: Option, pub max_open_retries: Option, pub open_retry_max_delay: Option, pub circuit_breaker_threshold: Option, pub circuit_breaker_cool_down: Option, + + /// Upgrades the per-batch log line from `debug!` to `info!`. Mirrors the + /// runtime's own `verbose` flag; keep the field name `verbose_logging` + /// (see `postgres_sink::PostgresSinkConfig::verbose_logging`). + pub verbose_logging: Option, } /// Rejects anything that isn't a plain alphanumeric/underscore identifier. @@ -158,6 +168,7 @@ pub struct TemplateSink { circuit_breaker: Arc, batch_size_limit: usize, retry_delay: Duration, + verbose: bool, state: Mutex, records_written_total: AtomicU64, } @@ -175,6 +186,7 @@ impl TemplateSink { ), )); let batch_size_limit = config.batch_size.unwrap_or(DEFAULT_BATCH_SIZE).max(1); + let verbose = config.verbose_logging.unwrap_or(false); Self { id, @@ -183,6 +195,7 @@ impl TemplateSink { circuit_breaker, batch_size_limit, retry_delay, + verbose, state: Mutex::new(State::default()), records_written_total: AtomicU64::new(0), } @@ -213,7 +226,11 @@ impl TemplateSink { /// (network error, 5xx, timeout — should retry) by returning /// `Error::PermanentHttpError` for the former; see the doc comment on /// that variant in `iggy_connector_sdk::Error` for why the distinction - /// matters to the circuit breaker. + /// matters. `consume()` drops and counts a `PermanentHttpError` batch but + /// keeps processing the rest; any other error stops `consume()` and is + /// returned, which the runtime treats as fatal for the whole connector — + /// so a permanent classification is what keeps one bad message from + /// taking every future batch down with it. async fn push_batch( &self, client: &ClientWithMiddleware, @@ -321,18 +338,33 @@ impl Sink for TemplateSink { let invocation = state.invocations_count; drop(state); - info!( - "{CONNECTOR_NAME} with ID: {} received: {} messages, schema: {}, stream: {}, topic: {}, \ - partition: {}, offset: {}, invocation: {}", - self.id, - messages.len(), - messages_metadata.schema, - topic_metadata.stream, - topic_metadata.topic, - messages_metadata.partition_id, - messages_metadata.current_offset, - invocation - ); + if self.verbose { + info!( + "{CONNECTOR_NAME} with ID: {} received: {} messages, schema: {}, stream: {}, \ + topic: {}, partition: {}, offset: {}, invocation: {}", + self.id, + messages.len(), + messages_metadata.schema, + topic_metadata.stream, + topic_metadata.topic, + messages_metadata.partition_id, + messages_metadata.current_offset, + invocation + ); + } else { + debug!( + "{CONNECTOR_NAME} with ID: {} received: {} messages, schema: {}, stream: {}, \ + topic: {}, partition: {}, offset: {}, invocation: {}", + self.id, + messages.len(), + messages_metadata.schema, + topic_metadata.stream, + topic_metadata.topic, + messages_metadata.partition_id, + messages_metadata.current_offset, + invocation + ); + } if self.circuit_breaker.is_open().await { warn!( @@ -349,13 +381,32 @@ impl Sink for TemplateSink { Error::Connection("client not initialized -- was open() called?".into()) })?; - let mut first_error: Option = None; + // Track the *last* transient error, not the first — an earlier chunk's + // failure may already be stale by the time later chunks run, and the + // circuit breaker below should react to the most recent signal. A + // `PermanentHttpError` batch (bad schema, will never succeed on retry) + // is dropped and counted here rather than kept as `last_err`: letting + // it propagate would return `Err` from `consume()`, and the runtime + // treats any `Err` here as fatal for the whole connector (see + // `runtime/src/sink.rs::consume_messages`), not just a retry of that + // one batch — so a single unprocessable message would take down every + // future batch instead of just being skipped. + let mut last_err: Option = None; let mut written = 0u64; let mut failed = 0u64; for batch in messages.chunks(self.batch_size_limit) { match self.push_batch(client, batch).await { Ok(()) => written += batch.len() as u64, + Err(Error::PermanentHttpError(message)) => { + failed += batch.len() as u64; + error!( + "{CONNECTOR_NAME} connector with ID: {} dropping a batch of {} \ + (permanent): {message}", + self.id, + batch.len() + ); + } Err(err) => { failed += batch.len() as u64; error!( @@ -363,23 +414,24 @@ impl Sink for TemplateSink { self.id, batch.len() ); - if first_error.is_none() { - first_error = Some(err); - } + last_err = Some(err); } } } // Record the circuit breaker outcome once per `consume()` call, not - // once per chunk — recording success partway through would reset - // the failure counter mid-consume and prevent the breaker from - // opening on a batch with sustained, mixed-success chunks. - match &first_error { - None => self.circuit_breaker.record_success(), - Some(e) if !matches!(e, Error::PermanentHttpError(_)) => { - self.circuit_breaker.record_failure().await; - } - Some(_) => {} + // once per chunk — recording success partway through would reset the + // failure counter mid-consume and prevent the breaker from opening on + // a batch with sustained, mixed-success chunks. A transient failure + // always trips it. A batch where every failure was permanent records + // neither success nor failure — nothing demonstrated the destination + // is reachable, but a schema/data problem isn't a connectivity signal + // either. A batch with at least one real write (or no messages at + // all) counts as success. + match &last_err { + Some(_) => self.circuit_breaker.record_failure().await, + None if written > 0 || messages.is_empty() => self.circuit_breaker.record_success(), + None => {} } let mut state = self.state.lock().await; @@ -389,7 +441,7 @@ impl Sink for TemplateSink { self.records_written_total .fetch_add(written, Ordering::Relaxed); - match first_error { + match last_err { None => Ok(()), Some(err) => Err(err), } @@ -428,9 +480,24 @@ mod tests { open_retry_max_delay: Some("100ms".to_string()), circuit_breaker_threshold: Some(3), circuit_breaker_cool_down: Some("50ms".to_string()), + verbose_logging: None, } } + #[test] + fn given_verbose_logging_enabled_should_set_verbose_flag() { + let mut config = test_config(); + config.verbose_logging = Some(true); + let sink = TemplateSink::new(1, config); + assert!(sink.verbose); + } + + #[test] + fn given_verbose_logging_disabled_should_not_set_verbose_flag() { + let sink = TemplateSink::new(1, test_config()); + assert!(!sink.verbose); + } + #[tokio::test] async fn open_rejects_empty_connection_string() { let mut config = test_config(); @@ -486,19 +553,57 @@ mod tests { sink.circuit_breaker.record_failure().await; assert!(sink.circuit_breaker.is_open().await); - let topic_metadata = TopicMetadata { - stream: "s".to_string(), - topic: "t".to_string(), - }; - let messages_metadata = MessagesMetadata { - partition_id: 1, - current_offset: 0, - schema: iggy_connector_sdk::Schema::Json, - }; + let (topic_metadata, messages_metadata) = topic_and_messages_metadata(); let result = sink .consume(&topic_metadata, messages_metadata, Vec::new()) .await; assert!(matches!(result, Err(Error::CannotStoreData(_)))); } + + fn consumed_message() -> ConsumedMessage { + ConsumedMessage { + id: 1, + offset: 0, + checksum: 0, + timestamp: 0, + origin_timestamp: 0, + headers: None, + payload: iggy_connector_sdk::Payload::Raw(b"payload".to_vec()), + } + } + + fn topic_and_messages_metadata() -> (TopicMetadata, MessagesMetadata) { + ( + TopicMetadata { + stream: "s".to_string(), + topic: "t".to_string(), + }, + MessagesMetadata { + partition_id: 1, + current_offset: 0, + schema: iggy_connector_sdk::Schema::Json, + }, + ) + } + + // push_batch() is an unimplemented TODO(ConnectorDeveloper) stub that + // always returns Error::InitError — not a PermanentHttpError — so a + // non-empty consume() call exercises the "last_err" (retryable) branch + // of the chunking loop end to end, including the failed-count and + // circuit-breaker bookkeeping. + #[tokio::test] + async fn consume_with_unimplemented_push_batch_returns_err_and_counts_failure() { + let mut sink = TemplateSink::new(1, test_config()); + sink.open().await.expect("open should succeed"); + let (topic_metadata, messages_metadata) = topic_and_messages_metadata(); + + let result = sink + .consume(&topic_metadata, messages_metadata, vec![consumed_message()]) + .await; + + assert!(matches!(result, Err(Error::InitError(_)))); + assert_eq!(sink.state.lock().await.messages_failed, 1); + assert_eq!(sink.state.lock().await.messages_written, 0); + } } diff --git a/core/connectors/sources/source_template/Cargo.toml b/core/connectors/sources/source_template/Cargo.toml index 392504dacf..dd29dbacd7 100644 --- a/core/connectors/sources/source_template/Cargo.toml +++ b/core/connectors/sources/source_template/Cargo.toml @@ -1,23 +1,23 @@ -// Licensed to the Apache Software Foundation (ASF) under one -// or more contributor license agreements. See the NOTICE file -// distributed with this work for additional information -// regarding copyright ownership. The ASF licenses this file -// to you under the Apache License, Version 2.0 (the -// "License"); you may not use this file except in compliance -// with the License. You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, -// software distributed under the License is distributed on an -// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY -// KIND, either express or implied. See the License for the -// specific language governing permissions and limitations -// under the License. -// -//! TEMPLATE — rename the package (and this directory) to -//! `iggy_connector__source` before publishing, and update the -//! `[[sources]]` entry you add to the workspace root Cargo.toml accordingly. +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# TEMPLATE — rename the package (and this directory) to +# `iggy_connector__source` before publishing, and update the +# `[[sources]]` entry you add to the workspace root Cargo.toml accordingly. [package] name = "iggy_connector_template_source" diff --git a/core/connectors/sources/source_template/README.md b/core/connectors/sources/source_template/README.md index 415e2c210a..0ae77ac2cc 100644 --- a/core/connectors/sources/source_template/README.md +++ b/core/connectors/sources/source_template/README.md @@ -25,32 +25,38 @@ blog post for the checklist this template is built against. on `Nack` so a failed delivery gets re-polled instead of silently lost. - The `source_connector!` FFI macro invocation and a `Cargo.toml` with the right `crate-type`, workspace-pinned dependencies, and license header. -- Tests for config validation and the Ack/Nack state-commit behavior. +- `verbose_logging: Option` upgrading the per-poll log line from + `debug!` to `info!`, mirroring the runtime's own `verbose` flag. +- The six canonical state/Ack-Nack tests, plus config validation and the + `verbose_logging` flag. ## What you need to fill in -Search for `TODO(ConnectorDeveloper)` in `src/lib.rs` — there are exactly two spots: +Search for `TODO(ConnectorDeveloper)` in `src/lib.rs` — there are exactly +three spots: one required, one conditional, one documentation-only. -1. **`build_raw_client()`** — if your source isn't HTTP, replace the +1. **`fetch_records()`** (required) — fetch up to `self.batch_size` new + records from your system, ordered after `cursor` (`None` = start from the + beginning, or from "now" — whichever fits your source). Map each result + to a `FetchedRecord { cursor_value, payload }`, using something + monotonically increasing as `cursor_value` (a timestamp, an ID, a page + token) — that's what lets the cursor-staging logic advance correctly. +2. **`build_raw_client()`** (only if your source isn't HTTP) — replace the `reqwest::Client` construction with your driver's connection/pool setup (see `core/connectors/sources/postgres_source` for a real non-HTTP example), store it on `TemplateSource` (you'll need to add a field — `client: Option` here is HTTP-specific), and adjust or remove the `check_connectivity_with_retry` call in `open()` in favor of whatever connectivity check your driver offers. -2. **`fetch_records()`** — fetch up to `self.batch_size` new records from - your system, ordered after `cursor` (`None` = start from the beginning, - or from "now" — whichever fits your source). Map each result to a - `FetchedRecord { cursor_value, payload }`, using something monotonically - increasing as `cursor_value` (a timestamp, an ID, a page token) — that's - what lets the cursor-staging logic advance correctly. +3. **`connection_string`'s doc comment** (documentation only) — describe + the exact shape your connector expects instead of the generic example. ## Using it 1. Copy this directory, rename it and the package in `Cargo.toml` (`iggy_connector__source`), and add it to the `members` list in the workspace root `Cargo.toml`. -2. Fill in the two `TODO(ConnectorDeveloper)` sections. +2. Fill in `fetch_records()` (and `build_raw_client()` if not HTTP). 3. Update `config.toml` with your real `connection_string` and any settings specific to your system; delete `auth_token` if you don't need it, or add fields of your own the same way (see `TemplateSourceConfig`). diff --git a/core/connectors/sources/source_template/config.toml b/core/connectors/sources/source_template/config.toml index 53ac14874e..721abb0464 100644 --- a/core/connectors/sources/source_template/config.toml +++ b/core/connectors/sources/source_template/config.toml @@ -44,4 +44,5 @@ max_open_retries = 10 open_retry_max_delay = "60s" circuit_breaker_threshold = 5 circuit_breaker_cool_down = "30s" +verbose_logging = false # auth_token = "replace-me" # uncomment if your source needs bearer/API-key auth diff --git a/core/connectors/sources/source_template/src/lib.rs b/core/connectors/sources/source_template/src/lib.rs index 57a2c576b0..9191f0fc2d 100644 --- a/core/connectors/sources/source_template/src/lib.rs +++ b/core/connectors/sources/source_template/src/lib.rs @@ -24,18 +24,20 @@ //! staged in `poll()` and only committed in `on_batch_result()` on an ACK so //! a dropped/nacked batch can be re-polled instead of silently lost. //! -//! There are exactly **two** places you need to touch, each marked -//! `TODO(ConnectorDeveloper)`: -//! 1. `TemplateSource::connect()` — build your actual client/connection -//! from `config.connection_string` (and `config.auth_token`, if used). -//! 2. `TemplateSource::fetch_records()` — fetch up to `batch_size` new -//! records from your external system, starting after `cursor`. +//! There is exactly **one** place you need to write code, marked +//! `TODO(ConnectorDeveloper)`: `TemplateSource::fetch_records()` — fetch up +//! to `batch_size` new records from your external system, starting after +//! `cursor`. `TemplateSource::build_raw_client()` carries a second +//! `TODO(ConnectorDeveloper)` too, but only applies if your source isn't +//! HTTP (see below); a third, on the `connection_string` field's doc +//! comment, just asks you to describe its expected shape — not code. +//! `grep` for the marker and you'll find all three. //! //! This template assumes an HTTP-ish source and uses `reqwest` wrapped by //! the SDK's retry middleware, because that's what `iggy_connector_sdk::retry` //! is built for and it covers the common case. If your source talks to //! something else (a database, a queue, a filesystem), swap the client type -//! in `connect()`/`fetch_records()` for your driver of choice and lean on +//! in `build_raw_client()`/`fetch_records()` for your driver of choice and lean on //! its own retry/pooling behavior — keep the surrounding shape (validation //! in `open()`, circuit breaker, cursor staging, batching) unchanged. See //! `core/connectors/sources/postgres_source` in this repo for a real @@ -58,7 +60,7 @@ use std::sync::Arc; use std::sync::atomic::{AtomicU64, Ordering}; use std::time::Duration; use tokio::sync::Mutex; -use tracing::{error, info, warn}; +use tracing::{debug, error, info, warn}; source_connector!(TemplateSource); @@ -96,7 +98,7 @@ pub struct TemplateSourceConfig { /// `Debug`/log output; delete this field if `connection_string` already /// carries all required auth. Read it with `.expose_secret()` (from the /// `secrecy::ExposeSecret` trait) at the one place you actually need the - /// plaintext — e.g. when building an auth header in `connect()`. + /// plaintext — e.g. when building an auth header in `build_raw_client()`. #[serde( default, serialize_with = "iggy_common::serde_secret::serialize_optional_secret" @@ -118,6 +120,11 @@ pub struct TemplateSourceConfig { pub open_retry_max_delay: Option, pub circuit_breaker_threshold: Option, pub circuit_breaker_cool_down: Option, + + /// Upgrades the per-poll log line from `debug!` to `info!`. Mirrors the + /// runtime's own `verbose` flag; keep the field name `verbose_logging` + /// (see `postgres_source::PostgresSourceConfig::verbose_logging`). + pub verbose_logging: Option, } // ── Internal state ────────────────────────────────────────────────────────── @@ -149,6 +156,7 @@ pub struct TemplateSource { batch_size: u32, poll_interval: Duration, retry_delay: Duration, + verbose: bool, state: Mutex, pending_state: Mutex>, records_produced: AtomicU64, @@ -173,6 +181,7 @@ impl TemplateSource { ), )); let batch_size = config.batch_size.unwrap_or(DEFAULT_BATCH_SIZE).max(1); + let verbose = config.verbose_logging.unwrap_or(false); let restored_state = state .and_then(|s| s.deserialize::(CONNECTOR_NAME, id)) @@ -191,6 +200,7 @@ impl TemplateSource { batch_size, poll_interval, retry_delay, + verbose, state: Mutex::new(restored_state.unwrap_or_default()), pending_state: Mutex::new(None), records_produced: AtomicU64::new(0), @@ -371,7 +381,6 @@ impl Source for TemplateSource { let mut messages = Vec::with_capacity(records.len()); let mut new_cursor = cursor; for record in records { - new_cursor = Some(record.cursor_value); let Ok(payload) = serde_json::to_vec(&record.payload) else { error!( "Failed to serialize a record fetched by {CONNECTOR_NAME} connector with ID: {}", @@ -379,6 +388,11 @@ impl Source for TemplateSource { ); continue; }; + // Only advance the candidate cursor once the record actually made it + // into `messages` - advancing on a dropped (unserializable) record + // would stage a cursor past data that was never produced, and a + // later Ack would commit past it permanently. + new_cursor = Some(record.cursor_value); messages.push(ProducedMessage { id: None, headers: None, @@ -389,6 +403,17 @@ impl Source for TemplateSource { }); } + if messages.is_empty() { + // Every fetched record failed to serialize - no progress was + // made, so there's nothing to stage and no reason to write + // state, same as the `records.is_empty()` case above. + return Ok(ProducedMessages { + schema: Schema::Json, + messages: Vec::new(), + state: None, + }); + } + let candidate_state = State { cursor: new_cursor }; let persisted_state = self.serialize_state(&candidate_state).ok_or_else(|| { Error::Serialization(format!( @@ -401,6 +426,20 @@ impl Source for TemplateSource { self.records_produced .fetch_add(messages.len() as u64, Ordering::Relaxed); + if self.verbose { + info!( + "{CONNECTOR_NAME} connector with ID: {} produced {} messages", + self.id, + messages.len() + ); + } else { + debug!( + "{CONNECTOR_NAME} connector with ID: {} produced {} messages", + self.id, + messages.len() + ); + } + Ok(ProducedMessages { schema: Schema::Json, messages, @@ -477,9 +516,24 @@ mod tests { open_retry_max_delay: Some("100ms".to_string()), circuit_breaker_threshold: Some(3), circuit_breaker_cool_down: Some("50ms".to_string()), + verbose_logging: None, } } + #[test] + fn given_verbose_logging_enabled_should_set_verbose_flag() { + let mut config = test_config(); + config.verbose_logging = Some(true); + let source = TemplateSource::new(1, config, None); + assert!(source.verbose); + } + + #[test] + fn given_verbose_logging_disabled_should_not_set_verbose_flag() { + let source = TemplateSource::new(1, test_config(), None); + assert!(!source.verbose); + } + #[tokio::test] async fn open_rejects_empty_connection_string() { let mut config = test_config(); @@ -568,7 +622,7 @@ mod tests { assert!(source.pending_state.lock().await.is_none()); } - #[tokio::test] + #[tokio::test(start_paused = true)] async fn poll_returns_empty_without_error_when_circuit_is_open() { let source = TemplateSource::new(1, test_config(), None); source.circuit_breaker.record_failure().await; From 9ab9c50164fa5c8dd3bb9916f95e866567796227 Mon Sep 17 00:00:00 2001 From: ryerraguntla Date: Sat, 29 Aug 2026 12:00:49 +0530 Subject: [PATCH 006/182] Fixing cargo machete --- Cargo.lock | 1 - core/connectors/sinks/sink_template/Cargo.toml | 1 - 2 files changed, 2 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index fa3a3efb13..af69db4cdd 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -7238,7 +7238,6 @@ dependencies = [ "reqwest-middleware", "secrecy", "serde", - "serde_json", "tokio", "tracing", ] diff --git a/core/connectors/sinks/sink_template/Cargo.toml b/core/connectors/sinks/sink_template/Cargo.toml index 6ee035f63b..886962dc6d 100644 --- a/core/connectors/sinks/sink_template/Cargo.toml +++ b/core/connectors/sinks/sink_template/Cargo.toml @@ -50,7 +50,6 @@ reqwest = { workspace = true } reqwest-middleware = { workspace = true } secrecy = { workspace = true } serde = { workspace = true } -serde_json = { workspace = true } tokio = { workspace = true } tracing = { workspace = true } From 6986980a26e42a6d4a32b0d52c8ec9a088ce8062 Mon Sep 17 00:00:00 2001 From: ryerraguntla Date: Sat, 29 Aug 2026 12:42:35 +0530 Subject: [PATCH 007/182] making clickable url --- core/connectors/sinks/sink_template/src/lib.rs | 2 +- core/connectors/sources/source_template/src/lib.rs | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/core/connectors/sinks/sink_template/src/lib.rs b/core/connectors/sinks/sink_template/src/lib.rs index f1ad735068..47bfa83676 100644 --- a/core/connectors/sinks/sink_template/src/lib.rs +++ b/core/connectors/sinks/sink_template/src/lib.rs @@ -89,7 +89,7 @@ const DEFAULT_CIRCUIT_BREAKER_COOL_DOWN: &str = "30s"; #[serde(deny_unknown_fields)] pub struct TemplateSinkConfig { /// TODO(ConnectorDeveloper): document the exact shape this connector expects, e.g. - /// "https://api.example.com" or "postgres://user:pass@host:5432/db". + /// "" or "postgres://user:pass@host:5432/db". /// `SecretString` because DSNs commonly embed credentials — never plain /// `String` for this field, see `PostgresSinkConfig::connection_string` /// in `sinks/postgres_sink` for the same pattern. diff --git a/core/connectors/sources/source_template/src/lib.rs b/core/connectors/sources/source_template/src/lib.rs index 9191f0fc2d..978811e3d1 100644 --- a/core/connectors/sources/source_template/src/lib.rs +++ b/core/connectors/sources/source_template/src/lib.rs @@ -87,7 +87,7 @@ const DEFAULT_CIRCUIT_BREAKER_COOL_DOWN: &str = "30s"; #[serde(deny_unknown_fields)] pub struct TemplateSourceConfig { /// TODO(ConnectorDeveloper): document the exact shape this connector expects, e.g. - /// "https://api.example.com" or "postgres://user:pass@host:5432/db". + /// "" or "postgres://user:pass@host:5432/db". /// `SecretString` because DSNs commonly embed credentials — never plain /// `String` for this field, see `PostgresSinkConfig::connection_string` /// in `sinks/postgres_sink` for the same pattern. From 162a1f1cd0986968372cd4c5dafd97d66377510b Mon Sep 17 00:00:00 2001 From: ryerraguntla Date: Sat, 29 Aug 2026 12:55:57 +0530 Subject: [PATCH 008/182] update lib.rs to retrigger prechecks build --- core/connectors/sources/source_template/src/lib.rs | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/core/connectors/sources/source_template/src/lib.rs b/core/connectors/sources/source_template/src/lib.rs index 978811e3d1..0b6af85715 100644 --- a/core/connectors/sources/source_template/src/lib.rs +++ b/core/connectors/sources/source_template/src/lib.rs @@ -45,8 +45,8 @@ use async_trait::async_trait; use iggy_connector_sdk::retry::{ - CircuitBreaker, ConnectivityConfig, build_retry_client, check_connectivity_with_retry, - parse_duration, + CircuitBreaker, ConnectivityConfig, build_retry_client, + check_connectivity_with_retry, parse_duration, }; use iggy_connector_sdk::{ ConnectorState, Error, ProducedMessage, ProducedMessages, Schema, Source, From 1d5d2809e438a8bd95f341d849b630760cdfe257 Mon Sep 17 00:00:00 2001 From: ryerraguntla Date: Sat, 29 Aug 2026 13:30:31 +0530 Subject: [PATCH 009/182] Removing trailing space --- core/connectors/sources/source_template/src/lib.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/core/connectors/sources/source_template/src/lib.rs b/core/connectors/sources/source_template/src/lib.rs index 0b6af85715..7421799d31 100644 --- a/core/connectors/sources/source_template/src/lib.rs +++ b/core/connectors/sources/source_template/src/lib.rs @@ -45,7 +45,7 @@ use async_trait::async_trait; use iggy_connector_sdk::retry::{ - CircuitBreaker, ConnectivityConfig, build_retry_client, + CircuitBreaker, ConnectivityConfig, build_retry_client, check_connectivity_with_retry, parse_duration, }; use iggy_connector_sdk::{ From e63f9e88134aac1fdb71a3994263ac22a91462a9 Mon Sep 17 00:00:00 2001 From: ryerraguntla Date: Sat, 29 Aug 2026 13:32:46 +0530 Subject: [PATCH 010/182] Update lib.rs --- core/connectors/sources/source_template/src/lib.rs | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/core/connectors/sources/source_template/src/lib.rs b/core/connectors/sources/source_template/src/lib.rs index 7421799d31..978811e3d1 100644 --- a/core/connectors/sources/source_template/src/lib.rs +++ b/core/connectors/sources/source_template/src/lib.rs @@ -45,8 +45,8 @@ use async_trait::async_trait; use iggy_connector_sdk::retry::{ - CircuitBreaker, ConnectivityConfig, build_retry_client, - check_connectivity_with_retry, parse_duration, + CircuitBreaker, ConnectivityConfig, build_retry_client, check_connectivity_with_retry, + parse_duration, }; use iggy_connector_sdk::{ ConnectorState, Error, ProducedMessage, ProducedMessages, Schema, Source, From 5916e6f586491c8b431be1a844daee73856921cd Mon Sep 17 00:00:00 2001 From: Hubert Gruszecki Date: Sun, 30 Aug 2026 17:44:08 +0200 Subject: [PATCH 011/182] feat(partitions): persist log and index concurrently, harden recovery (#3970) Every journal flush issued two serialized fdatasyncs, log then index, so an fsync-gated topic paid two device round trips per ack. Both files now persist under one futures::future::join: 13.4k to 26.1k msg/s and p50 5.83 to 3.01 ms on ext4 with enforce_fsync and messages_required_to_save = 1. Cursors advance only after both saves succeed, so a failed half keeps its slot and a later flush rewrites the same positions instead of appending a duplicate. Removing the write barrier forced a new recovery contract. An index entry is derived from the log and, without enforce_fsync, proves nothing about it, so recovery checksum-walks every segment from byte 0 and keeps a clean index untouched. Every entry must match its log batch's offset, timestamp, partition and extent, or the index is rebuilt: ascending columns alone let a corrupt entry point at the wrong valid batch and polls silently skipped offsets. Under enforce_fsync serialized flushes prove the prefix below the last entry, so a healthy boot still anchors there; a gap deeper than the one in-flight chunk refuses as FsyncedLogLoss, and a rebuild stopping below the last entry's position refuses as FsyncedRebuildShortfall rather than truncating bytes a completed flush made durable. Probe-budget exhaustion always refuses, since recovering as empty would serve the segment and re-mint offsets over bytes never scanned. Previously any index the log outran refused boot outright and tombstoned single-replica partitions with healthy logs. A persist failure on a cluster-committed op panicked on the shard pump, where compio::runtime::spawn swallows the unwind: the process stayed up with every partition on that shard dead. The partition now fences itself, the pump breaks, flips the shared shutdown flag before its final flush (the flush writes to the failed device and can stall forever; the flag is what arms the watchdog and the bounded drain), and a failed shutdown flush fences its partition too, so the exit is ShardFatal and non-zero rather than a clean report over unpersisted committed data. Also: the active segment fills the same read-state slot sealed segments use, ending one openat per head poll; latency extremes backfill only as a whole group; segment writer save methods are crate-private; index scans yield on a stride. --- core/bench/report/src/types/group_metrics.rs | 28 +- .../report/src/types/individual_metrics.rs | 94 +- core/bench/report/src/utils.rs | 30 +- core/bench/src/analytics/metrics/group.rs | 48 +- .../bench/src/analytics/metrics/individual.rs | 65 +- core/bench/src/args/common.rs | 17 +- core/bench/src/args/examples.rs | 6 + .../bench/src/args/kinds/balanced/producer.rs | 19 +- core/bench/src/benchmarks/benchmark.rs | 13 +- .../cluster/crash_recovery_corruption.rs | 269 ++ core/partitions/src/iggy_index.rs | 6 + core/partitions/src/iggy_index_writer.rs | 24 +- core/partitions/src/iggy_partition.rs | 583 +++- core/partitions/src/lib.rs | 6 +- core/partitions/src/log.rs | 74 +- core/partitions/src/messages_writer.rs | 29 +- core/partitions/src/poll_plan.rs | 72 +- core/partitions/src/state_transfer.rs | 8 + core/partitions/src/types.rs | 19 + core/server/config.toml | 77 +- core/server/src/bootstrap.rs | 78 +- core/server/src/dispatch.rs | 9 +- core/server/src/segment_recovery.rs | 2566 ++++++++++++++--- core/server/src/server_error.rs | 133 +- core/shard/src/lib.rs | 32 +- core/shard/src/router.rs | 110 +- core/simulator/src/lib.rs | 14 +- 27 files changed, 3687 insertions(+), 742 deletions(-) diff --git a/core/bench/report/src/types/group_metrics.rs b/core/bench/report/src/types/group_metrics.rs index e1a8c3eb87..5653005cc7 100644 --- a/core/bench/report/src/types/group_metrics.rs +++ b/core/bench/report/src/types/group_metrics.rs @@ -92,23 +92,21 @@ impl<'de> Deserialize<'de> for BenchmarkGroupMetrics { let mut updated_summary = summary.clone(); - // Calculate and populate missing statistics from the time series data + // Backfill for reports written before the three fields + // existed: `serde(default)` lands them at zero together, so + // all three being zero is what identifies such a report. + // Keyed on all three rather than on each alone -- a run whose + // samples are all equal reports a real zero std dev beside a + // nonzero min and max, and refilling that from the per-bucket + // moving average would put back the averaged-away extremes + // the summary is computed from raw samples to avoid. if updated_summary.min_latency_ms == 0.0 - && let Some(min_val) = min(&avg_latency_ts) + && updated_summary.max_latency_ms == 0.0 + && updated_summary.std_dev_latency_ms == 0.0 { - updated_summary.min_latency_ms = min_val; - } - - if updated_summary.max_latency_ms == 0.0 - && let Some(max_val) = max(&avg_latency_ts) - { - updated_summary.max_latency_ms = max_val; - } - - if updated_summary.std_dev_latency_ms == 0.0 - && let Some(std_dev_val) = std_dev(&avg_latency_ts) - { - updated_summary.std_dev_latency_ms = std_dev_val; + updated_summary.min_latency_ms = min(&avg_latency_ts).unwrap_or(0.0); + updated_summary.max_latency_ms = max(&avg_latency_ts).unwrap_or(0.0); + updated_summary.std_dev_latency_ms = std_dev(&avg_latency_ts).unwrap_or(0.0); } Ok(BenchmarkGroupMetrics { diff --git a/core/bench/report/src/types/individual_metrics.rs b/core/bench/report/src/types/individual_metrics.rs index ed3b302d09..d1290ab0c9 100644 --- a/core/bench/report/src/types/individual_metrics.rs +++ b/core/bench/report/src/types/individual_metrics.rs @@ -89,22 +89,21 @@ impl<'de> Deserialize<'de> for BenchmarkIndividualMetrics { let mut updated_summary = summary.clone(); + // Backfill for reports written before the three fields + // existed: `serde(default)` lands them at zero together, so + // all three being zero is what identifies such a report. + // Keyed on all three rather than on each alone -- a run whose + // samples are all equal reports a real zero std dev beside a + // nonzero min and max, and refilling that from the per-bucket + // moving average would put back the averaged-away extremes + // the summary is computed from raw samples to avoid. if updated_summary.min_latency_ms == 0.0 - && let Some(min_val) = min(&latency_ts) + && updated_summary.max_latency_ms == 0.0 + && updated_summary.std_dev_latency_ms == 0.0 { - updated_summary.min_latency_ms = min_val; - } - - if updated_summary.max_latency_ms == 0.0 - && let Some(max_val) = max(&latency_ts) - { - updated_summary.max_latency_ms = max_val; - } - - if updated_summary.std_dev_latency_ms == 0.0 - && let Some(std_dev_val) = std_dev(&latency_ts) - { - updated_summary.std_dev_latency_ms = std_dev_val; + updated_summary.min_latency_ms = min(&latency_ts).unwrap_or(0.0); + updated_summary.max_latency_ms = max(&latency_ts).unwrap_or(0.0); + updated_summary.std_dev_latency_ms = std_dev(&latency_ts).unwrap_or(0.0); } Ok(BenchmarkIndividualMetrics { @@ -121,3 +120,70 @@ impl<'de> Deserialize<'de> for BenchmarkIndividualMetrics { deserializer.deserialize_map(BenchmarkIndividualMetricsVisitor) } } + +#[cfg(test)] +mod tests { + use super::*; + + /// Report JSON carrying `summary` and the three time series, with the + /// summary's latency extremes spliced in as `extremes`. + fn report_json(extremes: &str) -> String { + format!( + r#"{{ + "summary": {{ + "benchmark_kind": "pinned_producer", + "actor_kind": "producer", + "actor_id": 0, + "total_time_secs": 1.0, + "total_user_data_bytes": 1000, + "total_bytes": 1000, + "total_messages": 3, + "total_message_batches": 3, + "throughput_megabytes_per_second": 1.0, + "throughput_messages_per_second": 3.0, + "p50_latency_ms": 5.0, + "p90_latency_ms": 5.0, + "p95_latency_ms": 5.0, + "p99_latency_ms": 5.0, + "p999_latency_ms": 5.0, + "p9999_latency_ms": 5.0, + "avg_latency_ms": 5.0, + "median_latency_ms": 5.0{extremes} + }}, + "throughput_mb_ts": {{ "points": [] }}, + "throughput_msg_ts": {{ "points": [] }}, + "latency_ts": {{ "points": [ + {{ "time_s": 0.0, "value": 4.0 }}, + {{ "time_s": 1.0, "value": 6.0 }} + ] }} + }}"# + ) + } + + #[test] + fn given_a_report_without_latency_extremes_when_deserialized_should_backfill_from_the_series() { + let metrics: BenchmarkIndividualMetrics = + serde_json::from_str(&report_json("")).expect("deserialize legacy report"); + + assert!((metrics.summary.min_latency_ms - 4.0).abs() < f64::EPSILON); + assert!((metrics.summary.max_latency_ms - 6.0).abs() < f64::EPSILON); + assert!(metrics.summary.std_dev_latency_ms > 0.0); + } + + /// A run whose samples are all equal reports a real zero std dev beside a + /// nonzero min and max. Backfilling it from the per-bucket moving average + /// would put back the averaged-away extremes the summary avoids. + #[test] + fn given_a_report_with_a_real_zero_std_dev_when_deserialized_should_keep_it() { + let json = report_json( + r#","min_latency_ms": 5.0, "max_latency_ms": 5.0, "std_dev_latency_ms": 0.0"#, + ); + + let metrics: BenchmarkIndividualMetrics = + serde_json::from_str(&json).expect("deserialize report"); + + assert!((metrics.summary.min_latency_ms - 5.0).abs() < f64::EPSILON); + assert!((metrics.summary.max_latency_ms - 5.0).abs() < f64::EPSILON); + assert_eq!(metrics.summary.std_dev_latency_ms, 0.0); + } +} diff --git a/core/bench/report/src/utils.rs b/core/bench/report/src/utils.rs index 6189ed4b87..3cd17a1587 100644 --- a/core/bench/report/src/utils.rs +++ b/core/bench/report/src/utils.rs @@ -116,32 +116,38 @@ pub fn lttb_downsample(points: &[TimePoint], threshold: usize) -> Vec result } -/// Calculate the standard deviation of values from a TimeSeries +/// Calculate the population standard deviation of `values` /// -/// Returns None if the TimeSeries has fewer than 2 points -pub fn std_dev(series: &TimeSeries) -> Option { - let points_count = series.points.len(); +/// Returns None if there are fewer than 2 values +pub fn std_dev_values(values: &[f64]) -> Option { + let count = values.len(); - if points_count < 2 { + if count < 2 { return None; } - let sum: f64 = series.points.iter().map(|p| p.value).sum(); - let mean = sum / points_count as f64; + let mean = values.iter().sum::() / count as f64; - let variance = series - .points + let variance = values .iter() - .map(|p| { - let diff = p.value - mean; + .map(|value| { + let diff = value - mean; diff * diff }) .sum::() - / points_count as f64; + / count as f64; Some(variance.sqrt()) } +/// Calculate the standard deviation of values from a TimeSeries +/// +/// Returns None if the TimeSeries has fewer than 2 points +pub fn std_dev(series: &TimeSeries) -> Option { + let values: Vec = series.points.iter().map(|p| p.value).collect(); + std_dev_values(&values) +} + #[cfg(test)] mod tests { use super::*; diff --git a/core/bench/src/analytics/metrics/group.rs b/core/bench/src/analytics/metrics/group.rs index 90364bb2a4..ae5672c32b 100644 --- a/core/bench/src/analytics/metrics/group.rs +++ b/core/bench/src/analytics/metrics/group.rs @@ -30,8 +30,9 @@ use bench_report::{ group_metrics_summary::BenchmarkGroupMetricsSummary, individual_metrics::BenchmarkIndividualMetrics, time_series::{TimeSeries, TimeSeriesKind}, - utils::{max, min, std_dev}, + utils::std_dev_values, }; +use std::cmp::Ordering; use std::thread; pub fn from_producers_and_consumers_statistics( @@ -59,8 +60,7 @@ pub fn from_individual_metrics( let latency_metrics = calculate_latency_metrics(stats); let kind = determine_group_kind(stats); let time_series = calculate_group_time_series(stats, moving_average_window); - let (min_latency_ms_value, max_latency_ms_value) = - calculate_min_max_latencies(stats, &time_series.2); + let (min_latency_ms, max_latency_ms) = calculate_min_max_latencies(stats); let mut all_latencies: Vec = stats .iter() @@ -88,9 +88,9 @@ pub fn from_individual_metrics( average_p9999_latency_ms: latency_metrics.p9999_latency, average_latency_ms: latency_metrics.average_latency, average_median_latency_ms: latency_metrics.median_latency, - min_latency_ms: min_latency_ms_value, - max_latency_ms: max_latency_ms_value, - std_dev_latency_ms: std_dev(&time_series.2).unwrap_or(0.0), + min_latency_ms, + max_latency_ms, + std_dev_latency_ms: std_dev_values(&all_latencies).unwrap_or(0.0), }; Some(BenchmarkGroupMetrics { @@ -235,30 +235,18 @@ fn extract_time_series_of_kind( ts_vec.swap_remove(position) } -fn calculate_min_max_latencies( - stats: &[BenchmarkIndividualMetrics], - avg_latency_ts: &TimeSeries, -) -> (f64, f64) { - let min_latency_ms = if stats.is_empty() { - None - } else { - stats - .iter() - .map(|s| s.summary.min_latency_ms) - .min_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal)) - }; - - let max_latency_ms = if stats.is_empty() { - None - } else { - stats - .iter() - .map(|s| s.summary.max_latency_ms) - .max_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal)) - }; +fn calculate_min_max_latencies(stats: &[BenchmarkIndividualMetrics]) -> (f64, f64) { + let min_latency_ms = stats + .iter() + .map(|s| s.summary.min_latency_ms) + .min_by(|a, b| a.partial_cmp(b).unwrap_or(Ordering::Equal)) + .unwrap_or(0.0); - let min_latency_ms_value = min_latency_ms.unwrap_or_else(|| min(avg_latency_ts).unwrap_or(0.0)); - let max_latency_ms_value = max_latency_ms.unwrap_or_else(|| max(avg_latency_ts).unwrap_or(0.0)); + let max_latency_ms = stats + .iter() + .map(|s| s.summary.max_latency_ms) + .max_by(|a, b| a.partial_cmp(b).unwrap_or(Ordering::Equal)) + .unwrap_or(0.0); - (min_latency_ms_value, max_latency_ms_value) + (min_latency_ms, max_latency_ms) } diff --git a/core/bench/src/analytics/metrics/individual.rs b/core/bench/src/analytics/metrics/individual.rs index be25e01748..5d46dec822 100644 --- a/core/bench/src/analytics/metrics/individual.rs +++ b/core/bench/src/analytics/metrics/individual.rs @@ -28,7 +28,7 @@ use bench_report::benchmark_kind::BenchmarkKind; use bench_report::individual_metrics::BenchmarkIndividualMetrics; use bench_report::individual_metrics_summary::BenchmarkIndividualMetricsSummary; use bench_report::time_series::TimeSeries; -use bench_report::utils::{max, min, std_dev}; +use bench_report::utils::std_dev_values; use iggy::prelude::IggyDuration; pub fn from_records( @@ -68,7 +68,7 @@ pub fn from_records( .collect(); raw_latencies_ms.sort_unstable_by(|a, b| a.partial_cmp(b).unwrap()); - let latency_metrics = calculate_latency_metrics(&raw_latencies_ms, &latency_ts); + let latency_metrics = calculate_latency_metrics(&raw_latencies_ms); BenchmarkIndividualMetrics { summary: BenchmarkIndividualMetricsSummary { @@ -219,10 +219,7 @@ struct LatencyMetrics { std_dev: f64, } -fn calculate_latency_metrics( - sorted_latencies_ms: &[f64], - latency_ts: &TimeSeries, -) -> LatencyMetrics { +fn calculate_latency_metrics(sorted_latencies_ms: &[f64]) -> LatencyMetrics { let p50 = calculate_percentile(sorted_latencies_ms, 50.0); let p90 = calculate_percentile(sorted_latencies_ms, 90.0); let p95 = calculate_percentile(sorted_latencies_ms, 95.0); @@ -238,9 +235,11 @@ fn calculate_latency_metrics( sorted_latencies_ms[len] }; - let min = min(latency_ts).unwrap_or(0.0); - let max = max(latency_ts).unwrap_or(0.0); - let std_dev = std_dev(latency_ts).unwrap_or(0.0); + // Taken from raw samples, not from `latency_ts`: that series is averaged + // per sampling bucket, which pulls the extremes inside the percentiles. + let min = sorted_latencies_ms[0]; + let max = sorted_latencies_ms[sorted_latencies_ms.len() - 1]; + let std_dev = std_dev_values(sorted_latencies_ms).unwrap_or(0.0); LatencyMetrics { p50, @@ -273,3 +272,51 @@ pub fn calculate_percentile(sorted_data: &[f64], percentile: f64) -> f64 { let weight = rank - lower as f64; sorted_data[lower].mul_add(1.0 - weight, sorted_data[upper] * weight) } + +#[cfg(test)] +mod tests { + use super::*; + use std::str::FromStr; + + fn record(elapsed_time_us: u64, latency_us: u64) -> BenchmarkRecord { + BenchmarkRecord { + elapsed_time_us, + latency_us, + messages: 1, + message_batches: 1, + user_data_bytes: 1_000, + total_bytes: 1_000, + } + } + + #[test] + fn given_latencies_averaged_into_one_bucket_when_building_metrics_should_report_raw_extremes() { + let records = [ + record(1_000, 1_000), + record(2_000, 2_000), + record(3_000, 10_000), + ]; + + let metrics = from_records( + &records, + BenchmarkKind::PinnedProducer, + ActorKind::Producer, + 0, + IggyDuration::from_str("1s").unwrap(), + 20, + ); + + // A single sampling bucket holds every record, so the series carries + // only their average and hides both the 1 ms and the 10 ms sample. + assert_eq!(metrics.latency_ts.points.len(), 1); + assert!((metrics.latency_ts.points[0].value - 4.333).abs() < 1e-9); + + let summary = &metrics.summary; + assert!((summary.min_latency_ms - 1.0).abs() < f64::EPSILON); + assert!((summary.max_latency_ms - 10.0).abs() < f64::EPSILON); + // Population std dev of [1, 2, 10]; the one-point series would give 0. + assert!((summary.std_dev_latency_ms - 4.027_682).abs() < 1e-6); + assert!(summary.min_latency_ms <= summary.p50_latency_ms); + assert!(summary.max_latency_ms >= summary.p9999_latency_ms); + } +} diff --git a/core/bench/src/args/common.rs b/core/bench/src/args/common.rs index a7c5f9f7d6..c11a32a8ea 100644 --- a/core/bench/src/args/common.rs +++ b/core/bench/src/args/common.rs @@ -101,11 +101,18 @@ pub struct IggyBenchArgs { #[arg(long, default_value_t = false)] pub reuse_streams: bool, - /// Fsync every write on the benchmark topic instead of leaving it to the - /// page cache. Set as a topic option at creation, so it has no effect with - /// `--reuse-streams` against an already-created topic. + /// Fsync each journal flush on the benchmark topic. Flush timing stays + /// governed by `--messages-required-to-save` (server default: 1024), so + /// acks are durability-gated only with `--messages-required-to-save 1`. + /// Topic option at creation, so it has no effect with `--reuse-streams`. #[arg(long, default_value_t = false)] pub enforce_fsync: bool, + + /// Topic journal flush threshold in messages (server default: 1024). + /// With `--enforce-fsync`, `1` makes every produce ack wait for the fsync. + /// Topic option at creation, so it has no effect with `--reuse-streams`. + #[arg(long)] + pub messages_required_to_save: Option, } impl IggyBenchArgs { @@ -337,6 +344,10 @@ impl IggyBenchArgs { self.enforce_fsync } + pub const fn messages_required_to_save(&self) -> Option { + self.messages_required_to_save + } + pub fn username(&self) -> &str { &self.username } diff --git a/core/bench/src/args/examples.rs b/core/bench/src/args/examples.rs index 7866eb41e8..8602085a7e 100644 --- a/core/bench/src/args/examples.rs +++ b/core/bench/src/args/examples.rs @@ -37,6 +37,12 @@ const EXAMPLES: &str = r#"EXAMPLES: $ cargo r -r --bin iggy-bench -- balanced-producer-and-consumer-group --partitions 24 --producers 6 --consumers 6 tcp $ cargo r -r --bin iggy-bench -- -T 10GB bpc tcp + Durability-matched run, where every produce ack waits for an fsync + (--partitions 1 routes every producer straight to that partition): + + $ cargo r -r --bin iggy-bench -- --enforce-fsync --messages-required-to-save 1 \ + balanced-producer --partitions 1 --producers 8 tcp + 3) End-to-End Benchmarking: Run end-to-end benchmarks that measure performance for a producer that is also a consumer: diff --git a/core/bench/src/args/kinds/balanced/producer.rs b/core/bench/src/args/kinds/balanced/producer.rs index 9657759b96..e693df656a 100644 --- a/core/bench/src/args/kinds/balanced/producer.rs +++ b/core/bench/src/args/kinds/balanced/producer.rs @@ -16,7 +16,6 @@ // under the License. use crate::args::{ - common::IggyBenchArgs, defaults::{ DEFAULT_BALANCED_NUMBER_OF_PARTITIONS, DEFAULT_BALANCED_NUMBER_OF_STREAMS, DEFAULT_NUMBER_OF_PRODUCERS, @@ -24,7 +23,7 @@ use crate::args::{ props::BenchmarkKindProps, transport::BenchmarkTransportCommand, }; -use clap::{CommandFactory, Parser, error::ErrorKind}; +use clap::Parser; use iggy::prelude::{IggyByteSize, IggyExpiry}; use std::num::NonZeroU32; @@ -88,16 +87,8 @@ impl BenchmarkKindProps for BalancedProducerArgs { self.message_expiry.unwrap_or(IggyExpiry::NeverExpire) } - fn validate(&self) { - let partitions = self.partitions(); - let mut cmd = IggyBenchArgs::command(); - - if partitions < 2 { - cmd.error( - ErrorKind::ArgumentConflict, - format!("For balanced producer, number of partitions must be at least 2, got {partitions}"), - ) - .exit(); - } - } + // No partition floor: a lone partition is a legal target (producers route + // straight to it and balance only across several), and it is the shape + // the fsync-gated benchmark needs, N producers into one partition. + fn validate(&self) {} } diff --git a/core/bench/src/benchmarks/benchmark.rs b/core/bench/src/benchmarks/benchmark.rs index 60675237a7..356b81279a 100644 --- a/core/bench/src/benchmarks/benchmark.rs +++ b/core/bench/src/benchmarks/benchmark.rs @@ -22,6 +22,7 @@ use async_trait::async_trait; use bench_report::benchmark_kind::BenchmarkKind; use bench_report::individual_metrics::BenchmarkIndividualMetrics; use iggy::prelude::*; +use std::num::NonZeroU32; use std::sync::Arc; use tokio::task::JoinSet; use tracing::{info, warn}; @@ -123,10 +124,17 @@ pub trait Benchmarkable: Send { .map_or(MaxTopicSize::Unlimited, MaxTopicSize::Custom); let message_expiry = self.args().message_expiry(); let enforce_fsync = self.args().enforce_fsync(); + let messages_required_to_save = + self.args().messages_required_to_save().map(NonZeroU32::get); info!( - "Creating the test topic '{}' for stream '{}' with max topic size: {:?}, message expiry: {}, enforce fsync: {}", - topic_name, stream_name, max_topic_size, message_expiry, enforce_fsync + "Creating the test topic '{}' for stream '{}' with max topic size: {:?}, message expiry: {}, enforce fsync: {}, messages required to save: {:?}", + topic_name, + stream_name, + max_topic_size, + message_expiry, + enforce_fsync, + messages_required_to_save ); client @@ -140,6 +148,7 @@ pub trait Benchmarkable: Send { max_topic_size: (max_topic_size != MaxTopicSize::ServerDefault) .then_some(max_topic_size), enforce_fsync: enforce_fsync.then_some(true), + messages_required_to_save, ..TopicCreateOptions::default() }, ) diff --git a/core/integration/tests/cluster/crash_recovery_corruption.rs b/core/integration/tests/cluster/crash_recovery_corruption.rs index 63a82d2156..c825c2f0c4 100644 --- a/core/integration/tests/cluster/crash_recovery_corruption.rs +++ b/core/integration/tests/cluster/crash_recovery_corruption.rs @@ -58,6 +58,29 @@ const GARBAGE_BYTE: u8 = 0xA5; /// the file provably lands ahead of many complete committed entries. const WAL_FODDER_STREAMS: usize = 20; +/// Sparse index entry layout, mirrored from `partitions::iggy_index::IggyIndex` +/// (which the `integration` crate does not depend on): three little-endian +/// u64s, `offset`, `timestamp`, `position`. +const INDEX_ENTRY_SIZE: usize = 3 * size_of::(); +const INDEX_ENTRY_POSITION_AT: usize = 2 * size_of::(); + +/// Batches produced before the index-ahead-of-log surgery. High enough that +/// the damaged node's index holds several flush chunks at either role's +/// cadence (a primary flushes per op, a backup per committed range). +const INDEX_AHEAD_BATCHES: u32 = 30; +/// Flush chunks the damaged node's index must hold for the surgery to leave a +/// real surviving prefix behind the one entry it strands past the log end. +const MIN_INDEX_ENTRIES: usize = 4; +/// Infix of the directory the refusal path renames a partition's segment files +/// into (`partitions::state_transfer::quarantine_segment_files`). +const FENCED_DIR_MARKER: &str = ".fenced."; +/// Boot log line recovery emits when the log cannot back the last entry of an +/// index (`server::segment_recovery::recover_segment_bounds`): the positive +/// evidence that path ran, as opposed to the clean anchored walk or a refusal. +/// Distinct from the line the self-contradicting-index check emits, which ends +/// "rebuilding it from the log". +const INDEX_REBUILD_MARKER: &str = "discarding the index and rebuilding it from a byte-0 walk"; + async fn create_stream_and_topic(client: &IggyClient) { client .create_stream(STREAM_NAME) @@ -285,6 +308,64 @@ fn append_garbage(path: &Path, count: usize) { fs::write(path, bytes).unwrap_or_else(|error| panic!("write {}: {error}", path.display())); } +/// Cut `path` down to exactly `length` bytes. The owning process is already +/// dead, so nothing can be holding the file open against the truncation. +fn truncate_to(path: &Path, length: u64) { + fs::OpenOptions::new() + .write(true) + .open(path) + .and_then(|file| file.set_len(length)) + .unwrap_or_else(|error| panic!("truncate {} to {length}: {error}", path.display())); +} + +/// The `position` of every whole entry in a segment `.index`, in file order. +/// +/// One entry is written per flushed chunk and points at that chunk's FIRST +/// batch, so every position is both an absolute byte offset into the paired +/// `.log` and a batch boundary in it. +fn index_positions(path: &Path) -> Vec { + let bytes = fs::read(path).unwrap_or_else(|error| panic!("read {}: {error}", path.display())); + bytes + .as_chunks::() + .0 + .iter() + .map(|entry| { + let mut position = [0u8; size_of::()]; + position.copy_from_slice(&entry[INDEX_ENTRY_POSITION_AT..]); + u64::from_le_bytes(position) + }) + .collect() +} + +/// Payloads from `expected` whose bytes appear anywhere in `haystack`. The +/// payloads are fixed-width, so none is a prefix of another and a plain +/// substring search cannot cross-match them. +fn payloads_within(haystack: &[u8], expected: &[String]) -> Vec { + expected + .iter() + .filter(|payload| { + haystack + .windows(payload.len()) + .any(|window| window == payload.as_bytes()) + }) + .cloned() + .collect() +} + +/// Segment files a boot refusal renamed aside on this node. Once written the +/// fenced directory stays, so an empty result is a stable verdict rather than +/// a snapshot that catch-up could invalidate. +fn fenced_segment_paths(data_path: &Path) -> Vec { + let mut fenced = Vec::new(); + let _ = disk::walk(data_path, &mut |path| { + if path.to_string_lossy().contains(FENCED_DIR_MARKER) { + fenced.push(path.to_path_buf()); + } + false + }); + fenced +} + /// Flip one interior byte at `numerator/denominator` of the file's length. fn flip_interior_byte(path: &Path, numerator: u64, denominator: u64) { let mut bytes = @@ -486,6 +567,194 @@ async fn given_a_torn_index_tail_when_a_node_recovers_should_not_misalign_subseq }); } +/// A crash can leave a node's segment `.log` shorter than its already durable +/// `.index` claims: the two files are persisted concurrently, so death between +/// them strands the entry of the chunk that was in flight even under +/// `enforce_fsync`. That one entry is the whole window: every earlier entry +/// belongs to a completed serialized flush whose log fdatasync finished before +/// the later flush began, so it is the shape the surgery reproduces. The index is a rebuildable local +/// artifact and the log is the authority, so recovery must discard the index, +/// rebuild it from a byte-0 walk of the log, and keep serving the batches the +/// walk proves - not refuse the chain, which fences every surviving byte aside +/// on a cluster and tombstones the partition outright on a single replica. The +/// walk starts at byte 0 rather than at the highest entry the log still backs +/// because an anchor above the damage would leave everything below it unread. +/// The spec pins the whole outcome: the node boots without a refusal, nothing +/// is fenced, its index no longer points past its log, catch-up refills the +/// truncated tail, every acked offset reads back, and the replicas end +/// byte-identical. +#[iggy_harness(cluster_nodes = 3)] +async fn given_a_durable_index_ahead_of_a_truncated_log_when_a_node_recovers_should_rebuild_the_index_and_serve_the_surviving_prefix( + harness: &mut TestHarness, +) { + let client = harness.tcp_root_client().await.unwrap(); + create_stream_and_topic(&client).await; + let acked = produce_acked(&client, "index-ahead", INDEX_AHEAD_BATCHES).await; + let payloads: Vec = acked.iter().map(|(_, payload)| payload.clone()).collect(); + for node in 0..harness.cluster_size() { + wait_until_node_holds_payloads( + harness, + node, + &payloads, + FLUSH_INSTALL_TIMEOUT, + "eager flush before the surgery", + ) + .await; + } + + drop(client); + + // The leader keeps quorum and serving throughout, so the damaged node + // always has a peer to catch up from. SIGKILL rather than a graceful stop: + // the shape under test is a crash, and no shutdown hook may run to + // reconcile the two files first. + let (_, backup) = pick_backup(harness).await; + harness.kill_node(backup).expect("SIGKILL the backup"); + + let backup_data = harness.node(backup).data_path(); + let log_path = find_active_segment_file(&backup_data, "log"); + // Paired by stem, not by a second `find_active_segment_file` sweep, so the + // index provably describes the log being cut. + let index_path = log_path.with_extension("index"); + let positions = index_positions(&index_path); + assert!( + positions.len() >= MIN_INDEX_ENTRIES, + "the backup's index holds only {} flush chunk(s), fewer than the {MIN_INDEX_ENTRIES} \ + the surgery needs to leave a real surviving prefix behind the stranded entry", + positions.len() + ); + let log_size = fs::metadata(&log_path) + .map(|meta| meta.len()) + .unwrap_or_else(|error| panic!("stat {}: {error}", log_path.display())); + // Cutting at the LAST entry's position lands the log end exactly on a + // batch boundary, keeps whole batches behind it, and strands exactly one + // entry past the end of the file - the only depth a crash can produce + // under `enforce_fsync`, where each serialized flush fdatasyncs the whole + // log before the next chunk's entry can exist. A deeper cut would + // fabricate previously durable data loss, which recovery refuses by + // design. + let cut_at = positions[positions.len() - 1]; + assert!( + cut_at > 0 && cut_at < log_size, + "the last entry's position {cut_at} must sit inside the {log_size}-byte log" + ); + truncate_to(&log_path, cut_at); + let truncated = + fs::read(&log_path).unwrap_or_else(|error| panic!("read {}: {error}", log_path.display())); + let surviving = payloads_within(&truncated, &payloads); + assert!( + !surviving.is_empty() && surviving.len() < payloads.len(), + "the surgery must keep a real prefix and remove a real tail; {} of {} payloads \ + survived the cut at byte {cut_at}", + surviving.len(), + payloads.len() + ); + + harness.restart_node(backup).unwrap_or_else(|error| { + panic!( + "an index ahead of its log is what a crash between the two writes leaves; \ + boot must rebuild the index instead of failing: {error}" + ) + }); + + // Read the index BEFORE the log, both right after boot: catch-up grows + // the two files together, so a log still at the cut proves the index was + // read before any append landed, and the rebuild is then the only shape it + // may have. Once the log has grown the refilled tail re-mints entries over + // the same bytes, and nothing on disk tells the two apart; the boot log + // marker below is the evidence that survives that. + // + // The rebuild's stride is its own, so the entry COUNT is not the spec. + // What is: the index describes only bytes the walk proved, which is the + // property the stranded entry violated. + let recovered_positions = index_positions(&index_path); + let recovered_index_len = fs::metadata(&index_path) + .map(|meta| meta.len()) + .unwrap_or_else(|error| panic!("stat {}: {error}", index_path.display())); + let recovered_log_len = fs::metadata(&log_path) + .map(|meta| meta.len()) + .unwrap_or_else(|error| panic!("stat {}: {error}", log_path.display())); + if recovered_log_len == cut_at { + assert_eq!( + recovered_index_len as usize % INDEX_ENTRY_SIZE, + 0, + "the recovered index must hold whole entries; it is {recovered_index_len} bytes" + ); + assert!( + !recovered_positions.is_empty(), + "the {cut_at}-byte log holds whole batches, so the rebuilt index must not be empty" + ); + assert!( + recovered_positions + .iter() + .all(|position| *position < cut_at), + "every rebuilt entry must open inside the {cut_at}-byte log, got \ + {recovered_positions:?}" + ); + } + + let fenced = fenced_segment_paths(&backup_data); + assert!( + fenced.is_empty(), + "recovery must rebuild the index from the log, keeping the {} \ + surviving batches in service; instead the chain was refused and fenced aside: {fenced:?}", + surviving.len() + ); + // `fenced_segment_paths` walks past unreadable directories, so an empty + // result alone could be vacuous: the files must still be where boot found + // them. + assert!( + log_path.exists() && index_path.exists(), + "the segment files must stay in place after recovery; missing under {}", + backup_data.display() + ); + // `restart_node` truncates the node's stdout log, so a marker found here + // was logged by the boot just performed. Under `IGGY_TEST_VERBOSE` the + // child's output is inherited and no file exists to read, which would make + // either check vacuous. + if stderr_is_captured() { + assert!( + !harness + .node(backup) + .stdout_contains("refusing the recovered segment chain"), + "boot must absorb an index that outruns its log, not refuse the chain" + ); + assert!( + harness.node(backup).stdout_contains(INDEX_REBUILD_MARKER), + "boot must log the byte-0 rebuild ({INDEX_REBUILD_MARKER:?}); recovery took \ + another path" + ); + } + + wait_until_node_holds_payloads( + harness, + backup, + &payloads, + CONVERGE_TIMEOUT, + "catch-up refilling the truncated tail", + ) + .await; + + let nodes: Vec = (0..harness.cluster_size()).collect(); + let client = wait_until_cluster_serves(harness, &nodes, CONVERGE_TIMEOUT).await; + wait_for_acked_readable(&client, &acked, CONVERGE_TIMEOUT) + .await + .unwrap_or_else(|state| { + panic!("every acked offset must poll back after the rebuilt-index recovery: {state}") + }); + + let data_paths: Vec = harness + .all_servers() + .iter() + .map(|server| server.data_path()) + .collect(); + harness + .stop() + .await + .expect("stop the cluster for the at-rest comparison"); + disk::assert_replica_data_identical(&data_paths, false); +} + /// An interior flip in the metadata WAL is bit-rot, not a torn append: a /// complete committed entry follows the damage, so truncating would discard /// acked history. Boot must refuse with the interior-corruption diagnostic, diff --git a/core/partitions/src/iggy_index.rs b/core/partitions/src/iggy_index.rs index f98c2a7d84..39715e63bc 100644 --- a/core/partitions/src/iggy_index.rs +++ b/core/partitions/src/iggy_index.rs @@ -17,6 +17,12 @@ pub const IGGY_INDEX_SIZE: usize = std::mem::size_of::() * 3; +/// One sparse index entry. +/// +/// On disk it is the three fields as little-endian u64s in declaration +/// order, `IGGY_INDEX_SIZE` bytes per entry. The server's segment recovery +/// and the cluster crash tests decode that layout by hand (the `integration` +/// crate does not depend on this one), so a change here must change them too. #[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] pub struct IggyIndex { pub offset: u64, diff --git a/core/partitions/src/iggy_index_writer.rs b/core/partitions/src/iggy_index_writer.rs index 939652360c..78ab2c0eba 100644 --- a/core/partitions/src/iggy_index_writer.rs +++ b/core/partitions/src/iggy_index_writer.rs @@ -97,14 +97,17 @@ impl IggyIndexWriter { }) } - /// Appends encoded sparse index bytes to the backing file. + /// Appends encoded sparse index bytes at the current write cursor and + /// returns how many bytes landed. The cursor is left where it was: the + /// caller advances it with `advance` once the companion segment save has + /// also succeeded. /// /// # Errors /// /// Returns an error if the index bytes cannot be written or synced to disk. - pub async fn save_indexes(&self, indexes: Vec) -> Result<(), IggyError> { + pub(crate) async fn save_indexes(&self, indexes: Vec) -> Result { if indexes.is_empty() { - return Ok(()); + return Ok(0); } let len = indexes.len(); @@ -120,12 +123,6 @@ impl IggyIndexWriter { self.fsync().await?; } - // Advance the write cursor last: if the write or fsync fails, the - // counter must stay put so the retry overwrites the same slot instead - // of appending a duplicate entry that boot recovery would refuse. - self.index_size_bytes - .fetch_add(len as u64, Ordering::Release); - trace!( target: "iggy.partitions.storage", file = self.file_path.as_str(), @@ -133,7 +130,14 @@ impl IggyIndexWriter { position, "saved sparse index bytes to file" ); - Ok(()) + Ok(len as u64) + } + + /// Move the write cursor forward over `bytes` that are now durable. Split + /// out of the save so the index and segment cursors advance together, only + /// once both halves have succeeded. + pub(crate) fn advance(&self, bytes: u64) { + self.index_size_bytes.fetch_add(bytes, Ordering::Release); } /// Flushes buffered index file contents to disk. diff --git a/core/partitions/src/iggy_partition.rs b/core/partitions/src/iggy_partition.rs index 34900e5d65..93eaf156f0 100644 --- a/core/partitions/src/iggy_partition.rs +++ b/core/partitions/src/iggy_partition.rs @@ -30,7 +30,7 @@ use crate::poll_plan::{ }; use crate::segment::Segment; use crate::state_transfer::{PartitionTransferSession, PendingTransferRearm}; -use crate::types::{RepairConclusion, RepairSession}; +use crate::types::{FatalCommit, RepairConclusion, RepairSession}; use crate::{ AppendResult, Partition, PartitionOffsets, PartitionsConfig, PollQueryResult, PollingArgs, PollingConsumer, @@ -83,7 +83,7 @@ use std::rc::Rc; use std::sync::Arc; use std::sync::atomic::{AtomicU64, Ordering}; use tokio::sync::Mutex as TokioMutex; -use tracing::{debug, warn}; +use tracing::{debug, error, warn}; // This struct aliases in terms of the code contained the `LocalPartition from `core/server/src/streaming/partitions/local_partition.rs`. // @@ -147,6 +147,11 @@ where /// also gates repaired-batch persistence -- overstating THAT field would /// silently drop the `(commit_op, commit_max]` replay window. pub installed_frontier: Option, + /// Set once a local commit fails for an op the cluster already committed. + /// Fences the partition: every path that would advance or serve it turns + /// into a no-op, so the shard's tick can observe the fault and shut the + /// server down without the partition moving again in the meantime. + fatal: Option, pub(crate) pending_consumer_offset_commits: HashMap, /// Committed-only mirror of each consumer's persisted offset file: the /// last value this replica durably wrote per (kind, consumer id). Fed @@ -474,6 +479,7 @@ where repair: None, recovered_durable_offset: None, installed_frontier: None, + fatal: None, pending_consumer_offset_commits: HashMap::new(), persisted_offsets: RefCell::new(HashMap::new()), observed_view, @@ -1643,20 +1649,22 @@ where let (start_segment, start_position) = self.disk_poll_start(&query); // Cap resident sealed read handles: touch this poll's start segment so // the LRU keeps the hot set and drops the least-recently-used fd + - // index (a no-op for the active segment, whose handle never caches). + // index (a no-op for the active segment, whose slot is bounded by + // rotation instead). self.log.touch_sealed_read_state(start_segment); // Snapshot only the segments the disk walk visits (`start_segment..`), - // so `start_position` applies to the first snapshotted segment. A sealed - // segment carries its shared read-state handle (fd + sparse index) so - // the off-borrow read reuses (or fills) it; the active segment opens - // fresh and resolves from its resident index. + // so `start_position` applies to the first snapshotted segment. Every + // segment carries its shared read-state handle so the off-borrow read + // reuses (or fills) the cached fd; only a sealed one also resolves its + // start byte from the shared sparse index. let segments = self.log.segments()[start_segment..] .iter() .zip(self.log.sealed_read_state()[start_segment..].iter()) .map(|(segment, read_state)| DiskSegment { start_offset: segment.start_offset, persisted: segment.size.as_bytes_u64(), - read_state: segment.sealed.then(|| Rc::clone(read_state)), + read_state: Rc::clone(read_state), + sealed: segment.sealed, }) .collect(); let disk = DiskReadPlan { @@ -1831,6 +1839,28 @@ where IggyNamespace::from_raw(self.consensus.group()) } + /// The commit fault that fenced this partition, if one has. + #[must_use] + pub const fn fatal(&self) -> Option<&FatalCommit> { + self.fatal.as_ref() + } + + /// Fence this partition after the shutdown flush failed to persist its + /// committed journal prefix: that data is cluster-committed and now lives + /// only in this process's memory, so the shard must not report a clean + /// exit over it. A fault the commit path already recorded is kept, since + /// it names the op that first diverged; `commit_min` here only bounds + /// where the unpersisted prefix ends. + pub fn fence_flush_failure(&mut self) { + if self.fatal.is_none() { + self.fatal = Some(FatalCommit { + namespace_raw: self.namespace().inner(), + op: self.consensus.commit_min(), + operation: Operation::SendMessages, + }); + } + } + fn partition_dir(&self) -> Option { if self.partition_dir.is_some() { return self.partition_dir.clone(); @@ -2478,6 +2508,9 @@ where #[allow(clippy::future_not_send)] pub async fn on_ack(&mut self, message: Message, config: &PartitionsConfig) { + if self.fatal.is_some() { + return; + } self.clear_pending_consumer_offset_commits_if_view_changed(); let header = *message.header(); { @@ -2532,6 +2565,9 @@ where #[allow(clippy::future_not_send)] pub async fn commit_journal(&mut self, config: &PartitionsConfig) { + if self.fatal.is_some() { + return; + } self.clear_pending_consumer_offset_commits_if_view_changed(); // The primary commits inline via `on_ack` (it drains its own pipeline). @@ -2927,12 +2963,13 @@ where // commit frontier; flushing the uncommitted tail would write // per-replica-timing bytes to its segment (cross-replica divergence) and // drop the headers those ops need when their own commit later lands - // (commit_min wedge). Eviction is deferred until the bytes are durable: - // on a persist failure the prefix stays resident so the next commit - // re-reads it instead of losing a committed batch (a live-process I/O - // fault only; the in-memory journal does not survive a crash). All - // segment range / stats / durable-offset accounting below is computed - // from the committed entries, not the resident-journal snapshot above. + // (commit_min wedge). Eviction is deferred until the bytes are durable, + // so a persist failure retains the prefix rather than losing a committed + // batch (a live-process I/O fault only; the in-memory journal does not + // survive a crash). Only the transfer-offer flush survives to re-read it; + // the commit path panics the shard pump instead. All segment range / + // stats / durable-offset accounting below is computed from the committed + // entries, not the resident-journal snapshot above. let commit_max = self.consensus.commit_max(); let committed_entries = self.log.journal().inner.committed_prefix(commit_max); if committed_entries.is_empty() { @@ -2965,8 +3002,8 @@ where // re-appends the whole retained tail, so a per-chunk call would // re-walk that tail once per segment crossed, quadratic in the flush // span -- all under the partition write lock. On an error mid-flush - // the accumulated prefix is evicted before propagating, so the retry - // re-reads only what did not land. + // the accumulated prefix is evicted before propagating, so any later + // flush attempt re-reads only what did not land. let mut evictable = 0usize; while entries.peek().is_some() { // A recovered active segment can already sit at or past the cap @@ -3089,12 +3126,21 @@ where }; // Persist BEFORE eviction so a write failure leaves the rest of the - // committed prefix resident for retry. The persist is idempotent on - // failure: a batch write that lands but whose index save then fails - // rewinds the segment write cursor, so the retry overwrites those - // bytes instead of appending a duplicate. Chunks already durable - // are evicted before the error propagates, so the retry cannot - // re-read them (and re-write them past a rotation). + // committed prefix resident instead of dropping it. On the commit + // path a failure fences the partition and stops the shard pump; the + // server then shuts down non-zero after one final best-effort flush + // of every partition. A shutdown-flush failure warns and moves to + // the next namespace, while the transfer offer turns it into + // `FlushFailed`. Recovery reopens each writer at the on-disk length + // boot recovery validated. + // + // The ordering therefore buys a non-corrupting failure, not an + // in-process one. Write cursors advance only once both the batch and the + // index are durable, so a later attempt, including the final + // shutdown flush, rewrites the same positions instead of appending + // a duplicate. Chunks already durable are evicted before the error + // propagates, so it cannot re-read them and write them past a + // rotation. if let Err(error) = self .persist_frozen_batches_to_disk(frozen_batches, index_bytes, batch_count) .await @@ -3104,9 +3150,9 @@ where } // Insert the flushed sparse-index entry into the in-mem cache only now // that the batch + index are durable. Inserting in the build loop (before - // persist) re-inserts a duplicate on a persist-failure retry, which - // re-reads the same prefix. The active segment has not rotated yet, so - // this targets the segment that received the batches. + // persist) re-inserts a duplicate on the next flush after a persist + // failure, which re-reads the same prefix. The active segment has not + // rotated yet, so this targets the segment that received the batches. if let Some(index) = flush_index { self.log.ensure_indexes(); let indexes = self.log.active_indexes_mut().expect("indexes must exist"); @@ -3236,12 +3282,32 @@ where // advanced; next advance_commit_min(op+1) would assert // op+1 == commit_min + 1, panics cryptically. // - // Fatal: better to suicide than serve stale or panic later. - // Operator restarts; recovery+repair re-syncs. - panic!( - "partition local commit failed at op={} ({:?}): replica is divergent from cluster commit; restart required", - prepare_header.op, prepare_header.operation + // Fatal, but NOT by panicking: this runs on the shard's + // message pump, and `compio::runtime::spawn` swallows a panic + // there, leaving every partition on the shard unable to + // commit, tick or reply while the process still reports + // healthy and answers on its other shards. Fence the + // partition and stop draining; the shard's tick picks the + // fault up and takes the server down through the ordinary + // shutdown path, so the remaining partitions flush and the + // exit code is non-zero. Operator restarts; recovery+repair + // re-syncs. + error!( + target: "iggy.partitions.diag", + plane = "partitions", + replica_id, + namespace_raw, + op = prepare_header.op, + operation = ?prepare_header.operation, + "partition local commit failed for a cluster-committed op; \ + replica is divergent, fencing the partition and shutting down" ); + self.fatal = Some(FatalCommit { + namespace_raw, + op: prepare_header.op, + operation: prepare_header.operation, + }); + return; } self.consensus.advance_commit_min(prepare_header.op); @@ -3711,38 +3777,58 @@ where let messages_writer = messages_writer.expect("checked above"); let index_writer = index_writer.expect("checked above"); - let saved = messages_writer - .save_frozen_batches(&stripped_batches) - .await - .map_err(|error| { - warn!( - target: "iggy.partitions.diag", - plane = "partitions", - namespace_raw = self.namespace().inner(), - batch_count, - %error, - "failed to save frozen batches" - ); - error - })?; + // Both writes are in flight before either completes, so under + // `enforce_fsync` the two fdatasync round trips overlap instead of + // serializing. `join` never cancels a half, so no write is dropped + // mid-flight when the other one fails. + let (log_result, index_result) = futures::future::join( + messages_writer.save_frozen_batches(&stripped_batches), + index_writer.save_indexes(index_bytes), + ) + .await; - if let Err(error) = index_writer.save_indexes(index_bytes).await { + // The match below collapses to the log's error (the durable record, the + // index being derived from it). The halves write different files and can + // fail for unrelated reasons, so name the index failure here instead of + // letting the log's error stand for both. + if let (Err(log_error), Err(index_error)) = (&log_result, &index_result) { warn!( target: "iggy.partitions.diag", plane = "partitions", namespace_raw = self.namespace().inner(), batch_count, - %error, - "failed to save sparse indexes; rewinding segment write cursor" + %log_error, + %index_error, + "failed to persist frozen batches: log and index both failed" ); - // The batch bytes landed but the index did not, so the whole persist - // fails and the committed prefix stays resident for retry. Rewind the - // writer cursor by exactly what this call advanced so the retry - // overwrites those bytes instead of appending a duplicate copy. - messages_writer.rewind(saved.as_bytes_u64()); - return Err(error); } + let (saved, saved_index_bytes) = match (log_result, index_result) { + (Ok(saved), Ok(saved_index_bytes)) => (saved, saved_index_bytes), + (Err(error), _) | (Ok(_), Err(error)) => { + warn!( + target: "iggy.partitions.diag", + plane = "partitions", + namespace_raw = self.namespace().inner(), + batch_count, + %error, + "failed to persist frozen batches" + ); + return Err(error); + } + }; + + // Advance both cursors only here, back to back with no await between. + // They are the writers' next-write positions, so a half whose save + // failed must keep its cursor: the pair then still describes one durable + // prefix, which is what a writer re-opened by boot recovery asserts + // against the length of the file it finds + // (`SegmentSizeMismatchAtOpen`). In-process it also keeps a later flush + // rewriting the same slot rather than appending a duplicate index entry + // or leaving a hole in the segment. + messages_writer.advance(saved.as_bytes_u64()); + index_writer.advance(saved_index_bytes); + debug!( target: "iggy.partitions.diag", plane = "partitions", @@ -3837,6 +3923,10 @@ where // segment's cache is ever read (the `commit_messages` flush staging), // so a sealed cache is dead weight. self.log.indexes_mut()[old_segment_index] = None; + // The read fd cached while this segment was active is not counted by + // the sealed LRU budget, so it must not survive the seal; the next + // sealed poll re-fills the fresh slot under the LRU's rules. + self.log.reset_read_state(old_segment_index); self.log .add_persisted_segment(segment, storage, Some(messages_writer), Some(index_writer)); @@ -5086,6 +5176,8 @@ fn nth_oldest_sealed_end(segments: &[Segment], count: u32) -> Option { #[cfg(test)] mod tests { use super::*; + use crate::iggy_index::{IGGY_INDEX_SIZE, IggyIndex, IggyIndexCache}; + use crate::iggy_index_reader::IggyIndexReader; use crate::poll_plan::{DiskReadOutcome, SealedSegmentHandle}; use bytes::Bytes; use compio::io::AsyncWriteAtExt; @@ -5931,12 +6023,14 @@ mod tests { DiskSegment { start_offset: 0, persisted: 512, - read_state: None, + read_state: SealedSegmentHandle::default(), + sealed: false, }, DiskSegment { start_offset: 5, persisted: later_len, - read_state: None, + read_state: SealedSegmentHandle::default(), + sealed: false, }, ], start_position: 0, @@ -6017,12 +6111,14 @@ mod tests { DiskSegment { start_offset: 0, persisted: corrupt_len, - read_state: None, + read_state: SealedSegmentHandle::default(), + sealed: false, }, DiskSegment { start_offset: 5, persisted: later_len, - read_state: None, + read_state: SealedSegmentHandle::default(), + sealed: false, }, ], start_position: 0, @@ -6089,7 +6185,8 @@ mod tests { segments: vec![DiskSegment { start_offset: 0, persisted: record_len, - read_state: None, + read_state: SealedSegmentHandle::default(), + sealed: false, }], start_position: 0, namespace_raw: namespace.inner(), @@ -6126,7 +6223,8 @@ mod tests { segments: vec![DiskSegment { start_offset: 0, persisted: 512, - read_state: None, + read_state: SealedSegmentHandle::default(), + sealed: false, }], start_position: 0, namespace_raw: IggyNamespace::new(1, 1, 0).inner(), @@ -6157,7 +6255,8 @@ mod tests { segments: vec![DiskSegment { start_offset: 0, persisted: 512, - read_state: None, + read_state: SealedSegmentHandle::default(), + sealed: false, }], start_position: 0, namespace_raw: IggyNamespace::new(1, 1, 0).inner(), @@ -6224,7 +6323,8 @@ mod tests { segments: vec![DiskSegment { start_offset: 0, persisted: record_len, - read_state: Some(Rc::clone(&handle)), + read_state: Rc::clone(&handle), + sealed: true, }], start_position: 0, namespace_raw: namespace.inner(), @@ -6255,7 +6355,8 @@ mod tests { segments: vec![DiskSegment { start_offset: 0, persisted: record_len, - read_state: Some(Rc::clone(&handle)), + read_state: Rc::clone(&handle), + sealed: true, }], start_position: 0, namespace_raw: namespace.inner(), @@ -6315,7 +6416,8 @@ mod tests { segments: vec![DiskSegment { start_offset: 0, persisted: record_len, - read_state: Some(Rc::clone(&handle)), + read_state: Rc::clone(&handle), + sealed: true, }], start_position: 0, namespace_raw: namespace.inner(), @@ -6399,7 +6501,8 @@ mod tests { segments: vec![DiskSegment { start_offset: 0, persisted: log_len, - read_state: Some(Rc::clone(&handle)), + read_state: Rc::clone(&handle), + sealed: true, }], // Byte 0, exactly what disk_poll_start returns for a sealed segment // whose resident index was dropped. @@ -6498,7 +6601,8 @@ mod tests { segments: vec![DiskSegment { start_offset: 0, persisted: log_len, - read_state: Some(Rc::clone(&handle)), + read_state: Rc::clone(&handle), + sealed: true, }], start_position: 0, namespace_raw: namespace.inner(), @@ -6575,7 +6679,8 @@ mod tests { segments: vec![DiskSegment { start_offset: 0, persisted: record_len, - read_state: Some(Rc::clone(&handle)), + read_state: Rc::clone(&handle), + sealed: true, }], start_position: 0, namespace_raw: namespace.inner(), @@ -6623,7 +6728,8 @@ mod tests { segments: vec![DiskSegment { start_offset: 0, persisted: record_len, - read_state: Some(Rc::clone(&handle)), + read_state: Rc::clone(&handle), + sealed: true, }], start_position: 0, namespace_raw: namespace.inner(), @@ -7343,6 +7449,349 @@ mod tests { fn given_unframed_operation_when_committed_should_reply_empty_body() { assert!(committed_reply_body(Operation::DeleteSegments).is_empty()); } + + /// Every write to this device fails with `ENOSPC`, which is how the + /// persist failure cases below inject a fault into one half of the flush + /// without any production-side plumbing. + #[cfg(target_os = "linux")] + const DEV_FULL: &str = "/dev/full"; + + const FIRST_PAYLOAD: &[u8] = b"first-chunk"; + const SECOND_PAYLOAD: &[u8] = b"second-chunk-is-longer"; + + /// Partition whose active segment carries real writers over the given + /// paths, both with fsync on. Point either path at [`DEV_FULL`] to make + /// that half's save fail. + struct PersistFixture { + partition: IggyPartition, + log_cursor: Rc, + index_cursor: Rc, + } + + impl PersistFixture { + async fn new(log_path: &str, index_path: &str) -> Self { + let log_cursor = Rc::new(AtomicU64::new(0)); + let index_cursor = Rc::new(AtomicU64::new(0)); + let messages_writer = + MessagesWriter::new(log_path, log_cursor.clone(), true, false, None) + .await + .expect("open segment log writer"); + let index_writer = IggyIndexWriter::new(index_path, index_cursor.clone(), true, false) + .await + .expect("open segment index writer"); + + let mut partition = test_partition(); + partition.log.add_persisted_segment( + Segment::new(0, IggyByteSize::from(1024 * 1024_u64)), + SegmentStorage::default(), + Some(Rc::new(messages_writer)), + Some(Rc::new(index_writer)), + ); + + Self { + partition, + log_cursor, + index_cursor, + } + } + + /// Mirrors one `commit_messages` chunk: the sparse entry addresses the + /// byte the batch is about to land on, taken from the segment size the + /// way production takes it, so a second persist can only index + /// correctly if the first one advanced the segment in step with the + /// writer's cursor. + async fn persist(&mut self, payload: &[u8], offset: u64) -> Result<(), IggyError> { + let index = IggyIndex::new(offset, offset + 1, self.segment_size()); + self.partition + .persist_frozen_batches_to_disk( + vec![prepare_framed(payload)], + IggyIndexCache::serialize(&index), + 1, + ) + .await + } + + fn cursors(&self) -> (u64, u64) { + ( + self.log_cursor.load(Ordering::Relaxed), + self.index_cursor.load(Ordering::Relaxed), + ) + } + + fn segment_size(&self) -> u64 { + self.partition.log.active_segment().size.as_bytes_u64() + } + } + + /// Journaled entry in the shape the persist path expects: a `PrepareHeader` + /// prefix it strips, followed by the bytes that reach the segment file. + fn prepare_framed(payload: &[u8]) -> Frozen<4096> { + let mut bytes = vec![0u8; size_of::()]; + bytes.extend_from_slice(payload); + Owned::<4096>::copy_from_slice(&bytes).into() + } + + fn file_len(path: &std::path::Path) -> u64 { + std::fs::metadata(path).expect("stat file").len() + } + + #[cfg(target_os = "linux")] + #[compio::test] + async fn given_index_save_failure_when_persisting_should_leave_both_cursors_and_segment_size_untouched() + { + let dir = tempfile::tempdir().expect("temp dir"); + let log_path = dir.path().join("segment.log"); + let mut fixture = + PersistFixture::new(log_path.to_str().expect("utf-8 path"), DEV_FULL).await; + + let result = fixture.persist(FIRST_PAYLOAD, 0).await; + + assert!( + matches!(result, Err(IggyError::CannotSaveIndexToSegment)), + "index save over {DEV_FULL} must fail, got {result:?}" + ); + assert_eq!( + file_len(&log_path), + FIRST_PAYLOAD.len() as u64, + "the segment bytes landed; only the cursor is withheld" + ); + assert_eq!( + fixture.cursors(), + (0, 0), + "neither cursor may advance when the index half failed" + ); + assert_eq!(fixture.segment_size(), 0, "segment size must not advance"); + } + + #[cfg(target_os = "linux")] + #[compio::test] + async fn given_log_save_failure_when_persisting_should_leave_both_cursors_and_segment_size_untouched() + { + let dir = tempfile::tempdir().expect("temp dir"); + let index_path = dir.path().join("segment.index"); + let mut fixture = + PersistFixture::new(DEV_FULL, index_path.to_str().expect("utf-8 path")).await; + + let result = fixture.persist(FIRST_PAYLOAD, 0).await; + + assert!( + matches!(result, Err(IggyError::CannotWriteToFile)), + "segment save over {DEV_FULL} must fail, got {result:?}" + ); + assert_eq!( + file_len(&index_path), + IGGY_INDEX_SIZE as u64, + "the index entry landed; only the cursor is withheld" + ); + assert_eq!( + fixture.cursors(), + (0, 0), + "neither cursor may advance when the segment half failed" + ); + assert_eq!(fixture.segment_size(), 0, "segment size must not advance"); + } + + #[cfg(target_os = "linux")] + #[compio::test] + async fn given_both_saves_failing_when_persisting_should_leave_both_cursors_untouched() { + let mut fixture = PersistFixture::new(DEV_FULL, DEV_FULL).await; + + let result = fixture.persist(FIRST_PAYLOAD, 0).await; + + assert!( + matches!(result, Err(IggyError::CannotWriteToFile)), + "a failed segment save must win over the index error, got {result:?}" + ); + assert_eq!(fixture.cursors(), (0, 0), "no cursor may advance"); + assert_eq!(fixture.segment_size(), 0, "segment size must not advance"); + } + + /// The two saves run concurrently, so each must read its own cursor + /// without observing the other's advance: the second chunk lands exactly + /// where the first one ended, and its index entry says so. + #[compio::test] + async fn given_two_successful_persists_when_reading_back_should_place_second_chunk_at_first_chunk_end() + { + let dir = tempfile::tempdir().expect("temp dir"); + let log_path = dir.path().join("segment.log"); + let index_path = dir.path().join("segment.index"); + let mut fixture = PersistFixture::new( + log_path.to_str().expect("utf-8 path"), + index_path.to_str().expect("utf-8 path"), + ) + .await; + + fixture + .persist(FIRST_PAYLOAD, 0) + .await + .expect("first persist"); + fixture + .persist(SECOND_PAYLOAD, 7) + .await + .expect("second persist"); + + let log_bytes = (FIRST_PAYLOAD.len() + SECOND_PAYLOAD.len()) as u64; + let index_bytes = 2 * IGGY_INDEX_SIZE as u64; + assert_eq!( + fixture.cursors(), + (log_bytes, index_bytes), + "both cursors must cover both persists" + ); + assert_eq!(file_len(&log_path), log_bytes, "segment file length"); + assert_eq!(file_len(&index_path), index_bytes, "index file length"); + assert_eq!(fixture.segment_size(), log_bytes, "segment size"); + + let reader = IggyIndexReader::new(index_path.to_str().expect("utf-8 path")) + .await + .expect("open index reader"); + let last = reader + .load_last() + .await + .expect("read last index entry") + .expect("index entry present"); + assert_eq!( + last, + IggyIndex::new(7, 8, FIRST_PAYLOAD.len() as u64), + "the second entry must address the first chunk's end" + ); + } + + /// A half that failed never advanced its cursor, so the retry rewrites + /// the same positions: one copy of the batch, one index entry. + #[cfg(target_os = "linux")] + #[compio::test] + async fn given_a_persist_failure_on_a_committed_op_should_fence_the_partition_not_panic() { + // The op is cluster-committed and the local write cannot be made, so + // the replica is divergent. This used to `panic!`, which the pump + // task swallows: the shard stopped serving every partition it owned + // while the process reported healthy. The partition must fence itself + // instead, so the shard's tick can see the fault and stop the server. + let dir = tempfile::tempdir().expect("temp dir"); + let log_path = dir.path().join("segment.log"); + let log_cursor = Rc::new(AtomicU64::new(0)); + let index_cursor = Rc::new(AtomicU64::new(0)); + let messages_writer = MessagesWriter::new( + log_path.to_str().expect("utf-8 path"), + log_cursor, + true, + false, + None, + ) + .await + .expect("open segment log writer"); + // The index half writes to a device that is always full, so the + // persist fails the way a full disk fails. + let index_writer = IggyIndexWriter::new(DEV_FULL, index_cursor, true, false) + .await + .expect("open segment index writer"); + + let mut partition = test_partition(); + partition.log.add_persisted_segment( + Segment::new(0, IggyByteSize::from(1024 * 1024_u64)), + SegmentStorage::default(), + Some(Rc::new(messages_writer)), + Some(Rc::new(index_writer)), + ); + + journal_send_batch(&mut partition, 1).await; + partition.consensus().advance_commit_max(1); + + // Would abort the test process before the fence existed. + partition.commit_journal(&repair_config()).await; + + let fault = partition + .fatal() + .expect("a failed commit of a cluster-committed op must fence the partition"); + assert_eq!(fault.op, 1); + assert_eq!(fault.operation, Operation::SendMessages); + + // The fence holds: a fenced partition must not advance again, or the + // pump's tail drain walks it into the `advance_commit_min` assert. + let commit_min = partition.consensus().commit_min(); + partition.commit_journal(&repair_config()).await; + assert_eq!( + partition.consensus().commit_min(), + commit_min, + "a fenced partition must not advance on a later commit" + ); + } + + #[compio::test] + async fn given_a_shutdown_flush_failure_when_fencing_should_keep_an_earlier_commit_fault() { + let mut partition = test_partition(); + + // A flush failure on a healthy partition fences it, so the pump's + // post-flush scan turns the exit non-zero instead of reporting a + // clean shutdown over unpersisted cluster-committed data. + partition.fence_flush_failure(); + let fault = partition + .fatal() + .expect("a failed shutdown flush must fence the partition"); + assert_eq!(fault.op, partition.consensus().commit_min()); + assert_eq!(fault.operation, Operation::SendMessages); + + // A partition the commit path already fenced keeps that fault: it + // names the op that first diverged. + let commit_fault = FatalCommit { + namespace_raw: fault.namespace_raw, + op: 42, + operation: Operation::StoreConsumerOffset, + }; + partition.fatal = Some(commit_fault); + partition.fence_flush_failure(); + let kept = partition.fatal().expect("the fence must hold"); + assert_eq!(kept.op, 42); + assert_eq!(kept.operation, Operation::StoreConsumerOffset); + } + + #[compio::test] + async fn given_failed_persist_when_retried_with_a_healthy_writer_should_overwrite_the_same_positions() + { + let dir = tempfile::tempdir().expect("temp dir"); + let log_path = dir.path().join("segment.log"); + let index_path = dir.path().join("segment.index"); + let mut fixture = + PersistFixture::new(log_path.to_str().expect("utf-8 path"), DEV_FULL).await; + + assert!( + fixture.persist(FIRST_PAYLOAD, 0).await.is_err(), + "index save over {DEV_FULL} must fail" + ); + + // The committed prefix stays resident, so the retry re-persists the + // identical bytes; only the broken index writer is swapped out. + let index_writer = IggyIndexWriter::new( + index_path.to_str().expect("utf-8 path"), + fixture.index_cursor.clone(), + true, + false, + ) + .await + .expect("open replacement index writer"); + let active = fixture.partition.log.index_writers().len() - 1; + fixture.partition.log.index_writers_mut()[active] = Some(Rc::new(index_writer)); + + fixture + .persist(FIRST_PAYLOAD, 0) + .await + .expect("retry persist"); + + assert_eq!( + file_len(&log_path), + FIRST_PAYLOAD.len() as u64, + "the retry must overwrite the batch, not append a second copy" + ); + assert_eq!( + file_len(&index_path), + IGGY_INDEX_SIZE as u64, + "the retry must write exactly one index entry" + ); + assert_eq!( + fixture.cursors(), + (FIRST_PAYLOAD.len() as u64, IGGY_INDEX_SIZE as u64), + "both cursors must advance once the retry succeeded" + ); + } } #[cfg(test)] diff --git a/core/partitions/src/lib.rs b/core/partitions/src/lib.rs index aa33f5c7a8..818872029d 100644 --- a/core/partitions/src/lib.rs +++ b/core/partitions/src/lib.rs @@ -46,9 +46,9 @@ pub use segment::Segment; use server_common::Message; pub use server_common::send_messages::{IggyMessage, IggyMessageHeader, IggyMessages}; pub use types::{ - AppendResult, Fragment, PartitionOffsets, PartitionPathLayout, PartitionsConfig, PollFragments, - PollQueryResult, PollingArgs, PollingConsumer, REPAIR_RETRY_TICKS, RepairConclusion, - RepairSession, SendMessagesResult, + AppendResult, FatalCommit, Fragment, PartitionOffsets, PartitionPathLayout, PartitionsConfig, + PollFragments, PollQueryResult, PollingArgs, PollingConsumer, REPAIR_RETRY_TICKS, + RepairConclusion, RepairSession, SendMessagesResult, }; /// A partition's message log, named so a caller can carry one across a rebuild. diff --git a/core/partitions/src/log.rs b/core/partitions/src/log.rs index 1dc0fcd3dd..92d9c1768d 100644 --- a/core/partitions/src/log.rs +++ b/core/partitions/src/log.rs @@ -148,9 +148,9 @@ where messages_writers: Vec>>, index_writers: Vec>>, // Parallel to `segments`: a shared read-state handle (fd + sparse index) - // per segment, filled lazily on the first sealed-segment poll and cloned - // into the off-borrow poll plan. Maintained in lockstep with `segments` - // (push/remove together). + // per segment, filled lazily on the first poll that reads the segment and + // cloned into the off-borrow poll plan. Maintained in lockstep with + // `segments` (push/remove together). sealed_read_state: Vec, // LRU of sealed-segment `start_offset`s (most-recently-used at the front) // bounding how many `sealed_read_state` handles stay resident, capped at @@ -199,7 +199,7 @@ where } /// Shared read-state handles, parallel to [`Self::segments`]. Cloned into - /// the poll plan for sealed segments (see [`SealedSegmentHandle`]). + /// the poll plan (see [`SealedSegmentHandle`]). pub fn sealed_read_state(&self) -> &[SealedSegmentHandle] { &self.sealed_read_state } @@ -207,15 +207,18 @@ where /// Record a sealed-segment access and enforce [`SEALED_READ_STATE_CAP`] /// (LRU). `slot` indexes [`Self::segments`]; an out-of-range slot or an /// unsealed (active) segment is a no-op, so the poll path passes its start - /// segment unconditionally. The LRU is keyed by `start_offset` - stable - /// across retire, unlike the slot index. The touched segment moves to the - /// most-recently-used front and its handle is marked tracked (eligible to - /// cache a read fd, see `SealedSegmentReadState::tracked`); once more than - /// the cap distinct sealed segments are tracked, the least-recently-used - /// one's handle is untracked and dropped (replaced with a fresh empty - /// handle) so its fd + sparse index free. An in-flight poll holding a clone - /// of the dropped handle keeps it alive until it finishes (see - /// [`SealedSegmentHandle`]). + /// segment unconditionally. Keeping the active segment out is deliberate: + /// its read fd must not be evictable by unrelated sealed traffic, and its + /// `start_offset` is not a stable LRU key across rotation. Its slot is + /// bounded by [`Self::reset_read_state`] instead. The LRU is keyed by + /// `start_offset` - stable across retire, unlike the slot index. The + /// touched segment moves to the most-recently-used front and its handle is + /// marked tracked (eligible to cache a read fd, see + /// `SealedSegmentReadState::tracked`); once more than the cap distinct + /// sealed segments are tracked, the least-recently-used one's handle is + /// untracked and dropped (replaced with a fresh empty handle) so its fd + + /// sparse index free. An in-flight poll holding a clone of the dropped + /// handle keeps it alive until it finishes (see [`SealedSegmentHandle`]). pub fn touch_sealed_read_state(&mut self, slot: usize) { let Some(touched) = self.segments.get(slot) else { return; @@ -250,6 +253,25 @@ where } } + /// Orphan `slot`'s read-state handle (replaced with a fresh empty one) and + /// purge its sealed-LRU entry. Called wherever a segment changes + /// sealed-ness, because the two states cache under different rules: the + /// active segment's fd lives outside the LRU budget and must not carry into + /// sealed tracking, and a sealed handle must not carry into active use + /// while an LRU entry survives that could evict the now-active fd. + /// Replacing (rather than clearing in place) also detaches an in-flight + /// poll that snapshotted the old sealed-ness, so its store-back lands in + /// the orphan and frees when the poll finishes. + pub fn reset_read_state(&mut self, slot: usize) { + let Some(segment) = self.segments.get(slot) else { + return; + }; + let start_offset = segment.start_offset; + self.sealed_read_state[slot].tracked.set(false); + self.sealed_read_state[slot] = SealedSegmentHandle::default(); + self.sealed_lru.retain(|&offset| offset != start_offset); + } + /// Wipe every shared read-state handle in place (cached fd + sparse index /// cleared, handle untracked) and reset the sealed LRU. In-flight polls /// hold `Rc` clones of these handles, so clearing the slots (not just @@ -577,6 +599,32 @@ mod tests { assert!(log.sealed_lru.is_empty()); } + #[test] + fn reset_read_state_orphans_the_handle_and_purges_its_lru_entry() { + let mut log = TestLog::default(); + let handle = push_resident_sealed(&mut log, 0); + push_resident_sealed(&mut log, 5); + log.touch_sealed_read_state(0); + log.touch_sealed_read_state(1); + + // Un-sealing slot 0 back into the active segment: its handle must not + // stay in the LRU, which could evict it while it is the active fd. + log.reset_read_state(0); + + assert!( + !Rc::ptr_eq(&handle, &log.sealed_read_state()[0]), + "an in-flight poll's clone must not keep filling the live slot", + ); + assert!(!handle.tracked.get()); + assert!(!log.sealed_lru.contains(&0)); + assert!(log.sealed_lru.contains(&5), "other slots are untouched"); + assert!(log.sealed_read_state()[0].index.borrow().is_none()); + + // Out-of-range slot (the purge drain window empties the vec across + // awaits): a no-op, not a panic. + log.reset_read_state(2); + } + #[test] fn retire_front_purges_lru_entry_and_keeps_vecs_lockstep() { let mut log = TestLog::default(); diff --git a/core/partitions/src/messages_writer.rs b/core/partitions/src/messages_writer.rs index 1d8e8ab24f..9e61ec1bef 100644 --- a/core/partitions/src/messages_writer.rs +++ b/core/partitions/src/messages_writer.rs @@ -104,12 +104,15 @@ impl MessagesWriter { }) } - /// Appends a batch of frozen message buffers to the segment file. + /// Appends a batch of frozen message buffers at the current write cursor + /// and returns how many bytes landed. The cursor is left where it was: the + /// caller advances it with `advance` once the companion index save has + /// also succeeded. /// /// # Errors /// /// Returns an error if any chunk cannot be written or synced to disk. - pub async fn save_frozen_batches( + pub(crate) async fn save_frozen_batches( &self, buffers: &[Frozen], ) -> Result { @@ -126,26 +129,14 @@ impl MessagesWriter { self.fsync().await?; } - self.messages_size_bytes - .fetch_add(messages_size, Ordering::Release); - Ok(IggyByteSize::from(messages_size)) } - /// Roll the in-memory write cursor back by `bytes`, undoing the advance of a - /// `save_frozen_batches` whose batch was written but whose companion index - /// save then failed. The committed prefix stays resident and is - /// re-persisted on the next `commit_messages`; rewinding the cursor makes - /// that retry overwrite the same region instead of appending a second copy - /// of the committed batch. Crash-safe without truncating: the rewound - /// bytes are whole batch records past the last index entry, so boot - /// recovery's indexed walk absorbs them and its truncate is a no-op. - pub(crate) fn rewind(&self, bytes: u64) { - debug_assert!( - bytes <= self.messages_size_bytes.load(Ordering::Relaxed), - "rewind underflow: bytes ({bytes}) exceeds the write cursor" - ); - self.messages_size_bytes.fetch_sub(bytes, Ordering::Release); + /// Move the write cursor forward over `bytes` that are now durable. Split + /// out of the save so the segment and index cursors advance together, only + /// once both halves have succeeded. + pub(crate) fn advance(&self, bytes: u64) { + self.messages_size_bytes.fetch_add(bytes, Ordering::Release); } #[must_use] diff --git a/core/partitions/src/poll_plan.rs b/core/partitions/src/poll_plan.rs index 4fd1ea9564..d982fa1bc7 100644 --- a/core/partitions/src/poll_plan.rs +++ b/core/partitions/src/poll_plan.rs @@ -86,12 +86,15 @@ pub enum PartitionDirResolution { /// wipes the slots in place (`SegmentedLog::invalidate_sealed_read_state`): /// it recreates the same paths, so a clone surviving in a suspended walk must /// re-open by path and observe the fresh files rather than serve purged data. -/// The active segment is never cached. +/// The active segment uses the [`Self::fd`] slot only: it grows under the +/// reader, so a size-derived memo on it would go stale, and its slot sits +/// outside the sealed LRU (`SegmentedLog::reset_read_state` drops it wherever +/// a segment changes sealed-ness). #[derive(Debug, Default)] pub struct SealedSegmentReadState { /// Read-only descriptor; compio `File` clones share the kernel fd, so a hit /// avoids the per-poll `openat` (an `io_uring` op prone to io-wq punts) and - /// preserves kernel readahead. `None` until the first sealed poll opens it. + /// preserves kernel readahead. `None` until the first poll opens it. pub(crate) fd: RefCell>, /// Sparse offset/timestamp index reloaded from the `.index` file the /// segment dropped at rotation, so a poll resolves the start byte in @@ -99,7 +102,9 @@ pub struct SealedSegmentReadState { /// `None` until the first sealed poll loads it. pub(crate) index: RefCell>, /// Whether the owning partition's sealed LRU currently tracks this handle. - /// Gates the fd store-back in `resolve_segment_file`: a walk crosses every + /// Gates the fd store-back in `resolve_segment_file` for a SEALED segment + /// (the active segment's slot is outside the LRU, so it always fills): a + /// walk crosses every /// sealed segment from the poll's start onward, but only the start segment /// is LRU-touched, so an untracked fill would retain a descriptor the /// `SEALED_READ_STATE_CAP` budget never counts. Set on touch, cleared on @@ -132,8 +137,8 @@ pub type SealedSegmentHandle = Rc; /// Owned, borrow-free inputs for the disk tier of a poll (see module docs). A /// sealed segment reuses its cached [`SealedSegmentReadState`] (read fd + sparse -/// index); the active segment (and any cache miss) opens by path and resolves -/// from its resident index, because sealed segments drop both at rotation. +/// index); the active segment reuses the cached fd but resolves from its +/// resident index, because sealed segments drop that index at rotation. pub struct DiskReadPlan { pub(crate) partition_dir: PartitionDirResolution, /// Segments to walk, snapshotted from the poll's starting segment onward @@ -150,10 +155,13 @@ pub struct DiskReadPlan { pub struct DiskSegment { pub(crate) start_offset: u64, pub(crate) persisted: u64, - /// Shared read state, cloned from the owning partition at plan time for a - /// SEALED segment; `None` for the active segment, which always opens fresh - /// and resolves from its resident index. See [`SealedSegmentReadState`]. - pub(crate) read_state: Option, + /// Shared read state, cloned from the owning partition at plan time. See + /// [`SealedSegmentReadState`]. + pub(crate) read_state: SealedSegmentHandle, + /// Whether the segment was sealed when the plan was built. Only a sealed + /// segment resolves its start byte from the shared sparse index; the active + /// one grows under the reader and uses its resident index instead. + pub(crate) sealed: bool, } /// Owned auto-commit input, applied off the partition borrow after a poll (see @@ -519,8 +527,8 @@ impl DiskReadPlan { // then cached) and resolve the start byte so the walk skips straight to // the target instead of scanning the whole segment - the poll stall. A // miss or load failure keeps `start_position` (the pre-existing - // full-scan fallback). The active segment carries no read state, so its - // resident-index-resolved `start_position` is left untouched. + // full-scan fallback). An active first segment keeps its + // resident-index-resolved `start_position` untouched. let mut position = match self.segments.first() { Some(first) => self .resolve_sealed_start(first, query, partition_dir) @@ -620,31 +628,30 @@ impl DiskReadPlan { } } - /// Resolve the read-only descriptor for `segment`'s file. A sealed segment - /// clones its cached fd on a hit (sharing the kernel fd, no syscall) and, on - /// a miss, opens by path and stores the fd back so later polls skip the - /// `openat`. The active segment (no cache slot) always opens fresh. Returns - /// `None` only when the open exhausts its retries (the caller fails closed). + /// Resolve the read-only descriptor for `segment`'s file. A hit clones the + /// cached fd (sharing the kernel fd, no syscall); a miss opens by path and + /// stores the fd back so later polls skip the `openat`. Returns `None` only + /// when the open exhausts its retries (the caller fails closed). async fn resolve_segment_file( &self, segment: &DiskSegment, path: &str, ) -> Option { - let Some(handle) = &segment.read_state else { - return self.open_segment_with_retry(path).await; - }; + let handle = &segment.read_state; // Borrow only to clone the `Option` out, never across the await. if let Some(cached) = handle.fd.borrow().clone() { return Some(cached); } let file = self.open_segment_with_retry(path).await?; - // Store back only while the pump tracks this handle; an untracked - // fill (walk-through segment, or a slot evicted mid-poll) would pin an - // fd outside the LRU budget, so it opens transiently instead. Benign - // race: a concurrent poll of the same segment may have filled the slot - // while this open was in flight; overwriting with an equivalent fd - // (same inode) is harmless. - if handle.tracked.get() { + // A sealed segment stores back only while the pump tracks its handle; + // an untracked fill (walk-through segment, or a slot evicted mid-poll) + // would pin an fd outside the LRU budget, so it opens transiently + // instead. The active segment's slot is not LRU-budgeted (one per + // partition, dropped when it seals), so it always fills. Benign race: a + // concurrent poll of the same segment may have filled the slot while + // this open was in flight; overwriting with an equivalent fd (same + // inode) is harmless, as is filling a slot the pump orphaned mid-poll. + if !segment.sealed || handle.tracked.get() { *handle.fd.borrow_mut() = Some(file.clone()); } Some(file) @@ -655,14 +662,20 @@ impl DiskReadPlan { /// whole on the first sealed poll and is cached on the shared handle; a /// larger one is binary-searched on file every poll and never materialized /// (see the constant). Returns `None` (keep the byte-0 fallback) for the - /// active segment (no handle), a below-range query, or an IO failure. + /// active segment, a below-range query, or an IO failure. async fn resolve_sealed_start( &self, segment: &DiskSegment, query: MessageLookup, partition_dir: &str, ) -> Option { - let handle = segment.read_state.as_ref()?; + // The active segment grows under the reader, so neither the shared + // sparse index nor the offset memo can describe it; its own resident + // index already resolved `start_position`. + if !segment.sealed { + return None; + } + let handle = &segment.read_state; // Cache hit: resolve under a short borrow, never across the await. let cached = handle .index @@ -1074,7 +1087,8 @@ mod tests { let segment = DiskSegment { start_offset: 0, persisted: u64::MAX, - read_state: Some(Rc::clone(&handle)), + read_state: Rc::clone(&handle), + sealed: true, }; let plan = DiskReadPlan { partition_dir: PartitionDirResolution::Resolved(dir.display().to_string()), diff --git a/core/partitions/src/state_transfer.rs b/core/partitions/src/state_transfer.rs index 2e0f00c919..5bd29807db 100644 --- a/core/partitions/src/state_transfer.rs +++ b/core/partitions/src/state_transfer.rs @@ -2347,6 +2347,10 @@ where self.log.index_writers_mut()[last] = Some(Rc::new(index_writer)); } self.log.segments_mut()[last].sealed = false; + // The active and sealed read-state caches live under different + // rules (see `SegmentedLog::reset_read_state`), so the slot cannot + // carry its sealed identity into active use. + self.log.reset_read_state(last); } // The installed segments supersede every journaled op; stale @@ -2663,6 +2667,10 @@ where minted_next_offset: u64, staged_was_empty: bool, ) -> Result<(), iggy_common::IggyError> { + // The empty plant below can land on a base offset this sweep unlinks, + // so an in-flight poll's cached read fd would keep serving the retired + // inodes as live data. Same hazard and same fix as `purge`. + self.log.invalidate_sealed_read_state(); while let Some((_, mut storage)) = self.log.retire_front() { let _ = storage.shutdown(); } diff --git a/core/partitions/src/types.rs b/core/partitions/src/types.rs index bf9d936f68..6c956f7e35 100644 --- a/core/partitions/src/types.rs +++ b/core/partitions/src/types.rs @@ -15,11 +15,30 @@ // specific language governing permissions and limitations // under the License. +use iggy_binary_protocol::Operation; use iggy_common::{EncryptorKind, IggyByteSize, PollingStrategy}; use server_common::iobuf::Frozen; use smallvec::SmallVec; use std::sync::Arc; +/// A local commit that failed for an op the cluster had already committed. +/// +/// The replica is divergent from here on: `drain_committable_prefix` popped +/// the op, its `commit_min` never advanced, and the next +/// `advance_commit_min` would assert on the gap. Continuing would either +/// serve a prefix the cluster has moved past or panic somewhere less +/// legible, so the partition fences itself on this and the shard brings the +/// server down through the ordinary shutdown path. Recorded rather than +/// panicked because a panic on the pump task is swallowed by the runtime, +/// which wedges every partition on the shard while the process reports +/// healthy. +#[derive(Debug, Clone)] +pub struct FatalCommit { + pub namespace_raw: u64, + pub op: u64, + pub operation: Operation, +} + #[derive(Debug, Clone)] pub struct Fragment { source: Frozen, diff --git a/core/server/config.toml b/core/server/config.toml index 2bac202b83..583df57320 100644 --- a/core/server/config.toml +++ b/core/server/config.toml @@ -443,9 +443,14 @@ path = "partitions" validate_checksum = true # Durability and flush cadence are per topic, set at CreateTopic: -# enforce_fsync - fsync each write (false when unset) +# enforce_fsync - fdatasync each flush (false when unset) # messages_required_to_save - flush after this many messages (1024) # size_of_messages_required_to_save - flush after this many bytes (1 MiB) +# enforce_fsync decides whether a flush syncs, never whether one happens. A +# write is acknowledged when it commits, which precedes its flush unless that +# commit trips a threshold, and the partition journal holding the unflushed +# remainder is in memory and is not replayed at boot. Acks are therefore +# durability-gated only at messages_required_to_save = 1. # Segment configuration [system.segment] @@ -471,30 +476,56 @@ recreate_missing_state = false # At boot, segment recovery walks each partition's segments: bytes after the # last decodable batch of a genuinely torn tail are physically truncated from -# the .log/.index files. A LOST index is rebuilt from the batches the walk -# proves -- lost meaning absent altogether, or holding no whole 24-byte -# entry. An index that still holds whole entries is NEVER rebuilt, only -# floored to whole entries, or the partition is refused when its entries -# contradict the log. Only the walk of a segment whose index was lost -# re-checksums every batch; with an intact index the walk trusts batch -# headers, except that a batch contradicting the walk (see below) must pass -# its batch checksum before the contradiction is believed. Bytes before the -# last index entry are not re-examined at boot at all -- at-rest damage -# there surfaces on the read path via validate_checksum. +# the .log/.index files. Every index entry is derived from a batch the log +# already held, so an index is never evidence about the log: one the log +# contradicts is repaired, not believed. It is DROPPED whole and rebuilt from +# a byte-0 walk when it holds no whole 24-byte entry, when its entries +# regress in offset or position, when an entry does not match the log batch's +# offset, timestamp, partition and position, when its first entry is not the +# segment's own start offset at position 0, or when the log cannot back its +# LAST entry. Only an index whose +# last entry the log proves anchors a walk, which then runs from that entry +# to the file end. Under the topic's enforce_fsync = true an index the log +# cannot back is measured against the log first: a gap of more than ONE entry +# refuses the partition instead of rebuilding, because fsync lets the index +# outrun the log by at most the chunk in flight at the crash, so a wider gap +# is previously durable data the log lost. # -# A restore that dropped every .index therefore boots by re-reading and -# re-checksumming every byte of every segment and rebuilding each index -# durably, two fsyncs per segment: budget minutes, not seconds, on a -# partition holding thousands of segments. A boot that looks hung after such -# a restore is doing that work. Before, that same shape refused the boot -# outright. +# The byte-0 walk is deliberately not narrowed to the entries the log still +# proves. An anchor picked that way can sit ABOVE damage, and the residue +# probe only ever looks forward from where a walk stopped, so a hole under +# the anchor would go unread while end_offset still advertised the offsets +# over it. Reaching that branch means the index and the log already disagree, +# so the extra bytes read are bounded by the segments a crash actually tore. # -# Either walk refuses a batch whose base offset does not continue the chain, -# or whose partition_id stamp is not this partition's. A contradiction whose -# batch fails its own checksum is damage rather than data: the walk stops -# there and the residue goes through the damage probe like any other torn -# tail, so a bit flip in a tail batch's offset truncates instead of refusing -# the partition, regardless of the flip's direction. +# Every batch a walk accepts passes its own checksum. With enforce_fsync = +# false, the index only locates data because page-cache writeback can preserve +# a later log page while losing an earlier one, so recovery walks every segment +# from byte 0. A clean walk preserves the existing index and performs no disk +# mutation. With enforce_fsync = true, completed serialized log fdatasyncs +# prove the prefix before the last index entry, so a healthy boot starts at +# that last entry and leaves earlier at-rest damage to the read path's +# validate_checksum. Wherever a recovery walk finds damage, it truncates only +# when nothing decodable follows and refuses when something does. +# +# The non-fsync verification pass therefore reads and checksums every log byte +# at boot, although it keeps a clean index in place. A byte-0 repair walk also +# rebuilds a missing or contradicted index durably, adding two fsyncs per +# repaired segment. Either path can take minutes over thousands of segments. +# A restore that dropped every .index repairs every segment; an ordinary +# fsynced crash usually repairs only its tail segment. A slow boot is therefore +# not by itself evidence of a stall. Before, the repair shapes refused boot. +# +# Either walk refuses a verifying batch whose base offset does not continue +# the chain, or whose partition_id stamp is not this partition's. The one +# exception is the first batch at an index-derived anchor: the offset it +# fails to match is the ENTRY's, not the log's, so a mismatch there breaks +# the walk as a stale entry, leaving the verdict to the byte-0 rebuild. A +# batch that fails its own checksum is damage rather than data +# whatever its fields claim: the walk stops there and the residue goes +# through the damage probe like any other torn tail, so a bit flip in a tail +# batch's offset truncates instead of refusing the partition, regardless of +# the flip's direction. # # Damage in the middle of a segment is never silently truncated: the # partition is refused, as is trailing residue whose verification cost diff --git a/core/server/src/bootstrap.rs b/core/server/src/bootstrap.rs index 7d8a277162..a5840277ef 100644 --- a/core/server/src/bootstrap.rs +++ b/core/server/src/bootstrap.rs @@ -86,7 +86,7 @@ use metadata::stm::snapshot::Snapshot; use metadata::stm::stream::{Partition, Streams}; use metadata::stm::user::Users; use partitions::{ - IggyIndexWriter, IggyPartition, IggyPartitions, MessagesWriter, PartitionsConfig, + FatalCommit, IggyIndexWriter, IggyPartition, IggyPartitions, MessagesWriter, PartitionsConfig, }; use rustls::pki_types::ServerName; use server_common::Message; @@ -1271,8 +1271,23 @@ async fn shard_main( // tracked pump would be cancelled by runtime teardown mid final-flush // and every graceful shutdown would silently drop the committed journal // tail that had not hit a flush threshold yet. + let pump_shutdown_flag = Arc::clone(&shutdown_flag_for_handoff); let mut pump_handle = Some(compio::runtime::spawn(async move { - pump_shard.run_message_pump(stop_rx).await; + // The pump itself flips the shared flag when a commit fault stops it, + // BEFORE its final flush, so a flush stalling on the failed device + // still reaches the watchdog and the bounded drain. Every sibling + // shard's watchdog drives its own graceful stop off the same flag; + // this shard's watchdog is what fires the token `shard_main` is + // parked on. The store below backstops the one fault the pump can + // only observe after that flip: a partition fenced by the final + // flush itself. + let fatal = pump_shard + .run_message_pump(stop_rx, Arc::clone(&pump_shutdown_flag)) + .await; + if fatal.is_some() { + pump_shutdown_flag.store(true, Ordering::Relaxed); + } + fatal })); let reconciler_ctx = Rc::new(crate::partition_reconciler::ReconcilerCtx::new( @@ -1513,7 +1528,7 @@ async fn shard_main( /// wrapper alone cannot see it, and a shard that swallows it prints /// "exited cleanly" over a corpse. async fn await_pump_drain( - pump_handle: Option>, + pump_handle: Option>>, config: &ServerConfig, shard_id: u16, ) -> Result<(), ServerError> { @@ -1541,7 +1556,26 @@ async fn await_pump_drain( // reaches the tracing sink too. let reason = match panic::catch_unwind(panic::AssertUnwindSafe(|| join_result.resume_unwind())) { - Ok(Some(())) => return Ok(()), + Ok(Some(None)) => return Ok(()), + // The pump drained and flushed; it just has nothing left to serve. + // Fail the shard so the process exits non-zero: a node that stopped + // because it could not persist a cluster-committed op must not look + // to an orchestrator like a clean shutdown. + Ok(Some(Some(fault))) => { + error!( + shard = shard_id, + namespace_raw = fault.namespace_raw, + op = fault.op, + operation = ?fault.operation, + "message pump stopped on a partition commit fault; \ + the server is shutting down" + ); + return Err(ServerError::ShardFatal { + shard_id, + namespace_raw: fault.namespace_raw, + op: fault.op, + }); + } Ok(None) => "task was cancelled".to_string(), Err(payload) => payload .downcast_ref::<&str>() @@ -2549,6 +2583,10 @@ fn restore_metadata_consensus( /// Recover this partition's persisted segment chain, stamping each segment /// with the topic's effective segment size (the per-topic value when the /// topic was created with one, else the shard-wide configured size). +/// +/// The topic's effective `enforce_fsync` goes in for the same reason: it is +/// what tells recovery whether a durable index entry the log cannot back is a +/// benign torn index or previously durable data the log lost. async fn recover_partition_segments( config: &ServerConfig, namespace: IggyNamespace, @@ -2561,12 +2599,16 @@ async fn recover_partition_segments( let segment_size = runtime_options .segment_size .unwrap_or_else(|| IggyByteSize::from(iggy_common::DEFAULT_SEGMENT_SIZE)); + let enforce_fsync = runtime_options + .enforce_fsync + .unwrap_or(iggy_common::DEFAULT_ENFORCE_FSYNC); load_persisted_segments( config, stream_id, topic_id, partition_id, segment_size, + enforce_fsync, stats, ) .await @@ -4740,7 +4782,7 @@ mod tests { .expect("a fresh ServerConfig owns its system config") .sharding .shutdown_drain_timeout = iggy_common::IggyDuration::new(timeout); - let pump = compio::runtime::spawn(std::future::pending::<()>()); + let pump = compio::runtime::spawn(std::future::pending::>()); let error = await_pump_drain(Some(pump), &config, 7) .await @@ -4754,6 +4796,32 @@ mod tests { )); } + #[compio::test] + async fn pump_stopped_by_a_commit_fault_is_not_reported_as_clean() { + // The pump drained and flushed, so the join succeeds. Reporting that + // as a clean exit would hand an orchestrator exit code 0 for a node + // that stopped because it could not persist a cluster-committed op. + let config = ServerConfig::default(); + let fault = FatalCommit { + namespace_raw: 42, + op: 7, + operation: iggy_binary_protocol::Operation::SendMessages, + }; + let pump = compio::runtime::spawn(async move { Some(fault) }); + + let error = await_pump_drain(Some(pump), &config, 3) + .await + .expect_err("a pump that stopped on a commit fault is not a clean exit"); + assert!(matches!( + error, + ServerError::ShardFatal { + shard_id: 3, + namespace_raw: 42, + op: 7, + } + )); + } + #[compio::test] async fn signal_bootstrap_complete_aborts_when_owner_drops_rx() { // Shard 0 aborted before draining and dropped its receiver; a peer's diff --git a/core/server/src/dispatch.rs b/core/server/src/dispatch.rs index f5ef7980f5..008f6f84cf 100644 --- a/core/server/src/dispatch.rs +++ b/core/server/src/dispatch.rs @@ -3854,6 +3854,7 @@ mod tests { use std::future::Future; use std::mem::size_of; use std::rc::Rc; + use std::sync::atomic::AtomicBool; type TestMux = MuxStateMachine; type TestShard = IggyShard; @@ -4514,7 +4515,9 @@ mod tests { let (stop_tx, stop_rx) = shard::channel::<()>(1); let pump_shard = Rc::clone(&shard); let pump = compio::runtime::spawn(async move { - pump_shard.run_message_pump(stop_rx).await; + pump_shard + .run_message_pump(stop_rx, Arc::new(AtomicBool::new(false))) + .await; }); lane_sender @@ -4560,7 +4563,9 @@ mod tests { // post-loop drain must still deliver the reply-lane frame. let (stop_tx, stop_rx) = shard::channel::<()>(1); stop_tx.try_send(()).expect("stop channel has capacity"); - shard.run_message_pump(stop_rx).await; + shard + .run_message_pump(stop_rx, Arc::new(AtomicBool::new(false))) + .await; let replies = bus.client_replies.borrow(); assert_eq!( diff --git a/core/server/src/segment_recovery.rs b/core/server/src/segment_recovery.rs index f4b630de8a..2c889fa92b 100644 --- a/core/server/src/segment_recovery.rs +++ b/core/server/src/segment_recovery.rs @@ -64,6 +64,36 @@ const SCAN_WINDOW_CAPACITY: usize = 4 * 1024 * 1024; /// batch always gets one. const REBUILT_INDEX_STRIDE_BYTES: u64 = 64 * 1024; +/// Sparse index entries either index scan steps through between reactor +/// yields, on top of the per-refill yield the anchor search shares with the +/// walks. Neither scan is bounded by disk reads: an entry whose position is +/// past the log end is rejected on arithmetic alone (an index floored back to +/// a short log is mostly those), and the consistency scan reads a whole window +/// of entries per pread and then compares them in memory. Without this a +/// megabyte-scale index would hold the shard core -- signal handling included +/// -- for its whole length on every clean boot. +const INDEX_SCAN_YIELD_STRIDE: u64 = 1024; + +/// Index entries the log may legitimately fail to back under `enforce_fsync`. +/// Persistence writes exactly one entry per flush chunk and chunks never +/// overlap. The two halves fdatasync concurrently WITHIN one flush, but +/// flushes are serialized, and the log's fdatasync covers the whole file: an +/// entry existing above entry N therefore proves the log was synced through +/// chunk N. Only the chunk in flight when the process died can leave an entry +/// the log never backed. See [`PartitionRecoveryRefusal::FsyncedLogLoss`] for +/// why a deeper step-back is evidence about the log rather than about the +/// index. +const MAX_FSYNCED_INDEX_STEP_BACK_ENTRIES: u64 = 1; + +/// Index entries the backward anchor search probes before giving up, at one +/// 24-byte pread each. Capping the walk cannot change a verdict: only the +/// FIRST entry probed can leave the step-back within +/// [`MAX_FSYNCED_INDEX_STEP_BACK_ENTRIES`], so past that entry the refusal is +/// already decided and deeper probes only sharpen the byte it names. The +/// refusal carries the depth actually searched, so a capped search never +/// reads as "the log backs nothing". +const MAX_INDEX_ANCHOR_PROBE_ENTRIES: u64 = 4096; + /// Units the damage probes of one partition load may spend per byte of /// residue they are asked to classify, at one unit per candidate offset /// examined. Candidates never outnumber residue bytes, so an honest @@ -134,11 +164,15 @@ pub struct RecoveredSegment { /// /// Transient I/O failures (listing, stat, open, read, truncate, fsync) are /// returned as-is and abort the boot so it can be retried. Structural -/// contradictions -- a holed chain, an index diverging from its log, damage -/// with intact batches after it, residue the damage probe could not classify -/// within its limits -- return +/// contradictions -- a holed chain, damage with intact batches after it, +/// residue the damage probe could not classify within its limits -- return /// [`ServerError::PartitionRecoveryRefused`] so the caller can fence this one /// partition instead of taking the node down. +/// +/// `enforce_fsync` is the topic's own effective value, not a hint: it is what +/// makes a durable index entry evidence about the log (see +/// [`PartitionRecoveryRefusal::FsyncedLogLoss`]), so passing it wrong either +/// refuses healthy chains or hides previously durable data loss. #[allow(clippy::too_many_lines)] pub async fn load_persisted_segments( config: &ServerConfig, @@ -146,6 +180,7 @@ pub async fn load_persisted_segments( topic_id: usize, partition_id: usize, segment_size: IggyByteSize, + enforce_fsync: bool, stats: &PartitionStats, ) -> Result, ServerError> { let partition_path = config @@ -193,6 +228,7 @@ pub async fn load_persisted_segments( &messages_path, start_offset, raw_messages_size, + enforce_fsync, &mut scratch, ) .await?; @@ -205,9 +241,9 @@ pub async fn load_persisted_segments( // bytes with `end_offset == start_offset` would fabricate one // phantom message for the bootstrap non-empty filters and strand // undecodable garbage inside the readable range. Note this is NOT - // tail-only -- a torn index is reachable mid-chain on the shipped - // `enforce_fsync = false`, which is why the walk exists rather than - // refusing the partition. + // tail-only -- the log and the index persist concurrently under + // every config, so a torn index is reachable mid-chain, which is why + // the walk exists rather than refusing the partition. let recovered_empty = bounds.is_none(); let bounds = bounds.unwrap_or_else(|| { if raw_messages_size > 0 { @@ -282,14 +318,18 @@ pub async fn load_persisted_segments( plan.segment.start_offset, )?; } - // Log first, index second: a walk only accepts bounds when a whole - // batch decodes at the last index entry's position, so the walked log - // length strictly exceeds that position and every surviving index - // entry still points inside the shortened log even if a crash lands - // between the two mutations. The staged-rebuild install keeps the - // same property: until its rename lands, the on-disk index still - // holds no whole entry, so a crash between the two re-runs the - // index-less walk over the already-truncated log. + // Log first, index second: an index is kept only when a whole batch + // verifies at its LAST entry's position, so the walked log length + // strictly exceeds that position and every kept + // entry still points inside the shortened log even if + // a crash lands between the two mutations -- and a crash that lands + // there anyway leaves an index running ahead of its log, which the + // next boot discards and rebuilds instead of refusing. The + // staged-rebuild install is reached from a populated index too, so + // a crash between the truncate and the rename leaves that unbacked + // index over the shortened log; the next boot re-discards it and + // rebuilds from the log again, and truncation is monotone, so the + // pair converges. truncate_to(&plan.messages_path, messages_size)?; if let Some(staging_path) = &plan.rebuilt_index_staging { install_rebuilt_index(staging_path, &plan.index_path, identity.partition_path)?; @@ -335,10 +375,13 @@ pub async fn load_persisted_segments( stats.increment_segments_count(1); stats.increment_size_bytes(messages_size); if messages_size > 0 { - // Offsets in a segment are contiguous (either walk refuses a - // discontinuity), so the message count is the inclusive span - // between the first (segment start) and last offset. Saturating: - // an end offset at u64::MAX must not wrap the counter. + // The count is the offset span this segment now advertises, which + // is what a consumer can ask for. A byte-0 walk proves every + // offset in a non-fsynced or rebuilt span. An fsynced anchored + // walk proves the final chunk and inherits the earlier prefix from + // completed serialized log fdatasyncs. Counting the whole span + // therefore does not widen the recovery claim. Saturating: an end + // offset at u64::MAX must not wrap the counter. stats.increment_messages_count( plan.segment .end_offset @@ -410,6 +453,13 @@ struct PlannedSegment { /// Scoped to the LOAD, not to one probe: pass A probes every segment before /// pass B can refuse the chain, so a per-probe budget would multiply the /// worst case by the segment count. +/// +/// The limits are therefore a sum of grants, not a function of any one +/// residue: the index anchor search grows them over the log spans it steps +/// across and the damage probe grows them again over residue that can overlap +/// those spans, so a segment whose index was walked backward carries a larger +/// allowance than its own residue would buy. Exhaustion only ever refuses and +/// never truncates, so the looser bound buys boot work, not a weaker verdict. #[derive(Default)] struct ProbeBudget { limit_units: u64, @@ -454,6 +504,16 @@ struct WalkedBounds { rebuilt_index: Option>, } +/// What the batch chain walked from one index entry proves. +struct AnchoredWalk { + /// Timestamp of the first non-empty batch proved by this walk. + start_timestamp: Option, + end_offset: u64, + end_timestamp: u64, + /// First byte past the last whole batch the walk proved. + position: u64, +} + /// Reusable buffers for the walk, probe, and index validation scans, plus the /// probe work budget they share, allocated once per partition load. #[derive(Default)] @@ -978,16 +1038,23 @@ async fn load_index_anchors( /// Derives a segment's readable bounds. `None` when the log holds no whole /// batch at all (the caller recovers the segment as empty). /// -/// With a whole index entry present, the last entry's `position` is only the -/// last flushed chunk's START byte, so the batch chain is walked from there to -/// prove where the segment really ends -- without `enforce_fsync` there is no -/// ordering barrier between the message write and the index write, and a tail -/// torn mid-flush would otherwise pass while `end_offset` claims offsets whose -/// bytes are incomplete. Without one, the log itself is walked from byte 0 and -/// the index is rebuilt from the batches found. Either way, bytes left past -/// the walked prefix go through the damage probe: a torn tail truncates, but -/// damage with intact batches after it -- or residue the probe cannot -/// classify within its limits -- refuses recovery. +/// Without `enforce_fsync`, a consistent index locates batches but cannot prove +/// any log page reached disk: page-cache writeback may preserve a later chunk +/// while losing an earlier one. Recovery therefore checksum-walks the log from +/// byte 0. A clean walk preserves the existing index, while a break falls +/// through to the rebuilding walk so every retained entry describes verified +/// bytes. +/// +/// With `enforce_fsync`, completed serialized flushes prove the prefix before +/// the final index entry, so the last entry's `position` anchors the walk that +/// proves where the segment really ends. An index whose last entry the log +/// cannot back, or whose entries contradict each other, is dropped whole: the +/// log is walked from byte 0 and the index is rebuilt from the batches that +/// walk proves. The unbacked-last path first measures the step-back, and a gap +/// deeper than one entry refuses as previously durable log loss rather than +/// rebuilding. Bytes left past either walked prefix go through the damage +/// probe: a torn tail truncates, while damage with intact batches after it, or +/// residue the probe cannot classify within its limits, refuses recovery. #[allow(clippy::too_many_lines)] async fn recover_segment_bounds( identity: PartitionIdentity<'_>, @@ -995,148 +1062,218 @@ async fn recover_segment_bounds( messages_path: &str, start_offset: u64, messages_size: u64, + enforce_fsync: bool, scratch: &mut ScanScratch, ) -> Result, ServerError> { let (entry_count, first, last) = load_index_anchors(identity, index_path).await?; match (first, last) { (Some(first), Some(last)) => { - // Interior entries were never validated before: a mis-strided - // index can decode to garbage entries that binary searches then - // trust. Monotonicity plus the walk's own anchor bound them: the - // walk below only accepts bounds when a whole batch decodes at - // the LAST entry's position, so ascending positions keep every - // surviving entry inside the truncated log. - validate_index_entries(identity, index_path, start_offset, entry_count, scratch)?; + // A mis-strided or foreign index decodes to garbage entries that + // binary searches would trust, so an index that contradicts itself + // is dropped whole and rebuilt from the log. No `enforce_fsync` + // gate here, unlike the step-back below: the writer cannot emit a + // non-ascending run, so this file is foreign or mis-strided and is + // no witness to what the log once held. + let validation = index_is_consistent( + identity, + index_path, + messages_path, + start_offset, + entry_count, + messages_size, + scratch, + ) + .await?; let messages = open_messages_file(identity, messages_path)?; let mut scanner = FileScanner::new(&messages, messages_size, scratch); + if !validation.structurally_consistent { + return recover_by_walking_log( + identity, + &mut scanner, + messages_path, + start_offset, + messages_size, + ) + .await; + } + // The sparse index holds ONE entry per flushed chunk, pointing // at the chunk's FIRST batch -- `last.offset` is where the last // chunk STARTS, not where the segment ends (a whole journal // flushed as one chunk indexes only its first offset). Walk the // batch chain from that position to the file end to recover the // true end offset. - let mut position = last.position; - let mut end_offset = last.offset; - let mut end_timestamp = last.timestamp; - let mut expected_offset = last.offset; - let mut walked_any = false; - // TODO(hubcio): for batches that continue the chain exactly this - // indexed walk trusts the header decode alone (batches that - // contradict the walk checksum-verify below), so a torn flush - // that persisted the header page but zeroed the body is absorbed - // silently; the index-less walk below checksums every batch. - // Decide whether the indexed arm should checksum too (boot cost) - // or leave body rot to protocol-aware repair. - while position < messages_size { - let header = match scanner.peek_header(position) { - Ok(Some(header)) => header, - Ok(None) => break, - Err(source) => { - return Err(scan_read_failure(identity, messages_path, &source)); - } - }; - let extent = position.saturating_add(header.total_size() as u64); - if extent > messages_size { - break; - } - // A header contradicting the walk -- a foreign partition_id, - // or a base offset that does not continue the chain in - // EITHER direction -- is either damage wearing a decodable - // header or a real record that does not belong at this - // position. The batch checksum tells them apart, because it - // covers both fields and the server mints every legal record - // with it: a batch that fails it is damage, so break and let - // the probe below classify the residue (a bit flip in a tail - // batch then truncates like any torn tail, regardless of the - // flip's direction), while a batch that VERIFIES is durable - // evidence -- of a misdirected or copied write when the - // partition stamp is foreign, of duplicated or lost offsets - // when the chain breaks -- and both absorbing it and - // truncating it would hide that, so refuse and keep the - // bytes. Refusing the verified forward gap over absorbing it - // is deliberate: state transfer's own walk refuses any gap, - // so an absorbed one would mint a segment no peer can ever - // install, and the offsets it fabricates seed current_offset - // under an advance-only superblock persist. - let foreign_partition = header.partition_id != identity.partition_id as u64; - if foreign_partition || header.base_offset != expected_offset { - let verifies = scanner - .slice_at(position, header.total_size()) - .map_err(|source| scan_read_failure(identity, messages_path, &source))? - .is_some_and(|batch| decode_batch_slice(batch).is_ok()); - if !verifies { - break; - } - if foreign_partition { - return Err(identity.refusal(PartitionRecoveryRefusal::ForeignBatch { + let walk = walk_chain_from_anchor( + identity, + &mut scanner, + messages_path, + start_offset, + messages_size, + last, + ) + .await?; + if walk.start_timestamp.is_none() { + // Under `enforce_fsync` the DEPTH of the step-back that would + // find a usable non-empty anchor is evidence about the LOG. + // Flushes are serialized and each one fdatasyncs the whole log + // file before the next one writes, so an entry existing above + // entry N proves the log's fdatasync through chunk N completed. + // Only the chunk in flight at the crash can strand an entry; a + // deeper gap is the log missing bytes a completed flush had + // already made durable. A checksum-valid empty anchor reaches + // this path too: it carries no message range recovery may + // advertise and the production writer never emits one. + // Refuse and keep every byte for the operator. + if enforce_fsync { + let search = find_provable_index_anchor( + identity, + index_path, + messages_path, + &mut scanner, + start_offset, + entry_count, + messages_size, + ) + .await?; + let step_back_entries = entry_count.saturating_sub(search.provable_entries); + if step_back_entries > MAX_FSYNCED_INDEX_STEP_BACK_ENTRIES { + return Err(identity.refusal(PartitionRecoveryRefusal::FsyncedLogLoss { start_offset, - batch_partition_id: header.partition_id, - position, + entry_count, + provable_entries: search.provable_entries, + provable_position: search.provable_position, + searched_entries: search.searched_entries, })); } - return Err( - identity.refusal(PartitionRecoveryRefusal::OffsetDiscontinuity { - start_offset, - expected_offset, - found_offset: header.base_offset, - position, - }), - ); - } - if header.message_count > 0 { - end_offset = header - .base_offset - .saturating_add(u64::from(header.message_count) - 1); - end_timestamp = header.base_timestamp; - expected_offset = end_offset.saturating_add(1); } - walked_any = true; - position = extent; - if scanner.take_refilled() { - yield_to_reactor().await; + // The log cannot back its own last entry, so this index is no + // longer a description of this log and nothing in it is a + // trustworthy anchor. Anchoring on the highest entry that + // still proves would leave `[0, anchor.position)` unread, and + // the damage probe only scans FORWARD from where a walk + // stopped: a hole below the anchor would be invisible, while + // `end_offset` still advertised the offsets over it. That hole + // is exactly what the crash reaching this branch can leave -- + // without `enforce_fsync` writeback order is arbitrary, so an + // unprovable tail entry is equally a torn INDEX tail and + // evidence the LOG lost an interior page. Walk from byte 0 + // instead: it reads the damage, finds the anchor's own batch + // as a survivor past it, and refuses with the bytes preserved, + // and where the log is whole every rebuilt entry is one the + // walk proved rather than one it inherited. + warn!( + stream_id = identity.stream_id, + topic_id = identity.topic_id, + partition_id = identity.partition_id, + start_offset, + messages_size, + entry_count, + last_entry_offset = last.offset, + last_entry_position = last.position, + "the log does not back its last sparse index entry with a non-empty batch; \ + discarding the index and rebuilding it from a byte-0 \ + walk of the log" + ); + let rebuilt = recover_by_walking_log( + identity, + &mut scanner, + messages_path, + start_offset, + messages_size, + ) + .await?; + if enforce_fsync { + ensure_fsynced_rebuild_reaches( + identity, + rebuilt.as_ref(), + start_offset, + entry_count, + last.position, + )?; } + return Ok(rebuilt); } - if !walked_any { - // Probe before concluding: a verifying batch past the bytes - // that broke the walk at the anchor is a survivor, and - // refusing it as a divergence would claim the log holds - // nothing while durable data sits in it. - refuse_if_survivor_past_damage( + + if !validation.mappings_match { + let rebuilt = recover_by_walking_log( identity, &mut scanner, messages_path, - position, - messages_size, - Some(end_offset), start_offset, + messages_size, ) .await?; - return Err( - identity.refusal(PartitionRecoveryRefusal::IndexLogDivergence { + if enforce_fsync { + ensure_fsynced_rebuild_reaches( + identity, + rebuilt.as_ref(), start_offset, - end_offset: last.offset, - messages_size_bytes: messages_size, - indexed_size_bytes: last.position, - }), - ); + entry_count, + last.position, + )?; + } + return Ok(rebuilt); + } + + if !enforce_fsync { + // The last entry proved that the index still describes this + // log. It does not prove earlier log pages reached disk. The + // first entry is exactly `(start_offset, position 0)` by the + // consistency check above, so use it only to seed the expected + // offset and checksum every batch, including unindexed spans. + let full_walk = walk_chain_from_anchor( + identity, + &mut scanner, + messages_path, + start_offset, + messages_size, + first, + ) + .await?; + if full_walk.position == messages_size + && let Some(start_timestamp) = full_walk.start_timestamp + { + return Ok(Some(WalkedBounds { + start_timestamp, + end_timestamp: full_walk.end_timestamp, + end_offset: full_walk.end_offset, + messages_size: full_walk.position, + index_size: entry_count * SPARSE_INDEX_ENTRY_SIZE as u64, + rebuilt_index: None, + })); + } + + // The full walk found damage or no non-empty batch. Re-run + // through the rebuilding path: it probes any residue before a + // destructive verdict and emits an index containing only the + // batches the log itself proves. + return recover_by_walking_log( + identity, + &mut scanner, + messages_path, + start_offset, + messages_size, + ) + .await; } + refuse_if_survivor_past_damage( identity, &mut scanner, messages_path, - position, + walk.position, messages_size, - Some(end_offset), + Some(walk.end_offset), start_offset, ) .await?; Ok(Some(WalkedBounds { start_timestamp: first.timestamp, - end_timestamp, - end_offset, - messages_size: position, + end_timestamp: walk.end_timestamp, + end_offset: walk.end_offset, + messages_size: walk.position, index_size: entry_count * SPARSE_INDEX_ENTRY_SIZE as u64, rebuilt_index: None, })) @@ -1145,10 +1282,10 @@ async fn recover_segment_bounds( // WALKING the log from byte 0 instead of declaring the segment empty. // // The index is not the only self-describing copy -- batch headers carry - // their own offsets, timestamps and lengths -- and with the shipped - // `enforce_fsync = false` there is no write ordering between a log and - // its index, so a torn index is reachable on default config for a - // MID-CHAIN segment too, not just the tail. Recovering that as empty + // their own offsets, timestamps and lengths -- and the log and its + // index persist concurrently under every config, so a torn index is + // reachable for a MID-CHAIN segment too, not just the tail. + // Recovering that as empty // then trips the contiguity guard and refuses the whole partition: // total serve loss (and offset reuse from 0) for a chain whose bytes // are all present. The walk keeps the torn-tail truncation the indexed @@ -1157,142 +1294,441 @@ async fn recover_segment_bounds( _ if messages_size > 0 => { let messages = open_messages_file(identity, messages_path)?; let mut scanner = FileScanner::new(&messages, messages_size, scratch); - let mut position = 0u64; - let mut start_timestamp = None; - let mut end_offset = start_offset; - let mut end_timestamp = 0; - let mut expected_offset = start_offset; - let mut rebuilt_index = Vec::new(); - let mut last_indexed_position: Option = None; - while position < messages_size { - let header = match scanner.peek_header(position) { - Ok(Some(header)) => header, - Ok(None) => break, - Err(source) => { - return Err(scan_read_failure(identity, messages_path, &source)); - } - }; - let extent = position.saturating_add(header.total_size() as u64); - if extent > messages_size { - break; - } - // The FILENAME is the only trustworthy anchor once the index - // is gone, and the header decode checks a length, not a - // checksum. So the batch has to verify before its header is - // believed, and the chain has to be contiguous from the - // filename onward. - let verifies = scanner - .slice_at(position, header.total_size()) - .map_err(|source| scan_read_failure(identity, messages_path, &source))? - .is_some_and(|batch| decode_batch_slice(batch).is_ok()); - if !verifies { - break; - } - if header.partition_id != identity.partition_id as u64 { - // Verified above, so this is a real record minted for - // another partition (a misdirected write, an operator - // copy), not damage: preserve it as evidence. - return Err(identity.refusal(PartitionRecoveryRefusal::ForeignBatch { - start_offset, - batch_partition_id: header.partition_id, - position, - })); - } - if header.base_offset != expected_offset { - // A batch that VERIFIES but does not continue the chain is - // durable data past a hole (or a duplicated range): the - // offsets in between are exactly what a truncation here - // would silently erase, so refuse instead. - return Err( - identity.refusal(PartitionRecoveryRefusal::OffsetDiscontinuity { - start_offset, - expected_offset, - found_offset: header.base_offset, - position, - }), - ); - } - if header.message_count > 0 { - end_offset = header - .base_offset - .saturating_add(u64::from(header.message_count) - 1); - end_timestamp = header.base_timestamp; - start_timestamp.get_or_insert(header.base_timestamp); - expected_offset = end_offset.saturating_add(1); - if last_indexed_position.is_none_or(|indexed| { - position.saturating_sub(indexed) >= REBUILT_INDEX_STRIDE_BYTES - }) { - push_index_entry( - &mut rebuilt_index, - header.base_offset, - header.base_timestamp, - position, - ); - last_indexed_position = Some(position); - } - } - position = extent; - if scanner.take_refilled() { - yield_to_reactor().await; - } - } - refuse_if_survivor_past_damage( + recover_by_walking_log( identity, &mut scanner, messages_path, - position, - messages_size, - start_timestamp.map(|_| end_offset), start_offset, + messages_size, ) - .await?; - let Some(start_timestamp) = start_timestamp else { - // Not one whole batch, and the probe above proved nothing - // decodable follows either: the bytes really are unusable, so - // the caller's empty recovery is right after all. - return Ok(None); - }; - warn!( - stream_id = identity.stream_id, - topic_id = identity.topic_id, - partition_id = identity.partition_id, + .await + } + _ => Ok(None), + } +} + +/// Refuse an fsynced rebuild that stops before the durable prefix implied by +/// the index entry whose flush could begin only after the preceding log +/// fdatasync completed. +fn ensure_fsynced_rebuild_reaches( + identity: PartitionIdentity<'_>, + rebuilt: Option<&WalkedBounds>, + start_offset: u64, + entry_count: u64, + durable_position: u64, +) -> Result<(), ServerError> { + let walked_position = rebuilt.map_or(0, |bounds| bounds.messages_size); + if walked_position < durable_position { + return Err( + identity.refusal(PartitionRecoveryRefusal::FsyncedRebuildShortfall { start_offset, - messages_size, - walked_size = position, - rebuilt_entries = rebuilt_index.len() / SPARSE_INDEX_ENTRY_SIZE, - "sparse index holds no whole entry; recovered segment bounds \ - by walking the log and rebuilding its index from the walked \ - batches" + entry_count, + walked_position, + durable_position, + }), + ); + } + Ok(()) +} + +/// Recovers a segment's bounds from the log alone, walking the batch chain +/// from byte 0 and rebuilding the sparse index from the batches it proves. +/// `None` when no whole batch decodes anywhere (the caller recovers the +/// segment as empty). +/// +/// Reached both when the index holds no whole entry and when it holds +/// entries the log cannot back. Every batch is checksum-verified, as in the +/// anchored walk; here the FILENAME is the only anchor at all, and the +/// header decode checks a length, not a checksum. +async fn recover_by_walking_log( + identity: PartitionIdentity<'_>, + scanner: &mut FileScanner<'_>, + messages_path: &str, + start_offset: u64, + messages_size: u64, +) -> Result, ServerError> { + let mut position = 0u64; + let mut start_timestamp = None; + let mut end_offset = start_offset; + let mut end_timestamp = 0; + let mut expected_offset = start_offset; + let mut rebuilt_index = Vec::new(); + let mut last_indexed_position: Option = None; + while position < messages_size { + let Some(header) = header_at(identity, scanner, messages_path, position)? else { + break; + }; + let extent = position.saturating_add(header.total_size() as u64); + if extent > messages_size { + break; + } + let verifies = batch_verifies( + identity, + scanner, + messages_path, + position, + header.total_size(), + )?; + if !verifies { + break; + } + if header.partition_id != identity.partition_id as u64 { + // Verified above, so this is a real record minted for another + // partition (a misdirected write, an operator copy), not damage: + // preserve it as evidence. + return Err(identity.refusal(PartitionRecoveryRefusal::ForeignBatch { + start_offset, + batch_partition_id: header.partition_id, + position, + })); + } + if header.base_offset != expected_offset { + // A batch that VERIFIES but does not continue the chain is + // durable data past a hole (or a duplicated range): the offsets + // in between are exactly what a truncation here would silently + // erase, so refuse instead. + return Err( + identity.refusal(PartitionRecoveryRefusal::OffsetDiscontinuity { + start_offset, + expected_offset, + found_offset: header.base_offset, + position, + }), ); - Ok(Some(WalkedBounds { - start_timestamp, - end_timestamp, - end_offset, - messages_size: position, - index_size: rebuilt_index.len() as u64, - rebuilt_index: Some(rebuilt_index), - })) } - _ => Ok(None), + if header.message_count > 0 { + end_offset = header + .base_offset + .saturating_add(u64::from(header.message_count) - 1); + end_timestamp = header.base_timestamp; + start_timestamp.get_or_insert(header.base_timestamp); + expected_offset = end_offset.saturating_add(1); + if last_indexed_position.is_none_or(|indexed| { + position.saturating_sub(indexed) >= REBUILT_INDEX_STRIDE_BYTES + }) { + push_index_entry( + &mut rebuilt_index, + header.base_offset, + header.base_timestamp, + position, + ); + last_indexed_position = Some(position); + } + } + position = extent; + if scanner.take_refilled() { + yield_to_reactor().await; + } + } + refuse_if_survivor_past_damage( + identity, + scanner, + messages_path, + position, + messages_size, + start_timestamp.map(|_| end_offset), + start_offset, + ) + .await?; + let Some(start_timestamp) = start_timestamp else { + // Not one whole batch, and the probe above proved nothing decodable + // follows either: the bytes really are unusable, so the caller's + // empty recovery is right after all. + return Ok(None); + }; + warn!( + stream_id = identity.stream_id, + topic_id = identity.topic_id, + partition_id = identity.partition_id, + start_offset, + messages_size, + walked_size = position, + rebuilt_entries = rebuilt_index.len() / SPARSE_INDEX_ENTRY_SIZE, + "recovered segment bounds by walking the log and rebuilding its \ + index from the walked batches" + ); + Ok(Some(WalkedBounds { + start_timestamp, + end_timestamp, + end_offset, + messages_size: position, + index_size: rebuilt_index.len() as u64, + rebuilt_index: Some(rebuilt_index), + })) +} + +/// Walks the batch chain forward from one sparse index entry to the end of +/// the log, proving where the segment really ends. +/// +/// The anchor is only a starting byte and the offset the chain must continue +/// from; nothing about it is assumed to hold. Every batch the walk accepts +/// passes its batch checksum: an index entry proves nothing about the bytes +/// under it, because the log and the index persist concurrently and a crash +/// can leave a batch's header page durable over a body that never landed. +/// The header decode still runs first, so a header that does not fit the +/// file breaks the walk before any checksum is paid, and the verify covers +/// one flushed chunk per segment in the ordinary boot (the last entry points +/// at the last chunk's first batch). When no non-empty batch verifies at the +/// anchor, `start_timestamp` remains `None`, which tells the caller the index +/// does not describe a recoverable message range; the other bounds are then +/// the anchor's own and must not be used. +async fn walk_chain_from_anchor( + identity: PartitionIdentity<'_>, + scanner: &mut FileScanner<'_>, + messages_path: &str, + start_offset: u64, + messages_size: u64, + anchor: IggyIndex, +) -> Result { + let mut position = anchor.position; + let mut start_timestamp = None; + let mut end_offset = anchor.offset; + let mut end_timestamp = anchor.timestamp; + let mut expected_offset = anchor.offset; + // The anchor's offset comes from the INDEX; only a batch this walk has + // already proved makes the expectation the LOG's own. + let mut expectation_from_log = false; + while position < messages_size { + let Some(header) = header_at(identity, scanner, messages_path, position)? else { + break; + }; + let extent = position.saturating_add(header.total_size() as u64); + if extent > messages_size { + break; + } + // The batch checksum covers every header field and the body, and the + // server mints every legal record with it, so it is what tells + // damage wearing a decodable header from a real record: a batch that + // fails it is damage whatever its fields claim -- a torn body under + // an intact header page, a bit flip in either direction -- so break + // and let the probe classify the residue (a torn tail truncates). A + // batch that VERIFIES is durable evidence, and only the verified + // contradictions below earn a refusal. + if !batch_verifies( + identity, + scanner, + messages_path, + position, + header.total_size(), + )? { + break; + } + if header.partition_id != identity.partition_id as u64 { + // A real record minted for another partition (a misdirected or + // copied write), not damage: adopting it would seed this + // partition's offsets from foreign data and truncating it would + // destroy the evidence, so refuse and keep the bytes. + return Err(identity.refusal(PartitionRecoveryRefusal::ForeignBatch { + start_offset, + batch_partition_id: header.partition_id, + position, + })); + } + if header.base_offset != expected_offset { + if !expectation_from_log { + // Still the anchor: the offset this batch failed to match is + // the INDEX ENTRY's, so a verifying batch here contradicts + // the entry, not the chain. Calling that a discontinuity + // would refuse a healthy log over a stale entry, so break and + // leave the verdict to the byte-0 rebuild, which reads the + // chain from the filename onward. + break; + } + // A verified batch that does not continue the chain is durable + // data past a hole or a duplicated range. Absorbing it would + // mint a segment no peer can ever install (state transfer's own + // walk refuses any gap) and seed current_offset with fabricated + // offsets under an advance-only superblock persist; truncating + // it would hide the loss. Refuse and keep the bytes. + return Err( + identity.refusal(PartitionRecoveryRefusal::OffsetDiscontinuity { + start_offset, + expected_offset, + found_offset: header.base_offset, + position, + }), + ); + } + if header.message_count > 0 { + start_timestamp.get_or_insert(header.base_timestamp); + end_offset = header + .base_offset + .saturating_add(u64::from(header.message_count) - 1); + end_timestamp = header.base_timestamp; + expected_offset = end_offset.saturating_add(1); + } + // Any verified batch that matched the expectation makes it the LOG's + // own, an empty batch included: its header carried the offset. Gating + // this on message_count would let a verified empty anchor batch turn + // a real discontinuity after it into the benign anchor-mismatch break + // above, which truncates instead of refusing. + expectation_from_log = true; + position = extent; + if scanner.take_refilled() { + yield_to_reactor().await; + } + } + Ok(AnchoredWalk { + start_timestamp, + end_offset, + end_timestamp, + position, + }) +} + +/// How far the index has outrun the log, in the terms the refusal reports. +struct IndexAnchorSearch { + /// Entries at or below the highest one the log still proves, that entry + /// included; 0 when nothing in the searched window proved. + provable_entries: u64, + /// Position of that entry, or 0 when nothing proved. + provable_position: u64, + /// Entries actually probed, capped by + /// [`MAX_INDEX_ANCHOR_PROBE_ENTRIES`]. + searched_entries: u64, +} + +/// Independent sparse-index validation verdicts. +/// +/// A structural contradiction means the index cannot be writer-produced and +/// carries no fsync evidence, so recovery drops it immediately. A mapping +/// mismatch can instead be the ordinary index-ahead-of-log crash shape, so an +/// fsynced topic must still run the step-back and durable-position gates before +/// rebuilding it. +struct IndexValidation { + structurally_consistent: bool, + mappings_match: bool, +} + +/// Buffered log-header reader used while validating sparse index mappings. +/// Index positions ascend, so each window moves forward and every log byte is +/// read at most once even for an index with one entry per batch. +struct IndexLogScanner<'scan> { + identity: PartitionIdentity<'scan>, + file: &'scan fs::File, + path: &'scan str, + file_len: u64, + window: &'scan mut Vec, + window_start: u64, +} + +impl<'scan> IndexLogScanner<'scan> { + fn new( + identity: PartitionIdentity<'scan>, + file: &'scan fs::File, + path: &'scan str, + file_len: u64, + window: &'scan mut Vec, + ) -> Self { + window.clear(); + Self { + identity, + file, + path, + file_len, + window, + window_start: 0, + } + } + + fn entry_matches( + &mut self, + offset: u64, + timestamp: u64, + position: u64, + ) -> Result { + let Some(header_end) = position.checked_add(COMMAND_HEADER_SIZE as u64) else { + return Ok(false); + }; + if header_end > self.file_len { + return Ok(false); + } + let window_end = self.window_start + self.window.len() as u64; + if position < self.window_start || header_end > window_end { + let fill = usize::try_from((self.file_len - position).min(SCAN_WINDOW_CAPACITY as u64)) + .unwrap_or(SCAN_WINDOW_CAPACITY); + self.window.resize(fill, 0); + self.file + .read_exact_at(&mut self.window[..], position) + .map_err(|source| { + error!( + stream_id = self.identity.stream_id, + topic_id = self.identity.topic_id, + partition_id = self.identity.partition_id, + path = %self.path, + position, + error = %source, + "failed to read a segment log header for sparse index validation" + ); + ServerError::from(IggyError::CannotReadFile) + })?; + self.window_start = position; + } + let at = usize::try_from(position - self.window_start).unwrap_or(0); + Ok( + BatchHeader::decode(&self.window[at..at + COMMAND_HEADER_SIZE]).is_ok_and(|header| { + header.partition_id == self.identity.partition_id as u64 + && header.base_offset == offset + && header.base_timestamp == timestamp + && header.message_count > 0 + && header.total_size() as u64 <= MAX_RECOVERABLE_BATCH_BYTES + && position.saturating_add(header.total_size() as u64) <= self.file_len + }), + ) } } -/// Validates every whole index entry: the first must not claim an offset -/// below the segment's own start, and offsets and positions must strictly -/// ascend (the writer appends one entry per flushed chunk over a growing -/// log, and every chunk covers at least one message and one byte). +/// Highest index entry below the last one that the log can still prove, +/// reported as the number of entries at or below it plus that entry's +/// position, alongside how deep the search went. +/// +/// Reached only under `enforce_fsync`, and only to measure how far the index +/// has outrun the log: the index is dropped whole either way, so nothing is +/// anchored on the entry this returns. What the DEPTH decides is whether the +/// gap is the one chunk a crash can strand or previously durable data the log lost +/// (see [`MAX_FSYNCED_INDEX_STEP_BACK_ENTRIES`]). /// -/// Timestamps are deliberately NOT validated: a primary clock rewind across a -/// restart can legitimately regress persisted `base_timestamp` today, and the -/// lower-bound searches degrade gracefully on a non-monotone run, so refusing -/// would trade availability for nothing. -fn validate_index_entries( +/// An entry proves when the log holds, at its position, the whole non-empty +/// batch it describes: decoding, fitting inside the log, carrying this +/// partition's stamp and the entry's own base offset, and passing its batch +/// checksum. +/// Entries are read one at a time from the back rather than +/// slurped: an index can run to megabytes, and the search stops at the first +/// entry that proves. The dominant shape costs no LOG read at all -- an +/// entry pointing past the log end is rejected on arithmetic alone -- so a +/// torn tail pays one header read for the entry it lands on; the index side +/// still costs a 24-byte pread per entry stepped over, which is what +/// [`MAX_INDEX_ANCHOR_PROBE_ENTRIES`] caps. +/// +/// The log reads are budgeted like the damage probe's: the log span between +/// one probed entry and the one probed before it is the residue that entry +/// classifies. An honest index holds one entry per flushed chunk, so a whole +/// batch fits inside every such span and its verify always fits the residue +/// multiple; entries packed closer together than the batches they claim +/// exhaust it and refuse, rather than paying a verify per entry over an index +/// that never proves. That budget bounds hashed BYTES, so the step count is +/// capped separately. +/// Yields once per window of disk reads, like the walks, and additionally +/// every [`INDEX_SCAN_YIELD_STRIDE`] entries: an entry that overshoots the +/// log reads nothing from it, so there is no refill to key the yield on, and +/// an index that outran a truncated log is full of exactly those. +async fn find_provable_index_anchor( identity: PartitionIdentity<'_>, index_path: &str, + messages_path: &str, + scanner: &mut FileScanner<'_>, start_offset: u64, entry_count: u64, - scratch: &mut ScanScratch, -) -> Result<(), ServerError> { + messages_size: u64, +) -> Result { + // The last entry is the one that just failed to prove out. + let Some(mut entry_index) = entry_count.checked_sub(2) else { + return Ok(IndexAnchorSearch { + provable_entries: 0, + provable_position: 0, + searched_entries: 0, + }); + }; let file = fs::File::open(index_path).map_err(|source| { error!( stream_id = identity.stream_id, @@ -1300,22 +1736,17 @@ fn validate_index_entries( partition_id = identity.partition_id, path = %index_path, error = %source, - "failed to open sparse index for validation during recovery" + "failed to open sparse index for anchor search during recovery" ); ServerError::from(IggyError::CannotReadFile) })?; - let window = &mut scratch.window; - let per_chunk_entries = SCAN_WINDOW_CAPACITY / SPARSE_INDEX_ENTRY_SIZE; - let mut previous: Option<(u64, u64)> = None; - let mut entry_index = 0u64; - let mut byte_position = 0u64; - while entry_index < entry_count { - let chunk_entries = (entry_count - entry_index).min(per_chunk_entries as u64); - // Bounded by the window capacity, so the try_from cannot fail. - let chunk_bytes = - usize::try_from(chunk_entries).unwrap_or(per_chunk_entries) * SPARSE_INDEX_ENTRY_SIZE; - window.resize(chunk_bytes, 0); - file.read_exact_at(&mut window[..], byte_position) + let mut raw = [0u8; SPARSE_INDEX_ENTRY_SIZE]; + // Lowest log byte a probed entry has already paid for; the next probed + // entry is budgeted by the span from its own position up to here. + let mut budgeted_down_to = messages_size; + let mut searched_entries = 0u64; + loop { + file.read_exact_at(&mut raw, entry_index * SPARSE_INDEX_ENTRY_SIZE as u64) .map_err(|source| { error!( stream_id = identity.stream_id, @@ -1323,44 +1754,238 @@ fn validate_index_entries( partition_id = identity.partition_id, path = %index_path, error = %source, - "failed to read sparse index entries for validation during recovery" + "failed to read a sparse index entry for anchor search during recovery" ); ServerError::from(IggyError::CannotReadFile) })?; - for entry in window.as_chunks::().0 { - let entry_offset = read_u64_le(entry, 0); - let entry_position = read_u64_le(entry, 16); - if let Some((previous_offset, previous_position)) = previous - && (entry_offset <= previous_offset || entry_position <= previous_position) + let entry = IggyIndex::new( + read_u64_le(&raw, 0), + read_u64_le(&raw, 8), + read_u64_le(&raw, 16), + ); + searched_entries += 1; + if entry.position < messages_size { + scanner + .budget + .grow_for_residue(budgeted_down_to.saturating_sub(entry.position)); + budgeted_down_to = entry.position; + if let Some(header) = header_at(identity, scanner, messages_path, entry.position)? + && header.partition_id == identity.partition_id as u64 + && header.base_offset == entry.offset + && header.message_count > 0 + && entry.position.saturating_add(header.total_size() as u64) <= messages_size { - return Err( - identity.refusal(PartitionRecoveryRefusal::IndexEntriesNotMonotone { - start_offset, - entry_index, - }), - ); - } - if previous.is_none() && entry_offset < start_offset { - return Err(identity.refusal( - PartitionRecoveryRefusal::IndexEntryBeforeSegmentStart { + if !scanner.budget.charge_verify(header.total_size() as u64) { + return Err(unverified_residue( + identity, + scanner, start_offset, - first_entry_offset: entry_offset, - }, - )); + entry.position, + messages_size, + )); + } + if batch_verifies( + identity, + scanner, + messages_path, + entry.position, + header.total_size(), + )? { + return Ok(IndexAnchorSearch { + provable_entries: entry_index + 1, + provable_position: entry.position, + searched_entries, + }); + } } - previous = Some((entry_offset, entry_position)); - entry_index += 1; } - byte_position += chunk_bytes as u64; + if scanner.take_refilled() || entry_index.is_multiple_of(INDEX_SCAN_YIELD_STRIDE) { + yield_to_reactor().await; + } + if searched_entries == MAX_INDEX_ANCHOR_PROBE_ENTRIES { + break; + } + let Some(next) = entry_index.checked_sub(1) else { + break; + }; + entry_index = next; } - Ok(()) + Ok(IndexAnchorSearch { + provable_entries: 0, + provable_position: 0, + searched_entries, + }) +} + +/// Checks every whole index entry: the first must name the segment's own +/// start -- its start offset, at byte 0 -- and offsets and positions must +/// strictly ascend (the writer appends one entry per flushed chunk over a +/// growing log, and every chunk covers at least one message and one byte). +/// Every entry must also point at a whole non-empty batch in this partition +/// whose offset and timestamp exactly match the entry. Monotonic columns alone +/// are insufficient: an ascending corrupt position can land on a different +/// valid batch, and an offset poll would then silently skip the requested +/// range instead of failing to decode. +/// Every producer of an index mints its first entry there: the writer at +/// `file_position == 0` on a fresh segment, [`recover_by_walking_log`]'s +/// rebuild, and the state-transfer install's own walk. A first entry +/// elsewhere means the file was written mis-strided or over foreign bytes +/// and describes no log at all -- and the accepted path reads that entry's +/// timestamp as the SEGMENT's start timestamp, so a head that belongs to +/// another segment is not merely unused. Logged here, where the entry is +/// known, and the caller rebuilds the index from the log. +/// +/// Timestamp MONOTONICITY is deliberately not required: a primary clock rewind +/// across a restart can legitimately regress persisted `base_timestamp` today, +/// and the lower-bound searches degrade gracefully on a non-monotone run. Each +/// individual entry must still equal the batch timestamp it indexes, or a +/// timestamp lookup can seek to an unrelated valid batch and skip data. +/// +/// Runs on EVERY clean boot over the whole index, and a window holds tens of +/// thousands of entries, so the per-read yield the walks rely on is far too +/// coarse here: yields every [`INDEX_SCAN_YIELD_STRIDE`] entries instead. +#[allow(clippy::too_many_lines)] +async fn index_is_consistent( + identity: PartitionIdentity<'_>, + index_path: &str, + messages_path: &str, + start_offset: u64, + entry_count: u64, + messages_size: u64, + scratch: &mut ScanScratch, +) -> Result { + let index_file = fs::File::open(index_path).map_err(|source| { + error!( + stream_id = identity.stream_id, + topic_id = identity.topic_id, + partition_id = identity.partition_id, + path = %index_path, + error = %source, + "failed to open sparse index for validation during recovery" + ); + ServerError::from(IggyError::CannotReadFile) + })?; + let messages_file = fs::File::open(messages_path).map_err(|source| { + error!( + stream_id = identity.stream_id, + topic_id = identity.topic_id, + partition_id = identity.partition_id, + path = %messages_path, + error = %source, + "failed to open segment log for sparse index validation" + ); + ServerError::from(IggyError::CannotReadFile) + })?; + let ScanScratch { + window: index_window, + spill: log_window, + .. + } = scratch; + let mut log_scanner = IndexLogScanner::new( + identity, + &messages_file, + messages_path, + messages_size, + log_window, + ); + let per_chunk_entries = SCAN_WINDOW_CAPACITY / SPARSE_INDEX_ENTRY_SIZE; + let mut previous: Option<(u64, u64)> = None; + let mut mappings_match = true; + let mut entry_index = 0u64; + let mut byte_position = 0u64; + while entry_index < entry_count { + let chunk_entries = (entry_count - entry_index).min(per_chunk_entries as u64); + // Bounded by the window capacity, so the try_from cannot fail. + let chunk_bytes = + usize::try_from(chunk_entries).unwrap_or(per_chunk_entries) * SPARSE_INDEX_ENTRY_SIZE; + index_window.resize(chunk_bytes, 0); + index_file + .read_exact_at(&mut index_window[..], byte_position) + .map_err(|source| { + error!( + stream_id = identity.stream_id, + topic_id = identity.topic_id, + partition_id = identity.partition_id, + path = %index_path, + error = %source, + "failed to read sparse index entries for validation during recovery" + ); + ServerError::from(IggyError::CannotReadFile) + })?; + for entry in index_window.as_chunks::().0 { + let entry_offset = read_u64_le(entry, 0); + let entry_timestamp = read_u64_le(entry, 8); + let entry_position = read_u64_le(entry, 16); + if let Some((previous_offset, previous_position)) = previous + && (entry_offset <= previous_offset || entry_position <= previous_position) + { + warn!( + stream_id = identity.stream_id, + topic_id = identity.topic_id, + partition_id = identity.partition_id, + start_offset, + entry_index, + "sparse index entry regresses in offset or position; \ + discarding the index and rebuilding it from the log" + ); + return Ok(IndexValidation { + structurally_consistent: false, + mappings_match: false, + }); + } + if previous.is_none() && (entry_offset != start_offset || entry_position != 0) { + warn!( + stream_id = identity.stream_id, + topic_id = identity.topic_id, + partition_id = identity.partition_id, + start_offset, + first_entry_offset = entry_offset, + first_entry_position = entry_position, + "sparse index does not start at the segment's own first batch; \ + discarding the index and rebuilding it from the log" + ); + return Ok(IndexValidation { + structurally_consistent: false, + mappings_match: false, + }); + } + + if mappings_match + && !log_scanner.entry_matches(entry_offset, entry_timestamp, entry_position)? + { + warn!( + stream_id = identity.stream_id, + topic_id = identity.topic_id, + partition_id = identity.partition_id, + start_offset, + entry_index, + entry_offset, + entry_timestamp, + entry_position, + "sparse index entry does not describe the log batch at its position; \ + discarding the index and rebuilding it from the log" + ); + mappings_match = false; + } + previous = Some((entry_offset, entry_position)); + entry_index += 1; + if entry_index.is_multiple_of(INDEX_SCAN_YIELD_STRIDE) { + yield_to_reactor().await; + } + } + byte_position += chunk_bytes as u64; + } + Ok(IndexValidation { + structurally_consistent: true, + mappings_match, + }) } /// Opens a segment's messages file for the recovery walk. Fail-stop on any /// failure, mirroring `file_len`: recovery truncates to the bounds the walk -/// produces, so folding an open failure into "walked nothing" would route a -/// healthy indexed segment into a divergence refusal -- or an index-less one -/// into recover-as-empty, fencing the whole log out of service. +/// produces, so folding an open failure into "walked nothing" would discard +/// a healthy indexed segment's index -- or route an index-less one into +/// recover-as-empty, fencing the whole log out of service. fn open_messages_file( identity: PartitionIdentity<'_>, messages_path: &str, @@ -1378,6 +2003,35 @@ fn open_messages_file( }) } +/// The batch header at `position`, or `None` when the walk must stop there +/// (nothing decodes, or the header runs past the end of the walked file). +fn header_at( + identity: PartitionIdentity<'_>, + scanner: &mut FileScanner<'_>, + messages_path: &str, + position: u64, +) -> Result, ServerError> { + scanner + .peek_header(position) + .map_err(|source| scan_read_failure(identity, messages_path, &source)) +} + +/// Whether the whole batch at `position` passes its batch checksum. `false` +/// when the claimed extent runs past the end of the walked file, which is +/// the torn-tail shape and not a read failure. +fn batch_verifies( + identity: PartitionIdentity<'_>, + scanner: &mut FileScanner<'_>, + messages_path: &str, + position: u64, + total_size: usize, +) -> Result { + Ok(scanner + .slice_at(position, total_size) + .map_err(|source| scan_read_failure(identity, messages_path, &source))? + .is_some_and(|batch| decode_batch_slice(batch).is_ok())) +} + /// A read failure inside the walk or probe is transient I/O, not evidence /// about the bytes: fail stop rather than classify it as a torn tail, which /// would truncate a healthy segment on an `EIO`. @@ -1404,6 +2058,11 @@ fn scan_read_failure( /// can only exist because an append completed after the damaged region -- so /// discarding it would hide real loss behind a silent boot-time repair. /// +/// Scans FORWARD from `damage_position` only, and a walk that consumed the +/// whole file leaves it nothing to do. Damage BELOW where the caller's walk +/// began is invisible to it, which is why no caller may start a walk part way +/// into a log it has reason to distrust. +/// /// The residue is deliberately NOT width-gated: a torn flush chunk is /// bounded by the CHUNK, not by one record, and with `enforce_fsync = false` /// delayed allocation routinely extends a file far past its written-back @@ -1413,12 +2072,11 @@ fn scan_read_failure( /// (candidates examined, bytes handed to verification), whose exhaustion /// REFUSES and keeps the bytes rather than truncating: past the limits the /// probe has proven nothing, and the cheapest input to construct must never -/// earn the destructive verdict. One exception folds exhaustion forward -/// instead: when the walk proved not a single batch (`chain_end_offset` is -/// `None`), the refusal and the no-survivor verdict converge on the same -/// recover-as-empty outcome -- the pair is fenced aside whole, bytes -/// preserved either way -- so exhaustion there returns `Ok` rather than -/// trading a fence for a tombstone. +/// earn the destructive verdict. That holds even when the walk proved not a +/// single batch: recover-as-empty SERVES the segment and re-mints offsets +/// from the previous frontier, so a residue the probe could not finish +/// scanning may still hide the very survivor that would have refused, and +/// only the refusal keeps the partition dark until an operator looks. async fn refuse_if_survivor_past_damage( identity: PartitionIdentity<'_>, scanner: &mut FileScanner<'_>, @@ -1446,22 +2104,38 @@ async fn refuse_if_survivor_past_damage( survivor_position: position, })) } - ProbeOutcome::BudgetExhausted if chain_end_offset.is_none() => Ok(()), - ProbeOutcome::BudgetExhausted => Err(identity.refusal( - PartitionRecoveryRefusal::UnverifiedResidue { - start_offset, - damage_position, - residue_bytes, - candidates_examined: scanner.budget.spent_units, - budget_units: scanner.budget.limit_units, - verified_bytes: scanner.budget.verify_spent_bytes, - verify_budget_bytes: scanner.budget.verify_limit_bytes, - }, + ProbeOutcome::BudgetExhausted => Err(unverified_residue( + identity, + scanner, + start_offset, + damage_position, + messages_size, )), ProbeOutcome::NoSurvivor => Ok(()), } } +/// Refusal for a scan that ran out of its work budget before classifying +/// the bytes from `damage_position` to the end of the walked file. The +/// counters are diagnostic: they say which budget broke and by how much. +fn unverified_residue( + identity: PartitionIdentity<'_>, + scanner: &FileScanner<'_>, + start_offset: u64, + damage_position: u64, + messages_size: u64, +) -> ServerError { + identity.refusal(PartitionRecoveryRefusal::UnverifiedResidue { + start_offset, + damage_position, + residue_bytes: messages_size.saturating_sub(damage_position), + candidates_examined: scanner.budget.spent_units, + budget_units: scanner.budget.limit_units, + verified_bytes: scanner.budget.verify_spent_bytes, + verify_budget_bytes: scanner.budget.verify_limit_bytes, + }) +} + /// Verdict of the damage probe over the residue past the walked prefix. /// `NoSurvivor` is the only verdict that permits truncation; running out of /// budget is deliberately NOT folded into it, so a residue that is expensive @@ -1476,8 +2150,9 @@ enum ProbeOutcome { BudgetExhausted, } -/// Forward-only buffered reads over one segment file for the recovery walk -/// and the damage probe. Parsing and checksumming happen against an in-memory +/// Buffered reads over one segment file for the recovery walks, the index +/// anchor search, and the damage probe. Parsing and checksumming happen +/// against an in-memory /// window so neither pays a syscall per batch -- the probe advances its /// candidate one byte at a time, and per-candidate preads would turn one /// damaged multi-GiB segment into a boot-length stall. @@ -1544,11 +2219,24 @@ impl<'scan> FileScanner<'scan> { } let window_end = self.window_start + self.window.len() as u64; if position < self.window_start || end > window_end { - let fill = usize::try_from((self.file_len - position).min(SCAN_WINDOW_CAPACITY as u64)) - .unwrap_or(SCAN_WINDOW_CAPACITY); + // A forward move anchors the window at `position`, so the walks + // and the probe stream ahead through it. A backward move (the + // index anchor search stepping down its entries) anchors the + // window to END at `end` instead, so the entries below it land in + // the same window and the refills stay linear in the file rather + // than costing one window per entry. + let window_start = if position < self.window_start { + end.saturating_sub(SCAN_WINDOW_CAPACITY as u64) + } else { + position + }; + let fill = + usize::try_from((self.file_len - window_start).min(SCAN_WINDOW_CAPACITY as u64)) + .unwrap_or(SCAN_WINDOW_CAPACITY); self.window.resize(fill, 0); - self.file.read_exact_at(&mut self.window[..], position)?; - self.window_start = position; + self.file + .read_exact_at(&mut self.window[..], window_start)?; + self.window_start = window_start; self.refilled = true; } // In-window by the branch above, and the window is capacity-bounded, @@ -1885,13 +2573,31 @@ mod tests { fs::read(path).expect("read fixture file") } + /// Bytes of a segment file after recovery moved it into the first fence + /// directory, keyed by the name it had at its original path. + fn fenced_bytes(partition_path: &str, original_path: &str) -> Vec { + let fenced_dir = format!("{partition_path}.fenced.0"); + let name = Path::new(original_path) + .file_name() + .expect("fixture file name"); + fs::read(Path::new(&fenced_dir).join(name)).expect("read fenced fixture file") + } + async fn recover(config: &ServerConfig) -> Result, ServerError> { + recover_under_fsync(config, false).await + } + + async fn recover_under_fsync( + config: &ServerConfig, + enforce_fsync: bool, + ) -> Result, ServerError> { load_persisted_segments( config, STREAM_ID, TOPIC_ID, PARTITION_ID, IggyByteSize::from(SEGMENT_MAX_SIZE), + enforce_fsync, &PartitionStats::default(), ) .await @@ -1959,17 +2665,13 @@ mod tests { assert_eq!(segment.end_offset, 0); assert_eq!(len_of(&messages_path), 0, "the served log must be empty"); assert_eq!(len_of(&index_path), 0, "the served index must be empty"); - let fenced_dir = format!("{partition_path}.fenced.0"); - let fenced = |original: &str| { - Path::new(&fenced_dir).join(Path::new(original).file_name().expect("fixture file name")) - }; assert_eq!( - fs::read(fenced(&messages_path)).expect("read fenced log"), + fenced_bytes(&partition_path, &messages_path), GARBAGE, "the unreadable log bytes must survive in the fence directory" ); assert_eq!( - fs::read(fenced(&index_path)).expect("read fenced index"), + fenced_bytes(&partition_path, &index_path), &GARBAGE[..10], "the unreadable index bytes must survive in the fence directory" ); @@ -2317,7 +3019,8 @@ mod tests { } #[compio::test] - async fn given_non_monotone_index_entries_when_recovering_should_refuse() { + async fn given_non_monotone_index_entries_when_recovering_should_rebuild_the_index_from_the_log() + { let tmp = tempdir().expect("tempdir"); let config = test_config(&tmp); prepare_partition_dir(&config); @@ -2329,63 +3032,179 @@ mod tests { log.extend_from_slice(&batch2); let last_position = (batch0.len() + batch1.len()) as u64; // Interior garbage entry: ascending against its predecessor, so only - // the offset regression to the (valid) last entry exposes it. + // the offset regression to the (valid) last entry exposes it. That + // last entry alone would anchor a clean walk and keep the garbage, + // so the interior check must run before the walk trusts the tail. let mut index = index_entry(0, 0); index.extend_from_slice(&index_entry(50, 100)); index.extend_from_slice(&index_entry(2, last_position)); let (messages_path, index_path) = write_segment(&config, 0, &log, &index); - let error = recover(&config) + let recovered = recover(&config) .await - .err() - .expect("a non-monotone index must refuse recovery"); + .expect("a self-contradicting index over a whole log must be rebuilt, not refused"); - assert!( - matches!( - &error, - ServerError::PartitionRecoveryRefused { - reason: PartitionRecoveryRefusal::IndexEntriesNotMonotone { - entry_index: 2, - .. - }, - .. - } - ), - "expected a non-monotone index refusal, got {error:?}" + assert_eq!(recovered.len(), 1); + assert_eq!(recovered[0].segment.end_offset, 2); + assert_eq!( + bytes_of(&index_path), + index_entry(0, 0), + "the index must be rebuilt from the walked batches" + ); + assert_eq!( + bytes_of(&messages_path), + log, + "rebuilding the index must not touch a whole log" + ); + } + + #[compio::test] + async fn given_ascending_index_entry_pointing_at_another_batch_when_recovering_should_rebuild_the_index() + { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + let batch0 = encoded_batch(0, 1); + let batch1 = encoded_batch(1, 1); + let batch2 = encoded_batch(2, 1); + let batch3 = encoded_batch(3, 1); + let batch2_position = (batch0.len() + batch1.len()) as u64; + let batch3_position = batch2_position + batch2.len() as u64; + let mut log = batch0.clone(); + log.extend_from_slice(&batch1); + log.extend_from_slice(&batch2); + log.extend_from_slice(&batch3); + + // Both columns ascend and the last entry is valid, but the interior + // entry for offset 1 points at the valid batch for offset 2. A poll + // starting from this entry decodes cleanly and can skip offset 1, so + // structural monotonicity alone cannot make the index safe to retain. + let mut index = index_entry(0, 0); + index.extend_from_slice(&index_entry(1, batch2_position)); + index.extend_from_slice(&index_entry(3, batch3_position)); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + + let recovered = recover(&config) + .await + .expect("an index entry pointing at another batch must be rebuilt"); + + assert_eq!(recovered.len(), 1); + assert_eq!(recovered[0].segment.end_offset, 3); + assert_eq!( + bytes_of(&index_path), + index_entry(0, 0), + "every retained index entry must match its log batch" ); assert_eq!(bytes_of(&messages_path), log); - assert_eq!(bytes_of(&index_path), index); } #[compio::test] - async fn given_index_entry_below_segment_start_when_recovering_should_refuse() { + async fn given_index_entry_below_segment_start_when_recovering_should_rebuild_the_index_from_the_log() + { let tmp = tempdir().expect("tempdir"); let config = test_config(&tmp); prepare_partition_dir(&config); - let log = encoded_batch(5, 1); - let index = index_entry(3, 0); + let first = encoded_batch(5, 1); + let second = encoded_batch(6, 1); + let mut log = first.clone(); + log.extend_from_slice(&second); + // The last entry is valid and would anchor a clean walk on its own; + // only the first-entry check catches the offset below the start. + let mut index = index_entry(3, 0); + index.extend_from_slice(&index_entry(6, first.len() as u64)); let (messages_path, index_path) = write_segment(&config, 5, &log, &index); - let error = recover(&config) + let recovered = recover(&config).await.expect( + "an index claiming offsets below the segment start must be rebuilt, not refused", + ); + + assert_eq!(recovered.len(), 1); + assert_eq!(recovered[0].segment.end_offset, 6); + assert_eq!( + bytes_of(&index_path), + index_entry(5, 0), + "the index must be rebuilt from the walked batches" + ); + assert_eq!( + bytes_of(&messages_path), + log, + "rebuilding the index must not touch a whole log" + ); + } + + #[compio::test] + async fn given_index_entry_above_segment_start_when_recovering_should_rebuild_the_index_from_the_log() + { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + let first = encoded_batch(5, 1); + let second = encoded_batch(6, 1); + let third = encoded_batch(7, 1); + let mut log = first.clone(); + log.extend_from_slice(&second); + log.extend_from_slice(&third); + // An index that lost its own head: the last entry is valid and + // anchors a clean walk, so the segment would be accepted with a start + // timestamp taken from a chunk that is not its first. + let mut index = index_entry(6, 0); + index.extend_from_slice(&index_entry(7, (first.len() + second.len()) as u64)); + let (messages_path, index_path) = write_segment(&config, 5, &log, &index); + + let recovered = recover(&config) .await - .err() - .expect("an index claiming offsets below the segment start must refuse recovery"); + .expect("an index starting above the segment start must be rebuilt, not refused"); - assert!( - matches!( - &error, - ServerError::PartitionRecoveryRefused { - reason: PartitionRecoveryRefusal::IndexEntryBeforeSegmentStart { - first_entry_offset: 3, - .. - }, - .. - } - ), - "expected a below-start index refusal, got {error:?}" + assert_eq!(recovered.len(), 1); + assert_eq!(recovered[0].segment.end_offset, 7); + assert_eq!( + bytes_of(&index_path), + index_entry(5, 0), + "the rebuilt index must start at the segment's own first batch" + ); + assert_eq!( + bytes_of(&messages_path), + log, + "rebuilding the index must not touch a whole log" + ); + } + + #[compio::test] + async fn given_index_first_entry_past_byte_zero_when_recovering_should_rebuild_the_index_from_the_log() + { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + let first = encoded_batch(0, 1); + let second = encoded_batch(1, 1); + let third = encoded_batch(2, 1); + let mut log = first.clone(); + log.extend_from_slice(&second); + log.extend_from_slice(&third); + // The offset column opens where the segment does and the entries + // ascend, so every other check reads this index as healthy. Only the + // position is wrong: byte 0 is described by nothing, while the last + // entry lands on its batch and would anchor a clean walk. + let mut index = index_entry(0, first.len() as u64); + index.extend_from_slice(&index_entry(2, (first.len() + second.len()) as u64)); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + + let recovered = recover(&config) + .await + .expect("an index whose first entry skips byte 0 must be rebuilt, not refused"); + + assert_eq!(recovered.len(), 1); + assert_eq!(recovered[0].segment.end_offset, 2); + assert_eq!( + bytes_of(&index_path), + index_entry(0, 0), + "the rebuilt index must open at byte 0" + ); + assert_eq!( + bytes_of(&messages_path), + log, + "rebuilding the index must not touch a whole log" ); - assert_eq!(bytes_of(&messages_path), log); - assert_eq!(bytes_of(&index_path), index); } #[compio::test] @@ -2420,18 +3239,18 @@ mod tests { } #[compio::test] - async fn given_zero_padded_records_when_probing_should_scan_whole_residue_and_recover_empty() { + async fn given_zero_padded_records_when_probing_should_refuse_on_verify_budget() { let tmp = tempdir().expect("tempdir"); let config = test_config(&tmp); - let partition_path = prepare_partition_dir(&config); + prepare_partition_dir(&config); // Torn index forces the index-less walk, and the garbage head keeps // it from decoding anything, so the whole file is probe residue. // Each record's header decodes and claims an 8 KiB batch that fits, - // so aligned candidates pay a (fast-failing) verify. Whether the - // verify budget survives all 128 claims or gives up partway, the - // outcome is the same by design: with no walked batch, exhaustion - // converges with survivor-free on recover-as-empty, and the empty - // recovery fences the pair whole. + // so aligned candidates pay a (fast-failing) verify; 128 claims + // exhaust the residue-derived verify budget partway. Exhaustion must + // refuse rather than recover empty: nothing past the budget horizon + // was scanned, so serving an empty segment would re-mint offsets + // over bytes the probe never classified. let mut log = GARBAGE.to_vec(); for record in 0..128u32 { log.extend_from_slice(&zero_padded_record( @@ -2441,22 +3260,25 @@ mod tests { } let (messages_path, _index_path) = write_segment(&config, 0, &log, &GARBAGE[..10]); - let recovered = recover(&config) + let error = recover(&config) .await - .expect("a survivor-free residue must recover as empty, not refuse"); + .err() + .expect("an exhausted probe over an unwalked residue must refuse"); - assert_eq!(recovered.len(), 1); - assert_eq!(recovered[0].segment.size, IggyByteSize::default()); - assert_eq!(len_of(&messages_path), 0, "the served log must be empty"); - let fenced_log = Path::new(&format!("{partition_path}.fenced.0")).join( - Path::new(&messages_path) - .file_name() - .expect("log file name"), - ); - assert_eq!( - fs::read(fenced_log).expect("read fenced log"), - log, - "the unclassifiable bytes must survive in the fence directory" + assert!( + matches!( + &error, + ServerError::PartitionRecoveryRefused { + reason: PartitionRecoveryRefusal::UnverifiedResidue { .. }, + .. + } + ), + "expected a budget-exhausted refusal, got {error:?}" + ); + assert_eq!( + bytes_of(&messages_path), + log, + "a refusal must leave the log byte-identical" ); } @@ -2540,37 +3362,40 @@ mod tests { } #[compio::test] - async fn given_overlapping_verify_claims_with_no_walked_batch_when_probing_should_recover_empty() - { + async fn given_overlapping_verify_claims_with_no_walked_batch_when_probing_should_refuse() { let tmp = tempdir().expect("tempdir"); let config = test_config(&tmp); - let partition_path = prepare_partition_dir(&config); - // Same bait with a garbage head, so the walk proves not one batch: - // exhaustion and survivor-free converge on recover-as-empty here, - // and the pair is fenced whole -- bytes preserved without minting a - // tombstone for a file that provably serves nothing. + prepare_partition_dir(&config); + // Same bait with a garbage head, so the walk proves not one batch. + // Exhaustion must still refuse: recover-as-empty would SERVE the + // segment and re-mint offsets while a real survivor may hide past + // the budget horizon, which is exactly the destructive verdict the + // cheapest-to-construct residue must never earn. let mut log = GARBAGE.to_vec(); for sequence in 0..64u32 { log.extend_from_slice(&bait_record(100, 8 * 1024, sequence + 1)); } let (messages_path, _index_path) = write_segment(&config, 0, &log, &GARBAGE[..10]); - let recovered = recover(&config) + let error = recover(&config) .await - .expect("verify-budget exhaustion with no walked batch must recover as empty"); + .err() + .expect("verify-budget exhaustion must refuse even with no walked batch"); - assert_eq!(recovered.len(), 1); - assert_eq!(recovered[0].segment.size, IggyByteSize::default()); - assert_eq!(len_of(&messages_path), 0, "the served log must be empty"); - let fenced_log = Path::new(&format!("{partition_path}.fenced.0")).join( - Path::new(&messages_path) - .file_name() - .expect("log file name"), + assert!( + matches!( + &error, + ServerError::PartitionRecoveryRefused { + reason: PartitionRecoveryRefusal::UnverifiedResidue { .. }, + .. + } + ), + "expected a budget-exhausted refusal, got {error:?}" ); assert_eq!( - fs::read(fenced_log).expect("read fenced log"), + bytes_of(&messages_path), log, - "the unclassifiable bytes must survive in the fence directory" + "a refusal must leave the log byte-identical" ); } @@ -2673,6 +3498,52 @@ mod tests { ); } + #[compio::test] + async fn given_exhausted_probe_budget_with_no_walked_batch_when_classifying_should_refuse() { + let tmp = tempdir().expect("tempdir"); + let log = GARBAGE.to_vec(); + let messages_path = tmp.path().join("00000000000000000000.log"); + fs::write(&messages_path, &log).expect("write log fixture"); + let file = fs::File::open(&messages_path).expect("open log fixture"); + let partition_path = tmp.path().to_string_lossy().into_owned(); + let identity = PartitionIdentity { + partition_path: &partition_path, + stream_id: STREAM_ID, + topic_id: TOPIC_ID, + partition_id: PARTITION_ID, + }; + let mut scratch = ScanScratch::default(); + scratch.probe_budget.spent_units = u64::MAX / 2; + let mut scanner = FileScanner::new(&file, log.len() as u64, &mut scratch); + + // A walk that proved nothing exhausts the probe over the whole file. + // Recover-as-empty here would SERVE the segment and re-mint offsets + // over bytes the probe never finished scanning; only refusal holds. + let error = refuse_if_survivor_past_damage( + identity, + &mut scanner, + &messages_path.to_string_lossy(), + 0, + log.len() as u64, + None, + 0, + ) + .await + .expect_err("an exhausted probe with no walked batch must refuse, not recover empty"); + + assert!( + matches!( + &error, + ServerError::PartitionRecoveryRefused { + reason: PartitionRecoveryRefusal::UnverifiedResidue { .. }, + .. + } + ), + "expected a budget-exhausted refusal, got {error:?}" + ); + assert_eq!(bytes_of(&messages_path.to_string_lossy()), log); + } + #[compio::test] async fn given_indexed_offset_regression_when_recovering_should_refuse_without_panicking() { let tmp = tempdir().expect("tempdir"); @@ -2850,10 +3721,10 @@ mod tests { let config = test_config(&tmp); prepare_partition_dir(&config); // The failing gap batch is the FIRST thing past the last index - // entry, so the walk breaks with nothing walked. The probe must - // still run: a verifying batch past the damage is a survivor and - // refuses as interior damage, not as a divergence claiming the log - // holds nothing. + // entry, so the anchored walk breaks with nothing walked and the log + // rebuild takes over. Its probe must still find the verifying batch + // past the damage and refuse as interior damage, rather than + // truncating a survivor away with the index. let mut corrupt = encoded_batch(5, 1); corrupt[HEADER_BASE_OFFSET_OFFSET] ^= 0x04; let mut log = corrupt; @@ -2881,38 +3752,909 @@ mod tests { } #[compio::test] - async fn given_failing_gap_batch_at_index_anchor_with_no_survivor_when_recovering_should_refuse_divergence() + async fn given_failing_gap_batch_at_index_anchor_with_no_survivor_when_recovering_should_fence_and_recover_empty() { let tmp = tempdir().expect("tempdir"); let config = test_config(&tmp); - prepare_partition_dir(&config); - // Same anchor shape with nothing verifying past it: the index claims - // a batch where none decodes and verifies, which is the divergence - // verdict -- non-destructive, bytes preserved. + let partition_path = prepare_partition_dir(&config); + // Same anchor shape with nothing verifying past it. The index backs + // nothing, so it is dropped and the log walked on its own; the log + // proves nothing either, which is the ordinary unreadable-pair + // verdict: bytes fenced aside, fresh empty files seeded. let mut log = encoded_batch(5, 1); log[HEADER_BASE_OFFSET_OFFSET] ^= 0x04; let index = index_entry(0, 0); let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + let recovered = recover(&config) + .await + .expect("an anchor batch that fails its checksum must not refuse the partition"); + + assert_eq!(recovered.len(), 1); + assert_eq!(recovered[0].segment.size, IggyByteSize::default()); + assert_eq!(len_of(&messages_path), 0, "the served log must be empty"); + assert_eq!(len_of(&index_path), 0, "the served index must be empty"); + assert_eq!( + fenced_bytes(&partition_path, &messages_path), + log, + "the undecodable log bytes must survive in the fence directory" + ); + assert_eq!( + fenced_bytes(&partition_path, &index_path), + index, + "the unbacked index bytes must survive in the fence directory" + ); + } + + #[compio::test] + async fn given_index_entry_past_the_log_end_when_recovering_should_rebuild_the_index_from_the_log() + { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + // The index reached disk one entry ahead of the log it indexes -- no + // barrier orders the two files -- so the last entry points exactly at + // the log end. The log is whole; only the index overshot. + let batch0 = encoded_batch(0, 1); + let batch1 = encoded_batch(1, 1); + let mut log = batch0.clone(); + log.extend_from_slice(&batch1); + let mut index = index_entry(0, 0); + index.extend_from_slice(&index_entry(1, batch0.len() as u64)); + index.extend_from_slice(&index_entry(2, log.len() as u64)); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + + let recovered = recover(&config) + .await + .expect("an index running ahead of a whole log must recover, not refuse"); + + assert_eq!(recovered.len(), 1); + assert_eq!(recovered[0].segment.end_offset, 1); + assert_eq!( + bytes_of(&index_path), + index_entry(0, 0), + "an index the log contradicts must be rebuilt from the walked batches" + ); + assert_eq!( + bytes_of(&messages_path), + log, + "rebuilding the index must not touch a whole log" + ); + } + + #[compio::test] + async fn given_index_anchor_on_torn_batch_when_recovering_should_rebuild_index_and_truncate_log_tail() + { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + // The last entry points at a batch whose header made it to disk but + // whose body did not: a torn tail on both files at once. + let batch0 = encoded_batch(0, 1); + let batch1 = encoded_batch(1, 1); + let batch2 = encoded_batch(2, 1); + let mut log = batch0.clone(); + log.extend_from_slice(&batch1); + let torn_position = log.len() as u64; + log.extend_from_slice(&batch2[..COMMAND_HEADER_SIZE + 8]); + let mut index = index_entry(0, 0); + index.extend_from_slice(&index_entry(1, batch0.len() as u64)); + index.extend_from_slice(&index_entry(2, torn_position)); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + + let recovered = recover(&config) + .await + .expect("a torn batch under the last index entry must truncate, not refuse"); + + assert_eq!(recovered.len(), 1); + assert_eq!(recovered[0].segment.end_offset, 1); + assert_eq!( + len_of(&messages_path), + torn_position, + "the torn batch must be gone from disk" + ); + assert_eq!( + bytes_of(&index_path), + index_entry(0, 0), + "the index must describe the batches the walk proved, not the torn one" + ); + drop(recovered); + + // A repaired pair must be a fixpoint: a boot loop that re-refused + // what it just repaired is the failure this replaces. + let reopened = recover(&config) + .await + .expect("re-recovering a repaired segment must not refuse"); + + assert_eq!(reopened[0].segment.end_offset, 1); + assert_eq!( + (len_of(&messages_path), len_of(&index_path)), + (torn_position, SPARSE_INDEX_ENTRY_SIZE as u64), + "a second recovery must not move the files" + ); + } + + #[compio::test] + async fn given_index_anchor_past_the_log_end_with_survivor_past_damage_when_recovering_should_refuse_interior_damage() + { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + // Dropping the index must not weaken the survivor rule: the byte-0 + // walk still ends at the garbage, and the batch verifying past it is + // durable data truncation would erase. + let batch0 = encoded_batch(0, 1); + let batch1 = encoded_batch(1, 1); + let mut log = batch0.clone(); + log.extend_from_slice(&batch1); + log.extend_from_slice(&GARBAGE); + log.extend_from_slice(&encoded_batch(6, 1)); + let mut index = index_entry(0, 0); + index.extend_from_slice(&index_entry(1, batch0.len() as u64)); + index.extend_from_slice(&index_entry(2, log.len() as u64)); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + let error = recover(&config) .await .err() - .expect("an anchor batch that fails its checksum must refuse recovery"); + .expect("a survivor past the damage must refuse recovery even after stepping back"); assert!( matches!( &error, ServerError::PartitionRecoveryRefused { - reason: PartitionRecoveryRefusal::IndexLogDivergence { .. }, + reason: PartitionRecoveryRefusal::InteriorDamage { .. }, .. } ), - "expected an index-log divergence refusal, got {error:?}" + "expected an interior-damage refusal, got {error:?}" + ); + assert_eq!(bytes_of(&messages_path), log); + assert_eq!(bytes_of(&index_path), index); + } + + /// The `.log` and `.index` fixtures for a segment whose SECOND chunk lost + /// its body while its header page survived, and whose last index entry + /// points past the log end. Anchoring on the highest entry the log still + /// proves lands on the third chunk, ABOVE the damage, and the damage probe + /// only ever looks forward from where a walk stopped -- so an anchored + /// walk runs clean to EOF and mints an `end_offset` covering an offset + /// whose bytes are gone. Returned with the damaged and surviving positions + /// so the tests can name them. + fn segment_damaged_below_its_last_provable_entry() -> (Vec, Vec, u64, u64) { + let mut torn = encoded_batch(1, 1); + torn[COMMAND_HEADER_SIZE..].fill(0); + let batch2 = encoded_batch(2, 1); + let mut log = encoded_batch(0, 1); + let damage_position = log.len() as u64; + log.extend_from_slice(&torn); + let survivor_position = log.len() as u64; + log.extend_from_slice(&batch2); + let mut index = index_entry(0, 0); + index.extend_from_slice(&index_entry(1, damage_position)); + index.extend_from_slice(&index_entry(2, survivor_position)); + index.extend_from_slice(&index_entry(3, log.len() as u64)); + (log, index, damage_position, survivor_position) + } + + #[compio::test] + async fn given_damage_below_the_highest_provable_index_entry_when_recovering_should_refuse() { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + let (log, index, damage_position, survivor_position) = + segment_damaged_below_its_last_provable_entry(); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + + let error = recover(&config) + .await + .err() + .expect("damage below the highest provable entry must refuse, not be walked past"); + + assert!( + matches!( + &error, + ServerError::PartitionRecoveryRefused { + reason: PartitionRecoveryRefusal::InteriorDamage { + damage_position: at, + survivor_position: past, + .. + }, + .. + } if *at == damage_position && *past == survivor_position + ), + "expected an interior-damage refusal naming the damaged chunk, got {error:?}" + ); + assert_eq!(bytes_of(&messages_path), log); + assert_eq!(bytes_of(&index_path), index); + } + + #[compio::test] + async fn given_damage_below_a_valid_last_index_entry_without_fsync_when_recovering_should_refuse() + { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + let (log, mut index, damage_position, survivor_position) = + segment_damaged_below_its_last_provable_entry(); + index.truncate(index.len() - SPARSE_INDEX_ENTRY_SIZE); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + + let error = recover(&config) + .await + .err() + .expect("a valid last entry must not hide earlier damage without fsync"); + + assert!( + matches!( + &error, + ServerError::PartitionRecoveryRefused { + reason: PartitionRecoveryRefusal::InteriorDamage { + damage_position: at, + survivor_position: past, + .. + }, + .. + } if *at == damage_position && *past == survivor_position + ), + "expected an interior-damage refusal below the valid last entry, got {error:?}" + ); + assert_eq!(bytes_of(&messages_path), log); + assert_eq!(bytes_of(&index_path), index); + } + + #[compio::test] + async fn given_damage_below_the_highest_provable_index_entry_under_fsync_when_recovering_should_refuse() + { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + // The gap between the index and the log is exactly one entry here, so + // the `enforce_fsync` guard reads this as the benign in-flight chunk + // and lets it through. That verdict is about the INDEX; the log is + // still walked from byte 0, which is what catches the damage. + let (log, index, damage_position, _) = segment_damaged_below_its_last_provable_entry(); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + + let error = recover_under_fsync(&config, true) + .await + .err() + .expect("a passing step-back depth must not stand in for reading the log"); + + assert!( + matches!( + &error, + ServerError::PartitionRecoveryRefused { + reason: PartitionRecoveryRefusal::InteriorDamage { + damage_position: at, + .. + }, + .. + } if *at == damage_position + ), + "expected an interior-damage refusal, got {error:?}" ); assert_eq!(bytes_of(&messages_path), log); assert_eq!(bytes_of(&index_path), index); } + #[compio::test] + async fn given_an_offset_gap_below_the_highest_provable_index_entry_when_recovering_should_refuse() + { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + // Nothing here is damaged: both batches verify. What is missing is + // offsets 1..4, and only a walk that starts at byte 0 can notice. + // An anchor on the entry for offset 5 takes that entry's own offset + // as the chain expectation, so the gap below it reads as continuous. + let mut log = encoded_batch(0, 1); + let gap_position = log.len() as u64; + log.extend_from_slice(&encoded_batch(5, 1)); + let mut index = index_entry(0, 0); + index.extend_from_slice(&index_entry(5, gap_position)); + index.extend_from_slice(&index_entry(6, log.len() as u64)); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + + let error = recover(&config) + .await + .err() + .expect("an offset gap below the highest provable entry must refuse"); + + assert!( + matches!( + &error, + ServerError::PartitionRecoveryRefused { + reason: PartitionRecoveryRefusal::OffsetDiscontinuity { + expected_offset: 1, + found_offset: 5, + position, + .. + }, + .. + } if *position == gap_position + ), + "expected an offset-discontinuity refusal at the gap, got {error:?}" + ); + assert_eq!(bytes_of(&messages_path), log); + assert_eq!(bytes_of(&index_path), index); + } + + #[compio::test] + async fn given_no_provable_index_entry_when_recovering_should_rebuild_the_index_from_the_log() { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + // Every entry points past the log end, so nothing in the index backs + // the walk -- and the first entry is not even at the segment's first + // byte, which the consistency scan catches before the walk runs. The + // log is whole and describes itself either way, so the index is + // dropped and rebuilt from it. + let log = encoded_batch(0, 2); + let mut index = index_entry(0, log.len() as u64); + index.extend_from_slice(&index_entry(2, log.len() as u64 + 64)); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + + let recovered = recover(&config) + .await + .expect("an index the log backs nowhere must be rebuilt, not refused"); + + assert_eq!(recovered.len(), 1); + assert_eq!(recovered[0].segment.end_offset, 1); + assert_eq!( + bytes_of(&index_path), + index_entry(0, 0), + "the index must be rebuilt from the walked batches" + ); + assert_eq!( + bytes_of(&messages_path), + log, + "rebuilding the index must not touch a whole log" + ); + } + + #[compio::test] + async fn given_an_index_entry_over_an_empty_log_when_recovering_should_fence_and_recover_empty() + { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + let partition_path = prepare_partition_dir(&config); + // The crash shape concurrent persists make ordinary: the first + // persist into a fresh segment fsyncs its one index entry and its log + // chunk at the same time, and the entry lands while the log does not. + // Nothing was acked for those bytes, so the segment is simply empty. + let index = index_entry(0, 0); + let (messages_path, index_path) = write_segment(&config, 0, &[], &index); + + let recovered = recover(&config) + .await + .expect("an index entry over an empty log must not refuse the partition"); + + assert_eq!(recovered.len(), 1); + assert_eq!(recovered[0].segment.size, IggyByteSize::default()); + assert_eq!(len_of(&messages_path), 0); + assert_eq!(len_of(&index_path), 0, "the unbacked entry must be gone"); + assert_eq!( + fenced_bytes(&partition_path, &index_path), + index, + "the unbacked index bytes must survive in the fence directory" + ); + drop(recovered); + + // Fixpoint: the seeded empty pair holds no bytes to fence, so a + // second boot must not open another fence directory. + recover(&config).await.expect("second recovery"); + + assert_eq!((len_of(&messages_path), len_of(&index_path)), (0, 0)); + assert!( + !Path::new(&format!("{partition_path}.fenced.1")).exists(), + "a repaired empty segment must not be fenced again" + ); + } + + #[compio::test] + async fn given_an_index_entry_over_a_torn_batch_when_recovering_should_fence_and_recover_empty() + { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + let partition_path = prepare_partition_dir(&config); + // The same crash one flush later: the entry is durable and the log + // holds only the head of the batch it points at. + let batch = encoded_batch(0, 1); + let log = batch[..COMMAND_HEADER_SIZE + 8].to_vec(); + let index = index_entry(0, 0); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + + let recovered = recover(&config) + .await + .expect("an index entry over a torn batch must not refuse the partition"); + + assert_eq!(recovered.len(), 1); + assert_eq!(recovered[0].segment.size, IggyByteSize::default()); + assert_eq!(len_of(&messages_path), 0, "the torn head must be gone"); + assert_eq!(len_of(&index_path), 0, "the unbacked entry must be gone"); + assert_eq!( + fenced_bytes(&partition_path, &messages_path), + log, + "the torn bytes must survive in the fence directory" + ); + assert_eq!(fenced_bytes(&partition_path, &index_path), index); + } + + #[compio::test] + async fn given_an_index_entry_contradicting_a_healthy_log_when_recovering_should_rebuild_the_index() + { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + // The entry claims offset 7 where the log holds offset 0. The log + // verifies and the entry is derived data, so the entry is what is + // wrong: rebuild the index rather than reading the mismatch as a + // break in the chain. + let batch0 = encoded_batch(0, 1); + let mut log = batch0.clone(); + log.extend_from_slice(&encoded_batch(1, 1)); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index_entry(7, 0)); + + let recovered = recover(&config) + .await + .expect("an index entry contradicting a healthy log must be rebuilt, not refused"); + + assert_eq!(recovered.len(), 1); + assert_eq!(recovered[0].segment.end_offset, 1); + assert_eq!( + bytes_of(&index_path), + index_entry(0, 0), + "the rebuilt entry must describe the batch the log actually holds" + ); + assert_eq!(bytes_of(&messages_path), log); + } + + #[compio::test] + async fn given_two_unprovable_index_entries_when_recovering_should_rebuild_the_index_from_the_log() + { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + // Two unprovable entries for two different reasons: one lands + // mid-batch inside the log, the next past its end. Neither the depth + // of the gap nor the reason for it changes the verdict without + // `enforce_fsync`: the whole index goes and the log speaks for itself. + let batch0 = encoded_batch(0, 1); + let batch1 = encoded_batch(1, 1); + let mut log = batch0.clone(); + log.extend_from_slice(&batch1); + let mut index = index_entry(0, 0); + index.extend_from_slice(&index_entry(1, batch0.len() as u64)); + index.extend_from_slice(&index_entry(2, batch0.len() as u64 + 8)); + index.extend_from_slice(&index_entry(3, log.len() as u64)); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + + let recovered = recover(&config) + .await + .expect("a multi-entry index overshoot must rebuild from what the log proves"); + + assert_eq!(recovered.len(), 1); + assert_eq!(recovered[0].segment.end_offset, 1); + assert_eq!( + bytes_of(&index_path), + index_entry(0, 0), + "every entry must come from the walk, not just the unbacked ones be dropped" + ); + assert_eq!(bytes_of(&messages_path), log); + } + + #[compio::test] + async fn given_two_unprovable_index_entries_under_fsync_when_recovering_should_refuse() { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + // The same fixture the lenient rebuild above accepts. Under + // `enforce_fsync` the second-from-last entry was durable before its + // chunk was acked, so a log that cannot back it lost acked bytes. + let batch0 = encoded_batch(0, 1); + let batch1 = encoded_batch(1, 1); + let mut log = batch0.clone(); + log.extend_from_slice(&batch1); + let mut index = index_entry(0, 0); + index.extend_from_slice(&index_entry(1, batch0.len() as u64)); + index.extend_from_slice(&index_entry(2, batch0.len() as u64 + 8)); + index.extend_from_slice(&index_entry(3, log.len() as u64)); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + + let error = recover_under_fsync(&config, true) + .await + .err() + .expect("a multi-entry overshoot under enforce_fsync must refuse"); + + assert!( + matches!( + &error, + ServerError::PartitionRecoveryRefused { + reason: PartitionRecoveryRefusal::FsyncedLogLoss { + entry_count: 4, + provable_entries: 2, + .. + }, + .. + } + ), + "expected an fsynced-log-loss refusal naming the step-back, got {error:?}" + ); + assert_eq!(bytes_of(&messages_path), log); + assert_eq!(bytes_of(&index_path), index); + } + + #[compio::test] + async fn given_one_unprovable_index_entry_under_fsync_when_recovering_should_rebuild_the_index() + { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + // The crash window `enforce_fsync` cannot close: the entry for the + // chunk still in flight reached disk, its log bytes did not, and no + // ack was ever sent for them. Exactly one entry deep, so it recovers. + let batch0 = encoded_batch(0, 1); + let batch1 = encoded_batch(1, 1); + let mut log = batch0.clone(); + log.extend_from_slice(&batch1); + let mut index = index_entry(0, 0); + index.extend_from_slice(&index_entry(1, batch0.len() as u64)); + index.extend_from_slice(&index_entry(2, log.len() as u64)); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + + let recovered = recover_under_fsync(&config, true) + .await + .expect("a one-entry overshoot is the crash window, not data loss"); + + assert_eq!(recovered.len(), 1); + assert_eq!(recovered[0].segment.end_offset, 1); + assert_eq!( + bytes_of(&index_path), + index_entry(0, 0), + "the index must be rebuilt from the walk, not floored to the entry below" + ); + assert_eq!(bytes_of(&messages_path), log); + } + + #[compio::test] + async fn given_a_mid_chunk_tear_below_the_last_entry_under_fsync_when_recovering_should_refuse() + { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + // Chunk 1 held two batches under one entry; its second batch is torn + // and the in-flight chunk 2's log bytes never landed. Entry 2 exists, + // so the log completed an fdatasync through its position: the tear + // sits inside bytes a completed flush made durable, a loss the + // entry-granular step-back (depth 1 here) cannot see. + let batch0 = encoded_batch(0, 1); + let batch1 = encoded_batch(1, 1); + let mut torn = encoded_batch(2, 1); + torn[COMMAND_HEADER_SIZE..].fill(0); + let mut log = batch0.clone(); + let chunk1_position = log.len() as u64; + log.extend_from_slice(&batch1); + let torn_position = log.len() as u64; + log.extend_from_slice(&torn); + let durable = log.len() as u64; + let mut index = index_entry(0, 0); + index.extend_from_slice(&index_entry(1, chunk1_position)); + index.extend_from_slice(&index_entry(3, durable)); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + + let error = recover_under_fsync(&config, true) + .await + .err() + .expect("a rebuild proving less than the last entry's position must refuse"); + + assert!( + matches!( + &error, + ServerError::PartitionRecoveryRefused { + reason: PartitionRecoveryRefusal::FsyncedRebuildShortfall { + walked_position, + durable_position, + .. + }, + .. + } if *walked_position == torn_position && *durable_position == durable + ), + "expected an fsynced-rebuild-shortfall refusal, got {error:?}" + ); + assert_eq!(bytes_of(&messages_path), log); + assert_eq!(bytes_of(&index_path), index); + } + + #[compio::test] + async fn given_a_verified_offset_gap_past_an_empty_anchor_batch_when_recovering_should_refuse() + { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + // The empty batch verifies and carries the anchor's own offset, so + // the expectation it confirms is the log's, not the index's. The + // verified batch after it that does not continue the chain is durable + // data past a hole; treating the mismatch as a stale-anchor break + // would truncate it and re-mint the offsets in between. + let empty = encoded_batch(0, 0); + let mut log = empty.clone(); + let survivor_position = log.len() as u64; + log.extend_from_slice(&encoded_batch(5, 1)); + let index = index_entry(0, 0); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + + let error = recover_under_fsync(&config, true) + .await + .err() + .expect("a verified offset gap past an empty anchor batch must refuse"); + + assert!( + matches!( + &error, + ServerError::PartitionRecoveryRefused { + reason: PartitionRecoveryRefusal::OffsetDiscontinuity { + expected_offset: 0, + found_offset: 5, + position, + .. + }, + .. + } if *position == survivor_position + ), + "expected an offset-discontinuity refusal past the empty anchor, got {error:?}" + ); + assert_eq!(bytes_of(&messages_path), log); + assert_eq!(bytes_of(&index_path), index); + } + + #[compio::test] + async fn given_only_an_empty_anchor_batch_under_fsync_when_recovering_should_not_count_a_phantom_message() + { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + let partition_path = prepare_partition_dir(&config); + let log = encoded_batch(0, 0); + let index = index_entry(0, 0); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + + let recovered = recover_under_fsync(&config, true) + .await + .expect("a checksum-valid empty record carries no message range"); + + assert_eq!(recovered.len(), 1); + assert_eq!(recovered[0].segment.size, IggyByteSize::default()); + assert_eq!(len_of(&messages_path), 0, "the served log must be empty"); + assert_eq!(len_of(&index_path), 0, "the served index must be empty"); + assert_eq!(fenced_bytes(&partition_path, &messages_path), log); + assert_eq!(fenced_bytes(&partition_path, &index_path), index); + } + + #[compio::test] + async fn given_an_index_the_log_backs_nowhere_under_fsync_when_recovering_should_refuse() { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + // The gap is measured the same way when no entry proves at all. The + // index still starts at the segment's own first byte, so it is not + // the mis-strided shape the consistency scan drops; the log simply + // lost the tail of its FIRST indexed chunk, which two entries the log + // backs nowhere is one more than the in-flight chunk can explain. + let whole = encoded_batch(0, 2); + let log = whole[..COMMAND_HEADER_SIZE + 8].to_vec(); + let mut index = index_entry(0, 0); + index.extend_from_slice(&index_entry(2, whole.len() as u64)); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + + let error = recover_under_fsync(&config, true) + .await + .err() + .expect("an index backed nowhere under enforce_fsync must refuse, not rebuild"); + + assert!( + matches!( + &error, + ServerError::PartitionRecoveryRefused { + reason: PartitionRecoveryRefusal::FsyncedLogLoss { + entry_count: 2, + provable_entries: 0, + .. + }, + .. + } + ), + "expected an fsynced-log-loss refusal over the dropped index, got {error:?}" + ); + assert_eq!(bytes_of(&messages_path), log); + assert_eq!(bytes_of(&index_path), index); + } + + #[compio::test] + async fn given_an_index_deeper_than_the_probe_cap_under_fsync_when_recovering_should_refuse_with_the_searched_depth() + { + // Entry 0 is the one entry the log does back, and it sits below the + // cap: the search stops before reaching it and proves nothing. The + // refusal is still the right verdict -- a step-back this deep is log + // loss whatever a deeper probe would find -- which is exactly why it + // must name the depth it searched instead of letting zero provable + // entries read as "the log backs nothing". + const OVERSHOOTING_ENTRIES: u64 = MAX_INDEX_ANCHOR_PROBE_ENTRIES + 1; + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + let log = encoded_batch(0, 1); + let mut index = index_entry(0, 0); + for step in 0..OVERSHOOTING_ENTRIES { + index.extend_from_slice(&index_entry(step + 1, log.len() as u64 + step)); + } + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + + let error = recover_under_fsync(&config, true) + .await + .err() + .expect("an index outrunning the log past the probe cap must refuse"); + + assert!( + matches!( + &error, + ServerError::PartitionRecoveryRefused { + reason: PartitionRecoveryRefusal::FsyncedLogLoss { + entry_count, + provable_entries: 0, + searched_entries, + .. + }, + .. + } if *entry_count == OVERSHOOTING_ENTRIES + 1 + && *searched_entries == MAX_INDEX_ANCHOR_PROBE_ENTRIES + ), + "expected a capped fsynced-log-loss refusal naming its search depth, got {error:?}" + ); + assert_eq!(bytes_of(&messages_path), log); + assert_eq!(bytes_of(&index_path), index); + } + + #[compio::test] + async fn given_a_lone_entry_over_an_empty_log_under_fsync_when_recovering_should_not_refuse() { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + // The first flush into a fresh segment, crashed between the two + // fsyncs: one entry and no log bytes. Whether any batches in that + // flush were acknowledged depends on the flush thresholds, but only + // one entry can belong to the interrupted flush. `enforce_fsync` must + // not turn that shape into a tombstone. + let index = index_entry(0, 0); + let (messages_path, index_path) = write_segment(&config, 0, &[], &index); + + let recovered = recover_under_fsync(&config, true) + .await + .expect("one entry over an empty log is the crash window, not data loss"); + + assert_eq!(recovered.len(), 1); + assert_eq!(recovered[0].segment.size, IggyByteSize::default()); + assert_eq!((len_of(&messages_path), len_of(&index_path)), (0, 0)); + } + + #[compio::test] + async fn given_index_anchor_on_batch_with_torn_body_when_recovering_should_truncate_it_and_rebuild_index() + { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + // The last entry points at a batch whose header page reached disk + // whole but whose body did not: the header decodes, fits the file, + // and continues the chain from the entry, so only the checksum can + // tell it from a durable batch. The entry is no evidence for the + // bytes under it, because the log and the index persist + // concurrently. + let batch0 = encoded_batch(0, 1); + let batch1 = encoded_batch(1, 1); + let mut torn = encoded_batch(2, 1); + torn[COMMAND_HEADER_SIZE..].fill(0); + let mut log = batch0.clone(); + log.extend_from_slice(&batch1); + let torn_position = log.len() as u64; + log.extend_from_slice(&torn); + let mut index = index_entry(0, 0); + index.extend_from_slice(&index_entry(1, batch0.len() as u64)); + index.extend_from_slice(&index_entry(2, torn_position)); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + + let recovered = recover(&config) + .await + .expect("a torn body under an intact header must truncate, not refuse"); + + assert_eq!(recovered.len(), 1); + assert_eq!( + recovered[0].segment.end_offset, 1, + "the torn batch's offset must not be minted from its header" + ); + assert_eq!( + len_of(&messages_path), + torn_position, + "the torn batch must be gone from disk" + ); + assert_eq!( + bytes_of(&index_path), + index_entry(0, 0), + "the entry pointing at the torn batch must be gone from disk" + ); + drop(recovered); + + let reopened = recover(&config) + .await + .expect("re-recovering a repaired segment must not refuse"); + + assert_eq!(reopened[0].segment.end_offset, 1); + assert_eq!( + (len_of(&messages_path), len_of(&index_path)), + (torn_position, SPARSE_INDEX_ENTRY_SIZE as u64), + "a second recovery must not move the files" + ); + } + + #[compio::test] + async fn given_index_entries_packed_closer_than_their_batches_under_fsync_when_measuring_the_gap_should_refuse_on_verify_budget() + { + // A mis-strided index over a log of back-to-back headers, each + // claiming a batch 8 KiB wide: every entry lands on a header that + // decodes, matches the entry, and fits the file, so each step back + // costs a whole-batch verify that never passes. The claims overlap + // many times over, so an unbudgeted search would hash close to + // entries x claim bytes; it must give up and refuse instead, leaving + // the files byte-identical. Only `enforce_fsync` walks the index + // backward at all -- without it the gap is not measured, so there is + // nothing here to bound. + const CLAIMED_BATCH_BYTES: usize = 8 * 1024; + const ENTRIES: u64 = 64; + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + let mut log = Vec::new(); + let mut index = Vec::new(); + for offset in 0..ENTRIES { + let mut header = encoded_batch(offset, 1)[..COMMAND_HEADER_SIZE].to_vec(); + header[HEADER_BATCH_LENGTH_OFFSET..HEADER_BATCH_LENGTH_OFFSET + 8] + .copy_from_slice(&(CLAIMED_BATCH_BYTES as u64).to_le_bytes()); + index.extend_from_slice(&index_entry(offset, log.len() as u64)); + log.extend_from_slice(&header); + } + // Padding so the last claim still fits inside the file. + log.resize(log.len() + CLAIMED_BATCH_BYTES, 0); + let (messages_path, index_path) = write_segment(&config, 0, &log, &index); + + let error = recover_under_fsync(&config, true) + .await + .err() + .expect("exhausting the verify budget while measuring the gap must refuse"); + + assert!( + matches!( + &error, + ServerError::PartitionRecoveryRefused { + reason: PartitionRecoveryRefusal::UnverifiedResidue { + verified_bytes, + verify_budget_bytes, + .. + }, + .. + } if verified_bytes > verify_budget_bytes + ), + "expected a verify-budget refusal, got {error:?}" + ); + assert_eq!( + bytes_of(&messages_path), + log, + "a refusal must leave the log byte-identical" + ); + assert_eq!( + bytes_of(&index_path), + index, + "a refusal must leave the index byte-identical" + ); + } + #[compio::test] async fn given_foreign_partition_batch_when_recovering_indexed_should_refuse() { let tmp = tempdir().expect("tempdir"); diff --git a/core/server/src/server_error.rs b/core/server/src/server_error.rs index 6ada1d006f..d5343d137b 100644 --- a/core/server/src/server_error.rs +++ b/core/server/src/server_error.rs @@ -77,6 +77,21 @@ pub enum ServerError { committed journal tail may not have flushed" )] ShardPumpDied { shard_id: u16, reason: String }, + /// A shard's message pump stopped because a partition could not commit + /// an op the cluster had already committed. The partition is fenced and + /// the server is shutting down; the exit is non-zero so an orchestrator + /// does not read a durability fault as a clean stop. + #[error( + "shard {shard_id} stopped: partition {namespace_raw} could not commit op {op}, \ + which the cluster had already committed. The replica is divergent and was \ + fenced; the server shut down so it cannot serve a prefix the cluster has \ + moved past" + )] + ShardFatal { + shard_id: u16, + namespace_raw: u64, + op: u64, + }, #[error( "shard {shard_id} message pump did not drain within {timeout:?}. \ Committed journal tail may not have flushed" @@ -284,9 +299,19 @@ pub enum ServerError { /// causes, and not all of them are at-rest corruption: an empty non-tail /// segment is a failed rebuild's orphan pairing, a hole is a stray or /// half-unlinked file, interior damage is bit rot (or a resurrected tail -/// appended over), a divergent index is a mis-strided or foreign write, and -/// offsets that do not continue the chain can be minted into byte-clean files -/// by an upstream crash window as well as by damage. +/// appended over), and offsets that do not continue the chain can be minted +/// into byte-clean files by an upstream crash window as well as by damage. +/// +/// An index that contradicts itself is deliberately NOT here, and neither is +/// one the log cannot back UNLESS the topic runs under `enforce_fsync` and the +/// gap is deeper than the single in-flight entry: entries are derived from the +/// log, so recovery drops such an index whole and rebuilds it from a byte-0 +/// walk of the log rather than believing any part of it. What `enforce_fsync` +/// adds is evidence from serialized completed flushes: an entry above chunk N +/// means the log fdatasync covering chunk N completed before the later flush +/// began. This is independent of reply timing and turns a deeper gap into +/// evidence about the LOG. Absent that evidence the index only locates data; +/// recovery verifies the log from byte 0. #[derive(Debug)] pub enum PartitionRecoveryRefusal { /// `recoverable_bytes` on the two chain-shape refusals is the sum of @@ -305,16 +330,6 @@ pub enum PartitionRecoveryRefusal { next_start: u64, recoverable_bytes: u64, }, - /// The index holds entries but no whole batch decodes AND verifies where - /// its last entry points, so index and log describe different files. The - /// damage probe ran first: anything verifying past the anchor's damage - /// refuses as [`Self::InteriorDamage`] instead. - IndexLogDivergence { - start_offset: u64, - end_offset: u64, - messages_size_bytes: u64, - indexed_size_bytes: u64, - }, /// A complete, checksum-verifying batch survives PAST bytes that do not /// decode. A torn tail has nothing after it, so this is interior damage, /// and truncating at it would silently discard the surviving batches. @@ -330,9 +345,12 @@ pub enum PartitionRecoveryRefusal { /// were re-examined -- a probe defect); the verification budget bounds /// the bytes handed to checksum verifies, whose claimed slices overlap, /// so residue packed with plausible headers can exhaust it from an - /// on-disk shape. Truncation is only ever sound for a proven torn tail, - /// so giving up keeps the bytes. The residue width is diagnostic only; - /// it is not a gate. + /// on-disk shape. The index anchor search charges the same verification + /// budget as it steps back through entries the log cannot back, so an + /// index packed with claims the log never proves ends here too instead + /// of paying a verify per entry. Truncation is only ever sound for a + /// proven torn tail, so giving up keeps the bytes. The residue width is + /// diagnostic only; it is not a gate. UnverifiedResidue { start_offset: u64, damage_position: u64, @@ -365,13 +383,43 @@ pub enum PartitionRecoveryRefusal { batch_partition_id: u64, position: u64, }, - /// Index entries must ascend in offset and position (they are appended, - /// one per flushed chunk, over a growing log); a regression means the - /// file was written mis-strided or over foreign bytes. - IndexEntriesNotMonotone { start_offset: u64, entry_index: u64 }, - IndexEntryBeforeSegmentStart { + /// The sparse index of a topic running under `enforce_fsync` outruns its + /// log by more than the one entry a crash can legitimately strand there. + /// Persistence writes exactly one entry per flush chunk, chunks never + /// overlap, and flushes are serialized, so every entry below the last one + /// names a chunk whose log bytes completed their fdatasync. A completed + /// chunk can contain batches acknowledged before the flush threshold was + /// reached, while the in-flight chunk can do so too. Reply timing is not + /// the proof. Only the chunk in flight when the process died can have an + /// entry the log never backed. A deeper step-back therefore says the LOG + /// lost bytes it had already made durable, and rebuilding from what remains + /// could re-mint offsets, including offsets already returned to clients. + FsyncedLogLoss { + start_offset: u64, + entry_count: u64, + provable_entries: u64, + /// Position of the highest entry the log still proves; 0 when it + /// proves none, which `provable_entries` disambiguates. + provable_position: u64, + /// Entries the backward search actually probed, which its own cap + /// holds below `entry_count` on a long index: `provable_entries == 0` + /// then means nothing proved in the searched window, not that the log + /// backs nothing. + searched_entries: u64, + }, + /// Under `enforce_fsync`, the byte-0 rebuild after a dropped index proved + /// the log only through `walked_position`, short of `durable_position`, + /// the byte the index's own last entry proves the log had already + /// fdatasynced through (the flush that wrote the entry began only after + /// the previous chunk's log sync completed). The step-back gate measures + /// loss at entry granularity; this catches the sub-chunk shape it cannot: + /// bytes a completed flush made durable are gone mid-chunk, so truncating + /// to the walked prefix would re-mint their offsets. + FsyncedRebuildShortfall { start_offset: u64, - first_entry_offset: u64, + entry_count: u64, + walked_position: u64, + durable_position: u64, }, /// A writer reopening over recovered bounds found the on-disk length /// diverging from the size recovery just validated and truncated to. @@ -408,18 +456,6 @@ impl std::fmt::Display for PartitionRecoveryRefusal { starts at {next_start}, leaving a hole in a chain holding \ {recoverable_bytes} recoverable bytes" ), - Self::IndexLogDivergence { - start_offset, - end_offset, - messages_size_bytes, - indexed_size_bytes, - } => write!( - f, - "segment {start_offset} has message/index divergence: the index ends \ - at offset {end_offset}, byte {indexed_size_bytes}, where the \ - {messages_size_bytes}-byte log holds no batch that decodes and \ - verifies" - ), Self::InteriorDamage { start_offset, damage_position, @@ -469,21 +505,34 @@ impl std::fmt::Display for PartitionRecoveryRefusal { stamped for partition {batch_partition_id}; a foreign record in \ this log is preserved as evidence, not truncated" ), - Self::IndexEntriesNotMonotone { + Self::FsyncedLogLoss { start_offset, - entry_index, + entry_count, + provable_entries, + provable_position, + searched_entries, } => write!( f, - "segment {start_offset} index entry {entry_index} regresses in \ - offset or position; the index was not appended over this log" + "segment {start_offset} runs under enforce_fsync with {entry_count} sparse \ + index entries, but its log backs only {provable_entries} of the \ + {searched_entries} searched from the top (up to byte {provable_position}); \ + every entry below the last describes a log chunk whose fdatasync had \ + completed, so the log has lost previously durable data rather than the \ + index having outrun it" ), - Self::IndexEntryBeforeSegmentStart { + Self::FsyncedRebuildShortfall { start_offset, - first_entry_offset, + entry_count, + walked_position, + durable_position, } => write!( f, - "segment {start_offset} index claims offset {first_entry_offset}, \ - below the segment's own start" + "segment {start_offset} runs under enforce_fsync with {entry_count} sparse \ + index entries, and the byte-0 rebuild proved its log only through byte \ + {walked_position}, short of byte {durable_position} which the last \ + entry's own fdatasync ordering proves the log had already made durable; \ + the log has lost previously durable bytes mid-chunk, so rebuilding \ + would re-mint their offsets" ), Self::StorageSizeMismatch { start_offset, diff --git a/core/shard/src/lib.rs b/core/shard/src/lib.rs index eadf6880df..2decb39341 100644 --- a/core/shard/src/lib.rs +++ b/core/shard/src/lib.rs @@ -63,7 +63,9 @@ use metadata::impls::metadata::StreamsFrontend; use metadata::stm::StateMachine; use metadata::{BoundSession, MetadataSubmitError}; use partitions::state_transfer::TransferArtifact; -use partitions::{IggyPartition, IggyPartitions, PollFragments, PollingArgs, PollingConsumer}; +use partitions::{ + FatalCommit, IggyPartition, IggyPartitions, PollFragments, PollingArgs, PollingConsumer, +}; use server_common::sharding::{IggyNamespace, PartitionLocation, ShardId}; use server_common::{MESSAGE_ALIGN, Message, MessageBag, iobuf::Frozen}; use shards_table::ShardsTable; @@ -6428,7 +6430,15 @@ where /// Tick partition consensuses. Loop partitions. No partitions-plane journal. #[allow(clippy::future_not_send)] #[allow(clippy::too_many_lines)] - pub async fn tick_partitions(&self, namespace_scratch: &mut Vec) + /// Returns the commit fault that fenced a partition on this shard, if one + /// has. The pump turns it into a server shutdown: a fenced partition is + /// divergent from the cluster and can never advance again, so the tick + /// stops driving it and the node stops rather than serving a prefix the + /// cluster has moved past. + pub async fn tick_partitions( + &self, + namespace_scratch: &mut Vec, + ) -> Option where B: MessageBus, MJ: JournalHandle, @@ -6498,10 +6508,19 @@ where // already accepts. let mut transfers_inflight: Option = None; + let mut fatal: Option = None; for namespace in namespace_scratch.drain(..) { let Some(partition) = partitions.get_by_ns(&namespace) else { continue; }; + // A fenced partition must not tick: its consensus would emit + // view-scoped sends for a log the cluster has already passed. + if let Some(fault) = partition.fatal() { + if fatal.is_none() { + fatal = Some(fault.clone()); + } + continue; + } let consensus = partition.consensus(); // Only while a view change is live. A `Normal` tick has no consumer: @@ -6705,6 +6724,8 @@ where } } } + + fatal } /// Flush every owned partition's committed journal prefix to segment @@ -6731,11 +6752,16 @@ where .flush_committed_messages(partitions.config()) .await { - tracing::warn!( + tracing::error!( namespace_raw = namespace.inner(), %error, "failed to flush partition journal on shutdown" ); + // The bytes left behind are cluster-committed, so the pump + // must not let this exit report clean (it re-scans for faults + // after this flush). A partition already fenced by the commit + // path keeps its original fault. + partition.fence_flush_failure(); } } } diff --git a/core/shard/src/router.rs b/core/shard/src/router.rs index 43f02c6274..c98e585303 100644 --- a/core/shard/src/router.rs +++ b/core/shard/src/router.rs @@ -27,8 +27,11 @@ use iggy_binary_protocol::{Command, ConsensusError, GenericHeader, Operation, Pr use journal::superblock::SuperblockStore; use journal::{Journal, JournalHandle}; use message_bus::{ConnectionInstaller, MessageBus, ReplicaHandshakeDoneFn}; +use partitions::FatalCommit; use server_common::sharding::{IggyNamespace, METADATA_GROUP}; use server_common::{Message, MessageBag}; +use std::sync::Arc; +use std::sync::atomic::{AtomicBool, Ordering}; /// How often the shard pump drives `VsrConsensus::tick`. /// @@ -254,8 +257,25 @@ where /// Drain this shard's inbox and process each frame locally until the /// `stop` signal fires or the inbox disconnects, then drain any frames /// still queued so in-flight requests still get a response. + /// + /// Returns the commit fault that ended the pump, if one did. A fault + /// skips that queued drain: the frames in it are requests for a shard + /// holding a divergent partition, and answering them means re-entering + /// the commit path that just failed. The final flush still runs, so + /// every partition that CAN still reach disk does. + /// + /// `shutdown_flag` is the cross-thread server shutdown signal. The pump + /// flips it as soon as a commit fault is resolved, BEFORE the final + /// flush: the flush writes to the device whose failure raised the fault, + /// so it can stall indefinitely, and only the flag arms the watchdog and + /// the bounded pump drain that turn a stalled flush into a timed-out + /// non-zero exit instead of a process that reports healthy forever. #[allow(clippy::future_not_send)] - pub async fn run_message_pump(&self, stop: Receiver<()>) + pub async fn run_message_pump( + &self, + stop: Receiver<()>, + shutdown_flag: Arc, + ) -> Option where B: MessageBus + 'static, MJ: JournalHandle, @@ -281,6 +301,7 @@ where // simulator (see `MessageBus::sleep`). let rearm_tick = || self.bus.sleep(CONSENSUS_TICK_INTERVAL).fuse(); let mut consensus_tick = std::pin::pin!(rearm_tick()); + let mut fatal: Option = None; loop { // `select_biased!`, not `select!`: the unbiased macro draws its // arm order from a process-random thread-local PRNG, which the @@ -299,7 +320,10 @@ where // decoupled from the pump again without reintroducing the // partition-ref-across-`.await` UB this fold closed. self.tick_metadata().await; - self.tick_partitions(&mut namespace_scratch).await; + if let Some(fault) = self.tick_partitions(&mut namespace_scratch).await { + fatal = Some(fault); + break; + } // Runs here, not inside `tick_metadata`: that early-returns // on shards without metadata consensus, and partition-plane // offers live on every shard that hosts a serving group -- @@ -366,27 +390,87 @@ where } } + // A stop can win the select immediately after a frame fenced a + // partition, before the next tick observes it. Preserve that fault so + // shutdown cannot turn a durability failure into a clean pump exit. + if fatal.is_none() { + fatal = self.first_partition_commit_fault(); + } + // Drain remaining frames so in-flight requests get a response, and // the reply lane so already-forwarded replies still reach their - // clients before the bus tears down. - while let Ok(frame) = self.inbox.try_recv() { - if self.accept_frame_for_self(&frame) { - self.process_frame(frame).await; - self.process_loopback(&mut loopback_buf, &mut namespace_scratch) - .await; - self.apply_reconcile_ops(); + // clients before the bus tears down. Skipped on a commit fault: a + // queued Ack or Commit frame for the fenced partition would re-enter + // the commit path that just failed, and `advance_commit_min` asserts + // on the gap the fault left. Those requests go unanswered and their + // clients time out, which is what a node stopping on a durability + // fault owes them. + if fatal.is_none() { + while let Ok(frame) = self.inbox.try_recv() { + if self.accept_frame_for_self(&frame) { + self.process_frame(frame).await; + self.process_loopback(&mut loopback_buf, &mut namespace_scratch) + .await; + self.apply_reconcile_ops(); + if let Some(fault) = self.first_partition_commit_fault() { + fatal = Some(fault); + break; + } + } } } - while let Ok(frame) = self.reply_inbox.try_recv() { - if self.accept_frame_for_self(&frame) { - self.process_frame(frame).await; + if fatal.is_none() { + while let Ok(frame) = self.reply_inbox.try_recv() { + if self.accept_frame_for_self(&frame) { + self.process_frame(frame).await; + if let Some(fault) = self.first_partition_commit_fault() { + fatal = Some(fault); + break; + } + } } } + if fatal.is_some() { + // Flipped BEFORE the final flush, not after the pump returns: the + // flush writes to the device whose failure raised the fault and + // can stall there indefinitely, and the flag is what arms the + // watchdog and the bounded pump drain. Siblings give up at most + // the flush window of extra serving. + shutdown_flag.store(true, Ordering::Relaxed); + } + // Final flush: committed messages still resident in the in-memory // journal must reach segment storage before the process exits, or a - // graceful restart recovers consumer offsets ahead of the data. + // graceful restart recovers consumer offsets ahead of the data. Runs + // on a fault too, the fenced partition included: its resident prefix + // is cluster-committed data, so writing what still reaches disk is + // strictly better than dropping it, and a second failure of an + // already-fenced partition is warned rather than propagated. self.flush_partitions().await; + + // A failed flush fences its partition: the data it could not write is + // cluster-committed and now lives only in this process's memory, so a + // clean exit here would report a durability loss as a good shutdown. + if fatal.is_none() { + fatal = self.first_partition_commit_fault(); + } + + fatal + } + + /// First partition commit fault currently fenced on this shard. + /// + /// The regular path observes faults in `tick_partitions`. This scan covers + /// the pump-exit edges where a stop signal wins before that tick, or a + /// queued frame fails while the pump is draining during shutdown. + fn first_partition_commit_fault(&self) -> Option { + let partitions = self.plane.partitions(); + partitions.namespaces().find_map(|namespace| { + partitions + .get_by_ns(namespace) + .and_then(|partition| partition.fatal().cloned()) + }) } /// Sanity check at pump entry: every Consensus frame routed through diff --git a/core/simulator/src/lib.rs b/core/simulator/src/lib.rs index 19d89dc879..128278899a 100644 --- a/core/simulator/src/lib.rs +++ b/core/simulator/src/lib.rs @@ -56,6 +56,8 @@ use std::cell::RefCell; use std::collections::{HashMap, HashSet}; use std::net::{IpAddr, Ipv4Addr, SocketAddr}; use std::rc::Rc; +use std::sync::Arc; +use std::sync::atomic::AtomicBool; /// Poll budget per [`DetExecutor::run_until_stalled`]. Pumps are event-driven, so /// hitting it means a task is spin-waking: a bug, panicked with the seed. @@ -433,8 +435,13 @@ impl Simulator { // the held stop channel or a crash abort. let (stop_tx, stop_rx) = shard::channel::<()>(1); let pump_shard = Rc::clone(&shard); + // The simulator crashes replicas explicitly, so nothing reads + // the shutdown flag a commit fault would flip. + let pump_shutdown_flag = Arc::new(AtomicBool::new(false)); pump_tasks.push(executor.spawn(async move { - pump_shard.run_message_pump(stop_rx).await; + pump_shard + .run_message_pump(stop_rx, pump_shutdown_flag) + .await; })); stop_txs.push(stop_tx); shards.push(shard); @@ -1117,8 +1124,11 @@ impl Simulator { } let (stop_tx, stop_rx) = shard::channel::<()>(1); let pump_shard = Rc::clone(&shard); + let pump_shutdown_flag = Arc::new(AtomicBool::new(false)); pump_tasks.push(self.executor.spawn(async move { - pump_shard.run_message_pump(stop_rx).await; + pump_shard + .run_message_pump(stop_rx, pump_shutdown_flag) + .await; })); stop_txs.push(stop_tx); shards.push(shard); From eb410a96adbe638800eff2bf3a1e5282e7947a79 Mon Sep 17 00:00:00 2001 From: Jogesh A Dinavahi Date: Sun, 30 Aug 2026 23:27:35 -0700 Subject: [PATCH 012/182] fix(server): apply thread pool limit carve-out to all of macOS (#3994) --- core/partitions/src/messages_writer.rs | 6 +++--- core/server_common/src/executor.rs | 2 +- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/core/partitions/src/messages_writer.rs b/core/partitions/src/messages_writer.rs index 9e61ec1bef..d689127190 100644 --- a/core/partitions/src/messages_writer.rs +++ b/core/partitions/src/messages_writer.rs @@ -178,9 +178,9 @@ fn preallocate_file(file: &File, file_path: &str, len: u64) { // sets `thread_pool_limit(0)` on the shard proactor, so `spawn_blocking` // has no worker to park a task on and compio panics the shard outright with // "the thread pool is needed but no worker thread is running". (That limit - // is skipped on macOS/aarch64, where the pool does exist -- see the FIXME - // there -- so the panic is Linux-and-most-targets, not universal. This arm - // is Linux-only regardless.) + // is skipped on macOS, whose polling driver routes fs through the pool, so + // the panic is Linux-and-most-targets, not universal. This arm is + // Linux-only regardless.) // // The cost is acceptable only because of what this call is: a metadata-only // extent reservation, microseconds on the local filesystems this option diff --git a/core/server_common/src/executor.rs b/core/server_common/src/executor.rs index 8fb8cc8422..8ad8f9920f 100644 --- a/core/server_common/src/executor.rs +++ b/core/server_common/src/executor.rs @@ -84,7 +84,7 @@ pub fn create_shard_executor() -> Result { // io_uring targets keep the zero limit: no blocking pool exists on shard // threads, which `core/partitions` messages_writer relies on to justify // running fallocate inline (`spawn_blocking` would hit that same panic). - #[cfg(not(all(target_os = "macos", target_arch = "aarch64")))] + #[cfg(not(target_os = "macos"))] proactor.thread_pool_limit(0); compio::runtime::RuntimeBuilder::new() From c8ef3b9a43c07b6ddc4a00e606537827a61b6bdb Mon Sep 17 00:00:00 2001 From: Hubert Gruszecki Date: Mon, 31 Aug 2026 08:45:34 +0200 Subject: [PATCH 013/182] fix(go): tear down only the connection a request failed on (#3986) --- foreign/go/client/tcp/tcp_core.go | 84 ++++++++++++++----- foreign/go/client/tcp/tcp_failover_test.go | 35 ++++++++ .../go/client/tcp/tcp_session_management.go | 31 ++++++- 3 files changed, 125 insertions(+), 25 deletions(-) diff --git a/foreign/go/client/tcp/tcp_core.go b/foreign/go/client/tcp/tcp_core.go index 0bd8b3600a..f52a9d8613 100644 --- a/foreign/go/client/tcp/tcp_core.go +++ b/foreign/go/client/tcp/tcp_core.go @@ -100,6 +100,11 @@ type IggyTcpClient struct { // connectAttempt is the attempt a Connect is running, shared with every // caller that arrives while it is in progress. Guarded by c.mtx. connectAttempt *connectAttempt + // connGeneration counts the connections this client installed. A request + // carries the generation it ran on, so the teardown that follows its + // failure closes that connection rather than one another caller + // established meanwhile. Guarded by c.mtx. + connGeneration uint64 // rememberedLogin holds the credentials a manual sign-in succeeded with, // so a reconnect -- on this node or, after a failover, another one -- can // re-establish the session instead of surfacing an unauthenticated error. @@ -514,7 +519,7 @@ func (e *localPreconditionError) Unwrap() error { return e.err } // exchange runs one request to completion, reconnecting and replaying it when // the failure is one a fresh connection recovers from. func (c *IggyTcpClient) exchange(ctx context.Context, code uint32, frame []byte) ([]byte, error) { - response, err := c.sendFrame(ctx, code, frame) + response, generation, err := c.sendFrame(ctx, code, frame) if err == nil || !isReconnectable(err) { return response, err } @@ -558,7 +563,7 @@ func (c *IggyTcpClient) exchange(ctx context.Context, code uint32, frame []byte) return nil, err } - if disconnectErr := c.disconnect(); disconnectErr != nil { + if _, disconnectErr := c.disconnectGeneration(generation); disconnectErr != nil { return nil, disconnectErr } reconnectCtx := ctx @@ -577,7 +582,8 @@ func (c *IggyTcpClient) exchange(ctx context.Context, code uint32, frame []byte) if reconnectErr := c.Connect(reconnectCtx); reconnectErr != nil { return nil, reconnectErr } - return c.sendFrame(ctx, code, frame) + response, _, err = c.sendFrame(ctx, code, frame) + return response, err } // canReplay reports whether re-issuing the request over a fresh connection @@ -631,12 +637,17 @@ func isReconnectable(err error) bool { // sendFrame runs the request against the current connection. One deadline // bounds it across every same-connection replay and every leader failover. -func (c *IggyTcpClient) sendFrame(ctx context.Context, code uint32, frame []byte) ([]byte, error) { +// It reports the connection generation the last attempt ran on, so a caller +// tearing the connection down after a failure can tell whether that +// connection is still the installed one. +func (c *IggyTcpClient) sendFrame( + ctx context.Context, code uint32, frame []byte, +) ([]byte, uint64, error) { if ctx == nil { - return nil, ierror.ErrNilContext + return nil, 0, ierror.ErrNilContext } if err := ctx.Err(); err != nil { - return nil, err + return nil, 0, err } deadline := time.Now().Add(responseReadTimeout) @@ -659,13 +670,13 @@ func (c *IggyTcpClient) sendFrame(ctx context.Context, code uint32, frame []byte } } - response, attemptStamped, err := c.attempt( + response, attemptStamped, generation, err := c.attempt( ctx, code, frame, stamped, transientDeadline, deadline) stamped = attemptStamped switch { case err == nil: - return response, nil + return response, generation, nil case errors.Is(err, ierror.ErrTransientNotAccepted) && !isRegisterCode(code) && time.Now().Before(deadline): // The server never admitted the request, so re-issuing it cannot @@ -675,9 +686,9 @@ func (c *IggyTcpClient) sendFrame(ctx context.Context, code uint32, frame []byte walked := false if !walkingRoster { var redirectErr error - redirect, redirectErr = c.HandleLeaderRedirection(ctx) + redirect, redirectErr = c.redirectToLeader(ctx, generation) if redirectErr != nil { - return nil, redirectErr + return nil, generation, redirectErr } } if !redirect { @@ -691,7 +702,7 @@ func (c *IggyTcpClient) sendFrame(ctx context.Context, code uint32, frame []byte var walkErr error walked, walkErr = c.settleOnNextEndpoint(visitedRosterEndpoints) if walkErr != nil { - return nil, walkErr + return nil, generation, walkErr } if walked { walkingRoster = true @@ -711,41 +722,43 @@ func (c *IggyTcpClient) sendFrame(ctx context.Context, code uint32, frame []byte redirectCtx = suppressAutoLogin(ctx) } if connectErr := c.Connect(redirectCtx); connectErr != nil { - return nil, connectErr + return nil, generation, connectErr } stamped = false } default: - return nil, err + return nil, generation, err } } } // attempt stamps the frame if it is not stamped yet and exchanges it once, -// replaying in place while the server answers transiently. +// replaying in place while the server answers transiently. It reports the +// connection generation it ran on alongside the outcome. func (c *IggyTcpClient) attempt( ctx context.Context, code uint32, frame []byte, stamped bool, transientDeadline, readDeadline time.Time, -) ([]byte, bool, error) { +) ([]byte, bool, uint64, error) { c.mtx.Lock() defer c.mtx.Unlock() + generation := c.connGeneration switch c.transportState { case iggcon.TransportStateShutdown: c.logger.Debug("Cannot send data. Client is shutdown.") - return nil, stamped, ierror.ErrClientShutdown + return nil, stamped, generation, ierror.ErrClientShutdown case iggcon.TransportStateDisconnected: c.logger.Debug("Cannot send data. Client is not connected.") - return nil, stamped, ierror.ErrNotConnected + return nil, stamped, generation, ierror.ErrNotConnected case iggcon.TransportStateConnecting: c.logger.Debug("Cannot send data. Client is still connecting.") - return nil, stamped, ierror.ErrNotConnected + return nil, stamped, generation, ierror.ErrNotConnected } if c.conn == nil { - return nil, stamped, ierror.ErrNotConnected + return nil, stamped, generation, ierror.ErrNotConnected } if !stamped { @@ -754,7 +767,7 @@ func (c *IggyTcpClient) attempt( // A stamp failure is local and pre-write, so it is marked as such: // the connection is healthy and must not be torn down for it. if err := vsr.StampRequestHeader(c.session, code, frame); err != nil { - return nil, false, &localPreconditionError{err} + return nil, false, generation, &localPreconditionError{err} } stamped = true } @@ -789,10 +802,10 @@ func (c *IggyTcpClient) attempt( if err != nil { if ctxErr := ctx.Err(); ctxErr != nil { - return nil, stamped, ctxErr + return nil, stamped, generation, ctxErr } } - return response, stamped, err + return response, stamped, generation, err } // exchangeLocked writes the frame and reads its reply, resending the identical @@ -1134,6 +1147,7 @@ func (c *IggyTcpClient) Connect(ctx context.Context) (err error) { } c.conn = conn c.reader = bufio.NewReaderSize(conn, 64*1024) + c.connGeneration++ c.transportState = iggcon.TransportStateConnected c.connectedAt = time.Now() // The server fence does not survive the old socket, so the new connection @@ -1388,7 +1402,33 @@ func (c *IggyTcpClient) createTLSConfig(address string) (*tls.Config, error) { func (c *IggyTcpClient) disconnect() error { c.mtx.Lock() defer c.mtx.Unlock() + return c.disconnectLocked() +} + +// disconnectGeneration tears down the connection a request ran on, and only +// while that connection is still the installed one. It reports whether the +// generation still was. +// +// A caller cannot go by the transport state alone: Connect marks the client +// connected before it signs in, so a caller parked on c.mtx for the length of +// that sign-in wakes to a state that looks healthy and would close the socket +// the reconnect just established -- leaving the requests replaying over it +// with nothing to send on, and starting a second attempt for a reconnect that +// already happened. +func (c *IggyTcpClient) disconnectGeneration(generation uint64) (bool, error) { + c.mtx.Lock() + defer c.mtx.Unlock() + + if c.connGeneration != generation { + c.logger.Debug("Not disconnecting; the connection was already replaced.", + slog.Uint64("request_generation", generation), + slog.Uint64("current_generation", c.connGeneration)) + return false, nil + } + return true, c.disconnectLocked() +} +func (c *IggyTcpClient) disconnectLocked() error { if c.transportState == iggcon.TransportStateDisconnected || c.transportState == iggcon.TransportStateShutdown { return nil } diff --git a/foreign/go/client/tcp/tcp_failover_test.go b/foreign/go/client/tcp/tcp_failover_test.go index 1a035deee6..9a0ca663e6 100644 --- a/foreign/go/client/tcp/tcp_failover_test.go +++ b/foreign/go/client/tcp/tcp_failover_test.go @@ -748,6 +748,41 @@ func TestConnect_ConcurrentReconnectsThroughExchangeShareOneAttempt(t *testing.T "the two failing requests reconnected separately") } +// The teardown that follows a failure belongs to the connection the request +// ran on. Connect marks the client connected before it signs in, so a caller +// parked on c.mtx for that sign-in wakes to a healthy-looking state that says +// nothing about which connection is installed. +func TestDisconnect_DoesNotCloseAConnectionItDidNotFailOn(t *testing.T) { + var server *testListener + server = listenVSR(t, nil, singleNodeHandler(t, func() string { return server.address() })) + + client := newDialingClient(t, server.address(), + WithAutoLogin(NewUsernamePasswordCredentials("iggy", "iggy"))) + require.NoError(t, client.Connect(context.Background())) + + // What a request that fails on this connection carries with it. + failed := client.connGeneration + + require.NoError(t, client.disconnect()) + require.NoError(t, client.Connect(context.Background())) + require.NotEqual(t, failed, client.connGeneration, + "a reconnect installs a connection of its own") + + torn, err := client.disconnectGeneration(failed) + require.NoError(t, err) + assert.False(t, torn, "the stale generation reported a teardown it did not make") + assert.Equal(t, iggcon.TransportStateConnected, client.transportState) + require.NoError(t, client.Ping(context.Background()), + "the reconnected client was torn down by a stale failure") + assert.Equal(t, 2, server.connections(), "the stale teardown forced a third connection") + + // The connection the caller did fail on is still torn down. + torn, err = client.disconnectGeneration(client.connGeneration) + require.NoError(t, err) + assert.True(t, torn) + assert.Equal(t, iggcon.TransportStateDisconnected, client.transportState) +} + // The sign-in transaction holds registerMtx across its reconnect, and an // attempt started by a plain request ends in a sign-in that needs that same // lock. Waiting for that attempt closes a cycle -- the owner blocked on diff --git a/foreign/go/client/tcp/tcp_session_management.go b/foreign/go/client/tcp/tcp_session_management.go index a34d055651..e62fbc9656 100644 --- a/foreign/go/client/tcp/tcp_session_management.go +++ b/foreign/go/client/tcp/tcp_session_management.go @@ -176,8 +176,11 @@ func (c *IggyTcpClient) settleOnLeader(ctx context.Context, code uint32, body [] // The roster read runs while register holds the sign-in lock, so it must // not enter the reconnect path: the reconnect's automatic sign-in would // deadlock on that lock. The connect scope fails it fast instead. - redirect, err := c.HandleLeaderRedirection( - context.WithValue(ctx, connectScoped{}, struct{}{})) + c.mtx.Lock() + generation := c.connGeneration + c.mtx.Unlock() + redirect, err := c.redirectToLeader( + context.WithValue(ctx, connectScoped{}, struct{}{}), generation) if err != nil || !redirect { return settled, err } @@ -253,7 +256,24 @@ func (c *IggyTcpClient) LogoutUser(ctx context.Context) error { return nil } +// HandleLeaderRedirection moves the client to the leader the cluster roster +// names, tearing down whichever connection it is on. func (c *IggyTcpClient) HandleLeaderRedirection(ctx context.Context) (bool, error) { + c.mtx.Lock() + generation := c.connGeneration + c.mtx.Unlock() + return c.redirectToLeader(ctx, generation) +} + +// redirectToLeader moves the client to the leader, tearing down the +// connection generation the redirect was decided on. +// +// A caller whose connection was replaced while the roster was being read has +// nothing left to redirect: closing the replacement would strand the requests +// running on it, and reporting a redirect would have the caller replay on a +// node it never chose. It is told no redirect happened, and its re-attempt +// reads the roster over the connection it now has. +func (c *IggyTcpClient) redirectToLeader(ctx context.Context, generation uint64) (bool, error) { // Clone current address c.mtx.Lock() currentAddress := c.currentServerAddress @@ -292,9 +312,14 @@ func (c *IggyTcpClient) HandleLeaderRedirection(ctx context.Context) (bool, erro } c.mtx.Unlock() - if err = c.disconnect(); err != nil { + torn, err := c.disconnectGeneration(generation) + if err != nil { return false, err } + if !torn { + c.logger.Debug("Dropping a redirect decided on a replaced connection.") + return false, nil + } c.mtx.Lock() c.leaderRedirectionState.IncrementRedirect(leaderAddress) From 8b7dbd9c675479ef0a96e9b1cf072ef833cddf99 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Mon, 31 Aug 2026 09:08:48 +0200 Subject: [PATCH 014/182] chore(deps): Bump the go group across 2 directories with 2 updates (#3990) --- bdd/go/go.mod | 4 ++-- bdd/go/go.sum | 19 ++++++------------- examples/go/go.sum | 10 ++++------ foreign/go/go.mod | 5 ++--- foreign/go/go.sum | 10 ++++------ 5 files changed, 18 insertions(+), 30 deletions(-) diff --git a/bdd/go/go.mod b/bdd/go/go.mod index 1106967af4..d25ad1c130 100644 --- a/bdd/go/go.mod +++ b/bdd/go/go.mod @@ -8,7 +8,7 @@ require ( github.com/apache/iggy/foreign/go v0.0.0-00010101000000-000000000000 github.com/cucumber/godog v0.16.0 github.com/google/uuid v1.6.0 - github.com/onsi/ginkgo/v2 v2.32.0 + github.com/onsi/ginkgo/v2 v2.32.1 github.com/onsi/gomega v1.42.1 ) @@ -28,7 +28,7 @@ require ( github.com/klauspost/cpuid/v2 v2.2.10 // indirect github.com/spf13/pflag v1.0.10 // indirect github.com/zeebo/xxh3 v1.1.0 // indirect - go.yaml.in/yaml/v3 v3.0.4 // indirect + go.yaml.in/yaml/v3 v3.0.5 // indirect golang.org/x/mod v0.36.0 // indirect golang.org/x/net v0.56.0 // indirect golang.org/x/sync v0.21.0 // indirect diff --git a/bdd/go/go.sum b/bdd/go/go.sum index 37673653b2..3ed6d2cc2c 100644 --- a/bdd/go/go.sum +++ b/bdd/go/go.sum @@ -8,8 +8,6 @@ github.com/cucumber/godog v0.16.0 h1:ezQbgItuWqZrjPUQwLJ3muwIlvzXBOfZso5QZfG7efE github.com/cucumber/godog v0.16.0/go.mod h1:EDUX9yCqANK+GpbftMDeu61sUDtdLuo1JJgXD2n3bbM= github.com/cucumber/messages/go/v34 v34.2.0 h1:VCbcNOMz+f8ccjjOOx1NLBNhwvE7/X49Atc8klJa+i8= github.com/cucumber/messages/go/v34 v34.2.0/go.mod h1:LYUPjqlTS1kS0pdkdf6sS5uirnjwiIzEGyXPezXNhL8= -github.com/davecgh/go-spew v1.1.1 h1:vj9j/u1bqnvCEfJOwUhtlOARqs3+rkHYY13jYWTU97c= -github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= github.com/gkampitakis/ciinfo v0.3.2 h1:JcuOPk8ZU7nZQjdUhctuhQofk7BGHuIy0c9Ez8BNhXs= github.com/gkampitakis/ciinfo v0.3.2/go.mod h1:1NIwaOcFChN4fa/B0hEBdAb6npDlFL8Bwx4dfRLRqAo= github.com/gkampitakis/go-diff v1.3.2 h1:Qyn0J9XJSDTgnsgHRdz9Zp24RaJeKMUHg2+PDZZdC4M= @@ -53,18 +51,16 @@ github.com/maruel/natural v1.1.1 h1:Hja7XhhmvEFhcByqDoHz9QZbkWey+COd9xWfCfn1ioo= github.com/maruel/natural v1.1.1/go.mod h1:v+Rfd79xlw1AgVBjbO0BEQmptqb5HvL/k9GRHB7ZKEg= github.com/mfridman/tparse v0.18.0 h1:wh6dzOKaIwkUGyKgOntDW4liXSo37qg5AXbIhkMV3vE= github.com/mfridman/tparse v0.18.0/go.mod h1:gEvqZTuCgEhPbYk/2lS3Kcxg1GmTxxU7kTC8DvP0i/A= -github.com/onsi/ginkgo/v2 v2.32.0 h1:Hw7s2pVrQo/8Yz5N77qdnpHaoc+c6cC9WIV1Jce+J6E= -github.com/onsi/ginkgo/v2 v2.32.0/go.mod h1:+aXOY+vzZ5mu2iI2HpTZUPmM//oQfsNFX6gU9kNcA44= +github.com/onsi/ginkgo/v2 v2.32.1 h1:6tlvcDm/3sE8lGJbZ4+d4mO3RLy24/tQWOFzVSQNIfw= +github.com/onsi/ginkgo/v2 v2.32.1/go.mod h1:+aXOY+vzZ5mu2iI2HpTZUPmM//oQfsNFX6gU9kNcA44= github.com/onsi/gomega v1.42.1 h1:iN1rCUX+44NZ1Dc97MPoeFYbFR0vh8zxoxMFwKdyZ6I= github.com/onsi/gomega v1.42.1/go.mod h1:REff/hsDsodHoKlWsP2mAPhu1+5/6hVYNf9rIEBpeSg= -github.com/pmezard/go-difflib v1.0.0 h1:4DBwDE0NGyQoBHbLQYPwSUPoCMWR5BEzIk/f1lZbAQM= -github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4= github.com/rogpeppe/go-internal v1.14.1 h1:UQB4HGPB6osV0SQTLymcB4TgvyWu6ZyliaW0tI/otEQ= github.com/rogpeppe/go-internal v1.14.1/go.mod h1:MaRKkUm5W0goXpeCfT7UZI6fk/L7L7so1lCWt35ZSgc= github.com/spf13/pflag v1.0.10 h1:4EBh2KAYBwaONj6b2Ye1GiHfwjqyROoF4RwYO+vPwFk= github.com/spf13/pflag v1.0.10/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg= -github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu7U= -github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U= +github.com/stretchr/testify v1.12.1 h1:EuwCh5fleGS7H32xRwO3wRGT7DxrDhLAT6FF8MpWDWE= +github.com/stretchr/testify v1.12.1/go.mod h1:MDEgiDPPsNp5cuIrHPPCyornHKgEVbtFUmoNlxoYthg= github.com/tidwall/gjson v1.18.0 h1:FIDeeyB800efLX89e5a8Y0BNH+LOngJyGrIWxG2FKQY= github.com/tidwall/gjson v1.18.0/go.mod h1:/wbyibRr2FHMks5tjHJ5F8dMZh3AcwJEMf5vlfC0lxk= github.com/tidwall/match v1.1.1 h1:+Ho715JplO36QYgwN9PGYNhgZvoUSc9X2c80KVTi+GA= @@ -77,8 +73,8 @@ github.com/zeebo/assert v1.3.0 h1:g7C04CbJuIDKNPFHmsk4hwZDO5O+kntRxzaUoNXj+IQ= github.com/zeebo/assert v1.3.0/go.mod h1:Pq9JiuJQpG8JLJdtkwrJESF0Foym2/D9XMU5ciN/wJ0= github.com/zeebo/xxh3 v1.1.0 h1:s7DLGDK45Dyfg7++yxI0khrfwq9661w9EN78eP/UZVs= github.com/zeebo/xxh3 v1.1.0/go.mod h1:IisAie1LELR4xhVinxWS5+zf1lA4p0MW4T+w+W07F5s= -go.yaml.in/yaml/v3 v3.0.4 h1:tfq32ie2Jv2UxXFdLJdh3jXuOzWiL1fo0bu/FbuKpbc= -go.yaml.in/yaml/v3 v3.0.4/go.mod h1:DhzuOOF2ATzADvBadXxruRBLzYTpT36CKvDb3+aBEFg= +go.yaml.in/yaml/v3 v3.0.5 h1:N6y/pJk8buWs9NY5ERU2HSMfm+IuD/OtfdAnq6kESPw= +go.yaml.in/yaml/v3 v3.0.5/go.mod h1:HVTZu1O7/Vkt2N+BFy8Zza+lnLsABggaTM2ZpNIGuKg= golang.org/x/mod v0.36.0 h1:JJjpVx6myfUsUdAzZuOSTTmRE0PfZeNWzzvKrP7amb4= golang.org/x/mod v0.36.0/go.mod h1:moc6ELqsWcOw5Ef3xVprK5ul/MvtVvkIXLziUOICjUQ= golang.org/x/net v0.56.0 h1:Rw8j/hFzGvJUZwNBXnAtf5sVDVt+65SK2C7IxCxZt5o= @@ -93,8 +89,5 @@ golang.org/x/tools v0.45.0 h1:18qN3FAooORvApf5XjCXgsuayZOEtXf6JK18I3+ONa8= golang.org/x/tools v0.45.0/go.mod h1:LuUGqqaXcXMEFEruIVJVm5mgDD8vww/z/SR1gQ4uE/0= google.golang.org/protobuf v1.36.7 h1:IgrO7UwFQGJdRNXH/sQux4R1Dj1WAKcLElzeeRaXV2A= google.golang.org/protobuf v1.36.7/go.mod h1:jduwjTPXsFjZGTmRluh+L6NjiWu7pchiJ2/5YcXBHnY= -gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= -gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c h1:Hei/4ADfdWqJk1ZMxUNpqntNwaWcugrBjAiHlqqRiVk= -gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c/go.mod h1:JHkPIbrfpd72SG/EVd6muEfDQjcINNoR0C8j2r3qZ4Q= gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA= gopkg.in/yaml.v3 v3.0.1/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM= diff --git a/examples/go/go.sum b/examples/go/go.sum index 438f23b54c..0b46ffc404 100644 --- a/examples/go/go.sum +++ b/examples/go/go.sum @@ -1,7 +1,5 @@ github.com/avast/retry-go/v5 v5.0.0 h1:kf1Qc2UsTZ4qq8elDymqfbISvkyMuhgRxuJqX2NHP7k= github.com/avast/retry-go/v5 v5.0.0/go.mod h1://d+usmKWio1agtZfS1H/ltTqwtIfBnRq9zEwjc3eH8= -github.com/davecgh/go-spew v1.1.1 h1:vj9j/u1bqnvCEfJOwUhtlOARqs3+rkHYY13jYWTU97c= -github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8= github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX3N/iU= github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0= @@ -10,14 +8,14 @@ github.com/klauspost/compress v1.19.2 h1:hMRETovs/pu/dVWN7zIT1PGG8t509MwT6bO7XSi github.com/klauspost/compress v1.19.2/go.mod h1:cwPg85FWrGar70rWktvGQj8/hthj3wpl0PGDogxkrSQ= github.com/klauspost/cpuid/v2 v2.2.10 h1:tBs3QSyvjDyFTq3uoc/9xFpCuOsJQFNPiAhYdw2skhE= github.com/klauspost/cpuid/v2 v2.2.10/go.mod h1:hqwkgyIinND0mEev00jJYCxPNVRVXFQeu1XKlok6oO0= -github.com/pmezard/go-difflib v1.0.0 h1:4DBwDE0NGyQoBHbLQYPwSUPoCMWR5BEzIk/f1lZbAQM= -github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4= -github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu7U= -github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U= +github.com/stretchr/testify v1.12.1 h1:EuwCh5fleGS7H32xRwO3wRGT7DxrDhLAT6FF8MpWDWE= +github.com/stretchr/testify v1.12.1/go.mod h1:MDEgiDPPsNp5cuIrHPPCyornHKgEVbtFUmoNlxoYthg= github.com/zeebo/assert v1.3.0 h1:g7C04CbJuIDKNPFHmsk4hwZDO5O+kntRxzaUoNXj+IQ= github.com/zeebo/assert v1.3.0/go.mod h1:Pq9JiuJQpG8JLJdtkwrJESF0Foym2/D9XMU5ciN/wJ0= github.com/zeebo/xxh3 v1.1.0 h1:s7DLGDK45Dyfg7++yxI0khrfwq9661w9EN78eP/UZVs= github.com/zeebo/xxh3 v1.1.0/go.mod h1:IisAie1LELR4xhVinxWS5+zf1lA4p0MW4T+w+W07F5s= +go.yaml.in/yaml/v3 v3.0.5 h1:N6y/pJk8buWs9NY5ERU2HSMfm+IuD/OtfdAnq6kESPw= +go.yaml.in/yaml/v3 v3.0.5/go.mod h1:HVTZu1O7/Vkt2N+BFy8Zza+lnLsABggaTM2ZpNIGuKg= golang.org/x/sys v0.30.0 h1:QjkSwP/36a20jFYWkSue1YwXzLmsV5Gfq7Eiy72C1uc= golang.org/x/sys v0.30.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA= gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA= diff --git a/foreign/go/go.mod b/foreign/go/go.mod index cfd8f6551d..ce2cbe178b 100644 --- a/foreign/go/go.mod +++ b/foreign/go/go.mod @@ -7,17 +7,16 @@ require ( github.com/google/go-cmp v0.7.0 github.com/google/uuid v1.6.0 github.com/klauspost/compress v1.19.2 - github.com/stretchr/testify v1.11.1 + github.com/stretchr/testify v1.12.1 github.com/zeebo/xxh3 v1.1.0 gopkg.in/yaml.v3 v3.0.1 ) require ( - github.com/davecgh/go-spew v1.1.1 // indirect github.com/klauspost/cpuid/v2 v2.2.10 // indirect github.com/kr/pretty v0.3.1 // indirect - github.com/pmezard/go-difflib v1.0.0 // indirect github.com/rogpeppe/go-internal v1.14.1 // indirect + go.yaml.in/yaml/v3 v3.0.5 // indirect golang.org/x/sys v0.30.0 // indirect gopkg.in/check.v1 v1.0.0-20201130134442-10cb98267c6c // indirect ) diff --git a/foreign/go/go.sum b/foreign/go/go.sum index 36e95e80c9..f1f23d423d 100644 --- a/foreign/go/go.sum +++ b/foreign/go/go.sum @@ -1,8 +1,6 @@ github.com/avast/retry-go/v5 v5.0.0 h1:kf1Qc2UsTZ4qq8elDymqfbISvkyMuhgRxuJqX2NHP7k= github.com/avast/retry-go/v5 v5.0.0/go.mod h1://d+usmKWio1agtZfS1H/ltTqwtIfBnRq9zEwjc3eH8= github.com/creack/pty v1.1.9/go.mod h1:oKZEueFk5CKHvIhNR5MUki03XCEU+Q6VDXinZuGJ33E= -github.com/davecgh/go-spew v1.1.1 h1:vj9j/u1bqnvCEfJOwUhtlOARqs3+rkHYY13jYWTU97c= -github.com/davecgh/go-spew v1.1.1/go.mod h1:J7Y8YcW2NihsgmVo/mv3lAwl/skON4iLHjSsI+c5H38= github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8= github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX3N/iU= github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0= @@ -19,17 +17,17 @@ github.com/kr/text v0.1.0/go.mod h1:4Jbv+DJW3UT/LiOwJeYQe1efqtUx/iVham/4vfdArNI= github.com/kr/text v0.2.0 h1:5Nx0Ya0ZqY2ygV366QzturHI13Jq95ApcVaJBhpS+AY= github.com/kr/text v0.2.0/go.mod h1:eLer722TekiGuMkidMxC/pM04lWEeraHUUmBw8l2grE= github.com/pkg/diff v0.0.0-20210226163009-20ebb0f2a09e/go.mod h1:pJLUxLENpZxwdsKMEsNbx1VGcRFpLqf3715MtcvvzbA= -github.com/pmezard/go-difflib v1.0.0 h1:4DBwDE0NGyQoBHbLQYPwSUPoCMWR5BEzIk/f1lZbAQM= -github.com/pmezard/go-difflib v1.0.0/go.mod h1:iKH77koFhYxTK1pcRnkKkqfTogsbg7gZNVY4sRDYZ/4= github.com/rogpeppe/go-internal v1.9.0/go.mod h1:WtVeX8xhTBvf0smdhujwtBcq4Qrzq/fJaraNFVN+nFs= github.com/rogpeppe/go-internal v1.14.1 h1:UQB4HGPB6osV0SQTLymcB4TgvyWu6ZyliaW0tI/otEQ= github.com/rogpeppe/go-internal v1.14.1/go.mod h1:MaRKkUm5W0goXpeCfT7UZI6fk/L7L7so1lCWt35ZSgc= -github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu7U= -github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U= +github.com/stretchr/testify v1.12.1 h1:EuwCh5fleGS7H32xRwO3wRGT7DxrDhLAT6FF8MpWDWE= +github.com/stretchr/testify v1.12.1/go.mod h1:MDEgiDPPsNp5cuIrHPPCyornHKgEVbtFUmoNlxoYthg= github.com/zeebo/assert v1.3.0 h1:g7C04CbJuIDKNPFHmsk4hwZDO5O+kntRxzaUoNXj+IQ= github.com/zeebo/assert v1.3.0/go.mod h1:Pq9JiuJQpG8JLJdtkwrJESF0Foym2/D9XMU5ciN/wJ0= github.com/zeebo/xxh3 v1.1.0 h1:s7DLGDK45Dyfg7++yxI0khrfwq9661w9EN78eP/UZVs= github.com/zeebo/xxh3 v1.1.0/go.mod h1:IisAie1LELR4xhVinxWS5+zf1lA4p0MW4T+w+W07F5s= +go.yaml.in/yaml/v3 v3.0.5 h1:N6y/pJk8buWs9NY5ERU2HSMfm+IuD/OtfdAnq6kESPw= +go.yaml.in/yaml/v3 v3.0.5/go.mod h1:HVTZu1O7/Vkt2N+BFy8Zza+lnLsABggaTM2ZpNIGuKg= golang.org/x/sys v0.30.0 h1:QjkSwP/36a20jFYWkSue1YwXzLmsV5Gfq7Eiy72C1uc= golang.org/x/sys v0.30.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA= gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0= From a1df38354bc6ba521f4440abab03b19643a763e8 Mon Sep 17 00:00:00 2001 From: Hubert Gruszecki Date: Mon, 31 Aug 2026 09:28:55 +0200 Subject: [PATCH 015/182] test(partitions): gate the persist-retry test on Linux (#3998) --- core/partitions/src/iggy_partition.rs | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/core/partitions/src/iggy_partition.rs b/core/partitions/src/iggy_partition.rs index 93eaf156f0..7b538e68da 100644 --- a/core/partitions/src/iggy_partition.rs +++ b/core/partitions/src/iggy_partition.rs @@ -7656,8 +7656,6 @@ mod tests { ); } - /// A half that failed never advanced its cursor, so the retry rewrites - /// the same positions: one copy of the batch, one index entry. #[cfg(target_os = "linux")] #[compio::test] async fn given_a_persist_failure_on_a_committed_op_should_fence_the_partition_not_panic() { @@ -7744,6 +7742,9 @@ mod tests { assert_eq!(kept.operation, Operation::StoreConsumerOffset); } + /// A half that failed never advanced its cursor, so the retry rewrites + /// the same positions: one copy of the batch, one index entry. + #[cfg(target_os = "linux")] #[compio::test] async fn given_failed_persist_when_retried_with_a_healthy_writer_should_overwrite_the_same_positions() { From 31b86ff2a9ef2e7802f55c9211eb3cfab541be47 Mon Sep 17 00:00:00 2001 From: Chengxi Luo Date: Mon, 31 Aug 2026 04:47:13 -0400 Subject: [PATCH 016/182] fix(server)!: reject a wildcard bind with no advertised address (#3923) --- .devcontainer/devcontainer.json | 3 +- README.md | 2 +- bdd/docker-compose.server.yml | 1 + core/configs/src/server_config/cluster.rs | 423 +++++++++++------- core/configs/src/server_config/defaults.rs | 2 + core/configs/src/server_config/mod.rs | 1 + core/configs/src/server_config/node.rs | 89 ++++ core/configs/src/server_config/server.rs | 54 +++ core/configs/src/server_config/validators.rs | 133 ++++++ .../server/cluster_metadata_advertised.rs | 80 ++++ core/integration/tests/server/mod.rs | 3 + core/server/README.md | 11 +- core/server/config.toml | 18 +- core/server/src/args.rs | 4 +- core/server/src/bootstrap.rs | 281 +++++++++++- core/server/src/cluster_meta.rs | 188 ++++++-- core/server/src/dispatch.rs | 9 +- core/server/src/http.rs | 7 +- core/server/src/http/error.rs | 19 +- core/server/src/http/forward.rs | 18 +- docker-compose.yml | 4 + examples/csharp/README.md | 2 +- examples/go/README.md | 2 +- examples/java/README.md | 2 +- examples/node/README.md | 5 +- examples/python/README.md | 5 +- examples/rust/README.md | 2 +- .../Fixtures/VsrCluster.cs | 1 + .../iggy-connector-flink/docker-compose.yml | 1 + .../iggy-connector-pinot/docker-compose.yml | 1 + .../iggy/client/BaseIntegrationTest.java | 1 + foreign/php/README.md | 1 + foreign/php/docker-compose.test.yml | 1 + foreign/python/docker-compose.test.yml | 1 + foreign/python/tests/test_tls.py | 1 + helm/charts/iggy/README.md | 20 +- helm/charts/iggy/README.md.gotmpl | 17 + helm/charts/iggy/templates/deployment.yaml | 16 + helm/charts/iggy/values.yaml | 5 + web/README.md | 3 +- web/docker-compose.yml | 1 + 41 files changed, 1184 insertions(+), 254 deletions(-) create mode 100644 core/configs/src/server_config/node.rs create mode 100644 core/integration/tests/server/cluster_metadata_advertised.rs diff --git a/.devcontainer/devcontainer.json b/.devcontainer/devcontainer.json index 450aad7972..3110ad371b 100644 --- a/.devcontainer/devcontainer.json +++ b/.devcontainer/devcontainer.json @@ -33,7 +33,8 @@ "IGGY_HTTP_ADDRESS": "0.0.0.0:3000", "IGGY_QUIC_ADDRESS": "0.0.0.0:8080", "IGGY_TCP_ADDRESS": "0.0.0.0:8090", - "IGGY_WEBSOCKET_ADDRESS": "0.0.0.0:8092" + "IGGY_WEBSOCKET_ADDRESS": "0.0.0.0:8092", + "IGGY_NODE_ADVERTISED_ADDRESS": "localhost" }, "forwardPorts": [3000, 3050, 8080, 8090, 8092], "portsAttributes": { diff --git a/README.md b/README.md index fd61c9a57d..74690f17e1 100644 --- a/README.md +++ b/README.md @@ -288,7 +288,7 @@ For configuration options and detailed help: You can also use environment variables to override any configuration setting: - Override TCP address - `IGGY_TCP_ADDRESS=0.0.0.0:8090 cargo run --bin iggy-server` + `IGGY_TCP_ADDRESS=127.0.0.1:8090 cargo run --bin iggy-server` - Set custom data path `IGGY_SYSTEM_PATH=/data/iggy cargo run --bin iggy-server` diff --git a/bdd/docker-compose.server.yml b/bdd/docker-compose.server.yml index e22484533c..5563040feb 100644 --- a/bdd/docker-compose.server.yml +++ b/bdd/docker-compose.server.yml @@ -65,6 +65,7 @@ services: - IGGY_ROOT_PASSWORD=iggy - IGGY_SYSTEM_PATH=local_data - IGGY_TCP_ADDRESS=0.0.0.0:8090 + - IGGY_NODE_ADVERTISED_ADDRESS=iggy-server - IGGY_HTTP_ADDRESS=0.0.0.0:3000 - IGGY_QUIC_ADDRESS=0.0.0.0:8080 - IGGY_WEBSOCKET_ADDRESS=0.0.0.0:8070 diff --git a/core/configs/src/server_config/cluster.rs b/core/configs/src/server_config/cluster.rs index 033ea6f77d..88d9b195fa 100644 --- a/core/configs/src/server_config/cluster.rs +++ b/core/configs/src/server_config/cluster.rs @@ -421,6 +421,7 @@ pub struct ClusterTlsConfig { #[serde(deny_unknown_fields)] pub struct ClusterNodeConfig { pub name: String, + /// Replica-plane address, dialed verbatim by every peer, a literal IP. pub ip: String, /// Optional client-facing address: a literal IP or a DNS hostname, /// validated as [`AdvertisedAddress`] at boot. Replica traffic continues @@ -473,10 +474,7 @@ pub struct AdvertisedAddressSelector { /// once, built wherever a roster is assembled for serving clients /// (listener/shard start). Per-request resolution never re-parses config /// strings: everything is snapshotted here, so mutating the source config -/// after conversion has no effect on what clients are told. Entries that do -/// not parse are dropped at build time; validation already rejects them -/// whenever the cluster is enabled, and a disabled cluster never consults -/// the roster. +/// after conversion has no effect on what clients are told. #[derive(Debug, Clone)] pub struct ResolvedClusterNode { config: ClusterNodeConfig, @@ -484,38 +482,82 @@ pub struct ResolvedClusterNode { /// addresses, in declaration order. selectors: Vec<(IpNet, AdvertisedAddress)>, /// Parsed catch-all: [`ClusterNodeConfig::advertised_address`], else the - /// roster [`ClusterNodeConfig::ip`]. `None` when the configured value - /// does not parse - a set `advertised_address` never falls through to - /// the private roster ip. - catch_all: Option, + /// roster [`ClusterNodeConfig::ip`]. A set `advertised_address` never + /// falls through to the private roster ip. + catch_all: AdvertisedAddress, /// Parsed roster [`ClusterNodeConfig::ip`], the replica-plane dial - /// address. `None` when the roster ip is not a literal IP (boot only - /// requires it non-empty); internal forwarding then has no dial target. - replica_ip: Option, + /// address. + replica_ip: IpAddr, } -impl From for ResolvedClusterNode { - fn from(config: ClusterNodeConfig) -> Self { - let selectors = config - .advertised_addresses - .iter() - .filter_map(|selector| { - let network = selector.client_cidr.parse::().ok()?; - let address = selector.address.parse::().ok()?; - Some((canonical_ip_net(network.trunc()), address)) - }) - .collect(); - let catch_all = match config.advertised_address.as_deref() { - Some(advertised_address) => advertised_address.parse().ok(), - None => config.ip.parse().ok(), - }; - let replica_ip = config.ip.parse().ok(); - Self { +impl TryFrom for ResolvedClusterNode { + type Error = ConfigurationError; + + fn try_from(config: ClusterNodeConfig) -> Result { + let replica_ip = config.ip.parse::().map_err(|error| { + eprintln!( + "Invalid cluster configuration: IP '{}' for node '{}' is not a literal IP \ + address: {error}; set cluster.nodes[*].advertised_address for the name clients \ + dial", + config.ip, config.name + ); + ConfigurationError::InvalidConfigurationValue + })?; + + if replica_ip.to_canonical().is_unspecified() { + eprintln!( + "Invalid cluster configuration: IP '{}' for node '{}' is the unspecified \ + address; declare the address peers and clients reach this node at", + config.ip, config.name + ); + return Err(ConfigurationError::InvalidConfigurationValue); + } + + let catch_all = + match config.advertised_address.as_deref() { + Some(advertised_address) => advertised_address + .parse::() + .map_err(|error| { + eprintln!( + "Invalid cluster configuration: advertised_address \ + '{advertised_address}' for node '{}': {error}", + config.name + ); + ConfigurationError::InvalidConfigurationValue + })?, + None => AdvertisedAddress::Ip(replica_ip.to_canonical()), + }; + + let mut selectors = Vec::with_capacity(config.advertised_addresses.len()); + for selector in &config.advertised_addresses { + let network = selector.client_cidr.parse::().map_err(|error| { + eprintln!( + "Invalid cluster configuration: advertised_addresses client_cidr '{}' for \ + node '{}': {error}", + selector.client_cidr, config.name + ); + ConfigurationError::InvalidConfigurationValue + })?; + let address = selector + .address + .parse::() + .map_err(|error| { + eprintln!( + "Invalid cluster configuration: advertised_addresses address '{}' for \ + node '{}': {error}", + selector.address, config.name + ); + ConfigurationError::InvalidConfigurationValue + })?; + selectors.push((canonical_ip_net(network.trunc()), address)); + } + + Ok(Self { config, selectors, catch_all, replica_ip, - } + }) } } @@ -531,33 +573,19 @@ impl ResolvedClusterNode { /// internal request forwarding. Never routed through the advertised /// ladder: this is what servers dial, not what clients are told. #[must_use] - pub fn replica_ip(&self) -> Option { + pub fn replica_ip(&self) -> IpAddr { self.replica_ip } /// The client-facing address for a client connecting from `client_ip`: - /// longest-prefix match over the selector networks, then the parsed - /// catch-all. `None` when no selector matches and the catch-all did not - /// parse; callers choose whether to fail closed (redirect URLs) or to - /// publish [`Self::raw_advertised_fallback`] verbatim (cluster metadata). + /// longest-prefix match over the selector networks, then the catch-all. + /// Always an address, since construction refused a node whose sources did + /// not parse. #[must_use] - pub fn advertised_for(&self, client_ip: Option) -> Option<&AdvertisedAddress> { + pub fn advertised_for(&self, client_ip: Option) -> &AdvertisedAddress { client_ip .and_then(|client_ip| self.selector_address(client_ip)) - .or(self.catch_all.as_ref()) - } - - /// The catch-all ladder ([`ClusterNodeConfig::advertised_address`], else - /// the roster [`ClusterNodeConfig::ip`]) as configured, unparsed. Cluster - /// metadata publishes this verbatim when [`Self::advertised_for`] finds - /// nothing: the roster `ip` is only validated non-empty, and Docker - /// service names with underscores exist in the wild. - #[must_use] - pub fn raw_advertised_fallback(&self) -> &str { - self.config - .advertised_address - .as_deref() - .unwrap_or(&self.config.ip) + .unwrap_or(&self.catch_all) } /// Longest-prefix match over the boot-parsed selector networks. The @@ -620,8 +648,9 @@ pub struct TransportPorts { /// consisting solely of digits and dots are rejected as malformed IPv4 rather /// than accepted as hostnames, so `10.0.0.256` fails loudly instead of being /// handed to DNS. Hostnames normalize to lowercase and IPs to their canonical -/// form ([`IpAddr`]), so textual variants of one address (`Broker.Example.COM`, -/// `2001:DB8::1`, `[2001:db8::1]`) compare equal. +/// form ([`IpAddr::to_canonical`]), so textual variants of one address +/// (`Broker.Example.COM`, `2001:DB8::1`, `[2001:db8::1]`, `::ffff:10.0.0.1`) +/// compare equal. #[derive(Debug, Clone, PartialEq, Eq, Hash)] pub enum AdvertisedAddress { Ip(IpAddr), @@ -637,6 +666,17 @@ impl AdvertisedAddress { Self::Hostname(hostname) => format!("{hostname}:{port}"), } } + + /// Canonicalize first, so the v4-mapped spellings of one address + /// (`::ffff:10.0.0.1`, `::ffff:0.0.0.0`) are the same value as their v4 + /// form for both the comparison and the wildcard refusal. + fn from_ip(ip: IpAddr) -> Result { + let ip = ip.to_canonical(); + if ip.is_unspecified() { + return Err(AdvertisedAddressError::Unspecified); + } + Ok(Self::Ip(ip)) + } } impl FromStr for AdvertisedAddress { @@ -647,7 +687,7 @@ impl FromStr for AdvertisedAddress { return Err(AdvertisedAddressError::Empty); } if let Ok(ip) = address.parse::() { - return Ok(Self::Ip(ip)); + return Self::from_ip(ip); } // URL-style bracketed IPv6 (`[2001:db8::1]`) is unambiguous; accept // it and store the inner address. @@ -656,7 +696,7 @@ impl FromStr for AdvertisedAddress { .and_then(|rest| rest.strip_suffix(']')) && let Ok(ip) = inner.parse::() { - return Ok(Self::Ip(IpAddr::V6(ip))); + return Self::from_ip(IpAddr::V6(ip)); } if let Some((host, port)) = address.rsplit_once(':') { // `host:port` and `[v6]:port` are the common misconfigurations; @@ -720,6 +760,7 @@ impl fmt::Display for AdvertisedAddress { #[derive(Debug, Clone, PartialEq, Eq)] pub enum AdvertisedAddressError { Empty, + Unspecified, PortNotAllowed, MalformedIpv4, MalformedIpv6, @@ -734,9 +775,15 @@ impl fmt::Display for AdvertisedAddressError { fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { match self { Self::Empty => write!(formatter, "address cannot be empty"), + Self::Unspecified => write!( + formatter, + "address is the unspecified address, which tells a client which interfaces this \ + node accepts on rather than where to reach it; declare a routable address" + ), Self::PortNotAllowed => write!( formatter, - "address must not include a port; ports are configured in cluster.nodes.ports" + "address must not include a port; each transport's port comes from its own \ + listener address in single-node mode and from cluster.nodes.ports in a cluster" ), Self::MalformedIpv4 => write!( formatter, @@ -990,13 +1037,7 @@ impl Validatable for ClusterConfig { return Err(ConfigurationError::InvalidConfigurationValue); } - if node.ip.trim().is_empty() { - eprintln!( - "Invalid cluster configuration: IP cannot be empty for node '{}'", - node.name - ); - return Err(ConfigurationError::InvalidConfigurationValue); - } + let resolved = ResolvedClusterNode::try_from(node.clone())?; if !seen_names.insert(node.name.clone()) { eprintln!( @@ -1040,8 +1081,12 @@ impl Validatable for ClusterConfig { return Err(ConfigurationError::InvalidConfigurationValue); } - let endpoint = format!("{}:{}", node.ip, port); - if !used_endpoints.insert(endpoint.clone()) { + // Keyed on the canonical form, and rendered from it, so + // two spellings of one address ('::ffff:10.0.0.1' and + // '10.0.0.1') cannot claim the same port twice and an IPv6 + // endpoint reads back with its port separable. + let endpoint = SocketAddr::new(resolved.replica_ip().to_canonical(), port); + if !used_endpoints.insert(endpoint) { eprintln!( "Invalid cluster configuration: port conflict - {endpoint} is already bound (node '{}', transport {name})", node.name @@ -1051,28 +1096,6 @@ impl Validatable for ClusterConfig { } } - // An advertised address must parse strictly (IP or RFC 1123 - // hostname): the value is handed verbatim to every client via - // cluster metadata and redirect URLs, so a bad one poisons them - // all. The roster `ip` predates this check and is only validated - // as non-empty (Docker service names with underscores exist in - // the wild), so when it backs the client endpoints an unparsable - // value falls back to raw-string comparison instead of failing - // boot. - let client_address = match node.advertised_address.as_deref() { - Some(advertised_address) => match advertised_address.parse::() { - Ok(address) => Some(address), - Err(error) => { - eprintln!( - "Invalid cluster configuration: advertised_address '{advertised_address}' for node '{}': {error}", - node.name - ); - return Err(ConfigurationError::InvalidConfigurationValue); - } - }, - None => node.ip.parse::().ok(), - }; - if node.advertised_addresses.len() > MAX_ADVERTISED_SELECTORS { eprintln!( "Invalid cluster configuration: node '{}' declares {} advertised_addresses \ @@ -1083,45 +1106,20 @@ impl Validatable for ClusterConfig { return Err(ConfigurationError::InvalidConfigurationValue); } - // Selector CIDRs and addresses feed clients the same way the - // catch-all advertised address does, so they get the same strict - // parse. Networks are compared truncated (`10.0.1.0/16` == + // Networks are compared truncated (`10.0.1.0/16` == // `10.0.0.0/16`) and canonicalized (`::ffff:10.0.0.0/104` == - // `10.0.0.0/8`) since matching truncates and canonicalizes too. - // Parsed before the catch-all enters the conflict pool because - // every entry's effective client set depends on the node's full - // selector list. - let mut selectors = Vec::with_capacity(node.advertised_addresses.len()); + // `10.0.0.0/8`), the form resolution matched them in, so two + // spellings of one network cannot both be declared. + let selectors = &resolved.selectors; let mut seen_selector_cidrs = std::collections::HashSet::new(); - for selector in &node.advertised_addresses { - let client_cidr = match selector.client_cidr.parse::() { - Ok(client_cidr) => canonical_ip_net(client_cidr.trunc()), - Err(error) => { - eprintln!( - "Invalid cluster configuration: advertised_addresses client_cidr '{}' for node '{}': {error}", - selector.client_cidr, node.name - ); - return Err(ConfigurationError::InvalidConfigurationValue); - } - }; - if !seen_selector_cidrs.insert(client_cidr) { + for ((client_cidr, _), declared) in selectors.iter().zip(&node.advertised_addresses) { + if !seen_selector_cidrs.insert(*client_cidr) { eprintln!( "Invalid cluster configuration: duplicate advertised_addresses client_cidr '{}' for node '{}'", - selector.client_cidr, node.name + declared.client_cidr, node.name ); return Err(ConfigurationError::InvalidConfigurationValue); } - let address = match selector.address.parse::() { - Ok(address) => address, - Err(error) => { - eprintln!( - "Invalid cluster configuration: advertised_addresses address '{}' for node '{}': {error}", - selector.address, node.name - ); - return Err(ConfigurationError::InvalidConfigurationValue); - } - }; - selectors.push((client_cidr, address)); } let selector_ranges: Vec = selectors .iter() @@ -1136,26 +1134,21 @@ impl Validatable for ClusterConfig { // override shadowing the same node's wider selector); a conflict // means some client wins both entries and would resolve both // nodes to one endpoint. The catch-all is an implicit - // match-everything-else selector, so it pools the same way. A - // roster ip that fails the strict parse skips the pool: it can - // never equal a parsed host, and two raw ips sharing host:port - // are already rejected by the bind-endpoint check above. - if let Some(address) = &client_address { - let catch_all_clients = EffectiveClients::for_catch_all(&selector_ranges); - for (name, port) in &client_ports { - if let Some(port) = port { - insert_advertised_endpoint( - &mut advertised_endpoints, - AdvertisedEndpoint { - node_name: &node.name, - transport: name, - network: None, - clients: catch_all_clients.clone(), - host: address.clone(), - port: *port, - }, - )?; - } + // match-everything-else selector, so it pools the same way. + let catch_all_clients = EffectiveClients::for_catch_all(&selector_ranges); + for (name, port) in &client_ports { + if let Some(port) = port { + insert_advertised_endpoint( + &mut advertised_endpoints, + AdvertisedEndpoint { + node_name: &node.name, + transport: name, + network: None, + clients: catch_all_clients.clone(), + host: resolved.catch_all.clone(), + port: *port, + }, + )?; } } @@ -1560,6 +1553,47 @@ mod advertised_address_tests { } } + #[test] + fn rejects_every_spelling_of_the_unspecified_address() { + for wildcard in [ + "0.0.0.0", + "::", + "::ffff:0.0.0.0", + "[::]", + "[::ffff:0.0.0.0]", + ] { + assert_eq!( + wildcard.parse::(), + Err(AdvertisedAddressError::Unspecified), + "'{wildcard}' names interfaces, not a dial target" + ); + } + for routable in ["203.0.113.1", "2001:db8::1", "::ffff:203.0.113.1", "broker"] { + assert!( + routable.parse::().is_ok(), + "'{routable}' is dialable" + ); + } + } + + /// A v4-mapped IPv6 address and its v4 form are one address, so they must + /// be one value: the conflict pool compares advertised endpoints for + /// equality, and two spellings would let one host:port slip past it twice. + #[test] + fn canonicalizes_a_v4_mapped_address_to_its_v4_form() { + assert_eq!( + "::ffff:10.0.0.1".parse::(), + "10.0.0.1".parse::() + ); + assert_eq!( + "::ffff:10.0.0.1" + .parse::() + .unwrap() + .authority(8090), + "10.0.0.1:8090" + ); + } + #[test] fn normalizes_hostname_to_lowercase() { let address = "Broker-1.Example.COM".parse::(); @@ -1641,7 +1675,7 @@ mod advertised_for_tests { } fn resolved(node: ClusterNodeConfig) -> ResolvedClusterNode { - node.into() + ResolvedClusterNode::try_from(node).expect("a roster node the validator would accept") } fn ip(address: &str) -> IpAddr { @@ -1653,7 +1687,7 @@ mod advertised_for_tests { let node = node_with_selectors(Vec::new()); assert_eq!( resolved(node).advertised_for(Some(ip("10.0.0.7"))), - Some(&AdvertisedAddress::Ip(ip("203.0.113.10"))) + &AdvertisedAddress::Ip(ip("203.0.113.10")) ); } @@ -1663,16 +1697,47 @@ mod advertised_for_tests { node.advertised_address = None; assert_eq!( resolved(node).advertised_for(Some(ip("10.0.0.7"))), - Some(&AdvertisedAddress::Ip(ip("10.0.1.5"))) + &AdvertisedAddress::Ip(ip("10.0.1.5")) ); } #[test] - fn is_none_when_no_fallback_parses() { + fn refuses_a_node_whose_ip_is_not_an_address() { + // Nothing downstream carries a fallback for an unparsable source, so + // the conversion is where such a node has to stop. let mut node = node_with_selectors(Vec::new()); node.advertised_address = None; node.ip = "iggy_node".to_owned(); - assert_eq!(resolved(node).advertised_for(Some(ip("10.0.0.7"))), None); + assert!(ResolvedClusterNode::try_from(node).is_err()); + } + + #[test] + fn refuses_a_node_advertising_the_unspecified_address() { + // The conversion is reachable without the validator, so the "every + // resolved node is dialable" invariant has to hold here rather than + // rely on validation having run first. + for wildcard in ["0.0.0.0", "::", "::ffff:0.0.0.0"] { + let mut roster_ip = node_with_selectors(Vec::new()); + roster_ip.advertised_address = None; + roster_ip.ip = wildcard.to_owned(); + assert!( + ResolvedClusterNode::try_from(roster_ip).is_err(), + "roster ip '{wildcard}' is not dialable" + ); + + let mut catch_all = node_with_selectors(Vec::new()); + catch_all.advertised_address = Some(wildcard.to_owned()); + assert!( + ResolvedClusterNode::try_from(catch_all).is_err(), + "advertised_address '{wildcard}' is not dialable" + ); + + let selectors = node_with_selectors(vec![selector("10.0.0.0/16", wildcard)]); + assert!( + ResolvedClusterNode::try_from(selectors).is_err(), + "selector address '{wildcard}' is not dialable" + ); + } } #[test] @@ -1680,7 +1745,7 @@ mod advertised_for_tests { let node = node_with_selectors(vec![selector("10.0.0.0/16", "10.0.1.5")]); assert_eq!( resolved(node).advertised_for(Some(ip("10.0.200.7"))), - Some(&AdvertisedAddress::Ip(ip("10.0.1.5"))) + &AdvertisedAddress::Ip(ip("10.0.1.5")) ); } @@ -1689,7 +1754,7 @@ mod advertised_for_tests { let node = node_with_selectors(vec![selector("10.0.0.0/16", "10.0.1.5")]); assert_eq!( resolved(node).advertised_for(Some(ip("192.168.0.7"))), - Some(&AdvertisedAddress::Ip(ip("203.0.113.10"))) + &AdvertisedAddress::Ip(ip("203.0.113.10")) ); } @@ -1698,7 +1763,7 @@ mod advertised_for_tests { let node = node_with_selectors(vec![selector("10.0.0.0/16", "10.0.1.5")]); assert_eq!( resolved(node).advertised_for(None), - Some(&AdvertisedAddress::Ip(ip("203.0.113.10"))) + &AdvertisedAddress::Ip(ip("203.0.113.10")) ); } @@ -1710,12 +1775,12 @@ mod advertised_for_tests { ])); assert_eq!( node.advertised_for(Some(ip("10.0.200.7"))), - Some(&AdvertisedAddress::Ip(ip("10.0.1.5"))), + &AdvertisedAddress::Ip(ip("10.0.1.5")), "the /16 must win over the /8 even though it is declared second" ); assert_eq!( node.advertised_for(Some(ip("10.9.0.7"))), - Some(&AdvertisedAddress::Ip(ip("10.255.255.1"))), + &AdvertisedAddress::Ip(ip("10.255.255.1")), "a client outside the /16 but inside the /8 must match the /8" ); } @@ -1732,7 +1797,7 @@ mod advertised_for_tests { ]); assert_eq!( resolved(node).advertised_for(Some(ip("10.0.200.7"))), - Some(&AdvertisedAddress::Ip(ip("10.0.1.5"))) + &AdvertisedAddress::Ip(ip("10.0.1.5")) ); } @@ -1742,7 +1807,7 @@ mod advertised_for_tests { let node = node_with_selectors(vec![selector("10.0.0.0/16", "10.0.1.5")]); assert_eq!( resolved(node).advertised_for(Some(ip("::ffff:10.0.0.7"))), - Some(&AdvertisedAddress::Ip(ip("10.0.1.5"))) + &AdvertisedAddress::Ip(ip("10.0.1.5")) ); } @@ -1754,7 +1819,7 @@ mod advertised_for_tests { let node = node_with_selectors(vec![selector("::ffff:10.0.0.0/104", "10.0.1.5")]); assert_eq!( resolved(node).advertised_for(Some(ip("10.0.0.7"))), - Some(&AdvertisedAddress::Ip(ip("10.0.1.5"))) + &AdvertisedAddress::Ip(ip("10.0.1.5")) ); } @@ -1763,7 +1828,7 @@ mod advertised_for_tests { let node = node_with_selectors(vec![selector("2001:db8::/32", "2001:db8::1")]); assert_eq!( resolved(node).advertised_for(Some(ip("2001:db8::7"))), - Some(&AdvertisedAddress::Ip(ip("2001:db8::1"))) + &AdvertisedAddress::Ip(ip("2001:db8::1")) ); } @@ -1772,9 +1837,7 @@ mod advertised_for_tests { let node = node_with_selectors(vec![selector("10.0.0.0/16", "Broker.Internal.Example")]); assert_eq!( resolved(node).advertised_for(Some(ip("10.0.0.7"))), - Some(&AdvertisedAddress::Hostname( - "broker.internal.example".to_owned() - )) + &AdvertisedAddress::Hostname("broker.internal.example".to_owned()) ); } } @@ -1982,6 +2045,45 @@ mod cluster_validate_tests { assert!(c.validate().is_err()); } + #[test] + fn validate_rejects_an_unspecified_node_ip() { + for wildcard in ["0.0.0.0", "::", "::ffff:0.0.0.0"] { + let mut nodes = vec![node("n1", 0), node("n2", 1)]; + nodes[0].ip = wildcard.to_owned(); + assert!(cfg(nodes).validate().is_err(), "{wildcard} is not dialable"); + } + } + + #[test] + fn validate_rejects_a_hostname_node_ip() { + for hostname in ["iggy_leader", "node-1.example.com"] { + let mut nodes = vec![node("n1", 0), node("n2", 1)]; + nodes[0].ip = hostname.to_owned(); + assert!( + cfg(nodes).validate().is_err(), + "'{hostname}' must not pass as a roster ip" + ); + } + } + + #[test] + fn validate_rejects_an_unspecified_advertised_address() { + for wildcard in ["0.0.0.0", "::", "::ffff:0.0.0.0"] { + let mut nodes = vec![node("n1", 0), node("n2", 1)]; + nodes[0].advertised_address = Some(wildcard.to_owned()); + assert!(cfg(nodes).validate().is_err(), "{wildcard} is not dialable"); + } + } + + #[test] + fn validate_rejects_an_unspecified_selector_address() { + for wildcard in ["0.0.0.0", "::", "::ffff:0.0.0.0"] { + let mut nodes = vec![node("n1", 0), node("n2", 1)]; + nodes[0].advertised_addresses = vec![selector("10.0.0.0/16", wildcard)]; + assert!(cfg(nodes).validate().is_err(), "{wildcard} is not dialable"); + } + } + #[test] fn validate_rejects_out_of_range_replica_id() { // 2 nodes total, so id 2 is out of range. @@ -2193,19 +2295,6 @@ mod cluster_validate_tests { assert!(cfg(vec![n1, n2]).validate().is_err()); } - #[test] - fn validate_rejects_node_ip_hostname_clashing_with_advertised_hostname() { - let mut n1 = node("n1", 0); - n1.ip = "10.0.0.1".to_owned(); - n1.advertised_address = Some("broker.example.com".to_owned()); - n1.ports.tcp = Some(8090); - let mut n2 = node("n2", 1); - n2.ip = "broker.example.com".to_owned(); - n2.ports.tcp = Some(8090); - - assert!(cfg(vec![n1, n2]).validate().is_err()); - } - #[test] fn validate_accepts_distinct_hostname_advertised_endpoints() { let mut n1 = node("n1", 0); diff --git a/core/configs/src/server_config/defaults.rs b/core/configs/src/server_config/defaults.rs index f752d32659..1569cfadad 100644 --- a/core/configs/src/server_config/defaults.rs +++ b/core/configs/src/server_config/defaults.rs @@ -29,6 +29,7 @@ use super::cluster::{ }; use super::message_bus::MessageBusConfig; use super::metadata::MetadataConfig; +use super::node::NodeConfig; use super::partition::PartitionConfig; use super::quic::{QuicCertificateConfig, QuicConfig}; use super::server::ServerConfig; @@ -52,6 +53,7 @@ impl Default for ServerConfig { consumer_group: ConsumerGroupConfig::default(), data_maintenance: DataMaintenanceConfig::default(), heartbeat: HeartbeatConfig::default(), + node: NodeConfig::default(), personal_access_token: PersonalAccessTokenConfig::default(), system: Arc::new(ServerSystemConfig::default()), quic: QuicConfig::default(), diff --git a/core/configs/src/server_config/mod.rs b/core/configs/src/server_config/mod.rs index 718aee5cdd..c2ff98a353 100644 --- a/core/configs/src/server_config/mod.rs +++ b/core/configs/src/server_config/mod.rs @@ -26,6 +26,7 @@ pub mod defaults; pub mod displays; pub mod message_bus; pub mod metadata; +pub mod node; pub mod partition; pub mod quic; pub mod server; diff --git a/core/configs/src/server_config/node.rs b/core/configs/src/server_config/node.rs new file mode 100644 index 0000000000..61d9f88950 --- /dev/null +++ b/core/configs/src/server_config/node.rs @@ -0,0 +1,89 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +// This node's own client-facing identity, for the cluster-disabled server. + +use super::COMPONENT; +use super::cluster::AdvertisedAddress; +use crate::ConfigurationError; +use configs::ConfigEnv; +use iggy_common::Validatable; +use serde::{Deserialize, Serialize}; + +/// Named to match its roster counterpart: `advertised_address` here and +/// `cluster.nodes[*].advertised_address` there are the same setting for the +/// same question, and an operator moving between the two modes should not have +/// to learn a second spelling. +#[derive(Debug, Default, Deserialize, Serialize, Clone, ConfigEnv)] +#[serde(deny_unknown_fields)] +pub struct NodeConfig { + /// Client-facing address: a literal IP or a DNS hostname. `None` leaves + /// the server deriving one from its bind address. + #[serde(default)] + pub advertised_address: Option, +} + +impl Validatable for NodeConfig { + fn validate(&self) -> Result<(), ConfigurationError> { + let Some(address) = self.advertised_address.as_deref() else { + return Ok(()); + }; + + address.parse::().map_err(|error| { + eprintln!("{COMPONENT} - node.advertised_address '{address}': {error}"); + ConfigurationError::InvalidConfigurationValue + })?; + + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn advertised(address: &str) -> NodeConfig { + NodeConfig { + advertised_address: Some(address.to_owned()), + } + } + + #[test] + fn validate_accepts_an_unset_address() { + assert!(NodeConfig::default().validate().is_ok()); + } + + #[test] + fn validate_accepts_a_routable_address() { + assert!(advertised("203.0.113.10").validate().is_ok()); + assert!(advertised("broker-1.example.com").validate().is_ok()); + assert!(advertised("2001:db8::1").validate().is_ok()); + } + + #[test] + fn validate_rejects_an_unspecified_address() { + assert!(advertised("0.0.0.0").validate().is_err()); + assert!(advertised("::").validate().is_err()); + } + + #[test] + fn validate_rejects_an_unparsable_address() { + assert!(advertised("broker-1.example.com:8090").validate().is_err()); + assert!(advertised("10.0.0.256").validate().is_err()); + assert!(advertised("").validate().is_err()); + } +} diff --git a/core/configs/src/server_config/server.rs b/core/configs/src/server_config/server.rs index e4c6910d01..bd79fdc7f3 100644 --- a/core/configs/src/server_config/server.rs +++ b/core/configs/src/server_config/server.rs @@ -19,6 +19,7 @@ use super::COMPONENT; use super::cluster::ClusterConfig; use super::message_bus::MessageBusConfig; use super::metadata::MetadataConfig; +use super::node::NodeConfig; use super::partition::PartitionConfig; use super::quic::QuicConfig; use super::tcp::TcpConfig; @@ -111,6 +112,8 @@ pub struct ServerConfig { pub consumer_group: ConsumerGroupConfig, pub data_maintenance: DataMaintenanceConfig, #[serde(default)] + pub node: NodeConfig, + #[serde(default)] pub personal_access_token: PersonalAccessTokenConfig, pub heartbeat: HeartbeatConfig, pub system: Arc, @@ -125,7 +128,58 @@ pub struct ServerConfig { pub message_bus: MessageBusConfig, } +/// One client-facing listener, as the client-facing address derivation and +/// boot validation see it: the config key naming its bind address, that +/// address as written, and whether the listener is switched on. +pub struct ClientListener<'a> { + pub key: &'static str, + pub address: &'a str, + pub enabled: bool, +} + impl ServerConfig { + /// The client-facing listeners, in the order the derived client-facing + /// address prefers them. TCP leads: it is the binary protocol every SDK + /// speaks, so it is the listener an address derived for clients should + /// describe whenever it is running. + #[must_use] + pub fn client_listeners(&self) -> [ClientListener<'_>; 4] { + [ + ClientListener { + key: "tcp.address", + address: &self.tcp.address, + enabled: self.tcp.enabled, + }, + ClientListener { + key: "websocket.address", + address: &self.websocket.address, + enabled: self.websocket.enabled, + }, + ClientListener { + key: "quic.address", + address: &self.quic.address, + enabled: self.quic.enabled, + }, + ClientListener { + key: "http.address", + address: &self.http.address, + enabled: self.http.enabled, + }, + ] + } + + /// The listener whose bind address cluster metadata derives this node's + /// client-facing address from when `node.advertised_address` is unset: + /// the first enabled one. `None` when every client-facing listener is + /// off, which leaves no address for a client to dial and nothing to + /// publish. + #[must_use] + pub fn derived_address_listener(&self) -> Option> { + self.client_listeners() + .into_iter() + .find(|listener| listener.enabled) + } + /// Load server configuration from file and environment variables. /// /// The path comes from `IGGY_CONFIG_PATH` or defaults to diff --git a/core/configs/src/server_config/validators.rs b/core/configs/src/server_config/validators.rs index 3fb3afcf09..a20a48b567 100644 --- a/core/configs/src/server_config/validators.rs +++ b/core/configs/src/server_config/validators.rs @@ -31,6 +31,7 @@ use crate::common::http::HMAC_JWT_ALGORITHMS; use crate::common::validators::SEGMENT_MAX_SIZE_BYTES; use err_trail::ErrContext; use iggy_common::{IggyExpiry, MAX_MESSAGE_SIZE_UPPER_BYTES, Validatable}; +use std::net::SocketAddr; /// compio-ws (tungstenite 0.29) `write_buffer_size` default. Used to /// evaluate the `max_write_buffer_size > write_buffer_size` invariant @@ -76,6 +77,11 @@ impl Validatable for ServerConfig { self.cluster.validate().error(|e: &ConfigurationError| { format!("{COMPONENT} (error: {e}) - failed to validate cluster config") })?; + self.node.validate().error(|e: &ConfigurationError| { + format!("{COMPONENT} (error: {e}) - failed to validate node config") + })?; + self.validate_tcp_bind_address()?; + self.validate_client_facing_address()?; self.metadata.validate().error(|e: &ConfigurationError| { format!("{COMPONENT} (error: {e}) - failed to validate metadata config") })?; @@ -424,6 +430,54 @@ fn reject_unsupported(config: &ServerConfig) -> Result<(), ConfigurationError> { Ok(()) } +impl ServerConfig { + fn validate_tcp_bind_address(&self) -> Result<(), ConfigurationError> { + parse_bind_address("tcp.address", &self.tcp.address)?; + Ok(()) + } + + /// The listener the client-facing address is derived from must not bind a + /// wildcard unless that address is declared outright. + fn validate_client_facing_address(&self) -> Result<(), ConfigurationError> { + if self.cluster.enabled || self.node.advertised_address.is_some() { + return Ok(()); + } + // No client-facing listener runs, so no client dials this node and + // there is no address to demand. + let Some(listener) = self.derived_address_listener() else { + return Ok(()); + }; + let bind = parse_bind_address(listener.key, listener.address)?; + if !bind.ip().to_canonical().is_unspecified() { + return Ok(()); + } + + eprintln!( + "{COMPONENT} - {} binds the wildcard {bind}, which says which interfaces this node \ + accepts on rather than where a client reaches it, so cluster metadata would carry no \ + address for this node. Set node.advertised_address to the address clients dial, or \ + bind a concrete address.", + listener.key + ); + Err(ConfigurationError::InvalidConfigurationValue) + } +} + +/// A listener's bind address, which is a literal IP and a port and nothing +/// else. `context` names the config key so the operator reads back the one +/// they wrote. +fn parse_bind_address(context: &str, address: &str) -> Result { + address.parse::().map_err(|error| { + eprintln!( + "{COMPONENT} - {context} '{address}' is not an address and port: {error}. The host \ + is required and must be a literal IP, so ':PORT' and 'hostname:PORT' are both \ + rejected; use 127.0.0.1:PORT for loopback or 0.0.0.0:PORT to accept on every \ + interface." + ); + ConfigurationError::InvalidConfigurationValue + }) +} + #[cfg(test)] mod tests { use super::super::cluster::{ClusterNodeConfig, TransportPorts}; @@ -470,6 +524,85 @@ mod tests { ); } + #[test] + fn given_wildcard_bind_without_advertised_address_when_validating_should_reject() { + for wildcard in ["0.0.0.0:8090", "[::]:8090", "[::ffff:0.0.0.0]:8090"] { + let config = config_with_override(&format!( + "[tcp]\naddress = \"{wildcard}\"\n[cluster]\nenabled = false\n" + )); + assert!( + config.validate().is_err(), + "{wildcard} names no address a client can dial" + ); + } + } + + #[test] + fn given_a_hostless_or_named_bind_address_when_validating_should_reject() { + for address in [":8090", "localhost:8090", "0.0.0.0", "not-an-address"] { + let config = config_with_override(&format!("[tcp]\naddress = \"{address}\"\n")); + assert!( + config.validate().is_err(), + "{address} does not name a bind address" + ); + } + } + + #[test] + fn given_wildcard_bind_with_advertised_address_when_validating_should_pass() { + let config = config_with_override( + "[tcp]\naddress = \"0.0.0.0:8090\"\n[cluster]\nenabled = false\n\ + [node]\nadvertised_address = \"broker-1.example.com\"\n", + ); + assert!(config.validate().is_ok()); + } + + #[test] + fn given_concrete_bind_without_advertised_address_when_validating_should_pass() { + let config = config_with_override( + "[tcp]\naddress = \"192.0.2.10:8090\"\n[cluster]\nenabled = false\n", + ); + assert!(config.validate().is_ok()); + } + + #[test] + fn given_wildcard_bind_on_a_disabled_listener_when_validating_should_pass() { + let config = config_with_override( + "[tcp]\nenabled = false\naddress = \"0.0.0.0:8090\"\n[cluster]\nenabled = false\n", + ); + assert!(config.validate().is_ok()); + } + + #[test] + fn given_wildcard_bind_on_the_first_enabled_listener_when_validating_should_reject() { + let config = config_with_override( + "[tcp]\nenabled = false\n[websocket]\nenabled = false\n[quic]\nenabled = false\n\ + [http]\naddress = \"0.0.0.0:3000\"\n[cluster]\nenabled = false\n", + ); + assert!( + config.validate().is_err(), + "an http-only server derives its address from http.address" + ); + } + + #[test] + fn given_every_client_listener_disabled_when_validating_should_pass() { + let config = config_with_override( + "[tcp]\nenabled = false\naddress = \"0.0.0.0:8090\"\n[websocket]\nenabled = false\n\ + [quic]\nenabled = false\n[http]\nenabled = false\n[cluster]\nenabled = false\n", + ); + assert!(config.validate().is_ok()); + } + + #[test] + fn given_clustered_wildcard_bind_without_advertised_address_when_validating_should_pass() { + // The roster answers the client-facing address per node, so the bind + // address is free to be a wildcard with nothing declared here. + let config = + config_with_override("[tcp]\naddress = \"0.0.0.0:8090\"\n[cluster]\nenabled = true\n"); + assert!(config.validate().is_ok()); + } + #[test] fn given_shipped_default_config_when_validating_should_pass() { let config: ServerConfig = Figment::new() diff --git a/core/integration/tests/server/cluster_metadata_advertised.rs b/core/integration/tests/server/cluster_metadata_advertised.rs new file mode 100644 index 0000000000..9ce7f78746 --- /dev/null +++ b/core/integration/tests/server/cluster_metadata_advertised.rs @@ -0,0 +1,80 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! `node.advertised_address` on a cluster-disabled server: what a client is +//! told about the one node in the roster. +//! +//! Without a roster the server has only its own bind address to reason from, +//! and a bind address answers which interfaces it accepts on - never where a +//! client reaches it. Behind any NAT (published container ports, a Service, a +//! load balancer) the two are different addresses, and this setting is the +//! only way to state the second one. The bind-derived fallback is pinned at +//! unit level (`cluster_meta.rs`); what needs a real server is that a declared +//! address survives config load and reaches the wire. + +use iggy::prelude::*; +use integration::iggy_harness; + +const ADVERTISED_ADDRESS: &str = "broker-1.example.com"; + +// The harness runs a VSR cluster by default, where the roster answers the +// client-facing address per node and this setting is deliberately ignored. One +// node with clustering off is the shape that reaches the self-synthesized +// path: `--replica-id 0` stays valid without a cluster, any higher id does not. +#[iggy_harness( + cluster_nodes = 1, + server(cluster.enabled = false, node.advertised_address = "broker-1.example.com") +)] +async fn given_a_declared_advertised_address_when_getting_cluster_metadata_should_publish_it( + harness: &TestHarness, +) { + let client = harness + .node(0) + .tcp_client() + .expect("tcp client") + .with_root_login() + .connect() + .await + .expect("connect"); + + let metadata = client + .get_cluster_metadata() + .await + .expect("get cluster metadata"); + + assert_eq!( + metadata.nodes.len(), + 1, + "a cluster-disabled server reports itself alone, got {metadata}" + ); + assert_eq!( + metadata.name, "single-node", + "a cluster-disabled server reports the single-node label, got {metadata}" + ); + let node = &metadata.nodes[0]; + // The harness binds a concrete loopback address, so this also pins the + // precedence: a declaration outranks an address the bind could vouch for, + // because only the declaration is a claim about reachability. + assert_eq!( + node.ip, ADVERTISED_ADDRESS, + "the declared address must reach the wire verbatim" + ); + assert_ne!( + node.endpoints.tcp, 0, + "the self node reports its real tcp port alongside the declared address" + ); +} diff --git a/core/integration/tests/server/mod.rs b/core/integration/tests/server/mod.rs index c880723c52..b8c832e474 100644 --- a/core/integration/tests/server/mod.rs +++ b/core/integration/tests/server/mod.rs @@ -50,6 +50,9 @@ mod http_rbac; mod http_tls; // Binary GetClusterMetadata must serve the real roster from a VSR cluster. mod cluster_metadata_vsr; +// A declared node.advertised_address outranks the bind address a +// cluster-disabled server would otherwise publish. +mod cluster_metadata_advertised; // A metadata view change must persist the advanced view and recover it from disk // across a replica restart. mod cluster_view_durability_vsr; diff --git a/core/server/README.md b/core/server/README.md index f0150b32eb..1c4fc81c4b 100644 --- a/core/server/README.md +++ b/core/server/README.md @@ -12,6 +12,15 @@ cargo run --bin iggy-server --release The Docker image `apache/iggy:latest` ships the server together with the CLI; the `edge` tag tracks the latest development build. +The image binds every listener to `0.0.0.0` so the container is reachable from outside it. A wildcard bind says which interfaces accept connections, not where a client reaches the server, so the address to publish in cluster metadata has to be supplied and the server refuses to start without it. On a single host with published ports that address is `localhost`: + +```sh +docker run -p 3000:3000 -p 8090:8090 \ + -e IGGY_NODE_ADVERTISED_ADDRESS=localhost apache/iggy:latest +``` + +Use the hostname or load balancer address clients actually dial when they are not on the same host. The Helm chart derives it from the Service DNS name. + To run one node of a cluster, pass its replica ID from the `cluster.nodes` roster: ```sh @@ -27,7 +36,7 @@ Settings are read from [config.toml](config.toml), resolved relative to the work Any single value can be overridden with an `IGGY_`-prefixed environment variable that mirrors the TOML path: ```sh -IGGY_TCP_ADDRESS=0.0.0.0:8090 IGGY_HTTP_ENABLED=false cargo run --bin iggy-server +IGGY_TCP_ADDRESS=127.0.0.1:8090 IGGY_HTTP_ENABLED=false cargo run --bin iggy-server ``` Cluster membership, quorum and replica addressing live under `[cluster]`. diff --git a/core/server/config.toml b/core/server/config.toml index 583df57320..e70cfacee3 100644 --- a/core/server/config.toml +++ b/core/server/config.toml @@ -28,6 +28,16 @@ cleaner_enabled = true # Interval for running the message cleaner. interval = "1 m" +# This node's own client-facing identity, used when 'cluster.enabled' is false. +[node] +# A literal IP or a DNS hostname, without a port. Named to match its roster +# counterpart 'cluster.nodes.advertised_address', which answers the same +# question per node. The unspecified address ("0.0.0.0", "::") is rejected: it +# is the bind address's answer to a question clients are not asking. Commented +# out here because the shipped 'tcp.address' binds a concrete address and needs +# no declaration. +#advertised_address = "broker-1.example.com" + # HTTP server configuration [http] # Determines if the HTTP server is active. @@ -712,9 +722,9 @@ key_file = "" # PEM trust anchor(s) the dialer verifies peer certificates against. # Unused when self_signed = true. # The dialer verifies each peer against the 'ip' string from that peer's -# cluster.nodes entry, so a roster of literal IPs needs peer certificates -# carrying matching IP SANs. Hostname-only certificates fail verification; -# use hostnames in the roster if the certificates only have DNS SANs. +# cluster.nodes entry, and that string is the TLS server name as well as a +# literal IP, so every peer certificate needs a matching IP SAN. A +# certificate carrying only DNS SANs fails verification. ca_file = "" # Shard-0 coordinator placement. @@ -788,7 +798,7 @@ skip_shard_zero_for_clients = false # # [[cluster.nodes]] # name = "iggy-node-1" -# ip = "10.0.1.5" # replica plane + last-resort fallback +# ip = "10.0.1.5" # replica plane, literal IP only # advertised_address = "203.0.113.10" # catch-all for unmatched clients # replica_id = 0 # ports = { tcp = 8090, http = 3000, tcp_replica = 9090 } diff --git a/core/server/src/args.rs b/core/server/src/args.rs index 94e0bd7079..b03789640e 100644 --- a/core/server/src/args.rs +++ b/core/server/src/args.rs @@ -54,8 +54,10 @@ ENVIRONMENT VARIABLES: Common examples: IGGY_SYSTEM_PATH=/data/iggy # Data directory - IGGY_TCP_ADDRESS=0.0.0.0:8090 # TCP listener address + IGGY_TCP_ADDRESS=127.0.0.1:8090 # TCP listener address IGGY_HTTP_ADDRESS=0.0.0.0:3000 # HTTP listener address + IGGY_NODE_ADVERTISED_ADDRESS=localhost # Address clients dial, required + # when a listener binds a wildcard IGGY_SYSTEM_LOGGING_LEVEL=debug # Log level IGGY_ROOT_USERNAME=iggy # Root user, set with the password IGGY_ROOT_PASSWORD=secret # Root password, set with the username diff --git a/core/server/src/bootstrap.rs b/core/server/src/bootstrap.rs index a5840277ef..3c15d647b1 100644 --- a/core/server/src/bootstrap.rs +++ b/core/server/src/bootstrap.rs @@ -16,7 +16,7 @@ // under the License. use crate::auth::warm_dummy_password_hash; -use crate::cluster_meta::ClusterRoster; +use crate::cluster_meta::{ClusterRoster, resolved_roster_nodes, self_advertised_address}; use crate::config_writer::write_current_config; use crate::dispatch::{ make_client_request_handler, make_deferred_client_request_handler, @@ -1786,32 +1786,44 @@ fn spawn_shutdown_watchdog( /// the shared [`ClusterRoster`] so the binary `GetClusterMetadata` read serves /// the real topology. `self_*` back only the cluster-disabled self-synthesis /// and carry the requested listener ports from the resolved topology, not the -/// bound ones (a `:0` wildcard is reported as 0). +/// bound ones (a `:0` wildcard is reported as 0). The self address resolves +/// through [`self_advertised_address`], which boot validation has already +/// guaranteed names somewhere a client can dial. fn build_cluster_roster( + shard_id: u16, config: &ServerConfig, topology: &TcpTopology, metadata_view: Arc, -) -> ClusterRoster { - ClusterRoster { +) -> Result { + let declared = config.node.advertised_address.as_deref(); + let self_advertised = self_advertised_address(declared, derived_bind_ip(topology, config)); + // The roster answers this per node, so a value here would be read by + // nobody. Silence would leave the operator believing it took effect. + // Every shard builds its own roster off the same config, so keep the + // operator-facing explanation to one line per process. + if declared.is_some() && config.cluster.enabled && shard_id == 0 { + warn!( + "node.advertised_address is set but cluster.enabled is true, so it is ignored; \ + the client-facing address of each node comes from its cluster.nodes entry" + ); + } + Ok(ClusterRoster { enabled: config.cluster.enabled, name: config.cluster.name.clone(), - nodes: config - .cluster - .nodes - .iter() - .cloned() - .map(Into::into) - .collect(), - self_ip: topology.client_listen_addr.ip().to_string(), + nodes: resolved_roster_nodes(&config.cluster).map_err(ServerError::Config)?, + self_advertised, self_ports: configs::cluster::TransportPorts { - tcp: Some(topology.client_listen_addr.port()), + tcp: config + .tcp + .enabled + .then(|| topology.client_listen_addr.port()), quic: topology.quic_listen_addr.map(|addr| addr.port()), http: topology.http_listen_addr.map(|addr| addr.port()), websocket: topology.ws_listen_addr.map(|addr| addr.port()), tcp_replica: None, }, metadata_view, - } + }) } #[allow(clippy::too_many_arguments, clippy::too_many_lines)] @@ -2134,10 +2146,11 @@ async fn build_shard_for_thread( sessions .borrow_mut() .set_cluster_roster(Rc::new(build_cluster_roster( + shard_id, config, topology, metadata_view, - ))); + )?)); let shard_name = format!("server-shard-{shard_id}"); let built = IggyShardBuilder::new( ShardIdentity::new(shard_id, shard_name), @@ -3124,15 +3137,75 @@ fn merge_roster_port_with_bind_ip( listen_addr } +/// The client-facing listeners paired with the config key naming their bind +/// address, in the order [`ServerConfig::client_listeners`] derives the +/// published client-facing address from them. `None` marks a listener that is +/// switched off and therefore binds nothing. +fn client_listeners( + topology: &TcpTopology, + config: &ServerConfig, +) -> [(&'static str, Option); 4] { + [ + ( + "tcp.address", + config.tcp.enabled.then_some(topology.client_listen_addr), + ), + ("websocket.address", topology.ws_listen_addr), + ("quic.address", topology.quic_listen_addr), + ("http.address", topology.http_listen_addr), + ] +} + +/// The bind interface the published client-facing address names when none is +/// declared: the first enabled listener's, the same one boot validation gated +/// its wildcard refusal on. With every client listener off nothing dials this +/// node, so the tcp bind address stands in for an answer no client reads. +fn derived_bind_ip(topology: &TcpTopology, config: &ServerConfig) -> IpAddr { + client_listeners(topology, config) + .into_iter() + .find_map(|(_, listen_addr)| listen_addr) + .unwrap_or(topology.client_listen_addr) + .ip() +} + +/// Whether the address cluster metadata publishes for this node misses one of +/// its own listeners. Only a derived address is judged: it names one +/// listener's bind interface, so a listener on a different one is unreachable +/// at the published address. A declared `node.advertised_address` is +/// deliberate (NAT, a public name) and says nothing about which local +/// interface serves a transport, so it stays quiet. +fn derived_address_misses_listener( + declared: Option<&str>, + self_advertised: &str, + listen_addr: SocketAddr, +) -> bool { + declared.is_none() && roster_ip_unreachable_from_bind_addr(self_advertised, listen_addr) +} + /// Whether a dialer aiming at the advertised roster ip misses `listen_addr`. An /// unspecified bind covers every interface, and a roster ip that parses as /// neither IPv4 nor IPv6 (a DNS name, say) can resolve to the bound interface, -/// so both cases stay quiet. +/// so both cases stay quiet. Both sides reduce to the canonical form first, so +/// the v4-mapped wildcard (`[::ffff:0.0.0.0]`, which a dual-stack host binds as +/// `0.0.0.0`) stays quiet as well and `10.0.0.5` matches `::ffff:10.0.0.5`. fn roster_ip_unreachable_from_bind_addr(roster_ip: &str, listen_addr: SocketAddr) -> bool { - !listen_addr.ip().is_unspecified() + let bind_ip = listen_addr.ip().to_canonical(); + !bind_ip.is_unspecified() && roster_ip .parse::() - .is_ok_and(|parsed| parsed != listen_addr.ip()) + .is_ok_and(|parsed| parsed.to_canonical() != bind_ip) +} + +fn wildcard_listener_under_loopback_address( + declared: Option<&str>, + self_advertised: &str, + listen_addr: SocketAddr, +) -> bool { + declared.is_none() + && listen_addr.ip().to_canonical().is_unspecified() + && self_advertised + .parse::() + .is_ok_and(|address| address.to_canonical().is_loopback()) } fn resolve_cluster_replica_peers( @@ -3190,6 +3263,44 @@ async fn start_tcp_runtime( .await?; } + // Cluster metadata carries one host for all four transports, so a listener + // the derived host does not reach is unreachable at it. Only a derived + // address is judged, and never against the listener it was derived from. + let declared = config.node.advertised_address.as_deref(); + let self_advertised = self_advertised_address(declared, derived_bind_ip(topology, config)); + // A roster entry answers this per node in cluster mode, so the derived + // address is never served and none of these listeners are judged against it. + let listeners = client_listeners(topology, config); + if !config.cluster.enabled + && let Some(derived_from) = listeners + .iter() + .find_map(|(key, listen_addr)| listen_addr.map(|_| *key)) + { + for (key, listen_addr) in listeners { + let Some(listen_addr) = listen_addr.filter(|_| key != derived_from) else { + continue; + }; + if derived_address_misses_listener(declared, &self_advertised, listen_addr) { + warn!( + "{key} binds {listen_addr} but cluster metadata publishes {self_advertised}, \ + derived from {derived_from}; a client reading that metadata would not reach \ + this listener. Set node.advertised_address to the address clients dial." + ); + } else if wildcard_listener_under_loopback_address( + declared, + &self_advertised, + listen_addr, + ) { + warn!( + "{key} binds the wildcard {listen_addr} but cluster metadata publishes the \ + loopback {self_advertised}, derived from {derived_from}; a client reaching \ + this listener from another host is told an address that points back at \ + itself. Set node.advertised_address to the address clients dial." + ); + } + } + } + // HTTP is served over TCP but sits outside the replica_io / manual client // reactor, so it binds independently. Shard-0 gating comes from the sole // caller of this function. @@ -3211,6 +3322,7 @@ async fn start_tcp_runtime( config.personal_access_token.max_tokens_per_user, &config.cluster, Arc::clone(&config.system), + &self_advertised, self_ports, shard_metrics_all, ) @@ -5053,4 +5165,137 @@ mod tests { addr("10.0.0.5:18070") )); } + + #[test] + fn derived_address_warns_only_when_it_misses_a_listener() { + // Derived from a loopback tcp.address while another transport serves + // an external interface: metadata would publish an address no client + // reaches. Every non-TCP listener carries the same exposure, since one + // host is published for all four. + for listener in ["10.0.0.5:3000", "10.0.0.5:8080", "10.0.0.5:8092"] { + assert!( + derived_address_misses_listener(None, "127.0.0.1", addr(listener)), + "{listener} is not reachable at 127.0.0.1" + ); + } + // Same interface, and a wildcard bind that covers any of them. + assert!(!derived_address_misses_listener( + None, + "10.0.0.5", + addr("10.0.0.5:3000") + )); + assert!(!derived_address_misses_listener( + None, + "127.0.0.1", + addr("0.0.0.0:3000") + )); + // A declared address is deliberate and unrelated to local interfaces. + assert!(!derived_address_misses_listener( + Some("broker-1.example.com"), + "broker-1.example.com", + addr("10.0.0.5:3000") + )); + } + + #[test] + fn wildcard_listener_warns_only_under_a_derived_loopback_address() { + // Metadata says 127.0.0.1 while this listener takes connections from + // anywhere: whoever arrives from another host is told to dial itself. + assert!(wildcard_listener_under_loopback_address( + None, + "127.0.0.1", + addr("0.0.0.0:3000") + )); + assert!(wildcard_listener_under_loopback_address( + None, + "127.0.0.1", + addr("[::]:3000") + )); + // A published address that is reachable from elsewhere is what the + // wildcard listener wants, so there is nothing to say. + assert!(!wildcard_listener_under_loopback_address( + None, + "10.0.0.5", + addr("0.0.0.0:3000") + )); + // A concrete bind is the other warning's business, not this one's. + assert!(!wildcard_listener_under_loopback_address( + None, + "127.0.0.1", + addr("10.0.0.5:3000") + )); + // A declared address is deliberate; a loopback one is a local setup. + assert!(!wildcard_listener_under_loopback_address( + Some("127.0.0.1"), + "127.0.0.1", + addr("0.0.0.0:3000") + )); + } + + #[test] + fn derived_bind_ip_follows_the_first_enabled_listener() { + let mut config: ServerConfig = + toml::from_str(include_str!("../config.toml")).expect("shipped config deserializes"); + let topology = |ws: Option<&str>, http: Option<&str>| TcpTopology { + cluster_id: 0, + self_replica_id: 0, + replica_count: 1, + client_listen_addr: addr("127.0.0.1:8090"), + replica_listen_addr: None, + ws_listen_addr: ws.map(addr), + quic_listen_addr: None, + http_listen_addr: http.map(addr), + tcp_tls_listen_addr: None, + peers: Vec::new(), + }; + let expected = |ip: &str| ip.parse::().unwrap(); + + assert_eq!( + derived_bind_ip( + &topology(Some("10.0.0.5:8092"), Some("10.0.0.6:3000")), + &config + ), + expected("127.0.0.1") + ); + + config.tcp.enabled = false; + assert_eq!( + derived_bind_ip( + &topology(Some("10.0.0.5:8092"), Some("10.0.0.6:3000")), + &config + ), + expected("10.0.0.5"), + "websocket is next in line once tcp is off" + ); + assert_eq!( + derived_bind_ip(&topology(None, Some("10.0.0.6:3000")), &config), + expected("10.0.0.6"), + "with websocket and quic off too, http answers" + ); + assert_eq!( + derived_bind_ip(&topology(None, None), &config), + expected("127.0.0.1"), + "with every client listener off the value reaches no client anyway" + ); + } + + #[test] + fn roster_mismatch_warning_is_silent_for_v4_mapped_binds() { + // `[::ffff:0.0.0.0]` is the v4 wildcard and `::ffff:10.0.0.5` is + // `10.0.0.5`, so neither reaches the dialer any differently than the + // plain spelling the case above covers. + assert!(!roster_ip_unreachable_from_bind_addr( + "10.0.0.5", + addr("[::ffff:0.0.0.0]:18070") + )); + assert!(!roster_ip_unreachable_from_bind_addr( + "10.0.0.5", + addr("[::ffff:10.0.0.5]:18070") + )); + // A genuine mismatch still warns through the mapped spelling. + assert!(roster_ip_unreachable_from_bind_addr( + "10.0.0.5", + addr("[::ffff:127.0.0.1]:18070") + )); + } } diff --git a/core/server/src/cluster_meta.rs b/core/server/src/cluster_meta.rs index c251ef6028..6fd0624534 100644 --- a/core/server/src/cluster_meta.rs +++ b/core/server/src/cluster_meta.rs @@ -28,7 +28,8 @@ //! leader, but the full roster is still returned). The self-synthesized single //! node is the cluster-disabled fallback, shared by both callers. -use configs::cluster::{ResolvedClusterNode, TransportPorts}; +use configs::ConfigurationError; +use configs::cluster::{AdvertisedAddress, ClusterConfig, ResolvedClusterNode, TransportPorts}; use iggy_common::{ ClusterMetadata, ClusterNode, ClusterNodeRole, ClusterNodeStatus, TransportEndpoints, }; @@ -43,6 +44,52 @@ const SELF_NODE_NAME: &str = "iggy-node"; /// single-node label. const SINGLE_NODE_CLUSTER_NAME: &str = "single-node"; +/// Client-facing host for the cluster-disabled single node, normalized the way +/// [`client_host`] normalizes the roster path so one address cannot publish in +/// two spellings. Normalizing matters beyond tidiness: an SDK that brackets an +/// IPv6 host before joining it to the port brackets a declared `[2001:db8::1]` +/// a second time unless it guards on a leading '[', yielding an address that no +/// longer parses. Go's `net.JoinHostPort` keys on the colon alone, so it is the +/// one that does. +/// +/// `NodeConfig::validate` has already accepted `declared`, so the parse only +/// fails on a caller that skipped validation; such a value is passed through +/// rather than dropped. +pub fn self_advertised_address(declared: Option<&str>, bind: IpAddr) -> String { + declared.map_or_else( + || bind.to_string(), + |declared| { + declared + .parse::() + .map_or_else(|_| declared.to_owned(), |address| address.to_string()) + }, + ) +} + +/// Resolve the roster a [`ClusterRoster`] serves, which is only the configured +/// one while the cluster is enabled. Gated on the same flag the config +/// validator gates itself on: a disabled cluster leaves `cluster.nodes` +/// unvalidated, and a stale entry left there must not fail a boot that never +/// consults it. +/// +/// # Errors +/// +/// Returns [`ConfigurationError`] when an enabled roster carries a node whose +/// address does not parse, which boot validation rejects first. +pub fn resolved_roster_nodes( + cluster: &ClusterConfig, +) -> Result, ConfigurationError> { + if !cluster.enabled { + return Ok(Vec::new()); + } + cluster + .nodes + .iter() + .cloned() + .map(ResolvedClusterNode::try_from) + .collect() +} + /// Config-derived cluster topology reported by cluster-metadata reads. /// /// Copied out of `ClusterConfig` at listener/shard start so both handlers stay @@ -54,8 +101,9 @@ pub struct ClusterRoster { /// Roster nodes with selectors parsed once at roster build, so the /// per-request address resolution never re-parses config strings. pub nodes: Vec, - /// This node's own address, reported for the synthesized self node. - pub self_ip: String, + /// This node's own client-facing address, reported for the synthesized + /// self node (see [`self_advertised_address`]). + pub self_advertised: String, /// This node's own client ports for the same self node (`None` = transport /// disabled). pub self_ports: TransportPorts, @@ -69,15 +117,16 @@ pub struct ClusterRoster { pub const METADATA_VIEW_UNKNOWN: u64 = u64::MAX; impl ClusterRoster { - /// A cluster-disabled roster with no self address. Used as the pre-bootstrap - /// default before the real roster is installed; [`Self::cluster_metadata`] - /// on it synthesizes a bare single node. + /// A cluster-disabled roster with no self address. The pre-bootstrap + /// placeholder a [`crate::session_manager::SessionManager`] holds until + /// bootstrap installs the real roster, which happens before any listener + /// accepts, so its blank address is never served to a client. pub fn disabled() -> Self { Self { enabled: false, name: String::new(), nodes: Vec::new(), - self_ip: String::new(), + self_advertised: String::new(), self_ports: TransportPorts::default(), metadata_view: Arc::new(AtomicU64::new(METADATA_VIEW_UNKNOWN)), } @@ -135,12 +184,14 @@ impl ClusterRoster { } } + /// The cluster-disabled single node, carrying the address + /// [`self_advertised_address`] resolved. fn self_metadata(&self) -> ClusterMetadata { ClusterMetadata { name: SINGLE_NODE_CLUSTER_NAME.to_owned(), nodes: vec![ClusterNode { name: SELF_NODE_NAME.to_owned(), - ip: self.self_ip.clone(), + ip: self.self_advertised.clone(), endpoints: ports_to_endpoints(&self.self_ports), role: ClusterNodeRole::Leader, status: ClusterNodeStatus::Healthy, @@ -153,16 +204,11 @@ impl ClusterRoster { /// matching what boot validation compared and what redirect URLs render, so /// textual config variants of one address publish identical metadata. The /// per-client-network selectors, the catch-all `advertised_address`, and the -/// roster `ip` are consulted in that order ([`ResolvedClusterNode::advertised_for`]). -/// Metadata deliberately does NOT fail closed like the redirect path: a host -/// that parses as neither IP nor hostname (the roster `ip` is only validated -/// non-empty - Docker service names with underscores exist in the wild) -/// publishes verbatim via [`ResolvedClusterNode::raw_advertised_fallback`]. +/// roster `ip` are consulted in that order +/// ([`ResolvedClusterNode::advertised_for`]), which always resolves: a node +/// whose sources do not parse never becomes a [`ResolvedClusterNode`]. fn client_host(node: &ResolvedClusterNode, client_ip: Option) -> String { - node.advertised_for(client_ip).map_or_else( - || node.raw_advertised_fallback().to_owned(), - ToString::to_string, - ) + node.advertised_for(client_ip).to_string() } const fn role_for(primary_index: Option, replica_id: u8) -> ClusterNodeRole { @@ -202,8 +248,8 @@ mod tests { ClusterRoster { enabled: true, name: "test-cluster".to_owned(), - nodes: vec![node.into()], - self_ip: "127.0.0.1".to_owned(), + nodes: vec![ResolvedClusterNode::try_from(node).expect("valid roster node")], + self_advertised: "127.0.0.1".to_owned(), self_ports: TransportPorts::default(), metadata_view: Arc::new(AtomicU64::new(METADATA_VIEW_UNKNOWN)), } @@ -248,16 +294,6 @@ mod tests { } } - #[test] - fn cluster_metadata_passes_unparsable_replica_ip_verbatim() { - let mut node = node_config(None); - node.ip = "iggy_node".to_owned(); - - let metadata = roster_of(node).cluster_metadata(Some(0), None); - - assert_eq!(metadata.nodes[0].ip, "iggy_node"); - } - #[test] fn cluster_metadata_serves_the_selector_address_to_a_matching_client() { let mut node = node_config(Some("203.0.113.10".to_owned())); @@ -293,4 +329,98 @@ mod tests { assert_eq!(metadata.nodes[0].ip, "broker.internal.test"); } + + #[test] + fn resolved_roster_nodes_ignores_a_roster_a_disabled_cluster_never_reads() { + // The shipped config carries a roster with the cluster off, so an + // entry that stopped parsing must not fail a boot that never serves + // it: the validator skips those entries for the same reason. + let mut cluster = ClusterConfig { + enabled: false, + ..ClusterConfig::default() + }; + cluster.nodes[0].ip = "iggy-server".to_owned(); + + assert!( + resolved_roster_nodes(&cluster) + .expect("no roster to resolve") + .is_empty() + ); + } + + #[test] + fn resolved_roster_nodes_refuses_an_enabled_roster_that_does_not_parse() { + let mut cluster = ClusterConfig { + enabled: true, + ..ClusterConfig::default() + }; + cluster.nodes[0].ip = "iggy-server".to_owned(); + + assert!(resolved_roster_nodes(&cluster).is_err()); + } + + #[test] + fn self_advertised_address_prefers_a_declared_address() { + assert_eq!( + self_advertised_address(Some("broker-1.example.com"), "192.0.2.10".parse().unwrap()), + "broker-1.example.com" + ); + } + + #[test] + fn self_advertised_address_falls_back_to_the_bind_address() { + assert_eq!( + self_advertised_address(None, "192.0.2.10".parse().unwrap()), + "192.0.2.10" + ); + assert_eq!( + self_advertised_address(None, "2001:db8::1".parse().unwrap()), + "2001:db8::1" + ); + } + + #[test] + fn self_advertised_address_normalizes_a_declared_address() { + let bind = "192.0.2.10".parse().unwrap(); + // The roster path renders through the same `Display`, so a config + // variant must not publish differently depending on which path served + // it. + assert_eq!( + self_advertised_address(Some("Broker.Example.COM"), bind), + "broker.example.com" + ); + // A client joins the published host to a port; leaving the brackets on + // would bracket it twice into an address that no longer parses. + assert_eq!( + self_advertised_address(Some("[2001:db8::1]"), bind), + "2001:db8::1" + ); + assert_eq!( + self_advertised_address(Some("2001:DB8::1"), bind), + "2001:db8::1" + ); + } + + #[test] + fn self_metadata_synthesizes_a_single_leader_node() { + let roster = ClusterRoster { + enabled: false, + name: String::new(), + nodes: Vec::new(), + self_advertised: "broker-1.example.com".to_owned(), + self_ports: TransportPorts { + tcp: Some(8090), + ..TransportPorts::default() + }, + metadata_view: Arc::new(AtomicU64::new(METADATA_VIEW_UNKNOWN)), + }; + + let metadata = roster.cluster_metadata(None, None); + + assert_eq!(metadata.nodes.len(), 1); + assert_eq!(metadata.nodes[0].ip, "broker-1.example.com"); + assert_eq!(metadata.nodes[0].endpoints.tcp, 8090); + assert_eq!(metadata.nodes[0].role, ClusterNodeRole::Leader); + assert_eq!(metadata.nodes[0].status, ClusterNodeStatus::Healthy); + } } diff --git a/core/server/src/dispatch.rs b/core/server/src/dispatch.rs index 008f6f84cf..5e7bca80e8 100644 --- a/core/server/src/dispatch.rs +++ b/core/server/src/dispatch.rs @@ -5106,8 +5106,13 @@ mod tests { let multi_node = Rc::new(ClusterRoster { enabled: true, name: "test-cluster".to_owned(), - nodes: vec![roster_node("node-0").into(), roster_node("node-1").into()], - self_ip: "127.0.0.1".to_owned(), + nodes: ["node-0", "node-1"] + .map(|name| { + configs::cluster::ResolvedClusterNode::try_from(roster_node(name)) + .expect("valid roster node") + }) + .to_vec(), + self_advertised: "127.0.0.1".to_owned(), self_ports: TransportPorts::default(), metadata_view: Arc::new(std::sync::atomic::AtomicU64::new( crate::cluster_meta::METADATA_VIEW_UNKNOWN, diff --git a/core/server/src/http.rs b/core/server/src/http.rs index b8180f726a..fc72acc4a7 100644 --- a/core/server/src/http.rs +++ b/core/server/src/http.rs @@ -65,7 +65,7 @@ use tower_http::cors::{AllowOrigin, CorsLayer}; use tracing::{error, info, warn}; use crate::bootstrap::ServerShard; -use crate::cluster_meta::ClusterRoster; +use crate::cluster_meta::{ClusterRoster, resolved_roster_nodes}; use crate::http::handlers::{ change_password, create_cg, create_partitions, create_pat, create_stream, create_topic, create_user, delete_cg, delete_consumer_offset, delete_partitions, delete_pat, delete_segments, @@ -103,6 +103,7 @@ pub async fn start( max_tokens_per_user: u32, cluster: &ClusterConfig, system_config: Arc, + self_advertised: &str, self_ports: TransportPorts, shard_metrics_all: &[shard::metrics::ShardMetrics], ) -> Result<(), ServerError> { @@ -149,8 +150,8 @@ pub async fn start( roster: ClusterRoster { enabled: cluster.enabled, name: cluster.name.clone(), - nodes: cluster.nodes.iter().cloned().map(Into::into).collect(), - self_ip: bound_addr.ip().to_string(), + nodes: resolved_roster_nodes(cluster).map_err(ServerError::Config)?, + self_advertised: self_advertised.to_owned(), // The self node reports the live bound HTTP port; the other client // ports arrive resolved from the caller. self_ports: TransportPorts { diff --git a/core/server/src/http/error.rs b/core/server/src/http/error.rs index 45ef05315c..db3edcc5ac 100644 --- a/core/server/src/http/error.rs +++ b/core/server/src/http/error.rs @@ -555,7 +555,7 @@ pub(in crate::http) fn primary_http_socket( primary_index: u8, ) -> Option { let (node, http_port) = primary_node(roster, primary_index)?; - Some(SocketAddr::new(node.replica_ip()?, http_port)) + Some(SocketAddr::new(node.replica_ip(), http_port)) } /// Resolve the client-facing HTTP authority (`host:port`) for a redirect @@ -563,18 +563,16 @@ pub(in crate::http) fn primary_http_socket( /// match first, then the catch-all advertised address, then the private /// roster IP as the compatibility fallback. `AdvertisedAddress::authority` /// brackets IPv6 hosts and passes hostnames through, so the redirect URL -/// stays valid. This is the fail-closed caller: a host that is neither a -/// valid IP nor a valid hostname yields `None` and the redirect becomes a -/// 503 rather than a `Location` pointing at an unparsable target (cluster -/// metadata makes the opposite choice and publishes such a host verbatim). +/// stays valid. `None` here means the roster has no node at `primary_index` +/// or that node declares no HTTP port, never that its address failed to +/// parse: such a node never becomes a [`ResolvedClusterNode`]. fn primary_advertised_http_authority( roster: &ClusterRoster, primary_index: u8, client_ip: Option, ) -> Option { let (node, http_port) = primary_node(roster, primary_index)?; - let address = node.advertised_for(client_ip)?; - Some(address.authority(http_port)) + Some(node.advertised_for(client_ip).authority(http_port)) } fn primary_node(roster: &ClusterRoster, primary_index: u8) -> Option<(&ResolvedClusterNode, u16)> { @@ -614,8 +612,11 @@ mod tests { ClusterRoster { enabled: true, name: "test-cluster".to_owned(), - nodes: nodes.into_iter().map(Into::into).collect(), - self_ip: "127.0.0.1".to_owned(), + nodes: nodes + .into_iter() + .map(|node| ResolvedClusterNode::try_from(node).expect("valid roster node")) + .collect(), + self_advertised: "127.0.0.1".to_owned(), self_ports: TransportPorts::default(), metadata_view: std::sync::Arc::new(std::sync::atomic::AtomicU64::new( crate::cluster_meta::METADATA_VIEW_UNKNOWN, diff --git a/core/server/src/http/forward.rs b/core/server/src/http/forward.rs index 8a153bfbdd..8c1acadcdb 100644 --- a/core/server/src/http/forward.rs +++ b/core/server/src/http/forward.rs @@ -602,8 +602,8 @@ fn primary_socket(state: &HttpInner) -> Option { } /// Private HTTP sockets for every other configured replica, in stable roster -/// order. The caller tries each once. Invalid or HTTP-disabled entries are -/// skipped because they cannot accept the forwarded request. +/// order. The caller tries each once. HTTP-disabled entries are skipped +/// because they cannot accept the forwarded request. fn partition_http_sockets( roster: &crate::cluster_meta::ClusterRoster, self_id: Option, @@ -614,7 +614,7 @@ fn partition_http_sockets( .filter(|node| Some(node.config().replica_id) != self_id) .filter_map(|node| { Some(SocketAddr::new( - node.replica_ip()?, + node.replica_ip(), node.config().ports.http?, )) }) @@ -715,7 +715,7 @@ impl ServerCertVerifier for PinnedCertVerifier { mod tests { use super::*; - use configs::cluster::{ClusterNodeConfig, TransportPorts}; + use configs::cluster::{ClusterNodeConfig, ResolvedClusterNode, TransportPorts}; fn node(replica_id: u8, ip: &str, http: Option) -> ClusterNodeConfig { ClusterNodeConfig { @@ -738,8 +738,11 @@ mod tests { crate::cluster_meta::ClusterRoster { enabled: true, name: "test-cluster".to_owned(), - nodes: nodes.into_iter().map(Into::into).collect(), - self_ip: "127.0.0.1".to_owned(), + nodes: nodes + .into_iter() + .map(|node| ResolvedClusterNode::try_from(node).expect("valid roster node")) + .collect(), + self_advertised: "127.0.0.1".to_owned(), self_ports: TransportPorts::default(), metadata_view: std::sync::Arc::new(std::sync::atomic::AtomicU64::new( crate::cluster_meta::METADATA_VIEW_UNKNOWN, @@ -789,8 +792,7 @@ mod tests { let roster = roster(vec![ node(0, "10.0.0.1", Some(8080)), node(1, "10.0.0.2", Some(8081)), - node(2, "not-an-ip", Some(8082)), - node(3, "10.0.0.4", None), + node(2, "10.0.0.3", None), ]); assert_eq!( diff --git a/docker-compose.yml b/docker-compose.yml index afc9e14e6a..c1409a4652 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -30,6 +30,10 @@ services: memlock: soft: -1 hard: -1 + environment: + # The image binds every listener to 0.0.0.0 (Dockerfile), so the address + # clients dial has to be declared here. + - IGGY_NODE_ADVERTISED_ADDRESS=localhost networks: - iggy ports: diff --git a/examples/csharp/README.md b/examples/csharp/README.md index 995859aff8..53d2afd5fc 100644 --- a/examples/csharp/README.md +++ b/examples/csharp/README.md @@ -16,7 +16,7 @@ You can also customize the server using environment variables: ```bash ## Example: Enable HTTP transport and set custom address -IGGY_HTTP_ENABLED=true IGGY_TCP_ADDRESS=0.0.0.0:8090 cargo run --bin iggy-server +IGGY_HTTP_ENABLED=true IGGY_TCP_ADDRESS=127.0.0.1:8090 cargo run --bin iggy-server ``` ## Basic Examples diff --git a/examples/go/README.md b/examples/go/README.md index d463dee8bc..3e6eed2f33 100644 --- a/examples/go/README.md +++ b/examples/go/README.md @@ -16,7 +16,7 @@ You can also customize the server using environment variables: ```bash ## Example: Enable HTTP transport and set custom address -IGGY_HTTP_ENABLED=true IGGY_TCP_ADDRESS=0.0.0.0:8090 cargo run --bin iggy-server +IGGY_HTTP_ENABLED=true IGGY_TCP_ADDRESS=127.0.0.1:8090 cargo run --bin iggy-server ``` You can run multiple producers and consumers simultaneously to observe how messages are distributed across clients. diff --git a/examples/java/README.md b/examples/java/README.md index 4507d90095..b506823735 100644 --- a/examples/java/README.md +++ b/examples/java/README.md @@ -39,7 +39,7 @@ You can also customize the server using environment variables: ```bash ## Example: set a custom TCP address -IGGY_TCP_ADDRESS=0.0.0.0:8090 cargo run --bin iggy-server +IGGY_TCP_ADDRESS=127.0.0.1:8090 cargo run --bin iggy-server ``` ## Basic Examples diff --git a/examples/node/README.md b/examples/node/README.md index 90447387ae..a49fc3f3f3 100644 --- a/examples/node/README.md +++ b/examples/node/README.md @@ -8,7 +8,8 @@ To run any example, first start the server with ```bash # Using latest release -docker run --rm -p 8080:8080 -p 3000:3000 -p 8090:8090 apache/iggy:latest +docker run --rm -p 8080:8080 -p 3000:3000 -p 8090:8090 \ + -e IGGY_NODE_ADVERTISED_ADDRESS=localhost apache/iggy:latest # Or build from source (recommended for development) cd ../../ && cargo run --bin iggy-server @@ -24,7 +25,7 @@ You can also customize the server using environment variables: ```bash ## Example: Enable HTTP transport and set custom address -IGGY_HTTP_ENABLED=true IGGY_TCP_ADDRESS=0.0.0.0:8090 cargo run --bin iggy-server +IGGY_HTTP_ENABLED=true IGGY_TCP_ADDRESS=127.0.0.1:8090 cargo run --bin iggy-server ``` and then install Node.js dependencies: diff --git a/examples/python/README.md b/examples/python/README.md index 7e5da180d4..4e5c0c740d 100644 --- a/examples/python/README.md +++ b/examples/python/README.md @@ -8,7 +8,8 @@ To run any example, first start the server with ```bash # Using latest release -docker run --rm -p 8080:8080 -p 3000:3000 -p 8090:8090 apache/iggy:latest +docker run --rm -p 8080:8080 -p 3000:3000 -p 8090:8090 \ + -e IGGY_NODE_ADVERTISED_ADDRESS=localhost apache/iggy:latest # Or build from source (recommended for development) cd ../../ && cargo run --bin iggy-server @@ -24,7 +25,7 @@ You can also customize the server using environment variables: ```bash ## Example: Enable HTTP transport and set custom address -IGGY_HTTP_ENABLED=true IGGY_TCP_ADDRESS=0.0.0.0:8090 cargo run --bin iggy-server +IGGY_HTTP_ENABLED=true IGGY_TCP_ADDRESS=127.0.0.1:8090 cargo run --bin iggy-server ``` and then install Python dependencies: diff --git a/examples/rust/README.md b/examples/rust/README.md index d14b5491f5..b7b296e77c 100644 --- a/examples/rust/README.md +++ b/examples/rust/README.md @@ -59,7 +59,7 @@ You can also customize the server using environment variables: ```bash ## Example: Enable HTTP transport and set custom address -IGGY_HTTP_ENABLED=true IGGY_TCP_ADDRESS=0.0.0.0:8090 cargo run --bin iggy-server +IGGY_HTTP_ENABLED=true IGGY_TCP_ADDRESS=127.0.0.1:8090 cargo run --bin iggy-server ``` You can run multiple producers and consumers simultaneously to observe how messages are distributed across clients. Most examples support configurable options via the [Args](https://github.com/apache/iggy/blob/master/examples/rust/src/shared/args.rs) struct, including transport protocol, stream/topic/partition settings, consumer ID, message size, and more. diff --git a/foreign/csharp/Iggy_SDK.Tests.Integration/Fixtures/VsrCluster.cs b/foreign/csharp/Iggy_SDK.Tests.Integration/Fixtures/VsrCluster.cs index 23001c9da5..e6cafc964d 100644 --- a/foreign/csharp/Iggy_SDK.Tests.Integration/Fixtures/VsrCluster.cs +++ b/foreign/csharp/Iggy_SDK.Tests.Integration/Fixtures/VsrCluster.cs @@ -291,6 +291,7 @@ private Dictionary BuildClusterEnvironment(IReadOnlyList if (!ClusterEnabled) { + environment["IGGY_NODE_ADVERTISED_ADDRESS"] = "127.0.0.1"; return environment; } diff --git a/foreign/java/external-processors/iggy-connector-flink/docker-compose.yml b/foreign/java/external-processors/iggy-connector-flink/docker-compose.yml index da7011c302..8b20600f8d 100644 --- a/foreign/java/external-processors/iggy-connector-flink/docker-compose.yml +++ b/foreign/java/external-processors/iggy-connector-flink/docker-compose.yml @@ -26,6 +26,7 @@ services: - "3000:3000" # HTTP API / Web UI environment: - IGGY_TCP_ADDRESS=0.0.0.0:8090 + - IGGY_NODE_ADVERTISED_ADDRESS=iggy - IGGY_HTTP_ADDRESS=0.0.0.0:3000 - IGGY_QUIC_ADDRESS=0.0.0.0:8080 - IGGY_SYSTEM_LOGGING_LEVEL=info diff --git a/foreign/java/external-processors/iggy-connector-pinot/docker-compose.yml b/foreign/java/external-processors/iggy-connector-pinot/docker-compose.yml index 8e50a9be38..49fed9a29f 100644 --- a/foreign/java/external-processors/iggy-connector-pinot/docker-compose.yml +++ b/foreign/java/external-processors/iggy-connector-pinot/docker-compose.yml @@ -27,6 +27,7 @@ services: environment: - IGGY_SYSTEM_LOGGING_LEVEL=info - IGGY_TCP_ADDRESS=0.0.0.0:8090 + - IGGY_NODE_ADVERTISED_ADDRESS=iggy - IGGY_HTTP_ENABLED=true - IGGY_HTTP_ADDRESS=0.0.0.0:3000 - IGGY_QUIC_ADDRESS=0.0.0.0:8080 diff --git a/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/BaseIntegrationTest.java b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/BaseIntegrationTest.java index 7c94728867..1e4fced76e 100644 --- a/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/BaseIntegrationTest.java +++ b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/BaseIntegrationTest.java @@ -91,6 +91,7 @@ static void setupContainer() { .withEnv("IGGY_ROOT_PASSWORD", "iggy") .withEnv("IGGY_TCP_ADDRESS", "0.0.0.0:8090") .withEnv("IGGY_HTTP_ADDRESS", "0.0.0.0:3000") + .withEnv("IGGY_NODE_ADVERTISED_ADDRESS", "127.0.0.1") .withCreateContainerCmdModifier(cmd -> cmd.getHostConfig() .withCapAdd(Capability.SYS_NICE) .withSecurityOpts(List.of("seccomp:unconfined")) diff --git a/foreign/php/README.md b/foreign/php/README.md index 73a31cfa42..d6bd9de37a 100644 --- a/foreign/php/README.md +++ b/foreign/php/README.md @@ -64,6 +64,7 @@ php -r 'var_dump(extension_loaded("iggy-php"));' docker run --rm --name iggy-php-test \ -p 8090:8090 \ -p 3000:3000 \ + -e IGGY_NODE_ADVERTISED_ADDRESS=localhost \ apache/iggy:latest ``` diff --git a/foreign/php/docker-compose.test.yml b/foreign/php/docker-compose.test.yml index 3831e39bde..19abcc2178 100644 --- a/foreign/php/docker-compose.test.yml +++ b/foreign/php/docker-compose.test.yml @@ -31,6 +31,7 @@ services: environment: - IGGY_HTTP_ADDRESS=0.0.0.0:3000 - IGGY_TCP_ADDRESS=0.0.0.0:8090 + - IGGY_NODE_ADVERTISED_ADDRESS=iggy-server - IGGY_QUIC_ADDRESS=0.0.0.0:8080 - IGGY_WEBSOCKET_ADDRESS=0.0.0.0:8092 - IGGY_ROOT_USERNAME=iggy diff --git a/foreign/python/docker-compose.test.yml b/foreign/python/docker-compose.test.yml index 78caed9929..665df9121c 100644 --- a/foreign/python/docker-compose.test.yml +++ b/foreign/python/docker-compose.test.yml @@ -36,6 +36,7 @@ services: environment: - IGGY_HTTP_ADDRESS=0.0.0.0:3000 - IGGY_TCP_ADDRESS=0.0.0.0:8090 + - IGGY_NODE_ADVERTISED_ADDRESS=iggy-server - IGGY_QUIC_ADDRESS=0.0.0.0:8080 - IGGY_WEBSOCKET_ADDRESS=0.0.0.0:8092 - IGGY_ROOT_USERNAME=iggy diff --git a/foreign/python/tests/test_tls.py b/foreign/python/tests/test_tls.py index f9be943797..f1daf889cf 100644 --- a/foreign/python/tests/test_tls.py +++ b/foreign/python/tests/test_tls.py @@ -66,6 +66,7 @@ def tls_container(): .with_env("IGGY_TCP_TLS_CERT_FILE", "/app/certs/iggy_cert.pem") .with_env("IGGY_TCP_TLS_KEY_FILE", "/app/certs/iggy_key.pem") .with_env("IGGY_TCP_ADDRESS", f"0.0.0.0:{CONTAINER_TCP_PORT}") + .with_env("IGGY_NODE_ADVERTISED_ADDRESS", "127.0.0.1") .with_volume_mapping(CERTS_DIR, "/app/certs", "ro") .with_kwargs(privileged=True) ) diff --git a/helm/charts/iggy/README.md b/helm/charts/iggy/README.md index 4e4cd42116..d9861ca13d 100644 --- a/helm/charts/iggy/README.md +++ b/helm/charts/iggy/README.md @@ -210,6 +210,23 @@ Ensure the server binds to `0.0.0.0` instead of `127.0.0.1`. This is configured * `IGGY_TCP_ADDRESS=0.0.0.0:8090` * `IGGY_QUIC_ADDRESS=0.0.0.0:8080` +A wildcard bind says which interfaces accept connections, not where clients +reach the pod, so the server also needs the address to publish in cluster +metadata. Server builds that carry the setting refuse to start without it; +older ones, including the `0.7.0` this chart pins by default, have no such +refusal and log the variable as unknown and ignored. Either way the chart +sets `IGGY_NODE_ADVERTISED_ADDRESS` to the in-cluster Service DNS name; +override it with `server.advertisedAddress` when clients arrive through a +LoadBalancer or an Ingress. + +Declaring `IGGY_NODE_ADVERTISED_ADDRESS` in `server.env` yourself works too: +the chart then leaves its own default out, so the variable is declared once. +Give that entry a non-empty value; an empty one suppresses the chart default +and reaches the server as unset, so the render fails instead. +Setting it in `server.env` and in `server.advertisedAddress` at the same time +is refused at render time rather than resolved silently, since only the +`server.env` entry would take effect. + ## Accessing the Server ### Port Forward @@ -312,7 +329,8 @@ pre-commit install | podSecurityContext | object | `{"seccompProfile":{"type":"Unconfined"}}` | Pod security context (server uses io_uring, requires unconfined seccomp) | | resources | object | `{}` | Resource limits and requests for server | | securityContext | object | `{"capabilities":{"add":["IPC_LOCK"]}}` | Container security context (server requires IPC_LOCK for io_uring) | -| server | object | `{"affinity":{},"enabled":true,"env":[{"name":"RUST_LOG","value":"info"},{"name":"IGGY_HTTP_ADDRESS","value":"0.0.0.0:3000"},{"name":"IGGY_TCP_ADDRESS","value":"0.0.0.0:8090"},{"name":"IGGY_QUIC_ADDRESS","value":"0.0.0.0:8080"},{"name":"IGGY_WEBSOCKET_ADDRESS","value":"0.0.0.0:8092"}],"image":{"pullPolicy":"Always","repository":"apache/iggy","tag":"0.7.0"},"ingress":{"annotations":{},"className":"","enabled":false,"hosts":[{"host":"chart-example.local","paths":[{"path":"/","pathType":"ImplementationSpecific"}]}],"tls":[]},"nodeSelector":{},"persistence":{"accessMode":"ReadWriteOnce","annotations":{},"enabled":false,"existingClaim":"","size":"8Gi","storageClass":""},"ports":{"http":3000,"quic":8080,"tcp":8090},"replicaCount":1,"service":{"port":3000,"type":"ClusterIP"},"serviceMonitor":{"additionalLabels":{},"authorization":{},"enabled":false,"honorLabels":false,"interval":"30s","namespace":"","path":"/metrics","scrapeTimeout":"10s"},"tolerations":[],"users":{"root":{"createSecret":true,"existingSecret":{"name":"","passwordKey":"password","usernameKey":"username"},"password":"changeit","username":"iggy"}}}` | Iggy server configuration | +| server | object | `{"advertisedAddress":"","affinity":{},"enabled":true,"env":[{"name":"RUST_LOG","value":"info"},{"name":"IGGY_HTTP_ADDRESS","value":"0.0.0.0:3000"},{"name":"IGGY_TCP_ADDRESS","value":"0.0.0.0:8090"},{"name":"IGGY_QUIC_ADDRESS","value":"0.0.0.0:8080"},{"name":"IGGY_WEBSOCKET_ADDRESS","value":"0.0.0.0:8092"}],"image":{"pullPolicy":"Always","repository":"apache/iggy","tag":"0.7.0"},"ingress":{"annotations":{},"className":"","enabled":false,"hosts":[{"host":"chart-example.local","paths":[{"path":"/","pathType":"ImplementationSpecific"}]}],"tls":[]},"nodeSelector":{},"persistence":{"accessMode":"ReadWriteOnce","annotations":{},"enabled":false,"existingClaim":"","size":"8Gi","storageClass":""},"ports":{"http":3000,"quic":8080,"tcp":8090},"replicaCount":1,"service":{"port":3000,"type":"ClusterIP"},"serviceMonitor":{"additionalLabels":{},"authorization":{},"enabled":false,"honorLabels":false,"interval":"30s","namespace":"","path":"/metrics","scrapeTimeout":"10s"},"tolerations":[],"users":{"root":{"createSecret":true,"existingSecret":{"name":"","passwordKey":"password","usernameKey":"username"},"password":"changeit","username":"iggy"}}}` | Iggy server configuration | +| server.advertisedAddress | string | `""` | Client-facing address published in cluster metadata. Declaring `IGGY_NODE_ADVERTISED_ADDRESS` in `server.env` instead also works, but setting both is refused at render time. Empty falls back to the in-cluster Service DNS name. | | server.affinity | object | `{}` | Affinity rules for server pods | | server.enabled | bool | `true` | Enable the Iggy server deployment | | server.env | list | `[{"name":"RUST_LOG","value":"info"},{"name":"IGGY_HTTP_ADDRESS","value":"0.0.0.0:3000"},{"name":"IGGY_TCP_ADDRESS","value":"0.0.0.0:8090"},{"name":"IGGY_QUIC_ADDRESS","value":"0.0.0.0:8080"},{"name":"IGGY_WEBSOCKET_ADDRESS","value":"0.0.0.0:8092"}]` | Environment variables for the server container | diff --git a/helm/charts/iggy/README.md.gotmpl b/helm/charts/iggy/README.md.gotmpl index a30365fe6c..03a2700ea7 100644 --- a/helm/charts/iggy/README.md.gotmpl +++ b/helm/charts/iggy/README.md.gotmpl @@ -228,6 +228,23 @@ Ensure the server binds to `0.0.0.0` instead of `127.0.0.1`. This is configured * `IGGY_TCP_ADDRESS=0.0.0.0:8090` * `IGGY_QUIC_ADDRESS=0.0.0.0:8080` +A wildcard bind says which interfaces accept connections, not where clients +reach the pod, so the server also needs the address to publish in cluster +metadata. Server builds that carry the setting refuse to start without it; +older ones, including the `0.7.0` this chart pins by default, have no such +refusal and log the variable as unknown and ignored. Either way the chart +sets `IGGY_NODE_ADVERTISED_ADDRESS` to the in-cluster Service DNS name; +override it with `server.advertisedAddress` when clients arrive through a +LoadBalancer or an Ingress. + +Declaring `IGGY_NODE_ADVERTISED_ADDRESS` in `server.env` yourself works too: +the chart then leaves its own default out, so the variable is declared once. +Give that entry a non-empty value; an empty one suppresses the chart default +and reaches the server as unset, so the render fails instead. +Setting it in `server.env` and in `server.advertisedAddress` at the same time +is refused at render time rather than resolved silently, since only the +`server.env` entry would take effect. + ## Accessing the Server ### Port Forward diff --git a/helm/charts/iggy/templates/deployment.yaml b/helm/charts/iggy/templates/deployment.yaml index 4f17b2f33d..4e7b6751ef 100644 --- a/helm/charts/iggy/templates/deployment.yaml +++ b/helm/charts/iggy/templates/deployment.yaml @@ -89,6 +89,22 @@ spec: name: {{ include "iggy.fullname" . }}-root-credentials key: password {{- end }}{{- end}} + {{- $declaredInEnv := false }} + {{- range .Values.server.env }} + {{- if eq .name "IGGY_NODE_ADVERTISED_ADDRESS" }} + {{- if not .value }} + {{- fail "IGGY_NODE_ADVERTISED_ADDRESS is declared in server.env with an empty value. The server reads that as unset and refuses to start behind a wildcard bind. Give the entry the address clients dial, or drop it and set server.advertisedAddress." }} + {{- end }} + {{- $declaredInEnv = true }} + {{- end }} + {{- end }} + {{- if and $declaredInEnv .Values.server.advertisedAddress }} + {{- fail "IGGY_NODE_ADVERTISED_ADDRESS is set in server.env and server.advertisedAddress is also set. The server.env entry wins and server.advertisedAddress is ignored. Please set only one." }} + {{- end }} + {{- if not $declaredInEnv }} + - name: IGGY_NODE_ADVERTISED_ADDRESS + value: {{ .Values.server.advertisedAddress | default (printf "%s.%s.svc.cluster.local" (include "iggy.fullname" .) .Release.Namespace) | quote }} + {{- end }} {{- if .Values.server.env }} {{- range .Values.server.env }} - name: {{ .name }} diff --git a/helm/charts/iggy/values.yaml b/helm/charts/iggy/values.yaml index 7e760ea547..be97571569 100644 --- a/helm/charts/iggy/values.yaml +++ b/helm/charts/iggy/values.yaml @@ -17,6 +17,11 @@ # -- Iggy server configuration server: + # -- Client-facing address published in cluster metadata. Declaring + # `IGGY_NODE_ADVERTISED_ADDRESS` in `server.env` instead also works, but + # setting both is refused at render time. Empty falls back to the in-cluster + # Service DNS name. + advertisedAddress: "" # -- Enable the Iggy server deployment enabled: true # -- Number of server replicas diff --git a/web/README.md b/web/README.md index f2ef7cdbc3..e3dcc4224c 100644 --- a/web/README.md +++ b/web/README.md @@ -25,7 +25,8 @@ The [docker image](https://hub.docker.com/r/apache/iggy-web-ui) is available, an ``` ```sh - docker run -p 3000:3000 -p 8090:8090 apache/iggy:latest + docker run -p 3000:3000 -p 8090:8090 \ + -e IGGY_NODE_ADVERTISED_ADDRESS=localhost apache/iggy:latest ``` 2. **Clone the repository:** diff --git a/web/docker-compose.yml b/web/docker-compose.yml index dca555202b..e8e9a2cf13 100644 --- a/web/docker-compose.yml +++ b/web/docker-compose.yml @@ -24,6 +24,7 @@ services: IGGY_HTTP_ADDRESS: 0.0.0.0:3000 IGGY_QUIC_ADDRESS: 0.0.0.0:8080 IGGY_TCP_ADDRESS: 0.0.0.0:8090 + IGGY_NODE_ADVERTISED_ADDRESS: localhost IGGY_WEBSOCKET_ADDRESS: 0.0.0.0:8092 cap_add: - SYS_NICE From dde679aea50ea980f057a62dbfc07a90d0c6c4ec Mon Sep 17 00:00:00 2001 From: Ethan Lin <103916325+ethanlin01x@users.noreply.github.com> Date: Mon, 31 Aug 2026 17:05:20 +0800 Subject: [PATCH 017/182] ci(python): fold the build task into test and check the stub (#3980) --- .../python-maturin/pre-merge/action.yml | 50 +++++++++++++------ .github/config/components.yml | 2 +- foreign/python/README.md | 3 +- foreign/python/apache_iggy.pyi | 50 +++++++++---------- foreign/python/src/bin/stub_gen.rs | 11 ++-- foreign/python/src/topic.rs | 4 +- 6 files changed, 66 insertions(+), 54 deletions(-) diff --git a/.github/actions/python-maturin/pre-merge/action.yml b/.github/actions/python-maturin/pre-merge/action.yml index f97462f34d..6a28d0e5be 100644 --- a/.github/actions/python-maturin/pre-merge/action.yml +++ b/.github/actions/python-maturin/pre-merge/action.yml @@ -20,7 +20,7 @@ description: Python pre-merge testing with maturin github iggy actions inputs: task: - description: "Task to run (lint, test, build)" + description: "Task to run (lint, test)" required: true runs: @@ -90,6 +90,39 @@ runs: echo "pyrefly version: $(uv run pyrefly --version)" shell: bash + - name: Build Python wheel + if: inputs.task == 'test' + run: | + cd foreign/python + + # Build the module + echo "Building Python wheel..." + uv run maturin build -o dist + + # List built artifacts + echo "" + echo "Build artifacts:" + ls -la dist/ + shell: bash + + - name: Check apache_iggy.pyi is up to date + if: inputs.task == 'test' + run: | + # stub_gen must run from foreign/python; a subdirectory corrupts the stub. + cd foreign/python + cargo run --bin stub_gen + + # The tracked stub is post-ruff. Format first: the raw output leaves + # whitespace on blank lines, which `check` reports but only `format` fixes. + uv run --no-sync ruff format apache_iggy.pyi + uv run --no-sync ruff check --fix apache_iggy.pyi + + if ! git diff --exit-code -- apache_iggy.pyi; then + echo "::error::apache_iggy.pyi is out of date. Run 'cargo run --bin stub_gen' from foreign/python, let the ruff hooks format it, and commit the result." + exit 1 + fi + shell: bash + - name: Build Python wheel with coverage instrumentation if: inputs.task == 'test' run: | @@ -105,21 +138,6 @@ runs: uv run --no-sync maturin develop shell: bash - - name: Build Python wheel - if: inputs.task == 'build' - run: | - cd foreign/python - - # Build the module - echo "Building Python wheel..." - uv run maturin build -o dist - - # List built artifacts - echo "" - echo "Build artifacts:" - ls -la dist/ - shell: bash - - name: Build server Docker image for TLS tests if: inputs.task == 'test' run: | diff --git a/.github/config/components.yml b/.github/config/components.yml index 5fef0e88a0..082aec83b5 100644 --- a/.github/config/components.yml +++ b/.github/config/components.yml @@ -232,7 +232,7 @@ components: - "ci-infrastructure" # CI changes trigger full regression paths: - "foreign/python/**" - tasks: ["lint", "test", "build"] + tasks: ["lint", "test"] sdk-php: depends_on: diff --git a/foreign/python/README.md b/foreign/python/README.md index dbbdbe02ca..e2a4fffc2d 100644 --- a/foreign/python/README.md +++ b/foreign/python/README.md @@ -92,12 +92,11 @@ Every installation below compiles the Rust extension, so you'll need: pytest tests/ -v # make sure iggy-server is running and the venv is activated ``` -4. To update the stubs, only after changing the pyo3 API surface (nothing in CI checks stub freshness, so unconditional regen just invites `.pyi` churn), use +4. To update the stubs, after changing the pyo3 API surface, use ```bash # run from foreign/python cargo run --bin stub_gen - # TODO: Known bug: running this from a subdirectory of `foreign/python` corrupts the tracked stub, see https://github.com/apache/iggy/pull/3825/changes/BASE..773a27971b4ddb7b44773ded395ed23afb1de4c9#r3727691619 ``` 5. Before committing, test the pre-commit and pre-push hooks. `prek` only inspects staged content, so stage your work first: diff --git a/foreign/python/apache_iggy.pyi b/foreign/python/apache_iggy.pyi index 254b23f41c..6c3715b54b 100644 --- a/foreign/python/apache_iggy.pyi +++ b/foreign/python/apache_iggy.pyi @@ -872,6 +872,27 @@ class IggyClient: Sends a ping request to the server to check connectivity. Raises `RuntimeError` if the connection fails. """ + def describe_options( + self, scope: builtins.str + ) -> collections.abc.Awaitable[list[OptionSpec]]: + r""" + Describe the option catalog for a resource scope. + + This is the discovery surface for the `options` argument on + `create_topic`/`update_topic`: a key outside the catalog is refused at + create, and the binary transports carry only the error code back. + + Args: + scope: One of `"topic"`, `"stream"`, `"user"`. + + Returns: + An awaitable that resolves to `list[OptionSpec]`, empty for a scope + with no keys yet. + + Raises: + ValueError: If the scope name is not one of the three above. + RuntimeError: If the request fails. + """ def login_user( self, username: builtins.str, password: builtins.str ) -> collections.abc.Awaitable[None]: @@ -1034,27 +1055,6 @@ class IggyClient: Returns the stream details, or `None` if the stream does not exist. Raises `RuntimeError` on failure. """ - def describe_options( - self, scope: builtins.str - ) -> collections.abc.Awaitable[builtins.list[OptionSpec]]: - r""" - Describe the option catalog for a resource scope. - - This is the discovery surface for the `options` argument on - `create_topic`/`update_topic`: a key outside the catalog is refused at - create, and the binary transports carry only the error code back. - - Args: - scope: One of `"topic"`, `"stream"`, `"user"`. - - Returns: - An awaitable that resolves to `list[OptionSpec]`, empty for a scope - with no keys yet. - - Raises: - ValueError: If the scope name is not one of the three above. - RuntimeError: If the request fails. - """ def create_topic( self, stream: builtins.str | builtins.int, @@ -2103,8 +2103,8 @@ class Topic: r""" Options admission resolved for the keys the client did not send. - Same shape as `options`. These would have resolved differently under - another server configuration. + Same shape as `options`. These would have resolved differently + under another server configuration. """ @typing.final @@ -2168,8 +2168,8 @@ class TopicDetails: r""" Options admission resolved for the keys the client did not send. - Same shape as `options`. These would have resolved differently under - another server configuration. + Same shape as `options`. These would have resolved differently + under another server configuration. """ @property def partitions(self) -> builtins.list[Partition]: diff --git a/foreign/python/src/bin/stub_gen.rs b/foreign/python/src/bin/stub_gen.rs index e9db76059d..82693da1a4 100644 --- a/foreign/python/src/bin/stub_gen.rs +++ b/foreign/python/src/bin/stub_gen.rs @@ -47,14 +47,9 @@ fn main() -> Result<()> { // `stub_info` is a function defined by `define_stub_info_gatherer!` macro. let stub = apache_iggy::client::stub_info()?; stub.generate()?; - let path = Path::new(file!()) - .parent() - .unwrap() - .parent() - .unwrap() - .parent() - .unwrap() - .join("apache_iggy.pyi"); + // Anchor on the manifest dir: `stub.generate()` writes to the crate root, so + // a cwd-relative path leaves the tracked stub without its license header. + let path = Path::new(env!("CARGO_MANIFEST_DIR")).join("apache_iggy.pyi"); let mut f = File::open(&path)?; let mut content = LICENSE.as_bytes().to_owned(); f.read_to_end(&mut content)?; diff --git a/foreign/python/src/topic.rs b/foreign/python/src/topic.rs index e5f86facd8..0d6e35c803 100644 --- a/foreign/python/src/topic.rs +++ b/foreign/python/src/topic.rs @@ -254,7 +254,7 @@ impl Topic { /// Options admission resolved for the keys the client did not send. /// - /// Same shape as [`Self::options`]. These would have resolved differently + /// Same shape as `options`. These would have resolved differently /// under another server configuration. #[getter] pub fn derived_options<'a>(&self, py: Python<'a>) -> PyResult> { @@ -345,7 +345,7 @@ impl TopicDetails { /// Options admission resolved for the keys the client did not send. /// - /// Same shape as [`Self::options`]. These would have resolved differently + /// Same shape as `options`. These would have resolved differently /// under another server configuration. #[getter] pub fn derived_options<'a>(&self, py: Python<'a>) -> PyResult> { From 3e4898ac67c4f60f4146e8313e0132017e59becd Mon Sep 17 00:00:00 2001 From: Chengxi Luo Date: Mon, 31 Aug 2026 06:33:29 -0400 Subject: [PATCH 018/182] fix(bdd): resolve server address and credentials only from env (#3935) --- .github/actions/go/pre-merge/action.yml | 2 + .github/actions/php/pre-merge/action.yml | 10 ++-- .github/workflows/coverage-baseline.yml | 4 ++ .../step_definitions/background_steps.cpp | 15 +++-- bdd/docker-compose.server.yml | 3 + bdd/docker-compose.yml | 9 +-- bdd/go/tests/basic_messaging.go | 10 ++-- bdd/go/tests/env/env.go | 55 +++++++++++++++++ bdd/go/tests/leader_redirection.go | 26 +++----- bdd/go/tests/raw_command.go | 11 ++-- .../tests/tcp_test/session_feature_login.go | 7 ++- bdd/go/tests/tcp_test/test_helpers.go | 11 ++-- .../apache/iggy/bdd/BasicMessagingSteps.java | 11 +--- .../iggy/bdd/LeaderRedirectionSteps.java | 15 ++--- .../org/apache/iggy/bdd/TestEnvironment.java | 59 +++++++++++++++++++ bdd/php/Dockerfile | 2 - bdd/php/tests/BasicMessagingFeatureTest.php | 5 +- bdd/php/tests/RawCommandFeatureTest.php | 7 ++- bdd/php/tests/TestEnvironment.php | 53 +++++++++++++++++ bdd/python/tests/conftest.py | 24 +++++++- bdd/python/tests/test_basic_messaging.py | 4 +- bdd/python/tests/test_raw_command.py | 4 +- bdd/rust/tests/helpers/cluster.rs | 17 ++---- bdd/rust/tests/helpers/env.rs | 50 ++++++++++++++++ bdd/rust/tests/helpers/mod.rs | 1 + bdd/rust/tests/steps/auth.rs | 3 +- bdd/rust/tests/steps/leader_redirection.rs | 3 +- bdd/rust/tests/steps/server.rs | 5 +- .../Iggy_SDK.Tests.BDD/Context/TestContext.cs | 6 +- .../Context/TestEnvironment.cs | 44 ++++++++++++++ .../Iggy_SDK.Tests.BDD/Context/TestHooks.cs | 3 - foreign/csharp/Iggy_SDK.Tests.BDD/README.md | 25 ++++++-- .../BasicMessagingOperationsSteps.cs | 3 +- .../StepDefinitions/LeaderRedirectionSteps.cs | 6 +- foreign/node/README.md | 20 +++++-- foreign/node/src/bdd/README.md | 23 ++++++-- foreign/node/src/bdd/auth.ts | 6 +- foreign/node/src/bdd/env.ts | 39 ++++++++++++ foreign/php/Dockerfile.test | 3 +- foreign/php/README.md | 2 +- foreign/php/docker-compose.test.yml | 7 +-- foreign/php/scripts/test.sh | 9 +-- foreign/php/tests/IggySdkTest.php | 2 +- foreign/php/tests/bootstrap.php | 17 ++++-- 44 files changed, 488 insertions(+), 153 deletions(-) create mode 100644 bdd/go/tests/env/env.go create mode 100644 bdd/java/src/test/java/org/apache/iggy/bdd/TestEnvironment.java create mode 100644 bdd/php/tests/TestEnvironment.php create mode 100644 bdd/rust/tests/helpers/env.rs create mode 100644 foreign/csharp/Iggy_SDK.Tests.BDD/Context/TestEnvironment.cs create mode 100644 foreign/node/src/bdd/env.ts diff --git a/.github/actions/go/pre-merge/action.yml b/.github/actions/go/pre-merge/action.yml index 56e0162b70..53a381edcd 100644 --- a/.github/actions/go/pre-merge/action.yml +++ b/.github/actions/go/pre-merge/action.yml @@ -141,6 +141,8 @@ runs: if: inputs.task == 'e2e' env: IGGY_TCP_ADDRESS: 127.0.0.1:8090 + IGGY_ROOT_USERNAME: iggy + IGGY_ROOT_PASSWORD: iggy run: | echo "🧪 Running Go e2e tests..." diff --git a/.github/actions/php/pre-merge/action.yml b/.github/actions/php/pre-merge/action.yml index 789e9db400..4205d465cf 100644 --- a/.github/actions/php/pre-merge/action.yml +++ b/.github/actions/php/pre-merge/action.yml @@ -161,10 +161,9 @@ runs: shell: bash working-directory: foreign/php env: - IGGY_HOST: 127.0.0.1 - IGGY_PORT: 8090 - IGGY_USERNAME: iggy - IGGY_PASSWORD: iggy + IGGY_TCP_ADDRESS: 127.0.0.1:8090 + IGGY_ROOT_USERNAME: iggy + IGGY_ROOT_PASSWORD: iggy run: | mkdir -p ../../reports ./scripts/test.sh --log-junit ../../reports/php-junit.xml @@ -194,8 +193,7 @@ runs: shell: bash working-directory: foreign/php env: - IGGY_HOST: 127.0.0.1 - IGGY_PORT: 8090 + IGGY_TCP_ADDRESS: 127.0.0.1:8090 IGGY_TLS_CONNECTION_STRING: iggy+tcp://iggy:iggy@127.0.0.1:8090?tls=true&tls_domain=localhost&tls_ca_file=${{ github.workspace }}/core/certs/iggy_ca_cert.pem IGGY_TLS_PLAINTEXT_ADDRESS: 127.0.0.1:8090 run: ./scripts/test.sh --log-junit ../../reports/php-tls-junit.xml tests/TlsTest.php diff --git a/.github/workflows/coverage-baseline.yml b/.github/workflows/coverage-baseline.yml index baa1ae160c..08f878c190 100644 --- a/.github/workflows/coverage-baseline.yml +++ b/.github/workflows/coverage-baseline.yml @@ -516,6 +516,10 @@ jobs: go test -v -race -coverprofile=../../reports/go-coverage-unit.out ./... - name: Run BDD tests with coverage + env: + IGGY_TCP_ADDRESS: 127.0.0.1:8090 + IGGY_ROOT_USERNAME: iggy + IGGY_ROOT_PASSWORD: iggy run: | cd bdd/go mkdir -p ../../reports diff --git a/bdd/cpp/features/step_definitions/background_steps.cpp b/bdd/cpp/features/step_definitions/background_steps.cpp index c4f593f596..eaa0e7b9e0 100644 --- a/bdd/cpp/features/step_definitions/background_steps.cpp +++ b/bdd/cpp/features/step_definitions/background_steps.cpp @@ -27,15 +27,20 @@ #include #include +#include #include #include #include "world.hpp" namespace { -std::string env_or(const char *name, const std::string &fallback) { +std::string required_env(const char *name) { const char *value = std::getenv(name); - return value != nullptr ? std::string(value) : fallback; + if (value == nullptr || *value == '\0') { + throw std::runtime_error(std::string(name) + + " must be set; run the suite via scripts/run-bdd-tests.sh"); + } + return std::string(value); } } // namespace @@ -44,7 +49,7 @@ GIVEN("^I have a running Iggy server$") { // Empty address makes the SDK fall back to its default TCP endpoint; in CI the address // is supplied via IGGY_TCP_ADDRESS (e.g. iggy-server:8090). - const std::string address = env_or("IGGY_TCP_ADDRESS", ""); + const std::string address = required_env("IGGY_TCP_ADDRESS"); iggy::ffi::IggyClientConfig config{}; config.server_address = address; iggy::ffi::Client *client = iggy::ffi::new_connection(std::move(config)); @@ -57,7 +62,7 @@ GIVEN("^I am authenticated as the root user$") { cucumber::ScenarioScope context; ASSERT_NE(context->client, nullptr); - const std::string username = env_or("IGGY_ROOT_USERNAME", "iggy"); - const std::string password = env_or("IGGY_ROOT_PASSWORD", "iggy"); + const std::string username = required_env("IGGY_ROOT_USERNAME"); + const std::string password = required_env("IGGY_ROOT_PASSWORD"); context->client->login_user(username, password); } diff --git a/bdd/docker-compose.server.yml b/bdd/docker-compose.server.yml index 5563040feb..c57b3cdd82 100644 --- a/bdd/docker-compose.server.yml +++ b/bdd/docker-compose.server.yml @@ -80,6 +80,9 @@ services: python-bdd: <<: *server-bdd-deps + php-bdd: + <<: *server-bdd-deps + go-bdd: <<: *server-bdd-deps diff --git a/bdd/docker-compose.yml b/bdd/docker-compose.yml index 9d577ab191..01dee92925 100644 --- a/bdd/docker-compose.yml +++ b/bdd/docker-compose.yml @@ -77,14 +77,9 @@ services: build: context: .. dockerfile: bdd/php/Dockerfile - depends_on: - iggy-server: - condition: service_healthy environment: - - IGGY_HOST=iggy-server - - IGGY_PORT=8090 - - IGGY_USERNAME=iggy - - IGGY_PASSWORD=iggy + - IGGY_ROOT_USERNAME=iggy + - IGGY_ROOT_PASSWORD=iggy - BDD_FEATURE=${BDD_FEATURE:-all} volumes: - ./scenarios/basic_messaging.feature:/app/features/basic_messaging.feature diff --git a/bdd/go/tests/basic_messaging.go b/bdd/go/tests/basic_messaging.go index e5d7089f68..21ad3e12b7 100644 --- a/bdd/go/tests/basic_messaging.go +++ b/bdd/go/tests/basic_messaging.go @@ -21,8 +21,8 @@ import ( "context" "errors" "fmt" - "os" + "github.com/apache/iggy/bdd/go/tests/env" "github.com/apache/iggy/foreign/go/client" "github.com/apache/iggy/foreign/go/client/tcp" iggcon "github.com/apache/iggy/foreign/go/contracts" @@ -52,10 +52,7 @@ type basicMessagingSteps struct{} func (s basicMessagingSteps) givenRunningServer(ctx context.Context) error { c := getBasicMessagingCtx(ctx) - addr := os.Getenv("IGGY_TCP_ADDRESS") - if addr == "" { - addr = "127.0.0.1:8090" - } + addr := env.ServerAddress() c.serverAddr = &addr return nil } @@ -80,7 +77,8 @@ func (s basicMessagingSteps) givenAuthenticationAsRoot(ctx context.Context) erro return fmt.Errorf("error pinging client: %w", err) } - if _, err = cli.LoginUser(ctx, "iggy", "iggy"); err != nil { + username, password := env.RootCredentials() + if _, err = cli.LoginUser(ctx, username, password); err != nil { return fmt.Errorf("error logging in: %v", err) } diff --git a/bdd/go/tests/env/env.go b/bdd/go/tests/env/env.go new file mode 100644 index 0000000000..002c9bfceb --- /dev/null +++ b/bdd/go/tests/env/env.go @@ -0,0 +1,55 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +// Package env reads the endpoints and credentials the BDD suites run +// against. A default here would turn a dropped compose variable into a run +// against whatever happens to listen on the fallback address, so a missing +// value aborts the suite instead. +package env + +import ( + "fmt" + "os" +) + +func required(name string) string { + value, ok := os.LookupEnv(name) + if !ok || value == "" { + panic(fmt.Sprintf("%s must be set; run the suite via scripts/run-bdd-tests.sh", name)) + } + return value +} + +// ServerAddress returns the address of the single-node server. +func ServerAddress() string { + return required("IGGY_TCP_ADDRESS") +} + +// LeaderAddress returns the address of the cluster leader. +func LeaderAddress() string { + return required("IGGY_TCP_ADDRESS_LEADER") +} + +// FollowerAddress returns the address of the cluster follower. +func FollowerAddress() string { + return required("IGGY_TCP_ADDRESS_FOLLOWER") +} + +// RootCredentials returns the root username and password. +func RootCredentials() (string, string) { + return required("IGGY_ROOT_USERNAME"), required("IGGY_ROOT_PASSWORD") +} diff --git a/bdd/go/tests/leader_redirection.go b/bdd/go/tests/leader_redirection.go index cca2aa66b7..8593f55127 100644 --- a/bdd/go/tests/leader_redirection.go +++ b/bdd/go/tests/leader_redirection.go @@ -22,12 +22,11 @@ import ( "errors" "fmt" "net" - "os" "regexp" - "strconv" "strings" "time" + "github.com/apache/iggy/bdd/go/tests/env" "github.com/apache/iggy/foreign/go/client" "github.com/apache/iggy/foreign/go/client/tcp" iggcon "github.com/apache/iggy/foreign/go/contracts" @@ -35,9 +34,6 @@ import ( "github.com/cucumber/godog" ) -const defaultRootUsername = "iggy" -const defaultRootPassword = "iggy" - type leaderCtxKey struct{} type leaderCtx struct { Clients map[string]iggcon.Client @@ -90,25 +86,16 @@ func resolveServerAddress(role string, port uint16) string { switch { case role == "leader" && port == 8091: - if addr, ok := os.LookupEnv("IGGY_TCP_ADDRESS_LEADER"); ok { - return addr - } - return "iggy-leader:8091" + return env.LeaderAddress() case role == "follower" && port == 8092: - if addr, ok := os.LookupEnv("IGGY_TCP_ADDRESS_FOLLOWER"); ok { - return addr - } - return "iggy-follower:8092" + return env.FollowerAddress() case port == 8090: - if addr, ok := os.LookupEnv("IGGY_TCP_ADDRESS"); ok { - return addr - } - return "iggy-server:8090" + return env.ServerAddress() default: - return "iggy-server:" + strconv.Itoa(int(port)) + panic(fmt.Sprintf("no address mapping for role %q on port %d", role, port)) } } @@ -323,12 +310,13 @@ func (s leaderSteps) whenAuthenticateRoot(ctx context.Context) error { names = []string{"main"} } + username, password := env.RootCredentials() for _, name := range names { cli, ok := c.Clients[name] if !ok { return fmt.Errorf("client %s should be created", name) } - if _, err := cli.LoginUser(ctx, defaultRootUsername, defaultRootPassword); err != nil { + if _, err := cli.LoginUser(ctx, username, password); err != nil { return err } // Small delay between multiple authentications to avoid race conditions diff --git a/bdd/go/tests/raw_command.go b/bdd/go/tests/raw_command.go index 8bafdf1c13..290dfe3252 100644 --- a/bdd/go/tests/raw_command.go +++ b/bdd/go/tests/raw_command.go @@ -21,8 +21,8 @@ import ( "context" "errors" "fmt" - "os" + "github.com/apache/iggy/bdd/go/tests/env" "github.com/apache/iggy/foreign/go/client" "github.com/apache/iggy/foreign/go/client/tcp" iggcon "github.com/apache/iggy/foreign/go/contracts" @@ -46,11 +46,7 @@ func getRawCommandCtx(ctx context.Context) *rawCommandCtx { type rawCommandSteps struct{} func (rawCommandSteps) givenRunningServer(ctx context.Context) error { - address := os.Getenv("IGGY_TCP_ADDRESS") - if address == "" { - address = "127.0.0.1:8090" - } - getRawCommandCtx(ctx).serverAddr = address + getRawCommandCtx(ctx).serverAddr = env.ServerAddress() return nil } @@ -63,7 +59,8 @@ func (rawCommandSteps) givenAuthenticationAsRoot(ctx context.Context) error { if err = iggyClient.Connect(ctx); err != nil { return fmt.Errorf("connect client: %w", err) } - if _, err = iggyClient.LoginUser(ctx, "iggy", "iggy"); err != nil { + username, password := env.RootCredentials() + if _, err = iggyClient.LoginUser(ctx, username, password); err != nil { return fmt.Errorf("authenticate client: %w", err) } state.client = iggyClient diff --git a/bdd/go/tests/tcp_test/session_feature_login.go b/bdd/go/tests/tcp_test/session_feature_login.go index ccb1b08e51..9d6315178c 100644 --- a/bdd/go/tests/tcp_test/session_feature_login.go +++ b/bdd/go/tests/tcp_test/session_feature_login.go @@ -20,16 +20,19 @@ package tcp_test import ( "context" + "github.com/apache/iggy/bdd/go/tests/env" iggcon "github.com/apache/iggy/foreign/go/contracts" "github.com/onsi/ginkgo/v2" "github.com/onsi/gomega" ) var _ = ginkgo.Describe("LOGIN FEATURE:", func() { + rootUsername, rootPassword := env.RootCredentials() + ginkgo.When("user is already logged in", func() { ginkgo.Context("and tries to log with correct data", func() { client := createAuthorizedConnection() - user, err := client.LoginUser(context.Background(), "iggy", "iggy") + user, err := client.LoginUser(context.Background(), rootUsername, rootPassword) itShouldNotReturnError(err) itShouldReturnUserId(user, 0) @@ -47,7 +50,7 @@ var _ = ginkgo.Describe("LOGIN FEATURE:", func() { ginkgo.When("user is not logged in", func() { ginkgo.Context("and tries to log with correct data", func() { client := createClient() - user, err := client.LoginUser(context.Background(), "iggy", "iggy") + user, err := client.LoginUser(context.Background(), rootUsername, rootPassword) itShouldNotReturnError(err) itShouldReturnUserId(user, 0) diff --git a/bdd/go/tests/tcp_test/test_helpers.go b/bdd/go/tests/tcp_test/test_helpers.go index efcdd18310..b122b8c4b4 100644 --- a/bdd/go/tests/tcp_test/test_helpers.go +++ b/bdd/go/tests/tcp_test/test_helpers.go @@ -20,10 +20,10 @@ package tcp_test import ( "context" "math/rand" - "os" "strings" "time" + "github.com/apache/iggy/bdd/go/tests/env" "github.com/apache/iggy/foreign/go/client" iggcon "github.com/apache/iggy/foreign/go/contracts" @@ -32,7 +32,8 @@ import ( func createAuthorizedConnection() iggcon.Client { cli := createClient() - _, err := cli.LoginUser(context.Background(), "iggy", "iggy") + username, password := env.RootCredentials() + _, err := cli.LoginUser(context.Background(), username, password) if err != nil { panic(err) } @@ -40,13 +41,9 @@ func createAuthorizedConnection() iggcon.Client { } func createClient() iggcon.Client { - addr := os.Getenv("IGGY_TCP_ADDRESS") - if addr == "" { - addr = "127.0.0.1:8090" - } cli, err := client.NewIggyClient( client.WithTcp( - tcp.WithServerAddress(addr), + tcp.WithServerAddress(env.ServerAddress()), ), ) if err != nil { diff --git a/bdd/java/src/test/java/org/apache/iggy/bdd/BasicMessagingSteps.java b/bdd/java/src/test/java/org/apache/iggy/bdd/BasicMessagingSteps.java index 8cc810960d..0fc05c28c0 100644 --- a/bdd/java/src/test/java/org/apache/iggy/bdd/BasicMessagingSteps.java +++ b/bdd/java/src/test/java/org/apache/iggy/bdd/BasicMessagingSteps.java @@ -54,7 +54,7 @@ public class BasicMessagingSteps { @Given("I have a running Iggy server") public void runningServer() { - context.serverAddr = getenvOrDefault("IGGY_TCP_ADDRESS", "127.0.0.1:8090"); + context.serverAddr = TestEnvironment.serverAddress(); HostPort hostPort = HostPort.parse(context.serverAddr); IggyTcpClient client = @@ -67,9 +67,7 @@ public void runningServer() { @Given("I am authenticated as the root user") public void authenticatedRootUser() { - String username = getenvOrDefault("IGGY_ROOT_USERNAME", "iggy"); - String password = getenvOrDefault("IGGY_ROOT_PASSWORD", "iggy"); - getClient().users().login(username, password); + getClient().users().login(TestEnvironment.rootUsername(), TestEnvironment.rootPassword()); } @Given("I have no streams in the system") @@ -301,11 +299,6 @@ private IggyBaseClient getClient() { return context.client; } - private static String getenvOrDefault(String key, String defaultValue) { - String value = System.getenv(key); - return value == null || value.isBlank() ? defaultValue : value; - } - private static final class HostPort { private final String host; private final int port; diff --git a/bdd/java/src/test/java/org/apache/iggy/bdd/LeaderRedirectionSteps.java b/bdd/java/src/test/java/org/apache/iggy/bdd/LeaderRedirectionSteps.java index 4f9645377a..23f556e8f4 100644 --- a/bdd/java/src/test/java/org/apache/iggy/bdd/LeaderRedirectionSteps.java +++ b/bdd/java/src/test/java/org/apache/iggy/bdd/LeaderRedirectionSteps.java @@ -208,8 +208,8 @@ private void createAndConnectClient(String name, String address) { } private void authenticateAllClients() { - String username = getenvOrDefault("IGGY_ROOT_USERNAME", "iggy"); - String password = getenvOrDefault("IGGY_ROOT_PASSWORD", "iggy"); + String username = TestEnvironment.rootUsername(); + String password = TestEnvironment.rootPassword(); for (IggyTcpClient client : clients.values()) { String initialAddress = client.getConnectionInfo().serverAddress(); client.users().login(username, password); @@ -275,8 +275,8 @@ private static void assertAddressMatchesPort(String address, int port, String de private static String addressForRole(String role) { return switch (role) { - case "leader" -> getenvOrDefault("IGGY_TCP_ADDRESS_LEADER", "127.0.0.1:8091"); - case "follower" -> getenvOrDefault("IGGY_TCP_ADDRESS_FOLLOWER", "127.0.0.1:8092"); + case "leader" -> TestEnvironment.leaderAddress(); + case "follower" -> TestEnvironment.followerAddress(); default -> throw new IllegalArgumentException("Unknown role: " + role); }; } @@ -291,11 +291,6 @@ private static String addressForPort(int port) { } private static String singleServerAddress() { - return getenvOrDefault("IGGY_TCP_ADDRESS", "127.0.0.1:8090"); - } - - private static String getenvOrDefault(String key, String defaultValue) { - String value = System.getenv(key); - return value == null || value.isBlank() ? defaultValue : value; + return TestEnvironment.serverAddress(); } } diff --git a/bdd/java/src/test/java/org/apache/iggy/bdd/TestEnvironment.java b/bdd/java/src/test/java/org/apache/iggy/bdd/TestEnvironment.java new file mode 100644 index 0000000000..0a3cdc9aae --- /dev/null +++ b/bdd/java/src/test/java/org/apache/iggy/bdd/TestEnvironment.java @@ -0,0 +1,59 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.iggy.bdd; + +/** + * Endpoints and credentials the BDD suite runs against. + * + *

A default here would turn a dropped compose variable into a run against whatever happens to + * listen on the fallback address, so a missing value aborts the suite instead. + */ +final class TestEnvironment { + + private TestEnvironment() {} + + static String require(String name) { + String value = System.getenv(name); + if (value == null || value.isBlank()) { + throw new IllegalStateException(name + " must be set; run the suite via scripts/run-bdd-tests.sh"); + } + return value; + } + + static String serverAddress() { + return require("IGGY_TCP_ADDRESS"); + } + + static String leaderAddress() { + return require("IGGY_TCP_ADDRESS_LEADER"); + } + + static String followerAddress() { + return require("IGGY_TCP_ADDRESS_FOLLOWER"); + } + + static String rootUsername() { + return require("IGGY_ROOT_USERNAME"); + } + + static String rootPassword() { + return require("IGGY_ROOT_PASSWORD"); + } +} diff --git a/bdd/php/Dockerfile b/bdd/php/Dockerfile index 5e9acc4778..b5d7adb749 100644 --- a/bdd/php/Dockerfile +++ b/bdd/php/Dockerfile @@ -45,8 +45,6 @@ ENV RUSTUP_HOME=/usr/local/rustup ENV PATH=/usr/local/cargo/bin:$PATH ENV PHP=/usr/bin/php8.3 ENV PHP_CONFIG=/usr/bin/php-config8.3 -ENV IGGY_HOST=iggy-server -ENV IGGY_PORT=8090 ENV BDD_FEATURE_FILE=/workspace/bdd/scenarios/basic_messaging.feature RUN rustup component add llvm-tools-preview \ diff --git a/bdd/php/tests/BasicMessagingFeatureTest.php b/bdd/php/tests/BasicMessagingFeatureTest.php index d47b09f609..8c4f8b068b 100644 --- a/bdd/php/tests/BasicMessagingFeatureTest.php +++ b/bdd/php/tests/BasicMessagingFeatureTest.php @@ -19,6 +19,7 @@ declare(strict_types=1); require_once __DIR__ . '/SharedFeatureParser.php'; +require_once __DIR__ . '/TestEnvironment.php'; use Iggy\Client as IggyClient; use Iggy\PollingStrategy; @@ -71,7 +72,7 @@ public static function scenarioCases(): array private function runStep(string $step): void { if ($step === 'I have a running Iggy server') { - $this->client = new IggyClient(server_host() . ':' . server_port()); + $this->client = new IggyClient(TestEnvironment::serverAddress()); $this->client->connect(); $this->client->ping(); @@ -79,7 +80,7 @@ private function runStep(string $step): void } if ($step === 'I am authenticated as the root user') { - $this->requireClient()->loginUser(env_or_default('IGGY_USERNAME', 'iggy'), env_or_default('IGGY_PASSWORD', 'iggy')); + $this->requireClient()->loginUser(TestEnvironment::rootUsername(), TestEnvironment::rootPassword()); return; } diff --git a/bdd/php/tests/RawCommandFeatureTest.php b/bdd/php/tests/RawCommandFeatureTest.php index db9aeb8f87..181ca9bcdb 100644 --- a/bdd/php/tests/RawCommandFeatureTest.php +++ b/bdd/php/tests/RawCommandFeatureTest.php @@ -19,6 +19,7 @@ declare(strict_types=1); require_once __DIR__ . '/SharedFeatureParser.php'; +require_once __DIR__ . '/TestEnvironment.php'; use Iggy\Client as IggyClient; use Iggy\Exception\IggyException; @@ -52,15 +53,15 @@ public static function scenarioCases(): array private function runStep(string $step): void { if ($step === 'I have a running Iggy server') { - $this->client = new IggyClient(server_host() . ':' . server_port()); + $this->client = new IggyClient(TestEnvironment::serverAddress()); $this->client->connect(); $this->client->ping(); return; } if ($step === 'I am authenticated as the root user') { $this->requireClient()->loginUser( - env_or_default('IGGY_USERNAME', 'iggy'), - env_or_default('IGGY_PASSWORD', 'iggy'), + TestEnvironment::rootUsername(), + TestEnvironment::rootPassword(), ); return; } diff --git a/bdd/php/tests/TestEnvironment.php b/bdd/php/tests/TestEnvironment.php new file mode 100644 index 0000000000..3c999fe2bc --- /dev/null +++ b/bdd/php/tests/TestEnvironment.php @@ -0,0 +1,53 @@ + str: + """Read a variable the suite cannot run without. + + A default here would turn a dropped compose variable into a run against + whatever happens to listen on the fallback address, so a missing value + aborts the suite instead. + """ + value = os.environ.get(name) + if not value: + raise RuntimeError( + f"{name} must be set; run the suite via scripts/run-bdd-tests.sh" + ) + return value + + +@pytest.fixture(scope="session") +def root_credentials() -> tuple[str, str]: + """Root username and password the server was started with.""" + return required_env("IGGY_ROOT_USERNAME"), required_env("IGGY_ROOT_PASSWORD") + + @pytest.fixture(scope="session") def event_loop(): """Create an instance of the default event loop for the test session.""" @@ -57,7 +78,6 @@ def context(): """Create a fresh context for each test scenario.""" ctx = GlobalContext() - # Get server address from environment or use default - ctx.server_addr = os.environ.get("IGGY_TCP_ADDRESS", "127.0.0.1:8090") + ctx.server_addr = required_env("IGGY_TCP_ADDRESS") yield ctx diff --git a/bdd/python/tests/test_basic_messaging.py b/bdd/python/tests/test_basic_messaging.py index cbaf3869a4..b34d0cdc79 100644 --- a/bdd/python/tests/test_basic_messaging.py +++ b/bdd/python/tests/test_basic_messaging.py @@ -52,11 +52,11 @@ async def _connect(): @given("I am authenticated as the root user") -def authenticated_root_user(context): +def authenticated_root_user(context, root_credentials): """Authenticate as root user""" async def _login(): - await context.client.login_user("iggy", "iggy") + await context.client.login_user(*root_credentials) asyncio.run(_login()) diff --git a/bdd/python/tests/test_raw_command.py b/bdd/python/tests/test_raw_command.py index cc0842530b..e2b040d30b 100644 --- a/bdd/python/tests/test_raw_command.py +++ b/bdd/python/tests/test_raw_command.py @@ -41,9 +41,9 @@ async def connect(): @given("I am authenticated as the root user") -def authenticated_root_user(context): +def authenticated_root_user(context, root_credentials): async def login(): - await context.client.login_user("iggy", "iggy") + await context.client.login_user(*root_credentials) asyncio.run(login()) diff --git a/bdd/rust/tests/helpers/cluster.rs b/bdd/rust/tests/helpers/cluster.rs index 8a954c22af..db636586ef 100644 --- a/bdd/rust/tests/helpers/cluster.rs +++ b/bdd/rust/tests/helpers/cluster.rs @@ -15,23 +15,18 @@ // specific language governing permissions and limitations // under the License. +use crate::helpers::env::{follower_address, leader_address, server_address}; use iggy::prelude::*; -use std::env; use std::net::{SocketAddr, ToSocketAddrs}; use std::sync::Arc; -/// Resolves server address based on role and port, checking environment variables first +/// Resolves the server address for a role and port from the environment pub fn resolve_server_address(role: &str, port: u16) -> String { match (role.to_lowercase().as_str(), port) { - ("leader", 8091) => { - env::var("IGGY_TCP_ADDRESS_LEADER").unwrap_or_else(|_| "iggy-leader:8091".to_string()) - } - ("follower", 8092) => env::var("IGGY_TCP_ADDRESS_FOLLOWER") - .unwrap_or_else(|_| "iggy-follower:8092".to_string()), - ("single", 8090) | (_, 8090) => { - env::var("IGGY_TCP_ADDRESS").unwrap_or_else(|_| "iggy-server:8090".to_string()) - } - _ => format!("iggy-server:{}", port), + ("leader", 8091) => leader_address(), + ("follower", 8092) => follower_address(), + (_, 8090) => server_address(), + _ => panic!("no address mapping for role '{role}' on port {port}"), } } diff --git a/bdd/rust/tests/helpers/env.rs b/bdd/rust/tests/helpers/env.rs new file mode 100644 index 0000000000..75c65683a0 --- /dev/null +++ b/bdd/rust/tests/helpers/env.rs @@ -0,0 +1,50 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use std::env; + +/// Reads a variable the suite cannot run without. +/// +/// A default here would turn a dropped compose variable into a run against +/// whatever happens to listen on the fallback address, so a missing value +/// aborts the suite instead. +fn required_env(name: &str) -> String { + match env::var(name) { + Ok(value) if !value.is_empty() => value, + _ => panic!("{name} must be set; run the suite via scripts/run-bdd-tests.sh"), + } +} + +pub fn server_address() -> String { + required_env("IGGY_TCP_ADDRESS") +} + +pub fn leader_address() -> String { + required_env("IGGY_TCP_ADDRESS_LEADER") +} + +pub fn follower_address() -> String { + required_env("IGGY_TCP_ADDRESS_FOLLOWER") +} + +pub fn root_username() -> String { + required_env("IGGY_ROOT_USERNAME") +} + +pub fn root_password() -> String { + required_env("IGGY_ROOT_PASSWORD") +} diff --git a/bdd/rust/tests/helpers/mod.rs b/bdd/rust/tests/helpers/mod.rs index 26173fc631..7521d25264 100644 --- a/bdd/rust/tests/helpers/mod.rs +++ b/bdd/rust/tests/helpers/mod.rs @@ -16,4 +16,5 @@ // under the License. pub mod cluster; +pub mod env; pub mod test_data; diff --git a/bdd/rust/tests/steps/auth.rs b/bdd/rust/tests/steps/auth.rs index f5fc554bee..bca0c06876 100644 --- a/bdd/rust/tests/steps/auth.rs +++ b/bdd/rust/tests/steps/auth.rs @@ -16,6 +16,7 @@ // under the License. use crate::common::global_context::GlobalContext; +use crate::helpers::env::{root_password, root_username}; use cucumber::given; use iggy::prelude::*; use std::sync::Arc; @@ -42,7 +43,7 @@ pub async fn given_authenticated_as_root(world: &mut GlobalContext) { client.ping().await.expect("Server should respond to ping"); client - .login_user(DEFAULT_ROOT_USERNAME, DEFAULT_ROOT_PASSWORD) + .login_user(&root_username(), &root_password()) .await .expect("Failed to login as root"); diff --git a/bdd/rust/tests/steps/leader_redirection.rs b/bdd/rust/tests/steps/leader_redirection.rs index c035bd848b..dd0f4f02c9 100644 --- a/bdd/rust/tests/steps/leader_redirection.rs +++ b/bdd/rust/tests/steps/leader_redirection.rs @@ -17,6 +17,7 @@ use crate::common::leader_context::LeaderContext; use crate::helpers::cluster; +use crate::helpers::env::{root_password, root_username}; use cucumber::{given, then, when}; use iggy::prelude::*; use std::time::Duration; @@ -160,7 +161,7 @@ async fn when_authenticate_root(world: &mut LeaderContext) { .unwrap_or_else(|| panic!("Client {} should be created", client_name)); client - .login_user(DEFAULT_ROOT_USERNAME, DEFAULT_ROOT_PASSWORD) + .login_user(&root_username(), &root_password()) .await .expect("Failed to login as root"); diff --git a/bdd/rust/tests/steps/server.rs b/bdd/rust/tests/steps/server.rs index d621b699cc..73e8b7491e 100644 --- a/bdd/rust/tests/steps/server.rs +++ b/bdd/rust/tests/steps/server.rs @@ -16,12 +16,11 @@ // under the License. use crate::common::global_context::GlobalContext; +use crate::helpers::env::server_address; use cucumber::given; #[given("I have a running Iggy server")] pub async fn given_running_server(world: &mut GlobalContext) { // External server mode - connect to server from environment - let server_addr = - std::env::var("IGGY_TCP_ADDRESS").unwrap_or_else(|_| "localhost:8090".to_string()); - world.server_addr = Some(server_addr); + world.server_addr = Some(server_address()); } diff --git a/foreign/csharp/Iggy_SDK.Tests.BDD/Context/TestContext.cs b/foreign/csharp/Iggy_SDK.Tests.BDD/Context/TestContext.cs index 81c12f2e82..27a08f5555 100644 --- a/foreign/csharp/Iggy_SDK.Tests.BDD/Context/TestContext.cs +++ b/foreign/csharp/Iggy_SDK.Tests.BDD/Context/TestContext.cs @@ -24,9 +24,9 @@ namespace Apache.Iggy.Tests.BDD.Context; public class TestContext { public IIggyClient IggyClient { get; set; } = null!; - public string TcpUrl { get; set; } = string.Empty; - public string LeaderTcpUrl { get; set; } = string.Empty; - public string FollowerTcpUrl { get; set; } = string.Empty; + public string TcpUrl => TestEnvironment.TcpAddress; + public string LeaderTcpUrl => TestEnvironment.LeaderTcpAddress; + public string FollowerTcpUrl => TestEnvironment.FollowerTcpAddress; public Dictionary Clients { get; } = new(); public StreamResponse? CreatedStream { get; set; } public TopicResponse? CreatedTopic { get; set; } diff --git a/foreign/csharp/Iggy_SDK.Tests.BDD/Context/TestEnvironment.cs b/foreign/csharp/Iggy_SDK.Tests.BDD/Context/TestEnvironment.cs new file mode 100644 index 0000000000..cfb89a9643 --- /dev/null +++ b/foreign/csharp/Iggy_SDK.Tests.BDD/Context/TestEnvironment.cs @@ -0,0 +1,44 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +namespace Apache.Iggy.Tests.BDD.Context; + +///

+/// Endpoints and credentials the BDD suite runs against. A default here would turn a dropped +/// compose variable into a run against whatever happens to listen on the fallback address, so a +/// missing value aborts the suite instead. +/// +public static class TestEnvironment +{ + public static string TcpAddress => Require("IGGY_TCP_ADDRESS"); + public static string LeaderTcpAddress => Require("IGGY_TCP_ADDRESS_LEADER"); + public static string FollowerTcpAddress => Require("IGGY_TCP_ADDRESS_FOLLOWER"); + public static string RootUsername => Require("IGGY_ROOT_USERNAME"); + public static string RootPassword => Require("IGGY_ROOT_PASSWORD"); + + private static string Require(string name) + { + var value = Environment.GetEnvironmentVariable(name); + if (string.IsNullOrWhiteSpace(value)) + { + throw new InvalidOperationException( + $"{name} must be set; run the suite via scripts/run-bdd-tests.sh"); + } + + return value; + } +} diff --git a/foreign/csharp/Iggy_SDK.Tests.BDD/Context/TestHooks.cs b/foreign/csharp/Iggy_SDK.Tests.BDD/Context/TestHooks.cs index 58c5debdb6..d63f2cd6b1 100644 --- a/foreign/csharp/Iggy_SDK.Tests.BDD/Context/TestHooks.cs +++ b/foreign/csharp/Iggy_SDK.Tests.BDD/Context/TestHooks.cs @@ -32,9 +32,6 @@ public TestHooks(TestContext context) [BeforeScenario] public void BeforeScenario() { - _context.TcpUrl = Environment.GetEnvironmentVariable("IGGY_TCP_ADDRESS") ?? "127.0.0.1:8090"; - _context.LeaderTcpUrl = Environment.GetEnvironmentVariable("IGGY_TCP_ADDRESS_LEADER") ?? "127.0.0.1:8091"; - _context.FollowerTcpUrl = Environment.GetEnvironmentVariable("IGGY_TCP_ADDRESS_FOLLOWER") ?? "127.0.0.1:8092"; _context.Clients.Clear(); _context.CreatedStream = null; _context.RedirectionOccurred = false; diff --git a/foreign/csharp/Iggy_SDK.Tests.BDD/README.md b/foreign/csharp/Iggy_SDK.Tests.BDD/README.md index 78ed769703..bf1de4ed42 100644 --- a/foreign/csharp/Iggy_SDK.Tests.BDD/README.md +++ b/foreign/csharp/Iggy_SDK.Tests.BDD/README.md @@ -4,20 +4,35 @@ Scenario are located at [/bdd/scenarios](../../../bdd/scenarios) ## env var -use env var `IGGY_TCP_ADDRESS="host:port"` to set expected server address for bdd test suite. +the bdd test suite has no defaults and fails when any of these is missing: -## Run via docker +- `IGGY_TCP_ADDRESS="host:port"` - server address +- `IGGY_ROOT_USERNAME` / `IGGY_ROOT_PASSWORD` - root credentials +- `IGGY_TCP_ADDRESS_LEADER` / `IGGY_TCP_ADDRESS_FOLLOWER` - cluster addresses, leader redirection scenarios only -see [/bdd/README.md](../../../bdd/README.md) +## Run (recommended) + +from the repository root run + +```bash +./scripts/run-bdd-tests.sh csharp +``` + +the script starts the server, brings up the leader and follower for the cluster +scenarios, and sets every variable above, so none of them have to be exported by +hand. see [/bdd/README.md](../../../bdd/README.md) for the sdk and feature +matrix. ## Run locally -note: bdd test expect an iggy-server at tcp://127.0.0.1:8090 +for iterating against a server you started yourself. note: bdd test expect an +iggy-server started with the same root credentials from [/foreign/csharp/Iggy_SDK.Tests.BDD](.) run ```bash -dotnet test +IGGY_TCP_ADDRESS=127.0.0.1:8090 IGGY_ROOT_USERNAME=iggy IGGY_ROOT_PASSWORD=iggy \ + dotnet test ``` ## Troubleshooting diff --git a/foreign/csharp/Iggy_SDK.Tests.BDD/StepDefinitions/BasicMessagingOperationsSteps.cs b/foreign/csharp/Iggy_SDK.Tests.BDD/StepDefinitions/BasicMessagingOperationsSteps.cs index 3012d3e995..f0f44b539a 100644 --- a/foreign/csharp/Iggy_SDK.Tests.BDD/StepDefinitions/BasicMessagingOperationsSteps.cs +++ b/foreign/csharp/Iggy_SDK.Tests.BDD/StepDefinitions/BasicMessagingOperationsSteps.cs @@ -26,6 +26,7 @@ using Shouldly; using Partitioning = Apache.Iggy.Kinds.Partitioning; using TestContext = Apache.Iggy.Tests.BDD.Context.TestContext; +using TestEnvironment = Apache.Iggy.Tests.BDD.Context.TestEnvironment; namespace Apache.Iggy.Tests.BDD.StepDefinitions; @@ -55,7 +56,7 @@ public async Task GivenIHaveARunningIggyServer() [Given(@"I am authenticated as the root user")] public async Task GivenIAmAuthenticatedAsTheRootUser() { - var loginResult = await _context.IggyClient.LoginUserAsync("iggy", "iggy"); + var loginResult = await _context.IggyClient.LoginUserAsync(TestEnvironment.RootUsername, TestEnvironment.RootPassword); loginResult.ShouldNotBeNull(); loginResult.UserId.ShouldBe(0); diff --git a/foreign/csharp/Iggy_SDK.Tests.BDD/StepDefinitions/LeaderRedirectionSteps.cs b/foreign/csharp/Iggy_SDK.Tests.BDD/StepDefinitions/LeaderRedirectionSteps.cs index a02934d03c..e23d2fc87f 100644 --- a/foreign/csharp/Iggy_SDK.Tests.BDD/StepDefinitions/LeaderRedirectionSteps.cs +++ b/foreign/csharp/Iggy_SDK.Tests.BDD/StepDefinitions/LeaderRedirectionSteps.cs @@ -25,15 +25,13 @@ using Reqnroll; using Shouldly; using TestContext = Apache.Iggy.Tests.BDD.Context.TestContext; +using TestEnvironment = Apache.Iggy.Tests.BDD.Context.TestEnvironment; namespace Apache.Iggy.Tests.BDD.StepDefinitions; [Binding] public class LeaderRedirectionSteps { - private const string RootUsername = "iggy"; - private const string RootPassword = "iggy"; - private readonly TestContext _context; public LeaderRedirectionSteps(TestContext context) @@ -141,7 +139,7 @@ private async Task AuthenticateAllClients() { var client = GetClient(name); var initialAddress = client.GetCurrentAddress(); - var result = await client.LoginUserAsync(RootUsername, RootPassword); + var result = await client.LoginUserAsync(TestEnvironment.RootUsername, TestEnvironment.RootPassword); result.ShouldNotBeNull("Failed to login as root"); // RedirectAsync runs inside LoginUserAsync; if address changed, redirection happened. diff --git a/foreign/node/README.md b/foreign/node/README.md index 5c0f9052fc..dfd315edb7 100644 --- a/foreign/node/README.md +++ b/foreign/node/README.md @@ -161,7 +161,7 @@ npm run build ### test note: use env var `IGGY_TCP_ADDRESS="host:port"` to set the server -address for bdd and e2e tests. +address for e2e tests. bdd tests need more variables, see below. #### unit tests @@ -179,15 +179,27 @@ npm run test:e2e #### bdd tests -bdd test expect an iggy-server at tcp://127.0.0.1:8090 +the bdd suite has no defaults and fails when `IGGY_TCP_ADDRESS`, +`IGGY_ROOT_USERNAME` or `IGGY_ROOT_PASSWORD` is missing. from the repository +root run ```bash -npm run test:bdd +./scripts/run-bdd-tests.sh node ``` +the script starts the server and sets every variable, so none of them have to be +exported by hand. to iterate against a server you started yourself, see +[src/bdd/README.md](./src/bdd/README.md). + #### run all test -`npm run test` runs unit, bdd and e2e tests suite (expect an iggy-server at tcp://127.0.0.1:8090) +`npm run test` runs unit, bdd and e2e tests suite against an iggy-server at +tcp://127.0.0.1:8090, started with the same root credentials + +```bash +IGGY_TCP_ADDRESS=127.0.0.1:8090 IGGY_ROOT_USERNAME=iggy IGGY_ROOT_PASSWORD=iggy \ + npm run test +``` ### lint diff --git a/foreign/node/src/bdd/README.md b/foreign/node/src/bdd/README.md index b7ec9eac04..4bdd5c81b3 100644 --- a/foreign/node/src/bdd/README.md +++ b/foreign/node/src/bdd/README.md @@ -6,19 +6,32 @@ scenario are located at [/bdd/scenarios](../../../../bdd/scenarios) ## env var -use env var `IGGY_TCP_ADDRESS="host:port"` to set expected server address for bdd test suite. +the bdd test suite has no defaults and fails when any of these is missing: -## Run via docker +- `IGGY_TCP_ADDRESS="host:port"` - server address +- `IGGY_ROOT_USERNAME` / `IGGY_ROOT_PASSWORD` - root credentials -see [/bdd/README.md](../../../../bdd/README.md) +## Run (recommended) + +from the repository root run + +```bash +./scripts/run-bdd-tests.sh node +``` + +the script starts the server and sets every variable above, so none of them have +to be exported by hand. see [/bdd/README.md](../../../../bdd/README.md) for the +sdk and feature matrix. ## Run locally -note: bdd test expect an iggy-server at tcp://127.0.0.1:8090 +for iterating against a server you started yourself. note: bdd test expect an +iggy-server started with the same root credentials from [/foreign/node](../../) run ```bash npm ci # if not already done -npm run test:bdd +IGGY_TCP_ADDRESS=127.0.0.1:8090 IGGY_ROOT_USERNAME=iggy IGGY_ROOT_PASSWORD=iggy \ + npm run test:bdd ``` diff --git a/foreign/node/src/bdd/auth.ts b/foreign/node/src/bdd/auth.ts index 2212893091..52b042677a 100644 --- a/foreign/node/src/bdd/auth.ts +++ b/foreign/node/src/bdd/auth.ts @@ -19,10 +19,10 @@ import assert from 'node:assert/strict'; import { Client } from '../client/index.js'; import { Given } from "@cucumber/cucumber"; import type { TestWorld } from './world.js'; -import { getIggyAddress } from '../tcp.sm.utils.js'; +import { getRootCredentials, getServerAddress } from './env.js'; -const credentials = { username: 'iggy', password: 'iggy' }; -const [host, port] = getIggyAddress(); +const credentials = getRootCredentials(); +const [host, port] = getServerAddress(); const opt = { transport: 'TCP' as const, diff --git a/foreign/node/src/bdd/env.ts b/foreign/node/src/bdd/env.ts new file mode 100644 index 0000000000..2c8bfc288f --- /dev/null +++ b/foreign/node/src/bdd/env.ts @@ -0,0 +1,39 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +// A default here would turn a dropped compose variable into a run against +// whatever happens to listen on the fallback address, so a missing value +// aborts the suite instead. +const requiredEnv = (name: string): string => { + const value = process.env[name]; + if (!value) + throw new Error(`${name} must be set; run the suite via scripts/run-bdd-tests.sh`); + return value; +}; + +export const getServerAddress = (): [string, number] => { + const address = requiredEnv('IGGY_TCP_ADDRESS'); + const [host, port] = address.split(':'); + if (!host || !port) + throw new Error(`IGGY_TCP_ADDRESS must be "host:port", got "${address}"`); + return [host, parseInt(port, 10)]; +}; + +export const getRootCredentials = () => ({ + username: requiredEnv('IGGY_ROOT_USERNAME'), + password: requiredEnv('IGGY_ROOT_PASSWORD') +}); diff --git a/foreign/php/Dockerfile.test b/foreign/php/Dockerfile.test index 7c9a74173f..e5ce15a7da 100644 --- a/foreign/php/Dockerfile.test +++ b/foreign/php/Dockerfile.test @@ -36,8 +36,7 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ ENV PHP=/usr/bin/php ENV PHP_CONFIG=/usr/bin/php-config -ENV IGGY_HOST=iggy-server -ENV IGGY_PORT=8090 +ENV IGGY_TCP_ADDRESS=iggy-server:8090 WORKDIR /workspace diff --git a/foreign/php/README.md b/foreign/php/README.md index d6bd9de37a..795b3bebbf 100644 --- a/foreign/php/README.md +++ b/foreign/php/README.md @@ -81,7 +81,7 @@ The tests assume: - username: `iggy` - password: `iggy` -Override them with `IGGY_HOST`, `IGGY_PORT`, `IGGY_USERNAME`, and `IGGY_PASSWORD`. +Override them with `IGGY_TCP_ADDRESS`, `IGGY_ROOT_USERNAME`, and `IGGY_ROOT_PASSWORD`. ## Usage diff --git a/foreign/php/docker-compose.test.yml b/foreign/php/docker-compose.test.yml index 19abcc2178..c80c759d0b 100644 --- a/foreign/php/docker-compose.test.yml +++ b/foreign/php/docker-compose.test.yml @@ -62,10 +62,9 @@ services: networks: - php-test-network environment: - - IGGY_HOST=iggy-server - - IGGY_PORT=8090 - - IGGY_USERNAME=iggy - - IGGY_PASSWORD=iggy + - IGGY_TCP_ADDRESS=iggy-server:8090 + - IGGY_ROOT_USERNAME=iggy + - IGGY_ROOT_PASSWORD=iggy volumes: - ./test-results:/workspace/foreign/php/test-results diff --git a/foreign/php/scripts/test.sh b/foreign/php/scripts/test.sh index effe9e7a06..f01ec8516b 100755 --- a/foreign/php/scripts/test.sh +++ b/foreign/php/scripts/test.sh @@ -21,12 +21,13 @@ set -euo pipefail echo "PHP SDK Test Runner" echo "===================" -IGGY_HOST="${IGGY_HOST:-127.0.0.1}" -IGGY_PORT="${IGGY_PORT:-8090}" +IGGY_TCP_ADDRESS="${IGGY_TCP_ADDRESS:-127.0.0.1:8090}" +host="${IGGY_TCP_ADDRESS%%:*}" +port="${IGGY_TCP_ADDRESS##*:}" -echo "Waiting for Iggy server at ${IGGY_HOST}:${IGGY_PORT}..." +echo "Waiting for Iggy server at ${host}:${port}..." timeout 60 bash -c " - until timeout 5 bash -c 'connect(); - $client->loginUser(env_or_default('IGGY_USERNAME', 'iggy'), env_or_default('IGGY_PASSWORD', 'iggy')); + $client->loginUser(env_or_default('IGGY_ROOT_USERNAME', 'iggy'), env_or_default('IGGY_ROOT_PASSWORD', 'iggy')); $client->ping(); assert_true($client instanceof IggyClient); diff --git a/foreign/php/tests/bootstrap.php b/foreign/php/tests/bootstrap.php index de0866f691..88f63f86e7 100644 --- a/foreign/php/tests/bootstrap.php +++ b/foreign/php/tests/bootstrap.php @@ -84,14 +84,19 @@ function env_or_default(string $name, string $default): string return $value === false || $value === '' ? $default : $value; } +function server_address(): string +{ + return env_or_default('IGGY_TCP_ADDRESS', '127.0.0.1:8090'); +} + function server_host(): string { - return env_or_default('IGGY_HOST', '127.0.0.1'); + return explode(':', server_address(), 2)[0]; } function server_port(): int { - return (int) env_or_default('IGGY_PORT', '8090'); + return (int) (explode(':', server_address(), 2)[1] ?? '8090'); } function wait_for_server(string $host, int $port, int $timeoutSeconds = 30): void @@ -116,9 +121,9 @@ function wait_for_server(string $host, int $port, int $timeoutSeconds = 30): voi function new_client(): IggyClient { - $client = new IggyClient(server_host() . ':' . server_port()); + $client = new IggyClient(server_address()); $client->connect(); - $client->loginUser(env_or_default('IGGY_USERNAME', 'iggy'), env_or_default('IGGY_PASSWORD', 'iggy')); + $client->loginUser(env_or_default('IGGY_ROOT_USERNAME', 'iggy'), env_or_default('IGGY_ROOT_PASSWORD', 'iggy')); return $client; } @@ -127,8 +132,8 @@ function new_connection_string_client(): IggyClient { $host = server_host(); $port = server_port(); - $username = rawurlencode(env_or_default('IGGY_USERNAME', 'iggy')); - $password = rawurlencode(env_or_default('IGGY_PASSWORD', 'iggy')); + $username = rawurlencode(env_or_default('IGGY_ROOT_USERNAME', 'iggy')); + $password = rawurlencode(env_or_default('IGGY_ROOT_PASSWORD', 'iggy')); $client = IggyClient::fromConnectionString("iggy+tcp://{$username}:{$password}@{$host}:{$port}"); $client->connect(); From e80992b6382a0a314d378454fc7332a2dda96ea6 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Mon, 31 Aug 2026 13:44:35 +0200 Subject: [PATCH 019/182] chore(deps): Bump the java group across 3 directories with 17 updates (#3966) --- bdd/java/Dockerfile | 2 +- bdd/java/build.gradle.kts | 10 +++++----- .../gradle/wrapper/gradle-wrapper.properties | 2 +- bdd/java/gradlew | 4 ++-- examples/java/build.gradle.kts | 4 ++-- .../gradle/wrapper/gradle-wrapper.properties | 2 +- examples/java/gradlew | 4 ++-- foreign/java/gradle/libs.versions.toml | 16 ++++++++-------- .../gradle/wrapper/gradle-wrapper.properties | 2 +- foreign/java/gradlew | 4 ++-- 10 files changed, 25 insertions(+), 25 deletions(-) diff --git a/bdd/java/Dockerfile b/bdd/java/Dockerfile index 63ee0cbb99..4fb9296697 100644 --- a/bdd/java/Dockerfile +++ b/bdd/java/Dockerfile @@ -15,7 +15,7 @@ # specific language governing permissions and limitations # under the License. -FROM gradle:9.7.0-jdk17 +FROM gradle:9.7.1-jdk17 WORKDIR /workspace COPY . . diff --git a/bdd/java/build.gradle.kts b/bdd/java/build.gradle.kts index f62ec9c44a..7f673b1f42 100644 --- a/bdd/java/build.gradle.kts +++ b/bdd/java/build.gradle.kts @@ -20,7 +20,7 @@ plugins { java jacoco - id("com.diffplug.spotless") version "8.9.0" + id("com.diffplug.spotless") version "8.10.0" } repositories { @@ -29,10 +29,10 @@ repositories { dependencies { testImplementation("org.apache.iggy:iggy") - testImplementation("io.cucumber:cucumber-java:7.34.6") - testImplementation("io.cucumber:cucumber-junit-platform-engine:7.34.6") - testImplementation("org.junit.jupiter:junit-jupiter:6.1.2") - testImplementation("org.junit.platform:junit-platform-suite:6.1.2") + testImplementation("io.cucumber:cucumber-java:7.34.7") + testImplementation("io.cucumber:cucumber-junit-platform-engine:7.34.7") + testImplementation("org.junit.jupiter:junit-jupiter:6.1.3") + testImplementation("org.junit.platform:junit-platform-suite:6.1.3") } spotless { diff --git a/bdd/java/gradle/wrapper/gradle-wrapper.properties b/bdd/java/gradle/wrapper/gradle-wrapper.properties index a9db11550c..ad7845be30 100644 --- a/bdd/java/gradle/wrapper/gradle-wrapper.properties +++ b/bdd/java/gradle/wrapper/gradle-wrapper.properties @@ -1,6 +1,6 @@ distributionBase=GRADLE_USER_HOME distributionPath=wrapper/dists -distributionUrl=https\://services.gradle.org/distributions/gradle-9.6.1-bin.zip +distributionUrl=https\://services.gradle.org/distributions/gradle-9.7.1-bin.zip networkTimeout=10000 retries=0 retryBackOffMs=500 diff --git a/bdd/java/gradlew b/bdd/java/gradlew index f8de29016f..f14608ef47 100755 --- a/bdd/java/gradlew +++ b/bdd/java/gradlew @@ -112,8 +112,8 @@ die () { # ############################################################################## -GRADLE_WRAPPER_VERSION=9.6.1 -REQUIRED_WRAPPER_JAR_CHECKSUM="497c8c2a7e5031f6aa847f88104aa80a93532ec32ee17bdb8d1d2f67a194a9c7" +GRADLE_WRAPPER_VERSION=9.7.1 +REQUIRED_WRAPPER_JAR_CHECKSUM="7a9ce74cff467ca1bf60a4fcd9f05185acceda4d0f382434d393e17864262c5d" for _ in 1 2 3; do if [ ! -e "$APP_HOME/gradle/wrapper/gradle-wrapper.jar" ]; then diff --git a/examples/java/build.gradle.kts b/examples/java/build.gradle.kts index ddc6e3c8bf..91c8e6fec1 100644 --- a/examples/java/build.gradle.kts +++ b/examples/java/build.gradle.kts @@ -19,7 +19,7 @@ plugins { java - id("com.diffplug.spotless") version "8.9.0" + id("com.diffplug.spotless") version "8.10.0" } repositories { @@ -29,7 +29,7 @@ repositories { dependencies { implementation("org.apache.iggy:iggy:local-dev") implementation("org.slf4j:slf4j-simple:2.0.18") - implementation("tools.jackson.core:jackson-databind:3.2.1") + implementation("tools.jackson.core:jackson-databind:3.2.2") } spotless { diff --git a/examples/java/gradle/wrapper/gradle-wrapper.properties b/examples/java/gradle/wrapper/gradle-wrapper.properties index a9db11550c..ad7845be30 100644 --- a/examples/java/gradle/wrapper/gradle-wrapper.properties +++ b/examples/java/gradle/wrapper/gradle-wrapper.properties @@ -1,6 +1,6 @@ distributionBase=GRADLE_USER_HOME distributionPath=wrapper/dists -distributionUrl=https\://services.gradle.org/distributions/gradle-9.6.1-bin.zip +distributionUrl=https\://services.gradle.org/distributions/gradle-9.7.1-bin.zip networkTimeout=10000 retries=0 retryBackOffMs=500 diff --git a/examples/java/gradlew b/examples/java/gradlew index f8de29016f..f14608ef47 100755 --- a/examples/java/gradlew +++ b/examples/java/gradlew @@ -112,8 +112,8 @@ die () { # ############################################################################## -GRADLE_WRAPPER_VERSION=9.6.1 -REQUIRED_WRAPPER_JAR_CHECKSUM="497c8c2a7e5031f6aa847f88104aa80a93532ec32ee17bdb8d1d2f67a194a9c7" +GRADLE_WRAPPER_VERSION=9.7.1 +REQUIRED_WRAPPER_JAR_CHECKSUM="7a9ce74cff467ca1bf60a4fcd9f05185acceda4d0f382434d393e17864262c5d" for _ in 1 2 3; do if [ ! -e "$APP_HOME/gradle/wrapper/gradle-wrapper.jar" ]; then diff --git a/foreign/java/gradle/libs.versions.toml b/foreign/java/gradle/libs.versions.toml index 2cf5ce3d79..5e01cdcf06 100644 --- a/foreign/java/gradle/libs.versions.toml +++ b/foreign/java/gradle/libs.versions.toml @@ -23,29 +23,29 @@ flink = "2.3.0" pinot = "1.5.1" # Jackson -jackson = "3.2.1" -jackson2 = "2.22.1" +jackson = "3.2.2" +jackson2 = "2.22.2" # Apache Commons commons-lang3 = "3.20.0" # Hashing -hash4j = "0.22.0" +hash4j = "0.30.0" # HTTP Client -httpclient5 = "5.6.3" +httpclient5 = "5.6.4" # Logging slf4j = "2.0.18" -logback = "1.6.1" +logback = "1.6.3" # Testing -junit = "6.1.2" +junit = "6.1.3" assertj = "3.27.7" testcontainers = "2.0.5" # Netty -netty = "4.2.16.Final" +netty = "4.2.17.Final" # Spotbugs spotbugs = "4.10.3" @@ -55,7 +55,7 @@ typesafe-config = "1.4.9" picocli = "4.7.7" # Build plugins -spotless = "8.9.0" +spotless = "8.10.0" shadow = "9.6.1" checkstyle = "12.3.1" jacoco = "0.8.15" diff --git a/foreign/java/gradle/wrapper/gradle-wrapper.properties b/foreign/java/gradle/wrapper/gradle-wrapper.properties index a9db11550c..ad7845be30 100644 --- a/foreign/java/gradle/wrapper/gradle-wrapper.properties +++ b/foreign/java/gradle/wrapper/gradle-wrapper.properties @@ -1,6 +1,6 @@ distributionBase=GRADLE_USER_HOME distributionPath=wrapper/dists -distributionUrl=https\://services.gradle.org/distributions/gradle-9.6.1-bin.zip +distributionUrl=https\://services.gradle.org/distributions/gradle-9.7.1-bin.zip networkTimeout=10000 retries=0 retryBackOffMs=500 diff --git a/foreign/java/gradlew b/foreign/java/gradlew index f8de29016f..f14608ef47 100755 --- a/foreign/java/gradlew +++ b/foreign/java/gradlew @@ -112,8 +112,8 @@ die () { # ############################################################################## -GRADLE_WRAPPER_VERSION=9.6.1 -REQUIRED_WRAPPER_JAR_CHECKSUM="497c8c2a7e5031f6aa847f88104aa80a93532ec32ee17bdb8d1d2f67a194a9c7" +GRADLE_WRAPPER_VERSION=9.7.1 +REQUIRED_WRAPPER_JAR_CHECKSUM="7a9ce74cff467ca1bf60a4fcd9f05185acceda4d0f382434d393e17864262c5d" for _ in 1 2 3; do if [ ! -e "$APP_HOME/gradle/wrapper/gradle-wrapper.jar" ]; then From dc1cdf23209c91fd703b59ceb47bdb9b4c662480 Mon Sep 17 00:00:00 2001 From: haubur Date: Mon, 31 Aug 2026 15:46:05 +0200 Subject: [PATCH 020/182] fix(sdk): respect max_buffer_size when merging batches (#3933) Merging adjacent producer batches released their byte permits before the write completed, allowing buffered and in-flight data to exceed max_buffer_size. Large permit totals could also overflow during the merge and permanently reduce available capacity. Retain each permit separately until send_internal completes. Validate batch sizes before u32 conversion, split merged writes at the wire limit, and handle zero buffer and in-flight limits without invalid semaphores. Update max_buffer_size documentation and add regression coverage for permit lifetime, unlimited limits, and budgets above u32::MAX. Keep the edge.6 package versions synchronized. Fixes #3932 --- Cargo.lock | 14 +- Cargo.toml | 8 +- bdd/python/uv.lock | 2 +- core/ai/mcp/Cargo.toml | 2 +- core/bench/Cargo.toml | 2 +- core/binary_protocol/Cargo.toml | 2 +- core/cli/Cargo.toml | 2 +- core/common/Cargo.toml | 2 +- core/connectors/runtime/Cargo.toml | 2 +- core/sdk/Cargo.toml | 2 +- core/sdk/src/clients/producer_config.rs | 3 +- core/sdk/src/clients/producer_dispatcher.rs | 146 +++++++--- core/sdk/src/clients/producer_sharding.rs | 282 +++++++++++++++++--- examples/python/uv.lock | 2 +- foreign/python/Cargo.toml | 4 +- foreign/python/pyproject.toml | 2 +- foreign/python/uv.lock | 2 +- 17 files changed, 376 insertions(+), 103 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index c6aa065741..251f7f309a 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -6659,7 +6659,7 @@ checksum = "cd62e6b5e86ea8eeeb8db1de02880a6abc01a397b2ebb64b5d74ac255318f5cb" [[package]] name = "iggy" -version = "0.11.0-edge.5" +version = "0.11.0-edge.6" dependencies = [ "async-broadcast", "async-dropper", @@ -6693,7 +6693,7 @@ dependencies = [ [[package]] name = "iggy-bench" -version = "0.6.0-edge.5" +version = "0.6.0-edge.6" dependencies = [ "async-trait", "bench-report", @@ -6750,7 +6750,7 @@ dependencies = [ [[package]] name = "iggy-cli" -version = "0.14.0-edge.5" +version = "0.14.0-edge.6" dependencies = [ "anyhow", "apple-native-keyring-store", @@ -6784,7 +6784,7 @@ dependencies = [ [[package]] name = "iggy-connectors" -version = "0.5.0-edge.5" +version = "0.5.0-edge.6" dependencies = [ "async-trait", "axum", @@ -6856,7 +6856,7 @@ dependencies = [ [[package]] name = "iggy-mcp" -version = "0.5.0-edge.4" +version = "0.5.0-edge.5" dependencies = [ "axum", "axum-server", @@ -6890,7 +6890,7 @@ dependencies = [ [[package]] name = "iggy_binary_protocol" -version = "0.11.0-edge.5" +version = "0.11.0-edge.6" dependencies = [ "aligned-vec", "bytemuck", @@ -6903,7 +6903,7 @@ dependencies = [ [[package]] name = "iggy_common" -version = "0.11.0-edge.5" +version = "0.11.0-edge.6" dependencies = [ "aes-gcm", "async-broadcast", diff --git a/Cargo.toml b/Cargo.toml index ea2d476acb..7872f40937 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -206,10 +206,10 @@ hyper-util = { version = "0.1.20", features = ["server-auto", "service"] } iceberg = "0.9.1" iceberg-catalog-rest = "0.9.1" iceberg-storage-opendal = "0.9.1" -iggy = { path = "core/sdk", version = "0.11.0-edge.5" } -iggy-cli = { path = "core/cli", version = "0.14.0-edge.5" } -iggy_binary_protocol = { path = "core/binary_protocol", version = "0.11.0-edge.5" } -iggy_common = { path = "core/common", version = "0.11.0-edge.5" } +iggy = { path = "core/sdk", version = "0.11.0-edge.6" } +iggy-cli = { path = "core/cli", version = "0.14.0-edge.6" } +iggy_binary_protocol = { path = "core/binary_protocol", version = "0.11.0-edge.6" } +iggy_common = { path = "core/common", version = "0.11.0-edge.6" } iggy_connector_sdk = { path = "core/connectors/sdk", version = "0.4.0-edge.3" } indexmap = "2.14.0" integration = { path = "core/integration" } diff --git a/bdd/python/uv.lock b/bdd/python/uv.lock index b401a9ae2c..9198d276fa 100644 --- a/bdd/python/uv.lock +++ b/bdd/python/uv.lock @@ -8,7 +8,7 @@ exclude-newer-span = "P7D" [[package]] name = "apache-iggy" -version = "0.9.0.dev5" +version = "0.9.0.dev6" source = { directory = "../../foreign/python" } [package.metadata] diff --git a/core/ai/mcp/Cargo.toml b/core/ai/mcp/Cargo.toml index bab4769b3e..5386c5c4a0 100644 --- a/core/ai/mcp/Cargo.toml +++ b/core/ai/mcp/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "iggy-mcp" -version = "0.5.0-edge.4" +version = "0.5.0-edge.5" description = "MCP Server for Iggy message streaming platform" edition = "2024" license = "Apache-2.0" diff --git a/core/bench/Cargo.toml b/core/bench/Cargo.toml index 60e521d910..e2a3c8d8d0 100644 --- a/core/bench/Cargo.toml +++ b/core/bench/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "iggy-bench" -version = "0.6.0-edge.5" +version = "0.6.0-edge.6" edition = "2024" license = "Apache-2.0" repository = "https://github.com/apache/iggy" diff --git a/core/binary_protocol/Cargo.toml b/core/binary_protocol/Cargo.toml index 79e99c768f..0a83a4c010 100644 --- a/core/binary_protocol/Cargo.toml +++ b/core/binary_protocol/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "iggy_binary_protocol" -version = "0.11.0-edge.5" +version = "0.11.0-edge.6" description = "Wire protocol types and codec for the Iggy binary protocol. Shared between server and SDK." edition = "2024" rust-version.workspace = true diff --git a/core/cli/Cargo.toml b/core/cli/Cargo.toml index bfecb9ef32..a2f8756bb7 100644 --- a/core/cli/Cargo.toml +++ b/core/cli/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "iggy-cli" -version = "0.14.0-edge.5" +version = "0.14.0-edge.6" edition = "2024" rust-version.workspace = true authors = ["bartosz.ciesla@gmail.com"] diff --git a/core/common/Cargo.toml b/core/common/Cargo.toml index 8055104fe7..16aaa621c5 100644 --- a/core/common/Cargo.toml +++ b/core/common/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "iggy_common" -version = "0.11.0-edge.5" +version = "0.11.0-edge.6" description = "Iggy is the persistent message streaming platform written in Rust, supporting QUIC, TCP and HTTP transport protocols, capable of processing millions of messages per second." edition = "2024" rust-version.workspace = true diff --git a/core/connectors/runtime/Cargo.toml b/core/connectors/runtime/Cargo.toml index 37b5351d5b..d791d7d26a 100644 --- a/core/connectors/runtime/Cargo.toml +++ b/core/connectors/runtime/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "iggy-connectors" -version = "0.5.0-edge.5" +version = "0.5.0-edge.6" description = "Connectors runtime for Iggy message streaming platform" edition = "2024" license = "Apache-2.0" diff --git a/core/sdk/Cargo.toml b/core/sdk/Cargo.toml index 0e3e504b51..09ef84d03c 100644 --- a/core/sdk/Cargo.toml +++ b/core/sdk/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "iggy" -version = "0.11.0-edge.5" +version = "0.11.0-edge.6" description = "Iggy is the persistent message streaming platform written in Rust, supporting QUIC, TCP and HTTP transport protocols, capable of processing millions of messages per second." edition = "2024" rust-version.workspace = true diff --git a/core/sdk/src/clients/producer_config.rs b/core/sdk/src/clients/producer_config.rs index 1e49e90f26..f2d5948527 100644 --- a/core/sdk/src/clients/producer_config.rs +++ b/core/sdk/src/clients/producer_config.rs @@ -105,7 +105,8 @@ pub struct BackgroundConfig { /// Action to apply when back-pressure limits are reached #[builder(default = BackpressureMode::Block)] pub failure_mode: BackpressureMode, - /// Upper bound for the **bytes held in memory** across *all* shards. + /// Upper bound for the **bytes buffered or in flight** across *all* shards. + /// Bytes remain charged until the corresponding write completes. /// `IggyByteSize::from(0)` ⇒ unlimited. #[builder(default = IggyByteSize::from(32 * MIB as u64))] pub max_buffer_size: IggyByteSize, diff --git a/core/sdk/src/clients/producer_dispatcher.rs b/core/sdk/src/clients/producer_dispatcher.rs index acb79c45bf..447a1b86cd 100644 --- a/core/sdk/src/clients/producer_dispatcher.rs +++ b/core/sdk/src/clients/producer_dispatcher.rs @@ -20,7 +20,7 @@ use crate::clients::producer_config::{BackgroundConfig, BackpressureMode}; use crate::clients::producer_error_callback::ErrorCtx; use crate::clients::producer_sharding::{Shard, ShardMessage, ShardMessageWithPermit}; use futures::FutureExt; -use iggy_common::{Identifier, IggyError, IggyMessage, Partitioning, Sizeable}; +use iggy_common::{Identifier, IggyByteSize, IggyError, IggyMessage, Partitioning, Sizeable}; use std::sync::Arc; use std::sync::atomic::{AtomicBool, Ordering}; use tokio::sync::{Semaphore, broadcast}; @@ -61,13 +61,16 @@ impl ProducerDispatcher { tracing::debug!("error-callback worker finished"); }); - let bytes_permit = { - let bytes = config.max_buffer_size.as_bytes_usize(); - if bytes == 0 { usize::MAX } else { bytes } - }; + let max_buffer_size = config.max_buffer_size.as_bytes_u64(); + assert!( + max_buffer_size == 0 || max_buffer_size <= Semaphore::MAX_PERMITS as u64, + "max_buffer_size cannot exceed {} bytes on this platform", + Semaphore::MAX_PERMITS + ); + let bytes_permit = Arc::new(Semaphore::new(max_buffer_size as usize)); let slots_permit = Arc::new(Semaphore::new(if config.max_in_flight == 0 { - usize::MAX + Semaphore::MAX_PERMITS } else { config.max_in_flight })); @@ -87,7 +90,7 @@ impl ProducerDispatcher { shards, config, closed: AtomicBool::new(false), - bytes_permit: Arc::new(Semaphore::new(bytes_permit)), + bytes_permit, stop_tx, join_handle: handle, } @@ -112,41 +115,45 @@ impl ProducerDispatcher { }; let batch_bytes = shard_message.get_size_bytes(); - if batch_bytes > self.config.max_buffer_size { + if self.config.max_buffer_size != 0 && batch_bytes > self.config.max_buffer_size { return Err(IggyError::BackgroundSendBufferOverflow); } - let permit_bytes = match self - .bytes_permit - .clone() - .try_acquire_many_owned(batch_bytes.as_bytes_u32()) - { - Ok(perm) => perm, - Err(_) => match self.config.failure_mode { - BackpressureMode::FailImmediately => { - return Err(IggyError::BackgroundSendBufferOverflow); - } - BackpressureMode::Block => self - .bytes_permit - .clone() - .acquire_many_owned(batch_bytes.as_bytes_u32()) - .await - .map_err(|_| IggyError::BackgroundSendError)?, - BackpressureMode::BlockWithTimeout(timeout_dur) => { - match tokio::time::timeout( - timeout_dur.get_duration(), - self.bytes_permit - .clone() - .acquire_many_owned(batch_bytes.as_bytes_u32()), - ) - .await - { - Ok(Ok(perm)) => perm, - Ok(Err(_)) => return Err(IggyError::BackgroundSendError), - Err(_) => return Err(IggyError::BackgroundSendTimeout), + let permit_count = Self::permit_count(batch_bytes)?; + let bytes_permit = if self.config.max_buffer_size == 0 { + None + } else { + let permit = match self + .bytes_permit + .clone() + .try_acquire_many_owned(permit_count) + { + Ok(permit) => permit, + Err(_) => match &self.config.failure_mode { + BackpressureMode::FailImmediately => { + return Err(IggyError::BackgroundSendBufferOverflow); } - } - }, + BackpressureMode::Block => self + .bytes_permit + .clone() + .acquire_many_owned(permit_count) + .await + .map_err(|_| IggyError::BackgroundSendError)?, + BackpressureMode::BlockWithTimeout(timeout_duration) => { + match tokio::time::timeout( + timeout_duration.get_duration(), + self.bytes_permit.clone().acquire_many_owned(permit_count), + ) + .await + { + Ok(Ok(permit)) => permit, + Ok(Err(_)) => return Err(IggyError::BackgroundSendError), + Err(_) => return Err(IggyError::BackgroundSendTimeout), + } + } + }, + }; + Some(permit) }; let shard_ix = self.config.sharding.pick_shard( @@ -159,10 +166,15 @@ impl ProducerDispatcher { let shard = &self.shards[shard_ix]; shard - .send(ShardMessageWithPermit::new(shard_message, permit_bytes)) + .send(ShardMessageWithPermit::new(shard_message, bytes_permit)) .await } + fn permit_count(batch_size: IggyByteSize) -> Result { + u32::try_from(batch_size.as_bytes_u64()) + .map_err(|_| IggyError::BackgroundSendBufferOverflow) + } + /// Flushes each shard's buffer and stops its worker. Dropping the /// dispatcher instead of calling this silently discards any buffered, /// not-yet-sent messages. @@ -174,7 +186,7 @@ impl ProducerDispatcher { let _ = self.stop_tx.send(()); for shard in self.shards.drain(..) { - if let Err(e) = shard._handle.await { + if let Err(e) = shard.handle.await { tracing::error!("shard panicked: {e:?}"); } } @@ -237,6 +249,60 @@ mod tests { assert!(result.is_ok()); } + #[tokio::test] + async fn test_dispatch_succeeds_with_unlimited_buffer_and_in_flight_requests() { + let mut mock = MockProducerCoreBackend::new(); + mock.expect_send_internal() + .times(1) + .returning(|_, _, _, _| Box::pin(async { Ok(no_confirmations()) })); + + let config = BackgroundConfig::builder() + .max_buffer_size(0.into()) + .max_in_flight(0) + .batch_length(1) + .build(); + let dispatcher = ProducerDispatcher::new(Arc::new(mock), config); + + assert_eq!(dispatcher.bytes_permit.available_permits(), 0); + dispatcher + .dispatch( + vec![dummy_message(5)], + dummy_identifier(), + dummy_identifier(), + None, + ) + .await + .unwrap(); + dispatcher.shutdown().await; + } + + #[cfg(target_pointer_width = "64")] + #[tokio::test] + async fn test_dispatcher_supports_buffer_budget_above_u32_max() { + let mock = MockProducerCoreBackend::new(); + let budget_size = u32::MAX as u64 + 1; + let config = BackgroundConfig::builder() + .max_buffer_size(budget_size.into()) + .build(); + let dispatcher = ProducerDispatcher::new(Arc::new(mock), config); + + assert_eq!( + dispatcher.bytes_permit.available_permits(), + budget_size as usize + ); + dispatcher.shutdown().await; + } + + #[test] + fn test_permit_count_rejects_batch_above_u32_max() { + let result = ProducerDispatcher::permit_count(IggyByteSize::from(u32::MAX as u64 + 1)); + + assert!(matches!( + result, + Err(IggyError::BackgroundSendBufferOverflow) + )); + } + #[tokio::test] async fn test_dispatch_fails_on_buffer_overflow_immediate() { let mock = MockProducerCoreBackend::new(); diff --git a/core/sdk/src/clients/producer_sharding.rs b/core/sdk/src/clients/producer_sharding.rs index 668256f83d..619470b31d 100644 --- a/core/sdk/src/clients/producer_sharding.rs +++ b/core/sdk/src/clients/producer_sharding.rs @@ -15,18 +15,20 @@ // specific language governing permissions and limitations // under the License. -use crate::clients::producer::ProducerCoreBackend; -use crate::clients::producer_config::BackgroundConfig; -use crate::clients::producer_error_callback::ErrorCtx; -use iggy_common::{Identifier, IggyByteSize, IggyError, IggyMessage, Partitioning, Sizeable}; use std::hash::DefaultHasher; use std::hash::{Hash, Hasher}; use std::sync::Arc; use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; + +use iggy_common::{Identifier, IggyByteSize, IggyError, IggyMessage, Partitioning, Sizeable}; use tokio::sync::{OwnedSemaphorePermit, Semaphore, broadcast}; use tokio::task::JoinHandle; use tracing::{debug, error}; +use crate::clients::producer::ProducerCoreBackend; +use crate::clients::producer_config::BackgroundConfig; +use crate::clients::producer_error_callback::ErrorCtx; + /// A strategy for distributing messages across shards. /// /// Implementors of this trait define how to choose a shard for a given batch of messages. @@ -101,8 +103,11 @@ impl Sizeable for ShardMessage { let mut total = IggyByteSize::new(0); total += self.stream.get_size_bytes(); total += self.topic.get_size_bytes(); - for msg in &self.messages { - total += msg.get_size_bytes(); + if let Some(partitioning) = &self.partitioning { + total += partitioning.get_size_bytes(); + } + for message in &self.messages { + total += message.get_size_bytes(); } total } @@ -110,22 +115,36 @@ impl Sizeable for ShardMessage { pub struct ShardMessageWithPermit { pub inner: ShardMessage, - _bytes_permit: Option, + size_bytes: u64, + bytes_permit: Option, + merged_bytes_permits: Vec, } impl ShardMessageWithPermit { - pub fn new(msg: ShardMessage, permit_bytes: OwnedSemaphorePermit) -> Self { + pub fn new(msg: ShardMessage, bytes_permit: Option) -> Self { + let size_bytes = msg.get_size_bytes().as_bytes_u64(); Self { inner: msg, - _bytes_permit: Some(permit_bytes), + size_bytes, + bytes_permit, + merged_bytes_permits: Vec::new(), } } + + fn merge(&mut self, other: Self) { + self.inner.messages.extend(other.inner.messages); + self.size_bytes += other.size_bytes; + // Tokio stores a merged permit count in a u32, so retain permits separately to avoid + // overflowing when the buffer budget exceeds u32::MAX. + self.merged_bytes_permits.extend(other.bytes_permit); + self.merged_bytes_permits.extend(other.merged_bytes_permits); + } } pub struct Shard { tx: flume::Sender, closed: Arc, - pub(crate) _handle: JoinHandle<()>, + pub(crate) handle: JoinHandle<()>, } impl Shard { @@ -151,7 +170,7 @@ impl Shard { maybe_msg = rx.recv_async() => { match maybe_msg { Ok(msg) => { - buffer_bytes += msg.inner.get_size_bytes().as_bytes_usize(); + buffer_bytes += msg.size_bytes as usize; buffer.push(msg); debug!( buffer_len = buffer.len(), @@ -193,7 +212,7 @@ impl Shard { _ = stop_rx.recv() => { closed_clone.store(true, Ordering::Release); while let Ok(msg) = rx.try_recv() { - buffer_bytes += msg.inner.get_size_bytes().as_bytes_usize(); + buffer_bytes += msg.size_bytes as usize; buffer.push(msg); } if !buffer.is_empty() { @@ -205,11 +224,26 @@ impl Shard { } }); - Self { - tx, - closed, - _handle: handle, + Self { tx, closed, handle } + } + + /// Drains the buffer and combines adjacent messages with the same destination. + fn merge_batches(buffer: &mut Vec) -> Vec { + let mut merged_batches: Vec = Vec::with_capacity(buffer.len()); + for message in buffer.drain(..) { + if let Some(last) = merged_batches.last_mut() + && Self::same_destination(&last.inner, &message.inner) + && last + .size_bytes + .checked_add(message.size_bytes) + .is_some_and(|size_bytes| size_bytes <= u32::MAX as u64) + { + last.merge(message); + continue; + } + merged_batches.push(message); } + merged_batches } async fn flush_buffer( @@ -223,18 +257,7 @@ impl Shard { return; } - let mut merged_batches: Vec = Vec::new(); - for msg in buffer.drain(..) { - if let Some(last) = merged_batches.last_mut() - && Self::same_destination(&last.inner, &msg.inner) - { - last.inner.messages.extend(msg.inner.messages); - continue; - } - merged_batches.push(msg); - } - - for msg in merged_batches { + for msg in Self::merge_batches(buffer) { let _slot_permit = slots_permit.acquire().await; let result = core @@ -312,6 +335,187 @@ mod tests { .unwrap() } + async fn charged_batch( + budget: &Arc, + stream: Arc, + topic: Arc, + payload_size: usize, + ) -> ShardMessageWithPermit { + let message = ShardMessage { + stream, + topic, + messages: vec![dummy_message(payload_size)], + partitioning: None, + }; + let permit = budget + .clone() + .acquire_many_owned(message.get_size_bytes().as_bytes_u32()) + .await + .unwrap(); + ShardMessageWithPermit::new(message, Some(permit)) + } + + #[tokio::test] + async fn test_merge_batches_keeps_permits_of_merged_batches_charged() { + let budget = Arc::new(Semaphore::new(10_000)); + let stream = dummy_identifier(); + let topic = dummy_identifier(); + + let mut buffer = Vec::new(); + for _ in 0..3 { + buffer.push(charged_batch(&budget, stream.clone(), topic.clone(), 10).await); + } + let charged = 10_000 - budget.available_permits(); + let merged = Shard::merge_batches(&mut buffer); + + assert!(buffer.is_empty()); + assert_eq!(merged.len(), 1); + assert_eq!(merged[0].inner.messages.len(), 3); + assert_eq!(budget.available_permits(), 10_000 - charged); + + drop(merged); + assert_eq!(budget.available_permits(), 10_000); + } + + #[cfg(target_pointer_width = "64")] + #[tokio::test] + async fn test_merge_batches_keeps_more_than_u32_max_permits_charged() { + let budget_size = u32::MAX as usize + 1; + let budget = Arc::new(Semaphore::new(budget_size)); + let stream = dummy_identifier(); + let topic = dummy_identifier(); + let first_permit = budget.clone().acquire_many_owned(u32::MAX).await.unwrap(); + let second_permit = budget.clone().acquire_owned().await.unwrap(); + let mut buffer = vec![ + ShardMessageWithPermit::new( + ShardMessage { + stream: stream.clone(), + topic: topic.clone(), + messages: vec![dummy_message(1)], + partitioning: None, + }, + Some(first_permit), + ), + ShardMessageWithPermit::new( + ShardMessage { + stream, + topic, + messages: vec![dummy_message(1)], + partitioning: None, + }, + Some(second_permit), + ), + ]; + + let merged = Shard::merge_batches(&mut buffer); + + assert_eq!(merged.len(), 1); + assert_eq!(budget.available_permits(), 0); + drop(merged); + assert_eq!(budget.available_permits(), budget_size); + } + + #[test] + fn test_merge_batches_splits_batches_above_u32_max_bytes() { + let stream = dummy_identifier(); + let topic = dummy_identifier(); + let mut first = ShardMessageWithPermit::new( + ShardMessage { + stream: stream.clone(), + topic: topic.clone(), + messages: vec![dummy_message(1)], + partitioning: None, + }, + None, + ); + first.size_bytes = u32::MAX as u64; + let mut second = ShardMessageWithPermit::new( + ShardMessage { + stream, + topic, + messages: vec![dummy_message(1)], + partitioning: None, + }, + None, + ); + second.size_bytes = 1; + let mut buffer = vec![first, second]; + + let merged = Shard::merge_batches(&mut buffer); + + assert_eq!(merged.len(), 2); + } + + #[tokio::test] + async fn test_shard_keeps_budget_charged_until_merged_batch_is_written() { + const BUDGET: usize = 10_000; + + let (write_started_tx, write_started_rx) = flume::unbounded::<()>(); + let (release_write_tx, release_write_rx) = flume::unbounded::<()>(); + + let mut mock = MockProducerCoreBackend::new(); + mock.expect_send_internal() + .times(1) + .returning(move |_, _, _, _| { + let write_started_tx = write_started_tx.clone(); + let release_write_rx = release_write_rx.clone(); + Box::pin(async move { + write_started_tx.send_async(()).await.unwrap(); + release_write_rx.recv_async().await.unwrap(); + Ok(no_confirmations()) + }) + }); + + let bb = BackgroundConfig::builder() + .batch_length(3) + .batch_size(0) + .linger_time(IggyDuration::new_from_secs(60)); + let config = Arc::new(bb.build()); + + let budget = Arc::new(Semaphore::new(BUDGET)); + let slots_permit = Arc::new(Semaphore::new(100)); + + let (stop_tx, stop_rx) = broadcast::channel(1); + let shard = Shard::new( + Arc::new(mock), + config, + slots_permit, + flume::unbounded().0, + stop_rx, + ); + + let stream = dummy_identifier(); + let topic = dummy_identifier(); + for _ in 0..3 { + let batch = charged_batch(&budget, stream.clone(), topic.clone(), 100).await; + shard.send(batch).await.unwrap(); + } + let charged = BUDGET - budget.available_permits(); + assert!(charged > 0); + + tokio::time::timeout(Duration::from_secs(1), write_started_rx.recv_async()) + .await + .expect("the merged write must start") + .unwrap(); + assert_eq!( + budget.available_permits(), + BUDGET - charged, + "the merged batch must hold every permit it absorbed until the write completes" + ); + + release_write_tx.send_async(()).await.unwrap(); + tokio::time::timeout(Duration::from_secs(1), async { + while budget.available_permits() != BUDGET { + sleep(Duration::from_millis(5)).await; + } + }) + .await + .expect("the written batch must give its permits back"); + + stop_tx.send(()).unwrap(); + shard.handle.await.unwrap(); + } + #[tokio::test] async fn test_shard_flushes_by_batch_length() { let mut mock = MockProducerCoreBackend::new(); @@ -346,7 +550,7 @@ mod tests { }; let wrapped = ShardMessageWithPermit::new( message, - permit_bytes.clone().acquire_many_owned(1).await.unwrap(), + Some(permit_bytes.clone().acquire_many_owned(1).await.unwrap()), ); shard.send(wrapped).await.unwrap(); } @@ -387,11 +591,13 @@ mod tests { }; let wrapped = ShardMessageWithPermit::new( message, - permit_bytes - .clone() - .acquire_many_owned(10_000) - .await - .unwrap(), + Some( + permit_bytes + .clone() + .acquire_many_owned(10_000) + .await + .unwrap(), + ), ); shard.send(wrapped).await.unwrap(); @@ -431,7 +637,7 @@ mod tests { }; let wrapped = ShardMessageWithPermit::new( message, - permit_bytes.clone().acquire_many_owned(1).await.unwrap(), + Some(permit_bytes.clone().acquire_many_owned(1).await.unwrap()), ); shard.send(wrapped).await.unwrap(); @@ -472,7 +678,7 @@ mod tests { }; let wrapped = ShardMessageWithPermit::new( message, - permit_bytes.clone().acquire_many_owned(1).await.unwrap(), + Some(permit_bytes.clone().acquire_many_owned(1).await.unwrap()), ); shard.send(wrapped).await.unwrap(); @@ -489,7 +695,7 @@ mod tests { let shard = Shard { tx, closed: Arc::new(AtomicBool::new(false)), - _handle: tokio::spawn(async {}), + handle: tokio::spawn(async {}), }; let permit_bytes = Arc::new(Semaphore::new(10_000)); @@ -502,7 +708,7 @@ mod tests { }; let wrapped = ShardMessageWithPermit::new( message, - permit_bytes.clone().acquire_many_owned(1).await.unwrap(), + Some(permit_bytes.clone().acquire_many_owned(1).await.unwrap()), ); let result = shard.send(wrapped).await; diff --git a/examples/python/uv.lock b/examples/python/uv.lock index 1f9c5b3132..eee0abbdd9 100644 --- a/examples/python/uv.lock +++ b/examples/python/uv.lock @@ -8,7 +8,7 @@ exclude-newer-span = "P7D" [[package]] name = "apache-iggy" -version = "0.9.0.dev5" +version = "0.9.0.dev6" source = { directory = "../../foreign/python" } [package.metadata] diff --git a/foreign/python/Cargo.toml b/foreign/python/Cargo.toml index eefeaa1c6f..ee723b4888 100644 --- a/foreign/python/Cargo.toml +++ b/foreign/python/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "apache-iggy" -version = "0.9.0-dev5" +version = "0.9.0-dev6" edition = "2024" authors = ["Iggy Committers "] license = "Apache-2.0" @@ -37,7 +37,7 @@ doc = false [dependencies] bytes = "1.12.1" futures = "0.3.33" -iggy = { path = "../../core/sdk", version = "0.11.0-edge.5" } +iggy = { path = "../../core/sdk", version = "0.11.0-edge.6" } paste = "1" pyo3 = "0.29.0" pyo3-async-runtimes = { version = "0.29.0", features = [ diff --git a/foreign/python/pyproject.toml b/foreign/python/pyproject.toml index 7a82108b6c..d9623d4312 100644 --- a/foreign/python/pyproject.toml +++ b/foreign/python/pyproject.toml @@ -22,7 +22,7 @@ build-backend = "maturin" [project] name = "apache-iggy" requires-python = ">=3.10" -version = "0.9.0.dev5" +version = "0.9.0.dev6" description = "Apache Iggy is the persistent message streaming platform written in Rust, supporting QUIC, TCP and HTTP transport protocols, capable of processing millions of messages per second." readme = "README.md" license = { file = "LICENSE" } diff --git a/foreign/python/uv.lock b/foreign/python/uv.lock index 963b58acf9..214d7023b5 100644 --- a/foreign/python/uv.lock +++ b/foreign/python/uv.lock @@ -8,7 +8,7 @@ exclude-newer-span = "P7D" [[package]] name = "apache-iggy" -version = "0.9.0.dev5" +version = "0.9.0.dev6" source = { editable = "." } [package.optional-dependencies] From 6ef0c167787c6fefd429cccf41501222ae75ccf2 Mon Sep 17 00:00:00 2001 From: Grzegorz Koszyk <112548209+numinnex@users.noreply.github.com> Date: Mon, 31 Aug 2026 21:01:10 +0200 Subject: [PATCH 021/182] feat(sdk): consume the request counter for partition operations (#3958) Our partition operations did not use the `request` counter which meant that partition ops did not participate in the deduplication process. This PR makes the partition operations bump the request counter for all of the SDKs --- core/binary_protocol/src/consensus/header.rs | 4 +- core/consensus/src/client_table.rs | 36 ++++- .../tests/cluster/client_table_adversarial.rs | 140 +++++++++++++++++- core/sdk/src/quic/quic_client.rs | 4 +- core/sdk/src/vsr.rs | 74 ++++++--- core/server/src/dispatch.rs | 5 +- core/server/src/http/session.rs | 5 +- core/simulator/src/client.rs | 109 ++++++-------- core/simulator/src/lib.rs | 6 +- .../csharp/Iggy_SDK/Vsr/ConsensusSession.cs | 10 +- foreign/csharp/Iggy_SDK/Vsr/VsrHeader.cs | 6 +- foreign/csharp/Iggy_SDK/Vsr/VsrOperation.cs | 11 -- .../VsrTests/ConsensusSessionTests.cs | 19 ++- .../Iggy_SDK_Tests/VsrTests/VsrHeaderTests.cs | 12 +- .../VsrTests/VsrOperationTests.cs | 5 +- foreign/go/internal/vsr/envelope.go | 11 +- foreign/go/internal/vsr/envelope_test.go | 18 ++- foreign/go/internal/vsr/header.go | 5 +- foreign/go/internal/vsr/operation.go | 11 +- foreign/go/internal/vsr/operation_test.go | 8 - .../go/internal/vsr/protocol_parity_test.go | 3 - foreign/go/internal/vsr/session.go | 11 +- .../async/tcp/vsr/ConsensusSession.java | 30 +++- .../iggy/client/async/tcp/vsr/VsrHeaders.java | 6 +- .../client/async/tcp/vsr/VsrOperation.java | 5 - .../async/tcp/vsr/VsrRequestEncoder.java | 9 +- .../async/tcp/vsr/VsrRequestEncoderTest.java | 12 +- .../async/tcp/vsr/VsrResponseHandlerTest.java | 72 +++++++++ foreign/node/scripts/check-vsr-protocol.mjs | 6 - foreign/node/src/wire/vsr/header.ts | 6 +- foreign/node/src/wire/vsr/index.ts | 13 +- foreign/node/src/wire/vsr/operation.test.ts | 3 - foreign/node/src/wire/vsr/operation.ts | 5 - foreign/node/src/wire/vsr/session.ts | 6 +- foreign/node/src/wire/vsr/vsr.test.ts | 43 ++++-- 35 files changed, 498 insertions(+), 231 deletions(-) diff --git a/core/binary_protocol/src/consensus/header.rs b/core/binary_protocol/src/consensus/header.rs index a9d20ed3c9..233396c182 100644 --- a/core/binary_protocol/src/consensus/header.rs +++ b/core/binary_protocol/src/consensus/header.rs @@ -287,7 +287,9 @@ pub struct RequestHeader { /// catch a `request` number reused for a different operation: a retry that /// disagrees with the stamp of the cached reply is refused rather than /// answered with the wrong reply. Zero means unstamped, which disables the - /// comparison; the wire currently sends zero. + /// comparison. The Rust SDK stamps the ops the table dedups; partition and + /// non-replicated ops, and the other SDKs, leave it zero. The server + /// verifies any nonzero stamp before routing. pub request_checksum: u128, pub timestamp: u64, pub request: u64, diff --git a/core/consensus/src/client_table.rs b/core/consensus/src/client_table.rs index 67cb42e25c..0737e23bf3 100644 --- a/core/consensus/src/client_table.rs +++ b/core/consensus/src/client_table.rs @@ -578,8 +578,10 @@ pub struct ClientTable { /// Whether two integrity stamps for the same request number disagree. /// -/// Zero means unstamped (the wire integrity fields are zeroed today), and an -/// unstamped side carries no evidence either way, so it never conflicts. +/// Zero means unstamped, and an unstamped side carries no evidence either way, +/// so it never conflicts. The Rust SDK stamps the ops this table dedups; +/// partition ops and the other SDKs still send zero, so a conflict is only ever +/// detectable between two stamped frames. const fn checksums_conflict(stored: u128, received: u128) -> bool { stored != 0 && received != 0 && stored != received } @@ -2681,6 +2683,36 @@ mod tests { assert_eq!(table.get_watermark(1), Some(9)); } + // The shape a client that spends request ids off the metadata plane + // produces: partition-plane ids never reach this table, so the next + // metadata request arrives with a gap under it. It executes, moves the + // watermark to itself, and its retry still replays the original reply -- + // gaps cost the skipped ids and nothing else. + #[test] + fn check_request_dedups_a_metadata_request_that_arrives_after_a_gap() { + let (mut table, epoch) = table_with_client(); + table.commit_reply(1, TEST_USER_ID, make_reply_for(1, 1, 11)); + // Requests 2..=5 went to the partition plane, which keeps no table. + assert!(matches!( + table.check_request(1, epoch, 6, 0), + RequestStatus::New + )); + + table.commit_reply(1, TEST_USER_ID, make_reply_for(1, 6, 12)); + assert_eq!(table.get_watermark(1), Some(6)); + match table.check_request(1, epoch, 6, 0) { + RequestStatus::Duplicate(cached) => { + assert_eq!(cached.header().request, 6); + assert_eq!( + cached.header().commit, + 12, + "the original reply, not a re-run" + ); + } + other => panic!("expected the gapped request to dedup, got {other:?}"), + } + } + #[test] fn check_request_duplicate_at_watermark() { let (mut table, epoch) = table_with_client(); diff --git a/core/integration/tests/cluster/client_table_adversarial.rs b/core/integration/tests/cluster/client_table_adversarial.rs index 5181c8fea0..bf6c3f573c 100644 --- a/core/integration/tests/cluster/client_table_adversarial.rs +++ b/core/integration/tests/cluster/client_table_adversarial.rs @@ -17,8 +17,9 @@ //! Adversarial specs against the VSR client table's at-most-once guarantees. //! -//! Both assert the dedup contract a retrying client needs at the table's two -//! resource edges. +//! Each asserts the dedup contract a retrying client needs: the first two at +//! the table's resource edges, the third across the request-id gaps a client +//! that also produces leaves behind. //! //! 1. Capacity: a full table evicts the entry with the oldest commit. Eviction //! keeps that client's request watermark (and the watermark's reply when the @@ -34,6 +35,10 @@ //! fact that neither a state transfer nor a restart carries the deeper //! history, are pinned by the consensus crate's unit tests; this file pins //! the depth the budget buys on a live server. +//! 3. Sequence gaps: partition sends spend request ids on a plane that keeps no +//! client table, so the next metadata request arrives with a gap under it. +//! The entry stores a watermark, not a contiguous sequence, so the gapped +//! request must execute once and its retry must replay that reply. //! //! The frames are hand-crafted on raw TCP sockets, same technique and frame //! builders as the clients-table restart tests, because the churn needs @@ -41,7 +46,7 @@ //! expose. The builders here are parameterized by client id, which is why the //! restart tests' fixed-identity helpers are not reused directly. -use bytes::Bytes; +use bytes::{Bytes, BytesMut}; use consensus::client_table::REPLY_RING_CAPACITY; use iggy::prelude::*; use iggy_binary_protocol::codec::{WireDecode, WireEncode}; @@ -49,11 +54,13 @@ use iggy_binary_protocol::consensus::{ Command, Operation, ReplyHeader, RequestHeader, read_size_field, result_code, result_section_len, }; +use iggy_binary_protocol::requests::messages::{RawMessage, SendMessagesEncoder}; use iggy_binary_protocol::requests::streams::CreateStreamRequest; use iggy_binary_protocol::requests::users::LoginRegisterRequest; use iggy_binary_protocol::responses::users::LoginRegisterResponse; use iggy_binary_protocol::{ - ClientVersionInfo, HEADER_SIZE, IGGY_PROTOCOL_VERSION, WireName, WireOptions, + ClientVersionInfo, HEADER_SIZE, IGGY_PROTOCOL_VERSION, WireIdentifier, WireName, WireOptions, + WirePartitioning, }; use integration::harness::TestHarness; use integration::iggy_harness; @@ -72,6 +79,13 @@ const CLIENT_A: u128 = 0xA11CE0001; /// eviction of `CLIENT_A`'s entry. const CHURN_CLIENTS: [u128; 3] = [0xB0B0001, 0xB0B0002, 0xB0B0003]; +/// The topic the gap spec produces into, and how many batches it sends before +/// its next metadata request. One batch already gaps the sequence; four makes +/// the distance the table has to tolerate unmistakable. +const GAP_STREAM: &str = "adv-m-gap"; +const GAP_TOPIC: &str = "adv-m-gap-topic"; +const GAP_BATCHES: u64 = 4; + /// Budget for one committed round-trip (covers transient replays while the /// single node elects itself after boot). const COMMIT_BUDGET: Duration = Duration::from_secs(15); @@ -196,6 +210,65 @@ async fn given_a_retry_past_the_reply_floor_when_the_retention_budget_still_hold } } +/// A metadata request that lands above a gap the partition plane opened must +/// still deduplicate. +/// +/// Every replicated request on a session spends an id, but only the metadata +/// plane keeps a client table, so a session that produces reaches its next +/// metadata request several ids above the watermark. The entry records the +/// highest committed request rather than a contiguous run, so the gapped +/// request executes once and the retry of its exact frame is answered from the +/// cache. A committed duplicate-name rejection is the proof of re-execution. +#[iggy_harness(cluster_nodes = 1)] +async fn given_partition_batches_spent_request_ids_when_a_metadata_request_is_retried_should_replay_from_cache( + harness: &mut TestHarness, +) { + // The topic the batches target is set up over the SDK on its own session, + // so the raw session below spends ids only on what this spec is about. + let setup = harness.tcp_root_client().await.unwrap(); + setup + .create_stream(GAP_STREAM) + .await + .expect("create stream"); + setup + .create_topic( + &Identifier::named(GAP_STREAM).unwrap(), + GAP_TOPIC, + &TopicCreateOptions { + partitions_count: Some(1), + ..TopicCreateOptions::default() + }, + ) + .await + .expect("create topic"); + drop(setup); + + let addr = tcp_addr(harness); + let (mut stream, session) = register(addr, CLIENT_A).await; + + for request in 1..=GAP_BATCHES { + let batch = send_messages_payload(u128::from(request)); + commit_batch(&mut stream, CLIENT_A, session, request, &batch).await; + } + + let payload = create_stream_payload("adv-m-after-gap"); + let gapped_request = GAP_BATCHES + 1; + let committed = commit_request(&mut stream, CLIENT_A, session, gapped_request, &payload).await; + let replay = replay_request(&mut stream, CLIENT_A, session, gapped_request, &payload).await; + + match replay { + Verdict::Success(replayed) => { + assert_replayed_from_cache(&committed, &replayed, gapped_request); + } + other => panic!( + "a metadata request above a partition-plane gap lost its dedup: request \ + {gapped_request} committed after {GAP_BATCHES} batches spent the ids below it, \ + so its retry must be answered from the cache rather than re-executed; the \ + watermark is a high-water mark, not a contiguity check; got {other:?}" + ), + } +} + fn tcp_addr(harness: &TestHarness) -> SocketAddr { harness .server() @@ -211,6 +284,31 @@ fn create_stream_payload(name: &str) -> Bytes { .to_bytes() } +/// One canonical batch for the gap spec's partition, in the shape +/// `SendMessagesEncoder` writes and admission verifies. `message_id` keeps +/// successive batches distinct on the wire. +fn send_messages_payload(message_id: u128) -> Bytes { + let stream_id = WireIdentifier::named(GAP_STREAM).unwrap(); + let topic_id = WireIdentifier::named(GAP_TOPIC).unwrap(); + let partitioning = WirePartitioning::PartitionId(0); + let messages = [RawMessage { + id: message_id, + origin_timestamp: 0, + headers: None, + payload: b"adv-m-gap-batch", + }]; + + let mut buf = BytesMut::with_capacity(SendMessagesEncoder::encoded_size( + &stream_id, + &topic_id, + &partitioning, + &messages, + )); + SendMessagesEncoder::encode(&mut buf, &stream_id, &topic_id, &partitioning, &messages) + .expect("send batch encodes"); + buf.freeze() +} + fn request_header(client: u128, session: u64, request: u64, body_len: usize) -> RequestHeader { RequestHeader { command: Command::Request, @@ -300,6 +398,40 @@ async fn commit_request( } } +/// Send one batch on the partition plane and require it committed. The reply +/// is not result-framed and carries no body, so a zero status is the whole +/// verdict; what this spec needs from it is only the request id it spends. +async fn commit_batch( + stream: &mut TcpStream, + client: u128, + session: u64, + request: u64, + body: &Bytes, +) { + let header = RequestHeader { + command: Command::Request, + operation: Operation::SendMessages, + size: u32::try_from(HEADER_SIZE + body.len()).unwrap(), + client, + session, + request, + ..Default::default() + }; + let deadline = Instant::now() + COMMIT_BUDGET; + loop { + match exchange(stream, &header, body).await { + Exchange::Reply { status: 0, .. } => return, + Exchange::Reply { status, .. } if is_transient(status) && Instant::now() < deadline => { + sleep(RETRY_PAUSE).await; + } + other => panic!( + "batch at request {request} did not commit: {:?}", + other.verdict() + ), + } + } +} + /// Replay `request` and return the first non-transient verdict. Unlike /// `commit_request` this never panics on a committed rejection: the rejection /// IS the observation the red specs are after. diff --git a/core/sdk/src/quic/quic_client.rs b/core/sdk/src/quic/quic_client.rs index 28fb6529da..e69e94af6c 100644 --- a/core/sdk/src/quic/quic_client.rs +++ b/core/sdk/src/quic/quic_client.rs @@ -919,8 +919,8 @@ impl QuicClient { // construction, so replaying the SAME request header on a fresh // bidi cannot double-commit), and it no longer abandons a bidi // whose op is still committing. Silence therefore is NOT a - // retry signal: partition ops share one request id and have no - // reply cache, so resending a silently-unanswered request whose + // retry signal: the partition plane has no dedup or reply + // cache, so resending a silently-unanswered request whose // first attempt was buffered and later commits would commit it // twice (duplicate `SendMessages`, or a succeeded delete coming // back as terminal `ConsumerOffsetNotFound`). A silent deadline diff --git a/core/sdk/src/vsr.rs b/core/sdk/src/vsr.rs index 2e8a210977..876457ad8a 100644 --- a/core/sdk/src/vsr.rs +++ b/core/sdk/src/vsr.rs @@ -94,26 +94,22 @@ pub(crate) fn encode_request_header( session.current_request_id(), session.session().unwrap_or(0), ) - } else if operation.is_partition() { - // Partition ops replicate in their own per-partition group, - // which is at-least-once with no `ClientTable` dedup -- the - // metadata table never records their request ids, so there - // is nothing for a consumed id to deduplicate against. Every - // partition request on a session therefore carries the id - // the next metadata op will claim, and a partition-plane - // replay is at-least-once. - let session_id = session.session().ok_or(IggyError::Unauthenticated)?; - (operation, session.current_request_id(), session_id) } else { + // Partition ops consume an id too, even though nothing dedups + // them yet: dedup needs each send to carry a distinct number, + // and the metadata watermark tolerates the resulting gaps + // (`client_table.rs`: "There is no `RequestGap`"). let session_id = session.session().ok_or(IggyError::Unauthenticated)?; (operation, session.next_request_id(), session_id) } } }; - // Stamped only for ops the server's `ClientTable` dedups. Partition ops are - // at-least-once with no reply cache to poison, and theirs are the large payloads, - // already covered client-side by `batch_checksum` over the same bytes. - // NonReplicated ops bypass dedup too. + // Stamped only for the ops the server's `ClientTable` dedups, and only by this + // SDK: the others leave the field zero, which the server reads as unstamped. + // Partition ops are the large payloads and already carry `batch_checksum` over + // the same bytes, and nothing dedups them, so hashing here would only buy the + // server a second full-payload pass in `verify_request_checksum`. NonReplicated + // ops bypass dedup too. let request_checksum = if operation.is_partition() || operation == Operation::NonReplicated { 0 } else { @@ -362,7 +358,9 @@ fn read_window_field(header_bytes: &[u8; HEADER_SIZE], offset: usize) -> u32 { mod tests { use super::*; use crate::session::ConsensusSession; - use iggy_binary_protocol::codes::{CREATE_STREAM_CODE, GET_STREAM_CODE, PING_CODE}; + use iggy_binary_protocol::codes::{ + CREATE_STREAM_CODE, GET_STREAM_CODE, PING_CODE, SEND_MESSAGES_CODE, + }; use iggy_binary_protocol::requests::streams::CreateStreamRequest; use iggy_binary_protocol::requests::users::LoginRegisterRequest; use iggy_binary_protocol::version::IGGY_PROTOCOL_VERSION; @@ -503,25 +501,57 @@ mod tests { #[test] fn request_checksum_is_stamped_only_for_deduped_operations() { - // The stamp exists to stop a reused `request` number returning the wrong - // cached reply, so it is worth its hashing pass only where `ClientTable` - // dedups. Partition payloads are the large ones and carry `batch_checksum` - // over the same bytes already; hashing them again is pure cost. + // The stamp exists to stop a reused `request` number matching a dedup + // entry recorded for different bytes, so it is worth its hashing pass only + // where `ClientTable` dedups. Partition ops are the large payloads and + // already carry `batch_checksum` over the same bytes; NonReplicated ops + // bypass dedup. Neither stamps. let mut session = ConsensusSession::with_client_id(42); session.bind(99); let payload = Bytes::from_static(b"payload"); - let deduped = + let metadata = encode_contiguous_request(&mut session, CREATE_STREAM_CODE, &payload).unwrap(); + // Hashed against the framed body rather than the `payload` the encoder was + // handed, so a slice mistake between what is stamped and what is sent fails + // here instead of reaching the server's `verify_request_checksum`. assert_eq!( - decode_request_header(&deduped).request_checksum, - u128::from(calculate_checksum(&payload)), + decode_request_header(&metadata).request_checksum, + u128::from(calculate_checksum(&metadata[HEADER_SIZE..])), ); + let partition = + encode_contiguous_request(&mut session, SEND_MESSAGES_CODE, &payload).unwrap(); + assert_eq!(decode_request_header(&partition).request_checksum, 0); + let ping = encode_contiguous_request(&mut session, PING_CODE, &Bytes::new()).unwrap(); assert_eq!(decode_request_header(&ping).request_checksum, 0); } + #[test] + fn partition_request_consumes_the_request_counter() { + // Dedup identity requires each send to carry a distinct id, so + // partition ops advance the counter exactly like metadata ops and + // the two planes interleave on one sequence. + let mut session = ConsensusSession::with_client_id(42); + session.bind(99); + let payload = Bytes::from_static(b"batch"); + + let first = encode_contiguous_request(&mut session, SEND_MESSAGES_CODE, &payload).unwrap(); + let second = encode_contiguous_request(&mut session, SEND_MESSAGES_CODE, &payload).unwrap(); + assert_eq!(decode_request_header(&first).request, 1); + assert_eq!(decode_request_header(&second).request, 2); + + let metadata_payload = CreateStreamRequest { + name: WireName::new("stream").unwrap(), + options: WireOptions::empty(), + } + .to_bytes(); + let metadata = + encode_contiguous_request(&mut session, CREATE_STREAM_CODE, &metadata_payload).unwrap(); + assert_eq!(decode_request_header(&metadata).request, 3); + } + #[test] fn ping_uses_non_replicated_operation() { let mut session = ConsensusSession::with_client_id(42); diff --git a/core/server/src/dispatch.rs b/core/server/src/dispatch.rs index 5e7bca80e8..44cfd93e2c 100644 --- a/core/server/src/dispatch.rs +++ b/core/server/src/dispatch.rs @@ -1598,8 +1598,9 @@ pub(crate) async fn dispatch_partition_request( // Header validation requires `session > 0 && request > 0` for // non-register ops. The partition plane itself is sessionless // (at-least-once, no `ClientTable` dedup), so the bound VSR - // session merely satisfies validation, and a zero request id - // (the SDK does not number data-plane ops) is normalized. + // session merely satisfies validation. Current SDKs do number + // partition ops, but older and internal callers may still send + // zero, so a zero id is normalized to the compatibility value 1. new_header.session = bound_session; new_header.request = new_header.request.max(1); }); diff --git a/core/server/src/http/session.rs b/core/server/src/http/session.rs index 2e104268d4..aa1db608c3 100644 --- a/core/server/src/http/session.rs +++ b/core/server/src/http/session.rs @@ -99,7 +99,10 @@ pub(in crate::http) struct HttpSession { /// Serializes this session's writes: the guarded value is the NEXT request /// id. A `tokio::sync::Mutex` because the write path holds it across the /// submit `.await` so each session's request numbers reach the primary in - /// order and stay gap-free for the depth-1 consensus dedup. + /// order. Ordering is what matters, not contiguity: the client table dedups + /// on a watermark (see `submit.rs`), so gaps are free but an id overtaken by + /// a larger one would arrive at or below the watermark and be refused as a + /// duplicate. pub(in crate::http) gate: Mutex, /// Next data-plane request id. A separate, gate-free counter: partition ops /// are at-least-once with no consensus dedup, so the id only correlates the diff --git a/core/simulator/src/client.rs b/core/simulator/src/client.rs index 6f19bd1077..1b1501aeed 100644 --- a/core/simulator/src/client.rs +++ b/core/simulator/src/client.rs @@ -55,24 +55,15 @@ use server_common::sharding::{IggyNamespace, METADATA_GROUP}; use server_common::{Message, iobuf::Owned}; use std::cell::Cell; -/// Partition-plane request ids are offset into the top half of the `u64` space -/// so they never collide with the small, contiguous metadata ids in the -/// auditor's `(client, request)` map. The metadata sequence would have to reach -/// `2^63` to overlap, which no run approaches. -const PARTITION_ID_BASE: u64 = 1 << 63; - // TODO: Proper client which implements the full client SDK API pub struct SimClient { client_id: u128, - /// Contiguous `1, 2, 3, …` request ids for metadata/replicated ops, the - /// sequence the server's `ClientTable` dedups and requires gap-free. + /// Monotonic `1, 2, 3, …` request ids shared by every replicated op, + /// metadata and partition alike, which is what the SDKs send. One sequence + /// also issues each id exactly once per client, so a delayed or duplicated + /// partition reply can never carry the `(client, request)` key of a live + /// metadata entry in the auditor's map. See [`SimClient::next_request_id`]. request_counter: Cell, - /// Separate id sequence for partition-plane ops, offset into a disjoint - /// range ([`PARTITION_ID_BASE`]). The partition plane has no client-table - /// dedup and treats the id as an opaque echo, so a partition id never - /// collides with a metadata id, even under reply duplication. See - /// [`SimClient::request_id_for`]. - partition_counter: Cell, /// Deterministic per-message id source for produced messages. The real SDK /// mints a random UUID for a zero message id before encoding; that mint is /// unseeded, so under the deterministic executor a produce's replicated @@ -100,7 +91,6 @@ impl SimClient { Self { client_id, request_counter: Cell::new(0), - partition_counter: Cell::new(0), message_counter: Cell::new(0), session: Cell::new(0), shell_wire: Cell::new(false), @@ -157,30 +147,21 @@ impl SimClient { self.session.set(session); } - /// Assign the wire request id for `operation`, keyed by plane. + /// Assign the wire request id for the next replicated op. /// - /// Metadata/replicated ops advance a contiguous `1, 2, 3, …` counter, matching - /// the real SDK. Gaps are admitted rather than fatal (`check_request` answers - /// `New` to anything above the watermark; there is no `RequestGap`), but the - /// dedup ring is sized for a contiguous sequence. Partition ops are at-least-once - /// with no dedup and the server - /// treats their id as an opaque echo, so they draw from a separate counter - /// offset into a disjoint range ([`PARTITION_ID_BASE`]). A partition id can - /// therefore never equal a metadata id, so a delayed or duplicated partition - /// reply is never misattributed to a metadata entry in the auditor's - /// `(client, request)` map (which would trip the group guard and drop a - /// live metadata op). This holds regardless of reply duplication, not only - /// while clients are one-in-flight. - fn request_id_for(&self, operation: Operation) -> u64 { - if operation.is_partition() { - let next = self.partition_counter.get() + 1; - self.partition_counter.set(next); - PARTITION_ID_BASE + next - } else { - let next = self.request_counter.get() + 1; - self.request_counter.set(next); - next - } + /// Every replicated op advances one counter, metadata and partition alike, + /// which is what the SDKs send: a partition op needs its own number for a + /// retry to be recognisable, and the ids it spends cost the metadata plane + /// nothing, because `ClientTable` admits anything above the watermark + /// (`client_table.rs`: "There is no `RequestGap`"). + /// + /// `NonReplicated` reads never reach here — the poll path builds its own + /// header and reads the counter without advancing it, matching the SDK, + /// since the server ignores the id for ops the table never sees. + fn next_request_id(&self) -> u64 { + let next = self.request_counter.get() + 1; + self.request_counter.set(next); + next } fn session_id(&self) -> u64 { @@ -623,10 +604,10 @@ impl SimClient { /// `count` messages from offset 0 of `group`'s partition. /// /// A `NonReplicated` read: the command code sits in the header's - /// `reserved` prefix, and the request id ECHOES the current metadata - /// counter without advancing it (matching the SDK), so a read never - /// gaps the replicated sequence the server's `ClientTable` requires - /// gap-free. Requires a bound session (polls are auth-gated). + /// `reserved` prefix, and the request id ECHOES the current counter + /// without advancing it (matching the SDK). The server ignores the id for + /// ops its `ClientTable` never sees, so burning one would buy nothing. + /// Requires a bound session (polls are auth-gated). /// /// # Panics /// Panics if the session is unbound or the request buffer is invalid. @@ -779,7 +760,7 @@ impl SimClient { request_checksum: 0, timestamp: 0, // TODO: Use actual timestamp session: self.session_id(), - request: self.request_id_for(operation), + request: self.next_request_id(), group, ..Default::default() } @@ -813,29 +794,29 @@ fn namespace_ids(ns: IggyNamespace) -> (WireIdentifier, WireIdentifier, Option = interleaved + .into_iter() + .map(|operation| client.header(operation, METADATA_GROUP, 0).request) + .collect(); - // Metadata ops advance (1, 2, 3); interleaved sends draw their own - // disjoint sequence and leave the metadata counter untouched. - assert_eq!(client.request_id_for(Operation::CreateStream), 1); - assert_eq!( - client.request_id_for(Operation::SendMessages), - PARTITION_ID_BASE + 1 - ); - assert_eq!(client.request_id_for(Operation::CreateStream), 2); - assert_eq!( - client.request_id_for(Operation::SendMessages), - PARTITION_ID_BASE + 2 - ); - assert_eq!(client.request_id_for(Operation::CreateStream), 3); + assert_eq!(ids, vec![1, 2, 3, 4, 5], "plane must not fork the sequence"); } } diff --git a/core/simulator/src/lib.rs b/core/simulator/src/lib.rs index 128278899a..4e006c573d 100644 --- a/core/simulator/src/lib.rs +++ b/core/simulator/src/lib.rs @@ -1982,9 +1982,11 @@ mod tests { // `WireConsumer` discriminant so every such request was dropped unparsed (see // `ops::sample_consumer_kind`); and per-stream seeds moving from XOR salts to // [`SimSeeds`] alongside `Xoshiro256Plus` becoming `Xoshiro256PlusPlus`, which - // together remap every stream. + // together remap every stream; and partition ops drawing from the one shared + // request counter instead of a separate sequence based at `1<<63`, which + // renumbers every partition request id and so every reply header in the trace. assert_eq!( - h1, 0x1376_D480_4A3F_E6A9, + h1, 0x5C2B_6057_2DA9_908B, "workload reply hash drifted from locked baseline" ); } diff --git a/foreign/csharp/Iggy_SDK/Vsr/ConsensusSession.cs b/foreign/csharp/Iggy_SDK/Vsr/ConsensusSession.cs index 58e1bed51f..b5996d7a52 100644 --- a/foreign/csharp/Iggy_SDK/Vsr/ConsensusSession.cs +++ b/foreign/csharp/Iggy_SDK/Vsr/ConsensusSession.cs @@ -203,14 +203,8 @@ private SessionFrame ReplicatedFrameLocked(VsrOperation operation) var sessionId = _session ?? throw VsrError.Exception(VsrError.UNAUTHENTICATED, "A replicated request requires a bound consensus session."); - // Partition ops replicate in their own per-partition group with no client-table dedup, so they too - // must leave the metadata counter untouched. Only metadata operations and logout consume an id: the - // server tracks request ids for those alone, and it accepts any id above the client's watermark. - if (operation.IsPartition()) - { - return new SessionFrame(_clientId, _requestCounter, sessionId); - } - + // Partition ops consume an id too, even though no partition-plane dedup exists yet: dedup needs + // each send to carry a distinct number, and the metadata watermark tolerates the gaps. var requestId = _requestCounter; _requestCounter = checked(_requestCounter + 1); diff --git a/foreign/csharp/Iggy_SDK/Vsr/VsrHeader.cs b/foreign/csharp/Iggy_SDK/Vsr/VsrHeader.cs index 3de62116b2..6dce9d3af3 100644 --- a/foreign/csharp/Iggy_SDK/Vsr/VsrHeader.cs +++ b/foreign/csharp/Iggy_SDK/Vsr/VsrHeader.cs @@ -21,8 +21,10 @@ namespace Apache.Iggy.Vsr; /// /// The 256-byte consensus header, read and written by wire offset. Offsets mirror -/// core/binary_protocol/src/consensus/header.rs. Checksums stay zero: the server does not verify -/// them for client frames. +/// core/binary_protocol/src/consensus/header.rs. Checksums stay zero: the frame and body checksums +/// are not read on the client request path, and request_checksum treats zero as unstamped, which +/// opts out of the server's payload comparison. Stamping it is optional -- the Rust SDK does so for +/// deduped operations, this SDK does not yet. /// internal static class VsrHeader { diff --git a/foreign/csharp/Iggy_SDK/Vsr/VsrOperation.cs b/foreign/csharp/Iggy_SDK/Vsr/VsrOperation.cs index f46eeca25c..de5d9ed726 100644 --- a/foreign/csharp/Iggy_SDK/Vsr/VsrOperation.cs +++ b/foreign/csharp/Iggy_SDK/Vsr/VsrOperation.cs @@ -68,7 +68,6 @@ internal static class VsrOperations { private const byte InternalStart = (byte)VsrOperation.CreateTopicWithAssignments; private const byte MetadataStart = (byte)VsrOperation.CreateStream; - private const byte PartitionStart = (byte)VsrOperation.SendMessages; /// /// Non-replicated codes this build knows to leave no server-side state behind, so re-sending one after a @@ -245,16 +244,6 @@ or VsrOperation.JoinConsumerGroup or VsrOperation.LeaveConsumerGroup; } - /// - /// Data-plane operations routed by namespace to the shard owning the partition. - /// is deliberately neither metadata nor partition: the - /// server resolves it to an internal TruncatePartition, yet it still carries a packed namespace. - /// - internal static bool IsPartition(this VsrOperation operation) - { - return (byte)operation >= PartitionStart; - } - /// /// Whether a reply for this operation leads its body with the committed result section. Metadata ops /// always do; on the partition plane only the consumer-offset ops do. Register is result-framed only diff --git a/foreign/csharp/Iggy_SDK_Tests/VsrTests/ConsensusSessionTests.cs b/foreign/csharp/Iggy_SDK_Tests/VsrTests/ConsensusSessionTests.cs index 3ff5256c85..aa91e558ff 100644 --- a/foreign/csharp/Iggy_SDK_Tests/VsrTests/ConsensusSessionTests.cs +++ b/foreign/csharp/Iggy_SDK_Tests/VsrTests/ConsensusSessionTests.cs @@ -119,17 +119,32 @@ public void NextRequestId_BeforeBindThrows() } [Fact] - public void Resolve_DoesNotConsumeAnIdForNonReplicatedOrPartitionOps() + public void Resolve_DoesNotConsumeAnIdForNonReplicatedOps() { var session = new ConsensusSession(1); session.Resolve(VsrOperation.Register); session.Bind(10); Assert.Equal(1UL, session.Resolve(VsrOperation.NonReplicated).RequestId); - Assert.Equal(1UL, session.Resolve(VsrOperation.SendMessages).RequestId); Assert.Equal(1UL, session.RequestCounter); } + [Fact] + public void Resolve_PartitionOpsConsumeADistinctIdPerSend() + { + // Dedup identity requires each send to carry a distinct number, so + // partition ops advance the counter exactly like metadata ops and the + // two planes interleave on one sequence. + var session = new ConsensusSession(1); + session.Resolve(VsrOperation.Register); + session.Bind(10); + + Assert.Equal(1UL, session.Resolve(VsrOperation.SendMessages).RequestId); + Assert.Equal(2UL, session.Resolve(VsrOperation.SendMessages).RequestId); + Assert.Equal(3UL, session.Resolve(VsrOperation.CreateStream).RequestId); + Assert.Equal(4UL, session.RequestCounter); + } + [Fact] public void Bind_TwiceThrows() { diff --git a/foreign/csharp/Iggy_SDK_Tests/VsrTests/VsrHeaderTests.cs b/foreign/csharp/Iggy_SDK_Tests/VsrTests/VsrHeaderTests.cs index 278037041a..4a6e54e0a9 100644 --- a/foreign/csharp/Iggy_SDK_Tests/VsrTests/VsrHeaderTests.cs +++ b/foreign/csharp/Iggy_SDK_Tests/VsrTests/VsrHeaderTests.cs @@ -162,16 +162,18 @@ public void Encode_LogoutAdvancesTheCounter() } [Fact] - public void Encode_PartitionOpDoesNotAdvanceTheCounter() + public void Encode_PartitionOpConsumesADistinctId() { var session = BoundSession(); var payload = VsrTestPayloads.SendMessagesToPartition(2, 3, 4); - var header = Encode(session, CommandCodes.SEND_MESSAGES_CODE, payload, out _); + var first = Encode(session, CommandCodes.SEND_MESSAGES_CODE, payload, out _); + Assert.Equal((byte)VsrOperation.SendMessages, first[VsrHeader.REQUEST_OPERATION_OFFSET]); + Assert.Equal(1UL, ReadUInt64(first, VsrHeader.REQUEST_ID_OFFSET)); - Assert.Equal((byte)VsrOperation.SendMessages, header[VsrHeader.REQUEST_OPERATION_OFFSET]); - Assert.Equal(1UL, ReadUInt64(header, VsrHeader.REQUEST_ID_OFFSET)); - Assert.Equal(1UL, session.RequestCounter); + var second = Encode(session, CommandCodes.SEND_MESSAGES_CODE, payload, out _); + Assert.Equal(2UL, ReadUInt64(second, VsrHeader.REQUEST_ID_OFFSET)); + Assert.Equal(3UL, session.RequestCounter); } [Fact] diff --git a/foreign/csharp/Iggy_SDK_Tests/VsrTests/VsrOperationTests.cs b/foreign/csharp/Iggy_SDK_Tests/VsrTests/VsrOperationTests.cs index 6888db37bd..ad191b3081 100644 --- a/foreign/csharp/Iggy_SDK_Tests/VsrTests/VsrOperationTests.cs +++ b/foreign/csharp/Iggy_SDK_Tests/VsrTests/VsrOperationTests.cs @@ -94,12 +94,9 @@ public void Classification_MatchesTheServerSideRanges() Assert.True(VsrOperation.CreateTopicWithAssignments.IsInternal()); Assert.True(VsrOperation.CreateTopicWithAssignments.IsMetadata()); Assert.True(VsrOperation.CreateStream.IsMetadata()); - Assert.False(VsrOperation.CreateStream.IsPartition()); - Assert.True(VsrOperation.SendMessages.IsPartition()); Assert.False(VsrOperation.SendMessages.IsMetadata()); - // Resolved server-side to an internal truncate, so it is neither plane despite carrying a namespace. - Assert.False(VsrOperation.DeleteSegments.IsPartition()); + // Resolved server-side to an internal truncate, so it is not metadata despite carrying a namespace. Assert.False(VsrOperation.DeleteSegments.IsMetadata()); } diff --git a/foreign/go/internal/vsr/envelope.go b/foreign/go/internal/vsr/envelope.go index 912881a2c1..75c2d1c94d 100644 --- a/foreign/go/internal/vsr/envelope.go +++ b/foreign/go/internal/vsr/envelope.go @@ -61,13 +61,10 @@ func StampRequestHeader(session *Session, code uint32, frame []byte) error { sessionID = session.SessionID() default: sessionID = session.SessionID() - if IsPartition(operation) { - // Partition operations replicate in per-partition groups that - // keep no client table, so there is nothing to deduplicate - // against: the watermark is read without being consumed and a - // partition-plane replay is at-least-once. - request = session.CurrentRequestID() - } else if request, err = session.NextRequestID(); err != nil { + // Partition operations consume an id too, even though no + // partition-plane dedup exists yet: dedup needs each send to carry a + // distinct number, and the metadata watermark tolerates the gaps. + if request, err = session.NextRequestID(); err != nil { return err } } diff --git a/foreign/go/internal/vsr/envelope_test.go b/foreign/go/internal/vsr/envelope_test.go index c497b07fc5..154f95a334 100644 --- a/foreign/go/internal/vsr/envelope_test.go +++ b/foreign/go/internal/vsr/envelope_test.go @@ -115,7 +115,7 @@ func TestEncodeRequest_SendsNonReplicatedCommandsBeforeRegister(t *testing.T) { assert.Equal(t, uint64(1), binary.LittleEndian.Uint64(header[requestOffsetRequest:])) } -func TestEncodeRequest_DoesNotAdvanceTheWatermarkOffTheMetadataPlane(t *testing.T) { +func TestEncodeRequest_DoesNotAdvanceTheWatermarkForNonReplicated(t *testing.T) { session := boundSession(t) for range 3 { @@ -123,12 +123,21 @@ func TestEncodeRequest_DoesNotAdvanceTheWatermarkOffTheMetadataPlane(t *testing. require.NoError(t, err) } assert.Equal(t, uint64(1), session.CurrentRequestID()) +} - for range 3 { - _, err := EncodeRequest(session, uint32(command.SendMessagesCode), []byte{1}) +func TestEncodeRequest_PartitionCommandConsumesTheWatermark(t *testing.T) { + // Dedup identity requires each send to carry a distinct id, so partition + // commands advance the counter exactly like metadata commands and the two + // planes interleave on one sequence. + session := boundSession(t) + + for expected := uint64(1); expected <= 3; expected++ { + frame, err := EncodeRequest(session, uint32(command.SendMessagesCode), []byte{1}) require.NoError(t, err) + header := frameHeader(t, frame) + assert.Equal(t, expected, binary.LittleEndian.Uint64(header[requestOffsetRequest:])) } - assert.Equal(t, uint64(1), session.CurrentRequestID()) + assert.Equal(t, uint64(4), session.CurrentRequestID()) } func TestEncodeRequest_AdvancesTheWatermarkPerMetadataCommand(t *testing.T) { @@ -157,6 +166,7 @@ func TestEncodeRequest_RejectsAPartitionCommandOnAnUnboundSession(t *testing.T) _, err := EncodeRequest(session, uint32(command.SendMessagesCode), []byte{1}) assert.ErrorIs(t, err, ierror.ErrUnauthenticated) + assert.Equal(t, uint64(1), session.CurrentRequestID(), "the rejection burns no id") } func TestEncodeRequest_EncodesASendMessagesFrame(t *testing.T) { diff --git a/foreign/go/internal/vsr/header.go b/foreign/go/internal/vsr/header.go index 789237efd1..93580ff839 100644 --- a/foreign/go/internal/vsr/header.go +++ b/foreign/go/internal/vsr/header.go @@ -116,7 +116,10 @@ func (c ClientID) IsZero() bool { } // RequestFields are the header fields a client fills in. The rest of the 256 -// bytes stay zero, including both checksums, which the server does not read. +// bytes stay zero. The frame and body checksums are not read on the client +// request path. request_checksum is read, but zero means unstamped and opts out +// of the server's payload comparison, so leaving it zero is legal; the Rust SDK +// stamps it for deduped ops, this SDK does not yet. type RequestFields struct { // Size is the header plus body total. Size uint32 diff --git a/foreign/go/internal/vsr/operation.go b/foreign/go/internal/vsr/operation.go index cc95c57720..a1f0140cce 100644 --- a/foreign/go/internal/vsr/operation.go +++ b/foreign/go/internal/vsr/operation.go @@ -67,9 +67,8 @@ const ( // Band boundaries. The internal band is never client-sent. const ( - internalBandStart = OperationCreateTopicWithAssignments - metadataBandStart = OperationCreateStream - partitionBandStart = OperationSendMessages + internalBandStart = OperationCreateTopicWithAssignments + metadataBandStart = OperationCreateStream ) // allOperations lists every declared discriminant in wire order. It backs both @@ -206,12 +205,6 @@ func IsMetadata(operation Operation) bool { return operation >= metadataBandStart && operation <= OperationLeaveConsumerGroup } -// IsPartition reports whether the operation is routed to the shard owning a -// partition. -func IsPartition(operation Operation) bool { - return operation >= partitionBandStart -} - // IsResultFramed reports whether the reply body leads with a committed result // section. Every metadata operation is framed; on the partition plane only the // consumer-offset operations are, because they reject with typed errors at diff --git a/foreign/go/internal/vsr/operation_test.go b/foreign/go/internal/vsr/operation_test.go index 0b7b4784f9..e064ac5250 100644 --- a/foreign/go/internal/vsr/operation_test.go +++ b/foreign/go/internal/vsr/operation_test.go @@ -158,14 +158,6 @@ func TestIsMetadata_ExcludesDeleteSegments(t *testing.T) { } } -func TestIsPartition_CoversTheDataPlaneBand(t *testing.T) { - assert.True(t, IsPartition(OperationSendMessages)) - assert.True(t, IsPartition(OperationStoreConsumerOffset)) - assert.True(t, IsPartition(OperationDeleteConsumerOffset)) - assert.False(t, IsPartition(OperationLeaveConsumerGroup)) - assert.False(t, IsPartition(OperationDeleteSegments)) -} - func TestIsResultFramed_ExcludesSendMessages(t *testing.T) { assert.False(t, IsResultFramed(OperationSendMessages), "a send confirmation is not preceded by a result section") diff --git a/foreign/go/internal/vsr/protocol_parity_test.go b/foreign/go/internal/vsr/protocol_parity_test.go index 6cddd7f167..c9a32700cb 100644 --- a/foreign/go/internal/vsr/protocol_parity_test.go +++ b/foreign/go/internal/vsr/protocol_parity_test.go @@ -428,10 +428,8 @@ func TestProtocolParity_OperationClassification(t *testing.T) { internalStart := rustValues["CreateTopicWithAssignments"] metadataStart := rustValues["CreateStream"] - partitionStart := rustValues["SendMessages"] require.NotZero(t, internalStart) require.NotZero(t, metadataStart) - require.NotZero(t, partitionStart) for name, value := range rustValues { operation := Operation(value) @@ -442,7 +440,6 @@ func TestProtocolParity_OperationClassification(t *testing.T) { assert.Equal(t, internal, IsInternal(operation), "IsInternal(%s)", name) assert.Equal(t, metadata, IsMetadata(operation), "IsMetadata(%s)", name) - assert.Equal(t, value >= partitionStart, IsPartition(operation), "IsPartition(%s)", name) assert.Equal(t, metadata || inResultFramedList, IsResultFramed(operation), "IsResultFramed(%s)", name) assert.True(t, IsKnownOperation(operation), "IsKnownOperation(%s)", name) diff --git a/foreign/go/internal/vsr/session.go b/foreign/go/internal/vsr/session.go index 78226f4a38..3226a998fb 100644 --- a/foreign/go/internal/vsr/session.go +++ b/foreign/go/internal/vsr/session.go @@ -128,14 +128,9 @@ func (s *Session) NextRequestID() (uint64, error) { } // CurrentRequestID returns the watermark without advancing it. Non-replicated -// and partition-plane requests use it because neither consults the client -// table: non-replicated requests route by transport identity, and the -// partition plane replicates in per-partition groups with no dedup table at -// all. The table accepts any id above the watermark with no contiguity -// requirement, so consuming one here would not gap anything; the invariant -// that matters is that every partition request on a session carries the id -// the next metadata operation will claim, and that a partition-plane replay -// is therefore at-least-once. +// requests use it because they never consult the client table: they route by +// transport identity, and the table accepts any id above the watermark with +// no contiguity requirement, so reading here gaps nothing. func (s *Session) CurrentRequestID() uint64 { return s.requestCounter } diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/vsr/ConsensusSession.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/vsr/ConsensusSession.java index ab79ea19ee..16498bd842 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/vsr/ConsensusSession.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/vsr/ConsensusSession.java @@ -52,12 +52,20 @@ public ConsensusSession() { * the whole identity re-arms with a fresh client id so the server sees a * brand-new registration. Returns the request id a Register carries, * which is always zero. + * + *

The request counter is deliberately not rewound. This SDK multiplexes + * a single pinned channel and correlates replies by (operation, request + * id), so a send still in flight when a re-login re-arms would share its + * key with the first send of the new session: the correlation map would + * refuse the second one and a late reply for the first could be handed to + * it. A re-arm registers a fresh client id, which the server admits at + * watermark zero and which accepts any id above it, so carrying the + * counter forward costs nothing on the wire. */ synchronized long beginRegister() { if (registerConsumed || session != null) { regenerateClientId(); session = null; - requestCounter = 1; } registerConsumed = true; return 0; @@ -71,17 +79,31 @@ synchronized void bind(long sessionEpoch) { this.session = sessionEpoch; } - /** Replicated metadata ops consume the monotonic VSR dedup counter. */ + /** + * Replicated ops (metadata and partition) consume the monotonic VSR dedup + * counter. The wire field is a u64 but Java has no unsigned long, so the + * counter is refused at {@link Long#MAX_VALUE} rather than wrapping + * negative and sending an id below the server's watermark. + * + *

Exhaustion is terminal for this instance. {@link #beginRegister()} + * deliberately carries the counter across a re-login to keep pending-reply + * correlation keys unique, so reconnecting cannot rewind it; only a new + * client instance starts a fresh sequence. + */ synchronized long nextRequestId() { if (session == null) { throw new IggyNotConnectedException("Not authenticated, call login first"); } + if (requestCounter == Long.MAX_VALUE) { + throw new IllegalStateException( + "VSR request counter exhausted, create a fresh client instance (reconnecting preserves the counter)"); + } return requestCounter++; } /** - * Partition and non-replicated ops use an independent sequence for reply - * correlation, so they do not create gaps in the metadata dedup sequence. + * Non-replicated ops use an independent sequence for reply correlation, + * so they do not create gaps in the dedup sequence. */ synchronized long nextCorrelationId() { return correlationCounter++; diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/vsr/VsrHeaders.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/vsr/VsrHeaders.java index 35b933ef55..c49244cb06 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/vsr/VsrHeaders.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/vsr/VsrHeaders.java @@ -26,7 +26,11 @@ * Byte offsets and readers for the 256-byte consensus headers, mirroring the * {@code #[repr(C)]} layouts in * {@code core/binary_protocol/src/consensus/header.rs}. All fields are - * little-endian; the checksum fields stay zero by protocol contract. + * little-endian. The checksum fields stay zero by choice, not by protocol + * requirement: the frame and body checksums are not read on the client request + * path, and {@code request_checksum} treats zero as unstamped, which opts out of + * the server's payload comparison. The Rust SDK stamps it for deduped + * operations; this SDK does not yet. */ public final class VsrHeaders { diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/vsr/VsrOperation.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/vsr/VsrOperation.java index 794dd6f286..5ada71abbb 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/vsr/VsrOperation.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/vsr/VsrOperation.java @@ -69,7 +69,6 @@ public final class VsrOperation { private static final int INTERNAL_START = 64; private static final int METADATA_START = 128; - private static final int PARTITION_START = 160; /** * Replicated command code to operation, from the server's @@ -143,10 +142,6 @@ static boolean isMetadata(int operation) { return operation >= METADATA_START && operation <= LEAVE_CONSUMER_GROUP; } - static boolean isPartition(int operation) { - return operation >= PARTITION_START; - } - /** * Whether a reply body for this operation starts with a committed result * section ({@code [count:u32][{index,result} x count]}). diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/vsr/VsrRequestEncoder.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/vsr/VsrRequestEncoder.java index 45859f06d7..4bfcbeed81 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/vsr/VsrRequestEncoder.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/vsr/VsrRequestEncoder.java @@ -76,12 +76,11 @@ public ByteBuf encode(ByteBufAllocator alloc, int commandCode, ByteBuf payload) // sessionless before login. requestId = session.nextCorrelationId(); sessionId = session.sessionOrZero(); - } else if (VsrOperation.isPartition(operation)) { - // Partition ops replicate in their own group without client - // table dedup, so use the independent correlation sequence. - sessionId = session.boundSession(); - requestId = session.nextCorrelationId(); } else { + // Partition ops consume the dedup counter too, even though no + // partition-plane dedup exists yet: dedup needs each send to + // carry a distinct number, and the metadata watermark + // tolerates the gaps. sessionId = session.boundSession(); requestId = session.nextRequestId(); } diff --git a/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/vsr/VsrRequestEncoderTest.java b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/vsr/VsrRequestEncoderTest.java index 426aa0b501..92c89243ba 100644 --- a/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/vsr/VsrRequestEncoderTest.java +++ b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/vsr/VsrRequestEncoderTest.java @@ -120,9 +120,10 @@ void shouldAdvanceRequestIdsForReplicatedCommands() { } @Test - void shouldCorrelatePartitionOpsWithoutAdvancingTheDedupRequestId() { - // Partition ops replicate in their own group with no client-table dedup, so - // they take a correlation id and leave the dedup counter where it was. + void shouldConsumeTheDedupRequestIdForPartitionOps() { + // Dedup identity requires each send to carry a distinct number, so + // partition ops advance the dedup counter exactly like metadata ops + // and the two planes interleave on one sequence. session.beginRegister(); session.bind(42); @@ -132,13 +133,16 @@ void shouldCorrelatePartitionOpsWithoutAdvancingTheDedupRequestId() { ByteBuf second = encoder.encode(alloc, SEND_MESSAGES_CODE, secondPayload); firstPayload.release(); secondPayload.release(); + ByteBuf metadata = encoder.encode(alloc, CREATE_STREAM_CODE, Unpooled.EMPTY_BUFFER); try { assertThat(first.getLongLE(VsrHeaders.REQUEST_ID_OFFSET)).isEqualTo(1); assertThat(second.getLongLE(VsrHeaders.REQUEST_ID_OFFSET)).isEqualTo(2); - assertThat(session.currentRequestId()).isEqualTo(1); + assertThat(metadata.getLongLE(VsrHeaders.REQUEST_ID_OFFSET)).isEqualTo(3); + assertThat(session.currentRequestId()).isEqualTo(4); } finally { first.release(); second.release(); + metadata.release(); } } diff --git a/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/vsr/VsrResponseHandlerTest.java b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/vsr/VsrResponseHandlerTest.java index 4a9e3bc2d6..51c4d3790d 100644 --- a/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/vsr/VsrResponseHandlerTest.java +++ b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/vsr/VsrResponseHandlerTest.java @@ -28,8 +28,10 @@ import org.junit.jupiter.api.AfterEach; import org.junit.jupiter.api.Test; +import java.nio.charset.StandardCharsets; import java.util.concurrent.CompletableFuture; import java.util.concurrent.ExecutionException; +import java.util.concurrent.TimeUnit; import java.util.concurrent.atomic.AtomicInteger; import static org.assertj.core.api.Assertions.assertThat; @@ -37,6 +39,9 @@ class VsrResponseHandlerTest { + private static final int LOGIN_USER_CODE = 38; + private static final int SEND_MESSAGES_CODE = 101; + private final ConsensusSession session = new ConsensusSession(); private final AtomicInteger evictions = new AtomicInteger(); private final AtomicInteger lastEvictionReason = new AtomicInteger(); @@ -251,6 +256,47 @@ void shouldCorrelateRepliesArrivingInReverseOrder() throws Exception { } } + @Test + void shouldCorrelateASendInFlightAcrossReLogin() throws Exception { + // A re-login re-arms the session while an earlier send is still + // pending. Replies correlate by (operation, request id), so the first + // send of the new session must not claim the in-flight one's key: + // registering it would be refused and a late reply for the older send + // would be handed to the newer one. + VsrRequestEncoder encoder = new VsrRequestEncoder(session); + session.beginRegister(); + session.bind(42); + + CompletableFuture inFlight = new CompletableFuture<>(); + long beforeLoginId = registerEncodedSend(encoder, inFlight); + + ByteBuf loginPayload = loginUserPayload(); + encoder.encode(channel.alloc(), LOGIN_USER_CODE, loginPayload).release(); + loginPayload.release(); + session.bind(43); + + CompletableFuture afterLogin = new CompletableFuture<>(); + long afterLoginId = registerEncodedSend(encoder, afterLogin); + + assertThat(afterLoginId).isNotEqualTo(beforeLoginId); + assertThat(channel.isActive()).isTrue(); + + channel.writeInbound( + replyFrame(VsrOperation.SEND_MESSAGES, afterLoginId, Unpooled.wrappedBuffer(new byte[] {2}))); + channel.writeInbound( + replyFrame(VsrOperation.SEND_MESSAGES, beforeLoginId, Unpooled.wrappedBuffer(new byte[] {1}))); + + ByteBuf inFlightResponse = inFlight.get(); + ByteBuf afterLoginResponse = afterLogin.get(); + try { + assertThat(inFlightResponse.readByte()).isEqualTo((byte) 1); + assertThat(afterLoginResponse.readByte()).isEqualTo((byte) 2); + } finally { + inFlightResponse.release(); + afterLoginResponse.release(); + } + } + @Test void shouldCorrelateRepliesForServerRewrittenOperations() throws Exception { int[][] rewrittenOperations = { @@ -296,6 +342,32 @@ private CompletableFuture enqueue(int operation, long requestId) { return future; } + /** + * Encodes a partition send off the live session and registers it the way + * the connection does, returning the request id the encoder minted. + */ + private long registerEncodedSend(VsrRequestEncoder encoder, CompletableFuture future) { + ByteBuf frame = encoder.encode(channel.alloc(), SEND_MESSAGES_CODE, Unpooled.EMPTY_BUFFER); + try { + handler.registerRequest( + channel, frame, future, System.nanoTime() + TimeUnit.MINUTES.toNanos(1), SEND_MESSAGES_CODE); + return VsrHeaders.readRequestId(frame); + } finally { + frame.release(); + } + } + + private static ByteBuf loginUserPayload() { + ByteBuf payload = Unpooled.buffer(); + payload.writeByte(4); + payload.writeBytes("iggy".getBytes(StandardCharsets.UTF_8)); + payload.writeByte(4); + payload.writeBytes("iggy".getBytes(StandardCharsets.UTF_8)); + payload.writeIntLE(0); + payload.writeIntLE(0); + return payload; + } + private static ByteBuf emptyFrame() { ByteBuf frame = Unpooled.buffer(VsrHeaders.HEADER_SIZE); frame.writeZero(VsrHeaders.HEADER_SIZE); diff --git a/foreign/node/scripts/check-vsr-protocol.mjs b/foreign/node/scripts/check-vsr-protocol.mjs index dcbe060fb3..41426e5bbc 100644 --- a/foreign/node/scripts/check-vsr-protocol.mjs +++ b/foreign/node/scripts/check-vsr-protocol.mjs @@ -252,7 +252,6 @@ const operationModule = await import( ); const internalStart = rustOperations.get('CreateTopicWithAssignments'); const metadataStart = rustOperations.get('CreateStream'); -const partitionStart = rustOperations.get('SendMessages'); const rustMetadataNames = new Set( [...(rustOperation.match( /fn is_metadata[\s\S]*?matches!\(\s*self,([\s\S]*?)\)\s*\n\s*\}/ @@ -276,11 +275,6 @@ for (const [name, value] of rustOperations) { metadata, `Node isMetadata(${name}) differs from Rust is_metadata` ); - assert.equal( - operationModule.isPartition(value), - value >= partitionStart, - `Node isPartition(${name}) differs from Rust is_partition` - ); assert.equal( operationModule.isResultFramed(value), metadata || rustResultFramedNames.has(name), diff --git a/foreign/node/src/wire/vsr/header.ts b/foreign/node/src/wire/vsr/header.ts index d54182c5ba..e7128a1159 100644 --- a/foreign/node/src/wire/vsr/header.ts +++ b/foreign/node/src/wire/vsr/header.ts @@ -113,8 +113,10 @@ const U64_MASK = 0xFFFFFFFFFFFFFFFFn; /** * Encodes a 256-byte request header. Only the six fields the server reads - * are written; the checksums stay zero, matching the Rust SDK's contract - * with the VSR server. + * are written. The checksums stay zero: the frame and body checksums are not + * read on the client request path, and `request_checksum` treats zero as + * unstamped, which opts out of the server's payload comparison. Stamping it is + * optional -- the Rust SDK does for deduped ops, this SDK does not yet. */ export const encodeRequestHeader = (fields: RequestHeaderFields): Buffer => { const header = Buffer.alloc(HEADER_SIZE); diff --git a/foreign/node/src/wire/vsr/index.ts b/foreign/node/src/wire/vsr/index.ts index 24091ca1fe..0fdee6f365 100644 --- a/foreign/node/src/wire/vsr/index.ts +++ b/foreign/node/src/wire/vsr/index.ts @@ -20,11 +20,7 @@ import type { CommandResponse } from '../../client/client.type.js'; import { COMMAND_CODE } from '../command.code.js'; import { responseError } from '../error.utils.js'; import { HEADER_SIZE, encodeRequestHeader } from './header.js'; -import { - Operation, - isPartition, - operationForCode, -} from './operation.js'; +import { Operation, operationForCode } from './operation.js'; import { deserializeLoginRegister, serializeLoginRegister, @@ -80,9 +76,10 @@ export class VsrSession { } else { if (this.state.session === null) throw responseError(command, UNAUTHENTICATED); - request = isPartition(operation) - ? this.state.currentRequestId() - : this.state.nextRequestId(); + // Partition ops consume an id too, even though no partition-plane dedup + // exists yet: dedup needs each send to carry a distinct number, and the + // metadata watermark tolerates the gaps. + request = this.state.nextRequestId(); session = this.state.session; } diff --git a/foreign/node/src/wire/vsr/operation.test.ts b/foreign/node/src/wire/vsr/operation.test.ts index daac920474..4ce539ffe9 100644 --- a/foreign/node/src/wire/vsr/operation.test.ts +++ b/foreign/node/src/wire/vsr/operation.test.ts @@ -22,7 +22,6 @@ import { isInternal, isKnownOperation, isMetadata, - isPartition, isResultFramed, Operation, operationForCode @@ -87,8 +86,6 @@ describe('VSR operation classification', () => { assert.equal(isMetadata(Operation.LeaveConsumerGroup), true); assert.equal(isMetadata(Operation.DeleteSegments), false); assert.equal(isMetadata(150), false); - assert.equal(isPartition(Operation.SendMessages), true); - assert.equal(isPartition(159), false); assert.equal(isResultFramed(Operation.StoreConsumerOffset), true); assert.equal(isResultFramed(Operation.DeleteConsumerOffset), true); assert.equal(isResultFramed(Operation.SendMessages), false); diff --git a/foreign/node/src/wire/vsr/operation.ts b/foreign/node/src/wire/vsr/operation.ts index 2781668885..030d9f0fef 100644 --- a/foreign/node/src/wire/vsr/operation.ts +++ b/foreign/node/src/wire/vsr/operation.ts @@ -63,7 +63,6 @@ export const Operation = { const INTERNAL_START = 64; const METADATA_START = 128; -const PARTITION_START = 160; /** * Replicated command code to `Operation` mapping, the client half of the @@ -126,10 +125,6 @@ export const isMetadata = (operation: number): boolean => { operation <= Operation.LeaveConsumerGroup; }; -/** Partition band is a bare range, mirroring `Operation::is_partition`. */ -export const isPartition = (operation: number): boolean => - operation >= PARTITION_START; - /** Whether a reply body leads with a committed result section. */ export const isResultFramed = (operation: number): boolean => isMetadata(operation) || diff --git a/foreign/node/src/wire/vsr/session.ts b/foreign/node/src/wire/vsr/session.ts index f61cbc3f0b..5f381fbbbf 100644 --- a/foreign/node/src/wire/vsr/session.ts +++ b/foreign/node/src/wire/vsr/session.ts @@ -24,9 +24,9 @@ const MAX_U64 = 0xFFFF_FFFF_FFFF_FFFFn; * * Each client instance generates an ephemeral random `clientId` (u128). * After a Register commits, the server assigns a `session` number (commit op - * number). Replicated metadata requests advance a monotonic request watermark. - * Non-replicated and partition-plane requests reuse the current value because - * the server only applies request sequencing to replicated metadata. + * number). Every replicated request (metadata and partition) advances a + * monotonic request watermark; non-replicated requests reuse the current + * value because they bypass server-side request sequencing. */ export class ConsensusSession { private _clientId: bigint; diff --git a/foreign/node/src/wire/vsr/vsr.test.ts b/foreign/node/src/wire/vsr/vsr.test.ts index 60e9c19088..cbc9feb7e4 100644 --- a/foreign/node/src/wire/vsr/vsr.test.ts +++ b/foreign/node/src/wire/vsr/vsr.test.ts @@ -79,7 +79,7 @@ describe('VSR custom request framing', () => { ); }); - it('does not advance for non-replicated or partition operations', () => { + it('does not advance for non-replicated operations', () => { const session = new VsrSession(7n); session.bind(42n); const custom = session.encode( @@ -88,26 +88,43 @@ describe('VSR custom request framing', () => { ); assert.equal(custom.readBigUInt64LE(REQUEST_OFFSET.request), 1n); - const partition = session.encode( - COMMAND_CODE.SendMessages, - serializeSendMessages( - 1, - 2, - [{ payload: 'x' }], - Partitioning.PartitionId(3) - ) + const metadata = session.encode( + COMMAND_CODE.CreateStream, + Buffer.alloc(0) ); + assert.equal(metadata.readBigUInt64LE(REQUEST_OFFSET.request), 1n); + assert.equal(metadata.length, HEADER_SIZE); + }); + + it('partition operations consume a distinct id per send', () => { + // Dedup identity requires each send to carry a distinct number, so + // partition ops advance the counter exactly like metadata ops and the + // two planes interleave on one sequence. + const session = new VsrSession(7n); + session.bind(42n); + const sendMessages = () => + session.encode( + COMMAND_CODE.SendMessages, + serializeSendMessages( + 1, + 2, + [{ payload: 'x' }], + Partitioning.PartitionId(3) + ) + ); + + const first = sendMessages(); assert.equal( - partition.readUInt8(REQUEST_OFFSET.operation), + first.readUInt8(REQUEST_OFFSET.operation), Operation.SendMessages ); - assert.equal(partition.readBigUInt64LE(REQUEST_OFFSET.request), 1n); + assert.equal(first.readBigUInt64LE(REQUEST_OFFSET.request), 1n); + assert.equal(sendMessages().readBigUInt64LE(REQUEST_OFFSET.request), 2n); const metadata = session.encode( COMMAND_CODE.CreateStream, Buffer.alloc(0) ); - assert.equal(metadata.readBigUInt64LE(REQUEST_OFFSET.request), 1n); - assert.equal(metadata.length, HEADER_SIZE); + assert.equal(metadata.readBigUInt64LE(REQUEST_OFFSET.request), 3n); }); }); From e9b1c9a12d0baaf74129d6925c2f114663536151 Mon Sep 17 00:00:00 2001 From: Hubert Gruszecki Date: Mon, 31 Aug 2026 21:43:07 +0200 Subject: [PATCH 022/182] fix(cluster): stop a late partition materialiser from wedging its group (#4007) A partition group with no history of its own seeds its view and log_view from the metadata view this replica happens to publish. Nothing bounds that view by the group's real one, so a replica that materialises the group late, after a further metadata election, starts Normal on an empty log above every peer. That empty log is canonical in the next DVC merge: as a backup it collects a nack quorum against committed ops and the group deadlocks for good, the superblock persisting the view so a restart re-wedges; as the primary-by-index it answers the real primary's heartbeat with a StartView at op 0 that resets the peers' heads under their commit floor, and every new write is fenced. Introduced with the seed in #3985. Record the view the admitting primary mints the create in on the committed metadata entry and seed from that instead. Every replica reads the same value, and the group's view never drops below it, so an empty log at the creation view is the floor a view-0 start used to be, never canonical over committed history. The reconciler no longer needs the published-view sentinel, its deferral, or the double atomic read that came with it. The value rides the request body, not the prepare header: a header's view is the one field a post-view-change retransmit restamps (replicate_preflight refuses the frame otherwise), so a header-derived value is a per-replica property of delivery history and durably diverges the metadata STM. The body is sealed by checksum_body and never restamped; entries journaled before the trailing field decode it as 0, the safe floor. Chosen over electing into the seed: no extra round trip per group, no consensus change, and a late materialiser joins as a backup at or below its peers instead of forcing an election it may win with an empty log. The snapshot format moves to version 5 with the new trailing field defaulted, so version 3 and 4 checkpoints still decode. --- .../create_partitions_with_assignments.rs | 44 ++++ .../topics/create_topic_with_assignments.rs | 46 ++++ core/consensus/src/impls.rs | 47 ++-- core/metadata/src/impls/metadata.rs | 7 + core/metadata/src/stm/authz.rs | 1 + core/metadata/src/stm/consumer_group.rs | 1 + core/metadata/src/stm/snapshot.rs | 33 ++- core/metadata/src/stm/stream.rs | 126 +++++++++- core/server/src/bootstrap.rs | 10 +- core/server/src/dispatch.rs | 1 + core/server/src/partition_helpers.rs | 33 +-- core/server/src/partition_reconciler.rs | 219 ++++-------------- core/server/src/responses.rs | 6 +- core/shard/src/lib.rs | 11 +- core/simulator/src/lib.rs | 217 ++++++++++++++--- core/simulator/src/replica.rs | 6 +- 16 files changed, 544 insertions(+), 264 deletions(-) diff --git a/core/binary_protocol/src/requests/partitions/create_partitions_with_assignments.rs b/core/binary_protocol/src/requests/partitions/create_partitions_with_assignments.rs index 1350541aff..904ed6e5fa 100644 --- a/core/binary_protocol/src/requests/partitions/create_partitions_with_assignments.rs +++ b/core/binary_protocol/src/requests/partitions/create_partitions_with_assignments.rs @@ -29,6 +29,10 @@ fn usize_to_u32(value: usize, context: &str) -> u32 { pub struct CreatePartitionsWithAssignmentsRequest { pub request: CreatePartitionsRequest, pub partitions: Vec, + /// View the admitting primary minted this create in. Rides the body, which + /// `checksum_body` seals, so unlike the header's `view` it is never restamped + /// by a post-view-change retransmit: every replica decodes the same value. + pub created_view: u32, } impl WireEncode for CreatePartitionsWithAssignmentsRequest { @@ -40,6 +44,7 @@ impl WireEncode for CreatePartitionsWithAssignmentsRequest { .iter() .map(WireEncode::encoded_size) .sum::() + + 4 } fn encode(&self, buf: &mut BytesMut) { @@ -55,6 +60,7 @@ impl WireEncode for CreatePartitionsWithAssignmentsRequest { for partition in &self.partitions { partition.encode(buf); } + buf.put_u32_le(self.created_view); } } @@ -81,10 +87,22 @@ impl WireDecode for CreatePartitionsWithAssignmentsRequest { partitions.push(partition); } + // Trailing field: entries journaled before it existed end here and read + // 0, the view every plane starts in. 1-3 leftover bytes are corruption, + // not an old shape, so they still error. + let created_view = if offset < buf.len() { + let view = read_u32_le(buf, offset)?; + offset += 4; + view + } else { + 0 + }; + Ok(( Self { request, partitions, + created_view, }, offset, )) @@ -117,10 +135,36 @@ mod tests { consensus_group_id: 12, }, ], + created_view: 9, }; let bytes = request.to_bytes(); let (decoded, consumed) = CreatePartitionsWithAssignmentsRequest::decode(&bytes).unwrap(); assert_eq!(consumed, bytes.len()); assert_eq!(decoded, request); } + + #[test] + fn decode_without_trailing_view_reads_zero() { + let request = CreatePartitionsWithAssignmentsRequest { + request: CreatePartitionsRequest { + stream_id: WireIdentifier::numeric(1), + topic_id: WireIdentifier::numeric(2), + partitions_count: 1, + }, + partitions: vec![CreatedPartitionAssignment { + partition_id: 3, + consensus_group_id: 11, + }], + created_view: 9, + }; + // Bytes journaled before the field existed: same shape, no trailing u32. + let bytes = request.to_bytes(); + let old_bytes = &bytes[..bytes.len() - 4]; + let (decoded, consumed) = + CreatePartitionsWithAssignmentsRequest::decode(old_bytes).unwrap(); + assert_eq!(consumed, old_bytes.len()); + assert_eq!(decoded.created_view, 0); + assert_eq!(decoded.partitions, request.partitions); + assert_eq!(decoded.request, request.request); + } } diff --git a/core/binary_protocol/src/requests/topics/create_topic_with_assignments.rs b/core/binary_protocol/src/requests/topics/create_topic_with_assignments.rs index b6665b6188..632286530d 100644 --- a/core/binary_protocol/src/requests/topics/create_topic_with_assignments.rs +++ b/core/binary_protocol/src/requests/topics/create_topic_with_assignments.rs @@ -38,6 +38,10 @@ pub struct CreateTopicWithAssignmentsRequest { pub request: CreateTopicRequest, pub derived_options: WireOptions, pub partitions: Vec, + /// View the admitting primary minted this create in. Rides the body, which + /// `checksum_body` seals, so unlike the header's `view` it is never restamped + /// by a post-view-change retransmit: every replica decodes the same value. + pub created_view: u32, } impl WireEncode for CreateTopicWithAssignmentsRequest { @@ -51,6 +55,7 @@ impl WireEncode for CreateTopicWithAssignmentsRequest { .iter() .map(WireEncode::encoded_size) .sum::() + + 4 } fn encode(&self, buf: &mut BytesMut) { @@ -71,6 +76,7 @@ impl WireEncode for CreateTopicWithAssignmentsRequest { for partition in &self.partitions { partition.encode(buf); } + buf.put_u32_le(self.created_view); } } @@ -116,11 +122,23 @@ impl WireDecode for CreateTopicWithAssignmentsRequest { partitions.push(partition); } + // Trailing field: entries journaled before it existed end here and read + // 0, the view every plane starts in. 1-3 leftover bytes are corruption, + // not an old shape, so they still error. + let created_view = if offset < buf.len() { + let view = read_u32_le(buf, offset)?; + offset += 4; + view + } else { + 0 + }; + Ok(( Self { request, derived_options, partitions, + created_view, }, offset, )) @@ -165,6 +183,7 @@ mod tests { consensus_group_id: 2, }, ], + created_view: 7, }; let bytes = request.to_bytes(); let (decoded, consumed) = CreateTopicWithAssignmentsRequest::decode(&bytes).unwrap(); @@ -172,6 +191,32 @@ mod tests { assert_eq!(decoded, request); } + #[test] + fn decode_without_trailing_view_reads_zero() { + let request = CreateTopicWithAssignmentsRequest { + request: CreateTopicRequest { + stream_id: WireIdentifier::numeric(1), + partitions_count: 1, + name: WireName::new("events").unwrap(), + options: WireOptions::empty(), + }, + derived_options: WireOptions::empty(), + partitions: vec![CreatedPartitionAssignment { + partition_id: 0, + consensus_group_id: 1, + }], + created_view: 7, + }; + // Bytes journaled before the field existed: same shape, no trailing u32. + let bytes = request.to_bytes(); + let old_bytes = &bytes[..bytes.len() - 4]; + let (decoded, consumed) = CreateTopicWithAssignmentsRequest::decode(old_bytes).unwrap(); + assert_eq!(consumed, old_bytes.len()); + assert_eq!(decoded.created_view, 0); + assert_eq!(decoded.partitions, request.partitions); + assert_eq!(decoded.request, request.request); + } + #[test] fn roundtrip_with_empty_blocks() { let request = CreateTopicWithAssignmentsRequest { @@ -183,6 +228,7 @@ mod tests { }, derived_options: WireOptions::empty(), partitions: vec![], + created_view: 0, }; let bytes = request.to_bytes(); let (decoded, consumed) = CreateTopicWithAssignmentsRequest::decode(&bytes).unwrap(); diff --git a/core/consensus/src/impls.rs b/core/consensus/src/impls.rs index 28865344f7..edc60c0861 100644 --- a/core/consensus/src/impls.rs +++ b/core/consensus/src/impls.rs @@ -1152,14 +1152,21 @@ pub struct VsrRestore<'a> { /// durable record exists; `log_view` cannot be inferred and stays 0. pub view_fallback: Option, /// Starting `(view, log_view)` for a group with NO history of its own, - /// consulted only when neither of the two above applies. + /// consulted only when neither of the two above applies: the view the + /// group was created in, as recorded by the metadata plane. + /// + /// Safe to start `Normal` in without a quorum only because every replica + /// seeds the same value and the group's real view never drops below it: + /// an empty log at the creation view is the floor a view-0 start used to + /// be, never canonical over committed history. Seeded above the group's + /// real view it would be canonical: the empty log wins the next DVC merge + /// and wedges the group on its committed ops. /// /// Sets `log_view` as well as `view`, unlike `view_fallback`: the log is /// empty, so there is no history to misattribute to the view, and a /// primary whose `log_view` lags its `view` is treated as mid-transition /// and answers no `RequestStartView` probe (see - /// [`VsrConsensus::handle_request_start_view`]) - it would hold its own group's - /// probes open forever. + /// [`VsrConsensus::handle_request_start_view`]). /// /// Deliberately NOT marked superblock-durable: nothing has been written /// yet, and claiming otherwise would let a restart resume a view no record @@ -1192,12 +1199,14 @@ pub struct FreshGroupStart { /// /// `restarted` is the caller's own evidence of a prior life, since the two /// paths read it differently: a partition directory already on disk for the -/// server, a retained in-memory log for the simulator. +/// server, a retained in-memory log for the simulator. `created_view` is the +/// view the metadata plane created the group in; see +/// [`VsrRestore::seed_view`]. #[must_use] pub const fn fresh_group_start( restarted: bool, durable_view: Option<(u32, u32)>, - metadata_view: Option, + created_view: u32, ) -> FreshGroupStart { let join = if restarted { JoinMode::ProbeAsBackup { @@ -1211,7 +1220,7 @@ pub const fn fresh_group_start( // `StartView` to move it forward, and seeded above that the reply reads as // stale and is dropped. let seed_view = match (durable_view, join) { - (None, JoinMode::Init) => metadata_view, + (None, JoinMode::Init) => Some(created_view), _ => None, }; FreshGroupStart { join, seed_view } @@ -1295,7 +1304,7 @@ impl> VsrConsensus { tracing::info!( group, view, - "seeded a group with no history of its own into the current view" + "seeded a group with no history of its own at its creation view" ); consensus.set_view(view); consensus.set_log_view(view); @@ -4061,33 +4070,23 @@ where mod fresh_group_start_tests { use super::{JoinMode, fresh_group_start}; - const METADATA_VIEW: u32 = 4; + const CREATED_VIEW: u32 = 4; /// The case the seed exists for: a group created after the metadata plane /// elected must not start at view 0, or it names a replica the roster does /// not advertise and no client can be routed to it. #[test] - fn given_a_fresh_group_when_the_metadata_view_moved_should_seed_that_view() { - let start = fresh_group_start(false, None, Some(METADATA_VIEW)); - assert_eq!(start.join, JoinMode::Init); - assert_eq!(start.seed_view, Some(METADATA_VIEW)); - } - - /// No published view is "no opinion", not "view 0". The caller defers - /// materialising rather than seeding; this only asserts nothing is - /// invented here. - #[test] - fn given_a_fresh_group_when_no_metadata_view_is_known_should_not_seed() { - let start = fresh_group_start(false, None, None); + fn given_a_fresh_group_when_created_after_an_election_should_seed_the_creation_view() { + let start = fresh_group_start(false, None, CREATED_VIEW); assert_eq!(start.join, JoinMode::Init); - assert_eq!(start.seed_view, None); + assert_eq!(start.seed_view, Some(CREATED_VIEW)); } /// A durable record outranks the seed: it is what this replica actually - /// promised, and the seed is a guess about someone else's plane. + /// promised, and the seed only describes where the group began. #[test] fn given_a_durable_record_when_seeding_should_prefer_the_record() { - let start = fresh_group_start(false, Some((7, 7)), Some(METADATA_VIEW)); + let start = fresh_group_start(false, Some((7, 7)), CREATED_VIEW); assert_eq!(start.seed_view, None); } @@ -4096,7 +4095,7 @@ mod fresh_group_start_tests { /// reads as stale and the replica never rejoins. #[test] fn given_a_prior_life_when_seeding_should_probe_without_a_seed() { - let start = fresh_group_start(true, None, Some(METADATA_VIEW)); + let start = fresh_group_start(true, None, CREATED_VIEW); assert!(matches!(start.join, JoinMode::ProbeAsBackup { .. })); assert_eq!(start.seed_view, None); } diff --git a/core/metadata/src/impls/metadata.rs b/core/metadata/src/impls/metadata.rs index b7b7f83375..6ff355e474 100644 --- a/core/metadata/src/impls/metadata.rs +++ b/core/metadata/src/impls/metadata.rs @@ -3425,6 +3425,11 @@ where request, derived_options, partitions, + // Minted once here, on the admitting primary, and sealed by + // `checksum_body`: the header's `view` is restamped on a + // post-view-change retransmit, so a body-carried copy is the + // only per-op view every replica commits identically. + created_view: consensus.view(), } .to_bytes(); Ok(build_prepare_message( @@ -3458,6 +3463,8 @@ where let body = PersistedCreatePartitionsRequest { request, partitions, + // Same body-carried view as `PersistedCreateTopicRequest`. + created_view: consensus.view(), } .to_bytes(); Ok(build_prepare_message( diff --git a/core/metadata/src/stm/authz.rs b/core/metadata/src/stm/authz.rs index 667b87f026..1e7ec1ec35 100644 --- a/core/metadata/src/stm/authz.rs +++ b/core/metadata/src/stm/authz.rs @@ -486,6 +486,7 @@ mod tests { fn create_topic_body(stream_id: u32, name: &str) -> bytes::Bytes { CreateTopicWithAssignmentsRequest { + created_view: 0, request: CreateTopicRequest { stream_id: WireIdentifier::numeric(stream_id), partitions_count: 1, diff --git a/core/metadata/src/stm/consumer_group.rs b/core/metadata/src/stm/consumer_group.rs index 22a66c7d40..0e70642bb0 100644 --- a/core/metadata/src/stm/consumer_group.rs +++ b/core/metadata/src/stm/consumer_group.rs @@ -969,6 +969,7 @@ mod tests { IggyTimestamp::now(), ); let create_topic = CreateTopicWithAssignmentsRequest { + created_view: 0, request: CreateTopicRequest { stream_id: WireIdentifier::numeric(0), partitions_count: 1, diff --git a/core/metadata/src/stm/snapshot.rs b/core/metadata/src/stm/snapshot.rs index b59a960000..2185b439eb 100644 --- a/core/metadata/src/stm/snapshot.rs +++ b/core/metadata/src/stm/snapshot.rs @@ -52,13 +52,17 @@ use crate::stm::user::UsersSnapshot; /// position out of place. The bump is what turns that into /// `UnsupportedFormatVersion` instead of an opaque deserializer error, or worse a /// silent misread. -pub const SNAPSHOT_FORMAT_VERSION: u32 = 4; +/// +/// Version 5: `PartitionSnapshot` gained `created_view` as a trailing, defaulted +/// field. +pub const SNAPSHOT_FORMAT_VERSION: u32 = 5; /// Oldest format version [`MetadataSnapshot::decode`] still reads. /// -/// Version 4 added the client table's dedup fences as a trailing, defaulted -/// field, so a version 3 checkpoint decodes under the current layout and simply -/// carries none. Accepting it is what lets a node upgrade in place instead of +/// Versions 4 and 5 each appended a trailing, defaulted field (the client +/// table's dedup fences, then `PartitionSnapshot::created_view`), so a version 3 +/// or 4 checkpoint decodes under the current layout and simply carries the +/// defaults. Accepting them is what lets a node upgrade in place instead of /// refusing its own last checkpoint; anything older than 3 changed field /// positions and is refused as before. pub const MIN_READABLE_SNAPSHOT_FORMAT_VERSION: u32 = 3; @@ -568,7 +572,7 @@ mod tests { // operator's boot reading one field's bytes as another's. Changing either // number is the reminder to change the other. const FIELD_COUNT: u32 = 7; - const PINNED_VERSION: u32 = 4; + const PINNED_VERSION: u32 = 5; let encoded = MetadataSnapshot::new(0).encode().unwrap(); let mut cursor = encoded.as_slice(); @@ -597,6 +601,8 @@ mod tests { // keeps version 3 readable, so a further append here needs the same // treatment or a bump. const CLIENT_TABLE_FIELD_COUNT: u32 = 2; + // Version 5 appended `created_view` the same way. + const PARTITION_FIELD_COUNT: u32 = 7; let client_table = consensus::ClientTableSnapshot { slots: Vec::new(), @@ -634,6 +640,22 @@ mod tests { "TopicSnapshot's field count changed; bump SNAPSHOT_FORMAT_VERSION with it" ); + let partition = PartitionSnapshot { + id: 0, + consensus_group_id: 0, + created_at: IggyTimestamp::default(), + created_revision: 0, + deleted_up_to_offset: 0, + purge_generation: 0, + created_view: 0, + }; + let encoded = rmp_serde::to_vec(&partition).unwrap(); + assert_eq!( + rmp::decode::read_array_len(&mut encoded.as_slice()).unwrap(), + PARTITION_FIELD_COUNT, + "PartitionSnapshot's field count changed; default the new field or bump the version" + ); + let stream = StreamSnapshot { id: 0, name: String::new(), @@ -861,6 +883,7 @@ mod tests { created_revision: 0, deleted_up_to_offset: 0, purge_generation: 0, + created_view: 0, }], consumer_groups: Vec::new(), // Nonzero and distinct from every id above so the diff --git a/core/metadata/src/stm/stream.rs b/core/metadata/src/stm/stream.rs index 5e6d7a4d9d..119a9f4762 100644 --- a/core/metadata/src/stm/stream.rs +++ b/core/metadata/src/stm/stream.rs @@ -89,6 +89,10 @@ pub struct PartitionSnapshot { /// `#[serde(default)]` so pre-purge snapshots restore at 0. #[serde(default)] pub purge_generation: u64, + /// `#[serde(default)]` so snapshots predating this field restore at 0, the + /// view every group started in before it existed. + #[serde(default)] + pub created_view: u32, } #[derive(Debug, Clone)] @@ -101,6 +105,15 @@ pub struct Partition { /// reused the slab key, so the local partition is stale and must be torn /// down before rebuild. pub created_revision: u64, + /// View the admitting primary minted the create in, read off the request + /// body (never the prepare header, whose `view` a post-view-change + /// retransmit restamps per delivery). Every replica seeds the partition's + /// consensus group from it, so they all name the same primary: the metadata + /// primary at creation, which is what the roster advertises. Never above + /// the group's current view, since the group starts here and views only + /// climb, so a replica materialising late joins at or below its peers, the + /// floor a view-0 start used to be. + pub created_view: u32, /// Replicated delete watermark: the reconciler on every replica removes /// sealed segments with `end_offset` below this. Advanced monotonically by /// `TruncatePartition` (the resolved form of a client `DeleteSegments`). @@ -123,12 +136,14 @@ impl Partition { consensus_group_id: u64, created_at: IggyTimestamp, created_revision: u64, + created_view: u32, ) -> Self { Self { id, consensus_group_id, created_at, created_revision, + created_view, deleted_up_to_offset: 0, purge_generation: 0, } @@ -1363,6 +1378,22 @@ impl Streams { /// nothing in the type enforces that density. #[must_use] pub fn created_revision_for_namespace(&self, namespace: IggyNamespace) -> Option { + self.with_committed_partition(namespace, |partition| partition.created_revision) + } + + /// Committed [`Partition::created_view`] for the exact partition the + /// `namespace` tuple denotes; resolved like + /// [`Self::created_revision_for_namespace`]. + #[must_use] + pub fn created_view_for_namespace(&self, namespace: IggyNamespace) -> Option { + self.with_committed_partition(namespace, |partition| partition.created_view) + } + + fn with_committed_partition( + &self, + namespace: IggyNamespace, + read: impl FnOnce(&Partition) -> T, + ) -> Option { self.inner.read(|inner| { let stream = inner.items.get(namespace.stream_id())?; let topic = stream.topics.get(namespace.topic_id())?; @@ -1370,13 +1401,13 @@ impl Streams { if let Some(partition) = topic.partitions.get(partition_id) && partition.id == partition_id { - return Some(partition.created_revision); + return Some(read(partition)); } topic .partitions .iter() .find(|partition| partition.id == partition_id) - .map(|partition| partition.created_revision) + .map(read) }) } @@ -1408,6 +1439,7 @@ impl Streams { stream_slab: usize, topic_slab: usize, target_partitions: &[CreatedPartitionAssignment], + created_view: u32, ) { let stream_wire = WireIdentifier::numeric(u32::try_from(stream_slab).expect("sim stream slab fits u32")); @@ -1439,6 +1471,7 @@ impl Streams { }, derived_options: WireOptions::empty(), partitions, + created_view, }, IggyTimestamp::from(1), )) @@ -1458,12 +1491,20 @@ impl Streams { /// the missing partition. Mirrors [`Users::ensure_root_user`](crate::stm::user::Users::ensure_root_user): a seed /// helper that bypasses consensus, never a production runtime path. /// + /// `created_view` stands in for the view a real create's admitting primary + /// mints into the request body; see [`Partition::created_view`]. + /// /// # Panics /// Panics if the apply is attempted on a reader handle rather than the /// writer (never for the simulator's writer-backed STM), or if a slab id /// exceeds the `u32` wire identifier space. #[cfg(any(test, feature = "simulator"))] - pub fn seed_namespace(&self, namespace: IggyNamespace, consensus_group_id: u64) { + pub fn seed_namespace( + &self, + namespace: IggyNamespace, + consensus_group_id: u64, + created_view: u32, + ) { let stream_slab = namespace.stream_id(); let topic_slab = namespace.topic_id(); let partition_id = @@ -1496,7 +1537,7 @@ impl Streams { }) .collect(); self.seed_stream_slabs(stream_slab); - self.seed_topic_slabs(stream_slab, topic_slab, &target_partitions); + self.seed_topic_slabs(stream_slab, topic_slab, &target_partitions, created_view); if self.created_revision_for_namespace(namespace).is_some() { return; @@ -1529,6 +1570,7 @@ impl Streams { .expect("sim partition count fits u32"), }, partitions, + created_view, }, IggyTimestamp::from(1), )) @@ -1850,6 +1892,7 @@ impl StateHandler for CreateTopicWithAssignmentsRequest { consensus_group_id: partition.consensus_group_id, created_at: timestamp, created_revision: new_revision, + created_view: self.created_view, deleted_up_to_offset: 0, purge_generation: 0, }; @@ -2140,6 +2183,7 @@ impl StateHandler for CreatePartitionsWithAssignmentsRequest { consensus_group_id: partition.consensus_group_id, created_at: timestamp, created_revision: new_revision, + created_view: self.created_view, deleted_up_to_offset: 0, purge_generation: 0, }); @@ -2247,6 +2291,7 @@ impl Snapshotable for Streams { consensus_group_id: p.consensus_group_id, created_at: p.created_at, created_revision: p.created_revision, + created_view: p.created_view, deleted_up_to_offset: p.deleted_up_to_offset, purge_generation: p.purge_generation, }) @@ -2365,6 +2410,7 @@ impl StreamsInner { consensus_group_id: p.consensus_group_id, created_at: p.created_at, created_revision: p.created_revision, + created_view: p.created_view, deleted_up_to_offset: p.deleted_up_to_offset, purge_generation: p.purge_generation, }) @@ -2489,6 +2535,7 @@ mod tests { ..TopicCreateOptions::default() }; let request = CreateTopicWithAssignmentsRequest { + created_view: 0, request: WireCreateTopicRequest { stream_id: WireIdentifier::numeric(0), partitions_count: 1, @@ -2523,6 +2570,51 @@ mod tests { ); } + /// The seed every replica materialises a group from is the body-carried + /// `created_view`, minted once by the admitting primary. A header-derived + /// view would differ per replica after a post-view-change retransmit. + #[test] + fn create_ops_record_the_body_carried_view_on_the_partition() { + let mut inner = StreamsInner::new(); + create_stream(&mut inner, "s"); + + let create_topic = CreateTopicWithAssignmentsRequest { + created_view: 7, + request: WireCreateTopicRequest { + stream_id: WireIdentifier::numeric(0), + partitions_count: 1, + name: WireName::new("t").unwrap(), + options: WireOptions::empty(), + }, + derived_options: WireOptions::empty(), + partitions: vec![CreatedPartitionAssignment { + partition_id: 0, + consensus_group_id: 1, + }], + }; + let reply = StateHandler::apply(&create_topic, &mut inner, IggyTimestamp::from(1)); + assert_eq!(reply.code, 0, "create topic must succeed"); + + let create_partitions = CreatePartitionsWithAssignmentsRequest { + created_view: 9, + request: CreatePartitionsRequest { + stream_id: WireIdentifier::numeric(0), + topic_id: WireIdentifier::numeric(0), + partitions_count: 1, + }, + partitions: vec![CreatedPartitionAssignment { + partition_id: 0, + consensus_group_id: 2, + }], + }; + let reply = StateHandler::apply(&create_partitions, &mut inner, IggyTimestamp::from(2)); + assert_eq!(reply.code, 0, "create partitions must succeed"); + + let topic = inner.items.get(0).unwrap().topics.get(0).unwrap(); + assert_eq!(topic.partitions[0].created_view, 7); + assert_eq!(topic.partitions[1].created_view, 9); + } + /// A client may send the literal 0 that means "resolve the default". The /// merge lets that explicit entry win over the derived one, so the map has /// to be rewritten from the resolved value: otherwise the persisted map @@ -2558,6 +2650,7 @@ mod tests { &mut sentinel, ); let request = CreateTopicWithAssignmentsRequest { + created_view: 0, request: WireCreateTopicRequest { stream_id: WireIdentifier::numeric(0), partitions_count: 1, @@ -2656,6 +2749,7 @@ mod tests { for stream_id in 0..2u32 { for topic_name in ["logs", "events"] { let create_topic = CreateTopicWithAssignmentsRequest { + created_view: 0, request: make_topic_request(stream_id, 2, topic_name), derived_options: WireOptions::empty(), partitions: vec![ @@ -2724,6 +2818,7 @@ mod tests { let mut inner = StreamsInner::new(); create_stream(&mut inner, "stream"); let create_topic = CreateTopicWithAssignmentsRequest { + created_view: 0, request: make_topic_request(0, 2, "topic"), derived_options: WireOptions::empty(), partitions: vec![ @@ -2752,6 +2847,7 @@ mod tests { let mut inner = StreamsInner::new(); create_stream(&mut inner, "stream"); let create_topic = CreateTopicWithAssignmentsRequest { + created_view: 0, request: make_topic_request(0, 2, "topic"), derived_options: WireOptions::empty(), partitions: vec![ @@ -2768,6 +2864,7 @@ mod tests { let _ = StateHandler::apply(&create_topic, &mut inner, IggyTimestamp::now()); let create_partitions = CreatePartitionsWithAssignmentsRequest { + created_view: 0, request: WireCreatePartitionsRequest { stream_id: WireIdentifier::numeric(0), topic_id: WireIdentifier::numeric(0), @@ -2807,6 +2904,7 @@ mod tests { let mut inner = StreamsInner::new(); create_stream(&mut inner, "stream"); let create_topic = CreateTopicWithAssignmentsRequest { + created_view: 0, request: make_topic_request(0, 2, "topic"), derived_options: WireOptions::empty(), partitions: vec![ @@ -2841,6 +2939,7 @@ mod tests { let mut inner = StreamsInner::new(); create_stream(&mut inner, "stream"); let create_topic = CreateTopicWithAssignmentsRequest { + created_view: 0, request: make_topic_request(0, 2, "topic"), derived_options: WireOptions::empty(), partitions: vec![ @@ -2857,6 +2956,7 @@ mod tests { let _ = StateHandler::apply(&create_topic, &mut inner, IggyTimestamp::now()); let create_partitions = CreatePartitionsWithAssignmentsRequest { + created_view: 0, request: WireCreatePartitionsRequest { stream_id: WireIdentifier::numeric(0), topic_id: WireIdentifier::numeric(0), @@ -2894,6 +2994,7 @@ mod tests { create_stream(&mut inner, "stream"); // Topic missing => validation failure path let create_partitions = CreatePartitionsWithAssignmentsRequest { + created_view: 0, request: WireCreatePartitionsRequest { stream_id: WireIdentifier::numeric(0), topic_id: WireIdentifier::numeric(99), @@ -2936,6 +3037,7 @@ mod tests { let mut inner = StreamsInner::new(); create_stream(&mut inner, "stream"); let create_topic = CreateTopicWithAssignmentsRequest { + created_view: 0, request: make_topic_request(0, partitions_count, "topic"), derived_options: WireOptions::empty(), partitions: (0..partitions_count) @@ -2991,6 +3093,7 @@ mod tests { let mut inner = StreamsInner::new(); create_stream(&mut inner, "stream"); let create_topic = CreateTopicWithAssignmentsRequest { + created_view: 0, request: make_topic_request(0, 1, "topic"), derived_options: WireOptions::empty(), partitions: vec![CreatedPartitionAssignment { @@ -3090,6 +3193,7 @@ mod tests { fn given_counted_partitions_when_apply_purge_stream_should_zero_every_topic() { let mut inner = inner_with_registered_partition(); let create_topic = CreateTopicWithAssignmentsRequest { + created_view: 0, request: make_topic_request(0, 1, "metrics"), derived_options: WireOptions::empty(), partitions: vec![CreatedPartitionAssignment { @@ -3188,6 +3292,7 @@ mod tests { let mut inner = StreamsInner::new(); create_stream(&mut inner, "alpha"); let create_topic = CreateTopicWithAssignmentsRequest { + created_view: 0, request: make_topic_request(0, 1, "logs"), derived_options: WireOptions::empty(), partitions: vec![CreatedPartitionAssignment { @@ -3302,6 +3407,7 @@ mod tests { let mut inner = StreamsInner::new(); create_stream(&mut inner, "alpha"); let create_topic = CreateTopicWithAssignmentsRequest { + created_view: 0, request: make_topic_request(0, 1, "logs"), derived_options: WireOptions::empty(), partitions: vec![CreatedPartitionAssignment { @@ -3396,6 +3502,7 @@ mod tests { for index in 0..MAX_TOPICS { let request = CreateTopicWithAssignmentsRequest { + created_view: 0, request: make_topic_request(0, 0, &format!("t{index}")), derived_options: WireOptions::empty(), partitions: Vec::new(), @@ -3405,6 +3512,7 @@ mod tests { } let overflow = CreateTopicWithAssignmentsRequest { + created_view: 0, request: make_topic_request(0, 0, "one-too-many"), derived_options: WireOptions::empty(), partitions: Vec::new(), @@ -3434,6 +3542,7 @@ mod tests { create_stream(&mut inner, "s"); let seed = CreateTopicWithAssignmentsRequest { + created_view: 0, request: make_topic_request(0, 1, "t"), derived_options: WireOptions::empty(), partitions: vec![CreatedPartitionAssignment { @@ -3448,6 +3557,7 @@ mod tests { // Resolves to MAX_PARTITIONS - 1: the last legal id. let last = CreatePartitionsWithAssignmentsRequest { + created_view: 0, request: WireCreatePartitionsRequest { stream_id: WireIdentifier::numeric(0), topic_id: WireIdentifier::numeric(0), @@ -3465,6 +3575,7 @@ mod tests { ); let overflow = CreatePartitionsWithAssignmentsRequest { + created_view: 0, request: WireCreatePartitionsRequest { stream_id: WireIdentifier::numeric(0), topic_id: WireIdentifier::numeric(0), @@ -3496,6 +3607,7 @@ mod tests { create_stream(&mut inner, "s"); let request = CreateTopicWithAssignmentsRequest { + created_view: 0, request: make_topic_request(0, 1, "t"), derived_options: WireOptions::empty(), partitions: vec![CreatedPartitionAssignment { @@ -3567,6 +3679,7 @@ mod tests { create_stream(&mut inner, "s"); for name in ["a", "b", "c"] { let request = CreateTopicWithAssignmentsRequest { + created_view: 0, request: make_topic_request(0, 0, name), derived_options: WireOptions::empty(), partitions: Vec::new(), @@ -3625,6 +3738,7 @@ mod tests { create_stream(&mut inner, "s"); let request = CreateTopicWithAssignmentsRequest { + created_view: 0, request: make_topic_request(0, 2, "t"), derived_options: WireOptions::empty(), partitions: vec![ @@ -3659,6 +3773,7 @@ mod tests { create_stream(&mut inner, "s"); let seed = CreateTopicWithAssignmentsRequest { + created_view: 0, request: make_topic_request(0, 1, "t"), derived_options: WireOptions::empty(), partitions: vec![CreatedPartitionAssignment { @@ -3672,6 +3787,7 @@ mod tests { ); let duplicate = CreatePartitionsWithAssignmentsRequest { + created_view: 0, request: WireCreatePartitionsRequest { stream_id: WireIdentifier::numeric(0), topic_id: WireIdentifier::numeric(0), @@ -3710,6 +3826,7 @@ mod tests { create_stream(&mut inner, "s"); let seed = CreateTopicWithAssignmentsRequest { + created_view: 0, request: make_topic_request(0, 1, "t"), derived_options: WireOptions::empty(), partitions: vec![CreatedPartitionAssignment { @@ -3723,6 +3840,7 @@ mod tests { ); let distinct = CreatePartitionsWithAssignmentsRequest { + created_view: 0, request: WireCreatePartitionsRequest { stream_id: WireIdentifier::numeric(0), topic_id: WireIdentifier::numeric(0), diff --git a/core/server/src/bootstrap.rs b/core/server/src/bootstrap.rs index 3c15d647b1..9a58b8b78a 100644 --- a/core/server/src/bootstrap.rs +++ b/core/server/src/bootstrap.rs @@ -1052,10 +1052,10 @@ async fn shard_main( |mux_stm| { ensure_default_root_user(mux_stm); }, - |mux_stm, client, timestamp| { + |mux_stm, client, stamp| { mux_stm .streams() - .remove_consumer_group_member(client, timestamp); + .remove_consumer_group_member(client, stamp); }, ) .await @@ -1297,7 +1297,6 @@ async fn shard_main( topology.cluster_id, topology.self_replica_id, topology.replica_count, - Arc::clone(&metadata_view), )); let reconcile_periodic = config .system @@ -2081,10 +2080,7 @@ async fn build_shard_for_thread( topology.cluster_id, topology.self_replica_id, topology.replica_count, - // Quarantine-and-rebuild always finds a partition - // directory already there, so this joins as a probing - // backup and learns the live view; nothing to seed. - None, + partition_metadata.created_view, Rc::clone(&bus), ) .await? diff --git a/core/server/src/dispatch.rs b/core/server/src/dispatch.rs index 44cfd93e2c..0a494f524b 100644 --- a/core/server/src/dispatch.rs +++ b/core/server/src/dispatch.rs @@ -4330,6 +4330,7 @@ mod tests { partition_id: 0, consensus_group_id: 1, }], + created_view: 0, }; md.mux_stm .update(prepare_message( diff --git a/core/server/src/partition_helpers.rs b/core/server/src/partition_helpers.rs index f548f5de71..a6f7098805 100644 --- a/core/server/src/partition_helpers.rs +++ b/core/server/src/partition_helpers.rs @@ -518,11 +518,11 @@ pub(crate) async fn open_partition_superblock( /// The namespace arrives packed, so its components are in range by /// construction. Metadata admission is what bounds them. /// -/// `view_seed` is the view a group with no durable record of its own starts -/// in, and it is the metadata plane's current view: see the `seed_view` -/// comment below for why a group left at view 0 is unreachable. `None` keeps -/// the historical view-0 start, and is also what a restart materialization -/// gets, since it probes for the live view instead. +/// `created_view` is the view a group with no durable record of its own starts +/// in: the metadata plane's view when it committed the create, recorded on +/// the committed partition so every replica seeds the same value. See the +/// `seed_view` comment below for why a group left at view 0 is unreachable. A +/// restart materialization ignores it and probes for the live view instead. /// /// The returned partition's `offset` / `dirty_offset` are `0` and /// `should_increment_offset` is `false`, mirroring a clean append starting @@ -542,7 +542,7 @@ pub async fn build_partition_fresh( cluster_id: u128, self_replica_id: u8, replica_count: u8, - view_seed: Option, + created_view: u32, bus: Rc, ) -> Result>, ServerError> { let stream_id = namespace.stream_id(); @@ -604,7 +604,8 @@ pub async fn build_partition_fresh( .map(|state| (state.view, state.log_view)); // Shared with the simulator's `init_partition`, which cannot call this // builder; see `fresh_group_start`. - let FreshGroupStart { join, seed_view } = fresh_group_start(restarted, durable_view, view_seed); + let FreshGroupStart { join, seed_view } = + fresh_group_start(restarted, durable_view, created_view); // Request queue holds 2x the prepare depth (buffered requests drain as // prepares commit); depth is the per-partition `[partition]` knob. let prepare_queue_depth = config.partition.prepare_queue_depth; @@ -626,16 +627,16 @@ pub async fn build_partition_fresh( // than the roster advertises as leader, and nothing routes a // partition write across that gap: the client is sent to the // metadata leader and refused there for the whole budget. Seeding - // from the metadata view keeps the two congruent for a group born - // after a metadata election. + // from the view the create was admitted in keeps the two congruent + // for a group born after a metadata election. // - // Replicas can still disagree on the seed: each publishes its own - // metadata view on a 100ms poll, so one may read V while another - // has yet to see the election. That resolves the way any view - // disagreement does, the higher view winning through `StartView` - - // but only because the seed sets `log_view` too. The caller is - // what keeps a replica from seeding a view it has no opinion on; - // see `ReconcilerCtx::partition_view_seed`. + // The seed comes off the committed partition, not this replica's + // live metadata view: the two differ once the metadata plane + // elects again, and a seed above the group's real view is an + // empty log that outranks committed history in the next DVC + // merge. The value rides the create's request body (a header view + // is restamped per delivery), so every replica commits the same + // one and a late materialiser lands at or below its peers. seed_view, incarnation: None, join, diff --git a/core/server/src/partition_reconciler.rs b/core/server/src/partition_reconciler.rs index e9114dabdb..dea36c9c02 100644 --- a/core/server/src/partition_reconciler.rs +++ b/core/server/src/partition_reconciler.rs @@ -170,7 +170,6 @@ //! -- `PrepareHeader.reserved` has room, but it is a `#[repr(C)]` wire change. use crate::bootstrap::ServerShard; -use crate::cluster_meta::METADATA_VIEW_UNKNOWN; use crate::partition_helpers::{build_partition_fresh, delete_partitions_from_disk}; use ahash::{AHashMap, AHashSet}; use configs::server::ServerConfig; @@ -188,7 +187,6 @@ use shard::{Receiver, Sender}; use std::cell::{Cell, RefCell}; use std::rc::Rc; use std::sync::Arc; -use std::sync::atomic::{AtomicU64, Ordering}; use std::time::{Duration, Instant}; use tracing::{debug, error, trace, warn}; @@ -227,11 +225,6 @@ pub struct ReconcilerCtx { pub cluster_id: u128, pub self_replica_id: u8, pub replica_count: u8, - /// The metadata plane's view, published by shard 0 and shared with every - /// shard's roster. Read when materialising a partition so its consensus - /// group starts in the same view the roster's advertised leader comes - /// from; the unknown-view sentinel until the first publish. - pub metadata_view: Arc, failure_state: RefCell>, /// `Streams::revision` observed at the end of the last pass that fully /// converged. Paired with `last_pass_noop` for the fast-skip in @@ -251,7 +244,6 @@ impl ReconcilerCtx { cluster_id: u128, self_replica_id: u8, replica_count: u8, - metadata_view: Arc, ) -> Self { Self { shard, @@ -260,48 +252,12 @@ impl ReconcilerCtx { cluster_id, self_replica_id, replica_count, - metadata_view, failure_state: RefCell::new(AHashMap::new()), last_revision: Cell::new(None), last_pass_noop: Cell::new(false), } } - /// The view a partition group materialised now should start in. - /// - /// `None` means this replica has no opinion yet, NOT "start at view 0": - /// shard 0 publishes the unknown-view sentinel before its first tick and - /// for as long as it has ceded a recovered view's primaryship. Treating - /// that as 0 is how a single replica of a group ends up view-0 while its - /// peers are at V, which is the split this seed exists to close. Callers - /// on a replicated group defer materialising instead; see - /// [`Self::partition_view_seed_ready`]. - /// - /// # Panics - /// If the published view does not fit a `u32`. The publisher writes - /// `u64::from(consensus.view())` or the sentinel handled above, so a value - /// past `u32::MAX` is memory corruption, not a state a retry can clear. - /// Silently falling back to `None` here would restore the view-0 start - /// this change exists to remove, on the one path nobody would look at. - fn partition_view_seed(&self) -> Option { - let view = self.metadata_view.load(Ordering::Relaxed); - if view == METADATA_VIEW_UNKNOWN { - return None; - } - Some(u32::try_from(view).expect("published metadata view must fit a u32")) - } - - /// Whether a group may be materialised now. - /// - /// A solo replica is always ready: with one replica every view names it - /// primary, so there is no peer to disagree with and no split to cause. - /// A replicated group waits until this replica knows the metadata view, - /// so it cannot seed a view-0 group underneath peers that already moved. - /// The wait is one reconcile pass; shard 0 republishes every 100ms. - fn partition_view_seed_ready(&self) -> bool { - self.replica_count <= 1 || self.partition_view_seed().is_some() - } - fn is_backed_off(&self, ns: IggyNamespace, cause: FailureCause, now: Instant) -> bool { let state = self.failure_state.borrow(); if state.is_empty() { @@ -461,12 +417,6 @@ struct PassCounters { /// has not applied yet. Counted so the pass does not arm the fast-skip /// while work is in flight; applying it bumps no revision. already_staged: usize, - /// Materialisations held back because this replica has no metadata view to - /// seed from yet. Counted so the pass does not arm the fast-skip: shard 0 - /// publishing a view bumps no `Streams::revision` and does not wake the - /// reconciler, so an armed skip would strand every deferred group until - /// some unrelated commit happened to bump the revision. - view_unpublished: usize, } impl PassCounters { @@ -483,7 +433,6 @@ impl PassCounters { + self.deferred + self.parked_reclaimed + self.already_staged - + self.view_unpublished } } @@ -529,7 +478,7 @@ async fn reconcile_once(ctx: &ReconcilerCtx) -> bool { } let target = snapshot_target_namespaces(ctx); - let target_set: AHashSet = target.iter().map(|(ns, _)| *ns).collect(); + let target_set: AHashSet = target.iter().map(|partition| partition.ns).collect(); let mut counters = PassCounters::default(); reconcile_additions(ctx, target, &mut counters).await; @@ -581,14 +530,19 @@ async fn reconcile_once(ctx: &ReconcilerCtx) -> bool { #[allow(clippy::too_many_lines)] async fn reconcile_additions( ctx: &ReconcilerCtx, - target: Vec<(IggyNamespace, u64)>, + target: Vec, counters: &mut PassCounters, ) { let shard_id = ctx.shard.id; let partitions = ctx.shard.plane.partitions(); let total_shards = u32::from(ctx.total_shards); - for (ns, epoch) in target { + for TargetPartition { + ns, + epoch, + created_view, + } in target + { if partitions.contains(&ns) { // Tombstoned but still in the map. Two cases, told apart by // whether teardown's disk delete succeeded: @@ -710,22 +664,6 @@ async fn reconcile_additions( continue; } - // Materialising before this replica knows the metadata view would seed - // the group at view 0 under peers that already elected past it. Defer - // the whole pass rather than the namespace: every group on this shard - // reads the same view, so none of them can be seeded correctly yet. - if !ctx.partition_view_seed_ready() { - debug!( - shard = shard_id, - "metadata view not published yet; deferring partition materialisation" - ); - // Every remaining namespace reads the same view, so none of them - // can be seeded either. Counted before breaking so the pass does - // not read as converged, which would fast-skip the retry. - counters.view_unpublished += 1; - break; - } - // Resolve the shared stats `Arc` only for namespaces actually // built, not once per committed partition every pass. A topic that // vanished between the target snapshot and this read defers to the @@ -743,7 +681,7 @@ async fn reconcile_additions( ctx.cluster_id, ctx.self_replica_id, ctx.replica_count, - ctx.partition_view_seed(), + created_view, Rc::clone(&ctx.shard.bus), ) .await @@ -1124,11 +1062,21 @@ fn snapshot_topic_live_groups(ctx: &ReconcilerCtx) -> AHashMap<(usize, usize), A }) } -/// Committed `(namespace, created_revision)` pairs. The epoch lets the -/// additions pass detect a stale local incarnation after slab-key reuse -/// without an `Arc` clone per partition; stats are fetched -/// lazily in [`fetch_partition_stats`] only for namespaces actually built. -fn snapshot_target_namespaces(ctx: &ReconcilerCtx) -> Vec<(IggyNamespace, u64)> { +/// One committed partition the additions pass has to account for. +struct TargetPartition { + ns: IggyNamespace, + /// The committed `created_revision`. Lets the pass detect a stale local + /// incarnation after slab-key reuse without an `Arc` clone per + /// partition; stats are fetched lazily in [`fetch_partition_stats`] only + /// for namespaces actually built. + epoch: u64, + /// The view a fresh materialisation seeds its consensus group with; see + /// `build_partition_fresh`. + created_view: u32, +} + +/// Every committed partition, as the additions pass needs it. +fn snapshot_target_namespaces(ctx: &ReconcilerCtx) -> Vec { ctx.shard.plane.metadata().mux_stm.streams().read(|inner| { // TODO(krishna): O(committed partitions) per non-skipped pass (here + // reconcile_removals). The revision fast-skip hides this in steady @@ -1138,8 +1086,11 @@ fn snapshot_target_namespaces(ctx: &ReconcilerCtx) -> Vec<(IggyNamespace, u64)> for (_, stream) in &inner.items { for (topic_id, topic) in &stream.topics { for partition in &topic.partitions { - let ns = IggyNamespace::new(stream.id, topic_id, partition.id); - entries.push((ns, partition.created_revision)); + entries.push(TargetPartition { + ns: IggyNamespace::new(stream.id, topic_id, partition.id), + epoch: partition.created_revision, + created_view: partition.created_view, + }); } } } @@ -1279,8 +1230,8 @@ pub fn install_tick_handler(shard: &Rc, wake_tx: WakeTx) { #[cfg(test)] mod tests { use super::{ - AtomicU64, FailureCause, FailureRecord, METADATA_VIEW_UNKNOWN, ReconcilerCtx, - build_partition_fresh, delete_partitions_from_disk, fetch_partition_stats, reconcile_once, + FailureCause, FailureRecord, ReconcilerCtx, build_partition_fresh, + delete_partitions_from_disk, fetch_partition_stats, reconcile_once, }; use configs::server::{ServerConfig, ServerSystemConfig}; use consensus::{MetadataHandle, PartitionsHandle}; @@ -1303,6 +1254,7 @@ mod tests { use metadata::IggyMetadata; use metadata::MuxStateMachine; use metadata::impls::metadata::IggySnapshot; + use metadata::impls::metadata::StreamsFrontend; use metadata::stm::StateMachine; use metadata::stm::stream::Streams; use metadata::stm::user::Users; @@ -1514,6 +1466,7 @@ mod tests { }, derived_options: WireOptions::empty(), partitions: assignments, + created_view: 0, }; mux.update(build_prepare( op, @@ -1732,32 +1685,19 @@ mod tests { CLUSTER_ID, 0, 1, - Arc::new(AtomicU64::new(METADATA_VIEW_UNKNOWN)), )) } - /// [`make_ctx`] for a REPLICATED group, sharing `metadata_view` with the - /// caller so a test can publish a view mid-run. - /// - /// Replica count is what decides whether materialisation waits on a - /// published view: a solo replica is primary in every view, so there is no - /// peer to disagree with and nothing to wait for. Every other test here - /// runs solo and takes that short circuit. - fn make_cluster_ctx( - shard: Rc, - total_shards: u16, - config: Rc, - metadata_view: Arc, - ) -> Rc { - Rc::new(ReconcilerCtx::new( - shard, - total_shards, - config, - CLUSTER_ID, - 0, - 3, - metadata_view, - )) + /// The committed `created_view` of `ns`, what a reconcile pass would seed + /// a fresh build with. + fn created_view(ctx: &ReconcilerCtx, ns: IggyNamespace) -> u32 { + ctx.shard + .plane + .metadata() + .mux_stm + .streams() + .created_view_for_namespace(ns) + .expect("committed namespace records its creation view") } /// Tests run reconcile + pump-side apply inline since no real pump exists. @@ -1902,7 +1842,7 @@ mod tests { CLUSTER_ID, 0, 1, - None, + created_view(&ctx, ns), Rc::clone(&ctx.shard.bus), ) .await @@ -1932,7 +1872,7 @@ mod tests { CLUSTER_ID, 0, 1, - None, + created_view(&ctx, ns), Rc::clone(&ctx.shard.bus), ) .await @@ -2056,73 +1996,6 @@ mod tests { /// `CreatePartitions` on an existing topic adds new namespaces; the /// reconciler picks them up on the next pass without touching the /// partitions it already materialised. - /// A replicated group is NOT materialised while this replica has no - /// metadata view to seed from, and IS once one is published. - /// - /// Seeding is a local read: shard 0 publishes the unknown sentinel before - /// its first tick and for as long as it has ceded a recovered view. Taking - /// that for view 0 would start the group naming replica 0 underneath peers - /// that already elected past it, which is the split the seed exists to - /// close, reintroduced one replica at a time. Waiting costs one reconcile - /// pass; the publisher reposts every 100ms. - #[compio::test] - async fn given_no_published_metadata_view_when_reconciling_should_defer_materialisation() { - let tmp = TempDir::new().expect("tempdir for system path"); - let config = test_config(&tmp); - let mux = TestMux::default(); - seed_stream(&mux, 1, "stream-deferred"); - seed_topic(&mux, 2, 0, "topic-deferred", vec![assignment(0, 1)]); - - let shard = build_test_shard(0, &config, mux); - let metadata_view = Arc::new(AtomicU64::new(METADATA_VIEW_UNKNOWN)); - let ctx = make_cluster_ctx( - Rc::clone(&shard), - 1, - Rc::new(config), - Arc::clone(&metadata_view), - ); - - reconcile_pass(&ctx).await; - assert_eq!( - shard.plane.partitions().len(), - 0, - "a replicated group must not materialise before this replica knows the metadata \ - view: seeded at 0 it names replica 0 whatever the metadata plane elected" - ); - - // The publisher posts a real view; the deferred pass now converges. - metadata_view.store(1, Ordering::Relaxed); - reconcile_pass(&ctx).await; - assert_eq!( - shard.plane.partitions().len(), - 1, - "once a view is published the deferred group must materialise on the next pass" - ); - } - - /// A SOLO replica never waits. It is primary in every view, so there is no - /// peer for it to disagree with, and blocking on a publisher that a - /// single-node deployment may never run would wedge materialisation - /// outright. - #[compio::test] - async fn given_a_solo_replica_when_no_view_is_published_should_still_materialise() { - let tmp = TempDir::new().expect("tempdir for system path"); - let config = test_config(&tmp); - let mux = TestMux::default(); - seed_stream(&mux, 1, "stream-solo"); - seed_topic(&mux, 2, 0, "topic-solo", vec![assignment(0, 1)]); - - let shard = build_test_shard(0, &config, mux); - let ctx = make_ctx(Rc::clone(&shard), 1, Rc::new(config)); - - reconcile_pass(&ctx).await; - assert_eq!( - shard.plane.partitions().len(), - 1, - "a solo replica must materialise without waiting on a published metadata view" - ); - } - #[compio::test] async fn reconcile_picks_up_create_partitions_increments() { let tmp = TempDir::new().expect("tempdir for system path"); @@ -2161,6 +2034,7 @@ mod tests { partitions_count: 2, }, partitions: vec![assignment(0, 3), assignment(1, 4)], + created_view: 0, }, )) .expect("CreatePartitions apply succeeds"); @@ -3353,6 +3227,7 @@ mod tests { partitions_count: 1, }, partitions: vec![assignment(0, 3)], + created_view: 0, }, )) .expect("CreatePartitions apply succeeds"); diff --git a/core/server/src/responses.rs b/core/server/src/responses.rs index 21b8e6497c..8165bdf63b 100644 --- a/core/server/src/responses.rs +++ b/core/server/src/responses.rs @@ -1869,7 +1869,7 @@ mod tests { use metadata::stm::stream::{Partition, StreamsInner}; let streams = StreamsInner::new(); - let partition = Partition::new(0, 1, IggyTimestamp::from(1u64), 0); + let partition = Partition::new(0, 1, IggyTimestamp::from(1u64), 0, 0); // Registry miss: the owning shard has not started building. let predicted = partition_response(&streams, 0, 0, &partition).expect("response builds"); @@ -1914,8 +1914,8 @@ mod tests { options: iggy_common::ResourceOptions::default(), stats: topic_stats.clone(), partitions: vec![ - Partition::new(0, 1, created_at, 0), - Partition::new(1, 1, created_at, 0), + Partition::new(0, 1, created_at, 0, 0), + Partition::new(1, 1, created_at, 0, 0), ], round_robin_counter: Arc::new(AtomicUsize::new(0)), consumer_groups: ahash::AHashMap::default(), diff --git a/core/shard/src/lib.rs b/core/shard/src/lib.rs index 2decb39341..98674232c8 100644 --- a/core/shard/src/lib.rs +++ b/core/shard/src/lib.rs @@ -3555,6 +3555,9 @@ where /// materialisation and wrong for a restart: a rebuilt partition with no data /// reports `commit_offset` 0, which reads as a regression rather than a /// harness that discarded the log. + /// + /// `created_view` is the view the metadata plane created the namespace in; + /// see `fresh_group_start`. // `feature = "simulator"` alone, unlike its neighbours: the body names items // `partitions` gates the same way, and a `test` arm cannot turn those on. // Under `cargo test -p shard` that arm fires from shard's own `cfg(test)` @@ -3569,7 +3572,7 @@ where recovered_state: Option, retained: Option, restore_frontier: bool, - metadata_view: Option, + created_view: u32, ) where B: MessageBus + Clone, { @@ -3590,15 +3593,15 @@ where // The SAME decision `build_partition_fresh` makes, not a copy of it. // This path cannot call that builder (it does real filesystem work and // this runs on in-memory storage), and while the two decided - // separately the simulator exercised neither the metadata-view seed nor - // the plane split it closes. `retained` is populated only by the + // separately the simulator exercised neither the creation-view seed + // nor the plane split it closes. `retained` is populated only by the // restart path, which is this path's evidence of a prior life. let durable_view = recovered_state .as_ref() .map(|state| (state.view, state.log_view)); let restarted = retained.is_some() && self.partition_consensus.replica_count > 1; let consensus::FreshGroupStart { join, seed_view } = - consensus::fresh_group_start(restarted, durable_view, metadata_view); + consensus::fresh_group_start(restarted, durable_view, created_view); // Recorded view first, exactly as the two boot paths order it: restoring // after `init` would advertise a view older than the recorded one. diff --git a/core/simulator/src/lib.rs b/core/simulator/src/lib.rs index 4e006c573d..7b722eb9bf 100644 --- a/core/simulator/src/lib.rs +++ b/core/simulator/src/lib.rs @@ -170,6 +170,12 @@ pub struct Simulator { /// a run with it on studies a system more durable than Iggy is. See /// `IggyShard::init_partition`. restore_partition_frontier: bool, + /// The view each namespace was created in, what production records on the + /// committed partition and every replica seeds its group from. Read from the + /// metadata plane once, at the first seed, and reused for every later seed + /// of that namespace, a restart's included, so no replica seeds a view its + /// peers did not. + partition_created_views: HashMap, /// Replies a setup handshake pulled off the wire that were not its own. /// /// `await_setup_reply` steps the whole simulator, so it sees every client's @@ -472,6 +478,7 @@ impl Simulator { entry_rng: Xoshiro256PlusPlus::seed_from_u64(SimSeeds::derive(seed).entry_shard), evicted: Vec::new(), restore_partition_frontier: false, + partition_created_views: HashMap::new(), deferred_client_replies: Vec::new(), seed, shell, @@ -494,11 +501,19 @@ impl Simulator { // segment storage so that branch can be deleted. #[allow(clippy::cast_possible_truncation)] pub fn init_partition(&mut self, namespace: IggyNamespace) { + let Some(created_view) = self.partition_created_view(namespace) else { + return; + }; for (i, replica) in self.replicas.iter().enumerate() { if self.crashed.contains(&(i as u8)) { continue; } - materialise_partition(replica, namespace, self.restore_partition_frontier); + materialise_partition( + replica, + namespace, + self.restore_partition_frontier, + created_view, + ); } } @@ -513,7 +528,10 @@ impl Simulator { /// dispatch-shell poll path. /// #[allow(clippy::cast_possible_truncation)] - pub fn seed_stream_topic_partition(&self, namespace: IggyNamespace) { + pub fn seed_stream_topic_partition(&mut self, namespace: IggyNamespace) { + let Some(created_view) = self.partition_created_view(namespace) else { + return; + }; for (i, replica) in self.replicas.iter().enumerate() { if self.crashed.contains(&(i as u8)) { continue; @@ -525,8 +543,20 @@ impl Simulator { .metadata() .mux_stm .streams() - .seed_namespace(namespace, namespace.inner()); + .seed_namespace(namespace, namespace.inner(), created_view); + } + } + + /// The view `namespace` was created in: the metadata plane's view at its + /// first seed, as the prepare of a real create carries it, and the same value + /// for every seed after. `None` with no live metadata consensus to read. + fn partition_created_view(&mut self, namespace: IggyNamespace) -> Option { + if let Some(&created_view) = self.partition_created_views.get(&namespace) { + return Some(created_view); } + let created_view = self.metadata_view()?; + self.partition_created_views.insert(namespace, created_view); + Some(created_view) } /// Log `client` in against the deterministic root user through the dispatch @@ -1063,9 +1093,13 @@ impl Simulator { let partition_logs = self.retain_partition_logs(idx, &partition_superblocks); // SORTED: seed order decides slab ids and `HashMap` order is per-process, so // an unsorted walk would stop replay being byte-identical. Also drives the - // re-materialisation loop below, which must agree with it. - let mut materialised: Vec = partition_superblocks.keys().copied().collect(); - materialised.sort_unstable_by_key(IggyNamespace::inner); + // re-materialisation loop below, which must agree with it. Each carries the + // view it was created in, as the metadata a real boot replays would. + let mut seed_namespaces: Vec<(IggyNamespace, u32)> = partition_superblocks + .keys() + .map(|&namespace| (namespace, self.partition_created_views[&namespace])) + .collect(); + seed_namespaces.sort_unstable_by_key(|(namespace, _)| namespace.inner()); // Durable VSR state from the retained superblock, before the rebuild, as // production reads it in `restore_metadata_consensus`. @@ -1116,7 +1150,7 @@ impl Simulator { recovered_state, metadata_incarnation, (shard_idx == 0).then(|| replica_data_dir.clone()).flatten(), - &materialised, + &seed_namespaces, ); if shard_idx == 0 { metadata_bundle = @@ -1153,11 +1187,12 @@ impl Simulator { // makes the carried-forward superblock load-bearing: the group recovers its // recorded `(view, log_view)` instead of re-entering view 0. The metadata half // of the seed already ran inside `new_shard`, ahead of the replay. - for namespace in materialised { + for (namespace, created_view) in seed_namespaces { materialise_partition( &self.replicas[idx], namespace, self.restore_partition_frontier, + created_view, ); } @@ -1334,6 +1369,21 @@ impl Simulator { }) } + /// The metadata plane's view, as the first live replica owning a metadata + /// consensus sees it. + fn metadata_view(&self) -> Option { + (0..self.replica_count) + .filter(|replica_idx| !self.crashed.contains(replica_idx)) + .find_map(|replica_idx| { + self.replicas[usize::from(replica_idx)].shards[0] + .plane + .metadata() + .consensus + .as_ref() + .map(consensus::VsrConsensus::view) + }) + } + #[must_use] pub(crate) fn primary_index(&self, namespace: IggyNamespace) -> Option { (0..self.replica_count) @@ -1357,14 +1407,19 @@ impl Simulator { /// re-opens every partition directory it owns, so the sim must re-materialise too, /// else the superblock a restart carries forward is never read back and the /// recovered-view branch is dead code. -fn materialise_partition(replica: &SimReplica, namespace: IggyNamespace, restore_frontier: bool) { +fn materialise_partition( + replica: &SimReplica, + namespace: IggyNamespace, + restore_frontier: bool, + created_view: u32, +) { let shard_count = u32::try_from(replica.shards.len()).expect("shard count fits u32"); let owner = calculate_shard_assignment(&namespace, shard_count); // Commit the namespace first: a partition the metadata plane never heard of is // a shape production cannot produce, and the shard refuses client traffic whose // routing-row epoch it cannot match against a committed `created_revision`. let streams = replica.shards[0].plane.metadata().mux_stm.streams(); - streams.seed_namespace(namespace, namespace.inner()); + streams.seed_namespace(namespace, namespace.inner(), created_view); // No committed revision means the seed could not re-add the namespace, which // happens once a metadata workload has deleted its stream or topic: the seed's // `CreatePartitions` is then a committed REJECTION rather than an error, so it @@ -1375,6 +1430,12 @@ fn materialise_partition(replica: &SimReplica, namespace: IggyNamespace, restore let Some(epoch) = streams.created_revision_for_namespace(namespace) else { return; }; + // Read back rather than trusted from the argument: the seed above is a no-op + // for a namespace a metadata op already committed, and production seeds from + // the committed partition, never from a live view. + let created_view = streams + .created_view_for_namespace(namespace) + .expect("a committed partition records its creation view"); // One store per group, minted on first materialisation and reused after, so the // recorded view survives a replica restart. let superblock = Rc::clone( @@ -1392,24 +1453,13 @@ fn materialise_partition(replica: &SimReplica, namespace: IggyNamespace, restore // a second materialisation with no restart between would otherwise resurrect a // log the live partition has moved past. let retained = replica.partition_logs.borrow_mut().remove(&namespace); - // The view this replica's metadata plane is in, which is what a fresh - // group seeds from. Production reads it off the roster value shard 0 - // publishes; here shard 0's consensus is right there. Without it the - // simulator materialises every group at view 0 and can never produce the - // plane split that costs production its writes. - let metadata_view = replica.shards[0] - .plane - .metadata() - .consensus - .as_ref() - .map(consensus::VsrConsensus::view); replica.shards[usize::from(owner)].init_partition( namespace, Some(superblock), recovered_state, retained, restore_frontier, - metadata_view, + created_view, ); for shard in &replica.shards { shard.shards_table().insert( @@ -1576,8 +1626,9 @@ mod tests { primary back at replica 0 both planes agree and the split cannot show" ); - // Materialise a brand-new group. Every live replica seeds from its own - // metadata view, which is the value production reads off the roster. + // Materialise a brand-new group. Every live replica seeds from the view + // the create was committed in, which production reads off the committed + // partition. let namespace = IggyNamespace::new(1, 1, 0); sim.init_partition(namespace); @@ -1617,6 +1668,120 @@ mod tests { } } + /// A replica that missed a group's creation materialises it after the + /// metadata plane elected again. It must seed the view the group was CREATED + /// in, not the live metadata view: seeded above the group's real view, its + /// empty log outranks every peer in the next DVC merge and the committed ops + /// collect a nack quorum, wedging the group for good. + #[test] + fn given_a_late_materialiser_when_the_metadata_view_moved_on_should_seed_the_creation_view() { + server_common::MemoryPool::init_pool(&server_common::MemoryPoolSettings { + enabled: false, + size: iggy_common::IggyByteSize::from(0u64), + bucket_capacity: 1, + }); + + let replica_count: u8 = 3; + let client_id: u128 = 1; + let network_opts = packet::PacketSimulatorOptions { + node_count: replica_count, + client_count: 1, + ..packet::PacketSimulatorOptions::default() + }; + let mut sim = Simulator::new( + replica_count as usize, + std::iter::once(client_id), + network_opts, + ); + let settle = |sim: &mut Simulator| { + for _ in 0..800 { + sim.step(); + } + }; + let reply_within = |sim: &mut Simulator, budget: usize| { + (0..budget).find_map(|_| { + let replies = sim.step(); + (!replies.is_empty()).then(|| replies[0].deep_copy()) + }) + }; + + // Replica 0 is down while the group is created, at a metadata view its + // crash moved off 0. + sim.replica_crash(0); + settle(&mut sim); + let namespace = IggyNamespace::new(1, 1, 0); + sim.init_partition(namespace); + let created_view = sim + .partition_consensus_state(1, namespace) + .expect("replica 1 materialised the group") + .view; + assert!( + created_view > 0, + "crashing replica 0 must have moved the metadata plane off view 0, or a \ + creation-view seed is indistinguishable from a view-0 start" + ); + + // Commit a write, so the group holds history a wrong seed could outrank. + let client = SimClient::new(client_id); + sim.register_client_with_primary(&client); + let primary = sim + .primary_index(namespace) + .expect("a live replica hosts the group"); + let request = client.send_messages(namespace, &[Bytes::from_static(b"before")]); + sim.submit_request(client_id, primary, request.into_generic()); + reply_within(&mut sim, 200).expect("the write commits on the fresh group"); + + // Replica 0 returns without the group, then the metadata plane elects + // again, so its live view now exceeds the view the group was created in. + sim.replica_restart(0); + settle(&mut sim); + let metadata_primary = sim + .metadata_primary_index() + .expect("a live replica owns metadata consensus"); + sim.replica_crash(metadata_primary); + settle(&mut sim); + let moved_view = sim + .metadata_view() + .expect("a live replica owns metadata consensus"); + assert!( + moved_view > created_view, + "the second election must move the metadata view ({moved_view}) past the \ + group's creation view ({created_view})" + ); + + // Replica 0 materialises the group now. + sim.init_partition(namespace); + let seeded = sim + .partition_consensus_state(0, namespace) + .expect("replica 0 materialised the group") + .view; + assert_eq!( + seeded, created_view, + "a late materialiser must seed the group's creation view, not the live metadata \ + view {moved_view}: above the group's real view its empty log wins the next merge" + ); + + // With replica 0 in, the group has a quorum again: it elects past its + // dead primary and commits a new write. + let settled_primary = (0..4) + .find_map(|_| { + settle(&mut sim); + (0..replica_count) + .filter(|replica_idx| !sim.is_crashed(*replica_idx)) + .find(|replica_idx| { + sim.partition_consensus_state(usize::from(*replica_idx), namespace) + .is_some_and(|state| state.status == Status::Normal && state.is_primary) + }) + }) + .expect("the group must elect a live primary once replica 0 joins it"); + let request = client.send_messages(namespace, &[Bytes::from_static(b"after")]); + sim.submit_request(client_id, settled_primary, request.into_generic()); + reply_within(&mut sim, 400).expect( + "a write must commit after the late materialiser joined; a wedged merge \ + answers nothing", + ); + } + /// A replica that advanced its view, persisted it through the superblock gate, /// then crashed recovers that view from its own disk, not a fresh 0. The /// split-brain guarantee: a replica never forgets a view it acted in. Impossible @@ -2533,7 +2698,7 @@ mod tests { executor.run_until_stalled(POLL_BUDGET); // borrow acquired; task parks let grow = Rc::clone(&sim.replicas[0].shards[0]); executor.spawn(async move { - grow.init_partition(ns_grow, None, None, None, false, None); + grow.init_partition(ns_grow, None, None, None, false, 0); }); executor.run_until_stalled(POLL_BUDGET); // grow while the borrow is live })) @@ -2568,7 +2733,7 @@ mod tests { executor.run_until_stalled(POLL_BUDGET); let grow = Rc::clone(&sim.replicas[0].shards[0]); executor.spawn(async move { - grow.init_partition(ns_grow, None, None, None, false, None); + grow.init_partition(ns_grow, None, None, None, false, 0); }); executor.run_until_stalled(POLL_BUDGET); diff --git a/core/simulator/src/replica.rs b/core/simulator/src/replica.rs index bdbeed1b57..638a61dce6 100644 --- a/core/simulator/src/replica.rs +++ b/core/simulator/src/replica.rs @@ -155,7 +155,7 @@ pub fn new_shard( recovered_state: Option, incarnation: u128, data_dir: Option, - seed_namespaces: &[server_common::sharding::IggyNamespace], + seed_namespaces: &[(server_common::sharding::IggyNamespace, u32)], ) -> (Rc, Option) { // Metadata is single-writer, mirroring the server bootstrap. Shard 0 owns // the only writable STM; every peer shard rebuilds a reader-mode mirror from @@ -327,8 +327,8 @@ pub fn new_shard( // `materialise_partition`'s own seed is then a no-op. if shard_idx == 0 { let streams = metadata.mux_stm.streams(); - for &namespace in seed_namespaces { - streams.seed_namespace(namespace, namespace.inner()); + for &(namespace, created_view) in seed_namespaces { + streams.seed_namespace(namespace, namespace.inner(), created_view); } } From bb973973d2158930a220dfb813cac613cf7bf599 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Tue, 1 Sep 2026 07:41:43 +0200 Subject: [PATCH 023/182] chore(deps): Bump the csharp group with 1 update (#4011) --- foreign/csharp/Directory.Packages.props | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/foreign/csharp/Directory.Packages.props b/foreign/csharp/Directory.Packages.props index d65cb8108d..fdc9ef880e 100644 --- a/foreign/csharp/Directory.Packages.props +++ b/foreign/csharp/Directory.Packages.props @@ -36,7 +36,7 @@ under the License. - + all runtime; build; native; contentfiles; analyzers; buildtransitive From b1b911ffb894fb9d29381fd7fb8b2ef5933302bb Mon Sep 17 00:00:00 2001 From: haubur Date: Tue, 1 Sep 2026 08:25:19 +0200 Subject: [PATCH 024/182] fix(sdk): stream should terminate after consumer shutdown (#3964) --- core/sdk/src/clients/consumer.rs | 68 +++++++++++++++++++++++++++++++- 1 file changed, 66 insertions(+), 2 deletions(-) diff --git a/core/sdk/src/clients/consumer.rs b/core/sdk/src/clients/consumer.rs index b9d23a684f..a4477e6c8f 100644 --- a/core/sdk/src/clients/consumer.rs +++ b/core/sdk/src/clients/consumer.rs @@ -1055,6 +1055,10 @@ impl Stream for IggyConsumer { type Item = Result; fn poll_next(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll> { + if self.shutdown.load(ORDERING) { + return Poll::Ready(None); + } + let partition_id = self.state.partition_id(); if let Some(message) = self.buffered_messages.pop_front() { { @@ -1314,12 +1318,14 @@ mod tests { use crate::clients::consumer_builder::IggyConsumerBuilder; use crate::tcp::tcp_client::TcpClient; use iggy_common::locking::IggyRwLockFn; + use std::str::FromStr; + use std::task::Waker; - fn builder() -> IggyConsumerBuilder { + fn builder_for(consumer: Consumer) -> IggyConsumerBuilder { IggyConsumerBuilder::new( IggyRwLock::new(ClientWrapper::Tcp(TcpClient::default())), "consumer".to_owned(), - Consumer::new(Identifier::numeric(1).unwrap()), + consumer, Identifier::numeric(1).unwrap(), Identifier::numeric(1).unwrap(), None, @@ -1328,6 +1334,64 @@ mod tests { ) } + fn builder() -> IggyConsumerBuilder { + builder_for(Consumer::new(Identifier::numeric(1).unwrap())) + } + + async fn assert_stream_terminates_after_shutdown(consumer: Consumer) { + let mut consumer = builder_for(consumer) + .partition(Some(1)) + .batch_length(1) + .auto_commit(AutoCommit::Disabled) + .build(); + consumer.buffered_messages.extend([ + IggyMessage::from_str("a").unwrap(), + IggyMessage::from_str("b").unwrap(), + ]); + let mut context = Context::from_waker(Waker::noop()); + + assert!(matches!( + Pin::new(&mut consumer).poll_next(&mut context), + Poll::Ready(Some(Ok(_))) + )); + + consumer.shutdown().await.unwrap(); + + assert_eq!(consumer.buffered_messages.len(), 1); + assert!(matches!( + Pin::new(&mut consumer).poll_next(&mut context), + Poll::Ready(None) + )); + } + + #[tokio::test] + async fn standalone_consumer_should_stop_yielding_messages_after_shutdown() { + assert_stream_terminates_after_shutdown(Consumer::new(Identifier::numeric(1).unwrap())) + .await; + } + + #[tokio::test] + async fn consumer_group_should_stop_yielding_messages_after_shutdown() { + assert_stream_terminates_after_shutdown(Consumer::group(Identifier::numeric(1).unwrap())) + .await; + } + + #[tokio::test] + async fn consumer_group_should_not_create_poll_future_after_shutdown() { + let mut consumer = builder_for(Consumer::group(Identifier::numeric(1).unwrap())) + .auto_commit(AutoCommit::Disabled) + .build(); + let mut context = Context::from_waker(Waker::noop()); + + consumer.shutdown().await.unwrap(); + + assert!(matches!( + Pin::new(&mut consumer).poll_next(&mut context), + Poll::Ready(None) + )); + assert!(consumer.poll_future.is_none()); + } + #[tokio::test] async fn should_accept_every_auto_commit_mode() { for auto_commit in [ From 72b5995c5d66aaf8ac62e0221361d6eb460c0a72 Mon Sep 17 00:00:00 2001 From: Hubert Gruszecki Date: Tue, 1 Sep 2026 08:32:30 +0200 Subject: [PATCH 025/182] docs(repo): add repo-wide team-review agent skill (#4004) This PR adds a new skill used to review changes. Contributors are encouraged to use it. It is placed under .claude/skills/ next to the connector skills. The report lands in the agent session scratchpad. Skills carry no disable-model-invocation flag, so the description states the skill is user-invoked only: one run spawns roughly ten subagents. --- .claude/skills/team-review/SKILL.md | 117 ++++++++++++++++++++++++++++ AGENTS.md | 17 ++-- 2 files changed, 128 insertions(+), 6 deletions(-) create mode 100644 .claude/skills/team-review/SKILL.md diff --git a/.claude/skills/team-review/SKILL.md b/.claude/skills/team-review/SKILL.md new file mode 100644 index 0000000000..5d43f6117a --- /dev/null +++ b/.claude/skills/team-review/SKILL.md @@ -0,0 +1,117 @@ +--- +name: team-review +description: Adversarial 4-expert review (storage, perf, distsys, ecosystem) of a PR, branch, or ref range, with clean-room validation of every finding. Experts work alone, no peer debate. Expensive, one run spawns ~10 subagents. +argument-hint: "[PR number | branch | ref range]" +disable-model-invocation: true +--- + +# Apache Iggy Team Review + +`` = `$ARGUMENTS`: a PR number, a branch, or a ref range. Empty means `origin/master..HEAD`. Mission critical code. + +You = **moderator**. You never open the diff or a source file: you route paths, merge claims, synthesize. Every token you load rides along every later turn. Reviewers and validators are one-shot agents that deliver by writing a file; nobody chats. + +## Charter (paste VERBATIM into every expert, validator, and tiebreak prompt) + +> You think big brain. You speak caveman. Separate things. +> +> **Thinking, unchanged.** Read the diff, then every changed file in full from the local checkout, then whatever call sites you need. Trace call chains. Verify invariants. Prove findings, don't guess. Cite exact `file:line`. Running tests or builds needs a stated justification: reading and tracing settles most claims, and parallel cargo runs block on one target-dir lock. +> +> **Output style.** Drop articles, filler, pleasantries, hedging. Fragments OK. Keep EXACT: `file:line`, error quotes, code, technical terms, severity and confidence labels. +> +> - Finding, one line each: `[sev] file:line - problem. Fix: action. (origin, conf:H|M|L)` +> - `sev`: `critical` = correctness/safety/data-loss/security, blocks merge; `warning` = real defect, perf hit, API issue; `nit` = style/naming; `simplify` = complexity/dead-code reduction, format `[simplify] file:line - what's complex. Simpler: alternative. Saves: ~N lines / removes indirection. (origin, conf)`. +> - `origin`: `intro` (PR introduced), `pre-surfaced` (existed, exposed by PR), `pre-untouched` (existed, not touched). +> - Never flag em dashes or other punctuation style as a finding. +> - Simplification mandate: less code > more code. Per changed file ask whether ~30% smaller keeps correctness: dead fields/params/branches/imports, duplication of an existing helper (cite it), single-impl traits, premature generics, checks for impossible states. Do not propose simplifications that change semantics or break public API. If nothing qualifies, write `Simplifications: none`. +> +> Caveman = output compression, not analysis compression. Dig deep. Write short. + +## Step 1: Identify the target (no reading) + +- Classify ``: matches `^#?(pr)?[0-9]+$` case-insensitively -> PR, the digits are ``. Anything else -> ref range or bare branch. Empty -> review `origin/master..HEAD`. +- ``: `` lowercased, chars outside `[a-z0-9-]` replaced by `-`, repeats collapsed, trimmed, max 40 chars (`PR3123` -> `pr3123`, `origin/master..HEAD` -> `origin-master-head`). Empty -> `date +%s`. +- `

` = `/review-`. `mkdir -p` it. +- PR: `gh pr view --json title,body,headRefOid > /pr.json`, `gh pr diff > /diff.patch`, `gh pr diff --name-only > /files.txt`. `` = first 8 of `headRefOid`. +- Ref range or bare branch: `git diff $(git merge-base origin/master HEAD)..HEAD > /diff.patch`, same with `--name-only`, `` = `git rev-parse --short=8 HEAD`. No `pr.json` on this path. +- Guard: `git rev-parse HEAD` must equal the reviewed head. Experts read the local checkout; if it differs, stop and ask the user to check out the reviewed head. +- ``: 1-3 word `snake_case` summary, `[a-z0-9_]`, <= 24 chars. From the PR title; no PR -> from `git log -1 --format=%s`. +- Report path: `/report.md`. + +Do not `cat` any of the files you just wrote. `wc -l /diff.patch` is the only look you take. + +## Step 2: Round 1, four one-shot experts (one message, parallel) + +Spawn 4 `Agent` calls in a single message: `subagent_type: general-purpose`, `name: -` (bare role names collide with concurrent sessions: one shared agent namespace), no `model` (inherits). Prompt = role block + Charter + this brief, with ``, ``, `` filled in: + +> Target: `` at ``. Diff: `/diff.patch`. Changed files: `/files.txt`. PR title and body: `/pr.json` (drop this sentence when there is no PR). Classify each finding's origin; check existing codebase conventions before calling a deviation `intro`. +> Deliverable = the file `/.md`, written with the Write tool BEFORE you end your turn: findings in Charter format, then `Simplifications: ...`, then `Verdict: APPROVE | REQUEST CHANGES - reason`. A previous worker finished reading and then idled without delivering; the Write call IS the delivery, your final message is just the path. Budget 3/4 reading, 1/4 writing; partial beats unshipped. +> You work alone: no teammates, no SendMessage, no questions back. + +Role blocks: + +- **storage**: Senior storage/DB engineer, 15 years of WAL, B-trees, LSM, crash recovery, fsync semantics. Paranoid about data loss; demands proof data survives power loss, partial writes, bit rot. Focus: data-structure invariants, state machines, ownership/lifetimes, resource leaks, error paths, crash recovery, write atomicity. Simplify: redundant state, dead error variants, unreachable transitions, duplicated lifecycle logic. +- **perf**: Performance engineer / kernel dev. Flamegraphs, cache lines, io_uring, allocators. Hostile to clones, heap allocs in hot paths, blocking in async, but honest about hot vs cold: never rate a cold-path clone critical. Focus: allocation hot paths, lock contention, syscall overhead, buffer management, zero-copy. Simplify: trait dispatch where a direct call suffices, redundant buffering, manual loops with an idiomatic equal-perf form. +- **distsys**: Distributed-systems architect, formal methods. TLA+, linearizability, "message arrives twice / out of order / never". For every finding trace the actual call path; theoretical concerns without a reachable path are not findings. Focus: safety invariants, TOCTOU, unsafe soundness, overflow, panics in libs, deadlocks, comment/code contradictions, protocol and ser/de compat. Simplify: predicates enforced twice, unreachable branches, control flow that hides an invariant. +- **ecosystem**: SDK and API ecosystem lead across the client languages. Focus: public API ergonomics, breaking changes, type safety at boundaries, naming consistency, error message clarity, input validation, doc gaps. Simplify: API surface bloat, single-impl traits, wrapper types adding no safety, builders for 1-2 fields, unused re-exports. + +Collect: wait for the completion notifications, then `ls /*.md`. A role with no file gets one `SendMessage` nudge to `-` ("Write `/.md` now, then stop."); still missing after that, respawn the role once with the same prompt. Never open a subagent transcript via `TaskOutput` (it is the whole JSONL). + +## Step 3: Merge into neutral claims (moderator) + +Read the 4 role files. Write `/claims.md`, one line per claim: `C [sev] file:line - claim. Fix: action. (origin)`. Strip role names, confidence, and argument. Same anchor + same defect from several roles = one claim at the highest severity; keep a private raised-by map for the report. Simplify items are claims too. + +No claims at all: skip Steps 4 and 5, go to Step 6 with empty sections and `Verdict: APPROVE`. The report file still gets written. + +## Step 4: Clean-room validation (one message, parallel) + +Shard claims ~5 per validator. Spawn one `Agent` per shard plus one sweep validator, all in one message: `subagent_type: general-purpose`, `model: opus`, `name: validator--` / `sweep-`. Each gets ONLY: its claims verbatim, `/files.txt`, `/diff.patch`, the target identity, the Charter. Not the role files, not raised-by, not your reasoning; the missing context is what removes the anchoring bias. + +Validator mandate (adversarial): for each claim open the cited `file:line`, trace call sites, then rate `C: PASS | FIX: | REMOVE: `; judge whether the severity is calibrated; re-check the anchor. Deliverable `/validate-.md` via Write, same idle rule as Step 2. + +Sweep mandate: all claims + the diff. Two questions only: which real defects in the diff are missing from the list, and which listed items wrongly clear a bug. Deliverable `/sweep.md`, additions in Charter format tagged `(sweep)`. + +Apply: drop REMOVE, apply FIX (wording, line, severity), fold sweep additions in as `(sweep, unvalidated)`. A `critical` sweep addition gets one extra validator before it may block the verdict. + +## Step 5: Contested items (only when triggered) + +Contested = a validator REMOVEs or downgrades a `critical` or `warning`, or a sweep addition contradicts a PASS. Per item spawn one `Agent` (`model: opus`) with the claim, the validator's verdict text, the expert's original line, and the paths; it writes `UPHELD | OVERTURNED - reason (cite path)` to `/contested-.md`. Cap 5 per run; past the cap you adjudicate and mark `(moderator call)`. + +## Step 6: Synthesize, write, done + +Output in caveman style: + +```text +## Review: [change desc] + +### Confirmed (expert + clean-room validator) +- [sev] file:line - problem. Fix: action. (raised: role[, role]; validated: PASS|FIX) + +### Contested +- file:line - problem. + Expert: position. Validator: counter. **Tiebreak**: UPHELD|OVERTURNED - why. + +### Retracted (validator REMOVE) +- finding - why. + +### Pre-existing (origin pre-*, not blocking) +- file:line - follows pattern in [ref]. + +### Simplification opportunities (non-blocking) +- file:line - current shape. Simpler: alternative. Saves: ~N lines / removes indirection. + +### Verdict: APPROVE | REQUEST CHANGES +Confirmed critical + warning only. Simplifications informational. Reason: one line. + +Counts: critical N, warning N, nit N, simplify N (Confirmed + Simplification sections) +``` + +Then write `/report.md` with: + +1. H1 `# Iggy Team Review - ()`. +2. Metadata, one line each: target ``, reviewed commit, ISO timestamp, roles, validator count, contested count. +3. The report above, verbatim. +4. Appendix `## Raw findings per expert`: each role file verbatim in a fenced block. +5. `## Validation record`: counts of PASS / FIX / REMOVE, sweep additions, contested outcomes. + +Last user-facing line: `Findings written: /report.md`. No cleanup: one-shot agents end themselves, `` stays in the scratchpad. diff --git a/AGENTS.md b/AGENTS.md index 5e74d5a6ff..ada37f1457 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -6,11 +6,10 @@ Transports: QUIC, WebSocket, TCP (custom binary), HTTP (REST). SDKs: Rust, .NET, Java, Python, Go, C++, Node.js. A connectors subsystem ingests from / egresses to external systems via dlopened plugins. -> Skills under `.claude/skills/` are currently scoped to the -> **connectors** subsystem (`core/connectors/`). Load +> Skills live under `.claude/skills/`. Load > [connectors-overview](.claude/skills/connectors-overview/SKILL.md) -> first for any change there. Other subsystems follow the repo-wide -> principles in this file. +> first for any change under `core/connectors/`. Other subsystems +> follow the repo-wide principles in this file. ## Contents @@ -121,8 +120,8 @@ iggy/ ## Skills -Connectors-scoped. Each `SKILL.md` has YAML frontmatter (name, -description). Load `connectors-overview` first as router. +Each `SKILL.md` has YAML frontmatter (name, description). For +connectors work, load `connectors-overview` first as router. - [connectors-overview](.claude/skills/connectors-overview/SKILL.md) - router + universal connector rules - [connector-runtime](.claude/skills/connector-runtime/SKILL.md) - FFI host, lifecycle, state, metrics @@ -132,6 +131,12 @@ description). Load `connectors-overview` first as router. - [connector-transform](.claude/skills/connector-transform/SKILL.md) - transform authoring - [connector-testing](.claude/skills/connector-testing/SKILL.md) - unit + integration test patterns +Repo-wide, user-invoked only. `disable-model-invocation: true` keeps it +out of the agent's context; do not replicate its steps. When a +non-trivial change passes verification, suggest `/team-review `. + +- [team-review](.claude/skills/team-review/SKILL.md) - adversarial 4-expert PR/branch review, ~10 subagents per run + ## Repo-wide principles 1. **Apache 2.0 header on every new source file:** Follow the comment style configured in `licenserc.toml`; common examples: `.rs` uses `// ...`, `Cargo.toml` and shell use `# ...`. From 328b289745e0e24bace1bb531fde4c0f9159d608 Mon Sep 17 00:00:00 2001 From: aias00 Date: Tue, 1 Sep 2026 14:56:53 +0800 Subject: [PATCH 026/182] fix(journal): reject recovery slot collisions (#4009) --- core/configs/src/server_config/metadata.rs | 4 +- core/journal/src/prepare_journal.rs | 85 +++++++++++++++++++++- core/server/config.toml | 4 +- 3 files changed, 89 insertions(+), 4 deletions(-) diff --git a/core/configs/src/server_config/metadata.rs b/core/configs/src/server_config/metadata.rs index 2ba8d821fb..fbec9853fa 100644 --- a/core/configs/src/server_config/metadata.rs +++ b/core/configs/src/server_config/metadata.rs @@ -100,7 +100,9 @@ pub struct MetadataConfig { /// Size of the metadata WAL's in-memory index, in slots (one /// committed-but-unsnapshotted op per slot). Headroom between forced /// checkpoints; more slots = rarer checkpoints, more memory, larger - /// per-checkpoint WAL rewrites. + /// per-checkpoint WAL rewrites. Reducing this for an existing data directory + /// is refused at boot when the live WAL suffix cannot fit without collisions; + /// restore the previous value to recover. pub journal_slots: usize, /// Slot count of the VSR client table: how many distinct clients diff --git a/core/journal/src/prepare_journal.rs b/core/journal/src/prepare_journal.rs index 278fb8e0ec..b00f24b560 100644 --- a/core/journal/src/prepare_journal.rs +++ b/core/journal/src/prepare_journal.rs @@ -575,8 +575,25 @@ impl PrepareJournal { let slot = slot_for_op(header.op, slot_count); - // Note: Regarding duplicate op in WAL. We rewrite it with whichever - // is the latest entry. + // Match append's collision fence while rebuilding the index. Reopening + // with fewer configured slots can otherwise hide an unsnapshotted op + // even though its bytes remain in the WAL. A duplicate op deliberately + // keeps the latest entry, and a snapshotted op is safe to evict. + if let Some(existing) = headers[slot] + && existing.op != header.op + && existing.op > snapshot_op + { + return Err(JournalError::Io(io::Error::new( + io::ErrorKind::InvalidData, + format!( + "journal slot collision while rebuilding the index: op {} and \ + unsnapshotted op {} map to slot {slot} with slot_count={slot_count} \ + (snapshot_op={snapshot_op}); restore the previous \ + metadata.journal_slots value or checkpoint before shrinking it", + header.op, existing.op, + ), + ))); + } headers[slot] = Some(header); offsets[slot] = Some(pos); @@ -2099,6 +2116,70 @@ mod tests { assert_eq!(journal.last_op(), Some(3)); } + #[compio::test] + async fn reopen_with_fewer_slots_rejects_unsnapshotted_collision_without_changing_wal() { + let dir = tempdir().unwrap(); + let path = dir.path().join("journal.wal"); + let journal = PrepareJournal::open_with_slots(&path, 0, 4).await.unwrap(); + journal.append(make_prepare(1, 32)).await.unwrap(); + journal.append(make_prepare(3, 32)).await.unwrap(); + drop(journal); + let wal_before = std::fs::read(&path).unwrap(); + + let error = PrepareJournal::open_with_slots(&path, 0, 2) + .await + .expect_err("shrinking the index must not hide an unsnapshotted entry"); + + assert!( + error.to_string().contains("journal slot collision"), + "unexpected error: {error}" + ); + assert_eq!( + std::fs::read(&path).unwrap(), + wal_before, + "a refused scan must leave the WAL intact" + ); + } + + #[compio::test] + async fn reopen_keeps_latest_duplicate_op() { + let dir = tempdir().unwrap(); + let path = dir.path().join("journal.wal"); + let journal = PrepareJournal::open_with_slots(&path, 0, 4).await.unwrap(); + journal.append(make_prepare(1, 32)).await.unwrap(); + journal.set_snapshot_op(1); + journal.append(make_prepare(1, 64)).await.unwrap(); + drop(journal); + + let journal = PrepareJournal::open_with_slots(&path, 0, 2).await.unwrap(); + let header = *journal.header(1).expect("duplicate op must remain indexed"); + assert_eq!(header.size as usize, HEADER_SIZE + 64); + assert_eq!( + journal + .entry_at(&header) + .await + .unwrap() + .unwrap() + .as_slice() + .len(), + HEADER_SIZE + 64 + ); + } + + #[compio::test] + async fn reopen_with_fewer_slots_can_evict_snapshotted_entry() { + let dir = tempdir().unwrap(); + let path = dir.path().join("journal.wal"); + let journal = PrepareJournal::open_with_slots(&path, 0, 4).await.unwrap(); + journal.append(make_prepare(1, 32)).await.unwrap(); + journal.append(make_prepare(3, 32)).await.unwrap(); + drop(journal); + + let journal = PrepareJournal::open_with_slots(&path, 1, 2).await.unwrap(); + assert!(journal.header(1).is_none()); + assert_eq!(journal.header(3).map(|header| header.op), Some(3)); + } + const POISON_REASON: &str = "test: simulated post-rename failure"; #[compio::test] diff --git a/core/server/config.toml b/core/server/config.toml index e70cfacee3..79bdc46e88 100644 --- a/core/server/config.toml +++ b/core/server/config.toml @@ -960,7 +960,9 @@ prepare_queue_depth = 32 # Size of the metadata WAL's in-memory index, in slots (one committed but # not-yet-snapshotted op per slot). Larger values buy more headroom # between forced checkpoints at the cost of memory and bigger WAL -# rewrites per checkpoint. +# rewrites per checkpoint. Reducing this for an existing data directory is +# refused at boot if the live WAL suffix collides in the smaller index; +# restore the previous value to recover. journal_slots = 1024 # Slot count of the VSR client table: how many distinct clients (TCP/QUIC/WS From 8c4986ac545c1b75afd40b11c425d35b7fbbe058 Mon Sep 17 00:00:00 2001 From: Grzegorz Koszyk <112548209+numinnex@users.noreply.github.com> Date: Tue, 1 Sep 2026 09:25:15 +0200 Subject: [PATCH 027/182] fix(io_uring): keep shard startup off blocking fallbacks Kernels without IORING_OP_FTRUNCATE, IORING_OP_BIND, or IORING_OP_LISTEN send compio to its blocking fallback. Shard executors disable that worker pool, so a torn metadata WAL panicked during recovery and synchronous repair alone only moved the panic to plain TCP, WebSocket, or HTTP startup. Repair torn WAL tails through std::fs using a duplicate of the open descriptor. The tracked write offset updates immediately after truncation so it matches the file even when the fsync fails, and the helper syncs the new length itself so the repair is durable without a separate fsync from the caller. Bind plain listeners through the existing synchronous socket2 path while preserving TCP_NODELAY. This keeps shard startup off the unsupported opcodes and restores the mainline Linux 5.19 floor. Regression tests pin no-pool truncation, descriptor identity, and listener configuration. --------- Co-authored-by: Piotr Gankiewicz --- core/journal/src/file_storage.rs | 78 ++++++++++++++++--- core/journal/src/prepare_journal.rs | 10 +-- core/message_bus/src/client_listener/mod.rs | 75 ++++++++++-------- core/message_bus/src/client_listener/tcp.rs | 8 +- core/message_bus/src/client_listener/ws.rs | 8 +- core/message_bus/src/replica/io.rs | 4 +- core/message_bus/tests/graceful_shutdown.rs | 4 +- .../message_bus/tests/tcp_client_roundtrip.rs | 4 +- core/message_bus/tests/ws_client_roundtrip.rs | 4 +- core/server/src/bootstrap.rs | 28 +++---- core/server/src/http.rs | 8 +- helm/charts/iggy/README.md | 11 ++- helm/charts/iggy/README.md.gotmpl | 11 ++- 13 files changed, 165 insertions(+), 88 deletions(-) diff --git a/core/journal/src/file_storage.rs b/core/journal/src/file_storage.rs index 1273d27b98..b8f07fb089 100644 --- a/core/journal/src/file_storage.rs +++ b/core/journal/src/file_storage.rs @@ -19,7 +19,9 @@ use crate::Storage; use compio::buf::IoBuf; use compio::io::{AsyncReadAtExt, AsyncWriteAtExt}; use std::cell::{Cell, UnsafeCell}; +use std::fs; use std::io; +use std::os::fd::AsFd; use std::path::{Path, PathBuf}; /// File-backed storage implementing the `Storage` trait. @@ -57,20 +59,30 @@ impl FileStorage { self.write_offset.get() } - /// Truncate the file to `len` bytes. + /// Truncate the file to `len` bytes and make the new length durable. + /// + /// Synchronous `std::fs` on a duplicate of the open descriptor, not compio: + /// compio's `set_len` submits `IORING_OP_FTRUNCATE`, which landed in + /// mainline Linux 6.9. When the opcode is unavailable, the driver falls + /// back to its blocking pool, and shard proactors run with + /// `thread_pool_limit(0)`, so the fallback panics the shard instead of + /// repairing the WAL. `std::fs` needs neither the opcode nor the pool. The + /// sole caller is boot-time torn-tail repair, so blocking the shard thread + /// here costs nothing. + /// + /// `sync_all` makes the durable-truncation contract explicit and matches + /// segment recovery. Its additional metadata synchronization is acceptable + /// because this runs only during boot-time repair. /// /// # Errors - /// Returns an I/O error if truncation fails. - // TODO(hubcio): compio `set_len` submits IORING_OP_FTRUNCATE, which kernels - // below 6.9 do not support; the driver then falls back to its blocking - // pool, and shard proactors run with `thread_pool_limit(0)`, so the torn - // WAL repair panics the shard on such kernels instead of repairing. Use a - // synchronous `std::fs` truncate here (boot-time path) or gate on a probe. - pub async fn truncate(&self, len: u64) -> io::Result<()> { + /// Returns an I/O error if the descriptor cannot be cloned, truncated, or synced. + pub(crate) fn truncate(&self, len: u64) -> io::Result<()> { + // SAFETY: single-threaded compio runtime, no concurrent access to the file. let file = unsafe { &*self.file.get() }; - file.set_len(len).await?; + let file = fs::File::from(file.as_fd().try_clone_to_owned()?); + file.set_len(len)?; self.write_offset.set(len); - Ok(()) + file.sync_all() } /// Fsync the file to disk. @@ -178,3 +190,49 @@ impl Storage for FileStorage { Ok(buffer) } } + +#[cfg(test)] +mod tests { + use super::FileStorage; + use server_common::executor::create_shard_executor; + use tempfile::tempdir; + + /// Pins the synchronous truncate signature and verifies it works inside a + /// shard executor with no blocking pool. A modern test kernel supports + /// `IORING_OP_FTRUNCATE`, so this does not reproduce compio's fallback. + #[test] + fn given_a_shard_executor_with_no_blocking_pool_when_truncating_should_repair_the_file() { + let runtime = create_shard_executor().unwrap(); + runtime.block_on(async { + let dir = tempdir().unwrap(); + let path = dir.path().join("journal.wal"); + let storage = FileStorage::open(&path).await.unwrap(); + storage.write_append(vec![0xAB_u8; 128]).await.unwrap(); + + storage.truncate(64).unwrap(); + + assert_eq!(storage.file_len(), 64); + assert_eq!(std::fs::metadata(&path).unwrap().len(), 64); + }); + } + + #[test] + fn given_a_replaced_path_when_truncating_should_truncate_the_open_file() { + let runtime = create_shard_executor().unwrap(); + runtime.block_on(async { + let dir = tempdir().unwrap(); + let path = dir.path().join("journal.wal"); + let renamed_path = dir.path().join("journal.renamed.wal"); + let storage = FileStorage::open(&path).await.unwrap(); + storage.write_append(vec![0xAB_u8; 128]).await.unwrap(); + std::fs::rename(&path, &renamed_path).unwrap(); + std::fs::write(&path, vec![0xCD_u8; 256]).unwrap(); + + storage.truncate(64).unwrap(); + + assert_eq!(storage.file_len(), 64); + assert_eq!(std::fs::metadata(&renamed_path).unwrap().len(), 64); + assert_eq!(std::fs::metadata(&path).unwrap().len(), 256); + }); + } +} diff --git a/core/journal/src/prepare_journal.rs b/core/journal/src/prepare_journal.rs index b00f24b560..d7de238752 100644 --- a/core/journal/src/prepare_journal.rs +++ b/core/journal/src/prepare_journal.rs @@ -243,12 +243,7 @@ async fn truncate_or_fail( reason, "truncating torn WAL tail; no complete entry follows the damage" ); - storage.truncate(pos).await?; - // The repair must be crash-durable. `FileStorage::truncate` is a - // bare `set_len`; without this fsync a power loss right after the - // repair re-presents the torn tail on the next boot. Mirrors the - // write-then-fsync the `append` path already does. - storage.fsync().await?; + storage.truncate(pos)?; Ok(()) } @@ -1802,8 +1797,7 @@ mod tests { let storage = FileStorage::open(&path).await.unwrap(); let full_len = storage.file_len(); // Remove the last 10 bytes (partial second entry) - storage.truncate(full_len - 10).await.unwrap(); - storage.fsync().await.unwrap(); + storage.truncate(full_len - 10).unwrap(); } // Reopen, should recover only the first entry diff --git a/core/message_bus/src/client_listener/mod.rs b/core/message_bus/src/client_listener/mod.rs index 1ed966bd22..f0f739496b 100644 --- a/core/message_bus/src/client_listener/mod.rs +++ b/core/message_bus/src/client_listener/mod.rs @@ -85,12 +85,16 @@ //! violation). DoS-shaped abuse is bounded instead by handshake-grace //! timeouts and the bus-wide [`crate::installer`] backpressure budget. -use crate::{GenericHeader, Message}; -use compio::net::{TcpListener, TcpSocket}; -use iggy_common::IggyError; use std::net::SocketAddr; use std::rc::Rc; +use compio::net::TcpListener; +use iggy_common::IggyError; +use socket2::SockRef; + +use crate::socket_opts::bind_reusable_tcp_listener; +use crate::{GenericHeader, Message}; + pub mod quic; pub mod tcp; pub mod tcp_tls; @@ -100,40 +104,22 @@ pub mod wss; /// Bind a TCP listener with `TCP_NODELAY` set, the shared shape used by /// the plain-TCP and WS pre-upgrade client listeners. /// -/// compio 0.19 replaced `TcpListener::bind_with_options(addr, SocketOpts)` -/// with the `TcpSocket` builder; this preserves the prior `nodelay(true)` -/// bind. `SO_REUSEADDR` is set so a restarted server can rebind the port -/// while a previous client connection lingers in `TIME_WAIT` (matching the -/// replica and TLS listeners; QUIC sets no reuse flag since UDP has no -/// `TIME_WAIT`); `SO_REUSEPORT` is intentionally not set: only shard 0 -/// binds the client listeners (see each caller). +/// Binding stays synchronous through `bind_reusable_tcp_listener` so shard +/// startup does not depend on compio's `IORING_OP_BIND` and +/// `IORING_OP_LISTEN` fallback path. `TCP_NODELAY` remains set on the listener +/// to preserve the previous plain-TCP and WebSocket bind configuration. /// /// # Errors /// -/// Returns [`IggyError::CannotBindToSocket`] if the bind/listen fails. -#[allow(clippy::future_not_send)] -pub async fn bind_nodelay_listener( - addr: SocketAddr, -) -> Result<(TcpListener, SocketAddr), IggyError> { - let socket = match addr { - SocketAddr::V4(_) => TcpSocket::new_v4().await, - SocketAddr::V6(_) => TcpSocket::new_v6().await, - } - .map_err(|e| IggyError::IoError(e.to_string()))?; - socket - .set_nodelay(true) - .map_err(|e| IggyError::IoError(e.to_string()))?; - socket - .set_reuseaddr(true) - .map_err(|e| IggyError::IoError(e.to_string()))?; - socket - .bind(addr) - .await - .map_err(|_| IggyError::CannotBindToSocket(addr.to_string()))?; - let listener = socket - .listen(libc::SOMAXCONN) - .await +/// Returns [`IggyError::CannotBindToSocket`] if the bind or listen fails, or +/// [`IggyError::IoError`] if configuring `TCP_NODELAY` or reading the bound +/// address fails. +pub fn bind_nodelay_listener(addr: SocketAddr) -> Result<(TcpListener, SocketAddr), IggyError> { + let listener = bind_reusable_tcp_listener(addr) .map_err(|_| IggyError::CannotBindToSocket(addr.to_string()))?; + SockRef::from(&listener) + .set_tcp_nodelay(true) + .map_err(|e| IggyError::IoError(e.to_string()))?; let actual = listener .local_addr() .map_err(|e| IggyError::IoError(e.to_string()))?; @@ -147,3 +133,26 @@ pub async fn bind_nodelay_listener( /// `RequestHeader` message with the same handler signature, regardless /// of whether the wire is plain TCP, TLS, WS, WSS, or QUIC. pub type RequestHandler = Rc)>; + +#[cfg(test)] +mod tests { + use std::net::{Ipv4Addr, SocketAddr}; + + use server_common::executor::create_shard_executor; + use socket2::SockRef; + + use super::bind_nodelay_listener; + + #[test] + fn given_a_shard_executor_when_binding_a_client_listener_should_preserve_nodelay() { + let runtime = create_shard_executor().unwrap(); + runtime.block_on(async { + let addr = SocketAddr::new(Ipv4Addr::LOCALHOST.into(), 0); + + let (listener, bound_addr) = bind_nodelay_listener(addr).unwrap(); + + assert_ne!(bound_addr.port(), 0); + assert!(SockRef::from(&listener).tcp_nodelay().unwrap()); + }); + } +} diff --git a/core/message_bus/src/client_listener/tcp.rs b/core/message_bus/src/client_listener/tcp.rs index c900d390cf..a14673774a 100644 --- a/core/message_bus/src/client_listener/tcp.rs +++ b/core/message_bus/src/client_listener/tcp.rs @@ -40,13 +40,13 @@ use tracing::{debug, error, info}; /// /// # Errors /// -/// Returns [`IggyError::CannotBindToSocket`] if the bind fails. -#[allow(clippy::future_not_send)] -pub async fn bind(addr: SocketAddr) -> Result<(TcpListener, SocketAddr), IggyError> { +/// Returns [`IggyError::CannotBindToSocket`] if the bind fails, or +/// [`IggyError::IoError`] if listener configuration fails. +pub fn bind(addr: SocketAddr) -> Result<(TcpListener, SocketAddr), IggyError> { // `SO_REUSEPORT` intentionally not set: only shard 0 binds the client // listener. The shard-0 coordinator round-robins accepts to owning // shards via `shard::LifecycleFrame::ClientConnectionSetup`. - bind_nodelay_listener(addr).await + bind_nodelay_listener(addr) } /// Run the client listener accept loop until the shutdown token fires. The diff --git a/core/message_bus/src/client_listener/ws.rs b/core/message_bus/src/client_listener/ws.rs index 9d2521209f..f61f8c7e40 100644 --- a/core/message_bus/src/client_listener/ws.rs +++ b/core/message_bus/src/client_listener/ws.rs @@ -54,13 +54,13 @@ use tracing::{debug, error, info}; /// /// # Errors /// -/// Returns [`IggyError::CannotBindToSocket`] if the bind fails. -#[allow(clippy::future_not_send)] -pub async fn bind(addr: SocketAddr) -> Result<(TcpListener, SocketAddr), IggyError> { +/// Returns [`IggyError::CannotBindToSocket`] if the bind fails, or +/// [`IggyError::IoError`] if listener configuration fails. +pub fn bind(addr: SocketAddr) -> Result<(TcpListener, SocketAddr), IggyError> { // `SO_REUSEPORT` intentionally not set: only shard 0 binds the WS // listener. The shard-0 coordinator round-robins accepts to owning // shards via `shard::LifecycleFrame::ClientWsConnectionSetup`. - bind_nodelay_listener(addr).await + bind_nodelay_listener(addr) } /// Run the WS pre-upgrade listener accept loop until the shutdown diff --git a/core/message_bus/src/replica/io.rs b/core/message_bus/src/replica/io.rs index 290db8e2d2..9368ef5a7e 100644 --- a/core/message_bus/src/replica/io.rs +++ b/core/message_bus/src/replica/io.rs @@ -212,7 +212,7 @@ pub async fn start_on_shard_zero( ); let (replica_listener, replica_bound) = bind_replica_listener(replica_listen_addr).await?; - let (clients_listener, client_bound) = client_listener::tcp::bind(client_listen_addr).await?; + let (clients_listener, client_bound) = client_listener::tcp::bind(client_listen_addr)?; let token_for_replica = bus.token(); let replica_handle = compio::runtime::spawn(async move { @@ -228,7 +228,7 @@ pub async fn start_on_shard_zero( let ws_bound = match (ws_listen_addr, on_accepted_ws_client) { (Some(addr), Some(on_accepted_ws)) => { - let (ws_listener, ws_bound) = client_listener::ws::bind(addr).await?; + let (ws_listener, ws_bound) = client_listener::ws::bind(addr)?; let token_for_ws = bus.token(); let ws_handle = compio::runtime::spawn(async move { client_listener::ws::run(ws_listener, token_for_ws, on_accepted_ws).await; diff --git a/core/message_bus/tests/graceful_shutdown.rs b/core/message_bus/tests/graceful_shutdown.rs index b37ac40488..cee1a8a4bd 100644 --- a/core/message_bus/tests/graceful_shutdown.rs +++ b/core/message_bus/tests/graceful_shutdown.rs @@ -32,7 +32,7 @@ use std::time::Duration; async fn drains_all_clients_within_timeout() { let bus = Rc::new(IggyMessageBus::new(0)); let on_request: RequestHandler = Rc::new(|_, _| {}); - let (listener, addr) = bind(loopback()).await.unwrap(); + let (listener, addr) = bind(loopback()).unwrap(); let token = bus.token(); let accept_delegate = install_clients_locally(bus.clone(), on_request); @@ -87,7 +87,7 @@ async fn drains_all_clients_within_timeout() { async fn connection_drain_precedes_slow_background() { let bus = Rc::new(IggyMessageBus::new(0)); let on_request: RequestHandler = Rc::new(|_, _| {}); - let (listener, addr) = bind(loopback()).await.unwrap(); + let (listener, addr) = bind(loopback()).unwrap(); let token = bus.token(); let accept_delegate = install_clients_locally(bus.clone(), on_request); diff --git a/core/message_bus/tests/tcp_client_roundtrip.rs b/core/message_bus/tests/tcp_client_roundtrip.rs index 9925026b23..b020ad8841 100644 --- a/core/message_bus/tests/tcp_client_roundtrip.rs +++ b/core/message_bus/tests/tcp_client_roundtrip.rs @@ -49,7 +49,7 @@ async fn request_reply_round_trip() { .detach(); }); - let (listener, addr) = bind(loopback()).await.expect("bind"); + let (listener, addr) = bind(loopback()).expect("bind"); let token = bus.token(); let accept_delegate = install_clients_locally(bus.clone(), on_request); let accept_handle = compio::runtime::spawn(async move { @@ -88,7 +88,7 @@ async fn unexpected_command_is_ignored() { let _ = tx.try_send(()); }); - let (listener, addr) = bind(loopback()).await.unwrap(); + let (listener, addr) = bind(loopback()).unwrap(); let token = bus.token(); let accept_delegate = install_clients_locally(bus.clone(), on_request); let accept_handle = compio::runtime::spawn(async move { diff --git a/core/message_bus/tests/ws_client_roundtrip.rs b/core/message_bus/tests/ws_client_roundtrip.rs index 3f10ff4e07..f14d9a88e6 100644 --- a/core/message_bus/tests/ws_client_roundtrip.rs +++ b/core/message_bus/tests/ws_client_roundtrip.rs @@ -88,7 +88,7 @@ async fn handshake_succeeds_and_round_trip_completes() { .detach(); }); - let (listener, server_addr) = bind(loopback()).await.expect("bind"); + let (listener, server_addr) = bind(loopback()).expect("bind"); let token = bus.token(); let on_accepted = install_ws_clients_locally(bus.clone(), on_request); let accept_handle = compio::runtime::spawn(async move { @@ -128,7 +128,7 @@ async fn handshake_succeeds_without_subprotocol_header() { let bus = Rc::new(IggyMessageBus::new(0)); let on_request: RequestHandler = Rc::new(|_, _| {}); - let (listener, server_addr) = bind(loopback()).await.expect("bind"); + let (listener, server_addr) = bind(loopback()).expect("bind"); let token = bus.token(); let on_accepted = install_ws_clients_locally(bus.clone(), on_request); let accept_handle = compio::runtime::spawn(async move { diff --git a/core/server/src/bootstrap.rs b/core/server/src/bootstrap.rs index 9a58b8b78a..4320c3bc5e 100644 --- a/core/server/src/bootstrap.rs +++ b/core/server/src/bootstrap.rs @@ -3319,10 +3319,9 @@ async fn start_tcp_runtime( &config.cluster, Arc::clone(&config.system), &self_advertised, - self_ports, + &self_ports, shard_metrics_all, - ) - .await?; + )?; } Ok(()) @@ -3484,7 +3483,7 @@ async fn start_manual_runtime( None }; - let bound_clients = start_client_listeners(shard, config, topology, &accepted_clients).await?; + let bound_clients = start_client_listeners(shard, config, topology, &accepted_clients)?; write_current_config( config, Some(topology.self_replica_id), @@ -3849,7 +3848,7 @@ fn mint_client_meta( ClientConnMeta::new(coord.mint_shard_zero_client_id(), peer_addr, transport) } -async fn start_client_listeners( +fn start_client_listeners( shard: &Rc, config: &ServerConfig, topology: &TcpTopology, @@ -3859,7 +3858,6 @@ async fn start_client_listeners( if config.tcp.enabled && !config.tcp.tls.enabled { let (listener, bound_addr) = client_listener::tcp::bind(topology.client_listen_addr) - .await .map_err(|source| { error!( addr = %topology.client_listen_addr, @@ -3878,7 +3876,12 @@ async fn start_client_listeners( } if let Some(ws_addr) = topology.ws_listen_addr { - bound.ws = Some(start_websocket_listener(shard, config, ws_addr, accepted_clients).await?); + bound.ws = Some(start_websocket_listener( + shard, + config, + ws_addr, + accepted_clients, + )?); } if let Some(quic_addr) = topology.quic_listen_addr { @@ -4080,7 +4083,7 @@ fn load_tcp_tls_server_credentials( /// `websocket.tls.enabled` (the plain-WS accept loop must not also bind the /// port -- a plain upgrade parser fed a TLS `ClientHello` rejects every /// connection with an httparse error), plain WS otherwise. -async fn start_websocket_listener( +fn start_websocket_listener( shard: &Rc, config: &ServerConfig, ws_addr: SocketAddr, @@ -4101,11 +4104,10 @@ async fn start_websocket_listener( shard.bus.track_background(wss_handle); Ok(bound_addr) } else { - let (listener, bound_addr) = - client_listener::ws::bind(ws_addr).await.map_err(|source| { - error!(addr = %ws_addr, error = %source, "failed to bind websocket listener"); - source - })?; + let (listener, bound_addr) = client_listener::ws::bind(ws_addr).map_err(|source| { + error!(addr = %ws_addr, error = %source, "failed to bind websocket listener"); + source + })?; let token = shard.bus.token(); let accepted_ws = accepted_clients.ws.clone(); let ws_handle = compio::runtime::spawn(async move { diff --git a/core/server/src/http.rs b/core/server/src/http.rs index fc72acc4a7..ed670883d8 100644 --- a/core/server/src/http.rs +++ b/core/server/src/http.rs @@ -95,7 +95,7 @@ use crate::server_error::ServerError; /// `http_config.jwt`, the `[http.cors]` config is invalid, the `[http.tls]` /// credentials cannot be loaded, or the listener cannot bind to `addr`. #[allow(clippy::too_many_arguments)] -pub async fn start( +pub fn start( shard: &Rc, addr: SocketAddr, http_config: &HttpConfig, @@ -104,7 +104,7 @@ pub async fn start( cluster: &ClusterConfig, system_config: Arc, self_advertised: &str, - self_ports: TransportPorts, + self_ports: &TransportPorts, shard_metrics_all: &[shard::metrics::ShardMetrics], ) -> Result<(), ServerError> { // In cluster mode with no configured JWT secret the signing key derives @@ -139,7 +139,7 @@ pub async fn start( // Same early-fail rule for the scrape path: axum panics on a route // without a leading '/', so reject it as a config error instead. let metrics_endpoint = metrics::validated_endpoint(&http_config.metrics)?; - let (listener, bound_addr) = client_listener::tcp::bind(addr).await?; + let (listener, bound_addr) = client_listener::tcp::bind(addr)?; let state: HttpState = SendWrapper::new(Rc::new(HttpInner { shard: Rc::clone(shard), @@ -156,7 +156,7 @@ pub async fn start( // ports arrive resolved from the caller. self_ports: TransportPorts { http: Some(bound_addr.port()), - ..self_ports + ..self_ports.clone() }, // The HTTP listener is shard-0-only, where the live consensus // handle supplies the leader; the published-view fallback is diff --git a/helm/charts/iggy/README.md b/helm/charts/iggy/README.md index d9861ca13d..bfd38cb7ae 100644 --- a/helm/charts/iggy/README.md +++ b/helm/charts/iggy/README.md @@ -15,8 +15,15 @@ A Helm chart for Apache Iggy server and web-ui Iggy server uses `io_uring` for high-performance async I/O. This requires: -1. **IPC_LOCK capability** - For locking memory required by io_uring -2. **Unconfined seccomp profile** - To allow io_uring syscalls +1. **Linux kernel 5.19 or newer on the node** + + * Shard rings require `IORING_SETUP_COOP_TASKRUN` and `IORING_SETUP_TASKRUN_FLAG`. + * Compio's asynchronous socket creation requires `IORING_OP_SOCKET`. + * Mainline Linux provides these features starting in 5.19. + * Older kernels fail during shard startup. The node kernel matters, not the container image. + +2. **IPC_LOCK capability** - For locking memory required by io_uring +3. **Unconfined seccomp profile** - To allow io_uring syscalls These are configured by default for the Iggy server via the chart's root-level `securityContext` and `podSecurityContext`. The web UI uses `ui.securityContext` diff --git a/helm/charts/iggy/README.md.gotmpl b/helm/charts/iggy/README.md.gotmpl index 03a2700ea7..a36d30adcd 100644 --- a/helm/charts/iggy/README.md.gotmpl +++ b/helm/charts/iggy/README.md.gotmpl @@ -33,8 +33,15 @@ under the License. Iggy server uses `io_uring` for high-performance async I/O. This requires: -1. **IPC_LOCK capability** - For locking memory required by io_uring -2. **Unconfined seccomp profile** - To allow io_uring syscalls +1. **Linux kernel 5.19 or newer on the node** + + * Shard rings require `IORING_SETUP_COOP_TASKRUN` and `IORING_SETUP_TASKRUN_FLAG`. + * Compio's asynchronous socket creation requires `IORING_OP_SOCKET`. + * Mainline Linux provides these features starting in 5.19. + * Older kernels fail during shard startup. The node kernel matters, not the container image. + +2. **IPC_LOCK capability** - For locking memory required by io_uring +3. **Unconfined seccomp profile** - To allow io_uring syscalls These are configured by default for the Iggy server via the chart's root-level `securityContext` and `podSecurityContext`. The web UI uses `ui.securityContext` From 0270b76b6b23f45d9fe0ab9a0c8a280f2e396d99 Mon Sep 17 00:00:00 2001 From: Hubert Gruszecki Date: Tue, 1 Sep 2026 14:59:56 +0200 Subject: [PATCH 028/182] refactor(server): make the module graph a DAG and enforce it (#4023) --- Cargo.lock | 16 +- Cargo.toml | 2 - core/partitions/src/send_messages2.rs | 17 - core/server/Cargo.toml | 7 +- core/server/build.rs | 23 ++ core/server/src/auth.rs | 14 +- core/server/src/bootstrap.rs | 187 +--------- core/server/src/consumer_group.rs | 6 +- core/server/src/dispatch.rs | 58 ++-- core/server/src/dispatch/authz.rs | 2 +- core/server/src/http.rs | 8 +- core/server/src/http/forward.rs | 19 +- core/server/src/http/handlers.rs | 67 ++++ core/server/src/http/metrics.rs | 101 +----- core/server/src/http/reads.rs | 2 +- core/server/src/http/state.rs | 23 +- core/server/src/http/submit.rs | 2 +- core/server/src/{ => http}/web.rs | 0 core/server/src/lib.rs | 53 +-- core/server/src/partition_helpers.rs | 4 +- core/server/src/partition_reconciler.rs | 2 +- core/server/src/pat.rs | 4 +- .../src/personal_access_token_cleaner.rs | 2 +- core/server/src/responses.rs | 40 +-- core/server/src/segment_cleaner.rs | 2 +- core/server/src/server_error.rs | 2 +- .../lib.rs => server/src/shard_allocator.rs} | 51 ++- core/server/src/shell.rs | 202 +++++++++++ core/server/src/users.rs | 4 +- core/server/src/wire.rs | 12 +- core/server/tests/module_graph.rs | 319 ++++++++++++++++++ core/shard_allocator/Cargo.toml | 41 --- core/shard_allocator/build.rs | 43 --- core/simulator/src/replica.rs | 3 +- 34 files changed, 811 insertions(+), 527 deletions(-) delete mode 100644 core/partitions/src/send_messages2.rs rename core/server/src/{ => http}/web.rs (100%) rename core/{shard_allocator/src/lib.rs => server/src/shard_allocator.rs} (93%) create mode 100644 core/server/src/shell.rs create mode 100644 core/server/tests/module_graph.rs delete mode 100644 core/shard_allocator/Cargo.toml delete mode 100644 core/shard_allocator/build.rs diff --git a/Cargo.lock b/Cargo.lock index 251f7f309a..5a622dab6d 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -12048,6 +12048,7 @@ dependencies = [ "compio", "configs", "consensus", + "cpu_allocation", "crossfire", "ctrlc", "cyper", @@ -12082,7 +12083,9 @@ dependencies = [ "opentelemetry_sdk", "papaya", "partitions", + "proc-macro2", "prometheus-client", + "quote", "rand 0.10.2", "ringbuffer", "rmp-serde", @@ -12097,9 +12100,9 @@ dependencies = [ "serde_json", "server_common", "shard", - "shard_allocator", "socket2 0.6.5", "strum 0.28.0", + "syn 2.0.119", "sysinfo 0.39.6", "system_stats", "tempfile", @@ -12233,17 +12236,6 @@ dependencies = [ "tracing", ] -[[package]] -name = "shard_allocator" -version = "0.1.0" -dependencies = [ - "cpu_allocation", - "hwlocality", - "nix", - "thiserror 2.0.19", - "tracing", -] - [[package]] name = "sharded-slab" version = "0.1.7" diff --git a/Cargo.toml b/Cargo.toml index 7872f40937..0f339f28cc 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -64,7 +64,6 @@ members = [ "core/server", "core/server_common", "core/shard", - "core/shard_allocator", "core/simulator", "core/system_stats", "core/tools", @@ -303,7 +302,6 @@ serial_test = "3.5.0" server = { path = "core/server" } server_common = { path = "core/server_common" } shard = { path = "core/shard" } -shard_allocator = { path = "core/shard_allocator" } simd-json = { version = "0.17.3", features = ["serde_impl"] } smallvec = "1.15" socket2 = "0.6.5" diff --git a/core/partitions/src/send_messages2.rs b/core/partitions/src/send_messages2.rs deleted file mode 100644 index 5cd17fb5a6..0000000000 --- a/core/partitions/src/send_messages2.rs +++ /dev/null @@ -1,17 +0,0 @@ -// Licensed to the Apache Software Foundation (ASF) under one -// or more contributor license agreements. See the NOTICE file -// distributed with this work for additional information -// regarding copyright ownership. The ASF licenses this file -// to you under the Apache License, Version 2.0 (the -// "License"); you may not use this file except in compliance -// with the License. You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, -// software distributed under the License is distributed on an -// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY -// KIND, either express or implied. See the License for the -// specific language governing permissions and limitations -// under the License. - diff --git a/core/server/Cargo.toml b/core/server/Cargo.toml index 22d6ba2f0e..1d59114880 100644 --- a/core/server/Cargo.toml +++ b/core/server/Cargo.toml @@ -45,12 +45,10 @@ ignored = [ "figlet-rs", "hash32", "human-repr", - "hwlocality", "jsonwebtoken", "left-right", "mimalloc", "mime_guess", - "nix", "opentelemetry", "opentelemetry-appender-tracing", "opentelemetry-otlp", @@ -103,6 +101,7 @@ clap = { workspace = true } compio = { workspace = true } configs = { workspace = true } consensus = { workspace = true } +cpu_allocation = { workspace = true } crossfire = { workspace = true } ctrlc = { workspace = true } cyper = { workspace = true } @@ -151,7 +150,6 @@ serde = { workspace = true } serde_json = { workspace = true } server_common = { workspace = true } shard = { workspace = true } -shard_allocator = { workspace = true } socket2 = { workspace = true } strum = { workspace = true } sysinfo = { workspace = true } @@ -179,6 +177,8 @@ vergen-git2 = { workspace = true } [dev-dependencies] assert_cmd = { workspace = true } bytemuck = { workspace = true } +proc-macro2 = { workspace = true } +quote = { workspace = true } # Reconciler unit tests assert on `ShardMetrics` snapshots and # `IggyShard::parked_frame_count`, gated to test/simulator so they cannot grow # production callers. `shard`'s own `cfg(test)` is false when compiled as our @@ -186,6 +186,7 @@ bytemuck = { workspace = true } # keeps a dev-dependency's features out of non-test targets, so a production # build still links `shard` without `simulator`. shard = { workspace = true, features = ["simulator"] } +syn = { workspace = true } tokio = { workspace = true, features = ["full", "test-util"] } [lints.clippy] diff --git a/core/server/build.rs b/core/server/build.rs index 421dd918ae..ad152b7230 100644 --- a/core/server/build.rs +++ b/core/server/build.rs @@ -23,11 +23,34 @@ const WEB_ASSETS_PATH: &str = "web/build/static"; const WEB_INDEX_FILE: &str = "web/build/static/index.html"; fn main() -> Result<(), Box> { + stub_libm_for_musl(); verify_web_assets_if_enabled(); emit_vergen_instructions()?; Ok(()) } +// Vendored `hwloc` references `cbrt`, which makes the linker pull in a +// `libm`. musl folds the math functions into its `libc`, so the Rust +// musl sysroot ships no `libm.a`. Without one, `-lm` falls through to +// the host glibc's `libm.a`, whose `cbrt` needs glibc-internal +// `__frexp`/`__ldexp` symbols that do not exist on musl, and the static +// link fails. Drop an empty `libm.a` stub on the search path so `-lm` +// resolves to nothing and `cbrt` is satisfied later by musl's own +// `libc`. No effect on non-musl targets. +fn stub_libm_for_musl() { + if env::var("CARGO_CFG_TARGET_ENV").as_deref() != Ok("musl") { + return; + } + + let out_dir = env::var("OUT_DIR").expect("OUT_DIR is set by cargo for build scripts"); + let stub = PathBuf::from(&out_dir).join("libm.a"); + + // `!\n` is the canonical header of an empty `ar` archive. + std::fs::write(&stub, b"!\n").expect("write empty libm.a stub"); + + println!("cargo:rustc-link-search=native={out_dir}"); +} + /// Returns the workspace root (iggy/), two levels up from core/server. fn workspace_root() -> PathBuf { PathBuf::from(env::var("CARGO_MANIFEST_DIR").unwrap()) diff --git a/core/server/src/auth.rs b/core/server/src/auth.rs index 589214852b..1851281c6a 100644 --- a/core/server/src/auth.rs +++ b/core/server/src/auth.rs @@ -22,11 +22,11 @@ //! surfaced as typed `Eviction` frames, transient ones as result-framed //! replay hints. -use crate::bootstrap::{ShellBus, ShellShard}; use crate::dispatch::{send_login_eviction, submit_register_on_owner}; use crate::login_register::LoginRegisterError; use crate::responses::{build_login_register_reply, current_metadata_commit}; use crate::session_manager::{ClientSdkInfo, SessionManager}; +use crate::shell::{ShellBus, ShellShard}; use consensus::{MetadataHandle, build_result_rejection_reply}; use iggy_binary_protocol::PrepareHeader; use iggy_binary_protocol::{ClientVersionInfo, EvictionReason, RoutedRequestHeader}; @@ -58,11 +58,11 @@ static DUMMY_PASSWORD_HASH: LazyLock = /// Pay the one-time Argon2 cost of [`DUMMY_PASSWORD_HASH`] at boot instead of /// inside the first unknown-username login request. -pub(crate) fn warm_dummy_password_hash() { +pub fn warm_dummy_password_hash() { LazyLock::force(&DUMMY_PASSWORD_HASH); } -pub(crate) fn verify_login_credentials( +pub fn verify_login_credentials( shard: &Rc>, username: &str, password: &str, @@ -109,7 +109,7 @@ where }) } -pub(crate) fn verify_pat_credentials( +pub fn verify_pat_credentials( shard: &Rc>, token: &str, ) -> Result @@ -127,7 +127,7 @@ where /// seconds, `u64::MAX` when the PAT never expires). The HTTP extractor keys a /// per-token VSR session table on this expiry for lazy eviction; the wire and /// login paths only need the user id and go through [`verify_pat_credentials`]. -pub(crate) fn verify_pat_credentials_with_expiry( +pub fn verify_pat_credentials_with_expiry( shard: &Rc>, token: &str, ) -> Result<(u32, u64), LoginRegisterError> @@ -179,7 +179,7 @@ where } #[allow(clippy::future_not_send)] -pub(crate) async fn complete_login_register( +pub async fn complete_login_register( shard: &Rc>, sessions: &Rc>, transport_client_id: u128, @@ -294,7 +294,7 @@ where /// SDK surfaces the real reason (every frame transport decodes /// `Command::Eviction`) instead of a decode error or a timeout. #[allow(clippy::future_not_send)] -pub(crate) async fn surface_login_failure( +pub async fn surface_login_failure( shard: &Rc>, transport_client_id: u128, request_header: &RoutedRequestHeader, diff --git a/core/server/src/bootstrap.rs b/core/server/src/bootstrap.rs index 4320c3bc5e..186325ff47 100644 --- a/core/server/src/bootstrap.rs +++ b/core/server/src/bootstrap.rs @@ -33,14 +33,19 @@ use crate::server_error::{ PartitionRecoveryRefusal, ServerError, ShardJoinFailure, ShardJoinFailureKind, }; use crate::session_manager::SessionManager; +use crate::shard_allocator::{ShardAllocator, ShardInfo}; +use crate::shell::{ + ServerMetadata, ServerMetadataBundle, ServerMuxStateMachine, ServerShard, ShellBus, + ShellHandlers, ShellShardHandle, consensus_timers, repair_retry_ticks, +}; use compio::runtime::ResumeUnwind; use configs::server::{ServerConfig, ServerSystemConfig}; use configs::sharding::{ INBOX_CAPACITY_MAX, SHUTDOWN_DRAIN_TIMEOUT_MAX, SHUTDOWN_POLL_INTERVAL_MAX, }; use consensus::{ - ClientTable, ConsensusTimers, JoinMode, LocalPipeline, MetadataHandle, PartitionsHandle, - PipelineEntry, Sequencer, VsrConsensus, VsrRestore, + ClientTable, JoinMode, LocalPipeline, MetadataHandle, PartitionsHandle, PipelineEntry, + Sequencer, VsrConsensus, VsrRestore, }; // `try_send` / `try_recv` resolve through these traits on `MAsyncTx` / // `MAsyncRx`; the metadata-handoff loops below depend on the @@ -53,8 +58,7 @@ use iggy_common::defaults::{ MIN_PASSWORD_LENGTH, MIN_USERNAME_LENGTH, }; use iggy_common::{ - Aes256GcmEncryptor, EncryptorKind, IggyByteSize, IggyError, PartitionStats, - TopicRuntimeOptions, variadic, + Aes256GcmEncryptor, EncryptorKind, IggyByteSize, IggyError, PartitionStats, TopicRuntimeOptions, }; use journal::prepare_journal::PrepareJournal; use journal::superblock::{PingPongSuperblock, SuperblockStore}; @@ -65,7 +69,7 @@ use message_bus::installer::conn_info::{ClientConnMeta, ClientTransportKind}; use message_bus::replica::auth::{self, ReplicaAuth}; use message_bus::replica::handshake::{ReplicaHandshakeCtx, ReplicaTlsCtx}; use message_bus::replica::io as replica_io; -use message_bus::replica::listener::{self as replica_listener, MessageHandler}; +use message_bus::replica::listener::{self as replica_listener}; use message_bus::transports::quic::server_config_with_cert; use message_bus::transports::tls::{ AcceptAnyServerCert, REPLICA_ALPN, TlsServerCredentials, install_default_crypto_provider, @@ -76,15 +80,11 @@ use message_bus::{ AcceptedWsClientFn, AcceptedWssClientFn, ConnectionInstaller, DialedReplicaFn, IggyMessageBus, MAX_INFLIGHT_REPLICA_HANDSHAKES, MessageBus, ReplicaOwnerTable, connector, }; -use metadata::IggyMetadata; -use metadata::MuxStateMachine; use metadata::ReplicaIdentity; use metadata::impls::metadata::{IggySnapshot, StreamsFrontend}; use metadata::impls::recovery::recover; -use metadata::stm::mux::WithFactory; use metadata::stm::snapshot::Snapshot; -use metadata::stm::stream::{Partition, Streams}; -use metadata::stm::user::Users; +use metadata::stm::stream::Partition; use partitions::{ FatalCommit, IggyIndexWriter, IggyPartition, IggyPartitions, MessagesWriter, PartitionsConfig, }; @@ -100,18 +100,16 @@ use shard::builder::IggyShardBuilder; use shard::metrics::{ShardMetrics, frame_drop_reason, frame_drop_variant}; use shard::shards_table::{PapayaShardsTable, ShardsTable, calculate_shard_assignment}; use shard::{ - CoordinatorConfig, IggyShard, LifecycleFrame, ListClientsHandler, MetadataSubmitHandler, - PartitionConsensusConfig, PartitionReadHandler, Receiver as ShardReceiver, ShardFrame, - ShardIdentity, TaggedSender, channel, shard_mesh_channels, + CoordinatorConfig, LifecycleFrame, PartitionConsensusConfig, Receiver as ShardReceiver, + ShardFrame, ShardIdentity, TaggedSender, channel, shard_mesh_channels, }; -use shard_allocator::{ShardAllocator, ShardInfo}; use std::cell::RefCell; use std::collections::HashMap; use std::env; use std::net::{IpAddr, SocketAddr}; use std::panic; use std::path::{Path, PathBuf}; -use std::rc::{Rc, Weak}; +use std::rc::Rc; use std::sync::Arc; use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; use std::thread; @@ -123,76 +121,6 @@ const SHARD_REPLICA_ID: u8 = 0; pub const IGGY_ROOT_USERNAME_ENV: &str = "IGGY_ROOT_USERNAME"; pub const IGGY_ROOT_PASSWORD_ENV: &str = "IGGY_ROOT_PASSWORD"; -type ServerMuxStateMachine = MuxStateMachine; - -/// Cross-thread bundle carrying one `ReadHandleFactory` per metadata -/// state. Shard 0 mints one after `recover()` and broadcasts a clone to -/// every peer shard; each peer rebuilds a reader-mode -/// [`ServerMuxStateMachine`] on its own runtime, skipping the WAL. -type ServerMetadataBundle = ::Bundle; - -pub(crate) type ServerMetadata = IggyMetadata< - VsrConsensus>, - PrepareJournal, - IggySnapshot, - ServerMuxStateMachine, ->; - -/// The shard type the dispatch layer is generic over. -/// -/// `B`/`MJ`/`S`/`SB` are free; the metadata state machine (`M`) and shards -/// table (`T`) are pinned, being identical in production and the simulator. -/// Production instantiates it as [`ServerShard`], defaulting `SB` to the -/// on-disk [`PingPongSuperblock`]; the simulator supplies its own -/// `B`/`MJ`/`S`/`SB`. -pub type ShellShard = - IggyShard; - -/// Late-bound self-reference the deferred dispatch handlers upgrade per frame. -pub type ShellShardHandle = - Rc>>>>; - -/// Bus bounds the dispatch/pump path needs (matches `run_message_pump`). -/// Blanket-impl'd, so it is only shorthand for the four underlying bounds. -pub trait ShellBus: MessageBus + ConnectionInstaller + Clone + 'static {} -impl ShellBus for B {} - -/// The five dispatch handlers a shard is built with, plus the -/// [`SessionManager`] the request-plane pair shares. -/// -/// Both production (`build_shard_for_thread`) and the simulator's shell -/// mode construct these through [`wire_shell_handlers`], so the request -/// plane is wired one way. The simulator's shell-off fast path uses -/// [`ShellHandlers::noop`] instead. -pub struct ShellHandlers { - pub on_replica_message: MessageHandler, - pub on_client_request: RequestHandler, - pub on_metadata_submit: MetadataSubmitHandler, - pub on_list_clients: ListClientsHandler, - pub on_partition_read: PartitionReadHandler, - /// Bound by the client-request handler, read by the get-clients - /// handler; the caller keeps it to reach locally-homed sessions. - pub sessions: Rc>, -} - -impl ShellHandlers { - /// Inert handlers for the shell-off fast path: every callback is a - /// no-op over an empty [`SessionManager`]. Behaviorally identical to - /// hand-written no-op closures, so a caller can keep one destructure - /// site across both toggle states. - #[must_use] - pub fn noop() -> Self { - Self { - on_replica_message: Rc::new(|_, _| {}), - on_client_request: Rc::new(|_, _| {}), - on_metadata_submit: Rc::new(|_| {}), - on_list_clients: Rc::new(|_| {}), - on_partition_read: Rc::new(|_, _, _| {}), - sessions: Rc::new(RefCell::new(SessionManager::new())), - } - } -} - /// Build the deferred dispatch handlers for `shard_handle` against `bus`. /// /// They share one fresh [`SessionManager`]. The caller must set the weak @@ -228,8 +156,6 @@ where } } -pub type ServerShard = ShellShard, PrepareJournal, IggySnapshot>; - /// Result of a multi-shard bootstrap. /// /// Carries the cross-thread shutdown flag and one OS-thread `JoinHandle` @@ -727,7 +653,7 @@ fn validate_sharding_runtime_knobs( /// `(senders, inboxes)` channels, and spawns one OS thread per shard. /// /// Each thread pins itself (`nix::sched::sched_setaffinity` on Linux via -/// [`ShardInfo::bind_cpu`]), binds memory to its NUMA node when +/// `ShardInfo::bind_cpu`), binds memory to its NUMA node when /// configured, builds a fresh `compio::runtime::Runtime` (one /// `io_uring` instance per shard), and runs `shard_main` inside it. /// @@ -2255,13 +2181,6 @@ const _: () = const _: () = assert!(consensus::DVC_HEADERS_MAX == iggy_binary_protocol::consensus::DVC_HEADERS_MAX); const _: () = assert!(consensus::DVC_HEADERS_MAX == u128::BITS as usize); -/// Convert a consensus-timer interval to whole ticks, floored at one tick so a -/// sub-tick value still fires and saturated on overflow. -fn duration_to_ticks(interval: Duration) -> u64 { - let ticks = interval.as_millis() / shard::CONSENSUS_TICK_INTERVAL.as_millis(); - u64::try_from(ticks.max(1)).unwrap_or(u64::MAX) -} - /// `[cluster] superblock_wedged_fatal_timeout` as a consecutive-failure count. /// Retries pin at the backoff cap after warmup, so the window divided by /// [`journal::superblock::SUPERBLOCK_RETRY_BACKOFF_MAX_MICROS`] bounds how @@ -2284,14 +2203,6 @@ fn superblock_window_to_failures(window: Duration) -> u64 { u64::try_from((window.as_micros() / cap_micros).max(1)).unwrap_or(u64::MAX) } -/// `[cluster] heartbeat_timeout` in consensus ticks. Every consensus group -/// (metadata and per-partition planes alike) gets the same window: the failure -/// it guards against - a primary that stopped heartbeating - is host-level, not -/// per-plane. -pub(crate) fn cluster_heartbeat_ticks(config: &ServerConfig) -> u64 { - duration_to_ticks(config.cluster.heartbeat_timeout.get_duration()) -} - /// Floor for the post-restart read-recovery deadline (see /// [`recovery_barrier_deadline`]). At and below the 5s default heartbeat the /// worst-case recovery is dominated by the heartbeat-independent term - the @@ -2327,76 +2238,6 @@ pub(crate) fn recovery_barrier_deadline( .max(RECOVERY_BARRIER_DEADLINE_FLOOR) } -/// `[cluster] commit_broadcast_interval` in consensus ticks: how often the -/// primary broadcasts its commit point, the cluster's liveness feed. Applied -/// to every consensus group, matching `cluster_heartbeat_ticks`. -pub(crate) fn commit_broadcast_ticks(config: &ServerConfig) -> u64 { - duration_to_ticks(config.cluster.commit_broadcast_interval.get_duration()) -} - -/// `[cluster] prepare_retransmit_interval` in consensus ticks: how often the -/// primary retransmits un-acked prepares. Applied to every consensus group, -/// matching `cluster_heartbeat_ticks`. -pub(crate) fn prepare_retransmit_ticks(config: &ServerConfig) -> u64 { - duration_to_ticks(config.cluster.prepare_retransmit_interval.get_duration()) -} - -/// `[cluster] view_change_retransmit_interval` in consensus ticks: how often a -/// replica retransmits its `StartViewChange` / `DoViewChange` during a view -/// change. Applied to every consensus group, matching `cluster_heartbeat_ticks`. -pub(crate) fn view_change_retransmit_ticks(config: &ServerConfig) -> u64 { - duration_to_ticks( - config - .cluster - .view_change_retransmit_interval - .get_duration(), - ) -} - -/// `[cluster] view_change_status_timeout` in consensus ticks: the stalled -/// view-change backstop before escalating to a fresh election. Applied to every -/// consensus group, matching `cluster_heartbeat_ticks`. -pub(crate) fn view_change_status_ticks(config: &ServerConfig) -> u64 { - duration_to_ticks(config.cluster.view_change_status_timeout.get_duration()) -} - -/// `[cluster] request_start_view_retransmit_interval` in consensus ticks: how -/// often a recovering or view-change backup re-requests the current `StartView`. -/// Applied to every consensus group, matching `cluster_heartbeat_ticks`. -pub(crate) fn request_start_view_ticks(config: &ServerConfig) -> u64 { - duration_to_ticks( - config - .cluster - .request_start_view_retransmit_interval - .get_duration(), - ) -} - -/// The full `[cluster]` timer set every consensus group boots with, built -/// once so the planes cannot diverge in what they apply. -pub(crate) fn consensus_timers(config: &ServerConfig) -> ConsensusTimers { - ConsensusTimers { - normal_heartbeat_ticks: cluster_heartbeat_ticks(config), - commit_message_ticks: commit_broadcast_ticks(config), - prepare_ticks: prepare_retransmit_ticks(config), - view_change_retransmit_ticks: view_change_retransmit_ticks(config), - view_change_status_ticks: view_change_status_ticks(config), - request_start_view_ticks: request_start_view_ticks(config), - probe_attempts_max: config.cluster.view_probe_attempts_max, - } -} - -/// `[cluster] repair_retry_interval` in consensus ticks: how long a stalled -/// journal-repair stream waits before re-requesting its window. Both planes' -/// repair loops share it, so it is applied once per shard (not per consensus -/// group). Clamped to `u32`, the width of the session idle-tick counter. -pub(crate) fn repair_retry_ticks(config: &ServerConfig) -> u32 { - u32::try_from(duration_to_ticks( - config.cluster.repair_retry_interval.get_duration(), - )) - .unwrap_or(u32::MAX) -} - /// Shard 0's half of a metadata recovery: everything [`recover`] produced except the /// state machine, which every shard receives through the factory bundle. /// diff --git a/core/server/src/consumer_group.rs b/core/server/src/consumer_group.rs index 00a6df5580..d4b77eb954 100644 --- a/core/server/src/consumer_group.rs +++ b/core/server/src/consumer_group.rs @@ -25,8 +25,8 @@ //! primary enriches the op here before replication, mirroring the PAT mint //! in [`crate::pat`] and the password hash in [`crate::users`]. -use crate::bootstrap::{ShellBus, ShellShard}; use crate::responses::resolve_partition_namespace; +use crate::shell::{ShellBus, ShellShard}; use crate::wire::{request_body, rewrite_request_body}; use consensus::MetadataHandle; use iggy_binary_protocol::PrepareHeader; @@ -60,7 +60,7 @@ use std::rc::Rc; /// (via the partition-read mesh), so the cooperative rebalance pending-revokes /// only those and hands off never-polled/drained partitions synchronously at /// join. Every other operation passes through. -pub(crate) async fn maybe_rewrite_consumer_group_request( +pub async fn maybe_rewrite_consumer_group_request( shard: &Rc>, request: Message, ) -> Result, IggyError> @@ -205,7 +205,7 @@ where /// purge agree, and a re-created group (new id) never inherits a stale offset. /// Individual-consumer ops and every other operation pass through untouched. #[allow(clippy::cast_possible_truncation)] -pub(crate) fn maybe_rewrite_consumer_offset_request( +pub fn maybe_rewrite_consumer_offset_request( shard: &Rc>, request: Message, ) -> Result, IggyError> diff --git a/core/server/src/dispatch.rs b/core/server/src/dispatch.rs index 0a494f524b..784b86bdf6 100644 --- a/core/server/src/dispatch.rs +++ b/core/server/src/dispatch.rs @@ -28,7 +28,6 @@ use crate::auth::{ complete_login_register, surface_login_failure, verify_login_credentials, verify_pat_credentials, }; -use crate::bootstrap::{ShellBus, ShellShard, ShellShardHandle}; use crate::cluster_meta::ClusterRoster; use crate::consumer_group::{ maybe_rewrite_consumer_group_request, maybe_rewrite_consumer_offset_request, @@ -48,6 +47,7 @@ use crate::responses::{ }; use crate::segment_cleaner::UNENFORCEABLE_TOPIC_SIZE_WARN; use crate::session_manager::SessionManager; +use crate::shell::{ShellBus, ShellShard, ShellShardHandle}; use crate::snapshot; use crate::users::maybe_rewrite_user_password_request; use crate::wire::{request_body, usize_to_u32, verify_request_checksum}; @@ -130,10 +130,10 @@ use std::sync::Arc; use std::time::Duration; use tracing::{debug, warn}; -pub(crate) type ClientRequestQueues = Rc>>>>; -pub(crate) type ActiveClientRequests = Rc>>; +pub type ClientRequestQueues = Rc>>>>; +pub type ActiveClientRequests = Rc>>; -pub(crate) fn make_client_request_handler( +pub fn make_client_request_handler( shard: &Rc>, sessions: &Rc>, system_config: Arc, @@ -181,9 +181,7 @@ where /// its `SessionManager` and push them back over the reply sender. The /// aggregation across all shards happens in /// [`shard::IggyShard::list_all_clients`]. -pub(crate) fn make_list_clients_handler( - sessions: &Rc>, -) -> ListClientsHandler { +pub fn make_list_clients_handler(sessions: &Rc>) -> ListClientsHandler { let sessions = Rc::clone(sessions); Rc::new(move |reply| { let clients: Vec = sessions.borrow().iter_clients().collect(); @@ -198,7 +196,7 @@ pub(crate) fn make_list_clients_handler( /// against the local partitions plane and push the result back over the /// carried reply sender. The requesting shard bounds the wait with a /// timeout, so a dropped reply degrades to a client-visible read failure. -pub(crate) fn make_partition_read_handler( +pub fn make_partition_read_handler( shard_handle: &ShellShardHandle, ) -> PartitionReadHandler where @@ -451,7 +449,7 @@ fn build_auto_commit_request( ) } -pub(crate) fn make_deferred_replica_message_handler( +pub fn make_deferred_replica_message_handler( shard_handle: &ShellShardHandle, ) -> MessageHandler where @@ -469,7 +467,7 @@ where }) } -pub(crate) fn make_deferred_client_request_handler( +pub fn make_deferred_client_request_handler( bus: &B, shard_handle: &ShellShardHandle, sessions: &Rc>, @@ -538,7 +536,7 @@ where /// proposal. Spawns a task so the awaiting peer is woken once the op /// commits. Submit failures are returned verbatim so the peer can preserve /// unknown-outcome retry semantics. -pub(crate) fn make_metadata_submit_handler( +pub fn make_metadata_submit_handler( shard_handle: &ShellShardHandle, ) -> shard::MetadataSubmitHandler where @@ -790,7 +788,7 @@ fn pop_next_client_request( /// Zero passes here because a zero-partition TOPIC is legal (legacy /// `create_topic` admits `0..=MAX`); the add/remove requests reject it in /// [`validate_partitions_change_count`]. -pub(crate) const fn validate_partitions_count(partitions_count: u32) -> Result<(), IggyError> { +pub const fn validate_partitions_count(partitions_count: u32) -> Result<(), IggyError> { if partitions_count > MAX_PARTITIONS_PER_REQUEST { return Err(IggyError::TooManyPartitions); } @@ -803,9 +801,7 @@ pub(crate) const fn validate_partitions_count(partitions_count: u32) -> Result<( /// force every shard through a rebalance pass. Legacy rejects it with /// `TooManyPartitions` in both handlers (`1..=MAX` on create, `== 0` on /// delete), so the code matches rather than inventing a new one. -pub(crate) const fn validate_partitions_change_count( - partitions_count: u32, -) -> Result<(), IggyError> { +pub const fn validate_partitions_change_count(partitions_count: u32) -> Result<(), IggyError> { if partitions_count == 0 { return Err(IggyError::TooManyPartitions); } @@ -820,7 +816,7 @@ pub(crate) const fn validate_partitions_change_count( /// `segment_size_bytes` is the topic's RESOLVED segment size (explicit /// option, else this node's default), so a per-topic segment above the /// global default still floors the topic cap. -pub(crate) fn validate_topic_bounds( +pub fn validate_topic_bounds( partitions_count: u32, max_topic_size: MaxTopicSize, segment_size_bytes: u64, @@ -832,7 +828,7 @@ pub(crate) fn validate_topic_bounds( /// A topic cap below one segment can never be enforced: the first segment /// already exceeds it. Split out of [`validate_topic_bounds`] because update /// admission checks the cap without a partitions count to check. -pub(crate) fn validate_topic_size_floor( +pub fn validate_topic_size_floor( max_topic_size: MaxTopicSize, segment_size_bytes: u64, ) -> Result<(), IggyError> { @@ -858,7 +854,7 @@ pub(crate) fn validate_topic_size_floor( /// /// Warns rather than rejects: which caps are accepted is client-visible wire /// behavior, and tightening it would break topics that already exist. -pub(crate) fn warn_unenforceable_topic_size( +pub fn warn_unenforceable_topic_size( max_topic_size: MaxTopicSize, segment_size_bytes: u64, max_message_size_bytes: usize, @@ -888,7 +884,7 @@ pub(crate) fn warn_unenforceable_topic_size( /// partition shrinks the share: a cap that cleared the floor when the topic was /// created can stop clearing it here. The request carries only the delta, so /// the stored cap, segment size and current partition count come from metadata. -pub(crate) fn warn_unenforceable_topic_size_on_partition_add( +pub fn warn_unenforceable_topic_size_on_partition_add( streams: &Streams, stream_id: &WireIdentifier, topic_id: &WireIdentifier, @@ -920,7 +916,7 @@ pub(crate) fn warn_unenforceable_topic_size_on_partition_add( /// keys are rejected rather than skipped: a silently ignored knob would hand /// the client server defaults without it ever learning. Streams and users /// have no catalog keys yet, so `known` is empty for both until one lands. -pub(crate) fn validate_option_keys(options: &WireOptions, known: &[&str]) -> Result<(), IggyError> { +pub fn validate_option_keys(options: &WireOptions, known: &[&str]) -> Result<(), IggyError> { for entry in options { // Wire validation already enforced UTF-8 string keys. let key = String::from_utf8_lossy(entry.key); @@ -1468,7 +1464,7 @@ async fn handle_get_me( /// `vsr_client_id` keys the consumer-group offset fence (the member id), /// not the transport id stamped into the partition-op header. #[allow(clippy::future_not_send)] -pub(crate) async fn dispatch_partition_request( +pub async fn dispatch_partition_request( shard: &Rc>, request: Message, vsr_client_id: u128, @@ -2007,7 +2003,7 @@ async fn send_unauthenticated_eviction( /// groups + rebalances via the replicated `Logout`) and sends a session- /// terminal `Eviction(StaleClient)` so the client fails fast and can reconnect. #[allow(clippy::future_not_send)] -pub(crate) async fn run_heartbeat_verifier( +pub async fn run_heartbeat_verifier( shard: Rc>, sessions: Rc>, interval: std::time::Duration, @@ -2474,14 +2470,14 @@ fn empty_polled_messages_body(partition_id: u32) -> Bytes { Bytes::from(body) } -pub(crate) type DecodedPollRequest = (IggyNamespace, u32, PollingConsumer, PollingArgs); +pub type DecodedPollRequest = (IggyNamespace, u32, PollingConsumer, PollingArgs); /// Resolve a decoded poll request into its owning-shard read: namespace, /// partition, polling consumer, and args. Shared by the TCP dispatch (client /// id = the connection's bound VSR client) and the HTTP route (client id 0, /// which fences group polls closed). #[allow(clippy::cast_possible_truncation)] -pub(crate) fn resolve_poll_request( +pub fn resolve_poll_request( shard: &Rc>, wire: &PollMessagesRequest, client_id: u128, @@ -2548,7 +2544,7 @@ where /// namespace, partition, and polling consumer. Shared by the TCP dispatch and /// the HTTP route; needs no client id because offset reads are not fenced /// (any client may read a group's offset, member or not). -pub(crate) fn resolve_consumer_offset_request( +pub fn resolve_consumer_offset_request( shard: &Rc>, wire: &GetConsumerOffsetRequest, ) -> Result<(IggyNamespace, u32, PollingConsumer), IggyError> @@ -3082,7 +3078,7 @@ fn build_forward_logout_result_message( /// reply (shard-0 inbox full / shutdown) maps to a transient `Canceled`, which /// the caller wraps so the SDK replays. #[allow(clippy::future_not_send)] -pub(crate) async fn submit_register_on_owner( +pub async fn submit_register_on_owner( shard: &Rc>, vsr_client_id: u128, user_id: u32, @@ -3112,7 +3108,7 @@ where /// Logout counterpart of [`submit_register_on_owner`]. #[allow(clippy::future_not_send)] -pub(crate) async fn submit_logout_on_owner( +pub async fn submit_logout_on_owner( shard: &Rc>, vsr_client_id: u128, session: u64, @@ -3283,7 +3279,7 @@ async fn handle_delete_segments_request( /// the error. #[allow(clippy::future_not_send)] #[allow(clippy::cast_possible_truncation)] -pub(crate) async fn resolve_delete_segments_truncate( +pub async fn resolve_delete_segments_truncate( shard: &Rc>, template: &RoutedRequestHeader, client_id: u128, @@ -3445,7 +3441,7 @@ fn submit_disconnect_logout( /// `client` id (it's the VSR id, not the transport/home-shard-encoding id). /// `None` = transient submit failure (SDK read-timeout replays). #[allow(clippy::future_not_send)] -pub(crate) async fn submit_client_request_on_owner( +pub async fn submit_client_request_on_owner( shard: &Rc>, request: Message, ) -> Option> @@ -3759,7 +3755,7 @@ async fn handle_login_register_request( /// metadata shard and zeroed elsewhere -- the SDK only reads the reason, /// plus the protocol window on `IncompatibleProtocol`. #[allow(clippy::future_not_send)] -pub(crate) async fn send_login_eviction( +pub async fn send_login_eviction( shard: &Rc>, transport_client_id: u128, vsr_client_id: u128, @@ -3799,7 +3795,7 @@ pub(crate) async fn send_login_eviction( } } -pub(crate) fn upgrade_shard_handle( +pub fn upgrade_shard_handle( shard_handle: &ShellShardHandle, ) -> Option>> where diff --git a/core/server/src/dispatch/authz.rs b/core/server/src/dispatch/authz.rs index 5e8790e7ae..ee113e0d1e 100644 --- a/core/server/src/dispatch/authz.rs +++ b/core/server/src/dispatch/authz.rs @@ -50,10 +50,10 @@ use metadata::permissioner::Permissioner; use server_common::Message; use tracing::warn; -use crate::bootstrap::{ShellBus, ShellShard}; use crate::responses::{ build_deny_reply, current_metadata_commit, resolve_stream_id, resolve_topic_id, }; +use crate::shell::{ShellBus, ShellShard}; /// Authorize a partition-plane op on its resolved (stream, topic) for the /// acting user, returning the deny status code or `None` to proceed. The diff --git a/core/server/src/http.rs b/core/server/src/http.rs index ed670883d8..25a91eff50 100644 --- a/core/server/src/http.rs +++ b/core/server/src/http.rs @@ -36,6 +36,8 @@ mod session; mod state; mod submit; mod tls; +#[cfg(feature = "iggy-web")] +mod web; mod wire; use std::cell::{Cell, RefCell}; @@ -64,7 +66,6 @@ use send_wrapper::SendWrapper; use tower_http::cors::{AllowOrigin, CorsLayer}; use tracing::{error, info, warn}; -use crate::bootstrap::ServerShard; use crate::cluster_meta::{ClusterRoster, resolved_roster_nodes}; use crate::http::handlers::{ change_password, create_cg, create_partitions, create_pat, create_stream, create_topic, @@ -80,6 +81,7 @@ use crate::http::jwt::JwtManager; use crate::http::session::RegistrationBarrier; use crate::http::state::{HttpInner, HttpState, insert_view_header}; use crate::server_error::ServerError; +use crate::shell::ServerShard; /// Bind the shard-0 HTTP listener and spawn the `cyper-axum` serve loop as a /// background task on shard 0's compio runtime. Serves HTTPS when @@ -300,7 +302,7 @@ fn router( .route("/clients", get(get_clients)) .route("/clients/{client_id}", get(get_client)); let local = match metrics_endpoint { - Some(endpoint) => local.route(endpoint, get(metrics::get_metrics)), + Some(endpoint) => local.route(endpoint, get(handlers::get_metrics)), None => local, }; let router = Router::new() @@ -458,7 +460,7 @@ fn merge_web_ui(router: Router, web_ui: bool) -> Router { #[cfg(feature = "iggy-web")] let router = if web_ui { info!("Web UI enabled at /ui"); - router.merge(crate::web::router()) + router.merge(web::router()) } else { router }; diff --git a/core/server/src/http/forward.rs b/core/server/src/http/forward.rs index 8c1acadcdb..b0ead9be58 100644 --- a/core/server/src/http/forward.rs +++ b/core/server/src/http/forward.rs @@ -82,7 +82,7 @@ use crate::http::error::{ CustomError, error_response, gateway_timeout_response, primary_http_socket, with_retry_after, }; use crate::http::extractor::{bearer_token, resolve_credential}; -use crate::http::state::{HttpInner, VIEW_HEADER}; +use crate::http::state::{ForwardState, HttpInner, VIEW_HEADER}; use crate::server_error::ServerError; /// Marker stamped on every forwarded request. Loop guard only: a node that is @@ -135,23 +135,6 @@ const RESPONSE_CAPACITY_HINT: usize = 64 * 1024; /// (the view layer only fills the header when absent). const RELAYED_RESPONSE_HEADERS: [HeaderName; 3] = [CONTENT_TYPE, RETRY_AFTER, VIEW_HEADER]; -/// Per-node forwarding context hung off `HttpInner`: the outbound client -/// (pinned-cert TLS when the listener serves HTTPS), the scheme it dials, the -/// request-body buffer bound, and the in-flight budget. -pub(in crate::http) struct ForwardState { - /// False when no cluster-wide bearer key material exists (no configured - /// JWT secret, no cluster PSK): a forwarded bearer would 401 on the - /// primary, so the middleware passes through and followers answer with - /// the transient 503 instead. - active: bool, - client: cyper::Client, - /// Also read by the 307 redirect builder: the primary is assumed to serve - /// the same scheme as this node (uniform cluster HTTP config). - pub(in crate::http) scheme: &'static str, - body_limit: usize, - in_flight: Cell, -} - /// Build the [`ForwardState`] at listener startup. /// /// With `http.tls.enabled` the forward hop dials `https` and verifies the peer diff --git a/core/server/src/http/handlers.rs b/core/server/src/http/handlers.rs index c1172a4ccc..cdce21b5c4 100644 --- a/core/server/src/http/handlers.rs +++ b/core/server/src/http/handlers.rs @@ -129,6 +129,7 @@ use crate::http::error::{ ReadError, WriteError, }; use crate::http::extractor::{Authenticated, Identity}; +use crate::http::metrics::gauge_value; use crate::http::reads::{ authorize_data_plane, authorize_read, read_local, resolve_gate_stream, resolve_gate_topic, resolve_gate_topic_ids, resolve_gate_user, @@ -600,6 +601,72 @@ pub(in crate::http) async fn get_stats( Ok(Json(Stats::from(response))) } +/// `GET `: the metric set in prometheus text +/// exposition. Auth-only, like `/stats`: the `Identity` extractor rejects a +/// missing or invalid bearer with 401, and any authenticated user may scrape +/// (no RBAC rule guards it). Scrapers present a JWT or a raw PAT the same way +/// every read route accepts them. +/// +/// The entity gauges sample the same reads `/stats` serves: the metadata STM +/// stream and user maps plus the stats-registry rollups, whose partition-plane +/// increments are relaxed, so scraped values are approximate while writes are +/// in flight. The clients count scatter-gathers the per-shard session managers +/// exactly like `GET /clients` and turns partial when a shard misses the reply +/// deadline. +pub(in crate::http) async fn get_metrics( + State(state): State, + _identity: Identity, +) -> String { + let (streams_count, topics_count, partitions_count, segments_count, messages_count) = state + .shard + .plane + .metadata() + .mux_stm + .streams() + .read(|streams| { + let mut topics_count = 0u64; + let mut partitions_count = 0u64; + let mut segments_count = 0u64; + let mut messages_count = 0u64; + for (_, stream) in &streams.items { + topics_count = topics_count.saturating_add(stream.topics.len() as u64); + segments_count = segments_count + .saturating_add(u64::from(stream.stats.segments_count_inconsistent())); + messages_count = + messages_count.saturating_add(stream.stats.messages_count_inconsistent()); + for (_, topic) in &stream.topics { + partitions_count = + partitions_count.saturating_add(topic.partitions.len() as u64); + } + } + ( + streams.items.len() as u64, + topics_count, + partitions_count, + segments_count, + messages_count, + ) + }); + let users_count = state + .shard + .plane + .metadata() + .mux_stm + .users() + .read(|users| users.items.len() as u64); + let clients_count = SendWrapper::new(state.shard.list_all_clients()).await.len() as u64; + + let metrics = &state.metrics; + metrics.streams.set(gauge_value(streams_count)); + metrics.topics.set(gauge_value(topics_count)); + metrics.partitions.set(gauge_value(partitions_count)); + metrics.segments.set(gauge_value(segments_count)); + metrics.messages.set(gauge_value(messages_count)); + metrics.users.set(gauge_value(users_count)); + metrics.clients.set(gauge_value(clients_count)); + metrics.formatted_output() +} + /// `POST /snapshot`: collect a diagnostic archive and return it as a ZIP /// download with the same headers the legacy server sets. /// diff --git a/core/server/src/http/metrics.rs b/core/server/src/http/metrics.rs index 3fdb1b4238..6546c00c15 100644 --- a/core/server/src/http/metrics.rs +++ b/core/server/src/http/metrics.rs @@ -16,40 +16,35 @@ // under the License. //! The `[http.metrics]` scrape surface: the legacy-parity metric registry -//! (entity gauges plus the request counter), its public scrape handler, and -//! the config gate deciding whether the route is mounted. +//! (entity gauges plus the request counter) and the config gate deciding +//! whether the route is mounted. The scrape handler itself lives with the +//! other route handlers so this leaf never imports the state hub. -use axum::extract::State; use configs::http::HttpMetricsConfig; -use consensus::MetadataHandle; use iggy_common::IggyError; -use metadata::impls::metadata::StreamsFrontend; use prometheus_client::encoding::text::encode; use prometheus_client::metrics::counter::Counter; use prometheus_client::metrics::gauge::Gauge; use prometheus_client::registry::Registry; -use send_wrapper::SendWrapper; use tracing::error; -use crate::http::extractor::Identity; -use crate::http::state::HttpState; - /// The legacy server's metric set, registered under the same names and help /// texts so existing dashboards and alerts keep working unchanged. /// /// Unlike the legacy server, the entity gauges are not counted at mutation -/// sites: [`get_metrics`] samples the live state on every scrape, so a gauge -/// can never drift from the state it describes. +/// sites: the scrape handler (`http::handlers::get_metrics`) samples the live +/// state on every scrape, so a gauge can never drift from the state it +/// describes. pub(in crate::http) struct HttpMetrics { registry: Registry, http_requests: Counter, - streams: Gauge, - topics: Gauge, - partitions: Gauge, - segments: Gauge, - messages: Gauge, - users: Gauge, - clients: Gauge, + pub(in crate::http) streams: Gauge, + pub(in crate::http) topics: Gauge, + pub(in crate::http) partitions: Gauge, + pub(in crate::http) segments: Gauge, + pub(in crate::http) messages: Gauge, + pub(in crate::http) users: Gauge, + pub(in crate::http) clients: Gauge, } impl HttpMetrics { @@ -111,7 +106,7 @@ impl HttpMetrics { self.http_requests.clone() } - fn formatted_output(&self) -> String { + pub(in crate::http) fn formatted_output(&self) -> String { let mut buffer = String::new(); if let Err(error) = encode(&mut buffer, &self.registry) { error!(%error, "failed to encode metrics"); @@ -147,75 +142,9 @@ pub(in crate::http) fn validated_endpoint( Ok(Some(config.endpoint.clone())) } -/// `GET `: the metric set in prometheus text -/// exposition. Auth-only, like `/stats`: the `Identity` extractor rejects a -/// missing or invalid bearer with 401, and any authenticated user may scrape -/// (no RBAC rule guards it). Scrapers present a JWT or a raw PAT the same way -/// every read route accepts them. -/// -/// The entity gauges sample the same reads `/stats` serves: the metadata STM -/// stream and user maps plus the stats-registry rollups, whose partition-plane -/// increments are relaxed, so scraped values are approximate while writes are -/// in flight. The clients count scatter-gathers the per-shard session managers -/// exactly like `GET /clients` and turns partial when a shard misses the reply -/// deadline. -pub(in crate::http) async fn get_metrics( - State(state): State, - _identity: Identity, -) -> String { - let (streams_count, topics_count, partitions_count, segments_count, messages_count) = state - .shard - .plane - .metadata() - .mux_stm - .streams() - .read(|streams| { - let mut topics_count = 0u64; - let mut partitions_count = 0u64; - let mut segments_count = 0u64; - let mut messages_count = 0u64; - for (_, stream) in &streams.items { - topics_count = topics_count.saturating_add(stream.topics.len() as u64); - segments_count = segments_count - .saturating_add(u64::from(stream.stats.segments_count_inconsistent())); - messages_count = - messages_count.saturating_add(stream.stats.messages_count_inconsistent()); - for (_, topic) in &stream.topics { - partitions_count = - partitions_count.saturating_add(topic.partitions.len() as u64); - } - } - ( - streams.items.len() as u64, - topics_count, - partitions_count, - segments_count, - messages_count, - ) - }); - let users_count = state - .shard - .plane - .metadata() - .mux_stm - .users() - .read(|users| users.items.len() as u64); - let clients_count = SendWrapper::new(state.shard.list_all_clients()).await.len() as u64; - - let metrics = &state.metrics; - metrics.streams.set(gauge_value(streams_count)); - metrics.topics.set(gauge_value(topics_count)); - metrics.partitions.set(gauge_value(partitions_count)); - metrics.segments.set(gauge_value(segments_count)); - metrics.messages.set(gauge_value(messages_count)); - metrics.users.set(gauge_value(users_count)); - metrics.clients.set(gauge_value(clients_count)); - metrics.formatted_output() -} - /// Clamp a count into the gauge's `i64` domain; only `messages` can pass /// `i64::MAX` even in theory, the rest are bounded far below it. -fn gauge_value(count: u64) -> i64 { +pub(in crate::http) fn gauge_value(count: u64) -> i64 { i64::try_from(count).unwrap_or(i64::MAX) } diff --git a/core/server/src/http/reads.rs b/core/server/src/http/reads.rs index 08e743f472..3859b28ae3 100644 --- a/core/server/src/http/reads.rs +++ b/core/server/src/http/reads.rs @@ -19,7 +19,7 @@ //! metadata-STM read entry, and the wire/domain identifier resolvers the read //! and data-plane routes ground their scopes through. -use crate::bootstrap::ServerShard; +use crate::shell::ServerShard; use bytes::Bytes; use consensus::MetadataHandle; use iggy_binary_protocol::WireIdentifier; diff --git a/core/server/src/http/state.rs b/core/server/src/http/state.rs index 8ea1bb4867..c0c67e7dec 100644 --- a/core/server/src/http/state.rs +++ b/core/server/src/http/state.rs @@ -37,17 +37,17 @@ use send_wrapper::SendWrapper; use tokio::sync::Mutex; use tracing::warn; -use crate::bootstrap::ServerShard; use crate::cluster_meta::ClusterRoster; use crate::dispatch::submit_register_on_owner; use crate::http::error::{AuthError, ReadError, primary_redirect_location}; -use crate::http::forward::ForwardState; + use crate::http::jwt::JwtManager; use crate::http::metrics::HttpMetrics; use crate::http::session::{ BarrierEntry, FIRST_REQUEST_ID, FRESH_ENTRY_WATERMARK, HttpSession, RegistrationBarrier, forget_if_same, live_entry, sweep_expired, }; +use crate::shell::ServerShard; /// Response header carrying the current VSR view number. Stamped by /// `insert_view_header` on success and redirect responses only (never on @@ -63,6 +63,25 @@ pub(in crate::http) const VIEW_HEADER: HeaderName = HeaderName::from_static("igg /// same thread that builds this state. Never touch it off that thread. pub(in crate::http) type HttpState = SendWrapper>; +/// Per-node forwarding context hung off `HttpInner`: the outbound client +/// (pinned-cert TLS when the listener serves HTTPS), the scheme it dials, the +/// request-body buffer bound, and the in-flight budget. Built by +/// `http::forward::build_forward_state`; lives here so the state hub never +/// imports the forwarding middleware. +pub(in crate::http) struct ForwardState { + /// False when no cluster-wide bearer key material exists (no configured + /// JWT secret, no cluster PSK): a forwarded bearer would 401 on the + /// primary, so the middleware passes through and followers answer with + /// the transient 503 instead. + pub(in crate::http) active: bool, + pub(in crate::http) client: cyper::Client, + /// Also read by the 307 redirect builder: the primary is assumed to serve + /// the same scheme as this node (uniform cluster HTTP config). + pub(in crate::http) scheme: &'static str, + pub(in crate::http) body_limit: usize, + pub(in crate::http) in_flight: Cell, +} + /// Shared shard-0 HTTP state. /// /// Groups the shard handle, the JWT issuer/verifier, and the per-credential VSR diff --git a/core/server/src/http/submit.rs b/core/server/src/http/submit.rs index ca763d8368..eeaf9dc1ad 100644 --- a/core/server/src/http/submit.rs +++ b/core/server/src/http/submit.rs @@ -33,7 +33,6 @@ use metadata::impls::metadata::StreamsFrontend; use server_common::Message; use tracing::warn; -use crate::bootstrap::ServerShard; use crate::dispatch::{ dispatch_partition_request, resolve_delete_segments_truncate, submit_client_request_on_owner, submit_logout_on_owner, @@ -47,6 +46,7 @@ use crate::http::session::HttpSession; use crate::http::state::HttpInner; use crate::http::wire::build_request_message; use crate::pat::rewrite_pat_request_for_user; +use crate::shell::ServerShard; use crate::users::maybe_rewrite_user_password_request; use crate::wire::request_body; diff --git a/core/server/src/web.rs b/core/server/src/http/web.rs similarity index 100% rename from core/server/src/web.rs rename to core/server/src/http/web.rs diff --git a/core/server/src/lib.rs b/core/server/src/lib.rs index 2f5a00932f..46ce02fb79 100644 --- a/core/server/src/lib.rs +++ b/core/server/src/lib.rs @@ -30,28 +30,43 @@ static GLOBAL: MiMalloc = MiMalloc; pub const VERSION: &str = env!("CARGO_PKG_VERSION"); pub const SEMANTIC_VERSION: SemanticVersion = SemanticVersion::parse_const(VERSION); -pub mod auth; +// Visibility rule: `pub` = named external consumer. main.rs consumes +// `bootstrap`, `server_error`, and `systemd`; the simulator consumes `shell`, +// `bootstrap::wire_shell_handlers`, and (through `ShellHandlers.sessions`) +// `session_manager`. Everything else is crate-internal. + +// boot: process entry, shard threads, recovery orchestration. pub mod bootstrap; -pub(crate) mod cluster_meta; -pub mod config_writer; -pub mod consumer_group; -pub mod dispatch; +pub(crate) mod config_writer; +pub(crate) mod shard_allocator; +#[cfg(feature = "systemd")] +pub mod systemd; + +// spine: the request path - shell vocabulary, dispatch funnel, per-domain ops. +pub(crate) mod auth; +pub(crate) mod consumer_group; +pub(crate) mod dispatch; +pub(crate) mod login_register; +pub(crate) mod pat; +pub(crate) mod responses; +pub mod session_manager; +pub mod shell; +pub(crate) mod users; +pub(crate) mod wire; + +// http: the REST spine (role-leaf tree). pub(crate) mod http; -pub mod login_register; -pub(crate) mod offset_recovery; -pub mod partition_helpers; -pub mod partition_reconciler; -pub mod pat; + +// background: per-shard maintenance loops. +pub(crate) mod partition_reconciler; pub(crate) mod personal_access_token_cleaner; -pub mod responses; pub(crate) mod segment_cleaner; +pub(crate) mod snapshot; + +// support: shared plumbing and the crash-recovery readers +// (the readers move beside their writers in core/partitions later). +pub(crate) mod cluster_meta; +pub(crate) mod offset_recovery; +pub(crate) mod partition_helpers; pub(crate) mod segment_recovery; pub mod server_error; -pub mod session_manager; -pub(crate) mod snapshot; -#[cfg(feature = "systemd")] -pub mod systemd; -pub mod users; -#[cfg(feature = "iggy-web")] -pub(crate) mod web; -pub mod wire; diff --git a/core/server/src/partition_helpers.rs b/core/server/src/partition_helpers.rs index a6f7098805..91610720b7 100644 --- a/core/server/src/partition_helpers.rs +++ b/core/server/src/partition_helpers.rs @@ -432,7 +432,7 @@ pub async fn ensure_initial_segment( /// [`ServerError::PartitionSuperblockIo`] when the directory or a slot /// cannot be read; the `VersionUnknown` / `Unverifiable` / `Undecodable` / /// `IdentityMismatch` variants when a record exists but cannot be trusted. -pub(crate) async fn open_partition_superblock( +pub async fn open_partition_superblock( partition_dir: &str, identity: ReplicaIdentity, ) -> Result<(Rc, Option), ServerError> { @@ -609,7 +609,7 @@ pub async fn build_partition_fresh( // Request queue holds 2x the prepare depth (buffered requests drain as // prepares commit); depth is the per-partition `[partition]` knob. let prepare_queue_depth = config.partition.prepare_queue_depth; - let timers = crate::bootstrap::consensus_timers(config); + let timers = crate::shell::consensus_timers(config); let consensus = VsrConsensus::restored( cluster_id, self_replica_id, diff --git a/core/server/src/partition_reconciler.rs b/core/server/src/partition_reconciler.rs index dea36c9c02..5cbd5f19ad 100644 --- a/core/server/src/partition_reconciler.rs +++ b/core/server/src/partition_reconciler.rs @@ -169,8 +169,8 @@ //! discriminator, like `checkpoint_id` on every prepare //! -- `PrepareHeader.reserved` has room, but it is a `#[repr(C)]` wire change. -use crate::bootstrap::ServerShard; use crate::partition_helpers::{build_partition_fresh, delete_partitions_from_disk}; +use crate::shell::ServerShard; use ahash::{AHashMap, AHashSet}; use configs::server::ServerConfig; use consensus::{MetadataHandle, PartitionsHandle}; diff --git a/core/server/src/pat.rs b/core/server/src/pat.rs index 26c8ea6db4..dfaf3f50fc 100644 --- a/core/server/src/pat.rs +++ b/core/server/src/pat.rs @@ -37,7 +37,7 @@ use server_common::Message; use std::cell::RefCell; use std::rc::Rc; -pub(crate) fn maybe_rewrite_pat_request( +pub fn maybe_rewrite_pat_request( sessions: &Rc>, transport_client_id: u128, max_tokens_per_user: u32, @@ -65,7 +65,7 @@ pub(crate) fn maybe_rewrite_pat_request( /// form scoped to `user_id` (`only_if_expired` false); every other operation /// passes through unchanged with `None` (the HTTP write core routes all of /// its ops here, so the non-PAT arm is a no-op, not `unreachable`). -pub(crate) fn rewrite_pat_request_for_user( +pub fn rewrite_pat_request_for_user( user_id: u32, max_tokens_per_user: u32, pat_count_of: impl FnOnce(u32) -> usize, diff --git a/core/server/src/personal_access_token_cleaner.rs b/core/server/src/personal_access_token_cleaner.rs index a55093605a..61e7f585ae 100644 --- a/core/server/src/personal_access_token_cleaner.rs +++ b/core/server/src/personal_access_token_cleaner.rs @@ -22,7 +22,7 @@ //! proposes it once and every replica applies the commit. Backups never //! propose, so cleanup cannot race across the cluster. -use crate::bootstrap::ServerShard; +use crate::shell::ServerShard; use consensus::MetadataHandle; use iggy_binary_protocol::WireName; use iggy_common::IggyTimestamp; diff --git a/core/server/src/responses.rs b/core/server/src/responses.rs index 8165bdf63b..6575456ae6 100644 --- a/core/server/src/responses.rs +++ b/core/server/src/responses.rs @@ -23,9 +23,9 @@ //! `NonReplicatedResponse` dispatch shim and the partition-namespace //! resolvers. -use crate::bootstrap::{ShellBus, ShellShard}; use crate::cluster_meta::ClusterRoster; use crate::session_manager::SessionManager; +use crate::shell::{ShellBus, ShellShard}; use crate::wire::{transport_kind_to_wire, usize_to_u32}; use bytes::{Bytes, BytesMut}; use consensus::{MetadataHandle, VsrConsensus}; @@ -106,7 +106,7 @@ use system_stats::SystemProbe; /// (`user_id`, transport kind, peer address) comes from the per-shard /// [`SessionManager`]; the `consumer_groups` list is read from the /// (replicated) consumer-group STM by the connection's bound VSR client id. -pub(crate) fn build_get_personal_access_tokens_response( +pub fn build_get_personal_access_tokens_response( shard: &Rc>, sessions: &Rc>, transport_client_id: u128, @@ -143,7 +143,7 @@ where }) } -pub(crate) fn build_get_me_response( +pub fn build_get_me_response( shard: &Rc>, sessions: &Rc>, transport_client_id: u128, @@ -218,7 +218,7 @@ where /// `consumer_groups_count` is resolved from the connection's bound VSR client /// id against the replicated `Streams` STM (memberships are keyed by VSR id, not /// transport id). Connections that never bound (pre-register) count 0. -pub(crate) fn connected_client_to_response( +pub fn connected_client_to_response( shard: &Rc>, info: &ConnectedClientInfo, ) -> ClientResponse @@ -326,7 +326,7 @@ where resolve_partition_namespace(shard, stream_id, topic_id, partition_id) } -pub(crate) fn resolve_partition_request_namespace( +pub fn resolve_partition_request_namespace( shard: &Rc>, operation: Operation, body: &[u8], @@ -431,7 +431,7 @@ where ) } -pub(crate) fn resolve_partition_namespace( +pub fn resolve_partition_namespace( shard: &Rc>, stream_id: &WireIdentifier, topic_id: &WireIdentifier, @@ -494,7 +494,7 @@ fn wire_identifier_for_display(id: &WireIdentifier) -> Identifier { /// connected-client total, used only by the stats read: it comes from the async /// `ListClients` scatter-gather, which this sync builder cannot run, so both /// transport callers gather it up front (0 for every other opcode). -pub(crate) fn build_non_replicated_response( +pub fn build_non_replicated_response( shard: &Rc>, code: u32, body: &[u8], @@ -886,7 +886,7 @@ static STATS_DATA_PATH: OnceLock = OnceLock::new(); /// Capture the configured data directory for `GetStats` disk reporting. /// Idempotent: only the first call (process bootstrap) takes effect. -pub(crate) fn init_stats_data_path(path: PathBuf) { +pub fn init_stats_data_path(path: PathBuf) { let _ = STATS_DATA_PATH.set(path); } @@ -1279,7 +1279,7 @@ fn topic_not_found(stream_id: &WireIdentifier, topic_id: &WireIdentifier) -> Igg ) } -pub(crate) fn resolve_stream_id( +pub fn resolve_stream_id( streams: &metadata::stm::stream::StreamsInner, identifier: &WireIdentifier, ) -> Option { @@ -1292,7 +1292,7 @@ pub(crate) fn resolve_stream_id( } } -pub(crate) fn resolve_topic_id( +pub fn resolve_topic_id( streams: &metadata::stm::stream::StreamsInner, stream_id: usize, identifier: &WireIdentifier, @@ -1409,7 +1409,7 @@ fn partition_response( }) } -pub(crate) enum NonReplicatedResponse { +pub enum NonReplicatedResponse { Empty, Bytes(Bytes), } @@ -1431,7 +1431,7 @@ impl NonReplicatedResponse { } } -pub(crate) fn build_empty_reply( +pub fn build_empty_reply( request_header: &RoutedRequestHeader, client_id: u128, session: u64, @@ -1447,7 +1447,7 @@ pub(crate) fn build_empty_reply( /// body, nonzero status); op carries the builder's session argument like every /// reply, and only the partition primary's pre-pipeline deny pins it to 0, /// stamped through `consensus::build_deny_reply_from_request`. -pub(crate) fn build_deny_reply( +pub fn build_deny_reply( request_header: &RoutedRequestHeader, client_id: u128, session: u64, @@ -1504,7 +1504,7 @@ fn build_result_framed_reply( ) } -pub(crate) fn build_login_register_reply( +pub fn build_login_register_reply( request_header: &RoutedRequestHeader, client_id: u128, session: u64, @@ -1523,7 +1523,7 @@ pub(crate) fn build_login_register_reply( build_result_framed_reply(request_header, client_id, session, commit, &payload) } -pub(crate) fn build_reply_from_bytes( +pub fn build_reply_from_bytes( request_header: &RoutedRequestHeader, client_id: u128, session: u64, @@ -1546,7 +1546,7 @@ pub(crate) fn build_reply_from_bytes( /// reusing the confirmed commit position from the committed reply. Otherwise /// (no token, a committed business rejection, or an eviction frame) the /// committed reply passes through unchanged. -pub(crate) fn build_raw_pat_reply( +pub fn build_raw_pat_reply( request_header: &RoutedRequestHeader, committed: Message, raw_token: Option, @@ -1600,7 +1600,7 @@ pub(crate) fn build_raw_pat_reply( Ok(reply.into_generic()) } -pub(crate) fn build_reply_with_body( +pub fn build_reply_with_body( request_header: &RoutedRequestHeader, client_id: u128, session: u64, @@ -1636,7 +1636,7 @@ pub(crate) fn build_reply_with_body( reply } -pub(crate) fn current_metadata_commit(shard: &Rc>) -> u64 +pub fn current_metadata_commit(shard: &Rc>) -> u64 where B: ShellBus, MJ: JournalHandle + 'static, @@ -1663,7 +1663,7 @@ where /// decrypt point, so encrypted records are rebuilt over the plaintext. /// /// Body layout: `[partition_id:4][current_offset:8][count:4][batch records...]`. -pub(crate) fn build_polled_messages_body( +pub fn build_polled_messages_body( partition_id: u32, current_offset: u64, fragments: PollFragments, @@ -1715,7 +1715,7 @@ pub(crate) fn build_polled_messages_body( /// Build the `ConsumerOffsetResponse` reply body: /// `[partition_id:4][current_offset:8][stored_offset:8]`. -pub(crate) fn build_consumer_offset_body( +pub fn build_consumer_offset_body( partition_id: u32, current_offset: u64, stored_offset: u64, diff --git a/core/server/src/segment_cleaner.rs b/core/server/src/segment_cleaner.rs index eac91c156e..724666b24e 100644 --- a/core/server/src/segment_cleaner.rs +++ b/core/server/src/segment_cleaner.rs @@ -26,7 +26,7 @@ //! with reads. This mirrors the legacy server's `MessagesCleaner` -> //! message-pump `CleanTopicMessages` path. -use crate::bootstrap::ServerShard; +use crate::shell::ServerShard; use consensus::{MetadataHandle, PartitionsHandle}; use iggy_common::{IggyExpiry, IggyTimestamp, MaxTopicSize}; use metadata::impls::metadata::StreamsFrontend; diff --git a/core/server/src/server_error.rs b/core/server/src/server_error.rs index d5343d137b..53e84e46d6 100644 --- a/core/server/src/server_error.rs +++ b/core/server/src/server_error.rs @@ -15,11 +15,11 @@ // specific language governing permissions and limitations // under the License. +use crate::shard_allocator::ShardingError; use consensus::VsrStateError; use metadata::impls::recovery::RecoveryError; use server_common::log::LogError; use shard::ShardCtorError; -use shard_allocator::ShardingError; use std::path::PathBuf; use thiserror::Error; diff --git a/core/shard_allocator/src/lib.rs b/core/server/src/shard_allocator.rs similarity index 93% rename from core/shard_allocator/src/lib.rs rename to core/server/src/shard_allocator.rs index f0aadae26b..e71ddbfacd 100644 --- a/core/shard_allocator/src/lib.rs +++ b/core/server/src/shard_allocator.rs @@ -15,10 +15,10 @@ // specific language governing permissions and limitations // under the License. -//! `shard_allocator`: decide which CPU cores each shard lives on. +//! Decide which CPU cores each shard lives on. //! //! The server makes many shards and wants each one to run on its own -//! core so they do not fight over CPU time. This crate reads the +//! core so they do not fight over CPU time. This module reads the //! operator's choice ([`CpuAllocation`] from the config), looks at the //! real machine with `hwloc`, and hands back one [`ShardInfo`] per //! shard. On Linux it also pins each shard's thread to its core and @@ -97,7 +97,7 @@ pub struct NumaTopology { impl NumaTopology { /// Ask `hwloc` to read this machine's NUMA layout right now. /// Errors if hwloc fails or the machine reports no NUMA nodes. - pub fn detect() -> Result { + pub fn detect() -> Result { let topology = Topology::new().map_err(|e| ShardingError::TopologyDetection { msg: e.to_string() })?; @@ -113,20 +113,19 @@ impl NumaTopology { let mut logical_cores_per_node = Vec::new(); for node in numa_nodes { - let cpuset = node.cpuset().ok_or(ShardingError::TopologyDetection { - msg: "NUMA node has no CPU set".to_string(), - })?; + let cpuset = node + .cpuset() + .ok_or_else(|| ShardingError::TopologyDetection { + msg: "NUMA node has no CPU set".to_string(), + })?; let logical_cores = cpuset.weight().unwrap_or(0); let physical_cores = topology .objects_with_type(ObjectType::Core) .filter(|core| { - if let Some(core_cpuset) = core.cpuset() { - !(cpuset & core_cpuset).is_empty() - } else { - false - } + core.cpuset() + .is_some_and(|core_cpuset| !(cpuset & core_cpuset).is_empty()) }) .count(); @@ -153,15 +152,15 @@ impl NumaTopology { self.logical_cores_per_node.get(node).copied().unwrap_or(0) } - fn filter_physical_cores(&self, node_cpuset: CpuSet) -> CpuSet { + fn filter_physical_cores(&self, node_cpuset: &CpuSet) -> CpuSet { let mut physical_cpuset = CpuSet::new(); for core in self.topology.objects_with_type(ObjectType::Core) { if let Some(core_cpuset) = core.cpuset() { - let intersection = node_cpuset.clone() & core_cpuset; + let intersection = node_cpuset & core_cpuset; if !intersection.is_empty() && let Some(first_cpu) = intersection.iter_set().min() { - physical_cpuset.set(first_cpu) + physical_cpuset.set(first_cpu); } } } @@ -182,14 +181,16 @@ impl NumaTopology { available: self.node_count, })?; - let cpuset_ref = node.cpuset().ok_or(ShardingError::TopologyDetection { - msg: format!("Node {} has no CPU set", node_id), - })?; + let cpuset_ref = node + .cpuset() + .ok_or_else(|| ShardingError::TopologyDetection { + msg: format!("Node {node_id} has no CPU set"), + })?; let cpuset = SpecializedBitmapRef::to_owned(&cpuset_ref); if avoid_hyperthread { - Ok(self.filter_physical_cores(cpuset)) + Ok(self.filter_physical_cores(&cpuset)) } else { Ok(cpuset) } @@ -199,6 +200,8 @@ impl NumaTopology { /// One shard's home: which CPU cores it may run on, and which NUMA /// node its memory should sit near (`None` means do not pin memory). #[derive(Debug, Clone)] +// The only readers are the Linux-gated bind paths below. +#[cfg_attr(not(target_os = "linux"), allow(dead_code))] pub struct ShardInfo { pub cpu_set: HashSet, pub numa_node: Option, @@ -249,7 +252,7 @@ impl ShardInfo { let node = topology .objects_with_type(ObjectType::NUMANode) .nth(node_id) - .ok_or(ShardingError::InvalidNode { + .ok_or_else(|| ShardingError::InvalidNode { requested: node_id, available: topology.objects_with_type(ObjectType::NUMANode).count(), })?; @@ -324,10 +327,7 @@ pub struct ShardAllocator { impl ShardAllocator { /// Build an allocator for the given choice. Only `NumaAware` reads /// the machine topology up front; the simpler modes do not. - pub fn new( - allocation: &CpuAllocation, - pin_cores: bool, - ) -> Result { + pub fn new(allocation: &CpuAllocation, pin_cores: bool) -> Result { let topology = if matches!(allocation, CpuAllocation::NumaAware(_)) { let numa_topology = NumaTopology::detect()?; @@ -353,7 +353,7 @@ impl ShardAllocator { // quota-restricted process gets proportionally fewer shards. let available_cpus = available_parallelism() .map_err(|err| ShardingError::Other { - msg: format!("Failed to get available_parallelism: {:?}", err), + msg: format!("Failed to get available_parallelism: {err:?}"), })? .get(); @@ -431,7 +431,7 @@ impl ShardAllocator { } CpuAllocation::NumaAware(numa_config) => { let topology = self.topology.as_ref().ok_or(ShardingError::NoTopology)?; - let assignments = self.compute_numa_assignments(topology, numa_config)?; + let assignments = Self::compute_numa_assignments(topology, numa_config)?; if !self.pin_cores { tracing::warn!( @@ -446,7 +446,6 @@ impl ShardAllocator { } fn compute_numa_assignments( - &self, topology: &NumaTopology, numa: &NumaConfig, ) -> Result, ShardingError> { diff --git a/core/server/src/shell.rs b/core/server/src/shell.rs new file mode 100644 index 0000000000..bc05b40a55 --- /dev/null +++ b/core/server/src/shell.rs @@ -0,0 +1,202 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! The shell vocabulary. +//! +//! The shard/metadata type aliases the dispatch layer is generic over, the +//! [`ShellBus`] bound, the [`ShellHandlers`] slot struct, and the +//! `[cluster]` timer-to-tick translation every consensus group boots with. +//! Everything here is type- and config-level; construction (wiring the +//! handlers against a live bus) stays in [`crate::bootstrap`]. + +use crate::session_manager::SessionManager; +use configs::server::ServerConfig; +use consensus::{ConsensusTimers, VsrConsensus}; +use iggy_common::variadic; +use journal::prepare_journal::PrepareJournal; +use journal::superblock::PingPongSuperblock; +use message_bus::client_listener::RequestHandler; +use message_bus::replica::listener::MessageHandler; +use message_bus::{ConnectionInstaller, IggyMessageBus, MessageBus}; +use metadata::IggyMetadata; +use metadata::MuxStateMachine; +use metadata::impls::metadata::IggySnapshot; +use metadata::stm::mux::WithFactory; +use metadata::stm::stream::Streams; +use metadata::stm::user::Users; +use shard::shards_table::PapayaShardsTable; +use shard::{IggyShard, ListClientsHandler, MetadataSubmitHandler, PartitionReadHandler}; +use std::cell::RefCell; +use std::rc::{Rc, Weak}; +use std::time::Duration; + +pub(crate) type ServerMuxStateMachine = MuxStateMachine; + +/// Cross-thread bundle carrying one `ReadHandleFactory` per metadata +/// state. Shard 0 mints one after `recover()` and broadcasts a clone to +/// every peer shard; each peer rebuilds a reader-mode +/// [`ServerMuxStateMachine`] on its own runtime, skipping the WAL. +pub(crate) type ServerMetadataBundle = ::Bundle; + +pub(crate) type ServerMetadata = IggyMetadata< + VsrConsensus>, + PrepareJournal, + IggySnapshot, + ServerMuxStateMachine, +>; + +/// The shard type the dispatch layer is generic over. +/// +/// `B`/`MJ`/`S`/`SB` are free; the metadata state machine (`M`) and shards +/// table (`T`) are pinned, being identical in production and the simulator. +/// Production instantiates it as [`ServerShard`], defaulting `SB` to the +/// on-disk [`PingPongSuperblock`]; the simulator supplies its own +/// `B`/`MJ`/`S`/`SB`. +pub type ShellShard = + IggyShard; + +/// Late-bound self-reference the deferred dispatch handlers upgrade per frame. +pub type ShellShardHandle = + Rc>>>>; + +/// Bus bounds the dispatch/pump path needs (matches `run_message_pump`). +/// Blanket-impl'd, so it is only shorthand for the four underlying bounds. +pub trait ShellBus: MessageBus + ConnectionInstaller + Clone + 'static {} +impl ShellBus for B {} + +/// The five dispatch handlers a shard is built with, plus the +/// [`SessionManager`] the request-plane pair shares. +/// +/// Both production (`build_shard_for_thread`) and the simulator's shell +/// mode construct these through [`crate::bootstrap::wire_shell_handlers`], +/// so the request plane is wired one way. The simulator's shell-off fast +/// path uses [`ShellHandlers::noop`] instead. +pub struct ShellHandlers { + pub on_replica_message: MessageHandler, + pub on_client_request: RequestHandler, + pub on_metadata_submit: MetadataSubmitHandler, + pub on_list_clients: ListClientsHandler, + pub on_partition_read: PartitionReadHandler, + /// Bound by the client-request handler, read by the get-clients + /// handler; the caller keeps it to reach locally-homed sessions. + pub sessions: Rc>, +} + +impl ShellHandlers { + /// Inert handlers for the shell-off fast path: every callback is a + /// no-op over an empty [`SessionManager`]. Behaviorally identical to + /// hand-written no-op closures, so a caller can keep one destructure + /// site across both toggle states. + #[must_use] + pub fn noop() -> Self { + Self { + on_replica_message: Rc::new(|_, _| {}), + on_client_request: Rc::new(|_, _| {}), + on_metadata_submit: Rc::new(|_| {}), + on_list_clients: Rc::new(|_| {}), + on_partition_read: Rc::new(|_, _, _| {}), + sessions: Rc::new(RefCell::new(SessionManager::new())), + } + } +} + +pub type ServerShard = ShellShard, PrepareJournal, IggySnapshot>; + +/// Convert a consensus-timer interval to whole ticks, floored at one tick so a +/// sub-tick value still fires and saturated on overflow. +fn duration_to_ticks(interval: Duration) -> u64 { + let ticks = interval.as_millis() / shard::CONSENSUS_TICK_INTERVAL.as_millis(); + u64::try_from(ticks.max(1)).unwrap_or(u64::MAX) +} + +/// `[cluster] heartbeat_timeout` in consensus ticks. Every consensus group +/// (metadata and per-partition planes alike) gets the same window: the failure +/// it guards against - a primary that stopped heartbeating - is host-level, not +/// per-plane. +pub(crate) fn cluster_heartbeat_ticks(config: &ServerConfig) -> u64 { + duration_to_ticks(config.cluster.heartbeat_timeout.get_duration()) +} + +/// `[cluster] commit_broadcast_interval` in consensus ticks: how often the +/// primary broadcasts its commit point, the cluster's liveness feed. Applied +/// to every consensus group, matching `cluster_heartbeat_ticks`. +pub(crate) fn commit_broadcast_ticks(config: &ServerConfig) -> u64 { + duration_to_ticks(config.cluster.commit_broadcast_interval.get_duration()) +} + +/// `[cluster] prepare_retransmit_interval` in consensus ticks: how often the +/// primary retransmits un-acked prepares. Applied to every consensus group, +/// matching `cluster_heartbeat_ticks`. +pub(crate) fn prepare_retransmit_ticks(config: &ServerConfig) -> u64 { + duration_to_ticks(config.cluster.prepare_retransmit_interval.get_duration()) +} + +/// `[cluster] view_change_retransmit_interval` in consensus ticks: how often a +/// replica retransmits its `StartViewChange` / `DoViewChange` during a view +/// change. Applied to every consensus group, matching `cluster_heartbeat_ticks`. +pub(crate) fn view_change_retransmit_ticks(config: &ServerConfig) -> u64 { + duration_to_ticks( + config + .cluster + .view_change_retransmit_interval + .get_duration(), + ) +} + +/// `[cluster] view_change_status_timeout` in consensus ticks: the stalled +/// view-change backstop before escalating to a fresh election. Applied to every +/// consensus group, matching `cluster_heartbeat_ticks`. +pub(crate) fn view_change_status_ticks(config: &ServerConfig) -> u64 { + duration_to_ticks(config.cluster.view_change_status_timeout.get_duration()) +} + +/// `[cluster] request_start_view_retransmit_interval` in consensus ticks: how +/// often a recovering or view-change backup re-requests the current `StartView`. +/// Applied to every consensus group, matching `cluster_heartbeat_ticks`. +pub(crate) fn request_start_view_ticks(config: &ServerConfig) -> u64 { + duration_to_ticks( + config + .cluster + .request_start_view_retransmit_interval + .get_duration(), + ) +} + +/// The full `[cluster]` timer set every consensus group boots with, built +/// once so the planes cannot diverge in what they apply. +pub(crate) fn consensus_timers(config: &ServerConfig) -> ConsensusTimers { + ConsensusTimers { + normal_heartbeat_ticks: cluster_heartbeat_ticks(config), + commit_message_ticks: commit_broadcast_ticks(config), + prepare_ticks: prepare_retransmit_ticks(config), + view_change_retransmit_ticks: view_change_retransmit_ticks(config), + view_change_status_ticks: view_change_status_ticks(config), + request_start_view_ticks: request_start_view_ticks(config), + probe_attempts_max: config.cluster.view_probe_attempts_max, + } +} + +/// `[cluster] repair_retry_interval` in consensus ticks: how long a stalled +/// journal-repair stream waits before re-requesting its window. Both planes' +/// repair loops share it, so it is applied once per shard (not per consensus +/// group). Clamped to `u32`, the width of the session idle-tick counter. +pub(crate) fn repair_retry_ticks(config: &ServerConfig) -> u32 { + u32::try_from(duration_to_ticks( + config.cluster.repair_retry_interval.get_duration(), + )) + .unwrap_or(u32::MAX) +} diff --git a/core/server/src/users.rs b/core/server/src/users.rs index 129bbfca8f..3b146cb888 100644 --- a/core/server/src/users.rs +++ b/core/server/src/users.rs @@ -35,7 +35,7 @@ //! the caller's request sequence stays contiguous; see //! `verify_and_rewrite_change_password`. -use crate::bootstrap::{ShellBus, ShellShard}; +use crate::shell::{ShellBus, ShellShard}; use crate::wire::{request_body, rewrite_request_body}; use bytes::Bytes; use consensus::MetadataHandle; @@ -58,7 +58,7 @@ use std::rc::Rc; /// rejection (see [`verify_and_rewrite_change_password`]). Every other operation /// passes through unchanged. Returns [`IggyError::InvalidCommand`] only on an /// undecodable password body. -pub(crate) fn maybe_rewrite_user_password_request( +pub fn maybe_rewrite_user_password_request( shard: &Rc>, request: Message, ) -> Result, IggyError> diff --git a/core/server/src/wire.rs b/core/server/src/wire.rs index d869a21551..737d724407 100644 --- a/core/server/src/wire.rs +++ b/core/server/src/wire.rs @@ -26,7 +26,7 @@ use iggy_common::IggyError; use message_bus::installer::conn_info::ClientTransportKind; use server_common::Message; -pub(crate) fn request_body(request: &Message) -> &[u8] { +pub fn request_body(request: &Message) -> &[u8] { &request.as_slice()[std::mem::size_of::()..request.header().size as usize] } @@ -38,9 +38,7 @@ pub(crate) fn request_body(request: &Message) -> &[u8] { /// /// # Errors /// [`IggyError::InvalidFormat`] when the stamp disagrees with the body. -pub(crate) fn verify_request_checksum( - request: &Message, -) -> Result<(), IggyError> { +pub fn verify_request_checksum(request: &Message) -> Result<(), IggyError> { let stamped = request.header().request_checksum; if stamped == 0 || u128::from(iggy_common::calculate_checksum(request_body(request))) == stamped { @@ -53,7 +51,7 @@ pub(crate) fn verify_request_checksum( /// (`1=TCP, 2=QUIC, 4=WebSocket`); TLS variants report their base /// transport. `ClientTransportKind` is `#[non_exhaustive]`, so any other /// (TCP, TCP-TLS, or a future) variant falls back to TCP. -pub(crate) const fn transport_kind_to_wire(kind: ClientTransportKind) -> u8 { +pub const fn transport_kind_to_wire(kind: ClientTransportKind) -> u8 { match kind { ClientTransportKind::Quic => 2, ClientTransportKind::Ws | ClientTransportKind::Wss => 4, @@ -61,7 +59,7 @@ pub(crate) const fn transport_kind_to_wire(kind: ClientTransportKind) -> u8 { } } -pub(crate) fn usize_to_u32(value: usize) -> Result { +pub fn usize_to_u32(value: usize) -> Result { u32::try_from(value).map_err(|_| IggyError::InvalidIdentifier) } @@ -69,7 +67,7 @@ pub(crate) fn usize_to_u32(value: usize) -> Result { /// preserving the header (and fixing `size`). Used by the primary-side /// request rewrites that swap a secret-bearing wire body for the /// hash-carrying replicated body before consensus. -pub(crate) fn rewrite_request_body( +pub fn rewrite_request_body( request: &Message, body: &Bytes, ) -> Result, IggyError> { diff --git a/core/server/tests/module_graph.rs b/core/server/tests/module_graph.rs new file mode 100644 index 0000000000..d9d978a56f --- /dev/null +++ b/core/server/tests/module_graph.rs @@ -0,0 +1,319 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Module-graph guard: the crate's module dependency graph must stay a DAG. +//! +//! The http/ role-leaf pattern rotted into two state-hub cycles while nothing +//! enforced it, and two of the historical bootstrap cycles were invisible to +//! `use`-greps (an inline-qualified `crate::...` call and `use super::` +//! imports). So this test parses every source file with `syn` and scans full +//! token streams - it sees qualified call sites and macro arguments, not just +//! use declarations. +//! +//! Granularity is the module FILE. Parent<->child edges are exempt: a root +//! composing its children (and children reaching items the root defines) is +//! the pattern working as intended. Sibling and cross-tree cycles are the +//! rot this guard exists to stop. +//! +//! `WHITELIST` carries the known survivors. Each entry must still be a live +//! cycle - a stale entry fails the test, so the list can only shrink. + +use proc_macro2::{TokenStream, TokenTree}; +use quote::ToTokens; +use std::collections::{BTreeMap, BTreeSet}; +use std::path::{Path, PathBuf}; + +/// Known mutual edges, as unordered pairs of module paths. Burned down per +/// refactor PR; the auth<->dispatch cycle dies with the dispatch/ login merge. +const WHITELIST: [(&str, &str); 1] = [("auth", "dispatch")]; + +type Module = Vec; + +#[test] +fn module_graph_is_a_dag_modulo_whitelist() { + let src = Path::new(env!("CARGO_MANIFEST_DIR")).join("src"); + let modules = collect_modules(&src); + let module_set: BTreeSet = modules.iter().map(|(module, _)| module.clone()).collect(); + + let mut edges: BTreeMap> = BTreeMap::new(); + for (module, file) in &modules { + let source = std::fs::read_to_string(file) + .unwrap_or_else(|error| panic!("cannot read {}: {error}", file.display())); + let ast = syn::parse_file(&source) + .unwrap_or_else(|error| panic!("cannot parse {}: {error}", file.display())); + let mut paths = Vec::new(); + scan_items(&ast.items, &mut paths); + let targets = edges.entry(module.clone()).or_default(); + for raw in paths { + if let Some(target) = resolve(module, &raw, &module_set) + && target != *module + && !is_ancestor(&target, module) + && !is_ancestor(module, &target) + { + targets.insert(target); + } + } + } + + let whitelist: BTreeSet<(Module, Module)> = WHITELIST + .iter() + .flat_map(|(a, b)| { + let a = parse_module(a); + let b = parse_module(b); + [(a.clone(), b.clone()), (b, a)] + }) + .collect(); + + for (from, to) in &whitelist { + assert!( + edges.get(from).is_some_and(|targets| targets.contains(to)), + "stale whitelist entry {} -> {}: the edge is gone, remove it", + render(from), + render(to), + ); + } + + for (from, targets) in &mut edges { + targets.retain(|to| !whitelist.contains(&(from.clone(), to.clone()))); + } + + if let Some(cycle) = find_cycle(&edges) { + let chain = cycle.iter().map(render).collect::>(); + panic!( + "module cycle: {}\nBreak it by moving the shared vocabulary into \ + a leaf both sides import, or point the edge one way.", + chain.join(" -> "), + ); + } +} + +/// Map every source file under `src/` to its module path. `lib.rs` is the +/// crate root `[]`; `main.rs`/`args.rs` belong to the bin target and are +/// skipped (their `crate::` is a different crate). +fn collect_modules(src: &Path) -> Vec<(Module, PathBuf)> { + let mut files = Vec::new(); + walk(src, &mut files); + files.sort(); + files + .into_iter() + .filter_map(|file| { + let relative = file + .strip_prefix(src) + .unwrap_or_else(|_| panic!("{} outside src", file.display())); + let mut module: Vec = relative + .components() + .map(|c| c.as_os_str().to_string_lossy().into_owned()) + .collect(); + let last = module.pop().unwrap_or_default(); + match last.as_str() { + "main.rs" | "args.rs" => return None, + "lib.rs" | "mod.rs" => {} + _ => module.push(last.trim_end_matches(".rs").to_owned()), + } + Some((module, file)) + }) + .collect() +} + +fn walk(dir: &Path, files: &mut Vec) { + let entries = std::fs::read_dir(dir) + .unwrap_or_else(|error| panic!("cannot read dir {}: {error}", dir.display())); + for entry in entries { + let path = entry + .unwrap_or_else(|error| panic!("cannot read entry in {}: {error}", dir.display())) + .path(); + if path.is_dir() { + walk(&path, files); + } else if path.extension().is_some_and(|ext| ext == "rs") { + files.push(path); + } + } +} + +/// Collect every `crate::`/`super::`-rooted path in the token streams of +/// `items`, recursing through nested modules and skipping `#[cfg(test)]` ones. +fn scan_items(items: &[syn::Item], paths: &mut Vec>) { + for item in items { + if let syn::Item::Mod(module) = item { + if is_cfg_test(&module.attrs) { + continue; + } + if let Some((_, nested)) = &module.content { + scan_items(nested, paths); + } + continue; + } + scan_tokens(item.to_token_stream(), paths); + } +} + +fn is_cfg_test(attrs: &[syn::Attribute]) -> bool { + attrs.iter().any(|attr| { + attr.path().is_ident("cfg") + && attr + .parse_args::() + .is_ok_and(tokens_mention_test) + }) +} + +/// True when the cfg predicate names the bare `test` ident anywhere, +/// `cfg(all(test, ...))` included. +fn tokens_mention_test(tokens: TokenStream) -> bool { + tokens.into_iter().any(|tree| match tree { + TokenTree::Ident(ident) => ident == "test", + TokenTree::Group(group) => tokens_mention_test(group.stream()), + _ => false, + }) +} + +/// Token-level scan: catches inline-qualified calls and macro arguments that +/// an AST `use`-only walk would miss. Doc comments and string literals are +/// literals, not idents, so they never produce edges. +fn scan_tokens(tokens: TokenStream, paths: &mut Vec>) { + let trees: Vec = tokens.into_iter().collect(); + let mut index = 0; + while index < trees.len() { + match &trees[index] { + TokenTree::Group(group) => { + scan_tokens(group.stream(), paths); + index += 1; + } + TokenTree::Ident(ident) => { + let name = ident.to_string(); + if name == "crate" || name == "super" { + let (segments, consumed) = read_path(&trees, index); + if segments.len() > 1 { + paths.push(segments); + } + index += consumed; + } else { + index += 1; + } + } + _ => index += 1, + } + } +} + +/// Read `ident (:: ident)*` starting at `start`, descending into no groups. +fn read_path(trees: &[TokenTree], start: usize) -> (Vec, usize) { + let mut segments = Vec::new(); + let mut index = start; + while let Some(TokenTree::Ident(ident)) = trees.get(index) { + segments.push(ident.to_string()); + index += 1; + let double_colon = matches!(trees.get(index), Some(TokenTree::Punct(p)) if p.as_char() == ':') + && matches!(trees.get(index + 1), Some(TokenTree::Punct(p)) if p.as_char() == ':'); + if double_colon { + index += 2; + } else { + break; + } + } + (segments, index - start) +} + +/// Resolve a raw `crate::`/`super::` path to the deepest known module it +/// names. Returns `None` for paths into the crate root's own items. +fn resolve(current: &Module, raw: &[String], modules: &BTreeSet) -> Option { + let (base, rest): (Module, &[String]) = match raw.first().map(String::as_str) { + Some("crate") => (Vec::new(), &raw[1..]), + Some("super") => { + let supers = raw.iter().take_while(|s| *s == "super").count(); + if supers > current.len() { + return None; + } + (current[..current.len() - supers].to_vec(), &raw[supers..]) + } + _ => return None, + }; + let mut best: Option = if base.is_empty() { + None + } else { + Some(base.clone()) + }; + let mut candidate = base; + for segment in rest { + candidate.push(segment.clone()); + if modules.contains(&candidate) { + best = Some(candidate.clone()); + } else { + break; + } + } + best +} + +fn is_ancestor(shorter: &Module, longer: &Module) -> bool { + shorter.len() < longer.len() && longer[..shorter.len()] == shorter[..] +} + +fn render(module: &Module) -> String { + if module.is_empty() { + "crate".to_owned() + } else { + module.join("::") + } +} + +fn parse_module(path: &str) -> Module { + path.split("::").map(str::to_owned).collect() +} + +/// DFS three-color cycle search; returns one cycle as a module chain. +fn find_cycle(edges: &BTreeMap>) -> Option> { + let mut visiting = BTreeSet::new(); + let mut done = BTreeSet::new(); + let mut stack = Vec::new(); + for start in edges.keys() { + if let Some(cycle) = dfs(start, edges, &mut visiting, &mut done, &mut stack) { + return Some(cycle); + } + } + None +} + +fn dfs( + node: &Module, + edges: &BTreeMap>, + visiting: &mut BTreeSet, + done: &mut BTreeSet, + stack: &mut Vec, +) -> Option> { + if done.contains(node) { + return None; + } + if visiting.contains(node) { + let from = stack.iter().position(|n| n == node).unwrap_or(0); + let mut cycle = stack[from..].to_vec(); + cycle.push(node.clone()); + return Some(cycle); + } + visiting.insert(node.clone()); + stack.push(node.clone()); + if let Some(targets) = edges.get(node) { + for target in targets { + if let Some(cycle) = dfs(target, edges, visiting, done, stack) { + return Some(cycle); + } + } + } + stack.pop(); + visiting.remove(node); + done.insert(node.clone()); + None +} diff --git a/core/shard_allocator/Cargo.toml b/core/shard_allocator/Cargo.toml deleted file mode 100644 index 1d51d3a8db..0000000000 --- a/core/shard_allocator/Cargo.toml +++ /dev/null @@ -1,41 +0,0 @@ -# Licensed to the Apache Software Foundation (ASF) under one -# or more contributor license agreements. See the NOTICE file -# distributed with this work for additional information -# regarding copyright ownership. The ASF licenses this file -# to you under the Apache License, Version 2.0 (the -# "License"); you may not use this file except in compliance -# with the License. You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, -# software distributed under the License is distributed on an -# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY -# KIND, either express or implied. See the License for the -# specific language governing permissions and limitations -# under the License. - -[package] -name = "shard_allocator" -version = "0.1.0" -description = "CPU and NUMA shard allocation for the iggy server, backed by hwloc." -edition = "2024" -license = "Apache-2.0" -publish = false - -[dependencies] -cpu_allocation = { workspace = true } -thiserror = { workspace = true } -tracing = { workspace = true } - -[target.'cfg(not(target_env = "musl"))'.dependencies] -hwlocality = { workspace = true } - -[target.'cfg(target_env = "musl")'.dependencies] -hwlocality = { workspace = true, features = ["vendored"] } - -[target.'cfg(target_os = "linux")'.dependencies] -nix = { workspace = true } - -[lints] -workspace = true diff --git a/core/shard_allocator/build.rs b/core/shard_allocator/build.rs deleted file mode 100644 index d66e93f072..0000000000 --- a/core/shard_allocator/build.rs +++ /dev/null @@ -1,43 +0,0 @@ -// Licensed to the Apache Software Foundation (ASF) under one -// or more contributor license agreements. See the NOTICE file -// distributed with this work for additional information -// regarding copyright ownership. The ASF licenses this file -// to you under the Apache License, Version 2.0 (the -// "License"); you may not use this file except in compliance -// with the License. You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, -// software distributed under the License is distributed on an -// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY -// KIND, either express or implied. See the License for the -// specific language governing permissions and limitations -// under the License. - -// Vendored `hwloc` references `cbrt`, which makes the linker pull in a -// `libm`. musl folds the math functions into its `libc`, so the Rust -// musl sysroot ships no `libm.a`. Without one, `-lm` falls through to -// the host glibc's `libm.a`, whose `cbrt` needs glibc-internal -// `__frexp`/`__ldexp` symbols that do not exist on musl, and the static -// link fails. Drop an empty `libm.a` stub on the search path so `-lm` -// resolves to nothing and `cbrt` is satisfied later by musl's own -// `libc`. No effect on non-musl targets. - -use std::env; -use std::fs; -use std::path::Path; - -fn main() { - if env::var("CARGO_CFG_TARGET_ENV").as_deref() != Ok("musl") { - return; - } - - let out_dir = env::var("OUT_DIR").expect("OUT_DIR is set by cargo for build scripts"); - let stub = Path::new(&out_dir).join("libm.a"); - - // `!\n` is the canonical header of an empty `ar` archive. - fs::write(&stub, b"!\n").expect("write empty libm.a stub"); - - println!("cargo:rustc-link-search=native={out_dir}"); -} diff --git a/core/simulator/src/replica.rs b/core/simulator/src/replica.rs index 638a61dce6..8e57773eab 100644 --- a/core/simulator/src/replica.rs +++ b/core/simulator/src/replica.rs @@ -32,7 +32,8 @@ use metadata::stm::stream::{Streams, StreamsInner}; use metadata::stm::user::{Users, UsersInner}; use metadata::{IggyMetadata, apply_committed_prepare}; use partitions::{IggyPartitions, PartitionPathLayout, PartitionsConfig}; -use server::bootstrap::{ShellHandlers, ShellShardHandle, wire_shell_handlers}; +use server::bootstrap::wire_shell_handlers; +use server::shell::{ShellHandlers, ShellShardHandle}; use server_common::crypto; use server_common::sharding::{METADATA_GROUP, ShardId}; use shard::shards_table::PapayaShardsTable; From 3aed9f458e545cb0de77bbe040e68ab5d5579a5a Mon Sep 17 00:00:00 2001 From: Grzegorz Koszyk <112548209+numinnex@users.noreply.github.com> Date: Tue, 1 Sep 2026 16:14:01 +0200 Subject: [PATCH 029/182] fix(shard): keep a parked prepare's stamp and op order on re-dispatch (#4005) Partition prepares can arrive before their namespace is materialised. Re-dispatch through the shard inbox lost each frame's original epoch stamp and placed it behind newer frames. A delete and recreate could therefore serve an old prepare against the replacement, while the backup gap check could drop a later operation before the parked prefix reached the plane. Stage parked frames in a shard-local FIFO and carry their epoch and park-pass provenance. A biased pump arm processes one staged frame per iteration after ticks and before inbox work. This preserves order without delaying heartbeats behind a bounded queue, and loopback runs after every delivery so a solo primary commits re-dispatched requests immediately. Shutdown uses the same ordering. Simulator materialisation now passes its resolved epoch and wakes the pump after off-pump staging. Seeded three-replica and solo-primary tests pin queue ordering, wakeup, loopback, and deterministic replay. Quiescence rejects stranded redispatch work. The integration oracle now filters gap markers to the partition plane, removes ambient RUST_LOG when an explicit logging level is configured, and falls back to the server log when verbose mode inherits stdout. Dropped or rejected staged frames are counted, and recovery documentation reflects same-view repair. The shared leader-redirection BDD fixture could elect replica 1 when a loaded runner delayed replica 0 during concurrent startup. Start replica 1 only after replica 0 is healthy and use a wider fixture heartbeat window so the scenarios test client redirection instead of container scheduling. --------- Co-authored-by: Piotr Gankiewicz --- bdd/docker-compose.cluster.yml | 6 + core/integration/src/harness/handle/server.rs | 49 +- core/integration/tests/cluster/mod.rs | 1 + .../tests/cluster/parked_frame_redispatch.rs | 363 +++++++++++++ core/server/src/partition_reconciler.rs | 396 ++++++++------ core/shard/src/lib.rs | 508 ++++++++++-------- core/shard/src/metrics.rs | 35 +- core/shard/src/router.rs | 109 +++- core/simulator/src/lib.rs | 271 +++++++++- 9 files changed, 1266 insertions(+), 472 deletions(-) create mode 100644 core/integration/tests/cluster/parked_frame_redispatch.rs diff --git a/bdd/docker-compose.cluster.yml b/bdd/docker-compose.cluster.yml index 9976d342b1..df773b8a5e 100644 --- a/bdd/docker-compose.cluster.yml +++ b/bdd/docker-compose.cluster.yml @@ -56,6 +56,9 @@ x-cluster-topology: &cluster-topology IGGY_ROOT_PASSWORD: iggy IGGY_CLUSTER_ENABLED: "true" IGGY_CLUSTER_NAME: test-cluster + # Keep role-specific scenarios on view 0 when a loaded CI runner delays + # one replica during startup. + IGGY_CLUSTER_HEARTBEAT_TIMEOUT: "30s" IGGY_CLUSTER_NODES_0_NAME: leader-node IGGY_CLUSTER_NODES_0_IP: 172.28.0.101 IGGY_CLUSTER_NODES_0_REPLICA_ID: "0" @@ -115,6 +118,9 @@ services: iggy-follower: <<: *cluster-node + depends_on: + iggy-leader: + condition: service_healthy command: [ "--replica-id", "1" ] healthcheck: test: [ "CMD", "/usr/local/bin/iggy", "--tcp-server-address", "172.28.0.102:8092", "ping" ] diff --git a/core/integration/src/harness/handle/server.rs b/core/integration/src/harness/handle/server.rs index ed015a1ec7..088b3b2617 100644 --- a/core/integration/src/harness/handle/server.rs +++ b/core/integration/src/harness/handle/server.rs @@ -187,35 +187,46 @@ impl ServerHandle { self.stdout_occurrences(marker) > 0 } - /// Number of times `marker` appears in this node's stdout log. The log - /// file is TRUNCATED on every (re)start (it captures one process run), - /// so counts never carry across a restart; a marker seen after - /// `restart_server` was logged by the new process. + /// Number of times `marker` appears in this node's captured stdout, or in + /// its own logs when verbose mode makes the child inherit stdout. Captured + /// stdout is truncated on restart. The server log may span restarts, so + /// restart-sensitive callers must compare against a pre-restart baseline. /// /// ANSI escape sequences are stripped before matching: the server colors /// its tracing fields, so a `key=value` marker never matches the raw /// bytes (`key\x1b[0m\x1b[2m=\x1b[0m value`). #[must_use] pub fn stdout_occurrences(&self, marker: &str) -> usize { - self.stdout_path - .as_ref() - .and_then(|path| fs::read_to_string(path).ok()) - .map_or(0, |log| strip_ansi(&log).matches(marker).count()) + self.stdout_plain().matches(marker).count() } - /// This node's stdout log with ANSI escapes stripped, the same text - /// [`Self::stdout_occurrences`] matches against. Empty when the log is - /// missing or unreadable. + /// This node's captured stdout with ANSI escapes stripped, or its own logs + /// when verbose mode disables the capture. Empty when neither source is + /// readable. /// /// For callers that need to PARSE a marker's fields (`checkpoint_op=193` /// reaches the file as `checkpoint_op\x1b[0m\x1b[2m=\x1b[0m193`) rather /// than just count occurrences of it. #[must_use] pub fn stdout_plain(&self) -> String { - self.stdout_path + let captured = self + .stdout_path .as_ref() .and_then(|path| fs::read_to_string(path).ok()) - .map_or_else(String::new, |log| strip_ansi(&log)) + .map(|log| strip_ansi(&log)); + captured.unwrap_or_else(|| { + let own_logs = self.data_path().join("logs"); + let Ok(entries) = fs::read_dir(own_logs) else { + return String::new(); + }; + let mut log = String::new(); + for entry in entries.flatten() { + if let Ok(contents) = fs::read_to_string(entry.path()) { + log.push_str(&contents); + } + } + log + }) } /// Returns a `ClientBuilder` using the test transport. @@ -882,6 +893,18 @@ impl TestBinary for ServerHandle { { command.env("IGGY_SHARD_RUNTIME_CAPACITY", "256"); } + // An explicit config-level override must be the test's logging + // contract. The server gives `RUST_LOG` precedence, so inheriting an + // ambient value could silently filter out markers the test asserts. + // A caller that explicitly puts `RUST_LOG` in `extra_envs` adds it back + // through `command.envs` below. + if self + .config + .extra_envs + .contains_key("IGGY_SYSTEM_LOGGING_LEVEL") + { + command.env_remove("RUST_LOG"); + } command.envs(&self.envs); // `--replica-id` is the single identity input expected by the diff --git a/core/integration/tests/cluster/mod.rs b/core/integration/tests/cluster/mod.rs index 940e1fcee8..b794264a0a 100644 --- a/core/integration/tests/cluster/mod.rs +++ b/core/integration/tests/cluster/mod.rs @@ -25,6 +25,7 @@ mod fast_primary_rejoin; mod metadata_checkpoint_restart; mod metadata_state_transfer; mod multi_shard_partition_convergence; +mod parked_frame_redispatch; mod partition_primary_routing; mod partition_state_transfer; mod register_forwarding; diff --git a/core/integration/tests/cluster/parked_frame_redispatch.rs b/core/integration/tests/cluster/parked_frame_redispatch.rs new file mode 100644 index 0000000000..3b1a525ed4 --- /dev/null +++ b/core/integration/tests/cluster/parked_frame_redispatch.rs @@ -0,0 +1,363 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! A BACKUP parking replicated partition prepares, and the frames it parked +//! reaching the plane in op order once its partition materialises. +//! +//! `multi_shard_partition_convergence` covers the same fence on one node and +//! says outright that it cannot tell a request served straight through from one +//! that parked. This test pins the park path positively, and on the replica +//! where getting it wrong creates a replica gap: a client request that never +//! reaches the plane is answered with a retriable status and the SDK replays it, +//! while a replicated PREPARE has no client behind it and must wait for a later +//! commit heartbeat to arm repair if this path drops it. +//! +//! What makes the window wide on a backup is the commit broadcast. A backup +//! learns a metadata commit from the `commit` field of the next prepare on that +//! plane or from the primary's `CommitMessage` heartbeat, whose interval is +//! `cluster.commit_broadcast_interval` (500ms by default). Nothing in +//! `create_topic`'s reply path waits for that, so a produce issued the instant +//! `create_topic` returns reaches the backups as a partition prepare for a +//! namespace they have not yet heard of, let alone built. Four producers on +//! their own connections keep a burst in flight across that gap, so several ops +//! of one partition park together and the order they leave in is observable. +//! +//! Three things are asserted, and they fail separately: +//! +//! - The path was entered on a backup. `redispatch_parked_frames` logs at +//! `debug`, hence the `system.logging.level` override; the marker on a node +//! that is not the leader is proof, because a fresh partition group seeds its +//! view from the metadata plane, so every partition primary here is the +//! metadata leader and no client request lands anywhere else. +//! - No park path degraded into a shed, an aged-out answer, or an incarnation +//! rejection, and no replica dropped a prepare for arriving out of order. +//! That last marker is the direct symptom of re-dispatch losing a frame's +//! arrival position. +//! - Every acked message is readable in dense offset order, each producer's own +//! sends stay in the order it made them, and all three replicas hold +//! byte-identical segments. A prepare lost to the gap check leaves a backup +//! permanently short, since the gap never closes on its own. +//! +//! The harness removes an ambient `RUST_LOG` when this test supplies its explicit +//! logging level, and the log oracle falls back from captured stdout to the +//! server's own log file so `IGGY_TEST_VERBOSE` cannot disable it. + +use std::collections::HashMap; +use std::path::PathBuf; +use std::str::FromStr; +use std::time::Duration; + +use futures::future::join_all; +use iggy::prelude::*; +use integration::harness::{ServerHandle, TestHarness, disk}; +use integration::iggy_harness; +use tokio::time::{Instant, sleep}; + +const STREAM: &str = "parked-redispatch-stream"; +/// Each topic is one shot at the race, and each costs about one commit +/// broadcast interval. +const TOPICS: u32 = 6; +const PARTITIONS: u32 = 4; +/// The reconciler builds a topic's namespaces in id order, so the last one has +/// the longest wait for its `InsertOwned`. +const TARGET_PARTITION: u32 = PARTITIONS - 1; +/// Separate connections, because one `IggyClient` serialises its requests and a +/// single in-flight prepare would never expose park ordering. +const PRODUCERS: usize = 4; +const PER_PRODUCER: usize = 8; +const TOTAL_MESSAGES: usize = PRODUCERS * PER_PRODUCER; + +/// Budget for eagerly flushed batches to reach every node's segment files. +const FLUSH_INSTALL_TIMEOUT: Duration = Duration::from_secs(20); +const POLL_INTERVAL: Duration = Duration::from_millis(250); + +/// `IggyShard::redispatch_parked_frames`, at `debug`. +const REDISPATCH_MARKER: &str = "re-dispatching parked partition frames after materialisation"; + +/// Modes in which the park path gives up instead of converging. All three cost +/// a frame: the first two shed or answer, the third refuses a namespace whose +/// incarnation moved under it. +const DEGRADED_MARKERS: [&str; 3] = [ + "park buffer at capacity", + "outlived their admission window", + "rejecting parked partition frame", +]; + +/// `IggyPartition::on_replicate`'s backup gap check. A re-dispatch that appends +/// behind an op already queued on the inbox surfaces here, and the dropped op +/// forces repair that correct redispatch ordering should never need. +const GAP_MARKER: &str = "dropping out-of-order prepare (gap)"; +const PARTITION_PLANE_FIELD: &str = "plane=\"partitions\""; + +fn topic_name(index: u32) -> String { + format!("parked-redispatch-topic-{index}") +} + +#[iggy_harness(cluster_nodes = 3, server(system.logging.level = "info,shard=debug"))] +async fn given_a_produce_burst_right_after_create_topic_when_backups_park_the_prepares_should_re_dispatch_them_in_order( + harness: &mut TestHarness, +) { + // Read once, before any topic exists: the leader is the primary of every + // partition group created below, so it is also the only node a produce can + // be admitted on. + let leader = disk::leader_node_index_via(harness, 0).await; + let setup = harness + .root_client_for_node(leader) + .await + .expect("root client on the metadata leader"); + setup.create_stream(STREAM).await.expect("create stream"); + let stream = Identifier::named(STREAM).expect("stream identifier"); + + // Connected and logged in before the first `create_topic`, so the burst + // costs one round trip rather than a handshake. + let mut producers = Vec::with_capacity(PRODUCERS); + for _ in 0..PRODUCERS { + producers.push( + harness + .root_client_for_node(leader) + .await + .expect("root client for a producer"), + ); + } + + let mut all_payloads = Vec::with_capacity(TOPICS as usize * TOTAL_MESSAGES); + for topic_index in 0..TOPICS { + let name = topic_name(topic_index); + create_topic(&setup, &stream, &name).await; + let topic = Identifier::named(&name).expect("topic identifier"); + + let sent = produce_burst(&producers, &stream, &topic, topic_index).await; + let polled = poll_payloads(&setup, &stream, &topic).await; + assert_eq!( + polled.len(), + TOTAL_MESSAGES, + "topic {topic_index} must serve every acked message, got {polled:?}" + ); + assert_producer_order(&polled, &sent, topic_index); + all_payloads.extend(polled); + } + + let data_paths: Vec = harness + .all_servers() + .iter() + .map(|server| server.data_path()) + .collect(); + wait_until_payloads_installed(harness, &all_payloads).await; + disk::wait_for_log_convergence(&data_paths).await; + // Also flushes each node's non-blocking log appender, so the markers below + // are read off a complete file. + harness + .stop() + .await + .expect("stop the cluster for the at-rest comparison"); + + assert_backup_re_dispatched(harness, leader); + assert_no_degraded_park_paths(harness); + disk::assert_replica_data_identical(&data_paths, false); +} + +/// `messages_required_to_save` + `enforce_fsync` persist every committed batch +/// on every replica, which is what makes the on-disk assertions mean anything +/// on a run this small; the default thresholds would ack from RAM alone. +async fn create_topic(client: &IggyClient, stream: &Identifier, name: &str) { + client + .create_topic( + stream, + name, + &TopicCreateOptions { + partitions_count: Some(PARTITIONS), + message_expiry: Some(IggyExpiry::NeverExpire), + messages_required_to_save: Some(1), + enforce_fsync: Some(true), + ..TopicCreateOptions::default() + }, + ) + .await + .unwrap_or_else(|error| panic!("create_topic {name}: {error}")); +} + +/// Fire every producer at once, returning each one's payloads in the order it +/// sent them. +async fn produce_burst( + producers: &[IggyClient], + stream: &Identifier, + topic: &Identifier, + topic_index: u32, +) -> Vec> { + let partitioning = Partitioning::partition_id(TARGET_PARTITION); + let sends = producers.iter().enumerate().map(|(producer, client)| { + let partitioning = &partitioning; + async move { + let mut sent = Vec::with_capacity(PER_PRODUCER); + for sequence in 0..PER_PRODUCER { + let payload = format!("t{topic_index}-p{producer}-{sequence}"); + let mut messages = vec![IggyMessage::from_str(&payload).expect("build message")]; + client + .send_messages(stream, topic, partitioning, &mut messages) + .await + .unwrap_or_else(|error| panic!("send_messages {payload}: {error}")); + sent.push(payload); + } + sent + } + }); + join_all(sends).await +} + +/// Payloads of the target partition in offset order, asserting the offsets are +/// dense on the way out: a hole would mean an acked op the leader itself cannot +/// serve. +async fn poll_payloads( + client: &IggyClient, + stream: &Identifier, + topic: &Identifier, +) -> Vec { + let polled = client + .poll_messages( + stream, + topic, + Some(TARGET_PARTITION), + &Consumer::default(), + &PollingStrategy::offset(0), + TOTAL_MESSAGES as u32, + false, + ) + .await + .unwrap_or_else(|error| panic!("poll_messages: {error}")); + for (expected, message) in polled.messages.iter().enumerate() { + assert_eq!( + message.header.offset, expected as u64, + "offsets must be dense from 0, got {} at position {expected}", + message.header.offset + ); + } + polled + .messages + .iter() + .map(|message| String::from_utf8_lossy(&message.payload).into_owned()) + .collect() +} + +/// Each producer's sends must appear in the order it made them. Nothing pins +/// the interleaving of four connections, but a partition that reordered one +/// producer's own ops reordered the log. +fn assert_producer_order(polled: &[String], sent: &[Vec], topic_index: u32) { + let positions: HashMap<&str, usize> = polled + .iter() + .enumerate() + .map(|(position, payload)| (payload.as_str(), position)) + .collect(); + for (producer, payloads) in sent.iter().enumerate() { + let mut previous: Option<(&str, usize)> = None; + for payload in payloads { + let position = *positions.get(payload.as_str()).unwrap_or_else(|| { + panic!("topic {topic_index}: {payload} was acked but never polled back") + }); + if let Some((earlier, earlier_position)) = previous { + assert!( + earlier_position < position, + "topic {topic_index}: producer {producer} sent {earlier} before {payload}, \ + but they polled back at {earlier_position} and {position}" + ); + } + previous = Some((payload.as_str(), position)); + } + } +} + +/// Poll until every node's segments hold every payload at a non-decreasing +/// position. A backup that lost a prepare to the gap check never gets it back, +/// so this is where the loss surfaces first, naming the node. +async fn wait_until_payloads_installed(harness: &TestHarness, payloads: &[String]) { + let deadline = Instant::now() + FLUSH_INSTALL_TIMEOUT; + loop { + let pending: Vec = (0..harness.cluster_size()) + .filter_map(|node| { + disk::installed_payloads_complete(&harness.node(node).data_path(), payloads) + .err() + .map(|error| format!("node {node}: {error}")) + }) + .collect(); + if pending.is_empty() { + return; + } + assert!( + Instant::now() < deadline, + "every acked payload must reach every replica's segments within \ + {FLUSH_INSTALL_TIMEOUT:?}: {pending:?}" + ); + sleep(POLL_INTERVAL).await; + } +} + +/// The point of the test. A non-leader node logging the re-dispatch is a backup +/// that parked REPLICATED prepares: client requests only ever reach the leader, +/// which is the primary of every partition group created here. +fn assert_backup_re_dispatched(harness: &TestHarness, leader: usize) { + let counts: Vec<(usize, usize)> = (0..harness.cluster_size()) + .map(|node| { + ( + node, + server_log_occurrences(harness.node(node), REDISPATCH_MARKER), + ) + }) + .collect(); + let on_backups: usize = counts + .iter() + .filter(|(node, _)| *node != leader) + .map(|(_, count)| *count) + .sum(); + assert!( + on_backups > 0, + "no backup logged {REDISPATCH_MARKER:?} (leader is node {leader}, per-node counts \ + {counts:?}); either the produce never raced materialisation, in which case this test \ + proves nothing, or the configured debug marker was not written" + ); +} + +fn assert_no_degraded_park_paths(harness: &TestHarness) { + for node in 0..harness.cluster_size() { + let server = harness.node(node); + let log = server_log_plain(server); + for marker in DEGRADED_MARKERS { + assert_eq!( + log.matches(marker).count(), + 0, + "node {node} logged {marker:?}: the park buffer degraded instead of converging" + ); + } + let partition_gaps = log + .lines() + .filter(|line| line.contains(GAP_MARKER) && line.contains(PARTITION_PLANE_FIELD)) + .count(); + assert_eq!( + partition_gaps, 0, + "node {node} logged {GAP_MARKER:?}: a re-dispatched prepare lost its arrival \ + position and forced avoidable partition repair" + ); + } +} + +fn server_log_occurrences(server: &ServerHandle, marker: &str) -> usize { + server_log_plain(server).matches(marker).count() +} + +/// The shared handle falls back to the server's own appender when +/// `IGGY_TEST_VERBOSE` makes the child inherit stdout. +fn server_log_plain(server: &ServerHandle) -> String { + server.stdout_plain() +} diff --git a/core/server/src/partition_reconciler.rs b/core/server/src/partition_reconciler.rs index 5cbd5f19ad..5de84bcc1b 100644 --- a/core/server/src/partition_reconciler.rs +++ b/core/server/src/partition_reconciler.rs @@ -40,20 +40,20 @@ //! "unroutable", and falls back to `calculate_shard_assignment`. The frame //! always reaches the shard that will own the partition. //! - `IggyShard::park_if_unmaterialised` holds it there until the matching -//! `InsertOwned` lands, then re-queues it onto this shard's inbox -- but not to -//! a DIFFERENT incarnation than the one it was addressed to. Each parked frame +//! `InsertOwned` lands, then hands it back to the pump -- but not to a +//! DIFFERENT incarnation than the one it was addressed to. Each parked frame //! carries the committed `created_revision` observed when it was parked, and a //! drain whose epoch disagrees with that stamp answers the client instead of //! serving it: recycled slab keys make the namespace byte-identical, so such a //! frame would otherwise land a dead topic's write inside the topic that -//! replaced it. One gap, recorded below: the stamp is re-derived if the frame -//! re-enters the park path from the inbox. A frame parked with NO stamp is -//! served; see `redispatch_parked_frames` for why a missing committed revision -//! is not evidence of a prior incarnation. Re-queuing appends, so a parked -//! frame is ordered behind whatever is already in the inbox. A frame the inbox -//! refuses is re-parked rather than answered, since the deny would ride the -//! same full sender, and the pump re-drives it (`retry_reparked_frames`) once -//! a slot frees. +//! replaced it. Production prevents a second park by ranking redispatch above +//! inbox work and applying incarnation changes only on the pump. Carrying the +//! stamp through re-delivery is defence in depth for off-pump staging such as +//! simulator materialisation. A frame parked with NO stamp is served; see +//! `redispatch_parked_frames` for why a missing committed revision is not +//! evidence of a prior incarnation. The redispatch select arm takes one frame +//! per iteration before the inbox arm, so a parked op is not ordered behind a +//! later op of the same partition already queued there. //! - `IggyShard::serves_committed_incarnation` refuses a namespace whose //! committed `created_revision` disagrees with the epoch on the local row, so //! a request arriving mid-teardown cannot be acked against the incarnation @@ -89,9 +89,9 @@ //! //! They apply asymmetrically, because the two frame classes fail differently. A //! shed request costs a retry: answered with a retriable status, re-issued by -//! the SDK. A shed prepare is permanent loss on this replica, with no client to -//! answer and `consensus::retransmit_targets` skipping any op that already -//! reached quorum. +//! the SDK. A shed prepare has no client to retry it and leaves the replica +//! behind until a later commit heartbeat exposes the gap and arms same-view +//! journal repair. //! //! All three bind a request: refused when admitting it would cross a byte budget //! or the frame cap, answered past `MAX_PARKED_PASSES`. Only the byte budgets @@ -115,34 +115,11 @@ //! materialization barrier this module used to promise, and the barrier is gone //! (see above) while these are not: //! -//! TODO(krishna): a shed or discarded *prepare* has no recovery once its op has -//! reached quorum. `consensus::retransmit_targets` skips entries with -//! `ok_quorum_received`, and the partition plane creates a repair session only -//! in `on_start_view` -- `tick_partitions` re-drives an existing session but -//! cannot open one -- so the backup stays behind `commit_max` until an unrelated -//! view change. It needs a normal-status repair driver. The park policy above -//! shrinks the exposure to two cases, a genuinely exhausted byte budget and a -//! namespace this shard cannot serve, but only the repair driver removes it. -//! -//! TODO(krishna): the park stamp is not stable across re-entry. A re-dispatched -//! frame still in the inbox when a delete + recreate completes (`ConfirmRemove` -//! removes and untombstones in one arm, then the rebuild lands) re-enters -//! `park_if_unmaterialised` and is re-stamped with the NEW revision and -//! `passes: 0`, then served against the replacement: the write the stamp exists -//! to block. Narrow (a full delete + recreate has to finish while one frame -//! waits), but the guarantee is not absolute the way the bullet above reads. -//! Closing it needs the frame to carry provenance through the inbox instead of -//! re-deriving it on arrival. -//! -//! TODO(krishna): re-dispatch APPENDS to the inbox, so a parked prepare loses its -//! arrival position. `router.rs`'s `select_biased!` puts the consensus tick (which -//! runs `apply_reconcile_ops`, and with it the re-dispatch) above the inbox arm, -//! so a parked op N is re-queued *behind* an op N+1 that was already sitting in -//! the inbox. The partition plane then sees N+1 first, rejects it against its -//! backup gap check, and N+1 is gone -- with no normal-status repair driver to -//! refetch it (see the TODO above). Ordering has to be restored at the plane, by -//! buffering out-of-order prepares rather than dropping them, or by re-dispatching -//! through a priority path that preserves op order. +//! A shed prepare is not retransmitted once its op reached quorum, but it is not +//! stranded until a view change. A later `CommitMessage` that advances the +//! backup's frontier runs `maybe_request_partition_repair`; an evicted repair +//! range escalates to partition state transfer. The park policy still avoids +//! manufacturing that recovery work unless a byte budget is already spent. //! //! TODO(krishna): `serves_committed_incarnation` and the park stamp both call //! `Streams::created_revision_for_namespace`, now on the per-request fence path. @@ -741,12 +718,11 @@ async fn reconcile_additions( /// snapshotted before `reconcile_additions` awaits `build_partition_fresh`, so a /// topic committing during those awaits is judged against a stale set. /// -/// Everything else is aged: building, backed off, still committing, genuinely -/// deleted, or materialised with frames the inbox refused. -/// [`shard::IggyShard::age_parked_partition_frames`] answers CLIENT REQUESTS past -/// `MAX_PARKED_PASSES` and leaves prepares alone, so no client waits out its read -/// timeout and no committed op dies on a local-convergence signal. Residency -/// only; see `ParkedFrame::passes`. +/// Everything else is aged: building, backed off, still committing, or +/// genuinely deleted. [`shard::IggyShard::age_parked_partition_frames`] answers +/// CLIENT REQUESTS past `MAX_PARKED_PASSES` and leaves prepares alone, so no +/// client waits out its read timeout and no committed op dies on a +/// local-convergence signal. Residency only; see `ParkedFrame::passes`. /// /// A namespace with a staged, unapplied `InsertOwned` is exempt: its partition /// is on the way but reads as un-materialised here. The queue is asked per @@ -783,11 +759,10 @@ fn reconcile_parked_frames(ctx: &ReconcilerCtx, counters: &mut PassCounters) { counters.parked_reclaimed += 1; continue; } - // Materialised with frames still parked means the re-dispatch hit a full - // inbox and re-parked them. The pump retries every iteration, so aging is - // only the backstop for an inbox that never drains. Without it they have - // no exit: `reconcile_additions` stages no second `InsertOwned` for a - // namespace already in `IggyPartitions`. + // Un-materialised and still ours: the build is on the way or backed off, + // so age the requests rather than hold them for the process lifetime. + // Materialisation hands its frames straight to the pump, so a namespace + // in `IggyPartitions` no longer reaches here with any parked. if ctx.shard.age_parked_partition_frames(ns) > 0 { counters.parked_reclaimed += 1; } @@ -1603,19 +1578,20 @@ mod tests { } /// [`build_test_shard`] with a sender mesh, for tests asserting on work - /// handed back to the pump (transient denies, parked-frame re-dispatch). + /// handed back to the pump (transient denies, frames queued on the inbox). /// Caller must keep the returned receiver alive; dropping it turns every /// `try_send` into `Disconnected`. /// /// Mesh covers `0..=shard_id` since consumers index `senders[shard_id]`. /// Peer receivers are dropped, so a misroute fails loudly instead of landing /// in this shard's inbox and reading as success. - /// Both receiving ends of a test shard's own sender-ring slot: parked - /// frames re-dispatch onto the main lane, staged client answers onto the - /// reply lane. + /// A test shard's own sender-ring slot: both receiving ends plus the + /// sending end, so a test can also put a frame on the main lane the way a + /// peer shard would. struct TestLanes { main: shard::Receiver, reply: shard::Receiver, + main_tx: shard::TaggedSender, } fn build_test_shard_with_inbox( @@ -1628,13 +1604,14 @@ mod tests { let mut own_rx = None; for peer in 0..=shard_id { let (tx, rx, reply_rx) = shard::shard_channel(peer, capacity, capacity); - senders.push(tx); if peer == shard_id { own_rx = Some(TestLanes { main: rx, reply: reply_rx, + main_tx: tx.clone(), }); } + senders.push(tx); } let mut shard = Rc::into_inner(build_test_shard(shard_id, config, mux)) .expect("freshly built shard is uniquely owned"); @@ -1645,8 +1622,8 @@ mod tests { ) } - /// Drain a test shard's lanes into `(re-dispatched frames, staged client - /// sends)`: served parked frames vs answers headed for a client. + /// Drain a test shard's lanes into `(consensus frames, staged client + /// sends)`: work headed for the pump vs answers headed for a client. fn drain_inbox(lanes: &TestLanes) -> (usize, usize) { let mut served = 0; while let Ok(frame) = lanes.main.try_recv() { @@ -1673,6 +1650,22 @@ mod tests { drain_inbox(lanes).1 } + /// Take the prepares off a test shard's main lane in arrival order, as the + /// pump would read them, and return their op numbers. + fn drain_main_lane_prepare_ops(lanes: &TestLanes) -> Vec { + let mut ops = Vec::new(); + while let Ok(frame) = lanes.main.try_recv() { + if let shard::ShardFrame::Consensus { + message: MessageBag::Prepare(prepare), + .. + } = frame + { + ops.push(prepare.header().op); + } + } + ops + } + fn make_ctx( shard: Rc, total_shards: u16, @@ -3408,10 +3401,15 @@ mod tests { 0, "materialisation must drain the park entry" ); + assert_eq!( + shard.redispatched_frame_count(), + 1, + "the unstamped frame must be staged for the pump, not rejected" + ); let (served, answered) = drain_inbox(&inbox); assert_eq!( - served, 1, - "the unstamped frame must be re-dispatched onto the pump, not rejected" + served, 0, + "the queue is the handoff, so nothing may be appended to the inbox" ); assert_eq!( answered, 0, @@ -3483,9 +3481,8 @@ mod tests { } /// The replicated-prepare shape, which no other test covers and where both - /// park critical are worst: a prepare has no client, so `deny_parked_frame` - /// no-ops on it and anything that discards it loses committed data silently, - /// with no normal-status repair driver to refetch it. + /// park critical are worst: a prepare has no client, so discarding it forces + /// the backup to wait for a later commit heartbeat and journal repair. /// /// A backup receives the prepare before its own metadata commits (so the frame /// parks unstamped), then applies the commit and materialises. The prepare must @@ -3521,11 +3518,16 @@ mod tests { ); reconcile_pass(&ctx).await; + assert_eq!( + shard.redispatched_frame_count(), + 1, + "the parked prepare must be staged for re-dispatch; discarding it is \ + an avoidable gap, since a prepare has no client to retry it" + ); let (served, answered) = drain_inbox(&inbox); assert_eq!( - served, 1, - "the parked prepare must be re-dispatched; discarding it is silent \ - committed-data loss, since a prepare has no client to answer" + served, 0, + "the queue is the handoff, so nothing may be appended to the inbox" ); assert_eq!(answered, 0, "a prepare has no client deny to send"); assert_eq!( @@ -3576,6 +3578,158 @@ mod tests { 1, "and the reject must be counted" ); + assert_eq!( + park_dropped_count(&shard), + 1, + "a rejected prepare has no client deny, so its destruction must be recorded" + ); + } + + /// Defence-in-depth for off-pump staging: materialisation stages the frame, + /// the delete half of a recreate leaves the namespace unmaterialised, and + /// explicit test delivery parks it a second time. Production cannot take + /// this interleaving because every pump-side reconcile apply returns to the + /// higher-ranked redispatch arm before another incarnation change can run. + /// The carried stamp still prevents simulator or test staging from deriving + /// the replacement's revision on that second park. + #[compio::test] + async fn given_a_staged_frame_when_a_recreate_lands_before_the_drain_should_reject_it_as_stale() + { + let tmp = TempDir::new().expect("tempdir for system path"); + let config = test_config(&tmp); + let mux = TestMux::default(); + seed_stream(&mux, 1, "stream-restamp"); + seed_topic(&mux, 2, 0, "topic-restamp-first", vec![assignment(0, 1)]); + + let (shard, inbox) = build_test_shard_with_inbox(0, &config, mux, 8); + let ctx = make_ctx(Rc::clone(&shard), 1, Rc::new(config)); + let ns = IggyNamespace::new(0, 0, 0); + + // Parked against the FIRST incarnation, so it carries that revision. + park_one_prepare(&shard, ns, 7).await; + assert_eq!(shard.parked_frame_count(ns), 1); + + // The matching incarnation materialises, so the frame stages. Nothing + // has drained it yet: that is the pump's next step. + reconcile_pass(&ctx).await; + assert_eq!( + shard.redispatched_frame_count(), + 1, + "a stamp matching the materialised epoch must stage for the pump" + ); + + // Delete + recreate the same tuple. The teardown pass removes and + // untombstones, which is the state that makes the drain below re-park. + seed_delete_topic(&shard.plane.metadata().mux_stm, 3, 0, 0); + seed_topic( + &shard.plane.metadata().mux_stm, + 4, + 0, + "topic-restamp-second", + vec![assignment(0, 2)], + ); + reconcile_pass(&ctx).await; + assert!( + !shard.plane.partitions().contains(&ns), + "the teardown pass must have dropped the first incarnation" + ); + assert!( + !shard.plane.partitions().is_tombstoned(&ns), + "and lifted the tombstone, or the drain takes the tombstone path" + ); + + assert!( + shard.dispatch_one_redispatched_frame_for_test().await, + "the defence-in-depth frame must be available for explicit delivery" + ); + assert_eq!( + shard.parked_frame_count(ns), + 1, + "an un-materialised namespace must park the re-delivered frame again" + ); + assert_eq!(shard.redispatched_frame_count(), 0); + + // The rebuild lands at the recreate's revision. + reconcile_pass(&ctx).await; + assert!( + shard.plane.partitions().contains(&ns), + "the replacement must materialise" + ); + assert_eq!( + shard.metrics().partition_frames_rejected_stale_value(), + 1, + "the re-parked frame must keep the stamp it first parked with, so the \ + replacement rejects it instead of serving a dead incarnation's op" + ); + assert_eq!( + shard.redispatched_frame_count(), + 0, + "and it must not be staged for the replacement" + ); + assert_eq!(drain_inbox(&inbox).0, 0, "nothing may reach the pump"); + } + + /// Re-dispatch must not append to the shard's own inbox. `select_biased!` + /// ranks the consensus tick, which is where materialisation runs, above the + /// inbox arm, so a later op of the same partition can already be queued + /// there. Appended, the parked op lands behind it and the plane's backup gap + /// check drops the later one for not being `current_op + 1`. Same-view repair + /// can heal that gap later, but redispatch must not manufacture it. + /// + /// The pump's order is the queue, then the inbox, so the assertions below + /// are on where each op sits at that moment. Whether the plane accepts them + /// is out of reach here: a solo primary never runs the gap check, and a + /// synthetic prepare cannot be applied. + #[compio::test] + async fn given_a_later_op_on_the_inbox_when_the_namespace_materialises_should_hand_back_the_parked_op_first() + { + let tmp = TempDir::new().expect("tempdir for system path"); + let config = test_config(&tmp); + let mux = TestMux::default(); + seed_stream(&mux, 1, "stream-order"); + seed_topic(&mux, 2, 0, "topic-order", vec![assignment(0, 1)]); + + let (shard, inbox) = build_test_shard_with_inbox(0, &config, mux, 8); + let ctx = make_ctx(Rc::clone(&shard), 1, Rc::new(config)); + let ns = IggyNamespace::new(0, 0, 0); + + park_one_prepare(&shard, ns, 5).await; + assert_eq!(shard.parked_frame_count(ns), 1); + + // Op 6 reaches the inbox while op 5 is still parked, as it does whenever + // the primary keeps replicating through the convergence window. + inbox + .main_tx + .try_send(shard::ShardFrame::consensus( + 0, + build_partition_prepare(ns, 6), + )) + .expect("capacity for one queued prepare"); + + reconcile_pass(&ctx).await; + + assert_eq!( + shard.parked_frame_count(ns), + 0, + "materialisation must drain the park entry" + ); + assert_eq!( + drain_main_lane_prepare_ops(&inbox), + vec![6], + "op 5 must not be appended behind op 6; the plane would then see 6 \ + first and drop it as a gap" + ); + assert_eq!( + shard.redispatched_frame_count(), + 1, + "op 5 must be handed back through the queue the pump drains before it \ + reads the inbox" + ); + assert_eq!( + park_dropped_count(&shard), + 0, + "and neither op may be counted as a drop" + ); } /// Parking does not bump `Streams::revision` and does not wake the reconciler, @@ -3673,7 +3827,7 @@ mod tests { assert_eq!( park_dropped_count(&shard), 0, - "the prepare must be retained: destroying it is unrecoverable" + "the prepare must be retained instead of forcing journal repair" ); // Retained across an unbounded number of further passes. @@ -3850,9 +4004,9 @@ mod tests { } /// The age bound answers requests and steps over prepares. Expiring a - /// prepare is permanent loss (no client, and `retransmit_targets` skips an op - /// already at quorum), and passes are commit-driven, so a create burst - /// elapses four in milliseconds across every parked namespace at once. + /// prepare would force same-view repair despite the bytes still being held + /// locally, and passes are commit-driven, so a create burst elapses four in + /// milliseconds across every parked namespace at once. #[compio::test] async fn aging_answers_requests_and_never_expires_a_prepare() { let tmp = TempDir::new().expect("tempdir for system path"); @@ -3927,8 +4081,8 @@ mod tests { } /// A frame larger than the per-namespace cap failed the check even against - /// an empty entry, so it could never park. Unrecoverable for a prepare: - /// `retransmit_targets` skips an op already at quorum. + /// an empty entry, so it could never park. A prepare already at quorum may + /// no longer retransmit, so shedding it forces later same-view repair. #[compio::test] async fn a_frame_over_the_namespace_byte_cap_still_parks_into_an_empty_entry() { const OVER_NAMESPACE_CAP: usize = 5 * 1024 * 1024; @@ -4194,62 +4348,9 @@ mod tests { assert_eq!(park_overflow_count(&shard), 2); } - /// A refused re-dispatch re-parks the frame, and by then the namespace is - /// materialised, closing every other exit: the sweep skips a namespace in - /// `IggyPartitions` and `reconcile_additions` stages no second - /// `InsertOwned`. Before the pump retry the frame sat until a topic delete, - /// unanswered, with its bytes charged and the fast-skip never re-arming. - #[compio::test] - async fn a_re_parked_frame_is_re_dispatched_once_the_inbox_drains() { - let tmp = TempDir::new().expect("tempdir for system path"); - let config = test_config(&tmp); - let mux = TestMux::default(); - seed_stream(&mux, 1, "stream-repark"); - seed_topic(&mux, 2, 0, "topic-repark", vec![assignment(0, 1)]); - - // Capacity 1: `enqueue_reconcile_op`'s `ReconcileApply` marker takes the - // only slot, so the re-dispatch that follows is refused with `Full`. - let (shard, inbox) = build_test_shard_with_inbox(0, &config, mux, 1); - let ctx = make_ctx(Rc::clone(&shard), 1, Rc::new(config)); - let ns = IggyNamespace::new(0, 0, 0); - - park_one_request(&shard, ns).await; - reconcile_pass(&ctx).await; - - assert!( - shard.plane.partitions().contains(&ns), - "the namespace must have materialised" - ); - assert_eq!( - shard.parked_frame_count(ns), - 1, - "the full inbox must have re-parked the frame rather than dropping it" - ); - - // What the pump does every iteration: consume a frame, then re-drive. - let (_served, _answered) = drain_inbox(&inbox); - shard.apply_reconcile_ops(); - - assert_eq!( - shard.parked_frame_count(ns), - 0, - "the freed slot must let the retry drain the entry" - ); - assert!( - !shard.has_parked_partition_frames(), - "and the byte budget must return, so the revision fast-skip can re-arm" - ); - assert_eq!( - drain_inbox(&inbox).0, - 1, - "the frame must reach the pump as a consensus frame, not be answered away" - ); - } - - /// A namespace mid-teardown is still in `IggyPartitions`, so it reads as - /// materialised while the fence forbids serving it. `ConfirmRemove` would - /// answer its frames, but a disk delete that keeps failing never enqueues - /// one, so the sweep has to. + /// A frame parks while the namespace is un-materialised, then teardown + /// fences it. `ConfirmRemove` would answer the frame, but a disk delete that + /// keeps failing never enqueues one, so the sweep has to. #[compio::test] async fn parked_frames_of_a_tombstoned_namespace_are_reclaimed_without_confirm_remove() { let tmp = TempDir::new().expect("tempdir for system path"); @@ -4258,20 +4359,20 @@ mod tests { seed_stream(&mux, 1, "stream-tombstone-park"); seed_topic(&mux, 2, 0, "topic-tombstone-park", vec![assignment(0, 1)]); - let (shard, _inbox) = build_test_shard_with_inbox(0, &config, mux, 1); + let (shard, _inbox) = build_test_shard_with_inbox(0, &config, mux, 8); let ctx = make_ctx(Rc::clone(&shard), 1, Rc::new(config)); let ns = IggyNamespace::new(0, 0, 0); park_one_request(&shard, ns).await; - reconcile_pass(&ctx).await; assert_eq!( shard.parked_frame_count(ns), 1, - "the full inbox must have re-parked the frame" + "the request must park while the namespace is un-materialised" ); // Teardown's synchronous fence, without the `ConfirmRemove` a wedged - // disk delete never reaches. + // disk delete never reaches. It also stops the pass below from building + // the namespace, which is what would otherwise drain the entry. shard.plane.partitions().tombstone(ns); shard.shards_table().remove(&ns); @@ -4287,35 +4388,6 @@ mod tests { ); } - /// Residency backstop: an inbox that never drains must not hold a re-parked - /// frame forever, so `MAX_PARKED_PASSES` covers a materialised namespace too. - #[compio::test] - async fn a_re_parked_frame_ages_out_when_the_inbox_never_drains() { - let tmp = TempDir::new().expect("tempdir for system path"); - let config = test_config(&tmp); - let mux = TestMux::default(); - seed_stream(&mux, 1, "stream-repark-age"); - seed_topic(&mux, 2, 0, "topic-repark-age", vec![assignment(0, 1)]); - - let (shard, _inbox) = build_test_shard_with_inbox(0, &config, mux, 1); - let ctx = make_ctx(Rc::clone(&shard), 1, Rc::new(config)); - let ns = IggyNamespace::new(0, 0, 0); - - park_one_request(&shard, ns).await; - reconcile_pass(&ctx).await; - assert_eq!(shard.parked_frame_count(ns), 1, "re-parked on a full inbox"); - - for _ in 0..=PARK_MAX_PASSES { - reconcile_pass(&ctx).await; - } - assert_eq!( - shard.parked_frame_count(ns), - 0, - "a materialised namespace must still be aged, or the frame is stranded" - ); - assert!(!shard.has_parked_partition_frames()); - } - /// Mirrors `MAX_PARKED_PER_NAMESPACE` in `shard::park_if_unmaterialised`. const PARK_CAP: usize = 128; /// Mirrors `MAX_PARKED_BYTES`. diff --git a/core/shard/src/lib.rs b/core/shard/src/lib.rs index 98674232c8..a2ffd20a30 100644 --- a/core/shard/src/lib.rs +++ b/core/shard/src/lib.rs @@ -37,7 +37,6 @@ use consensus::{ }; #[cfg(any(test, feature = "simulator"))] use crossfire::AsyncRxTrait; -use crossfire::TrySendError; use futures::FutureExt; use iggy_binary_protocol::{ CHECKSUM_UNSEALED, Command, CommitHeader, ConsensusHeader, DoViewChangeHeader, @@ -70,7 +69,7 @@ use server_common::sharding::{IggyNamespace, PartitionLocation, ShardId}; use server_common::{MESSAGE_ALIGN, Message, MessageBag, iobuf::Frozen}; use shards_table::ShardsTable; use std::cell::{Cell, RefCell}; -use std::collections::{BTreeMap, BTreeSet, HashMap, VecDeque}; +use std::collections::{BTreeMap, HashMap, VecDeque}; use std::future::Future; use std::rc::Rc; #[cfg(feature = "simulator")] @@ -107,6 +106,27 @@ where pub clock: ConsensusClock, } +/// Committed metadata the simulator carries into one partition +/// materialisation. Named because both values are `u64`-compatible revision or +/// view stamps and swapping positional arguments would compile. +#[cfg(feature = "simulator")] +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct PartitionMaterialisation { + epoch: u64, + created_view: u32, +} + +#[cfg(feature = "simulator")] +impl PartitionMaterialisation { + #[must_use] + pub const fn new(epoch: u64, created_view: u32) -> Self { + Self { + epoch, + created_view, + } + } +} + /// Replica id + count bundle. /// /// Adjacent `u8` params (`self_replica_id`, `replica_count`) were a @@ -1395,18 +1415,17 @@ where /// admission, on the reactor thread inside the map's `borrow_mut`. parked_partition_bytes: Cell, - /// Namespaces holding frames [`Self::redispatch_parked_frames`] could not - /// re-queue, for the pump to retry. - /// - /// Without it a re-parked frame has no exit: its namespace is materialised - /// by then, so the sweep skips it and `reconcile_additions` stages no second - /// `InsertOwned`. Only a topic delete would reach it. The pump drains the - /// inbox, so a refusal usually clears on its next iteration. + /// Frames [`Self::redispatch_parked_frames`] handed back for the pump to + /// deliver, in park order. /// - /// [`BTreeSet`] for the reason [`Self::pending_partition_frames`] is a - /// [`BTreeMap`]: fixed-seed simulator replay needs iteration order to be a - /// function of the namespaces alone. - reparked_partition_namespaces: RefCell>, + /// Staging exists because re-dispatch runs inside the synchronous + /// [`Self::apply_reconcile_ops`] while the plane is reachable only through + /// an async path. A biased select arm takes one staged frame per pump + /// iteration and ranks above the inbox, so a parked op reaches the plane + /// ahead of a later op already sitting there. One-at-a-time delivery lets + /// consensus ticks and commit broadcasts run between frames instead of + /// stalling behind the whole bounded queue. + redispatch_queue: RefCell>, /// Set while the shard-wide budget is shedding for namespaces holding no /// park entry of their own, which have no [`ParkEntry::shed`] to warn once @@ -1589,7 +1608,7 @@ where reconcile_queue: RefCell::new(VecDeque::new()), pending_partition_frames: RefCell::new(BTreeMap::new()), parked_partition_bytes: Cell::new(0), - reparked_partition_namespaces: RefCell::new(BTreeSet::new()), + redispatch_queue: RefCell::new(VecDeque::new()), shard_park_shedding: Cell::new(false), metadata_repair: RefCell::new(None), metadata_transfer: RefCell::new(None), @@ -1919,7 +1938,7 @@ where reconcile_queue: RefCell::new(VecDeque::new()), pending_partition_frames: RefCell::new(BTreeMap::new()), parked_partition_bytes: Cell::new(0), - reparked_partition_namespaces: RefCell::new(BTreeSet::new()), + redispatch_queue: RefCell::new(VecDeque::new()), shard_park_shedding: Cell::new(false), metadata_repair: RefCell::new(None), metadata_transfer: RefCell::new(None), @@ -2010,6 +2029,13 @@ where /// queue never strands ops for longer than one tick. pub fn enqueue_reconcile_op(&self, op: ReconcileOp) { self.reconcile_queue.borrow_mut().push_back(op); + self.wake_reconcile_apply(); + } + + /// Wake the pump after off-pump work becomes visible. A refused marker is + /// safe because a full inbox has already woken the pump, whose frame and + /// tick arms both apply staged reconciliation work. + fn wake_reconcile_apply(&self) { let Some(sender) = self.senders.get(self.id as usize) else { return; }; @@ -2109,46 +2135,18 @@ where })); } - /// Re-drive the re-dispatch for namespaces whose frames the inbox refused. - /// - /// Runs on the pump, wherever [`Self::apply_reconcile_ops`] does, so it - /// fires right after a frame was consumed and a slot freed. Only another - /// refusal puts a namespace back, so the set empties itself. - /// - /// Epoch comes from the routing row, which `InsertOwned` writes alongside - /// the partition. Skipped when the row is gone or the namespace is fenced; - /// teardown does both, and the reconciler sweep retires the frames. - fn retry_reparked_frames(&self) { - let pending: Vec = { - let mut reparked = self.reparked_partition_namespaces.borrow_mut(); - if reparked.is_empty() { - return; - } - std::mem::take(&mut *reparked).into_iter().collect() - }; - let partitions = self.plane.partitions(); - for namespace in pending { - if partitions.is_tombstoned(&namespace) { - continue; - } - let Some(epoch) = self.shards_table.epoch_for(namespace) else { - continue; - }; - self.redispatch_parked_frames(namespace, epoch); - } - } - /// Drain and apply staged [`ReconcileOp`]s on the pump task. - /// Synchronous: every arm is in-memory only. `ConfirmRemove`'s fsync + - /// blocking close is offloaded to a detached task so the pump doesn't - /// stall on bulk teardown. + /// + /// Synchronous: every arm is in-memory only. `ConfirmRemove`'s fsync and + /// blocking close are offloaded to a detached task so the pump does not + /// stall on bulk teardown. An `InsertOwned` can stage parked frames, so a + /// live-pump caller must return to the ranked redispatch select arm before + /// reading the inbox again. The graceful-shutdown caller drains the queue + /// explicitly because it has already left the select loop. pub fn apply_reconcile_ops(&self) where B: MessageBus + 'static, { - // Ahead of the staged ops and outside their empty-queue early return: a - // re-parked frame waits on inbox capacity, not on a reconcile op. - self.retry_reparked_frames(); let staged: Vec> = { let mut q = self.reconcile_queue.borrow_mut(); if q.is_empty() { @@ -2386,7 +2384,8 @@ enum ParkOutcome { /// Namespace is unmaterialised and its park buffer is at capacity. Client /// requests must be denied with a transient status: the frame is gone, and /// silence would leave a lockstep transport waiting out its response - /// read-timeout. Replicated traffic is dropped, recovered by retransmit. + /// read-timeout. Replicated traffic is dropped and recovers through either + /// retransmit or the same-view repair armed by a later commit heartbeat. Overflow(Message), } @@ -2406,9 +2405,9 @@ struct ParkedFrame { /// answers CLIENT REQUESTS past [`MAX_PARKED_PASSES`], in units the /// simulator's virtual clock controls. /// - /// Never expires a replicated prepare: no client to answer, and - /// `consensus::retransmit_targets` skips an op that already reached quorum, - /// so expiry is silent permanent loss. Byte budgets bound those instead. + /// Never expires a replicated prepare: no client can retry it, and forcing + /// the same-view repair backstop to recover a gap is worse than retaining + /// the frame. Byte budgets bound those instead. /// /// Bounds RESIDENCY, not staleness. The SDK replays the identical request /// for the rest of its response timeout, so an absolute-offset @@ -2430,6 +2429,20 @@ impl ParkedFrame { } } +/// What a frame keeps if it parks again after the pump re-delivers it. +/// +/// Production prevents that race by ranking redispatch above inbox work and by +/// applying reconcile operations only on the pump. Carrying the original stamp +/// is defence in depth for off-pump staging such as simulator materialisation. +/// There, re-deriving on a second park could stamp the frame with a replacement +/// incarnation. `None` also stays `None`, since absence of a committed revision +/// is not evidence that the frame belongs to a prior incarnation. +#[derive(Clone, Copy)] +struct ParkProvenance { + epoch: Option, + passes: u32, +} + /// One namespace's parked frames plus their running footprint. /// /// Carried, not re-summed: `park_if_unmaterialised` reads it per arriving frame @@ -2492,11 +2505,11 @@ const MAX_PARKED_BYTES: usize = 16 * 1024 * 1024; /// cannot spend the whole shard's budget and shed everyone else's frames. /// /// Applied only to an entry that already holds something. Sized against an -/// empty entry a larger frame could never park at all, and for a prepare that is -/// unrecoverable loss: `consensus::retransmit_targets` skips an op that already -/// reached quorum. Shipped `message_bus.max_message_size` is 64 MiB, so an -/// ordinary batched append exceeds this. Cost of the waiver is one convergence -/// window of shard budget; cost of the loss is the replica. +/// empty entry a larger frame could never park at all. For a prepare, shedding +/// also forces a later commit heartbeat to discover the gap and run same-view +/// repair. Shipped `message_bus.max_message_size` is 64 MiB, so an ordinary +/// batched append exceeds this. The waiver costs one convergence window of +/// shard budget and avoids unnecessary recovery work. const MAX_PARKED_BYTES_PER_NAMESPACE: usize = MAX_PARKED_BYTES / 4; /// Resident cost of parking a frame of `len` bytes. @@ -2553,14 +2566,99 @@ where MJ: JournalHandle, ::Target: Journal, Header = PrepareHeader>, - M: StateMachine< - Input = Message, - Output = metadata::stm::result::ApplyReply, - Error = iggy_common::IggyError, - > + StreamsFrontend - + metadata::stm::snapshot::RestoreSnapshotInPlace< - metadata::stm::snapshot::MetadataSnapshot, - >, + M: RestorableMetadataStm, + T: ShardsTable, + { + self.dispatch_message(message, None).await; + } + + /// Remove and classify one staged frame for the pump's ranked redispatch + /// arm. The queue borrow ends before dispatch awaits, so simulator + /// materialisation can append off-pump without colliding with a suspended + /// `RefCell` guard. + fn pop_redispatched_frame(&self) -> Option<(MessageBag, ParkProvenance)> { + loop { + let ParkedFrame { + epoch, + passes, + message, + } = self.redispatch_queue.borrow_mut().pop_front()?; + let provenance = ParkProvenance { epoch, passes }; + // Parked frames are stored generic (the buffer holds every variant + // in one Vec), so re-entering the pump costs one classify. That is + // the rare path - a post-`CreateTopic` convergence window, not the + // per-message steady state the bag handoff exists for. + match MessageBag::try_from(message) { + Ok(bag) => return Some((bag, provenance)), + Err(error) => { + // The frame classified once already, on the way in, so this + // is unreachable short of memory corruption. The consumed + // bytes can no longer produce a client deny, but the drop + // still needs the same operator-visible record as any other + // parked frame retired unserved. + self.metrics.record_frame_drop( + crate::metrics::frame_drop_variant::PARTITION, + crate::metrics::frame_drop_reason::PARK_DROPPED, + ); + tracing::error!( + shard = self.id, + %error, + "re-dispatched partition frame no longer classifies; dropping it" + ); + } + } + } + } + + /// Test-only delivery of one staged frame. Production obtains frames through + /// the router's ranked select arm, which also processes loopback after each + /// one. This hook exists for the reconciler's defence-in-depth interleaving. + #[cfg(feature = "simulator")] + #[allow(clippy::future_not_send)] + pub async fn dispatch_one_redispatched_frame_for_test(&self) -> bool + where + B: MessageBus + 'static, + MJ: JournalHandle, + ::Target: + Journal, Header = PrepareHeader>, + M: RestorableMetadataStm, + T: ShardsTable, + { + let Some((message, provenance)) = self.pop_redispatched_frame() else { + return false; + }; + self.dispatch_message(message, Some(provenance)).await; + true + } + + /// Retire staged frames the pump is no longer going to deliver, on its way + /// out. Client requests get a transient deny; the rest are counted as drops, + /// which is the only record a replicated frame leaves. + fn retire_redispatched_frames(&self) { + let staged: Vec = self.redispatch_queue.borrow_mut().drain(..).collect(); + if staged.is_empty() { + return; + } + let (answered, dropped) = self.retire_parked_frames(staged); + tracing::debug!( + shard = self.id, + answered, + dropped, + "retiring re-dispatched partition frames the pump will not deliver" + ); + } + + /// [`Self::on_message`] carrying the park provenance of a frame the pump is + /// re-delivering, so a second park keeps the stamp and age the first one + /// derived instead of deriving them again against newer committed state. + #[allow(clippy::future_not_send)] + async fn dispatch_message(&self, message: MessageBag, provenance: Option) + where + B: MessageBus + 'static, + MJ: JournalHandle, + ::Target: + Journal, Header = PrepareHeader>, + M: RestorableMetadataStm, T: ShardsTable, { match message { @@ -2570,7 +2668,7 @@ where let header = request.header(); (header.operation, header.group) }; - match self.park_if_unmaterialised(request, routing.0, routing.1) { + match self.park_if_unmaterialised(request, routing.0, routing.1, provenance) { // The incarnation fence runs only here, on client traffic. // A backup denying what the primary admitted would diverge // the replicas, so replicated frames are never fenced. @@ -2601,7 +2699,7 @@ where // A tombstoned prepare still flows to the plane: replicated // traffic has no client awaiting a reply on this node, and // the plane's own tombstone guard drops it. - match self.park_if_unmaterialised(prepare, routing.0, routing.1) { + match self.park_if_unmaterialised(prepare, routing.0, routing.1, provenance) { ParkOutcome::Deliver(prepare) | ParkOutcome::Tombstoned(prepare) => { self.on_replicate(prepare).await; // A follower learns the cluster commit point from the @@ -2806,13 +2904,10 @@ where } } - /// Remove a namespace's entry, debiting [`Self::parked_partition_bytes`] and - /// disarming the pump retry. Single place an entry leaves the map, so - /// neither can drift out of step with it. + /// Remove a namespace's entry, debiting [`Self::parked_partition_bytes`]. + /// Single place an entry leaves the map, so the two cannot drift out of step + /// with each other. fn take_parked_partition_frames(&self, namespace: IggyNamespace) -> Option> { - self.reparked_partition_namespaces - .borrow_mut() - .remove(&namespace); let (entry, converged) = { let mut pending = self.pending_partition_frames.borrow_mut(); let entry = pending.remove(&namespace)?; @@ -2839,19 +2934,28 @@ where let mut answered = 0; let mut dropped = 0; for frame in frames { - if self.deny_parked_client_request(frame) { + if self.retire_parked_frame(frame) { answered += 1; } else { dropped += 1; - self.metrics.record_frame_drop( - crate::metrics::frame_drop_variant::PARTITION, - crate::metrics::frame_drop_reason::PARK_DROPPED, - ); } } (answered, dropped) } + /// Answer one parked request or count one replicated frame as destroyed. + /// Returns `true` only when a client deny reached the pump. + fn retire_parked_frame(&self, frame: ParkedFrame) -> bool { + if self.deny_parked_client_request(frame) { + return true; + } + self.metrics.record_frame_drop( + crate::metrics::frame_drop_variant::PARTITION, + crate::metrics::frame_drop_reason::PARK_DROPPED, + ); + false + } + /// Whether any frame is parked. Cheap enough for the reconciler's per-tick /// fast-skip guard: a non-empty buffer means the shard is by definition not /// converged, so the skip must not fire. @@ -2877,9 +2981,8 @@ where .collect() } - /// Re-queue the frames parked for `namespace` now that its partition exists - /// at `epoch`, onto this shard's own inbox so the pump serves them after the - /// current drain. + /// Hand the frames parked for `namespace` back to the pump, in park order, + /// now that its partition exists at `epoch`. /// /// A frame stamped with a DIFFERENT incarnation never makes it back: the /// namespace is byte-identical across incarnations, so serving it would land @@ -2893,35 +2996,27 @@ where /// buffer exists to absorb -- the partition primary materialises and /// replicates as soon as its own metadata commits, well before a lagging /// backup applies the same commit. Treating that as "prior incarnation" - /// destroys live traffic: a replicated prepare has no client to answer, so - /// it would be dropped with no recovery until an unrelated view change. + /// destroys live traffic and forces the backup to recover a gap that the + /// park path could have delivered directly. /// The residual is unchanged from before the stamp existed -- a frame parked /// while the namespace was absent, then recreated under a new incarnation, /// is served against the replacement -- and closing it needs a wire-level /// discriminator (see the `TODO(krishna)` in /// `partition_reconciler`'s module docs), not a `None`-means-stale rule. /// - /// A frame the inbox refuses is re-parked: retained, so not counted as a - /// drop. Re-queuing appends, so a pass materialising many namespaces can - /// overrun the inbox; staging a deny is futile because it rides the same - /// sender with no await in between. One namespace alone can now do it, since - /// [`MAX_PARKED_BYTES_PER_NAMESPACE`] admits 1024 header-only prepares - /// against a default `inbox_capacity` of 1024. That costs other namespaces a - /// later convergence, not a frame. The first `Full` ends the loop, since - /// the sole consumer of `senders[self.id]` is the pump task running this - /// call and no later frame can find a slot the first one could not. + /// Staged onto [`Self::redispatch_queue`] rather than sent: the shard's own + /// inbox can already hold a LATER op of this partition, and the plane's + /// backup gap check drops anything that is not `current_op + 1`, so + /// appending would strand the parked op behind an op that will be dropped + /// for arriving too early. The pump's biased redispatch arm ranks above its + /// inbox arm and delivers one staged frame per iteration. /// - /// [`MAX_PARKED_PASSES`] does not bound a re-parked frame: the sweep ages a - /// namespace only while un-materialised, and by here it is materialised. - /// [`Self::repark_partition_frames`] arms the pump retry instead; - /// `partition_reconciler::reconcile_parked_frames` is the backstop for an - /// inbox that never drains. - fn redispatch_parked_frames(&self, namespace: IggyNamespace, epoch: u64) - where - B: MessageBus + 'static, - { + /// [`MAX_PARKED_PASSES`] does not bound a staged frame: it has left the park + /// buffer, and the pump selects the queue on the iteration after it was + /// filled. Returns whether at least one frame was staged. + fn redispatch_parked_frames(&self, namespace: IggyNamespace, epoch: u64) -> bool { let Some(frames) = self.take_parked_partition_frames(namespace) else { - return; + return false; }; tracing::debug!( shard = self.id, @@ -2930,8 +3025,6 @@ where epoch, "re-dispatching parked partition frames after materialisation" ); - // Incarnation filter first, independent of the sender: a prior - // incarnation is rejected whether or not this shard can re-queue. let mut servable: Vec = Vec::with_capacity(frames.len()); for frame in frames { // Only a stamp that exists and disagrees is evidence of a prior @@ -2944,119 +3037,21 @@ where servable.push(frame); } } - let Some(sender) = self.senders.get(self.id as usize) else { - self.retire_parked_frames(servable); - return; - }; - let mut refused_frames: Vec = Vec::new(); - let mut remaining = servable.into_iter(); - while let Some(frame) = remaining.next() { - let passes = frame.passes; - let parked_epoch = frame.epoch; - // Parked frames are stored generic (the buffer holds every variant - // in one Vec), so re-entering the pump costs one classify. That is - // the rare path -- a post-`CreateTopic` convergence window, not the - // per-message steady state the bag handoff exists for. - let bag = match MessageBag::try_from(frame.message) { - Ok(bag) => bag, - Err(error) => { - // The frame classified once already, on the way in, so this - // is unreachable short of memory corruption. Dropping it - // costs a client retry; panicking on the reconciler's path - // would take the shard down. - tracing::error!( - shard = self.id, - namespace_raw = namespace.inner(), - %error, - "parked partition frame no longer classifies; dropping it" - ); - continue; - } - }; - let Err(error) = sender.try_send(ShardFrame::consensus(self.id, bag)) else { - continue; - }; - let (refused, disconnected) = match error { - TrySendError::Full(frame) => (frame, false), - TrySendError::Disconnected(frame) => (frame, true), - }; - let ShardFrame::Consensus { message, .. } = refused else { - unreachable!("try_send returns the frame it was handed"); - }; - let refused_frame = ParkedFrame { - epoch: parked_epoch, - passes, - message: message.into_generic(), - }; - if disconnected { - // Pump gone: re-parking holds the frame until process exit, and - // every later send hits the same dead channel. - self.metrics.record_frame_drop( - crate::metrics::frame_drop_variant::PARTITION, - crate::metrics::frame_drop_reason::DISCONNECTED, - ); - tracing::warn!( - shard = self.id, - namespace_raw = namespace.inner(), - "re-dispatch of parked partition frames refused: inbox disconnected" - ); - let mut stranded = vec![refused_frame]; - stranded.extend(remaining); - self.retire_parked_frames(stranded); - return; - } - refused_frames.push(refused_frame); - refused_frames.extend(remaining); - tracing::debug!( - shard = self.id, - namespace_raw = namespace.inner(), - count = refused_frames.len(), - passes, - "re-parking parked partition frames: inbox full" - ); - break; - } - if !refused_frames.is_empty() { - self.repark_partition_frames(namespace, refused_frames); - } - } - - /// Put frames back under `namespace` after a refused re-dispatch, keeping - /// [`Self::parked_partition_bytes`] in step and arming the pump-side retry. - /// - /// Deliberately not budget-checked: these bytes were already counted while - /// parked, so re-admitting them cannot grow the total past what it held a - /// moment ago, and shedding here would answer a frame the inbox merely - /// deferred. - /// - /// Arming [`Self::reparked_partition_namespaces`] is what makes it a - /// deferral. Every other exit is closed once materialised: the sweep only - /// ages a namespace it has not built, and `reconcile_additions` stages no - /// second `InsertOwned` for one already in `IggyPartitions`. - fn repark_partition_frames(&self, namespace: IggyNamespace, frames: Vec) { - let restored: usize = frames.iter().map(ParkedFrame::footprint).sum(); - let mut pending = self.pending_partition_frames.borrow_mut(); - let entry = pending.entry(namespace).or_default(); - for frame in frames { - entry.push(frame); - } - drop(pending); - self.parked_partition_bytes - .set(self.parked_partition_bytes.get().saturating_add(restored)); - self.reparked_partition_namespaces - .borrow_mut() - .insert(namespace); + let staged = !servable.is_empty(); + self.redispatch_queue.borrow_mut().extend(servable); + staged } /// Age every frame under `namespace` by one pass, answering CLIENT REQUESTS /// past `MAX_PARKED_PASSES`. Returns the number answered. /// - /// Prepares age but never expire. Expiry destroys a committed op with - /// nothing to recover it (see `ParkedFrame::passes`), and passes are - /// commit-driven: a non-empty buffer defeats the reconciler fast-skip, so a - /// create burst elapses four in milliseconds, across every parked namespace - /// rather than the one it concerns. Byte budgets bound them instead. Only - /// [`Self::discard_parked_partition_frames`] still destroys a prepare. + /// Prepares age but never expire. Expiry would manufacture a gap that a + /// later commit heartbeat must repair (see `ParkedFrame::passes`), and + /// passes are commit-driven: a non-empty buffer defeats the reconciler + /// fast-skip, so a create burst elapses four in milliseconds across every + /// parked namespace rather than the one it concerns. Byte budgets bound + /// them instead. Only [`Self::discard_parked_partition_frames`] still + /// destroys a prepare. /// /// Passes, not wall-clock, so the simulator's virtual clock governs it. pub fn age_parked_partition_frames(&self, namespace: IggyNamespace) -> usize { @@ -3073,7 +3068,7 @@ where let emptied = entry.frames.is_empty(); drop(pending); if emptied { - // Through the shared remover so the pump-retry set is disarmed + // Through the shared remover so the shed-episode flag clears // with it; the entry is already empty, so this only unhooks it. self.take_parked_partition_frames(namespace); } @@ -3112,6 +3107,17 @@ where .map_or(0, |entry| entry.frames.len()) } + /// How many frames are staged for the pump to re-deliver. + /// + /// Test/simulator accessor, gated for the same reason as + /// [`Self::parked_frame_count`]: the pump consumes this queue through a + /// dedicated select arm, so no production caller has a depth to branch on. + #[cfg(any(test, feature = "simulator"))] + #[must_use] + pub fn redispatched_frame_count(&self) -> usize { + self.redispatch_queue.borrow().len() + } + /// Retire a frame that will never be served: a client request gets a /// transient deny, replicated traffic is destroyed. Returns `true` only when /// a reply reached the pump. @@ -3121,9 +3127,9 @@ where /// with no sender stages nothing, hence forwarding /// [`Self::stage_transient_deny`]'s verdict rather than assuming success. /// - /// No reply must not mean no record: the primary retransmits only what has - /// not reached quorum, so a destroyed prepare is invisible loss. The `false` - /// return is what makes callers bump + /// No reply must not mean no record: the primary may no longer retransmit an + /// op that reached quorum, so a destroyed prepare creates a gap that later + /// repair must fill. The `false` return is what makes callers bump /// `frame_drops_total{variant=partition,reason=park_dropped}`. fn deny_parked_client_request(&self, frame: ParkedFrame) -> bool { if frame.message.header().command == Command::Request @@ -3175,7 +3181,7 @@ where "rejecting parked partition frame from a prior incarnation" ); } - self.deny_parked_client_request(frame); + self.retire_parked_frame(frame); } /// Park a partition-plane frame whose namespace this shard has not yet @@ -3186,16 +3192,21 @@ where /// disk delete) report [`ParkOutcome::Tombstoned`] so the caller can deny /// client requests instead of feeding them to the plane's silent-drop /// guard, while replicated traffic still flows there. Parked frames are - /// re-dispatched by [`Self::apply_reconcile_ops`] once the matching + /// staged for the pump by [`Self::apply_reconcile_ops`] once the matching /// `ReconcileOp::InsertOwned` lands, and only if the epoch stamped here /// still matches (see [`ParkedFrame`]); a full buffer reports /// [`ParkOutcome::Overflow`] so the caller can answer rather than shed /// silently. + /// + /// `provenance` is `None` for a frame arriving off the wire and `Some` for + /// one the pump is re-delivering, which must keep the stamp and the age it + /// parked with (see [`ParkProvenance`]). fn park_if_unmaterialised( &self, message: Message, operation: Operation, namespace_raw: u64, + provenance: Option, ) -> ParkOutcome where H: iggy_binary_protocol::ConsensusHeader, @@ -3217,13 +3228,18 @@ where } // Read the committed revision before taking the borrow below: the frame // is stamped with the incarnation it was addressed to, so a later drain - // can tell it apart from a same-key replacement. - let epoch = self - .plane - .metadata() - .mux_stm - .streams() - .created_revision_for_namespace(namespace); + // can tell it apart from a same-key replacement. A re-delivered frame + // brings its own, since by now the committed revision can describe the + // replacement rather than the incarnation the frame was addressed to. + let ParkProvenance { epoch, passes } = provenance.unwrap_or_else(|| ParkProvenance { + epoch: self + .plane + .metadata() + .mux_stm + .streams() + .created_revision_for_namespace(namespace), + passes: 0, + }); let frame_cost = parked_footprint(message.as_slice().len()); let replicated = message.header().command() != Command::Request; let mut pending = self.pending_partition_frames.borrow_mut(); @@ -3234,13 +3250,12 @@ where let existing = pending.get_mut(&namespace); let parked_len = existing.as_ref().map_or(0, |entry| entry.frames.len()); let namespace_bytes = existing.as_ref().map_or(0, |entry| entry.bytes); - // A prepare is never shed on a byte budget. No client to answer, and no - // recovery: `consensus::retransmit_targets` skips an op that already - // reached quorum and the plane opens a repair session only in - // `on_start_view`, so shedding one is permanent loss where shedding a - // request costs a retry. A request is refused the moment admitting it - // would cross a budget; a prepare only once one is already spent. Caps - // prepare residency at one frame of overshoot per budget (worst case + // A prepare is never shed on a byte budget before the budget is spent. + // It has no client to retry it, and recovery requires a later commit + // heartbeat to expose the gap and arm same-view repair. A request costs + // only a retry, so it is refused the moment admitting it would cross a + // budget. This caps prepare residency at one frame of overshoot per + // budget (worst case // `MAX_PARKED_BYTES` + `max_message_size`, 80 MiB per shard) instead of // at the budget, and is what makes an oversize frame parkable at all. let namespace_budget_spent = parked_len > 0 @@ -3317,7 +3332,7 @@ where ); pending.entry(namespace).or_default().push(ParkedFrame { epoch, - passes: 0, + passes, message: message.into_generic(), }); drop(pending); @@ -3556,8 +3571,12 @@ where /// reports `commit_offset` 0, which reads as a regression rather than a /// harness that discarded the log. /// - /// `created_view` is the view the metadata plane created the namespace in; - /// see `fresh_group_start`. + /// `materialisation` carries the committed `created_revision` and the view + /// the metadata plane created the namespace in; see `fresh_group_start`. + /// + /// Once inserted, this also runs the same parked-frame redispatch as + /// `ReconcileOp::InsertOwned`. The simulator bypasses the reconciler build, + /// but it must not bypass the pump handoff that follows materialisation. // `feature = "simulator"` alone, unlike its neighbours: the body names items // `partitions` gates the same way, and a `test` arm cannot turn those on. // Under `cargo test -p shard` that arm fires from shard's own `cfg(test)` @@ -3572,10 +3591,15 @@ where recovered_state: Option, retained: Option, restore_frontier: bool, - created_view: u32, + materialisation: PartitionMaterialisation, ) where - B: MessageBus + Clone, + B: MessageBus + Clone + 'static, + T: ShardsTable, { + let PartitionMaterialisation { + epoch, + created_view, + } = materialisation; let partitions = self.plane.partitions(); if partitions.contains(&namespace) { return; @@ -3678,6 +3702,12 @@ where // store resumes minting at 0 while its group is at N. partition.restore_offset_frontier(recovered_state.as_ref()); partitions.insert(namespace, partition); + if self.redispatch_parked_frames(namespace, epoch) { + // This mutation occurs outside the pump, unlike production's + // `InsertOwned`. Wake the ranked redispatch arm so quiescence does + // not leave real work staged without a poll source. + self.wake_reconcile_apply(); + } } /// Resolve the single partition a VSR control frame addresses, keyed by diff --git a/core/shard/src/metrics.rs b/core/shard/src/metrics.rs index cd05dba02a..16007ffec1 100644 --- a/core/shard/src/metrics.rs +++ b/core/shard/src/metrics.rs @@ -27,8 +27,8 @@ //! by plane. //! - `IggyShard::park_if_unmaterialised` - partition frames shed because the //! park buffer is at its frame or byte cap. -//! - `IggyShard::apply_reconcile_ops` - parked frames whose re-dispatch onto -//! this shard's own inbox was refused. +//! - `IggyShard::retire_parked_frames` - parked frames retired with no client +//! to answer. //! //! The counter uses atomic interior mutability, safe to bump from `!Send` //! compio reactor contexts. Each shard owns its own instance, and the server @@ -77,18 +77,14 @@ pub struct FrameDropLabel { /// /// `PARTITION` covers the partition plane: a frame shed because the namespace /// had not materialised and its park buffer was at capacity -/// (`reason=park_overflow`), a re-dispatch the shard's own inbox refused, or a -/// routing send the target inbox refused. A shed client request is answered with -/// a retriable status, so the client recovers -- but a shed *prepare* is not -/// covered by retransmit once its op has reached quorum -/// (`consensus::retransmit_targets` skips `ok_quorum_received`, and the -/// partition plane creates a repair session only in `on_start_view`), so it -/// leaves that backup behind until an unrelated view change. -// -// TODO(krishna): give the partition plane a normal-status repair driver so a -// shed or refused prepare is repaired without waiting for a view change. Until -// then `variant=partition` is the only signal that a backup may be stranded -// behind `commit_max`. +/// (`reason=park_overflow`), a parked frame retired with no client to answer +/// (`reason=park_dropped`), an incarnation rejection, or a routing send the +/// target inbox refused. A shed client request is answered with a retriable +/// status. A shed prepare may no longer be covered by retransmit once its op +/// reached quorum, but a later `CommitMessage` that advances the backup's +/// frontier arms same-view journal repair. If the primary evicted the range, +/// repair escalates to partition state transfer. The counter therefore signals +/// a data-plane gap or recovery burden, not a requirement for a view change. pub mod frame_drop_variant { pub const CONSENSUS: &str = "consensus"; pub const FD_TRANSFER: &str = "fd_transfer"; @@ -114,11 +110,12 @@ pub mod frame_drop_variant { /// `PARK_OVERFLOW` ticks when a partition frame arrives for a namespace this /// shard has not materialised and the per-namespace park buffer is already at /// its cap, so the frame is shed with no reply. `PARK_DROPPED` ticks when a -/// frame that did park leaves unserved: its namespace became unreachable, or a -/// request outlived `MAX_PARKED_PASSES` with no pump to take the deny. A request -/// also bumps `partition_requests_denied_transient_total` when answered; -/// replicated traffic has nobody to answer, so this is the only record the op -/// was destroyed. +/// frame that did park leaves unserved: its namespace became unreachable, its +/// incarnation stamp was rejected, reclassification failed, or a request +/// outlived `MAX_PARKED_PASSES` with no pump to take the deny. A request also +/// bumps `partition_requests_denied_transient_total` when answered; replicated +/// traffic has nobody to answer, so this is the direct record that local bytes +/// were destroyed and repair may be required. pub mod frame_drop_reason { /// Operation discriminant unknown to this build: the sender is newer. /// diff --git a/core/shard/src/router.rs b/core/shard/src/router.rs index c98e585303..a866a9027c 100644 --- a/core/shard/src/router.rs +++ b/core/shard/src/router.rs @@ -30,8 +30,10 @@ use message_bus::{ConnectionInstaller, MessageBus, ReplicaHandshakeDoneFn}; use partitions::FatalCommit; use server_common::sharding::{IggyNamespace, METADATA_GROUP}; use server_common::{Message, MessageBag}; +use std::future::poll_fn; use std::sync::Arc; use std::sync::atomic::{AtomicBool, Ordering}; +use std::task::Poll; /// How often the shard pump drives `VsrConsensus::tick`. /// @@ -306,7 +308,8 @@ where // `select_biased!`, not `select!`: the unbiased macro draws its // arm order from a process-random thread-local PRNG, which the // deterministic simulator cannot seed. The listed order is the - // intended priority anyway: stop, then tick, then frames. + // intended priority anyway: stop, then tick, redispatch, then + // newly received frames. futures::select_biased! { _ = stop.recv().fuse() => break, () = consensus_tick.as_mut() => { @@ -348,6 +351,27 @@ where self.apply_reconcile_ops(); consensus_tick.set(rearm_tick()); } + (message, provenance) = poll_fn(|_| { + self.pop_redispatched_frame().map_or(Poll::Pending, Poll::Ready) + }).fuse() => { + // One frame per select iteration. Ranking this arm above + // the inbox preserves park order without making a full + // queue stall ticks and commit broadcasts for other groups. + self.dispatch_message(message, Some(provenance)).await; + // A request handled by a solo primary self-acks here. If + // loopback waited for another inbox frame, the request + // would remain uncommitted indefinitely on a quiet shard. + self.process_loopback(&mut loopback_buf, &mut namespace_scratch).await; + self.apply_reconcile_ops(); + // Same guaranteed reply-lane service as the inbox arm: this + // arm outranks both lanes, so a deep drain would otherwise + // starve forwarded client replies for its whole duration. + if let Ok(reply) = self.reply_inbox.try_recv() + && self.accept_frame_for_self(&reply) + { + self.process_frame(reply).await; + } + } frame = self.inbox.recv().fuse() => { match frame { Ok(frame) => { @@ -406,29 +430,9 @@ where // clients time out, which is what a node stopping on a durability // fault owes them. if fatal.is_none() { - while let Ok(frame) = self.inbox.try_recv() { - if self.accept_frame_for_self(&frame) { - self.process_frame(frame).await; - self.process_loopback(&mut loopback_buf, &mut namespace_scratch) - .await; - self.apply_reconcile_ops(); - if let Some(fault) = self.first_partition_commit_fault() { - fatal = Some(fault); - break; - } - } - } - } - if fatal.is_none() { - while let Ok(frame) = self.reply_inbox.try_recv() { - if self.accept_frame_for_self(&frame) { - self.process_frame(frame).await; - if let Some(fault) = self.first_partition_commit_fault() { - fatal = Some(fault); - break; - } - } - } + fatal = self + .drain_queued_frames_for_shutdown(&mut loopback_buf, &mut namespace_scratch) + .await; } if fatal.is_some() { @@ -459,6 +463,59 @@ where fatal } + /// Process queued work after the select loop has stopped. Redispatch keeps + /// its live-pump rank over the inbox, and every delivered frame gets its + /// loopback before another frame can run. + #[allow(clippy::future_not_send)] + async fn drain_queued_frames_for_shutdown( + &self, + loopback_buf: &mut Vec>, + namespace_scratch: &mut Vec, + ) -> Option + where + B: MessageBus + 'static, + MJ: JournalHandle, + ::Target: + Journal, Header = PrepareHeader>, + M: RestorableMetadataStm, + { + loop { + while let Some((message, provenance)) = self.pop_redispatched_frame() { + self.dispatch_message(message, Some(provenance)).await; + self.process_loopback(loopback_buf, namespace_scratch).await; + self.apply_reconcile_ops(); + if let Some(fault) = self.first_partition_commit_fault() { + return Some(fault); + } + } + let Ok(frame) = self.inbox.try_recv() else { + break; + }; + if self.accept_frame_for_self(&frame) { + self.process_frame(frame).await; + self.process_loopback(loopback_buf, namespace_scratch).await; + self.apply_reconcile_ops(); + if let Some(fault) = self.first_partition_commit_fault() { + return Some(fault); + } + } + } + + // Retired, not delivered: the pump is going away, so a staged frame has + // no later arm to reach the plane through. This runs before the reply + // lane drain, which is what carries client denies out. + self.retire_redispatched_frames(); + while let Ok(frame) = self.reply_inbox.try_recv() { + if self.accept_frame_for_self(&frame) { + self.process_frame(frame).await; + if let Some(fault) = self.first_partition_commit_fault() { + return Some(fault); + } + } + } + None + } + /// First partition commit fault currently fenced on this shard. /// /// The regular path observes faults in `tick_partitions`. This scan covers @@ -527,6 +584,10 @@ where async fn process_lifecycle(&self, payload: LifecycleFrame) where B: MessageBus + 'static, + MJ: JournalHandle, + ::Target: + Journal, Header = PrepareHeader>, + M: RestorableMetadataStm, { match payload { LifecycleFrame::ReplicaInboundSetup { fd, slot } => { diff --git a/core/simulator/src/lib.rs b/core/simulator/src/lib.rs index 7b722eb9bf..fda8e5bcfd 100644 --- a/core/simulator/src/lib.rs +++ b/core/simulator/src/lib.rs @@ -50,8 +50,8 @@ use replica::{Replica, SIM_INBOX_CAPACITY, new_shard}; use seeds::SimSeeds; use server_common::Message; use server_common::sharding::{IggyNamespace, PartitionLocation, ShardId}; -use shard::CONSENSUS_TICK_INTERVAL; use shard::shards_table::{ShardsTable, calculate_shard_assignment}; +use shard::{CONSENSUS_TICK_INTERVAL, PartitionMaterialisation}; use std::cell::RefCell; use std::collections::{HashMap, HashSet}; use std::net::{IpAddr, Ipv4Addr, SocketAddr}; @@ -902,12 +902,13 @@ impl Simulator { } } - /// Lost-wake tripwire. At executor quiescence every live pump must have drained - /// its inbox; a non-empty one on a non-crashed replica means a frame reached the - /// channel without waking the target pump. Every pump holds a standing + /// Lost-work tripwire. At executor quiescence every live pump must have drained + /// both inbox lanes and its parked-frame redispatch queue. A non-empty lane on a + /// non-crashed replica means a frame reached the channel without waking the + /// target pump. A non-empty redispatch queue means a materialisation staged work + /// without the pump's priority drain completing. Every pump holds a standing /// `CONSENSUS_TICK_INTERVAL` timer, so the next `advance_time` would re-poll and - /// silently drain it, masking the exact wake-loss class this harness exists to - /// catch. Trip here instead. + /// silently drain either case, masking the fault. Trip here instead. /// /// Incomplete by construction: only catches a lost wakeup while the un-woken /// frame is still queued at quiescence, since a later frame that does wake the @@ -943,6 +944,16 @@ impl Simulator { self.seed, self.executor.schedule_hash(), ); + let pending_redispatch = shard.redispatched_frame_count(); + assert_eq!( + pending_redispatch, + 0, + "lost redispatch: replica {replica_id} shard {} redispatch queue holds \ + {pending_redispatch} frame(s) at quiescence (seed {:#x}, schedule hash {:#x})", + shard.id, + self.seed, + self.executor.schedule_hash(), + ); } } } @@ -1459,7 +1470,7 @@ fn materialise_partition( recovered_state, retained, restore_frontier, - created_view, + PartitionMaterialisation::new(epoch, created_view), ); for shard in &replica.shards { shard.shards_table().insert( @@ -2432,11 +2443,10 @@ mod tests { /// behind VSR retransmit. /// /// `park_overflow` is deliberately NOT excluded. The reconciler is unwired here - /// (`init_partition` mirrors its outcome directly), so nothing drains the park - /// buffer and a parked frame is never re-dispatched or swept. Non-zero - /// `park_overflow` therefore means frames shed for a namespace that will never - /// materialise, which is the fault class this catches rather than the - /// back-pressure it would be in production. + /// and only an explicit `init_partition` mirrors its materialisation outcome and + /// drains the matching park entry. Non-zero `park_overflow` therefore means + /// frames shed for a namespace the driver never materialised, which is the fault + /// class this catches rather than the back-pressure it would be in production. /// /// Same reason the park buffer must be empty at quiescence: a frame still parked /// has no drainer, so it is neither delivered nor answered. @@ -2451,7 +2461,8 @@ mod tests { assert!( shard.parked_namespaces().is_empty(), "replica {replica_idx} shard {shard_idx} left partition frames parked; the \ - simulator wires no reconciler, so nothing will deliver or answer them" + simulator wires no reconciler, so only explicit materialisation can \ + deliver or answer them" ); } } @@ -2539,6 +2550,222 @@ mod tests { ); } + fn successful_send_reply_count(replies: &[Message]) -> usize { + replies + .iter() + .filter(|reply| { + reply.header().operation == iggy_binary_protocol::Operation::SendMessages + && reply.header().status == 0 + }) + .count() + } + + fn retained_prepare( + sim: &Simulator, + replica: usize, + namespace: IggyNamespace, + op: u64, + ) -> Message { + let partitions = sim.replicas[replica] + .partition_shard(namespace) + .plane + .partitions(); + let partition = partitions.get_by_ns(&namespace).expect("partition"); + let stored = partition + .log + .journal() + .inner + .repair_entry(op) + .unwrap_or_else(|| panic!("replica {replica} retains prepare op {op} for repair")); + Message::::try_from(server_common::iobuf::Owned::< + { server_common::MESSAGE_ALIGN }, + >::copy_from_slice( + stored.as_slice() + )) + .expect("journaled prepare remains a valid frame") + } + + /// Leave one backup without the partition while the primary commits two + /// sends through the other backup. The lagging replica parks both replicated + /// prepares. A third prepare is withheld from it, then placed on its inbox + /// after simulator materialisation stages the parked prefix. Running the pump + /// without advancing its tick forces the inbox arm to drain the prefix first. + fn parked_prepare_redispatch_trace(seed: u64) -> (usize, Vec, u64) { + const CLIENT_ID: u128 = 1; + + server_common::MemoryPool::init_pool(&server_common::MemoryPoolSettings { + enabled: false, + size: iggy_common::IggyByteSize::from(0u64), + bucket_capacity: 1, + }); + + let network_opts = packet::PacketSimulatorOptions { + node_count: 3, + client_count: 1, + seed, + ..packet::PacketSimulatorOptions::default() + }; + let mut sim = Simulator::with_shards_shell(3, 1, std::iter::once(CLIENT_ID), network_opts); + let namespace = IggyNamespace::new(0, 0, 0); + sim.seed_stream_topic_partition(namespace); + let created_view = sim.partition_created_views[&namespace]; + + // Replica 0 is the view-0 primary. Replica 2 supplies quorum while + // replica 1 has committed metadata but no local partition yet. + materialise_partition(&sim.replicas[0], namespace, false, created_view); + materialise_partition(&sim.replicas[2], namespace, false, created_view); + + let client = SimClient::new(CLIENT_ID); + sim.shell_login(&client); + for payload in [ + Bytes::from_static(b"parked-redispatch-0"), + Bytes::from_static(b"parked-redispatch-1"), + ] { + let request = client.send_messages(namespace, std::slice::from_ref(&payload)); + sim.submit_request(CLIENT_ID, 0, request.into_generic()); + } + + let lagging_shard = Rc::clone(&sim.replicas[1].shards[0]); + let mut successful_replies = 0usize; + let mut parked = 0usize; + for _ in 0..500 { + successful_replies += successful_send_reply_count(&sim.step()); + parked = lagging_shard.parked_frame_count(namespace); + if successful_replies >= 2 && parked >= 2 { + break; + } + } + assert!( + successful_replies >= 2, + "primary never committed both sends" + ); + assert!(parked >= 2, "lagging backup never parked both prepares"); + + // Keep the third prepare off replica 1's network path. Replica 0 can + // still commit it through replica 2, after which its journal supplies + // the exact wire frame to put on the lagging backup's inbox below. + sim.network.process_disable(ProcessId::Replica(1)); + let third = client.send_messages(namespace, &[Bytes::from_static(b"parked-redispatch-2")]); + sim.submit_request(CLIENT_ID, 0, third.into_generic()); + for _ in 0..500 { + successful_replies += successful_send_reply_count(&sim.step()); + if successful_replies >= 3 { + break; + } + } + assert!( + successful_replies >= 3, + "primary never committed the later send" + ); + let later_prepare = retained_prepare(&sim, 0, namespace, 3); + sim.network.process_enable(ProcessId::Replica(1)); + + materialise_partition(&sim.replicas[1], namespace, false, created_view); + assert_eq!(lagging_shard.parked_frame_count(namespace), 0); + assert_eq!( + lagging_shard.redispatched_frame_count(), + parked, + "materialisation must move every parked prepare to the pump queue" + ); + + // No virtual-time advance here. The materialisation marker and inbox + // wake poll the pump, whose ranked redispatch arm must put ops 1 and 2 + // ahead of this op 3 frame. + lagging_shard.dispatch(later_prepare.into_generic()); + sim.run_pumps(); + let lagging_holds_later = sim.replicas[1] + .partition_shard(namespace) + .plane + .partitions() + .get_by_ns(&namespace) + .is_some_and(|partition| partition.log.journal().inner.header_by_op(3).is_some()); + assert!( + lagging_holds_later, + "inbox op 3 reached the gap check before the redispatched prefix" + ); + + let expected = sim + .offsets(0, namespace) + .expect("primary partition offsets"); + let mut converged = false; + for _ in 0..500 { + sim.step(); + converged = (0..3).all(|replica| sim.offsets(replica, namespace) == Some(expected)); + if converged { + break; + } + } + assert!( + converged, + "redispatched prepares did not converge the backup" + ); + assert_eq!(lagging_shard.redispatched_frame_count(), 0); + assert_no_frame_drops(&sim); + + let offsets = (0..3) + .map(|replica| sim.offsets(replica, namespace).expect("partition offsets")) + .collect(); + (parked, offsets, sim.schedule_hash()) + } + + #[test] + fn parked_prepare_redispatch_is_seed_replayable() { + let first = parked_prepare_redispatch_trace(0x5CED_4005); + let second = parked_prepare_redispatch_trace(0x5CED_4005); + assert_eq!(first, second, "same seed diverged on parked redispatch"); + } + + #[test] + fn redispatched_solo_request_commits_without_another_inbox_frame() { + const CLIENT_ID: u128 = 1; + + server_common::MemoryPool::init_pool(&server_common::MemoryPoolSettings { + enabled: false, + size: iggy_common::IggyByteSize::from(0u64), + bucket_capacity: 1, + }); + + let network_opts = packet::PacketSimulatorOptions { + node_count: 1, + client_count: 1, + seed: 0x5CED_4006, + ..packet::PacketSimulatorOptions::default() + }; + let mut sim = Simulator::with_shards_shell(1, 1, std::iter::once(CLIENT_ID), network_opts); + let namespace = IggyNamespace::new(0, 0, 0); + sim.seed_stream_topic_partition(namespace); + + let client = SimClient::new(CLIENT_ID); + sim.shell_login(&client); + let request = client.send_messages(namespace, &[Bytes::from_static(b"solo-redispatch")]); + let shard = Rc::clone(&sim.replicas[0].shards[0]); + let request = server_common::MessageBag::try_from(request.into_generic()) + .expect("valid send request"); + futures::executor::block_on(shard.on_message(request)); + assert_eq!( + shard.parked_frame_count(namespace), + 1, + "the request must park before materialisation" + ); + + sim.init_partition(namespace); + assert_eq!( + shard.redispatched_frame_count(), + 1, + "materialisation must stage and wake the parked request" + ); + + // No network or virtual-time step follows materialisation. The ranked + // redispatch arm must deliver this request and process its self-ack in + // the same pump iteration. + sim.run_pumps(); + let state = sim + .partition_consensus_state(0, namespace) + .expect("the materialised solo partition has consensus state"); + assert_eq!(state.commit_min, 1, "the self-ack must commit the request"); + assert_eq!(shard.redispatched_frame_count(), 0); + } + /// The dispatch shell's reason to exist: detect the PR #3557 async-concurrency /// class, a partition reference held across an `.await` while a sibling task /// mutates the partitions vec. Under the deterministic executor a parked read with @@ -2698,7 +2925,14 @@ mod tests { executor.run_until_stalled(POLL_BUDGET); // borrow acquired; task parks let grow = Rc::clone(&sim.replicas[0].shards[0]); executor.spawn(async move { - grow.init_partition(ns_grow, None, None, None, false, 0); + grow.init_partition( + ns_grow, + None, + None, + None, + false, + PartitionMaterialisation::new(0, 0), + ); }); executor.run_until_stalled(POLL_BUDGET); // grow while the borrow is live })) @@ -2733,7 +2967,14 @@ mod tests { executor.run_until_stalled(POLL_BUDGET); let grow = Rc::clone(&sim.replicas[0].shards[0]); executor.spawn(async move { - grow.init_partition(ns_grow, None, None, None, false, 0); + grow.init_partition( + ns_grow, + None, + None, + None, + false, + PartitionMaterialisation::new(0, 0), + ); }); executor.run_until_stalled(POLL_BUDGET); From e362c0f0dcfcd1b6b52b071f6de3d6727f309c9a Mon Sep 17 00:00:00 2001 From: Hubert Gruszecki Date: Tue, 1 Sep 2026 17:23:25 +0200 Subject: [PATCH 030/182] fix(connectors): iceberg sink writes real partitions on iceberg 0.10 (#3969) Co-authored-by: Ashutosh Prajapati --- Cargo.lock | 919 +++++++++--------- Cargo.toml | 14 +- .../sinks/clickhouse_sink/Cargo.toml | 1 + core/connectors/sinks/iceberg_sink/Cargo.toml | 4 + core/connectors/sinks/iceberg_sink/README.md | 10 + .../connectors/sinks/iceberg_sink/config.toml | 1 + .../sinks/iceberg_sink/src/catalog.rs | 1 - core/connectors/sinks/iceberg_sink/src/lib.rs | 8 + .../sinks/iceberg_sink/src/props.rs | 16 + .../sinks/iceberg_sink/src/router/mod.rs | 459 +++++++-- .../connectors/fixtures/iceberg/container.rs | 64 +- .../tests/connectors/fixtures/iceberg/mod.rs | 3 +- .../tests/connectors/fixtures/mod.rs | 3 +- .../tests/connectors/iceberg/iceberg_sink.rs | 82 +- 14 files changed, 989 insertions(+), 596 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 5a622dab6d..be2a7673f4 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -67,7 +67,7 @@ dependencies = [ "actix-rt", "actix-service", "actix-utils", - "base64", + "base64 0.22.1", "bitflags 2.13.1", "brotli", "bytes", @@ -229,6 +229,16 @@ version = "2.0.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "320119579fcad9c21884f5c4861d16174d0e06250625266f50fe6898340abefa" +[[package]] +name = "aead" +version = "0.5.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d122413f284cf2d62fb1b7db97e02edb8cda96d769b16e443a4f6195e35662b0" +dependencies = [ + "crypto-common 0.1.7", + "generic-array", +] + [[package]] name = "aead" version = "0.6.1" @@ -261,17 +271,31 @@ dependencies = [ "cpufeatures 0.3.0", ] +[[package]] +name = "aes-gcm" +version = "0.10.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "831010a0f742e1209b3bcea8fab6a8e149051ba6099432c8cb2cc117dec3ead1" +dependencies = [ + "aead 0.5.2", + "aes 0.8.4", + "cipher 0.4.4", + "ctr 0.9.2", + "ghash 0.5.1", + "subtle", +] + [[package]] name = "aes-gcm" version = "0.11.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "fdf011db2e21ce0d575593d749db5554b47fed37aff429e4dc50bc91ac93a028" dependencies = [ - "aead", + "aead 0.6.1", "aes 0.9.1", "cipher 0.5.2", - "ctr", - "ghash", + "ctr 0.10.1", + "ghash 0.6.0", "subtle", ] @@ -422,6 +446,7 @@ checksum = "36fa98bc79671c7981272d91a8753a928ff6a1cd8e4f20a44c45bd5d313840bf" dependencies = [ "bigdecimal", "bon", + "crc32fast", "digest 0.10.7", "log", "miniz_oxide", @@ -432,6 +457,7 @@ dependencies = [ "serde", "serde_bytes", "serde_json", + "snap", "strum 0.27.2", "strum_macros 0.27.2", "thiserror 2.0.19", @@ -517,102 +543,49 @@ checksum = "d3fb67a6e08acf24fdeccbac2cb6ac4305825bd1f117462e0e6f2f193345ad56" [[package]] name = "arrow" -version = "57.3.1" +version = "58.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3bd47f2a6ddc39244bd722a27ee5da66c03369d087b9e024eafdb03e98b98ea7" +checksum = "6cfdd0833e32a9874d2b55089333ad310c0be208aafa277385ce2461dec90be3" dependencies = [ - "arrow-arith 57.3.1", - "arrow-array 57.3.1", - "arrow-buffer 57.3.1", - "arrow-cast 57.3.1", - "arrow-csv 57.3.1", - "arrow-data 57.3.1", - "arrow-ipc 57.3.1", - "arrow-json 57.3.1", - "arrow-ord 57.3.1", - "arrow-row 57.3.1", - "arrow-schema 57.3.1", - "arrow-select 57.3.1", - "arrow-string 57.3.1", -] - -[[package]] -name = "arrow" -version = "58.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "378530e55cd479eda3c14eb345310799717e6f76d0c332041e8487022166b471" -dependencies = [ - "arrow-arith 58.3.0", - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-cast 58.3.0", - "arrow-csv 58.3.0", - "arrow-data 58.3.0", - "arrow-ipc 58.3.0", - "arrow-json 58.3.0", - "arrow-ord 58.3.0", - "arrow-row 58.3.0", - "arrow-schema 58.3.0", - "arrow-select 58.3.0", - "arrow-string 58.3.0", -] - -[[package]] -name = "arrow-arith" -version = "57.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7c7bbd679c5418b8639b92be01f361d60013c4906574b578b77b63c78356594c" -dependencies = [ - "arrow-array 57.3.1", - "arrow-buffer 57.3.1", - "arrow-data 57.3.1", - "arrow-schema 57.3.1", - "chrono", - "num-traits", + "arrow-arith", + "arrow-array", + "arrow-buffer", + "arrow-cast", + "arrow-csv", + "arrow-data", + "arrow-ipc", + "arrow-json", + "arrow-ord", + "arrow-row", + "arrow-schema", + "arrow-select", + "arrow-string", ] [[package]] name = "arrow-arith" -version = "58.3.0" +version = "58.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a0ab212d2c1886e802f51c5212d78ebbcbb0bec980fff9dadc1eb8d45cd0b738" +checksum = "0a41203398f0eaa6f7ec8e62c0da742a21abf282c148fc157f6c35c90e29981a" dependencies = [ - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-data 58.3.0", - "arrow-schema 58.3.0", + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-schema", "chrono", "num-traits", ] [[package]] name = "arrow-array" -version = "57.3.1" +version = "58.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c8a4ab47b3f3eac60f7fd31b81e9028fda018607bcc63451aca4f2b755269862" +checksum = "ae33dad492b7df00a217563a7b0ef2874df68a0deea1b1a3acf628152f7f7a69" dependencies = [ "ahash 0.8.12", - "arrow-buffer 57.3.1", - "arrow-data 57.3.1", - "arrow-schema 57.3.1", - "chrono", - "half", - "hashbrown 0.16.1", - "num-complex", - "num-integer", - "num-traits", -] - -[[package]] -name = "arrow-array" -version = "58.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cfd33d3e92f207444098c75b42de99d329562be0cf686b307b097cc52b4e999e" -dependencies = [ - "ahash 0.8.12", - "arrow-buffer 58.3.0", - "arrow-data 58.3.0", - "arrow-schema 58.3.0", + "arrow-buffer", + "arrow-data", + "arrow-schema", "chrono", "chrono-tz", "half", @@ -624,21 +597,9 @@ dependencies = [ [[package]] name = "arrow-buffer" -version = "57.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0d18b89b4c4f4811d0858175e79541fe98e33e18db3b011708bc287b1240593f" -dependencies = [ - "bytes", - "half", - "num-bigint", - "num-traits", -] - -[[package]] -name = "arrow-buffer" -version = "58.3.0" +version = "58.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0c6cd424c2693bcdbc150d843dc9d4d137dd2de4782ce6df491ad11a3a0416c0" +checksum = "b9552f96391c005e6ab449fa941420935e7e062489b12b8b1b08879b2163f5b5" dependencies = [ "bytes", "half", @@ -648,39 +609,18 @@ dependencies = [ [[package]] name = "arrow-cast" -version = "57.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "722b5c41dd1d14d0a879a1bce92c6fe33f546101bb2acce57a209825edd075b3" -dependencies = [ - "arrow-array 57.3.1", - "arrow-buffer 57.3.1", - "arrow-data 57.3.1", - "arrow-ord 57.3.1", - "arrow-schema 57.3.1", - "arrow-select 57.3.1", - "atoi", - "base64", - "chrono", - "half", - "lexical-core", - "num-traits", - "ryu", -] - -[[package]] -name = "arrow-cast" -version = "58.3.0" +version = "58.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4c5aefb56a2c02e9e2b30746241058b85f8983f0fcff2ba0c6d09006e1cded7f" +checksum = "3a8a327c9649f30d8406995f27642b68df354713cca3baaaf100f076f18d5f34" dependencies = [ - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-data 58.3.0", - "arrow-ord 58.3.0", - "arrow-schema 58.3.0", - "arrow-select 58.3.0", + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-ord", + "arrow-schema", + "arrow-select", "atoi", - "base64", + "base64 0.22.1", "chrono", "half", "lexical-core", @@ -690,28 +630,13 @@ dependencies = [ [[package]] name = "arrow-csv" -version = "57.3.1" +version = "58.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "27ddb80a4848e03b1655af496d5ac2563a779e5742fcb48f2ca2e089c9cd2197" +checksum = "af0dd6d90d1955e9f9a014c1e563ee8aeffc21909085d25623e1da44d96eca26" dependencies = [ - "arrow-array 57.3.1", - "arrow-cast 57.3.1", - "arrow-schema 57.3.1", - "chrono", - "csv", - "csv-core", - "regex", -] - -[[package]] -name = "arrow-csv" -version = "58.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e94e8cf7e517657a52b91ea1263acf38c4ca62a84655d72458a3359b12ab97de" -dependencies = [ - "arrow-array 58.3.0", - "arrow-cast 58.3.0", - "arrow-schema 58.3.0", + "arrow-array", + "arrow-cast", + "arrow-schema", "chrono", "csv", "csv-core", @@ -720,25 +645,12 @@ dependencies = [ [[package]] name = "arrow-data" -version = "57.3.1" +version = "58.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c1683705c63dcf0d18972759eda48489028cbbff67af7d6bef2c6b7b74ab778a" +checksum = "2b24852db04738907e06c04ea61e42fe7fda962a34513022dc0d0e754fb7976b" dependencies = [ - "arrow-buffer 57.3.1", - "arrow-schema 57.3.1", - "half", - "num-integer", - "num-traits", -] - -[[package]] -name = "arrow-data" -version = "58.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3c88210023a2bfee1896af366309a3028fc3bcbd6515fa29a7990ee1baa08ee0" -dependencies = [ - "arrow-buffer 58.3.0", - "arrow-schema 58.3.0", + "arrow-buffer", + "arrow-schema", "half", "num-integer", "num-traits", @@ -746,43 +658,30 @@ dependencies = [ [[package]] name = "arrow-ipc" -version = "57.3.1" +version = "58.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8cf72d04c07229fbf4dbebe7145cac37d7cf7ec582fe705c6b92cb314af096ab" +checksum = "29a908a11fcfb3fb2f6730f4ac15e367bc644e419155e96238f68cf3adde572b" dependencies = [ - "arrow-array 57.3.1", - "arrow-buffer 57.3.1", - "arrow-data 57.3.1", - "arrow-schema 57.3.1", - "arrow-select 57.3.1", - "flatbuffers", -] - -[[package]] -name = "arrow-ipc" -version = "58.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "238438f0834483703d88896db6fe5a7138b2230debc31b34c0336c2996e3c64f" -dependencies = [ - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-data 58.3.0", - "arrow-schema 58.3.0", - "arrow-select 58.3.0", + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-schema", + "arrow-select", "flatbuffers", ] [[package]] name = "arrow-json" -version = "57.3.1" +version = "58.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a84a905f41fedfcd7679813c89a61dc369c0f932b27aa8dcc6aa051cc781a97d" +checksum = "b8a96aed3931c076adee39ec2a40d8219fc7f09e79bcdaca1df16272993e1e14" dependencies = [ - "arrow-array 57.3.1", - "arrow-buffer 57.3.1", - "arrow-cast 57.3.1", - "arrow-data 57.3.1", - "arrow-schema 57.3.1", + "arrow-array", + "arrow-buffer", + "arrow-cast", + "arrow-ord", + "arrow-schema", + "arrow-select", "chrono", "half", "indexmap 2.14.0", @@ -796,94 +695,37 @@ dependencies = [ "simdutf8", ] -[[package]] -name = "arrow-json" -version = "58.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "205ca2119e6d679d5c133c6f30e68f027738d95ed948cf77677ea69c7800036b" -dependencies = [ - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-cast 58.3.0", - "arrow-ord 58.3.0", - "arrow-schema 58.3.0", - "arrow-select 58.3.0", - "chrono", - "half", - "indexmap 2.14.0", - "itoa", - "lexical-core", - "memchr", - "num-traits", - "ryu", - "serde_core", - "serde_json", - "simdutf8", -] - -[[package]] -name = "arrow-ord" -version = "57.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "082342947d4e5a2bcccf029a0a0397e21cb3bb8421edd9571d34fb5dd2670256" -dependencies = [ - "arrow-array 57.3.1", - "arrow-buffer 57.3.1", - "arrow-data 57.3.1", - "arrow-schema 57.3.1", - "arrow-select 57.3.1", -] - [[package]] name = "arrow-ord" -version = "58.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1bffd8fd2579286a5d63bac898159873e5094a79009940bcb42bbfce4f19f1d0" -dependencies = [ - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-data 58.3.0", - "arrow-schema 58.3.0", - "arrow-select 58.3.0", -] - -[[package]] -name = "arrow-row" -version = "57.3.1" +version = "58.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e3a931b520a2a5e22033e01a6f2486b4cdc26f9106b759abeebc320f125e94d7" +checksum = "63a083ec750f5c043f02946b4baf05fcdbb55f4560a3277055caca5cc99f3eb0" dependencies = [ - "arrow-array 57.3.1", - "arrow-buffer 57.3.1", - "arrow-data 57.3.1", - "arrow-schema 57.3.1", - "half", + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-schema", + "arrow-select", ] [[package]] name = "arrow-row" -version = "58.3.0" +version = "58.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bab5994731204603c73ba69267616c50f80780774c6bb0476f1f830625115e0c" +checksum = "514ba0ef0d4c5896202dae736251ce415abb43a950bed570fb7981b8716c0e4c" dependencies = [ - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-data 58.3.0", - "arrow-schema 58.3.0", + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-schema", "half", ] [[package]] name = "arrow-schema" -version = "57.3.1" +version = "58.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e4cf0d4a6609679e03002167a61074a21d7b1ad9ea65e462b2c0a97f8a3b2bc6" - -[[package]] -name = "arrow-schema" -version = "58.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f633dbfdf39c039ada1bf9e34c694816eb71fbb7dc78f613993b7245e078a1ed" +checksum = "21ca356ad6425cecb6eb7b28e4f659f1ee7880fbb1a16127de7dd62901efee9e" dependencies = [ "bitflags 2.13.1", "serde", @@ -892,60 +734,29 @@ dependencies = [ [[package]] name = "arrow-select" -version = "57.3.1" +version = "58.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0b320d86a9806923663bb0fd9baa65ecaba81cb0cd77ff8c1768b9716b4ef891" +checksum = "c58da39eb3d8350ad4a549e5c2bc49284dac554016c69829310350f1731b0aad" dependencies = [ "ahash 0.8.12", - "arrow-array 57.3.1", - "arrow-buffer 57.3.1", - "arrow-data 57.3.1", - "arrow-schema 57.3.1", - "num-traits", -] - -[[package]] -name = "arrow-select" -version = "58.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8cd065c54172ac787cf3f2f8d4107e0d3fdc26edba76fdf4f4cc170258942222" -dependencies = [ - "ahash 0.8.12", - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-data 58.3.0", - "arrow-schema 58.3.0", - "num-traits", -] - -[[package]] -name = "arrow-string" -version = "57.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b493e99162e5764077e7823e50ba284858d365922631c7aaefe9487b1abd02c2" -dependencies = [ - "arrow-array 57.3.1", - "arrow-buffer 57.3.1", - "arrow-data 57.3.1", - "arrow-schema 57.3.1", - "arrow-select 57.3.1", - "memchr", + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-schema", "num-traits", - "regex", - "regex-syntax", ] [[package]] name = "arrow-string" -version = "58.3.0" +version = "58.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "29dd7cda3ab9692f43a2e4acc444d760cc17b12bb6d8232ddf64e9bab7c06b42" +checksum = "b6789b388467525e3271326b6b4915666ecfdf5142aef09779445c954b67543c" dependencies = [ - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-data 58.3.0", - "arrow-schema 58.3.0", - "arrow-select 58.3.0", + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-schema", + "arrow-select", "memchr", "num-traits", "regex", @@ -1333,7 +1144,7 @@ version = "0.30.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "16e2cdb6d5ed835199484bb92bb8b3edd526effe995c61732580439c1a67e2e9" dependencies = [ - "base64", + "base64 0.22.1", "http 1.4.2", "log", "rustls", @@ -1966,6 +1777,12 @@ version = "0.22.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "72b3254f16251a8381aa12e40e3c4d2f0199f8c6508fbecb9d91f575e0fbb8c6" +[[package]] +name = "base64" +version = "0.23.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ac07cdecf99051d9a5238b80f35af32cdeba5b336e55d957b318b50137e18da5" + [[package]] name = "base64-simd" version = "0.8.0" @@ -2294,7 +2111,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ee04c4c84f1f811b017f2fbb7dd8815c976e7ca98593de9c1e2afad0f636bff4" dependencies = [ "async-stream", - "base64", + "base64 0.22.1", "bitflags 2.13.1", "bollard-buildkit-proto", "bollard-stubs", @@ -2351,7 +2168,7 @@ version = "1.52.1-rc.29.1.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0f0a8ca8799131c1837d1282c3f81f31e76ceb0ce426e04a7fe1ccee3287c066" dependencies = [ - "base64", + "base64 0.22.1", "bollard-buildkit-proto", "bytes", "prost", @@ -2447,7 +2264,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7969a9ba84b0ff843813e7249eed1678d9b6607ce5a3b8f0a47af3fcf7978e6e" dependencies = [ "ahash 0.8.12", - "base64", + "base64 0.22.1", "bitvec", "getrandom 0.2.17", "getrandom 0.3.4", @@ -2492,7 +2309,7 @@ version = "0.22.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "2235eb320cd7178862a32dd111bd0c0f71a368e393add4914c50129add478eab" dependencies = [ - "arrow 58.3.0", + "arrow", "buoyant_kernel_derive", "bytes", "chrono", @@ -2501,7 +2318,7 @@ dependencies = [ "indexmap 2.14.0", "itertools 0.14.0", "object_store", - "parquet 58.3.0", + "parquet", "percent-encoding", "rand 0.9.5", "reqwest 0.13.4", @@ -3605,6 +3422,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a" dependencies = [ "generic-array", + "rand_core 0.6.4", "typenum", ] @@ -3667,6 +3485,15 @@ version = "0.0.13" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7a949c44fcacbbbb7ada007dc7acb34603dd97cd47de5d054f2b6493ecebb483" +[[package]] +name = "ctr" +version = "0.9.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0369ee1ad671834580515889b80f2ea915f23b8be8d0daa4bbaf2ac5c7590835" +dependencies = [ + "cipher 0.4.4", +] + [[package]] name = "ctr" version = "0.10.1" @@ -3787,7 +3614,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "5f8c8abaf78cfe3cb7838d7cfcf75e4e3bf02fe3e54571e0a1f10bb485cf8fbf" dependencies = [ "async-stream", - "base64", + "base64 0.22.1", "cfg-if", "cfg_aliases", "compio", @@ -4091,17 +3918,17 @@ version = "0.32.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "4588e95ff3b2ccdba56d9ec262bd3467c0593000f729402528706f62be8be1ca" dependencies = [ - "arrow 58.3.0", - "arrow-arith 58.3.0", - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-cast 58.3.0", - "arrow-ipc 58.3.0", - "arrow-json 58.3.0", - "arrow-ord 58.3.0", - "arrow-row 58.3.0", - "arrow-schema 58.3.0", - "arrow-select 58.3.0", + "arrow", + "arrow-arith", + "arrow-array", + "arrow-buffer", + "arrow-cast", + "arrow-ipc", + "arrow-json", + "arrow-ord", + "arrow-row", + "arrow-schema", + "arrow-select", "async-trait", "buoyant_kernel", "bytes", @@ -4118,7 +3945,7 @@ dependencies = [ "num_cpus", "object_store", "parking_lot", - "parquet 58.3.0", + "parquet", "percent-encoding", "percent-encoding-rfc3986", "pin-project-lite", @@ -4526,7 +4353,7 @@ version = "1.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "29547a1dc60885a552306986316bc9701ba120c1a8db6769fa68691529ad373d" dependencies = [ - "base64", + "base64 0.22.1", "serde", "serde_json", ] @@ -4638,7 +4465,7 @@ version = "9.1.0-alpha.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "12bb303aa6e1d28c0c86b6fbfe484fd0fd3f512629aeed1ac4f6b85f81d9834a" dependencies = [ - "base64", + "base64 0.22.1", "bytes", "dyn-clone", "flate2", @@ -5384,13 +5211,23 @@ dependencies = [ "wasm-bindgen", ] +[[package]] +name = "ghash" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f0d8a4362ccb29cb0b265253fb0a2728f592895ee6854fd9bc13f2ffda266ff1" +dependencies = [ + "opaque-debug", + "polyval 0.6.2", +] + [[package]] name = "ghash" version = "0.6.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "2eecf2d5dc9b66b732b97707a0210906b1d30523eb773193ab777c0c84b3e8d5" dependencies = [ - "polyval", + "polyval 0.7.3", ] [[package]] @@ -6390,7 +6227,7 @@ version = "0.1.20" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "96547c2556ec9d12fb1578c4eaf448b04993e7fb79cbaad930a656880a6bdfa0" dependencies = [ - "base64", + "base64 0.22.1", "bytes", "futures-channel", "futures-util", @@ -6450,25 +6287,26 @@ dependencies = [ [[package]] name = "iceberg" -version = "0.9.1" +version = "0.10.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4d9c3fc1f55c84ff64645c0d2ee35159f5574d33f729fc0860783bc247c8b8c5" +checksum = "6e50fb53c7480f414911ab76b6e14c0042d54ad5c30f6ac2bf2f52a487a25716" dependencies = [ + "aes-gcm 0.10.3", "anyhow", "apache-avro", "array-init", - "arrow-arith 57.3.1", - "arrow-array 57.3.1", - "arrow-buffer 57.3.1", - "arrow-cast 57.3.1", - "arrow-ord 57.3.1", - "arrow-schema 57.3.1", - "arrow-select 57.3.1", - "arrow-string 57.3.1", + "arrow-arith", + "arrow-array", + "arrow-buffer", + "arrow-cast", + "arrow-ord", + "arrow-schema", + "arrow-select", + "arrow-string", "as-any", "async-trait", "backon", - "base64", + "base64 0.22.1", "bimap", "bytes", "chrono", @@ -6483,7 +6321,7 @@ dependencies = [ "murmur3", "once_cell", "ordered-float 4.6.0", - "parquet 57.3.1", + "parquet", "rand 0.9.5", "reqwest 0.12.28", "roaring", @@ -6495,18 +6333,20 @@ dependencies = [ "serde_with", "strum 0.27.2", "tokio", + "tracing", "typed-builder 0.20.1", "typetag", "url", "uuid", + "zeroize", "zstd", ] [[package]] name = "iceberg-catalog-rest" -version = "0.9.1" +version = "0.10.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a49dfef578060c3a2a3f619522dcb25382a7ce85336dcb500495d4b8d72ae714" +checksum = "d1500a26a9b18f286a319e914ce7290dddb7e2eef810edf535afc4ce0c8a6f45" dependencies = [ "async-trait", "chrono", @@ -6518,25 +6358,25 @@ dependencies = [ "serde_derive", "serde_json", "tokio", - "tracing", "typed-builder 0.20.1", "uuid", ] [[package]] name = "iceberg-storage-opendal" -version = "0.9.1" +version = "0.10.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "30acae4698949eea6cfbd3e64ddf1c3ec39e6b7191acb0b0da2dcd1a25dfa842" +checksum = "983efbae430ed40c074af69a8b0aff99bee73e54686cf26ecde97f84127812a2" dependencies = [ "anyhow", "async-trait", "bytes", "cfg-if", + "futures", "iceberg", "opendal", - "reqsign", - "reqwest 0.12.28", + "reqsign-aws-v4", + "reqsign-core", "serde", "typetag", "url", @@ -6905,10 +6745,10 @@ dependencies = [ name = "iggy_common" version = "0.11.0-edge.6" dependencies = [ - "aes-gcm", + "aes-gcm 0.11.0", "async-broadcast", "async-trait", - "base64", + "base64 0.22.1", "blake3", "bon", "byte-unit", @@ -6976,7 +6816,7 @@ name = "iggy_connector_doris_sink" version = "0.2.0-edge.4" dependencies = [ "async-trait", - "base64", + "base64 0.22.1", "blake3", "bytes", "humantime", @@ -6996,7 +6836,7 @@ name = "iggy_connector_elasticsearch_sink" version = "0.5.0-edge.4" dependencies = [ "async-trait", - "base64", + "base64 0.22.1", "dashmap", "elasticsearch", "iggy_common", @@ -7032,7 +6872,7 @@ name = "iggy_connector_http_sink" version = "0.5.0-edge.4" dependencies = [ "async-trait", - "base64", + "base64 0.22.1", "bytes", "humantime", "iggy_connector_sdk", @@ -7053,17 +6893,19 @@ dependencies = [ name = "iggy_connector_iceberg_sink" version = "0.5.0-edge.4" dependencies = [ - "arrow-json 57.3.1", + "arrow-array", + "arrow-json", "async-trait", "dashmap", "iceberg", "iceberg-catalog-rest", "iceberg-storage-opendal", "iggy_connector_sdk", - "parquet 57.3.1", + "parquet", "serde", "simd-json", "strum 0.28.0", + "tokio", "tracing", "uuid", ] @@ -7074,7 +6916,7 @@ version = "0.5.0-edge.4" dependencies = [ "async-trait", "axum", - "base64", + "base64 0.22.1", "bytes", "iggy_common", "iggy_connector_sdk", @@ -7096,7 +6938,7 @@ dependencies = [ "ahash 0.8.12", "async-trait", "axum", - "base64", + "base64 0.22.1", "chrono", "csv", "dashmap", @@ -7120,7 +6962,7 @@ name = "iggy_connector_meilisearch_sink" version = "0.5.0-edge.4" dependencies = [ "async-trait", - "base64", + "base64 0.22.1", "iggy_common", "iggy_connector_sdk", "meilisearch-sdk", @@ -7173,7 +7015,7 @@ name = "iggy_connector_postgres_source" version = "0.5.0-edge.4" dependencies = [ "async-trait", - "base64", + "base64 0.22.1", "chrono", "dashmap", "futures", @@ -7225,13 +7067,13 @@ dependencies = [ name = "iggy_connector_redshift_sink" version = "0.4.1-edge.1" dependencies = [ - "arrow 57.3.1", + "arrow", "async-trait", "chrono", "humantime", "iggy_common", "iggy_connector_sdk", - "parquet 57.3.1", + "parquet", "rust-s3", "secrecy", "serde", @@ -7248,7 +7090,7 @@ name = "iggy_connector_s3_sink" version = "0.5.0-edge.4" dependencies = [ "async-trait", - "base64", + "base64 0.22.1", "byte-unit", "chrono", "dashmap", @@ -7271,7 +7113,7 @@ dependencies = [ "anyhow", "apache-avro", "async-trait", - "base64", + "base64 0.22.1", "dashmap", "flatbuffers", "http 1.4.2", @@ -7317,7 +7159,7 @@ name = "iggy_connector_surrealdb_sink" version = "0.5.0-edge.4" dependencies = [ "async-trait", - "base64", + "base64 0.22.1", "bytes", "iggy_common", "iggy_connector_sdk", @@ -7523,10 +7365,10 @@ checksum = "8bb03732005da905c88227371639bf1ad885cc712789c011c31c5fb3ab3ccf02" name = "integration" version = "0.0.1" dependencies = [ - "arrow 57.3.1", + "arrow", "assert_cmd", "async-trait", - "base64", + "base64 0.22.1", "bon", "bytemuck", "bytes", @@ -7555,7 +7397,7 @@ dependencies = [ "lazy_static", "libc", "mongodb", - "parquet 57.3.1", + "parquet", "pgwire", "predicates", "rand 0.10.2", @@ -7691,10 +7533,12 @@ dependencies = [ "jiff-core", "jiff-static", "jiff-tzdb-platform", + "js-sys", "log", "portable-atomic", "portable-atomic-util", "serde_core", + "wasm-bindgen", "windows-link 0.2.1", ] @@ -7824,7 +7668,7 @@ version = "10.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "eba32bfb4ffdeaca3e34431072faf01745c9b26d25504aa7a6cf5684334fc4fc" dependencies = [ - "base64", + "base64 0.22.1", "ed25519-dalek", "getrandom 0.2.17", "hmac 0.12.1", @@ -8355,15 +8199,6 @@ dependencies = [ "libc", ] -[[package]] -name = "lz4_flex" -version = "0.12.2" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "90071f8077f8e40adfc4b7fe9cd495ce316263f19e75c2211eeff3fdf475a3d9" -dependencies = [ - "twox-hash", -] - [[package]] name = "lz4_flex" version = "0.13.1" @@ -8492,6 +8327,16 @@ version = "0.8.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7ebb8d8732c6a6df3d8f032a82911cfc747e00efb95cc46e8d0acd5b5b88570c" +[[package]] +name = "mea" +version = "0.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c709842c4ce65cb91e2666ad5319dfc1efc3af0d34f02075eddca9000d9f8afb" +dependencies = [ + "hashbrown 0.17.1", + "slab", +] + [[package]] name = "meilisearch-index-setting-macro" version = "0.33.0" @@ -8775,7 +8620,7 @@ version = "3.8.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b814038f367d212f55de0a630cb35102a9b8ca23785a86955d62c0087c93846d" dependencies = [ - "base64", + "base64 0.22.1", "bitflags 2.13.1", "bson", "derive-where", @@ -9200,7 +9045,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "622acbc9100d3c10e2ee15804b0caa40e55c933d5aa53814cd520805b7958a49" dependencies = [ "async-trait", - "base64", + "base64 0.22.1", "bytes", "chrono", "form_urlencoded", @@ -9241,7 +9086,7 @@ checksum = "27be39870c558e1fbc5a5f8d4a29aa916268c1dbcb9d467f2a9bbab35598060e" dependencies = [ "arc-swap", "async-trait", - "base64", + "base64 0.22.1", "bytes", "cargo_metadata", "cfg-if", @@ -9299,33 +9144,132 @@ version = "1.70.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe" +[[package]] +name = "opaque-debug" +version = "0.3.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c08d65885ee38876c4f86fa503fb49d7b507c2b62552df7c70b2fce627e06381" + [[package]] name = "opendal" -version = "0.55.0" +version = "0.57.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "96c9c85ce253ff87225e7669979d877a20c98a06604ec9d6dd5f4473e08f1ae1" +dependencies = [ + "ctor 1.0.9", + "opendal-core", + "opendal-layer-concurrent-limit", + "opendal-layer-logging", + "opendal-layer-retry", + "opendal-layer-timeout", + "opendal-service-fs", + "opendal-service-s3", +] + +[[package]] +name = "opendal-core" +version = "0.57.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d075ab8a203a6ab4bc1bce0a4b9fe486a72bf8b939037f4b78d95386384bc80a" +checksum = "c4f8607c90e2c963a91467f50fb49fbc7fb3d573f88cea219ca59ccd3740b309" dependencies = [ "anyhow", - "backon", - "base64", + "base64 0.22.1", "bytes", - "crc32c", "futures", - "getrandom 0.2.17", "http 1.4.2", "http-body 1.1.0", "jiff", "log", - "md-5 0.10.6", + "md-5 0.11.0", + "mea", "percent-encoding", - "quick-xml 0.38.4", - "reqsign", - "reqwest 0.12.28", + "quick-xml 0.39.4", + "reqsign-core", + "reqwest 0.13.4", "serde", "serde_json", "tokio", "url", "uuid", + "web-time", +] + +[[package]] +name = "opendal-layer-concurrent-limit" +version = "0.57.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0d6f81ba6960e3fae1882f253b114b21d7e444e1534f209c7737a79f6243eb6f" +dependencies = [ + "futures", + "http 1.4.2", + "mea", + "opendal-core", +] + +[[package]] +name = "opendal-layer-logging" +version = "0.57.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "58ada45c6d81d1aa4c9305d0c7d4bc317c59c85866a0908a2d75a7a978aa5ee2" +dependencies = [ + "log", + "opendal-core", +] + +[[package]] +name = "opendal-layer-retry" +version = "0.57.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7b2a25a718afb81fad81cb9a0580a1cb989221fa2317f888c6a37f8dad408eb7" +dependencies = [ + "backon", + "log", + "opendal-core", +] + +[[package]] +name = "opendal-layer-timeout" +version = "0.57.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1e91f731724c213af81e9d03517859c8fc47b4578e64ad61ae4f099f10fe36e3" +dependencies = [ + "opendal-core", + "tokio", +] + +[[package]] +name = "opendal-service-fs" +version = "0.57.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "22e89a665fef0e6bd249cf5ea47fc174b7ba892159bee4b9382528b1ca873a2c" +dependencies = [ + "bytes", + "log", + "opendal-core", + "serde", + "tokio", + "xattr", +] + +[[package]] +name = "opendal-service-s3" +version = "0.57.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "313d46c9f5ae70bca26b7c3e3fbb9b639292625f28af73aa016f47e788af9deb" +dependencies = [ + "base64 0.22.1", + "bytes", + "crc32c", + "http 1.4.2", + "log", + "md-5 0.11.0", + "opendal-core", + "quick-xml 0.39.4", + "reqsign-aws-v4", + "reqsign-core", + "reqsign-file-read-tokio", + "serde", + "url", ] [[package]] @@ -9545,54 +9489,18 @@ dependencies = [ [[package]] name = "parquet" -version = "57.3.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2e832c6aa20310fc6de7ea5a3f4e20d34fd83e3b43229d32b81ffe5c14d74692" -dependencies = [ - "ahash 0.8.12", - "arrow-array 57.3.1", - "arrow-buffer 57.3.1", - "arrow-cast 57.3.1", - "arrow-data 57.3.1", - "arrow-ipc 57.3.1", - "arrow-schema 57.3.1", - "arrow-select 57.3.1", - "base64", - "brotli", - "bytes", - "chrono", - "flate2", - "futures", - "half", - "hashbrown 0.16.1", - "lz4_flex 0.12.2", - "num-bigint", - "num-integer", - "num-traits", - "paste", - "seq-macro", - "simdutf8", - "snap", - "thrift", - "tokio", - "twox-hash", - "zstd", -] - -[[package]] -name = "parquet" -version = "58.3.0" +version = "58.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5dafa7d01085b62a47dd0c1829550a0a36710ea9c4fe358a05a85477cec8a908" +checksum = "d298093b2dec60289dce0684c986d0f7679e9dd15771c2c65406e1aaf604a704" dependencies = [ "ahash 0.8.12", - "arrow-array 58.3.0", - "arrow-buffer 58.3.0", - "arrow-data 58.3.0", - "arrow-ipc 58.3.0", - "arrow-schema 58.3.0", - "arrow-select 58.3.0", - "base64", + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-ipc", + "arrow-schema", + "arrow-select", + "base64 0.22.1", "brotli", "bytes", "chrono", @@ -9606,6 +9514,7 @@ dependencies = [ "num-traits", "object_store", "paste", + "ring", "seq-macro", "simdutf8", "snap", @@ -9768,7 +9677,7 @@ version = "3.0.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1d30c53c26bc5b31a98cd02d20f25a7c8567146caf63ed593a9d87b2775291be" dependencies = [ - "base64", + "base64 0.22.1", "serde_core", ] @@ -9854,7 +9763,7 @@ checksum = "7981cfde34009be689a05a30c497ad5fbb552531d3d54230b3627264ff1bc384" dependencies = [ "async-trait", "aws-lc-rs", - "base64", + "base64 0.22.1", "bytes", "chrono", "derive-new", @@ -10049,6 +9958,18 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "polyval" +version = "0.6.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9d1fe60d06143b2430aa532c94cfe9e29783047f06c0d7fd359a9a51b729fa25" +dependencies = [ + "cfg-if", + "cpufeatures 0.2.17", + "opaque-debug", + "universal-hash 0.5.1", +] + [[package]] name = "polyval" version = "0.7.3" @@ -10057,7 +9978,7 @@ checksum = "f0fa31d631f2b2cb2a544d0aa321ce847a94764d701ca2becc411138b93d49cd" dependencies = [ "cpubits", "cpufeatures 0.3.0", - "universal-hash", + "universal-hash 0.6.1", ] [[package]] @@ -10094,7 +10015,7 @@ version = "0.6.12" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "08808e3c483c46e999108051c78334f473d5adb59d78bb80a1268c7e6aa6c514" dependencies = [ - "base64", + "base64 0.22.1", "byteorder", "bytes", "fallible-iterator", @@ -10525,9 +10446,9 @@ checksum = "a993555f31e5a609f617c12db6250dedcac1b0a85076912c436e6fc9b2c8e6a3" [[package]] name = "quick-xml" -version = "0.37.5" +version = "0.38.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "331e97a1af0bf59823e6eadffe373d7b27f485be8748f71471c662c1f269b7fb" +checksum = "b66c2058c55a409d601666cffe35f04333cf1013010882cec174a7467cd4e21c" dependencies = [ "memchr", "serde", @@ -10535,9 +10456,9 @@ dependencies = [ [[package]] name = "quick-xml" -version = "0.38.4" +version = "0.39.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b66c2058c55a409d601666cffe35f04333cf1013010882cec174a7467cd4e21c" +checksum = "cdcc8dd4e2f670d309a5f0e83fe36dfdc05af317008fea29144da1a2ac858e5e" dependencies = [ "memchr", "serde", @@ -10545,9 +10466,9 @@ dependencies = [ [[package]] name = "quick-xml" -version = "0.39.4" +version = "0.41.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cdcc8dd4e2f670d309a5f0e83fe36dfdc05af317008fea29144da1a2ac858e5e" +checksum = "e660451e55124f798a69a5af3f49ccfbefbd41910eefd25caf2393e1f3473ec1" dependencies = [ "memchr", "serde", @@ -10937,31 +10858,71 @@ dependencies = [ ] [[package]] -name = "reqsign" -version = "0.16.5" +name = "reqsign-aws-core" +version = "3.1.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "43451dbf3590a7590684c25fb8d12ecdcc90ed3ac123433e500447c7d77ed701" +checksum = "bac4749b7dfa7bfaccd01eb03e9dc795ed37e3f20d6f0f38e2c67ee85ad6bc86" dependencies = [ - "anyhow", - "async-trait", - "base64", - "chrono", + "bytes", "form_urlencoded", - "getrandom 0.2.17", "hex", - "hmac 0.12.1", - "home", "http 1.4.2", "log", "percent-encoding", - "quick-xml 0.37.5", - "rand 0.8.7", - "reqwest 0.12.28", + "quick-xml 0.41.0", + "reqsign-core", "rust-ini", "serde", "serde_json", - "sha1 0.10.7", - "sha2 0.10.9", + "serde_urlencoded", + "sha1 0.11.0", +] + +[[package]] +name = "reqsign-aws-v4" +version = "3.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ff250f0fd0b913fbd565e405acc553da0f13bde30bfb5403178c9d0313cdc15f" +dependencies = [ + "bytes", + "http 1.4.2", + "log", + "quick-xml 0.41.0", + "reqsign-aws-core", + "reqsign-core", + "serde", +] + +[[package]] +name = "reqsign-core" +version = "3.3.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ff052daffb0599681c50f85c59e7236438976efe991ab864edd9f3b235501a0f" +dependencies = [ + "anyhow", + "base64 0.23.1", + "bytes", + "futures", + "hex", + "hmac 0.13.0", + "http 1.4.2", + "jiff", + "log", + "mea", + "percent-encoding", + "sha1 0.11.0", + "sha2 0.11.0", + "windows-sys 0.61.2", +] + +[[package]] +name = "reqsign-file-read-tokio" +version = "3.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b3235df90a6bca681aa47dd86f2393d122a6d77042aa8a7c81e218cd45c5bfc0" +dependencies = [ + "anyhow", + "reqsign-core", "tokio", ] @@ -10971,7 +10932,7 @@ version = "0.12.28" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "eddd3ca559203180a307f12d114c268abf583f59b03cb906fd0b3ff8646c1147" dependencies = [ - "base64", + "base64 0.22.1", "bytes", "futures-core", "futures-util", @@ -11014,7 +10975,7 @@ version = "0.13.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "219c5811de6525e5416c7d5d53bb656d3afdbc6c5af816e0802bcfa42dbdc1c3" dependencies = [ - "base64", + "base64 0.22.1", "bytes", "encoding_rs", "futures-channel", @@ -11212,7 +11173,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "14db48ee17a9ba61810ab1a9c1beb7d06d8136ae39ac25a1137f10d357af01af" dependencies = [ "async-trait", - "base64", + "base64 0.22.1", "bytes", "chrono", "futures", @@ -11374,7 +11335,7 @@ dependencies = [ "async-trait", "aws-creds", "aws-region", - "base64", + "base64 0.22.1", "bytes", "cfg-if", "futures-util", @@ -11965,7 +11926,7 @@ version = "3.21.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "76a5c54c7310e7b8b9577c286d7e399ddd876c3e12b3ed917a8aabc4b96e9e8c" dependencies = [ - "base64", + "base64 0.22.1", "bs58", "chrono", "hex", @@ -12581,7 +12542,7 @@ version = "0.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "05b44e85bf579a8eeb4ceaa77a3a523baf2bf0e9bac7e40f405d537b5d2d5ccb" dependencies = [ - "base64", + "base64 0.22.1", "bytes", "cfg-if", "chrono", @@ -12687,7 +12648,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "87a2bdd6e83f6b3ea525ca9fee568030508b58355a43d0b2c1674d5f79dcd65e" dependencies = [ "atoi", - "base64", + "base64 0.22.1", "bitflags 2.13.1", "byteorder", "chrono", @@ -13631,7 +13592,7 @@ checksum = "ac2a5518c70fa84342385732db33fb3f44bc4cc748936eb5833d2df34d6445ef" dependencies = [ "async-trait", "axum", - "base64", + "base64 0.22.1", "bytes", "h2 0.4.15", "http 1.4.2", @@ -14165,6 +14126,16 @@ version = "0.2.6" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ebc1c04c71510c7f702b52b7c350734c9ff1295c464a03335b00bb84fc54f853" +[[package]] +name = "universal-hash" +version = "0.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fc1de2c688dc15305988b563c3854064043356019f97a4b46276fe734c4f07ea" +dependencies = [ + "crypto-common 0.1.7", + "subtle", +] + [[package]] name = "universal-hash" version = "0.6.1" @@ -14199,7 +14170,7 @@ version = "3.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "dea7109cdcd5864d4eeb1b58a1648dc9bf520360d7af16ec26d0a9354bafcfc0" dependencies = [ - "base64", + "base64 0.22.1", "flate2", "log", "percent-encoding", @@ -14216,7 +14187,7 @@ version = "0.6.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e994ba84b0bd1b1b0cf92878b7ef898a5c1760108fe7b6010327e274917a808c" dependencies = [ - "base64", + "base64 0.22.1", "http 1.4.2", "httparse", "log", @@ -14247,7 +14218,7 @@ version = "0.45.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "80be9b06fbae3b8b303400ab20778c80bbaf338f563afe567cf3c9eea17b47ef" dependencies = [ - "base64", + "base64 0.22.1", "data-url", "flate2", "fontdb", @@ -15163,7 +15134,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "08db1edfb05d9b3c1542e521aea074442088292f00b5f28e435c714a98f85031" dependencies = [ "assert-json-diff", - "base64", + "base64 0.22.1", "deadpool", "futures", "http 1.4.2", diff --git a/Cargo.toml b/Cargo.toml index 0f339f28cc..fd6424b3b2 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -95,9 +95,9 @@ apple-native-keyring-store = { version = "1.0.1", features = ["keychain"] } # build graph happens to enable it via feature unification. argon2 = { version = "0.5.3", features = ["std"] } # Pinned to 57 because iceberg 0.9.1 still requires arrow/parquet 57. -arrow = "57.3.1" -arrow-array = "57.3.1" -arrow-json = "57.3.1" +arrow = "58.4.0" +arrow-array = "58.4.0" +arrow-json = "58.4.0" assert_cmd = "2.2.2" async-broadcast = "0.7.2" async-channel = "2.5.0" @@ -202,9 +202,9 @@ humantime = "2.4.0" hwlocality = "1.0.0-alpha.12" hyper = "1.11.0" hyper-util = { version = "0.1.20", features = ["server-auto", "service"] } -iceberg = "0.9.1" -iceberg-catalog-rest = "0.9.1" -iceberg-storage-opendal = "0.9.1" +iceberg = "0.10.1" +iceberg-catalog-rest = "0.10.1" +iceberg-storage-opendal = "0.10.1" iggy = { path = "core/sdk", version = "0.11.0-edge.6" } iggy-cli = { path = "core/cli", version = "0.14.0-edge.6" } iggy_binary_protocol = { path = "core/binary_protocol", version = "0.11.0-edge.6" } @@ -256,7 +256,7 @@ opentelemetry_sdk = { version = "0.32.1", features = [ "experimental_trace_batch_span_processor_with_async_runtime", ] } papaya = "0.2.4" -parquet = "57.3.1" +parquet = "58.4.0" partitions = { path = "core/partitions" } passterm = "2.0.6" paste = "1.0" diff --git a/core/connectors/sinks/clickhouse_sink/Cargo.toml b/core/connectors/sinks/clickhouse_sink/Cargo.toml index 07cf33456b..3f0d07b0e5 100644 --- a/core/connectors/sinks/clickhouse_sink/Cargo.toml +++ b/core/connectors/sinks/clickhouse_sink/Cargo.toml @@ -27,6 +27,7 @@ homepage = "https://iggy.apache.org" documentation = "https://iggy.apache.org/docs" repository = "https://github.com/apache/iggy" readme = "../../README.md" +publish = false [lib] crate-type = ["cdylib", "lib"] diff --git a/core/connectors/sinks/iceberg_sink/Cargo.toml b/core/connectors/sinks/iceberg_sink/Cargo.toml index ec65742148..9e5e21cd92 100644 --- a/core/connectors/sinks/iceberg_sink/Cargo.toml +++ b/core/connectors/sinks/iceberg_sink/Cargo.toml @@ -35,6 +35,7 @@ ignored = ["dashmap"] crate-type = ["cdylib", "lib"] [dependencies] +arrow-array = { workspace = true } arrow-json = { workspace = true } async-trait = { workspace = true } dashmap = { workspace = true } @@ -49,5 +50,8 @@ strum = { workspace = true } tracing = { workspace = true } uuid = { workspace = true } +[dev-dependencies] +tokio = { workspace = true } + [lints] workspace = true diff --git a/core/connectors/sinks/iceberg_sink/README.md b/core/connectors/sinks/iceberg_sink/README.md index c9079e7136..386a6d417b 100644 --- a/core/connectors/sinks/iceberg_sink/README.md +++ b/core/connectors/sinks/iceberg_sink/README.md @@ -25,6 +25,7 @@ store_access_key_id = "admin" store_secret_access_key = "password" store_region = "us-east-1" store_class = "s3" +store_path_style_access = true ``` ## Configuration Options @@ -40,6 +41,7 @@ store_class = "s3" - **store_secret_access_key**: The secret key used to upload data to the object storage. - **store_region**: The region of the object storage, if applicable. - **store_class**: The storage class to use. **Currently, only S3-compatible storage is supported.** +- **store_path_style_access**: Use path-style S3 URLs (`http://host/bucket/key`). Defaults to `true`, which MinIO-style endpoints require; set to `false` for stores that only accept virtual-hosted-style URLs. ## Dynamic Routing @@ -66,6 +68,7 @@ store_access_key_id = "admin" store_secret_access_key = "password" store_region = "us-east-1" store_class = "s3" +store_path_style_access = true [sinks.iceberg.transforms.add_fields] enabled = true @@ -81,6 +84,13 @@ Example: - Namespace: `nyc` - Table name: `users` +## Partitioned Tables + +Data files follow the table's default partition spec. Each batch is split by the spec's transforms +(`identity`, `bucket`, `truncate`, `year`, `month`, `day`, `hour`) and every partition value present +in the batch gets at least one Parquet file under its partition path. Unpartitioned tables are +written into a single file per batch. + ## Source Compatibility The Iceberg sink expects **flat JSON** where each top-level key maps directly to a column in the diff --git a/core/connectors/sinks/iceberg_sink/config.toml b/core/connectors/sinks/iceberg_sink/config.toml index 3a8b21596d..7ab06127f3 100644 --- a/core/connectors/sinks/iceberg_sink/config.toml +++ b/core/connectors/sinks/iceberg_sink/config.toml @@ -43,3 +43,4 @@ store_access_key_id = "admin" store_secret_access_key = "password" store_region = "us-east-1" store_class = "s3" +store_path_style_access = true diff --git a/core/connectors/sinks/iceberg_sink/src/catalog.rs b/core/connectors/sinks/iceberg_sink/src/catalog.rs index d6b6962776..53d4043c5e 100644 --- a/core/connectors/sinks/iceberg_sink/src/catalog.rs +++ b/core/connectors/sinks/iceberg_sink/src/catalog.rs @@ -48,7 +48,6 @@ async fn get_rest_catalog( let storage_factory: Arc = match &config.store_class { IcebergSinkStoreClass::S3 => Arc::new(OpenDalStorageFactory::S3 { - configured_scheme: "s3".to_string(), customized_credential_load: None, }), other => { diff --git a/core/connectors/sinks/iceberg_sink/src/lib.rs b/core/connectors/sinks/iceberg_sink/src/lib.rs index 33ab537d6f..6c6315968c 100644 --- a/core/connectors/sinks/iceberg_sink/src/lib.rs +++ b/core/connectors/sinks/iceberg_sink/src/lib.rs @@ -63,6 +63,14 @@ pub struct IcebergSinkConfig { pub store_secret_access_key: Option, pub store_region: String, pub store_class: IcebergSinkStoreClass, + /// Path-style S3 addressing (`http://host/bucket/key`). MinIO-style endpoints + /// need it; iceberg 0.10 defaults to the virtual-host style otherwise. + #[serde(default = "default_true")] + pub store_path_style_access: bool, +} + +fn default_true() -> bool { + true } fn slice_user_table(table: &str) -> Vec { diff --git a/core/connectors/sinks/iceberg_sink/src/props.rs b/core/connectors/sinks/iceberg_sink/src/props.rs index 4815c530c0..4c1635aa41 100644 --- a/core/connectors/sinks/iceberg_sink/src/props.rs +++ b/core/connectors/sinks/iceberg_sink/src/props.rs @@ -30,6 +30,10 @@ fn get_props_s3(config: &IcebergSinkConfig) -> Result, E let mut props: HashMap = HashMap::new(); props.insert("s3.region".to_string(), config.store_region.clone()); props.insert("s3.endpoint".to_string(), config.store_url.clone()); + props.insert( + "s3.path-style-access".to_string(), + config.store_path_style_access.to_string(), + ); match (&config.store_access_key_id, &config.store_secret_access_key) { (Some(access_key_id), Some(secret_access_key)) => { props.insert("s3.access-key-id".to_string(), access_key_id.clone()); @@ -64,6 +68,7 @@ mod tests { store_secret_access_key: None, store_region: "us-east-1".to_string(), store_class: IcebergSinkStoreClass::S3, + store_path_style_access: true, } } @@ -73,10 +78,21 @@ mod tests { let props = get_props_s3(&config).expect("Should succeed without credentials"); assert_eq!(props.get("s3.region").unwrap(), "us-east-1"); assert_eq!(props.get("s3.endpoint").unwrap(), "http://localhost:9000"); + assert_eq!(props.get("s3.path-style-access").unwrap(), "true"); assert!(!props.contains_key("s3.access-key-id")); assert!(!props.contains_key("s3.secret-access-key")); } + #[test] + fn test_get_props_s3_path_style_disabled() { + let config = IcebergSinkConfig { + store_path_style_access: false, + ..base_config() + }; + let props = get_props_s3(&config).expect("Should succeed with virtual-host addressing"); + assert_eq!(props.get("s3.path-style-access").unwrap(), "false"); + } + #[test] fn test_get_props_s3_full_credentials() { let config = IcebergSinkConfig { diff --git a/core/connectors/sinks/iceberg_sink/src/router/mod.rs b/core/connectors/sinks/iceberg_sink/src/router/mod.rs index 3f43b94013..14b19717b8 100644 --- a/core/connectors/sinks/iceberg_sink/src/router/mod.rs +++ b/core/connectors/sinks/iceberg_sink/src/router/mod.rs @@ -17,16 +17,16 @@ use crate::router::arrow_streamer::JsonArrowReader; use crate::slice_user_table; +use arrow_array::RecordBatch; use arrow_json::ReaderBuilder; use async_trait::async_trait; use iceberg::TableIdent; -use iceberg::arrow::schema_to_arrow_schema; -use iceberg::spec::{ - Literal, PartitionKey, PartitionSpec, PrimitiveLiteral, PrimitiveType, Struct, StructType, -}; +use iceberg::arrow::{RecordBatchPartitionSplitter, schema_to_arrow_schema}; +use iceberg::io::FileIO; +use iceberg::spec::{DataFile, PartitionKey, Struct}; use iceberg::table::Table; use iceberg::transaction::{ApplyTransactionAction, Transaction}; -use iceberg::writer::base_writer::data_file_writer::DataFileWriterBuilder; +use iceberg::writer::base_writer::data_file_writer::{DataFileWriter, DataFileWriterBuilder}; use iceberg::writer::file_writer::ParquetWriterBuilder; use iceberg::writer::file_writer::rolling_writer::RollingFileWriterBuilder; use iceberg::writer::{IcebergWriter, IcebergWriterBuilder}; @@ -36,10 +36,16 @@ use iceberg::{ }; use iggy_connector_sdk::{ConsumedMessage, Error, MessagesMetadata, Payload, Schema}; use parquet::file::properties::WriterProperties; +use std::collections::HashMap; use std::sync::Arc; use tracing::{error, warn}; use uuid::Uuid; +type ParquetDataFileWriterBuilder = + DataFileWriterBuilder; +type ParquetDataFileWriter = + DataFileWriter; + fn format_error_chain(err: &dyn std::error::Error) -> String { let mut chain = err.to_string(); let mut source = err.source(); @@ -67,56 +73,51 @@ async fn table_exists(route_field_val: &str, catalog: &dyn Catalog) -> Option Result { - match pt { - PrimitiveType::Boolean => Ok(PrimitiveLiteral::Boolean(false)), - PrimitiveType::Int => Ok(PrimitiveLiteral::Int(0)), - PrimitiveType::Long => Ok(PrimitiveLiteral::Long(0)), - PrimitiveType::Decimal { .. } => Ok(PrimitiveLiteral::Int128(0)), - PrimitiveType::Date => Ok(PrimitiveLiteral::Int(0)), // e.g. days since epoch - PrimitiveType::Time => Ok(PrimitiveLiteral::Long(0)), // microseconds since midnight - PrimitiveType::Timestamp => Ok(PrimitiveLiteral::Long(0)), // microseconds since epoch - PrimitiveType::Timestamptz => Ok(PrimitiveLiteral::Long(0)), - PrimitiveType::TimestampNs => Ok(PrimitiveLiteral::Long(0)), - PrimitiveType::TimestamptzNs => Ok(PrimitiveLiteral::Long(0)), - PrimitiveType::String => Ok(PrimitiveLiteral::String(String::new())), - PrimitiveType::Uuid => Ok(PrimitiveLiteral::Binary(vec![0; 16])), - PrimitiveType::Fixed(len) => Ok(PrimitiveLiteral::Binary(vec![0; *len as usize])), - PrimitiveType::Binary => Ok(PrimitiveLiteral::Binary(Vec::new())), - _ => { - error!("Partition type not supported"); - Err(Error::InvalidConfig) - } - } -} - -fn get_partition_type_value(default_partition_type: &StructType) -> Result, Error> { - let mut fields: Vec> = Vec::new(); +async fn write_data( + messages: &[Payload], + table: &Table, + catalog: &dyn Catalog, + messages_schema: Schema, +) -> Result<(), Error> { + let data_files = write_data_files(messages, table, messages_schema).await?; - if default_partition_type.fields().is_empty() { - return Ok(None); - }; + // Data files are kept on commit failure: the catalog may have applied the + // commit before the error surfaced, and deleting files referenced by a + // committed snapshot would corrupt the table. + let table_commit = Transaction::new(table); - for field in default_partition_type.fields() { - let field_type = field.field_type.as_primitive_type().ok_or_else(|| { - error!("The partition type of the configured iceberg table is not a primitive type"); - Error::InvalidConfig - })?; + let action = table_commit.fast_append().add_data_files(data_files); - let value = Some(Literal::Primitive(primitive_type_to_literal(field_type)?)); + let tx = action.apply(table_commit).map_err(|err| { + let chain = format_error_chain(&err); + error!( + "Failed to apply transaction on table with UUID: {}, Error: {}", + table.metadata().uuid(), + chain + ); + Error::TransactionApplyError(chain) + })?; - fields.push(value); - } - Ok(Some(Struct::from_iter(fields))) + tx.commit(catalog).await.map_err(|err| { + let chain = format_error_chain(&err); + error!( + "Failed to commit transaction on table with UUID: {}, Error: {}", + table.metadata().uuid(), + chain + ); + Error::CatalogCommitError(chain) + })?; + Ok(()) } -async fn write_data( +/// Writes the JSON payloads as Parquet data files under the table location, +/// laid out by the table's default partition spec. Nothing is committed here. +async fn write_data_files( messages: &[Payload], table: &Table, - catalog: &dyn Catalog, messages_schema: Schema, -) -> Result<(), Error> { - let location = DefaultLocationGenerator::new(table.metadata().clone()).map_err(|err| { +) -> Result, Error> { + let location = DefaultLocationGenerator::new(table.metadata()).map_err(|err| { error!( "Failed to get location on table: {}. Error: {}", table.metadata().uuid(), @@ -145,27 +146,6 @@ async fn write_data( let data_file_writer_builder = DataFileWriterBuilder::new(rolling_file_writer_builder); - let partition_spec = PartitionSpec::builder(table.current_schema_ref()); - - let partition_type = get_partition_type_value(table.metadata().default_partition_type())?; - - let mut writer = data_file_writer_builder - .build(match partition_type { - None => None, - Some(p_type) => Some(PartitionKey::new( - partition_spec - .build() - .map_err(|err| Error::InitError(err.to_string()))?, - table.current_schema_ref(), - p_type, - )), - }) - .await - .map_err(|err| { - error!("Error while constructing data file writer: {}", err); - Error::InitError(err.to_string()) - })?; - let msgs: Vec<&simd_json::OwnedValue> = messages .iter() .filter_map(|payload| match payload { @@ -180,6 +160,14 @@ async fn write_data( }) .collect(); + if msgs.is_empty() { + error!( + "Batch of {} messages has no JSON payloads, the Iceberg sink requires schema = json", + messages.len() + ); + return Err(Error::InvalidPayloadType); + } + let cursor = JsonArrowReader::new(msgs.as_slice()); let reader = ReaderBuilder::new(Arc::new( schema_to_arrow_schema(&table.metadata().current_schema().clone()).map_err(|err| { @@ -201,6 +189,12 @@ async fn write_data( Error::InitError(err.to_string()) })?; + let mut writer = TableWriter::new(table, data_file_writer_builder).map_err(|err| { + let chain = format_error_chain(&err); + error!("Error while constructing data file writer: {}", chain); + Error::InitError(chain) + })?; + let write_result: Result<(), Error> = async { for batch in reader { let batch_data = batch.map_err(|err| { @@ -208,60 +202,158 @@ async fn write_data( error!("Error while getting record batch: {}", chain); Error::InvalidRecordValue(chain) })?; - writer.write(batch_data).await.map_err(|err| { - let chain = format_error_chain(&err); - error!("Error while writing record batch: {}", chain); - Error::WriteFailure(chain) - })?; + writer.write(batch_data).await?; } Ok(()) } .await; - if let Err(e) = &write_result { + let (data_files, close_error) = writer.close().await; + if let Some(err) = write_result.err().or(close_error) { error!( - "Batch loop failed ({}), closing writer to release resources", - e + "Writing data files failed ({}), deleting the uncommitted ones", + err ); - if let Err(close_err) = writer.close().await { - error!("Failed to close writer after batch error: {}", close_err); + delete_uncommitted_files(table.file_io(), &data_files).await; + return Err(err); + } + Ok(data_files) +} + +/// Files finalized by a failed batch are never committed, so they are removed +/// instead of being left as orphans on the object store. +async fn delete_uncommitted_files(file_io: &FileIO, data_files: &[DataFile]) { + for data_file in data_files { + if let Err(err) = file_io.delete(data_file.file_path()).await { + warn!( + "Failed to delete uncommitted data file {}: {}", + data_file.file_path(), + err + ); } - return Err(write_result.unwrap_err()); } +} - let data_files = writer.close().await.map_err(|err| { - let chain = format_error_chain(&err); - error!( - "Error while writing data records to Parquet file: {}", - chain - ); - Error::WriteFailure(chain) - })?; +/// Data file writer bound to the table's default partition spec. +/// +/// Every batch is mapped to partition keys and fanned out to one data file +/// writer per key, so a mixed batch lands in the right partitions. An +/// unpartitioned table maps every batch to the single key of its spec. +/// +/// The per-partition writers are owned here rather than by iceberg's +/// `FanoutWriter` because its `close` stops at the first failing writer and +/// loses the files the other writers already finalized. +struct TableWriter { + partitioner: Partitioner, + builder: ParquetDataFileWriterBuilder, + writers: HashMap, +} - let table_commit = Transaction::new(table); +enum Partitioner { + Unpartitioned(PartitionKey), + /// Boxed to keep the enum small (`clippy::large_enum_variant`). + Partitioned(Box), +} - let action = table_commit.fast_append().add_data_files(data_files); +impl TableWriter { + fn new(table: &Table, builder: ParquetDataFileWriterBuilder) -> iceberg::Result { + let partition_spec = table.metadata().default_partition_spec(); + let schema = table.metadata().current_schema().clone(); + let partitioner = if partition_spec.is_unpartitioned() { + // A spec whose fields are all `Transform::Void` counts as unpartitioned, + // but its partition type still has one field per void field and the + // commit rejects data files whose partition value has a different arity. + let void_field_count = table.metadata().default_partition_type().fields().len(); + Partitioner::Unpartitioned(PartitionKey::new( + partition_spec.as_ref().clone(), + schema, + Struct::from_iter(vec![None; void_field_count]), + )) + } else { + let splitter = RecordBatchPartitionSplitter::try_new_with_computed_values( + schema, + partition_spec.clone(), + )?; + Partitioner::Partitioned(Box::new(splitter)) + }; + Ok(Self { + partitioner, + builder, + writers: HashMap::new(), + }) + } - let tx = action.apply(table_commit).map_err(|err| { - let chain = format_error_chain(&err); - error!( - "Failed to apply transaction on table with UUID: {}, Error: {}", - table.metadata().uuid(), - chain - ); - Error::TransactionApplyError(chain) - })?; + async fn write(&mut self, batch: RecordBatch) -> Result<(), Error> { + let Self { + partitioner, + builder, + writers, + } = self; + match partitioner { + Partitioner::Unpartitioned(partition_key) => { + write_partition(writers, builder, partition_key, batch).await + } + Partitioner::Partitioned(splitter) => { + let partitioned_batches = splitter.split(&batch).map_err(|err| { + let chain = format_error_chain(&err); + error!("Error while splitting record batch by partition: {}", chain); + Error::InvalidRecordValue(chain) + })?; + for (partition_key, partition_batch) in partitioned_batches { + write_partition(writers, builder, &partition_key, partition_batch).await?; + } + Ok(()) + } + } + } - let _table = tx.commit(catalog).await.map_err(|err| { - let chain = format_error_chain(&err); - error!( - "Failed to commit transaction on table with UUID: {}, Error: {}", - table.metadata().uuid(), - chain - ); - Error::CatalogCommitError(chain) - })?; - Ok(()) + /// Closes every partition writer even after one of them fails, so the + /// caller learns about all finalized files and can commit or delete them. + /// Returns the first close failure alongside the files. + async fn close(self) -> (Vec, Option) { + let mut data_files = Vec::new(); + let mut first_error = None; + for mut writer in self.writers.into_values() { + match writer.close().await { + Ok(files) => data_files.extend(files), + Err(err) => { + let close_error = write_failure(err); + if first_error.is_none() { + first_error = Some(close_error); + } + } + } + } + (data_files, first_error) + } +} + +/// The partition key is cloned only when a writer is created, not on every batch. +async fn write_partition( + writers: &mut HashMap, + builder: &ParquetDataFileWriterBuilder, + partition_key: &PartitionKey, + batch: RecordBatch, +) -> Result<(), Error> { + if let Some(writer) = writers.get_mut(partition_key.data()) { + return writer.write(batch).await.map_err(write_failure); + } + let writer = builder + .build(Some(partition_key.clone())) + .await + .map_err(write_failure)?; + writers + .entry(partition_key.data().clone()) + .or_insert(writer) + .write(batch) + .await + .map_err(write_failure) +} + +fn write_failure(err: iceberg::Error) -> Error { + let chain = format_error_chain(&err); + error!("Error while writing data file: {}", chain); + Error::WriteFailure(chain) } #[async_trait] @@ -272,3 +364,156 @@ pub trait Router: std::fmt::Debug + Sync + Send { messages: Vec, ) -> Result<(), crate::Error>; } + +#[cfg(test)] +mod tests { + use super::*; + use iceberg::Runtime as IcebergRuntime; + use iceberg::spec::{ + FormatVersion, Literal, NestedField, PrimitiveLiteral, PrimitiveType, + Schema as IcebergSchema, SortOrder, TableMetadataBuilder, Transform, Type, + UnboundPartitionSpec, + }; + use std::collections::HashMap; + use std::sync::OnceLock; + + const REGION_FIELD_ID: i32 = 2; + + // Table keeps only tokio handles, so the runtime must outlive every table. + fn test_runtime() -> &'static tokio::runtime::Runtime { + static RUNTIME: OnceLock = OnceLock::new(); + RUNTIME.get_or_init(|| tokio::runtime::Runtime::new().expect("Failed to create runtime")) + } + + fn in_memory_table(partition_spec: UnboundPartitionSpec) -> Table { + let schema = IcebergSchema::builder() + .with_fields(vec![ + NestedField::required(1, "id", Type::Primitive(PrimitiveType::Long)).into(), + NestedField::required( + REGION_FIELD_ID, + "region", + Type::Primitive(PrimitiveType::String), + ) + .into(), + ]) + .build() + .expect("Failed to build schema"); + let metadata = TableMetadataBuilder::new( + schema, + partition_spec, + SortOrder::unsorted_order(), + "memory://warehouse/test/events".to_string(), + FormatVersion::V2, + HashMap::new(), + ) + .expect("Failed to create table metadata") + .build() + .expect("Failed to build table metadata") + .metadata; + + Table::builder() + .identifier(TableIdent::from_strs(["test", "events"]).expect("Failed to build ident")) + .metadata(metadata) + .file_io(FileIO::new_with_memory()) + .runtime(IcebergRuntime::new(test_runtime())) + .build() + .expect("Failed to build table") + } + + fn json_payloads(rows: &[(i64, &str)]) -> Vec { + rows.iter() + .map(|(id, region)| Payload::Json(simd_json::json!({ "id": *id, "region": *region }))) + .collect() + } + + fn write(table: &Table, payloads: &[Payload]) -> Vec { + test_runtime() + .block_on(write_data_files(payloads, table, Schema::Json)) + .expect("Failed to write data files") + } + + fn partition_region(data_file: &DataFile) -> String { + match &data_file.partition()[0] { + Some(Literal::Primitive(PrimitiveLiteral::String(region))) => region.clone(), + other => panic!("Expected string partition value, got {other:?}"), + } + } + + #[test] + fn given_partitioned_table_should_write_one_data_file_per_partition_value() { + let partition_spec = UnboundPartitionSpec::builder() + .add_partition_field(REGION_FIELD_ID, "region", Transform::Identity) + .expect("Failed to add partition field") + .build(); + let table = in_memory_table(partition_spec); + let payloads = json_payloads(&[(1, "eu"), (2, "us"), (3, "eu")]); + + let data_files = write(&table, &payloads); + + assert_eq!(data_files.len(), 2); + let by_region: HashMap = data_files + .iter() + .map(|data_file| (partition_region(data_file), data_file)) + .collect(); + assert_eq!(by_region["eu"].record_count(), 2); + assert!(by_region["eu"].file_path().contains("/data/region=eu/")); + assert_eq!(by_region["us"].record_count(), 1); + assert!(by_region["us"].file_path().contains("/data/region=us/")); + } + + #[test] + fn given_void_only_partition_spec_should_write_null_partition_values() { + let partition_spec = UnboundPartitionSpec::builder() + .add_partition_field(REGION_FIELD_ID, "region_void", Transform::Void) + .expect("Failed to add partition field") + .build(); + let table = in_memory_table(partition_spec); + let payloads = json_payloads(&[(1, "eu"), (2, "us")]); + + let data_files = write(&table, &payloads); + + assert_eq!(data_files.len(), 1); + assert_eq!(data_files[0].partition().fields(), &[None]); + assert!(!data_files[0].file_path().contains("region_void=")); + } + + #[test] + fn given_uncommitted_files_should_be_deleted_from_store() { + let table = in_memory_table(UnboundPartitionSpec::builder().build()); + let payloads = json_payloads(&[(1, "eu")]); + let data_files = write(&table, &payloads); + let file_path = data_files[0].file_path().to_string(); + + let exists_after_delete = test_runtime().block_on(async { + assert!(table.file_io().exists(&file_path).await.expect("exists")); + delete_uncommitted_files(table.file_io(), &data_files).await; + table.file_io().exists(&file_path).await.expect("exists") + }); + + assert!(!exists_after_delete); + } + + #[test] + fn given_unpartitioned_table_should_write_single_data_file_without_partition() { + let table = in_memory_table(UnboundPartitionSpec::builder().build()); + let payloads = json_payloads(&[(1, "eu"), (2, "us")]); + + let data_files = write(&table, &payloads); + + assert_eq!(data_files.len(), 1); + assert_eq!(data_files[0].record_count(), 2); + assert!(data_files[0].partition().fields().is_empty()); + assert!(data_files[0].file_path().contains("/data/")); + assert!(!data_files[0].file_path().contains("region=")); + } + + #[test] + fn given_batch_without_json_payloads_should_fail_with_invalid_payload_type() { + let table = in_memory_table(UnboundPartitionSpec::builder().build()); + let payloads = vec![Payload::Text("not json".to_string())]; + + let result = test_runtime().block_on(write_data_files(&payloads, &table, Schema::Text)); + + assert!(matches!(result, Err(Error::InvalidPayloadType))); + } +} diff --git a/core/integration/tests/connectors/fixtures/iceberg/container.rs b/core/integration/tests/connectors/fixtures/iceberg/container.rs index cf5386c8bb..71332f0ab9 100644 --- a/core/integration/tests/connectors/fixtures/iceberg/container.rs +++ b/core/integration/tests/connectors/fixtures/iceberg/container.rs @@ -275,10 +275,11 @@ pub trait IcebergOps: Sync { &self, namespace: &str, table: &str, + partition_spec: Option, ) -> impl std::future::Future> + Send { async move { let url = format!("{}/v1/namespaces/{namespace}/tables", self.catalog_url()); - let body = serde_json::json!({ + let mut body = serde_json::json!({ "name": table, "schema": { "type": "struct", @@ -292,6 +293,9 @@ pub trait IcebergOps: Sync { ] } }); + if let Some(partition_spec) = partition_spec { + body["partition-spec"] = partition_spec; + } let response = self .http_client() @@ -431,6 +435,8 @@ pub struct SnapshotSummary { pub added_data_files: Option, #[serde(rename = "added-records")] pub added_records: Option, + #[serde(rename = "changed-partition-count")] + pub changed_partition_count: Option, } impl IcebergOps for IcebergFixture { @@ -502,6 +508,8 @@ fn create_http_client() -> HttpClient { pub const DEFAULT_NAMESPACE: &str = "test"; pub const DEFAULT_TABLE: &str = "messages"; +/// Field id of the `active` column in the schema created by `IcebergOps::create_table`. +const ACTIVE_FIELD_ID: i32 = 5; pub struct IcebergPreCreatedFixture { inner: IcebergFixture, @@ -524,13 +532,61 @@ impl IcebergOps for IcebergPreCreatedFixture { } } +impl IcebergPreCreatedFixture { + async fn setup_with_partition_spec( + partition_spec: Option, + ) -> Result { + let inner = IcebergFixture::setup().await?; + + inner.create_namespace(DEFAULT_NAMESPACE).await?; + inner + .create_table(DEFAULT_NAMESPACE, DEFAULT_TABLE, partition_spec) + .await?; + + Ok(Self { inner }) + } +} + #[async_trait] impl TestFixture for IcebergPreCreatedFixture { async fn setup() -> Result { - let inner = IcebergFixture::setup().await?; + Self::setup_with_partition_spec(None).await + } - inner.create_namespace(DEFAULT_NAMESPACE).await?; - inner.create_table(DEFAULT_NAMESPACE, DEFAULT_TABLE).await?; + fn connectors_runtime_envs(&self) -> HashMap { + self.inner.connectors_runtime_envs() + } +} + +/// Same table as `IcebergPreCreatedFixture`, partitioned by identity(`active`). +pub struct IcebergPartitionedTableFixture { + inner: IcebergPreCreatedFixture, +} + +impl IcebergOps for IcebergPartitionedTableFixture { + fn catalog_url(&self) -> &str { + self.inner.catalog_url() + } + + fn http_client(&self) -> &HttpClient { + self.inner.http_client() + } +} + +#[async_trait] +impl TestFixture for IcebergPartitionedTableFixture { + async fn setup() -> Result { + let partition_spec = serde_json::json!({ + "spec-id": 0, + "fields": [{ + "source-id": ACTIVE_FIELD_ID, + "field-id": 1000, + "name": "active", + "transform": "identity" + }] + }); + let inner = + IcebergPreCreatedFixture::setup_with_partition_spec(Some(partition_spec)).await?; Ok(Self { inner }) } diff --git a/core/integration/tests/connectors/fixtures/iceberg/mod.rs b/core/integration/tests/connectors/fixtures/iceberg/mod.rs index 670c2513cf..0b5dd18774 100644 --- a/core/integration/tests/connectors/fixtures/iceberg/mod.rs +++ b/core/integration/tests/connectors/fixtures/iceberg/mod.rs @@ -18,5 +18,6 @@ mod container; pub use container::{ - DEFAULT_NAMESPACE, DEFAULT_TABLE, IcebergEnvAuthFixture, IcebergOps, IcebergPreCreatedFixture, + DEFAULT_NAMESPACE, DEFAULT_TABLE, IcebergEnvAuthFixture, IcebergOps, + IcebergPartitionedTableFixture, IcebergPreCreatedFixture, }; diff --git a/core/integration/tests/connectors/fixtures/mod.rs b/core/integration/tests/connectors/fixtures/mod.rs index 7eaf6fe510..e9d7a1332e 100644 --- a/core/integration/tests/connectors/fixtures/mod.rs +++ b/core/integration/tests/connectors/fixtures/mod.rs @@ -63,7 +63,8 @@ pub use http::{ HttpSinkNdjsonFixture, HttpSinkNoMetadataFixture, HttpSinkRawFixture, }; pub use iceberg::{ - DEFAULT_NAMESPACE, DEFAULT_TABLE, IcebergEnvAuthFixture, IcebergOps, IcebergPreCreatedFixture, + DEFAULT_NAMESPACE, DEFAULT_TABLE, IcebergEnvAuthFixture, IcebergOps, + IcebergPartitionedTableFixture, IcebergPreCreatedFixture, }; pub use influxdb::{ InfluxDb3SinkFixture, InfluxDb3SourceFixture, InfluxDbSinkBase64Fixture, InfluxDbSinkFixture, diff --git a/core/integration/tests/connectors/iceberg/iceberg_sink.rs b/core/integration/tests/connectors/iceberg/iceberg_sink.rs index ceb082f6fc..2e969775c1 100644 --- a/core/integration/tests/connectors/iceberg/iceberg_sink.rs +++ b/core/integration/tests/connectors/iceberg/iceberg_sink.rs @@ -17,7 +17,8 @@ use crate::connectors::create_test_messages; use crate::connectors::fixtures::{ - DEFAULT_NAMESPACE, DEFAULT_TABLE, IcebergEnvAuthFixture, IcebergOps, IcebergPreCreatedFixture, + DEFAULT_NAMESPACE, DEFAULT_TABLE, IcebergEnvAuthFixture, IcebergOps, + IcebergPartitionedTableFixture, IcebergPreCreatedFixture, }; use bytes::Bytes; use iggy::prelude::{IggyMessage, Partitioning}; @@ -33,6 +34,9 @@ const ICEBERG_SINK_KEY: &str = "iceberg"; const SNAPSHOT_POLL_ATTEMPTS: usize = 30; const SNAPSHOT_POLL_INTERVAL_MS: u64 = 500; const BULK_SNAPSHOT_POLL_ATTEMPTS: usize = 60; +/// `create_test_messages` alternates `active`, so this many messages span both partition values. +const PARTITIONED_MESSAGE_COUNT: usize = 5; +const ACTIVE_PARTITION_COUNT: usize = 2; #[iggy_harness( server(connectors_runtime(config_path = "tests/connectors/iceberg/sink.toml")), @@ -206,6 +210,82 @@ async fn iceberg_sink_handles_bulk_messages( assert!(sinks[0].last_error.is_none()); } +#[iggy_harness( + server(connectors_runtime(config_path = "tests/connectors/iceberg/sink.toml")), + seed = seeds::connector_stream +)] +async fn iceberg_sink_routes_messages_to_table_partitions( + harness: &TestHarness, + fixture: IcebergPartitionedTableFixture, +) { + let client = harness.root_client().await.unwrap(); + + let stream_id: Identifier = seeds::names::STREAM.try_into().unwrap(); + let topic_id: Identifier = seeds::names::TOPIC.try_into().unwrap(); + + let test_messages = create_test_messages(PARTITIONED_MESSAGE_COUNT); + let mut messages: Vec = test_messages + .iter() + .enumerate() + .map(|(i, msg)| { + let payload = serde_json::to_vec(msg).expect("Failed to serialize message"); + IggyMessage::builder() + .id((i + 1) as u128) + .payload(Bytes::from(payload)) + .build() + .expect("Failed to build message") + }) + .collect(); + + client + .send_messages( + &stream_id, + &topic_id, + &Partitioning::partition_id(0), + &mut messages, + ) + .await + .expect("Failed to send messages"); + + fixture + .wait_for_snapshots( + DEFAULT_NAMESPACE, + DEFAULT_TABLE, + 1, + SNAPSHOT_POLL_ATTEMPTS, + SNAPSHOT_POLL_INTERVAL_MS, + ) + .await + .expect("Data should be written to Iceberg table"); + + let metadata = fixture + .get_table_metadata(DEFAULT_NAMESPACE, DEFAULT_TABLE) + .await + .expect("Failed to get table metadata"); + let snapshots = metadata + .metadata + .snapshots + .expect("Table should have a snapshot"); + assert_eq!(snapshots.len(), 1, "One batch must produce one snapshot"); + let summary = snapshots[0] + .summary + .as_ref() + .expect("Snapshot should have a summary"); + + assert_eq!( + summary.added_records.as_deref(), + Some(PARTITIONED_MESSAGE_COUNT.to_string().as_str()) + ); + assert_eq!( + summary.changed_partition_count.as_deref(), + Some(ACTIVE_PARTITION_COUNT.to_string().as_str()) + ); + assert_eq!( + summary.added_data_files.as_deref(), + Some(ACTIVE_PARTITION_COUNT.to_string().as_str()) + ); +} + #[iggy_harness( server(connectors_runtime(config_path = "tests/connectors/iceberg/sink_default_credentials.toml")), seed = seeds::connector_stream From 9b7cc5c4cf38189f394e50c9fa53fe6f099c68db Mon Sep 17 00:00:00 2001 From: haubur Date: Tue, 1 Sep 2026 20:09:40 +0200 Subject: [PATCH 031/182] chore(docs): core/sdk lib.rs and IggyClient docs (#3809) --- core/common/src/traits/partitioner.rs | 16 +- core/common/src/types/message/partitioning.rs | 17 +- core/sdk/src/clients/client.rs | 736 +++++++++++++++++- core/sdk/src/lib.rs | 275 +++++++ 4 files changed, 1017 insertions(+), 27 deletions(-) diff --git a/core/common/src/traits/partitioner.rs b/core/common/src/traits/partitioner.rs index c665c82d38..0617a9c985 100644 --- a/core/common/src/traits/partitioner.rs +++ b/core/common/src/traits/partitioner.rs @@ -21,8 +21,22 @@ use crate::types::message::IggyMessage; use std::fmt::Debug; /// The trait represent the logic responsible for calculating the partition ID and is used by the `IggyClient`. -/// This might be especially useful when the partition ID is not constant and might be calculated based on the stream ID, topic ID and other parameters. +/// +/// Iggy uses a hierarchical model for append-only logs. A stream contains topics which hold partitions. Each partition is an append-only log.[^note] +/// A producer of messages such as an `IggyProducer`, that appends messages to the log, may want to choose which partition to write the messages into. +/// To do that, a producer can take a type that implements this trait. +/// This is especially useful when computing the partition ID requires some client side info, i.e. stream ID, topic ID and/ or [`IggyMessage`] attributes. +/// +/// Note the difference between [`Partitioning`] and [`Partitioner`]. [`Partitioning`] is a type used to set the _partitioning strategy_ for a producer. +/// If you use both, the [`Partitioner`] overwrites the strategy, sets it to [`PartitioningKind::PartitionId`] and the partition ID is +/// calculated with the logic implemented in [`Partitioner::calculate_partition_id()`]. +/// +/// [^note]: [Website docs on how Iggy organizes data.](https://iggy.apache.org/docs/#how-iggy-organizes-data) +/// +/// [`Partitioning`]: crate::Partitioning +/// [`PartitioningKind::PartitionId`]: crate::PartitioningKind::PartitionId pub trait Partitioner: Send + Sync + Debug { + /// Calculate a partition ID. fn calculate_partition_id( &self, stream_id: &Identifier, diff --git a/core/common/src/types/message/partitioning.rs b/core/common/src/types/message/partitioning.rs index 34694e47c8..615aa8ffb6 100644 --- a/core/common/src/types/message/partitioning.rs +++ b/core/common/src/types/message/partitioning.rs @@ -25,11 +25,21 @@ use std::{ hash::{Hash, Hasher}, }; -/// `Partitioning` is used to specify to which partition the messages should be sent. -/// It has the following kinds: +/// A type that defines a what strategy the server should choose to partition the messages. +/// +/// Iggy uses a hierarchical model for append-only logs. A stream contains topics which hold partitions. Each partition is an append-only log.[^note] +/// A producer of messages such as an `IggyProducer`, that appends messages to the log can choose between three partitioning strategies. /// - `Balanced` - the partition ID is calculated by the server using the round-robin algorithm. -/// - `PartitionId` - the partition ID is provided by the client. /// - `MessagesKey` - the partition ID is calculated by the server using the hash of the provided messages key. +/// - `PartitionId` - the partition ID is provided by the client. +/// +/// Note, that using a [`Partitioner`] on top of [`Partitioning`] sets the strategy to [`PartitioningKind::PartitionId`]. The value is then computed +/// based on your concrete implementation of [`Partitioner::calculate_partition_id()`]. +/// +/// [^note]: [Website docs on how Iggy organizes data.](https://iggy.apache.org/docs/#how-iggy-organizes-data) +/// +/// [`Partitioner`]: crate::Partitioner +/// [`Partitioner::calculate_partition_id()`]: crate::Partitioner::calculate_partition_id #[serde_as] #[derive(Debug, Serialize, Deserialize, PartialEq, Eq, Clone)] pub struct Partitioning { @@ -160,6 +170,7 @@ impl Partitioning { } /// Maximum size of the Partitioning struct + #[doc(hidden)] pub const fn maximum_byte_size() -> usize { 2 + 255 } diff --git a/core/sdk/src/clients/client.rs b/core/sdk/src/clients/client.rs index e91a601a5d..bc6a6382b1 100644 --- a/core/sdk/src/clients/client.rs +++ b/core/sdk/src/clients/client.rs @@ -56,9 +56,173 @@ const SESSION_CONTROL_CODES: [u32; 5] = [ LOGIN_REGISTER_WITH_PAT_CODE, ]; -/// The main client struct which implements all the `Client` traits and wraps the underlying low-level client for the specific transport. +/// A high-level, transport-agnostic client for an Iggy server. /// -/// It also provides the additional builders for the standalone consumer, consumer group, and producer. +/// `IggyClient` wraps a transport-specific low-level **client** ([`ClientWrapper`]) and +/// **provides access to the full server API**. +/// Iggy comes with four options for client-server communication: TCP, QUIC, WebSocket and HTTP. +/// The `IggyClient` is configured with one of these transport modes, hence abstracting +/// transport specific implementations away. +/// +/// The [`ClientWrapper`] lives behind an [`IggyRwLock`] so that the connection +/// can be shared safely. You create a single client and use it from many tasks +/// at once (e.g. producers, consumers). Every operation takes a shared read +/// guard. +/// +/// A [`Partitioner`] and a client-side [`EncryptorKind`] are optional, and both +/// default to disabled. The [`Partitioner`] computes on the client-side the target +/// partition for messages published without an explicit partition. Hence, you can +/// configure routing to partitions within a topic yourself dependent on the stream, +/// topic, and/ or message contents. +/// +/// The [`EncryptorKind`] encrypts each message payload before it leaves the client and decrypts it on +/// the way back, keeping payloads opaque to the server. Attach either through +/// [`create()`]. +/// +/// # What you can do +/// +/// Configure a connection with an Iggy server and interact with it. +/// The `IggyClient` provides various methods to setup the connection using connection strings, +/// builder patterns or an already existing [`ClientWrapper`]. +/// You can spawn [`IggyConsumer`]s and [`IggyProducer`]s that share that connection. +/// +/// The full server API is split into domain-specific traits. +/// `IggyClient` implements [`Client`], the supertrait, which pulls every domain-specific trait. +/// Bring the one you need into scope to call its methods. +/// `use iggy::prelude::*` brings all of them in at once. +/// +/// - [`SystemClient`]: ping, server statistics, snapshots, and connected-client info. +/// - [`UserClient`]: create, inspect, update, and delete users and their permissions. +/// - [`PersonalAccessTokenClient`]: create, list, and delete personal access tokens, log in with one. +/// - [`StreamClient`]: create, get, update, delete, and purge streams. +/// - [`TopicClient`]: create, get, update, delete, and purge topics within a stream. +/// - [`PartitionClient`]: add and remove partitions on a topic. +/// - [`SegmentClient`]: delete closed segments from a partition. +/// - [`ConsumerGroupClient`]: create, get, delete, and join or leave consumer groups. +/// - [`ConsumerOffsetClient`]: store, read, and delete consumer offsets. +/// - [`MessageClient`]: send and poll messages, and flush the unsaved buffer. +/// +/// Additionally, you can bypass invoking methods from these traits and directly talk to the server with [`send_binary_request`] and [`send_http_request`] for http. +/// Both trade typed API's safety for low-level control. You need to know the server codes and the wire format. +/// +/// # Usage +/// +/// The typical lifecycle of an `IggyClient` is construct, [`connect()`], use, and finally shutdown. +/// +/// 1. Construct a client from a connection string ([`from_connection_string()`]), +/// from the [`builder()`], or by wrapping an existing transport client with +/// [`new()`] / [`create()`]. +/// 2. Call [`connect()`] to establish the transport-level connection. If the transport was +/// configured with auto-login, this also authenticates. Otherwise call +/// [`login_user()`] afterwards. For HTTP always call [`login_user()`] instead of [`connect()`]. +/// 3. Spawn [`IggyConsumer`]s and [`IggyProducer`]s to write to, and consume messages +/// from, the server. +/// 4. To shut everything down, call [`IggyConsumer::shutdown()`] on each consumer to store their +/// final offset and leave consumer groups. Then, call [`IggyProducer::shutdown()`] on each +/// [`IggyProducer`] so that _background_ producers flush the latest state. Finally, +/// call [`shutdown()`] on the [`IggyClient`] which closes the connection. +/// Use [`disconnect()`] rather than [`shutdown()`] to close the connection but keep the client usable, as a +/// client that has been shut down cannot reconnect. Note, if `auto-login` is configured, the client +/// will reconnect automatically and undo the disconnect. +/// +/// # Examples +/// +/// Build a client from a connection string, connect, publish through a +/// background batching producer, consume with a standalone consumer, and shut down cleanly. +/// +/// ```no_run +/// use iggy::prelude::*; +/// use futures_util::StreamExt; +/// use std::str::FromStr; +/// +/// # async fn run() -> Result<(), IggyError> { +/// // Auto-logs in from the credentials in the string and retries forever on disconnect. +/// let client = IggyClient::builder_from_connection_string( +/// "iggy+tcp://user:secret@localhost:8090\ +/// ?reconnection_retries=unlimited&reconnection_interval=1s&heartbeat_interval=5s&nodelay=true", +/// )? +/// .build()?; +/// client.connect().await?; +/// +/// // A background producer batches in the background, retries failed sends, +/// // and creates a topic. +/// let producer = client +/// .producer("stream_name", "topic_name")? +/// .background( +/// BackgroundConfig::builder() +/// .batch_length(1000) +/// .linger_time(IggyDuration::ONE_SECOND) +/// .build(), +/// ) +/// .partitioning(Partitioning::balanced()) +/// .send_retries(Some(3), Some(NonZeroIggyDuration::ONE_SECOND)) +/// .create_topic_if_not_exists( +/// 3, +/// IggyExpiry::ServerDefault, +/// MaxTopicSize::ServerDefault, +/// ) +/// .build(); +/// producer.init().await?; +/// producer +/// .send(vec![IggyMessage::from_str("our-first-message")?]) +/// .await?; +/// +/// // A consumer pinned to one partition, committing its offset +/// // automatically when messages are polled (not consumed!). +/// let mut consumer = client +/// .consumer("consumer_name", "stream_name", "topic_name", 1)? +/// .auto_commit(AutoCommit::When(AutoCommitWhen::PollingMessages)) +/// .polling_strategy(PollingStrategy::next()) +/// .poll_interval(IggyDuration::ONE_SECOND) +/// .batch_length(1000) +/// .build(); +/// consumer.init().await?; +/// +/// while let Some(message) = consumer.next().await { +/// let message = message?; +/// // Handle `message.message.payload` here however required +/// break; +/// } +/// +/// // Finish commit before stopping (comp. method docs). +/// consumer.shutdown().await?; +/// // Finish flush before stopping (comp. method docs). +/// producer.shutdown().await; +/// +/// client.shutdown().await?; +/// # Ok(()) +/// # } +/// ``` +/// +/// [`IggyConsumer`]: crate::prelude::IggyConsumer +/// [`IggyProducer`]: crate::prelude::IggyProducer +/// [`IggyConsumer::shutdown()`]: crate::prelude::IggyConsumer::shutdown +/// [`IggyProducer::shutdown()`]: crate::prelude::IggyProducer::shutdown +/// [`new()`]: IggyClient::new +/// [`create()`]: IggyClient::create +/// [`send_binary_request`]: IggyClient::send_binary_request +/// [`send_http_request`]: IggyClient::send_http_request +/// [`shutdown()`]: IggyClient::shutdown +/// [`login_user()`]: crate::prelude::UserClient::login_user +/// [`connect()`]: IggyClient::connect +/// [`disconnect()`]: IggyClient::disconnect +/// [`producer()`]: IggyClient::producer +/// [`consumer()`]: IggyClient::consumer +/// [`builder()`]: IggyClient::builder +/// [`from_connection_string()`]: IggyClient::from_connection_string +/// [`consumer_group()`]: IggyClient::consumer_group +/// [`Client`]: crate::prelude::Client +/// [`SystemClient`]: crate::prelude::SystemClient +/// [`UserClient`]: crate::prelude::UserClient +/// [`PersonalAccessTokenClient`]: crate::prelude::PersonalAccessTokenClient +/// [`StreamClient`]: crate::prelude::StreamClient +/// [`TopicClient`]: crate::prelude::TopicClient +/// [`PartitionClient`]: crate::prelude::PartitionClient +/// [`SegmentClient`]: crate::prelude::SegmentClient +/// [`MessageClient`]: crate::prelude::MessageClient +/// [`ConsumerOffsetClient`]: crate::prelude::ConsumerOffsetClient +/// [`ConsumerGroupClient`]: crate::prelude::ConsumerGroupClient +/// [`ClusterClient`]: crate::prelude::ClusterClient #[derive(Debug)] #[allow(dead_code)] pub struct IggyClient { @@ -75,19 +239,211 @@ impl Default for IggyClient { } impl IggyClient { - /// Creates a new `IggyClientBuilder`. + /// Returns an empty [`IggyClientBuilder`]. + /// + /// The returned builder is not ready to be [`IggyClientBuilder::build()`]. + /// It sill needs to configure a mode of transport. pub fn builder() -> IggyClientBuilder { IggyClientBuilder::new() } - /// Creates a new `IggyClientBuilder` from the provided connection string. + /// Creates an [`IggyClientBuilder`] with the transport preconfigured from a + /// connection string. + /// + /// The transport is selected from the scheme: + /// - `iggy://` defaults to TCP. + /// - `iggy+tcp://` for TCP. + /// - `iggy+quic://` for QUIC. + /// - `iggy+http://` for HTTP. + /// - `iggy+ws://` for WebSocket. + /// + /// Authentication at the server is mandatory. + /// - user + password: `:@:` + /// - personal access token: `@host:port` + /// + /// Optional `?key=value&key=value` queries carry transport specific + /// configuration. The query is parsed per transport, so the accepted keys differ + /// by scheme. An unknown key is rejected. + /// If no queries are provided optional configurations are automatically set to their default values. + /// + /// # Examples + /// + /// Each example lists configuration options per transport mode and shows + /// one concrete example. + /// + /// If the value of an option is + /// - a duration pass a `humantime` string such as `5s`, `500ms`, or `1h 1m 1s`. + /// These are cast into an [`IggyDuration`]/[`NonZeroIggyDuration`]. + /// For [`IggyDuration`], `unlimited`, `none`, `disabled`, and `0` parse to zero. + /// - a retry count use either the literal `unlimited` or a number such as `5`. + /// - a bool use the literal `true` or `false`. + /// - bytes and millisecond options provide a number such as `1024`. + /// + /// ## TCP + /// + /// The same options apply for `iggy://` and `iggy+tcp://`. + /// + /// - `tls`: bool. Enable/disable TLS. Default: `false`. + /// - `tls_domain`: string. Server name to validate the certificate against. Default: unset. + /// - `tls_ca_file`: filesystem path. Extra CA certificate to trust. Default: unset. + /// - `reconnection_retries`: "unlimited" or u32. Number of attempts to connect. Default: `unlimited`. + /// - `reconnection_interval`: [`NonZeroIggyDuration`]. Wait between reconnection attempts. Default: `1s`. + /// - `reestablish_after`: [`IggyDuration`]. Grace period before reconnecting. Default: `5s`. + /// - `heartbeat_interval`: [`NonZeroIggyDuration`]. Client heartbeat period. Default: `5s`. + /// - `nodelay`: `bool`. Disable Nagle's algorithm (`TCP_NODELAY`). Default: `false`. + /// + /// ```no_run + /// use iggy::prelude::*; + /// + /// # fn run() -> Result<(), IggyError> { + /// let client = IggyClient::builder_from_connection_string( + /// "iggy+tcp://user:secret@localhost:8090\ + /// ?tls=true&tls_domain=localhost&tls_ca_file=/etc/iggy/ca.pem\ + /// &reconnection_retries=unlimited&reconnection_interval=1s&reestablish_after=5s\ + /// &heartbeat_interval=5s&nodelay=true", + /// )? + /// .build()?; + /// # Ok(()) + /// # } + /// ``` + /// + /// ## QUIC + /// + /// - `validate_certificate`: bool. Verify the server certificate. Default: `false`. + /// - `heartbeat_interval`: [`NonZeroIggyDuration`]. Client heartbeat period. Default: `5s`. + /// - `reconnection_max_retries`: "unlimited" or u32. Number of attempts to connect. Default: `unlimited`. + /// - `reconnection_interval`: [`NonZeroIggyDuration`]. Wait between reconnection attempts. Default: `1s`. + /// - `reconnection_reestablish_after`: [`IggyDuration`]. Grace period before reconnecting. Default: `5s`. + /// - `response_buffer_size`: u64. Number of bytes in the response receive buffer. Default: `10000000`. + /// - `max_concurrent_bidi_streams`: u64. Number of concurrent bidirectional streams. Default: `10000`. + /// - `datagram_send_buffer_size`: u64. Number of bytes in the datagram send buffer. Default: `100000`. + /// - `initial_mtu`: u16. Initial MTU estimate (in bytes). Default: `1200`. + /// - `send_window`: u64. Number of bytes bytes of the flow-control send window. Default: `100000`. + /// - `receive_window`: u64. Number of bytes of the flow-control receive window. Default: `100000`. + /// - `keep_alive_interval`: u64. QUIC keep-alive period (in milliseconds). Default: `5000`. + /// - `max_idle_timeout`: u64. Close after this much idle time (in middleseconds). Default: `10000`. + /// + /// ```no_run + /// use iggy::prelude::*; + /// + /// # fn run() -> Result<(), IggyError> { + /// let client = IggyClient::builder_from_connection_string( + /// "iggy+quic://user:secret@localhost:8080\ + /// ?validate_certificate=true&heartbeat_interval=5s\ + /// &reconnection_max_retries=unlimited&reconnection_interval=1s&reconnection_reestablish_after=5s\ + /// &response_buffer_size=10000000&max_concurrent_bidi_streams=10000\ + /// &datagram_send_buffer_size=100000&initial_mtu=1200\ + /// &send_window=100000&receive_window=100000\ + /// &keep_alive_interval=5000&max_idle_timeout=10000", + /// )? + /// .build()?; + /// # Ok(()) + /// # } + /// ``` + /// + /// ## HTTP + /// + /// A REST transport. Iggy uses the `reqwest` crate to manage the HTTP client. + /// Hence, transport specific configuration is abstracted away. + /// + /// Configurable options are: + /// - `heartbeat_interval`: [`NonZeroIggyDuration`]. Client heartbeat period. Default: `5s`. + /// - `retries`: u32. Number of retries when sending a request. Default: `3`. + /// + /// ```no_run + /// use iggy::prelude::*; + /// + /// # async fn run() -> Result<(), IggyError> { + /// let client = IggyClient::builder_from_connection_string( + /// "iggy+http://localhost:3000?heartbeat_interval=5s&retries=3", + /// )? + /// .build()?; + /// client.login_user("user", "password").await?; + /// # Ok(()) + /// # } + /// ``` + /// + /// ## WebSocket + /// + /// - `heartbeat_interval`: [`NonZeroIggyDuration`]. Client heartbeat period. Default: `5s`. + /// - `reconnection_retries`: "unlimited" or u32. Number of attempts to connect. Default: `unlimited`. + /// - `reconnection_interval`: [`NonZeroIggyDuration`]. Wait between reconnection attempts. Default: `1s`. + /// - `reestablish_after`: [`IggyDuration`]. Grace period before reconnecting. Default: `5s`. + /// - `read_buffer_size`: usize. Size of the read buffer in bytes. Default: `131072`. + /// - `write_buffer_size`: usize. Size of the write buffer in bytes. Default: `131072`. + /// - `max_write_buffer_size`: usize. Maximum size of the write buffer in bytes. Default: `usize::MAX`. + /// - `max_message_size`: usize. Maximum accepted message size in bytes. Default: `67108864`. + /// - `max_frame_size`: usize. Maximum accepted frame size in bytes. Default: `16777216`. + /// - `accept_unmasked_frames`: bool. Accept/ decline unmasked frames. Default: `false`. + /// - `tls`: `bool`. Enable/disbale TLS. Default: `false`. + /// - `tls_domain`: string. Server name to validate the certificate against. Default: unset. + /// - `tls_ca_file`: filesystem path. Extra CA certificate to trust. Default: unset. + /// - `tls_validate_certificate`: bool. Whether to verify the server certificate. Default: `false`. + /// + /// ```no_run + /// use iggy::prelude::*; + /// + /// # fn run() -> Result<(), IggyError> { + /// let client = IggyClient::builder_from_connection_string( + /// "iggy+ws://user:secret@localhost:8092\ + /// ?heartbeat_interval=5s&reconnection_retries=unlimited&reconnection_interval=1s&reestablish_after=5s\ + /// &read_buffer_size=131072&write_buffer_size=131072\ + /// &max_message_size=67108864&max_frame_size=16777216&accept_unmasked_frames=false\ + /// &tls=true&tls_domain=localhost&tls_ca_file=/etc/iggy/ca.pem&tls_validate_certificate=true", + /// )? + /// .build()?; + /// # Ok(()) + /// # } + /// ``` + /// + /// # Errors + /// + /// Returns [`IggyError::InvalidConnectionString`] if the connection string is malformed. + /// + /// [`IggyDuration`]: crate::prelude::IggyDuration + /// [`NonZeroIggyDuration`]: crate::prelude::NonZeroIggyDuration pub fn builder_from_connection_string( connection_string: &str, ) -> Result { IggyClientBuilder::from_connection_string(connection_string) } - /// Creates a new `IggyClient` with the provided client implementation for the specific transport. + /// Creates a new `IggyClient` from an already-constructed transport client. + /// + /// Use this when you built the transport client yourself + /// and want the full high-level `IggyClient` surface on top of it. To start + /// from a connection string instead, prefer + /// [`from_connection_string`](IggyClient::from_connection_string) or + /// [`builder_from_connection_string`](IggyClient::builder_from_connection_string). + /// To also attach a [`Partitioner`] or client-side [`EncryptorKind`], use + /// [`create`]. + /// + /// # Examples + /// + /// Build a [`TcpClient`] from a config, wrap it in a [`ClientWrapper`], and + /// hand it to `IggyClient` for the full server API. + /// + /// ```no_run + /// use iggy::prelude::*; + /// use std::sync::Arc; + /// + /// # fn run() -> Result<(), IggyError> { + /// let config = TcpClientConfigBuilder::new() + /// .with_server_address("127.0.0.1:8090".to_owned()) + /// .build()?; + /// let tcp_client = TcpClient::create(Arc::new(config))?; + /// + /// let client = IggyClient::new(ClientWrapper::Tcp(tcp_client)); + /// # let _ = client; + /// # Ok(()) + /// # } + /// ``` + /// + /// [`create`]: IggyClient::create + /// [`Partitioner`]: crate::prelude::Partitioner + /// [`EncryptorKind`]: crate::prelude::EncryptorKind + /// [`TcpClient`]: crate::prelude::TcpClient + /// [`ClientWrapper`]: crate::prelude::ClientWrapper pub fn new(client: ClientWrapper) -> Self { let client = IggyRwLock::new(client); IggyClient { @@ -98,7 +454,24 @@ impl IggyClient { } } - /// Creates a new `IggyClient` from the provided connection string. + /// Creates a new `IggyClient` directly from a connection string. + /// + /// This is a shortcut for [`builder_from_connection_string`] followed by + /// [`build`](IggyClientBuilder::build) when no partitioner or encryptor is + /// needed. + /// + /// Refer to [`builder_from_connection_string`] for concise examples on how + /// to use a connection string. + /// To also attach a [`Partitioner`] or client-side [`EncryptorKind`], use + /// [`create`]. + /// + /// # Errors + /// + /// Returns [`IggyError::InvalidConnectionString`] if the connection string is + /// malformed. + /// + /// [`builder_from_connection_string`]: IggyClient::builder_from_connection_string + /// [`create`]: IggyClient::create pub fn from_connection_string(connection_string: &str) -> Result { match ConnectionStringUtils::parse_protocol(connection_string)? { TransportProtocol::Tcp => Ok(IggyClient::new(ClientWrapper::Tcp( @@ -116,7 +489,62 @@ impl IggyClient { } } - /// Creates a new `IggyClient` with the provided client implementation for the specific transport and the optional implementations for the `partitioner` and `encryptor`. + /// Creates a new `IggyClient` from a transport client, with an optional + /// [`Partitioner`] and client-side [`EncryptorKind`]. + /// + /// The partitioner picks the target partition for messages published without + /// an explicit partition assigned to them. Note, that setting a [`Partitioner`] overrides a producer's + /// [`Partitioning`](crate::prelude::Partitioning) the partition id is + /// computed client-side and the partitioning strategy is forced to [`PartitioningKind::PartitionId`] using the computed ID. + /// The encryptor encrypts payloads before they leave the client. Pass [`None`] for either to + /// disable it, just as [`new`](IggyClient::new) does for both. + /// + /// # Examples + /// + /// Wrap a [`TcpClient`] together with a custom [`Partitioner`] and an + /// AES-256-GCM payload [`EncryptorKind`]. + /// + /// ```no_run + /// use iggy::prelude::*; + /// use std::sync::Arc; + /// + /// // Routes every message to partition 1. + /// #[derive(Debug)] + /// struct FixedPartitioner; + /// + /// impl Partitioner for FixedPartitioner { + /// fn calculate_partition_id( + /// &self, + /// _stream_id: &Identifier, + /// _topic_id: &Identifier, + /// _messages: &[IggyMessage], + /// ) -> Result { + /// Ok(1) + /// } + /// } + /// + /// # fn run() -> Result<(), IggyError> { + /// let tcp_client = + /// TcpClient::from_connection_string("iggy+tcp://user:secret@localhost:8090")?; + /// + /// let partitioner: Arc = Arc::new(FixedPartitioner); + /// let encryptor = Arc::new(EncryptorKind::Aes256Gcm(Aes256GcmEncryptor::new(&[0u8; 32])?)); + /// + /// let client = IggyClient::create( + /// ClientWrapper::Tcp(tcp_client), + /// Some(partitioner), + /// Some(encryptor), + /// ); + /// # let _ = client; + /// # Ok(()) + /// # } + /// ``` + /// + /// [`Partitioner`]: crate::prelude::Partitioner + /// [`PartitioningKind::PartitionId`]: iggy_common::PartitioningKind::PartitionId + /// [`EncryptorKind`]: crate::prelude::EncryptorKind + /// [`TcpClient`]: crate::prelude::TcpClient + /// [`ClientWrapper`]: crate::prelude::ClientWrapper pub fn create( client: ClientWrapper, partitioner: Option>, @@ -138,12 +566,78 @@ impl IggyClient { } } - /// Returns the underlying client implementation for the specific transport. + /// Returns a handle to the underlying transport client. + /// + /// The returned [`ClientWrapper`] is behind an [`IggyRwLock`]. + /// Thus, the returned type shares ownership with this `IggyClient`, meaning + /// changes made through either are visible to both. + /// + /// # Examples + /// + /// Take the shared handle, acquire a read guard, and reach the underlying + /// transport directly, here to ping the server over the raw connection. + /// + /// ```no_run + /// use iggy::prelude::*; + /// use crate::iggy::prelude::locking::IggyRwLockFn; + /// + /// # async fn run(client: IggyClient) -> Result<(), IggyError> { + /// let handle = client.client(); + /// handle.read().await.ping().await?; + /// # Ok(()) + /// # } + /// ``` pub fn client(&self) -> IggyRwLock { self.client.clone() } - /// Returns the builder for the standalone consumer. + /// Returns an [`IggyConsumerBuilder`] to build a standalone consumer. + /// + /// Copies the client and the encryptor registered with the [`IggyClient`] + /// into a [`IggyConsumerBuilder`] and returns it. + /// Sets the consumer to [`ConsumerKind::Consumer`], i.e. a single consumer, + /// with the provided name. + /// Registers a consumer for `stream`, `topic` and `partition`. + /// + /// To get builder for a load-balanced consumer group, use + /// [`consumer_group`](IggyClient::consumer_group) instead. + /// + /// Refer to the [`IggyConsumer`] type and the [`IggyConsumerBuilder`] + /// for details on how a consumer can be configured. + /// + /// # Examples + /// + /// Connect a client, build a consumer pinned to partition 1, and configure + /// it to auto-commit when messages are polled. + /// + /// ```no_run + /// use iggy::prelude::*; + /// + /// # async fn run() -> Result<(), IggyError> { + /// let client = IggyClient::from_connection_string( + /// "iggy+tcp://user:secret@localhost:8090", + /// )?; + /// client.connect().await?; + /// + /// let consumer = client + /// .consumer("consumer_name", "stream_name", "topic_name", 1)? // returns IggyConsumerBuilder from IggyClient + /// .auto_commit(AutoCommit::When(AutoCommitWhen::PollingMessages)) + /// .polling_strategy(PollingStrategy::next()) + /// .poll_interval(IggyDuration::ONE_SECOND) + /// .batch_length(1000) + /// .build(); // returns IggyConsumer from IggyConsumerBuilder + /// # let _ = consumer; + /// # Ok(()) + /// # } + /// ``` + /// + /// # Errors + /// + /// Returns [`IggyError::InvalidIdentifier`] if `name`, `stream`, or `topic` + /// is not a valid identifier. + /// + /// [`ConsumerKind::Consumer`]: crate::prelude::ConsumerKind::Consumer + /// [`IggyConsumer`]: crate::prelude::IggyConsumer pub fn consumer( &self, name: &str, @@ -163,7 +657,59 @@ impl IggyClient { )) } - /// Returns the builder for the consumer group. + /// Returns an [`IggyConsumerBuilder`] for a member of a consumer group. + /// + /// Copies the client and the encryptor registered with the [`IggyClient`] + /// into a [`IggyConsumerBuilder`] and returns it. + /// Sets the consumer to [`ConsumerKind::ConsumerGroup`], i.e. a member of a + /// load-balanced group. The provided name identifies the group. + /// Registers the member for every partition of `topic` in `stream`. The + /// group then balances those partitions across its members, so each message is + /// delivered to exactly one member. When consumers leave or join a group, + /// rebalancing (re-assigning consumers to partitions) might deliver messages again + /// in cases where they were polled, but the read did not commit yet. + /// Since the new consumer starts reading from the last commit, messages are delivered + /// at-least-once. + /// + /// For a consumer pinned to a single partition, use + /// [`consumer`](IggyClient::consumer) instead. + /// + /// Refer to the [`IggyConsumer`] type and the [`IggyConsumerBuilder`] + /// for details on how a consumer can be configured. + /// + /// # Examples + /// + /// Connect a client, build a member of a consumer group, and configure it to + /// auto-commit when polled. + /// + /// ```no_run + /// use iggy::prelude::*; + /// + /// # async fn run() -> Result<(), IggyError> { + /// let client = IggyClient::from_connection_string( + /// "iggy+tcp://user:secret@localhost:8090", + /// )?; + /// client.connect().await?; + /// + /// let consumer = client + /// .consumer_group("group_name", "stream_name", "topic_name")? // returns IggyConsumerBuilder from IggyClient + /// .auto_commit(AutoCommit::When(AutoCommitWhen::PollingMessages)) + /// .polling_strategy(PollingStrategy::next()) + /// .poll_interval(IggyDuration::ONE_SECOND) + /// .batch_length(1000) + /// .build(); // returns IggyConsumer from IggyConsumerBuilder + /// # let _ = consumer; + /// # Ok(()) + /// # } + /// ``` + /// + /// # Errors + /// + /// Returns [`IggyError::InvalidIdentifier`] if `name`, `stream`, or `topic` + /// is not a valid identifier. + /// + /// [`ConsumerKind::ConsumerGroup`]: crate::prelude::ConsumerKind::ConsumerGroup + /// [`IggyConsumer`]: crate::prelude::IggyConsumer pub fn consumer_group( &self, name: &str, @@ -182,7 +728,54 @@ impl IggyClient { )) } - /// Returns the builder for the producer. + /// Returns an [`IggyProducerBuilder`]. + /// + /// Copies the client and the encryptor registered with the [`IggyClient`] + /// into a [`IggyProducerBuilder`] and returns it. + /// + /// Binds a producer to the provided stream and topic. + /// Refer to the [`IggyProducer`] type and the [`IggyProducerBuilder`] + /// for details on how a producer can be configured. + /// + /// # Examples + /// + /// Connect a client, build a producer that creates the topic if it is + /// missing and retries failed sends, then publish a message. + /// + /// ```no_run + /// use iggy::prelude::*; + /// use std::str::FromStr; + /// + /// # async fn run() -> Result<(), IggyError> { + /// let client = IggyClient::from_connection_string( + /// "iggy+tcp://user:secret@localhost:8090", + /// )?; + /// client.connect().await?; + /// + /// let producer = client + /// .producer("stream_name", "topic_name")? // returns IggyProducerBuilder from IggyClient + /// .partitioning(Partitioning::balanced()) + /// .send_retries(Some(3), Some(NonZeroIggyDuration::ONE_SECOND)) + /// .create_topic_if_not_exists( + /// 3, + /// IggyExpiry::ServerDefault, + /// MaxTopicSize::ServerDefault, + /// ) + /// .build(); // returns IggyProducer from IggyProducerBuilder + /// producer.init().await?; + /// producer + /// .send(vec![IggyMessage::from_str("our-first-message")?]) + /// .await?; + /// # Ok(()) + /// # } + /// ``` + /// + /// # Errors + /// + /// Returns [`IggyError::InvalidIdentifier`] if `stream` or `topic` is not a + /// valid identifier. + /// + /// [`IggyProducer`]: crate::prelude::IggyProducer pub fn producer(&self, stream: &str, topic: &str) -> Result { Ok(IggyProducerBuilder::new( self.client.clone(), @@ -191,25 +784,87 @@ impl IggyClient { topic.try_into()?, topic.to_owned(), self.encryptor.clone(), - None, + self.partitioner.clone(), )) } - /// Returns the current connection information including the transport protocol and server address. - /// This is useful for verifying which server the client is connected to, especially after - /// leader redirection in a clustered environment. + /// Returns the current [`ConnectionInfo`]. + /// + /// The transport protocol and the server address the client is connected to. + /// + /// # Examples + /// + /// Connect a client and print the transport protocol and server address it + /// is connected to. + /// + /// ```no_run + /// use iggy::prelude::*; + /// + /// # async fn run() -> Result<(), IggyError> { + /// let client = IggyClient::from_connection_string( + /// "iggy+tcp://user:secret@localhost:8090", + /// )?; + /// client.connect().await?; + /// + /// let info = client.get_connection_info().await; + /// println!("connected to {} over {}", info.server_address, info.protocol); + /// # Ok(()) + /// # } + /// ``` pub async fn get_connection_info(&self) -> ConnectionInfo { self.client.read().await.get_connection_info().await } - /// Send a raw binary command (`code` + serialized `payload`), returning the - /// raw response. Binary transports only (HTTP yields `FeatureUnavailable`). + /// Sends a raw binary command (`code` plus serialized `payload`) and returns + /// the raw response payload. + /// + /// Use this method for commands the typed API does not cover, for + /// example a command you added to a forked server. + /// + /// `code` selects the command. Available codes are defined in + /// [`iggy_binary_protocol::codes`] /// - /// Login and logout codes are rejected with `InvalidCommand`. Use the - /// `login_user` / `logout_user` methods so SDK session state stays correct. + /// `payload` is the command body already serialized in the Iggy wire format, + /// and the returned [`Bytes`] is the raw response body in that same format, + /// which you need to decode yourself. The wire frame that carries both (length, code, + /// status) is documented at the [`iggy_binary_protocol`]. You pass + /// and receive only the payload, the transport configured with the client frames it. /// - /// Custom codes are forwarded to the server, which is the authority on - /// whether it implements them. + /// Only the binary transports (TCP, QUIC, WebSocket) have a raw binary path. + /// The HTTP counterpart is + /// [`send_http_request`](IggyClient::send_http_request). + /// + /// # Examples + /// + /// Ping the server over the raw binary path. `PING_CODE` takes an empty + /// payload and the server replies with an empty payload. + /// + /// ```no_run + /// use iggy::prelude::*; + /// use iggy_binary_protocol::codes::PING_CODE; + /// use bytes::Bytes; + /// + /// # async fn run() -> Result<(), IggyError> { + /// let client = IggyClient::from_connection_string( + /// "iggy+tcp://user:secret@localhost:8090", + /// )?; + /// client.connect().await?; + /// + /// let response = client.send_binary_request(PING_CODE, Bytes::new()).await?; + /// assert!(response.is_empty()); + /// # Ok(()) + /// # } + /// ``` + /// + /// # Errors + /// [`IggyError::InvalidCommand`] if `code` is one of the session-control + /// codes (login, logout, and register). Use the typed `login_user` / + /// `logout_user` methods so the SDK's session state stays correct. + /// [`IggyError::FeatureUnavailable`] on the HTTP transport, which has no + /// binary path. + /// + /// [`iggy_binary_protocol`]: iggy_binary_protocol + /// [`iggy_binary_protocol::codes`]: iggy_binary_protocol::codes pub async fn send_binary_request(&self, code: u32, payload: Bytes) -> Result { if SESSION_CONTROL_CODES.contains(&code) { return Err(IggyError::InvalidCommand); @@ -222,8 +877,43 @@ impl IggyClient { } } - /// Invoke an arbitrary HTTP endpoint and return the raw response body. HTTP - /// transport only; binary transports yield `FeatureUnavailable`. + /// Invokes a HTTP endpoint and returns the raw response body. + /// + /// This is the HTTP counterpart to + /// [`send_binary_request`](IggyClient::send_binary_request). + /// + /// `method` is the HTTP verb and `path` is joined onto the + /// client's configured API URL, e.g. `/streams`. + /// `body` (if needed) is sent as-is as the request body. The client + /// attaches its bearer token. The returned [`Bytes`] is the raw response body + /// for a response, which you decode yourself. + /// + /// # Examples + /// + /// Fetch the server's stats over the raw HTTP path with a `GET` and no body. + /// + /// ```no_run + /// use iggy::prelude::*; + /// + /// # async fn run() -> Result<(), IggyError> { + /// let client = IggyClient::from_connection_string( + /// "iggy+http://user:secret@localhost:3000", + /// )?; + /// client.login_user("user", "password").await?; + /// + /// let response = client + /// .send_http_request(HttpMethod::Get, "/stats", None) + /// .await?; + /// // `response` is the raw JSON body, decode it however required. + /// # let _ = response; + /// # Ok(()) + /// # } + /// ``` + /// + /// # Errors + /// + /// [`IggyError::FeatureUnavailable`] on the TCP, QUIC, and WebSocket + /// transports, which have no HTTP path. pub async fn send_http_request( &self, method: HttpMethod, diff --git a/core/sdk/src/lib.rs b/core/sdk/src/lib.rs index fd8b22efb2..ce2b4708f9 100644 --- a/core/sdk/src/lib.rs +++ b/core/sdk/src/lib.rs @@ -15,6 +15,281 @@ // specific language governing permissions and limitations // under the License. +//! Apache Iggy is a high-performance, persistent message streaming platform written in +//! Rust, capable of processing millions of messages per second with ultra-low latency. +//! It is part of the [`Apache Software Foundation`] (ASF). +//! +//! **This library is the Apache Iggy SDK.** +//! It exposes a low-level and a high-level API for the Apache Iggy message streaming +//! infrastructure for the Rust programming language. +//! SDKs for other programming languages can be found in [`foreign`] of the root +//! repository on GitHub. +//! +//! The core of Iggy is the message streaming server. +//! In essence it is a persisted append-only log data structure concerned with making +//! reads and writes highly efficient. +//! For that, the server exposes *commands* that can be triggered to change its state, +//! such as adding users, setting permissions, adding new streams and topics or reading +//! and writing messages to and from the log. +//! A comprehensive overview of commands can be found in the [`schema spec`] on the +//! website, or in the [`server command enum`] within the source code. +//! +//! The SDK provides tools to build production ready message-streaming applications. +//! It exposes its functionality at two levels. The [high-level API](#high-level-api) +//! is transport-agnostic and already ships with useful features that a production +//! application needs, such as message batching, retry policies and offset-tracking. +//! The [low-level API](#low-level-api) is the set of concrete transport clients that +//! speak the wire protocol directly and that the high-level API is built on top of. +//! It is recommended to start with the high-level API, and utilize the low-level API +//! in case the high-level API cannot satisfy your requirements. +//! +//! # High-level API +//! +//! The high-level API is most likely what you are looking for, especially if you are new +//! to building message-streaming applications with Iggy. +//! High-level API clients already provide common message-streaming features that +//! you would otherwise need to build yourself. +//! +//! There are three client types: +//! - [`IggyClient`] is the entry point and the full API surface. It owns the +//! connection and implements every domain trait, including [`MessageClient`] +//! with the raw [`send_messages`] and [`poll_messages`] primitives. For both, each call +//! ignores producer and consumer level policies, i.e. there is no batching, retries, offset tracking, +//! or polling loop. +//! - [`IggyProducer`] exposes all configuration and functionality to produce (send) +//! messages to a specific topic in a stream. It shares the connection from the +//! [`IggyClient`]. +//! - [`IggyConsumer`] exposes all configuration and functionality to consume (read) +//! messages from a specific topic and stream. It also shares the connection from the +//! [`IggyClient`]. +//! +//! You do not construct the producer and consumer independently. Spawn their builders +//! from an [`IggyClient`] with [`IggyClient::producer`] and +//! [`IggyClient::consumer`] so they share its connection. +//! +//! ## When to use each +//! +//! Reach for [`IggyClient`] directly for administrative tasks such as +//! creating streams, topics, users, and consumer groups, reading or storing +//! offsets, or sending and polling a handful of messages in a script. +//! Reach for [`IggyProducer`] and [`IggyConsumer`] when producing and consuming messages. +//! +//! The [`IggyProducer`] has two modes you can pick from when building it: +//! **direct** ([`DirectConfig`]), where a send goes out on the calling task, or +//! **background** ([`BackgroundConfig`]), where a send is buffered and sending +//! is offloaded to worker tasks. +//! +//! Additionally it provides the following features: +//! +//! - **Retries** with a configurable count and interval. +//! - A pluggable **partitioning strategy** ([`Partitioner`]). +//! - **Encryption** of message payloads before they leave the process. +//! - **Auto-creation** of the stream and topic if they do not exist yet. +//! +//! Only in **direct** mode: +//! - **Chunking** splits an input larger than `batch_length` into several +//! requests. +//! - **Spacing** applies `linger_time` between consecutive sends. +//! +//! Only in **background** mode: +//! - **Batching** collects until the batch size in bytes, the number of sends, +//! or the linger interval is reached, whichever comes first. +//! - **Shard workers** run several send loops in parallel (`num_shards`), which +//! helps when one producer writes to several streams or topics. +//! - A **sharding strategy** ([`Sharding`]) decides which worker a batch goes to: +//! [`OrderedSharding`] keeps messages for the same stream and topic in order, +//! [`BalancedSharding`] spreads them round-robin for throughput. +//! - **Backpressure** bounds the bytes buffered across all workers and the number +//! of in-flight batches. With a [`BackpressureMode`] you can decide whether a full buffer +//! blocks, blocks with a timeout, or fails immediately. +//! - **Ordering control** through `max_in_flight`. +//! - An **error callback** ([`ErrorCallback`]) receives the messages a background +//! send could not deliver, together with the confirmations that did commit. +//! - **Graceful shutdown** flushes what is still buffered, which dropping the +//! producer does not. +//! +//! The [`IggyConsumer`] has: +//! - A [`futures::Stream`] implementation, so a `while let Some(message) = +//! consumer.next().await` loop drives polling, paging, and the poll interval +//! for you. +//! - A **polling strategy** ([`PollingStrategy`]: `next`, `offset`, or +//! `timestamp`) that tracks position. +//! - **Auto-commit** ([`AutoCommit`]) stores the offset on an interval, or when +//! messages are polled, consumed one by one, consumed in full, or every Nth +//! message. A restart resumes from that last commit. +//! - **Manual offset control** to store, read, or delete the offset of any +//! partition yourself when auto-commit is disabled or not enough. +//! - A **shared state handle** ([`IggyConsumerState`]) that another task can clone +//! to read offsets or commit while the consuming loop holds the consumer. +//! - **Replay control** drops messages this consumer already consumed, unless you +//! opt into replaying them with `allow_replay`. +//! - **Auto-join** of the consumer group, optionally creating it, plus a rejoin +//! once the server revokes membership or the connection comes back. +//! - **Reconnection handling** pauses polling while the client is disconnected and +//! resumes it after the reconnect and the rejoin have gone through. +//! - **Init retries** wait for the stream and topic to appear instead of failing +//! right away when the consumer starts before they exist. +//! - Payload **decryption**. +//! - **Graceful shutdown** flushes pending offsets and leaves the consumer group. +//! +//! For details on the specific behavior of each feature, reach for the type-level +//! documentation. +//! +//! # Stream builder API +//! +//! The stream builder API is a convenient way to use the high-level API. +//! [`IggyStream`], [`IggyStreamProducer`], and [`IggyStreamConsumer`] construct +//! everything at once. You can pass an [`IggyClient`] (or just a connection string) +//! together with a config, and they hand back a ready, connected +//! [`IggyProducer`] / [`IggyConsumer`]. +//! Compared to the **high-level API**, it changes how you construct +//! producers and consumers, not what they can do. Instead of chaining an +//! [`IggyProducerBuilder`] / [`IggyConsumerBuilder`] and setting each option +//! with a method call, you describe the whole setup once in an +//! [`IggyStreamConfig`] (or in a single [`IggyProducerConfig`] / +//! [`IggyConsumerConfig`] when you only need one side) and build from it. +//! However, both provide a subset of available configurations only. +//! If you need full control use the builders instead. +//! +//! # Low-level API +//! +//! The low-level API is the set of concrete transport clients: [`TcpClient`], +//! [`QuicClient`], [`WebSocketClient`], and [`HttpClient`]. Each one implements +//! [`Client`], the supertrait that pulls in every domain-specific trait, so a +//! transport client on its own can already drive the full server API. The +//! high-level [`IggyClient`] is one more layer over exactly these types. +//! +//! ## Differences to the high-level API +//! +//! - **Transport is fixed at compile time.** You name a concrete type +//! ([`TcpClient`], [`QuicClient`], and so on) instead of configuring a +//! transport-agnostic [`IggyClient`]. Swapping transports means swapping the +//! type, not changing a connection-string scheme. +//! - **No producer or consumer helpers.** [`IggyProducer`] and [`IggyConsumer`] +//! are spawned from an [`IggyClient`], so a raw transport client gives you no +//! background batching, retries, polling loop, auto-commit, consumer-group +//! auto-join, or payload encryption. You get the request-response primitives +//! ([`send_messages`], [`poll_messages`]) and nothing layered on top. +//! - **Raw wire access.** [`BinaryTransport::send_raw_with_response`] sends an +//! arbitrary command code and payload and returns the raw response bytes. +//! The high-level equivalents are [`IggyClient::send_binary_request`] and +//! [`IggyClient::send_http_request`]. +//! Either way you need to know the server command codes and the wire format. +//! +//! ## When to use it +//! +//! Prefer the high-level API. Reach for the low-level API only when you need one +//! of the things it exposes that [`IggyClient`] deliberately hides: +//! +//! - You want to own the connection lifecycle yourself, with custom pooling, +//! supervision, or a different heartbeat strategy, rather than let +//! [`IggyClient`] manage it. +//! - You are building your own abstraction on top of the SDK, for example a +//! different producer or consumer, and want the primitives. +//! - You forked the server and need to issue a command the typed API does not recognize +//! and want the raw [`send_raw_with_response`][`BinaryTransport::send_raw_with_response`] +//! instruction. +//! +//! If none of these apply, the high-level API gives you the same reach with far +//! less to get wrong. +//! +//! # Async runtime +//! +//! The SDK is async and runs on the [Tokio] runtime. Note that this is a hard +//! requirement and not optional. The SDK uses [quinn] (for QUIC), [reqwest] (for HTTP), +//! [tokio-tungstenite] (for WebSocket) and [tokio-rustls] (for TLS) which all build on +//! Tokio. +//! The SDK also spawns its own background work with [`tokio::spawn`] (the +//! [`IggyClient::connect`] heartbeat, and the [`IggyProducer`] and +//! [`IggyConsumer`] tasks) and drives timeouts, retries, and poll intervals with +//! [`tokio::time`]. +//! Note that dropping down to the low-level transport clients does not change this. +//! **Thus, everything you do with the Rust SDK must happen inside a Tokio runtime.** +//! +//! ```no_run +//! use iggy::prelude::*; +//! use futures_util::StreamExt; +//! use std::error::Error; +//! use std::str::FromStr; +//! +//! // `#[tokio::main]` starts the runtime the SDK requires. +//! #[tokio::main] +//! async fn main() -> Result<(), Box> { +//! let client = IggyClient::from_connection_string( +//! "iggy://iggy:iggy@localhost:8090", +//! )?; +//! client.connect().await?; +//! +//! let producer = client.producer("stream_name", "topic_name")?.build(); +//! producer.init().await?; +//! producer +//! .send(vec![IggyMessage::from_str("some_message_payload")?]) +//! .await?; +//! +//! let mut consumer = client +//! .consumer("consumer_name", "stream_name", "topic_name", 1)? +//! .build(); +//! consumer.init().await?; +//! while let Some(message) = consumer.next().await { +//! let _message = message?; +//! break; +//! } +//! +//! client.shutdown().await?; +//! Ok(()) +//! } +//! ``` +//! +//! [`IggyClient`]: crate::prelude::IggyClient +//! [`IggyClient::producer`]: crate::prelude::IggyClient::producer +//! [`IggyClient::consumer`]: crate::prelude::IggyClient::consumer +//! [`IggyProducer`]: crate::prelude::IggyProducer +//! [`IggyConsumer`]: crate::prelude::IggyConsumer +//! [`MessageClient`]: crate::prelude::MessageClient +//! [`send_messages`]: crate::prelude::MessageClient::send_messages +//! [`poll_messages`]: crate::prelude::MessageClient::poll_messages +//! [`DirectConfig`]: crate::prelude::DirectConfig +//! [`BackgroundConfig`]: crate::prelude::BackgroundConfig +//! [`BackpressureMode`]: crate::clients::producer_config::BackpressureMode +//! [`Sharding`]: crate::prelude::Sharding +//! [`OrderedSharding`]: crate::prelude::OrderedSharding +//! [`BalancedSharding`]: crate::prelude::BalancedSharding +//! [`ErrorCallback`]: crate::clients::producer_error_callback::ErrorCallback +//! [`Partitioner`]: crate::prelude::Partitioner +//! [`PollingStrategy`]: crate::prelude::PollingStrategy +//! [`AutoCommit`]: crate::prelude::AutoCommit +//! [`IggyConsumerState`]: crate::prelude::IggyConsumerState +//! [`futures::Stream`]: https://docs.rs/futures/latest/futures/stream/trait.Stream.html +//! [`TcpClient`]: crate::prelude::TcpClient +//! [`QuicClient`]: crate::quic::quic_client::QuicClient +//! [`WebSocketClient`]: crate::prelude::WebSocketClient +//! [`HttpClient`]: crate::http::http_client::HttpClient +//! [`Client`]: crate::prelude::Client +//! [`BinaryTransport::send_raw_with_response`]: crate::binary::BinaryTransport::send_raw_with_response +//! [`IggyClient::send_binary_request`]: crate::prelude::IggyClient::send_binary_request +//! [`IggyClient::send_http_request`]: crate::prelude::IggyClient::send_http_request +//! [`IggyStream`]: crate::prelude::IggyStream +//! [`IggyStreamProducer`]: crate::prelude::IggyStreamProducer +//! [`IggyStreamConsumer`]: crate::prelude::IggyStreamConsumer +//! [`IggyStreamConfig`]: crate::prelude::IggyStreamConfig +//! [`IggyProducerConfig`]: crate::prelude::IggyProducerConfig +//! [`IggyConsumerConfig`]: crate::prelude::IggyConsumerConfig +//! [`IggyProducerBuilder`]: crate::prelude::IggyProducerBuilder +//! [`IggyConsumerBuilder`]: crate::prelude::IggyConsumerBuilder +//! [`IggyClient::connect`]: crate::prelude::Client::connect +//! +//! [Tokio]: https://tokio.rs +//! [`tokio::spawn`]: https://docs.rs/tokio/latest/tokio/task/fn.spawn.html +//! [`tokio::time`]: https://docs.rs/tokio/latest/tokio/time/index.html +//! [quinn]: https://docs.rs/quinn +//! [reqwest]: https://docs.rs/reqwest +//! [tokio-tungstenite]: https://docs.rs/tokio-tungstenite +//! [tokio-rustls]: https://docs.rs/tokio-rustls +//! +//! [`Apache Software Foundation`]: https://www.apache.org/ +//! [`foreign`]: https://github.com/apache/iggy/tree/master/foreign +//! [`schema spec`]: https://iggy.apache.org/docs/server/schema/ +//! [`server command enum`]: https://github.com/apache/iggy/blob/3e27ebc8dd5dbf257b816993908dc0747c4f8849/core/server/src/binary/command.rs#L74 pub mod binary; pub mod client_provider; pub mod client_wrappers; From 831b2a652ba71af89497aca9a08e2b8742d61e03 Mon Sep 17 00:00:00 2001 From: Hubert Gruszecki Date: Tue, 1 Sep 2026 20:33:03 +0200 Subject: [PATCH 032/182] perf(server): serve poll replies without copying record bytes (#4025) --- core/consensus/src/impls.rs | 9 +- core/consensus/src/metadata_helpers.rs | 8 +- core/consensus/src/plane_helpers.rs | 6 +- core/message_bus/src/lib.rs | 26 +- .../src/lifecycle/connection_registry.rs | 53 ++- core/message_bus/src/transports/mod.rs | 18 +- core/message_bus/src/transports/quic.rs | 39 +- core/message_bus/src/transports/tcp.rs | 143 +++++- core/message_bus/src/transports/tcp_tls.rs | 18 +- core/message_bus/src/transports/ws.rs | 14 +- core/message_bus/src/transports/wss.rs | 20 +- .../tests/tcp_tls_client_listener.rs | 7 +- .../tests/tcp_tls_client_roundtrip.rs | 7 +- core/message_bus/tests/wss_client_listener.rs | 7 +- .../message_bus/tests/wss_client_roundtrip.rs | 7 +- core/metadata/src/impls/metadata.rs | 8 +- core/partitions/src/iggy_partition.rs | 8 +- core/partitions/src/journal.rs | 243 ++++++++++- core/partitions/src/messages_writer.rs | 6 +- core/partitions/src/poll_plan.rs | 7 +- core/partitions/src/types.rs | 22 + core/server/src/dispatch.rs | 72 +++- core/server/src/http/reply.rs | 13 +- core/server/src/http/submit.rs | 10 +- core/server/src/partition_reconciler.rs | 2 +- core/server/src/responses.rs | 408 +++++++++++++++++- core/server_common/src/consensus_message.rs | 75 +++- core/server_common/src/iobuf.rs | 27 ++ core/server_common/src/lib.rs | 1 + core/shard/src/lib.rs | 9 +- core/simulator/src/bus.rs | 9 +- 31 files changed, 1128 insertions(+), 174 deletions(-) diff --git a/core/consensus/src/impls.rs b/core/consensus/src/impls.rs index edc60c0861..3f732bb48e 100644 --- a/core/consensus/src/impls.rs +++ b/core/consensus/src/impls.rs @@ -4377,6 +4377,7 @@ mod timestamp_clamp_tests { use super::*; use crate::LocalPipeline; + use message_bus::BusMessage; use server_common::MESSAGE_ALIGN; use server_common::iobuf::Frozen; @@ -4398,7 +4399,7 @@ mod timestamp_clamp_tests { async fn send_to_client( &self, _client_id: u128, - _data: Frozen, + _data: impl Into, ) -> Result<(), message_bus::SendError> { Ok(()) } @@ -4711,6 +4712,7 @@ mod timestamp_clamp_tests { #[cfg(test)] mod vsr_consensus_tests { use super::*; + use message_bus::BusMessage; #[test] fn stage_transitions_follow_the_machine() { @@ -4753,7 +4755,7 @@ mod vsr_consensus_tests { async fn send_to_client( &self, _client_id: u128, - _data: server_common::iobuf::Frozen<{ server_common::MESSAGE_ALIGN }>, + _data: impl Into, ) -> Result<(), message_bus::SendError> { Ok(()) } @@ -5284,6 +5286,7 @@ mod quorum_tests { use super::*; use crate::LocalPipeline; + use message_bus::BusMessage; use server_common::MESSAGE_ALIGN; use server_common::iobuf::Frozen; @@ -5293,7 +5296,7 @@ mod quorum_tests { async fn send_to_client( &self, _client_id: u128, - _data: Frozen, + _data: impl Into, ) -> Result<(), message_bus::SendError> { Ok(()) } diff --git a/core/consensus/src/metadata_helpers.rs b/core/consensus/src/metadata_helpers.rs index 43af22a878..188375f4b5 100644 --- a/core/consensus/src/metadata_helpers.rs +++ b/core/consensus/src/metadata_helpers.rs @@ -471,7 +471,7 @@ mod tests { use crate::client_table::{REGISTER_REQUEST_ID, REPLY_RING_RETENTION_BYTES}; use crate::{CLIENTS_TABLE_MAX, LocalPipeline}; use iggy_binary_protocol::{Command, Operation, ReplyHeader}; - use message_bus::SendError; + use message_bus::{BusMessage, SendError}; /// Acting user for register fixtures; these tests exercise preflight / /// replay, not user resolution, so the exact value is immaterial. @@ -502,9 +502,11 @@ mod tests { async fn send_to_client( &self, client_id: u128, - data: Frozen, + data: impl Into, ) -> Result<(), SendError> { - self.client_sends.borrow_mut().push((client_id, data)); + self.client_sends + .borrow_mut() + .push((client_id, data.into().into_contiguous())); Ok(()) } diff --git a/core/consensus/src/plane_helpers.rs b/core/consensus/src/plane_helpers.rs index ff7df19cc8..195120af88 100644 --- a/core/consensus/src/plane_helpers.rs +++ b/core/consensus/src/plane_helpers.rs @@ -877,7 +877,7 @@ mod tests { use aligned_vec::{AVec, ConstAlign}; use iggy_binary_protocol::{ConsensusHeader, Operation, StartViewChangeHeader}; use iggy_common::calculate_checksum; - use message_bus::SendError; + use message_bus::{BusMessage, SendError}; use server_common::{MESSAGE_ALIGN, iobuf::Frozen}; use std::collections::BTreeMap; @@ -896,7 +896,7 @@ mod tests { async fn send_to_client( &self, _client_id: u128, - _data: Frozen, + _data: impl Into, ) -> Result<(), SendError> { Ok(()) } @@ -1942,7 +1942,7 @@ mod tests { async fn send_to_client( &self, _client_id: u128, - _data: Frozen, + _data: impl Into, ) -> Result<(), SendError> { Ok(()) } diff --git a/core/message_bus/src/lib.rs b/core/message_bus/src/lib.rs index f41f50e30e..74836bb8e8 100644 --- a/core/message_bus/src/lib.rs +++ b/core/message_bus/src/lib.rs @@ -53,8 +53,9 @@ //! `Ready` on first poll. Consensus code relies on this for //! reentrancy reasoning; any `.await` in the body breaks it. //! - The TCP transport's writer task coalesces up to -//! `MessageBusConfig::max_batch` (default 256) `Frozen` -//! into one `write_vectored_all`. Don't introduce per-message +//! `MessageBusConfig::max_batch` (default 256) [`BusMessage`] frames +//! (their `Frozen` fragments flattened, chunked at +//! `IOV_MAX`) into `write_vectored_all`. Don't introduce per-message //! syscalls or per-message encryption on the plaintext TCP plane. //! - fd-delegation ([`fd_transfer`]) is TCP-only. TLS / QUIC //! connections have no dupable plaintext fd, so shard 0 terminates @@ -279,7 +280,7 @@ pub type ReplicaForwardFn = Box) -> Result /// shift. `client_id` is carried so the receiver can rebuild the /// `ShardFrame::Lifecycle`'s `ForwardClientSend { client_id, msg }` /// payload. -pub type ClientForwardFn = Box) -> Result<(), SendError>>; +pub type ClientForwardFn = Box Result<(), SendError>>; /// Callback invoked when a client connection metadata entry is removed. pub type ClientConnectionLostFn = std::rc::Rc; @@ -496,10 +497,13 @@ pub type ConnectionLostFn = std::rc::Rc; /// /// A bus impl must preserve this divergence - see each method. pub trait MessageBus { + /// Queue one frame for `client_id`. Takes anything that converts into a + /// [`BusMessage`]: a single `Frozen` buffer or an + /// already fragmented frame. fn send_to_client( &self, client_id: u128, - data: Frozen, + data: impl Into, ) -> impl Future>; fn send_to_replica( @@ -1261,7 +1265,7 @@ impl MessageBus for std::rc::Rc { fn send_to_client( &self, client_id: u128, - data: Frozen, + data: impl Into, ) -> impl Future> { (**self).send_to_client(client_id, data) } @@ -1313,11 +1317,12 @@ impl MessageBus for IggyMessageBus { async fn send_to_client( &self, client_id: u128, - message: Frozen, + message: impl Into, ) -> Result<(), SendError> { if self.is_shutting_down() { return Err(SendError::BusShuttingDown); } + let message = message.into(); // Owning shard is encoded in the top 16 bits of client_id. let owning_shard = client_id_owning_shard(client_id); if owning_shard == self.shard_id { @@ -1357,9 +1362,9 @@ impl MessageBus for IggyMessageBus { // Fast path: this shard owns a connection to the replica. On no-slot // the registry returns the message unchanged so the slow path can // forward it via the inter-shard channel without a wasted clone. - let message = match self.replicas.try_send_or_return(replica, message) { + let message = match self.replicas.try_send_or_return(replica, message.into()) { Ok(send_result) => return send_result.map_err(map_try_send_err), - Err(message) => message, + Err(message) => message.into_contiguous(), }; // Slow path: route via the inter-shard channel to the owning shard. // The shard-shared `owner_table` is the authoritative routing @@ -1438,7 +1443,7 @@ pub const fn is_auto_commit_client(client_id: u128) -> bool { /// Shape-matches `Result::map_err` (takes the error by value) so it can be /// used directly as a function reference rather than a closure. #[allow(clippy::needless_pass_by_value)] // signature required by map_err -fn map_try_send_err(e: async_channel::TrySendError>) -> SendError { +fn map_try_send_err(e: async_channel::TrySendError) -> SendError { match e { async_channel::TrySendError::Full(_) => SendError::Backpressure, async_channel::TrySendError::Closed(_) => SendError::ConnectionClosed, @@ -1448,9 +1453,10 @@ fn map_try_send_err(e: async_channel::TrySendError>) -> Se /// Peek `ReplyHeader.request` (the originating request id) from a reply /// buffer at its fixed header offset. Only the in-process reply path calls /// this; the socket path never decodes. -fn reply_request_id(reply: &Frozen) -> u64 { +fn reply_request_id(reply: &BusMessage) -> u64 { const OFFSET: usize = std::mem::offset_of!(ReplyHeader, request); reply + .first() .as_slice() .get(OFFSET..OFFSET + std::mem::size_of::()) .and_then(|bytes| bytes.try_into().ok()) diff --git a/core/message_bus/src/lifecycle/connection_registry.rs b/core/message_bus/src/lifecycle/connection_registry.rs index 0fa89a13b6..24c709742d 100644 --- a/core/message_bus/src/lifecycle/connection_registry.rs +++ b/core/message_bus/src/lifecycle/connection_registry.rs @@ -34,7 +34,7 @@ use compio::runtime::JoinHandle; use futures::channel::oneshot; -use server_common::{MESSAGE_ALIGN, iobuf::Frozen}; +use server_common::ResponseBacking; use std::cell::{Cell, RefCell}; use std::collections::HashMap; use std::collections::hash_map::Entry as HmEntry; @@ -70,12 +70,14 @@ pub const READER_DRAIN_FLOOR: Duration = Duration::from_millis(250); /// Payload type carried over every per-peer queue. /// -/// Consensus messages are `Frozen` by the time they hit -/// this queue: the dispatch layer freezes once and fan-out becomes a -/// refcount bump per target. The writer task reads `Frozen` out of the -/// queue and passes it straight to `write_vectored_all`, so no -/// conversion happens on the hot path. -pub type BusMessage = Frozen; +/// One frame as a list of `Frozen` fragments written back +/// to back. Consensus messages and most replies are a single fragment: the +/// dispatch layer freezes once and fan-out becomes a refcount bump per +/// target. A poll reply is a header fragment followed by the stored batch +/// buffers as they are, so record bytes are never copied. The writer task +/// hands the fragments straight to the socket write, so no conversion +/// happens on the hot path. +pub type BusMessage = ResponseBacking; /// Producer side of a per-peer queue. Cloned out of the registry by /// `send_to_*` and used with `try_send`. @@ -960,6 +962,7 @@ mod tests { h.size = HEADER_SIZE as u32; }) .into_frozen() + .into() } fn spawn_dummy_writer(rx: BusReceiver) -> JoinHandle<()> { @@ -1289,7 +1292,7 @@ mod tests { // Receiver drains the message; closing the writer's rx end via // close_peer is not necessary for this assertion. let received = rx.recv().await.expect("queue had message"); - assert_eq!(received.len(), HEADER_SIZE); + assert_eq!(received.total_len(), HEADER_SIZE); } /// No slot for `key`: the message is handed back to the caller so the @@ -1298,13 +1301,13 @@ mod tests { async fn try_send_or_return_returns_msg_when_slot_missing() { let reg: ConnectionRegistry = ConnectionRegistry::new(); let msg = make_bus_msg(); - let original_len = msg.len(); + let original_len = msg.total_len(); let outcome = reg.try_send_or_return(99u8, msg); let ReplyRoute::NoSlot(returned) = outcome else { panic!("missing slot should return msg"); }; - assert_eq!(returned.len(), original_len); + assert_eq!(returned.total_len(), original_len); } /// Slot present but queue full: inner `Err(TrySendError::Full(msg))` so @@ -1345,11 +1348,11 @@ mod tests { async fn replica_try_send_or_return_returns_msg_when_slot_missing() { let reg = ReplicaRegistry::new(); let msg = make_bus_msg(); - let original_len = msg.len(); + let original_len = msg.total_len(); let outcome = reg.try_send_or_return(7u8, msg); let returned = outcome.expect_err("missing slot should return msg"); - assert_eq!(returned.len(), original_len); + assert_eq!(returned.total_len(), original_len); } /// End-to-end registry seam: install entry + slot, route, fire, await. @@ -1366,7 +1369,7 @@ mod tests { reg.fire_in_process(1u8, 42, msg).expect("waiter present"); let received = reply_rx.await.expect("reply delivered"); - assert_eq!(received.len(), HEADER_SIZE); + assert_eq!(received.total_len(), HEADER_SIZE); drop(guard); assert!(reg.contains(1u8), "guard drop must not remove the entry"); } @@ -1418,7 +1421,10 @@ mod tests { reg.fire_in_process(1u8, 42, make_bus_msg()) .expect("first waiter still registered"); - assert_eq!(reply_rx.await.expect("reply delivered").len(), HEADER_SIZE); + assert_eq!( + reply_rx.await.expect("reply delivered").total_len(), + HEADER_SIZE + ); } /// Reply for a request id nobody waits on: handed back, no panic, and @@ -1430,15 +1436,18 @@ mod tests { let (_guard, reply_rx) = reg.install_reply_slot(1u8, 42).expect("slot installs"); let msg = make_bus_msg(); - let original_len = msg.len(); + let original_len = msg.total_len(); let returned = reg .fire_in_process(1u8, 7, msg) .expect_err("no waiter for request 7"); - assert_eq!(returned.len(), original_len); + assert_eq!(returned.total_len(), original_len); reg.fire_in_process(1u8, 42, make_bus_msg()) .expect("slot 42 untouched"); - assert_eq!(reply_rx.await.expect("reply delivered").len(), HEADER_SIZE); + assert_eq!( + reply_rx.await.expect("reply delivered").total_len(), + HEADER_SIZE + ); } /// Timeout / cancellation path: dropping the guard removes the slot, @@ -1473,7 +1482,10 @@ mod tests { .expect("waiter present"); drop(guard); - assert_eq!(reply_rx.await.expect("reply delivered").len(), HEADER_SIZE); + assert_eq!( + reply_rx.await.expect("reply delivered").total_len(), + HEADER_SIZE + ); let (_guard, _reply_rx) = reg .install_reply_slot(1u8, 42) .expect("request id reusable after fire + drop"); @@ -1497,6 +1509,9 @@ mod tests { drop(stale_guard); reg.fire_in_process(1u8, 42, make_bus_msg()) .expect("stale guard must not evict the new slot"); - assert_eq!(reply_rx.await.expect("reply delivered").len(), HEADER_SIZE); + assert_eq!( + reply_rx.await.expect("reply delivered").total_len(), + HEADER_SIZE + ); } } diff --git a/core/message_bus/src/transports/mod.rs b/core/message_bus/src/transports/mod.rs index 121afaa8b9..7a703884e0 100644 --- a/core/message_bus/src/transports/mod.rs +++ b/core/message_bus/src/transports/mod.rs @@ -39,20 +39,22 @@ //! - **Batch ordering**: each transport drains the per-peer //! `BusReceiver` in FIFO order, but the dispatch shape differs: //! * **TCP (vectored batch)** assembles up to `max_batch` frames -//! into one `write_vectored_all` call. The kernel may short-write -//! the iovec set (so `writev` is not atomic), but FIFO order -//! across the batch is preserved and any short or failed write -//! tears the connection down rather than retrying on a +//! (every fragment of each [`BusMessage`](crate::BusMessage), chunked at `IOV_MAX` +//! iovecs) into `write_vectored_all` calls. The kernel may +//! short-write the iovec set (so `writev` is not atomic), but FIFO +//! order across the batch is preserved and any short or failed +//! write tears the connection down rather than retrying on a //! half-written batch. //! * **TCP-TLS, WS, WSS (drain-and-flush)** drain the same //! `max_batch` window into a `Vec` and then write //! each frame through the per-record API (`AsyncWriteExt::write_all` -//! for TLS, `WebSocketStream::send` for WS / WSS) followed by ONE -//! trailing `flush()` per batch. This avoids per-frame TCP/TLS +//! per fragment for TLS, `WebSocketStream::send` for WS / WSS, which +//! joins a multi-fragment frame into one payload first) followed by +//! ONE trailing `flush()` per batch. This avoids per-frame TCP/TLS //! record overhead while staying inside the per-record API //! constraints of those transports. //! * **QUIC (per-frame)** uses one bidirectional stream per peer -//! and writes each `BusMessage` with a separate +//! and writes each `BusMessage` fragment with a separate //! `SendStream::write_all` call; quinn coalesces at the datagram //! layer. //! @@ -127,7 +129,7 @@ pub struct ActorContext { /// message to the bus's per-plane handler (`MessageHandler` / /// `RequestHandler`). pub in_tx: Sender>, - /// Outbound channel: the bus's `send_to_*` path pushes `Frozen` + /// Outbound channel: the bus's `send_to_*` path pushes [`BusMessage`](crate::BusMessage) /// frames into the matching `Sender`; the transport drains here. pub rx: BusReceiver, /// Cooperative cancellation. Fires on bus shutdown OR on diff --git a/core/message_bus/src/transports/quic.rs b/core/message_bus/src/transports/quic.rs index 8968a18e4d..da9296cb6d 100644 --- a/core/message_bus/src/transports/quic.rs +++ b/core/message_bus/src/transports/quic.rs @@ -319,7 +319,7 @@ enum ReplyRoute { /// Decide how to route an outbound frame. Only `Command::Reply` carries a /// `request` id; everything else must pass through unfiltered. fn reply_route(frame: &BusMessage) -> ReplyRoute { - let bytes = frame.as_slice(); + let bytes = frame.first().as_slice(); let is_reply = bytes .get(COMMAND_OFFSET) .is_some_and(|&command| command == Command::Reply as u8); @@ -503,11 +503,19 @@ impl TransportConn for QuicTransportConn { continue; }; - // Write the reply, then `finish()` so quinn-proto flushes pending - // data + a FIN, signalling the SDK that the reply is complete. - let BufResult(result, _frozen) = send.write_all(first).await; - if let Err(e) = result { - debug!(%label, %peer, error = ?e, "quic: write failed"); + // Write the reply fragment by fragment, then `finish()` so + // quinn-proto flushes pending data + a FIN, signalling the SDK + // that the reply is complete. + let mut write_failed = false; + for fragment in first.into_fragments() { + let BufResult(result, _frozen) = send.write_all(fragment).await; + if let Err(e) = result { + debug!(%label, %peer, error = ?e, "quic: write failed"); + write_failed = true; + break; + } + } + if write_failed { continue; } if let Err(e) = send.finish() { @@ -611,12 +619,12 @@ mod tests { fn drive( conn: QuicTransportConn, ) -> ( - async_channel::Sender>, + async_channel::Sender, async_channel::Receiver>, Shutdown, compio::runtime::JoinHandle<()>, ) { - let (out_tx, out_rx) = bounded::>(16); + let (out_tx, out_rx) = bounded::(16); let (in_tx, in_rx) = bounded::>(16); let (shutdown, token) = Shutdown::new(); let ctx = ActorContext { @@ -654,11 +662,20 @@ mod tests { // and then awaits exactly one reply on `rx` before // accepting the next bidi. Feed three replies in step. let a = in_rx.recv().await.unwrap(); - out_tx.send(header_only(Command::Reply)).await.unwrap(); + out_tx + .send(header_only(Command::Reply).into()) + .await + .unwrap(); let b = in_rx.recv().await.unwrap(); - out_tx.send(header_only(Command::Reply)).await.unwrap(); + out_tx + .send(header_only(Command::Reply).into()) + .await + .unwrap(); let c = in_rx.recv().await.unwrap(); - out_tx.send(header_only(Command::Reply)).await.unwrap(); + out_tx + .send(header_only(Command::Reply).into()) + .await + .unwrap(); shutdown.trigger(); let _ = handle.await; (a.header().command, b.header().command, c.header().command) diff --git a/core/message_bus/src/transports/tcp.rs b/core/message_bus/src/transports/tcp.rs index 87a310380c..53845b41e1 100644 --- a/core/message_bus/src/transports/tcp.rs +++ b/core/message_bus/src/transports/tcp.rs @@ -35,6 +35,8 @@ use compio::io::AsyncWriteExt; use compio::net::{TcpListener, TcpStream}; use compio::runtime::fd::PollFd; use futures::FutureExt; +use server_common::MESSAGE_ALIGN; +use server_common::iobuf::{Frozen, IOV_MAX}; use std::io; use std::mem; use std::net::SocketAddr; @@ -242,6 +244,9 @@ async fn writer_loop( ) { let _wake_reader = scopeguard::guard(conn_shutdown, |s| s.trigger()); let mut batch: Vec = Vec::with_capacity(max_batch); + let mut iovecs: Vec> = Vec::with_capacity(max_batch); + // Sized by the first wide batch; most connections never need it. + let mut scratch: Vec> = Vec::new(); let mut shutdown_fut = Box::pin(shutdown.wait().fuse()); loop { @@ -271,10 +276,12 @@ async fn writer_loop( let drained = batch.len(); trace!(%label, %peer, batch = drained, "writev batch"); - let owned = mem::take(&mut batch); - let compio::BufResult(result, mut returned) = write_half.write_vectored_all(owned).await; - returned.clear(); - batch = returned; + // `drain(..)` keeps the batch allocation for the next round. + #[allow(clippy::iter_with_drain)] + for frame in batch.drain(..) { + iovecs.extend(frame.into_fragments()); + } + let result = write_iovecs(&mut write_half, &mut iovecs, &mut scratch).await; if let Err(e) = result { error!( @@ -289,6 +296,43 @@ async fn writer_loop( } } +/// `write_vectored_all` over `iovecs`, at most `IOV_MAX` entries per call so +/// a frame fragmented past the syscall limit (a wide poll reply) cannot fail +/// with `EMSGSIZE` (the socket path is `io_uring` `sendmsg(2)`). `scratch` +/// carries each chunk in flight. Leaves both vecs empty with their +/// allocations intact for the next batch. +#[allow(clippy::future_not_send)] +async fn write_iovecs( + write_half: &mut TcpStream, + iovecs: &mut Vec>, + scratch: &mut Vec>, +) -> io::Result<()> { + if iovecs.len() <= IOV_MAX { + let compio::BufResult(result, mut returned) = + write_half.write_vectored_all(mem::take(iovecs)).await; + result?; + returned.clear(); + *iovecs = returned; + return Ok(()); + } + // A full-range drain shifts nothing, so the batch moves into `scratch` in + // O(n). Draining `IOV_MAX` at a time from the front would memmove the tail + // per syscall, O(n^2) over a client-chosen n (an unclamped poll count fans + // out to 1-2 iovecs per batch) on the shard's single-threaded executor. + let mut source = iovecs.drain(..); + loop { + scratch.extend(source.by_ref().take(IOV_MAX)); + if scratch.is_empty() { + return Ok(()); + } + let compio::BufResult(result, mut returned) = + write_half.write_vectored_all(mem::take(scratch)).await; + result?; + returned.clear(); + *scratch = returned; + } +} + #[cfg(test)] mod tests { use super::*; @@ -325,12 +369,12 @@ mod tests { fn drive( conn: TcpTransportConn, ) -> ( - async_channel::Sender>, + async_channel::Sender, async_channel::Receiver>, Shutdown, compio::runtime::JoinHandle<()>, ) { - let (out_tx, out_rx) = bounded::>(16); + let (out_tx, out_rx) = bounded::(16); let (in_tx, in_rx) = bounded::>(16); let (shutdown, token) = Shutdown::new(); let ctx = ActorContext { @@ -370,7 +414,7 @@ mod tests { drive(TcpTransportConn::new(server)); for cmd in [Command::Ping, Command::Prepare, Command::Request] { - client_out.send(header_only(cmd)).await.unwrap(); + client_out.send(header_only(cmd).into()).await.unwrap(); } let recv_with_timeout = |rx: &async_channel::Receiver>| { @@ -395,6 +439,91 @@ mod tests { let _ = server_handle.await; } + /// A frame fragmented to `IOV_MAX` or past it must reach the peer whole + /// and in order: `sendmsg` rejects more than `IOV_MAX` iovecs with + /// `EMSGSIZE`, so the writer has to chunk the flattened batch instead of + /// failing it. Each frame goes out alone in its batch (the follow-up frame + /// is sent only once the peer has it), so the iovec count is exactly the + /// fragment count and the chunk boundaries land where the cases say. + #[compio::test] + #[allow(clippy::future_not_send, clippy::cast_possible_truncation)] + async fn run_writes_frames_fragmented_at_and_past_iov_max() { + let (client, server) = local_pair().await; + let (client_out, _client_in, client_shutdown, client_handle) = + drive(TcpTransportConn::new(client)); + let (_server_out, server_in, server_shutdown, server_handle) = + drive(TcpTransportConn::new(server)); + let recv_with_timeout = || async { + compio::time::timeout(Duration::from_secs(2), server_in.recv()) + .await + .expect("recv within 2s") + .expect("ok") + }; + + // One connection for all cases so a chunked batch is followed by both + // a single-iovec one and a wider one. A full single chunk, one iovec + // over it, the original `2 * IOV_MAX`-byte repro, and an exact multiple + // of the chunk size. + for fragment_count in [ + IOV_MAX, + IOV_MAX + 1, + 2 * IOV_MAX - HEADER_SIZE + 1, + 2 * IOV_MAX, + ] { + // Header whole in the first fragment (the frame contract), body as + // one byte per fragment. + let frame_len = HEADER_SIZE + fragment_count - 1; + let whole = Message::::new(frame_len) + .transmute_header(|_, h: &mut GenericHeader| { + h.command = Command::Ping; + h.size = frame_len as u32; + }) + .into_frozen(); + let mut fragments = server_common::ResponseFragments::with_capacity(fragment_count); + fragments.push(whole.slice(..HEADER_SIZE)); + fragments.extend((HEADER_SIZE..frame_len).map(|at| whole.slice(at..=at))); + assert_eq!(fragments.len(), fragment_count); + let fragmented = + Message::::try_from(fragments) + .expect("fragments form a valid frame") + .into_inner(); + client_out.send(fragmented).await.unwrap(); + + let ping = recv_with_timeout().await; + assert_eq!( + ping.header().command, + Command::Ping, + "{fragment_count} fragments" + ); + assert_eq!( + ping.header().size as usize, + frame_len, + "{fragment_count} fragments" + ); + assert_eq!( + ping.as_slice(), + whole.as_slice(), + "{fragment_count} fragments" + ); + + client_out + .send(header_only(Command::Request).into()) + .await + .unwrap(); + let request = recv_with_timeout().await; + assert_eq!( + request.header().command, + Command::Request, + "{fragment_count} fragments" + ); + } + + client_shutdown.trigger(); + server_shutdown.trigger(); + let _ = client_handle.await; + let _ = server_handle.await; + } + #[compio::test] #[allow(clippy::future_not_send)] async fn run_exits_on_shutdown_signal() { diff --git a/core/message_bus/src/transports/tcp_tls.rs b/core/message_bus/src/transports/tcp_tls.rs index a248fb85c6..c3fa5ac21b 100644 --- a/core/message_bus/src/transports/tcp_tls.rs +++ b/core/message_bus/src/transports/tcp_tls.rs @@ -573,9 +573,9 @@ async fn run_pump(tls: &mut TlsStream, ctx: ActorContext) { // preserving the Vec's allocation for the next // iteration; `into_iter()` would move the buffer out. #[allow(clippy::iter_with_drain)] - for msg in batch.drain(..) { - let len = msg.buf_len(); - let compio::BufResult(result, _) = tls.write_all(msg).await; + for fragment in batch.drain(..).flat_map(BusMessage::into_fragments) { + let len = fragment.buf_len(); + let compio::BufResult(result, _) = tls.write_all(fragment).await; if let Err(e) = result { warn!( %label, @@ -742,12 +742,12 @@ mod tests { fn drive( conn: TcpTlsTransportConn, ) -> ( - Sender>, + Sender, Receiver>, Shutdown, compio::runtime::JoinHandle<()>, ) { - let (out_tx, out_rx) = bounded::>(16); + let (out_tx, out_rx) = bounded::(16); let (in_tx, in_rx) = bounded::>(16); let (shutdown, token) = Shutdown::new(); let ctx = ActorContext { @@ -780,7 +780,7 @@ mod tests { let (client_out, client_in, client_shutdown, client_handle) = drive(client_conn); client_out - .send(header_only(Command::Request)) + .send(header_only(Command::Request).into()) .await .expect("client send"); let received = compio::time::timeout(Duration::from_secs(5), server_in.recv()) @@ -790,7 +790,7 @@ mod tests { assert_eq!(received.header().command, Command::Request); server_out - .send(header_only(Command::Reply)) + .send(header_only(Command::Reply).into()) .await .expect("server send"); let reply = compio::time::timeout(Duration::from_secs(5), client_in.recv()) @@ -854,7 +854,7 @@ mod tests { let (client_out, _client_in, client_shutdown, client_handle) = drive(client_conn); client_out - .send(padded(Command::Request, total)) + .send(padded(Command::Request, total).into()) .await .expect("client send 1 MiB"); let received = compio::time::timeout(Duration::from_secs(10), server_in.recv()) @@ -924,7 +924,7 @@ mod tests { let (client_out, _client_in, client_shutdown, client_handle) = drive(client_conn); client_out - .send(header_only(Command::Request)) + .send(header_only(Command::Request).into()) .await .expect("client send"); let received = compio::time::timeout(Duration::from_secs(5), server_in.recv()) diff --git a/core/message_bus/src/transports/ws.rs b/core/message_bus/src/transports/ws.rs index f985191bbc..15b5a963a8 100644 --- a/core/message_bus/src/transports/ws.rs +++ b/core/message_bus/src/transports/ws.rs @@ -234,8 +234,12 @@ async fn run_pump(ws: &mut WebSocketStream, ctx: ActorContext) { } let drained = batch.len(); #[allow(clippy::iter_with_drain)] + // One WS frame per message: tungstenite takes a single payload + // buffer, so a multi-fragment frame is joined here (the only + // record copy left on this plane; single-fragment frames move). for msg in batch.drain(..) { - if let Err(e) = ws.send(WsMessage::Binary(Bytes::from_owner(msg))).await { + let payload = Bytes::from_owner(msg.into_contiguous()); + if let Err(e) = ws.send(WsMessage::Binary(payload)).await { warn!(%label, %peer, error = ?e, batch_len = drained, "ws writer: send failed"); return; } @@ -364,7 +368,7 @@ mod tests { fn drive( conn: WsTransportConn, ) -> ( - Sender>, + Sender, Receiver>, Shutdown, compio::runtime::JoinHandle<()>, @@ -377,12 +381,12 @@ mod tests { conn: WsTransportConn, max_message_size: usize, ) -> ( - Sender>, + Sender, Receiver>, Shutdown, compio::runtime::JoinHandle<()>, ) { - let (out_tx, out_rx) = bounded::>(16); + let (out_tx, out_rx) = bounded::(16); let (in_tx, in_rx) = bounded::>(16); let (shutdown, token) = Shutdown::new(); let ctx = ActorContext { @@ -442,7 +446,7 @@ mod tests { // Server replies via its outbound mailbox; the serial pump writes // the reply on the same bidi the request arrived on, client reads. server_out - .send(header_only(Command::Reply)) + .send(header_only(Command::Reply).into()) .await .expect("server send"); let reply = compio::time::timeout(Duration::from_secs(5), raw_recv(&mut client_ws)) diff --git a/core/message_bus/src/transports/wss.rs b/core/message_bus/src/transports/wss.rs index e30b6a21f5..20168d008f 100644 --- a/core/message_bus/src/transports/wss.rs +++ b/core/message_bus/src/transports/wss.rs @@ -381,8 +381,12 @@ async fn run_pump(ws: &mut WebSocketStream, ctx: ActorContext) { // preserving the Vec's allocation for the next // iteration; `into_iter()` would move the buffer out. #[allow(clippy::iter_with_drain)] + // One WS frame per message: tungstenite takes a single payload + // buffer, so a multi-fragment frame is joined here (the only + // record copy left on this plane; single-fragment frames move). for msg in batch.drain(..) { - if let Err(e) = ws.send(WsMessage::Binary(Bytes::from_owner(msg))).await { + let payload = Bytes::from_owner(msg.into_contiguous()); + if let Err(e) = ws.send(WsMessage::Binary(payload)).await { warn!(%label, %peer, error = ?e, batch_len = drained, "wss writer: send failed"); return; } @@ -554,7 +558,7 @@ mod tests { fn drive( conn: WssTransportConn, ) -> ( - Sender>, + Sender, Receiver>, Shutdown, compio::runtime::JoinHandle<()>, @@ -567,12 +571,12 @@ mod tests { conn: WssTransportConn, max_message_size: usize, ) -> ( - Sender>, + Sender, Receiver>, Shutdown, compio::runtime::JoinHandle<()>, ) { - let (out_tx, out_rx) = bounded::>(16); + let (out_tx, out_rx) = bounded::(16); let (in_tx, in_rx) = bounded::>(16); let (shutdown, token) = Shutdown::new(); let ctx = ActorContext { @@ -604,7 +608,7 @@ mod tests { let (client_out, client_in, client_shutdown, client_handle) = drive(client_conn); client_out - .send(header_only(Command::Request)) + .send(header_only(Command::Request).into()) .await .expect("client send"); let received = compio::time::timeout(Duration::from_secs(5), server_in.recv()) @@ -614,7 +618,7 @@ mod tests { assert_eq!(received.header().command, Command::Request); server_out - .send(header_only(Command::Reply)) + .send(header_only(Command::Reply).into()) .await .expect("server send"); let reply = compio::time::timeout(Duration::from_secs(5), client_in.recv()) @@ -647,7 +651,7 @@ mod tests { let (client_out, _client_in, client_shutdown, client_handle) = drive(client_conn); client_out - .send(padded(Command::Request, total)) + .send(padded(Command::Request, total).into()) .await .expect("client send 1 MiB"); let received = compio::time::timeout(Duration::from_secs(15), server_in.recv()) @@ -683,7 +687,7 @@ mod tests { drive_with_cap(client_conn, framing::MAX_MESSAGE_SIZE); client_out - .send(padded(Command::Request, OVER_CAP)) + .send(padded(Command::Request, OVER_CAP).into()) .await .expect("client send oversize"); diff --git a/core/message_bus/tests/tcp_tls_client_listener.rs b/core/message_bus/tests/tcp_tls_client_listener.rs index 0ce3ed030e..d477d1cd45 100644 --- a/core/message_bus/tests/tcp_tls_client_listener.rs +++ b/core/message_bus/tests/tcp_tls_client_listener.rs @@ -22,6 +22,7 @@ use common::{header_only, install_tls_clients_locally, loopback}; use compio::net::TcpStream; use iggy_binary_protocol::Command; use iggy_binary_protocol::GenericHeader; +use message_bus::BusMessage; use message_bus::client_listener::RequestHandler; use message_bus::client_listener::tcp_tls::{bind, run}; use message_bus::transports::tcp_tls::TcpTlsTransportConn; @@ -30,7 +31,7 @@ use message_bus::transports::{ActorContext, TransportConn}; use message_bus::{FusedShutdown, IggyMessageBus, MessageBus, MessageBusConfig, Shutdown, framing}; use rustls::RootCertStore; use rustls::pki_types::ServerName; -use server_common::{MESSAGE_ALIGN, Message, iobuf::Frozen}; +use server_common::Message; use std::rc::Rc; use std::sync::Arc; use std::time::{Duration, Instant}; @@ -83,7 +84,7 @@ async fn tcp_tls_client_listener_accepts_and_round_trips() { let client_tcp = TcpStream::connect(server_addr).await.expect("client dial"); let conn = TcpTlsTransportConn::new_client(client_tcp, client_cfg, server_name); - let (out_tx, out_rx) = bounded::>(8); + let (out_tx, out_rx) = bounded::(8); let (in_tx, in_rx) = bounded::>(8); let (client_shutdown, client_token) = Shutdown::new(); let ctx = ActorContext { @@ -99,7 +100,7 @@ async fn tcp_tls_client_listener_accepts_and_round_trips() { let client_handle = compio::runtime::spawn(async move { conn.run(ctx).await }); let request = header_only(Command::Request, 42, 0).into_frozen(); - out_tx.send(request).await.expect("client send"); + out_tx.send(request.into()).await.expect("client send"); let reply = compio::time::timeout(Duration::from_secs(5), in_rx.recv()) .await diff --git a/core/message_bus/tests/tcp_tls_client_roundtrip.rs b/core/message_bus/tests/tcp_tls_client_roundtrip.rs index f4d55f5c52..b3154f99c9 100644 --- a/core/message_bus/tests/tcp_tls_client_roundtrip.rs +++ b/core/message_bus/tests/tcp_tls_client_roundtrip.rs @@ -25,6 +25,7 @@ use common::{ use compio::net::TcpStream; use iggy_binary_protocol::Command; use iggy_binary_protocol::GenericHeader; +use message_bus::BusMessage; use message_bus::client_listener::RequestHandler; use message_bus::connector::DEFAULT_RECONNECT_PERIOD; use message_bus::replica::io::start_on_shard_zero; @@ -35,7 +36,7 @@ use message_bus::transports::{ActorContext, TransportConn}; use message_bus::{FusedShutdown, IggyMessageBus, MessageBus, Shutdown, framing}; use rustls::RootCertStore; use rustls::pki_types::ServerName; -use server_common::{MESSAGE_ALIGN, Message, iobuf::Frozen}; +use server_common::Message; use std::rc::Rc; use std::sync::Arc; use std::time::Duration; @@ -116,7 +117,7 @@ async fn start_on_shard_zero_tcp_tls_round_trip() { let client_tcp = TcpStream::connect(server_addr).await.expect("client dial"); let conn = TcpTlsTransportConn::new_client(client_tcp, client_cfg, server_name); - let (out_tx, out_rx) = bounded::>(8); + let (out_tx, out_rx) = bounded::(8); let (in_tx, in_rx) = bounded::>(8); let (client_shutdown, client_token) = Shutdown::new(); let ctx = ActorContext { @@ -132,7 +133,7 @@ async fn start_on_shard_zero_tcp_tls_round_trip() { let client_handle = compio::runtime::spawn(async move { conn.run(ctx).await }); let request = header_only(Command::Request, CLUSTER, 0).into_frozen(); - out_tx.send(request).await.expect("client send"); + out_tx.send(request.into()).await.expect("client send"); let reply = compio::time::timeout(Duration::from_secs(5), in_rx.recv()) .await diff --git a/core/message_bus/tests/wss_client_listener.rs b/core/message_bus/tests/wss_client_listener.rs index 56566a6c3e..f280d8ebf2 100644 --- a/core/message_bus/tests/wss_client_listener.rs +++ b/core/message_bus/tests/wss_client_listener.rs @@ -22,6 +22,7 @@ use common::{header_only, install_wss_clients_locally, loopback}; use compio::net::TcpStream; use iggy_binary_protocol::Command; use iggy_binary_protocol::GenericHeader; +use message_bus::BusMessage; use message_bus::client_listener::RequestHandler; use message_bus::client_listener::wss::{bind, run}; use message_bus::transports::tls::{install_default_crypto_provider, self_signed_for_loopback}; @@ -30,7 +31,7 @@ use message_bus::transports::{ActorContext, TransportConn}; use message_bus::{FusedShutdown, IggyMessageBus, MessageBus, MessageBusConfig, Shutdown, framing}; use rustls::RootCertStore; use rustls::pki_types::ServerName; -use server_common::{MESSAGE_ALIGN, Message, iobuf::Frozen}; +use server_common::Message; use std::rc::Rc; use std::sync::Arc; use std::time::{Duration, Instant}; @@ -78,7 +79,7 @@ async fn wss_client_listener_accepts_and_round_trips() { let client_tcp = TcpStream::connect(server_addr).await.expect("client dial"); let conn = WssTransportConn::new_client(client_tcp, client_cfg, server_name); - let (out_tx, out_rx) = bounded::>(8); + let (out_tx, out_rx) = bounded::(8); let (in_tx, in_rx) = bounded::>(8); let (client_shutdown, client_token) = Shutdown::new(); let ctx = ActorContext { @@ -94,7 +95,7 @@ async fn wss_client_listener_accepts_and_round_trips() { let client_handle = compio::runtime::spawn(async move { conn.run(ctx).await }); let request = header_only(Command::Request, 7, 0).into_frozen(); - out_tx.send(request).await.expect("client send"); + out_tx.send(request.into()).await.expect("client send"); let reply = compio::time::timeout(Duration::from_secs(5), in_rx.recv()) .await diff --git a/core/message_bus/tests/wss_client_roundtrip.rs b/core/message_bus/tests/wss_client_roundtrip.rs index 24cb272737..a1e2950fd6 100644 --- a/core/message_bus/tests/wss_client_roundtrip.rs +++ b/core/message_bus/tests/wss_client_roundtrip.rs @@ -25,6 +25,7 @@ use common::{ use compio::net::TcpStream; use iggy_binary_protocol::Command; use iggy_binary_protocol::GenericHeader; +use message_bus::BusMessage; use message_bus::client_listener::RequestHandler; use message_bus::connector::DEFAULT_RECONNECT_PERIOD; use message_bus::replica::io::start_on_shard_zero; @@ -35,7 +36,7 @@ use message_bus::transports::{ActorContext, TransportConn}; use message_bus::{FusedShutdown, IggyMessageBus, MessageBus, Shutdown, framing}; use rustls::RootCertStore; use rustls::pki_types::ServerName; -use server_common::{MESSAGE_ALIGN, Message, iobuf::Frozen}; +use server_common::Message; use std::rc::Rc; use std::sync::Arc; use std::time::Duration; @@ -117,7 +118,7 @@ async fn start_on_shard_zero_wss_round_trip() { let client_tcp = TcpStream::connect(server_addr).await.expect("client dial"); let conn = WssTransportConn::new_client(client_tcp, client_cfg, server_name); - let (out_tx, out_rx) = bounded::>(8); + let (out_tx, out_rx) = bounded::(8); let (in_tx, in_rx) = bounded::>(8); let (client_shutdown, client_token) = Shutdown::new(); let ctx = ActorContext { @@ -133,7 +134,7 @@ async fn start_on_shard_zero_wss_round_trip() { let client_handle = compio::runtime::spawn(async move { conn.run(ctx).await }); let request = header_only(Command::Request, CLUSTER, 0).into_frozen(); - out_tx.send(request).await.expect("client send"); + out_tx.send(request.into()).await.expect("client send"); let reply = compio::time::timeout(Duration::from_secs(5), in_rx.recv()) .await diff --git a/core/metadata/src/impls/metadata.rs b/core/metadata/src/impls/metadata.rs index 6ff355e474..31ce620cda 100644 --- a/core/metadata/src/impls/metadata.rs +++ b/core/metadata/src/impls/metadata.rs @@ -4056,7 +4056,9 @@ mod tests { use iggy_binary_protocol::requests::topics::CreateTopicRequest; use iggy_common::variadic; use journal::prepare_journal::PrepareJournal; - use message_bus::{ClientForwardFn, ConnectionLostFn, JoinHandle, ReplicaForwardFn, SendError}; + use message_bus::{ + BusMessage, ClientForwardFn, ConnectionLostFn, JoinHandle, ReplicaForwardFn, SendError, + }; use server_common::MESSAGE_ALIGN; use server_common::iobuf::Frozen; use std::cell::RefCell; @@ -4389,7 +4391,7 @@ mod tests { async fn send_to_client( &self, _client_id: u128, - _data: Frozen, + _data: impl Into, ) -> Result<(), SendError> { Ok(()) } @@ -4545,7 +4547,7 @@ mod tests { async fn send_to_client( &self, _client_id: u128, - _data: Frozen, + _data: impl Into, ) -> Result<(), SendError> { if self.stall.get() { self.stall_hits.set(self.stall_hits.get() + 1); diff --git a/core/partitions/src/iggy_partition.rs b/core/partitions/src/iggy_partition.rs index 7b538e68da..b08a4e2db5 100644 --- a/core/partitions/src/iggy_partition.rs +++ b/core/partitions/src/iggy_partition.rs @@ -5183,7 +5183,7 @@ mod tests { use compio::io::AsyncWriteAtExt; use consensus::LocalPipeline; use iggy_binary_protocol::{Command, ReplyHeader, WireConsumer, WireEncode}; - use message_bus::SendError; + use message_bus::{BusMessage, SendError}; use server_common::MESSAGE_ALIGN; use server_common::send_messages::{ COMMAND_HEADER_SIZE, IggyMessage, IggyMessageHeader, IggyMessages, SendMessagesOwned, @@ -5570,9 +5570,11 @@ mod tests { async fn send_to_client( &self, client_id: u128, - data: Frozen, + data: impl Into, ) -> Result<(), SendError> { - self.sent_to_clients.borrow_mut().push((client_id, data)); + self.sent_to_clients + .borrow_mut() + .push((client_id, data.into().into_contiguous())); Ok(()) } diff --git a/core/partitions/src/journal.rs b/core/partitions/src/journal.rs index fcb3a9af78..45f3b6d1d8 100644 --- a/core/partitions/src/journal.rs +++ b/core/partitions/src/journal.rs @@ -1153,6 +1153,68 @@ pub fn push_selected_batch_fragments( *matched_messages += selection.matched_messages; } +/// A fragment sliced from a storage buffer keeps the WHOLE allocation alive +/// until the reply frame is written out, and a reply can sit in a +/// per-connection mailbox for a while. Copy the matched bytes out when they +/// cover less than this fraction of the source, so a sparse match (a +/// `count=1` poll off a cold partition, a short poll into a large resident +/// batch) cannot pin a ~1 MiB chunk or a whole prepare per queued reply; a +/// dense match keeps the zero-copy path. +/// +/// On the disk tier, the chunk allocations a poll reply keeps alive are +/// bounded by `SPARSE_CHUNK_PIN_DIVISOR` times the record bytes it serves +/// from disk, plus one page per compacted chunk. This is a ratio, not an +/// absolute cap: a grown chunk can still retain tens of MiB, and it does not +/// cover the reply's absolute size. The resident tier applies the same ratio +/// per prepare entry but only copies up to [`RESIDENT_SPARSE_COPY_MAX_BYTES`]; +/// a sparse selection past that stays a slice and pins its prepare. +const SPARSE_CHUNK_PIN_DIVISOR: usize = 4; + +/// Resident copies run inline on the shard pump (the disk walk is detached), +/// and a poll selects at most two partial batches (its first and last), so +/// this caps the pump's per-poll memcpy at about twice this many bytes. +const RESIDENT_SPARSE_COPY_MAX_BYTES: usize = 64 * 1024; + +/// Rewrite the fragments pushed from index `pushed_from` on that slice +/// `source` to slices of one compact copy when their combined length is a +/// sparse fraction of `source` and at most `copy_max_bytes`. Fragments that +/// own their bytes (rewritten batch headers) are left alone. See +/// [`SPARSE_CHUNK_PIN_DIVISOR`]. +pub fn unpin_sparse_source( + fragments: &mut PollFragments<4096>, + pushed_from: usize, + source: &Frozen<4096>, + copy_max_bytes: usize, +) { + let pushed = &mut fragments[pushed_from..]; + let borrowed: usize = pushed + .iter() + .filter(|fragment| fragment.borrows_from(source)) + .map(Fragment::len) + .sum(); + if borrowed == 0 + || borrowed >= source.len() / SPARSE_CHUNK_PIN_DIVISOR + || borrowed > copy_max_bytes + { + return; + } + + let mut compact = Owned::<4096>::with_capacity(borrowed); + for fragment in pushed.iter().filter(|f| f.borrows_from(source)) { + compact.extend_from_slice(fragment.as_slice()); + } + let compact = Frozen::from(compact); + let mut cursor = 0; + for fragment in pushed.iter_mut() { + if !fragment.borrows_from(source) { + continue; + } + let len = fragment.len(); + *fragment = Fragment::slice(compact.clone(), cursor, cursor + len); + cursor += len; + } +} + /// Decode one resident `Frozen` entry and push its matching fragments. Shared by /// the live storage walk and the owned-snapshot walk so the corrupt-header skip /// and `SendMessages` filter live in one place. Skips (never panics) on a short @@ -1188,6 +1250,7 @@ fn try_push_resident_entry( }; // The batch's 256B header sits right after the prepare header in a resident // entry (see `decode_prepare_slice`), so the batch base is `PREPARE_HEADER_SIZE`. + let pushed_from = fragments.len(); push_selected_batch_fragments( fragments, last_matching_offset, @@ -1197,6 +1260,16 @@ fn try_push_resident_entry( &batch, selection, ); + // `evict_prefix` drains the storage on the routine commit flush, so a + // queued reply that still slices this prepare becomes its sole owner. + // Accounted per entry here; the disk walk accounts per chunk, where many + // batches share one allocation. + unpin_sparse_source( + fragments, + pushed_from, + prepare, + RESIDENT_SPARSE_COPY_MAX_BYTES, + ); } /// Poll an owned, point-in-time snapshot of the resident journal tail. @@ -1251,7 +1324,8 @@ mod tests { use journal::Journal; use server_common::Message; use server_common::send_messages::{ - IggyMessage, IggyMessageHeader, IggyMessages, SendMessagesOwned, decode_batch_slice, + BatchHeader, IggyMessage, IggyMessageHeader, IggyMessages, SendMessagesOwned, + decode_batch_slice, }; use server_common::sharding::IggyNamespace; @@ -1516,14 +1590,58 @@ mod tests { let mut owned = SendMessagesOwned::from_messages(IggyNamespace::new(1, 1, 0), &messages) .expect("build send_messages batch"); owned.header.base_timestamp = base_timestamp; - owned.header.batch_checksum = owned.header.checksum_for_blob(&owned.blob); + stamped_batch_record(owned) + } + /// Stamp `owned`'s checksum and lay it out as the `[256B batch header][blob]` + /// record a batch occupies in storage. + fn stamped_batch_record(mut owned: SendMessagesOwned) -> Vec { + owned.header.batch_checksum = owned.header.checksum_for_blob(&owned.blob); let mut record = vec![0u8; COMMAND_HEADER_SIZE + owned.blob.len()]; owned.header.encode_into(&mut record[..COMMAND_HEADER_SIZE]); record[COMMAND_HEADER_SIZE..].copy_from_slice(&owned.blob); record } + /// A resident `SendMessages` prepare entry holding one batch of + /// `message_count` records with distinct `payload_len`-byte payloads, in + /// the `[PrepareHeader][256B batch header][blob]` layout + /// `try_push_resident_entry` decodes. + fn build_resident_prepare(message_count: usize, payload_len: usize) -> Frozen<4096> { + let mut messages = IggyMessages::with_capacity(message_count); + for index in 0..message_count { + let fill = u8::try_from(index % usize::from(u8::MAX)).expect("bounded by u8::MAX"); + messages.push(IggyMessage { + header: IggyMessageHeader { + payload_length: u32::try_from(payload_len).expect("payload_len fits u32"), + ..Default::default() + }, + payload: Bytes::from(vec![fill; payload_len]), + user_headers: None, + }); + } + let owned = SendMessagesOwned::from_messages(IggyNamespace::new(1, 1, 0), &messages) + .expect("build send_messages batch"); + let record = stamped_batch_record(owned); + + let mut prepare = build_prepare(1, PREPARE_HEADER_SIZE + record.len()).transmute_header( + |header: PrepareHeader, send_messages: &mut PrepareHeader| { + *send_messages = header; + send_messages.operation = Operation::SendMessages; + }, + ); + prepare.as_mut_slice()[PREPARE_HEADER_SIZE..].copy_from_slice(&record); + prepare.into_frozen() + } + + fn offset_lookup(offset: u64, count: u32) -> MessageLookup { + MessageLookup::Offset { + offset, + count, + ceiling: u64::MAX, + } + } + #[test] fn timestamp_poll_at_exact_broker_timestamp_includes_the_batch() { // A client polls with a timestamp read from a previous reply, which is @@ -1562,4 +1680,125 @@ mod tests { "poll past the broker timestamp must match nothing" ); } + + /// A short poll into a large resident batch ships a rewritten header plus + /// a body slice. Left as a slice of the prepare, that body would keep the + /// whole entry alive after `evict_prefix` drained it from the journal. + #[test] + fn resident_partial_selection_copies_out_of_a_large_prepare() { + let prepare = build_resident_prepare(1_000, 300); + let query = offset_lookup(500, 1); + let batch = decode_prepare_slice_trusted(prepare.as_slice()).expect("prepare decodes"); + let selection = select_batch_slice(&batch, query, 0).expect("record 500 is selected"); + let expected_body = &batch.blob()[selection.start..selection.end]; + assert!( + expected_body.len() < prepare.len() / SPARSE_CHUNK_PIN_DIVISOR, + "fixture must select a sparse fraction of the prepare" + ); + + let (fragments, last_matching_offset) = + select_resident(std::slice::from_ref(&prepare), query).expect("one record matches"); + assert_eq!(last_matching_offset, Some(500)); + assert_eq!(fragments.len(), 2, "rewritten header plus body slice"); + let (header, body) = (&fragments[0], &fragments[1]); + assert!( + !body.borrows_from(&prepare), + "sparse body must be copied out of the prepare" + ); + assert_eq!(body.as_slice(), expected_body); + assert_eq!(header.len(), COMMAND_HEADER_SIZE); + assert!( + !header.borrows_from(&body.clone().into_frozen()), + "the owned header must stay out of the compact copy" + ); + let rewritten = BatchHeader::decode(header.as_slice()).expect("rewritten header decodes"); + assert_eq!(rewritten.message_count, 1); + } + + #[test] + fn resident_whole_batch_selection_keeps_the_zero_copy_slice() { + let prepare = build_resident_prepare(1_000, 300); + + let (fragments, last_matching_offset) = + select_resident(std::slice::from_ref(&prepare), offset_lookup(0, 1_000)) + .expect("whole batch matches"); + assert_eq!(last_matching_offset, Some(999)); + assert_eq!(fragments.len(), 1, "a whole batch ships its original bytes"); + assert!( + fragments[0].borrows_from(&prepare), + "dense selection keeps the zero-copy path" + ); + assert_eq!(fragments[0].len(), prepare.len() - PREPARE_HEADER_SIZE); + } + + /// The resident copy runs inline on the shard pump, so past the byte cap + /// a sparse selection stays a zero-copy slice even though it pins the + /// prepare. + #[test] + fn resident_sparse_copy_stops_at_the_byte_cap() { + let prepare = build_resident_prepare(2_000, 300); + let query = offset_lookup(0, 200); + let batch = decode_prepare_slice_trusted(prepare.as_slice()).expect("prepare decodes"); + let selection = select_batch_slice(&batch, query, 0).expect("200 records are selected"); + let selected = selection.end - selection.start; + assert!( + selected > RESIDENT_SPARSE_COPY_MAX_BYTES + && selected < prepare.len() / SPARSE_CHUNK_PIN_DIVISOR, + "fixture must select a sparse fraction that is over the copy cap" + ); + + let (fragments, _) = + select_resident(std::slice::from_ref(&prepare), query).expect("records match"); + assert_eq!(fragments.len(), 2, "rewritten header plus body slice"); + assert!( + fragments[1].borrows_from(&prepare), + "over the cap the body must stay a slice of the prepare" + ); + } + + /// A sparse match must not pin the whole disk chunk: the matched bytes + /// are copied out byte-for-byte and the fragments stop borrowing the + /// chunk allocation. A dense match keeps the zero-copy slices, and + /// fragments that already own their bytes (rewritten batch headers) are + /// never touched. + #[test] + fn unpin_sparse_source_bounds_chunk_retention() { + let chunk_len = 1 << 20; + let mut backing = Owned::<4096>::zeroed(chunk_len); + for (position, byte) in backing.as_mut_slice().iter_mut().enumerate() { + *byte = u8::try_from(position % 251).unwrap(); + } + let chunk = Frozen::from(backing); + + let mut fragments = PollFragments::<4096>::new(); + fragments.push(Fragment::whole(Owned::<4096>::zeroed(256).into())); + fragments.push(Fragment::slice(chunk.clone(), 512, 512 + 600)); + fragments.push(Fragment::slice(chunk.clone(), 4096, 4096 + 300)); + let first = fragments[1].as_slice().to_vec(); + let second = fragments[2].as_slice().to_vec(); + unpin_sparse_source(&mut fragments, 0, &chunk, usize::MAX); + assert!( + !fragments[1].borrows_from(&chunk) && !fragments[2].borrows_from(&chunk), + "sparse slices must be copied out of the chunk" + ); + assert_eq!(fragments[1].as_slice(), &first[..]); + assert_eq!(fragments[2].as_slice(), &second[..]); + assert!( + fragments[1].borrows_from(&fragments[2].clone().into_frozen()), + "copies pack into one compact allocation" + ); + assert_eq!(fragments[0].len(), 256); + + let mut fragments = PollFragments::<4096>::new(); + fragments.push(Fragment::slice( + chunk.clone(), + 0, + chunk_len / SPARSE_CHUNK_PIN_DIVISOR, + )); + unpin_sparse_source(&mut fragments, 0, &chunk, usize::MAX); + assert!( + fragments[0].borrows_from(&chunk), + "dense slice keeps the zero-copy path" + ); + } } diff --git a/core/partitions/src/messages_writer.rs b/core/partitions/src/messages_writer.rs index d689127190..63ad93c113 100644 --- a/core/partitions/src/messages_writer.rs +++ b/core/partitions/src/messages_writer.rs @@ -20,7 +20,7 @@ use compio::{ io::AsyncWriteAtExt, }; use iggy_common::{IggyByteSize, IggyError}; -use server_common::iobuf::Frozen; +use server_common::iobuf::{Frozen, IOV_MAX}; use std::{ rc::Rc, sync::atomic::{AtomicU64, Ordering}, @@ -30,8 +30,6 @@ use tracing::{error, warn}; #[cfg(target_os = "linux")] use nix::fcntl::{FallocateFlags, fallocate}; -const MAX_IOV_COUNT: usize = 1024; - #[derive(Debug)] pub struct MessagesWriter { file_path: String, @@ -217,7 +215,7 @@ async fn write_frozen_chunked( mut position: u64, buffers: &[Frozen], ) -> Result<(), IggyError> { - for chunk in buffers.chunks(MAX_IOV_COUNT) { + for chunk in buffers.chunks(IOV_MAX) { let chunk_size: usize = chunk.iter().map(Frozen::len).sum(); let chunk_vec: Vec<_> = chunk.to_vec(); diff --git a/core/partitions/src/poll_plan.rs b/core/partitions/src/poll_plan.rs index d982fa1bc7..7d3fae869d 100644 --- a/core/partitions/src/poll_plan.rs +++ b/core/partitions/src/poll_plan.rs @@ -33,7 +33,9 @@ use crate::PollFragments; use crate::iggy_index::{IGGY_INDEX_SIZE, IggyIndexCache}; use crate::iggy_index_reader::IggyIndexReader; -use crate::journal::{MessageLookup, push_selected_batch_fragments, select_batch_slice}; +use crate::journal::{ + MessageLookup, push_selected_batch_fragments, select_batch_slice, unpin_sparse_source, +}; use compio::io::AsyncReadAtExt; use iggy_common::{ ConsumerGroupId, ConsumerGroupOffsets, ConsumerKind, ConsumerOffset, ConsumerOffsets, IggyError, @@ -572,6 +574,7 @@ impl DiskReadPlan { faulted = true; break 'walk; }; + let fragments_before_chunk = fragments.len(); let ChunkWalk { consumed, corrupt } = walk_disk_chunk( &chunk, query, @@ -586,6 +589,8 @@ impl DiskReadPlan { }, self.namespace_raw, ); + // Detached from the pump, so the ratio alone bounds the copy. + unpin_sparse_source(&mut fragments, fragments_before_chunk, &chunk, usize::MAX); if corrupt { // A batch that does not match its own checksum. Fail closed like // an IO fault: serving it hands a consumer data provably not what diff --git a/core/partitions/src/types.rs b/core/partitions/src/types.rs index 6c956f7e35..4762b772a7 100644 --- a/core/partitions/src/types.rs +++ b/core/partitions/src/types.rs @@ -75,6 +75,28 @@ impl Fragment { self.source.slice(self.start..self.end) } } + + #[must_use] + pub fn as_slice(&self) -> &[u8] { + &self.source[self.start..self.end] + } + + #[must_use] + pub const fn len(&self) -> usize { + self.end - self.start + } + + #[must_use] + pub const fn is_empty(&self) -> bool { + self.start == self.end + } + + /// Whether this fragment refcounts `source`'s allocation (and thus keeps + /// all of it alive, not just the sliced window). + #[must_use] + pub fn borrows_from(&self, source: &Frozen) -> bool { + self.source.shares_allocation(source) + } } /// Arguments for polling messages from a partition. diff --git a/core/server/src/dispatch.rs b/core/server/src/dispatch.rs index 784b86bdf6..3bf5af28ae 100644 --- a/core/server/src/dispatch.rs +++ b/core/server/src/dispatch.rs @@ -41,7 +41,7 @@ use crate::pat::maybe_rewrite_pat_request; use crate::responses::{ NonReplicatedResponse, build_consumer_offset_body, build_deny_reply, build_empty_reply, build_get_me_response, build_get_personal_access_tokens_response, - build_non_replicated_response, build_polled_messages_body, build_raw_pat_reply, + build_non_replicated_response, build_polled_messages_reply, build_raw_pat_reply, connected_client_to_response, current_metadata_commit, resolve_partition_namespace, resolve_partition_request_namespace, }; @@ -103,10 +103,10 @@ use iggy_common::{ }; use journal::superblock::SuperblockStore; use journal::{Journal, JournalHandle}; -use message_bus::AUTO_COMMIT_CLIENT_ID; use message_bus::client_listener::RequestHandler; use message_bus::framing::MAX_MESSAGE_SIZE; use message_bus::replica::listener::MessageHandler; +use message_bus::{AUTO_COMMIT_CLIENT_ID, BusMessage}; use metadata::impls::metadata::{ BoundSession, MetadataSubmitError, StreamsFrontend, build_truncate_partition_client_message, build_truncate_partition_client_message_with_identifiers, @@ -1941,11 +1941,30 @@ async fn send_non_replicated_bytes( request.header().session, commit, ); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { + send_reply_frame( + shard, + transport_client_id, + reply.into_generic().into_frozen(), + label, + ) + .await; +} + +/// Hand a built reply frame to the bus for `transport_client_id`. +#[allow(clippy::future_not_send)] +async fn send_reply_frame( + shard: &Rc>, + transport_client_id: u128, + frame: impl Into, + label: &'static str, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + if let Err(error) = shard.bus.send_to_client(transport_client_id, frame).await { warn!(transport_client_id, label, error = %error, "failed to send non-replicated reply"); } } @@ -2169,20 +2188,27 @@ async fn handle_poll_messages( Some(PartitionReadReply::Poll { fragments, current_offset, - }) => build_polled_messages_body( + }) => match build_polled_messages_reply( + request.header(), + current_metadata_commit(shard), partition_id, current_offset, fragments, shard.plane.partitions().config().encryptor.as_deref(), - ) - .unwrap_or_else(|error| { - warn!( - transport_client_id, - error = %error, - "failed to re-encode polled batches; replying empty poll" - ); - empty_polled_messages_body(partition_id) - }), + ) { + Ok(reply) => { + send_reply_frame(shard, transport_client_id, reply, "poll_messages").await; + return; + } + Err(error) => { + warn!( + transport_client_id, + error = %error, + "failed to re-encode polled batches; replying empty poll" + ); + empty_polled_messages_body(partition_id) + } + }, other => { warn!( transport_client_id, @@ -3823,6 +3849,7 @@ mod tests { use iggy_common::defaults::DEFAULT_ROOT_USER_ID; use iggy_common::variadic; use journal::prepare_journal::PrepareJournal; + use message_bus::BusMessage; use message_bus::client_listener::RequestHandler; use message_bus::fd_transfer::DupedFd; use message_bus::installer::ConnectionInstaller; @@ -3895,11 +3922,11 @@ mod tests { async fn send_to_client( &self, client_id: u128, - data: Frozen, + data: impl Into, ) -> Result<(), SendError> { self.client_replies .borrow_mut() - .push((client_id, data.as_slice().to_vec())); + .push((client_id, data.into().into_contiguous().as_slice().to_vec())); Ok(()) } async fn send_to_replica( @@ -4496,7 +4523,10 @@ mod tests { fn reply_lane_forward(client_id: u128) -> ShardFrame { ShardFrame::lifecycle(LifecycleFrame::ForwardClientSend { client_id, - msg: server_common::iobuf::Owned::::zeroed(64).into(), + msg: server_common::iobuf::Frozen::from( + server_common::iobuf::Owned::::zeroed(64), + ) + .into(), }) } @@ -4646,7 +4676,7 @@ mod tests { if let ShardFrame::Lifecycle(LifecycleFrame::ForwardClientSend { client_id, msg }) = frame { - denies.push((client_id, msg.as_slice().to_vec())); + denies.push((client_id, msg.into_contiguous().as_slice().to_vec())); } } assert_eq!( diff --git a/core/server/src/http/reply.rs b/core/server/src/http/reply.rs index c17f5ce788..0ce0b14f4b 100644 --- a/core/server/src/http/reply.rs +++ b/core/server/src/http/reply.rs @@ -33,8 +33,7 @@ use iggy_common::{ ConsumerGroupDetails, IggyError, StreamDetails, TopicDetails, UserInfoDetails, eviction_reason_to_error, }; -use message_bus::BusMessage; -use server_common::Message; +use server_common::{MESSAGE_ALIGN, Message, iobuf::Frozen}; use tracing::warn; use crate::http::error::{PartitionWriteError, WriteError}; @@ -60,7 +59,7 @@ use crate::login_register::LoginRegisterError; /// keeps that a single parse whose failure mode is already decided here, rather /// than a second one whose fallback would have to invent a body. pub(in crate::http) fn classify_partition_reply( - reply: &BusMessage, + reply: &Frozen, ) -> Result { let header = reply .as_slice() @@ -93,7 +92,7 @@ pub(in crate::http) fn classify_partition_reply( /// `header` is the one [`classify_partition_reply`] graded, so this reads the /// body without re-deriving its extent. pub(in crate::http) fn send_confirmations( - reply: &BusMessage, + reply: &Frozen, header: &ReplyHeader, ) -> Option { let body = partition_reply_body(reply, header); @@ -115,7 +114,7 @@ pub(in crate::http) fn send_confirmations( /// A partition reply's body past the header, bounded by the header's `size` /// rather than by the buffer length: `size` is the frame's authoritative /// extent, and the typed decoders reject trailing bytes. -fn partition_reply_body<'a>(reply: &'a BusMessage, header: &ReplyHeader) -> &'a [u8] { +fn partition_reply_body<'a>(reply: &'a Frozen, header: &ReplyHeader) -> &'a [u8] { reply .as_slice() .get(HEADER_SIZE..header.size as usize) @@ -362,7 +361,7 @@ mod tests { ); } - fn frozen(reply: Message) -> BusMessage { + fn frozen(reply: Message) -> Frozen { reply.into_generic().into_frozen() } @@ -486,7 +485,7 @@ mod tests { )); } - fn send_reply(body: &Bytes) -> BusMessage { + fn send_reply(body: &Bytes) -> Frozen { let prepare = PrepareHeader { command: Command::Prepare, operation: Operation::SendMessages, diff --git a/core/server/src/http/submit.rs b/core/server/src/http/submit.rs index eeaf9dc1ad..e7dd75efd3 100644 --- a/core/server/src/http/submit.rs +++ b/core/server/src/http/submit.rs @@ -28,9 +28,8 @@ use futures::channel::oneshot; use iggy_binary_protocol::consensus::Command; use iggy_binary_protocol::{GenericHeader, Operation, ReplyHeader, RoutedRequestHeader}; use iggy_common::IggyError; -use message_bus::BusMessage; use metadata::impls::metadata::StreamsFrontend; -use server_common::Message; +use server_common::{MESSAGE_ALIGN, Message, iobuf::Frozen}; use tracing::warn; use crate::dispatch::{ @@ -370,7 +369,7 @@ pub(in crate::http) async fn partition_write_replicated( session: &HttpSession, operation: Operation, body: &[u8], -) -> Result<(BusMessage, ReplyHeader), PartitionWriteError> { +) -> Result<(Frozen, ReplyHeader), PartitionWriteError> { // Admission sits here rather than before body decode: axum's extractors // already buffered and deserialized the body (bounded by the router-wide // `DefaultBodyLimit`) before the handler ran, so the caps gate what is @@ -414,7 +413,10 @@ pub(in crate::http) async fn partition_write_replicated( // reply after a timeout sheds at the bus instead of leaking a waiter. drop(guard); match outcome { - Ok(Ok(reply)) => classify_partition_reply(&reply).map(|header| (reply, header)), + Ok(Ok(reply)) => { + let reply = reply.into_contiguous(); + classify_partition_reply(&reply).map(|header| (reply, header)) + } // Cancelled (reply target torn down by session eviction mid-wait) or // elapsed: same caller contract either way - outcome unknown, 504. Ok(Err(_)) | Err(_) => Err(PartitionWriteError::Timeout(operation)), diff --git a/core/server/src/partition_reconciler.rs b/core/server/src/partition_reconciler.rs index 5de84bcc1b..d0216aea7d 100644 --- a/core/server/src/partition_reconciler.rs +++ b/core/server/src/partition_reconciler.rs @@ -4470,7 +4470,7 @@ mod tests { fn reply_status(reply: &message_bus::BusMessage) -> u32 { bytemuck::checked::try_from_bytes::( - &reply.as_ref()[..size_of::()], + &reply.first().as_slice()[..size_of::()], ) .expect("deny reply carries a valid ReplyHeader") .status diff --git a/core/server/src/responses.rs b/core/server/src/responses.rs index 6575456ae6..4c48c0514b 100644 --- a/core/server/src/responses.rs +++ b/core/server/src/responses.rs @@ -88,11 +88,13 @@ use iggy_common::{ }; use journal::superblock::SuperblockStore; use journal::{Journal, JournalHandle}; +use message_bus::BusMessage; use metadata::impls::metadata::StreamsFrontend; -use partitions::PollFragments; -use server_common::Message; +use partitions::{Fragment, PollFragments}; +use server_common::iobuf::{Frozen, Owned}; use server_common::send_messages; use server_common::sharding::IggyNamespace; +use server_common::{MESSAGE_ALIGN, Message, ResponseBacking, ResponseFragments}; use shard::ConnectedClientInfo; use std::cell::RefCell; use std::net::IpAddr; @@ -1610,15 +1612,25 @@ pub fn build_reply_with_body( ) -> Message { let header_len = std::mem::size_of::(); let total_size = header_len + body_len; + let size = u32::try_from(total_size).expect("reply size must fit into u32"); let mut reply = Message::::new(total_size); - let header_size = u32::try_from(total_size).expect("reply size must fit into u32"); - let header = bytemuck::checked::try_from_bytes_mut::( - &mut reply.as_mut_slice()[..header_len], - ) - .expect("zeroed bytes are valid"); - *header = ReplyHeader { + let header = reply_header(request_header, client_id, session, commit, size); + reply.as_mut_slice()[..header_len].copy_from_slice(bytemuck::bytes_of(&header)); + write_body(&mut reply.as_mut_slice()[header_len..total_size]); + reply +} + +/// The header of a `size`-byte reply frame answering `request_header`. +fn reply_header( + request_header: &RoutedRequestHeader, + client_id: u128, + session: u64, + commit: u64, + size: u32, +) -> ReplyHeader { + ReplyHeader { cluster: request_header.cluster, - size: header_size, + size, view: request_header.view, release: request_header.release, command: Command::Reply, @@ -1631,9 +1643,7 @@ pub fn build_reply_with_body( request: request_header.request, operation: request_header.operation, ..Default::default() - }; - write_body(&mut reply.as_mut_slice()[header_len..total_size]); - reply + } } pub fn current_metadata_commit(shard: &Rc>) -> u64 @@ -1652,8 +1662,148 @@ where .map_or(0, VsrConsensus::commit_max) } +/// Body head of a `PolledMessages` reply: +/// `[partition_id:4][current_offset:8][count:4]`, before the batch records. +const POLLED_HEAD_LEN: usize = 16; + +/// Build the `PolledMessages` reply for the wire as a vectored frame: one +/// buffer holding the reply header and the body head, then the poll +/// fragments as they are. The record bytes are never copied or gathered; +/// their reply encoding IS the storage encoding (see +/// [`build_polled_messages_body`]), so `count` comes from walking the batch +/// headers in place. At-rest decryption is the one case that must rewrite +/// records, and it takes the flattening builder instead. +pub fn build_polled_messages_reply( + request_header: &RoutedRequestHeader, + commit: u64, + partition_id: u32, + current_offset: u64, + fragments: PollFragments, + encryptor: Option<&EncryptorKind>, +) -> Result { + let client_id = request_header.client; + let session = request_header.session; + if encryptor.is_some() { + let body = build_polled_messages_body(partition_id, current_offset, fragments, encryptor)?; + let reply = build_reply_from_bytes(request_header, client_id, session, commit, &body); + return Ok(reply.into_generic().into_frozen().into()); + } + + let mut frames = ResponseFragments::with_capacity(fragments.len() + 1); + frames.extend(fragments.into_iter().map(Fragment::into_frozen)); + let count = polled_message_count(&frames)?; + let records_len: usize = frames.iter().map(Frozen::len).sum(); + + let header_len = std::mem::size_of::(); + let size = u32::try_from(header_len + POLLED_HEAD_LEN + records_len) + .map_err(|_| IggyError::InvalidCommand)?; + let header = reply_header(request_header, client_id, session, commit, size); + let mut head = Owned::::zeroed(header_len + POLLED_HEAD_LEN); + let (header_bytes, body_head) = head.as_mut_slice().split_at_mut(header_len); + header_bytes.copy_from_slice(bytemuck::bytes_of(&header)); + body_head[..4].copy_from_slice(&partition_id.to_le_bytes()); + body_head[4..12].copy_from_slice(¤t_offset.to_le_bytes()); + body_head[12..].copy_from_slice(&count.to_le_bytes()); + frames.insert(0, head.into()); + + // Re-checks the header and that the fragments cover `size`. + Message::::try_from(frames) + .map(Message::into_inner) + .map_err(|_| IggyError::InvalidCommand) +} + +/// Sum of `message_count` over the batch records spanning `fragments`, read +/// from each batch header in place. Rejects a stream that is not a whole +/// number of batches, as [`build_polled_messages_body`] does. +fn polled_message_count(fragments: &[Frozen]) -> Result { + let mut cursor = FragmentCursor::new(fragments); + let mut count = 0u32; + let mut header = [0u8; send_messages::COMMAND_HEADER_SIZE]; + while !cursor.is_exhausted() { + cursor.read_exact(&mut header)?; + let batch = + send_messages::BatchHeader::decode(&header).map_err(|_| IggyError::InvalidCommand)?; + cursor.skip(batch.blob_len().map_err(|_| IggyError::InvalidCommand)?)?; + count = count + .checked_add(batch.message_count) + .ok_or(IggyError::InvalidCommand)?; + } + Ok(count) +} + +/// Byte cursor over the virtual concatenation of `fragments`. Rests on an +/// unread byte or at the end of the stream, never inside an exhausted +/// fragment, so a batch header split across fragments reads the same as one +/// stored whole. +struct FragmentCursor<'a> { + fragments: &'a [Frozen], + index: usize, + offset: usize, +} + +impl<'a> FragmentCursor<'a> { + fn new(fragments: &'a [Frozen]) -> Self { + let mut cursor = Self { + fragments, + index: 0, + offset: 0, + }; + cursor.settle(); + cursor + } + + const fn is_exhausted(&self) -> bool { + self.index == self.fragments.len() + } + + fn read_exact(&mut self, out: &mut [u8]) -> Result<(), IggyError> { + let mut filled = 0; + while filled < out.len() { + let available = self.available()?; + let take = available.len().min(out.len() - filled); + out[filled..filled + take].copy_from_slice(&available[..take]); + filled += take; + self.advance(take); + } + Ok(()) + } + + fn skip(&mut self, mut len: usize) -> Result<(), IggyError> { + while len > 0 { + let take = self.available()?.len().min(len); + len -= take; + self.advance(take); + } + Ok(()) + } + + /// Unread bytes of the current fragment; `Err` past the end of the stream. + fn available(&self) -> Result<&'a [u8], IggyError> { + self.fragments + .get(self.index) + .map(|fragment| &fragment.as_slice()[self.offset..]) + .ok_or(IggyError::InvalidCommand) + } + + fn advance(&mut self, len: usize) { + self.offset += len; + self.settle(); + } + + /// Step past the current fragment once it is used up, and past empty ones. + fn settle(&mut self) { + while let Some(fragment) = self.fragments.get(self.index) { + if self.offset < fragment.len() { + break; + } + self.offset = 0; + self.index += 1; + } + } +} + /// Build the `PolledMessages` reply body from the owning shard's poll -/// fragments. +/// fragments, gathered into one buffer. /// /// Fragments carry the stored batch records (a 256-byte batch header plus /// `[48B header][payload][user_headers]` frames, deltas resolved against the @@ -1662,6 +1812,10 @@ where /// decryption: stored sections are ciphertext, and this reply is the single /// decrypt point, so encrypted records are rebuilt over the plaintext. /// +/// The binary transports reply through [`build_polled_messages_reply`], which +/// ships the fragments without gathering them; this builder serves the +/// decrypt path and the HTTP handler, which decodes the body into JSON. +/// /// Body layout: `[partition_id:4][current_offset:8][count:4][batch records...]`. pub fn build_polled_messages_body( partition_id: u32, @@ -2005,4 +2159,232 @@ mod tests { assert_eq!(unlimited.max_topic_size, u64::MAX); assert_eq!(unlimited.message_expiry, u64::MAX); } + + // Vectored `PolledMessages` replies against the flattening builder as the + // byte-for-byte oracle. + + use iggy_common::Aes256GcmEncryptor; + use server_common::send_messages::{ + BatchHeader, COMMAND_HEADER_SIZE, IggyMessage, IggyMessageHeader, IggyMessages, + PREPARE_SPLIT_POINT, SendMessagesOwned, encrypt_batch_request, frozen_batch_header, + }; + use server_common::sharding::IggyNamespace; + + const POLL_PARTITION_ID: u32 = 9; + const POLL_CURRENT_OFFSET: u64 = 1_234; + const POLL_COMMIT: u64 = 17; + + fn poll_request_header() -> RoutedRequestHeader { + pat_request_header() + } + + /// A stored batch record over an opaque blob. Both builders decode only + /// the 256-byte batch header, so the blob needs no message framing. + fn batch_record(base_offset: u64, message_count: u32, blob: &[u8]) -> Frozen { + let batch_length = u64::try_from(COMMAND_HEADER_SIZE + blob.len()).expect("fits u64"); + let mut header = + BatchHeader::new(u64::from(POLL_PARTITION_ID), 5, batch_length, message_count); + header.base_offset = base_offset; + let mut bytes = vec![0u8; COMMAND_HEADER_SIZE + blob.len()]; + header.encode_into(&mut bytes[..COMMAND_HEADER_SIZE]); + bytes[COMMAND_HEADER_SIZE..].copy_from_slice(blob); + Owned::::copy_from_slice(&bytes).into() + } + + /// The wire bytes the flattening builder ships for `fragments`. + fn flattened_reply(fragments: PollFragments, encryptor: Option<&EncryptorKind>) -> Vec { + let header = poll_request_header(); + let body = build_polled_messages_body( + POLL_PARTITION_ID, + POLL_CURRENT_OFFSET, + fragments, + encryptor, + ) + .expect("flattening builder accepts the fragments"); + build_reply_from_bytes(&header, header.client, header.session, POLL_COMMIT, &body) + .into_generic() + .into_frozen() + .as_slice() + .to_vec() + } + + fn vectored_reply( + fragments: PollFragments, + encryptor: Option<&EncryptorKind>, + ) -> Result { + build_polled_messages_reply( + &poll_request_header(), + POLL_COMMIT, + POLL_PARTITION_ID, + POLL_CURRENT_OFFSET, + fragments, + encryptor, + ) + } + + /// The vectored reply must be byte-identical to the flattened one and + /// ship exactly `fragment_count` buffers. + fn assert_vectored_matches_flattened(fragments: PollFragments, fragment_count: usize) { + let expected = flattened_reply(fragments.clone(), None); + let reply = + vectored_reply(fragments, None).expect("vectored builder accepts the fragments"); + assert_eq!(reply.fragments().len(), fragment_count); + assert_eq!(reply.total_len(), expected.len()); + assert_eq!(reply.into_contiguous().as_slice(), expected.as_slice()); + } + + fn polled_count(reply: &[u8]) -> u32 { + let count_at = std::mem::size_of::() + 12; + u32::from_le_bytes(reply[count_at..count_at + 4].try_into().expect("4 bytes")) + } + + #[test] + fn polled_reply_single_fragment_matches_flattened_builder() { + let record = batch_record(0, 3, &[0xAB; 100]); + let fragments = PollFragments::from_iter([Fragment::whole(record)]); + assert_vectored_matches_flattened(fragments.clone(), 2); + + let reply = vectored_reply(fragments, None) + .expect("reply") + .into_contiguous(); + let header = bytemuck::checked::try_from_bytes::( + &reply.as_slice()[..std::mem::size_of::()], + ) + .expect("reply header decodes"); + assert_eq!(header.size as usize, reply.len()); + assert_eq!(header.client, 42); + assert_eq!(header.op, 7); + assert_eq!(header.commit, POLL_COMMIT); + assert_eq!(polled_count(reply.as_slice()), 3); + } + + #[test] + fn polled_reply_split_batch_matches_flattened_builder() { + // The journal slices a partially selected batch into a rewritten header + // plus a blob slice, exactly how `push_selected_batch_fragments` does. + let source = batch_record(10, 4, &[0x11; 400]); + let (start, end) = (100, 300); + let batch_length = u64::try_from(COMMAND_HEADER_SIZE + (end - start)).expect("fits u64"); + let mut rewritten = BatchHeader::new(u64::from(POLL_PARTITION_ID), 5, batch_length, 2); + rewritten.base_offset = 10; + let fragments = PollFragments::from_iter([ + Fragment::whole(frozen_batch_header(&rewritten)), + Fragment::slice( + source, + COMMAND_HEADER_SIZE + start, + COMMAND_HEADER_SIZE + end, + ), + ]); + assert_vectored_matches_flattened(fragments.clone(), 3); + let reply = vectored_reply(fragments, None) + .expect("reply") + .into_contiguous(); + assert_eq!(polled_count(reply.as_slice()), 2); + } + + #[test] + fn polled_reply_multiple_batches_counts_every_header() { + let first = batch_record(0, 1, &[0x01; 50]); + let second = batch_record(1, 4, &[0x02; 700]); + let third = batch_record(5, 7, &[0x03; 20]); + // `second` arrives cut mid-header so the count walk has to read a batch + // header spanning two fragments. + let fragments = PollFragments::from_iter([ + Fragment::whole(first), + Fragment::slice(second.clone(), 0, 100), + Fragment::slice(second.clone(), 100, second.len()), + Fragment::whole(third), + ]); + assert_vectored_matches_flattened(fragments.clone(), 5); + let reply = vectored_reply(fragments, None) + .expect("reply") + .into_contiguous(); + assert_eq!(polled_count(reply.as_slice()), 12); + } + + #[test] + fn polled_reply_empty_poll_is_the_head_alone() { + assert_vectored_matches_flattened(PollFragments::new(), 1); + let reply = vectored_reply(PollFragments::new(), None) + .expect("reply") + .into_contiguous(); + assert_eq!( + reply.len(), + std::mem::size_of::() + POLLED_HEAD_LEN + ); + assert_eq!(polled_count(reply.as_slice()), 0); + } + + #[test] + fn polled_reply_rejects_a_truncated_record() { + let record = batch_record(0, 3, &[0xAB; 100]); + let truncated = + PollFragments::from_iter([Fragment::slice(record, 0, COMMAND_HEADER_SIZE + 99)]); + assert!(matches!( + build_polled_messages_body( + POLL_PARTITION_ID, + POLL_CURRENT_OFFSET, + truncated.clone(), + None + ), + Err(IggyError::InvalidCommand) + )); + assert!(matches!( + vectored_reply(truncated, None), + Err(IggyError::InvalidCommand) + )); + } + + /// A stored record encrypted the way the primary encrypts at ingestion. + fn encrypted_record(encryptor: &EncryptorKind) -> Frozen { + let namespace = IggyNamespace::new(1, 1, 3); + let mut messages = IggyMessages::with_capacity(2); + for (id, payload) in [(7u128, &b"first-payload"[..]), (8, &b"second-payload"[..])] { + messages.push(IggyMessage { + header: IggyMessageHeader { + id, + origin_timestamp: 1_000, + ..Default::default() + }, + payload: Bytes::copy_from_slice(payload), + user_headers: None, + }); + } + let owned = SendMessagesOwned::from_messages(namespace, &messages).expect("build batch"); + let header_size = std::mem::size_of::(); + let total = header_size + owned.header.total_size(); + let mut buffer = Owned::::zeroed(total); + { + let header: &mut RoutedRequestHeader = + bytemuck::checked::try_from_bytes_mut(&mut buffer.as_mut_slice()[..header_size]) + .expect("zeroed bytes form a valid RoutedRequestHeader"); + header.command = Command::Request; + header.operation = Operation::SendMessages; + header.client = 1; + header.session = 1; + header.request = 1; + header.size = u32::try_from(total).expect("size fits u32"); + } + let bytes = buffer.as_mut_slice(); + owned + .header + .encode_into(&mut bytes[header_size..header_size + COMMAND_HEADER_SIZE]); + bytes[PREPARE_SPLIT_POINT..].copy_from_slice(&owned.blob); + let canonical = Message::try_from(buffer).expect("request message is valid"); + let encrypted = encrypt_batch_request(canonical, encryptor).expect("encrypt batch"); + let record = &encrypted.as_slice()[header_size..encrypted.header().size as usize]; + Owned::::copy_from_slice(record).into() + } + + #[test] + fn polled_reply_encrypted_records_take_the_flattening_path() { + let encryptor = + EncryptorKind::Aes256Gcm(Aes256GcmEncryptor::new(&[7u8; 32]).expect("valid 32B key")); + let fragments = PollFragments::from_iter([Fragment::whole(encrypted_record(&encryptor))]); + let expected = flattened_reply(fragments.clone(), Some(&encryptor)); + let reply = vectored_reply(fragments, Some(&encryptor)).expect("decrypting reply"); + assert_eq!(reply.fragments().len(), 1); + assert_eq!(reply.into_contiguous().as_slice(), expected.as_slice()); + assert_eq!(polled_count(&expected), 2); + } } diff --git a/core/server_common/src/consensus_message.rs b/core/server_common/src/consensus_message.rs index ec7ff6ba06..c6e428b777 100644 --- a/core/server_common/src/consensus_message.rs +++ b/core/server_common/src/consensus_message.rs @@ -17,6 +17,7 @@ use crate::iobuf::{Frozen, Owned}; use crate::sharding::METADATA_GROUP; +use aligned_vec::{AVec, ConstAlign}; use iggy_binary_protocol::{ Command, CommitHeader, ConsensusError, ConsensusHeader, DoViewChangeHeader, ForwardLogoutHeader, ForwardLogoutResultHeader, ForwardRegisterHeader, @@ -35,6 +36,12 @@ use std::{ pub const MESSAGE_ALIGN: usize = 4096; +/// Fragment list behind a [`ResponseBacking`]. Inline for the single-buffer +/// frame every reply but a poll is, so the per-connection mailboxes and the +/// inter-shard reply lane stay one word wider than a bare [`Frozen`]; a +/// vectored poll reply spills its fragment table to the heap once. +pub type ResponseFragments = SmallVec<[Frozen; 1]>; + pub trait MessageBacking where H: ConsensusHeader, @@ -73,14 +80,68 @@ pub struct RequestBacking { owned: Owned, } +/// An outbound frame as a list of buffers written back to back: the wire bytes +/// are the concatenation of `fragments`. Never empty; the first fragment holds +/// the whole frame header. #[derive(Debug, Clone)] pub struct ResponseBacking { - fragments: SmallVec<[Frozen; 4]>, + fragments: ResponseFragments, } impl RequestBackingKind for RequestBacking {} impl ResponseBackingKind for ResponseBacking {} +impl ResponseBacking { + /// The fragment carrying the frame header. + #[must_use] + pub fn first(&self) -> &Frozen { + self.fragments + .first() + .expect("response backing is never empty") + } + + #[must_use] + pub fn fragments(&self) -> &[Frozen] { + &self.fragments + } + + #[must_use] + pub fn into_fragments(self) -> ResponseFragments { + self.fragments + } + + #[must_use] + pub fn total_len(&self) -> usize { + self.fragments.iter().map(Frozen::len).sum() + } + + /// The frame as one buffer: the single fragment as is, or the fragments + /// copied back to back. For writers whose record layer needs a contiguous + /// payload (WebSocket frames, in-process reply decoding). + #[must_use] + pub fn into_contiguous(self) -> Frozen { + match self.fragments.as_slice() { + [single] => single.clone(), + fragments => { + let mut joined: AVec> = + AVec::with_capacity(MESSAGE_ALIGN, self.total_len()); + for fragment in fragments { + joined.extend_from_slice(fragment); + } + Owned::from(joined).into() + } + } + } +} + +impl From> for ResponseBacking { + fn from(frozen: Frozen) -> Self { + Self { + fragments: smallvec::smallvec![frozen], + } + } +} + impl RequestBacking { fn into_owned(self) -> Owned { self.owned @@ -503,13 +564,13 @@ where } } -impl TryFrom; 4]>> for Message +impl TryFrom for Message where H: ConsensusHeader, { type Error = ConsensusError; - fn try_from(fragments: SmallVec<[Frozen; 4]>) -> Result { + fn try_from(fragments: ResponseFragments) -> Result { let Some(first) = fragments.first() else { return Err(ConsensusError::InvalidCommand { expected: H::COMMAND, @@ -1600,7 +1661,7 @@ mod tests { fn response_backing_single_fragment_roundtrip() { let owned = header_bytes(Command::Reply, 256); let frozen: Frozen = owned.into(); - let fragments: smallvec::SmallVec<[Frozen; 4]> = smallvec![frozen]; + let fragments: ResponseFragments = smallvec![frozen]; let msg = Message::::try_from(fragments).expect("valid"); assert_eq!(msg.header().command, Command::Reply); assert_eq!(msg.fragments().len(), 1); @@ -1608,7 +1669,7 @@ mod tests { #[test] fn response_backing_empty_fragments_returns_err() { - let fragments: smallvec::SmallVec<[Frozen; 4]> = smallvec![]; + let fragments: ResponseFragments = smallvec![]; let result = Message::::try_from(fragments); assert!(matches!(result, Err(ConsensusError::InvalidCommand { .. }))); } @@ -1617,7 +1678,7 @@ mod tests { fn response_backing_first_fragment_too_short_returns_err() { let owned = Owned::::zeroed(100); let frozen: Frozen = owned.into(); - let fragments: smallvec::SmallVec<[Frozen; 4]> = smallvec![frozen]; + let fragments: ResponseFragments = smallvec![frozen]; let result = Message::::try_from(fragments); assert!(matches!(result, Err(ConsensusError::InvalidCommand { .. }))); } @@ -1628,7 +1689,7 @@ mod tests { // the header; the floor must reject before any consumer slices a body. let owned = header_bytes(Command::Reply, size_of::() as u32 - 1); let frozen: Frozen = owned.into(); - let fragments: smallvec::SmallVec<[Frozen; 4]> = smallvec![frozen]; + let fragments: ResponseFragments = smallvec![frozen]; let result = Message::::try_from(fragments); assert!(matches!(result, Err(ConsensusError::InvalidCommand { .. }))); } diff --git a/core/server_common/src/iobuf.rs b/core/server_common/src/iobuf.rs index 3c793ec98f..2337a81651 100644 --- a/core/server_common/src/iobuf.rs +++ b/core/server_common/src/iobuf.rs @@ -23,6 +23,12 @@ use std::ptr::NonNull; use std::slice; use std::sync::atomic::{AtomicUsize, Ordering, fence}; +/// Linux `IOV_MAX`: the most iovecs one vectored IO syscall accepts. +/// Vectored writers must chunk their buffer lists at this many entries or +/// the syscall fails: `EINVAL` from `writev(2)`/`pwritev(2)` (segment file +/// writes), `EMSGSIZE` from `sendmsg(2)` (socket writes). +pub const IOV_MAX: usize = 1024; + #[derive(Debug, Clone)] pub struct Owned { inner: AVec>, @@ -115,6 +121,10 @@ impl Owned { &mut self.inner } + pub fn extend_from_slice(&mut self, bytes: &[u8]) { + self.inner.extend_from_slice(bytes); + } + pub fn split_at(self, split_at: usize) -> (Prefix, Frozen) { assert!(split_at <= self.inner.len()); @@ -276,6 +286,13 @@ impl Frozen { self.inner.as_slice() } + /// Whether `self` and `other` refcount the same backing allocation, + /// regardless of their sliced windows. Any hit keeps the whole + /// allocation alive, not just the window. + pub fn shares_allocation(&self, other: &Self) -> bool { + self.inner.ctrlb == other.inner.ctrlb + } + pub fn split_at(self, split_at: usize) -> (Prefix, Frozen) { assert!(split_at <= self.inner.len); @@ -648,6 +665,16 @@ mod tests { assert_eq!(prefix.as_slice(), b"abc"); } + #[test] + fn owned_extend_from_slice_fills_reserved_capacity_in_place() { + let mut o: Owned = Owned::with_capacity(6); + let capacity = o.buf_capacity(); + o.extend_from_slice(b"abc"); + o.extend_from_slice(b"def"); + assert_eq!(o.as_slice(), b"abcdef"); + assert_eq!(o.buf_capacity(), capacity); + } + // Owned::split_at: shared control block / disjoint views #[test] diff --git a/core/server_common/src/lib.rs b/core/server_common/src/lib.rs index 16070aeed9..9c7c1a55bc 100644 --- a/core/server_common/src/lib.rs +++ b/core/server_common/src/lib.rs @@ -38,6 +38,7 @@ pub use certificates::generate_self_signed_certificate; pub use consensus_message::{ ConsensusMessage, FragmentedBacking, MESSAGE_ALIGN, Message, MessageBacking, MessageBag, MutableBacking, RequestBacking, RequestBackingKind, ResponseBacking, ResponseBackingKind, + ResponseFragments, }; pub use executor::create_shard_executor; pub use memory_pool::{MEMORY_POOL, MemoryPool, MemoryPoolSettings, memory_pool}; diff --git a/core/shard/src/lib.rs b/core/shard/src/lib.rs index a2ffd20a30..8b56a415a4 100644 --- a/core/shard/src/lib.rs +++ b/core/shard/src/lib.rs @@ -52,11 +52,11 @@ use iggy_common::variadic; use iggy_common::{IggyError, IggyExpiry, IggyTimestamp}; use journal::superblock::{PingPongSuperblock, SuperblockStore}; use journal::{Journal, JournalHandle}; -use message_bus::MessageBus; use message_bus::client_listener::RequestHandler; use message_bus::fd_transfer::DupedFd; use message_bus::installer::conn_info::{ClientConnMeta, ClientTransportKind}; use message_bus::replica::listener::MessageHandler; +use message_bus::{BusMessage, MessageBus}; use metadata::IggyMetadata; use metadata::impls::metadata::StreamsFrontend; use metadata::stm::StateMachine; @@ -681,10 +681,7 @@ pub enum LifecycleFrame { }, /// A shard that doesn't hold the client's TCP connection forwards a /// client send to the owning shard (top 16 bits of `client_id`). - ForwardClientSend { - client_id: u128, - msg: Frozen, - }, + ForwardClientSend { client_id: u128, msg: BusMessage }, /// A peer shard hands a metadata consensus submit (login/logout) to /// shard 0, the metadata consensus owner. The committed op returns over /// the `reply` sender carried in [`MetadataSubmit`]. Always addressed to @@ -3391,7 +3388,7 @@ where ); let frame = ShardFrame::lifecycle(LifecycleFrame::ForwardClientSend { client_id: request_header.client, - msg: reply.into_generic().into_frozen(), + msg: reply.into_generic().into_frozen().into(), }); let Some(sender) = self.senders.get(self.id as usize) else { return false; diff --git a/core/simulator/src/bus.rs b/core/simulator/src/bus.rs index 39ff8feb74..daa65c8dc1 100644 --- a/core/simulator/src/bus.rs +++ b/core/simulator/src/bus.rs @@ -24,7 +24,8 @@ use message_bus::fd_transfer::DupedFd; use message_bus::installer::conn_info::ClientConnMeta; use message_bus::replica::listener::MessageHandler; use message_bus::{ - ClientConnectionLostFn, ConnectionInstaller, MessageBus, ReplicaHandshakeDoneFn, SendError, + BusMessage, ClientConnectionLostFn, ConnectionInstaller, MessageBus, ReplicaHandshakeDoneFn, + SendError, }; use server_common::{ MESSAGE_ALIGN, Message, @@ -189,7 +190,7 @@ impl MessageBus for SimOutbox { async fn send_to_client( &self, client_id: u128, - data: Frozen, + data: impl Into, ) -> Result<(), SendError> { if !self.clients.borrow().contains(&client_id) { return Err(SendError::ClientNotFound(client_id)); @@ -199,7 +200,7 @@ impl MessageBus for SimOutbox { from_replica: Some(self.self_id), to_replica: None, to_client: Some(client_id), - payload: EnvelopePayload::Client(frozen_to_message(&data)), + payload: EnvelopePayload::Client(frozen_to_message(&data.into().into_contiguous())), }); Ok(()) @@ -264,7 +265,7 @@ impl MessageBus for SharedSimOutbox { async fn send_to_client( &self, client_id: u128, - data: Frozen, + data: impl Into, ) -> Result<(), SendError> { self.0.send_to_client(client_id, data).await } From 2783907c7a90cd095b9d88b62646d906fedc1e2b Mon Sep 17 00:00:00 2001 From: Richard Cocks <50965970+richardcocks@users.noreply.github.com> Date: Tue, 1 Sep 2026 20:13:45 +0100 Subject: [PATCH 033/182] fix(csharp): set default socket buffer size to OS default (#4013) Closes #4010 --------- Co-authored-by: spetz --- foreign/csharp/Benchmarks/Program.cs | 10 ------ .../Configuration/IggyClientConfigurator.cs | 12 ++++--- .../Iggy_SDK/Consumers/IggyConsumerBuilder.cs | 36 +++++++++++++------ .../Iggy_SDK/Consumers/IggyConsumerConfig.cs | 12 ++++--- .../Iggy_SDK/Factory/IggyClientFactory.cs | 15 +++++++- .../Implementations/TcpMessageStream.cs | 11 ++++-- .../Publishers/IggyPublisherBuilder.cs | 36 +++++++++++++------ .../Publishers/IggyPublisherConfig.cs | 14 ++++---- .../ClientTests/IggyClientFactoryTests.cs | 20 +++++++++++ .../ConsumerTests/IggyConsumerBuilderTests.cs | 27 ++++++++++++++ .../IggyPublisherBuilderTests.cs | 27 ++++++++++++++ foreign/csharp/README.md | 6 ++-- 12 files changed, 176 insertions(+), 50 deletions(-) diff --git a/foreign/csharp/Benchmarks/Program.cs b/foreign/csharp/Benchmarks/Program.cs index ea796fa3de..b3e7a42ade 100644 --- a/foreign/csharp/Benchmarks/Program.cs +++ b/foreign/csharp/Benchmarks/Program.cs @@ -44,16 +44,6 @@ BaseAddress = "127.0.0.1:8090", Protocol = Protocol.Tcp, LoggerFactory = loggerFactory, -#if OS_LINUX - ReceiveBufferSize = Int32.MaxValue, - SendBufferSize = Int32.MaxValue, -#elif OS_WINDOWS - ReceiveBufferSize = int.MaxValue, - SendBufferSize = int.MaxValue, -#elif OS_MAC - ReceiveBufferSize = 7280 * 1024, - SendBufferSize = 7280 * 1024, -#endif }); await bus.LoginUserAsync("iggy", "iggy"); diff --git a/foreign/csharp/Iggy_SDK/Configuration/IggyClientConfigurator.cs b/foreign/csharp/Iggy_SDK/Configuration/IggyClientConfigurator.cs index 48e50d3b7e..d959886c07 100644 --- a/foreign/csharp/Iggy_SDK/Configuration/IggyClientConfigurator.cs +++ b/foreign/csharp/Iggy_SDK/Configuration/IggyClientConfigurator.cs @@ -44,14 +44,18 @@ public sealed class IggyClientConfigurator public int MaxResponseFrameSize { get; set; } = 64 * 1024 * 1024; /// - /// The size of the receive buffer in bytes. Default is 4096. + /// The size of the receive buffer in bytes. When null, the size is not set on the socket, so the + /// operating system default is used. On Linux, this keeps TCP auto-tuning enabled. The initial size + /// is the middle value in /proc/sys/net/ipv4/tcp_rmem. /// - public int ReceiveBufferSize { get; set; } = 4096; + public int? ReceiveBufferSize { get; set; } = null; /// - /// The size of the send buffer in bytes. Default is 4096. + /// The size of the send buffer in bytes. When null, the size is not set on the socket, so the + /// operating system default is used. On Linux, this keeps TCP auto-tuning enabled. The initial size + /// is the middle value in /proc/sys/net/ipv4/tcp_wmem. /// - public int SendBufferSize { get; set; } = 4096; + public int? SendBufferSize { get; set; } = null; /// /// Interval between the pings the client sends on its own while diff --git a/foreign/csharp/Iggy_SDK/Consumers/IggyConsumerBuilder.cs b/foreign/csharp/Iggy_SDK/Consumers/IggyConsumerBuilder.cs index 400cc72342..985cd10bfe 100644 --- a/foreign/csharp/Iggy_SDK/Consumers/IggyConsumerBuilder.cs +++ b/foreign/csharp/Iggy_SDK/Consumers/IggyConsumerBuilder.cs @@ -90,12 +90,20 @@ public static IggyConsumerBuilder Create(IIggyClient iggyClient, Identifier stre /// The address of the server to connect to. /// The login username for authentication. /// The password for authentication. - /// The size of the receive buffer. - /// The size of the send buffer. + /// + /// The size of the receive buffer in bytes. When null, the size is not set on the socket, so the + /// operating system default is used. On Linux, this keeps TCP auto-tuning enabled. The initial size + /// is the middle value in /proc/sys/net/ipv4/tcp_rmem. + /// + /// + /// The size of the send buffer in bytes. When null, the size is not set on the socket, so the + /// operating system default is used. On Linux, this keeps TCP auto-tuning enabled. The initial size + /// is the middle value in /proc/sys/net/ipv4/tcp_wmem. + /// /// Reconnection settings for the client. /// The current instance of to allow method chaining. public IggyConsumerBuilder WithConnection(Protocol protocol, string address, string login, string password, - int receiveBufferSize = 4096, int sendBufferSize = 4096, ReconnectionSettings? reconnectionSettings = null) + int? receiveBufferSize = null, int? sendBufferSize = null, ReconnectionSettings? reconnectionSettings = null) { Config.Protocol = protocol; Config.Address = address; @@ -114,12 +122,20 @@ public IggyConsumerBuilder WithConnection(Protocol protocol, string address, str /// The protocol to use for the connection (e.g., TCP, UDP). /// The address of the server to connect to. /// The personal access token to authenticate with. - /// The size of the receive buffer. - /// The size of the send buffer. + /// + /// The size of the receive buffer in bytes. When null, the size is not set on the socket, so the + /// operating system default is used. On Linux, this keeps TCP auto-tuning enabled. The initial size + /// is the middle value in /proc/sys/net/ipv4/tcp_rmem. + /// + /// + /// The size of the send buffer in bytes. When null, the size is not set on the socket, so the + /// operating system default is used. On Linux, this keeps TCP auto-tuning enabled. The initial size + /// is the middle value in /proc/sys/net/ipv4/tcp_wmem. + /// /// Reconnection settings for the client. /// The current instance of to allow method chaining. public IggyConsumerBuilder WithConnection(Protocol protocol, string address, string personalAccessToken, - int receiveBufferSize = 4096, int sendBufferSize = 4096, ReconnectionSettings? reconnectionSettings = null) + int? receiveBufferSize = null, int? sendBufferSize = null, ReconnectionSettings? reconnectionSettings = null) { Config.Protocol = protocol; Config.Address = address; @@ -347,14 +363,14 @@ protected virtual void Validate() "AutoCommitMode.Auto with a message encryptor risks silent message loss: the offset is committed before decryption. Use AutoCommitMode.AfterReceive or AutoCommitMode.Disabled."); } - if (Config.ReceiveBufferSize <= 0) + if (Config.ReceiveBufferSize is <= 0) { - throw new InvalidOperationException("ReceiveBufferSize must be greater than 0."); + throw new InvalidOperationException("ReceiveBufferSize must be greater than 0 when set."); } - if (Config.SendBufferSize <= 0) + if (Config.SendBufferSize is <= 0) { - throw new InvalidOperationException("SendBufferSize must be greater than 0."); + throw new InvalidOperationException("SendBufferSize must be greater than 0 when set."); } if (Config.BatchSize == 0) diff --git a/foreign/csharp/Iggy_SDK/Consumers/IggyConsumerConfig.cs b/foreign/csharp/Iggy_SDK/Consumers/IggyConsumerConfig.cs index d814301101..50edf2efcc 100644 --- a/foreign/csharp/Iggy_SDK/Consumers/IggyConsumerConfig.cs +++ b/foreign/csharp/Iggy_SDK/Consumers/IggyConsumerConfig.cs @@ -76,14 +76,18 @@ public class IggyConsumerConfig public string PersonalAccessToken { get; set; } = string.Empty; /// - /// The size of the receive buffer in bytes. Default is 4096. + /// The size of the receive buffer in bytes. When null, the size is not set on the socket, so the + /// operating system default is used. On Linux, this keeps TCP auto-tuning enabled. The initial size + /// is the middle value in /proc/sys/net/ipv4/tcp_rmem. /// - public int ReceiveBufferSize { get; set; } = 4096; + public int? ReceiveBufferSize { get; set; } = null; /// - /// The size of the send buffer in bytes. Default is 4096. + /// The size of the send buffer in bytes. When null, the size is not set on the socket, so the + /// operating system default is used. On Linux, this keeps TCP auto-tuning enabled. The initial size + /// is the middle value in /proc/sys/net/ipv4/tcp_wmem. /// - public int SendBufferSize { get; set; } = 4096; + public int? SendBufferSize { get; set; } = null; /// /// The identifier of the stream to consume from diff --git a/foreign/csharp/Iggy_SDK/Factory/IggyClientFactory.cs b/foreign/csharp/Iggy_SDK/Factory/IggyClientFactory.cs index 515a06dd88..091565ace4 100644 --- a/foreign/csharp/Iggy_SDK/Factory/IggyClientFactory.cs +++ b/foreign/csharp/Iggy_SDK/Factory/IggyClientFactory.cs @@ -46,7 +46,8 @@ public static class IggyClientFactory /// supported. /// /// - /// Thrown when is below the 256-byte header. + /// Thrown when is below the 256-byte header or a + /// configured socket buffer size is not positive. /// public static IIggyClient CreateClient(IggyClientConfigurator options) { @@ -68,6 +69,18 @@ private static void Validate(IggyClientConfigurator options) $"MaxResponseFrameSize must be at least {VsrHeader.HEADER_SIZE} bytes."); } + if (options.ReceiveBufferSize is <= 0) + { + throw new ArgumentOutOfRangeException(nameof(IggyClientConfigurator.ReceiveBufferSize), + options.ReceiveBufferSize, "ReceiveBufferSize must be greater than 0 when set."); + } + + if (options.SendBufferSize is <= 0) + { + throw new ArgumentOutOfRangeException(nameof(IggyClientConfigurator.SendBufferSize), + options.SendBufferSize, "SendBufferSize must be greater than 0 when set."); + } + // The bounds PeriodicTimer accepts; anything outside them would fault the heartbeat task at start // instead of failing the caller here. if (options.HeartbeatInterval < TimeSpan.FromMilliseconds(1) || diff --git a/foreign/csharp/Iggy_SDK/IggyClient/Implementations/TcpMessageStream.cs b/foreign/csharp/Iggy_SDK/IggyClient/Implementations/TcpMessageStream.cs index 928d5e9ff8..c8e0163d66 100644 --- a/foreign/csharp/Iggy_SDK/IggyClient/Implementations/TcpMessageStream.cs +++ b/foreign/csharp/Iggy_SDK/IggyClient/Implementations/TcpMessageStream.cs @@ -1066,8 +1066,15 @@ private async Task TryEstablishConnectionAsync(bool autoLogin, bool settleOnLead try { socket = new Socket(ServerAddress.AddressFamilyOf(host), SocketType.Stream, ProtocolType.Tcp); - socket.SendBufferSize = _configuration.SendBufferSize; - socket.ReceiveBufferSize = _configuration.ReceiveBufferSize; + if (_configuration.SendBufferSize.HasValue) + { + socket.SendBufferSize = _configuration.SendBufferSize.Value; + } + + if (_configuration.ReceiveBufferSize.HasValue) + { + socket.ReceiveBufferSize = _configuration.ReceiveBufferSize.Value; + } // The protocol is request/reply, so a write is always the last one before // the client blocks on the answer and Nagle has nothing to coalesce it with - it only delays the diff --git a/foreign/csharp/Iggy_SDK/Publishers/IggyPublisherBuilder.cs b/foreign/csharp/Iggy_SDK/Publishers/IggyPublisherBuilder.cs index 81ec6c2972..2dc98a6382 100644 --- a/foreign/csharp/Iggy_SDK/Publishers/IggyPublisherBuilder.cs +++ b/foreign/csharp/Iggy_SDK/Publishers/IggyPublisherBuilder.cs @@ -109,12 +109,20 @@ public static IggyPublisherBuilder Create(IggyPublisherConfig config) /// The server address to connect to (format depends on protocol). /// The login username for authentication. /// The password for authentication. - /// The size of the receive buffer in bytes. Default is 4096. - /// The size of the send buffer in bytes. Default is 4096. + /// + /// The size of the receive buffer in bytes. When null, the size is not set on the socket, so the + /// operating system default is used. On Linux, this keeps TCP auto-tuning enabled. The initial size + /// is the middle value in /proc/sys/net/ipv4/tcp_rmem. + /// + /// + /// The size of the send buffer in bytes. When null, the size is not set on the socket, so the + /// operating system default is used. On Linux, this keeps TCP auto-tuning enabled. The initial size + /// is the middle value in /proc/sys/net/ipv4/tcp_wmem. + /// /// Reconnection settings for the client. /// The builder instance for method chaining. public IggyPublisherBuilder WithConnection(Protocol protocol, string address, string login, string password, - int receiveBufferSize = 4096, int sendBufferSize = 4096, ReconnectionSettings? reconnectionSettings = null) + int? receiveBufferSize = null, int? sendBufferSize = null, ReconnectionSettings? reconnectionSettings = null) { Config.Protocol = protocol; Config.Address = address; @@ -133,12 +141,20 @@ public IggyPublisherBuilder WithConnection(Protocol protocol, string address, st /// The protocol to use for the connection (e.g., TCP, UDP). /// The address of the server to connect to. /// The personal access token to authenticate with. - /// The size of the receive buffer. - /// The size of the send buffer. + /// + /// The size of the receive buffer in bytes. When null, the size is not set on the socket, so the + /// operating system default is used. On Linux, this keeps TCP auto-tuning enabled. The initial size + /// is the middle value in /proc/sys/net/ipv4/tcp_rmem. + /// + /// + /// The size of the send buffer in bytes. When null, the size is not set on the socket, so the + /// operating system default is used. On Linux, this keeps TCP auto-tuning enabled. The initial size + /// is the middle value in /proc/sys/net/ipv4/tcp_wmem. + /// /// Reconnection settings for the client. /// The current instance of to allow method chaining. public IggyPublisherBuilder WithConnection(Protocol protocol, string address, string personalAccessToken, - int receiveBufferSize = 4096, int sendBufferSize = 4096, ReconnectionSettings? reconnectionSettings = null) + int? receiveBufferSize = null, int? sendBufferSize = null, ReconnectionSettings? reconnectionSettings = null) { Config.Protocol = protocol; Config.Address = address; @@ -422,14 +438,14 @@ protected virtual void Validate() } } - if (Config.ReceiveBufferSize <= 0) + if (Config.ReceiveBufferSize is <= 0) { - throw new InvalidOperationException("ReceiveBufferSize must be greater than 0."); + throw new InvalidOperationException("ReceiveBufferSize must be greater than 0 when set."); } - if (Config.SendBufferSize <= 0) + if (Config.SendBufferSize is <= 0) { - throw new InvalidOperationException("SendBufferSize must be greater than 0."); + throw new InvalidOperationException("SendBufferSize must be greater than 0 when set."); } if (Config.EnableBackgroundSending) diff --git a/foreign/csharp/Iggy_SDK/Publishers/IggyPublisherConfig.cs b/foreign/csharp/Iggy_SDK/Publishers/IggyPublisherConfig.cs index 36604663bc..8f9335a6ef 100644 --- a/foreign/csharp/Iggy_SDK/Publishers/IggyPublisherConfig.cs +++ b/foreign/csharp/Iggy_SDK/Publishers/IggyPublisherConfig.cs @@ -93,16 +93,18 @@ public class IggyPublisherConfig public Identifier TopicId { get; set; } /// - /// Gets or sets the size of the receive buffer in bytes. - /// Default is 4096 bytes (4 KB). + /// The size of the receive buffer in bytes. When null, the size is not set on the socket, so the + /// operating system default is used. On Linux, this keeps TCP auto-tuning enabled. The initial size + /// is the middle value in /proc/sys/net/ipv4/tcp_rmem. /// - public int ReceiveBufferSize { get; set; } = 4096; + public int? ReceiveBufferSize { get; set; } = null; /// - /// Gets or sets the size of the send buffer in bytes. - /// Default is 4096 bytes (4 KB). + /// The size of the send buffer in bytes. When null, the size is not set on the socket, so the + /// operating system default is used. On Linux, this keeps TCP auto-tuning enabled. The initial size + /// is the middle value in /proc/sys/net/ipv4/tcp_wmem. /// - public int SendBufferSize { get; set; } = 4096; + public int? SendBufferSize { get; set; } = null; /// /// Gets or sets the partitioning strategy for messages. diff --git a/foreign/csharp/Iggy_SDK_Tests/ClientTests/IggyClientFactoryTests.cs b/foreign/csharp/Iggy_SDK_Tests/ClientTests/IggyClientFactoryTests.cs index 68f32d2649..9028445ef0 100644 --- a/foreign/csharp/Iggy_SDK_Tests/ClientTests/IggyClientFactoryTests.cs +++ b/foreign/csharp/Iggy_SDK_Tests/ClientTests/IggyClientFactoryTests.cs @@ -33,11 +33,31 @@ public void CreateClient_CreatesTcpClient() }; Assert.Equal(64 * 1024 * 1024, options.MaxResponseFrameSize); + Assert.Null(options.ReceiveBufferSize); + Assert.Null(options.SendBufferSize); using var client = IggyClientFactory.CreateClient(options) as IDisposable; Assert.NotNull(client); } + [Theory] + [InlineData(0, null)] + [InlineData(-1, null)] + [InlineData(null, 0)] + [InlineData(null, -1)] + public void CreateClient_RejectsNonPositiveSocketBufferSize(int? receiveBufferSize, int? sendBufferSize) + { + var options = new IggyClientConfigurator + { + BaseAddress = "127.0.0.1:8090", + Protocol = Protocol.Tcp, + ReceiveBufferSize = receiveBufferSize, + SendBufferSize = sendBufferSize + }; + + Assert.Throws(() => IggyClientFactory.CreateClient(options)); + } + [Fact] public void CreateClient_RejectsMaxResponseFrameSizeBelowHeader() { diff --git a/foreign/csharp/Iggy_SDK_Tests/ConsumerTests/IggyConsumerBuilderTests.cs b/foreign/csharp/Iggy_SDK_Tests/ConsumerTests/IggyConsumerBuilderTests.cs index f7d01c11fe..6070a00024 100644 --- a/foreign/csharp/Iggy_SDK_Tests/ConsumerTests/IggyConsumerBuilderTests.cs +++ b/foreign/csharp/Iggy_SDK_Tests/ConsumerTests/IggyConsumerBuilderTests.cs @@ -30,6 +30,33 @@ public class IggyConsumerBuilderTests private static readonly Identifier StreamId = Identifier.Numeric(1); private static readonly Identifier TopicId = Identifier.Numeric(1); + [Fact] + public void Build_WithDefaultSocketBufferSizes_CreatesTheClient() + { + var builder = IggyConsumerBuilder + .Create(StreamId, TopicId, Consumer.New(1)) + .WithConnection(Protocol.Tcp, "127.0.0.1:8090", "user", "pass"); + + Assert.Null(builder.Config.ReceiveBufferSize); + Assert.Null(builder.Config.SendBufferSize); + Assert.NotNull(builder.Build()); + } + + [Theory] + [InlineData(0, null)] + [InlineData(-1, null)] + [InlineData(null, 0)] + [InlineData(null, -1)] + public void Build_WithNonPositiveSocketBufferSize_Throws(int? receiveBufferSize, int? sendBufferSize) + { + var builder = IggyConsumerBuilder + .Create(StreamId, TopicId, Consumer.New(1)) + .WithConnection(Protocol.Tcp, "127.0.0.1:8090", "user", "pass", + receiveBufferSize: receiveBufferSize, sendBufferSize: sendBufferSize); + + Assert.Throws(() => builder.Build()); + } + [Fact] public void Build_WithEncryptorAndAutoCommit_Throws() { diff --git a/foreign/csharp/Iggy_SDK_Tests/PublisherTests/IggyPublisherBuilderTests.cs b/foreign/csharp/Iggy_SDK_Tests/PublisherTests/IggyPublisherBuilderTests.cs index 1172a752dd..fdd4168c1f 100644 --- a/foreign/csharp/Iggy_SDK_Tests/PublisherTests/IggyPublisherBuilderTests.cs +++ b/foreign/csharp/Iggy_SDK_Tests/PublisherTests/IggyPublisherBuilderTests.cs @@ -27,6 +27,33 @@ public class IggyPublisherBuilderTests private static readonly Identifier StreamId = Identifier.Numeric(1); private static readonly Identifier TopicId = Identifier.Numeric(1); + [Fact] + public void Build_WithDefaultSocketBufferSizes_CreatesTheClient() + { + var builder = IggyPublisherBuilder + .Create(StreamId, TopicId) + .WithConnection(Protocol.Tcp, "127.0.0.1:8090", "user", "pass"); + + Assert.Null(builder.Config.ReceiveBufferSize); + Assert.Null(builder.Config.SendBufferSize); + Assert.NotNull(builder.Build()); + } + + [Theory] + [InlineData(0, null)] + [InlineData(-1, null)] + [InlineData(null, 0)] + [InlineData(null, -1)] + public void Build_WithNonPositiveSocketBufferSize_Throws(int? receiveBufferSize, int? sendBufferSize) + { + var builder = IggyPublisherBuilder + .Create(StreamId, TopicId) + .WithConnection(Protocol.Tcp, "127.0.0.1:8090", "user", "pass", + receiveBufferSize: receiveBufferSize, sendBufferSize: sendBufferSize); + + Assert.Throws(() => builder.Build()); + } + [Fact] public void TypedBuild_OverTcp_CreatesTheClient() { diff --git a/foreign/csharp/README.md b/foreign/csharp/README.md index 3fed133b1b..a067df8c4b 100644 --- a/foreign/csharp/README.md +++ b/foreign/csharp/README.md @@ -80,9 +80,9 @@ var client = IggyClientFactory.CreateClient(new IggyClientConfigurator BaseAddress = "127.0.0.1:8090", Protocol = Protocol.Tcp, - // Buffer sizes (optional, default: 4096) - ReceiveBufferSize = 4096, - SendBufferSize = 4096, + // Socket buffer sizes in bytes (optional, null = OS default) + ReceiveBufferSize = null, + SendBufferSize = null, // TLS/SSL configuration TlsSettings = new TlsSettings From f45f9dda9bcfb4b74723650fafd8933cc7ed19c9 Mon Sep 17 00:00:00 2001 From: Hubert Gruszecki Date: Tue, 1 Sep 2026 21:45:55 +0200 Subject: [PATCH 034/182] refactor(server): split the dispatch spine into plane-named leaves (#4026) --- core/server/src/auth.rs | 457 -- core/server/src/bootstrap.rs | 16 +- core/server/src/dispatch.rs | 5275 ----------------- core/server/src/dispatch/authz.rs | 14 +- .../login_error.rs} | 25 +- core/server/src/dispatch/mod.rs | 1413 +++++ core/server/src/dispatch/partition.rs | 1497 +++++ core/server/src/dispatch/reads.rs | 491 ++ core/server/src/dispatch/session_ops.rs | 2096 +++++++ core/server/src/dispatch/submit.rs | 194 + core/server/src/dispatch/test_support.rs | 279 + core/server/src/http/extractor.rs | 2 +- core/server/src/http/handlers.rs | 8 +- core/server/src/http/reply.rs | 2 +- core/server/src/http/state.rs | 2 +- core/server/src/http/submit.rs | 7 +- core/server/src/lib.rs | 2 - core/server/src/partition_reconciler.rs | 2 +- core/server/tests/module_graph.rs | 45 +- 19 files changed, 6043 insertions(+), 5784 deletions(-) delete mode 100644 core/server/src/auth.rs delete mode 100644 core/server/src/dispatch.rs rename core/server/src/{login_register.rs => dispatch/login_error.rs} (75%) create mode 100644 core/server/src/dispatch/mod.rs create mode 100644 core/server/src/dispatch/partition.rs create mode 100644 core/server/src/dispatch/reads.rs create mode 100644 core/server/src/dispatch/session_ops.rs create mode 100644 core/server/src/dispatch/submit.rs create mode 100644 core/server/src/dispatch/test_support.rs diff --git a/core/server/src/auth.rs b/core/server/src/auth.rs deleted file mode 100644 index 1851281c6a..0000000000 --- a/core/server/src/auth.rs +++ /dev/null @@ -1,457 +0,0 @@ -// Licensed to the Apache Software Foundation (ASF) under one -// or more contributor license agreements. See the NOTICE file -// distributed with this work for additional information -// regarding copyright ownership. The ASF licenses this file -// to you under the Apache License, Version 2.0 (the -// "License"); you may not use this file except in compliance -// with the License. You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, -// software distributed under the License is distributed on an -// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY -// KIND, either express or implied. See the License for the -// specific language governing permissions and limitations -// under the License. - -//! Local credential verification and login/register completion. -//! -//! Verifies password + PAT credentials locally, then runs the consensus -//! `Register` proposal on the metadata owner; terminal failures are -//! surfaced as typed `Eviction` frames, transient ones as result-framed -//! replay hints. - -use crate::dispatch::{send_login_eviction, submit_register_on_owner}; -use crate::login_register::LoginRegisterError; -use crate::responses::{build_login_register_reply, current_metadata_commit}; -use crate::session_manager::{ClientSdkInfo, SessionManager}; -use crate::shell::{ShellBus, ShellShard}; -use consensus::{MetadataHandle, build_result_rejection_reply}; -use iggy_binary_protocol::PrepareHeader; -use iggy_binary_protocol::{ClientVersionInfo, EvictionReason, RoutedRequestHeader}; -use iggy_common::defaults::{ - MAX_PASSWORD_LENGTH, MAX_USERNAME_LENGTH, MIN_PASSWORD_LENGTH, MIN_USERNAME_LENGTH, -}; -use iggy_common::{IggyError, IggyTimestamp, PersonalAccessToken, UserStatus}; -use journal::superblock::SuperblockStore; -use journal::{Journal, JournalHandle}; -use metadata::MetadataSubmitError; -use metadata::impls::metadata::StreamsFrontend; -use server_common::Message; -use server_common::crypto; -use std::cell::RefCell; -use std::rc::Rc; -use std::sync::LazyLock; -use tracing::warn; - -/// A well-formed Argon2 hash to verify against on the unknown-user login -/// branch, so a missing username costs the same single `verify_password` a real -/// user's wrong-password branch costs. Closes the username-existence timing -/// oracle without changing the returned error. On the unknown-username branch -/// the verify result is discarded, so even a request presenting the exact dummy -/// plaintext cannot authenticate; the literal only needs to be a fixed input -/// hashed by the same Argon2 hasher real users use, so the dummy verify runs an -/// identical Argon2 KDF. -static DUMMY_PASSWORD_HASH: LazyLock = - LazyLock::new(|| crypto::hash_password("http-login-timing-guard")); - -/// Pay the one-time Argon2 cost of [`DUMMY_PASSWORD_HASH`] at boot instead of -/// inside the first unknown-username login request. -pub fn warm_dummy_password_hash() { - LazyLock::force(&DUMMY_PASSWORD_HASH); -} - -pub fn verify_login_credentials( - shard: &Rc>, - username: &str, - password: &str, -) -> Result -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - // Same bounds the legacy server enforces before any lookup or hashing; - // also keeps arbitrary-length input out of the password hash. Collapsed - // to InvalidCredentials on purpose (legacy: InvalidUsername / - // InvalidPassword): don't leak which field failed. - if !(MIN_USERNAME_LENGTH..=MAX_USERNAME_LENGTH).contains(&username.len()) - || !(MIN_PASSWORD_LENGTH..=MAX_PASSWORD_LENGTH).contains(&password.len()) - { - return Err(LoginRegisterError::InvalidCredentials); - } - shard.plane.metadata().mux_stm.users().read(|users| { - let user = users - .index - .get(username) - .copied() - .and_then(|user_id| users.items.get(user_id as usize)); - let Some(user) = user else { - // Constant-cost path: verify against a dummy hash so a missing - // username is indistinguishable by response timing from a wrong - // password (both return InvalidCredentials). - let _ = crypto::verify_password(password, DUMMY_PASSWORD_HASH.as_str()); - return Err(LoginRegisterError::InvalidCredentials); - }; - // Verify before the status check and collapse inactive to - // InvalidCredentials: an inactive account must answer exactly like a - // wrong password (same error, same Argon2 cost), or login could probe - // which accounts exist but are disabled. - if !crypto::verify_password(password, user.password_hash.as_ref()) - || user.status != UserStatus::Active - { - return Err(LoginRegisterError::InvalidCredentials); - } - Ok(user.id) - }) -} - -pub fn verify_pat_credentials( - shard: &Rc>, - token: &str, -) -> Result -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - verify_pat_credentials_with_expiry(shard, token).map(|(user_id, _)| user_id) -} - -/// Like [`verify_pat_credentials`] but also surfaces the token's expiry (unix -/// seconds, `u64::MAX` when the PAT never expires). The HTTP extractor keys a -/// per-token VSR session table on this expiry for lazy eviction; the wire and -/// login paths only need the user id and go through [`verify_pat_credentials`]. -pub fn verify_pat_credentials_with_expiry( - shard: &Rc>, - token: &str, -) -> Result<(u32, u64), LoginRegisterError> -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let token_hash = PersonalAccessToken::hash_token(token); - // PAT expiry gates the login accept/reject, and that outcome folds into - // the reply, so read the environment-injected bus clock (seed-derived - // under the simulator), not the wall clock, or a replayed login diverges. - // The two sibling wall-clock reads in `dispatch` stay direct because both - // are off the reply path (diagnostic-only). The bus seam exists on every - // shard, so this holds even when login lands on an entry shard that does - // not own the metadata consensus. - let now = IggyTimestamp::from(shard.bus.realtime_micros()); - shard.plane.metadata().mux_stm.users().read(|users| { - let Some((user_id, token_name)) = - users.personal_access_token_index.get(token_hash.as_str()) - else { - return Err(LoginRegisterError::InvalidToken); - }; - let Some(pat) = users - .personal_access_tokens - .get(user_id) - .and_then(|tokens| tokens.get(token_name)) - else { - return Err(LoginRegisterError::InvalidToken); - }; - if pat.is_expired(now) { - return Err(LoginRegisterError::InvalidToken); - } - let Some(user) = users.items.get(*user_id as usize) else { - return Err(LoginRegisterError::InvalidToken); - }; - if user.status != UserStatus::Active { - return Err(LoginRegisterError::UserInactive); - } - // `expiry_at == None` is a never-expiring PAT; map it to `u64::MAX` so - // the HTTP session table never expiry-evicts its entry. - let expiry = pat - .expiry_at - .map_or(u64::MAX, |expiry_at| expiry_at.to_secs()); - Ok((user.id, expiry)) - }) -} - -#[allow(clippy::future_not_send)] -pub async fn complete_login_register( - shard: &Rc>, - sessions: &Rc>, - transport_client_id: u128, - vsr_client_id: u128, - request_header: &RoutedRequestHeader, - user_id: u32, - client_version: &ClientVersionInfo, -) -> Result<(), LoginRegisterError> -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let sdk_info = ClientSdkInfo { - sdk_name: client_version.sdk_name.as_str().to_owned(), - sdk_version: client_version.sdk_version.as_str().to_owned(), - protocol_version: client_version.protocol_version, - }; - let existing_session = { - let sessions = sessions.borrow(); - sessions - .get_session(transport_client_id) - .map(|(_, session)| session) - }; - if let Some(session) = existing_session { - // Re-login on a bound connection: refresh the recorded SDK info - // (a reconnecting client may have been upgraded) and replay. - sessions - .borrow_mut() - .record_sdk_info(transport_client_id, sdk_info); - // A lagging backup's commit_max can sit below the epoch this session - // already bound; never advertise a commit behind the session itself. - let commit = current_metadata_commit(shard).max(session); - let reply = - build_login_register_reply(request_header, vsr_client_id, session, commit, user_id); - let _ = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await; - return Ok(()); - } - - // Submit Register and await the commit. The SessionManager is left - // untouched until the op commits cluster-wide (post-quorum): there is no - // optimistic Authenticated transition, so a transient submit failure - // needs no rollback -- the connection stays Connected and the SDK - // read-timeout replays. - let session = match submit_register_on_owner(shard, vsr_client_id, user_id).await { - // The wire reply carries only the fence epoch; the SDK numbers its - // own requests, so the bind watermark is not surfaced (see the - // BoundSession doc for who does consume it). - Ok(bound) => bound.epoch, - Err(error) => { - return Err(LoginRegisterError::Transient(error)); - } - }; - - // Post-commit: Connected -> Authenticated -> Bound in a single borrow with - // no await in between, so the intermediate Authenticated state is never - // observable to a concurrent request on this connection. - { - let mut sessions = sessions.borrow_mut(); - sessions - .login(transport_client_id, user_id) - .map_err(LoginRegisterError::Session)?; - sessions.record_sdk_info(transport_client_id, sdk_info); - if let Err(error) = sessions.bind_session(transport_client_id, vsr_client_id, session) { - // No local rollback: `submit_register_in_process` above has - // already committed cluster-wide. A local-only - // `remove_client_session` here would diverge peers (they retain - // the slot until they evict the client themselves). The - // transport-disconnect callback owns local cleanup once the - // socket closes. - return Err(LoginRegisterError::Session(error)); - } - } - - // `session` IS the register's commit op, and on a backup that forwarded - // the proposal the local applied commit still lags it. Reporting the - // lower number would make one frame contradict itself. - let commit = current_metadata_commit(shard).max(session); - let reply = build_login_register_reply(request_header, vsr_client_id, session, commit, user_id); - let send_result = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await; - if let Err(error) = send_result { - warn!( - transport_client_id, - error = %error, - "failed to send login/register reply" - ); - } - - Ok(()) -} - -/// Decide whether a failed login/register gets a terminal eviction or a -/// transient replay hint. -/// -/// A transient consensus failure ([`LoginRegisterError::is_terminal`] is -/// `false`) means the cluster could not commit *right now* (a freshly booted -/// primary still catching up, or a cross-shard submit canceled). Those get a -/// result-framed replay hint instead of silence, so the SDK replays at once -/// rather than waiting out its read-timeout; replying empty would surface as -/// a hard `InvalidFormat` decode failure and break the replay. -/// -/// Terminal auth errors (`InvalidCredentials` / `InvalidToken` / -/// `UserInactive` / `Session`) fast-fail with a typed `Eviction` frame so the -/// SDK surfaces the real reason (every frame transport decodes -/// `Command::Eviction`) instead of a decode error or a timeout. -#[allow(clippy::future_not_send)] -pub async fn surface_login_failure( - shard: &Rc>, - transport_client_id: u128, - request_header: &RoutedRequestHeader, - error: &LoginRegisterError, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - if error.is_terminal() { - send_login_eviction( - shard, - transport_client_id, - request_header.client, - eviction_reason_for(error), - ) - .await; - } else { - // Which code the hint carries is what tells the client whether the - // replay may move to another node: see `transient_login_code`. - send_login_transient_reply( - shard, - transport_client_id, - request_header, - transient_login_code(error), - ) - .await; - } -} - -/// Result-framed transient Reply on a non-terminal failed Register. The SDK -/// decodes the nonzero result code and replays the same login on the same -/// connection. Only call for transient errors -- see -/// [`surface_login_failure`]. -#[allow(clippy::future_not_send)] -async fn send_login_transient_reply( - shard: &Rc>, - transport_client_id: u128, - request_header: &RoutedRequestHeader, - code: IggyError, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let commit = current_metadata_commit(shard); - let reply = build_result_rejection_reply(request_header, commit, code.as_code()); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - "failed to send login transient reply" - ); - } -} - -/// Wire code for a transient (non-terminal) login/register failure. -/// -/// `TransientNotAccepted` asserts nothing was committed: the register never -/// entered a pipeline (not primary / not caught up / pipeline full) or never -/// left this node (primary unreachable). The client may re-issue it anywhere, -/// including under a fresh identity after failing over to another node. -/// -/// A forward timeout, an in-progress proposal, or a canceled proposal has an -/// UNKNOWN outcome, so none can ride that assertion. `TransientNotCommitted` -/// pins the replay to this connection and its client id, where a register that -/// did commit rebinds its own client-table entry. Re-issuing under a freshly -/// minted id would instead orphan that entry until capacity eviction reclaims -/// it. -const fn transient_login_code(error: &LoginRegisterError) -> IggyError { - match error { - LoginRegisterError::Transient( - MetadataSubmitError::ForwardTimedOut - | MetadataSubmitError::InProgress - | MetadataSubmitError::Canceled, - ) => IggyError::TransientNotCommitted, - _ => IggyError::TransientNotAccepted, - } -} - -/// Wire reason for a terminal login/register failure. Session-level -/// rejections (including the non-retryable submit refusal, where the -/// presented client id belongs to another user) collapse to -/// `SessionError`; the SDK maps it to `Unauthenticated`. -const fn eviction_reason_for(error: &LoginRegisterError) -> EvictionReason { - match error { - LoginRegisterError::InvalidCredentials => EvictionReason::InvalidCredentials, - LoginRegisterError::InvalidToken => EvictionReason::InvalidToken, - LoginRegisterError::UserInactive => EvictionReason::UserInactive, - _ => EvictionReason::SessionError, - } -} - -#[cfg(test)] -mod tests { - use super::*; - use iggy_common::{IggyError, eviction_reason_to_error}; - - /// The reasons this module emits must map back to the credential - /// errors the SDK is expected to surface (the shared - /// `eviction_reason_to_error` grading both ends). - #[test] - fn terminal_login_errors_map_to_typed_sdk_errors() { - let cases = [ - ( - eviction_reason_for(&LoginRegisterError::InvalidCredentials), - IggyError::InvalidCredentials, - ), - ( - eviction_reason_for(&LoginRegisterError::InvalidToken), - IggyError::InvalidPersonalAccessToken, - ), - ( - eviction_reason_for(&LoginRegisterError::UserInactive), - IggyError::Unauthenticated, - ), - ]; - for (reason, expected) in cases { - let error = eviction_reason_to_error(reason, 0, 0); - assert_eq!( - error.as_code(), - expected.as_code(), - "reason {reason:?} must surface as {expected:?}" - ); - } - } - - #[test] - fn unknown_register_outcomes_pin_the_client_identity() { - for error in [ - MetadataSubmitError::ForwardTimedOut, - MetadataSubmitError::InProgress, - MetadataSubmitError::Canceled, - ] { - assert_eq!( - transient_login_code(&LoginRegisterError::Transient(error)), - IggyError::TransientNotCommitted, - ); - } - for error in [ - MetadataSubmitError::NotPrimary, - MetadataSubmitError::NotCaughtUp, - MetadataSubmitError::PipelineFull, - MetadataSubmitError::PrimaryUnreachable, - ] { - assert_eq!( - transient_login_code(&LoginRegisterError::Transient(error)), - IggyError::TransientNotAccepted, - ); - } - } -} diff --git a/core/server/src/bootstrap.rs b/core/server/src/bootstrap.rs index 186325ff47..5903aebafc 100644 --- a/core/server/src/bootstrap.rs +++ b/core/server/src/bootstrap.rs @@ -15,13 +15,14 @@ // specific language governing permissions and limitations // under the License. -use crate::auth::warm_dummy_password_hash; use crate::cluster_meta::{ClusterRoster, resolved_roster_nodes, self_advertised_address}; use crate::config_writer::write_current_config; +use crate::dispatch::partition::make_partition_read_handler; +use crate::dispatch::session_ops::warm_dummy_password_hash; +use crate::dispatch::submit::make_metadata_submit_handler; use crate::dispatch::{ make_client_request_handler, make_deferred_client_request_handler, - make_deferred_replica_message_handler, make_list_clients_handler, make_metadata_submit_handler, - make_partition_read_handler, + make_deferred_replica_message_handler, make_list_clients_handler, }; use crate::http; use crate::partition_helpers::{ @@ -1252,8 +1253,13 @@ async fn shard_main( let hb_sessions = Rc::clone(&sessions); let hb_interval = config.heartbeat.interval.get_duration(); let hb_handle = compio::runtime::spawn(async move { - crate::dispatch::run_heartbeat_verifier(hb_shard, hb_sessions, hb_interval, hb_stop_rx) - .await; + crate::dispatch::session_ops::run_heartbeat_verifier( + hb_shard, + hb_sessions, + hb_interval, + hb_stop_rx, + ) + .await; }); bus.track_background(hb_handle); Some(hb_stop_tx) diff --git a/core/server/src/dispatch.rs b/core/server/src/dispatch.rs deleted file mode 100644 index 3bf5af28ae..0000000000 --- a/core/server/src/dispatch.rs +++ /dev/null @@ -1,5275 +0,0 @@ -// Licensed to the Apache Software Foundation (ASF) under one -// or more contributor license agreements. See the NOTICE file -// distributed with this work for additional information -// regarding copyright ownership. The ASF licenses this file -// to you under the Apache License, Version 2.0 (the -// "License"); you may not use this file except in compliance -// with the License. You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, -// software distributed under the License is distributed on an -// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY -// KIND, either express or implied. See the License for the -// specific language governing permissions and limitations -// under the License. - -//! Per-shard request dispatch. -//! -//! Client-request queue plumbing, the transport / replica / -//! metadata-submit handler factories, the owner-forwarding helpers that -//! run consensus on shard 0, and the login/register, logout, and -//! non-replicated request handlers. - -mod authz; - -use crate::auth::{ - complete_login_register, surface_login_failure, verify_login_credentials, - verify_pat_credentials, -}; -use crate::cluster_meta::ClusterRoster; -use crate::consumer_group::{ - maybe_rewrite_consumer_group_request, maybe_rewrite_consumer_offset_request, -}; -use crate::dispatch::authz::{ - authorize_default_read, authorize_partition_op, authorize_partition_read, authorize_uid, - send_deny_reply, send_non_replicated_deny, send_unbound_deny_reply, -}; -use crate::login_register::LoginRegisterError; -use crate::pat::maybe_rewrite_pat_request; -use crate::responses::{ - NonReplicatedResponse, build_consumer_offset_body, build_deny_reply, build_empty_reply, - build_get_me_response, build_get_personal_access_tokens_response, - build_non_replicated_response, build_polled_messages_reply, build_raw_pat_reply, - connected_client_to_response, current_metadata_commit, resolve_partition_namespace, - resolve_partition_request_namespace, -}; -use crate::segment_cleaner::UNENFORCEABLE_TOPIC_SIZE_WARN; -use crate::session_manager::SessionManager; -use crate::shell::{ShellBus, ShellShard, ShellShardHandle}; -use crate::snapshot; -use crate::users::maybe_rewrite_user_password_request; -use crate::wire::{request_body, usize_to_u32, verify_request_checksum}; -use bytes::Bytes; -use configs::server::ServerSystemConfig; -use consensus::{ - Consensus, DISCONNECT_LOGOUT_REQUEST_ID, EvictionContext, MetadataHandle, PartitionsHandle, - build_eviction_message, build_incompatible_protocol_eviction_message, - build_result_rejection_reply, -}; -use iggy_binary_protocol::PrepareHeader; -use iggy_binary_protocol::codes::{ - GET_CLIENT_CODE, GET_CLIENTS_CODE, GET_CLUSTER_METADATA_CODE, GET_CONSUMER_OFFSET_CODE, - GET_ME_CODE, GET_PERSONAL_ACCESS_TOKENS_CODE, GET_SNAPSHOT_FILE_CODE, GET_STATS_CODE, - LOGIN_USER_CODE, LOGIN_WITH_PERSONAL_ACCESS_TOKEN_CODE, PING_CODE, POLL_MESSAGES_CODE, - SYNC_CONSUMER_GROUP_CODE, -}; -use iggy_binary_protocol::primitives::consumer::WireConsumer; -use iggy_binary_protocol::primitives::polling_strategy::WirePollingStrategy; -use iggy_binary_protocol::requests::consumer_groups::SyncConsumerGroupRequest; -use iggy_binary_protocol::requests::consumer_offsets::{ - GetConsumerOffsetRequest, StoreConsumerOffsetRequest, -}; -use iggy_binary_protocol::requests::messages::PollMessagesRequest; -use iggy_binary_protocol::requests::partitions::{ - CreatePartitionsRequest, DeletePartitionsRequest, -}; -use iggy_binary_protocol::requests::segments::DeleteSegmentsRequest; -use iggy_binary_protocol::requests::streams::{CreateStreamRequest, UpdateStreamRequest}; -use iggy_binary_protocol::requests::system::get_client::GetClientRequest; -use iggy_binary_protocol::requests::system::get_snapshot::GetSnapshotRequest; -use iggy_binary_protocol::requests::topics::{CreateTopicRequest, UpdateTopicRequest}; -use iggy_binary_protocol::requests::users::{ - CreateUserRequest, LoginRegisterRequest, LoginRegisterWithPatRequest, UpdateUserRequest, -}; -use iggy_binary_protocol::responses::clients::client_response::ConsumerGroupInfoResponse; -use iggy_binary_protocol::responses::clients::get_client::ClientDetailsResponse; -use iggy_binary_protocol::responses::clients::get_clients::GetClientsResponse; -use iggy_binary_protocol::responses::consumer_groups::SyncConsumerGroupResponse; -use iggy_binary_protocol::responses::system::get_snapshot::GetSnapshotResponse; -use iggy_binary_protocol::{ - AckLevel, ClientVersionInfo, Command, ConsensusHeader, EvictionReason, ForwardLogoutHeader, - ForwardLogoutOutcome, ForwardLogoutResultHeader, ForwardRegisterHeader, ForwardRegisterOutcome, - ForwardRegisterResultHeader, GenericHeader, HEADER_SIZE, KIND_CONSUMER_GROUP, - MAX_PARTITIONS_PER_REQUEST, Operation, ProtocolVersion, RequestHeader, RoutedRequestHeader, - WireDecode, WireEncode, WireIdentifier, WireOptions, is_protocol_compatible, -}; -use iggy_common::{ - IggyByteSize, IggyError, MaxTopicSize, PollingStrategy, SnapshotCompression, - SystemSnapshotType, TopicCreateOptions, UPDATABLE_STREAM_OPTION_KEYS, - UPDATABLE_TOPIC_OPTION_KEYS, UPDATABLE_USER_OPTION_KEYS, validate_preallocated_topic_bytes, - validate_topic_segment_size, -}; -use journal::superblock::SuperblockStore; -use journal::{Journal, JournalHandle}; -use message_bus::client_listener::RequestHandler; -use message_bus::framing::MAX_MESSAGE_SIZE; -use message_bus::replica::listener::MessageHandler; -use message_bus::{AUTO_COMMIT_CLIENT_ID, BusMessage}; -use metadata::impls::metadata::{ - BoundSession, MetadataSubmitError, StreamsFrontend, build_truncate_partition_client_message, - build_truncate_partition_client_message_with_identifiers, -}; -use metadata::permissioner::Permissioner; -use metadata::stm::stream::Streams; -use partitions::{AutoCommitApplied, PollPlan, PollingArgs, PollingConsumer}; -use secrecy::ExposeSecret; -use server_common::Message; -use server_common::sharding::IggyNamespace; -use shard::shards_table::ShardsTable; -use shard::{ - ConnectedClientInfo, ListClientsHandler, PartitionRead, PartitionReadHandler, - PartitionReadReply, -}; -use std::cell::RefCell; -use std::collections::{HashMap, HashSet, VecDeque}; -use std::net::IpAddr; -use std::rc::Rc; -use std::sync::Arc; -use std::time::Duration; -use tracing::{debug, warn}; - -pub type ClientRequestQueues = Rc>>>>; -pub type ActiveClientRequests = Rc>>; - -pub fn make_client_request_handler( - shard: &Rc>, - sessions: &Rc>, - system_config: Arc, - max_tokens_per_user: u32, -) -> RequestHandler -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let shard = Rc::clone(shard); - let sessions = Rc::clone(sessions); - let queues: ClientRequestQueues = Rc::new(RefCell::new(HashMap::new())); - let active: ActiveClientRequests = Rc::new(RefCell::new(HashSet::new())); - let sessions_for_disconnect = Rc::clone(&sessions); - let shard_for_disconnect = Rc::clone(&shard); - shard - .bus - .set_client_connection_lost_fn(Rc::new(move |client_id| { - if let Some((vsr_client_id, session)) = sessions_for_disconnect - .borrow_mut() - .remove_connection(client_id) - { - submit_disconnect_logout(Rc::clone(&shard_for_disconnect), vsr_client_id, session); - } - })); - Rc::new(move |client_id, message| { - enqueue_client_request( - Rc::clone(&shard), - Rc::clone(&sessions), - Arc::clone(&system_config), - max_tokens_per_user, - Rc::clone(&queues), - Rc::clone(&active), - client_id, - message, - ); - }) -} - -/// Build the per-shard [`ListClientsHandler`]: on a `ListClients` -/// broadcast, serialize this shard's locally-homed connected clients from -/// its `SessionManager` and push them back over the reply sender. The -/// aggregation across all shards happens in -/// [`shard::IggyShard::list_all_clients`]. -pub fn make_list_clients_handler(sessions: &Rc>) -> ListClientsHandler { - let sessions = Rc::clone(sessions); - Rc::new(move |reply| { - let clients: Vec = sessions.borrow().iter_clients().collect(); - // Best-effort: the gather side bounds itself by count + timeout, so - // a dropped reply (receiver gone) just means this shard is omitted. - let _ = reply.try_send(clients); - }) -} - -/// Build the per-shard [`PartitionReadHandler`]: on a `PartitionRead` frame -/// (this shard owns the namespace), run the poll / consumer-offset lookup -/// against the local partitions plane and push the result back over the -/// carried reply sender. The requesting shard bounds the wait with a -/// timeout, so a dropped reply degrades to a client-visible read failure. -pub fn make_partition_read_handler( - shard_handle: &ShellShardHandle, -) -> PartitionReadHandler -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let shard_handle = Rc::clone(shard_handle); - // Runs synchronously on the shard pump (see `process_lifecycle` -> - // `on_partition_read`). `build_poll_snapshot` takes a pump-only `&mut` - // partition borrow (synchronous, so no sibling task can realloc under it) and - // returns an owned `PollPlan`; only owned data crosses into `spawn_poll_io`. A - // fully-resident poll replies here without spawning. See the `poll_plan` module docs. - Rc::new(move |namespace, read, reply| { - let Some(shard) = upgrade_shard_handle(&shard_handle) else { - return; - }; - let partitions = shard.plane.partitions(); - match read { - PartitionRead::Poll { consumer, args } => { - match partitions.build_poll_snapshot(&namespace, consumer, &args) { - None => { - let _ = reply.try_send(PartitionReadReply::NotFound); - } - Some(plan) if plan.needs_off_pump_io() => { - spawn_poll_io(Rc::clone(&shard), namespace, plan, reply); - } - Some(plan) => { - let (fragments, current_offset, auto_commit) = plan.execute_resident(); - if let Some(applied) = auto_commit { - submit_auto_commit(&shard, namespace, &applied); - } - let _ = reply.try_send(PartitionReadReply::Poll { - fragments, - current_offset, - }); - } - } - } - PartitionRead::ConsumerOffset { consumer } => { - let result = match partitions.consumer_offset_read(&namespace, consumer) { - Some((stored, current_offset)) => PartitionReadReply::ConsumerOffset { - stored, - current_offset, - }, - None => PartitionReadReply::NotFound, - }; - let _ = reply.try_send(result); - } - PartitionRead::GroupOffsetState { group_id } => { - let result = match partitions.group_offset_state(&namespace, group_id) { - Some((last_polled, committed)) => PartitionReadReply::GroupOffsetState { - last_polled, - committed, - }, - None => PartitionReadReply::NotFound, - }; - let _ = reply.try_send(result); - } - PartitionRead::ClearGroupLastPolled { group_id } => { - let result = match partitions.clear_group_last_polled(&namespace, group_id) { - Some(()) => PartitionReadReply::Ack, - None => PartitionReadReply::NotFound, - }; - let _ = reply.try_send(result); - } - PartitionRead::ResolveSegmentDeleteOffset { count } => { - let result = partitions - .segment_delete_resolution(&namespace, count) - .map_or_else( - || PartitionReadReply::NotFound, - |(up_to_offset, lagging)| PartitionReadReply::SegmentDeleteOffset { - up_to_offset, - lagging, - }, - ); - let _ = reply.try_send(result); - } - } - }) -} - -/// Spawn the off-pump leg of a partition poll: disk read + auto-commit apply on -/// the OWNED plan (disk descriptors, resident-tail `Frozen` clones, `Arc` offset -/// map), then replicate the auto-committed offset and send the reply. Holds no -/// partition reference across the IO, so it is sound concurrently with the -/// pump's `&mut` writes; the auto-commit submit re-borrows synchronously after. -fn spawn_poll_io( - shard: Rc>, - namespace: IggyNamespace, - plan: PollPlan, - reply: shard::Sender, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let bus = shard.bus.clone(); - bus.spawn(async move { - // Diagnostic-only wall clock: `elapsed` gates the slow-poll `warn!` - // below and is never folded into a reply or the deterministic schedule, - // so it stays sound under the simulator's virtual clock (there it just - // measures near-zero real time and never fires). Do not derive any - // replicated or reply value from it, or replay determinism breaks. - let poll_started = std::time::Instant::now(); - let (fragments, current_offset, auto_commit) = plan.execute().await; - let elapsed = poll_started.elapsed(); - if elapsed > std::time::Duration::from_secs(1) { - warn!( - namespace_raw = namespace.inner(), - elapsed_ms = u64::try_from(elapsed.as_millis()).unwrap_or(u64::MAX), - "slow partition poll; gather side may have timed out" - ); - } - // Fire-and-forget: the poll reply is not gated on the offset commit. - if let Some(applied) = auto_commit { - submit_auto_commit(&shard, namespace, &applied); - } - let _ = reply.try_send(PartitionReadReply::Poll { - fragments, - current_offset, - }); - }); -} - -/// Replicate a poll's auto-committed offset through the partition consensus so -/// it survives failover, mirroring the explicit `StoreConsumerOffset` path: the -/// same op code, submitted onto the owning shard's own pipeline. Best-effort and -/// fire-and-forget -- the poll reply never waits on it, and a full inbox drops -/// the op at WARN rather than backpressuring the reply. -/// -/// The partition plane admits writes on the primary only (it asserts so), and a -/// poll is served on whichever node owns the namespace locally, which may be a -/// backup. So gate on primary status here and drop at WARN otherwise; auto-commit -/// is server-managed best-effort (at-least-once delivery), so a follower-served -/// poll simply does not advance the durable offset. -/// -/// Coalescing: an offset the partition's committed high-water already covers is -/// dropped without a consensus op (the steady state for a re-poll of committed -/// data, hence no log). The gate reads committed state only, so an offset that -/// merely sits in flight keeps resubmitting until its covering op commits -- a -/// dropped op self-heals on the next poll instead of being suppressed forever. -fn submit_auto_commit( - shard: &Rc>, - namespace: IggyNamespace, - applied: &AutoCommitApplied, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - enum AutoCommitGate { - Submit, - Covered, - NotPrimary, - } - let gate = shard - .plane - .partitions() - .with_partition(&namespace, |partition| { - let consensus = partition.consensus(); - if !(consensus.is_primary() && consensus.is_normal() && !consensus.is_transferring()) { - AutoCommitGate::NotPrimary - } else if partition.is_auto_commit_offset_covered( - applied.kind, - applied.consumer_id, - applied.offset, - ) { - AutoCommitGate::Covered - } else { - AutoCommitGate::Submit - } - }); - match gate { - Some(AutoCommitGate::Submit) => {} - Some(AutoCommitGate::Covered) => return, - Some(AutoCommitGate::NotPrimary) | None => { - warn!( - namespace_raw = namespace.inner(), - "auto-commit offset not replicated: partition not primary on this node (best-effort)" - ); - return; - } - } - let message = match build_auto_commit_request(namespace, applied) { - Ok(message) => message, - Err(error) => { - warn!( - namespace_raw = namespace.inner(), - error = %error, - "failed to build auto-commit store-offset request" - ); - return; - } - }; - // Routes by namespace to this same (owning, primary) shard's inbox; the pump - // admits it next turn exactly like a client store. `dispatch` never blocks. - shard.dispatch(message.into_generic()); -} - -/// Build the synthetic `StoreConsumerOffset` request for an auto-commit, keyed -/// to the resolved numeric consumer/group id and stamped with the reserved -/// [`AUTO_COMMIT_CLIENT_ID`] so the commit path skips the (unwaited) reply. The -/// wire stream/topic ids are cosmetic here -- admission and apply key off the -/// header namespace and the consumer id -- but are set from the namespace for a -/// well-formed body. `ack` is `Quorum` so the offset actually replicates. -fn build_auto_commit_request( - namespace: IggyNamespace, - applied: &AutoCommitApplied, -) -> Result, IggyError> { - let request = StoreConsumerOffsetRequest { - consumer: WireConsumer { - kind: applied.kind.as_code(), - id: WireIdentifier::Numeric(applied.consumer_id), - }, - stream_id: WireIdentifier::Numeric(usize_to_u32(namespace.stream_id())?), - topic_id: WireIdentifier::Numeric(usize_to_u32(namespace.topic_id())?), - partition_id: Some(usize_to_u32(namespace.partition_id())?), - offset: applied.offset, - ack: AckLevel::Quorum, - }; - let body = request.to_bytes(); - let header_size = std::mem::size_of::(); - let total_size = header_size + body.len(); - let size = u32::try_from(total_size).map_err(|_| IggyError::InvalidConfiguration)?; - let mut message = Message::::new(total_size); - message.as_mut_slice()[header_size..].copy_from_slice(&body); - Ok( - message.transmute_header(|_, header: &mut RoutedRequestHeader| { - *header = RoutedRequestHeader { - command: Command::Request, - operation: Operation::StoreConsumerOffset, - size, - client: AUTO_COMMIT_CLIENT_ID, - // The partition plane is sessionless (no `ClientTable` dedup); a - // nonzero session + request just satisfy the wire header - // validation. - session: 1, - request: 1, - group: namespace.inner(), - ..Default::default() - }; - }), - ) -} - -pub fn make_deferred_replica_message_handler( - shard_handle: &ShellShardHandle, -) -> MessageHandler -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let shard_handle = Rc::clone(shard_handle); - Rc::new(move |_replica_id, message| { - if let Some(shard) = upgrade_shard_handle(&shard_handle) { - shard.dispatch(message); - } - }) -} - -pub fn make_deferred_client_request_handler( - bus: &B, - shard_handle: &ShellShardHandle, - sessions: &Rc>, - system_config: Arc, - max_tokens_per_user: u32, -) -> RequestHandler -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let shard_handle = Rc::clone(shard_handle); - let sessions = Rc::clone(sessions); - let queues: ClientRequestQueues = Rc::new(RefCell::new(HashMap::new())); - let active: ActiveClientRequests = Rc::new(RefCell::new(HashSet::new())); - let sessions_for_disconnect = Rc::clone(&sessions); - let shard_handle_for_disconnect = Rc::clone(&shard_handle); - let bus_for_spawn = (*bus).clone(); - bus.set_client_connection_lost_fn(Rc::new(move |client_id| { - if let Some((vsr_client_id, session)) = sessions_for_disconnect - .borrow_mut() - .remove_connection(client_id) - && let Some(shard) = upgrade_shard_handle(&shard_handle_for_disconnect) - { - submit_disconnect_logout(shard, vsr_client_id, session); - } - })); - Rc::new(move |client_id, message| { - let shard_handle = Rc::clone(&shard_handle); - let sessions = Rc::clone(&sessions); - let system_config = Arc::clone(&system_config); - let queues = Rc::clone(&queues); - let active = Rc::clone(&active); - queues - .borrow_mut() - .entry(client_id) - .or_default() - .push_back(message); - if !active.borrow_mut().insert(client_id) { - return; - } - bus_for_spawn.spawn(async move { - let Some(shard) = upgrade_shard_handle(&shard_handle) else { - active.borrow_mut().remove(&client_id); - return; - }; - drain_client_requests( - shard, - sessions, - system_config, - max_tokens_per_user, - queues, - active, - client_id, - ) - .await; - }); - }) -} - -/// Handler shard 0 runs for an inbound [`shard::MetadataSubmit`]: a peer -/// shard has verified credentials and owns the session locally, and asks -/// shard 0 (the metadata consensus owner) to run only the consensus -/// proposal. Spawns a task so the awaiting peer is woken once the op -/// commits. Submit failures are returned verbatim so the peer can preserve -/// unknown-outcome retry semantics. -pub fn make_metadata_submit_handler( - shard_handle: &ShellShardHandle, -) -> shard::MetadataSubmitHandler -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let shard_handle = Rc::clone(shard_handle); - Rc::new(move |submit| { - let Some(shard) = upgrade_shard_handle(&shard_handle) else { - return; - }; - let bus = shard.bus.clone(); - bus.spawn(async move { - match submit { - shard::MetadataSubmit::Register { - vsr_client_id, - user_id, - reply, - } => { - let bound = - submit_register_local_or_forward(&shard, vsr_client_id, user_id).await; - let _ = reply.try_send(bound); - } - shard::MetadataSubmit::ForwardedRegister { - vsr_client_id, - user_id, - nonce, - origin_replica, - } => { - answer_forwarded_register( - &shard, - vsr_client_id, - user_id, - nonce, - origin_replica, - ) - .await; - } - shard::MetadataSubmit::ForwardedLogout { - vsr_client_id, - session, - request, - nonce, - origin_replica, - } => { - answer_forwarded_logout( - &shard, - vsr_client_id, - session, - request, - nonce, - origin_replica, - ) - .await; - } - shard::MetadataSubmit::Logout { - vsr_client_id, - session, - request, - reply, - } => { - let outcome = - submit_logout_local_or_forward(&shard, vsr_client_id, session, request) - .await; - let _ = reply.try_send(outcome); - } - shard::MetadataSubmit::ClientRequest { request, reply } => { - let committed = match request.try_into_typed::() { - Ok(typed) => shard - .plane - .metadata() - .submit_request_in_process(typed) - .await - .ok(), - Err(error) => { - warn!(?error, "ClientRequest submit: undecodable request header"); - None - } - }; - let _ = reply.try_send(committed); - } - shard::MetadataSubmit::CompleteRevocation { - stream_id, - topic_id, - group_id, - source_client_id, - partition_id, - reply, - } => { - let commit = shard - .plane - .metadata() - .submit_complete_revocation_in_process( - stream_id, - topic_id, - group_id, - source_client_id, - partition_id, - ) - .await - .ok(); - let _ = reply.try_send(commit); - } - } - }); - }) -} - -// Session resume is performed BY THE LOGIN PATH, not by a separate -// credential-free rebind. -// -// A reconnecting client re-authenticates on the new connection and presents -// its previous `client_id` in the login frame; `submit_register_in_process` -// finds the existing table entry, verifies the authenticated user owns it, -// and returns its epoch, so `bind_session` binds the new transport to the -// old entry with its watermark and reply ring intact. That IS the resume. -// -// An earlier revision instead rebound an *unbound* transport straight from -// the table whenever a replicated frame carried a matching -// `(client, session)`, treating that pair as a bearer token. That was wrong -// in four ways, and the combination was a pre-auth session takeover: -// -// - it called `SessionManager::login` itself, so no credential was ever -// presented, and the connection was logged in as the entry's cached -// `user_id`; authority for replicated ops then resolves from the table -// (`resolve_acting_user_id`) and for partition ops from the session -// manager, so BOTH planes ran as the original registrant; -// - the pair carries far less entropy than "client-generated random -// u128" implies: HTTP mints `client_id` from the shard-0 sequential -// counter (`mint_shard_zero_client_id`, seeded at 1 per process) and no -// live path ever bumps an epoch past 1, so the token was `client=N, -// session=1` for small N; -// - `ClientEntry` carries no transport or plane tag, so a raw TCP peer -// could bind an HTTP-originated session; -// - `bind_session` demotes the evicted holder to `Connected`, the one -// state `login` accepts, so the loser's next replicated frame -// re-resumed and stole the session back, unbounded and with no eviction -// frame either way. -// -// Routing resume through login also restores the checks that path owns: -// password / PAT verification, `UserStatus::Active`, PAT expiry, the -// protocol-version gate, and SDK-info recording. -// -// An unbound transport sending a replicated frame therefore gets the typed -// `Eviction(NoSession)` fail-fast below and must log in. - -#[allow(clippy::too_many_arguments)] -fn enqueue_client_request( - shard: Rc>, - sessions: Rc>, - system_config: Arc, - max_tokens_per_user: u32, - queues: ClientRequestQueues, - active: ActiveClientRequests, - client_id: u128, - message: Message, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - queues - .borrow_mut() - .entry(client_id) - .or_default() - .push_back(message); - if !active.borrow_mut().insert(client_id) { - return; - } - - let bus = shard.bus.clone(); - bus.spawn(async move { - drain_client_requests( - shard, - sessions, - system_config, - max_tokens_per_user, - queues, - active, - client_id, - ) - .await; - }); -} - -#[allow(clippy::future_not_send)] -async fn drain_client_requests( - shard: Rc>, - sessions: Rc>, - system_config: Arc, - max_tokens_per_user: u32, - queues: ClientRequestQueues, - active: ActiveClientRequests, - client_id: u128, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - loop { - let Some(message) = pop_next_client_request(&queues, &active, client_id) else { - return; - }; - handle_client_request( - &shard, - &sessions, - &system_config, - max_tokens_per_user, - client_id, - message, - ) - .await; - } -} - -fn pop_next_client_request( - queues: &ClientRequestQueues, - active: &ActiveClientRequests, - client_id: u128, -) -> Option> { - let mut queues = queues.borrow_mut(); - let Some(queue) = queues.get_mut(&client_id) else { - active.borrow_mut().remove(&client_id); - return None; - }; - let message = queue.pop_front(); - if queue.is_empty() { - queues.remove(&client_id); - } - if message.is_none() { - active.borrow_mut().remove(&client_id); - } - message -} - -/// Per-request partitions-count cap, shared by create-topic, create-partitions -/// and delete-partitions admission. Runs pre-consensus like -/// [`validate_topic_bounds`]: an oversized count must not burn a replicated -/// log entry (create-partitions admission would also allocate that many -/// consensus-group ids before replicating). -/// -/// Zero passes here because a zero-partition TOPIC is legal (legacy -/// `create_topic` admits `0..=MAX`); the add/remove requests reject it in -/// [`validate_partitions_change_count`]. -pub const fn validate_partitions_count(partitions_count: u32) -> Result<(), IggyError> { - if partitions_count > MAX_PARTITIONS_PER_REQUEST { - return Err(IggyError::TooManyPartitions); - } - Ok(()) -} - -/// [`validate_partitions_count`] plus the zero rejection that create-partitions -/// and delete-partitions carry: adding or removing zero partitions is a no-op -/// that would still burn a replicated log entry, bump `Streams::revision` and -/// force every shard through a rebalance pass. Legacy rejects it with -/// `TooManyPartitions` in both handlers (`1..=MAX` on create, `== 0` on -/// delete), so the code matches rather than inventing a new one. -pub const fn validate_partitions_change_count(partitions_count: u32) -> Result<(), IggyError> { - if partitions_count == 0 { - return Err(IggyError::TooManyPartitions); - } - validate_partitions_count(partitions_count) -} - -/// Static create-topic bounds shared by the TCP and HTTP ingresses. Runs -/// pre-consensus: a rejected request must not burn a replicated log entry, -/// and `prepare_request` errors evict the session instead of denying typed. -/// `ServerDefault` is exempt from the size floor (it resolves against server -/// config at admission, matching legacy); `Unlimited` passes numerically. -/// `segment_size_bytes` is the topic's RESOLVED segment size (explicit -/// option, else this node's default), so a per-topic segment above the -/// global default still floors the topic cap. -pub fn validate_topic_bounds( - partitions_count: u32, - max_topic_size: MaxTopicSize, - segment_size_bytes: u64, -) -> Result<(), IggyError> { - validate_partitions_count(partitions_count)?; - validate_topic_size_floor(max_topic_size, segment_size_bytes) -} - -/// A topic cap below one segment can never be enforced: the first segment -/// already exceeds it. Split out of [`validate_topic_bounds`] because update -/// admission checks the cap without a partitions count to check. -pub fn validate_topic_size_floor( - max_topic_size: MaxTopicSize, - segment_size_bytes: u64, -) -> Result<(), IggyError> { - if !matches!(max_topic_size, MaxTopicSize::ServerDefault) - && max_topic_size.as_bytes_u64() < segment_size_bytes - { - return Err(IggyError::InvalidTopicSize( - max_topic_size, - IggyByteSize::from(segment_size_bytes), - )); - } - Ok(()) -} - -/// Announce an accepted `max_topic_size` the server cannot enforce as written. -/// -/// [`validate_topic_size_floor`] admits any cap of one segment or more, but -/// retention runs PER PARTITION and floors each partition's share at one SEALED -/// segment, which reaches up to one maximum bus frame past `segment_size`. A cap -/// between the two is stored and echoed back verbatim while the server actually -/// keeps `(segment_size + max_message_size) * partitions_count`, so the only -/// moment an operator can be told is the one where they set it. -/// -/// Warns rather than rejects: which caps are accepted is client-visible wire -/// behavior, and tightening it would break topics that already exist. -pub fn warn_unenforceable_topic_size( - max_topic_size: MaxTopicSize, - segment_size_bytes: u64, - max_message_size_bytes: usize, - partitions_count: u32, -) { - let MaxTopicSize::Custom(configured) = max_topic_size else { - return; - }; - let max_message_size_bytes = u64::try_from(max_message_size_bytes).unwrap_or(u64::MAX); - let per_partition_floor = segment_size_bytes.saturating_add(max_message_size_bytes); - let topic_floor = per_partition_floor.saturating_mul(u64::from(partitions_count)); - if configured.as_bytes_u64() >= topic_floor { - return; - } - warn!( - max_topic_size = configured.as_bytes_u64(), - partitions_count, - segment_size = segment_size_bytes, - enforced_per_partition = per_partition_floor, - "{UNENFORCEABLE_TOPIC_SIZE_WARN}" - ); -} - -/// Announce the same unenforceable cap when partitions are ADDED to a topic. -/// -/// The cap is topic-wide but enforcement is per partition, so every added -/// partition shrinks the share: a cap that cleared the floor when the topic was -/// created can stop clearing it here. The request carries only the delta, so -/// the stored cap, segment size and current partition count come from metadata. -pub fn warn_unenforceable_topic_size_on_partition_add( - streams: &Streams, - stream_id: &WireIdentifier, - topic_id: &WireIdentifier, - max_message_size_bytes: usize, - added_partitions_count: u32, -) { - let Some(((stream_slab, topic_slab), _)) = streams.partition_count_context(stream_id, topic_id) - else { - return; - }; - let Some((_, max_topic_size, partitions_count, segment_size)) = - streams.topic_retention_config(stream_slab, topic_slab) - else { - return; - }; - warn_unenforceable_topic_size( - max_topic_size, - segment_size.map_or(iggy_common::DEFAULT_SEGMENT_SIZE, |segment_size| { - segment_size.as_bytes_u64() - }), - max_message_size_bytes, - u32::try_from(partitions_count) - .unwrap_or(u32::MAX) - .saturating_add(added_partitions_count), - ); -} - -/// Reject option keys outside the resource's catalog, pre-consensus. Unknown -/// keys are rejected rather than skipped: a silently ignored knob would hand -/// the client server defaults without it ever learning. Streams and users -/// have no catalog keys yet, so `known` is empty for both until one lands. -pub fn validate_option_keys(options: &WireOptions, known: &[&str]) -> Result<(), IggyError> { - for entry in options { - // Wire validation already enforced UTF-8 string keys. - let key = String::from_utf8_lossy(entry.key); - if !known.contains(&key.as_ref()) { - return Err(IggyError::UnsupportedOptionKey(key.into_owned())); - } - } - Ok(()) -} - -/// Reject a request before it reaches consensus: warn, then send the typed -/// deny reply. A silent drop would wedge every later request on the -/// connection until the socket read timeout. `context` labels the rejection -/// site in both log lines. -#[allow(clippy::future_not_send)] -async fn send_pre_consensus_deny( - shard: &Rc>, - header: &RoutedRequestHeader, - transport_client_id: u128, - error: &IggyError, - context: &'static str, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - warn!( - transport_client_id, - error = %error, - operation = ?header.operation, - context, - "denying request pre-consensus" - ); - let commit = current_metadata_commit(shard); - let reply = build_deny_reply(header, transport_client_id, 0, commit, error.as_code()); - if let Err(send_error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %send_error, - context, - "failed to send pre-consensus deny reply" - ); - } -} - -#[allow(clippy::future_not_send, clippy::too_many_lines)] -async fn handle_client_request( - shard: &Rc>, - sessions: &Rc>, - system_config: &Arc, - max_tokens_per_user: u32, - transport_client_id: u128, - message: Message, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let request = match message.try_into_typed::() { - Ok(request) => request, - Err(error) => { - warn!( - transport_client_id, - error = %error, - "dropping client request with invalid header" - ); - return; - } - }; - // Promote to the server-internal routed shape at the boundary: the - // client wire carries no group (it is derived -- plane from `operation`, - // partition target from the payload), so it starts unset here and the - // resolution sites below stamp it before anything routes on it. - let request = request.into_routed(); - - // The last point that still sees the body the CLIENT sent; every rewrite below - // substitutes server-chosen bytes and carries the stamp through unchanged. - if let Err(error) = verify_request_checksum(&request) { - warn!( - transport_client_id, - operation = ?request.header().operation, - request = request.header().request, - "dropping client request whose body does not match its own checksum" - ); - send_deny_reply( - shard, - transport_client_id, - request.header(), - error.as_code(), - ) - .await; - return; - } - - ensure_transport_connection(shard, sessions, transport_client_id); - - // Any request is liveness proof, not just PING: an idle-but-active client - // (e.g. an admin issuing reads between long sleeps) must not be evicted by - // the heartbeat verifier. A genuinely dead connection sends nothing, so the - // intended stale-client eviction still fires. No-ops for an unbound client. - sessions.borrow_mut().record_heartbeat(transport_client_id); - - let header = *request.header(); - if header.operation == Operation::NonReplicated { - // Auth bypass guard: `PING`, the liveness probe, is the only pre-auth - // code, on every roster shape. `GET_CLUSTER_METADATA` describes the - // private replica network and is not something an unauthenticated - // caller gets to read; a client that dialed a backup no longer needs - // it to find the leader, because the backup authenticates the login - // locally and forwards only the consensus proposal - // (`submit_register_local_or_forward`). Every other non-replicated - // code MUST go through Register first, which binds the acting user - // the per-op authz gates resolve. - let nr_code = u32::from_le_bytes(request.header().reserved[..4].try_into().unwrap()); - // Legacy (pre-register) login codes. The server authenticates only via - // the Register handshake (LOGIN_REGISTER / LOGIN_REGISTER_WITH_PAT, - // Operation::Register); the vsr SDK funnels both logins there and never - // emits these. Reject them uniformly with a typed MalformedLogin (the - // SDK maps it to InvalidFormat) before the session gate, so a legacy or - // foreign client fails fast instead of getting the generic - // Unauthenticated deny the pre-auth guard would send unbound, or the - // silent empty-ok Reply the bound non-replicated path would send. - if matches!( - nr_code, - LOGIN_USER_CODE | LOGIN_WITH_PERSONAL_ACCESS_TOKEN_CODE - ) { - warn!( - transport_client_id, - code = nr_code, - "rejecting legacy login code; server requires the register handshake" - ); - send_login_eviction( - shard, - transport_client_id, - header.client, - EvictionReason::MalformedLogin, - ) - .await; - return; - } - let allowed_pre_auth = nr_code == PING_CODE; - if !allowed_pre_auth && sessions.borrow().get_session(transport_client_id).is_none() { - // Foreign SDKs still probe `GET_CLUSTER_METADATA` before login - // until they are fixed, so that rejection is routine traffic and - // logs at debug rather than warn. - if nr_code == GET_CLUSTER_METADATA_CODE { - debug!( - transport_client_id, - "denying pre-auth cluster-metadata read with Unauthenticated" - ); - } else { - warn!( - transport_client_id, - code = nr_code, - "denying pre-auth non-replicated read with Unauthenticated" - ); - } - // A plain deny Reply, not an Eviction: there is no session to - // evict, and an Eviction is session-terminal by wire contract, - // so SDKs would tear down the very connection their login is - // about to use. The status channel carries the error the same - // way the request-checksum denial above does. - send_unbound_deny_reply( - shard, - transport_client_id, - request.header(), - IggyError::Unauthenticated.as_code(), - ) - .await; - return; - } - handle_non_replicated_request(shard, sessions, system_config, transport_client_id, request) - .await; - return; - } - - if header.operation == Operation::Register && header.session == 0 && header.request == 0 { - handle_login_register_request(shard, sessions, transport_client_id, request).await; - return; - } - - if header.operation == Operation::Logout { - handle_logout_request(shard, sessions, transport_client_id, request).await; - return; - } - - let bound = sessions.borrow().get_session(transport_client_id); - if bound.is_none() { - // Replicated request on an unbound transport. Without this short- - // circuit, the rewrite below overwrites `header.client` with - // `transport_client_id` and dispatches; the request_preflight then - // rejects with `NoSession`/`Fenced` and the failure disappears - // silently, wedging the SDK until the socket timeout. A typed - // `Eviction(NoSession)` is right here, unlike the pre-auth read - // guard above: a replicated request implies the client believes it - // has a session, and that session is gone, so it must register - // again. An empty status-0 Reply is not safe here, because - // SendMessages is the one replicated operation without a result - // section, and its decoder would read the empty body as a - // successful send. - warn!( - transport_client_id, - operation = ?header.operation, - "rejecting replicated request from unbound transport with Eviction(NoSession)" - ); - send_unauthenticated_eviction(shard, transport_client_id).await; - return; - } - - // DeleteSegments is neither a partition nor a metadata consensus op: the - // owning shard resolves the requested count to a concrete offset, then a - // `TruncatePartition` is replicated through metadata (Option A). Each - // replica's reconciler trims to the committed watermark. Handle it here, - // ahead of the partition/metadata routing below. - if header.operation == Operation::DeleteSegments { - handle_delete_segments_request(shard, transport_client_id, bound, &request).await; - return; - } - - if header.operation.is_partition() { - // `bound` is Some here (unbound transports returned above). - let (vsr_client_id, bound_session) = bound.unwrap_or((0, 0)); - // `get_session` discards the acting user id the partition gate needs; - // resolve it from the same bound connection. A bound transport always - // has one, but the gate fails closed on `None` rather than trust that. - let acting_user_id = sessions.borrow().get_user_id(transport_client_id); - dispatch_partition_request( - shard, - request, - vsr_client_id, - bound_session, - transport_client_id, - acting_user_id, - ) - .await; - return; - } - - let request = request.transmute_header(|header, new_header: &mut RoutedRequestHeader| { - *new_header = header; - // Metadata-plane ops route by operation: stamp the sentinel group. - new_header.group = server_common::sharding::METADATA_GROUP; - // `bound` is always Some here (unbound transports early-return above); - // this sets the consensus client id + session for the replicated op. - if let Some((bound_client_id, bound_session)) = bound { - new_header.client = bound_client_id; - new_header.session = bound_session; - } - }); - let (request, raw_pat_token) = match maybe_rewrite_pat_request( - sessions, - transport_client_id, - max_tokens_per_user, - |user_id| { - shard - .plane - .metadata() - .mux_stm - .users() - .read(|users| users.pat_count_of(user_id)) - }, - request, - ) { - Ok(rewritten) => rewritten, - Err(error) => { - // Token cap reached, malformed body, or a lost session binding. - send_pre_consensus_deny( - shard, - &header, - transport_client_id, - &error, - "personal-access-token", - ) - .await; - return; - } - }; - // Hash raw passwords and, for ChangePassword, verify the current password - // on the primary before replication; see `crate::users`. Replicas store the - // hash directly. A wrong current password is not denied here: it rides - // consensus and applies as a committed InvalidCredentials no-op, so the only - // Err returned is a malformed body. - let request = match maybe_rewrite_user_password_request(shard, request) { - Ok(rewritten) => rewritten, - Err(error) => { - // Malformed body: deny fast with InvalidCommand. - send_pre_consensus_deny(shard, &header, transport_client_id, &error, "user-password") - .await; - return; - } - }; - // Static bounds run pre-consensus so a rejected request burns no - // replicated log entry; HTTP covers the same bounds via - // `command.validate()`. A body that fails to decode denies typed too - // (`InvalidCommand`), instead of riding consensus just to fail there. - let bounds = match header.operation { - Operation::CreateTopic => CreateTopicRequest::decode_from(request_body(&request)) - .map_err(|_| IggyError::InvalidCommand) - .and_then(|create_topic| { - // `parse` doubles as the catalog gate: an unknown key or a - // malformed value denies typed here, pre-consensus. - let options = TopicCreateOptions::parse(&create_topic.options)?; - if let Some(segment_size) = options.segment_size { - validate_topic_segment_size( - segment_size.as_bytes_u64(), - iggy_common::MAX_TOPIC_SEGMENT_SIZE, - )?; - } - let segment_size = options.segment_size.map_or_else( - || iggy_common::DEFAULT_SEGMENT_SIZE, - |segment_size| segment_size.as_bytes_u64(), - ); - if options - .preallocate_segments - .unwrap_or(iggy_common::DEFAULT_PREALLOCATE_SEGMENTS) - { - validate_preallocated_topic_bytes(segment_size, create_topic.partitions_count)?; - } - let max_topic_size = options - .max_topic_size - .unwrap_or(MaxTopicSize::ServerDefault); - validate_topic_bounds(create_topic.partitions_count, max_topic_size, segment_size)?; - warn_unenforceable_topic_size( - max_topic_size, - segment_size, - shard.bus_max_message_size(), - create_topic.partitions_count, - ); - Ok(()) - }), - Operation::CreatePartitions => CreatePartitionsRequest::decode_from(request_body(&request)) - .map_err(|_| IggyError::InvalidCommand) - .and_then(|create_partitions| { - validate_partitions_change_count(create_partitions.partitions_count)?; - let metadata = shard.plane.metadata(); - warn_unenforceable_topic_size_on_partition_add( - metadata.mux_stm.streams(), - &create_partitions.stream_id, - &create_partitions.topic_id, - shard.bus_max_message_size(), - create_partitions.partitions_count, - ); - Ok(()) - }), - Operation::DeletePartitions => DeletePartitionsRequest::decode_from(request_body(&request)) - .map_err(|_| IggyError::InvalidCommand) - .and_then(|delete_partitions| { - validate_partitions_change_count(delete_partitions.partitions_count) - }), - // Only the updatable subset: the create-time knobs are pushed to - // partitions when the topic is built and nothing re-pushes them, so - // accepting one here would store a value no partition ever sees. - Operation::UpdateTopic => UpdateTopicRequest::decode_from(request_body(&request)) - .map_err(|_| IggyError::InvalidCommand) - .and_then(|update_topic| { - validate_option_keys(&update_topic.options, UPDATABLE_TOPIC_OPTION_KEYS)?; - let options = TopicCreateOptions::parse(&update_topic.options)?; - let Some(max_topic_size) = options.max_topic_size else { - return Ok(()); - }; - // An update can lower the cap below one segment just as a - // create can, and the stored map would then report a size the - // topic can never enforce. The floor is this topic's own - // segment size, since that key is create-only. - let metadata = shard.plane.metadata(); - let streams = metadata.mux_stm.streams(); - let segment_size = streams - .topic_segment_size(&update_topic.stream_id, &update_topic.topic_id) - .map_or_else( - || iggy_common::DEFAULT_SEGMENT_SIZE, - |segment_size| segment_size.as_bytes_u64(), - ); - validate_topic_size_floor(max_topic_size, segment_size)?; - let partitions_count = streams - .topic_partitions_count(&update_topic.stream_id, &update_topic.topic_id) - .unwrap_or(0); - warn_unenforceable_topic_size( - max_topic_size, - segment_size, - shard.bus_max_message_size(), - u32::try_from(partitions_count).unwrap_or(u32::MAX), - ); - Ok(()) - }), - Operation::UpdateStream => UpdateStreamRequest::decode_from(request_body(&request)) - .map_err(|_| IggyError::InvalidCommand) - .and_then(|update_stream| { - validate_option_keys(&update_stream.options, UPDATABLE_STREAM_OPTION_KEYS) - }), - Operation::UpdateUser => UpdateUserRequest::decode_from(request_body(&request)) - .map_err(|_| IggyError::InvalidCommand) - .and_then(|update_user| { - validate_option_keys(&update_user.options, UPDATABLE_USER_OPTION_KEYS) - }), - Operation::CreateStream => CreateStreamRequest::decode_from(request_body(&request)) - .map_err(|_| IggyError::InvalidCommand) - .and_then(|create_stream| validate_option_keys(&create_stream.options, &[])), - Operation::CreateUser => CreateUserRequest::decode_from(request_body(&request)) - .map_err(|_| IggyError::InvalidCommand) - .and_then(|create_user| validate_option_keys(&create_user.options, &[])), - _ => Ok(()), - }; - if let Err(error) = bounds { - send_pre_consensus_deny(shard, &header, transport_client_id, &error, "static-bounds").await; - return; - } - // Enrich consumer-group Join/Leave with the client's VSR id (+ topic - // partition count for Join) before replication; see `crate::consumer_group`. - let request = match maybe_rewrite_consumer_group_request(shard, request).await { - Ok(rewritten) => rewritten, - Err(error) => { - warn!( - transport_client_id, - error = %error, - operation = ?header.operation, - "dropping consumer-group request with invalid payload" - ); - return; - } - }; - let request_header = *request.header(); - // Replicated request: run consensus on the metadata owner (shard 0) and - // bring the committed reply back here. This shard owns the connection, - // so it writes the reply to the socket via the transport client id -- - // shard 0 can't route by the consensus client id (no home-shard bits). - match submit_client_request_on_owner(shard, request).await { - Some(reply) => { - // The raw PAT token never enters consensus (it is non-deterministic - // and secret), so the committed reply body is empty. Substitute the - // raw-token response here, on the minting client's home shard, using - // the confirmed commit position from the committed reply. - let reply = match build_raw_pat_reply(&request_header, reply, raw_pat_token) { - Ok(reply) => reply, - Err(error) => { - warn!( - transport_client_id, - error = %error, - "failed to build raw PAT reply" - ); - return; - } - }; - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - operation = ?header.operation, - "failed to deliver committed reply to client" - ); - } - } - None => { - // Transient submit failure (not primary / not caught up / dedup - // absorbed). Stay silent; the SDK read-timeout replays. - warn!( - transport_client_id, - operation = ?header.operation, - "replicated request not committed (transient); client will replay" - ); - } - } -} - -/// Per-user PATs, resolved from this shard's session (like `get_me`) and read -/// out of the Users STM. Built here rather than in `build_non_replicated_response` -/// which has no session context. -#[allow(clippy::future_not_send)] -async fn handle_get_personal_access_tokens( - shard: &Rc>, - sessions: &Rc>, - transport_client_id: u128, - request: &Message, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let response = build_get_personal_access_tokens_response(shard, sessions, transport_client_id); - send_non_replicated_bytes( - shard, - request, - transport_client_id, - response.to_bytes(), - "get_personal_access_tokens", - ) - .await; -} - -/// The requesting connection's own identity, sourced from this shard's -/// `SessionManager` (not `IggyMetadata`), so built here rather than in -/// `build_non_replicated_response`. -#[allow(clippy::future_not_send)] -async fn handle_get_me( - shard: &Rc>, - sessions: &Rc>, - transport_client_id: u128, - request: &Message, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let response = build_get_me_response(shard, sessions, transport_client_id); - send_non_replicated_bytes( - shard, - request, - transport_client_id, - response.to_bytes(), - "get_me", - ) - .await; -} - -/// Route a partition data-plane op (`SendMessages` / consumer-offset writes) -/// through the shard mesh by namespace: the op belongs to the partition's -/// own consensus group, not the metadata group. The owning shard's -/// partitions plane runs at-least-once consensus and replies directly via -/// `send_to_client`. `header.client` therefore stays the TRANSPORT id -/// (home-shard routing bits), not the VSR session id -- partition ops are -/// sessionless ("session lifecycle is metadata-only"). -/// -/// Callers must have authenticated the transport already: `vsr_client_id` / -/// `bound_session` come from its bound VSR session. Every failure before -/// dispatch replies with a nonzero status -- unresolvable namespace, -/// authorization denial, exhausted routable wait -- so the client fails fast -/// instead of wedging on a silent drop or reading a status-0 frame as a -/// committed write. -/// -/// `vsr_client_id` keys the consumer-group offset fence (the member id), -/// not the transport id stamped into the partition-op header. -#[allow(clippy::future_not_send)] -pub async fn dispatch_partition_request( - shard: &Rc>, - request: Message, - vsr_client_id: u128, - bound_session: u64, - transport_client_id: u128, - acting_user_id: Option, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let header = *request.header(); - let namespace = match resolve_partition_request_namespace( - shard, - header.operation, - request_body(&request), - vsr_client_id, - ) { - Ok(namespace) => namespace, - Err(error) => { - // A partition op against a stream/topic that no longer resolves - // (e.g. a consumer's trailing auto-commit racing a `delete_stream`, - // or an explicit partition id that skipped the client-side - // resolve). The op never reached the partition plane, so a status-0 - // reply would read as a committed ack for work that never happened. - // A silent drop is no better: the SDK connection processes replies - // in lockstep and would wedge forever. - warn!( - transport_client_id, - error = %error, - operation = ?header.operation, - "partition request with unresolved namespace; replying denied" - ); - send_deny_reply( - shard, - transport_client_id, - &header, - IggyError::ResourceNotFound(String::new()).as_code(), - ) - .await; - return; - } - }; - // Dispatch-time RBAC. The partition plane is not replicated through the - // metadata STM, so the in-apply gate cannot cover it; authorize here, on - // the connection's own shard, before burning the routable wait or touching - // the plane. The namespace resolved above, so its stream/topic are the - // committed slab ids the permissioner keys on directly. A denial replies - // the op's frame with an empty body and a nonzero `status` the SDK peeks. - // - // Consistency: this reads THIS shard's local committed permissioner. On a - // peer shard that is a replicated read-mirror, so a permission revocation - // takes effect on the partition plane only once this shard applies the - // revoking commit -- an apply-lag window bounded by replication lag. - // Control-plane ops are exact (gated in-apply, in the same committed order - // on every replica); this local-read relaxation on the data plane is the - // accepted trade for keeping partition ops off the metadata consensus. - let scope = IggyNamespace::from_raw(namespace); - if let Some(status) = authorize_partition_op( - shard, - header.operation, - acting_user_id, - scope.stream_id(), - scope.topic_id(), - ) { - warn!( - transport_client_id, - status, - operation = ?header.operation, - "partition request denied by authorization; replying with status" - ); - send_deny_reply(shard, transport_client_id, &header, status).await; - return; - } - // Convergence wait: a CreateTopic commit returns to the client before the - // per-shard reconcilers seed routing rows and materialise the partition - // (next wake/periodic tick). An op arriving inside that window is not lost - // if it skips this wait -- `router::route_typed` falls back to the hash - // assignment, and the owning shard parks it -- so this is an admission - // courtesy that keeps the steady state off that park buffer, not a - // correctness gate. See `wait_for_partition_routable`, which spells out why - // there is no owner-readiness probe here any more. - if !wait_for_partition_routable(shard, IggyNamespace::from_raw(namespace)).await { - // The op never reached the partition plane, so it is safe to re-issue - // anywhere -- the same contract the plane itself answers for a - // non-primary routing artifact. A status-0 empty reply here would - // fabricate a success ack for a write that hit no partition at all. - warn!( - transport_client_id, - namespace, - operation = ?header.operation, - "partition request not routable within budget; replying transient" - ); - send_deny_reply( - shard, - transport_client_id, - &header, - IggyError::TransientNotAccepted.as_code(), - ) - .await; - return; - } - // A group consumer-offset op carries the group NAME on the wire; the - // partition plane keys the offset by the group's monotonic id (the same - // key the poll path auto-commits under and the read path resolves), so - // rewrite the consumer id before replication -- the apply layer has no - // metadata access to resolve it. - let request = match maybe_rewrite_consumer_offset_request(shard, request) { - Ok(rewritten) => rewritten, - Err(error) => { - warn!( - transport_client_id, - error = %error, - operation = ?header.operation, - "failed to rewrite consumer-offset request; replying empty" - ); - send_empty_partition_reply(shard, transport_client_id, &header).await; - return; - } - }; - let request = request.transmute_header(|header, new_header: &mut RoutedRequestHeader| { - *new_header = header; - new_header.group = namespace; - new_header.client = transport_client_id; - // Header validation requires `session > 0 && request > 0` for - // non-register ops. The partition plane itself is sessionless - // (at-least-once, no `ClientTable` dedup), so the bound VSR - // session merely satisfies validation. Current SDKs do number - // partition ops, but older and internal callers may still send - // zero, so a zero id is normalized to the compatibility value 1. - new_header.session = bound_session; - new_header.request = new_header.request.max(1); - }); - shard.dispatch(request.into_generic()); -} - -#[allow(clippy::future_not_send, clippy::too_many_lines)] -async fn handle_non_replicated_request( - shard: &Rc>, - sessions: &Rc>, - system_config: &Arc, - transport_client_id: u128, - request: Message, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - const CODE_RANGE: std::ops::Range = 0..4; - let code = u32::from_le_bytes(request.header().reserved[CODE_RANGE].try_into().unwrap()); - // Acting user and peer address for the read gates below, resolved in one - // connection lookup. `user_id` is `None` only on the pre-auth path - // (PING), which serves ungated codes; the gated arms fail closed on it. - let (user_id, client_address) = sessions.borrow().read_context(transport_client_id); - match code { - PING_CODE => { - // A ping is the client's liveness proof; reset its staleness clock - // so the heartbeat verifier doesn't evict an active connection. - sessions.borrow_mut().record_heartbeat(transport_client_id); - let commit = current_metadata_commit(shard); - let reply = build_empty_reply( - request.header(), - request.header().client, - request.header().session, - commit, - ); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - "failed to send non-replicated ping reply" - ); - } - } - GET_ME_CODE => { - handle_get_me(shard, sessions, transport_client_id, &request).await; - } - GET_PERSONAL_ACCESS_TOKENS_CODE => { - handle_get_personal_access_tokens(shard, sessions, transport_client_id, &request).await; - } - GET_CLIENTS_CODE => { - if let Err(error) = authorize_uid(shard, user_id, Permissioner::get_clients) { - send_non_replicated_deny(shard, &request, transport_client_id, error.as_code()) - .await; - return; - } - // Shared-nothing: each shard knows only its own connections, so - // gather across all shards (scatter-gather over the mesh). - let infos = shard.list_all_clients().await; - let response = GetClientsResponse { - clients: infos - .iter() - .map(|info| connected_client_to_response(shard, info)) - .collect(), - }; - send_non_replicated_bytes( - shard, - &request, - transport_client_id, - response.to_bytes(), - "get_clients", - ) - .await; - } - GET_CLIENT_CODE => { - if let Err(error) = authorize_uid(shard, user_id, Permissioner::get_client) { - send_non_replicated_deny(shard, &request, transport_client_id, error.as_code()) - .await; - return; - } - // No reverse map from the wire u32 id to a u128 transport id / - // home shard (the u32 is just the seq tail), so gather all and - // filter -- same fan-out as `get_clients`. - let target = GetClientRequest::decode_from(request_body(&request)) - .ok() - .map(|req| req.client_id); - let infos = shard.list_all_clients().await; - #[allow(clippy::cast_possible_truncation)] - let found = target.and_then(|id| infos.iter().find(|info| info.client_id as u32 == id)); - // The SDK decodes an empty body as `None` (client not found). - let bytes = found.map_or_else(Bytes::new, |info| { - let consumer_groups = info.vsr_client_id.map_or_else(Vec::new, |vsr_client_id| { - shard - .plane - .metadata() - .mux_stm - .streams() - .consumer_group_memberships(vsr_client_id) - .into_iter() - .map( - |(stream_id, topic_id, group_id)| ConsumerGroupInfoResponse { - stream_id, - topic_id, - group_id, - }, - ) - .collect() - }); - ClientDetailsResponse { - client: connected_client_to_response(shard, info), - consumer_groups, - } - .to_bytes() - }); - send_non_replicated_bytes(shard, &request, transport_client_id, bytes, "get_client") - .await; - } - GET_SNAPSHOT_FILE_CODE => { - handle_get_snapshot(shard, system_config, transport_client_id, &request, user_id).await; - } - POLL_MESSAGES_CODE => { - handle_poll_messages(shard, transport_client_id, &request, user_id).await; - } - GET_CONSUMER_OFFSET_CODE => { - handle_get_consumer_offset(shard, transport_client_id, &request, user_id).await; - } - SYNC_CONSUMER_GROUP_CODE => { - // Self-scoped: serves the caller's own assignment keyed by the - // header client id, so it carries no permissioner rule. - handle_sync_consumer_group(shard, transport_client_id, &request).await; - } - _ => { - let roster = sessions.borrow().cluster_roster(); - let client_ip = client_address.map(|address| address.ip()); - if client_ip.is_none() { - debug!( - transport_client_id, - code, - "no peer address recorded; advertised-address resolution degrades to the catch-all" - ); - } - handle_default_non_replicated( - shard, - transport_client_id, - code, - &request, - user_id, - &roster, - client_ip, - ) - .await; - } - } -} - -#[allow(clippy::future_not_send, clippy::too_many_arguments)] -async fn handle_default_non_replicated( - shard: &Rc>, - transport_client_id: u128, - code: u32, - request: &Message, - user_id: Option, - roster: &ClusterRoster, - client_ip: Option, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - // Gate by command code before the shared builder runs. The builder stays - // authz-free (it is byte-shared with the HTTP read path, which gates - // separately); a denial replies status!=0 with an empty body. - if let Err(error) = authorize_default_read(shard, code, request_body(request), user_id) { - send_non_replicated_deny(shard, request, transport_client_id, error.as_code()).await; - return; - } - // Stats is the one default read with an async input: the cross-shard - // connected-client gather. Run it here so the shared builder stays sync. - let clients_count = if code == GET_STATS_CODE { - u32::try_from(shard.list_all_clients().await.len()).unwrap_or(u32::MAX) - } else { - 0 - }; - match build_non_replicated_response( - shard, - code, - request_body(request), - user_id, - roster, - client_ip, - clients_count, - ) { - Ok(response) => { - let commit = current_metadata_commit(shard); - let reply = response.into_reply( - request.header(), - request.header().client, - request.header().session, - commit, - ); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - code, - error = %error, - "failed to send non-replicated VSR reply" - ); - } - } - Err(error) => { - // Surface the builder's typed error (unsupported op, undecodable - // body, or a not-found parity read) on the same deny channel the - // authz gate uses; a silent drop would wedge the client until its - // read timeout. - warn!( - transport_client_id, - code, - error = %error, - "denying non-replicated VSR request" - ); - send_non_replicated_deny(shard, request, transport_client_id, error.as_code()).await; - } - } -} - -/// Serve `GET_SNAPSHOT_FILE`: gate on the snapshot rule (`read_servers || -/// manage_servers`, the legacy gate - the archive dumps host diagnostics, so -/// plain authentication must not suffice), then await the off-thread -/// collection (see `snapshot::collect`) and reply with the raw ZIP bytes. -#[allow(clippy::future_not_send)] -async fn handle_get_snapshot( - shard: &Rc>, - system_config: &Arc, - transport_client_id: u128, - request: &Message, - user_id: Option, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - if let Err(error) = authorize_uid(shard, user_id, Permissioner::get_snapshot) { - send_non_replicated_deny(shard, request, transport_client_id, error.as_code()).await; - return; - } - let result = match decode_get_snapshot(request_body(request)) { - Ok((compression, snapshot_types)) => { - snapshot::collect(Arc::clone(system_config), compression, snapshot_types).await - } - Err(error) => Err(error), - }; - match result { - Ok(archive) => { - // The reply frames as `[256-byte header][archive]`. The client's - // `message_bus::read_message` rejects any frame past `MAX_MESSAGE_SIZE` - // (64 MiB) by tearing the connection down untyped, and a frame past - // `u32::MAX` would panic `build_reply_with_body`. The archive is the - // only unbounded non-replicated body, so refuse an oversized one with a - // typed error the SDK decodes. The HTTP path streams via `Body` (not - // this framing), so it stays uncapped. - let frame_size = HEADER_SIZE + archive.len(); - if frame_size > MAX_MESSAGE_SIZE { - warn!( - transport_client_id, - frame_size, - max = MAX_MESSAGE_SIZE, - "snapshot archive exceeds the client frame limit; refusing to send" - ); - send_non_replicated_deny( - shard, - request, - transport_client_id, - IggyError::SnapshotFileCompletionFailed.as_code(), - ) - .await; - return; - } - send_non_replicated_bytes( - shard, - request, - transport_client_id, - GetSnapshotResponse { data: archive }.to_bytes(), - "get_snapshot", - ) - .await; - } - Err(error) => { - warn!(transport_client_id, error = %error, "denying snapshot request"); - send_non_replicated_deny(shard, request, transport_client_id, error.as_code()).await; - } - } -} - -fn decode_get_snapshot( - body: &[u8], -) -> Result<(SnapshotCompression, Vec), IggyError> { - let request = GetSnapshotRequest::decode_from(body).map_err(|_| IggyError::InvalidCommand)?; - let compression = SnapshotCompression::from_code(request.compression)?; - let snapshot_types = request - .snapshot_types - .iter() - .map(|&code| SystemSnapshotType::from_code(code)) - .collect::, _>>()?; - Ok((compression, snapshot_types)) -} - -/// Send a non-replicated reply body to a client, stamping the current -/// metadata commit. Shared by the `get_me` / `get_clients` / `get_client` -/// arms. -#[allow(clippy::future_not_send)] -async fn send_non_replicated_bytes( - shard: &Rc>, - request: &Message, - transport_client_id: u128, - bytes: Bytes, - label: &'static str, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let commit = current_metadata_commit(shard); - let reply = NonReplicatedResponse::Bytes(bytes).into_reply( - request.header(), - request.header().client, - request.header().session, - commit, - ); - send_reply_frame( - shard, - transport_client_id, - reply.into_generic().into_frozen(), - label, - ) - .await; -} - -/// Hand a built reply frame to the bus for `transport_client_id`. -#[allow(clippy::future_not_send)] -async fn send_reply_frame( - shard: &Rc>, - transport_client_id: u128, - frame: impl Into, - label: &'static str, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - if let Err(error) = shard.bus.send_to_client(transport_client_id, frame).await { - warn!(transport_client_id, label, error = %error, "failed to send non-replicated reply"); - } -} - -/// Reject a replicated request from an unbound transport with a typed -/// `Eviction(NoSession)` frame: the session the client believes it has is -/// gone, so it must register again. Pre-auth non-replicated reads get a -/// deny Reply instead (no session exists, so nothing is evicted). -/// -/// The SDK's reply decoder maps eviction reasons to typed errors -/// (`NoSession` -> `Unauthenticated`), so clients fail fast with the same -/// error the legacy server returns instead of a body-decode failure. The -/// eviction context is best-effort off the metadata consensus (peer shards -/// have none; zeroes are cosmetic -- the SDK only reads the reason). -#[allow(clippy::future_not_send)] -async fn send_unauthenticated_eviction( - shard: &Rc>, - transport_client_id: u128, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let ctx = shard.plane.metadata().consensus.as_ref().map_or( - consensus::EvictionContext { - cluster: 0, - view: 0, - replica: 0, - }, - consensus::EvictionContext::from_consensus, - ); - let eviction = consensus::build_eviction_message( - ctx, - transport_client_id, - iggy_binary_protocol::EvictionReason::NoSession, - ); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, eviction.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - "failed to send unauthenticated eviction" - ); - } -} - -/// Per-shard heartbeat verifier: evict connections that have not pinged within -/// `1.2 x interval`. Mirrors the legacy `verify_heartbeats` periodic task. -/// Eviction reuses the disconnect path (drops the client from its consumer -/// groups + rebalances via the replicated `Logout`) and sends a session- -/// terminal `Eviction(StaleClient)` so the client fails fast and can reconnect. -#[allow(clippy::future_not_send)] -pub async fn run_heartbeat_verifier( - shard: Rc>, - sessions: Rc>, - interval: std::time::Duration, - stop_rx: shard::Receiver<()>, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - // Legacy `MAX_THRESHOLD`: a client is stale once it misses 1.2 intervals. - // Integer 6/5 rather than `mul_f64`, which panics on an absurd interval. - let max_age = interval.saturating_mul(6) / 5; - loop { - // `Ok(_)`: stop signalled -> exit. `Err(_)`: interval elapsed -> pass. - // Waiting on the stop channel rather than sleeping past it keeps this - // task inside the shutdown drain budget, which is shorter than the - // heartbeat interval. - let stop_signal = compio::time::timeout(interval, stop_rx.recv()).await; - if stop_signal.is_ok() { - break; - } - // Production-only wall clock: the heartbeat verifier is spawned solely - // by `build_shard_for_thread`, never by the simulator's - // `wire_shell_handlers`, so neither the interval wait above nor this - // read is on a deterministic path. Driving this task under the - // deterministic executor means routing both through the injected clock. - let stale = sessions - .borrow() - .collect_stale(max_age, std::time::Instant::now()); - for transport_client_id in stale { - // The heartbeat verifier exists to release a dead client's - // consumer-group membership (so the group rebalances off it). A - // connection that holds no membership has nothing for the eviction - // to clean up; reaping it would only drop a still-usable session - // (e.g. an idle admin connection that polls between long gaps), - // which the legacy server tolerates. The real transport-disconnect - // path still reaps it on socket close. So only evict a stale - // connection that is actually a group member. - let is_group_member = sessions - .borrow() - .bound_client_id(transport_client_id) - .is_some_and(|vsr_client_id| { - !shard - .plane - .metadata() - .mux_stm - .streams() - .consumer_group_memberships(vsr_client_id) - .is_empty() - }); - if is_group_member { - evict_stale_client(&shard, &sessions, transport_client_id).await; - } - } - } -} - -/// Evict one stale connection: drop its session (releasing consumer-group -/// membership through a replicated `Logout`) and notify the client with a -/// session-terminal `Eviction(StaleClient)`. -#[allow(clippy::future_not_send)] -async fn evict_stale_client( - shard: &Rc>, - sessions: &Rc>, - transport_client_id: u128, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let bound = sessions.borrow_mut().remove_connection(transport_client_id); - if let Some((vsr_client_id, session)) = bound { - submit_disconnect_logout(Rc::clone(shard), vsr_client_id, session); - } - let ctx = shard.plane.metadata().consensus.as_ref().map_or( - consensus::EvictionContext { - cluster: 0, - view: 0, - replica: 0, - }, - consensus::EvictionContext::from_consensus, - ); - let eviction = consensus::build_eviction_message( - ctx, - transport_client_id, - iggy_binary_protocol::EvictionReason::StaleClient, - ); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, eviction.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - "failed to send stale-client eviction" - ); - } else { - warn!( - transport_client_id, - "evicted stale client (missed heartbeat)" - ); - } -} - -/// Serve `poll_messages`: resolve the partition namespace, run the read on -/// the owning shard ([`shard::IggyShard::partition_read`]), and re-encode -/// the stored batches into the legacy wire `PolledMessages` body. -/// -/// Failures reply with an empty body so the SDK fails fast on decode -/// instead of hanging until its read timeout. -#[allow(clippy::future_not_send)] -async fn handle_poll_messages( - shard: &Rc>, - transport_client_id: u128, - request: &Message, - user_id: Option, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let Ok(wire) = PollMessagesRequest::decode_from(request_body(request)) else { - // Undecodable poll: keep the fail-fast empty-poll shape. - send_non_replicated_bytes( - shard, - request, - transport_client_id, - empty_polled_messages_body(0), - "poll_messages", - ) - .await; - return; - }; - // Gate on (stream, topic) before touching the partition plane. A resolution - // miss falls through to the resolve path below (empty-poll / not-found); a - // denial replies status!=0 with an empty body, distinct from the empty-poll - // "0 messages" shape. - if let Some(status) = authorize_partition_read( - shard, - &wire.stream_id, - &wire.topic_id, - user_id, - |permissioner, uid, stream_id, topic_id| { - permissioner.poll_messages(uid, stream_id, topic_id) - }, - ) { - send_non_replicated_deny(shard, request, transport_client_id, status).await; - return; - } - let body = match resolve_poll_request(shard, &wire, request.header().client) { - Ok((namespace, partition_id, consumer, args)) => { - match shard - .partition_read(namespace, PartitionRead::Poll { consumer, args }) - .await - { - Some(PartitionReadReply::Poll { - fragments, - current_offset, - }) => match build_polled_messages_reply( - request.header(), - current_metadata_commit(shard), - partition_id, - current_offset, - fragments, - shard.plane.partitions().config().encryptor.as_deref(), - ) { - Ok(reply) => { - send_reply_frame(shard, transport_client_id, reply, "poll_messages").await; - return; - } - Err(error) => { - warn!( - transport_client_id, - error = %error, - "failed to re-encode polled batches; replying empty poll" - ); - empty_polled_messages_body(partition_id) - } - }, - other => { - warn!( - transport_client_id, - namespace = namespace.inner(), - reply_was_none = other.is_none(), - "partition read failed; replying empty poll" - ); - empty_polled_messages_body(partition_id) - } - } - } - Err(error) => { - // A stream, topic, or partition id that does not resolve is a - // client addressing error and must surface as a typed rejection, - // not an empty poll a consumer would read as end-of-partition. - if matches!( - error, - IggyError::PartitionNotFound(..) - | IggyError::StreamIdNotFound(_) - | IggyError::TopicIdNotFound(..) - ) { - warn!( - transport_client_id, - error = %error, - "poll_messages rejected: target not found" - ); - send_non_replicated_deny(shard, request, transport_client_id, error.as_code()) - .await; - return; - } - // A zero-byte body would panic the SDK's `PolledMessages` - // decoder; reply the 16-byte empty-poll shape instead. A generation - // fence (the client's cached assignment is stale after a rebalance) - // carries the re-sync sentinel so the SDK re-syncs and retries - // rather than treating the empty poll as end-of-partition. - warn!( - transport_client_id, - error = %error, - "poll_messages request rejected; replying empty poll" - ); - let partition_id = if matches!(error, IggyError::ConsumerGroupPartitionNotOwned(..)) { - iggy_common::RESYNC_REQUIRED_PARTITION_SENTINEL - } else { - 0 - }; - empty_polled_messages_body(partition_id) - } - }; - send_non_replicated_bytes(shard, request, transport_client_id, body, "poll_messages").await; -} - -/// Serve `get_consumer_offset`. An empty body decodes as `None` on the SDK -/// side (no offset stored / partition unknown). -// TODO(hubcio): plain local partition_read with no primary gate, so a -// follower answers from its own (possibly lagging) offset state. Needs the -// same is-caught-up-primary gate the auto-commit path has, or an explicit -// read-from-follower contract. -#[allow(clippy::future_not_send)] -async fn handle_get_consumer_offset( - shard: &Rc>, - transport_client_id: u128, - request: &Message, - user_id: Option, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let Ok(wire) = GetConsumerOffsetRequest::decode_from(request_body(request)) else { - // Undecodable: an empty body decodes as None (no offset) on the SDK. - send_non_replicated_bytes( - shard, - request, - transport_client_id, - Bytes::new(), - "get_consumer_offset", - ) - .await; - return; - }; - if let Some(status) = authorize_partition_read( - shard, - &wire.stream_id, - &wire.topic_id, - user_id, - |permissioner, uid, stream_id, topic_id| { - permissioner.get_consumer_offset(uid, stream_id, topic_id) - }, - ) { - send_non_replicated_deny(shard, request, transport_client_id, status).await; - return; - } - let body = match resolve_consumer_offset_request(shard, &wire) { - Ok((namespace, partition_id, consumer)) => { - match shard - .partition_read(namespace, PartitionRead::ConsumerOffset { consumer }) - .await - { - Some(PartitionReadReply::ConsumerOffset { - stored: Some(stored_offset), - current_offset, - }) => build_consumer_offset_body(partition_id, current_offset, stored_offset), - _ => Bytes::new(), - } - } - // A partition id that does not exist in a resolvable topic is a client - // addressing error, the same one the poll path denies typed. An empty - // body decodes as `None` -- indistinguishable from "this consumer has - // no stored offset yet" -- so the caller cannot tell a typo from a - // fresh consumer. - Err(error @ IggyError::PartitionNotFound(..)) => { - warn!( - transport_client_id, - error = %error, - "get_consumer_offset rejected: partition not found" - ); - send_non_replicated_deny(shard, request, transport_client_id, error.as_code()).await; - return; - } - Err(error) => { - warn!( - transport_client_id, - error = %error, - "get_consumer_offset request rejected; replying empty" - ); - Bytes::new() - } - }; - send_non_replicated_bytes( - shard, - request, - transport_client_id, - body, - "get_consumer_offset", - ) - .await; -} - -/// Serve `SyncConsumerGroup`: return the requesting member's current partition -/// assignment + group generation so the client can select partitions locally. -/// The member is keyed by the connection's bound VSR client id -/// (`header().client`). An empty body decodes as "no assignment" on the SDK. -#[allow(clippy::future_not_send)] -async fn handle_sync_consumer_group( - shard: &Rc>, - transport_client_id: u128, - request: &Message, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let body = match SyncConsumerGroupRequest::decode_from(request_body(request)) { - Ok(wire) => shard - .plane - .metadata() - .mux_stm - .streams() - .consumer_group_member_assignment( - &wire.stream_id, - &wire.topic_id, - &wire.group_id, - request.header().client, - ) - .map_or_else(Bytes::new, |(generation, partitions)| { - SyncConsumerGroupResponse { - generation, - partitions, - } - .to_bytes() - }), - Err(error) => { - warn!( - transport_client_id, - error = %error, - "sync_consumer_group request rejected; replying empty" - ); - Bytes::new() - } - }; - send_non_replicated_bytes( - shard, - request, - transport_client_id, - body, - "sync_consumer_group", - ) - .await; -} - -/// Ack a consumer-offset op whose body could not be rewritten for the -/// partition plane with an empty Reply. The SDK connection processes replies -/// in lockstep, so a silent drop wedges every subsequent request on that -/// connection. -#[allow(clippy::future_not_send)] -async fn send_empty_partition_reply( - shard: &Rc>, - transport_client_id: u128, - request_header: &RoutedRequestHeader, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let commit = current_metadata_commit(shard); - let reply = build_empty_reply(request_header, transport_client_id, 0, commit); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - operation = ?request_header.operation, - "failed to surface empty partition reply" - ); - } -} - -/// Wait (bounded) until this shard holds a routing row for `namespace`. Fast -/// path: row already present -> no wait. -/// -/// Covers the post-`CreateTopic` convergence window where the metadata commit -/// has returned to the client but the per-shard reconcilers have not yet seeded -/// routing rows. This is an admission courtesy, not a correctness gate: the row -/// is a cache of the deterministic hash assignment and may exist before the -/// owner has materialised anything, so its presence proves only where the -/// partition belongs. What makes an early arrival safe is the owning shard -/// itself - `park_if_unmaterialised` holds the frame until its partition lands, -/// and `serves_committed_incarnation` refuses to serve a mismatched -/// incarnation. Waiting here simply keeps the steady state off that park -/// buffer, whose overflow is the one path that still sheds a request without -/// replying (`frame_drops_total{variant=partition,reason=park_overflow}`). -/// -/// Deliberately no owner-readiness probe. One used to run here, on the theory -/// that the table could not be trusted; it could not close the window either, -/// because the fast path above skipped it in exactly the case it was meant to -/// cover - a row seeded from the hash by a shard that owns nothing. Readiness -/// belongs to the owner, which is where it is now enforced. -#[allow(clippy::future_not_send)] -async fn wait_for_partition_routable( - shard: &Rc>, - namespace: IggyNamespace, -) -> bool -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - const ATTEMPT_DELAY: std::time::Duration = std::time::Duration::from_millis(50); - // 3s budget at 50ms per attempt. Counting attempts, not reading a - // wall-clock deadline, keeps the wait virtual under the simulator: the - // bus sleep advances virtual time, whereas `Instant::now` would not. - const MAX_ATTEMPTS: u32 = 60; - - let mut attempts = 0u32; - while shard.shards_table().shard_for(namespace).is_none() { - if attempts >= MAX_ATTEMPTS { - return false; - } - attempts += 1; - shard.bus.sleep(ATTEMPT_DELAY).await; - } - true -} - -/// The 16-byte `PolledMessages` body with zero messages -/// (`[partition_id:4][current_offset:8][count:4]`). The SDK decoder -/// requires at least this header, so failure paths must never reply a -/// zero-byte body. -fn empty_polled_messages_body(partition_id: u32) -> Bytes { - let mut body = Vec::with_capacity(16); - body.extend_from_slice(&partition_id.to_le_bytes()); - body.extend_from_slice(&0u64.to_le_bytes()); - body.extend_from_slice(&0u32.to_le_bytes()); - Bytes::from(body) -} - -pub type DecodedPollRequest = (IggyNamespace, u32, PollingConsumer, PollingArgs); - -/// Resolve a decoded poll request into its owning-shard read: namespace, -/// partition, polling consumer, and args. Shared by the TCP dispatch (client -/// id = the connection's bound VSR client) and the HTTP route (client id 0, -/// which fences group polls closed). -#[allow(clippy::cast_possible_truncation)] -pub fn resolve_poll_request( - shard: &Rc>, - wire: &PollMessagesRequest, - client_id: u128, -) -> Result -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let strategy = polling_strategy_from_wire(&wire.strategy)?; - let args = PollingArgs::new(strategy, wire.count, wire.auto_commit); - - // Consumer-group poll: the client selects which of its assigned partitions - // to read and sends it explicitly. The coordinator FENCES ownership (a stale - // client whose partition was reassigned is rejected with - // `ConsumerGroupPartitionNotOwned`, prompting a re-sync) and resolves the - // group's monotonic id -- the offset key the store rewrite and read path - // both use, so `next()` reads back the offset it just committed. - if wire.consumer.kind == KIND_CONSUMER_GROUP { - let partition_id = wire.partition_id.ok_or(IggyError::InvalidIdentifier)?; - let group_id = shard - .plane - .metadata() - .mux_stm - .streams() - .consumer_group_fence( - &wire.stream_id, - &wire.topic_id, - &wire.consumer.id, - client_id, - partition_id, - // Poll fence: reject a pending-revoked partition so the source - // re-syncs and skips it (it still commits it via the offset fence). - true, - ) - .ok_or(IggyError::ConsumerGroupPartitionNotOwned( - client_id as u32, - partition_id, - ))?; - let namespace = resolve_partition_namespace( - shard, - &wire.stream_id, - &wire.topic_id, - Some(partition_id), - )?; - #[allow(clippy::cast_possible_truncation)] - let consumer = PollingConsumer::ConsumerGroup(group_id as usize, partition_id as usize); - return Ok((namespace, partition_id, consumer, args)); - } - - // Plain-consumer poll: an omitted partition selects partition 0, matching - // the legacy resolver (`resolve_consumer_with_partition_id` uses - // `unwrap_or(0)` for `ConsumerKind::Consumer`). - let partition_id = wire.partition_id.unwrap_or(0); - let namespace = - resolve_partition_namespace(shard, &wire.stream_id, &wire.topic_id, Some(partition_id))?; - let consumer = polling_consumer_from_wire(&wire.consumer, partition_id)?; - Ok((namespace, partition_id, consumer, args)) -} - -/// Resolve a decoded consumer-offset read into its owning-shard read: -/// namespace, partition, and polling consumer. Shared by the TCP dispatch and -/// the HTTP route; needs no client id because offset reads are not fenced -/// (any client may read a group's offset, member or not). -pub fn resolve_consumer_offset_request( - shard: &Rc>, - wire: &GetConsumerOffsetRequest, -) -> Result<(IggyNamespace, u32, PollingConsumer), IggyError> -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - // Omitted partition reads partition 0, matching the legacy resolver for - // both consumer kinds (`unwrap_or(0)`). - let partition_id = wire.partition_id.unwrap_or(0); - let namespace = - resolve_partition_namespace(shard, &wire.stream_id, &wire.topic_id, Some(partition_id))?; - // A group offset is keyed by the group's monotonic id (any client may read - // it, member or not), the same key the write path is rewritten to. An - // unresolved group (e.g. deleted) has no offset, so the read reports None. - let consumer = if wire.consumer.kind == KIND_CONSUMER_GROUP { - let group_id = shard - .plane - .metadata() - .mux_stm - .streams() - .resolve_consumer_group_id(&wire.stream_id, &wire.topic_id, &wire.consumer.id) - .ok_or(IggyError::InvalidIdentifier)?; - #[allow(clippy::cast_possible_truncation)] - PollingConsumer::ConsumerGroup(group_id as usize, partition_id as usize) - } else { - polling_consumer_from_wire(&wire.consumer, partition_id)? - }; - Ok((namespace, partition_id, consumer)) -} - -fn polling_consumer_from_wire( - consumer: &WireConsumer, - partition_id: u32, -) -> Result { - // Mirrors the legacy server's `PollingConsumer::resolve_consumer_id`: - // numeric ids pass through, named consumers hash to a stable u32 so - // reads derive the same offset-table key the write path stores under. - let consumer_id = match &consumer.id { - iggy_binary_protocol::WireIdentifier::Numeric(id) => *id, - iggy_binary_protocol::WireIdentifier::String(name) => { - iggy_common::calculate_32(name.as_str().as_bytes()) - } - } as usize; - match consumer.kind { - 1 => Ok(PollingConsumer::Consumer( - consumer_id, - partition_id as usize, - )), - KIND_CONSUMER_GROUP => Ok(PollingConsumer::ConsumerGroup( - consumer_id, - partition_id as usize, - )), - _ => Err(IggyError::InvalidCommand), - } -} - -fn polling_strategy_from_wire( - strategy: &WirePollingStrategy, -) -> Result { - let mut mapped = match strategy.kind { - 1 => PollingStrategy::offset(0), - 2 => PollingStrategy::timestamp(iggy_common::IggyTimestamp::from(strategy.value)), - 3 => PollingStrategy::first(), - 4 => PollingStrategy::last(), - 5 => PollingStrategy::next(), - _ => return Err(IggyError::InvalidCommand), - }; - mapped.set_value(strategy.value); - Ok(mapped) -} - -/// Answer a backup's forwarded `Register` from the node it named primary. -/// -/// Proposes in process, never through [`submit_register_local_or_forward`]: -/// that is what bounds a forward at one hop. A node that has since lost -/// primaryship answers `NotPrimary`, and the origin's client replays against -/// whichever node it names next. -#[allow(clippy::future_not_send)] -async fn answer_forwarded_register( - shard: &Rc>, - vsr_client_id: u128, - user_id: u32, - nonce: u128, - origin_replica: u8, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let Some((cluster, view, replica)) = shard - .plane - .metadata() - .consensus - .as_ref() - .map(|consensus| (consensus.cluster(), consensus.view(), consensus.replica())) - else { - warn!("ForwardedRegister submit reached a shard without metadata consensus"); - return; - }; - let bound = shard - .plane - .metadata() - .submit_register_in_process(vsr_client_id, user_id) - .await; - // `view` predates the await above, which parks with no deadline, so the - // sealed value can be stale by send time. The origin routes the result by - // `(nonce, client)` alone; this field must never become a freshness fence. - let result = - build_forward_register_result_message(cluster, view, replica, vsr_client_id, nonce, &bound); - if let Err(error) = shard - .bus - .send_to_replica(origin_replica, result.into_generic().into_frozen()) - .await - { - warn!( - origin_replica, - error = %error, - "failed to answer a forwarded register" - ); - } -} - -/// How long a login waits for the primary's verdict on a forwarded register. -/// -/// Expiry does NOT prove the peer or the frame was lost. The primary answers -/// only once the proposal resolves, and its own submit parks with no deadline: -/// a primary that is not caught up, or whose pipeline is full, absorbs the -/// register into its request queue and answers when that drains. So a slow but -/// healthy primary commits the register after this node has stopped waiting, -/// which is why expiry surfaces as `TransientNotCommitted` rather than the -/// not-accepted flavor. -/// -/// The budget stays well under the SDK's response-read timeout on purpose: the -/// client only replays a login while it is still reading, so a longer wait -/// here turns a transient into a torn-down socket. -const FORWARD_SUBMIT_TIMEOUT: Duration = Duration::from_secs(5); - -/// Run the `Register` proposal for a login this node has already -/// authenticated, wherever the metadata primary currently is. Shard 0 only. -/// -/// A client may dial any node in the cluster. Credentials verify against the -/// replicated users table, which every node holds, so the whole login except -/// the consensus proposal already works on a backup. Only the verified -/// identity crosses the replica interconnect -- never the client's frame and -/// never its credentials -- and the session bind, the reply, and the -/// connection all stay on the node the client dialed. -/// -/// The hop does not move any credential decision: -/// - `verify_login_credentials` reads the backup's applied replicated user -/// state. -/// - `verify_pat_credentials` reads the same state, so a PAT minted on the -/// primary that has not replicated here yet is refused until it does. -/// Fail-closed on purpose, the same parity the HTTP forward keeps: it too -/// answers 401 until replication catches up rather than relaying an -/// unverified bearer. -/// - `ClientIdOwnedByAnotherUser` stays a decision of the caught-up primary -/// and round-trips as a terminal refusal. -/// -/// Verification is point-in-time on the backup. A password change, PAT -/// revocation, or user deactivation committed on the primary but not yet -/// applied on the backup can therefore admit a login during the backup's apply -/// lag. The forward cannot complete while the backup is partitioned from the -/// primary, which bounds this to a connected replica's replication lag. This -/// is the same stale-read window as the existing HTTP forward. -/// -/// The session binds here before this node applies the commit locally. That -/// is the window a primary-side login already has against every other node's -/// apply lag, not a new one. -#[allow(clippy::future_not_send)] -async fn submit_register_local_or_forward( - shard: &Rc>, - vsr_client_id: u128, - user_id: u32, -) -> Result -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let Some(consensus) = shard.plane.metadata().consensus.as_ref() else { - return Err(MetadataSubmitError::NotPrimary); - }; - let (cluster, view, self_replica) = - (consensus.cluster(), consensus.view(), consensus.replica()); - let target = consensus.primary_index(view); - // Forward only as a healthy backup. Everything else answers locally: the - // in-process submit proposes when this node is the serving primary and - // re-derives `NotPrimary` otherwise -- mid view change there is nobody to - // forward to (the node the view names has not finished taking over, the - // SDK replays once it settles), and the view's own primary under state - // transfer has nowhere to forward to and nothing to commit yet. - if target == self_replica || !consensus.is_normal() { - return shard - .plane - .metadata() - .submit_register_in_process(vsr_client_id, user_id) - .await; - } - - let nonce = shard.next_forward_nonce(self_replica); - let (reply, outcome) = shard::channel::(1); - shard.park_register_forward(nonce, vsr_client_id, reply); - let forward = - build_forward_register_message(cluster, view, self_replica, vsr_client_id, nonce, user_id); - if let Err(error) = shard - .bus - .send_to_replica(target, forward.into_generic().into_frozen()) - .await - { - shard.cancel_register_forward(nonce, vsr_client_id); - warn!( - target, - error = %error, - "failed to forward register to the metadata primary" - ); - return Err(MetadataSubmitError::PrimaryUnreachable); - } - - match shard::bus_timeout(&shard.bus, FORWARD_SUBMIT_TIMEOUT, outcome.recv()).await { - Some(Ok(result)) => forward_register_result(&result), - // Shard-0 teardown dropped the sender without answering. - Some(Err(_)) => Err(MetadataSubmitError::Canceled), - None => { - shard.cancel_register_forward(nonce, vsr_client_id); - warn!(target, "forwarded register timed out"); - Err(MetadataSubmitError::ForwardTimedOut) - } - } -} - -/// The primary's verdict, back in the vocabulary the login path speaks. -const fn forward_register_result( - result: &ForwardRegisterResultHeader, -) -> Result { - match result.outcome { - ForwardRegisterOutcome::Ok => Ok(BoundSession { - epoch: result.epoch, - watermark: result.watermark, - }), - ForwardRegisterOutcome::NotPrimary => Err(MetadataSubmitError::NotPrimary), - ForwardRegisterOutcome::NotCaughtUp => Err(MetadataSubmitError::NotCaughtUp), - ForwardRegisterOutcome::PipelineFull => Err(MetadataSubmitError::PipelineFull), - ForwardRegisterOutcome::InProgress => Err(MetadataSubmitError::InProgress), - ForwardRegisterOutcome::Canceled => Err(MetadataSubmitError::Canceled), - ForwardRegisterOutcome::ClientIdOwnedByAnotherUser => { - Err(MetadataSubmitError::ClientIdOwnedByAnotherUser) - } - } -} - -/// Inverse of [`forward_register_result`], for the answering primary. -const fn forward_register_outcome( - bound: &Result, -) -> (BoundSession, ForwardRegisterOutcome) { - let zero = BoundSession { - epoch: 0, - watermark: 0, - }; - match bound { - Ok(bound) => (*bound, ForwardRegisterOutcome::Ok), - Err(MetadataSubmitError::NotPrimary) => (zero, ForwardRegisterOutcome::NotPrimary), - Err(MetadataSubmitError::NotCaughtUp) => (zero, ForwardRegisterOutcome::NotCaughtUp), - Err(MetadataSubmitError::PipelineFull) => (zero, ForwardRegisterOutcome::PipelineFull), - Err(MetadataSubmitError::InProgress) => (zero, ForwardRegisterOutcome::InProgress), - Err(MetadataSubmitError::ClientIdOwnedByAnotherUser) => { - (zero, ForwardRegisterOutcome::ClientIdOwnedByAnotherUser) - } - // `MetadataSubmitError` is `#[non_exhaustive]`. Every variant but the - // ownership refusal is transient by contract, and `Canceled` is the - // transient answer that claims nothing beyond "retry". - Err(_) => (zero, ForwardRegisterOutcome::Canceled), - } -} - -#[allow(clippy::cast_possible_truncation)] -fn build_forward_register_message( - cluster: u128, - view: u32, - replica: u8, - client: u128, - nonce: u128, - user_id: u32, -) -> Message { - Message::::new(HEADER_SIZE).transmute_header( - |_, header: &mut ForwardRegisterHeader| { - header.command = Command::ForwardRegister; - header.cluster = cluster; - header.view = view; - header.replica = replica; - header.client = client; - header.nonce = nonce; - header.user_id = user_id; - header.size = HEADER_SIZE as u32; - header.seal(); - }, - ) -} - -#[allow(clippy::cast_possible_truncation)] -fn build_forward_register_result_message( - cluster: u128, - view: u32, - replica: u8, - client: u128, - nonce: u128, - bound: &Result, -) -> Message { - let (session, outcome) = forward_register_outcome(bound); - Message::::new(HEADER_SIZE).transmute_header( - |_, header: &mut ForwardRegisterResultHeader| { - header.command = Command::ForwardRegisterResult; - header.cluster = cluster; - header.view = view; - header.replica = replica; - header.client = client; - header.nonce = nonce; - header.epoch = session.epoch; - header.watermark = session.watermark; - header.outcome = outcome; - header.size = HEADER_SIZE as u32; - header.seal(); - }, - ) -} - -/// Answer a backup's forwarded Logout from the node it named primary. -#[allow(clippy::future_not_send)] -async fn answer_forwarded_logout( - shard: &Rc>, - vsr_client_id: u128, - session: u64, - request: u64, - nonce: u128, - origin_replica: u8, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let Some((cluster, view, replica)) = shard - .plane - .metadata() - .consensus - .as_ref() - .map(|consensus| (consensus.cluster(), consensus.view(), consensus.replica())) - else { - warn!("ForwardedLogout submit reached a shard without metadata consensus"); - return; - }; - let outcome = shard - .plane - .metadata() - .submit_logout_in_process(vsr_client_id, session, request) - .await; - let result = - build_forward_logout_result_message(cluster, view, replica, vsr_client_id, nonce, &outcome); - if let Err(error) = shard - .bus - .send_to_replica(origin_replica, result.into_generic().into_frozen()) - .await - { - warn!( - origin_replica, - error = %error, - "failed to answer a forwarded logout" - ); - } -} - -/// Commit a Logout locally when this node is primary, otherwise forward it -/// once to the primary named by the current normal view. -#[allow(clippy::future_not_send)] -async fn submit_logout_local_or_forward( - shard: &Rc>, - vsr_client_id: u128, - session: u64, - request: u64, -) -> Result -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let Some(consensus) = shard.plane.metadata().consensus.as_ref() else { - return Err(MetadataSubmitError::NotPrimary); - }; - let (cluster, view, self_replica) = - (consensus.cluster(), consensus.view(), consensus.replica()); - let target = consensus.primary_index(view); - if target == self_replica || !consensus.is_normal() { - return shard - .plane - .metadata() - .submit_logout_in_process(vsr_client_id, session, request) - .await; - } - - let nonce = shard.next_forward_nonce(self_replica); - let (reply, outcome) = shard::channel::(1); - shard.park_logout_forward(nonce, vsr_client_id, reply); - let forward = build_forward_logout_message( - cluster, - view, - self_replica, - vsr_client_id, - nonce, - session, - request, - ); - if let Err(error) = shard - .bus - .send_to_replica(target, forward.into_generic().into_frozen()) - .await - { - shard.cancel_logout_forward(nonce, vsr_client_id); - warn!( - target, - error = %error, - "failed to forward logout to the metadata primary" - ); - return Err(MetadataSubmitError::PrimaryUnreachable); - } - - match shard::bus_timeout(&shard.bus, FORWARD_SUBMIT_TIMEOUT, outcome.recv()).await { - Some(Ok(result)) => forward_logout_result(&result), - Some(Err(_)) => Err(MetadataSubmitError::Canceled), - None => { - shard.cancel_logout_forward(nonce, vsr_client_id); - warn!(target, "forwarded logout timed out"); - Err(MetadataSubmitError::ForwardTimedOut) - } - } -} - -const fn forward_logout_result( - result: &ForwardLogoutResultHeader, -) -> Result { - match result.outcome { - ForwardLogoutOutcome::Ok => Ok(result.commit), - ForwardLogoutOutcome::NotPrimary => Err(MetadataSubmitError::NotPrimary), - ForwardLogoutOutcome::PipelineFull => Err(MetadataSubmitError::PipelineFull), - ForwardLogoutOutcome::InProgress => Err(MetadataSubmitError::InProgress), - ForwardLogoutOutcome::Canceled => Err(MetadataSubmitError::Canceled), - } -} - -const fn forward_logout_outcome( - outcome: &Result, -) -> (u64, ForwardLogoutOutcome) { - match outcome { - Ok(commit) => (*commit, ForwardLogoutOutcome::Ok), - Err(MetadataSubmitError::NotPrimary) => (0, ForwardLogoutOutcome::NotPrimary), - Err(MetadataSubmitError::PipelineFull) => (0, ForwardLogoutOutcome::PipelineFull), - Err(MetadataSubmitError::InProgress) => (0, ForwardLogoutOutcome::InProgress), - Err(_) => (0, ForwardLogoutOutcome::Canceled), - } -} - -#[allow(clippy::cast_possible_truncation, clippy::too_many_arguments)] -fn build_forward_logout_message( - cluster: u128, - view: u32, - replica: u8, - client: u128, - nonce: u128, - session: u64, - request: u64, -) -> Message { - Message::::new(HEADER_SIZE).transmute_header( - |_, header: &mut ForwardLogoutHeader| { - header.command = Command::ForwardLogout; - header.cluster = cluster; - header.view = view; - header.replica = replica; - header.client = client; - header.nonce = nonce; - header.session = session; - header.request = request; - header.size = HEADER_SIZE as u32; - header.seal(); - }, - ) -} - -#[allow(clippy::cast_possible_truncation)] -fn build_forward_logout_result_message( - cluster: u128, - view: u32, - replica: u8, - client: u128, - nonce: u128, - result: &Result, -) -> Message { - let (commit, outcome) = forward_logout_outcome(result); - Message::::new(HEADER_SIZE).transmute_header( - |_, header: &mut ForwardLogoutResultHeader| { - header.command = Command::ForwardLogoutResult; - header.cluster = cluster; - header.view = view; - header.replica = replica; - header.client = client; - header.nonce = nonce; - header.commit = commit; - header.outcome = outcome; - header.size = HEADER_SIZE as u32; - header.seal(); - }, - ) -} - -/// Run the consensus `Register` proposal on the metadata owner (shard 0) -/// and return the committed session. -/// -/// Credential verification and session binding stay on the calling (home) -/// shard -- only this consensus step must execute where the metadata -/// consensus group lives. On shard 0 it goes straight to -/// [`submit_register_local_or_forward`]; on a peer it forwards a -/// [`shard::MetadataSubmit`] to shard 0 and awaits the committed op. A dropped -/// reply (shard-0 inbox full / shutdown) maps to a transient `Canceled`, which -/// the caller wraps so the SDK replays. -#[allow(clippy::future_not_send)] -pub async fn submit_register_on_owner( - shard: &Rc>, - vsr_client_id: u128, - user_id: u32, -) -> Result -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - if shard.id == 0 { - return submit_register_local_or_forward(shard, vsr_client_id, user_id).await; - } - let (reply, rx) = shard::channel::>(1); - shard.forward_metadata_submit(shard::MetadataSubmit::Register { - vsr_client_id, - user_id, - reply, - }); - // The owner's outcome, verbatim in both directions. `Canceled` is only for a - // dropped channel, where nothing came back to classify. - rx.recv() - .await - .unwrap_or(Err(MetadataSubmitError::Canceled)) -} - -/// Logout counterpart of [`submit_register_on_owner`]. -#[allow(clippy::future_not_send)] -pub async fn submit_logout_on_owner( - shard: &Rc>, - vsr_client_id: u128, - session: u64, - request: u64, -) -> Result -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - if shard.id == 0 { - return submit_logout_local_or_forward(shard, vsr_client_id, session, request).await; - } - let (reply, rx) = shard::channel::>(1); - shard.forward_metadata_submit(shard::MetadataSubmit::Logout { - vsr_client_id, - session, - request, - reply, - }); - rx.recv() - .await - .unwrap_or(Err(MetadataSubmitError::Canceled)) -} - -/// Handle a client `DeleteSegments`: resolve the requested count to an offset -/// on the owning shard, replicate a `TruncatePartition` through metadata so -/// every replica trims to the same watermark, then ack the client. The local -/// deletion happens later, when each replica's reconciler observes the commit. -/// -/// The consensus reply is forwarded verbatim: nothing-to-delete commits a -/// no-op `TruncatePartition(0)` and acks, while a not-primary rejection -/// reaches the client as `TransientNotCommitted` so the SDK replays instead -/// of mistaking a dropped delete for success. Only a malformed / unresolvable -/// request is acked empty without a commit. -#[allow(clippy::future_not_send)] -async fn handle_delete_segments_request( - shard: &Rc>, - transport_client_id: u128, - bound: Option<(u128, u64)>, - request: &Message, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let header = *request.header(); - let body = request_body(request); - - // An unbound transport cannot be attributed a VSR request sequence; the - // outer handler already short-circuits these, so this is defensive. - let Some((vsr_client_id, session)) = bound else { - return; - }; - - // The client numbers DeleteSegments in the same monotonic request sequence - // as every other metadata op. So resolve the requested count to a concrete - // offset on the owning shard, then replicate a `TruncatePartition(offset)` - // AS the client's own request through the standard owner path: the commit - // records (client, session, request) in the `ClientTable` on every replica, - // advancing the watermark. Skipping the commit (or attributing it to an - // internal id) leaves this request id unrecorded, so the SDK's own retry - // of it would re-execute instead of deduping. A no-op delete still - // commits `up_to_offset = 0` (monotonic apply) for the same reason. - let truncate = match resolve_delete_segments_truncate( - shard, - &header, - vsr_client_id, - session, - body, - ) - .await - { - Ok(truncate) => Some(truncate), - // The owning partition has not converged on the committed log yet, so - // the delete cannot be resolved to a watermark. Reply with the - // result-framed transient rejection (under the TruncatePartition - // operation, which the SDK decodes) so the client replays the same - // request once the partition catches up. Nothing was submitted, hence - // the re-issuable-anywhere flavor. - Err(IggyError::TransientNotAccepted) => { - let template = build_truncate_partition_client_message( - &header, - vsr_client_id, - session, - 0, - 0, - 0, - 0, - ); - let reply = build_result_rejection_reply( - template.header(), - current_metadata_commit(shard), - IggyError::TransientNotAccepted.as_code(), - ); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - "delete_segments: failed to send transient rejection" - ); - } - return; - } - Err(_) => None, - }; - - let reply = if let Some(truncate) = truncate { - // Forward the consensus reply verbatim, exactly like the generic - // metadata path: a committed success acks the delete, and a - // result-framed `TransientNotCommitted` rejection makes the SDK - // replay the request. Acking unconditionally here would swallow a - // not-primary rejection and drop the delete on the floor while the - // client believes it succeeded. - let Some(reply) = submit_client_request_on_owner(shard, truncate).await else { - // Transient submit failure (not primary / view change). Stay - // silent; the SDK read-timeout replays the same request id, - // which re-resolves and commits. Acking here would advance the - // client past an unrecorded request and gap the next metadata - // op. - warn!( - transport_client_id, - "delete_segments: transient submit; client will replay" - ); - return; - }; - reply - } else { - // Undecodable body (never produced by the SDK): ack empty so the - // lockstep stream stays framed; the typed decoder surfaces the - // failure client-side. Unresolvable-but-well-formed targets commit a - // typed rejection instead (see the resolve), so only a wire-corrupt - // request can gap the sequence here. - let commit = current_metadata_commit(shard); - build_empty_reply(&header, transport_client_id, session, commit).into_generic() - }; - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - "delete_segments: failed to send reply" - ); - } -} - -/// Resolve a client `DeleteSegments` to the `TruncatePartition` that commits the -/// trim. Shared by the TCP dispatch and the HTTP listener so both resolve the -/// requested segment count to a concrete watermark identically. -/// -/// `template` supplies the wire `cluster` / `view` / `release` and the client's -/// `request` number; `client_id` / `session` are the bound VSR identity the -/// truncate commits under. A resolvable namespace with nothing sealed to delete -/// still yields a `TruncatePartition(up_to_offset = 0)` so the metadata request -/// sequence stays contiguous. `Err` on a malformed body or an unresolved -/// namespace: the TCP caller drops it to a silent replay, the HTTP caller renders -/// the error. -#[allow(clippy::future_not_send)] -#[allow(clippy::cast_possible_truncation)] -pub async fn resolve_delete_segments_truncate( - shard: &Rc>, - template: &RoutedRequestHeader, - client_id: u128, - session: u64, - body: &[u8], -) -> Result, IggyError> -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let parsed = DeleteSegmentsRequest::decode_from(body).map_err(|_| IggyError::InvalidCommand)?; - let namespace_raw = match resolve_partition_request_namespace( - shard, - Operation::DeleteSegments, - body, - client_id, - ) { - Ok(namespace_raw) => namespace_raw, - // Unresolvable stream/topic: still commit the truncate, against the - // client's raw identifiers -- the apply rejects it as a committed - // result, so the failure is recorded against the client's request id - // and its retry dedups, while the client gets the typed error an - // empty ack would swallow. - Err(error) => { - debug!( - client_id, - %error, - "delete_segments: unresolved target; committing typed rejection" - ); - return Ok(build_truncate_partition_client_message_with_identifiers( - template, - client_id, - session, - parsed.stream_id, - parsed.topic_id, - parsed.partition_id, - 0, - )); - } - }; - let namespace = IggyNamespace::from_raw(namespace_raw); - let up_to_offset = match shard - .partition_read( - namespace, - PartitionRead::ResolveSegmentDeleteOffset { - count: parsed.segments_count, - }, - ) - .await - { - Some(PartitionReadReply::SegmentDeleteOffset { - up_to_offset: Some(offset), - .. - }) => offset, - // Nothing sealed to delete on a replica that has not converged on the - // replicated log (a backup behind the commit frontier may be missing - // whole sealed segments). Answering now would commit a no-op truncate - // and silently drop the delete, so surface a transient and let the - // client replay once the partition catches up. A converged primary - // whose resident tail is merely unflushed settles as a no-op below. - Some(PartitionReadReply::SegmentDeleteOffset { - up_to_offset: None, - lagging: true, - }) => { - debug!( - client_id, - namespace_raw, "delete_segments: partition not converged; transient" - ); - return Err(IggyError::TransientNotAccepted); - } - other => { - debug!( - client_id, - namespace_raw, - reply = ?other, - "delete_segments: nothing to delete; committing no-op truncate" - ); - 0 - } - }; - Ok(build_truncate_partition_client_message( - template, - client_id, - session, - namespace.stream_id() as u32, - namespace.topic_id() as u32, - namespace.partition_id() as u32, - up_to_offset, - )) -} - -/// Release the client-table slot for a disconnected transport, cluster-wide. -/// -/// The local `SessionManager` connection is already dropped by the caller; -/// this is what drops the replicated entry, so a peer replica does not keep an -/// orphaned session until it evicts one under capacity pressure. -/// -/// Unconditional, and deliberately so. Holding the slot open for a grace -/// window would let a reconnecting client resume onto its entry with its -/// watermark and reply ring intact, but nothing in tree re-presents a -/// `client_id` after a disconnect (the Rust SDK mints a fresh one on -/// re-login), so the window buys nothing today and the slot it holds is not -/// free: the client table's eviction point moves from concurrent connections -/// to CUMULATIVE connects, and every capacity eviction silently erases a -/// dedup watermark. -/// -/// A resume window becomes worth having once SDK-side identity stability -/// lands, at which point it needs a timer of its own -- riding the heartbeat -/// verifier would tie the grace period to heartbeat configuration, since -/// `collect_stale` keys off `heartbeat.interval` and the verifier does not run -/// at all when `heartbeat.enabled` is false. -/// Deliberately does NOT drop the local `ClientTable` slot first: -/// `submit_logout_*` short-circuits when the slot is already gone, so a -/// pre-emptive local removal would suppress the `Logout` and leave peer -/// replicas with an orphaned session until they evict it themselves -- the -/// exact divergence this avoids. `submit_logout_on_owner` runs in-process on -/// shard 0 and forwards for peer-homed connections; its session guard drops a -/// stale logout for a reused client id. -#[allow(clippy::future_not_send)] -fn submit_disconnect_logout( - shard: Rc>, - vsr_client_id: u128, - session: u64, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - // The sentinel request id is what the apply path reads to keep, rather - // than drop, the session's dedup fence: the client may be reconnecting - // under the same key, and its retry must still be answered. - let bus = shard.bus.clone(); - bus.spawn(async move { - if let Err(error) = - submit_logout_on_owner(&shard, vsr_client_id, session, DISCONNECT_LOGOUT_REQUEST_ID) - .await - { - warn!( - vsr_client_id, - ?error, - "disconnect logout submit failed; peer slots may linger until eviction" - ); - } - }); -} - -/// Submit a replicated client request to the metadata owner (shard 0) and -/// return the committed reply. -/// -/// The metadata consensus group lives on shard 0, but the connection lives -/// on the home shard (this shard). Run consensus where it belongs and bring -/// the committed reply back here so the caller can write it to the -/// originating socket -- shard 0 cannot route the reply by the consensus -/// `client` id (it's the VSR id, not the transport/home-shard-encoding id). -/// `None` = transient submit failure (SDK read-timeout replays). -#[allow(clippy::future_not_send)] -pub async fn submit_client_request_on_owner( - shard: &Rc>, - request: Message, -) -> Option> -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - if shard.id == 0 { - return shard - .plane - .metadata() - .submit_request_in_process(request) - .await - .ok(); - } - let (reply, rx) = shard::channel::>>(1); - shard.forward_metadata_submit(shard::MetadataSubmit::ClientRequest { - request: request.into_generic(), - reply, - }); - rx.recv().await.ok().flatten() -} - -#[allow(clippy::future_not_send)] -async fn handle_logout_request( - shard: &Rc>, - sessions: &Rc>, - transport_client_id: u128, - request: Message, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let Some((vsr_client_id, session)) = sessions.borrow().get_session(transport_client_id) else { - // Logout on an unbound transport: the desired state already holds, - // so answer ok. A silent drop would wedge the lockstep SDK on this - // connection until its socket read timeout, and the SDK routinely - // sends a logout before each re-login. - warn!( - transport_client_id, - "logout for unbound VSR session; answering ok" - ); - let commit = current_metadata_commit(shard); - let reply = build_empty_reply(request.header(), transport_client_id, 0, commit); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - "failed to send unbound logout reply" - ); - } - return; - }; - - let request_id = request.header().request; - let commit = match submit_logout_on_owner(shard, vsr_client_id, session, request_id).await { - Ok(commit) => commit, - Err(error) => { - // Deny as transient instead of dropping the frame: the submit - // usually fails because this replica is not the metadata owner - // right now, and the SDK replays a transient rejection. - warn!(transport_client_id, error = %error, "logout/unregister failed; denying transient"); - let commit = current_metadata_commit(shard); - let reply = build_deny_reply( - request.header(), - vsr_client_id, - session, - commit, - transient_logout_code(&error).as_code(), - ); - if let Err(send_error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %send_error, - "failed to send logout deny reply" - ); - } - return; - } - }; - - sessions.borrow_mut().remove_connection(transport_client_id); - - let reply = build_empty_reply(request.header(), vsr_client_id, session, commit); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - "failed to send logout reply" - ); - } -} - -/// Preserve the client identity when a Logout may already have entered the -/// primary's pipeline. Moving an unknown-outcome replay to another connection -/// could race a later Register and obscure whether the old epoch was removed. -const fn transient_logout_code(error: &MetadataSubmitError) -> IggyError { - match error { - MetadataSubmitError::ForwardTimedOut - | MetadataSubmitError::InProgress - | MetadataSubmitError::Canceled => IggyError::TransientNotCommitted, - _ => IggyError::TransientNotAccepted, - } -} - -fn ensure_transport_connection( - shard: &Rc>, - sessions: &Rc>, - transport_client_id: u128, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let Some(meta) = shard.bus.client_meta(transport_client_id) else { - return; - }; - sessions - .borrow_mut() - .ensure_connection(transport_client_id, meta.peer_addr, meta.transport); -} - -#[allow(clippy::future_not_send, clippy::too_many_lines)] -async fn handle_login_register_request( - shard: &Rc>, - sessions: &Rc>, - transport_client_id: u128, - request: Message, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let body = request_body(&request); - let vsr_client_id = request.header().client; - - // Both login-register shapes share the ClientVersionInfo prefix, so the - // protocol gate decodes it once and runs before any credential work; the - // body shapes below parse from past the prefix. Only VSR clients reach - // this gate -- legacy SDKs use LOGIN_USER_CODE, a separate path. A - // pre-versioning VSR client sends the old prefix-less body, which fails - // ClientVersionInfo::decode (-> MalformedLogin) or the version gate - // (-> IncompatibleProtocol) right here, not dropped earlier. - let Ok((version_info, prefix_len)) = ClientVersionInfo::decode(body) else { - warn!( - transport_client_id, - "rejecting login: body has no decodable version prefix" - ); - send_login_eviction( - shard, - transport_client_id, - vsr_client_id, - EvictionReason::MalformedLogin, - ) - .await; - return; - }; - if !is_protocol_compatible(version_info.protocol_version) { - warn!( - transport_client_id, - client_protocol_version = %ProtocolVersion(version_info.protocol_version), - sdk_name = %version_info.sdk_name, - sdk_version = %version_info.sdk_version, - "rejecting login: incompatible protocol version" - ); - send_login_eviction( - shard, - transport_client_id, - vsr_client_id, - EvictionReason::IncompatibleProtocol, - ) - .await; - return; - } - - let body_tail = &body[prefix_len..]; - let mut credentials_rejected = false; - if let Ok((wire_request, _)) = - LoginRegisterRequest::decode_after_prefix(version_info.clone(), body_tail) - { - match verify_login_credentials( - shard, - wire_request.username.as_str(), - wire_request.password.expose_secret(), - ) { - Ok(user_id) => { - if let Err(error) = complete_login_register( - shard, - sessions, - transport_client_id, - vsr_client_id, - request.header(), - user_id, - &wire_request.version_info, - ) - .await - { - warn!(transport_client_id, error = %error, "login/register failed"); - surface_login_failure(shard, transport_client_id, request.header(), &error) - .await; - } - return; - } - Err(LoginRegisterError::InvalidCredentials) => { - // Fall through to PAT attempt so a credential payload that - // collides with a valid PAT payload shape still gets a - // chance. A password-shaped body rarely parses as a PAT - // body, so remember the rejection: the final fall-through - // must surface InvalidCredentials, not MalformedLogin. - credentials_rejected = true; - } - Err(error) => { - warn!(transport_client_id, error = %error, "login/register failed"); - surface_login_failure(shard, transport_client_id, request.header(), &error).await; - return; - } - } - } - - if let Ok((wire_request, _)) = - LoginRegisterWithPatRequest::decode_after_prefix(version_info, body_tail) - { - match verify_pat_credentials(shard, wire_request.token.expose_secret()) { - Ok(user_id) => { - if let Err(error) = complete_login_register( - shard, - sessions, - transport_client_id, - vsr_client_id, - request.header(), - user_id, - &wire_request.version_info, - ) - .await - { - warn!( - transport_client_id, - error = %error, - "login/register with PAT failed" - ); - surface_login_failure(shard, transport_client_id, request.header(), &error) - .await; - } - return; - } - Err(error) => { - warn!( - transport_client_id, - error = %error, - "login/register with PAT failed" - ); - surface_login_failure(shard, transport_client_id, request.header(), &error).await; - return; - } - } - } - - if credentials_rejected { - warn!( - transport_client_id, - "rejecting register request: invalid credentials" - ); - send_login_eviction( - shard, - transport_client_id, - request.header().client, - EvictionReason::InvalidCredentials, - ) - .await; - return; - } - - warn!( - transport_client_id, - "rejecting register request with unsupported payload shape" - ); - send_login_eviction( - shard, - transport_client_id, - request.header().client, - EvictionReason::MalformedLogin, - ) - .await; -} - -/// Best-effort login-rejection eviction. Terminal one-way frame; a gone -/// connection has nothing to recover, so the send error is logged and -/// dropped. Consensus context (cluster/view/replica) is stamped on the -/// metadata shard and zeroed elsewhere -- the SDK only reads the reason, -/// plus the protocol window on `IncompatibleProtocol`. -#[allow(clippy::future_not_send)] -pub async fn send_login_eviction( - shard: &Rc>, - transport_client_id: u128, - vsr_client_id: u128, - reason: EvictionReason, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let ctx = shard.plane.metadata().consensus.as_ref().map_or( - EvictionContext { - cluster: 0, - view: 0, - replica: 0, - }, - EvictionContext::from_consensus, - ); - let eviction = match reason { - EvictionReason::IncompatibleProtocol => { - build_incompatible_protocol_eviction_message(ctx, vsr_client_id) - } - _ => build_eviction_message(ctx, vsr_client_id, reason), - }; - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, eviction.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - reason = ?reason, - "failed to send login eviction" - ); - } -} - -pub fn upgrade_shard_handle( - shard_handle: &ShellShardHandle, -) -> Option>> -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - shard_handle - .borrow() - .as_ref() - .and_then(std::rc::Weak::upgrade) -} - -#[cfg(test)] -mod tests { - use super::*; - use consensus::{LocalPipeline, Plane as _, PlaneKind, VsrConsensus}; - use iggy_binary_protocol::primitives::partition_assignment::CreatedPartitionAssignment; - use iggy_binary_protocol::requests::messages::SendMessagesHeader; - use iggy_binary_protocol::requests::streams::CreateStreamRequest; - use iggy_binary_protocol::requests::topics::CreateTopicWithAssignmentsRequest; - use iggy_binary_protocol::{PrepareOkHeader, ReplyHeader, WireName, WirePartitioning}; - use iggy_common::defaults::DEFAULT_ROOT_USER_ID; - use iggy_common::variadic; - use journal::prepare_journal::PrepareJournal; - use message_bus::BusMessage; - use message_bus::client_listener::RequestHandler; - use message_bus::fd_transfer::DupedFd; - use message_bus::installer::ConnectionInstaller; - use message_bus::installer::conn_info::{ClientConnMeta, ClientTransportKind}; - use message_bus::replica::listener::MessageHandler; - use message_bus::{ - ClientConnectionLostFn, ClientForwardFn, ConnectionLostFn, JoinHandle, MessageBus, - ReplicaForwardFn, ReplicaHandshakeDoneFn, SendError, - }; - use metadata::impls::metadata::IggySnapshot; - use metadata::stm::StateMachine as _; - use metadata::stm::stream::Streams; - use metadata::stm::user::Users; - use metadata::{IggyMetadata, MuxStateMachine}; - use partitions::{IggyPartitions, PartitionPathLayout, PartitionsConfig}; - use server_common::iobuf::Frozen; - use server_common::sharding::ShardId; - use server_common::{MESSAGE_ALIGN, Message, MessageBag}; - use shard::metrics::ShardMetrics; - use shard::shards_table::PapayaShardsTable; - use shard::{ - IggyShard, LifecycleFrame, PartitionConsensusConfig, ReconcileOp, ReplicaTopology, - ShardFrame, ShardIdentity, shard_channel, - }; - use std::cell::{Cell, RefCell}; - use std::future::Future; - use std::mem::size_of; - use std::rc::Rc; - use std::sync::atomic::AtomicBool; - - type TestMux = MuxStateMachine; - type TestShard = IggyShard; - /// `(target client id, reply frame bytes)` per `send_to_client` call. - type RecordedReplies = Rc)>>>; - /// `(target replica id, frame bytes)` per `send_to_replica` call. - type RecordedReplicaSends = Rc)>>>; - - /// Records every client-bound reply and replica-bound frame (target + - /// bytes) instead of writing to a socket; everything else is a no-op. The - /// two `ShellBus` halves are stubbed. - #[derive(Debug, Clone, Default)] - struct SpyBus { - client_replies: RecordedReplies, - replica_sends: RecordedReplicaSends, - /// Resolve [`MessageBus::sleep`] immediately instead of arming a real - /// timer. The register forward is the only path here that races a - /// timer, and its budget is five seconds -- too long to wait for in a - /// unit test, and too long to shorten in production for one. - instant_timers: Rc>, - } - - impl SpyBus { - /// Decode the single frame this bus sent to a replica. - fn sole_replica_send(&self) -> (u8, H) { - let sends = self.replica_sends.borrow(); - assert_eq!(sends.len(), 1, "expected exactly one replica-bound frame"); - let (target, frame) = &sends[0]; - let mut aligned = server_common::iobuf::Owned::::zeroed(frame.len()); - aligned.as_mut_slice().copy_from_slice(frame); - let header = - *bytemuck::checked::try_from_bytes::(&aligned.as_slice()[..size_of::()]) - .expect("replica frame decodes into the expected header"); - (*target, header) - } - } - - #[allow(clippy::future_not_send)] - impl MessageBus for SpyBus { - fn track_background(&self, _handle: JoinHandle<()>) {} - async fn send_to_client( - &self, - client_id: u128, - data: impl Into, - ) -> Result<(), SendError> { - self.client_replies - .borrow_mut() - .push((client_id, data.into().into_contiguous().as_slice().to_vec())); - Ok(()) - } - async fn send_to_replica( - &self, - replica: u8, - data: Frozen, - ) -> Result<(), SendError> { - self.replica_sends - .borrow_mut() - .push((replica, data.as_slice().to_vec())); - Ok(()) - } - async fn sleep(&self, duration: std::time::Duration) { - if !self.instant_timers.get() { - compio::time::sleep(duration).await; - } - } - fn set_connection_lost_fn(&self, _f: ConnectionLostFn) {} - fn set_replica_forward_fn(&self, _f: ReplicaForwardFn) {} - fn set_client_forward_fn(&self, _f: ClientForwardFn) {} - } - - impl ConnectionInstaller for SpyBus { - fn install_replica_inbound_fd( - &self, - _fd: DupedFd, - _on_message: MessageHandler, - _on_done: ReplicaHandshakeDoneFn, - ) { - } - fn install_replica_outbound_fd( - &self, - _fd: DupedFd, - _replica_id: u8, - _on_message: MessageHandler, - _on_done: ReplicaHandshakeDoneFn, - ) { - } - fn release_replica_handshake_slot(&self, _slot: u64) {} - fn clear_replica_dial_pending(&self, _replica_id: u8) {} - fn install_client_fd( - &self, - _fd: DupedFd, - _meta: ClientConnMeta, - _on_request: RequestHandler, - ) { - } - fn install_client_ws_fd( - &self, - _fd: DupedFd, - _meta: ClientConnMeta, - _on_request: RequestHandler, - ) { - } - fn client_meta(&self, _client_id: u128) -> Option> { - None - } - fn set_client_connection_lost_fn(&self, _f: ClientConnectionLostFn) {} - } - - /// Consensus incarnations standing for two successive boots of one node, as - /// far apart as the random draw at bootstrap makes them. - const FIRST_BOOT: u128 = 0x5EED_0001; - const SECOND_BOOT: u128 = 0x9E37_79B9_7F4A_7C15; - - /// Shard 0 carrying a metadata consensus group of `replica_count` - /// replicas in which this node is `replica`. No journal: every test using - /// it either never proposes, or is a backup that cannot. - /// - /// `incarnation` stands for one boot of this node: the shard seeds its - /// forward-nonce counter from it, so passing a different value models a - /// restart. - fn test_shard(bus: &SpyBus, replica: u8, replica_count: u8, incarnation: u128) -> TestShard { - let consensus = VsrConsensus::new( - 1, - replica, - replica_count, - server_common::sharding::METADATA_GROUP, - bus.clone(), - LocalPipeline::new(), - ); - consensus.set_incarnation(incarnation); - consensus.init(); - let metadata: IggyMetadata<_, PrepareJournal, IggySnapshot, TestMux> = - IggyMetadata::new(Some(consensus), None, None, None, TestMux::default(), None); - let partitions = IggyPartitions::new( - ShardId::new(0), - PartitionsConfig { - messages_required_to_save: 1, - size_of_messages_required_to_save: iggy_common::IggyByteSize::from(1024_u64), - enforce_fsync: false, - validate_checksum: true, - segment_size: iggy_common::IggyByteSize::from(1_048_576_u64), - preallocate_segments: false, - encryptor: None, - path_layout: PartitionPathLayout::default(), - }, - ); - TestShard::without_inbox( - ShardIdentity::new(0, "dispatch-test".to_string()), - bus.clone(), - metadata, - partitions, - PapayaShardsTable::new(), - PartitionConsensusConfig::new( - 1, - ReplicaTopology::new(replica, replica_count), - bus.clone(), - ), - ) - } - - /// Minimal committed `Register` reply for `ClientTable::commit_register` - /// (reads only `client` and `commit`). - fn register_reply(client: u128, session: u64) -> Message { - let header_size = size_of::(); - let mut reply = Message::::new(header_size); - let header = bytemuck::checked::try_from_bytes_mut::( - &mut reply.as_mut_slice()[..header_size], - ) - .expect("zeroed bytes are a valid ReplyHeader"); - *header = ReplyHeader { - client, - request: 0, - commit: session, - command: Command::Reply, - operation: Operation::Register, - ..Default::default() - }; - reply - } - - fn request_message( - operation: Operation, - client: u128, - session: u64, - request: u64, - body: &[u8], - ) -> Message { - let header_size = size_of::(); - let total = header_size + body.len(); - let mut message = Message::::new(total); - { - let slice = message.as_mut_slice(); - slice[header_size..total].copy_from_slice(body); - let header = - bytemuck::checked::from_bytes_mut::(&mut slice[..header_size]); - *header = RoutedRequestHeader { - command: Command::Request, - operation, - size: u32::try_from(total).expect("test request fits u32"), - client, - session, - request, - user_id: 0, - group: server_common::sharding::METADATA_GROUP, - ..Default::default() - }; - } - message - } - - /// Raw prepare for the sibling op, standing in for the crate-private - /// `prepare_request` projection. `user_id` 0 skips the in-apply RBAC - /// gate (server-originated convention), so the op applies cleanly. - fn prepare_message( - operation: Operation, - client: u128, - request: u64, - body: &[u8], - ) -> Message { - let header_size = size_of::(); - let total = header_size + body.len(); - let mut message = Message::::new(total); - { - let slice = message.as_mut_slice(); - slice[header_size..total].copy_from_slice(body); - let header = - bytemuck::checked::from_bytes_mut::(&mut slice[..header_size]); - *header = PrepareHeader { - command: Command::Prepare, - operation, - size: u32::try_from(total).expect("test prepare fits u32"), - op: 1, - view: 0, - client, - request, - user_id: 0, - group: server_common::sharding::METADATA_GROUP, - ..Default::default() - }; - } - // A real identity, not a placeholder: `on_replicate` recomputes it before the - // prepare reaches the WAL, so an arbitrary value reads as transit corruption. - consensus::seal_prepare_checksum(message) - } - - /// Regression test for the production failure chain "CLI stream - /// create succeeded, logout failed: Disconnected". - /// - /// Why the logout of a CLI invocation used to fail during ITS OWN - /// successful `stream create`: the catch-up gate was GLOBAL. The suite - /// runs many CLI invocations against one shared single-node server; - /// each one is three replicated ops (Register, work, Logout). When - /// THIS client's logout frame arrived, some SIBLING client's op was - /// regularly sitting between quorum-ack (`commit_max` advanced inside - /// `on_ack`) and apply (`commit_min` still behind, driver parked at - /// the journal read). `submit_logout_in_process` then rejected - /// `NotCaughtUp`, and `handle_logout_request` swallowed the error: no - /// reply frame, session left bound. A one-shot CLI saw only a dead - /// connection — "Problem with server logout / Disconnected" — and - /// exited non-zero although its create committed; the harness retry - /// then tripped "already exists". - /// - /// This test rebuilds that interleaving deterministically (client B = - /// the sibling parked mid-commit; client A = the CLI logging out) and - /// pins the contract that fixed it (non-register ops carry no - /// catch-up gate, see `submit_logout_in_process`): - /// - /// a client-initiated logout must always produce a reply frame and - /// unbind the transport session, even while a sibling's commit is - /// in flight — the logout simply pipelines behind it. - #[compio::test] - async fn logout_rejected_by_closed_gate_must_still_reply_to_client() { - const CLIENT_A: u128 = 1; - const CLIENT_B: u128 = 2; - const SESSION: u64 = 1; - const ACTING_USER: u32 = 7; - const TRANSPORT_A: u128 = 77; - - let dir = tempfile::tempdir().unwrap(); - let journal = PrepareJournal::open(&dir.path().join("journal.wal"), 0) - .await - .unwrap(); - let bus = SpyBus::default(); - let consensus = VsrConsensus::new( - 1, - 0, - 1, - server_common::sharding::METADATA_GROUP, - bus.clone(), - LocalPipeline::new(), - ); - consensus.init(); - let metadata: IggyMetadata<_, PrepareJournal, IggySnapshot, TestMux> = IggyMetadata::new( - Some(consensus), - Some(journal), - None, - None, - TestMux::default(), - None, - ); - let partitions = IggyPartitions::new( - ShardId::new(0), - PartitionsConfig { - messages_required_to_save: 1, - size_of_messages_required_to_save: iggy_common::IggyByteSize::from(1024_u64), - enforce_fsync: false, - validate_checksum: true, - segment_size: iggy_common::IggyByteSize::from(1_048_576_u64), - preallocate_segments: false, - encryptor: None, - path_layout: PartitionPathLayout::default(), - }, - ); - let shard = Rc::new(TestShard::without_inbox( - ShardIdentity::new(0, "logout-window-test".to_string()), - bus.clone(), - metadata, - partitions, - PapayaShardsTable::new(), - PartitionConsensusConfig::new(1, ReplicaTopology::new(0, 1), bus.clone()), - )); - let md = shard.plane.metadata(); - let consensus = md.consensus.as_ref().unwrap(); - - // A and B hold committed sessions (as after their CLI logins). - for client in [CLIENT_A, CLIENT_B] { - md.client_table.borrow_mut().commit_register( - client, - ACTING_USER, - register_reply(client, SESSION), - ); - } - // A's transport connection, authenticated + bound — the state a - // CLI connection is in right after its create-stream reply. - let sessions = Rc::new(RefCell::new(SessionManager::new())); - sessions.borrow_mut().ensure_connection( - TRANSPORT_A, - "127.0.0.1:34567".parse().unwrap(), - ClientTransportKind::Tcp, - ); - sessions - .borrow_mut() - .login(TRANSPORT_A, ACTING_USER) - .unwrap(); - sessions - .borrow_mut() - .bind_session(TRANSPORT_A, CLIENT_A, SESSION) - .unwrap(); - - // Sibling B's op: prepared, journaled, self-acked through the real - // replicate path. (The public submit API cannot be used to open - // the window: `dispatch_prepare_and_await` pumps its own loopback - // inline, committing before it returns. Production's window is a - // sibling submit task parked INSIDE `on_ack`'s awaits — modeled - // below by driving `on_ack` by hand.) - let create_body = CreateStreamRequest { - name: iggy_binary_protocol::primitives::identifier::WireName::new("s1").unwrap(), - options: WireOptions::empty(), - } - .to_bytes(); - let prepare = prepare_message(Operation::CreateStream, CLIENT_B, 1, &create_body); - consensus.pipeline_message(PlaneKind::Metadata, &prepare); - md.on_replicate(prepare).await; - let mut loopback = Vec::new(); - consensus.drain_loopback_into(&mut loopback); - let ack = loopback - .pop() - .expect("one self-ack per replicated prepare") - .try_into_typed::() - .expect("loopback holds self PrepareOks"); - - // Open the window: first poll of `on_ack` advances commit_max at - // quorum, then parks at the journal read — commit_min unchanged. - // Every production NotCaughtUp logout was submitted exactly here. - let waker = std::task::Waker::noop(); - let mut cx = std::task::Context::from_waker(waker); - let mut driver = Box::pin(md.on_ack(ack)); - assert!( - driver.as_mut().poll(&mut cx).is_pending(), - "driver must park mid-commit at the journal read" - ); - assert_eq!(consensus.commit_max(), 1); - assert_eq!(consensus.commit_min(), 0); - - // A's logout lands in the window, through the real dispatch path. - let logout = request_message(Operation::Logout, CLIENT_A, SESSION, 2, &[]); - handle_logout_request(&shard, &sessions, TRANSPORT_A, logout).await; - - // DESIRED CONTRACT (red on current code): the client must never be - // left in silence — that silence is what a one-shot CLI reports as - // "Problem with server logout / Disconnected". - assert!( - bus.client_replies - .borrow() - .iter() - .any(|(client, _)| *client == TRANSPORT_A), - "logout must produce a reply frame to the client even while the \ - catch-up gate is closed (silence = CLI 'Disconnected', exit 1)" - ); - assert_eq!( - sessions.borrow().get_session(TRANSPORT_A), - None, - "transport session must be unbound by a client-initiated logout; \ - the VSR slot may lapse to the eviction sweep" - ); - } - - /// A partition write whose routable wait exhausts (namespace committed, - /// but no reconciler ever seeds this shard's routing row -- the state a - /// teardown/rematerialise churn leaves behind) must answer a nonzero - /// retriable status. A status-0 empty reply is a fabricated success: the - /// SDK grades the send as acknowledged while zero bytes reached any - /// partition. - #[compio::test] - async fn unroutable_partition_send_must_reply_transient_error_not_success() { - const VSR_CLIENT: u128 = 1; - const SESSION: u64 = 1; - const TRANSPORT: u128 = 91; - const STATUS_OFFSET: usize = std::mem::offset_of!(ReplyHeader, status); - - let bus = SpyBus::default(); - let metadata = IggyMetadata::new(None, None, None, None, TestMux::default(), None); - let partitions = IggyPartitions::new( - ShardId::new(0), - PartitionsConfig { - messages_required_to_save: 1, - size_of_messages_required_to_save: iggy_common::IggyByteSize::from(1024_u64), - enforce_fsync: false, - validate_checksum: true, - segment_size: iggy_common::IggyByteSize::from(1_048_576_u64), - preallocate_segments: false, - encryptor: None, - path_layout: PartitionPathLayout::default(), - }, - ); - let shard = Rc::new(TestShard::without_inbox( - ShardIdentity::new(0, "unroutable-send-test".to_string()), - bus.clone(), - metadata, - partitions, - PapayaShardsTable::new(), - PartitionConsensusConfig::new(1, ReplicaTopology::new(0, 1), bus.clone()), - )); - let md = shard.plane.metadata(); - - // Committed stream 0 / topic 0 / partition 0, applied straight into - // the STM: the namespace resolves and root authorizes, but no - // reconciler runs, so the shards table never gains a routing row and - // the routable wait exhausts its budget. - md.mux_stm.users().ensure_root_user("iggy", "hash"); - let create_stream = CreateStreamRequest { - name: WireName::new("stream").unwrap(), - options: WireOptions::empty(), - }; - md.mux_stm - .update(prepare_message( - Operation::CreateStream, - VSR_CLIENT, - 1, - &create_stream.to_bytes(), - )) - .unwrap(); - let create_topic = CreateTopicWithAssignmentsRequest { - request: CreateTopicRequest { - stream_id: WireIdentifier::numeric(0), - partitions_count: 1, - name: WireName::new("topic").unwrap(), - options: WireOptions::empty(), - }, - derived_options: WireOptions::empty(), - partitions: vec![CreatedPartitionAssignment { - partition_id: 0, - consensus_group_id: 1, - }], - created_view: 0, - }; - md.mux_stm - .update(prepare_message( - Operation::CreateTopicWithAssignments, - VSR_CLIENT, - 2, - &create_topic.to_bytes(), - )) - .unwrap(); - assert!( - md.mux_stm - .streams() - .namespace_from_partition( - &WireIdentifier::numeric(0), - &WireIdentifier::numeric(0), - 0 - ) - .is_some(), - "seeded namespace must resolve, or the unresolved-namespace path \ - would reply instead of the exhausted routable wait" - ); - - let send_header = SendMessagesHeader { - stream_id: WireIdentifier::numeric(0), - topic_id: WireIdentifier::numeric(0), - partitioning: WirePartitioning::PartitionId(0), - messages_count: 1, - }; - let send_metadata = send_header.to_bytes(); - let mut send_body = Vec::with_capacity(4 + send_metadata.len()); - send_body.extend_from_slice(&u32::try_from(send_metadata.len()).unwrap().to_le_bytes()); - send_body.extend_from_slice(&send_metadata); - let request = request_message(Operation::SendMessages, VSR_CLIENT, SESSION, 1, &send_body); - - dispatch_partition_request( - &shard, - request, - VSR_CLIENT, - SESSION, - TRANSPORT, - Some(DEFAULT_ROOT_USER_ID), - ) - .await; - - let replies = bus.client_replies.borrow(); - assert_eq!(replies.len(), 1, "one reply frame for the failed send"); - let (client, frame) = &replies[0]; - assert_eq!(*client, TRANSPORT, "reply must target the transport id"); - let status = - u32::from_le_bytes(frame[STATUS_OFFSET..STATUS_OFFSET + 4].try_into().unwrap()); - assert_eq!( - status, - IggyError::TransientNotAccepted.as_code(), - "an unroutable partition write must surface the retriable \ - transient status; status 0 with an empty body grades as a \ - successfully acknowledged send" - ); - } - - /// A send that reaches the owning shard while its namespace is - /// tombstoned (the teardown fence a delete/recreate churn sets before - /// the disk delete) must answer the retriable transient status. The - /// partition plane's own tombstone guard drops the frame without any - /// reply; the transports decode replies in lockstep, so that silence - /// wedges the connection until the SDK's response read-timeout. - #[compio::test] - async fn tombstoned_partition_send_must_reply_transient_error_not_silence() { - const TRANSPORT: u128 = 91; - const SESSION: u64 = 1; - const STATUS_OFFSET: usize = std::mem::offset_of!(ReplyHeader, status); - - let bus = SpyBus::default(); - let metadata = IggyMetadata::new(None, None, None, None, TestMux::default(), None); - let partitions = IggyPartitions::new( - ShardId::new(0), - PartitionsConfig { - messages_required_to_save: 1, - size_of_messages_required_to_save: iggy_common::IggyByteSize::from(1024_u64), - enforce_fsync: false, - validate_checksum: true, - segment_size: iggy_common::IggyByteSize::from(1_048_576_u64), - preallocate_segments: false, - encryptor: None, - path_layout: PartitionPathLayout::default(), - }, - ); - let shard = Rc::new(TestShard::without_inbox( - ShardIdentity::new(0, "tombstoned-send-test".to_string()), - bus.clone(), - metadata, - partitions, - PapayaShardsTable::new(), - PartitionConsensusConfig::new(1, ReplicaTopology::new(0, 1), bus.clone()), - )); - - let namespace = IggyNamespace::new(0, 0, 0); - shard.plane.partitions().tombstone(namespace); - - let request = request_message(Operation::SendMessages, TRANSPORT, SESSION, 1, &[]) - .transmute_header(|header, new_header: &mut RoutedRequestHeader| { - *new_header = header; - new_header.group = namespace.inner(); - }); - shard.on_message(MessageBag::Request(request)).await; - - let replies = bus.client_replies.borrow(); - assert_eq!( - replies.len(), - 1, - "a send into a tombstoned namespace must produce a reply frame; \ - silence wedges the connection's lockstep decode" - ); - let (client, frame) = &replies[0]; - assert_eq!(*client, TRANSPORT, "reply must target the request's client"); - let status = - u32::from_le_bytes(frame[STATUS_OFFSET..STATUS_OFFSET + 4].try_into().unwrap()); - assert_eq!( - status, - IggyError::TransientNotAccepted.as_code(), - "a tombstoned-namespace send must surface the retriable transient \ - status so the SDK replays it after the partition rematerialises" - ); - } - - /// A test shard wired to its own lanes (the held sender feeds them), - /// for the reply-lane pump tests below. - fn reply_lane_test_shard(name: &str) -> (SpyBus, shard::TaggedSender, Rc) { - let bus = SpyBus::default(); - let metadata = IggyMetadata::new(None, None, None, None, TestMux::default(), None); - let partitions = IggyPartitions::new( - ShardId::new(0), - PartitionsConfig { - messages_required_to_save: 1, - size_of_messages_required_to_save: iggy_common::IggyByteSize::from(1024_u64), - enforce_fsync: false, - validate_checksum: true, - segment_size: iggy_common::IggyByteSize::from(1_048_576_u64), - preallocate_segments: false, - encryptor: None, - path_layout: PartitionPathLayout::default(), - }, - ); - let (sender, inbox_rx, reply_inbox_rx) = shard_channel(0, 16, 16); - let lane_sender = sender.clone(); - let shard = TestShard::new( - ShardIdentity::new(0, name.to_string()), - bus.clone(), - Rc::new(|_, _| {}), - Rc::new(|_, _| {}), - Rc::new(|_| {}), - Rc::new(|_| {}), - Rc::new(|_, _, _| {}), - metadata, - partitions, - vec![sender], - inbox_rx, - reply_inbox_rx, - PapayaShardsTable::new(), - PartitionConsensusConfig::new(1, ReplicaTopology::new(0, 1), bus.clone()), - None, - ShardMetrics::for_shard(), - ) - .expect("single-sender ring is canonically ordered"); - (bus, lane_sender, Rc::new(shard)) - } - - fn reply_lane_forward(client_id: u128) -> ShardFrame { - ShardFrame::lifecycle(LifecycleFrame::ForwardClientSend { - client_id, - msg: server_common::iobuf::Frozen::from( - server_common::iobuf::Owned::::zeroed(64), - ) - .into(), - }) - } - - /// A frame on the reply lane must reach the client through the RUNNING - /// pump's reply arm: the lane split moved `ForwardClientSend` off the - /// main inbox, so a pump that forgot to service the new lane would - /// strand every cross-shard reply while the send sites happily report - /// success. - #[compio::test] - async fn pump_live_arm_delivers_reply_lane_forwards() { - const TRANSPORT: u128 = 92; - let (bus, lane_sender, shard) = reply_lane_test_shard("reply-lane-live-arm-test"); - - let (stop_tx, stop_rx) = shard::channel::<()>(1); - let pump_shard = Rc::clone(&shard); - let pump = compio::runtime::spawn(async move { - pump_shard - .run_message_pump(stop_rx, Arc::new(AtomicBool::new(false))) - .await; - }); - - lane_sender - .reply_sender() - .try_send(reply_lane_forward(TRANSPORT)) - .expect("reply lane has capacity"); - - // The pump is idle on the main lane, so its bottom reply arm must - // serve the frame without any main-lane traffic or shutdown drain. - let mut delivered = false; - for _ in 0..500 { - if !bus.client_replies.borrow().is_empty() { - delivered = true; - break; - } - compio::time::sleep(std::time::Duration::from_millis(1)).await; - } - stop_tx.try_send(()).expect("stop channel has capacity"); - let _ = pump.await; - - assert!( - delivered, - "the live reply arm must deliver a forward while the pump runs" - ); - let replies = bus.client_replies.borrow(); - assert_eq!(replies[0].0, TRANSPORT, "forward must reach its client"); - } - - /// The shutdown path must ALSO deliver reply-lane frames: a forward - /// already accepted by the lane when the stop signal wins the biased - /// select would otherwise be silently destroyed at teardown. - #[compio::test] - async fn pump_shutdown_drain_delivers_reply_lane_forwards() { - const TRANSPORT: u128 = 93; - let (bus, lane_sender, shard) = reply_lane_test_shard("reply-lane-drain-test"); - - lane_sender - .reply_sender() - .try_send(reply_lane_forward(TRANSPORT)) - .expect("reply lane has capacity"); - - // Pre-armed stop: the pump exits through the biased stop arm and the - // post-loop drain must still deliver the reply-lane frame. - let (stop_tx, stop_rx) = shard::channel::<()>(1); - stop_tx.try_send(()).expect("stop channel has capacity"); - shard - .run_message_pump(stop_rx, Arc::new(AtomicBool::new(false))) - .await; - - let replies = bus.client_replies.borrow(); - assert_eq!( - replies.len(), - 1, - "the pump's reply-lane drain must deliver the forwarded reply" - ); - assert_eq!( - replies[0].0, TRANSPORT, - "the forward must reach the client it was addressed to" - ); - } - - /// A send parked for a namespace that is torn down before materialising - /// (create -> delete before the reconciler's `InsertOwned`) is discarded - /// on `ConfirmRemove`. The discard must stage the same retriable - /// transient deny toward the client -- through the shard's own pump as a - /// `ForwardClientSend` -- instead of dropping the request without any - /// reply. - #[compio::test] - async fn discarded_parked_partition_send_must_reply_transient_error_not_silence() { - const TRANSPORT: u128 = 91; - const SESSION: u64 = 1; - const STATUS_OFFSET: usize = std::mem::offset_of!(ReplyHeader, status); - - let bus = SpyBus::default(); - let metadata = IggyMetadata::new(None, None, None, None, TestMux::default(), None); - let partitions = IggyPartitions::new( - ShardId::new(0), - PartitionsConfig { - messages_required_to_save: 1, - size_of_messages_required_to_save: iggy_common::IggyByteSize::from(1024_u64), - enforce_fsync: false, - validate_checksum: true, - segment_size: iggy_common::IggyByteSize::from(1_048_576_u64), - preallocate_segments: false, - encryptor: None, - path_layout: PartitionPathLayout::default(), - }, - ); - // Real sender ring so the staged deny is observable: the test holds - // the receiving ends of this shard's own lanes. The deny is a client - // Reply forward, so it lands on the REPLY lane. - let (sender, _pump_rx, reply_rx) = shard_channel(0, 16, 16); - let (_inbox_tx, inbox_rx, reply_inbox_rx) = shard_channel(0, 1, 1); - let shard = TestShard::new( - ShardIdentity::new(0, "discarded-parked-send-test".to_string()), - bus.clone(), - Rc::new(|_, _| {}), - Rc::new(|_, _| {}), - Rc::new(|_| {}), - Rc::new(|_| {}), - Rc::new(|_, _, _| {}), - metadata, - partitions, - vec![sender], - inbox_rx, - reply_inbox_rx, - PapayaShardsTable::new(), - PartitionConsensusConfig::new(1, ReplicaTopology::new(0, 1), bus.clone()), - None, - ShardMetrics::for_shard(), - ) - .expect("single-sender ring is canonically ordered"); - - let namespace = IggyNamespace::new(0, 0, 0); - let request = request_message(Operation::SendMessages, TRANSPORT, SESSION, 1, &[]) - .transmute_header(|header, new_header: &mut RoutedRequestHeader| { - *new_header = header; - new_header.group = namespace.inner(); - }); - // Namespace neither materialised nor tombstoned: the frame parks. - shard.on_message(MessageBag::Request(request)).await; - - shard.enqueue_reconcile_op(ReconcileOp::ConfirmRemove { namespace }); - shard.apply_reconcile_ops(); - - let mut denies = Vec::new(); - while let Ok(frame) = reply_rx.try_recv() { - if let ShardFrame::Lifecycle(LifecycleFrame::ForwardClientSend { client_id, msg }) = - frame - { - denies.push((client_id, msg.into_contiguous().as_slice().to_vec())); - } - } - assert_eq!( - denies.len(), - 1, - "discarding a parked client request must stage exactly one deny \ - reply; silence wedges the connection's lockstep decode" - ); - let (client, frame) = &denies[0]; - assert_eq!(*client, TRANSPORT, "deny must target the request's client"); - let status = - u32::from_le_bytes(frame[STATUS_OFFSET..STATUS_OFFSET + 4].try_into().unwrap()); - assert_eq!( - status, - IggyError::TransientNotAccepted.as_code(), - "a discarded parked send must surface the retriable transient \ - status so the SDK replays it instead of timing out" - ); - } - - /// A backup's login: it forwards the register it authenticated to the - /// view's primary and completes on the primary's verdict, with the whole - /// round trip going through the real shard ingest arm. - #[compio::test] - async fn backup_forwards_register_and_completes_on_the_primary_verdict() { - const CLIENT: u128 = 0xCAFE; - const USER: u32 = 7; - const EPOCH: u64 = 41; - const WATERMARK: u64 = 9; - - let bus = SpyBus::default(); - // Replica 1 of 3, view 0: `primary_index(0)` is replica 0. - let shard = Rc::new(test_shard(&bus, 1, 3, FIRST_BOOT)); - let login = { - let shard = Rc::clone(&shard); - compio::runtime::spawn(async move { - submit_register_local_or_forward(&shard, CLIENT, USER).await - }) - }; - await_forward(&bus).await; - let (target, forward) = bus.sole_replica_send::(); - assert_eq!(target, 0, "forward must address the view's primary"); - assert_eq!(forward.command, Command::ForwardRegister); - assert_eq!(forward.client, CLIENT); - assert_eq!( - forward.user_id, USER, - "the forwarded identity is the payload" - ); - assert_eq!(forward.replica, 1, "the origin names itself for the answer"); - assert_ne!(forward.nonce, 0); - assert_eq!(forward.verify_frame(), Ok(()), "the frame must be sealed"); - assert_eq!(forward.validate(), Ok(())); - - shard - .on_message(forward_register_result( - &forward, - ForwardRegisterOutcome::Ok, - EPOCH, - WATERMARK, - )) - .await; - assert_eq!( - login.await.expect("the login task ran to completion"), - Ok(BoundSession { - epoch: EPOCH, - watermark: WATERMARK, - }) - ); - } - - #[compio::test] - async fn backup_forwards_logout_and_completes_on_the_primary_verdict() { - const CLIENT: u128 = 0xCAFE; - const SESSION: u64 = 41; - const REQUEST: u64 = 9; - const COMMIT: u64 = 42; - - let bus = SpyBus::default(); - let shard = Rc::new(test_shard(&bus, 1, 3, FIRST_BOOT)); - let logout = { - let shard = Rc::clone(&shard); - compio::runtime::spawn(async move { - submit_logout_local_or_forward(&shard, CLIENT, SESSION, REQUEST).await - }) - }; - await_forward(&bus).await; - let (target, forward) = bus.sole_replica_send::(); - assert_eq!(target, 0, "forward must address the view's primary"); - assert_eq!(forward.command, Command::ForwardLogout); - assert_eq!(forward.client, CLIENT); - assert_eq!(forward.session, SESSION); - assert_eq!(forward.request, REQUEST); - assert_eq!(forward.replica, 1); - assert_ne!(forward.nonce, 0); - assert_eq!(forward.verify_frame(), Ok(())); - assert_eq!(forward.validate(), Ok(())); - - shard - .on_message(forward_logout_result_message(&forward, &Ok(COMMIT))) - .await; - assert_eq!( - logout.await.expect("the logout task ran to completion"), - Ok(COMMIT) - ); - } - - #[compio::test] - async fn unanswered_logout_forward_times_out_and_clears_the_waiter() { - let bus = SpyBus::default(); - bus.instant_timers.set(true); - let shard = Rc::new(test_shard(&bus, 1, 3, FIRST_BOOT)); - - let outcome = submit_logout_local_or_forward(&shard, 0xCAFE, 41, 9).await; - assert_eq!(outcome, Err(MetadataSubmitError::ForwardTimedOut)); - - let (_, forward) = bus.sole_replica_send::(); - shard - .on_message(forward_logout_result_message(&forward, &Ok(42))) - .await; - } - - #[test] - fn unknown_logout_outcomes_pin_the_session() { - for error in [ - MetadataSubmitError::ForwardTimedOut, - MetadataSubmitError::InProgress, - MetadataSubmitError::Canceled, - ] { - assert_eq!( - transient_logout_code(&error), - IggyError::TransientNotCommitted - ); - } - for error in [ - MetadataSubmitError::NotPrimary, - MetadataSubmitError::PipelineFull, - MetadataSubmitError::PrimaryUnreachable, - ] { - assert_eq!( - transient_logout_code(&error), - IggyError::TransientNotAccepted - ); - } - } - - /// The ownership refusal is the one terminal verdict, and it has to stay - /// terminal across the hop or the SDK replays a login that cannot succeed. - #[compio::test] - async fn forwarded_register_keeps_the_ownership_refusal_terminal() { - let bus = SpyBus::default(); - let shard = Rc::new(test_shard(&bus, 1, 3, FIRST_BOOT)); - let login = { - let shard = Rc::clone(&shard); - compio::runtime::spawn(async move { - submit_register_local_or_forward(&shard, 0xCAFE, 7).await - }) - }; - await_forward(&bus).await; - let (_, forward) = bus.sole_replica_send::(); - shard - .on_message(forward_register_result( - &forward, - ForwardRegisterOutcome::ClientIdOwnedByAnotherUser, - 0, - 0, - )) - .await; - let error = login - .await - .expect("the login task ran to completion") - .expect_err("the refusal must surface"); - assert_eq!(error, MetadataSubmitError::ClientIdOwnedByAnotherUser); - assert!(!error.is_transient(), "the refusal must stay terminal"); - } - - /// A primary that never answers must not strand the login or leak its - /// parked entry; the client gets a transient failure and replays. - #[compio::test] - async fn unanswered_forward_times_out_and_clears_the_parked_login() { - let bus = SpyBus::default(); - bus.instant_timers.set(true); - let shard = Rc::new(test_shard(&bus, 1, 3, FIRST_BOOT)); - - let outcome = submit_register_local_or_forward(&shard, 0xCAFE, 7).await; - assert_eq!(outcome, Err(MetadataSubmitError::ForwardTimedOut)); - assert!( - outcome.unwrap_err().is_transient(), - "a lost answer is replayable" - ); - - // The parked entry is gone: the answer that arrives late finds nothing - // and is dropped rather than completing a login nobody is waiting on. - let (_, forward) = bus.sole_replica_send::(); - shard - .on_message(forward_register_result( - &forward, - ForwardRegisterOutcome::Ok, - 41, - 0, - )) - .await; - } - - /// The reply frame is where an unknown outcome has to be told apart from a - /// refusal: a forward that timed out may still commit, so the client must - /// replay under the same client id instead of failing over under a fresh - /// one. A verdict that refused the register carries no such doubt. - #[compio::test] - async fn transient_login_reply_marks_a_timed_out_forward_not_committed() { - const TRANSPORT: u128 = 91; - const VSR_CLIENT: u128 = 0xCAFE; - const RESULT_OFFSET: usize = size_of::() + 8; - - let bus = SpyBus::default(); - let shard = Rc::new(test_shard(&bus, 1, 3, FIRST_BOOT)); - let request = request_message(Operation::Register, VSR_CLIENT, 0, 0, &[]); - - for (submit_error, expected) in [ - ( - MetadataSubmitError::ForwardTimedOut, - IggyError::TransientNotCommitted, - ), - ( - MetadataSubmitError::NotPrimary, - IggyError::TransientNotAccepted, - ), - ] { - let error = LoginRegisterError::Transient(submit_error); - surface_login_failure(&shard, TRANSPORT, request.header(), &error).await; - - let replies = bus.client_replies.borrow(); - assert_eq!(replies.len(), 1, "a transient login must answer a frame"); - let (client, frame) = &replies[0]; - assert_eq!(*client, TRANSPORT, "reply must target the transport id"); - let result = - u32::from_le_bytes(frame[RESULT_OFFSET..RESULT_OFFSET + 4].try_into().unwrap()); - assert_eq!(result, expected.as_code(), "{error} must reply {expected}"); - drop(replies); - bus.client_replies.borrow_mut().clear(); - } - } - - /// A restart must not re-mint the nonce sequence of the boot before it. The - /// nonce is never persisted, and an answer to a pre-restart forward can - /// still be in flight: routed by a repeated nonce it would confirm a login - /// the cluster never committed, with another client's epoch. - #[compio::test] - async fn a_restart_moves_the_forward_nonce_sequence() { - assert_ne!( - first_forward_nonce(FIRST_BOOT).await, - first_forward_nonce(SECOND_BOOT).await, - "each boot must start its nonce sequence somewhere the other did not" - ); - } - - /// Seeding the counter from the incarnation means it can start one step - /// short of wrapping, and a zero nonce is a frame every replica rejects. - #[compio::test] - async fn wrapping_forward_nonce_counter_skips_zero() { - let nonce = first_forward_nonce(u128::from(u64::MAX)).await; - assert_ne!( - nonce & u128::from(u64::MAX), - 0, - "a counter that wrapped must not contribute a zero nonce half" - ); - } - - /// An answer echoing a client the nonce was never parked for must neither - /// complete that login nor evict it, since a repeated nonce is exactly what - /// a late cross-boot answer carries. - #[compio::test] - async fn forward_result_for_another_client_leaves_the_login_parked() { - const CLIENT: u128 = 0xCAFE; - const EPOCH: u64 = 41; - const WATERMARK: u64 = 9; - const FOREIGN_EPOCH: u64 = 77; - - let bus = SpyBus::default(); - let shard = Rc::new(test_shard(&bus, 1, 3, FIRST_BOOT)); - let login = { - let shard = Rc::clone(&shard); - compio::runtime::spawn(async move { - submit_register_local_or_forward(&shard, CLIENT, 7).await - }) - }; - await_forward(&bus).await; - let (_, forward) = bus.sole_replica_send::(); - - let mut foreign = forward; - foreign.client = CLIENT + 1; - shard - .on_message(forward_register_result( - &foreign, - ForwardRegisterOutcome::Ok, - FOREIGN_EPOCH, - 0, - )) - .await; - shard - .on_message(forward_register_result( - &forward, - ForwardRegisterOutcome::Ok, - EPOCH, - WATERMARK, - )) - .await; - assert_eq!( - login.await.expect("the login task ran to completion"), - Ok(BoundSession { - epoch: EPOCH, - watermark: WATERMARK, - }), - "the login must bind the epoch addressed to it, and must still be \ - parked to receive it" - ); - } - - /// A node that is primary itself never forwards -- that is what bounds a - /// forward at one hop. - #[compio::test] - async fn primary_proposes_locally_instead_of_forwarding() { - let bus = SpyBus::default(); - // Replica 0 of 3, view 0: this node IS the primary. - let shard = Rc::new(test_shard(&bus, 0, 3, FIRST_BOOT)); - - // No journal on the test shard, so the proposal cannot commit; what - // matters is that nothing left over the interconnect. - let _ = compio::time::timeout( - Duration::from_millis(50), - submit_register_local_or_forward(&shard, 0xCAFE, 7), - ) - .await; - assert!( - bus.replica_sends.borrow().is_empty(), - "a primary must propose in process" - ); - } - - /// The nonce a shard booted at `incarnation` stamps on its first forward. - /// Nobody answers, so the login abandons on the instant timer; the frame it - /// left on the bus is what the caller is after. - async fn first_forward_nonce(incarnation: u128) -> u128 { - let bus = SpyBus::default(); - bus.instant_timers.set(true); - let shard = Rc::new(test_shard(&bus, 1, 3, incarnation)); - let outcome = submit_register_local_or_forward(&shard, 0xCAFE, 7).await; - assert_eq!(outcome, Err(MetadataSubmitError::ForwardTimedOut)); - bus.sole_replica_send::().1.nonce - } - - /// Let a spawned login run until it has parked on the primary's answer. - async fn await_forward(bus: &SpyBus) { - for _ in 0..1000 { - if !bus.replica_sends.borrow().is_empty() { - return; - } - compio::time::sleep(Duration::from_millis(1)).await; - } - panic!("the login never forwarded a register"); - } - - /// A sealed `ForwardRegisterResult` addressed to `forward`'s nonce. - fn forward_register_result( - forward: &ForwardRegisterHeader, - outcome: ForwardRegisterOutcome, - epoch: u64, - watermark: u64, - ) -> MessageBag { - let bound = match outcome { - ForwardRegisterOutcome::Ok => Ok(BoundSession { epoch, watermark }), - ForwardRegisterOutcome::ClientIdOwnedByAnotherUser => { - Err(MetadataSubmitError::ClientIdOwnedByAnotherUser) - } - _ => Err(MetadataSubmitError::NotPrimary), - }; - MessageBag::ForwardRegisterResult(build_forward_register_result_message( - forward.cluster, - forward.view, - 0, - forward.client, - forward.nonce, - &bound, - )) - } - - fn forward_logout_result_message( - forward: &ForwardLogoutHeader, - outcome: &Result, - ) -> MessageBag { - MessageBag::ForwardLogoutResult(build_forward_logout_result_message( - forward.cluster, - forward.view, - 0, - forward.client, - forward.nonce, - outcome, - )) - } - - /// The `GET_CLUSTER_METADATA` auth gate holds on every roster shape: it - /// describes the private replica network, and a client that dialed a - /// backup reaches the cluster by logging in there (the backup forwards - /// the register), not by reading the topology first. - /// - /// The denial must be a plain Reply on the status channel, not an - /// Eviction: no session exists yet, and a session-terminal frame makes - /// SDKs drop the connection their login is about to use. - #[compio::test] - async fn pre_auth_cluster_metadata_denied_on_every_roster() { - use configs::cluster::{ClusterNodeConfig, TransportPorts}; - use iggy_binary_protocol::codes::GET_CLUSTER_METADATA_CODE; - use iggy_binary_protocol::{GenericHeader, ReplyHeader}; - - const TRANSPORT: u128 = 91; - const COMMAND_OFFSET: usize = std::mem::offset_of!(GenericHeader, command); - const STATUS_OFFSET: usize = std::mem::offset_of!(ReplyHeader, status); - const OP_OFFSET: usize = std::mem::offset_of!(ReplyHeader, op); - const COMMIT_OFFSET: usize = std::mem::offset_of!(ReplyHeader, commit); - - fn metadata_read() -> Message { - let header_size = size_of::(); - let mut message = Message::::new(header_size); - { - let header = bytemuck::checked::from_bytes_mut::( - &mut message.as_mut_slice()[..header_size], - ); - *header = RequestHeader { - command: Command::Request, - operation: Operation::NonReplicated, - size: u32::try_from(header_size).expect("header fits u32"), - client: TRANSPORT, - ..Default::default() - }; - header.reserved[..4].copy_from_slice(&GET_CLUSTER_METADATA_CODE.to_le_bytes()); - } - message.into_generic() - } - - fn roster_node(name: &str) -> ClusterNodeConfig { - ClusterNodeConfig { - name: name.to_owned(), - ip: "127.0.0.1".to_owned(), - advertised_address: None, - advertised_addresses: Vec::new(), - replica_id: 0, - ports: TransportPorts::default(), - } - } - - let bus = SpyBus::default(); - let shard = Rc::new(test_shard(&bus, 0, 1, FIRST_BOOT)); - let sessions = Rc::new(RefCell::new(SessionManager::new())); - let system_config = Arc::new(ServerSystemConfig::default()); - - let multi_node = Rc::new(ClusterRoster { - enabled: true, - name: "test-cluster".to_owned(), - nodes: ["node-0", "node-1"] - .map(|name| { - configs::cluster::ResolvedClusterNode::try_from(roster_node(name)) - .expect("valid roster node") - }) - .to_vec(), - self_advertised: "127.0.0.1".to_owned(), - self_ports: TransportPorts::default(), - metadata_view: Arc::new(std::sync::atomic::AtomicU64::new( - crate::cluster_meta::METADATA_VIEW_UNKNOWN, - )), - }); - // Default roster is disabled / single node; the installed one is a - // real cluster. Neither serves an unbound caller. - for roster in [None, Some(multi_node)] { - if let Some(roster) = roster { - sessions.borrow_mut().set_cluster_roster(roster); - } - handle_client_request( - &shard, - &sessions, - &system_config, - 1, - TRANSPORT, - metadata_read(), - ) - .await; - let replies = bus.client_replies.borrow(); - assert_eq!(replies.len(), 1, "gated read must still produce a frame"); - let (client, frame) = &replies[0]; - assert_eq!(*client, TRANSPORT); - assert_eq!( - frame[COMMAND_OFFSET], - Command::Reply as u8, - "an unbound cluster-metadata read must be denied with a Reply, not evicted" - ); - let status = - u32::from_le_bytes(frame[STATUS_OFFSET..STATUS_OFFSET + 4].try_into().unwrap()); - assert_eq!( - status, - IggyError::Unauthenticated.as_code(), - "deny reply status must be Unauthenticated" - ); - let op = u64::from_le_bytes(frame[OP_OFFSET..OP_OFFSET + 8].try_into().unwrap()); - assert_eq!(op, 0, "pre-auth deny carries no session, so op must be 0"); - let commit = - u64::from_le_bytes(frame[COMMIT_OFFSET..COMMIT_OFFSET + 8].try_into().unwrap()); - assert_eq!(commit, 0, "pre-auth deny must not disclose commit activity"); - drop(replies); - bus.client_replies.borrow_mut().clear(); - } - } - - #[test] - fn create_topic_bounds_deny_pre_consensus() { - let segment_size = iggy_common::DEFAULT_SEGMENT_SIZE; - assert!(segment_size > 0, "default segment size must be nonzero"); - - assert!( - validate_topic_bounds( - MAX_PARTITIONS_PER_REQUEST, - MaxTopicSize::ServerDefault, - segment_size - ) - .is_ok(), - "the partition cap itself is admissible" - ); - assert!( - matches!( - validate_topic_bounds( - MAX_PARTITIONS_PER_REQUEST + 1, - MaxTopicSize::ServerDefault, - segment_size - ), - Err(IggyError::TooManyPartitions) - ), - "one past the partition cap must deny" - ); - // ServerDefault is numerically 0 yet exempt from the segment-size - // floor: it resolves against server config, matching legacy. - assert!(validate_topic_bounds(1, MaxTopicSize::ServerDefault, segment_size).is_ok()); - assert!(validate_topic_bounds(1, MaxTopicSize::Unlimited, segment_size).is_ok()); - let below_floor = MaxTopicSize::Custom((segment_size - 1).into()); - assert!( - matches!( - validate_topic_bounds(1, below_floor, segment_size), - Err(IggyError::InvalidTopicSize(size, floor)) - if size == below_floor && floor == IggyByteSize::from(segment_size) - ), - "custom size below the segment size must deny with the bounds" - ); - let at_floor = MaxTopicSize::Custom(IggyByteSize::from(segment_size)); - assert!( - validate_topic_bounds(1, at_floor, segment_size).is_ok(), - "a topic exactly one segment large is admissible" - ); - } - - #[test] - fn partitions_count_cap_denies_pre_consensus() { - assert!( - validate_partitions_count(MAX_PARTITIONS_PER_REQUEST).is_ok(), - "the cap itself is admissible" - ); - assert!( - matches!( - validate_partitions_count(MAX_PARTITIONS_PER_REQUEST + 1), - Err(IggyError::TooManyPartitions) - ), - "one past the cap must deny" - ); - // Zero passes the shared cap because a zero-partition TOPIC is legal - // (legacy `create_topic` admits `0..=MAX`). - assert!(validate_partitions_count(0).is_ok()); - } - - #[test] - fn zero_partitions_change_denies_pre_consensus() { - // Adding or removing zero partitions is a no-op that would still burn - // a replicated log entry and force a rebalance. Legacy rejects it with - // `TooManyPartitions` in both handlers, so the code matches. - assert!( - matches!( - validate_partitions_change_count(0), - Err(IggyError::TooManyPartitions) - ), - "adding or removing zero partitions must deny" - ); - assert!(validate_partitions_change_count(1).is_ok()); - assert!(validate_partitions_change_count(MAX_PARTITIONS_PER_REQUEST).is_ok()); - assert!( - matches!( - validate_partitions_change_count(MAX_PARTITIONS_PER_REQUEST + 1), - Err(IggyError::TooManyPartitions) - ), - "the cap still applies" - ); - } -} diff --git a/core/server/src/dispatch/authz.rs b/core/server/src/dispatch/authz.rs index ee113e0d1e..436aa2222e 100644 --- a/core/server/src/dispatch/authz.rs +++ b/core/server/src/dispatch/authz.rs @@ -60,7 +60,7 @@ use crate::shell::{ShellBus, ShellShard}; /// namespace already resolved, so the entity exists; a `None` user id (which /// the bound-session gate should preclude) fails closed with `Unauthenticated` /// rather than allow an unattributed write. -pub(super) fn authorize_partition_op( +pub(in crate::dispatch) fn authorize_partition_op( shard: &Rc>, operation: Operation, user_id: Option, @@ -139,7 +139,7 @@ where /// better, the connection decodes replies in lockstep and would wedge on every /// later request. #[allow(clippy::future_not_send)] -pub(super) async fn send_deny_reply( +pub(in crate::dispatch) async fn send_deny_reply( shard: &Rc>, transport_client_id: u128, request_header: &RoutedRequestHeader, @@ -172,7 +172,7 @@ pub(super) async fn send_deny_reply( /// commit frontier. The status is the only field a pre-authenticated caller /// needs, while the live commit would expose cluster write activity. #[allow(clippy::future_not_send)] -pub(super) async fn send_unbound_deny_reply( +pub(in crate::dispatch) async fn send_unbound_deny_reply( shard: &Rc>, transport_client_id: u128, request_header: &RoutedRequestHeader, @@ -202,7 +202,7 @@ pub(super) async fn send_unbound_deny_reply( /// Run an unscoped non-replicated-read rule for the acting user. A `None` user /// id (only the pre-auth path, which serves ungated codes) fails closed. -pub(super) fn authorize_uid( +pub(in crate::dispatch) fn authorize_uid( shard: &Rc>, user_id: Option, rule: impl FnOnce(&Permissioner, u32) -> Result<(), IggyError>, @@ -227,7 +227,7 @@ where /// (stream, topic). `None` proceeds (allowed, or a resolution miss the caller's /// own not-found path handles); `Some(status)` denies. A `None` user id fails /// closed. -pub(super) fn authorize_partition_read( +pub(in crate::dispatch) fn authorize_partition_read( shard: &Rc>, stream_id: &WireIdentifier, topic_id: &WireIdentifier, @@ -263,7 +263,7 @@ where /// topic]) against committed state first. The PAT list is self-scoped, so /// authentication is its whole rule, and `GET_CLUSTER_METADATA` -- which /// describes the private replica network -- is gated the same way. -pub(super) fn authorize_default_read( +pub(in crate::dispatch) fn authorize_default_read( shard: &Rc>, code: u32, body: &[u8], @@ -337,7 +337,7 @@ where /// surfaces the typed error, so a poll denial never reaches the empty-poll /// "0 messages" body path. #[allow(clippy::future_not_send)] -pub(super) async fn send_non_replicated_deny( +pub(in crate::dispatch) async fn send_non_replicated_deny( shard: &Rc>, request: &Message, transport_client_id: u128, diff --git a/core/server/src/login_register.rs b/core/server/src/dispatch/login_error.rs similarity index 75% rename from core/server/src/login_register.rs rename to core/server/src/dispatch/login_error.rs index bcc7d8c753..c5e984258f 100644 --- a/core/server/src/login_register.rs +++ b/core/server/src/dispatch/login_error.rs @@ -15,12 +15,12 @@ // specific language governing permissions and limitations // under the License. -//! Login/register failure taxonomy. +//! Login/register failure taxonomy, shared by both spines. //! -//! The login/register flow itself lives in `dispatch` + `auth`; this module -//! owns the error type those handlers return and the terminal-vs-transient -//! split that decides a fast-fail reply versus a silent close (recoverable -//! failures stay silent so the SDK replays). +//! Its own leaf because the TCP spine raises these ([`super::session_ops`]) +//! while the HTTP spine maps them to wire errors (`crate::http::reply`). +//! Parked in either spine it would make one import the other's session module +//! for a type neither owns. use crate::session_manager::SessionError; use metadata::MetadataSubmitError; @@ -71,18 +71,3 @@ impl std::fmt::Display for LoginRegisterError { } impl std::error::Error for LoginRegisterError {} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn terminal_vs_transient() { - assert!(LoginRegisterError::InvalidCredentials.is_terminal()); - assert!(LoginRegisterError::InvalidToken.is_terminal()); - assert!(LoginRegisterError::UserInactive.is_terminal()); - assert!(LoginRegisterError::Session(SessionError::ConnectionNotFound(0)).is_terminal()); - // Transient is the only recoverable variant: never terminal. - assert!(!LoginRegisterError::Transient(MetadataSubmitError::PipelineFull).is_terminal()); - } -} diff --git a/core/server/src/dispatch/mod.rs b/core/server/src/dispatch/mod.rs new file mode 100644 index 0000000000..0385fdebac --- /dev/null +++ b/core/server/src/dispatch/mod.rs @@ -0,0 +1,1413 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Per-shard request dispatch: queue plumbing and the request funnel. +//! +//! The tree: [`session_ops`] (login/register/logout and their replica +//! forwards), [`partition`] (the partition data plane, both mesh ends), +//! [`reads`] (the non-replicated read router), [`submit`] (the shard-0 +//! metadata-submit RPC), `authz` (the wire-path authorization gates). +//! +//! Deliberate asymmetry (the two authz gates): replicated metadata ops are +//! authorized in-apply by the STM, in committed order on every replica; +//! partition and non-replicated ops never enter the metadata log, so `authz` +//! gates them pre-dispatch against this shard's applied permissioner. The +//! HTTP spine keeps its own equivalent gates (see `crate::http`) because its +//! error contract (404-before-403) is pinned client-visible behavior. + +mod authz; +pub mod login_error; +pub mod partition; +mod reads; +pub mod session_ops; +pub mod submit; +#[cfg(test)] +mod test_support; + +use crate::consumer_group::maybe_rewrite_consumer_group_request; +use crate::dispatch::authz::{send_deny_reply, send_unbound_deny_reply}; +use crate::dispatch::partition::{dispatch_partition_request, handle_delete_segments_request}; +use crate::dispatch::reads::handle_non_replicated_request; +use crate::dispatch::session_ops::{ + handle_login_register_request, handle_logout_request, send_login_eviction, + send_unauthenticated_eviction, submit_disconnect_logout, +}; +use crate::dispatch::submit::submit_client_request_on_owner; +use crate::pat::maybe_rewrite_pat_request; +use crate::responses::{ + NonReplicatedResponse, build_deny_reply, build_raw_pat_reply, current_metadata_commit, +}; +use crate::segment_cleaner::UNENFORCEABLE_TOPIC_SIZE_WARN; +use crate::session_manager::SessionManager; +use crate::shell::{ShellBus, ShellShard, ShellShardHandle}; +use crate::users::maybe_rewrite_user_password_request; +use crate::wire::{request_body, verify_request_checksum}; +use bytes::Bytes; +use configs::server::ServerSystemConfig; +use consensus::MetadataHandle; +use iggy_binary_protocol::PrepareHeader; +use iggy_binary_protocol::codes::{ + GET_CLUSTER_METADATA_CODE, LOGIN_USER_CODE, LOGIN_WITH_PERSONAL_ACCESS_TOKEN_CODE, PING_CODE, +}; +use iggy_binary_protocol::requests::partitions::{ + CreatePartitionsRequest, DeletePartitionsRequest, +}; +use iggy_binary_protocol::requests::streams::{CreateStreamRequest, UpdateStreamRequest}; +use iggy_binary_protocol::requests::topics::{CreateTopicRequest, UpdateTopicRequest}; +use iggy_binary_protocol::requests::users::{CreateUserRequest, UpdateUserRequest}; +use iggy_binary_protocol::{ + EvictionReason, GenericHeader, MAX_PARTITIONS_PER_REQUEST, Operation, RequestHeader, + RoutedRequestHeader, WireDecode, WireIdentifier, WireOptions, +}; +use iggy_common::{ + IggyByteSize, IggyError, MaxTopicSize, TopicCreateOptions, UPDATABLE_STREAM_OPTION_KEYS, + UPDATABLE_TOPIC_OPTION_KEYS, UPDATABLE_USER_OPTION_KEYS, validate_preallocated_topic_bytes, + validate_topic_segment_size, +}; +use journal::superblock::SuperblockStore; +use journal::{Journal, JournalHandle}; +use message_bus::BusMessage; +use message_bus::client_listener::RequestHandler; +use message_bus::replica::listener::MessageHandler; +use metadata::impls::metadata::StreamsFrontend; +use metadata::stm::stream::Streams; +use server_common::Message; +use shard::{ConnectedClientInfo, ListClientsHandler}; +use std::cell::RefCell; +use std::collections::{HashMap, HashSet, VecDeque}; +use std::rc::Rc; +use std::sync::Arc; +use tracing::{debug, warn}; + +type ClientRequestQueues = Rc>>>>; +type ActiveClientRequests = Rc>>; + +pub fn make_client_request_handler( + shard: &Rc>, + sessions: &Rc>, + system_config: Arc, + max_tokens_per_user: u32, +) -> RequestHandler +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let shard = Rc::clone(shard); + let sessions = Rc::clone(sessions); + let queues: ClientRequestQueues = Rc::new(RefCell::new(HashMap::new())); + let active: ActiveClientRequests = Rc::new(RefCell::new(HashSet::new())); + let sessions_for_disconnect = Rc::clone(&sessions); + let shard_for_disconnect = Rc::clone(&shard); + shard + .bus + .set_client_connection_lost_fn(Rc::new(move |client_id| { + if let Some((vsr_client_id, session)) = sessions_for_disconnect + .borrow_mut() + .remove_connection(client_id) + { + submit_disconnect_logout(Rc::clone(&shard_for_disconnect), vsr_client_id, session); + } + })); + Rc::new(move |client_id, message| { + enqueue_client_request( + Rc::clone(&shard), + Rc::clone(&sessions), + Arc::clone(&system_config), + max_tokens_per_user, + Rc::clone(&queues), + Rc::clone(&active), + client_id, + message, + ); + }) +} + +/// Build the per-shard [`ListClientsHandler`]: on a `ListClients` +/// broadcast, serialize this shard's locally-homed connected clients from +/// its `SessionManager` and push them back over the reply sender. The +/// aggregation across all shards happens in +/// [`shard::IggyShard::list_all_clients`]. +pub fn make_list_clients_handler(sessions: &Rc>) -> ListClientsHandler { + let sessions = Rc::clone(sessions); + Rc::new(move |reply| { + let clients: Vec = sessions.borrow().iter_clients().collect(); + // Best-effort: the gather side bounds itself by count + timeout, so + // a dropped reply (receiver gone) just means this shard is omitted. + let _ = reply.try_send(clients); + }) +} + +pub fn make_deferred_replica_message_handler( + shard_handle: &ShellShardHandle, +) -> MessageHandler +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let shard_handle = Rc::clone(shard_handle); + Rc::new(move |_replica_id, message| { + if let Some(shard) = upgrade_shard_handle(&shard_handle) { + shard.dispatch(message); + } + }) +} + +pub fn make_deferred_client_request_handler( + bus: &B, + shard_handle: &ShellShardHandle, + sessions: &Rc>, + system_config: Arc, + max_tokens_per_user: u32, +) -> RequestHandler +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let shard_handle = Rc::clone(shard_handle); + let sessions = Rc::clone(sessions); + let queues: ClientRequestQueues = Rc::new(RefCell::new(HashMap::new())); + let active: ActiveClientRequests = Rc::new(RefCell::new(HashSet::new())); + let sessions_for_disconnect = Rc::clone(&sessions); + let shard_handle_for_disconnect = Rc::clone(&shard_handle); + let bus_for_spawn = (*bus).clone(); + bus.set_client_connection_lost_fn(Rc::new(move |client_id| { + if let Some((vsr_client_id, session)) = sessions_for_disconnect + .borrow_mut() + .remove_connection(client_id) + && let Some(shard) = upgrade_shard_handle(&shard_handle_for_disconnect) + { + submit_disconnect_logout(shard, vsr_client_id, session); + } + })); + Rc::new(move |client_id, message| { + let shard_handle = Rc::clone(&shard_handle); + let sessions = Rc::clone(&sessions); + let system_config = Arc::clone(&system_config); + let queues = Rc::clone(&queues); + let active = Rc::clone(&active); + queues + .borrow_mut() + .entry(client_id) + .or_default() + .push_back(message); + if !active.borrow_mut().insert(client_id) { + return; + } + bus_for_spawn.spawn(async move { + let Some(shard) = upgrade_shard_handle(&shard_handle) else { + active.borrow_mut().remove(&client_id); + return; + }; + drain_client_requests( + shard, + sessions, + system_config, + max_tokens_per_user, + queues, + active, + client_id, + ) + .await; + }); + }) +} + +// Session resume is performed BY THE LOGIN PATH, not by a separate +// credential-free rebind. +// +// A reconnecting client re-authenticates on the new connection and presents +// its previous `client_id` in the login frame; `submit_register_in_process` +// finds the existing table entry, verifies the authenticated user owns it, +// and returns its epoch, so `bind_session` binds the new transport to the +// old entry with its watermark and reply ring intact. That IS the resume. +// +// An earlier revision instead rebound an *unbound* transport straight from +// the table whenever a replicated frame carried a matching +// `(client, session)`, treating that pair as a bearer token. That was wrong +// in four ways, and the combination was a pre-auth session takeover: +// +// - it called `SessionManager::login` itself, so no credential was ever +// presented, and the connection was logged in as the entry's cached +// `user_id`; authority for replicated ops then resolves from the table +// (`resolve_acting_user_id`) and for partition ops from the session +// manager, so BOTH planes ran as the original registrant; +// - the pair carries far less entropy than "client-generated random +// u128" implies: HTTP mints `client_id` from the shard-0 sequential +// counter (`mint_shard_zero_client_id`, seeded at 1 per process) and no +// live path ever bumps an epoch past 1, so the token was `client=N, +// session=1` for small N; +// - `ClientEntry` carries no transport or plane tag, so a raw TCP peer +// could bind an HTTP-originated session; +// - `bind_session` demotes the evicted holder to `Connected`, the one +// state `login` accepts, so the loser's next replicated frame +// re-resumed and stole the session back, unbounded and with no eviction +// frame either way. +// +// Routing resume through login also restores the checks that path owns: +// password / PAT verification, `UserStatus::Active`, PAT expiry, the +// protocol-version gate, and SDK-info recording. +// +// An unbound transport sending a replicated frame therefore gets the typed +// `Eviction(NoSession)` fail-fast below and must log in. + +#[allow(clippy::too_many_arguments)] +fn enqueue_client_request( + shard: Rc>, + sessions: Rc>, + system_config: Arc, + max_tokens_per_user: u32, + queues: ClientRequestQueues, + active: ActiveClientRequests, + client_id: u128, + message: Message, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + queues + .borrow_mut() + .entry(client_id) + .or_default() + .push_back(message); + if !active.borrow_mut().insert(client_id) { + return; + } + + let bus = shard.bus.clone(); + bus.spawn(async move { + drain_client_requests( + shard, + sessions, + system_config, + max_tokens_per_user, + queues, + active, + client_id, + ) + .await; + }); +} + +#[allow(clippy::future_not_send)] +async fn drain_client_requests( + shard: Rc>, + sessions: Rc>, + system_config: Arc, + max_tokens_per_user: u32, + queues: ClientRequestQueues, + active: ActiveClientRequests, + client_id: u128, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + loop { + let Some(message) = pop_next_client_request(&queues, &active, client_id) else { + return; + }; + handle_client_request( + &shard, + &sessions, + &system_config, + max_tokens_per_user, + client_id, + message, + ) + .await; + } +} + +fn pop_next_client_request( + queues: &ClientRequestQueues, + active: &ActiveClientRequests, + client_id: u128, +) -> Option> { + let mut queues = queues.borrow_mut(); + let Some(queue) = queues.get_mut(&client_id) else { + active.borrow_mut().remove(&client_id); + return None; + }; + let message = queue.pop_front(); + if queue.is_empty() { + queues.remove(&client_id); + } + if message.is_none() { + active.borrow_mut().remove(&client_id); + } + message +} + +/// Per-request partitions-count cap, shared by create-topic, create-partitions +/// and delete-partitions admission. Runs pre-consensus like +/// [`validate_topic_bounds`]: an oversized count must not burn a replicated +/// log entry (create-partitions admission would also allocate that many +/// consensus-group ids before replicating). +/// +/// Zero passes here because a zero-partition TOPIC is legal (legacy +/// `create_topic` admits `0..=MAX`); the add/remove requests reject it in +/// [`validate_partitions_change_count`]. +const fn validate_partitions_count(partitions_count: u32) -> Result<(), IggyError> { + if partitions_count > MAX_PARTITIONS_PER_REQUEST { + return Err(IggyError::TooManyPartitions); + } + Ok(()) +} + +/// [`validate_partitions_count`] plus the zero rejection that create-partitions +/// and delete-partitions carry: adding or removing zero partitions is a no-op +/// that would still burn a replicated log entry, bump `Streams::revision` and +/// force every shard through a rebalance pass. Legacy rejects it with +/// `TooManyPartitions` in both handlers (`1..=MAX` on create, `== 0` on +/// delete), so the code matches rather than inventing a new one. +const fn validate_partitions_change_count(partitions_count: u32) -> Result<(), IggyError> { + if partitions_count == 0 { + return Err(IggyError::TooManyPartitions); + } + validate_partitions_count(partitions_count) +} + +/// Static create-topic bounds shared by the TCP and HTTP ingresses. Runs +/// pre-consensus: a rejected request must not burn a replicated log entry, +/// and `prepare_request` errors evict the session instead of denying typed. +/// `ServerDefault` is exempt from the size floor (it resolves against server +/// config at admission, matching legacy); `Unlimited` passes numerically. +/// `segment_size_bytes` is the topic's RESOLVED segment size (explicit +/// option, else this node's default), so a per-topic segment above the +/// global default still floors the topic cap. +pub fn validate_topic_bounds( + partitions_count: u32, + max_topic_size: MaxTopicSize, + segment_size_bytes: u64, +) -> Result<(), IggyError> { + validate_partitions_count(partitions_count)?; + validate_topic_size_floor(max_topic_size, segment_size_bytes) +} + +/// A topic cap below one segment can never be enforced: the first segment +/// already exceeds it. Split out of [`validate_topic_bounds`] because update +/// admission checks the cap without a partitions count to check. +pub fn validate_topic_size_floor( + max_topic_size: MaxTopicSize, + segment_size_bytes: u64, +) -> Result<(), IggyError> { + if !matches!(max_topic_size, MaxTopicSize::ServerDefault) + && max_topic_size.as_bytes_u64() < segment_size_bytes + { + return Err(IggyError::InvalidTopicSize( + max_topic_size, + IggyByteSize::from(segment_size_bytes), + )); + } + Ok(()) +} + +/// Announce an accepted `max_topic_size` the server cannot enforce as written. +/// +/// [`validate_topic_size_floor`] admits any cap of one segment or more, but +/// retention runs PER PARTITION and floors each partition's share at one SEALED +/// segment, which reaches up to one maximum bus frame past `segment_size`. A cap +/// between the two is stored and echoed back verbatim while the server actually +/// keeps `(segment_size + max_message_size) * partitions_count`, so the only +/// moment an operator can be told is the one where they set it. +/// +/// Warns rather than rejects: which caps are accepted is client-visible wire +/// behavior, and tightening it would break topics that already exist. +pub fn warn_unenforceable_topic_size( + max_topic_size: MaxTopicSize, + segment_size_bytes: u64, + max_message_size_bytes: usize, + partitions_count: u32, +) { + let MaxTopicSize::Custom(configured) = max_topic_size else { + return; + }; + let max_message_size_bytes = u64::try_from(max_message_size_bytes).unwrap_or(u64::MAX); + let per_partition_floor = segment_size_bytes.saturating_add(max_message_size_bytes); + let topic_floor = per_partition_floor.saturating_mul(u64::from(partitions_count)); + if configured.as_bytes_u64() >= topic_floor { + return; + } + warn!( + max_topic_size = configured.as_bytes_u64(), + partitions_count, + segment_size = segment_size_bytes, + enforced_per_partition = per_partition_floor, + "{UNENFORCEABLE_TOPIC_SIZE_WARN}" + ); +} + +/// Announce the same unenforceable cap when partitions are ADDED to a topic. +/// +/// The cap is topic-wide but enforcement is per partition, so every added +/// partition shrinks the share: a cap that cleared the floor when the topic was +/// created can stop clearing it here. The request carries only the delta, so +/// the stored cap, segment size and current partition count come from metadata. +pub fn warn_unenforceable_topic_size_on_partition_add( + streams: &Streams, + stream_id: &WireIdentifier, + topic_id: &WireIdentifier, + max_message_size_bytes: usize, + added_partitions_count: u32, +) { + let Some(((stream_slab, topic_slab), _)) = streams.partition_count_context(stream_id, topic_id) + else { + return; + }; + let Some((_, max_topic_size, partitions_count, segment_size)) = + streams.topic_retention_config(stream_slab, topic_slab) + else { + return; + }; + warn_unenforceable_topic_size( + max_topic_size, + segment_size.map_or(iggy_common::DEFAULT_SEGMENT_SIZE, |segment_size| { + segment_size.as_bytes_u64() + }), + max_message_size_bytes, + u32::try_from(partitions_count) + .unwrap_or(u32::MAX) + .saturating_add(added_partitions_count), + ); +} + +/// Reject option keys outside the resource's catalog, pre-consensus. Unknown +/// keys are rejected rather than skipped: a silently ignored knob would hand +/// the client server defaults without it ever learning. Streams and users +/// have no catalog keys yet, so `known` is empty for both until one lands. +pub fn validate_option_keys(options: &WireOptions, known: &[&str]) -> Result<(), IggyError> { + for entry in options { + // Wire validation already enforced UTF-8 string keys. + let key = String::from_utf8_lossy(entry.key); + if !known.contains(&key.as_ref()) { + return Err(IggyError::UnsupportedOptionKey(key.into_owned())); + } + } + Ok(()) +} + +/// Reject a request before it reaches consensus: warn, then send the typed +/// deny reply. A silent drop would wedge every later request on the +/// connection until the socket read timeout. `context` labels the rejection +/// site in both log lines. +#[allow(clippy::future_not_send)] +async fn send_pre_consensus_deny( + shard: &Rc>, + header: &RoutedRequestHeader, + transport_client_id: u128, + error: &IggyError, + context: &'static str, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + warn!( + transport_client_id, + error = %error, + operation = ?header.operation, + context, + "denying request pre-consensus" + ); + let commit = current_metadata_commit(shard); + let reply = build_deny_reply(header, transport_client_id, 0, commit, error.as_code()); + if let Err(send_error) = shard + .bus + .send_to_client(transport_client_id, reply.into_generic().into_frozen()) + .await + { + warn!( + transport_client_id, + error = %send_error, + context, + "failed to send pre-consensus deny reply" + ); + } +} + +#[allow(clippy::future_not_send, clippy::too_many_lines)] +async fn handle_client_request( + shard: &Rc>, + sessions: &Rc>, + system_config: &Arc, + max_tokens_per_user: u32, + transport_client_id: u128, + message: Message, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let request = match message.try_into_typed::() { + Ok(request) => request, + Err(error) => { + warn!( + transport_client_id, + error = %error, + "dropping client request with invalid header" + ); + return; + } + }; + // Promote to the server-internal routed shape at the boundary: the + // client wire carries no group (it is derived -- plane from `operation`, + // partition target from the payload), so it starts unset here and the + // resolution sites below stamp it before anything routes on it. + let request = request.into_routed(); + + // The last point that still sees the body the CLIENT sent; every rewrite below + // substitutes server-chosen bytes and carries the stamp through unchanged. + if let Err(error) = verify_request_checksum(&request) { + warn!( + transport_client_id, + operation = ?request.header().operation, + request = request.header().request, + "dropping client request whose body does not match its own checksum" + ); + send_deny_reply( + shard, + transport_client_id, + request.header(), + error.as_code(), + ) + .await; + return; + } + + ensure_transport_connection(shard, sessions, transport_client_id); + + // Any request is liveness proof, not just PING: an idle-but-active client + // (e.g. an admin issuing reads between long sleeps) must not be evicted by + // the heartbeat verifier. A genuinely dead connection sends nothing, so the + // intended stale-client eviction still fires. No-ops for an unbound client. + sessions.borrow_mut().record_heartbeat(transport_client_id); + + let header = *request.header(); + if header.operation == Operation::NonReplicated { + // Auth bypass guard: `PING`, the liveness probe, is the only pre-auth + // code, on every roster shape. `GET_CLUSTER_METADATA` describes the + // private replica network and is not something an unauthenticated + // caller gets to read; a client that dialed a backup no longer needs + // it to find the leader, because the backup authenticates the login + // locally and forwards only the consensus proposal + // (`submit_register_local_or_forward`). Every other non-replicated + // code MUST go through Register first, which binds the acting user + // the per-op authz gates resolve. + let nr_code = u32::from_le_bytes(request.header().reserved[..4].try_into().unwrap()); + // Legacy (pre-register) login codes. The server authenticates only via + // the Register handshake (LOGIN_REGISTER / LOGIN_REGISTER_WITH_PAT, + // Operation::Register); the vsr SDK funnels both logins there and never + // emits these. Reject them uniformly with a typed MalformedLogin (the + // SDK maps it to InvalidFormat) before the session gate, so a legacy or + // foreign client fails fast instead of getting the generic + // Unauthenticated deny the pre-auth guard would send unbound, or the + // silent empty-ok Reply the bound non-replicated path would send. + if matches!( + nr_code, + LOGIN_USER_CODE | LOGIN_WITH_PERSONAL_ACCESS_TOKEN_CODE + ) { + warn!( + transport_client_id, + code = nr_code, + "rejecting legacy login code; server requires the register handshake" + ); + send_login_eviction( + shard, + transport_client_id, + header.client, + EvictionReason::MalformedLogin, + ) + .await; + return; + } + let allowed_pre_auth = nr_code == PING_CODE; + if !allowed_pre_auth && sessions.borrow().get_session(transport_client_id).is_none() { + // Foreign SDKs still probe `GET_CLUSTER_METADATA` before login + // until they are fixed, so that rejection is routine traffic and + // logs at debug rather than warn. + if nr_code == GET_CLUSTER_METADATA_CODE { + debug!( + transport_client_id, + "denying pre-auth cluster-metadata read with Unauthenticated" + ); + } else { + warn!( + transport_client_id, + code = nr_code, + "denying pre-auth non-replicated read with Unauthenticated" + ); + } + // A plain deny Reply, not an Eviction: there is no session to + // evict, and an Eviction is session-terminal by wire contract, + // so SDKs would tear down the very connection their login is + // about to use. The status channel carries the error the same + // way the request-checksum denial above does. + send_unbound_deny_reply( + shard, + transport_client_id, + request.header(), + IggyError::Unauthenticated.as_code(), + ) + .await; + return; + } + handle_non_replicated_request(shard, sessions, system_config, transport_client_id, request) + .await; + return; + } + + if header.operation == Operation::Register && header.session == 0 && header.request == 0 { + handle_login_register_request(shard, sessions, transport_client_id, request).await; + return; + } + + if header.operation == Operation::Logout { + handle_logout_request(shard, sessions, transport_client_id, request).await; + return; + } + + let bound = sessions.borrow().get_session(transport_client_id); + if bound.is_none() { + // Replicated request on an unbound transport. Without this short- + // circuit, the rewrite below overwrites `header.client` with + // `transport_client_id` and dispatches; the request_preflight then + // rejects with `NoSession`/`Fenced` and the failure disappears + // silently, wedging the SDK until the socket timeout. A typed + // `Eviction(NoSession)` is right here, unlike the pre-auth read + // guard above: a replicated request implies the client believes it + // has a session, and that session is gone, so it must register + // again. An empty status-0 Reply is not safe here, because + // SendMessages is the one replicated operation without a result + // section, and its decoder would read the empty body as a + // successful send. + warn!( + transport_client_id, + operation = ?header.operation, + "rejecting replicated request from unbound transport with Eviction(NoSession)" + ); + send_unauthenticated_eviction(shard, transport_client_id).await; + return; + } + + // DeleteSegments is neither a partition nor a metadata consensus op: the + // owning shard resolves the requested count to a concrete offset, then a + // `TruncatePartition` is replicated through metadata (Option A). Each + // replica's reconciler trims to the committed watermark. Handle it here, + // ahead of the partition/metadata routing below. + if header.operation == Operation::DeleteSegments { + handle_delete_segments_request(shard, transport_client_id, bound, &request).await; + return; + } + + if header.operation.is_partition() { + // `bound` is Some here (unbound transports returned above). + let (vsr_client_id, bound_session) = bound.unwrap_or((0, 0)); + // `get_session` discards the acting user id the partition gate needs; + // resolve it from the same bound connection. A bound transport always + // has one, but the gate fails closed on `None` rather than trust that. + let acting_user_id = sessions.borrow().get_user_id(transport_client_id); + dispatch_partition_request( + shard, + request, + vsr_client_id, + bound_session, + transport_client_id, + acting_user_id, + ) + .await; + return; + } + + let request = request.transmute_header(|header, new_header: &mut RoutedRequestHeader| { + *new_header = header; + // Metadata-plane ops route by operation: stamp the sentinel group. + new_header.group = server_common::sharding::METADATA_GROUP; + // `bound` is always Some here (unbound transports early-return above); + // this sets the consensus client id + session for the replicated op. + if let Some((bound_client_id, bound_session)) = bound { + new_header.client = bound_client_id; + new_header.session = bound_session; + } + }); + let (request, raw_pat_token) = match maybe_rewrite_pat_request( + sessions, + transport_client_id, + max_tokens_per_user, + |user_id| { + shard + .plane + .metadata() + .mux_stm + .users() + .read(|users| users.pat_count_of(user_id)) + }, + request, + ) { + Ok(rewritten) => rewritten, + Err(error) => { + // Token cap reached, malformed body, or a lost session binding. + send_pre_consensus_deny( + shard, + &header, + transport_client_id, + &error, + "personal-access-token", + ) + .await; + return; + } + }; + // Hash raw passwords and, for ChangePassword, verify the current password + // on the primary before replication; see `crate::users`. Replicas store the + // hash directly. A wrong current password is not denied here: it rides + // consensus and applies as a committed InvalidCredentials no-op, so the only + // Err returned is a malformed body. + let request = match maybe_rewrite_user_password_request(shard, request) { + Ok(rewritten) => rewritten, + Err(error) => { + // Malformed body: deny fast with InvalidCommand. + send_pre_consensus_deny(shard, &header, transport_client_id, &error, "user-password") + .await; + return; + } + }; + // Static bounds run pre-consensus so a rejected request burns no + // replicated log entry; HTTP covers the same bounds via + // `command.validate()`. A body that fails to decode denies typed too + // (`InvalidCommand`), instead of riding consensus just to fail there. + let bounds = match header.operation { + Operation::CreateTopic => CreateTopicRequest::decode_from(request_body(&request)) + .map_err(|_| IggyError::InvalidCommand) + .and_then(|create_topic| { + // `parse` doubles as the catalog gate: an unknown key or a + // malformed value denies typed here, pre-consensus. + let options = TopicCreateOptions::parse(&create_topic.options)?; + if let Some(segment_size) = options.segment_size { + validate_topic_segment_size( + segment_size.as_bytes_u64(), + iggy_common::MAX_TOPIC_SEGMENT_SIZE, + )?; + } + let segment_size = options.segment_size.map_or_else( + || iggy_common::DEFAULT_SEGMENT_SIZE, + |segment_size| segment_size.as_bytes_u64(), + ); + if options + .preallocate_segments + .unwrap_or(iggy_common::DEFAULT_PREALLOCATE_SEGMENTS) + { + validate_preallocated_topic_bytes(segment_size, create_topic.partitions_count)?; + } + let max_topic_size = options + .max_topic_size + .unwrap_or(MaxTopicSize::ServerDefault); + validate_topic_bounds(create_topic.partitions_count, max_topic_size, segment_size)?; + warn_unenforceable_topic_size( + max_topic_size, + segment_size, + shard.bus_max_message_size(), + create_topic.partitions_count, + ); + Ok(()) + }), + Operation::CreatePartitions => CreatePartitionsRequest::decode_from(request_body(&request)) + .map_err(|_| IggyError::InvalidCommand) + .and_then(|create_partitions| { + validate_partitions_change_count(create_partitions.partitions_count)?; + let metadata = shard.plane.metadata(); + warn_unenforceable_topic_size_on_partition_add( + metadata.mux_stm.streams(), + &create_partitions.stream_id, + &create_partitions.topic_id, + shard.bus_max_message_size(), + create_partitions.partitions_count, + ); + Ok(()) + }), + Operation::DeletePartitions => DeletePartitionsRequest::decode_from(request_body(&request)) + .map_err(|_| IggyError::InvalidCommand) + .and_then(|delete_partitions| { + validate_partitions_change_count(delete_partitions.partitions_count) + }), + // Only the updatable subset: the create-time knobs are pushed to + // partitions when the topic is built and nothing re-pushes them, so + // accepting one here would store a value no partition ever sees. + Operation::UpdateTopic => UpdateTopicRequest::decode_from(request_body(&request)) + .map_err(|_| IggyError::InvalidCommand) + .and_then(|update_topic| { + validate_option_keys(&update_topic.options, UPDATABLE_TOPIC_OPTION_KEYS)?; + let options = TopicCreateOptions::parse(&update_topic.options)?; + let Some(max_topic_size) = options.max_topic_size else { + return Ok(()); + }; + // An update can lower the cap below one segment just as a + // create can, and the stored map would then report a size the + // topic can never enforce. The floor is this topic's own + // segment size, since that key is create-only. + let metadata = shard.plane.metadata(); + let streams = metadata.mux_stm.streams(); + let segment_size = streams + .topic_segment_size(&update_topic.stream_id, &update_topic.topic_id) + .map_or_else( + || iggy_common::DEFAULT_SEGMENT_SIZE, + |segment_size| segment_size.as_bytes_u64(), + ); + validate_topic_size_floor(max_topic_size, segment_size)?; + let partitions_count = streams + .topic_partitions_count(&update_topic.stream_id, &update_topic.topic_id) + .unwrap_or(0); + warn_unenforceable_topic_size( + max_topic_size, + segment_size, + shard.bus_max_message_size(), + u32::try_from(partitions_count).unwrap_or(u32::MAX), + ); + Ok(()) + }), + Operation::UpdateStream => UpdateStreamRequest::decode_from(request_body(&request)) + .map_err(|_| IggyError::InvalidCommand) + .and_then(|update_stream| { + validate_option_keys(&update_stream.options, UPDATABLE_STREAM_OPTION_KEYS) + }), + Operation::UpdateUser => UpdateUserRequest::decode_from(request_body(&request)) + .map_err(|_| IggyError::InvalidCommand) + .and_then(|update_user| { + validate_option_keys(&update_user.options, UPDATABLE_USER_OPTION_KEYS) + }), + Operation::CreateStream => CreateStreamRequest::decode_from(request_body(&request)) + .map_err(|_| IggyError::InvalidCommand) + .and_then(|create_stream| validate_option_keys(&create_stream.options, &[])), + Operation::CreateUser => CreateUserRequest::decode_from(request_body(&request)) + .map_err(|_| IggyError::InvalidCommand) + .and_then(|create_user| validate_option_keys(&create_user.options, &[])), + _ => Ok(()), + }; + if let Err(error) = bounds { + send_pre_consensus_deny(shard, &header, transport_client_id, &error, "static-bounds").await; + return; + } + // Enrich consumer-group Join/Leave with the client's VSR id (+ topic + // partition count for Join) before replication; see `crate::consumer_group`. + let request = match maybe_rewrite_consumer_group_request(shard, request).await { + Ok(rewritten) => rewritten, + Err(error) => { + warn!( + transport_client_id, + error = %error, + operation = ?header.operation, + "dropping consumer-group request with invalid payload" + ); + return; + } + }; + let request_header = *request.header(); + // Replicated request: run consensus on the metadata owner (shard 0) and + // bring the committed reply back here. This shard owns the connection, + // so it writes the reply to the socket via the transport client id -- + // shard 0 can't route by the consensus client id (no home-shard bits). + match submit_client_request_on_owner(shard, request).await { + Some(reply) => { + // The raw PAT token never enters consensus (it is non-deterministic + // and secret), so the committed reply body is empty. Substitute the + // raw-token response here, on the minting client's home shard, using + // the confirmed commit position from the committed reply. + let reply = match build_raw_pat_reply(&request_header, reply, raw_pat_token) { + Ok(reply) => reply, + Err(error) => { + warn!( + transport_client_id, + error = %error, + "failed to build raw PAT reply" + ); + return; + } + }; + if let Err(error) = shard + .bus + .send_to_client(transport_client_id, reply.into_frozen()) + .await + { + warn!( + transport_client_id, + error = %error, + operation = ?header.operation, + "failed to deliver committed reply to client" + ); + } + } + None => { + // Transient submit failure (not primary / not caught up / dedup + // absorbed). Stay silent; the SDK read-timeout replays. + warn!( + transport_client_id, + operation = ?header.operation, + "replicated request not committed (transient); client will replay" + ); + } + } +} + +/// Send a non-replicated reply body to a client, stamping the current +/// metadata commit. Shared by the `get_me` / `get_clients` / `get_client` +/// arms. +#[allow(clippy::future_not_send)] +async fn send_non_replicated_bytes( + shard: &Rc>, + request: &Message, + transport_client_id: u128, + bytes: Bytes, + label: &'static str, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let commit = current_metadata_commit(shard); + let reply = NonReplicatedResponse::Bytes(bytes).into_reply( + request.header(), + request.header().client, + request.header().session, + commit, + ); + send_reply_frame( + shard, + transport_client_id, + reply.into_generic().into_frozen(), + label, + ) + .await; +} + +/// Hand a built reply frame to the bus for `transport_client_id`. +#[allow(clippy::future_not_send)] +async fn send_reply_frame( + shard: &Rc>, + transport_client_id: u128, + frame: impl Into, + label: &'static str, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + if let Err(error) = shard.bus.send_to_client(transport_client_id, frame).await { + warn!(transport_client_id, label, error = %error, "failed to send non-replicated reply"); + } +} + +fn ensure_transport_connection( + shard: &Rc>, + sessions: &Rc>, + transport_client_id: u128, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let Some(meta) = shard.bus.client_meta(transport_client_id) else { + return; + }; + sessions + .borrow_mut() + .ensure_connection(transport_client_id, meta.peer_addr, meta.transport); +} + +pub(in crate::dispatch) fn upgrade_shard_handle( + shard_handle: &ShellShardHandle, +) -> Option>> +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + shard_handle + .borrow() + .as_ref() + .and_then(std::rc::Weak::upgrade) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::cluster_meta::ClusterRoster; + use crate::dispatch::test_support::{FIRST_BOOT, SpyBus, TestMux, TestShard, test_shard}; + use iggy_binary_protocol::Command; + use metadata::IggyMetadata; + use partitions::{IggyPartitions, PartitionPathLayout, PartitionsConfig}; + use server_common::MESSAGE_ALIGN; + use server_common::sharding::ShardId; + use shard::metrics::ShardMetrics; + use shard::shards_table::PapayaShardsTable; + use shard::{ + LifecycleFrame, PartitionConsensusConfig, ReplicaTopology, ShardFrame, ShardIdentity, + shard_channel, + }; + use std::mem::size_of; + use std::sync::atomic::AtomicBool; + + /// A test shard wired to its own lanes (the held sender feeds them), + /// for the reply-lane pump tests below. + fn reply_lane_test_shard(name: &str) -> (SpyBus, shard::TaggedSender, Rc) { + let bus = SpyBus::default(); + let metadata = IggyMetadata::new(None, None, None, None, TestMux::default(), None); + let partitions = IggyPartitions::new( + ShardId::new(0), + PartitionsConfig { + messages_required_to_save: 1, + size_of_messages_required_to_save: iggy_common::IggyByteSize::from(1024_u64), + enforce_fsync: false, + validate_checksum: true, + segment_size: iggy_common::IggyByteSize::from(1_048_576_u64), + preallocate_segments: false, + encryptor: None, + path_layout: PartitionPathLayout::default(), + }, + ); + let (sender, inbox_rx, reply_inbox_rx) = shard_channel(0, 16, 16); + let lane_sender = sender.clone(); + let shard = TestShard::new( + ShardIdentity::new(0, name.to_string()), + bus.clone(), + Rc::new(|_, _| {}), + Rc::new(|_, _| {}), + Rc::new(|_| {}), + Rc::new(|_| {}), + Rc::new(|_, _, _| {}), + metadata, + partitions, + vec![sender], + inbox_rx, + reply_inbox_rx, + PapayaShardsTable::new(), + PartitionConsensusConfig::new(1, ReplicaTopology::new(0, 1), bus.clone()), + None, + ShardMetrics::for_shard(), + ) + .expect("single-sender ring is canonically ordered"); + (bus, lane_sender, Rc::new(shard)) + } + + fn reply_lane_forward(client_id: u128) -> ShardFrame { + ShardFrame::lifecycle(LifecycleFrame::ForwardClientSend { + client_id, + msg: server_common::iobuf::Frozen::from( + server_common::iobuf::Owned::::zeroed(64), + ) + .into(), + }) + } + + /// A frame on the reply lane must reach the client through the RUNNING + /// pump's reply arm: the lane split moved `ForwardClientSend` off the + /// main inbox, so a pump that forgot to service the new lane would + /// strand every cross-shard reply while the send sites happily report + /// success. + #[compio::test] + async fn pump_live_arm_delivers_reply_lane_forwards() { + const TRANSPORT: u128 = 92; + let (bus, lane_sender, shard) = reply_lane_test_shard("reply-lane-live-arm-test"); + + let (stop_tx, stop_rx) = shard::channel::<()>(1); + let pump_shard = Rc::clone(&shard); + let pump = compio::runtime::spawn(async move { + pump_shard + .run_message_pump(stop_rx, Arc::new(AtomicBool::new(false))) + .await; + }); + + lane_sender + .reply_sender() + .try_send(reply_lane_forward(TRANSPORT)) + .expect("reply lane has capacity"); + + // The pump is idle on the main lane, so its bottom reply arm must + // serve the frame without any main-lane traffic or shutdown drain. + let mut delivered = false; + for _ in 0..500 { + if !bus.client_replies.borrow().is_empty() { + delivered = true; + break; + } + compio::time::sleep(std::time::Duration::from_millis(1)).await; + } + stop_tx.try_send(()).expect("stop channel has capacity"); + let _ = pump.await; + + assert!( + delivered, + "the live reply arm must deliver a forward while the pump runs" + ); + let replies = bus.client_replies.borrow(); + assert_eq!(replies[0].0, TRANSPORT, "forward must reach its client"); + } + + /// The shutdown path must ALSO deliver reply-lane frames: a forward + /// already accepted by the lane when the stop signal wins the biased + /// select would otherwise be silently destroyed at teardown. + #[compio::test] + async fn pump_shutdown_drain_delivers_reply_lane_forwards() { + const TRANSPORT: u128 = 93; + let (bus, lane_sender, shard) = reply_lane_test_shard("reply-lane-drain-test"); + + lane_sender + .reply_sender() + .try_send(reply_lane_forward(TRANSPORT)) + .expect("reply lane has capacity"); + + // Pre-armed stop: the pump exits through the biased stop arm and the + // post-loop drain must still deliver the reply-lane frame. + let (stop_tx, stop_rx) = shard::channel::<()>(1); + stop_tx.try_send(()).expect("stop channel has capacity"); + shard + .run_message_pump(stop_rx, Arc::new(AtomicBool::new(false))) + .await; + + let replies = bus.client_replies.borrow(); + assert_eq!( + replies.len(), + 1, + "the pump's reply-lane drain must deliver the forwarded reply" + ); + assert_eq!( + replies[0].0, TRANSPORT, + "the forward must reach the client it was addressed to" + ); + } + + /// The `GET_CLUSTER_METADATA` auth gate holds on every roster shape: it + /// describes the private replica network, and a client that dialed a + /// backup reaches the cluster by logging in there (the backup forwards + /// the register), not by reading the topology first. + /// + /// The denial must be a plain Reply on the status channel, not an + /// Eviction: no session exists yet, and a session-terminal frame makes + /// SDKs drop the connection their login is about to use. + #[compio::test] + async fn pre_auth_cluster_metadata_denied_on_every_roster() { + use configs::cluster::{ClusterNodeConfig, TransportPorts}; + use iggy_binary_protocol::codes::GET_CLUSTER_METADATA_CODE; + use iggy_binary_protocol::{GenericHeader, ReplyHeader}; + + const TRANSPORT: u128 = 91; + const COMMAND_OFFSET: usize = std::mem::offset_of!(GenericHeader, command); + const STATUS_OFFSET: usize = std::mem::offset_of!(ReplyHeader, status); + const OP_OFFSET: usize = std::mem::offset_of!(ReplyHeader, op); + const COMMIT_OFFSET: usize = std::mem::offset_of!(ReplyHeader, commit); + + fn metadata_read() -> Message { + let header_size = size_of::(); + let mut message = Message::::new(header_size); + { + let header = bytemuck::checked::from_bytes_mut::( + &mut message.as_mut_slice()[..header_size], + ); + *header = RequestHeader { + command: Command::Request, + operation: Operation::NonReplicated, + size: u32::try_from(header_size).expect("header fits u32"), + client: TRANSPORT, + ..Default::default() + }; + header.reserved[..4].copy_from_slice(&GET_CLUSTER_METADATA_CODE.to_le_bytes()); + } + message.into_generic() + } + + fn roster_node(name: &str) -> ClusterNodeConfig { + ClusterNodeConfig { + name: name.to_owned(), + ip: "127.0.0.1".to_owned(), + advertised_address: None, + advertised_addresses: Vec::new(), + replica_id: 0, + ports: TransportPorts::default(), + } + } + + let bus = SpyBus::default(); + let shard = Rc::new(test_shard(&bus, 0, 1, FIRST_BOOT)); + let sessions = Rc::new(RefCell::new(SessionManager::new())); + let system_config = Arc::new(ServerSystemConfig::default()); + + let multi_node = Rc::new(ClusterRoster { + enabled: true, + name: "test-cluster".to_owned(), + nodes: ["node-0", "node-1"] + .map(|name| { + configs::cluster::ResolvedClusterNode::try_from(roster_node(name)) + .expect("valid roster node") + }) + .to_vec(), + self_advertised: "127.0.0.1".to_owned(), + self_ports: TransportPorts::default(), + metadata_view: Arc::new(std::sync::atomic::AtomicU64::new( + crate::cluster_meta::METADATA_VIEW_UNKNOWN, + )), + }); + // Default roster is disabled / single node; the installed one is a + // real cluster. Neither serves an unbound caller. + for roster in [None, Some(multi_node)] { + if let Some(roster) = roster { + sessions.borrow_mut().set_cluster_roster(roster); + } + handle_client_request( + &shard, + &sessions, + &system_config, + 1, + TRANSPORT, + metadata_read(), + ) + .await; + let replies = bus.client_replies.borrow(); + assert_eq!(replies.len(), 1, "gated read must still produce a frame"); + let (client, frame) = &replies[0]; + assert_eq!(*client, TRANSPORT); + assert_eq!( + frame[COMMAND_OFFSET], + Command::Reply as u8, + "an unbound cluster-metadata read must be denied with a Reply, not evicted" + ); + let status = + u32::from_le_bytes(frame[STATUS_OFFSET..STATUS_OFFSET + 4].try_into().unwrap()); + assert_eq!( + status, + IggyError::Unauthenticated.as_code(), + "deny reply status must be Unauthenticated" + ); + let op = u64::from_le_bytes(frame[OP_OFFSET..OP_OFFSET + 8].try_into().unwrap()); + assert_eq!(op, 0, "pre-auth deny carries no session, so op must be 0"); + let commit = + u64::from_le_bytes(frame[COMMIT_OFFSET..COMMIT_OFFSET + 8].try_into().unwrap()); + assert_eq!(commit, 0, "pre-auth deny must not disclose commit activity"); + drop(replies); + bus.client_replies.borrow_mut().clear(); + } + } + + #[test] + fn create_topic_bounds_deny_pre_consensus() { + let segment_size = iggy_common::DEFAULT_SEGMENT_SIZE; + assert!(segment_size > 0, "default segment size must be nonzero"); + + assert!( + validate_topic_bounds( + MAX_PARTITIONS_PER_REQUEST, + MaxTopicSize::ServerDefault, + segment_size + ) + .is_ok(), + "the partition cap itself is admissible" + ); + assert!( + matches!( + validate_topic_bounds( + MAX_PARTITIONS_PER_REQUEST + 1, + MaxTopicSize::ServerDefault, + segment_size + ), + Err(IggyError::TooManyPartitions) + ), + "one past the partition cap must deny" + ); + // ServerDefault is numerically 0 yet exempt from the segment-size + // floor: it resolves against server config, matching legacy. + assert!(validate_topic_bounds(1, MaxTopicSize::ServerDefault, segment_size).is_ok()); + assert!(validate_topic_bounds(1, MaxTopicSize::Unlimited, segment_size).is_ok()); + let below_floor = MaxTopicSize::Custom((segment_size - 1).into()); + assert!( + matches!( + validate_topic_bounds(1, below_floor, segment_size), + Err(IggyError::InvalidTopicSize(size, floor)) + if size == below_floor && floor == IggyByteSize::from(segment_size) + ), + "custom size below the segment size must deny with the bounds" + ); + let at_floor = MaxTopicSize::Custom(IggyByteSize::from(segment_size)); + assert!( + validate_topic_bounds(1, at_floor, segment_size).is_ok(), + "a topic exactly one segment large is admissible" + ); + } + + #[test] + fn partitions_count_cap_denies_pre_consensus() { + assert!( + validate_partitions_count(MAX_PARTITIONS_PER_REQUEST).is_ok(), + "the cap itself is admissible" + ); + assert!( + matches!( + validate_partitions_count(MAX_PARTITIONS_PER_REQUEST + 1), + Err(IggyError::TooManyPartitions) + ), + "one past the cap must deny" + ); + // Zero passes the shared cap because a zero-partition TOPIC is legal + // (legacy `create_topic` admits `0..=MAX`). + assert!(validate_partitions_count(0).is_ok()); + } + + #[test] + fn zero_partitions_change_denies_pre_consensus() { + // Adding or removing zero partitions is a no-op that would still burn + // a replicated log entry and force a rebalance. Legacy rejects it with + // `TooManyPartitions` in both handlers, so the code matches. + assert!( + matches!( + validate_partitions_change_count(0), + Err(IggyError::TooManyPartitions) + ), + "adding or removing zero partitions must deny" + ); + assert!(validate_partitions_change_count(1).is_ok()); + assert!(validate_partitions_change_count(MAX_PARTITIONS_PER_REQUEST).is_ok()); + assert!( + matches!( + validate_partitions_change_count(MAX_PARTITIONS_PER_REQUEST + 1), + Err(IggyError::TooManyPartitions) + ), + "the cap still applies" + ); + } +} diff --git a/core/server/src/dispatch/partition.rs b/core/server/src/dispatch/partition.rs new file mode 100644 index 0000000000..4da515f8e5 --- /dev/null +++ b/core/server/src/dispatch/partition.rs @@ -0,0 +1,1497 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! The partition data plane, both sides of the mesh on one page. +//! +//! Server side: [`make_partition_read_handler`] answers `PartitionRead` +//! frames on the owning shard (poll snapshots, consumer offsets, segment +//! deletes). Client side: the funnel routes partition writes through +//! [`dispatch_partition_request`], and the non-replicated read arms call +//! [`handle_poll_messages`] / [`handle_get_consumer_offset`], which read via +//! the shard mesh. +//! +//! Deliberate asymmetry (the plane-split reply trio): the partitions engine +//! replies to committed writes itself, straight from the owning shard, while +//! everything the host builds here -- read bodies, denies, empty-poll shapes +//! -- goes out on the connection's home shard. The reply path is therefore +//! split by plane, not unified, and the deny helpers in `authz` are the +//! third leg (typed status replies for requests that never reach a plane). + +use crate::consumer_group::maybe_rewrite_consumer_offset_request; +use crate::dispatch::authz::{ + authorize_partition_op, authorize_partition_read, send_deny_reply, send_non_replicated_deny, +}; +use crate::dispatch::submit::submit_client_request_on_owner; +use crate::dispatch::{send_non_replicated_bytes, send_reply_frame, upgrade_shard_handle}; +use crate::responses::{ + build_consumer_offset_body, build_empty_reply, build_polled_messages_reply, + current_metadata_commit, resolve_partition_namespace, resolve_partition_request_namespace, +}; +use crate::shell::{ShellBus, ShellShard, ShellShardHandle}; +use crate::wire::{request_body, usize_to_u32}; +use bytes::Bytes; +use consensus::{Consensus, MetadataHandle, PartitionsHandle, build_result_rejection_reply}; +use iggy_binary_protocol::PrepareHeader; +use iggy_binary_protocol::primitives::consumer::WireConsumer; +use iggy_binary_protocol::primitives::polling_strategy::WirePollingStrategy; +use iggy_binary_protocol::requests::consumer_offsets::{ + GetConsumerOffsetRequest, StoreConsumerOffsetRequest, +}; +use iggy_binary_protocol::requests::messages::PollMessagesRequest; +use iggy_binary_protocol::requests::segments::DeleteSegmentsRequest; +use iggy_binary_protocol::{ + AckLevel, Command, KIND_CONSUMER_GROUP, Operation, RoutedRequestHeader, WireDecode, WireEncode, + WireIdentifier, +}; +use iggy_common::{IggyError, PollingStrategy}; +use journal::superblock::SuperblockStore; +use journal::{Journal, JournalHandle}; +use message_bus::AUTO_COMMIT_CLIENT_ID; +use metadata::impls::metadata::{ + StreamsFrontend, build_truncate_partition_client_message, + build_truncate_partition_client_message_with_identifiers, +}; +use partitions::{AutoCommitApplied, PollPlan, PollingArgs, PollingConsumer}; +use server_common::Message; +use server_common::sharding::IggyNamespace; +use shard::shards_table::ShardsTable; +use shard::{PartitionRead, PartitionReadHandler, PartitionReadReply}; +use std::rc::Rc; +use tracing::{debug, warn}; + +/// Build the per-shard [`PartitionReadHandler`]: on a `PartitionRead` frame +/// (this shard owns the namespace), run the poll / consumer-offset lookup +/// against the local partitions plane and push the result back over the +/// carried reply sender. The requesting shard bounds the wait with a +/// timeout, so a dropped reply degrades to a client-visible read failure. +pub fn make_partition_read_handler( + shard_handle: &ShellShardHandle, +) -> PartitionReadHandler +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let shard_handle = Rc::clone(shard_handle); + // Runs synchronously on the shard pump (see `process_lifecycle` -> + // `on_partition_read`). `build_poll_snapshot` takes a pump-only `&mut` + // partition borrow (synchronous, so no sibling task can realloc under it) and + // returns an owned `PollPlan`; only owned data crosses into `spawn_poll_io`. A + // fully-resident poll replies here without spawning. See the `poll_plan` module docs. + Rc::new(move |namespace, read, reply| { + let Some(shard) = upgrade_shard_handle(&shard_handle) else { + return; + }; + let partitions = shard.plane.partitions(); + match read { + PartitionRead::Poll { consumer, args } => { + match partitions.build_poll_snapshot(&namespace, consumer, &args) { + None => { + let _ = reply.try_send(PartitionReadReply::NotFound); + } + Some(plan) if plan.needs_off_pump_io() => { + spawn_poll_io(Rc::clone(&shard), namespace, plan, reply); + } + Some(plan) => { + let (fragments, current_offset, auto_commit) = plan.execute_resident(); + if let Some(applied) = auto_commit { + submit_auto_commit(&shard, namespace, &applied); + } + let _ = reply.try_send(PartitionReadReply::Poll { + fragments, + current_offset, + }); + } + } + } + PartitionRead::ConsumerOffset { consumer } => { + let result = match partitions.consumer_offset_read(&namespace, consumer) { + Some((stored, current_offset)) => PartitionReadReply::ConsumerOffset { + stored, + current_offset, + }, + None => PartitionReadReply::NotFound, + }; + let _ = reply.try_send(result); + } + PartitionRead::GroupOffsetState { group_id } => { + let result = match partitions.group_offset_state(&namespace, group_id) { + Some((last_polled, committed)) => PartitionReadReply::GroupOffsetState { + last_polled, + committed, + }, + None => PartitionReadReply::NotFound, + }; + let _ = reply.try_send(result); + } + PartitionRead::ClearGroupLastPolled { group_id } => { + let result = match partitions.clear_group_last_polled(&namespace, group_id) { + Some(()) => PartitionReadReply::Ack, + None => PartitionReadReply::NotFound, + }; + let _ = reply.try_send(result); + } + PartitionRead::ResolveSegmentDeleteOffset { count } => { + let result = partitions + .segment_delete_resolution(&namespace, count) + .map_or_else( + || PartitionReadReply::NotFound, + |(up_to_offset, lagging)| PartitionReadReply::SegmentDeleteOffset { + up_to_offset, + lagging, + }, + ); + let _ = reply.try_send(result); + } + } + }) +} + +/// Spawn the off-pump leg of a partition poll: disk read + auto-commit apply on +/// the OWNED plan (disk descriptors, resident-tail `Frozen` clones, `Arc` offset +/// map), then replicate the auto-committed offset and send the reply. Holds no +/// partition reference across the IO, so it is sound concurrently with the +/// pump's `&mut` writes; the auto-commit submit re-borrows synchronously after. +fn spawn_poll_io( + shard: Rc>, + namespace: IggyNamespace, + plan: PollPlan, + reply: shard::Sender, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let bus = shard.bus.clone(); + bus.spawn(async move { + // Diagnostic-only wall clock: `elapsed` gates the slow-poll `warn!` + // below and is never folded into a reply or the deterministic schedule, + // so it stays sound under the simulator's virtual clock (there it just + // measures near-zero real time and never fires). Do not derive any + // replicated or reply value from it, or replay determinism breaks. + let poll_started = std::time::Instant::now(); + let (fragments, current_offset, auto_commit) = plan.execute().await; + let elapsed = poll_started.elapsed(); + if elapsed > std::time::Duration::from_secs(1) { + warn!( + namespace_raw = namespace.inner(), + elapsed_ms = u64::try_from(elapsed.as_millis()).unwrap_or(u64::MAX), + "slow partition poll; gather side may have timed out" + ); + } + // Fire-and-forget: the poll reply is not gated on the offset commit. + if let Some(applied) = auto_commit { + submit_auto_commit(&shard, namespace, &applied); + } + let _ = reply.try_send(PartitionReadReply::Poll { + fragments, + current_offset, + }); + }); +} + +/// Replicate a poll's auto-committed offset through the partition consensus so +/// it survives failover, mirroring the explicit `StoreConsumerOffset` path: the +/// same op code, submitted onto the owning shard's own pipeline. Best-effort and +/// fire-and-forget -- the poll reply never waits on it, and a full inbox drops +/// the op at WARN rather than backpressuring the reply. +/// +/// The partition plane admits writes on the primary only (it asserts so), and a +/// poll is served on whichever node owns the namespace locally, which may be a +/// backup. So gate on primary status here and drop at WARN otherwise; auto-commit +/// is server-managed best-effort (at-least-once delivery), so a follower-served +/// poll simply does not advance the durable offset. +/// +/// Coalescing: an offset the partition's committed high-water already covers is +/// dropped without a consensus op (the steady state for a re-poll of committed +/// data, hence no log). The gate reads committed state only, so an offset that +/// merely sits in flight keeps resubmitting until its covering op commits -- a +/// dropped op self-heals on the next poll instead of being suppressed forever. +fn submit_auto_commit( + shard: &Rc>, + namespace: IggyNamespace, + applied: &AutoCommitApplied, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + enum AutoCommitGate { + Submit, + Covered, + NotPrimary, + } + let gate = shard + .plane + .partitions() + .with_partition(&namespace, |partition| { + let consensus = partition.consensus(); + if !(consensus.is_primary() && consensus.is_normal() && !consensus.is_transferring()) { + AutoCommitGate::NotPrimary + } else if partition.is_auto_commit_offset_covered( + applied.kind, + applied.consumer_id, + applied.offset, + ) { + AutoCommitGate::Covered + } else { + AutoCommitGate::Submit + } + }); + match gate { + Some(AutoCommitGate::Submit) => {} + Some(AutoCommitGate::Covered) => return, + Some(AutoCommitGate::NotPrimary) | None => { + warn!( + namespace_raw = namespace.inner(), + "auto-commit offset not replicated: partition not primary on this node (best-effort)" + ); + return; + } + } + let message = match build_auto_commit_request(namespace, applied) { + Ok(message) => message, + Err(error) => { + warn!( + namespace_raw = namespace.inner(), + error = %error, + "failed to build auto-commit store-offset request" + ); + return; + } + }; + // Routes by namespace to this same (owning, primary) shard's inbox; the pump + // admits it next turn exactly like a client store. `dispatch` never blocks. + shard.dispatch(message.into_generic()); +} + +/// Build the synthetic `StoreConsumerOffset` request for an auto-commit, keyed +/// to the resolved numeric consumer/group id and stamped with the reserved +/// [`AUTO_COMMIT_CLIENT_ID`] so the commit path skips the (unwaited) reply. The +/// wire stream/topic ids are cosmetic here -- admission and apply key off the +/// header namespace and the consumer id -- but are set from the namespace for a +/// well-formed body. `ack` is `Quorum` so the offset actually replicates. +fn build_auto_commit_request( + namespace: IggyNamespace, + applied: &AutoCommitApplied, +) -> Result, IggyError> { + let request = StoreConsumerOffsetRequest { + consumer: WireConsumer { + kind: applied.kind.as_code(), + id: WireIdentifier::Numeric(applied.consumer_id), + }, + stream_id: WireIdentifier::Numeric(usize_to_u32(namespace.stream_id())?), + topic_id: WireIdentifier::Numeric(usize_to_u32(namespace.topic_id())?), + partition_id: Some(usize_to_u32(namespace.partition_id())?), + offset: applied.offset, + ack: AckLevel::Quorum, + }; + let body = request.to_bytes(); + let header_size = std::mem::size_of::(); + let total_size = header_size + body.len(); + let size = u32::try_from(total_size).map_err(|_| IggyError::InvalidConfiguration)?; + let mut message = Message::::new(total_size); + message.as_mut_slice()[header_size..].copy_from_slice(&body); + Ok( + message.transmute_header(|_, header: &mut RoutedRequestHeader| { + *header = RoutedRequestHeader { + command: Command::Request, + operation: Operation::StoreConsumerOffset, + size, + client: AUTO_COMMIT_CLIENT_ID, + // The partition plane is sessionless (no `ClientTable` dedup); a + // nonzero session + request just satisfy the wire header + // validation. + session: 1, + request: 1, + group: namespace.inner(), + ..Default::default() + }; + }), + ) +} + +/// Route a partition data-plane op (`SendMessages` / consumer-offset writes) +/// through the shard mesh by namespace: the op belongs to the partition's +/// own consensus group, not the metadata group. The owning shard's +/// partitions plane runs at-least-once consensus and replies directly via +/// `send_to_client`. `header.client` therefore stays the TRANSPORT id +/// (home-shard routing bits), not the VSR session id -- partition ops are +/// sessionless ("session lifecycle is metadata-only"). +/// +/// Callers must have authenticated the transport already: `vsr_client_id` / +/// `bound_session` come from its bound VSR session. Every failure before +/// dispatch replies with a nonzero status -- unresolvable namespace, +/// authorization denial, exhausted routable wait -- so the client fails fast +/// instead of wedging on a silent drop or reading a status-0 frame as a +/// committed write. +/// +/// `vsr_client_id` keys the consumer-group offset fence (the member id), +/// not the transport id stamped into the partition-op header. +#[allow(clippy::future_not_send)] +pub async fn dispatch_partition_request( + shard: &Rc>, + request: Message, + vsr_client_id: u128, + bound_session: u64, + transport_client_id: u128, + acting_user_id: Option, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let header = *request.header(); + let namespace = match resolve_partition_request_namespace( + shard, + header.operation, + request_body(&request), + vsr_client_id, + ) { + Ok(namespace) => namespace, + Err(error) => { + // A partition op against a stream/topic that no longer resolves + // (e.g. a consumer's trailing auto-commit racing a `delete_stream`, + // or an explicit partition id that skipped the client-side + // resolve). The op never reached the partition plane, so a status-0 + // reply would read as a committed ack for work that never happened. + // A silent drop is no better: the SDK connection processes replies + // in lockstep and would wedge forever. + warn!( + transport_client_id, + error = %error, + operation = ?header.operation, + "partition request with unresolved namespace; replying denied" + ); + send_deny_reply( + shard, + transport_client_id, + &header, + IggyError::ResourceNotFound(String::new()).as_code(), + ) + .await; + return; + } + }; + // Dispatch-time RBAC. The partition plane is not replicated through the + // metadata STM, so the in-apply gate cannot cover it; authorize here, on + // the connection's own shard, before burning the routable wait or touching + // the plane. The namespace resolved above, so its stream/topic are the + // committed slab ids the permissioner keys on directly. A denial replies + // the op's frame with an empty body and a nonzero `status` the SDK peeks. + // + // Consistency: this reads THIS shard's local committed permissioner. On a + // peer shard that is a replicated read-mirror, so a permission revocation + // takes effect on the partition plane only once this shard applies the + // revoking commit -- an apply-lag window bounded by replication lag. + // Control-plane ops are exact (gated in-apply, in the same committed order + // on every replica); this local-read relaxation on the data plane is the + // accepted trade for keeping partition ops off the metadata consensus. + let scope = IggyNamespace::from_raw(namespace); + if let Some(status) = authorize_partition_op( + shard, + header.operation, + acting_user_id, + scope.stream_id(), + scope.topic_id(), + ) { + warn!( + transport_client_id, + status, + operation = ?header.operation, + "partition request denied by authorization; replying with status" + ); + send_deny_reply(shard, transport_client_id, &header, status).await; + return; + } + // Convergence wait: a CreateTopic commit returns to the client before the + // per-shard reconcilers seed routing rows and materialise the partition + // (next wake/periodic tick). An op arriving inside that window is not lost + // if it skips this wait -- `router::route_typed` falls back to the hash + // assignment, and the owning shard parks it -- so this is an admission + // courtesy that keeps the steady state off that park buffer, not a + // correctness gate. See `wait_for_partition_routable`, which spells out why + // there is no owner-readiness probe here any more. + if !wait_for_partition_routable(shard, IggyNamespace::from_raw(namespace)).await { + // The op never reached the partition plane, so it is safe to re-issue + // anywhere -- the same contract the plane itself answers for a + // non-primary routing artifact. A status-0 empty reply here would + // fabricate a success ack for a write that hit no partition at all. + warn!( + transport_client_id, + namespace, + operation = ?header.operation, + "partition request not routable within budget; replying transient" + ); + send_deny_reply( + shard, + transport_client_id, + &header, + IggyError::TransientNotAccepted.as_code(), + ) + .await; + return; + } + // A group consumer-offset op carries the group NAME on the wire; the + // partition plane keys the offset by the group's monotonic id (the same + // key the poll path auto-commits under and the read path resolves), so + // rewrite the consumer id before replication -- the apply layer has no + // metadata access to resolve it. + let request = match maybe_rewrite_consumer_offset_request(shard, request) { + Ok(rewritten) => rewritten, + Err(error) => { + warn!( + transport_client_id, + error = %error, + operation = ?header.operation, + "failed to rewrite consumer-offset request; replying empty" + ); + send_empty_partition_reply(shard, transport_client_id, &header).await; + return; + } + }; + let request = request.transmute_header(|header, new_header: &mut RoutedRequestHeader| { + *new_header = header; + new_header.group = namespace; + new_header.client = transport_client_id; + // Header validation requires `session > 0 && request > 0` for + // non-register ops. The partition plane itself is sessionless + // (at-least-once, no `ClientTable` dedup), so the bound VSR + // session merely satisfies validation. Current SDKs do number + // partition ops, but older and internal callers may still send + // zero, so a zero id is normalized to the compatibility value 1. + new_header.session = bound_session; + new_header.request = new_header.request.max(1); + }); + shard.dispatch(request.into_generic()); +} + +/// Serve `poll_messages`: resolve the partition namespace, run the read on +/// the owning shard ([`shard::IggyShard::partition_read`]), and re-encode +/// the stored batches into the legacy wire `PolledMessages` body. +/// +/// Failures reply with an empty body so the SDK fails fast on decode +/// instead of hanging until its read timeout. +#[allow(clippy::future_not_send)] +pub(in crate::dispatch) async fn handle_poll_messages( + shard: &Rc>, + transport_client_id: u128, + request: &Message, + user_id: Option, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let Ok(wire) = PollMessagesRequest::decode_from(request_body(request)) else { + // Undecodable poll: keep the fail-fast empty-poll shape. + send_non_replicated_bytes( + shard, + request, + transport_client_id, + empty_polled_messages_body(0), + "poll_messages", + ) + .await; + return; + }; + // Gate on (stream, topic) before touching the partition plane. A resolution + // miss falls through to the resolve path below (empty-poll / not-found); a + // denial replies status!=0 with an empty body, distinct from the empty-poll + // "0 messages" shape. + if let Some(status) = authorize_partition_read( + shard, + &wire.stream_id, + &wire.topic_id, + user_id, + |permissioner, uid, stream_id, topic_id| { + permissioner.poll_messages(uid, stream_id, topic_id) + }, + ) { + send_non_replicated_deny(shard, request, transport_client_id, status).await; + return; + } + let body = match resolve_poll_request(shard, &wire, request.header().client) { + Ok((namespace, partition_id, consumer, args)) => { + match shard + .partition_read(namespace, PartitionRead::Poll { consumer, args }) + .await + { + Some(PartitionReadReply::Poll { + fragments, + current_offset, + }) => match build_polled_messages_reply( + request.header(), + current_metadata_commit(shard), + partition_id, + current_offset, + fragments, + shard.plane.partitions().config().encryptor.as_deref(), + ) { + Ok(reply) => { + send_reply_frame(shard, transport_client_id, reply, "poll_messages").await; + return; + } + Err(error) => { + warn!( + transport_client_id, + error = %error, + "failed to re-encode polled batches; replying empty poll" + ); + empty_polled_messages_body(partition_id) + } + }, + other => { + warn!( + transport_client_id, + namespace = namespace.inner(), + reply_was_none = other.is_none(), + "partition read failed; replying empty poll" + ); + empty_polled_messages_body(partition_id) + } + } + } + Err(error) => { + // A stream, topic, or partition id that does not resolve is a + // client addressing error and must surface as a typed rejection, + // not an empty poll a consumer would read as end-of-partition. + if matches!( + error, + IggyError::PartitionNotFound(..) + | IggyError::StreamIdNotFound(_) + | IggyError::TopicIdNotFound(..) + ) { + warn!( + transport_client_id, + error = %error, + "poll_messages rejected: target not found" + ); + send_non_replicated_deny(shard, request, transport_client_id, error.as_code()) + .await; + return; + } + // A zero-byte body would panic the SDK's `PolledMessages` + // decoder; reply the 16-byte empty-poll shape instead. A generation + // fence (the client's cached assignment is stale after a rebalance) + // carries the re-sync sentinel so the SDK re-syncs and retries + // rather than treating the empty poll as end-of-partition. + warn!( + transport_client_id, + error = %error, + "poll_messages request rejected; replying empty poll" + ); + let partition_id = if matches!(error, IggyError::ConsumerGroupPartitionNotOwned(..)) { + iggy_common::RESYNC_REQUIRED_PARTITION_SENTINEL + } else { + 0 + }; + empty_polled_messages_body(partition_id) + } + }; + send_non_replicated_bytes(shard, request, transport_client_id, body, "poll_messages").await; +} + +/// Serve `get_consumer_offset`. An empty body decodes as `None` on the SDK +/// side (no offset stored / partition unknown). +// TODO(hubcio): plain local partition_read with no primary gate, so a +// follower answers from its own (possibly lagging) offset state. Needs the +// same is-caught-up-primary gate the auto-commit path has, or an explicit +// read-from-follower contract. +#[allow(clippy::future_not_send)] +pub(in crate::dispatch) async fn handle_get_consumer_offset( + shard: &Rc>, + transport_client_id: u128, + request: &Message, + user_id: Option, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let Ok(wire) = GetConsumerOffsetRequest::decode_from(request_body(request)) else { + // Undecodable: an empty body decodes as None (no offset) on the SDK. + send_non_replicated_bytes( + shard, + request, + transport_client_id, + Bytes::new(), + "get_consumer_offset", + ) + .await; + return; + }; + if let Some(status) = authorize_partition_read( + shard, + &wire.stream_id, + &wire.topic_id, + user_id, + |permissioner, uid, stream_id, topic_id| { + permissioner.get_consumer_offset(uid, stream_id, topic_id) + }, + ) { + send_non_replicated_deny(shard, request, transport_client_id, status).await; + return; + } + let body = match resolve_consumer_offset_request(shard, &wire) { + Ok((namespace, partition_id, consumer)) => { + match shard + .partition_read(namespace, PartitionRead::ConsumerOffset { consumer }) + .await + { + Some(PartitionReadReply::ConsumerOffset { + stored: Some(stored_offset), + current_offset, + }) => build_consumer_offset_body(partition_id, current_offset, stored_offset), + _ => Bytes::new(), + } + } + // A partition id that does not exist in a resolvable topic is a client + // addressing error, the same one the poll path denies typed. An empty + // body decodes as `None` -- indistinguishable from "this consumer has + // no stored offset yet" -- so the caller cannot tell a typo from a + // fresh consumer. + Err(error @ IggyError::PartitionNotFound(..)) => { + warn!( + transport_client_id, + error = %error, + "get_consumer_offset rejected: partition not found" + ); + send_non_replicated_deny(shard, request, transport_client_id, error.as_code()).await; + return; + } + Err(error) => { + warn!( + transport_client_id, + error = %error, + "get_consumer_offset request rejected; replying empty" + ); + Bytes::new() + } + }; + send_non_replicated_bytes( + shard, + request, + transport_client_id, + body, + "get_consumer_offset", + ) + .await; +} + +/// Ack a consumer-offset op whose body could not be rewritten for the +/// partition plane with an empty Reply. The SDK connection processes replies +/// in lockstep, so a silent drop wedges every subsequent request on that +/// connection. +#[allow(clippy::future_not_send)] +async fn send_empty_partition_reply( + shard: &Rc>, + transport_client_id: u128, + request_header: &RoutedRequestHeader, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let commit = current_metadata_commit(shard); + let reply = build_empty_reply(request_header, transport_client_id, 0, commit); + if let Err(error) = shard + .bus + .send_to_client(transport_client_id, reply.into_generic().into_frozen()) + .await + { + warn!( + transport_client_id, + error = %error, + operation = ?request_header.operation, + "failed to surface empty partition reply" + ); + } +} + +/// Wait (bounded) until this shard holds a routing row for `namespace`. Fast +/// path: row already present -> no wait. +/// +/// Covers the post-`CreateTopic` convergence window where the metadata commit +/// has returned to the client but the per-shard reconcilers have not yet seeded +/// routing rows. This is an admission courtesy, not a correctness gate: the row +/// is a cache of the deterministic hash assignment and may exist before the +/// owner has materialised anything, so its presence proves only where the +/// partition belongs. What makes an early arrival safe is the owning shard +/// itself - `park_if_unmaterialised` holds the frame until its partition lands, +/// and `serves_committed_incarnation` refuses to serve a mismatched +/// incarnation. Waiting here simply keeps the steady state off that park +/// buffer, whose overflow is the one path that still sheds a request without +/// replying (`frame_drops_total{variant=partition,reason=park_overflow}`). +/// +/// Deliberately no owner-readiness probe. One used to run here, on the theory +/// that the table could not be trusted; it could not close the window either, +/// because the fast path above skipped it in exactly the case it was meant to +/// cover - a row seeded from the hash by a shard that owns nothing. Readiness +/// belongs to the owner, which is where it is now enforced. +#[allow(clippy::future_not_send)] +async fn wait_for_partition_routable( + shard: &Rc>, + namespace: IggyNamespace, +) -> bool +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + const ATTEMPT_DELAY: std::time::Duration = std::time::Duration::from_millis(50); + // 3s budget at 50ms per attempt. Counting attempts, not reading a + // wall-clock deadline, keeps the wait virtual under the simulator: the + // bus sleep advances virtual time, whereas `Instant::now` would not. + const MAX_ATTEMPTS: u32 = 60; + + let mut attempts = 0u32; + while shard.shards_table().shard_for(namespace).is_none() { + if attempts >= MAX_ATTEMPTS { + return false; + } + attempts += 1; + shard.bus.sleep(ATTEMPT_DELAY).await; + } + true +} + +/// The 16-byte `PolledMessages` body with zero messages +/// (`[partition_id:4][current_offset:8][count:4]`). The SDK decoder +/// requires at least this header, so failure paths must never reply a +/// zero-byte body. +fn empty_polled_messages_body(partition_id: u32) -> Bytes { + let mut body = Vec::with_capacity(16); + body.extend_from_slice(&partition_id.to_le_bytes()); + body.extend_from_slice(&0u64.to_le_bytes()); + body.extend_from_slice(&0u32.to_le_bytes()); + Bytes::from(body) +} + +type DecodedPollRequest = (IggyNamespace, u32, PollingConsumer, PollingArgs); + +/// Resolve a decoded poll request into its owning-shard read: namespace, +/// partition, polling consumer, and args. Shared by the TCP dispatch (client +/// id = the connection's bound VSR client) and the HTTP route (client id 0, +/// which fences group polls closed). +#[allow(clippy::cast_possible_truncation)] +pub fn resolve_poll_request( + shard: &Rc>, + wire: &PollMessagesRequest, + client_id: u128, +) -> Result +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let strategy = polling_strategy_from_wire(&wire.strategy)?; + let args = PollingArgs::new(strategy, wire.count, wire.auto_commit); + + // Consumer-group poll: the client selects which of its assigned partitions + // to read and sends it explicitly. The coordinator FENCES ownership (a stale + // client whose partition was reassigned is rejected with + // `ConsumerGroupPartitionNotOwned`, prompting a re-sync) and resolves the + // group's monotonic id -- the offset key the store rewrite and read path + // both use, so `next()` reads back the offset it just committed. + if wire.consumer.kind == KIND_CONSUMER_GROUP { + let partition_id = wire.partition_id.ok_or(IggyError::InvalidIdentifier)?; + let group_id = shard + .plane + .metadata() + .mux_stm + .streams() + .consumer_group_fence( + &wire.stream_id, + &wire.topic_id, + &wire.consumer.id, + client_id, + partition_id, + // Poll fence: reject a pending-revoked partition so the source + // re-syncs and skips it (it still commits it via the offset fence). + true, + ) + .ok_or(IggyError::ConsumerGroupPartitionNotOwned( + client_id as u32, + partition_id, + ))?; + let namespace = resolve_partition_namespace( + shard, + &wire.stream_id, + &wire.topic_id, + Some(partition_id), + )?; + #[allow(clippy::cast_possible_truncation)] + let consumer = PollingConsumer::ConsumerGroup(group_id as usize, partition_id as usize); + return Ok((namespace, partition_id, consumer, args)); + } + + // Plain-consumer poll: an omitted partition selects partition 0, matching + // the legacy resolver (`resolve_consumer_with_partition_id` uses + // `unwrap_or(0)` for `ConsumerKind::Consumer`). + let partition_id = wire.partition_id.unwrap_or(0); + let namespace = + resolve_partition_namespace(shard, &wire.stream_id, &wire.topic_id, Some(partition_id))?; + let consumer = polling_consumer_from_wire(&wire.consumer, partition_id)?; + Ok((namespace, partition_id, consumer, args)) +} + +/// Resolve a decoded consumer-offset read into its owning-shard read: +/// namespace, partition, and polling consumer. Shared by the TCP dispatch and +/// the HTTP route; needs no client id because offset reads are not fenced +/// (any client may read a group's offset, member or not). +pub fn resolve_consumer_offset_request( + shard: &Rc>, + wire: &GetConsumerOffsetRequest, +) -> Result<(IggyNamespace, u32, PollingConsumer), IggyError> +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + // Omitted partition reads partition 0, matching the legacy resolver for + // both consumer kinds (`unwrap_or(0)`). + let partition_id = wire.partition_id.unwrap_or(0); + let namespace = + resolve_partition_namespace(shard, &wire.stream_id, &wire.topic_id, Some(partition_id))?; + // A group offset is keyed by the group's monotonic id (any client may read + // it, member or not), the same key the write path is rewritten to. An + // unresolved group (e.g. deleted) has no offset, so the read reports None. + let consumer = if wire.consumer.kind == KIND_CONSUMER_GROUP { + let group_id = shard + .plane + .metadata() + .mux_stm + .streams() + .resolve_consumer_group_id(&wire.stream_id, &wire.topic_id, &wire.consumer.id) + .ok_or(IggyError::InvalidIdentifier)?; + #[allow(clippy::cast_possible_truncation)] + PollingConsumer::ConsumerGroup(group_id as usize, partition_id as usize) + } else { + polling_consumer_from_wire(&wire.consumer, partition_id)? + }; + Ok((namespace, partition_id, consumer)) +} + +fn polling_consumer_from_wire( + consumer: &WireConsumer, + partition_id: u32, +) -> Result { + // Mirrors the legacy server's `PollingConsumer::resolve_consumer_id`: + // numeric ids pass through, named consumers hash to a stable u32 so + // reads derive the same offset-table key the write path stores under. + let consumer_id = match &consumer.id { + iggy_binary_protocol::WireIdentifier::Numeric(id) => *id, + iggy_binary_protocol::WireIdentifier::String(name) => { + iggy_common::calculate_32(name.as_str().as_bytes()) + } + } as usize; + match consumer.kind { + 1 => Ok(PollingConsumer::Consumer( + consumer_id, + partition_id as usize, + )), + KIND_CONSUMER_GROUP => Ok(PollingConsumer::ConsumerGroup( + consumer_id, + partition_id as usize, + )), + _ => Err(IggyError::InvalidCommand), + } +} + +fn polling_strategy_from_wire( + strategy: &WirePollingStrategy, +) -> Result { + let mut mapped = match strategy.kind { + 1 => PollingStrategy::offset(0), + 2 => PollingStrategy::timestamp(iggy_common::IggyTimestamp::from(strategy.value)), + 3 => PollingStrategy::first(), + 4 => PollingStrategy::last(), + 5 => PollingStrategy::next(), + _ => return Err(IggyError::InvalidCommand), + }; + mapped.set_value(strategy.value); + Ok(mapped) +} + +/// Handle a client `DeleteSegments`: resolve the requested count to an offset +/// on the owning shard, replicate a `TruncatePartition` through metadata so +/// every replica trims to the same watermark, then ack the client. The local +/// deletion happens later, when each replica's reconciler observes the commit. +/// +/// The consensus reply is forwarded verbatim: nothing-to-delete commits a +/// no-op `TruncatePartition(0)` and acks, while a not-primary rejection +/// reaches the client as `TransientNotCommitted` so the SDK replays instead +/// of mistaking a dropped delete for success. Only a malformed / unresolvable +/// request is acked empty without a commit. +#[allow(clippy::future_not_send)] +pub(in crate::dispatch) async fn handle_delete_segments_request( + shard: &Rc>, + transport_client_id: u128, + bound: Option<(u128, u64)>, + request: &Message, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let header = *request.header(); + let body = request_body(request); + + // An unbound transport cannot be attributed a VSR request sequence; the + // outer handler already short-circuits these, so this is defensive. + let Some((vsr_client_id, session)) = bound else { + return; + }; + + // The client numbers DeleteSegments in the same monotonic request sequence + // as every other metadata op. So resolve the requested count to a concrete + // offset on the owning shard, then replicate a `TruncatePartition(offset)` + // AS the client's own request through the standard owner path: the commit + // records (client, session, request) in the `ClientTable` on every replica, + // advancing the watermark. Skipping the commit (or attributing it to an + // internal id) leaves this request id unrecorded, so the SDK's own retry + // of it would re-execute instead of deduping. A no-op delete still + // commits `up_to_offset = 0` (monotonic apply) for the same reason. + let truncate = match resolve_delete_segments_truncate( + shard, + &header, + vsr_client_id, + session, + body, + ) + .await + { + Ok(truncate) => Some(truncate), + // The owning partition has not converged on the committed log yet, so + // the delete cannot be resolved to a watermark. Reply with the + // result-framed transient rejection (under the TruncatePartition + // operation, which the SDK decodes) so the client replays the same + // request once the partition catches up. Nothing was submitted, hence + // the re-issuable-anywhere flavor. + Err(IggyError::TransientNotAccepted) => { + let template = build_truncate_partition_client_message( + &header, + vsr_client_id, + session, + 0, + 0, + 0, + 0, + ); + let reply = build_result_rejection_reply( + template.header(), + current_metadata_commit(shard), + IggyError::TransientNotAccepted.as_code(), + ); + if let Err(error) = shard + .bus + .send_to_client(transport_client_id, reply.into_generic().into_frozen()) + .await + { + warn!( + transport_client_id, + error = %error, + "delete_segments: failed to send transient rejection" + ); + } + return; + } + Err(_) => None, + }; + + let reply = if let Some(truncate) = truncate { + // Forward the consensus reply verbatim, exactly like the generic + // metadata path: a committed success acks the delete, and a + // result-framed `TransientNotCommitted` rejection makes the SDK + // replay the request. Acking unconditionally here would swallow a + // not-primary rejection and drop the delete on the floor while the + // client believes it succeeded. + let Some(reply) = submit_client_request_on_owner(shard, truncate).await else { + // Transient submit failure (not primary / view change). Stay + // silent; the SDK read-timeout replays the same request id, + // which re-resolves and commits. Acking here would advance the + // client past an unrecorded request and gap the next metadata + // op. + warn!( + transport_client_id, + "delete_segments: transient submit; client will replay" + ); + return; + }; + reply + } else { + // Undecodable body (never produced by the SDK): ack empty so the + // lockstep stream stays framed; the typed decoder surfaces the + // failure client-side. Unresolvable-but-well-formed targets commit a + // typed rejection instead (see the resolve), so only a wire-corrupt + // request can gap the sequence here. + let commit = current_metadata_commit(shard); + build_empty_reply(&header, transport_client_id, session, commit).into_generic() + }; + if let Err(error) = shard + .bus + .send_to_client(transport_client_id, reply.into_frozen()) + .await + { + warn!( + transport_client_id, + error = %error, + "delete_segments: failed to send reply" + ); + } +} + +/// Resolve a client `DeleteSegments` to the `TruncatePartition` that commits the +/// trim. Shared by the TCP dispatch and the HTTP listener so both resolve the +/// requested segment count to a concrete watermark identically. +/// +/// `template` supplies the wire `cluster` / `view` / `release` and the client's +/// `request` number; `client_id` / `session` are the bound VSR identity the +/// truncate commits under. A resolvable namespace with nothing sealed to delete +/// still yields a `TruncatePartition(up_to_offset = 0)` so the metadata request +/// sequence stays contiguous. `Err` on a malformed body or an unresolved +/// namespace: the TCP caller drops it to a silent replay, the HTTP caller renders +/// the error. +#[allow(clippy::future_not_send)] +#[allow(clippy::cast_possible_truncation)] +pub async fn resolve_delete_segments_truncate( + shard: &Rc>, + template: &RoutedRequestHeader, + client_id: u128, + session: u64, + body: &[u8], +) -> Result, IggyError> +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let parsed = DeleteSegmentsRequest::decode_from(body).map_err(|_| IggyError::InvalidCommand)?; + let namespace_raw = match resolve_partition_request_namespace( + shard, + Operation::DeleteSegments, + body, + client_id, + ) { + Ok(namespace_raw) => namespace_raw, + // Unresolvable stream/topic: still commit the truncate, against the + // client's raw identifiers -- the apply rejects it as a committed + // result, so the failure is recorded against the client's request id + // and its retry dedups, while the client gets the typed error an + // empty ack would swallow. + Err(error) => { + debug!( + client_id, + %error, + "delete_segments: unresolved target; committing typed rejection" + ); + return Ok(build_truncate_partition_client_message_with_identifiers( + template, + client_id, + session, + parsed.stream_id, + parsed.topic_id, + parsed.partition_id, + 0, + )); + } + }; + let namespace = IggyNamespace::from_raw(namespace_raw); + let up_to_offset = match shard + .partition_read( + namespace, + PartitionRead::ResolveSegmentDeleteOffset { + count: parsed.segments_count, + }, + ) + .await + { + Some(PartitionReadReply::SegmentDeleteOffset { + up_to_offset: Some(offset), + .. + }) => offset, + // Nothing sealed to delete on a replica that has not converged on the + // replicated log (a backup behind the commit frontier may be missing + // whole sealed segments). Answering now would commit a no-op truncate + // and silently drop the delete, so surface a transient and let the + // client replay once the partition catches up. A converged primary + // whose resident tail is merely unflushed settles as a no-op below. + Some(PartitionReadReply::SegmentDeleteOffset { + up_to_offset: None, + lagging: true, + }) => { + debug!( + client_id, + namespace_raw, "delete_segments: partition not converged; transient" + ); + return Err(IggyError::TransientNotAccepted); + } + other => { + debug!( + client_id, + namespace_raw, + reply = ?other, + "delete_segments: nothing to delete; committing no-op truncate" + ); + 0 + } + }; + Ok(build_truncate_partition_client_message( + template, + client_id, + session, + namespace.stream_id() as u32, + namespace.topic_id() as u32, + namespace.partition_id() as u32, + up_to_offset, + )) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::dispatch::test_support::{ + SpyBus, TestMux, TestShard, prepare_message, request_message, + }; + use iggy_binary_protocol::ReplyHeader; + use iggy_binary_protocol::primitives::partition_assignment::CreatedPartitionAssignment; + use iggy_binary_protocol::requests::messages::SendMessagesHeader; + use iggy_binary_protocol::requests::streams::CreateStreamRequest; + use iggy_binary_protocol::requests::topics::{ + CreateTopicRequest, CreateTopicWithAssignmentsRequest, + }; + use iggy_binary_protocol::{WireName, WireOptions, WirePartitioning}; + use iggy_common::defaults::DEFAULT_ROOT_USER_ID; + use metadata::IggyMetadata; + use metadata::stm::StateMachine as _; + use partitions::{IggyPartitions, PartitionPathLayout, PartitionsConfig}; + use server_common::MessageBag; + use server_common::sharding::ShardId; + use shard::metrics::ShardMetrics; + use shard::shards_table::PapayaShardsTable; + use shard::{ + LifecycleFrame, PartitionConsensusConfig, ReconcileOp, ReplicaTopology, ShardFrame, + ShardIdentity, shard_channel, + }; + + /// A partition write whose routable wait exhausts (namespace committed, + /// but no reconciler ever seeds this shard's routing row -- the state a + /// teardown/rematerialise churn leaves behind) must answer a nonzero + /// retriable status. A status-0 empty reply is a fabricated success: the + /// SDK grades the send as acknowledged while zero bytes reached any + /// partition. + #[compio::test] + async fn unroutable_partition_send_must_reply_transient_error_not_success() { + const VSR_CLIENT: u128 = 1; + const SESSION: u64 = 1; + const TRANSPORT: u128 = 91; + const STATUS_OFFSET: usize = std::mem::offset_of!(ReplyHeader, status); + + let bus = SpyBus::default(); + let metadata = IggyMetadata::new(None, None, None, None, TestMux::default(), None); + let partitions = IggyPartitions::new( + ShardId::new(0), + PartitionsConfig { + messages_required_to_save: 1, + size_of_messages_required_to_save: iggy_common::IggyByteSize::from(1024_u64), + enforce_fsync: false, + validate_checksum: true, + segment_size: iggy_common::IggyByteSize::from(1_048_576_u64), + preallocate_segments: false, + encryptor: None, + path_layout: PartitionPathLayout::default(), + }, + ); + let shard = Rc::new(TestShard::without_inbox( + ShardIdentity::new(0, "unroutable-send-test".to_string()), + bus.clone(), + metadata, + partitions, + PapayaShardsTable::new(), + PartitionConsensusConfig::new(1, ReplicaTopology::new(0, 1), bus.clone()), + )); + let md = shard.plane.metadata(); + + // Committed stream 0 / topic 0 / partition 0, applied straight into + // the STM: the namespace resolves and root authorizes, but no + // reconciler runs, so the shards table never gains a routing row and + // the routable wait exhausts its budget. + md.mux_stm.users().ensure_root_user("iggy", "hash"); + let create_stream = CreateStreamRequest { + name: WireName::new("stream").unwrap(), + options: WireOptions::empty(), + }; + md.mux_stm + .update(prepare_message( + Operation::CreateStream, + VSR_CLIENT, + 1, + &create_stream.to_bytes(), + )) + .unwrap(); + let create_topic = CreateTopicWithAssignmentsRequest { + request: CreateTopicRequest { + stream_id: WireIdentifier::numeric(0), + partitions_count: 1, + name: WireName::new("topic").unwrap(), + options: WireOptions::empty(), + }, + derived_options: WireOptions::empty(), + partitions: vec![CreatedPartitionAssignment { + partition_id: 0, + consensus_group_id: 1, + }], + created_view: 0, + }; + md.mux_stm + .update(prepare_message( + Operation::CreateTopicWithAssignments, + VSR_CLIENT, + 2, + &create_topic.to_bytes(), + )) + .unwrap(); + assert!( + md.mux_stm + .streams() + .namespace_from_partition( + &WireIdentifier::numeric(0), + &WireIdentifier::numeric(0), + 0 + ) + .is_some(), + "seeded namespace must resolve, or the unresolved-namespace path \ + would reply instead of the exhausted routable wait" + ); + + let send_header = SendMessagesHeader { + stream_id: WireIdentifier::numeric(0), + topic_id: WireIdentifier::numeric(0), + partitioning: WirePartitioning::PartitionId(0), + messages_count: 1, + }; + let send_metadata = send_header.to_bytes(); + let mut send_body = Vec::with_capacity(4 + send_metadata.len()); + send_body.extend_from_slice(&u32::try_from(send_metadata.len()).unwrap().to_le_bytes()); + send_body.extend_from_slice(&send_metadata); + let request = request_message(Operation::SendMessages, VSR_CLIENT, SESSION, 1, &send_body); + + dispatch_partition_request( + &shard, + request, + VSR_CLIENT, + SESSION, + TRANSPORT, + Some(DEFAULT_ROOT_USER_ID), + ) + .await; + + let replies = bus.client_replies.borrow(); + assert_eq!(replies.len(), 1, "one reply frame for the failed send"); + let (client, frame) = &replies[0]; + assert_eq!(*client, TRANSPORT, "reply must target the transport id"); + let status = + u32::from_le_bytes(frame[STATUS_OFFSET..STATUS_OFFSET + 4].try_into().unwrap()); + assert_eq!( + status, + IggyError::TransientNotAccepted.as_code(), + "an unroutable partition write must surface the retriable \ + transient status; status 0 with an empty body grades as a \ + successfully acknowledged send" + ); + } + + /// A send that reaches the owning shard while its namespace is + /// tombstoned (the teardown fence a delete/recreate churn sets before + /// the disk delete) must answer the retriable transient status. The + /// partition plane's own tombstone guard drops the frame without any + /// reply; the transports decode replies in lockstep, so that silence + /// wedges the connection until the SDK's response read-timeout. + #[compio::test] + async fn tombstoned_partition_send_must_reply_transient_error_not_silence() { + const TRANSPORT: u128 = 91; + const SESSION: u64 = 1; + const STATUS_OFFSET: usize = std::mem::offset_of!(ReplyHeader, status); + + let bus = SpyBus::default(); + let metadata = IggyMetadata::new(None, None, None, None, TestMux::default(), None); + let partitions = IggyPartitions::new( + ShardId::new(0), + PartitionsConfig { + messages_required_to_save: 1, + size_of_messages_required_to_save: iggy_common::IggyByteSize::from(1024_u64), + enforce_fsync: false, + validate_checksum: true, + segment_size: iggy_common::IggyByteSize::from(1_048_576_u64), + preallocate_segments: false, + encryptor: None, + path_layout: PartitionPathLayout::default(), + }, + ); + let shard = Rc::new(TestShard::without_inbox( + ShardIdentity::new(0, "tombstoned-send-test".to_string()), + bus.clone(), + metadata, + partitions, + PapayaShardsTable::new(), + PartitionConsensusConfig::new(1, ReplicaTopology::new(0, 1), bus.clone()), + )); + + let namespace = IggyNamespace::new(0, 0, 0); + shard.plane.partitions().tombstone(namespace); + + let request = request_message(Operation::SendMessages, TRANSPORT, SESSION, 1, &[]) + .transmute_header(|header, new_header: &mut RoutedRequestHeader| { + *new_header = header; + new_header.group = namespace.inner(); + }); + shard.on_message(MessageBag::Request(request)).await; + + let replies = bus.client_replies.borrow(); + assert_eq!( + replies.len(), + 1, + "a send into a tombstoned namespace must produce a reply frame; \ + silence wedges the connection's lockstep decode" + ); + let (client, frame) = &replies[0]; + assert_eq!(*client, TRANSPORT, "reply must target the request's client"); + let status = + u32::from_le_bytes(frame[STATUS_OFFSET..STATUS_OFFSET + 4].try_into().unwrap()); + assert_eq!( + status, + IggyError::TransientNotAccepted.as_code(), + "a tombstoned-namespace send must surface the retriable transient \ + status so the SDK replays it after the partition rematerialises" + ); + } + + /// A send parked for a namespace that is torn down before materialising + /// (create -> delete before the reconciler's `InsertOwned`) is discarded + /// on `ConfirmRemove`. The discard must stage the same retriable + /// transient deny toward the client -- through the shard's own pump as a + /// `ForwardClientSend` -- instead of dropping the request without any + /// reply. + #[compio::test] + async fn discarded_parked_partition_send_must_reply_transient_error_not_silence() { + const TRANSPORT: u128 = 91; + const SESSION: u64 = 1; + const STATUS_OFFSET: usize = std::mem::offset_of!(ReplyHeader, status); + + let bus = SpyBus::default(); + let metadata = IggyMetadata::new(None, None, None, None, TestMux::default(), None); + let partitions = IggyPartitions::new( + ShardId::new(0), + PartitionsConfig { + messages_required_to_save: 1, + size_of_messages_required_to_save: iggy_common::IggyByteSize::from(1024_u64), + enforce_fsync: false, + validate_checksum: true, + segment_size: iggy_common::IggyByteSize::from(1_048_576_u64), + preallocate_segments: false, + encryptor: None, + path_layout: PartitionPathLayout::default(), + }, + ); + // Real sender ring so the staged deny is observable: the test holds + // the receiving ends of this shard's own lanes. The deny is a client + // Reply forward, so it lands on the REPLY lane. + let (sender, _pump_rx, reply_rx) = shard_channel(0, 16, 16); + let (_inbox_tx, inbox_rx, reply_inbox_rx) = shard_channel(0, 1, 1); + let shard = TestShard::new( + ShardIdentity::new(0, "discarded-parked-send-test".to_string()), + bus.clone(), + Rc::new(|_, _| {}), + Rc::new(|_, _| {}), + Rc::new(|_| {}), + Rc::new(|_| {}), + Rc::new(|_, _, _| {}), + metadata, + partitions, + vec![sender], + inbox_rx, + reply_inbox_rx, + PapayaShardsTable::new(), + PartitionConsensusConfig::new(1, ReplicaTopology::new(0, 1), bus.clone()), + None, + ShardMetrics::for_shard(), + ) + .expect("single-sender ring is canonically ordered"); + + let namespace = IggyNamespace::new(0, 0, 0); + let request = request_message(Operation::SendMessages, TRANSPORT, SESSION, 1, &[]) + .transmute_header(|header, new_header: &mut RoutedRequestHeader| { + *new_header = header; + new_header.group = namespace.inner(); + }); + // Namespace neither materialised nor tombstoned: the frame parks. + shard.on_message(MessageBag::Request(request)).await; + + shard.enqueue_reconcile_op(ReconcileOp::ConfirmRemove { namespace }); + shard.apply_reconcile_ops(); + + let mut denies = Vec::new(); + while let Ok(frame) = reply_rx.try_recv() { + if let ShardFrame::Lifecycle(LifecycleFrame::ForwardClientSend { client_id, msg }) = + frame + { + denies.push((client_id, msg.into_contiguous().as_slice().to_vec())); + } + } + assert_eq!( + denies.len(), + 1, + "discarding a parked client request must stage exactly one deny \ + reply; silence wedges the connection's lockstep decode" + ); + let (client, frame) = &denies[0]; + assert_eq!(*client, TRANSPORT, "deny must target the request's client"); + let status = + u32::from_le_bytes(frame[STATUS_OFFSET..STATUS_OFFSET + 4].try_into().unwrap()); + assert_eq!( + status, + IggyError::TransientNotAccepted.as_code(), + "a discarded parked send must surface the retriable transient \ + status so the SDK replays it instead of timing out" + ); + } +} diff --git a/core/server/src/dispatch/reads.rs b/core/server/src/dispatch/reads.rs new file mode 100644 index 0000000000..2c8b43c0d8 --- /dev/null +++ b/core/server/src/dispatch/reads.rs @@ -0,0 +1,491 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Non-replicated read router and its per-code arms. +//! +//! Every `Operation::NonReplicated` request lands in +//! [`handle_non_replicated_request`] after the funnel's pre-auth gate; the +//! poll and consumer-offset arms live in `partition` (they read through the +//! shard mesh), everything else is served here from local shard state. The +//! catch-all arm delegates to the shared `responses` builder, which is +//! byte-shared with the HTTP read path -- authorization happens HERE (and in +//! the HTTP layer), never in the builder. + +use crate::cluster_meta::ClusterRoster; +use crate::dispatch::authz::{authorize_default_read, authorize_uid, send_non_replicated_deny}; +use crate::dispatch::partition::{handle_get_consumer_offset, handle_poll_messages}; +use crate::dispatch::send_non_replicated_bytes; +use crate::responses::{ + build_empty_reply, build_get_me_response, build_get_personal_access_tokens_response, + build_non_replicated_response, connected_client_to_response, current_metadata_commit, +}; +use crate::session_manager::SessionManager; +use crate::shell::{ShellBus, ShellShard}; +use crate::snapshot; +use crate::wire::request_body; +use bytes::Bytes; +use configs::server::ServerSystemConfig; +use consensus::MetadataHandle; +use iggy_binary_protocol::PrepareHeader; +use iggy_binary_protocol::codes::{ + GET_CLIENT_CODE, GET_CLIENTS_CODE, GET_CONSUMER_OFFSET_CODE, GET_ME_CODE, + GET_PERSONAL_ACCESS_TOKENS_CODE, GET_SNAPSHOT_FILE_CODE, GET_STATS_CODE, PING_CODE, + POLL_MESSAGES_CODE, SYNC_CONSUMER_GROUP_CODE, +}; +use iggy_binary_protocol::requests::consumer_groups::SyncConsumerGroupRequest; +use iggy_binary_protocol::requests::system::get_client::GetClientRequest; +use iggy_binary_protocol::requests::system::get_snapshot::GetSnapshotRequest; +use iggy_binary_protocol::responses::clients::client_response::ConsumerGroupInfoResponse; +use iggy_binary_protocol::responses::clients::get_client::ClientDetailsResponse; +use iggy_binary_protocol::responses::clients::get_clients::GetClientsResponse; +use iggy_binary_protocol::responses::consumer_groups::SyncConsumerGroupResponse; +use iggy_binary_protocol::responses::system::get_snapshot::GetSnapshotResponse; +use iggy_binary_protocol::{HEADER_SIZE, RoutedRequestHeader, WireDecode, WireEncode}; +use iggy_common::{IggyError, SnapshotCompression, SystemSnapshotType}; +use journal::superblock::SuperblockStore; +use journal::{Journal, JournalHandle}; +use message_bus::framing::MAX_MESSAGE_SIZE; +use metadata::impls::metadata::StreamsFrontend; +use metadata::permissioner::Permissioner; +use server_common::Message; +use std::cell::RefCell; +use std::net::IpAddr; +use std::rc::Rc; +use std::sync::Arc; +use tracing::{debug, warn}; + +/// Per-user PATs, resolved from this shard's session (like `get_me`) and read +/// out of the Users STM. Built here rather than in `build_non_replicated_response` +/// which has no session context. +#[allow(clippy::future_not_send)] +async fn handle_get_personal_access_tokens( + shard: &Rc>, + sessions: &Rc>, + transport_client_id: u128, + request: &Message, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let response = build_get_personal_access_tokens_response(shard, sessions, transport_client_id); + send_non_replicated_bytes( + shard, + request, + transport_client_id, + response.to_bytes(), + "get_personal_access_tokens", + ) + .await; +} + +/// The requesting connection's own identity, sourced from this shard's +/// `SessionManager` (not `IggyMetadata`), so built here rather than in +/// `build_non_replicated_response`. +#[allow(clippy::future_not_send)] +async fn handle_get_me( + shard: &Rc>, + sessions: &Rc>, + transport_client_id: u128, + request: &Message, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let response = build_get_me_response(shard, sessions, transport_client_id); + send_non_replicated_bytes( + shard, + request, + transport_client_id, + response.to_bytes(), + "get_me", + ) + .await; +} + +#[allow(clippy::future_not_send, clippy::too_many_lines)] +pub(in crate::dispatch) async fn handle_non_replicated_request( + shard: &Rc>, + sessions: &Rc>, + system_config: &Arc, + transport_client_id: u128, + request: Message, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + const CODE_RANGE: std::ops::Range = 0..4; + let code = u32::from_le_bytes(request.header().reserved[CODE_RANGE].try_into().unwrap()); + // Acting user and peer address for the read gates below, resolved in one + // connection lookup. `user_id` is `None` only on the pre-auth path + // (PING), which serves ungated codes; the gated arms fail closed on it. + let (user_id, client_address) = sessions.borrow().read_context(transport_client_id); + match code { + PING_CODE => { + // A ping is the client's liveness proof; reset its staleness clock + // so the heartbeat verifier doesn't evict an active connection. + sessions.borrow_mut().record_heartbeat(transport_client_id); + let commit = current_metadata_commit(shard); + let reply = build_empty_reply( + request.header(), + request.header().client, + request.header().session, + commit, + ); + if let Err(error) = shard + .bus + .send_to_client(transport_client_id, reply.into_generic().into_frozen()) + .await + { + warn!( + transport_client_id, + error = %error, + "failed to send non-replicated ping reply" + ); + } + } + GET_ME_CODE => { + handle_get_me(shard, sessions, transport_client_id, &request).await; + } + GET_PERSONAL_ACCESS_TOKENS_CODE => { + handle_get_personal_access_tokens(shard, sessions, transport_client_id, &request).await; + } + GET_CLIENTS_CODE => { + if let Err(error) = authorize_uid(shard, user_id, Permissioner::get_clients) { + send_non_replicated_deny(shard, &request, transport_client_id, error.as_code()) + .await; + return; + } + // Shared-nothing: each shard knows only its own connections, so + // gather across all shards (scatter-gather over the mesh). + let infos = shard.list_all_clients().await; + let response = GetClientsResponse { + clients: infos + .iter() + .map(|info| connected_client_to_response(shard, info)) + .collect(), + }; + send_non_replicated_bytes( + shard, + &request, + transport_client_id, + response.to_bytes(), + "get_clients", + ) + .await; + } + GET_CLIENT_CODE => { + if let Err(error) = authorize_uid(shard, user_id, Permissioner::get_client) { + send_non_replicated_deny(shard, &request, transport_client_id, error.as_code()) + .await; + return; + } + // No reverse map from the wire u32 id to a u128 transport id / + // home shard (the u32 is just the seq tail), so gather all and + // filter -- same fan-out as `get_clients`. + let target = GetClientRequest::decode_from(request_body(&request)) + .ok() + .map(|req| req.client_id); + let infos = shard.list_all_clients().await; + #[allow(clippy::cast_possible_truncation)] + let found = target.and_then(|id| infos.iter().find(|info| info.client_id as u32 == id)); + // The SDK decodes an empty body as `None` (client not found). + let bytes = found.map_or_else(Bytes::new, |info| { + let consumer_groups = info.vsr_client_id.map_or_else(Vec::new, |vsr_client_id| { + shard + .plane + .metadata() + .mux_stm + .streams() + .consumer_group_memberships(vsr_client_id) + .into_iter() + .map( + |(stream_id, topic_id, group_id)| ConsumerGroupInfoResponse { + stream_id, + topic_id, + group_id, + }, + ) + .collect() + }); + ClientDetailsResponse { + client: connected_client_to_response(shard, info), + consumer_groups, + } + .to_bytes() + }); + send_non_replicated_bytes(shard, &request, transport_client_id, bytes, "get_client") + .await; + } + GET_SNAPSHOT_FILE_CODE => { + handle_get_snapshot(shard, system_config, transport_client_id, &request, user_id).await; + } + POLL_MESSAGES_CODE => { + handle_poll_messages(shard, transport_client_id, &request, user_id).await; + } + GET_CONSUMER_OFFSET_CODE => { + handle_get_consumer_offset(shard, transport_client_id, &request, user_id).await; + } + SYNC_CONSUMER_GROUP_CODE => { + // Self-scoped: serves the caller's own assignment keyed by the + // header client id, so it carries no permissioner rule. + handle_sync_consumer_group(shard, transport_client_id, &request).await; + } + _ => { + let roster = sessions.borrow().cluster_roster(); + let client_ip = client_address.map(|address| address.ip()); + if client_ip.is_none() { + debug!( + transport_client_id, + code, + "no peer address recorded; advertised-address resolution degrades to the catch-all" + ); + } + handle_default_non_replicated( + shard, + transport_client_id, + code, + &request, + user_id, + &roster, + client_ip, + ) + .await; + } + } +} + +#[allow(clippy::future_not_send, clippy::too_many_arguments)] +async fn handle_default_non_replicated( + shard: &Rc>, + transport_client_id: u128, + code: u32, + request: &Message, + user_id: Option, + roster: &ClusterRoster, + client_ip: Option, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + // Gate by command code before the shared builder runs. The builder stays + // authz-free (it is byte-shared with the HTTP read path, which gates + // separately); a denial replies status!=0 with an empty body. + if let Err(error) = authorize_default_read(shard, code, request_body(request), user_id) { + send_non_replicated_deny(shard, request, transport_client_id, error.as_code()).await; + return; + } + // Stats is the one default read with an async input: the cross-shard + // connected-client gather. Run it here so the shared builder stays sync. + let clients_count = if code == GET_STATS_CODE { + u32::try_from(shard.list_all_clients().await.len()).unwrap_or(u32::MAX) + } else { + 0 + }; + match build_non_replicated_response( + shard, + code, + request_body(request), + user_id, + roster, + client_ip, + clients_count, + ) { + Ok(response) => { + let commit = current_metadata_commit(shard); + let reply = response.into_reply( + request.header(), + request.header().client, + request.header().session, + commit, + ); + if let Err(error) = shard + .bus + .send_to_client(transport_client_id, reply.into_generic().into_frozen()) + .await + { + warn!( + transport_client_id, + code, + error = %error, + "failed to send non-replicated VSR reply" + ); + } + } + Err(error) => { + // Surface the builder's typed error (unsupported op, undecodable + // body, or a not-found parity read) on the same deny channel the + // authz gate uses; a silent drop would wedge the client until its + // read timeout. + warn!( + transport_client_id, + code, + error = %error, + "denying non-replicated VSR request" + ); + send_non_replicated_deny(shard, request, transport_client_id, error.as_code()).await; + } + } +} + +/// Serve `GET_SNAPSHOT_FILE`: gate on the snapshot rule (`read_servers || +/// manage_servers`, the legacy gate - the archive dumps host diagnostics, so +/// plain authentication must not suffice), then await the off-thread +/// collection (see `snapshot::collect`) and reply with the raw ZIP bytes. +#[allow(clippy::future_not_send)] +async fn handle_get_snapshot( + shard: &Rc>, + system_config: &Arc, + transport_client_id: u128, + request: &Message, + user_id: Option, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + if let Err(error) = authorize_uid(shard, user_id, Permissioner::get_snapshot) { + send_non_replicated_deny(shard, request, transport_client_id, error.as_code()).await; + return; + } + let result = match decode_get_snapshot(request_body(request)) { + Ok((compression, snapshot_types)) => { + snapshot::collect(Arc::clone(system_config), compression, snapshot_types).await + } + Err(error) => Err(error), + }; + match result { + Ok(archive) => { + // The reply frames as `[256-byte header][archive]`. The client's + // `message_bus::read_message` rejects any frame past `MAX_MESSAGE_SIZE` + // (64 MiB) by tearing the connection down untyped, and a frame past + // `u32::MAX` would panic `build_reply_with_body`. The archive is the + // only unbounded non-replicated body, so refuse an oversized one with a + // typed error the SDK decodes. The HTTP path streams via `Body` (not + // this framing), so it stays uncapped. + let frame_size = HEADER_SIZE + archive.len(); + if frame_size > MAX_MESSAGE_SIZE { + warn!( + transport_client_id, + frame_size, + max = MAX_MESSAGE_SIZE, + "snapshot archive exceeds the client frame limit; refusing to send" + ); + send_non_replicated_deny( + shard, + request, + transport_client_id, + IggyError::SnapshotFileCompletionFailed.as_code(), + ) + .await; + return; + } + send_non_replicated_bytes( + shard, + request, + transport_client_id, + GetSnapshotResponse { data: archive }.to_bytes(), + "get_snapshot", + ) + .await; + } + Err(error) => { + warn!(transport_client_id, error = %error, "denying snapshot request"); + send_non_replicated_deny(shard, request, transport_client_id, error.as_code()).await; + } + } +} + +fn decode_get_snapshot( + body: &[u8], +) -> Result<(SnapshotCompression, Vec), IggyError> { + let request = GetSnapshotRequest::decode_from(body).map_err(|_| IggyError::InvalidCommand)?; + let compression = SnapshotCompression::from_code(request.compression)?; + let snapshot_types = request + .snapshot_types + .iter() + .map(|&code| SystemSnapshotType::from_code(code)) + .collect::, _>>()?; + Ok((compression, snapshot_types)) +} + +/// Serve `SyncConsumerGroup`: return the requesting member's current partition +/// assignment + group generation so the client can select partitions locally. +/// The member is keyed by the connection's bound VSR client id +/// (`header().client`). An empty body decodes as "no assignment" on the SDK. +#[allow(clippy::future_not_send)] +async fn handle_sync_consumer_group( + shard: &Rc>, + transport_client_id: u128, + request: &Message, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let body = match SyncConsumerGroupRequest::decode_from(request_body(request)) { + Ok(wire) => shard + .plane + .metadata() + .mux_stm + .streams() + .consumer_group_member_assignment( + &wire.stream_id, + &wire.topic_id, + &wire.group_id, + request.header().client, + ) + .map_or_else(Bytes::new, |(generation, partitions)| { + SyncConsumerGroupResponse { + generation, + partitions, + } + .to_bytes() + }), + Err(error) => { + warn!( + transport_client_id, + error = %error, + "sync_consumer_group request rejected; replying empty" + ); + Bytes::new() + } + }; + send_non_replicated_bytes( + shard, + request, + transport_client_id, + body, + "sync_consumer_group", + ) + .await; +} diff --git a/core/server/src/dispatch/session_ops.rs b/core/server/src/dispatch/session_ops.rs new file mode 100644 index 0000000000..d43bcc17b6 --- /dev/null +++ b/core/server/src/dispatch/session_ops.rs @@ -0,0 +1,2096 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Session lifecycle: login/register, logout, evictions, and their +//! replication plumbing. +//! +//! Credentials (password + PAT) verify locally against replicated state on +//! whichever node the client dialed; only the consensus `Register` / `Logout` +//! proposal runs on the metadata owner, forwarded at most one hop to the +//! current primary (`submit_*_local_or_forward`, shard 0 only) with both ends +//! of that replica protocol on this page. Terminal failures surface as typed +//! `Eviction` frames, transient ones as result-framed replay hints. +//! +//! Deliberate asymmetry (the two session stores): the per-shard +//! `SessionManager` owns transport sessions (connection -> user binding, +//! heartbeats, SDK info) and is never replicated; the consensus `ClientTable` +//! owns replicated VSR sessions and their dedup watermarks. Login binds the +//! two together, so logout and eviction must release BOTH -- every teardown +//! path below pairs `remove_connection` with a replicated `Logout`. + +use crate::dispatch::login_error::LoginRegisterError; +use crate::responses::{ + build_deny_reply, build_empty_reply, build_login_register_reply, current_metadata_commit, +}; +use crate::session_manager::{ClientSdkInfo, SessionManager}; +use crate::shell::{ShellBus, ShellShard}; +use crate::wire::request_body; +use consensus::{ + Consensus, DISCONNECT_LOGOUT_REQUEST_ID, EvictionContext, MetadataHandle, + build_eviction_message, build_incompatible_protocol_eviction_message, + build_result_rejection_reply, +}; +use iggy_binary_protocol::PrepareHeader; +use iggy_binary_protocol::requests::users::{LoginRegisterRequest, LoginRegisterWithPatRequest}; +use iggy_binary_protocol::{ + ClientVersionInfo, Command, ConsensusHeader, EvictionReason, ForwardLogoutHeader, + ForwardLogoutOutcome, ForwardLogoutResultHeader, ForwardRegisterHeader, ForwardRegisterOutcome, + ForwardRegisterResultHeader, HEADER_SIZE, ProtocolVersion, RoutedRequestHeader, WireDecode, + is_protocol_compatible, +}; +use iggy_common::defaults::{ + MAX_PASSWORD_LENGTH, MAX_USERNAME_LENGTH, MIN_PASSWORD_LENGTH, MIN_USERNAME_LENGTH, +}; +use iggy_common::{IggyError, IggyTimestamp, PersonalAccessToken, UserStatus}; +use journal::superblock::SuperblockStore; +use journal::{Journal, JournalHandle}; +use metadata::MetadataSubmitError; +use metadata::impls::metadata::{BoundSession, StreamsFrontend}; +use secrecy::ExposeSecret; +use server_common::Message; +use server_common::crypto; +use std::cell::RefCell; +use std::rc::Rc; +use std::sync::LazyLock; +use std::time::Duration; +use tracing::warn; + +/// A well-formed Argon2 hash to verify against on the unknown-user login +/// branch, so a missing username costs the same single `verify_password` a real +/// user's wrong-password branch costs. Closes the username-existence timing +/// oracle without changing the returned error. On the unknown-username branch +/// the verify result is discarded, so even a request presenting the exact dummy +/// plaintext cannot authenticate; the literal only needs to be a fixed input +/// hashed by the same Argon2 hasher real users use, so the dummy verify runs an +/// identical Argon2 KDF. +static DUMMY_PASSWORD_HASH: LazyLock = + LazyLock::new(|| crypto::hash_password("http-login-timing-guard")); + +/// Pay the one-time Argon2 cost of [`DUMMY_PASSWORD_HASH`] at boot instead of +/// inside the first unknown-username login request. +pub fn warm_dummy_password_hash() { + LazyLock::force(&DUMMY_PASSWORD_HASH); +} + +pub fn verify_login_credentials( + shard: &Rc>, + username: &str, + password: &str, +) -> Result +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + // Same bounds the legacy server enforces before any lookup or hashing; + // also keeps arbitrary-length input out of the password hash. Collapsed + // to InvalidCredentials on purpose (legacy: InvalidUsername / + // InvalidPassword): don't leak which field failed. + if !(MIN_USERNAME_LENGTH..=MAX_USERNAME_LENGTH).contains(&username.len()) + || !(MIN_PASSWORD_LENGTH..=MAX_PASSWORD_LENGTH).contains(&password.len()) + { + return Err(LoginRegisterError::InvalidCredentials); + } + shard.plane.metadata().mux_stm.users().read(|users| { + let user = users + .index + .get(username) + .copied() + .and_then(|user_id| users.items.get(user_id as usize)); + let Some(user) = user else { + // Constant-cost path: verify against a dummy hash so a missing + // username is indistinguishable by response timing from a wrong + // password (both return InvalidCredentials). + let _ = crypto::verify_password(password, DUMMY_PASSWORD_HASH.as_str()); + return Err(LoginRegisterError::InvalidCredentials); + }; + // Verify before the status check and collapse inactive to + // InvalidCredentials: an inactive account must answer exactly like a + // wrong password (same error, same Argon2 cost), or login could probe + // which accounts exist but are disabled. + if !crypto::verify_password(password, user.password_hash.as_ref()) + || user.status != UserStatus::Active + { + return Err(LoginRegisterError::InvalidCredentials); + } + Ok(user.id) + }) +} + +pub fn verify_pat_credentials( + shard: &Rc>, + token: &str, +) -> Result +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + verify_pat_credentials_with_expiry(shard, token).map(|(user_id, _)| user_id) +} + +/// Like [`verify_pat_credentials`] but also surfaces the token's expiry (unix +/// seconds, `u64::MAX` when the PAT never expires). The HTTP extractor keys a +/// per-token VSR session table on this expiry for lazy eviction; the wire and +/// login paths only need the user id and go through [`verify_pat_credentials`]. +pub fn verify_pat_credentials_with_expiry( + shard: &Rc>, + token: &str, +) -> Result<(u32, u64), LoginRegisterError> +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let token_hash = PersonalAccessToken::hash_token(token); + // PAT expiry gates the login accept/reject, and that outcome folds into + // the reply, so read the environment-injected bus clock (seed-derived + // under the simulator), not the wall clock, or a replayed login diverges. + // The two sibling wall-clock reads in `dispatch` stay direct because both + // are off the reply path (diagnostic-only). The bus seam exists on every + // shard, so this holds even when login lands on an entry shard that does + // not own the metadata consensus. + let now = IggyTimestamp::from(shard.bus.realtime_micros()); + shard.plane.metadata().mux_stm.users().read(|users| { + let Some((user_id, token_name)) = + users.personal_access_token_index.get(token_hash.as_str()) + else { + return Err(LoginRegisterError::InvalidToken); + }; + let Some(pat) = users + .personal_access_tokens + .get(user_id) + .and_then(|tokens| tokens.get(token_name)) + else { + return Err(LoginRegisterError::InvalidToken); + }; + if pat.is_expired(now) { + return Err(LoginRegisterError::InvalidToken); + } + let Some(user) = users.items.get(*user_id as usize) else { + return Err(LoginRegisterError::InvalidToken); + }; + if user.status != UserStatus::Active { + return Err(LoginRegisterError::UserInactive); + } + // `expiry_at == None` is a never-expiring PAT; map it to `u64::MAX` so + // the HTTP session table never expiry-evicts its entry. + let expiry = pat + .expiry_at + .map_or(u64::MAX, |expiry_at| expiry_at.to_secs()); + Ok((user.id, expiry)) + }) +} + +#[allow(clippy::future_not_send)] +async fn complete_login_register( + shard: &Rc>, + sessions: &Rc>, + transport_client_id: u128, + vsr_client_id: u128, + request_header: &RoutedRequestHeader, + user_id: u32, + client_version: &ClientVersionInfo, +) -> Result<(), LoginRegisterError> +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let sdk_info = ClientSdkInfo { + sdk_name: client_version.sdk_name.as_str().to_owned(), + sdk_version: client_version.sdk_version.as_str().to_owned(), + protocol_version: client_version.protocol_version, + }; + let existing_session = { + let sessions = sessions.borrow(); + sessions + .get_session(transport_client_id) + .map(|(_, session)| session) + }; + if let Some(session) = existing_session { + // Re-login on a bound connection: refresh the recorded SDK info + // (a reconnecting client may have been upgraded) and replay. + sessions + .borrow_mut() + .record_sdk_info(transport_client_id, sdk_info); + // A lagging backup's commit_max can sit below the epoch this session + // already bound; never advertise a commit behind the session itself. + let commit = current_metadata_commit(shard).max(session); + let reply = + build_login_register_reply(request_header, vsr_client_id, session, commit, user_id); + let _ = shard + .bus + .send_to_client(transport_client_id, reply.into_generic().into_frozen()) + .await; + return Ok(()); + } + + // Submit Register and await the commit. The SessionManager is left + // untouched until the op commits cluster-wide (post-quorum): there is no + // optimistic Authenticated transition, so a transient submit failure + // needs no rollback -- the connection stays Connected and the SDK + // read-timeout replays. + let session = match submit_register_on_owner(shard, vsr_client_id, user_id).await { + // The wire reply carries only the fence epoch; the SDK numbers its + // own requests, so the bind watermark is not surfaced (see the + // BoundSession doc for who does consume it). + Ok(bound) => bound.epoch, + Err(error) => { + return Err(LoginRegisterError::Transient(error)); + } + }; + + // Post-commit: Connected -> Authenticated -> Bound in a single borrow with + // no await in between, so the intermediate Authenticated state is never + // observable to a concurrent request on this connection. + { + let mut sessions = sessions.borrow_mut(); + sessions + .login(transport_client_id, user_id) + .map_err(LoginRegisterError::Session)?; + sessions.record_sdk_info(transport_client_id, sdk_info); + if let Err(error) = sessions.bind_session(transport_client_id, vsr_client_id, session) { + // No local rollback: `submit_register_in_process` above has + // already committed cluster-wide. A local-only + // `remove_client_session` here would diverge peers (they retain + // the slot until they evict the client themselves). The + // transport-disconnect callback owns local cleanup once the + // socket closes. + return Err(LoginRegisterError::Session(error)); + } + } + + // `session` IS the register's commit op, and on a backup that forwarded + // the proposal the local applied commit still lags it. Reporting the + // lower number would make one frame contradict itself. + let commit = current_metadata_commit(shard).max(session); + let reply = build_login_register_reply(request_header, vsr_client_id, session, commit, user_id); + let send_result = shard + .bus + .send_to_client(transport_client_id, reply.into_generic().into_frozen()) + .await; + if let Err(error) = send_result { + warn!( + transport_client_id, + error = %error, + "failed to send login/register reply" + ); + } + + Ok(()) +} + +/// Decide whether a failed login/register gets a terminal eviction or a +/// transient replay hint. +/// +/// A transient consensus failure ([`LoginRegisterError::is_terminal`] is +/// `false`) means the cluster could not commit *right now* (a freshly booted +/// primary still catching up, or a cross-shard submit canceled). Those get a +/// result-framed replay hint instead of silence, so the SDK replays at once +/// rather than waiting out its read-timeout; replying empty would surface as +/// a hard `InvalidFormat` decode failure and break the replay. +/// +/// Terminal auth errors (`InvalidCredentials` / `InvalidToken` / +/// `UserInactive` / `Session`) fast-fail with a typed `Eviction` frame so the +/// SDK surfaces the real reason (every frame transport decodes +/// `Command::Eviction`) instead of a decode error or a timeout. +#[allow(clippy::future_not_send)] +async fn surface_login_failure( + shard: &Rc>, + transport_client_id: u128, + request_header: &RoutedRequestHeader, + error: &LoginRegisterError, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + if error.is_terminal() { + send_login_eviction( + shard, + transport_client_id, + request_header.client, + eviction_reason_for(error), + ) + .await; + } else { + // Which code the hint carries is what tells the client whether the + // replay may move to another node: see `transient_login_code`. + send_login_transient_reply( + shard, + transport_client_id, + request_header, + transient_login_code(error), + ) + .await; + } +} + +/// Result-framed transient Reply on a non-terminal failed Register. The SDK +/// decodes the nonzero result code and replays the same login on the same +/// connection. Only call for transient errors -- see +/// [`surface_login_failure`]. +#[allow(clippy::future_not_send)] +async fn send_login_transient_reply( + shard: &Rc>, + transport_client_id: u128, + request_header: &RoutedRequestHeader, + code: IggyError, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let commit = current_metadata_commit(shard); + let reply = build_result_rejection_reply(request_header, commit, code.as_code()); + if let Err(error) = shard + .bus + .send_to_client(transport_client_id, reply.into_generic().into_frozen()) + .await + { + warn!( + transport_client_id, + error = %error, + "failed to send login transient reply" + ); + } +} + +/// Wire code for a transient (non-terminal) login/register failure. +/// +/// `TransientNotAccepted` asserts nothing was committed: the register never +/// entered a pipeline (not primary / not caught up / pipeline full) or never +/// left this node (primary unreachable). The client may re-issue it anywhere, +/// including under a fresh identity after failing over to another node. +/// +/// A forward timeout, an in-progress proposal, or a canceled proposal has an +/// UNKNOWN outcome, so none can ride that assertion. `TransientNotCommitted` +/// pins the replay to this connection and its client id, where a register that +/// did commit rebinds its own client-table entry. Re-issuing under a freshly +/// minted id would instead orphan that entry until capacity eviction reclaims +/// it. +const fn transient_login_code(error: &LoginRegisterError) -> IggyError { + match error { + LoginRegisterError::Transient( + MetadataSubmitError::ForwardTimedOut + | MetadataSubmitError::InProgress + | MetadataSubmitError::Canceled, + ) => IggyError::TransientNotCommitted, + _ => IggyError::TransientNotAccepted, + } +} + +/// Wire reason for a terminal login/register failure. Session-level +/// rejections (including the non-retryable submit refusal, where the +/// presented client id belongs to another user) collapse to +/// `SessionError`; the SDK maps it to `Unauthenticated`. +const fn eviction_reason_for(error: &LoginRegisterError) -> EvictionReason { + match error { + LoginRegisterError::InvalidCredentials => EvictionReason::InvalidCredentials, + LoginRegisterError::InvalidToken => EvictionReason::InvalidToken, + LoginRegisterError::UserInactive => EvictionReason::UserInactive, + _ => EvictionReason::SessionError, + } +} + +/// Reject a replicated request from an unbound transport with a typed +/// `Eviction(NoSession)` frame: the session the client believes it has is +/// gone, so it must register again. Pre-auth non-replicated reads get a +/// deny Reply instead (no session exists, so nothing is evicted). +/// +/// The SDK's reply decoder maps eviction reasons to typed errors +/// (`NoSession` -> `Unauthenticated`), so clients fail fast with the same +/// error the legacy server returns instead of a body-decode failure. The +/// eviction context is best-effort off the metadata consensus (peer shards +/// have none; zeroes are cosmetic -- the SDK only reads the reason). +#[allow(clippy::future_not_send)] +pub(in crate::dispatch) async fn send_unauthenticated_eviction( + shard: &Rc>, + transport_client_id: u128, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let ctx = shard.plane.metadata().consensus.as_ref().map_or( + consensus::EvictionContext { + cluster: 0, + view: 0, + replica: 0, + }, + consensus::EvictionContext::from_consensus, + ); + let eviction = consensus::build_eviction_message( + ctx, + transport_client_id, + iggy_binary_protocol::EvictionReason::NoSession, + ); + if let Err(error) = shard + .bus + .send_to_client(transport_client_id, eviction.into_generic().into_frozen()) + .await + { + warn!( + transport_client_id, + error = %error, + "failed to send unauthenticated eviction" + ); + } +} + +/// Per-shard heartbeat verifier: evict connections that have not pinged within +/// `1.2 x interval`. Mirrors the legacy `verify_heartbeats` periodic task. +/// Eviction reuses the disconnect path (drops the client from its consumer +/// groups + rebalances via the replicated `Logout`) and sends a session- +/// terminal `Eviction(StaleClient)` so the client fails fast and can reconnect. +#[allow(clippy::future_not_send)] +pub async fn run_heartbeat_verifier( + shard: Rc>, + sessions: Rc>, + interval: std::time::Duration, + stop_rx: shard::Receiver<()>, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + // Legacy `MAX_THRESHOLD`: a client is stale once it misses 1.2 intervals. + // Integer 6/5 rather than `mul_f64`, which panics on an absurd interval. + let max_age = interval.saturating_mul(6) / 5; + loop { + // `Ok(_)`: stop signalled -> exit. `Err(_)`: interval elapsed -> pass. + // Waiting on the stop channel rather than sleeping past it keeps this + // task inside the shutdown drain budget, which is shorter than the + // heartbeat interval. + let stop_signal = compio::time::timeout(interval, stop_rx.recv()).await; + if stop_signal.is_ok() { + break; + } + // Production-only wall clock: the heartbeat verifier is spawned solely + // by `build_shard_for_thread`, never by the simulator's + // `wire_shell_handlers`, so neither the interval wait above nor this + // read is on a deterministic path. Driving this task under the + // deterministic executor means routing both through the injected clock. + let stale = sessions + .borrow() + .collect_stale(max_age, std::time::Instant::now()); + for transport_client_id in stale { + // The heartbeat verifier exists to release a dead client's + // consumer-group membership (so the group rebalances off it). A + // connection that holds no membership has nothing for the eviction + // to clean up; reaping it would only drop a still-usable session + // (e.g. an idle admin connection that polls between long gaps), + // which the legacy server tolerates. The real transport-disconnect + // path still reaps it on socket close. So only evict a stale + // connection that is actually a group member. + let is_group_member = sessions + .borrow() + .bound_client_id(transport_client_id) + .is_some_and(|vsr_client_id| { + !shard + .plane + .metadata() + .mux_stm + .streams() + .consumer_group_memberships(vsr_client_id) + .is_empty() + }); + if is_group_member { + evict_stale_client(&shard, &sessions, transport_client_id).await; + } + } + } +} + +/// Evict one stale connection: drop its session (releasing consumer-group +/// membership through a replicated `Logout`) and notify the client with a +/// session-terminal `Eviction(StaleClient)`. +#[allow(clippy::future_not_send)] +async fn evict_stale_client( + shard: &Rc>, + sessions: &Rc>, + transport_client_id: u128, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let bound = sessions.borrow_mut().remove_connection(transport_client_id); + if let Some((vsr_client_id, session)) = bound { + submit_disconnect_logout(Rc::clone(shard), vsr_client_id, session); + } + let ctx = shard.plane.metadata().consensus.as_ref().map_or( + consensus::EvictionContext { + cluster: 0, + view: 0, + replica: 0, + }, + consensus::EvictionContext::from_consensus, + ); + let eviction = consensus::build_eviction_message( + ctx, + transport_client_id, + iggy_binary_protocol::EvictionReason::StaleClient, + ); + if let Err(error) = shard + .bus + .send_to_client(transport_client_id, eviction.into_generic().into_frozen()) + .await + { + warn!( + transport_client_id, + error = %error, + "failed to send stale-client eviction" + ); + } else { + warn!( + transport_client_id, + "evicted stale client (missed heartbeat)" + ); + } +} + +/// Answer a backup's forwarded `Register` from the node it named primary. +/// +/// Proposes in process, never through [`submit_register_local_or_forward`]: +/// that is what bounds a forward at one hop. A node that has since lost +/// primaryship answers `NotPrimary`, and the origin's client replays against +/// whichever node it names next. +#[allow(clippy::future_not_send)] +pub(in crate::dispatch) async fn answer_forwarded_register( + shard: &Rc>, + vsr_client_id: u128, + user_id: u32, + nonce: u128, + origin_replica: u8, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let Some((cluster, view, replica)) = shard + .plane + .metadata() + .consensus + .as_ref() + .map(|consensus| (consensus.cluster(), consensus.view(), consensus.replica())) + else { + warn!("ForwardedRegister submit reached a shard without metadata consensus"); + return; + }; + let bound = shard + .plane + .metadata() + .submit_register_in_process(vsr_client_id, user_id) + .await; + // `view` predates the await above, which parks with no deadline, so the + // sealed value can be stale by send time. The origin routes the result by + // `(nonce, client)` alone; this field must never become a freshness fence. + let result = + build_forward_register_result_message(cluster, view, replica, vsr_client_id, nonce, &bound); + if let Err(error) = shard + .bus + .send_to_replica(origin_replica, result.into_generic().into_frozen()) + .await + { + warn!( + origin_replica, + error = %error, + "failed to answer a forwarded register" + ); + } +} + +/// How long a login waits for the primary's verdict on a forwarded register. +/// +/// Expiry does NOT prove the peer or the frame was lost. The primary answers +/// only once the proposal resolves, and its own submit parks with no deadline: +/// a primary that is not caught up, or whose pipeline is full, absorbs the +/// register into its request queue and answers when that drains. So a slow but +/// healthy primary commits the register after this node has stopped waiting, +/// which is why expiry surfaces as `TransientNotCommitted` rather than the +/// not-accepted flavor. +/// +/// The budget stays well under the SDK's response-read timeout on purpose: the +/// client only replays a login while it is still reading, so a longer wait +/// here turns a transient into a torn-down socket. +const FORWARD_SUBMIT_TIMEOUT: Duration = Duration::from_secs(5); + +/// Run the `Register` proposal for a login this node has already +/// authenticated, wherever the metadata primary currently is. Shard 0 only. +/// +/// A client may dial any node in the cluster. Credentials verify against the +/// replicated users table, which every node holds, so the whole login except +/// the consensus proposal already works on a backup. Only the verified +/// identity crosses the replica interconnect -- never the client's frame and +/// never its credentials -- and the session bind, the reply, and the +/// connection all stay on the node the client dialed. +/// +/// The hop does not move any credential decision: +/// - `verify_login_credentials` reads the backup's applied replicated user +/// state. +/// - `verify_pat_credentials` reads the same state, so a PAT minted on the +/// primary that has not replicated here yet is refused until it does. +/// Fail-closed on purpose, the same parity the HTTP forward keeps: it too +/// answers 401 until replication catches up rather than relaying an +/// unverified bearer. +/// - `ClientIdOwnedByAnotherUser` stays a decision of the caught-up primary +/// and round-trips as a terminal refusal. +/// +/// Verification is point-in-time on the backup. A password change, PAT +/// revocation, or user deactivation committed on the primary but not yet +/// applied on the backup can therefore admit a login during the backup's apply +/// lag. The forward cannot complete while the backup is partitioned from the +/// primary, which bounds this to a connected replica's replication lag. This +/// is the same stale-read window as the existing HTTP forward. +/// +/// The session binds here before this node applies the commit locally. That +/// is the window a primary-side login already has against every other node's +/// apply lag, not a new one. +#[allow(clippy::future_not_send)] +pub(in crate::dispatch) async fn submit_register_local_or_forward( + shard: &Rc>, + vsr_client_id: u128, + user_id: u32, +) -> Result +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let Some(consensus) = shard.plane.metadata().consensus.as_ref() else { + return Err(MetadataSubmitError::NotPrimary); + }; + let (cluster, view, self_replica) = + (consensus.cluster(), consensus.view(), consensus.replica()); + let target = consensus.primary_index(view); + // Forward only as a healthy backup. Everything else answers locally: the + // in-process submit proposes when this node is the serving primary and + // re-derives `NotPrimary` otherwise -- mid view change there is nobody to + // forward to (the node the view names has not finished taking over, the + // SDK replays once it settles), and the view's own primary under state + // transfer has nowhere to forward to and nothing to commit yet. + if target == self_replica || !consensus.is_normal() { + return shard + .plane + .metadata() + .submit_register_in_process(vsr_client_id, user_id) + .await; + } + + let nonce = shard.next_forward_nonce(self_replica); + let (reply, outcome) = shard::channel::(1); + shard.park_register_forward(nonce, vsr_client_id, reply); + let forward = + build_forward_register_message(cluster, view, self_replica, vsr_client_id, nonce, user_id); + if let Err(error) = shard + .bus + .send_to_replica(target, forward.into_generic().into_frozen()) + .await + { + shard.cancel_register_forward(nonce, vsr_client_id); + warn!( + target, + error = %error, + "failed to forward register to the metadata primary" + ); + return Err(MetadataSubmitError::PrimaryUnreachable); + } + + match shard::bus_timeout(&shard.bus, FORWARD_SUBMIT_TIMEOUT, outcome.recv()).await { + Some(Ok(result)) => forward_register_result(&result), + // Shard-0 teardown dropped the sender without answering. + Some(Err(_)) => Err(MetadataSubmitError::Canceled), + None => { + shard.cancel_register_forward(nonce, vsr_client_id); + warn!(target, "forwarded register timed out"); + Err(MetadataSubmitError::ForwardTimedOut) + } + } +} + +/// The primary's verdict, back in the vocabulary the login path speaks. +const fn forward_register_result( + result: &ForwardRegisterResultHeader, +) -> Result { + match result.outcome { + ForwardRegisterOutcome::Ok => Ok(BoundSession { + epoch: result.epoch, + watermark: result.watermark, + }), + ForwardRegisterOutcome::NotPrimary => Err(MetadataSubmitError::NotPrimary), + ForwardRegisterOutcome::NotCaughtUp => Err(MetadataSubmitError::NotCaughtUp), + ForwardRegisterOutcome::PipelineFull => Err(MetadataSubmitError::PipelineFull), + ForwardRegisterOutcome::InProgress => Err(MetadataSubmitError::InProgress), + ForwardRegisterOutcome::Canceled => Err(MetadataSubmitError::Canceled), + ForwardRegisterOutcome::ClientIdOwnedByAnotherUser => { + Err(MetadataSubmitError::ClientIdOwnedByAnotherUser) + } + } +} + +/// Inverse of [`forward_register_result`], for the answering primary. +const fn forward_register_outcome( + bound: &Result, +) -> (BoundSession, ForwardRegisterOutcome) { + let zero = BoundSession { + epoch: 0, + watermark: 0, + }; + match bound { + Ok(bound) => (*bound, ForwardRegisterOutcome::Ok), + Err(MetadataSubmitError::NotPrimary) => (zero, ForwardRegisterOutcome::NotPrimary), + Err(MetadataSubmitError::NotCaughtUp) => (zero, ForwardRegisterOutcome::NotCaughtUp), + Err(MetadataSubmitError::PipelineFull) => (zero, ForwardRegisterOutcome::PipelineFull), + Err(MetadataSubmitError::InProgress) => (zero, ForwardRegisterOutcome::InProgress), + Err(MetadataSubmitError::ClientIdOwnedByAnotherUser) => { + (zero, ForwardRegisterOutcome::ClientIdOwnedByAnotherUser) + } + // `MetadataSubmitError` is `#[non_exhaustive]`. Every variant but the + // ownership refusal is transient by contract, and `Canceled` is the + // transient answer that claims nothing beyond "retry". + Err(_) => (zero, ForwardRegisterOutcome::Canceled), + } +} + +#[allow(clippy::cast_possible_truncation)] +fn build_forward_register_message( + cluster: u128, + view: u32, + replica: u8, + client: u128, + nonce: u128, + user_id: u32, +) -> Message { + Message::::new(HEADER_SIZE).transmute_header( + |_, header: &mut ForwardRegisterHeader| { + header.command = Command::ForwardRegister; + header.cluster = cluster; + header.view = view; + header.replica = replica; + header.client = client; + header.nonce = nonce; + header.user_id = user_id; + header.size = HEADER_SIZE as u32; + header.seal(); + }, + ) +} + +#[allow(clippy::cast_possible_truncation)] +fn build_forward_register_result_message( + cluster: u128, + view: u32, + replica: u8, + client: u128, + nonce: u128, + bound: &Result, +) -> Message { + let (session, outcome) = forward_register_outcome(bound); + Message::::new(HEADER_SIZE).transmute_header( + |_, header: &mut ForwardRegisterResultHeader| { + header.command = Command::ForwardRegisterResult; + header.cluster = cluster; + header.view = view; + header.replica = replica; + header.client = client; + header.nonce = nonce; + header.epoch = session.epoch; + header.watermark = session.watermark; + header.outcome = outcome; + header.size = HEADER_SIZE as u32; + header.seal(); + }, + ) +} + +/// Answer a backup's forwarded Logout from the node it named primary. +#[allow(clippy::future_not_send)] +pub(in crate::dispatch) async fn answer_forwarded_logout( + shard: &Rc>, + vsr_client_id: u128, + session: u64, + request: u64, + nonce: u128, + origin_replica: u8, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let Some((cluster, view, replica)) = shard + .plane + .metadata() + .consensus + .as_ref() + .map(|consensus| (consensus.cluster(), consensus.view(), consensus.replica())) + else { + warn!("ForwardedLogout submit reached a shard without metadata consensus"); + return; + }; + let outcome = shard + .plane + .metadata() + .submit_logout_in_process(vsr_client_id, session, request) + .await; + let result = + build_forward_logout_result_message(cluster, view, replica, vsr_client_id, nonce, &outcome); + if let Err(error) = shard + .bus + .send_to_replica(origin_replica, result.into_generic().into_frozen()) + .await + { + warn!( + origin_replica, + error = %error, + "failed to answer a forwarded logout" + ); + } +} + +/// Commit a Logout locally when this node is primary, otherwise forward it +/// once to the primary named by the current normal view. +#[allow(clippy::future_not_send)] +pub(in crate::dispatch) async fn submit_logout_local_or_forward( + shard: &Rc>, + vsr_client_id: u128, + session: u64, + request: u64, +) -> Result +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let Some(consensus) = shard.plane.metadata().consensus.as_ref() else { + return Err(MetadataSubmitError::NotPrimary); + }; + let (cluster, view, self_replica) = + (consensus.cluster(), consensus.view(), consensus.replica()); + let target = consensus.primary_index(view); + if target == self_replica || !consensus.is_normal() { + return shard + .plane + .metadata() + .submit_logout_in_process(vsr_client_id, session, request) + .await; + } + + let nonce = shard.next_forward_nonce(self_replica); + let (reply, outcome) = shard::channel::(1); + shard.park_logout_forward(nonce, vsr_client_id, reply); + let forward = build_forward_logout_message( + cluster, + view, + self_replica, + vsr_client_id, + nonce, + session, + request, + ); + if let Err(error) = shard + .bus + .send_to_replica(target, forward.into_generic().into_frozen()) + .await + { + shard.cancel_logout_forward(nonce, vsr_client_id); + warn!( + target, + error = %error, + "failed to forward logout to the metadata primary" + ); + return Err(MetadataSubmitError::PrimaryUnreachable); + } + + match shard::bus_timeout(&shard.bus, FORWARD_SUBMIT_TIMEOUT, outcome.recv()).await { + Some(Ok(result)) => forward_logout_result(&result), + Some(Err(_)) => Err(MetadataSubmitError::Canceled), + None => { + shard.cancel_logout_forward(nonce, vsr_client_id); + warn!(target, "forwarded logout timed out"); + Err(MetadataSubmitError::ForwardTimedOut) + } + } +} + +const fn forward_logout_result( + result: &ForwardLogoutResultHeader, +) -> Result { + match result.outcome { + ForwardLogoutOutcome::Ok => Ok(result.commit), + ForwardLogoutOutcome::NotPrimary => Err(MetadataSubmitError::NotPrimary), + ForwardLogoutOutcome::PipelineFull => Err(MetadataSubmitError::PipelineFull), + ForwardLogoutOutcome::InProgress => Err(MetadataSubmitError::InProgress), + ForwardLogoutOutcome::Canceled => Err(MetadataSubmitError::Canceled), + } +} + +const fn forward_logout_outcome( + outcome: &Result, +) -> (u64, ForwardLogoutOutcome) { + match outcome { + Ok(commit) => (*commit, ForwardLogoutOutcome::Ok), + Err(MetadataSubmitError::NotPrimary) => (0, ForwardLogoutOutcome::NotPrimary), + Err(MetadataSubmitError::PipelineFull) => (0, ForwardLogoutOutcome::PipelineFull), + Err(MetadataSubmitError::InProgress) => (0, ForwardLogoutOutcome::InProgress), + Err(_) => (0, ForwardLogoutOutcome::Canceled), + } +} + +#[allow(clippy::cast_possible_truncation, clippy::too_many_arguments)] +fn build_forward_logout_message( + cluster: u128, + view: u32, + replica: u8, + client: u128, + nonce: u128, + session: u64, + request: u64, +) -> Message { + Message::::new(HEADER_SIZE).transmute_header( + |_, header: &mut ForwardLogoutHeader| { + header.command = Command::ForwardLogout; + header.cluster = cluster; + header.view = view; + header.replica = replica; + header.client = client; + header.nonce = nonce; + header.session = session; + header.request = request; + header.size = HEADER_SIZE as u32; + header.seal(); + }, + ) +} + +#[allow(clippy::cast_possible_truncation)] +fn build_forward_logout_result_message( + cluster: u128, + view: u32, + replica: u8, + client: u128, + nonce: u128, + result: &Result, +) -> Message { + let (commit, outcome) = forward_logout_outcome(result); + Message::::new(HEADER_SIZE).transmute_header( + |_, header: &mut ForwardLogoutResultHeader| { + header.command = Command::ForwardLogoutResult; + header.cluster = cluster; + header.view = view; + header.replica = replica; + header.client = client; + header.nonce = nonce; + header.commit = commit; + header.outcome = outcome; + header.size = HEADER_SIZE as u32; + header.seal(); + }, + ) +} + +/// Run the consensus `Register` proposal on the metadata owner (shard 0) +/// and return the committed session. +/// +/// Credential verification and session binding stay on the calling (home) +/// shard -- only this consensus step must execute where the metadata +/// consensus group lives. On shard 0 it goes straight to +/// [`submit_register_local_or_forward`]; on a peer it forwards a +/// [`shard::MetadataSubmit`] to shard 0 and awaits the committed op. A dropped +/// reply (shard-0 inbox full / shutdown) maps to a transient `Canceled`, which +/// the caller wraps so the SDK replays. +#[allow(clippy::future_not_send)] +pub async fn submit_register_on_owner( + shard: &Rc>, + vsr_client_id: u128, + user_id: u32, +) -> Result +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + if shard.id == 0 { + return submit_register_local_or_forward(shard, vsr_client_id, user_id).await; + } + let (reply, rx) = shard::channel::>(1); + shard.forward_metadata_submit(shard::MetadataSubmit::Register { + vsr_client_id, + user_id, + reply, + }); + // The owner's outcome, verbatim in both directions. `Canceled` is only for a + // dropped channel, where nothing came back to classify. + rx.recv() + .await + .unwrap_or(Err(MetadataSubmitError::Canceled)) +} + +/// Logout counterpart of [`submit_register_on_owner`]. +#[allow(clippy::future_not_send)] +pub async fn submit_logout_on_owner( + shard: &Rc>, + vsr_client_id: u128, + session: u64, + request: u64, +) -> Result +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + if shard.id == 0 { + return submit_logout_local_or_forward(shard, vsr_client_id, session, request).await; + } + let (reply, rx) = shard::channel::>(1); + shard.forward_metadata_submit(shard::MetadataSubmit::Logout { + vsr_client_id, + session, + request, + reply, + }); + rx.recv() + .await + .unwrap_or(Err(MetadataSubmitError::Canceled)) +} + +/// Release the client-table slot for a disconnected transport, cluster-wide. +/// +/// The local `SessionManager` connection is already dropped by the caller; +/// this is what drops the replicated entry, so a peer replica does not keep an +/// orphaned session until it evicts one under capacity pressure. +/// +/// Unconditional, and deliberately so. Holding the slot open for a grace +/// window would let a reconnecting client resume onto its entry with its +/// watermark and reply ring intact, but nothing in tree re-presents a +/// `client_id` after a disconnect (the Rust SDK mints a fresh one on +/// re-login), so the window buys nothing today and the slot it holds is not +/// free: the client table's eviction point moves from concurrent connections +/// to CUMULATIVE connects, and every capacity eviction silently erases a +/// dedup watermark. +/// +/// A resume window becomes worth having once SDK-side identity stability +/// lands, at which point it needs a timer of its own -- riding the heartbeat +/// verifier would tie the grace period to heartbeat configuration, since +/// `collect_stale` keys off `heartbeat.interval` and the verifier does not run +/// at all when `heartbeat.enabled` is false. +/// Deliberately does NOT drop the local `ClientTable` slot first: +/// `submit_logout_*` short-circuits when the slot is already gone, so a +/// pre-emptive local removal would suppress the `Logout` and leave peer +/// replicas with an orphaned session until they evict it themselves -- the +/// exact divergence this avoids. `submit_logout_on_owner` runs in-process on +/// shard 0 and forwards for peer-homed connections; its session guard drops a +/// stale logout for a reused client id. +#[allow(clippy::future_not_send)] +pub(in crate::dispatch) fn submit_disconnect_logout( + shard: Rc>, + vsr_client_id: u128, + session: u64, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + // The sentinel request id is what the apply path reads to keep, rather + // than drop, the session's dedup fence: the client may be reconnecting + // under the same key, and its retry must still be answered. + let bus = shard.bus.clone(); + bus.spawn(async move { + if let Err(error) = + submit_logout_on_owner(&shard, vsr_client_id, session, DISCONNECT_LOGOUT_REQUEST_ID) + .await + { + warn!( + vsr_client_id, + ?error, + "disconnect logout submit failed; peer slots may linger until eviction" + ); + } + }); +} + +#[allow(clippy::future_not_send)] +pub(in crate::dispatch) async fn handle_logout_request( + shard: &Rc>, + sessions: &Rc>, + transport_client_id: u128, + request: Message, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let Some((vsr_client_id, session)) = sessions.borrow().get_session(transport_client_id) else { + // Logout on an unbound transport: the desired state already holds, + // so answer ok. A silent drop would wedge the lockstep SDK on this + // connection until its socket read timeout, and the SDK routinely + // sends a logout before each re-login. + warn!( + transport_client_id, + "logout for unbound VSR session; answering ok" + ); + let commit = current_metadata_commit(shard); + let reply = build_empty_reply(request.header(), transport_client_id, 0, commit); + if let Err(error) = shard + .bus + .send_to_client(transport_client_id, reply.into_generic().into_frozen()) + .await + { + warn!( + transport_client_id, + error = %error, + "failed to send unbound logout reply" + ); + } + return; + }; + + let request_id = request.header().request; + let commit = match submit_logout_on_owner(shard, vsr_client_id, session, request_id).await { + Ok(commit) => commit, + Err(error) => { + // Deny as transient instead of dropping the frame: the submit + // usually fails because this replica is not the metadata owner + // right now, and the SDK replays a transient rejection. + warn!(transport_client_id, error = %error, "logout/unregister failed; denying transient"); + let commit = current_metadata_commit(shard); + let reply = build_deny_reply( + request.header(), + vsr_client_id, + session, + commit, + transient_logout_code(&error).as_code(), + ); + if let Err(send_error) = shard + .bus + .send_to_client(transport_client_id, reply.into_generic().into_frozen()) + .await + { + warn!( + transport_client_id, + error = %send_error, + "failed to send logout deny reply" + ); + } + return; + } + }; + + sessions.borrow_mut().remove_connection(transport_client_id); + + let reply = build_empty_reply(request.header(), vsr_client_id, session, commit); + if let Err(error) = shard + .bus + .send_to_client(transport_client_id, reply.into_generic().into_frozen()) + .await + { + warn!( + transport_client_id, + error = %error, + "failed to send logout reply" + ); + } +} + +/// Preserve the client identity when a Logout may already have entered the +/// primary's pipeline. Moving an unknown-outcome replay to another connection +/// could race a later Register and obscure whether the old epoch was removed. +const fn transient_logout_code(error: &MetadataSubmitError) -> IggyError { + match error { + MetadataSubmitError::ForwardTimedOut + | MetadataSubmitError::InProgress + | MetadataSubmitError::Canceled => IggyError::TransientNotCommitted, + _ => IggyError::TransientNotAccepted, + } +} + +#[allow(clippy::future_not_send, clippy::too_many_lines)] +pub(in crate::dispatch) async fn handle_login_register_request( + shard: &Rc>, + sessions: &Rc>, + transport_client_id: u128, + request: Message, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let body = request_body(&request); + let vsr_client_id = request.header().client; + + // Both login-register shapes share the ClientVersionInfo prefix, so the + // protocol gate decodes it once and runs before any credential work; the + // body shapes below parse from past the prefix. Only VSR clients reach + // this gate -- legacy SDKs use LOGIN_USER_CODE, a separate path. A + // pre-versioning VSR client sends the old prefix-less body, which fails + // ClientVersionInfo::decode (-> MalformedLogin) or the version gate + // (-> IncompatibleProtocol) right here, not dropped earlier. + let Ok((version_info, prefix_len)) = ClientVersionInfo::decode(body) else { + warn!( + transport_client_id, + "rejecting login: body has no decodable version prefix" + ); + send_login_eviction( + shard, + transport_client_id, + vsr_client_id, + EvictionReason::MalformedLogin, + ) + .await; + return; + }; + if !is_protocol_compatible(version_info.protocol_version) { + warn!( + transport_client_id, + client_protocol_version = %ProtocolVersion(version_info.protocol_version), + sdk_name = %version_info.sdk_name, + sdk_version = %version_info.sdk_version, + "rejecting login: incompatible protocol version" + ); + send_login_eviction( + shard, + transport_client_id, + vsr_client_id, + EvictionReason::IncompatibleProtocol, + ) + .await; + return; + } + + let body_tail = &body[prefix_len..]; + let mut credentials_rejected = false; + if let Ok((wire_request, _)) = + LoginRegisterRequest::decode_after_prefix(version_info.clone(), body_tail) + { + match verify_login_credentials( + shard, + wire_request.username.as_str(), + wire_request.password.expose_secret(), + ) { + Ok(user_id) => { + if let Err(error) = complete_login_register( + shard, + sessions, + transport_client_id, + vsr_client_id, + request.header(), + user_id, + &wire_request.version_info, + ) + .await + { + warn!(transport_client_id, error = %error, "login/register failed"); + surface_login_failure(shard, transport_client_id, request.header(), &error) + .await; + } + return; + } + Err(LoginRegisterError::InvalidCredentials) => { + // Fall through to PAT attempt so a credential payload that + // collides with a valid PAT payload shape still gets a + // chance. A password-shaped body rarely parses as a PAT + // body, so remember the rejection: the final fall-through + // must surface InvalidCredentials, not MalformedLogin. + credentials_rejected = true; + } + Err(error) => { + warn!(transport_client_id, error = %error, "login/register failed"); + surface_login_failure(shard, transport_client_id, request.header(), &error).await; + return; + } + } + } + + if let Ok((wire_request, _)) = + LoginRegisterWithPatRequest::decode_after_prefix(version_info, body_tail) + { + match verify_pat_credentials(shard, wire_request.token.expose_secret()) { + Ok(user_id) => { + if let Err(error) = complete_login_register( + shard, + sessions, + transport_client_id, + vsr_client_id, + request.header(), + user_id, + &wire_request.version_info, + ) + .await + { + warn!( + transport_client_id, + error = %error, + "login/register with PAT failed" + ); + surface_login_failure(shard, transport_client_id, request.header(), &error) + .await; + } + return; + } + Err(error) => { + warn!( + transport_client_id, + error = %error, + "login/register with PAT failed" + ); + surface_login_failure(shard, transport_client_id, request.header(), &error).await; + return; + } + } + } + + if credentials_rejected { + warn!( + transport_client_id, + "rejecting register request: invalid credentials" + ); + send_login_eviction( + shard, + transport_client_id, + request.header().client, + EvictionReason::InvalidCredentials, + ) + .await; + return; + } + + warn!( + transport_client_id, + "rejecting register request with unsupported payload shape" + ); + send_login_eviction( + shard, + transport_client_id, + request.header().client, + EvictionReason::MalformedLogin, + ) + .await; +} + +/// Best-effort login-rejection eviction. Terminal one-way frame; a gone +/// connection has nothing to recover, so the send error is logged and +/// dropped. Consensus context (cluster/view/replica) is stamped on the +/// metadata shard and zeroed elsewhere -- the SDK only reads the reason, +/// plus the protocol window on `IncompatibleProtocol`. +#[allow(clippy::future_not_send)] +pub(in crate::dispatch) async fn send_login_eviction( + shard: &Rc>, + transport_client_id: u128, + vsr_client_id: u128, + reason: EvictionReason, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let ctx = shard.plane.metadata().consensus.as_ref().map_or( + EvictionContext { + cluster: 0, + view: 0, + replica: 0, + }, + EvictionContext::from_consensus, + ); + let eviction = match reason { + EvictionReason::IncompatibleProtocol => { + build_incompatible_protocol_eviction_message(ctx, vsr_client_id) + } + _ => build_eviction_message(ctx, vsr_client_id, reason), + }; + if let Err(error) = shard + .bus + .send_to_client(transport_client_id, eviction.into_generic().into_frozen()) + .await + { + warn!( + transport_client_id, + error = %error, + reason = ?reason, + "failed to send login eviction" + ); + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::dispatch::test_support::{ + FIRST_BOOT, SECOND_BOOT, SpyBus, TestMux, TestShard, prepare_message, register_reply, + request_message, test_shard, + }; + use crate::session_manager::SessionError; + use consensus::{LocalPipeline, Plane as _, PlaneKind, VsrConsensus}; + use iggy_binary_protocol::Operation; + use iggy_binary_protocol::requests::streams::CreateStreamRequest; + use iggy_binary_protocol::{PrepareOkHeader, ReplyHeader, WireEncode, WireOptions}; + use iggy_common::eviction_reason_to_error; + use journal::prepare_journal::PrepareJournal; + use message_bus::installer::conn_info::ClientTransportKind; + use metadata::IggyMetadata; + use metadata::impls::metadata::IggySnapshot; + use partitions::{IggyPartitions, PartitionPathLayout, PartitionsConfig}; + use server_common::MessageBag; + use server_common::sharding::ShardId; + use shard::shards_table::PapayaShardsTable; + use shard::{PartitionConsensusConfig, ReplicaTopology, ShardIdentity}; + use std::future::Future; + use std::mem::size_of; + + #[test] + fn terminal_vs_transient() { + assert!(LoginRegisterError::InvalidCredentials.is_terminal()); + assert!(LoginRegisterError::InvalidToken.is_terminal()); + assert!(LoginRegisterError::UserInactive.is_terminal()); + assert!(LoginRegisterError::Session(SessionError::ConnectionNotFound(0)).is_terminal()); + // Transient is the only recoverable variant: never terminal. + assert!(!LoginRegisterError::Transient(MetadataSubmitError::PipelineFull).is_terminal()); + } + + /// The reasons this module emits must map back to the credential + /// errors the SDK is expected to surface (the shared + /// `eviction_reason_to_error` grading both ends). + #[test] + fn terminal_login_errors_map_to_typed_sdk_errors() { + let cases = [ + ( + eviction_reason_for(&LoginRegisterError::InvalidCredentials), + IggyError::InvalidCredentials, + ), + ( + eviction_reason_for(&LoginRegisterError::InvalidToken), + IggyError::InvalidPersonalAccessToken, + ), + ( + eviction_reason_for(&LoginRegisterError::UserInactive), + IggyError::Unauthenticated, + ), + ]; + for (reason, expected) in cases { + let error = eviction_reason_to_error(reason, 0, 0); + assert_eq!( + error.as_code(), + expected.as_code(), + "reason {reason:?} must surface as {expected:?}" + ); + } + } + + #[test] + fn unknown_register_outcomes_pin_the_client_identity() { + for error in [ + MetadataSubmitError::ForwardTimedOut, + MetadataSubmitError::InProgress, + MetadataSubmitError::Canceled, + ] { + assert_eq!( + transient_login_code(&LoginRegisterError::Transient(error)), + IggyError::TransientNotCommitted, + ); + } + for error in [ + MetadataSubmitError::NotPrimary, + MetadataSubmitError::NotCaughtUp, + MetadataSubmitError::PipelineFull, + MetadataSubmitError::PrimaryUnreachable, + ] { + assert_eq!( + transient_login_code(&LoginRegisterError::Transient(error)), + IggyError::TransientNotAccepted, + ); + } + } + + /// Regression test for the production failure chain "CLI stream + /// create succeeded, logout failed: Disconnected". + /// + /// Why the logout of a CLI invocation used to fail during ITS OWN + /// successful `stream create`: the catch-up gate was GLOBAL. The suite + /// runs many CLI invocations against one shared single-node server; + /// each one is three replicated ops (Register, work, Logout). When + /// THIS client's logout frame arrived, some SIBLING client's op was + /// regularly sitting between quorum-ack (`commit_max` advanced inside + /// `on_ack`) and apply (`commit_min` still behind, driver parked at + /// the journal read). `submit_logout_in_process` then rejected + /// `NotCaughtUp`, and `handle_logout_request` swallowed the error: no + /// reply frame, session left bound. A one-shot CLI saw only a dead + /// connection — "Problem with server logout / Disconnected" — and + /// exited non-zero although its create committed; the harness retry + /// then tripped "already exists". + /// + /// This test rebuilds that interleaving deterministically (client B = + /// the sibling parked mid-commit; client A = the CLI logging out) and + /// pins the contract that fixed it (non-register ops carry no + /// catch-up gate, see `submit_logout_in_process`): + /// + /// a client-initiated logout must always produce a reply frame and + /// unbind the transport session, even while a sibling's commit is + /// in flight — the logout simply pipelines behind it. + #[compio::test] + async fn logout_rejected_by_closed_gate_must_still_reply_to_client() { + const CLIENT_A: u128 = 1; + const CLIENT_B: u128 = 2; + const SESSION: u64 = 1; + const ACTING_USER: u32 = 7; + const TRANSPORT_A: u128 = 77; + + let dir = tempfile::tempdir().unwrap(); + let journal = PrepareJournal::open(&dir.path().join("journal.wal"), 0) + .await + .unwrap(); + let bus = SpyBus::default(); + let consensus = VsrConsensus::new( + 1, + 0, + 1, + server_common::sharding::METADATA_GROUP, + bus.clone(), + LocalPipeline::new(), + ); + consensus.init(); + let metadata: IggyMetadata<_, PrepareJournal, IggySnapshot, TestMux> = IggyMetadata::new( + Some(consensus), + Some(journal), + None, + None, + TestMux::default(), + None, + ); + let partitions = IggyPartitions::new( + ShardId::new(0), + PartitionsConfig { + messages_required_to_save: 1, + size_of_messages_required_to_save: iggy_common::IggyByteSize::from(1024_u64), + enforce_fsync: false, + validate_checksum: true, + segment_size: iggy_common::IggyByteSize::from(1_048_576_u64), + preallocate_segments: false, + encryptor: None, + path_layout: PartitionPathLayout::default(), + }, + ); + let shard = Rc::new(TestShard::without_inbox( + ShardIdentity::new(0, "logout-window-test".to_string()), + bus.clone(), + metadata, + partitions, + PapayaShardsTable::new(), + PartitionConsensusConfig::new(1, ReplicaTopology::new(0, 1), bus.clone()), + )); + let md = shard.plane.metadata(); + let consensus = md.consensus.as_ref().unwrap(); + + // A and B hold committed sessions (as after their CLI logins). + for client in [CLIENT_A, CLIENT_B] { + md.client_table.borrow_mut().commit_register( + client, + ACTING_USER, + register_reply(client, SESSION), + ); + } + // A's transport connection, authenticated + bound — the state a + // CLI connection is in right after its create-stream reply. + let sessions = Rc::new(RefCell::new(SessionManager::new())); + sessions.borrow_mut().ensure_connection( + TRANSPORT_A, + "127.0.0.1:34567".parse().unwrap(), + ClientTransportKind::Tcp, + ); + sessions + .borrow_mut() + .login(TRANSPORT_A, ACTING_USER) + .unwrap(); + sessions + .borrow_mut() + .bind_session(TRANSPORT_A, CLIENT_A, SESSION) + .unwrap(); + + // Sibling B's op: prepared, journaled, self-acked through the real + // replicate path. (The public submit API cannot be used to open + // the window: `dispatch_prepare_and_await` pumps its own loopback + // inline, committing before it returns. Production's window is a + // sibling submit task parked INSIDE `on_ack`'s awaits — modeled + // below by driving `on_ack` by hand.) + let create_body = CreateStreamRequest { + name: iggy_binary_protocol::primitives::identifier::WireName::new("s1").unwrap(), + options: WireOptions::empty(), + } + .to_bytes(); + let prepare = prepare_message(Operation::CreateStream, CLIENT_B, 1, &create_body); + consensus.pipeline_message(PlaneKind::Metadata, &prepare); + md.on_replicate(prepare).await; + let mut loopback = Vec::new(); + consensus.drain_loopback_into(&mut loopback); + let ack = loopback + .pop() + .expect("one self-ack per replicated prepare") + .try_into_typed::() + .expect("loopback holds self PrepareOks"); + + // Open the window: first poll of `on_ack` advances commit_max at + // quorum, then parks at the journal read — commit_min unchanged. + // Every production NotCaughtUp logout was submitted exactly here. + let waker = std::task::Waker::noop(); + let mut cx = std::task::Context::from_waker(waker); + let mut driver = Box::pin(md.on_ack(ack)); + assert!( + driver.as_mut().poll(&mut cx).is_pending(), + "driver must park mid-commit at the journal read" + ); + assert_eq!(consensus.commit_max(), 1); + assert_eq!(consensus.commit_min(), 0); + + // A's logout lands in the window, through the real dispatch path. + let logout = request_message(Operation::Logout, CLIENT_A, SESSION, 2, &[]); + handle_logout_request(&shard, &sessions, TRANSPORT_A, logout).await; + + // DESIRED CONTRACT (red on current code): the client must never be + // left in silence — that silence is what a one-shot CLI reports as + // "Problem with server logout / Disconnected". + assert!( + bus.client_replies + .borrow() + .iter() + .any(|(client, _)| *client == TRANSPORT_A), + "logout must produce a reply frame to the client even while the \ + catch-up gate is closed (silence = CLI 'Disconnected', exit 1)" + ); + assert_eq!( + sessions.borrow().get_session(TRANSPORT_A), + None, + "transport session must be unbound by a client-initiated logout; \ + the VSR slot may lapse to the eviction sweep" + ); + } + + /// A backup's login: it forwards the register it authenticated to the + /// view's primary and completes on the primary's verdict, with the whole + /// round trip going through the real shard ingest arm. + #[compio::test] + async fn backup_forwards_register_and_completes_on_the_primary_verdict() { + const CLIENT: u128 = 0xCAFE; + const USER: u32 = 7; + const EPOCH: u64 = 41; + const WATERMARK: u64 = 9; + + let bus = SpyBus::default(); + // Replica 1 of 3, view 0: `primary_index(0)` is replica 0. + let shard = Rc::new(test_shard(&bus, 1, 3, FIRST_BOOT)); + let login = { + let shard = Rc::clone(&shard); + compio::runtime::spawn(async move { + submit_register_local_or_forward(&shard, CLIENT, USER).await + }) + }; + await_forward(&bus).await; + let (target, forward) = bus.sole_replica_send::(); + assert_eq!(target, 0, "forward must address the view's primary"); + assert_eq!(forward.command, Command::ForwardRegister); + assert_eq!(forward.client, CLIENT); + assert_eq!( + forward.user_id, USER, + "the forwarded identity is the payload" + ); + assert_eq!(forward.replica, 1, "the origin names itself for the answer"); + assert_ne!(forward.nonce, 0); + assert_eq!(forward.verify_frame(), Ok(()), "the frame must be sealed"); + assert_eq!(forward.validate(), Ok(())); + + shard + .on_message(forward_register_result( + &forward, + ForwardRegisterOutcome::Ok, + EPOCH, + WATERMARK, + )) + .await; + assert_eq!( + login.await.expect("the login task ran to completion"), + Ok(BoundSession { + epoch: EPOCH, + watermark: WATERMARK, + }) + ); + } + + #[compio::test] + async fn backup_forwards_logout_and_completes_on_the_primary_verdict() { + const CLIENT: u128 = 0xCAFE; + const SESSION: u64 = 41; + const REQUEST: u64 = 9; + const COMMIT: u64 = 42; + + let bus = SpyBus::default(); + let shard = Rc::new(test_shard(&bus, 1, 3, FIRST_BOOT)); + let logout = { + let shard = Rc::clone(&shard); + compio::runtime::spawn(async move { + submit_logout_local_or_forward(&shard, CLIENT, SESSION, REQUEST).await + }) + }; + await_forward(&bus).await; + let (target, forward) = bus.sole_replica_send::(); + assert_eq!(target, 0, "forward must address the view's primary"); + assert_eq!(forward.command, Command::ForwardLogout); + assert_eq!(forward.client, CLIENT); + assert_eq!(forward.session, SESSION); + assert_eq!(forward.request, REQUEST); + assert_eq!(forward.replica, 1); + assert_ne!(forward.nonce, 0); + assert_eq!(forward.verify_frame(), Ok(())); + assert_eq!(forward.validate(), Ok(())); + + shard + .on_message(forward_logout_result_message(&forward, &Ok(COMMIT))) + .await; + assert_eq!( + logout.await.expect("the logout task ran to completion"), + Ok(COMMIT) + ); + } + + #[compio::test] + async fn unanswered_logout_forward_times_out_and_clears_the_waiter() { + let bus = SpyBus::default(); + bus.instant_timers.set(true); + let shard = Rc::new(test_shard(&bus, 1, 3, FIRST_BOOT)); + + let outcome = submit_logout_local_or_forward(&shard, 0xCAFE, 41, 9).await; + assert_eq!(outcome, Err(MetadataSubmitError::ForwardTimedOut)); + + let (_, forward) = bus.sole_replica_send::(); + shard + .on_message(forward_logout_result_message(&forward, &Ok(42))) + .await; + } + + #[test] + fn unknown_logout_outcomes_pin_the_session() { + for error in [ + MetadataSubmitError::ForwardTimedOut, + MetadataSubmitError::InProgress, + MetadataSubmitError::Canceled, + ] { + assert_eq!( + transient_logout_code(&error), + IggyError::TransientNotCommitted + ); + } + for error in [ + MetadataSubmitError::NotPrimary, + MetadataSubmitError::PipelineFull, + MetadataSubmitError::PrimaryUnreachable, + ] { + assert_eq!( + transient_logout_code(&error), + IggyError::TransientNotAccepted + ); + } + } + + /// The ownership refusal is the one terminal verdict, and it has to stay + /// terminal across the hop or the SDK replays a login that cannot succeed. + #[compio::test] + async fn forwarded_register_keeps_the_ownership_refusal_terminal() { + let bus = SpyBus::default(); + let shard = Rc::new(test_shard(&bus, 1, 3, FIRST_BOOT)); + let login = { + let shard = Rc::clone(&shard); + compio::runtime::spawn(async move { + submit_register_local_or_forward(&shard, 0xCAFE, 7).await + }) + }; + await_forward(&bus).await; + let (_, forward) = bus.sole_replica_send::(); + shard + .on_message(forward_register_result( + &forward, + ForwardRegisterOutcome::ClientIdOwnedByAnotherUser, + 0, + 0, + )) + .await; + let error = login + .await + .expect("the login task ran to completion") + .expect_err("the refusal must surface"); + assert_eq!(error, MetadataSubmitError::ClientIdOwnedByAnotherUser); + assert!(!error.is_transient(), "the refusal must stay terminal"); + } + + /// A primary that never answers must not strand the login or leak its + /// parked entry; the client gets a transient failure and replays. + #[compio::test] + async fn unanswered_forward_times_out_and_clears_the_parked_login() { + let bus = SpyBus::default(); + bus.instant_timers.set(true); + let shard = Rc::new(test_shard(&bus, 1, 3, FIRST_BOOT)); + + let outcome = submit_register_local_or_forward(&shard, 0xCAFE, 7).await; + assert_eq!(outcome, Err(MetadataSubmitError::ForwardTimedOut)); + assert!( + outcome.unwrap_err().is_transient(), + "a lost answer is replayable" + ); + + // The parked entry is gone: the answer that arrives late finds nothing + // and is dropped rather than completing a login nobody is waiting on. + let (_, forward) = bus.sole_replica_send::(); + shard + .on_message(forward_register_result( + &forward, + ForwardRegisterOutcome::Ok, + 41, + 0, + )) + .await; + } + + /// The reply frame is where an unknown outcome has to be told apart from a + /// refusal: a forward that timed out may still commit, so the client must + /// replay under the same client id instead of failing over under a fresh + /// one. A verdict that refused the register carries no such doubt. + #[compio::test] + async fn transient_login_reply_marks_a_timed_out_forward_not_committed() { + const TRANSPORT: u128 = 91; + const VSR_CLIENT: u128 = 0xCAFE; + const RESULT_OFFSET: usize = size_of::() + 8; + + let bus = SpyBus::default(); + let shard = Rc::new(test_shard(&bus, 1, 3, FIRST_BOOT)); + let request = request_message(Operation::Register, VSR_CLIENT, 0, 0, &[]); + + for (submit_error, expected) in [ + ( + MetadataSubmitError::ForwardTimedOut, + IggyError::TransientNotCommitted, + ), + ( + MetadataSubmitError::NotPrimary, + IggyError::TransientNotAccepted, + ), + ] { + let error = LoginRegisterError::Transient(submit_error); + surface_login_failure(&shard, TRANSPORT, request.header(), &error).await; + + let replies = bus.client_replies.borrow(); + assert_eq!(replies.len(), 1, "a transient login must answer a frame"); + let (client, frame) = &replies[0]; + assert_eq!(*client, TRANSPORT, "reply must target the transport id"); + let result = + u32::from_le_bytes(frame[RESULT_OFFSET..RESULT_OFFSET + 4].try_into().unwrap()); + assert_eq!(result, expected.as_code(), "{error} must reply {expected}"); + drop(replies); + bus.client_replies.borrow_mut().clear(); + } + } + + /// A restart must not re-mint the nonce sequence of the boot before it. The + /// nonce is never persisted, and an answer to a pre-restart forward can + /// still be in flight: routed by a repeated nonce it would confirm a login + /// the cluster never committed, with another client's epoch. + #[compio::test] + async fn a_restart_moves_the_forward_nonce_sequence() { + assert_ne!( + first_forward_nonce(FIRST_BOOT).await, + first_forward_nonce(SECOND_BOOT).await, + "each boot must start its nonce sequence somewhere the other did not" + ); + } + + /// Seeding the counter from the incarnation means it can start one step + /// short of wrapping, and a zero nonce is a frame every replica rejects. + #[compio::test] + async fn wrapping_forward_nonce_counter_skips_zero() { + let nonce = first_forward_nonce(u128::from(u64::MAX)).await; + assert_ne!( + nonce & u128::from(u64::MAX), + 0, + "a counter that wrapped must not contribute a zero nonce half" + ); + } + + /// An answer echoing a client the nonce was never parked for must neither + /// complete that login nor evict it, since a repeated nonce is exactly what + /// a late cross-boot answer carries. + #[compio::test] + async fn forward_result_for_another_client_leaves_the_login_parked() { + const CLIENT: u128 = 0xCAFE; + const EPOCH: u64 = 41; + const WATERMARK: u64 = 9; + const FOREIGN_EPOCH: u64 = 77; + + let bus = SpyBus::default(); + let shard = Rc::new(test_shard(&bus, 1, 3, FIRST_BOOT)); + let login = { + let shard = Rc::clone(&shard); + compio::runtime::spawn(async move { + submit_register_local_or_forward(&shard, CLIENT, 7).await + }) + }; + await_forward(&bus).await; + let (_, forward) = bus.sole_replica_send::(); + + let mut foreign = forward; + foreign.client = CLIENT + 1; + shard + .on_message(forward_register_result( + &foreign, + ForwardRegisterOutcome::Ok, + FOREIGN_EPOCH, + 0, + )) + .await; + shard + .on_message(forward_register_result( + &forward, + ForwardRegisterOutcome::Ok, + EPOCH, + WATERMARK, + )) + .await; + assert_eq!( + login.await.expect("the login task ran to completion"), + Ok(BoundSession { + epoch: EPOCH, + watermark: WATERMARK, + }), + "the login must bind the epoch addressed to it, and must still be \ + parked to receive it" + ); + } + + /// A node that is primary itself never forwards -- that is what bounds a + /// forward at one hop. + #[compio::test] + async fn primary_proposes_locally_instead_of_forwarding() { + let bus = SpyBus::default(); + // Replica 0 of 3, view 0: this node IS the primary. + let shard = Rc::new(test_shard(&bus, 0, 3, FIRST_BOOT)); + + // No journal on the test shard, so the proposal cannot commit; what + // matters is that nothing left over the interconnect. + let _ = compio::time::timeout( + Duration::from_millis(50), + submit_register_local_or_forward(&shard, 0xCAFE, 7), + ) + .await; + assert!( + bus.replica_sends.borrow().is_empty(), + "a primary must propose in process" + ); + } + + /// The nonce a shard booted at `incarnation` stamps on its first forward. + /// Nobody answers, so the login abandons on the instant timer; the frame it + /// left on the bus is what the caller is after. + async fn first_forward_nonce(incarnation: u128) -> u128 { + let bus = SpyBus::default(); + bus.instant_timers.set(true); + let shard = Rc::new(test_shard(&bus, 1, 3, incarnation)); + let outcome = submit_register_local_or_forward(&shard, 0xCAFE, 7).await; + assert_eq!(outcome, Err(MetadataSubmitError::ForwardTimedOut)); + bus.sole_replica_send::().1.nonce + } + + /// Let a spawned login run until it has parked on the primary's answer. + async fn await_forward(bus: &SpyBus) { + for _ in 0..1000 { + if !bus.replica_sends.borrow().is_empty() { + return; + } + compio::time::sleep(Duration::from_millis(1)).await; + } + panic!("the login never forwarded a register"); + } + + /// A sealed `ForwardRegisterResult` addressed to `forward`'s nonce. + fn forward_register_result( + forward: &ForwardRegisterHeader, + outcome: ForwardRegisterOutcome, + epoch: u64, + watermark: u64, + ) -> MessageBag { + let bound = match outcome { + ForwardRegisterOutcome::Ok => Ok(BoundSession { epoch, watermark }), + ForwardRegisterOutcome::ClientIdOwnedByAnotherUser => { + Err(MetadataSubmitError::ClientIdOwnedByAnotherUser) + } + _ => Err(MetadataSubmitError::NotPrimary), + }; + MessageBag::ForwardRegisterResult(build_forward_register_result_message( + forward.cluster, + forward.view, + 0, + forward.client, + forward.nonce, + &bound, + )) + } + + fn forward_logout_result_message( + forward: &ForwardLogoutHeader, + outcome: &Result, + ) -> MessageBag { + MessageBag::ForwardLogoutResult(build_forward_logout_result_message( + forward.cluster, + forward.view, + 0, + forward.client, + forward.nonce, + outcome, + )) + } +} diff --git a/core/server/src/dispatch/submit.rs b/core/server/src/dispatch/submit.rs new file mode 100644 index 0000000000..db3529b1f4 --- /dev/null +++ b/core/server/src/dispatch/submit.rs @@ -0,0 +1,194 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! The shard-0 metadata-submit RPC, both ends on one page. +//! +//! The metadata consensus group lives on shard 0, but connections live on +//! their home shards. Peer shards send a [`shard::MetadataSubmit`] and await +//! the committed outcome; [`make_metadata_submit_handler`] is what shard 0 +//! runs for those frames. The session-lifecycle arms (register / logout and +//! their replica forwards) delegate to `session_ops`, which owns that +//! machinery. + +use crate::dispatch::session_ops::{ + answer_forwarded_logout, answer_forwarded_register, submit_logout_local_or_forward, + submit_register_local_or_forward, +}; +use crate::dispatch::upgrade_shard_handle; +use crate::shell::{ShellBus, ShellShard, ShellShardHandle}; +use consensus::MetadataHandle; +use iggy_binary_protocol::{GenericHeader, PrepareHeader, RoutedRequestHeader}; +use journal::superblock::SuperblockStore; +use journal::{Journal, JournalHandle}; +use server_common::Message; +use std::rc::Rc; +use tracing::warn; + +/// Handler shard 0 runs for an inbound [`shard::MetadataSubmit`]: a peer +/// shard has verified credentials and owns the session locally, and asks +/// shard 0 (the metadata consensus owner) to run only the consensus +/// proposal. Spawns a task so the awaiting peer is woken once the op +/// commits. Submit failures are returned verbatim so the peer can preserve +/// unknown-outcome retry semantics. +pub fn make_metadata_submit_handler( + shard_handle: &ShellShardHandle, +) -> shard::MetadataSubmitHandler +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let shard_handle = Rc::clone(shard_handle); + Rc::new(move |submit| { + let Some(shard) = upgrade_shard_handle(&shard_handle) else { + return; + }; + let bus = shard.bus.clone(); + bus.spawn(async move { + match submit { + shard::MetadataSubmit::Register { + vsr_client_id, + user_id, + reply, + } => { + let bound = + submit_register_local_or_forward(&shard, vsr_client_id, user_id).await; + let _ = reply.try_send(bound); + } + shard::MetadataSubmit::ForwardedRegister { + vsr_client_id, + user_id, + nonce, + origin_replica, + } => { + answer_forwarded_register( + &shard, + vsr_client_id, + user_id, + nonce, + origin_replica, + ) + .await; + } + shard::MetadataSubmit::ForwardedLogout { + vsr_client_id, + session, + request, + nonce, + origin_replica, + } => { + answer_forwarded_logout( + &shard, + vsr_client_id, + session, + request, + nonce, + origin_replica, + ) + .await; + } + shard::MetadataSubmit::Logout { + vsr_client_id, + session, + request, + reply, + } => { + let outcome = + submit_logout_local_or_forward(&shard, vsr_client_id, session, request) + .await; + let _ = reply.try_send(outcome); + } + shard::MetadataSubmit::ClientRequest { request, reply } => { + let committed = match request.try_into_typed::() { + Ok(typed) => shard + .plane + .metadata() + .submit_request_in_process(typed) + .await + .ok(), + Err(error) => { + warn!(?error, "ClientRequest submit: undecodable request header"); + None + } + }; + let _ = reply.try_send(committed); + } + shard::MetadataSubmit::CompleteRevocation { + stream_id, + topic_id, + group_id, + source_client_id, + partition_id, + reply, + } => { + let commit = shard + .plane + .metadata() + .submit_complete_revocation_in_process( + stream_id, + topic_id, + group_id, + source_client_id, + partition_id, + ) + .await + .ok(); + let _ = reply.try_send(commit); + } + } + }); + }) +} + +/// Submit a replicated client request to the metadata owner (shard 0) and +/// return the committed reply. +/// +/// The metadata consensus group lives on shard 0, but the connection lives +/// on the home shard (this shard). Run consensus where it belongs and bring +/// the committed reply back here so the caller can write it to the +/// originating socket -- shard 0 cannot route the reply by the consensus +/// `client` id (it's the VSR id, not the transport/home-shard-encoding id). +/// `None` = transient submit failure (SDK read-timeout replays). +#[allow(clippy::future_not_send)] +pub async fn submit_client_request_on_owner( + shard: &Rc>, + request: Message, +) -> Option> +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + if shard.id == 0 { + return shard + .plane + .metadata() + .submit_request_in_process(request) + .await + .ok(); + } + let (reply, rx) = shard::channel::>>(1); + shard.forward_metadata_submit(shard::MetadataSubmit::ClientRequest { + request: request.into_generic(), + reply, + }); + rx.recv().await.ok().flatten() +} diff --git a/core/server/src/dispatch/test_support.rs b/core/server/src/dispatch/test_support.rs new file mode 100644 index 0000000000..e52578c64d --- /dev/null +++ b/core/server/src/dispatch/test_support.rs @@ -0,0 +1,279 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Shared harness for the dispatch tree's unit tests: the [`SpyBus`] records +//! client- and replica-bound frames instead of writing to sockets, and the +//! builders below assemble minimal shards and wire frames. + +use consensus::{LocalPipeline, VsrConsensus}; +use iggy_binary_protocol::{Command, Operation, PrepareHeader, ReplyHeader, RoutedRequestHeader}; +use iggy_common::variadic; +use journal::prepare_journal::PrepareJournal; +use message_bus::client_listener::RequestHandler; +use message_bus::fd_transfer::DupedFd; +use message_bus::installer::ConnectionInstaller; +use message_bus::installer::conn_info::ClientConnMeta; +use message_bus::replica::listener::MessageHandler; +use message_bus::{ + BusMessage, ClientConnectionLostFn, ClientForwardFn, ConnectionLostFn, JoinHandle, MessageBus, + ReplicaForwardFn, ReplicaHandshakeDoneFn, SendError, +}; +use metadata::impls::metadata::IggySnapshot; +use metadata::stm::stream::Streams; +use metadata::stm::user::Users; +use metadata::{IggyMetadata, MuxStateMachine}; +use partitions::{IggyPartitions, PartitionPathLayout, PartitionsConfig}; +use server_common::iobuf::Frozen; +use server_common::sharding::ShardId; +use server_common::{MESSAGE_ALIGN, Message}; +use shard::shards_table::PapayaShardsTable; +use shard::{IggyShard, PartitionConsensusConfig, ReplicaTopology, ShardIdentity}; +use std::cell::{Cell, RefCell}; +use std::mem::size_of; +use std::rc::Rc; + +pub type TestMux = MuxStateMachine; +pub type TestShard = IggyShard; +/// `(target client id, reply frame bytes)` per `send_to_client` call. +pub type RecordedReplies = Rc)>>>; +/// `(target replica id, frame bytes)` per `send_to_replica` call. +pub type RecordedReplicaSends = Rc)>>>; + +/// Records every client-bound reply and replica-bound frame (target + +/// bytes) instead of writing to a socket; everything else is a no-op. The +/// two `ShellBus` halves are stubbed. +#[derive(Debug, Clone, Default)] +pub struct SpyBus { + pub client_replies: RecordedReplies, + pub replica_sends: RecordedReplicaSends, + /// Resolve [`MessageBus::sleep`] immediately instead of arming a real + /// timer. The register forward is the only path here that races a + /// timer, and its budget is five seconds -- too long to wait for in a + /// unit test, and too long to shorten in production for one. + pub instant_timers: Rc>, +} + +impl SpyBus { + /// Decode the single frame this bus sent to a replica. + pub fn sole_replica_send(&self) -> (u8, H) { + let sends = self.replica_sends.borrow(); + assert_eq!(sends.len(), 1, "expected exactly one replica-bound frame"); + let (target, frame) = &sends[0]; + let mut aligned = server_common::iobuf::Owned::::zeroed(frame.len()); + aligned.as_mut_slice().copy_from_slice(frame); + let header = *bytemuck::checked::try_from_bytes::(&aligned.as_slice()[..size_of::()]) + .expect("replica frame decodes into the expected header"); + (*target, header) + } +} + +#[allow(clippy::future_not_send)] +impl MessageBus for SpyBus { + fn track_background(&self, _handle: JoinHandle<()>) {} + async fn send_to_client( + &self, + client_id: u128, + data: impl Into, + ) -> Result<(), SendError> { + self.client_replies + .borrow_mut() + .push((client_id, data.into().into_contiguous().as_slice().to_vec())); + Ok(()) + } + async fn send_to_replica( + &self, + replica: u8, + data: Frozen, + ) -> Result<(), SendError> { + self.replica_sends + .borrow_mut() + .push((replica, data.as_slice().to_vec())); + Ok(()) + } + async fn sleep(&self, duration: std::time::Duration) { + if !self.instant_timers.get() { + compio::time::sleep(duration).await; + } + } + fn set_connection_lost_fn(&self, _f: ConnectionLostFn) {} + fn set_replica_forward_fn(&self, _f: ReplicaForwardFn) {} + fn set_client_forward_fn(&self, _f: ClientForwardFn) {} +} + +impl ConnectionInstaller for SpyBus { + fn install_replica_inbound_fd( + &self, + _fd: DupedFd, + _on_message: MessageHandler, + _on_done: ReplicaHandshakeDoneFn, + ) { + } + fn install_replica_outbound_fd( + &self, + _fd: DupedFd, + _replica_id: u8, + _on_message: MessageHandler, + _on_done: ReplicaHandshakeDoneFn, + ) { + } + fn release_replica_handshake_slot(&self, _slot: u64) {} + fn clear_replica_dial_pending(&self, _replica_id: u8) {} + fn install_client_fd(&self, _fd: DupedFd, _meta: ClientConnMeta, _on_request: RequestHandler) {} + fn install_client_ws_fd( + &self, + _fd: DupedFd, + _meta: ClientConnMeta, + _on_request: RequestHandler, + ) { + } + fn client_meta(&self, _client_id: u128) -> Option> { + None + } + fn set_client_connection_lost_fn(&self, _f: ClientConnectionLostFn) {} +} + +/// Consensus incarnations standing for two successive boots of one node, as +/// far apart as the random draw at bootstrap makes them. +pub const FIRST_BOOT: u128 = 0x5EED_0001; +pub const SECOND_BOOT: u128 = 0x9E37_79B9_7F4A_7C15; + +/// Shard 0 carrying a metadata consensus group of `replica_count` +/// replicas in which this node is `replica`. No journal: every test using +/// it either never proposes, or is a backup that cannot. +/// +/// `incarnation` stands for one boot of this node: the shard seeds its +/// forward-nonce counter from it, so passing a different value models a +/// restart. +pub fn test_shard(bus: &SpyBus, replica: u8, replica_count: u8, incarnation: u128) -> TestShard { + let consensus = VsrConsensus::new( + 1, + replica, + replica_count, + server_common::sharding::METADATA_GROUP, + bus.clone(), + LocalPipeline::new(), + ); + consensus.set_incarnation(incarnation); + consensus.init(); + let metadata: IggyMetadata<_, PrepareJournal, IggySnapshot, TestMux> = + IggyMetadata::new(Some(consensus), None, None, None, TestMux::default(), None); + let partitions = IggyPartitions::new( + ShardId::new(0), + PartitionsConfig { + messages_required_to_save: 1, + size_of_messages_required_to_save: iggy_common::IggyByteSize::from(1024_u64), + enforce_fsync: false, + validate_checksum: true, + segment_size: iggy_common::IggyByteSize::from(1_048_576_u64), + preallocate_segments: false, + encryptor: None, + path_layout: PartitionPathLayout::default(), + }, + ); + TestShard::without_inbox( + ShardIdentity::new(0, "dispatch-test".to_string()), + bus.clone(), + metadata, + partitions, + PapayaShardsTable::new(), + PartitionConsensusConfig::new(1, ReplicaTopology::new(replica, replica_count), bus.clone()), + ) +} + +/// Minimal committed `Register` reply for `ClientTable::commit_register` +/// (reads only `client` and `commit`). +pub fn register_reply(client: u128, session: u64) -> Message { + let header_size = size_of::(); + let mut reply = Message::::new(header_size); + let header = bytemuck::checked::try_from_bytes_mut::( + &mut reply.as_mut_slice()[..header_size], + ) + .expect("zeroed bytes are a valid ReplyHeader"); + *header = ReplyHeader { + client, + request: 0, + commit: session, + command: Command::Reply, + operation: Operation::Register, + ..Default::default() + }; + reply +} + +pub fn request_message( + operation: Operation, + client: u128, + session: u64, + request: u64, + body: &[u8], +) -> Message { + let header_size = size_of::(); + let total = header_size + body.len(); + let mut message = Message::::new(total); + { + let slice = message.as_mut_slice(); + slice[header_size..total].copy_from_slice(body); + let header = + bytemuck::checked::from_bytes_mut::(&mut slice[..header_size]); + *header = RoutedRequestHeader { + command: Command::Request, + operation, + size: u32::try_from(total).expect("test request fits u32"), + client, + session, + request, + user_id: 0, + group: server_common::sharding::METADATA_GROUP, + ..Default::default() + }; + } + message +} + +/// Raw prepare for the sibling op, standing in for the crate-private +/// `prepare_request` projection. `user_id` 0 skips the in-apply RBAC +/// gate (server-originated convention), so the op applies cleanly. +pub fn prepare_message( + operation: Operation, + client: u128, + request: u64, + body: &[u8], +) -> Message { + let header_size = size_of::(); + let total = header_size + body.len(); + let mut message = Message::::new(total); + { + let slice = message.as_mut_slice(); + slice[header_size..total].copy_from_slice(body); + let header = bytemuck::checked::from_bytes_mut::(&mut slice[..header_size]); + *header = PrepareHeader { + command: Command::Prepare, + operation, + size: u32::try_from(total).expect("test prepare fits u32"), + op: 1, + view: 0, + client, + request, + user_id: 0, + group: server_common::sharding::METADATA_GROUP, + ..Default::default() + }; + } + // A real identity, not a placeholder: `on_replicate` recomputes it before the + // prepare reaches the WAL, so an arbitrary value reads as transit corruption. + consensus::seal_prepare_checksum(message) +} diff --git a/core/server/src/http/extractor.rs b/core/server/src/http/extractor.rs index e818a4e69a..92c3170ca3 100644 --- a/core/server/src/http/extractor.rs +++ b/core/server/src/http/extractor.rs @@ -31,7 +31,7 @@ use tracing::debug; use super::HttpState; use super::error::AuthError; use super::session::HttpSession; -use crate::auth::verify_pat_credentials_with_expiry; +use crate::dispatch::session_ops::verify_pat_credentials_with_expiry; use crate::http::ClientAddr; /// Bearer scheme prefix in the `Authorization` header. diff --git a/core/server/src/http/handlers.rs b/core/server/src/http/handlers.rs index cdce21b5c4..8483138096 100644 --- a/core/server/src/http/handlers.rs +++ b/core/server/src/http/handlers.rs @@ -118,11 +118,11 @@ use send_wrapper::SendWrapper; use serde::Deserialize; use shard::{PartitionRead, PartitionReadReply}; -use crate::auth::{verify_login_credentials, verify_pat_credentials}; +use crate::dispatch::partition::{resolve_consumer_offset_request, resolve_poll_request}; +use crate::dispatch::session_ops::{verify_login_credentials, verify_pat_credentials}; use crate::dispatch::{ - resolve_consumer_offset_request, resolve_poll_request, validate_option_keys, - validate_topic_bounds, validate_topic_size_floor, warn_unenforceable_topic_size, - warn_unenforceable_topic_size_on_partition_add, + validate_option_keys, validate_topic_bounds, validate_topic_size_floor, + warn_unenforceable_topic_size, warn_unenforceable_topic_size_on_partition_add, }; use crate::http::error::{ Consistency, ConsistencyQuery, CustomError, PartitionWriteError, ProduceAck, ProduceQuery, diff --git a/core/server/src/http/reply.rs b/core/server/src/http/reply.rs index 0ce0b14f4b..29babb6b3f 100644 --- a/core/server/src/http/reply.rs +++ b/core/server/src/http/reply.rs @@ -36,8 +36,8 @@ use iggy_common::{ use server_common::{MESSAGE_ALIGN, Message, iobuf::Frozen}; use tracing::warn; +use crate::dispatch::login_error::LoginRegisterError; use crate::http::error::{PartitionWriteError, WriteError}; -use crate::login_register::LoginRegisterError; /// Discriminate a partition write reply. Partition replies carry no result /// section - a denial is empty-bodied and a committed body, where there is one, diff --git a/core/server/src/http/state.rs b/core/server/src/http/state.rs index c0c67e7dec..5f61f82897 100644 --- a/core/server/src/http/state.rs +++ b/core/server/src/http/state.rs @@ -38,7 +38,7 @@ use tokio::sync::Mutex; use tracing::warn; use crate::cluster_meta::ClusterRoster; -use crate::dispatch::submit_register_on_owner; +use crate::dispatch::session_ops::submit_register_on_owner; use crate::http::error::{AuthError, ReadError, primary_redirect_location}; use crate::http::jwt::JwtManager; diff --git a/core/server/src/http/submit.rs b/core/server/src/http/submit.rs index e7dd75efd3..d73b2005ee 100644 --- a/core/server/src/http/submit.rs +++ b/core/server/src/http/submit.rs @@ -32,10 +32,9 @@ use metadata::impls::metadata::StreamsFrontend; use server_common::{MESSAGE_ALIGN, Message, iobuf::Frozen}; use tracing::warn; -use crate::dispatch::{ - dispatch_partition_request, resolve_delete_segments_truncate, submit_client_request_on_owner, - submit_logout_on_owner, -}; +use crate::dispatch::partition::{dispatch_partition_request, resolve_delete_segments_truncate}; +use crate::dispatch::session_ops::submit_logout_on_owner; +use crate::dispatch::submit::submit_client_request_on_owner; use crate::http::admission::admit_partition_write; use crate::http::error::{PartitionWriteError, WriteError}; use crate::http::reply::{ diff --git a/core/server/src/lib.rs b/core/server/src/lib.rs index 46ce02fb79..6f0c5ad296 100644 --- a/core/server/src/lib.rs +++ b/core/server/src/lib.rs @@ -43,10 +43,8 @@ pub(crate) mod shard_allocator; pub mod systemd; // spine: the request path - shell vocabulary, dispatch funnel, per-domain ops. -pub(crate) mod auth; pub(crate) mod consumer_group; pub(crate) mod dispatch; -pub(crate) mod login_register; pub(crate) mod pat; pub(crate) mod responses; pub mod session_manager; diff --git a/core/server/src/partition_reconciler.rs b/core/server/src/partition_reconciler.rs index d0216aea7d..0a7fd5e625 100644 --- a/core/server/src/partition_reconciler.rs +++ b/core/server/src/partition_reconciler.rs @@ -67,7 +67,7 @@ //! `shards_table` is therefore a **cache of a deterministic hash**, never a //! readiness proof: every shard derives the same rows from the same committed //! metadata, and a row may exist before its partition does. Nothing may treat -//! presence as "the owner is ready" - `dispatch::wait_for_partition_routable` +//! presence as "the owner is ready" - `dispatch::partition::wait_for_partition_routable` //! documents why the owner-readiness probe that used to live there was both //! unnecessary and ineffective. //! diff --git a/core/server/tests/module_graph.rs b/core/server/tests/module_graph.rs index d9d978a56f..555eadb8cf 100644 --- a/core/server/tests/module_graph.rs +++ b/core/server/tests/module_graph.rs @@ -27,7 +27,8 @@ //! Granularity is the module FILE. Parent<->child edges are exempt: a root //! composing its children (and children reaching items the root defines) is //! the pattern working as intended. Sibling and cross-tree cycles are the -//! rot this guard exists to stop. +//! rot this guard exists to stop. `#[cfg(test)]` modules are outside the +//! graph whether they are inline or their own file. //! //! `WHITELIST` carries the known survivors. Each entry must still be a live //! cycle - a stale entry fails the test, so the list can only shrink. @@ -37,9 +38,9 @@ use quote::ToTokens; use std::collections::{BTreeMap, BTreeSet}; use std::path::{Path, PathBuf}; -/// Known mutual edges, as unordered pairs of module paths. Burned down per -/// refactor PR; the auth<->dispatch cycle dies with the dispatch/ login merge. -const WHITELIST: [(&str, &str); 1] = [("auth", "dispatch")]; +/// Known mutual edges, as unordered pairs of module paths. Burned down to +/// empty by the dispatch/ login merge; new entries need a written ruling. +const WHITELIST: [(&str, &str); 0] = []; type Module = Vec; @@ -104,11 +105,15 @@ fn module_graph_is_a_dag_modulo_whitelist() { /// Map every source file under `src/` to its module path. `lib.rs` is the /// crate root `[]`; `main.rs`/`args.rs` belong to the bin target and are /// skipped (their `crate::` is a different crate). +/// +/// Modules a parent declares under `#[cfg(test)]` are dropped with their whole +/// subtree: `scan_items` already skips inline `#[cfg(test)] mod`, and a test +/// harness in its own file must not become a production graph node. fn collect_modules(src: &Path) -> Vec<(Module, PathBuf)> { let mut files = Vec::new(); walk(src, &mut files); files.sort(); - files + let mut modules: Vec<(Module, PathBuf)> = files .into_iter() .filter_map(|file| { let relative = file @@ -126,7 +131,35 @@ fn collect_modules(src: &Path) -> Vec<(Module, PathBuf)> { } Some((module, file)) }) - .collect() + .collect(); + + let gated = cfg_test_file_modules(&modules); + modules.retain(|(module, _)| { + !gated.contains(module) && !gated.iter().any(|root| is_ancestor(root, module)) + }); + modules +} + +/// Module paths a parent declares as `#[cfg(test)] mod name;` (no inline body). +fn cfg_test_file_modules(modules: &[(Module, PathBuf)]) -> BTreeSet { + let mut gated = BTreeSet::new(); + for (module, file) in modules { + let source = std::fs::read_to_string(file) + .unwrap_or_else(|error| panic!("cannot read {}: {error}", file.display())); + let ast = syn::parse_file(&source) + .unwrap_or_else(|error| panic!("cannot parse {}: {error}", file.display())); + for item in &ast.items { + if let syn::Item::Mod(declaration) = item + && declaration.content.is_none() + && is_cfg_test(&declaration.attrs) + { + let mut child = module.clone(); + child.push(declaration.ident.to_string()); + gated.insert(child); + } + } + } + gated } fn walk(dir: &Path, files: &mut Vec) { From 4dd877267f9d07c85dd0b42aabba39d645e305f1 Mon Sep 17 00:00:00 2001 From: haubur Date: Tue, 1 Sep 2026 22:26:33 +0200 Subject: [PATCH 035/182] chore(sdk): IggyProducer docs for Rust SDK (#3989) Co-authored-by: Piotr Gankiewicz Co-authored-by: Hubert Gruszecki --- core/sdk/src/clients/producer.rs | 439 +++++++++++++++++- core/sdk/src/clients/producer_builder.rs | 2 +- core/sdk/src/clients/producer_config.rs | 236 +++++++--- core/sdk/src/clients/producer_dispatcher.rs | 434 ++++++++++++++++- .../src/clients/producer_error_callback.rs | 139 +++++- core/sdk/src/clients/producer_sharding.rs | 131 +++++- 6 files changed, 1274 insertions(+), 107 deletions(-) diff --git a/core/sdk/src/clients/producer.rs b/core/sdk/src/clients/producer.rs index a6c3e57918..23b945c6a3 100644 --- a/core/sdk/src/clients/producer.rs +++ b/core/sdk/src/clients/producer.rs @@ -474,6 +474,321 @@ impl ProducerCoreBackend for ProducerCore { unsafe impl Send for IggyProducer {} unsafe impl Sync for IggyProducer {} +/// Appends messages to one topic of one stream. +/// +/// A topic is split into partitions, and a partition is an ordered log that producers append to. +/// `IggyProducer` lets you configure [where](#where-messages-land-and-ordering) and +/// [how](#how-messages-are-sent) messages are sent. +/// +/// # Creating a producer +/// +/// The easiest way to create a producer is through an [`IggyClient`] with a configured connection. +/// [`IggyClient::producer()`] returns an [`IggyProducerBuilder`] that uses that client's connection. +/// +/// Building never talks to the server. [`init()`](Self::init) must be awaited before the first send. +/// +/// # Examples +/// +/// A producer with the defaults, sending one batch and reading the confirmations: +/// +/// ```rust,no_run +/// use iggy::prelude::*; +/// use std::str::FromStr; +/// +/// # async fn example() -> Result<(), IggyError> { +/// let client = IggyClient::from_connection_string("iggy://iggy:iggy@localhost:8090")?; +/// client.connect().await?; +/// +/// let producer = client.producer("my-stream", "my-topic")?.build(); +/// producer.init().await?; +/// +/// let messages = vec![IggyMessage::from_str("hello")?, IggyMessage::from_str("world")?]; +/// let response = producer.send(messages).await?; +/// for confirmation in &response.confirmations { +/// println!( +/// "Partition: {}, base offset: {}", +/// confirmation.partition_id, confirmation.base_offset +/// ); +/// } +/// # Ok(()) +/// # } +/// ``` +/// +/// A `background` producer, which queues a batch and sends it later, and shuts down cleanly: +/// +/// ```rust,no_run +/// use iggy::prelude::*; +/// use std::str::FromStr; +/// +/// # async fn example() -> Result<(), IggyError> { +/// let client = IggyClient::from_connection_string("iggy://iggy:iggy@localhost:8090")?; +/// client.connect().await?; +/// +/// let producer = client +/// .producer("my-stream", "my-topic")? +/// .background( +/// BackgroundConfig::builder() +/// .linger_time(IggyDuration::new_from_secs(1)) +/// .batch_size(64 * 1024) +/// .build(), +/// ) +/// .build(); +/// producer.init().await?; +/// +/// // Returns once the batch is queued. The send itself happens on a background worker. +/// producer.send_one(IggyMessage::from_str("hello")?).await?; +/// +/// // Without graceful shutdown, buffered messages have no completion guarantee and may be lost. +/// producer.shutdown().await; +/// # Ok(()) +/// # } +/// ``` +/// +/// Keying messages to a partition and separating confirmed chunks from the unconfirmed tail: +/// +/// ```rust,no_run +/// use iggy::prelude::*; +/// use std::str::FromStr; +/// +/// # async fn example() -> Result<(), IggyError> { +/// let client = IggyClient::from_connection_string("iggy://iggy:iggy@localhost:8090")?; +/// client.connect().await?; +/// +/// let producer = client +/// .producer("my-stream", "my-topic")? +/// // Every message of this producer goes to the partition the server derives from the key. +/// .partitioning(Partitioning::messages_key_str("my-key")?) +/// .send_retries(Some(5), Some(NonZeroIggyDuration::ONE_SECOND)) +/// .build(); +/// producer.init().await?; +/// +/// let messages = vec![IggyMessage::from_str("hello")?]; +/// if let Err(IggyError::ProducerSendFailed { cause, failed, committed, .. }) = +/// producer.send(messages).await +/// { +/// // `committed` holds confirmations for earlier chunks, `failed` the unconfirmed tail. See +/// // "Retrying and what a failure means" before resending `failed`. +/// eprintln!("{} messages have no usable confirmation: {cause}", failed.len()); +/// eprintln!("{} chunk(s) were confirmed before the failure", committed.len()); +/// } +/// # Ok(()) +/// # } +/// ``` +/// +/// # How messages are sent +/// +/// There are two options: [`direct()`] and [`background()`]. You will pick one of the two modes, +/// and a producer stays in that mode for its lifetime. +/// +/// A **direct** producer (the default) sends from the calling task. [`send()`](Self::send) awaits +/// the server and returns its [confirmations](#confirmations). A batch longer than +/// [`DirectConfig::batch_length`] is split into that many messages per request, and the requests are +/// awaited one after another, so a failure in the middle leaves confirmations for the successful +/// prefix. [`DirectConfig::linger_time`] requests a minimum gap between sequential send calls. It +/// does not space out chunks within one call and does not serialize concurrent callers. +/// +/// A **background** producer hands the batch to a [`ProducerDispatcher`] and returns after queueing +/// it without waiting for the write. A worker may already have started or completed the write by +/// then, but [`send()`](Self::send) reports no write result and returns no confirmations. The +/// dispatcher runs [`BackgroundConfig::num_shards`] workers, each buffering the batches routed to it +/// and flushing them when one of three limits is hit: [`BackgroundConfig::batch_length`] queued +/// sends, [`BackgroundConfig::batch_size`] bytes, or [`BackgroundConfig::linger_time`] since the +/// first of them was buffered. Adjacent buffered sends that share a stream, topic, and partitioning +/// are merged into one request. +/// +/// The sharding strategy decides how background dispatch affects message order: +/// - [`BackgroundConfig::sharding`] routes a batch to a worker. The default [`OrderedSharding`] picks +/// it from the stream and the topic, so everything going to one topic stays on one worker and keeps +/// its order. [`BalancedSharding`] spreads batches round-robin and gives up ordering, which can +/// improve throughput when multiple shards are allowed to write concurrently. +/// - [`BackgroundConfig::max_in_flight`] bounds concurrent writes across workers. A worker itself +/// remains sequential for every value and awaits retries before starting its next request. Raising +/// this setting does not break the per-topic order provided by [`OrderedSharding`], but it allows +/// strategies that spread one topic across workers to write those shards concurrently. +/// +/// Since a background send is queued rather than written, the dispatcher charges queued and +/// in-flight sends against [`BackgroundConfig::max_buffer_size`] over all workers. The default is +/// bounded, while a value of zero disables the byte budget. +/// [`BackgroundConfig::failure_mode`] decides what a send does when that budget is exhausted. You can block +/// until it frees up (the default), block with a timeout and fail with +/// [`IggyError::BackgroundSendTimeout`], or fail right away with +/// [`IggyError::BackgroundSendBufferOverflow`]. A single send larger than the whole budget always +/// fails with [`IggyError::BackgroundSendBufferOverflow`], whatever the mode. +/// +/// In contrast to the direct mode, which awaits the confirmations from the server, a background send +/// has no caller left to return to. Write failures are reported to +/// [`BackgroundConfig::error_callback`] instead. It receives the cause, the unconfirmed tail, and +/// confirmations returned for earlier chunks, see +/// [Retrying and what a failure means](#retrying-and-what-a-failure-means). The default callback +/// logs the context and drops it. A custom [`ErrorCallback`] can retain or persist failed sends +/// according to the application's at-least-once policy. +/// +/// # Where messages land and ordering +/// +/// The partition is the unit of order in Iggy. Inside one partition, messages receive increasing +/// offsets in the server's append order. Between partitions there is no global order, so a consumer +/// reading several partitions can observe their messages interleaved. +/// +/// Two messages therefore stay in order only if both of these hold: +/// 1. they are appended to the **same partition**, and +/// 2. their requests reach the server **one after another**, rather than at the same time. +/// +/// Point 1 is what the partitioning strategy and point 2 is what the send mode and the client's own +/// concurrency decide. +/// +/// | Setting | Order | What happens | +/// | --- | --- | --- | +/// | [`Partitioning::balanced()`], the default | not guaranteed with multiple partitions | the server chooses a partition per request, so consecutive sends and chunks may land in different logs | +/// | [`Partitioning::partition_id()`], [`Partitioning::messages_key()`] with a stable key, [`partitioner()`] returning a stable id | same partition | every batch is routed to the same log, satisfying the first ordering requirement | +/// | one task, awaiting each [`send()`](Self::send) before the next | sequential | requests reach a partition in call order | +/// | several tasks sharing the producer, or overlapping sends | not guaranteed | requests race, so call order does not determine append order | +/// | `direct` producer | sequential within one call | the calling task writes that call's chunks one after another | +/// | `background` producer with [`OrderedSharding`], the default | sequential per stream/topic pair | one pair is bound to one worker, which writes its queue in order | +/// | `background` producer with [`BalancedSharding`] and multiple shards | not guaranteed | consecutive batches can reach different workers, so a later batch may be written first | +/// | [`BackgroundConfig::max_in_flight`] | depends on sharding | it controls concurrency between workers, while each worker remains sequential | +/// +/// In short, per-partition order survives if you name the partition and let one sequential writer +/// write to it, for example keyed or fixed partitioning from one task, or a background producer with +/// ordered sharding. [`Partitioning::messages_key()`] maps the same key consistently to a partition +/// for a fixed topic layout. Different keys can share a partition, so this does not create a +/// separate physical log per key. +/// +/// One thing this does not protect against is a message appearing twice. Delivery is at-least-once, +/// so a retried batch can be appended a second time at a higher offset, see +/// [Retrying and what a failure means](#retrying-and-what-a-failure-means). +/// +/// # Confirmations +/// +/// A successful direct send returns [`SendMessagesResponse`], normally holding one +/// [`SendMessagesConfirmationResponse`] per chunk. Each confirmation records a partition and the +/// `base_offset` assigned to the first message in that chunk. A legacy server may return no +/// confirmation payload, and a background producer always returns an empty confirmation list. +/// +/// A confirmation means the server committed the batch in memory. It does not mean the batch was +/// fsynced. After a crash and restart, a later batch can receive an offset that a client recorded +/// before the crash. Delivery is also at least once, so a retry can commit the same messages at +/// another offset. +/// +/// # Retrying and what a failure means +/// +/// A retry is the same request sent again unchanged, meaning a producer never rewrites, splits, or +/// reorders a batch to retry a send. [`send_retries()`] sets how many times it may try and how +/// later retries are paced. The default allows three retries and configures a one-second interval. +/// The first retry is immediate. Later retries wait for the next tick of an interval timer, so they +/// are at most one interval apart, and an attempt that outlasts the interval is followed by the next +/// one right away. Passing `None` as the interval retries back-to-back without any delay. The retry +/// policy applies to both direct and background producers. +/// +/// A request can fail after the server has already appended it, so the first request of an +/// unconfirmed tail may already be in the partition. Sending the tail again can therefore write the +/// same messages twice, leaving the batch in the partition at multiple offsets. A consumer that +/// cannot accept duplicates must recognize them itself. +/// +/// What is retried, and what is not: +/// +/// | Situation | Retried | What the caller ends up with | +/// | --- | --- | --- | +/// | the client is not signed in, so nothing may be sent yet | yes | the send goes ahead as soon as the client is signed in, or fails with [`IggyError::CannotSendMessagesDueToClientDisconnection`] once the retry budget is spent | +/// | the request failed without an indication that it committed | yes, the identical request is sent again | the confirmation of the attempt that finally succeeds, or the last error once the retry budget is spent | +/// | the batch committed, but its confirmation could not be read ([`IggyError::InvalidBytesResponse`] or [`IggyError::InvalidJsonResponse`], raised on the HTTP transport only) | no | the write did happen and retrying would duplicate it on purpose | +/// | encrypting or partitioning the batch failed | no | that cause, with the whole batch returned as unconfirmed because nothing was sent, although earlier messages may already have been encrypted in place | +/// | [`send_retries()`] passed `None` or `0` | no | the outcome of the single attempt, which skips the sign-in check above and fails with the transport error when the client is disconnected | +/// +/// To be precise: the budget is spent per request. Every chunk of a split is a request that gets its own retry budget. +/// Also, waiting for the client to sign in is counted separately from retrying the write itself. +/// +/// Where a failure surfaces is the main difference between the two modes. Either way it names the +/// same three pieces: the `cause`, the unconfirmed tail, and confirmations returned for earlier +/// chunks. +/// +/// - A **direct** send returns them to the caller as [`IggyError::ProducerSendFailed`]. `committed` +/// contains confirmations returned for earlier chunks, while `failed` is the unconfirmed tail. +/// Inspect `cause` before resending it. Encryption mutates messages before sending or +/// partitioning, so `failed` contains encrypted messages when an encryptor is configured. Passing +/// them back to the same producer would encrypt them again. +/// - A **background** send returned to its caller long before the write, so they go to +/// [`BackgroundConfig::error_callback`] as an +/// [`ErrorCtx`](crate::clients::producer_error_callback::ErrorCtx) instead, once no further +/// automatic retry will be attempted. The default callback logs the failure and drops the context. +/// Implement [`ErrorCallback`] when failed sends must be retained. +/// +/// # Options and defaults +/// +/// Everything is configured on the [`IggyProducerBuilder`] before [`build()`] and is fixed +/// afterwards. +/// +/// | Option | Default | Controls | +/// | --- | --- | --- | +/// | [`stream()`], [`topic()`] | the values passed to [`IggyClient::producer()`] | where messages are appended | +/// | [`direct()`] / [`background()`] | [`direct()`] with the [`DirectConfig`] defaults, 1000 messages per request and no linger time | whether a send waits for the write | +/// | [`partitioning()`] | [`Partitioning::balanced()`] | which partition a batch lands in | +/// | [`partitioner()`] | none | computing the partition on the client instead | +/// | [`send_retries()`] | three retries, one-second interval after the immediate first retry | retrying a failed request | +/// | [`create_stream_if_not_exists()`] | on | creating the stream during [`init()`](Self::init) | +/// | [`create_topic_if_not_exists()`] | on, one partition, server defaults for expiry and max size | creating the topic during [`init()`](Self::init) | +/// | [`encryptor()`] | inherited from the client | encrypting payloads and user headers | +/// +/// There are inverse setters as well, such as [`without_partitioning()`], +/// [`without_partitioner()`], [`without_encryptor()`], [`do_not_create_stream_if_not_exists()`] and +/// [`do_not_create_topic_if_not_exists()`]. +/// +/// # Encryption +/// +/// When the [`IggyClient`] was created with an encryptor, a producer built from that client inherits +/// it and encrypts payloads and user headers before a batch leaves the producer. A consumer must use +/// a matching key to decrypt them. Producers and consumers built from the same client inherit the +/// same encryptor unless either builder overrides or clears it. An encryption failure fails the +/// whole send before any request leaves the producer. Encryption runs before a custom +/// [`partitioner()`], so that partitioner observes the encrypted payload and user headers. +/// +/// # Concurrency +/// +/// `IggyProducer` is `Send` and `Sync` but not `Clone`, and every send method takes `&self`. An +/// `Arc` can therefore be shared across tasks without further wrapping. A background +/// dispatcher routes those calls into worker queues and preserves order only within each worker. +/// +/// Concurrent direct sends are independent requests, so nothing orders them against each other. The +/// chunk order described above holds within one call to [`send()`](Self::send) only. +/// +/// # Shutting down +/// +/// Call [`shutdown()`](Self::shutdown) when production is complete. It takes the producer by value, +/// drains a background producer's queues, flushes its remaining buffers, and waits for the worker +/// and error tasks. Dropping a background producer provides no completion guarantee and can lose +/// buffered messages. A direct producer has nothing buffered, so shutdown is a no-op. +/// +/// [`IggyClient`]: crate::clients::client::IggyClient +/// [`IggyClient::producer()`]: crate::clients::client::IggyClient::producer +/// [`IggyProducerBuilder`]: crate::clients::producer_builder::IggyProducerBuilder +/// [`BackgroundConfig`]: crate::clients::producer_config::BackgroundConfig +/// [`BackgroundConfig::batch_length`]: crate::clients::producer_config::BackgroundConfig::batch_length +/// [`BackgroundConfig::batch_size`]: crate::clients::producer_config::BackgroundConfig::batch_size +/// [`BackgroundConfig::error_callback`]: crate::clients::producer_config::BackgroundConfig::error_callback +/// [`BackgroundConfig::failure_mode`]: crate::clients::producer_config::BackgroundConfig::failure_mode +/// [`BackgroundConfig::linger_time`]: crate::clients::producer_config::BackgroundConfig::linger_time +/// [`BackgroundConfig::max_buffer_size`]: crate::clients::producer_config::BackgroundConfig::max_buffer_size +/// [`BackgroundConfig::max_in_flight`]: crate::clients::producer_config::BackgroundConfig::max_in_flight +/// [`BackgroundConfig::num_shards`]: crate::clients::producer_config::BackgroundConfig::num_shards +/// [`BackgroundConfig::sharding`]: crate::clients::producer_config::BackgroundConfig::sharding +/// [`BalancedSharding`]: crate::clients::producer_sharding::BalancedSharding +/// [`ErrorCallback`]: crate::clients::producer_error_callback::ErrorCallback +/// [`OrderedSharding`]: crate::clients::producer_sharding::OrderedSharding +/// [`background()`]: crate::clients::producer_builder::IggyProducerBuilder::background +/// [`build()`]: crate::clients::producer_builder::IggyProducerBuilder::build +/// [`create_stream_if_not_exists()`]: crate::clients::producer_builder::IggyProducerBuilder::create_stream_if_not_exists +/// [`create_topic_if_not_exists()`]: crate::clients::producer_builder::IggyProducerBuilder::create_topic_if_not_exists +/// [`direct()`]: crate::clients::producer_builder::IggyProducerBuilder::direct +/// [`do_not_create_stream_if_not_exists()`]: crate::clients::producer_builder::IggyProducerBuilder::do_not_create_stream_if_not_exists +/// [`do_not_create_topic_if_not_exists()`]: crate::clients::producer_builder::IggyProducerBuilder::do_not_create_topic_if_not_exists +/// [`encryptor()`]: crate::clients::producer_builder::IggyProducerBuilder::encryptor +/// [`partitioner()`]: crate::clients::producer_builder::IggyProducerBuilder::partitioner +/// [`partitioning()`]: crate::clients::producer_builder::IggyProducerBuilder::partitioning +/// [`send_retries()`]: crate::clients::producer_builder::IggyProducerBuilder::send_retries +/// [`stream()`]: crate::clients::producer_builder::IggyProducerBuilder::stream +/// [`topic()`]: crate::clients::producer_builder::IggyProducerBuilder::topic +/// [`without_encryptor()`]: crate::clients::producer_builder::IggyProducerBuilder::without_encryptor +/// [`without_partitioner()`]: crate::clients::producer_builder::IggyProducerBuilder::without_partitioner +/// [`without_partitioning()`]: crate::clients::producer_builder::IggyProducerBuilder::without_partitioning pub struct IggyProducer { core: Arc, dispatcher: Option, @@ -532,37 +847,100 @@ impl IggyProducer { Self { core, dispatcher } } + /// Returns the identifier of the stream this producer appends to. pub fn stream(&self) -> &Identifier { &self.core.stream_id } + /// Returns the identifier of the topic this producer appends to. pub fn topic(&self) -> &Identifier { &self.core.topic_id } - /// Initializes the producer by subscribing to diagnostic events, creating the stream and topic if they do not exist etc. + /// Initializes the producer and makes it ready to send messages. + /// + /// This must be called before the first send. Calling it again after successful initialization + /// does nothing and returns immediately. + /// + /// Initialization ensures that: + /// - The producer subscribes to client [`DiagnosticEvent`] values and tracks whether sending is + /// currently allowed. The gate starts open and follows the events observed after this call: + /// connect, disconnect, sign-out, and shutdown close it, and only a sign-in reopens it. It is + /// checked only when + /// [`send_retries()`](crate::clients::producer_builder::IggyProducerBuilder::send_retries) + /// allows at least one retry. + /// - the stream exists, creating it when `create_stream_if_not_exists` is set (the default). + /// - the topic exists, creating it when `create_topic_if_not_exists` is set (the default), with + /// the partitions count, message expiry and max size passed to + /// [`IggyProducerBuilder::create_topic_if_not_exists`](crate::clients::producer_builder::IggyProducerBuilder::create_topic_if_not_exists). + /// These are the only topic options controlled by producer initialization. All other settings + /// come from [`TopicCreateOptions::default`]. /// - /// Note: This method must be invoked before producing messages. + /// # Errors + /// + /// - [`IggyError::StreamNameNotFound`] or [`IggyError::TopicNameNotFound`] when the stream or + /// the topic does not exist and its auto creation is disabled. + /// - Any other error the server raised while looking up or creating the stream or the topic. pub async fn init(&self) -> Result<(), IggyError> { self.core.init().await } - /// Sends `messages` and returns the commit confirmations of every chunk the - /// send was split into, concatenated in chunk order. A retried chunk - /// contributes only the confirmation of the attempt that finally succeeded. + /// Sends `messages` to the stream and topic this producer was built for, with the partitioning it + /// was built with. + /// + /// What a returned `Ok` tells you depends on the send mode: + /// + /// | | `direct` producer | `background` producer | + /// | --- | --- | --- | + /// | the call returns | once the server has answered every request the batch was split into | once the batch is queued on a worker | + /// | `Ok` means | the server returned success for every request | the messages were accepted into a worker queue, nothing more | + /// | confirmations | normally one per request, in order, but legacy servers may return none | always empty | + /// | a write that fails | comes back as [`IggyError::ProducerSendFailed`] | goes to [`error_callback`] later | + /// + /// An empty `messages` vector is a no-op in both modes and returns an empty confirmation list. + /// + /// # Confirmations + /// + /// A [`SendMessagesConfirmationResponse`] names the partition a chunk of the batch landed in and + /// the `base_offset` its first message was given. + /// An offset is a position, not an identity. Delivery is at-least-once, so an earlier retry may + /// have committed the same messages at a lower offset, see + /// [Retrying and what a failure means](IggyProducer#retrying-and-what-a-failure-means). + /// A confirmation reports an in-memory commit, not an fsync. A crash and restart can therefore + /// lose an acknowledged batch and later reuse an offset the client already observed. + /// + /// # How long the call takes /// - /// Delivery is at-least-once. An earlier retry may already have committed - /// the same messages at a lower offset, so `base_offset` never implies - /// uniqueness. + /// Both modes can wait: /// - /// A batch is confirmed once it is committed in memory, not once it is - /// fsynced. A crash-restart can stamp a later batch with an offset a client - /// has already recorded. + /// - A `direct` send first waits out whatever is left of [`DirectConfig::linger_time`] since the + /// previous send, then awaits its requests one after another, so it returns no earlier than the + /// last one is written. + /// - A `background` send waits when the dispatcher has no room for the batch, which + /// [`failure_mode`] configures. The default [`BackpressureMode::Block`] waits for as long as it + /// takes. It can wait on the queue of the worker it was routed to as well, which holds 256 + /// queued sends. /// - /// The confirmation list is empty whenever the server sends no confirmation - /// payload, as the legacy server never sends one, and for a `background` - /// producer, which hands the messages to a dispatcher and returns before the - /// send happens. Branch on `confirmations.is_empty()` instead of indexing. + /// # Errors + /// + /// A `direct` producer wraps a send failure in [`IggyError::ProducerSendFailed`]. The cause can be + /// [`IggyError::CannotSendMessagesDueToClientDisconnection`] after the readiness retry budget is + /// exhausted, an encryption or partitioning failure raised before anything left the producer, + /// a server or transport error after write retries, or an unreadable confirmation for a request + /// that may already have committed. + /// + /// A `background` producer only reports what queueing the batch ran into: + /// [`IggyError::ProducerClosed`] after [`shutdown()`](Self::shutdown), + /// [`IggyError::BackgroundSendBufferOverflow`] or [`IggyError::BackgroundSendTimeout`] under + /// back pressure, and [`IggyError::BackgroundSendError`] when a worker is gone. A batch larger + /// than [`max_buffer_size`] always fails with [`IggyError::BackgroundSendBufferOverflow`], + /// however idle the producer is. Failures of the write itself reach [`error_callback`] instead, + /// which drops the messages unless it is implemented to keep them. + /// + /// [`BackpressureMode::Block`]: crate::clients::producer_config::BackpressureMode::Block + /// [`error_callback`]: crate::clients::producer_config::BackgroundConfig::error_callback + /// [`failure_mode`]: crate::clients::producer_config::BackgroundConfig::failure_mode + /// [`max_buffer_size`]: crate::clients::producer_config::BackgroundConfig::max_buffer_size pub async fn send( &self, messages: Vec, @@ -588,12 +966,20 @@ impl IggyProducer { } } - /// See [`IggyProducer::send`] for the confirmation semantics. + /// Sends one message. + /// + /// This has the same mode-dependent confirmation, backpressure, retry, and error semantics as + /// [`IggyProducer::send`]. pub async fn send_one(&self, message: IggyMessage) -> Result { self.send(vec![message]).await } - /// See [`IggyProducer::send`] for the confirmation semantics. + /// Sends `messages` to the partition `partitioning` selects, overriding the partitioning this + /// producer was built with for this call only. `None` falls back to that configured + /// partitioning. + /// + /// A [`partitioner()`](crate::clients::producer_builder::IggyProducerBuilder::partitioner) still + /// wins over the argument, since it computes the partition from the messages themselves. pub async fn send_with_partitioning( &self, messages: Vec, @@ -620,7 +1006,13 @@ impl IggyProducer { } } - /// See [`IggyProducer::send`] for the confirmation semantics. + /// Sends `messages` to any stream and topic, not only the pair this producer was built for. + /// + /// The target has to exist, since [`init()`](Self::init) only creates the producer's own stream + /// and topic. Everything else stays in force: encryption, partitioning, retries, and for a + /// `background` producer the routing to a worker [`Shard`]. + /// + /// [`Shard`]: crate::clients::producer_sharding::Shard pub async fn send_to( &self, stream: Arc, @@ -646,10 +1038,15 @@ impl IggyProducer { } } - /// Flushes buffered messages in `background` mode before returning. A - /// `direct`-mode producer has nothing to flush. Dropping the producer - /// instead of calling this silently discards unflushed `background` + /// Shuts the producer down. + /// + /// For a background producer, this drains the dispatcher queues, flushes the remaining shard + /// buffers, waits for writes and error callbacks to finish, and then returns. Stop every sender + /// first: a send racing the shutdown can be queued after the drain and is lost without an error. + /// Dropping a background producer instead provides no such guarantee and can lose buffered /// messages. + /// + /// A direct producer has nothing buffered, so calling `shutdown()` is a no-op. pub async fn shutdown(self) { if let Some(dispatcher) = self.dispatcher { dispatcher.shutdown().await; diff --git a/core/sdk/src/clients/producer_builder.rs b/core/sdk/src/clients/producer_builder.rs index 0d05adbcb5..97e575ab66 100644 --- a/core/sdk/src/clients/producer_builder.rs +++ b/core/sdk/src/clients/producer_builder.rs @@ -91,7 +91,7 @@ impl IggyProducerBuilder { Self { stream, ..self } } - /// Sets the stream name. + /// Sets the topic identifier. pub fn topic(self, topic: Identifier) -> Self { Self { topic, ..self } } diff --git a/core/sdk/src/clients/producer_config.rs b/core/sdk/src/clients/producer_config.rs index f2d5948527..6da8553c27 100644 --- a/core/sdk/src/clients/producer_config.rs +++ b/core/sdk/src/clients/producer_config.rs @@ -22,132 +22,244 @@ use bon::Builder; use iggy_common::{IggyByteSize, IggyDuration}; use std::sync::Arc; -/// Determines how the `send_messages` API should behave when problem is encountered +/// What a background send does when the dispatcher's byte budget, +/// [`BackgroundConfig::max_buffer_size`], is exhausted. +/// +/// Set it through [`BackgroundConfig::failure_mode`]. These modes govern waiting for buffer capacity, +/// not retries of a server write. A single batch larger than the whole budget fails with +/// [`IggyError::BackgroundSendBufferOverflow`] under all of them. +/// +/// [`IggyError::BackgroundSendBufferOverflow`]: iggy_common::IggyError::BackgroundSendBufferOverflow #[derive(Debug, Clone)] pub enum BackpressureMode { - /// Block until the send succeeds + /// Waits until enough byte-budget capacity is released (default). + /// + /// This wait has no timeout. It can last indefinitely if queued or in-flight writes do not + /// complete and release their permits. Block, - /// Block with a timeout, after which the send fails + /// Waits for the given duration, then fails the send with + /// [`IggyError::BackgroundSendTimeout`](iggy_common::IggyError::BackgroundSendTimeout). BlockWithTimeout(IggyDuration), - /// Fail immediately without retrying + /// Gives up at once with + /// [`IggyError::BackgroundSendBufferOverflow`](iggy_common::IggyError::BackgroundSendBufferOverflow), + /// leaving the batch unqueued. FailImmediately, } -// Configuration for the *background* (asynchronous) producer +/// Configuration for a producer that sends messages in the background. +/// +/// A background producer passes every non-empty send to a [`ProducerDispatcher`]. The dispatcher +/// returns once the batch is queued, and one of its worker [`Shard`]s writes the batch later. This type controls +/// how many workers exist, which worker receives a batch, when workers flush, how many bytes may be +/// queued or in flight, and where write failures are reported. +/// +/// # Defaults +/// +/// | Field | Default | Controls | +/// | --- | --- | --- | +/// | [`num_shards`](Self::num_shards) | 1 | how many worker queues exist | +/// | [`sharding`](Self::sharding) | [`OrderedSharding`] | which worker a batch is queued on | +/// | [`batch_size`](Self::batch_size) | 1 MiB | flush once a worker holds this many bytes | +/// | [`batch_length`](Self::batch_length) | 1000 | flush once a worker holds this many queued sends | +/// | [`linger_time`](Self::linger_time) | 1 ms | how long a worker holds a non-empty buffer before flushing it | +/// | [`max_buffer_size`](Self::max_buffer_size) | 32 MiB | bytes the whole producer may hold | +/// | [`failure_mode`](Self::failure_mode) | [`BackpressureMode::Block`] | what a send does when that budget is full | +/// | [`max_in_flight`](Self::max_in_flight) | 1 | requests being written at once | +/// | [`error_callback`](Self::error_callback) | [`LogErrorCallback`] | what happens to a failed write | +/// +/// The default [`OrderedSharding`] strategy preserves dispatch order for each stream/topic pair by +/// routing that pair to one worker. That worker awaits each request, including its retries, before +/// starting the next. [`BalancedSharding`] can route consecutive batches for one destination to +/// different workers and therefore gives up that ordering. [`max_in_flight`](Self::max_in_flight) +/// only controls how many workers may write concurrently. It does not make a single worker process +/// more than one request at a time. +/// +/// # Zero values +/// +/// `0` disables the [`batch_size`](Self::batch_size) and +/// [`batch_length`](Self::batch_length) flush thresholds. A zero +/// [`linger_time`](Self::linger_time) flushes as soon as the worker picks up a send. A zero +/// [`max_buffer_size`](Self::max_buffer_size) is treated as unbounded. A zero +/// [`max_in_flight`](Self::max_in_flight) uses `Semaphore::MAX_PERMITS`, and a zero +/// [`num_shards`](Self::num_shards) is read as one worker. +/// /// # Examples /// /// ``` +/// use iggy::clients::producer_config::BackpressureMode; /// use iggy::prelude::*; -/// use iggy_common::{IggyDuration, IggyByteSize}; -/// -/// // Use default config -/// let config = BackgroundConfig::builder() -/// .build(); +/// use std::time::Duration; /// -/// // Set custom batch size and disable length limit -/// let config = BackgroundConfig::builder() -/// .batch_size(256 * 1024) // 256 KiB -/// .batch_length(0) // unlimited -/// .build(); +/// // Ordered and bounded, as described above. +/// let ordered = BackgroundConfig::builder().build(); /// -/// // Configure low-latency flush -/// let config = BackgroundConfig::builder() -/// .linger_time(IggyDuration::from(200)) // 200ms +/// // Throughput: four workers share one topic, flushing at 4 MiB, with a 256 MiB byte budget. +/// // Up to 4 requests can be in flight at the same time, one per worker. +/// let fast = BackgroundConfig::builder() +/// .num_shards(4) +/// .sharding(Box::new(BalancedSharding::default())) +/// .batch_size(4 * 1024 * 1024) +/// .max_buffer_size(IggyByteSize::from(256 * 1024 * 1024)) +/// .max_in_flight(4) /// .build(); /// -/// // Disable all limits (not recommended for production) -/// let config = BackgroundConfig::builder() -/// .batch_size(0) -/// .batch_length(0) -/// .max_buffer_size(IggyByteSize::from(0)) -/// .max_in_flight(0) +/// // Latency: flush within 5 ms or after 50 queued sends. Do not wait for byte-budget capacity. +/// // The bounded per-worker channel can still make dispatch wait when its 256 slots are occupied. +/// let responsive = BackgroundConfig::builder() +/// .linger_time(IggyDuration::new(Duration::from_millis(5))) +/// .batch_length(50) +/// .failure_mode(BackpressureMode::FailImmediately) /// .build(); /// ``` +/// +/// [`BalancedSharding`]: crate::clients::producer_sharding::BalancedSharding +/// [`ProducerDispatcher`]: crate::clients::producer_dispatcher::ProducerDispatcher +/// [`Shard`]: crate::clients::producer_sharding::Shard #[derive(Debug, Builder)] pub struct BackgroundConfig { - /// Number of shard-workers that run in parallel. + /// Number of worker [`Shard`]s the dispatcher runs, each with a queue of its own. /// - /// With the default `OrderedSharding` strategy, messages to the same - /// stream/topic are always routed to the same shard, preserving ordering. - /// Increasing shards improves throughput only when sending to multiple streams/topics. + /// Every batch is routed to exactly one worker by [`sharding`](Self::sharding), and each worker + /// buffers and writes independently. /// - /// With `BalancedSharding`, messages are distributed round-robin across all shards - /// for maximum single-destination throughput, but ordering is **not** preserved. + /// `0` is read as one worker. + /// + /// [`Shard`]: crate::clients::producer_sharding::Shard #[builder(default = 1)] pub num_shards: usize, - /// How long a shard may wait before flushing an *incomplete* batch. + /// Upper bound on how long a worker holds a non-empty buffer before flushing it. + /// + /// The window starts when a send enters an empty buffer, so an idle worker does not wake up. + /// A worker flushes as soon as any of `linger_time`, [`batch_length`](Self::batch_length) or + /// [`batch_size`](Self::batch_size) is reached. Lowering it reduces the time the write is delayed + /// at the price of smaller writes. `0` flushes as soon as the worker picks up a send. /// - /// Combines with `batch_size` / `batch_length`: whichever limit fires - /// first triggers the flush. + /// Note that [`IggyDuration::from`] reads a plain number as **microseconds**, so the default of + /// `1000` is 1 ms. #[builder(default = IggyDuration::from(1000))] pub linger_time: IggyDuration, - /// User-supplied asynchronous callback that will be executed whenever - /// the producer encounters an error it cannot automatically recover from - /// (e.g. network failure). + /// Where a background write that failed ends up. + /// + /// The dispatcher runs one task that owns this callback. A worker invokes it with an [`ErrorCtx`] + /// when its backend returns [`IggyError::ProducerSendFailed`]. Other error variants from a custom + /// backend are logged by the worker without invoking this callback. The context contains the + /// cause, destination, unconfirmed tail, and confirmations returned for earlier chunks. + /// + /// The default [`LogErrorCallback`] logs the failure and drops the messages with the context. + /// Implement [`ErrorCallback`] with your own logic to keep them. + /// + /// [`ErrorCtx`]: crate::clients::producer_error_callback::ErrorCtx + /// [`IggyError::ProducerSendFailed`]: iggy_common::IggyError::ProducerSendFailed #[builder(default = Arc::new(Box::new(LogErrorCallback)))] pub error_callback: Arc>, - /// Strategy that maps a message to a shard. + /// Picks the worker a batch is queued on, out of [`num_shards`](Self::num_shards). /// - /// Default is `OrderedSharding` which routes all messages for the same - /// stream/topic to the same shard, preserving message ordering. + /// The default [`OrderedSharding`] hashes stream and topic, so every batch for one topic queues + /// on one worker and keeps the order it was dispatched in. [`BalancedSharding`] hands them out + /// round-robin, which lets a single topic occupy all workers but gives up that order. Implement + /// [`Sharding`] to build your own logic for picking a `Shard`. /// - /// Use `BalancedSharding` for maximum throughput when ordering doesn't matter. + /// [`BalancedSharding`]: crate::clients::producer_sharding::BalancedSharding #[builder(default = Box::new(OrderedSharding))] pub sharding: Box, - /// Maximum **total size in bytes** of a batch. - /// `0` ⇒ unlimited (size-based batching disabled). + /// Flush threshold in bytes buffered on one worker. + /// + /// Sends accumulate until their reported sizes reach or exceed this value. The threshold is + /// checked after a send is added, so it is a flush trigger rather than a hard size ceiling. + /// `0` disables the threshold + /// and leaves [`batch_length`](Self::batch_length) and [`linger_time`](Self::linger_time) to + /// trigger the flush. + /// + /// This is a per-worker flush trigger. It is separate from + /// [`max_buffer_size`](Self::max_buffer_size), which caps the producer as a whole. #[builder(default = MIB)] pub batch_size: usize, - /// Maximum **number of messages** per batch. - /// `0` ⇒ unlimited (length-based batching disabled). + /// Flush threshold in number of queued batches on one worker. + /// + /// Counts the queued sends a worker holds, not the individual messages inside them. A worker + /// flushes after this many dispatches have been routed to it. `0` disables the threshold. + /// #[builder(default = 1000)] pub batch_length: usize, - /// Action to apply when back-pressure limits are reached + /// What a send does once [`max_buffer_size`](Self::max_buffer_size) is exhausted. #[builder(default = BackpressureMode::Block)] pub failure_mode: BackpressureMode, /// Upper bound for the **bytes buffered or in flight** across *all* shards. /// Bytes remain charged until the corresponding write completes. - /// `IggyByteSize::from(0)` ⇒ unlimited. + /// `IggyByteSize::from(0)` means unlimited. A nonzero value greater than + /// `Semaphore::MAX_PERMITS` makes [`ProducerDispatcher::new`] panic. + /// + /// [`ProducerDispatcher::new`]: crate::clients::producer_dispatcher::ProducerDispatcher::new #[builder(default = IggyByteSize::from(32 * MIB as u64))] pub max_buffer_size: IggyByteSize, - /// Maximum number of **in-flight requests** (batches being sent). + /// Upper bound on the requests being written concurrently, shared by *all* workers. /// - /// **WARNING**: Using more than 1 may cause message reordering if retries occur. - /// With max_in_flight > 1, a failed batch could be retried after later batches succeed. + /// A worker takes one permit before each request and holds it until the write returns. This + /// bounds write concurrency across the producer rather than per worker, and it does not bound + /// queued bytes. /// - /// The default is `1` to preserve message ordering. - /// `0` ⇒ unlimited (no ordering guarantee). + /// Each worker still sends sequentially regardless of this value. Raising it only lets different + /// workers write concurrently. Per-destination order is therefore governed by + /// [`sharding`](Self::sharding): [`OrderedSharding`] keeps one destination on one sequential + /// worker, while strategies that spread a destination across workers may reorder it. + /// A nonzero value greater than `Semaphore::MAX_PERMITS` makes + /// [`ProducerDispatcher::new`] panic. + /// + /// [`ProducerDispatcher::new`]: crate::clients::producer_dispatcher::ProducerDispatcher::new #[builder(default = 1)] pub max_in_flight: usize, } -/// Configuration for the *synchronous* (blocking) producer. +/// Configuration for a direct producer. +/// +/// A direct producer writes from the calling task. [`send()`] splits a batch into requests of at +/// most [`batch_length`](Self::batch_length) messages, awaits them one after another and returns +/// their confirmations. Nothing is buffered between calls. Unlike background mode, there is no +/// queue to bound, no worker to route to, and nothing to flush on shutdown. +/// +/// A send that fails part way through returns [`IggyError::ProducerSendFailed`], where `committed` +/// holds the confirmations returned for earlier requests and `failed` holds the unconfirmed tail. +/// See [`send()`] for what resending that tail means. +/// /// # Examples /// /// ```rust /// use iggy::prelude::*; -/// use iggy_common::IggyDuration; +/// use std::time::Duration; /// -/// // Send messages one-by-one (max latency, min memory per request) -/// let cfg = DirectConfig::builder() +/// // One request per message, with no configured delay between sequential calls. +/// let low_latency = DirectConfig::builder() /// .batch_length(1) /// .linger_time(IggyDuration::from(0)) /// .build(); /// -/// // Send in chunks of up to 500 messages, -/// // with a delay of at least 200 ms between consecutive sends. -/// let cfg = DirectConfig::builder() +/// // Up to 500 messages per request, with a 200 ms minimum gap between sequential sends. +/// let paced = DirectConfig::builder() /// .batch_length(500) -/// .linger_time(IggyDuration::from(200)) +/// .linger_time(IggyDuration::new(Duration::from_millis(200))) /// .build(); /// ``` +/// +/// [`send()`]: crate::clients::producer::IggyProducer::send +/// [`IggyError::ProducerSendFailed`]: iggy_common::IggyError::ProducerSendFailed #[derive(Clone, Builder)] pub struct DirectConfig { - /// Maximum number of messages to pack into **one** synchronous request. - /// `0` ⇒ MAX_BATCH_LENGTH(). + /// Maximum number of messages in one request. + /// + /// A send carrying more than this is split into consecutive requests of this size, each awaited + /// before the next one starts. A batch of 2500 therefore becomes three requests at the default. + /// + /// `0` limits to 1,000,000 messages per request. #[builder(default = 1000)] pub batch_length: u32, - /// How long to wait for more messages before flushing the current set. + /// Requested minimum gap between sequential direct sends. + /// + /// A send waits out whatever is left of this interval since the previous request completed + /// successfully. + /// Concurrent callers can wait against the same timestamp and then proceed together, so this is + /// not a global rate limiter. + /// When one call is split by [`batch_length`](Self::batch_length), the linger interval is applied + /// before the call rather than between its chunks. The default of zero does not wait. #[builder(default = IggyDuration::from(0))] pub linger_time: IggyDuration, } diff --git a/core/sdk/src/clients/producer_dispatcher.rs b/core/sdk/src/clients/producer_dispatcher.rs index 447a1b86cd..09d2e29562 100644 --- a/core/sdk/src/clients/producer_dispatcher.rs +++ b/core/sdk/src/clients/producer_dispatcher.rs @@ -17,15 +17,139 @@ use crate::clients::producer::ProducerCoreBackend; use crate::clients::producer_config::{BackgroundConfig, BackpressureMode}; -use crate::clients::producer_error_callback::ErrorCtx; +use crate::clients::producer_error_callback::{ErrorCallback, ErrorCtx}; use crate::clients::producer_sharding::{Shard, ShardMessage, ShardMessageWithPermit}; use futures::FutureExt; use iggy_common::{Identifier, IggyByteSize, IggyError, IggyMessage, Partitioning, Sizeable}; +use std::any::Any; +use std::panic::AssertUnwindSafe; use std::sync::Arc; use std::sync::atomic::{AtomicBool, Ordering}; use tokio::sync::{Semaphore, broadcast}; use tokio::task::JoinHandle; +/// The background machinery of an [`IggyProducer`](crate::clients::producer::IggyProducer), built from a +/// [`BackgroundConfig`] when the producer is configured with +/// [`IggyProducerBuilder::background`](crate::clients::producer_builder::IggyProducerBuilder::background). +/// +/// The dispatcher owns the background workers responsible for writing messages (the [`Shard`]s) +/// and coordinates two limits, both configured on [`BackgroundConfig`]: +/// [`max_buffer_size`](BackgroundConfig::max_buffer_size), the upper bound on the message bytes it +/// holds at once, and [`max_in_flight`](BackgroundConfig::max_in_flight), the upper bound on the +/// requests being written at once across all of its workers. +/// +/// A background send dispatches messages to a [`Shard`] through a channel. The batch is queued and +/// the write happens later on one of the shard workers owned by this type. Write failures therefore +/// surface on a shard after the queueing caller has returned. The shard forwards each failure to the +/// dispatcher's error task, which invokes [`BackgroundConfig::error_callback`]. The default +/// [`LogErrorCallback`] logs the failure, and applications can provide another [`ErrorCallback`] +/// implementation. +/// +/// Which worker a batch lands on, and what that means for ordering, is described in +/// [`IggyProducer`](crate::clients::producer::IggyProducer). +/// +/// # Examples +/// +/// Configuring background sending through the producer builder: +/// +/// ```rust,no_run +/// use iggy::prelude::*; +/// use std::str::FromStr; +/// +/// # async fn example() -> Result<(), IggyError> { +/// let client = IggyClient::from_connection_string("iggy://iggy:iggy@localhost:8090")?; +/// client.connect().await?; +/// +/// let producer = client +/// .producer("my-stream", "my-topic")? +/// .background(BackgroundConfig::builder().num_shards(4).build()) +/// .build(); +/// producer.init().await?; +/// +/// producer.send_one(IggyMessage::from_str("hello")?).await?; +/// producer.shutdown().await; +/// # Ok(()) +/// # } +/// ``` +/// +/// You can also implement a backend and wrap the dispatcher around it. This example prints instead +/// of sending anything to a server. +/// +/// ```rust,no_run +/// use iggy::clients::producer::ProducerCoreBackend; +/// use iggy::clients::producer_dispatcher::ProducerDispatcher; +/// use iggy::prelude::*; +/// use std::str::FromStr; +/// use std::sync::Arc; +/// +/// #[derive(Debug)] +/// struct CountingBackend; +/// +/// impl ProducerCoreBackend for CountingBackend { +/// async fn send_internal( +/// &self, +/// stream: &Identifier, +/// topic: &Identifier, +/// messages: Vec, +/// _partitioning: Option>, +/// ) -> Result { +/// println!("{} messages to {stream}/{topic}", messages.len()); +/// Ok(SendMessagesResponse { confirmations: Vec::new() }) +/// } +/// } +/// +/// # async fn example() -> Result<(), IggyError> { +/// let dispatcher = ProducerDispatcher::new( +/// Arc::new(CountingBackend), +/// BackgroundConfig::builder().num_shards(2).build(), +/// ); +/// +/// let stream = Arc::new(Identifier::named("my-stream")?); +/// let topic = Arc::new(Identifier::named("my-topic")?); +/// +/// // Returns once the batch is queued, not once it is written. +/// dispatcher +/// .dispatch(vec![IggyMessage::from_str("hello")?], stream, topic, None) +/// .await?; +/// +/// // Writes what is still queued. Dropping the dispatcher instead may lose buffered messages. +/// dispatcher.shutdown().await; +/// # Ok(()) +/// # } +/// ``` +/// +/// # Write constraints +/// +/// ## Memory budget +/// +/// [`dispatch()`](Self::dispatch) charges a batch against a budget of +/// [`BackgroundConfig::max_buffer_size`] bytes before queueing it. The charge is released when +/// [`ProducerCoreBackend::send_internal`] returns, so the budget covers queued sends and requests +/// whose result is still pending. What gets charged is the size [`ShardMessage`] reports, which +/// counts stream and topic identifiers alongside the messages rather than payloads alone. +/// +/// [`BackgroundConfig::failure_mode`] decides how the byte budget backpressures the caller. The +/// bounded channel feeding each shard is a separate source of backpressure and can also make a +/// dispatch wait. The second configured limit, [`BackgroundConfig::max_in_flight`], is shared by the +/// workers rather than the callers. A worker takes one of its permits for the batch it is about to +/// write, so it bounds concurrent writes across all workers, not queued bytes. +/// +/// ## In-flight limit +/// +/// [`BackgroundConfig::max_in_flight`] limits how many workers may write concurrently. The permit +/// pool is shared by every worker and defaults to one. A shard itself remains sequential for every +/// value: it awaits a request and all of its retries before starting its next request. Raising the +/// limit therefore affects concurrency between shards. It can expose reordering only when the +/// configured sharding strategy sends one ordered destination to different shards. +/// +/// # Shutdown +/// +/// [`shutdown()`](Self::shutdown) broadcasts a stop signal, drains every shard channel, flushes each +/// remaining buffer, and waits for the shard and error tasks. Dropping the dispatcher provides no +/// such completion guarantee and can lose batches that a shard has buffered but not written. +/// +/// [`ErrorCallback`]: crate::clients::producer_error_callback::ErrorCallback +/// [`LogErrorCallback`]: crate::clients::producer_error_callback::LogErrorCallback pub struct ProducerDispatcher { shards: Vec, config: Arc, @@ -36,6 +160,108 @@ pub struct ProducerDispatcher { } impl ProducerDispatcher { + /// Spawns the [`Shard`] workers that write messages to the server. + /// + /// The dispatcher owns the shards over which writes are distributed. + /// [`BackgroundConfig::sharding`] decides which of the + /// [`BackgroundConfig::num_shards`] workers receives each batch. + /// + /// The dispatcher also starts an error task. When a shard receives + /// [`IggyError::ProducerSendFailed`] from its backend, it sends the context to this task, which + /// invokes the [`ErrorCallback`] configured through [`BackgroundConfig::error_callback`]. A + /// custom backend error that is not `ProducerSendFailed` is logged by the shard and does not + /// invoke the callback. + /// + /// Two semaphores enforce [`BackgroundConfig::max_buffer_size`] and + /// [`BackgroundConfig::max_in_flight`]. Both are shared by every shard and therefore apply to + /// the entire producer rather than per worker. They + /// count different things at different points of a batch's life: `max_buffer_size` is charged in + /// bytes by [`dispatch()`](Self::dispatch) before the batch is queued and released once it has + /// completed, so it bounds queued and in-flight bytes together, while `max_in_flight` is taken + /// as one permit by a worker that is about to write and released when that request returns, so + /// it bounds concurrent requests and says nothing about their size. A batch therefore has to pass + /// the byte budget to enter a queue, and to take a request slot to leave it. + /// + /// # Panics + /// + /// Panics when a nonzero [`BackgroundConfig::max_buffer_size`] or + /// [`BackgroundConfig::max_in_flight`] exceeds `Semaphore::MAX_PERMITS`. + /// + /// # Examples + /// + /// Four workers allowed to write in parallel, fed round-robin, with a 64 MiB budget for queued + /// and in-flight bytes: + /// + /// ```rust,no_run + /// # use iggy::clients::producer::ProducerCoreBackend; + /// use iggy::clients::producer_dispatcher::ProducerDispatcher; + /// use iggy::prelude::*; + /// use std::sync::Arc; + /// + /// # #[derive(Debug)] + /// # struct Backend; + /// # impl ProducerCoreBackend for Backend { + /// # async fn send_internal( + /// # &self, + /// # _stream: &Identifier, + /// # _topic: &Identifier, + /// # _messages: Vec, + /// # _partitioning: Option>, + /// # ) -> Result { + /// # Ok(SendMessagesResponse { confirmations: Vec::new() }) + /// # } + /// # } + /// # fn example(backend: Arc) { + /// let dispatcher = ProducerDispatcher::new( + /// backend, + /// BackgroundConfig::builder() + /// .num_shards(4) + /// .max_in_flight(4) + /// // Ordering is given up for throughput: a batch can land on any of the four workers. + /// .sharding(Box::new(BalancedSharding::default())) + /// .max_buffer_size(IggyByteSize::from(64 * 1024 * 1024)) + /// .build(), + /// ); + /// # } + /// ``` + /// + /// `num_shards(0)` is read as one worker. A zero byte budget is treated as unbounded and a zero + /// in-flight limit uses the semaphore maximum. This dispatcher therefore never refuses a batch + /// for lack of byte-budget capacity, although its bounded shard channel can still make dispatch + /// wait: + /// + /// ```rust,no_run + /// # use iggy::clients::producer::ProducerCoreBackend; + /// use iggy::clients::producer_dispatcher::ProducerDispatcher; + /// use iggy::prelude::*; + /// use std::sync::Arc; + /// + /// # #[derive(Debug)] + /// # struct Backend; + /// # impl ProducerCoreBackend for Backend { + /// # async fn send_internal( + /// # &self, + /// # _stream: &Identifier, + /// # _topic: &Identifier, + /// # _messages: Vec, + /// # _partitioning: Option>, + /// # ) -> Result { + /// # Ok(SendMessagesResponse { confirmations: Vec::new() }) + /// # } + /// # } + /// # fn example(backend: Arc) { + /// let dispatcher = ProducerDispatcher::new( + /// backend, + /// BackgroundConfig::builder() + /// .num_shards(0) + /// .max_buffer_size(IggyByteSize::from(0)) + /// .max_in_flight(0) + /// .build(), + /// ); + /// # } + /// ``` + /// + /// [`ErrorCallback`]: crate::clients::producer_error_callback::ErrorCallback pub fn new(core: Arc, config: BackgroundConfig) -> Self { let num_shards = if config.num_shards == 0 { 1 @@ -51,10 +277,7 @@ impl ProducerDispatcher { let handle = tokio::spawn(async move { while let Ok(ctx) = err_rx.recv_async().await { - if let Err(panic) = std::panic::AssertUnwindSafe(err_callback.call(ctx)) - .catch_unwind() - .await - { + if let Err(panic) = call_error_callback(&**err_callback, ctx).await { tracing::error!("error_callback panicked: {:?}", panic); } } @@ -96,6 +319,133 @@ impl ProducerDispatcher { } } + /// Queues a batch on one of the worker [`Shard`]s and returns without waiting for it to be written. + /// + /// The batch is charged against the [`BackgroundConfig::max_buffer_size`] semaphore before it is + /// queued. Its permit travels with it, so those bytes stay charged until + /// [`ProducerCoreBackend::send_internal`] returns. A batch larger than the entire budget can + /// never be charged and fails with + /// [`IggyError::BackgroundSendBufferOverflow`]. + /// + /// When the budget is exhausted, [`BackgroundConfig::failure_mode`] decides what happens to the + /// caller. [`BackpressureMode::FailImmediately`] gives up with + /// [`IggyError::BackgroundSendBufferOverflow`], [`BackpressureMode::Block`] waits until enough + /// capacity is released, and [`BackpressureMode::BlockWithTimeout`] waits for its duration before + /// failing with [`IggyError::BackgroundSendTimeout`]. These modes do not control retries of the + /// server write. + /// + /// [`BackgroundConfig::sharding`] then picks the worker, and the batch is handed to that worker's + /// queue. The queue holds 256 entries, so a full queue can make the caller wait independently of + /// the byte-budget failure mode. + /// + /// # Errors + /// + /// [`IggyError::ProducerClosed`] once [`shutdown()`](Self::shutdown) has begun, and + /// [`IggyError::BackgroundSendError`] if the picked worker is already gone, which leaves the + /// batch unqueued and unsent in both cases. The budget can additionally fail the call with + /// [`IggyError::BackgroundSendBufferOverflow`] or [`IggyError::BackgroundSendTimeout`] as + /// described above. `BackgroundSendBufferOverflow` is also returned when the batch's reported + /// size does not fit the semaphore API's `u32` permit count, including when the configured byte + /// budget is unbounded. + /// + /// # Panics + /// + /// Panics if the configured [`Sharding`](crate::clients::producer_sharding::Sharding) + /// implementation returns an index outside the dispatcher's shard list. + /// + /// # Examples + /// + /// Dispatching with a strategy for partitioning. + /// + /// ```rust,no_run + /// # use iggy::clients::producer::ProducerCoreBackend; + /// use iggy::clients::producer_dispatcher::ProducerDispatcher; + /// use iggy::prelude::*; + /// use std::str::FromStr; + /// use std::sync::Arc; + /// + /// # #[derive(Debug)] + /// # struct Backend; + /// # impl ProducerCoreBackend for Backend { + /// # async fn send_internal( + /// # &self, + /// # _stream: &Identifier, + /// # _topic: &Identifier, + /// # _messages: Vec, + /// # _partitioning: Option>, + /// # ) -> Result { + /// # Ok(SendMessagesResponse { confirmations: Vec::new() }) + /// # } + /// # } + /// # async fn example(dispatcher: ProducerDispatcher) -> Result<(), IggyError> { + /// let stream = Arc::new(Identifier::named("orders")?); + /// let topic = Arc::new(Identifier::named("created")?); + /// let partitioning = Arc::new(Partitioning::messages_key_str("order-42")?); + /// + /// dispatcher + /// .dispatch( + /// vec![IggyMessage::from_str("order created")?], + /// stream.clone(), + /// topic.clone(), + /// Some(partitioning), + /// ) + /// .await?; + /// + /// // Dispatch only guarantees that the first batch was queued. A worker may already have + /// // started or completed its write. + /// dispatcher + /// .dispatch(vec![IggyMessage::from_str("order updated")?], stream, topic, None) + /// .await?; + /// # Ok(()) + /// # } + /// ``` + /// + /// Fail immediately if the `max_buffer_size` is exceeded. + /// + /// ```rust,no_run + /// # use iggy::clients::producer::ProducerCoreBackend; + /// use iggy::clients::producer_config::BackpressureMode; + /// use iggy::clients::producer_dispatcher::ProducerDispatcher; + /// use iggy::prelude::*; + /// use std::str::FromStr; + /// use std::sync::Arc; + /// + /// # #[derive(Debug)] + /// # struct Backend; + /// # impl ProducerCoreBackend for Backend { + /// # async fn send_internal( + /// # &self, + /// # _stream: &Identifier, + /// # _topic: &Identifier, + /// # _messages: Vec, + /// # _partitioning: Option>, + /// # ) -> Result { + /// # Ok(SendMessagesResponse { confirmations: Vec::new() }) + /// # } + /// # } + /// # async fn example(backend: Arc) -> Result<(), IggyError> { + /// let dispatcher = ProducerDispatcher::new( + /// backend, + /// BackgroundConfig::builder() + /// .max_buffer_size(IggyByteSize::from(1024 * 1024)) + /// .failure_mode(BackpressureMode::FailImmediately) + /// .build(), + /// ); + /// + /// let messages = vec![IggyMessage::from_str("hello")?]; + /// let stream = Arc::new(Identifier::named("orders")?); + /// let topic = Arc::new(Identifier::named("created")?); + /// + /// match dispatcher.dispatch(messages, stream, topic, None).await { + /// Ok(()) => println!("queued"), + /// // The workers are behind, or this single batch is larger than the whole budget. + /// Err(IggyError::BackgroundSendBufferOverflow) => println!("dropped, budget is full"), + /// Err(IggyError::ProducerClosed) => println!("dropped, dispatcher is shutting down"), + /// Err(error) => return Err(error), + /// } + /// # Ok(()) + /// # } + /// ``` pub async fn dispatch( &self, messages: Vec, @@ -162,7 +512,9 @@ impl ProducerDispatcher { &shard_message.stream, &shard_message.topic, ); + debug_assert!(shard_ix < self.shards.len()); + let shard = &self.shards[shard_ix]; shard @@ -199,6 +551,16 @@ impl ProducerDispatcher { } } +/// Catches a panic in `call()` itself as well as in the future it returns, so one misbehaving +/// callback cannot end the error task and silently drop every later failure. +async fn call_error_callback( + callback: &(dyn ErrorCallback + Send + Sync), + ctx: ErrorCtx, +) -> Result<(), Box> { + let future = std::panic::catch_unwind(AssertUnwindSafe(|| callback.call(ctx)))?; + AssertUnwindSafe(future).catch_unwind().await +} + #[cfg(test)] mod tests { use std::pin::Pin; @@ -209,7 +571,6 @@ mod tests { use tokio::time::sleep; use crate::clients::producer::{MockProducerCoreBackend, no_confirmations}; - use crate::clients::producer_error_callback::ErrorCallback; use crate::clients::producer_sharding::Sharding; use super::*; @@ -536,4 +897,65 @@ mod tests { assert_eq!(error_called.load(Ordering::SeqCst), 1); assert_eq!(last_batch_len.load(Ordering::SeqCst), 1); } + + /// Panics inside `call()` itself, before any future exists, on the first invocation only. + #[derive(Debug)] + struct PanicOnceErrorCallback { + called: Arc, + } + + impl ErrorCallback for PanicOnceErrorCallback { + fn call(&self, _ctx: ErrorCtx) -> Pin + Send + 'static>> { + if self.called.fetch_add(1, Ordering::SeqCst) == 0 { + panic!("first failure panics before returning a future"); + } + Box::pin(async {}) + } + } + + #[tokio::test] + async fn test_error_task_survives_panic_in_error_callback_call() { + let mut mock = MockProducerCoreBackend::new(); + mock.expect_send_internal().returning(|_, _, _, _| { + Box::pin(async { + Err(IggyError::ProducerSendFailed { + cause: Box::new(IggyError::Error), + failed: Arc::new(vec![dummy_message(10)]), + committed: Arc::new(Vec::new()), + stream_name: "1".to_string(), + topic_name: "1".to_string(), + }) + }) + }); + + let called = Arc::new(AtomicUsize::new(0)); + let config = BackgroundConfig::builder() + .num_shards(1) + .error_callback(Arc::new(Box::new(PanicOnceErrorCallback { + called: called.clone(), + }))) + .build(); + let dispatcher = ProducerDispatcher::new(Arc::new(mock), config); + + // Distinct topics keep the two sends from merging into one request, whichever branch of + // the worker ends up flushing them. + for topic_id in 1..=2 { + dispatcher + .dispatch( + vec![dummy_message(10)], + dummy_identifier(), + Arc::new(Identifier::numeric(topic_id).unwrap()), + None, + ) + .await + .unwrap(); + } + dispatcher.shutdown().await; + + assert_eq!( + called.load(Ordering::SeqCst), + 2, + "the failure after the panicking one must still reach the callback" + ); + } } diff --git a/core/sdk/src/clients/producer_error_callback.rs b/core/sdk/src/clients/producer_error_callback.rs index 2d2bc8514a..0c3a4758b4 100644 --- a/core/sdk/src/clients/producer_error_callback.rs +++ b/core/sdk/src/clients/producer_error_callback.rs @@ -23,32 +23,155 @@ use std::pin::Pin; use std::sync::Arc; use tracing::error; +/// Everything known about a background write that did not return a usable confirmation. +/// +/// A [`background()`] producer returns from [`send()`] once the batch is queued. The write happens +/// later on one of the dispatcher's worker [`Shard`]s, when there is no caller waiting for its +/// result. The worker therefore sends an `ErrorCtx` to the dedicated error task, which invokes the +/// [`ErrorCallback`] configured in [`BackgroundConfig::error_callback`]. +/// +/// # What to do with it +/// +/// [`messages`](Self::messages) is the unconfirmed tail of the send. No further automatic retry will +/// be attempted. Depending on [`cause`](Self::cause), the retry budget may be exhausted, the failure +/// may be deliberately non-retriable, or encryption or partitioning may have failed before a request +/// was sent. The tail is not proof that nothing committed. A request can commit before its response +/// is lost, and an HTTP confirmation decoding failure specifically occurs after a successful status. +/// Resending `messages` is therefore an at-least-once operation and can create duplicates. +/// Encryption mutates messages before the write, so this tail contains encrypted messages when the +/// producer uses an encryptor. Passing them back through the same producer would encrypt them again. +/// +/// [`stream`](Self::stream) and [`topic`](Self::topic) identify the destination. +/// [`partitioning`](Self::partitioning) contains only the per-send override passed to the dispatcher. +/// `None` means the producer's configured or default partitioning was used, not that the request had +/// no partitioning. A callback can retain the context, persist it in a dead-letter store, alert an +/// operator, or retry it when duplicate delivery is acceptable. The default [`LogErrorCallback`] +/// only logs the failure and then drops the context. +/// +/// [`background()`]: crate::clients::producer_builder::IggyProducerBuilder::background +/// [`send()`]: crate::clients::producer::IggyProducer::send +/// [`BackgroundConfig::error_callback`]: crate::clients::producer_config::BackgroundConfig::error_callback +/// [`Shard`]: crate::clients::producer_sharding::Shard #[derive(Debug)] pub struct ErrorCtx { + /// Error that ended the send. No further automatic retry will be attempted. pub cause: Box, + /// Stream identifier used by the failed request. pub stream: Arc, + /// Stream name configured when the producer was built. + /// + /// For a failure from + /// [`IggyProducer::send_to`](crate::clients::producer::IggyProducer::send_to), this may not name + /// [`Self::stream`]. pub stream_name: String, + /// Topic identifier used by the failed request. pub topic: Arc, + /// Topic name configured when the producer was built. + /// + /// For a failure from + /// [`IggyProducer::send_to`](crate::clients::producer::IggyProducer::send_to), this may not name + /// [`Self::topic`]. pub topic_name: String, + /// Per-send partitioning override, or `None` when the producer configuration was used. pub partitioning: Option>, + /// Unconfirmed tail of the send, see [`ErrorCtx`] for what resending it means. pub messages: Arc>, - /// Confirmations of the chunks that committed before the failure; `messages` - /// is the tail that did not. + /// Confirmations returned for chunks before the failure. pub committed: Arc>, } -/// A trait for handling background sending errors. +/// Handles a background write failure after the queueing caller has returned. +/// +/// A [`background()`](crate::clients::producer_builder::IggyProducerBuilder::background) producer +/// acknowledges a send once it is queued, so a later write failure cannot be returned by +/// [`IggyProducer::send`]. The dispatcher owns one implementation of this trait, set with +/// [`BackgroundConfig::error_callback`], and invokes it with an [`ErrorCtx`] whenever a worker's +/// backend returns [`IggyError::ProducerSendFailed`]. Other error variants from a custom backend are +/// logged by the worker without invoking this callback. +/// +/// # Implementing it +/// +/// - [`call()`](Self::call) returns a boxed future that the error task awaits, so the callback may do +/// asynchronous I/O. +/// - It runs on its own task rather than on a shard worker, so awaiting it does not stall batching. +/// Calls are serialized. One failure is handled at a time, and the unbounded error channel can +/// grow while a callback is slow. +/// - A panic inside it, whether in [`call()`](Self::call) itself or in the returned future, is +/// caught and logged, and the next failure is still delivered. +/// - `Send + Sync + Debug + 'static` is required because the dispatcher's task owns the callback for +/// the producer's lifetime and [`BackgroundConfig`] implements [`Debug`]. +/// +/// # Example /// -/// This is used when a message batch fails to send in an asynchronous background task. -/// Implementors can define custom logic such as logging, retrying, alerting, etc. +/// Forward each failed batch to a separate task instead of dropping it. The callback only enqueues +/// the context, so a slow store does not hold up later callbacks: +/// +/// ```no_run +/// use iggy::clients::producer_error_callback::{ErrorCallback, ErrorCtx}; +/// use iggy::prelude::*; +/// use std::pin::Pin; +/// use std::sync::Arc; +/// use tokio::sync::mpsc::{UnboundedReceiver, UnboundedSender}; +/// use tracing::warn; +/// +/// #[derive(Debug)] +/// struct FailedMessages { +/// failures: UnboundedSender, +/// } +/// +/// impl ErrorCallback for FailedMessages { +/// fn call(&self, ctx: ErrorCtx) -> Pin + Send + 'static>> { +/// let failures = self.failures.clone(); +/// Box::pin(async move { +/// let num_messages = ctx.messages.len(); +/// if failures.send(ctx).is_err() { +/// warn!(num_messages, "Failed messages task is gone, dropping messages"); +/// } +/// }) +/// } +/// } +/// +/// // Replace this warning with durable storage or another application-specific policy. +/// async fn drain(mut failures: UnboundedReceiver) { +/// while let Some(ctx) = failures.recv().await { +/// warn!( +/// cause = %ctx.cause, +/// stream_name = ctx.stream_name, +/// topic_name = ctx.topic_name, +/// num_messages = ctx.messages.len(), +/// "Received failed batch", +/// ); +/// } +/// } +/// +/// # async fn example() { +/// let (failures, receiver) = tokio::sync::mpsc::unbounded_channel(); +/// tokio::spawn(drain(receiver)); +/// +/// let config = BackgroundConfig::builder() +/// .error_callback(Arc::new(Box::new(FailedMessages { failures }))) +/// .build(); +/// # } +/// ``` +/// +/// [`BackgroundConfig`]: crate::clients::producer_config::BackgroundConfig +/// [`BackgroundConfig::error_callback`]: crate::clients::producer_config::BackgroundConfig::error_callback +/// [`IggyProducer::send`]: crate::clients::producer::IggyProducer::send pub trait ErrorCallback: Send + Sync + Debug + 'static { + /// Handles one failed request described by `ctx`. + /// + /// The dispatcher's error task calls this once per failed request and awaits the returned future + /// before taking the next failure from the queue. fn call(&self, ctx: ErrorCtx) -> Pin + Send + 'static>>; } -/// Default implementation of [`ErrorCallback`] that logs the error using `tracing::error!`. +/// Default [`ErrorCallback`] implementation that logs the error using `tracing::error!`. +/// +/// Logs include stream, topic, optional partitioning, number of messages, how many earlier chunks +/// returned confirmations, and the cause. /// -/// Logs include stream, topic, optional partitioning, number of messages, how -/// many chunks committed before the failure, and the cause. +/// The messages themselves are dropped with the context, so a background producer that keeps this +/// callback has no way to recover them. Implement [`ErrorCallback`] to hold on to them. #[derive(Debug, Default)] pub struct LogErrorCallback; diff --git a/core/sdk/src/clients/producer_sharding.rs b/core/sdk/src/clients/producer_sharding.rs index 619470b31d..26656f69a6 100644 --- a/core/sdk/src/clients/producer_sharding.rs +++ b/core/sdk/src/clients/producer_sharding.rs @@ -34,7 +34,12 @@ use crate::clients::producer_error_callback::ErrorCtx; /// Implementors of this trait define how to choose a shard for a given batch of messages. /// This allows customizing message routing based on message content, stream/topic identifiers, /// or round-robin load balancing. +/// +/// [`pick_shard`](Self::pick_shard) must return an index smaller than `num_shards`. The dispatcher +/// normalizes a configured shard count of zero to one, so `num_shards` passed here is never zero. +/// Returning an out-of-range index makes dispatch panic when it indexes the shard list. pub trait Sharding: Send + Sync + std::fmt::Debug + 'static { + /// Chooses the zero-based shard index for one batch. fn pick_shard( &self, num_shards: usize, @@ -90,11 +95,19 @@ impl Sharding for OrderedSharding { } } +/// One producer send after the dispatcher has selected its destination metadata. +/// +/// A sharding strategy sees the messages, stream, and topic before this value is queued. The +/// optional partitioning value overrides the producer's configured partitioning for this send. #[derive(Debug)] pub struct ShardMessage { + /// Target stream. pub stream: Arc, + /// Target topic. pub topic: Arc, + /// Messages carried by this send. pub messages: Vec, + /// Per-send partitioning override, or `None` to use the producer configuration. pub partitioning: Option>, } @@ -113,7 +126,13 @@ impl Sizeable for ShardMessage { } } +/// A [`ShardMessage`] together with its charge against the dispatcher's byte budget. +/// +/// The optional permit is acquired before the message enters a shard queue and remains owned by +/// this value until the worker finishes the write or drops the message. Keeping permits from merged +/// messages separate avoids the `u32` permit-count limit in Tokio's semaphore API. pub struct ShardMessageWithPermit { + /// Routed send and destination metadata. pub inner: ShardMessage, size_bytes: u64, bytes_permit: Option, @@ -121,6 +140,9 @@ pub struct ShardMessageWithPermit { } impl ShardMessageWithPermit { + /// Wraps `msg` with its previously acquired byte-budget permit. + /// + /// `bytes_permit` is `None` when the dispatcher's byte budget is unbounded. pub fn new(msg: ShardMessage, bytes_permit: Option) -> Self { let size_bytes = msg.get_size_bytes().as_bytes_u64(); Self { @@ -141,6 +163,36 @@ impl ShardMessageWithPermit { } } +/// Represents one background worker of a +/// [`ProducerDispatcher`](crate::clients::producer_dispatcher::ProducerDispatcher), +/// together with the channel (queue) that feeds it. +/// +/// Each shard owns a task that buffers sends routed to it, merges adjacent sends that share a +/// destination, and writes them through [`ProducerCoreBackend::send_internal`]. +/// +/// The dispatcher enqueues a batch by sending it on that channel, which returns once the batch is +/// queued. Shards are created and owned by the dispatcher, so they are rarely handled directly. +/// +/// # Worker loop +/// +/// The task selects over three sources: +/// +/// - **An enqueued send.** [`ShardMessageWithPermit`]s are appended to the buffer, which is flushed +/// once it reaches [`batch_length`](BackgroundConfig::batch_length) queued sends or +/// [`batch_size`](BackgroundConfig::batch_size) reported bytes. Either threshold is disabled when +/// configured as `0`. +/// - **The linger deadline.** Armed when a send enters an empty buffer and due +/// [`linger_time`](BackgroundConfig::linger_time) later, it flushes the buffer whether or not +/// either batching threshold was reached. An empty buffer arms nothing, so an idle shard does not +/// wake up. +/// - **The stop broadcast** sent by the dispatcher on shutdown. Marks the shard closed +/// so later sends fail with [`IggyError::ProducerClosed`], drains what is still +/// queued, flushes once, then ends the loop. A send that races the stop signal can be queued +/// after that drain and is lost without an error. +/// Dropping the [`ProducerDispatcher`] instead of shutting it down provides no completion +/// guarantee and can lose buffered messages. +/// +/// [`ProducerDispatcher`]: crate::clients::producer_dispatcher::ProducerDispatcher pub struct Shard { tx: flume::Sender, closed: Arc, @@ -148,6 +200,7 @@ pub struct Shard { } impl Shard { + /// Spawns the worker task described in the [`Shard`] type documentation. pub fn new( core: Arc, config: Arc, @@ -162,14 +215,18 @@ impl Shard { let handle = tokio::spawn(async move { let mut buffer = Vec::new(); let mut buffer_bytes = 0; - let mut last_flush = tokio::time::Instant::now(); + // Armed by the first send buffered after a flush and polled only while the buffer is + // non-empty, so an idle shard never wakes up and a zero linger cannot spin. + let mut linger_deadline = tokio::time::Instant::now(); loop { - let deadline = last_flush + config.linger_time.get_duration(); tokio::select! { maybe_msg = rx.recv_async() => { match maybe_msg { Ok(msg) => { + if buffer.is_empty() { + linger_deadline = tokio::time::Instant::now() + config.linger_time.get_duration(); + } buffer_bytes += msg.size_bytes as usize; buffer.push(msg); debug!( @@ -196,18 +253,13 @@ impl Shard { new_buffer_bytes = buffer_bytes, "Buffer flushed" ); - - last_flush = tokio::time::Instant::now(); } } Err(_) => break, } } - _ = tokio::time::sleep_until(deadline) => { - if !buffer.is_empty() { - Self::flush_buffer(&core, &slots_permit, &mut buffer, &mut buffer_bytes, &err_sender).await; - } - last_flush = tokio::time::Instant::now(); + _ = tokio::time::sleep_until(linger_deadline), if !buffer.is_empty() => { + Self::flush_buffer(&core, &slots_permit, &mut buffer, &mut buffer_bytes, &err_sender).await; } _ = stop_rx.recv() => { closed_clone.store(true, Ordering::Release); @@ -297,6 +349,10 @@ impl Shard { *buffer_bytes = 0; } + /// Queues one [`ShardMessageWithPermit`] on this worker. + /// + /// Returns [`IggyError::ProducerClosed`] after graceful shutdown has closed the shard, or + /// [`IggyError::BackgroundSendError`] if its worker channel is disconnected. pub(crate) async fn send(&self, message: ShardMessageWithPermit) -> Result<(), IggyError> { if self.closed.load(Ordering::Acquire) { return Err(IggyError::ProducerClosed); @@ -644,6 +700,63 @@ mod tests { sleep(Duration::from_millis(100)).await; } + #[tokio::test] + async fn test_shard_flushes_at_once_with_zero_linger() { + let sends = Arc::new(AtomicUsize::new(0)); + let sends_seen_by_mock = sends.clone(); + let mut mock = MockProducerCoreBackend::new(); + mock.expect_send_internal().returning(move |_, _, _, _| { + sends_seen_by_mock.fetch_add(1, Ordering::SeqCst); + Box::pin(async { Ok(no_confirmations()) }) + }); + + let config = Arc::new( + BackgroundConfig::builder() + .batch_length(10) + .batch_size(10_000) + .linger_time(IggyDuration::from(0)) + .build(), + ); + let permit_bytes = Arc::new(Semaphore::new(10_000)); + let slots_permit = Arc::new(Semaphore::new(100)); + + let (stop_tx, stop_rx) = broadcast::channel(1); + let shard = Shard::new( + Arc::new(mock), + config, + slots_permit, + flume::unbounded().0, + stop_rx, + ); + + let message = ShardMessage { + stream: dummy_identifier(), + topic: dummy_identifier(), + messages: vec![dummy_message(1)], + partitioning: None, + }; + let wrapped = ShardMessageWithPermit::new( + message, + Some(permit_bytes.clone().acquire_many_owned(1).await.unwrap()), + ); + shard.send(wrapped).await.unwrap(); + + sleep(Duration::from_millis(50)).await; + assert_eq!( + sends.load(Ordering::SeqCst), + 1, + "a zero linger must flush without waiting for a batching threshold" + ); + + stop_tx.send(()).unwrap(); + shard.handle.await.unwrap(); + assert_eq!( + sends.load(Ordering::SeqCst), + 1, + "the stop flush must find nothing left" + ); + } + #[tokio::test] async fn test_shard_forwards_error() { let mut mock = MockProducerCoreBackend::new(); From 7ff4eb170e8fec4f9ae8ac4dc60d43fded451622 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Tue, 1 Sep 2026 22:38:07 +0200 Subject: [PATCH 036/182] chore(deps): Bump com.github.spotbugs:spotbugs-annotations from 4.10.3 to 4.10.4 in /foreign/java in the java group across 1 directory (#4029) --- foreign/java/gradle/libs.versions.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/foreign/java/gradle/libs.versions.toml b/foreign/java/gradle/libs.versions.toml index 5e01cdcf06..700eec36e6 100644 --- a/foreign/java/gradle/libs.versions.toml +++ b/foreign/java/gradle/libs.versions.toml @@ -48,7 +48,7 @@ testcontainers = "2.0.5" netty = "4.2.17.Final" # Spotbugs -spotbugs = "4.10.3" +spotbugs = "4.10.4" # Config typesafe-config = "1.4.9" From 9dbc5f303754ad30d61977502ef37fd906a0603f Mon Sep 17 00:00:00 2001 From: Hubert Gruszecki Date: Tue, 1 Sep 2026 22:55:55 +0200 Subject: [PATCH 037/182] test(integration): stop harness servers pinning shards to cores (#4030) --- core/integration/src/harness/handle/server.rs | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/core/integration/src/harness/handle/server.rs b/core/integration/src/harness/handle/server.rs index 088b3b2617..8221a50a31 100644 --- a/core/integration/src/harness/handle/server.rs +++ b/core/integration/src/harness/handle/server.rs @@ -307,6 +307,12 @@ impl ServerHandle { self.envs .entry("IGGY_SYSTEM_SHARDING_CPU_ALLOCATION".to_string()) .or_insert(cpu_allocation); + // On a 4-core CI runner every server computes the same `0..4` range, so + // pinned shards of concurrently running tests pile onto the same cores + // and starve each other. Leave thread placement to the scheduler. + self.envs + .entry("IGGY_SYSTEM_SHARDING_PIN_CORES".to_string()) + .or_insert_with(|| "false".to_string()); self.envs .entry("IGGY_ROOT_USERNAME".to_string()) From a7d5a2894953cfac073a45741bc6cd5851d9a2ff Mon Sep 17 00:00:00 2001 From: haubur Date: Tue, 1 Sep 2026 23:16:11 +0200 Subject: [PATCH 038/182] chore(sdk): IggyConsumer docs (#3913) Co-authored-by: Hubert Gruszecki --- core/sdk/src/clients/consumer.rs | 563 ++++++++++++++++++++++- core/sdk/src/clients/consumer_builder.rs | 7 +- 2 files changed, 551 insertions(+), 19 deletions(-) diff --git a/core/sdk/src/clients/consumer.rs b/core/sdk/src/clients/consumer.rs index a4477e6c8f..a7bf549836 100644 --- a/core/sdk/src/clients/consumer.rs +++ b/core/sdk/src/clients/consumer.rs @@ -270,6 +270,355 @@ impl IggyConsumerState { // 4. All `&self` methods only access Sync-safe fields unsafe impl Sync for IggyConsumer {} +/// Reads messages from the partitions of one topic and yields them one at a time. +/// +/// A topic is split into partitions, and a partition is an ordered log that producers append to. +/// Every message sits at an *offset*, its position in that log. Reading is therefore always the +/// same three decisions: which partition to read, where in it to start, and how to keep track of +/// how far you got so the next run can continue there. +/// +/// `IggyConsumer` handles all three. It fetches batches of messages from the server, keeps them in +/// an in-memory buffer, decrypts them when it has an encryptor, and records how far it has read. +/// **It implements [`Stream`], so consuming is a loop over [`StreamExt::next`].** +/// +/// You can use a consumer as a worker draining a topic, a reader that replays a +/// partition from a chosen point, and a pool of consumers sharing a workload through a consumer +/// group. +/// +/// # Creating a consumer +/// +/// Easiest way is to use the [`IggyClient`] with a configured connection. Then: +/// - [`IggyClient::consumer()`] builds a standalone consumer, bound to the one partition passed +/// in. +/// - [`IggyClient::consumer_group()`] builds a member of a consumer group. The server gives every +/// partition to exactly one member, so several consumers using the same group name split the +/// topic between them and share one set of offsets. +/// +/// Note, building never talks to the server. [`init()`](Self::init) must be awaited once before the +/// first message is read. +/// +/// # Examples +/// +/// A standalone consumer reading partition 1 with the defaults: +/// +/// ```rust,no_run +/// use futures_util::StreamExt; +/// use iggy::prelude::*; +/// +/// # async fn example() -> Result<(), IggyError> { +/// let client = IggyClient::from_connection_string("iggy://iggy:iggy@localhost:8090")?; +/// client.connect().await?; +/// +/// let mut consumer = client +/// .consumer("my-consumer", "my-stream", "my-topic", 1)? +/// .batch_length(100) +/// .poll_interval(IggyDuration::new_from_secs(1)) +/// .build(); +/// consumer.init().await?; +/// +/// while let Some(received) = consumer.next().await { +/// match received { +/// Ok(received) => println!("Offset: {}", received.message.header.offset), +/// Err(error) => eprintln!("Failed to read a message: {error}"), +/// } +/// } +/// # Ok(()) +/// # } +/// ``` +/// +/// A group member that queues a commit for every message just before handing it over, and shuts +/// down cleanly: +/// +/// ```rust,no_run +/// use futures_util::StreamExt; +/// use iggy::prelude::*; +/// +/// # async fn handle(message: &IggyMessage) {} +/// # async fn example() -> Result<(), IggyError> { +/// let client = IggyClient::from_connection_string("iggy://iggy:iggy@localhost:8090")?; +/// client.connect().await?; +/// +/// let mut consumer = client +/// .consumer_group("order-workers", "my-stream", "my-topic")? +/// .auto_commit(AutoCommit::When(AutoCommitWhen::ConsumingEachMessage)) +/// .polling_strategy(PollingStrategy::next()) +/// .build(); +/// consumer.init().await?; +/// +/// let mut consumed = 0; +/// while let Some(received) = consumer.next().await { +/// match received { +/// Ok(received) => { +/// handle(&received.message).await; +/// consumed += 1; +/// } +/// Err(error) => eprintln!("Failed to read a message: {error}"), +/// } +/// if consumed == 100 { +/// break; +/// } +/// } +/// +/// consumer.shutdown().await?; +/// # Ok(()) +/// # } +/// ``` +/// +/// Committing by hand, so that a message the handler could not process comes back on the next +/// run. Every commit is one round trip, and no auto-commit setting substitutes, since each of them +/// also commits a message whose handler failed: +/// +/// ```rust,no_run +/// use futures_util::StreamExt; +/// use iggy::prelude::*; +/// +/// # async fn handle(message: &IggyMessage) -> Result<(), IggyError> { Ok(()) } +/// # async fn example() -> Result<(), IggyError> { +/// let client = IggyClient::from_connection_string("iggy://iggy:iggy@localhost:8090")?; +/// client.connect().await?; +/// +/// let mut consumer = client +/// .consumer("my-consumer", "my-stream", "my-topic", 1)? +/// .auto_commit(AutoCommit::Disabled) +/// .polling_strategy(PollingStrategy::next()) +/// .build(); +/// consumer.init().await?; +/// +/// while let Some(received) = consumer.next().await { +/// let received = match received { +/// Ok(received) => received, +/// Err(error) => { +/// eprintln!("Failed to read a message: {error}"); +/// continue; +/// } +/// }; +/// // Leaving a failed message uncommitted is what brings it back on the next run. +/// if handle(&received.message).await.is_err() { +/// break; +/// } +/// consumer +/// .store_offset(received.message.header.offset, Some(received.partition_id)) +/// .await?; +/// } +/// // No shutdown() here: it would commit the reading position, failed message included. +/// # Ok(()) +/// # } +/// ``` +/// +/// # Which partitions are read +/// +/// A **standalone consumer** reads exactly one partition, the one passed to +/// [`IggyClient::consumer()`]. Covering a whole topic with several partitions +/// means running one consumer per partition and dividing the work yourself. +/// +/// A **consumer group member** does not choose. The server hands every partition of the topic to +/// exactly one member, so consumers sharing a group name split the topic between them without +/// coordinating. [`ReceivedMessage::partition_id`] tells where a message came from. +/// +/// What to know when working with consumer groups: +/// - With [`auto_join_consumer_group()`] (the default) a member joins during [`init()`](Self::init), +/// creating the group first if [`create_consumer_group_if_not_exists()`] is set (the default). +/// It rejoins on its own after a reconnect and whenever the server reports that its membership +/// is gone. +/// - Until the join has succeeded the consumer does not poll. It re-checks every +/// [`polling_retry_interval()`] and polls once joined. +/// - Partitions are redistributed whenever members join or leave, so a member reads different +/// partitions over time and messages from several partitions interleave in its stream. +/// - More members than partitions leaves the surplus members without partitions. Such a member +/// still polls, one group sync round trip per attempt, so give it a [`poll_interval()`]. The +/// partition count of the topic is the ceiling on how far one group can be scaled out. +/// - The group shares one set of stored offsets, kept under the group name. Thus, +/// a partition taken over by another member continues where the previous one +/// committed. +/// +/// # How messages are read +/// +/// Reading is done by polling. One request fetches up to [`batch_length()`] messages. The consumer +/// passes the first one to the caller and buffers the rest. The next request is sent once that buffer +/// is empty. +/// +/// [`poll_interval()`] sets the smallest gap between two requests, measured from one send to the +/// next. Without it the next request goes out as soon as the previous one is answered, which is +/// the fastest option but keeps a busy loop running against an idle topic. +/// +/// [`polling_strategy()`] decides **where** in the partition reading begins: +/// +/// | Strategy | Starts at | +/// | --- | --- | +/// | [`PollingStrategy::next()`] (default) | the message after the offset stored on the server, or the first message when nothing is stored yet | +/// | [`PollingStrategy::first()`] | the oldest message in the partition | +/// | [`PollingStrategy::last()`] | the end of the partition (returns up to [`batch_length()`] of the most recent messages) | +/// | [`PollingStrategy::offset()`] | a custom offset | +/// | [`PollingStrategy::timestamp()`] | the first message at or after a given point in time | +/// +/// Only [`PollingStrategy::next()`] consults the offset stored on the server. +/// Use this if you want to resume where a previous run stopped. The other four are starting points +/// for the first request only. From the second request onwards, the consumer asks for whatever +/// follows the last message it handed over, and it keeps one such continuation point for all +/// partitions. That makes them fit for standalone consumers only: a group member polls a different +/// partition on every request, so a continuation point taken from one partition is applied to the +/// next, where it skips or repeats messages, and under the default [`auto_commit()`] the skipped +/// range is committed as read. +/// +/// [`StreamExt::next`] yields `None` once [`shutdown()`](Self::shutdown) has been called, and never +/// otherwise: not when the topic is empty and not while the client is disconnected. A request that +/// comes back empty is not an error and not the end of the stream, it just means nothing new has +/// arrived yet. +/// +/// A failed request is yielded as `Some(Err(..))` and leaves the consumer usable, while the next call +/// retries. Connection and authentication failures pause polling until the client has reconnected +/// and signed in again, which the consumer handles automatically. Hence, deciding when to give up on +/// repeated errors is up to you. +/// +/// For a boilerplate implementation of such a loop Iggy provides [`IggyConsumerMessageExt::consume_messages`]. +/// +/// # Tracking what has been read +/// +/// An offset is the index tracking what has been already read from a partition by the consumer. +/// Managing the offset has implications on where consumers resume reading messages. +/// +/// There are two positions (offsets) tracked in two different places: +/// - The **reading position** is held by the consumer, one per partition, and is the offset of the +/// last message handed over +/// ([`get_last_consumed_offset()`](Self::get_last_consumed_offset)). It dies with the process. +/// - The **stored offset** lives on the server under the consumer name, or the group name for a +/// group. This offset survives restarts. Writing it is called *storing* or *committing* an offset. +/// +/// Committing matters because [`PollingStrategy::next()`] resumes from the stored offset. A +/// consumer that never commits starts over from the same place on every run. Within a run it +/// stalls instead: the server serves the same messages again on every request, messages at or +/// below the reading position are dropped (see [Guarantees](#guarantees)), so the stream goes +/// quiet once the reading position is [`batch_length()`] or more ahead of the stored offset. Under +/// [`PollingStrategy::next()`], keep commits within [`batch_length()`] of the reading position. +/// +/// [`auto_commit()`] decides when the consumer commits by itself: +/// +/// | Setting | Commits | +/// | --- | --- | +/// | [`AutoCommit::Disabled`] | never on its own, decide manually with [`store_offset()`](Self::store_offset). [`shutdown()`](Self::shutdown) still commits the reading position | +/// | [`AutoCommit::Interval`] | on every tick, the reading position of every partition read so far | +/// | [`AutoCommitWhen::PollingMessages`] | sends the commit with the poll request itself, before your code sees the batch | +/// | [`AutoCommitWhen::ConsumingEachMessage`] | queued just before every message is handed over to the calling code, at one round trip per message and without backpressure | +/// | [`AutoCommitWhen::ConsumingEveryNthMessage`] | queued just before a message whose offset divides by `n` is handed over | +/// | [`AutoCommitWhen::ConsumingAllMessages`] | queued when the buffer of the current batch runs empty | +/// | [`AutoCommitAfter`] variants | once the handler returned, `Ok` or `Err`, and only under [`IggyConsumerMessageExt::consume_messages`], see below | +/// +/// [`AutoCommit::IntervalOrWhen`] and [`AutoCommit::IntervalOrAfter`] combine an interval with a +/// message trigger. The default is [`AutoCommit::IntervalOrWhen`] with one second and +/// [`AutoCommitWhen::PollingMessages`]. +/// Important implications of these settings: +/// - [`AutoCommitWhen::PollingMessages`] marks a batch as consumed while it is being delivered, +/// before your code has seen any of it. For a crash-safe option configure with [`AutoCommit::Disabled`] +/// and manually store the offset with [`Self::store_offset()`]. +/// - [`AutoCommitWhen::ConsumingEveryNthMessage`] tests the offset of a message, not a counter of +/// messages this process handled, so it commits at every `n`-th offset of the partition. With +/// `n = 0` the trigger never fires, so without an interval only [`shutdown()`](Self::shutdown) +/// commits. +/// - [`AutoCommitAfter::ConsumingAllMessages`] fires for the message whose offset equals the +/// partition head seen by the poll ([`ReceivedMessage::current_offset`]), not when the buffer +/// runs empty, so a consumer that lags behind commits nothing until it has caught up. Every +/// [`AutoCommitAfter`] variant commits after a handler that returned `Err` as well. +/// +/// ## Guarantees +/// +/// - **Each message is handed over once per consumer.** Messages whose offset is not greater than +/// the reading position of their partition are dropped before they reach the stream. Re-reading +/// a partition, or seeing a failed message again within the same consumer, needs +/// [`allow_replay()`], which turns that filter off. A new consumer starts with an empty filter. +/// - **Delivering at-least-once.** If you cannot tolerate missing any messages, use [`AutoCommit::Disabled`] +/// and store the offset using [`Self::store_offset()`] after handling a message. Every other +/// setting except the plain [`AutoCommit::After`] variants can commit a message before your +/// handler is done with it, so a crash in the handler loses it. [`AutoCommit::IntervalOrAfter`] +/// still commits on its interval tick. [`shutdown()`](Self::shutdown) commits the reading +/// position under every setting, [`AutoCommit::Disabled`] included, so it also commits a message +/// whose handler failed. +/// +/// # Options and defaults +/// +/// Everything is configured on the [`IggyConsumerBuilder`] before [`build()`] and is fixed +/// afterwards. +/// +/// | Option | Default | Controls | +/// | --- | --- | --- | +/// | [`stream()`], [`topic()`], [`partition()`] | the values passed to the entry point | what is read. [`partition()`] is for standalone consumers, on a group member it pins every poll to that partition instead of the server's assignment | +/// | [`batch_length()`] | 1000 | messages fetched per request | +/// | [`poll_interval()`] | none | smallest gap between two requests | +/// | [`polling_strategy()`] | [`PollingStrategy::next()`] | where reading starts. Anything but [`PollingStrategy::next()`] is for standalone consumers only | +/// | [`auto_commit()`] | [`AutoCommit::IntervalOrWhen`], one second, [`AutoCommitWhen::PollingMessages`] | when offsets are committed. [`commit_failed_messages()`] is a synonym for [`AutoCommit::Disabled`] | +/// | [`allow_replay()`] | off | whether a message can be handed over again | +/// | [`auto_join_consumer_group()`] | on | joining the group during [`init()`](Self::init) and after a reconnect. A group member built with [`do_not_auto_join_consumer_group()`] never polls, since polling waits for the join | +/// | [`create_consumer_group_if_not_exists()`] | on | creating the group when it is missing | +/// | [`polling_retry_interval()`] | one second | wait between attempts while polling is blocked | +/// | [`init_retries()`] | none, one second apart | retries when the stream or topic is missing at [`init()`](Self::init) | +/// | [`offset_drain_timeout()`] | five seconds | how long [`shutdown()`](Self::shutdown) waits for pending commits | +/// | [`encryptor()`] | inherited from the client | decrypting payloads and user headers | +/// +/// The switches have inverse setters as well, such as [`without_poll_interval()`], +/// [`without_encryptor()`], [`do_not_auto_join_consumer_group()`] and +/// [`do_not_create_consumer_group_if_not_exists()`]. +/// +/// # Encryption +/// +/// A consumer with an encryptor, inherited from the [`IggyClient`] or set with [`encryptor()`], +/// decrypts payloads and user headers before a message is yielded. That only works if the producer +/// encrypted them with the same key, which an [`IggyProducer`] and an `IggyConsumer` from the same +/// client share unless one of them overrides it on its builder. Without an encryptor the consumer +/// yields payloads as stored, encrypted or not. +/// +/// A message that cannot be decrypted is yielded as an `Err` and the whole batch is dropped. What +/// happens next depends on [`auto_commit()`]. Under [`AutoCommitWhen::PollingMessages`] (the +/// default) the server committed the batch with the poll, so it is skipped for good. Under every +/// other setting the next request fetches the same batch and fails the same way until +/// [`store_offset()`](Self::store_offset) moves the offset past it. Pick a setting other than +/// [`AutoCommitWhen::PollingMessages`] if a batch that fails to decrypt must not be lost silently. +/// +/// # Concurrency +/// +/// `IggyConsumer` is `Send` and `Sync` but not `Clone`. Driving the stream +/// ([`StreamExt::next`]) and [`shutdown()`](Self::shutdown) take `&mut self`, so one task owns and +/// drives a consumer end to end. To read offsets or commit from another task, take an +/// [`IggyConsumerState`] via [`state()`](Self::state): it is a cheap clone of the shared +/// bookkeeping and needs no lock. +/// +/// Besides the stream, a consumer runs background tasks: one watching the connection lifecycle, +/// one sending queued commits, and an interval commit task for the [`AutoCommit`] variants that +/// carry an interval. See [`init()`](Self::init). +/// +/// # Shutting down +/// +/// Call [`shutdown()`](Self::shutdown) once done consuming. It drains the commit tasks, commits +/// the reading position of every partition, [`AutoCommit::Disabled`] included, and leaves the +/// consumer group. Dropping an `IggyConsumer` instead skips that final commit and the group +/// leave, so the server reassigns the member's partitions only once the connection is gone. +/// Commits already queued are still sent. Neither stops the lifecycle task, which runs until the +/// client shuts down. +/// +/// [`IggyClient`]: crate::prelude::IggyClient +/// [`IggyClient::consumer()`]: crate::prelude::IggyClient::consumer +/// [`IggyClient::consumer_group()`]: crate::prelude::IggyClient::consumer_group +/// [`IggyProducer`]: crate::prelude::IggyProducer +/// [`IggyConsumerBuilder`]: crate::prelude::IggyConsumerBuilder +/// [`IggyConsumerMessageExt::consume_messages`]: crate::prelude::IggyConsumerMessageExt::consume_messages +/// [`allow_replay()`]: crate::prelude::IggyConsumerBuilder::allow_replay +/// [`auto_commit()`]: crate::prelude::IggyConsumerBuilder::auto_commit +/// [`auto_join_consumer_group()`]: crate::prelude::IggyConsumerBuilder::auto_join_consumer_group +/// [`batch_length()`]: crate::prelude::IggyConsumerBuilder::batch_length +/// [`build()`]: crate::prelude::IggyConsumerBuilder::build +/// [`commit_failed_messages()`]: crate::prelude::IggyConsumerBuilder::commit_failed_messages +/// [`create_consumer_group_if_not_exists()`]: crate::prelude::IggyConsumerBuilder::create_consumer_group_if_not_exists +/// [`do_not_auto_join_consumer_group()`]: crate::prelude::IggyConsumerBuilder::do_not_auto_join_consumer_group +/// [`do_not_create_consumer_group_if_not_exists()`]: crate::prelude::IggyConsumerBuilder::do_not_create_consumer_group_if_not_exists +/// [`encryptor()`]: crate::prelude::IggyConsumerBuilder::encryptor +/// [`init_retries()`]: crate::prelude::IggyConsumerBuilder::init_retries +/// [`offset_drain_timeout()`]: crate::prelude::IggyConsumerBuilder::offset_drain_timeout +/// [`partition()`]: crate::prelude::IggyConsumerBuilder::partition +/// [`poll_interval()`]: crate::prelude::IggyConsumerBuilder::poll_interval +/// [`polling_retry_interval()`]: crate::prelude::IggyConsumerBuilder::polling_retry_interval +/// [`polling_strategy()`]: crate::prelude::IggyConsumerBuilder::polling_strategy +/// [`stream()`]: crate::prelude::IggyConsumerBuilder::stream +/// [`topic()`]: crate::prelude::IggyConsumerBuilder::topic +/// [`without_encryptor()`]: crate::prelude::IggyConsumerBuilder::without_encryptor +/// [`without_poll_interval()`]: crate::prelude::IggyConsumerBuilder::without_poll_interval pub struct IggyConsumer { initialized: bool, shutdown: Arc, @@ -407,21 +756,28 @@ impl IggyConsumer { } /// Returns the name of the consumer. + /// + /// For a consumer group this is also the name of the group. pub fn name(&self) -> &str { &self.consumer_name } - /// Returns the topic ID of the consumer. + /// Returns the identifier of the topic this consumer reads from. pub fn topic(&self) -> &Identifier { &self.topic_id } - /// Returns the stream ID of the consumer. + /// Returns the identifier of the stream this consumer reads from. pub fn stream(&self) -> &Identifier { &self.stream_id } - /// Returns the current partition ID of the consumer. + /// Returns the partition the most recent poll response came from. + /// + /// This is `0` before the first response, and an empty response can report `0` as well. For a + /// consumer group the value changes over time, as the server hands different partitions to + /// this member. To commit for the partition a message came from, pass + /// [`ReceivedMessage::partition_id`] to [`store_offset()`](Self::store_offset) instead. pub fn partition_id(&self) -> u32 { self.state.partition_id() } @@ -431,7 +787,28 @@ impl IggyConsumer { self.state.clone() } - /// Stores the consumer offset on the server either for the current partition or the provided partition ID. + /// Stores an offset on the server, marking every message up to and including it as consumed. + /// + /// This is the manual counterpart to [`AutoCommit`] and is meant for + /// [`AutoCommit::Disabled`]. + /// + /// Pass `None` as `partition_id` to use [`partition_id()`](Self::partition_id), which for a + /// consumer group can already point at another partition. Prefer passing + /// [`ReceivedMessage::partition_id`]. + /// + /// An offset that is not ahead of the last one this consumer stored for that partition is + /// skipped and `Ok(())` is returned without a request, unless the consumer was built with + /// [`allow_replay`](crate::prelude::IggyConsumerBuilder::allow_replay). Offset `0` is always + /// sent, yet [`PollingStrategy::next()`] then resumes at offset `1`, so it does not rewind to + /// the start. To start over from the first message, delete the stored offset with + /// [`delete_offset()`](Self::delete_offset) after [`shutdown()`](Self::shutdown) and build a new + /// consumer. A new consumer built with [`PollingStrategy::offset()`] re-reads from any point. + /// + /// # Errors + /// + /// Returns any error the server raised while storing the offset, for example + /// [`IggyError::Disconnected`] or a permission error. The offset is then not stored and the + /// call can be retried. pub async fn store_offset( &self, offset: u64, @@ -440,26 +817,102 @@ impl IggyConsumer { self.state.store_offset(offset, partition_id).await } - /// Retrieves the last consumed offset for the specified partition ID. - /// To get the current partition ID use `partition_id()` + /// Returns the offset of the last message this consumer handed over for the given partition, + /// or `None` while it has not polled that partition yet. The first poll of a partition seeds + /// the entry with `0`, so `Some(0)` also covers "polled, nothing handed over yet". + /// + /// This is the local reading position, which can be ahead of what has been stored on the + /// server. pub fn get_last_consumed_offset(&self, partition_id: u32) -> Option { self.state.get_last_consumed_offset(partition_id) } - /// Deletes the consumer offset on the server either for the current partition or the provided partition ID. + /// Deletes the offset stored on the server, so that the next consumer polling with + /// [`PollingStrategy::next()`] starts at the first message. + /// + /// `None` as `partition_id` means [`partition_id()`](Self::partition_id) for a standalone + /// consumer. A consumer group passes `None` through to the server. This consumer's own records + /// are untouched, so its auto-commit or [`shutdown()`](Self::shutdown) can store the offset + /// again. When starting over, call it after [`shutdown()`](Self::shutdown). + /// + /// # Errors + /// + /// Returns [`IggyError::ConsumerOffsetNotFound`] when nothing is stored for that partition, or + /// any other error the server raised, for example [`IggyError::Disconnected`]. pub async fn delete_offset(&self, partition_id: Option) -> Result<(), IggyError> { self.state.delete_offset(partition_id).await } - /// Retrieves the last stored offset (on the server) for the specified partition ID. - /// To get the current partition ID use `partition_id()` + /// Returns the offset this consumer last stored on the server for the given partition, or + /// `None` while it has neither polled nor stored for that partition. The first poll or store + /// seeds the entry, so `Some(0)` also covers "seen, nothing stored yet". + /// + /// The value is this consumer's own record of what it committed, kept in memory rather than + /// read back from the server. + /// Under auto-commit-on-poll (the default) this can trail the server by up to one batch. pub fn get_last_stored_offset(&self, partition_id: u32) -> Option { self.state.get_last_stored_offset(partition_id) } - /// Initializes the consumer by subscribing to diagnostic events, initializing the consumer group if needed, storing the offsets in the background etc. + /// Initializes the consumer and makes it ready to poll messages. + /// + /// This must be called before the consumer can start polling messages. Calling it again on an + /// initialized consumer does nothing and returns immediately. + /// + /// Initialization ensures that: + /// - the consumer's `stream_id` and `topic_id` exist on the server. + /// It retries for a number of `init_retries` (defaults to `None`, which is treated as no + /// retry) with `init_retry_interval` (defaults to one + /// second) time in between retries. Both can be set together through + /// [`IggyConsumerBuilder::init_retries`](crate::prelude::IggyConsumerBuilder::init_retries). + /// - the consumer subscribes to connection lifecycle events ([`DiagnosticEvent`]) in order to + /// update its state, should it receive a shutdown, connected, disconnected, log in or log out event. + /// - if the consumer belongs to a group and `auto_join_consumer_group` is enabled, the group is + /// initialized if it does not exist yet, and the consumer joins that group. + /// - the tasks that store the offset on the server are spawned. + /// + /// # Lifecycle events /// - /// Note: This method must be called before polling messages. + /// Calling init spawns a background task that listens for lifecycle changes ([`DiagnosticEvent`]s) of the + /// client connection. It runs until the client shuts down and is not stopped by + /// [`shutdown()`](Self::shutdown). + /// - [`DiagnosticEvent::Connected`]: a fresh connection has not joined anything yet. + /// Polling resumes immediately only for a consumer that is not a group member. + /// - [`DiagnosticEvent::SignedIn`]: re-enables polling. A group member signing in after a + /// reconnect rejoins its group first and only polls once that succeeded. A failed rejoin is + /// logged and leaves polling disabled until the next reconnect or an explicit login. + /// - [`DiagnosticEvent::Disconnected`] and [`DiagnosticEvent::SignedOut`] disable polling. + /// - [`DiagnosticEvent::Shutdown`] disables polling and terminates the background task listening + /// for lifecycle changes. It does not flush in-flight commits; that only happens when + /// [`shutdown()`](Self::shutdown) itself is called. + /// + /// # Storing offsets + /// + /// When the consumer commits is decided by + /// [`auto_commit()`](crate::prelude::IggyConsumerBuilder::auto_commit), see + /// [Tracking what has been read](IggyConsumer#tracking-what-has-been-read). `init()` spawns the + /// tasks behind it: + /// - An interval task, only for the variants that carry an interval ([`AutoCommit::Interval`], + /// [`AutoCommit::IntervalOrWhen`], [`AutoCommit::IntervalOrAfter`]). Every tick it stores the + /// reading position of every partition read so far. + /// - An offset store task, always. It sends the commits queued by the [`AutoCommitWhen`] and + /// [`AutoCommitAfter`] triggers one at a time and stays idle under [`AutoCommit::Disabled`]. + /// + /// Both skip an offset that is not ahead of this consumer's own record of what it stored + /// ([`get_last_stored_offset()`](Self::get_last_stored_offset)). Only offset `0` is always sent. + /// Under auto-commit-on-poll (the default) that record trails the server by one batch, so every + /// tick re-sends the reading position and the server, which takes an explicit store as is, + /// moves its offset back to it until the next poll. + /// + /// # Errors + /// + /// - [`IggyError::StreamNameNotFound`] or [`IggyError::TopicNameNotFound`] when the + /// stream or the topic still does not exist once the retries are exhausted. + /// - [`IggyError::ConsumerGroupNameNotFound`] when the consumer group does not exist + /// and its auto creation is disabled. + /// - Any error returned by the server while looking up the stream or the topic, or + /// while creating or joining the consumer group. Such an error ends initialization + /// immediately instead of consuming a retry. pub async fn init(&mut self) -> Result<(), IggyError> { if self.initialized { return Ok(()); @@ -485,6 +938,9 @@ impl IggyConsumer { let mut stream_exists = client.get_stream(&stream_id).await?.is_some(); let mut topic_exists = client.get_topic(&stream_id, &topic_id).await?.is_some(); + // Absent streams or topics are not necessarily permanent failures. + // It may happen that get_stream/ get_topic races the initial setup of the stream/ topic. + // Retry for init_retries times, while waiting interval between retries. loop { if stream_exists && topic_exists { info!( @@ -538,6 +994,7 @@ impl IggyConsumer { } self.subscribe_events().await; + // No-op if either is_consumer_group or auto_join_consumer_group is false self.init_consumer_group().await?; match self.auto_commit { @@ -553,6 +1010,9 @@ impl IggyConsumer { let (store_offset_sender, store_offset_receiver) = flume::unbounded(); self.store_offset_sender = store_offset_sender; + // Message-triggered commits from `poll_next` and `consume_messages` queue here and go out + // one at a time. The interval task above and the poll request's own `auto_commit` flag + // are the other commit paths. self.store_offset_task = Some(tokio::spawn(async move { while let Ok((partition_id, offset)) = store_offset_receiver.recv_async().await { trace!( @@ -579,13 +1039,16 @@ impl IggyConsumer { let notify = self.background_commit_notify.clone(); tokio::spawn(async move { loop { + // Wait for the task until either the interval has passed or + // the task is explicitly notified, which happens when shutdown() is called. tokio::select! { _ = sleep(interval.get_duration()) => {} _ = notify.notified() => {} } - // Checked before storing: `shutdown` already ran its own final - // flush as a group member, so a store past that point would - // hit a group we've since left. + + // Checked before storing: `shutdown()` runs its own final flush as a + // group member and then leaves, so a store past this point would hit + // a group we've since left. After a bare `Drop` nothing flushes. if shutdown.load(ORDERING) { trace!("Shutdown signal received, stopping background offset storage"); break; @@ -1035,13 +1498,29 @@ impl IggyConsumer { } } +/// A single message handed over by an [`IggyConsumer`]. pub struct ReceivedMessage { + /// The message itself, with its payload already decrypted when the client uses an encryptor. + /// + /// Its own offset is `message.header.offset`, which is the value to pass to + /// [`IggyConsumer::store_offset`] when committing by hand. pub message: IggyMessage, + /// The offset of the newest message in the partition at the time it was polled. + /// + /// Comparing it with `message.header.offset` shows how far this consumer lags behind the end + /// of the partition. It is a snapshot taken per request, so it does not change while the + /// buffered messages of that request are handed over. pub current_offset: u64, + /// The partition this message was read from. + /// + /// For a consumer group this varies between messages, since the server hands different + /// partitions to the same member. pub partition_id: u32, } impl ReceivedMessage { + /// Creates a received message from a message, the partition head at poll time and the + /// partition it was read from. pub fn new(message: IggyMessage, current_offset: u64, partition_id: u32) -> Self { Self { message, @@ -1051,6 +1530,9 @@ impl ReceivedMessage { } } +/// Yields messages one at a time, from the buffer first and from a fresh poll once it is empty. +/// +/// See [How messages are read](IggyConsumer#how-messages-are-read) for errors, `None` and polling. impl Stream for IggyConsumer { type Item = Result; @@ -1062,6 +1544,9 @@ impl Stream for IggyConsumer { let partition_id = self.state.partition_id(); if let Some(message) = self.buffered_messages.pop_front() { { + // Since a consumer can be standalone or a member of a consumer group, in which case + // it can be reassigned to another partition, either update the offset of a partition + // the consumer already worked with or add a new record, if it got reassigned. if let Some(last_consumed_offset_entry) = self.state.last_consumed_offsets.get(&partition_id) { @@ -1080,6 +1565,11 @@ impl Stream for IggyConsumer { } } + // Popping above may have left the buffer empty. + // The next turn will therefore poll messages from the server. + // With `PollingStrategy` the user defines the starting point where to poll from. + // After that, each poll must read the next sequential offset. Hence, strategy is + // set to `PollingKind::Offset` and the next offset to read from is the last consumed message + 1. if self.buffered_messages.is_empty() { if self.polling_strategy.kind != PollingKind::Next { self.polling_strategy = PollingStrategy::offset(message.header.offset + 1); @@ -1090,6 +1580,8 @@ impl Stream for IggyConsumer { } } + // Not the position of this message but the newest offset the partition had when the + // batch was polled. So every message of a batch reports the same value. let current_offset; if let Some(current_offset_entry) = self.current_offsets.get(&partition_id) { current_offset = current_offset_entry.load(ORDERING); @@ -1104,6 +1596,7 @@ impl Stream for IggyConsumer { )))); } + // A used (and therefore invalid) future was dropped, thus create a fresh one. if self.poll_future.is_none() { let future = self.create_poll_messages_future(); self.poll_future = Some(Box::pin(future)); @@ -1192,6 +1685,7 @@ impl Stream for IggyConsumer { ); } + // Drop future since it is [invalid after being ready](https://doc.rust-lang.org/std/future/trait.Future.html#panics) self.poll_future = None; return Poll::Ready(Some(Ok(ReceivedMessage::new( message, @@ -1213,17 +1707,42 @@ impl Stream for IggyConsumer { } impl IggyConsumer { + /// Shuts the consumer down. + /// + /// Specifically, run shutdown and await before dropping the consumer to + /// - finish storing the offsets that are currently in-flight. + /// The interval task and the offset store task (see [`init()`](Self::init)) can both have + /// commits in flight. The consumer waits for `offset_drain_timeout` on each in turn before + /// forcing it to abort. + /// - commit the reading position of every partition where it is ahead of this consumer's own + /// record of what it stored, under every [`AutoCommit`] setting, [`AutoCommit::Disabled`] + /// included. Under auto-commit-on-poll (the default) the poll already committed the whole + /// batch, so this store moves the server offset back to the last message handed over, and + /// the next run resumes right after it instead of after the last batch fetched. + /// - leave the consumer group, if this consumer is a group member. This lets the server give its partitions to + /// the remaining members immediately instead of waiting for the connection to time out. + /// + /// The lifecycle event task is not stopped. It runs until the client shuts down. + /// + /// # Errors + /// + /// Returns `Ok(())` even when the final commits or the group leave failed, since those + /// failures are logged and do not leave anything for the caller to undo. The + /// [`Result`] is part of the signature for forward compatibility. pub async fn shutdown(&mut self) -> Result<(), IggyError> { + // Swap so background tasks see that the consumer got shut down. if self.shutdown.swap(true, ORDERING) { return Ok(()); } info!("Shutting down consumer: {}...", self.consumer_name); - // Drain the background commit tasks while still a group member, - // before leaving below — otherwise a store they send afterward hits - // a group we've already left. + // Drain the background commit tasks while still a group member, before + // leaving below. Otherwise a store they send afterward hits a group + // we've already left. self.background_commit_notify.notify_one(); + + // A background_commit_task exists, if auto_commit is configured with an interval option. if let Some(mut task) = self.background_commit_task.take() && time::timeout(self.offset_drain_timeout.get_duration(), &mut task) .await @@ -1238,11 +1757,17 @@ impl IggyConsumer { ); } + // Drop the sending end of the store offset task to end the `recv_async()` loop in `init()`. + // Offsets in queue will still be committed. This prevents loading additional offsets into a channel + // that is not read anymore. + // Replace with a new (hanging) channel, since `store_offset_sender` is not optional. let (closed_sender, _) = flume::bounded(0); drop(std::mem::replace( &mut self.store_offset_sender, closed_sender, )); + + // This task never sleeps, so no need to notify. if let Some(mut task) = self.store_offset_task.take() && time::timeout(self.offset_drain_timeout.get_duration(), &mut task) .await @@ -1300,6 +1825,10 @@ impl IggyConsumer { } } +/// Wakes the interval commit task so it exits. Commits already queued still go out. +/// +/// Nothing is flushed and the consumer group is not left. Await [`IggyConsumer::shutdown`] first, +/// see [Shutting down](IggyConsumer#shutting-down). impl Drop for IggyConsumer { fn drop(&mut self) { self.shutdown.store(true, ORDERING); diff --git a/core/sdk/src/clients/consumer_builder.rs b/core/sdk/src/clients/consumer_builder.rs index 5f31cdd7a0..87b427d834 100644 --- a/core/sdk/src/clients/consumer_builder.rs +++ b/core/sdk/src/clients/consumer_builder.rs @@ -92,7 +92,9 @@ impl IggyConsumerBuilder { Self { topic, ..self } } - /// Sets the partition identifier. + /// Sets the partition to read. `None` lets a consumer group read its assigned partitions and + /// makes the server read partition `0` for a standalone consumer. `Some(n)` on a group member + /// pins every poll to that partition instead of the assignment. pub fn partition(self, partition: Option) -> Self { Self { partition, ..self } } @@ -105,7 +107,7 @@ impl IggyConsumerBuilder { } } - /// Sets the batch size for polling messages. + /// Sets how many messages one poll request fetches at most. Defaults to 1000. pub fn batch_length(self, batch_length: u32) -> Self { Self { batch_length, @@ -121,6 +123,7 @@ impl IggyConsumerBuilder { } } + /// Same as [`auto_commit`](Self::auto_commit) with [`AutoCommit::Disabled`]. pub fn commit_failed_messages(self) -> Self { Self { auto_commit: AutoCommit::Disabled, From bbfad594efebdf8e08da5c684c59d5628d308af6 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Tue, 1 Sep 2026 23:49:48 +0200 Subject: [PATCH 039/182] chore(deps): Bump taiki-e/install-action from 2.86.3 to 2.86.7 in the github-actions group (#4028) --- .github/workflows/_common.yml | 2 +- .github/workflows/coverage-baseline.yml | 4 ++-- .github/workflows/post-merge.yml | 2 +- 3 files changed, 4 insertions(+), 4 deletions(-) diff --git a/.github/workflows/_common.yml b/.github/workflows/_common.yml index f9e75dab39..b6681b6bfb 100644 --- a/.github/workflows/_common.yml +++ b/.github/workflows/_common.yml @@ -93,7 +93,7 @@ jobs: run: echo "version=$(cat .github/config/hawkeye.version)" >> "$GITHUB_OUTPUT" - name: Install HawkEye - uses: taiki-e/install-action@v2.86.3 + uses: taiki-e/install-action@v2.86.7 with: tool: hawkeye@${{ steps.hawkeye-version.outputs.version }} diff --git a/.github/workflows/coverage-baseline.yml b/.github/workflows/coverage-baseline.yml index 08f878c190..9daa683559 100644 --- a/.github/workflows/coverage-baseline.yml +++ b/.github/workflows/coverage-baseline.yml @@ -104,7 +104,7 @@ jobs: free-disk-space-aggressive: "true" - name: Install cargo-llvm-cov - uses: taiki-e/install-action@v2.86.3 + uses: taiki-e/install-action@v2.86.7 with: tool: cargo-llvm-cov @@ -299,7 +299,7 @@ jobs: save-cache: "false" - name: Install cargo-llvm-cov - uses: taiki-e/install-action@v2.86.3 + uses: taiki-e/install-action@v2.86.7 with: tool: cargo-llvm-cov diff --git a/.github/workflows/post-merge.yml b/.github/workflows/post-merge.yml index f45a4b64d5..8d6c76007e 100644 --- a/.github/workflows/post-merge.yml +++ b/.github/workflows/post-merge.yml @@ -65,7 +65,7 @@ jobs: # preinstalled cargo (version pinned by rust-toolchain.toml) and skips the # heavyweight build-cache restore. - name: Install cargo-rail - uses: taiki-e/install-action@v2.86.3 + uses: taiki-e/install-action@v2.86.7 with: tool: cargo-rail From e7f03879d5c26aa4243263c0a3ae1fbb8e5bd35c Mon Sep 17 00:00:00 2001 From: Justin Mclean Date: Wed, 2 Sep 2026 16:25:48 +1000 Subject: [PATCH 040/182] docs: wait for the issue to be assigned, not for a label (#4032) --- CONTRIBUTING.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index a8fac6a2f8..f4540816f6 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -6,7 +6,7 @@ Every new PR that introduces new functionality must link to an approved issue. PRs without one may be closed at maintainer's discretion. 1. Create an issue or comment under existing -2. Wait for maintainer approval (`good-first-issue` label or comment) +2. Wait for the issue to be assigned to you - Maintainer may request for more details or a different approach 3. Then code From 38d58ca13243f2ef81b2758547f55ae54d6c5f2e Mon Sep 17 00:00:00 2001 From: Hubert Gruszecki Date: Wed, 2 Sep 2026 10:56:28 +0200 Subject: [PATCH 041/182] refactor(server): split bootstrap.rs into the boot/ tree (#4031) --- core/integration/src/harness/config/common.rs | 11 + core/integration/src/harness/config/server.rs | 5 + core/integration/src/harness/handle/server.rs | 107 +- .../src/harness/orchestrator/builder.rs | 15 +- core/integration/src/harness/port_reserver.rs | 8 +- .../tests/server/http_view_header.rs | 228 + core/integration/tests/server/mod.rs | 6 + .../tests/server/port_discovery.rs | 118 + core/metadata/src/impls/metadata.rs | 2 +- core/server/src/boot/credentials.rs | 502 ++ core/server/src/boot/handoff.rs | 434 ++ core/server/src/boot/listeners.rs | 636 ++ core/server/src/boot/mod.rs | 1038 ++++ core/server/src/boot/recovery.rs | 1388 +++++ core/server/src/{ => boot}/systemd.rs | 0 core/server/src/boot/threads.rs | 708 +++ core/server/src/boot/topology.rs | 729 +++ core/server/src/bootstrap.rs | 5146 ----------------- core/server/src/cluster_meta.rs | 89 +- core/server/src/config_writer.rs | 75 +- core/server/src/dispatch/mod.rs | 3 +- core/server/src/http.rs | 150 +- core/server/src/http/error.rs | 3 +- core/server/src/http/forward.rs | 20 +- core/server/src/http/state.rs | 2 +- core/server/src/http/tls.rs | 10 +- core/server/src/lib.rs | 10 +- core/server/src/main.rs | 6 +- core/server/src/partition_helpers.rs | 2 +- core/server/src/server_error.rs | 6 +- core/server/src/shell.rs | 4 +- core/simulator/src/replica.rs | 2 +- 32 files changed, 6165 insertions(+), 5298 deletions(-) create mode 100644 core/integration/tests/server/http_view_header.rs create mode 100644 core/integration/tests/server/port_discovery.rs create mode 100644 core/server/src/boot/credentials.rs create mode 100644 core/server/src/boot/handoff.rs create mode 100644 core/server/src/boot/listeners.rs create mode 100644 core/server/src/boot/mod.rs create mode 100644 core/server/src/boot/recovery.rs rename core/server/src/{ => boot}/systemd.rs (100%) create mode 100644 core/server/src/boot/threads.rs create mode 100644 core/server/src/boot/topology.rs delete mode 100644 core/server/src/bootstrap.rs diff --git a/core/integration/src/harness/config/common.rs b/core/integration/src/harness/config/common.rs index 9a7bad383f..a008f435a9 100644 --- a/core/integration/src/harness/config/common.rs +++ b/core/integration/src/harness/config/common.rs @@ -15,6 +15,7 @@ // specific language governing permissions and limitations // under the License. +use std::net::{IpAddr, Ipv4Addr, Ipv6Addr}; use std::path::PathBuf; #[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] @@ -24,6 +25,16 @@ pub enum IpAddrKind { V6, } +impl IpAddrKind { + /// The loopback address of this family, which every harness listener binds. + pub fn loopback(self) -> IpAddr { + match self { + Self::V4 => IpAddr::V4(Ipv4Addr::LOCALHOST), + Self::V6 => IpAddr::V6(Ipv6Addr::LOCALHOST), + } + } +} + #[derive(Debug, Clone)] pub struct EncryptionConfig { pub key: String, diff --git a/core/integration/src/harness/config/server.rs b/core/integration/src/harness/config/server.rs index 07c7f8e940..94d7550119 100644 --- a/core/integration/src/harness/config/server.rs +++ b/core/integration/src/harness/config/server.rs @@ -38,6 +38,11 @@ pub struct TestServerConfig { pub extra_envs: HashMap, #[builder(into)] pub executable_path: Option, + /// Bind every enabled transport to port 0 and discover the bound addresses + /// from `runtime/current_config.toml` instead of pre-reserving ports. + /// Single node only: a cluster roster names every port before boot. + #[builder(default)] + pub ephemeral_ports: bool, } impl Default for TestServerConfig { diff --git a/core/integration/src/harness/handle/server.rs b/core/integration/src/harness/handle/server.rs index 8221a50a31..da6b75af33 100644 --- a/core/integration/src/harness/handle/server.rs +++ b/core/integration/src/harness/handle/server.rs @@ -42,6 +42,10 @@ use toml::Value; const SLEEP_INTERVAL_MS: u64 = 20; const MAX_PORT_WAIT_DURATION_S: u64 = 60; const TEST_VERBOSITY_ENV_VAR: &str = "IGGY_TEST_VERBOSE"; +/// The server truncates the dump and then writes it in one go, so a read in +/// between parses as valid TOML with sections or keys missing; readiness +/// retries on this message. +const INCOMPLETE_DUMP_MESSAGE: &str = "Failed to parse server config: the dump is incomplete"; #[derive(Debug, Clone)] struct ServerProtocolAddr { @@ -410,6 +414,30 @@ impl ServerHandle { return Ok(()); } + // Ephemeral ports: every enabled transport binds `:0` and + // `verify_bound_ports` adopts what the OS chose from the dumped runtime + // config. Ahead of the restart branch so a restart discovers afresh + // instead of pinning a port the OS may have handed out elsewhere since. + if self.config.ephemeral_ports { + self.addrs = ServerProtocolAddr::empty(); + let unbound = SocketAddr::new(self.config.ip_kind.loopback(), 0).to_string(); + self.envs + .insert("IGGY_TCP_ADDRESS".to_string(), unbound.clone()); + if self.config.http_enabled { + self.envs + .insert("IGGY_HTTP_ADDRESS".to_string(), unbound.clone()); + } + if self.config.quic_enabled { + self.envs + .insert("IGGY_QUIC_ADDRESS".to_string(), unbound.clone()); + } + if self.config.websocket_enabled { + self.envs + .insert("IGGY_WEBSOCKET_ADDRESS".to_string(), unbound); + } + return Ok(()); + } + // Restart case: reuse existing addresses to maintain consistency if self.addrs.tcp.is_some() || self.addrs.http.is_some() @@ -509,7 +537,11 @@ impl ServerHandle { }) } - fn verify_bound_ports(&self, config_path: &Path) -> Result<(), TestBinaryError> { + /// Reconcile the dumped runtime config with the addresses this handle + /// holds: a pre-reserved address must be the one the server bound, and an + /// address the handle does not know yet (`ephemeral_ports`) is adopted + /// from the dump. + fn verify_bound_ports(&mut self, config_path: &Path) -> Result<(), TestBinaryError> { let content = fs::read_to_string(config_path).map_err(|e| TestBinaryError::InvalidState { message: format!( @@ -523,32 +555,21 @@ impl ServerHandle { message: format!("Failed to parse server config: {e}"), })?; - let bound_tcp = Self::extract_address(&config, "tcp"); - let bound_http = Self::extract_address(&config, "http"); - let bound_quic = Self::extract_address(&config, "quic"); - let bound_websocket = Self::extract_address(&config, "websocket"); - + let transports = [ + ("TCP", "tcp", &mut self.addrs.tcp), + ("HTTP", "http", &mut self.addrs.http), + ("QUIC", "quic", &mut self.addrs.quic), + ("WebSocket", "websocket", &mut self.addrs.websocket), + ]; let mut mismatches = Vec::new(); - - if let (Some(expected), Some(bound)) = (self.addrs.tcp, bound_tcp) - && expected != bound - { - mismatches.push(format!("TCP: expected {expected}, got {bound}")); - } - if let (Some(expected), Some(bound)) = (self.addrs.http, bound_http) - && expected != bound - { - mismatches.push(format!("HTTP: expected {expected}, got {bound}")); - } - if let (Some(expected), Some(bound)) = (self.addrs.quic, bound_quic) - && expected != bound - { - mismatches.push(format!("QUIC: expected {expected}, got {bound}")); - } - if let (Some(expected), Some(bound)) = (self.addrs.websocket, bound_websocket) - && expected != bound - { - mismatches.push(format!("WebSocket: expected {expected}, got {bound}")); + for (label, section, expected) in transports { + match (*expected, Self::extract_address(&config, section)?) { + (Some(expected), Some(bound)) if expected != bound => { + mismatches.push(format!("{label}: expected {expected}, got {bound}")); + } + (None, Some(bound)) => *expected = Some(bound), + _ => {} + } } if !mismatches.is_empty() { @@ -563,8 +584,38 @@ impl ServerHandle { Ok(()) } - fn extract_address(config: &Value, protocol: &str) -> Option { - config.get(protocol)?.get("address")?.as_str()?.parse().ok() + /// The address a transport section reports, when it marks the transport + /// enabled: a disabled transport keeps its configured address in the dump + /// without ever binding it. Every section is serialized with both keys + /// whether or not the transport is enabled, so a section or key still + /// absent is a dump being written, not a disabled transport. + fn extract_address( + config: &Value, + protocol: &str, + ) -> Result, TestBinaryError> { + let incomplete = || TestBinaryError::InvalidState { + message: INCOMPLETE_DUMP_MESSAGE.to_string(), + }; + let section = config.get(protocol).ok_or_else(incomplete)?; + let enabled = section + .get("enabled") + .and_then(Value::as_bool) + .ok_or_else(incomplete)?; + if !enabled { + return Ok(None); + } + let address = section + .get("address") + .and_then(Value::as_str) + .ok_or_else(incomplete)?; + address + .parse() + .map(Some) + .map_err(|error| TestBinaryError::InvalidState { + message: format!( + "Server config {protocol}.address {address:?} is not a socket address: {error}" + ), + }) } fn start_watchdog(&mut self) { diff --git a/core/integration/src/harness/orchestrator/builder.rs b/core/integration/src/harness/orchestrator/builder.rs index 23de6d23a8..fbd855631e 100644 --- a/core/integration/src/harness/orchestrator/builder.rs +++ b/core/integration/src/harness/orchestrator/builder.rs @@ -254,6 +254,16 @@ fn build_servers( return Ok(vec![ServerHandle::with_config(config, context.clone())]); } + // The roster below names every node's ports before any node boots, which + // is exactly what ephemeral ports cannot do. + if config.ephemeral_ports { + return Err(TestBinaryError::InvalidState { + message: format!( + "ephemeral_ports needs a single node, but the harness has {node_count} nodes" + ), + }); + } + // Multi-node cluster: pre-reserve all ports let cluster_ports = ClusterPortReserver::reserve(node_count, config.ip_kind, &config)?; let all_addrs = cluster_ports.all_addresses(); @@ -317,10 +327,7 @@ fn build_cluster_envs( ) -> HashMap { let mut envs = HashMap::new(); - let loopback = match ip_kind { - IpAddrKind::V4 => "127.0.0.1", - IpAddrKind::V6 => "::1", - }; + let loopback = ip_kind.loopback(); envs.insert("IGGY_CLUSTER_ENABLED".to_string(), "true".to_string()); envs.insert("IGGY_CLUSTER_NAME".to_string(), cluster_name.to_string()); diff --git a/core/integration/src/harness/port_reserver.rs b/core/integration/src/harness/port_reserver.rs index 395d8dc232..6b096325f2 100644 --- a/core/integration/src/harness/port_reserver.rs +++ b/core/integration/src/harness/port_reserver.rs @@ -46,7 +46,7 @@ use crate::harness::config::{IpAddrKind, TestServerConfig}; use crate::harness::error::TestBinaryError; use std::fs::{File, OpenOptions, TryLockError}; -use std::net::{IpAddr, Ipv4Addr, Ipv6Addr, SocketAddr}; +use std::net::SocketAddr; use std::path::{Path, PathBuf}; use std::sync::OnceLock; @@ -262,11 +262,7 @@ impl SlotGuard { let port = self.base_port + self.next_offset; self.next_offset += 1; - let ip: IpAddr = match ip_kind { - IpAddrKind::V4 => Ipv4Addr::LOCALHOST.into(), - IpAddrKind::V6 => Ipv6Addr::LOCALHOST.into(), - }; - Ok(SocketAddr::new(ip, port)) + Ok(SocketAddr::new(ip_kind.loopback(), port)) } } diff --git a/core/integration/tests/server/http_view_header.rs b/core/integration/tests/server/http_view_header.rs new file mode 100644 index 0000000000..8ed175a9b2 --- /dev/null +++ b/core/integration/tests/server/http_view_header.rs @@ -0,0 +1,228 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! The `iggy-view` response header: the serving node's current VSR view, +//! stamped on the success and redirect responses of authenticated flows and +//! withheld wherever it could reach a caller that proved no credential. Raw +//! `reqwest`, because the header itself is the contract. + +use std::future::Future; +use std::time::Instant; + +use iggy::prelude::*; +use integration::harness::TestHarness; +use integration::iggy_harness; +use reqwest::{Response, StatusCode}; +use serde_json::json; +use tokio::time::sleep; + +use crate::server::http_client::{HttpClient, LOGIN_RETRY_INTERVAL, LOGIN_TIMEOUT}; + +const VIEW_HEADER: &str = "iggy-view"; + +/// The view number a response carries, if any. A header that is present but +/// not a view number is a contract violation, not an absence. +fn view_number(response: &Response) -> Option { + response.headers().get(VIEW_HEADER).map(|value| { + value + .to_str() + .expect("iggy-view must be ASCII") + .parse() + .expect("iggy-view must be a view number") + }) +} + +#[iggy_harness(cluster_nodes = 1)] +async fn given_an_authenticated_request_when_it_succeeds_should_carry_the_iggy_view_header( + harness: &TestHarness, +) { + let http = HttpClient::login_root(harness).await; + + let response = http.get("/streams").await; + + assert_eq!(response.status(), StatusCode::OK); + assert!( + view_number(&response).is_some(), + "a successful authenticated response must carry {VIEW_HEADER}" + ); +} + +#[iggy_harness(cluster_nodes = 1)] +async fn given_a_request_without_credentials_when_it_is_rejected_should_omit_the_iggy_view_header( + harness: &TestHarness, +) { + let http = HttpClient::login_root(harness).await; + + let response = http.get_anonymous("/streams").await; + + assert_eq!(response.status(), StatusCode::UNAUTHORIZED); + assert!( + view_number(&response).is_none(), + "an error response must not leak {VIEW_HEADER} to an unauthenticated caller" + ); +} + +#[iggy_harness(cluster_nodes = 1)] +async fn given_the_ping_route_when_it_succeeds_should_omit_the_iggy_view_header( + harness: &TestHarness, +) { + let http = HttpClient::login_root(harness).await; + + let response = http.get_anonymous("/ping").await; + + assert_eq!(response.status(), StatusCode::OK); + assert!( + view_number(&response).is_none(), + "the pre-auth probe must not carry {VIEW_HEADER}" + ); +} + +/// `http://host:port` of a harness node's HTTP listener. +fn node_url(harness: &TestHarness, node: usize) -> String { + let addr = harness.node(node).http_addr().expect("node http address"); + format!("http://{addr}") +} + +/// Harness indexes of the node the roster marks `Leader` and of one it marks +/// `Follower`. The harness emits the roster in node order, so a roster +/// position is a harness index. Every node reads `Follower` until shard 0 +/// publishes its first view, so the roster is polled within the shared +/// warmup budget until it marks a leader. +async fn leader_and_follower(harness: &TestHarness) -> (usize, usize) { + let client = harness + .root_client_for_node(0) + .await + .expect("connect to node 0"); + let deadline = Instant::now() + LOGIN_TIMEOUT; + loop { + let metadata = client + .get_cluster_metadata() + .await + .expect("get cluster metadata"); + let position = + |role: ClusterNodeRole| metadata.nodes.iter().position(|node| node.role == role); + if let (Some(leader), Some(follower)) = ( + position(ClusterNodeRole::Leader), + position(ClusterNodeRole::Follower), + ) { + return (leader, follower); + } + assert!( + Instant::now() < deadline, + "the roster did not mark a leader within {LOGIN_TIMEOUT:?}, got {metadata}" + ); + sleep(LOGIN_RETRY_INTERVAL).await; + } +} + +/// Repeat `request` while the follower answers 503, which it does until it +/// can resolve the primary from its own view; bounded by the shared warmup +/// budget, as cluster_metadata_vsr does. A 503 is the retry-safe class: the +/// request provably never entered a pipeline. +async fn until_primary_resolved(request: F) -> Response +where + F: Fn() -> Fut, + Fut: Future, +{ + let deadline = Instant::now() + LOGIN_TIMEOUT; + loop { + let response = request().await; + if response.status() != StatusCode::SERVICE_UNAVAILABLE { + return response; + } + assert!( + Instant::now() < deadline, + "follower did not resolve the primary within {LOGIN_TIMEOUT:?}" + ); + sleep(LOGIN_RETRY_INTERVAL).await; + } +} + +/// The view the primary stamps on its own successful response: the value a +/// follower's redirect or relay must agree with. +async fn primary_view(primary: &HttpClient) -> u64 { + view_number(&primary.get("/streams").await).expect("the primary stamps its own view") +} + +/// Three nodes, the smallest cluster that keeps a quorum through a leader +/// change, one shard each so every request is served by shard 0, where the +/// metadata consensus lives (see cluster_metadata_vsr.rs). No `http.jwt` +/// secret and no `cluster.auth`: bearers are node-local, forwarding is off, +/// and a follower answers a linearizable read with the 307 primary redirect. +#[iggy_harness(cluster_nodes = 3, server(system.sharding.cpu_allocation = "0..1"))] +async fn given_a_follower_when_it_redirects_a_linearizable_read_should_carry_the_iggy_view_header( + harness: &TestHarness, +) { + let (leader, follower) = leader_and_follower(harness).await; + let primary = HttpClient::login_root_no_redirect(node_url(harness, leader)).await; + let http = HttpClient::login_root_no_redirect(node_url(harness, follower)).await; + // Views only grow, so bracketing the request lets a view change in flight + // widen the accepted range instead of failing the test. + let view_before = primary_view(&primary).await; + + let response = until_primary_resolved(|| http.get("/streams?consistency=linearizable")).await; + let view_after = primary_view(&primary).await; + + assert_eq!( + response.status(), + StatusCode::TEMPORARY_REDIRECT, + "a keyless follower must redirect a linearizable read to the primary" + ); + let view = view_number(&response); + assert!( + view.is_some_and(|view| (view_before..=view_after).contains(&view)), + "the primary redirect must carry the cluster's current view in {VIEW_HEADER}: \ + got {view:?}, expected within {view_before}..={view_after}" + ); +} + +/// The same three-node shape with cluster-wide bearer key material, which +/// switches follower-to-primary forwarding on: a control-plane write posted +/// to the follower is answered by the primary, and the relayed response must +/// carry the header the primary stamped. +#[iggy_harness( + cluster_nodes = 3, + server( + system.sharding.cpu_allocation = "0..1", + http.jwt.encoding_secret = "0123456789abcdef0123456789abcdef", + http.jwt.decoding_secret = "0123456789abcdef0123456789abcdef" + ) +)] +async fn given_a_follower_when_it_relays_a_forwarded_write_should_carry_the_iggy_view_header( + harness: &TestHarness, +) { + let (leader, follower) = leader_and_follower(harness).await; + let primary = HttpClient::login_root_no_redirect(node_url(harness, leader)).await; + let http = HttpClient::login_root_no_redirect(node_url(harness, follower)).await; + let body = json!({ "name": "forwarded-stream" }); + let view_before = primary_view(&primary).await; + + let response = until_primary_resolved(|| http.post_json("/streams", &body)).await; + let view_after = primary_view(&primary).await; + + assert_eq!( + response.status(), + StatusCode::OK, + "the follower must relay the primary's answer to a control-plane write" + ); + let view = view_number(&response); + assert!( + view.is_some_and(|view| (view_before..=view_after).contains(&view)), + "a relayed response must carry the view the primary stamped in {VIEW_HEADER}: \ + got {view:?}, expected within {view_before}..={view_after}" + ); +} diff --git a/core/integration/tests/server/mod.rs b/core/integration/tests/server/mod.rs index b8c832e474..a1da26b1e3 100644 --- a/core/integration/tests/server/mod.rs +++ b/core/integration/tests/server/mod.rs @@ -48,11 +48,17 @@ mod http_rbac; // End-to-end HTTPS: the server serves the REST listener over TLS and negotiates // HTTP/2 via ALPN. mod http_tls; +// The iggy-view response header: on authenticated success and redirect +// responses only, never on errors or /ping, relayed from the primary. +mod http_view_header; // Binary GetClusterMetadata must serve the real roster from a VSR cluster. mod cluster_metadata_vsr; // A declared node.advertised_address outranks the bind address a // cluster-disabled server would otherwise publish. mod cluster_metadata_advertised; +// Listeners bound to :0 must publish the OS-chosen ports through the runtime +// config dump and both cluster-metadata spines. +mod port_discovery; // A metadata view change must persist the advanced view and recover it from disk // across a replica restart. mod cluster_view_durability_vsr; diff --git a/core/integration/tests/server/port_discovery.rs b/core/integration/tests/server/port_discovery.rs new file mode 100644 index 0000000000..812aaec7dc --- /dev/null +++ b/core/integration/tests/server/port_discovery.rs @@ -0,0 +1,118 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Ephemeral listener ports end to end. The harness binds every enabled +//! transport to `:0` and learns the OS-chosen ports from the dumped runtime +//! config, which makes this the tripwire for two server contracts at once: +//! the dump must carry every bound address (the HTTP one included, or the +//! harness could not even find that listener), and cluster metadata must +//! report the ports as bound rather than as configured, on the binary and +//! the HTTP spine alike. + +use std::collections::HashMap; + +use iggy::prelude::*; +use integration::harness::{TestHarness, TestServerConfig}; +use reqwest::StatusCode; + +use crate::server::http_client::HttpClient; + +/// One node with clustering off: a cluster roster names every port before +/// boot and cannot be ephemeral, and the cluster-disabled roster is the one +/// that synthesizes its self node from the bound ports. +#[tokio::test] +#[serial_test::parallel] +async fn given_ephemeral_ports_when_getting_cluster_metadata_should_report_bound_ports() { + let mut harness = TestHarness::builder() + .server( + TestServerConfig::builder() + .ephemeral_ports(true) + .extra_envs(HashMap::from([( + "IGGY_CLUSTER_ENABLED".to_string(), + "false".to_string(), + )])) + .build(), + ) + .cluster_nodes(1) + .build() + .expect("build harness"); + harness + .start() + .await + .expect("start server on ephemeral ports"); + + let server = harness.server(); + let discovered = [ + ("tcp", server.tcp_addr()), + ("http", server.http_addr()), + ("quic", server.quic_addr()), + ("websocket", server.websocket_addr()), + ] + .map(|(transport, addr)| { + let addr = addr.unwrap_or_else(|| panic!("{transport} address must be discovered")); + assert_ne!( + addr.port(), + 0, + "{transport} must report the port the OS chose, not the configured 0" + ); + addr.port() + }); + let [tcp, http, quic, websocket] = discovered; + + let client = server + .tcp_client() + .expect("tcp client") + .with_root_login() + .connect() + .await + .expect("connect over the discovered tcp port"); + let binary = client + .get_cluster_metadata() + .await + .expect("get cluster metadata"); + assert_eq!( + binary.nodes.len(), + 1, + "a cluster-disabled server reports itself alone, got {binary}" + ); + let binary_endpoints = &binary.nodes[0].endpoints; + assert_eq!(binary_endpoints.tcp, tcp, "binary metadata tcp port"); + assert_eq!(binary_endpoints.http, http, "binary metadata http port"); + assert_eq!(binary_endpoints.quic, quic, "binary metadata quic port"); + assert_eq!( + binary_endpoints.websocket, websocket, + "binary metadata websocket port" + ); + + let session = HttpClient::login_root(&harness).await; + let response = session.get("/cluster/metadata").await; + assert_eq!(response.status(), StatusCode::OK); + let over_http: ClusterMetadata = response.json().await.expect("decode cluster metadata"); + assert_eq!( + over_http.nodes.len(), + 1, + "HTTP metadata must report the same single node, got {over_http}" + ); + let http_endpoints = &over_http.nodes[0].endpoints; + assert_eq!(http_endpoints.tcp, tcp, "HTTP metadata tcp port"); + assert_eq!(http_endpoints.http, http, "HTTP metadata http port"); + assert_eq!(http_endpoints.quic, quic, "HTTP metadata quic port"); + assert_eq!( + http_endpoints.websocket, websocket, + "HTTP metadata websocket port" + ); +} diff --git a/core/metadata/src/impls/metadata.rs b/core/metadata/src/impls/metadata.rs index 31ce620cda..0c19eb1563 100644 --- a/core/metadata/src/impls/metadata.rs +++ b/core/metadata/src/impls/metadata.rs @@ -688,7 +688,7 @@ pub struct IggyMetadata { /// the WAL at all. They receive a `MetadataHandoff::Waiter` factory /// bundle from shard 0 over the bootstrap broadcast channel and /// reconstruct `mux_stm` from the in-memory snapshot it carries (see - /// `server/src/bootstrap.rs` `await_metadata_bundle` / + /// `server/src/boot/handoff.rs` `await_metadata_bundle` / /// `broadcast_metadata_bundle`). pub journal: Option, /// `Some` on shard 0, `None` on other shards. diff --git a/core/server/src/boot/credentials.rs b/core/server/src/boot/credentials.rs new file mode 100644 index 0000000000..aeff093868 --- /dev/null +++ b/core/server/src/boot/credentials.rs @@ -0,0 +1,502 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Root user credentials and TLS/PSK material loaded at boot. + +use crate::boot::topology::TcpTopology; +use crate::server_error::ServerError; +use crate::shell::ServerMuxStateMachine; +use configs::server::ServerConfig; +use iggy_common::defaults::{ + DEFAULT_ROOT_PASSWORD, DEFAULT_ROOT_USERNAME, MAX_PASSWORD_LENGTH, MAX_USERNAME_LENGTH, + MIN_PASSWORD_LENGTH, MIN_USERNAME_LENGTH, +}; +use message_bus::replica::auth::ReplicaAuth; +use message_bus::replica::handshake::ReplicaTlsCtx; +use message_bus::replica::io as replica_io; +use message_bus::transports::tls::{ + AcceptAnyServerCert, REPLICA_ALPN, TlsServerCredentials, load_ca_pem, load_pem, + self_signed_for_loopback, +}; +use metadata::impls::metadata::StreamsFrontend; +use rustls::pki_types::ServerName; +use server_common::crypto; +use std::collections::HashMap; +use std::env; +use std::path::Path; +use std::sync::Arc; +use tracing::{info, warn}; + +const IGGY_ROOT_USERNAME_ENV: &str = "IGGY_ROOT_USERNAME"; + +const IGGY_ROOT_PASSWORD_ENV: &str = "IGGY_ROOT_PASSWORD"; + +pub(in crate::boot) fn ensure_default_root_user(mux_stm: &ServerMuxStateMachine) { + if !mux_stm.users().read(|users| users.items.is_empty()) { + return; + } + + let (username, password_hash) = create_root_credentials(); + mux_stm.users().ensure_root_user(&username, &password_hash); +} + +/// Apply `--with-default-root-credentials`. +/// +/// Fills in whichever of `IGGY_ROOT_USERNAME_ENV` / +/// `IGGY_ROOT_PASSWORD_ENV` the operator did not export, so the flag is +/// exactly the sugar for setting both by hand and the environment keeps +/// winning over it. +/// +/// # Safety +/// +/// Mutates the process environment, so the caller must still be +/// single-threaded. +pub unsafe fn apply_default_root_credentials(enabled: bool) { + if !enabled { + return; + } + + let username_set = env::var(IGGY_ROOT_USERNAME_ENV).is_ok(); + let password_set = env::var(IGGY_ROOT_PASSWORD_ENV).is_ok(); + if username_set && password_set { + warn!( + "--with-default-root-credentials ignored: {IGGY_ROOT_USERNAME_ENV} and \ + {IGGY_ROOT_PASSWORD_ENV} are already set" + ); + return; + } + + // SAFETY: single-threaded caller, per this function's contract. + unsafe { + if !username_set { + env::set_var(IGGY_ROOT_USERNAME_ENV, DEFAULT_ROOT_USERNAME); + } + if !password_set { + env::set_var(IGGY_ROOT_PASSWORD_ENV, DEFAULT_ROOT_PASSWORD); + } + } + warn!( + "--with-default-root-credentials: a newly created root user will use the \ + well-known development credentials; INSECURE outside development" + ); +} + +/// Resolve the root user credentials from `IGGY_ROOT_USERNAME` / +/// `IGGY_ROOT_PASSWORD`, falling back to the default username with a +/// generated password. +/// +/// Returns `(username, password_hash)`; the plaintext password never +/// leaves this function. +fn create_root_credentials() -> (String, String) { + if let Some((username, password)) = root_credentials_from_env() { + info!("Using the custom root user credentials."); + return (username, crypto::hash_password(&password)); + } + + info!("Using the default root user credentials..."); + let password = crypto::generate_secret(20..40); + // Through tracing, not stdout: this is the only time the operator can read + // the password, so it has to reach the log file too. + warn!("Generated root user password: {password}"); + ( + DEFAULT_ROOT_USERNAME.to_string(), + crypto::hash_password(&password), + ) +} + +/// The credentials the operator supplied, `None` when neither variable is +/// set. A half-set pair never reaches here: [`validate_root_credentials`] +/// rejects it at boot. +fn root_credentials_from_env() -> Option<(String, String)> { + match ( + env::var(IGGY_ROOT_USERNAME_ENV), + env::var(IGGY_ROOT_PASSWORD_ENV), + ) { + (Ok(username), Ok(password)) => Some((username, password)), + _ => None, + } +} + +/// Reject root-credential misconfiguration before any shard thread exists. +/// +/// Shard 0 seeds the root user from inside `recover`'s baseline closure, +/// which cannot fail, so every operator-facing check has to run here or it +/// would have to panic a shard thread instead. +pub(in crate::boot) fn validate_root_credentials_env( + config: &ServerConfig, +) -> Result<(), ServerError> { + // `recover` creates the metadata directory, so its absence is what tells a + // first cluster boot (root must come out identical on every replica, hence + // explicit credentials) apart from a restart that recovers the root user it + // already stored. `--fresh` has already wiped by this point, so a wiped + // replica is correctly treated as a first boot. + let fresh_cluster = config.cluster.enabled + && !Path::new(&config.system.path) + .join(metadata::impls::METADATA_DIR) + .exists(); + + validate_root_credentials( + fresh_cluster, + env::var(IGGY_ROOT_USERNAME_ENV).ok().as_deref(), + env::var(IGGY_ROOT_PASSWORD_ENV).ok().as_deref(), + ) +} + +fn validate_root_credentials( + explicit_required: bool, + username: Option<&str>, + password: Option<&str>, +) -> Result<(), ServerError> { + match (username, password) { + (Some(username), Some(password)) => { + validate_credential_length( + IGGY_ROOT_USERNAME_ENV, + username, + MIN_USERNAME_LENGTH, + MAX_USERNAME_LENGTH, + )?; + validate_credential_length( + IGGY_ROOT_PASSWORD_ENV, + password, + MIN_PASSWORD_LENGTH, + MAX_PASSWORD_LENGTH, + ) + } + (Some(_), None) => Err(ServerError::RootCredentialsIncomplete { + provided_env: IGGY_ROOT_USERNAME_ENV, + missing_env: IGGY_ROOT_PASSWORD_ENV, + }), + (None, Some(_)) => Err(ServerError::RootCredentialsIncomplete { + provided_env: IGGY_ROOT_PASSWORD_ENV, + missing_env: IGGY_ROOT_USERNAME_ENV, + }), + (None, None) if explicit_required => Err(ServerError::ClusterRootCredentialsRequired { + username_env: IGGY_ROOT_USERNAME_ENV, + password_env: IGGY_ROOT_PASSWORD_ENV, + }), + (None, None) => Ok(()), + } +} + +fn validate_credential_length( + env_name: &'static str, + value: &str, + min: usize, + max: usize, +) -> Result<(), ServerError> { + if (min..=max).contains(&value.len()) { + Ok(()) + } else { + Err(ServerError::RootCredentialLength { + env_name, + length: value.len(), + min, + max, + }) + } +} + +/// Build the replica auth context from cluster config. Returns `None` when the +/// cluster or replica auth is disabled, keeping the handshake in legacy mode. +/// Only the derived MAC keys are carried onward in [`ReplicaAuth`]; the raw +/// secrets (masked in config logs via `config_env(secret)`) are read here only +/// to derive them. A non-empty `previous_shared_secret` opens the verify-only +/// rotation acceptance window (see the [`ReplicaAuth`] rustdoc for the rolling +/// rotation procedure). `ClusterConfig::validate` guarantees a non-empty +/// secret whenever both `cluster.enabled` and `cluster.auth.enabled` are set +/// (validate early-returns `Ok` while `cluster.enabled` is false). +pub(in crate::boot) fn load_replica_auth(config: &ServerConfig) -> Option { + if !config.cluster.enabled || !config.cluster.auth.enabled { + return None; + } + let auth = ReplicaAuth::new(config.cluster.auth.shared_secret.as_bytes()); + let previous_shared_secret = &config.cluster.auth.previous_shared_secret; + if previous_shared_secret.is_empty() { + return Some(auth); + } + Some(auth.with_previous_secret(previous_shared_secret.as_bytes())) +} + +/// Build the replica TLS context from cluster config. Returns `None` when +/// the cluster or replica TLS is disabled. Every shard calls this once at +/// boot: CA mode re-reads the same PEM files per shard; self-signed mode +/// mints a per-shard throwaway certificate. Neither mode carries client +/// certificates, so TLS authenticates the acceptor only; peer +/// authentication comes from the PSK handshake (`ClusterConfig::validate` +/// enforces `cluster.auth.enabled` whenever `cluster.tls.enabled`). +/// +/// Both rustls configs are TLS 1.3 only with the [`REPLICA_ALPN`] +/// protocol pinned. The dialer's SNI / certificate-verify name for each +/// peer is the roster entry's `ip` field (a hostname or IP literal, the +/// same string the connector dials). +pub(in crate::boot) fn load_replica_tls_ctx( + config: &ServerConfig, + topology: &TcpTopology, +) -> Result, ServerError> { + let tls = &config.cluster.tls; + if !config.cluster.enabled || !tls.enabled { + return Ok(None); + } + let credential_error = |source: std::io::Error| ServerError::ListenerCredentials { + transport: "cluster.tls", + source, + }; + + let credentials = if tls.self_signed { + warn_ignored_certificate_files("cluster.tls", &tls.cert_file, &tls.key_file); + let san = config + .cluster + .nodes + .iter() + .find(|node| node.replica_id == topology.self_replica_id) + .map(|node| node.ip.as_str()) + .ok_or_else(|| { + credential_error(std::io::Error::other(format!( + "replica id {} not present in cluster.nodes", + topology.self_replica_id + ))) + })?; + let (cert_chain, key_der) = server_common::generate_self_signed_certificate(san) + .map_err(|error| credential_error(std::io::Error::other(error.to_string())))?; + TlsServerCredentials { + cert_chain, + key_der, + } + } else { + load_pem(Path::new(&tls.cert_file), Path::new(&tls.key_file)).map_err(credential_error)? + }; + + let mut server = + rustls::ServerConfig::builder_with_protocol_versions(&[&rustls::version::TLS13]) + .with_no_client_auth() + .with_single_cert(credentials.cert_chain, credentials.key_der) + .map_err(|error| { + credential_error(std::io::Error::other(format!( + "replica TLS server config rejected credentials: {error}" + ))) + })?; + server.alpn_protocols = vec![REPLICA_ALPN.to_vec()]; + + let client_builder = + rustls::ClientConfig::builder_with_protocol_versions(&[&rustls::version::TLS13]); + let mut client = if tls.self_signed { + client_builder + .dangerous() + .with_custom_certificate_verifier(Arc::new(AcceptAnyServerCert)) + .with_no_client_auth() + } else { + let roots = load_ca_pem(Path::new(&tls.ca_file)).map_err(credential_error)?; + client_builder + .with_root_certificates(Arc::new(roots)) + .with_no_client_auth() + }; + client.alpn_protocols = vec![REPLICA_ALPN.to_vec()]; + + // Keyed by replica id, never by roster position: sparse ids (dynamic + // replica join) would make a positional lookup verify against another + // peer's SNI name. + let peer_names = config + .cluster + .nodes + .iter() + .map(|node| { + let name = ServerName::try_from(node.ip.clone()).map_err(|error| { + credential_error(std::io::Error::new( + std::io::ErrorKind::InvalidInput, + format!( + "cluster node '{}' ip '{}' is not a valid TLS server name: {error}", + node.name, node.ip + ), + )) + })?; + Ok((node.replica_id, name)) + }) + .collect::, ServerError>>()?; + + Ok(Some(ReplicaTlsCtx { + server: Arc::new(server), + client: Arc::new(client), + peer_names, + })) +} + +pub(in crate::boot) fn load_tcp_tls_server_credentials( + config: &ServerConfig, +) -> Result { + let tls = &config.tcp.tls; + if ephemeral_certificate("tcp.tls", tls.self_signed, &tls.cert_file) { + return Ok(self_signed_for_loopback()); + } + + load_pem(Path::new(&tls.cert_file), Path::new(&tls.key_file)).map_err(|source| { + ServerError::ListenerCredentials { + transport: "tcp.tls", + source, + } + }) +} + +pub(in crate::boot) fn load_wss_server_credentials( + config: &ServerConfig, +) -> Result { + let tls = &config.websocket.tls; + if ephemeral_certificate("websocket.tls", tls.self_signed, &tls.cert_file) { + return Ok(self_signed_for_loopback()); + } + + load_pem(Path::new(&tls.cert_file), Path::new(&tls.key_file)).map_err(|source| { + ServerError::ListenerCredentials { + transport: "websocket.tls", + source, + } + }) +} + +pub(in crate::boot) fn load_quic_server_credentials( + config: &ServerConfig, +) -> Result { + let certificate = &config.quic.certificate; + if certificate.self_signed { + warn_ignored_certificate_files( + "quic.certificate", + &certificate.cert_file, + &certificate.key_file, + ); + let (cert_chain, key_der) = server_common::generate_self_signed_certificate("localhost") + .map_err(|error| ServerError::ListenerCredentials { + transport: "quic", + source: std::io::Error::other(error.to_string()), + })?; + return Ok(replica_io::QuicServerCredentials { + cert_chain, + key_der, + }); + } + + let credentials = load_pem( + Path::new(&certificate.cert_file), + Path::new(&certificate.key_file), + ) + .map_err(|source| ServerError::ListenerCredentials { + transport: "quic", + source, + })?; + Ok(replica_io::QuicServerCredentials { + cert_chain: credentials.cert_chain, + key_der: credentials.key_der, + }) +} + +/// Client-listener certificate precedence: `self_signed = true` mints an +/// ephemeral loopback certificate only while `cert_file` is absent from disk. +/// An existing PEM pair wins, so a deployment that lays certificates down +/// serves them without also having to unset the flag - the contract every +/// SDK test lane relies on when it points the server at `core/certs/`. +fn ephemeral_certificate(section: &str, self_signed: bool, cert_file: &str) -> bool { + if !self_signed { + return false; + } + if Path::new(cert_file).exists() { + info!( + "{section}.self_signed = true but cert_file = {cert_file} exists on disk; loading it - remove the file or clear the path to serve an ephemeral certificate" + ); + return false; + } + true +} + +/// `self_signed = true` never reads the PEM pair (cluster and QUIC keep the +/// flag authoritative: their generated certificates carry non-loopback SANs), +/// so a cert path resolving on disk looks active to an operator who never +/// asked for it. +fn warn_ignored_certificate_files(section: &str, cert_file: &str, key_file: &str) { + let found: Vec = [("cert_file", cert_file), ("key_file", key_file)] + .into_iter() + .filter(|(_, path)| Path::new(path).exists()) + .map(|(field, path)| format!("{field} = {path}")) + .collect(); + if found.is_empty() { + return; + } + + warn!( + "{section}.self_signed = true, ignoring certificate files found on disk ({}); set {section}.self_signed = false to load them", + found.join(", ") + ); +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn fresh_cluster_bootstrap_requires_explicit_root_credentials() { + assert!(matches!( + validate_root_credentials(true, None, None), + Err(ServerError::ClusterRootCredentialsRequired { + username_env: IGGY_ROOT_USERNAME_ENV, + password_env: IGGY_ROOT_PASSWORD_ENV, + }) + )); + validate_root_credentials(true, Some("root"), Some("secret")) + .expect("both credentials supplied must satisfy the fresh-cluster guard"); + } + + #[test] + fn single_node_bootstrap_generates_root_credentials_when_unset() { + validate_root_credentials(false, None, None) + .expect("a single node mints its own root password"); + } + + #[test] + fn half_set_root_credentials_are_rejected_in_both_directions() { + assert!(matches!( + validate_root_credentials(false, Some("root"), None), + Err(ServerError::RootCredentialsIncomplete { + provided_env: IGGY_ROOT_USERNAME_ENV, + missing_env: IGGY_ROOT_PASSWORD_ENV, + }) + )); + assert!(matches!( + validate_root_credentials(false, None, Some("secret")), + Err(ServerError::RootCredentialsIncomplete { + provided_env: IGGY_ROOT_PASSWORD_ENV, + missing_env: IGGY_ROOT_USERNAME_ENV, + }) + )); + } + + #[test] + fn out_of_range_root_credentials_are_rejected() { + assert!(matches!( + validate_root_credentials(false, Some(""), Some("secret")), + Err(ServerError::RootCredentialLength { + env_name: IGGY_ROOT_USERNAME_ENV, + length: 0, + .. + }) + )); + let too_long = "x".repeat(MAX_PASSWORD_LENGTH + 1); + assert!(matches!( + validate_root_credentials(false, Some("root"), Some(&too_long)), + Err(ServerError::RootCredentialLength { + env_name: IGGY_ROOT_PASSWORD_ENV, + .. + }) + )); + } +} diff --git a/core/server/src/boot/handoff.rs b/core/server/src/boot/handoff.rs new file mode 100644 index 0000000000..10e8daf5ae --- /dev/null +++ b/core/server/src/boot/handoff.rs @@ -0,0 +1,434 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Cross-shard boot handoff: the metadata bundle broadcast and the +//! listener barrier. + +use crate::server_error::ServerError; +use crate::shell::ServerMetadataBundle; +// `try_send` / `try_recv` resolve through these traits on `MAsyncTx` / +// `MAsyncRx`; the metadata-handoff loops below depend on the +// non-blocking variants for cancel-safe shutdown polling. +use crossfire::{AsyncRxTrait, AsyncTxTrait}; +use std::sync::Arc; +use std::sync::atomic::{AtomicBool, Ordering}; +use std::time::Duration; + +/// Shard-local end of the metadata bundle handoff. +/// +/// Shard 0 owns the WAL writer and runs `recover()` to build the only +/// `WriteHandle`-bearing [`crate::shell::ServerMuxStateMachine`]. It then mints a +/// [`ServerMetadataBundle`] (a tuple of `Send + Sync` +/// `ReadHandleFactory`s) and pushes one clone per peer onto `bundle_tx`. +/// Every other shard receives the bundle and rebuilds a reader-mode +/// `MuxStateMachine` on its own runtime - no WAL access, no replay, no +/// `RecoverySync` two-phase fence. The old phase-2 WAL fence is gone +/// because peers no longer scan the WAL. They do still scan live shared +/// metadata to load their on-disk partitions, so a separate listener +/// fence is still required - see [`BootstrapBarrier`]. +/// +/// The channel is bounded to the peer count so shard 0's `send` never +/// blocks beyond a peer drain. A peer that dies before recv drops its +/// `bundle_rx`, so shard 0's `send` eventually sees a disconnected +/// channel; the cross-thread shutdown flag drives every waiter out of +/// its `recv` loop if shard 0 panics before broadcasting. +pub(in crate::boot) enum MetadataHandoff { + Owner { + bundle_tx: crossfire::MAsyncTx>, + }, + Waiter { + bundle_rx: crossfire::MAsyncRx>, + }, +} + +/// Reverse handshake to [`MetadataHandoff`]: gates shard 0's client +/// listeners until every peer has loaded its on-disk partitions. +/// +/// Peers build their owned-partition set from live shared metadata and +/// load each segment from disk in `build_shard_for_thread`. If shard 0 +/// opened listeners the instant `broadcast_metadata_bundle` returned +/// (peers have only *received* the bundle, not *loaded* partitions), a +/// client could create a partition before a peer's load scan finished. +/// That freshly committed partition would surface in the peer's scan +/// with no segment dir on disk yet, and `load_partition`'s `walk_dir` +/// would fail with `CannotReadPartitions`, aborting the whole node. A +/// partition created after boot must take the runtime reconciler path +/// (which creates its dir), never the bootstrap load path. +/// +/// Shard 0 (`Owner`) drains one signal per peer before binding +/// listeners; each peer (`Waiter`) sends one once its load completes. +/// The cross-thread shutdown flag drives both sides out of their poll +/// loop if any shard dies mid-boot. +pub(in crate::boot) enum BootstrapBarrier { + Owner { + ready_rx: crossfire::MAsyncRx>, + }, + Waiter { + ready_tx: crossfire::MAsyncTx>, + }, +} + +/// Block until shard 0 broadcasts the metadata factory bundle, or the +/// cross-thread shutdown flag flips. Polled in a `poll_interval` loop +/// so a shard 0 that panics before it broadcasts cannot strand peer +/// shards: the shutdown path flips the flag, every waiter observes it +/// on the next tick, and the server tears down instead of hanging. +/// +/// Uses `try_recv` + sleep rather than `timeout(recv())`. Crossfire 3.x +/// documents `recv()` as cancellation-safe (no leak/deadlock) but does +/// not guarantee atomicity for the dropped future's result; `try_recv` +/// keeps each tick fully synchronous and side-effect-free, so the +/// shutdown poll cadence cannot ambiguously consume a bundle. +pub(in crate::boot) async fn await_metadata_bundle( + shard_id: u16, + bundle_rx: &crossfire::MAsyncRx>, + shutdown_flag: &Arc, + poll_interval: Duration, +) -> Result { + loop { + match bundle_rx.try_recv() { + Ok(bundle) => return Ok(bundle), + Err(crossfire::TryRecvError::Disconnected) => { + return Err(ServerError::MetadataHandoffAborted { shard_id }); + } + Err(crossfire::TryRecvError::Empty) => { + if shutdown_flag.load(Ordering::Relaxed) { + return Err(ServerError::MetadataHandoffAborted { shard_id }); + } + compio::time::sleep(poll_interval).await; + } + } + } +} + +/// Push `peers` cloned bundles onto `bundle_tx`, polling each send in a +/// `poll_interval` loop so the cross-thread shutdown flag can interrupt +/// a stalled handoff. Symmetric to [`await_metadata_bundle`]: shutdown +/// observed mid-handshake aborts cleanly rather than stalling on a +/// `send` future that can no longer make progress. +/// +/// Uses `try_send` + sleep rather than `timeout(send())`. Crossfire 3.x +/// documents `send()` as cancellation-safe in the leak/deadlock sense +/// but explicitly warns the true result is unknown when `SendFuture` is +/// dropped on cancellation. For a retry loop that re-clones on every +/// tick that would risk publishing the same bundle twice, stuffing the +/// bounded channel past `peers` and stranding a follow-up `send`. +/// `try_send` returns the bundle back inside `TrySendError::Full`, so +/// the loop reuses it instead of re-cloning when the channel is full. +pub(in crate::boot) async fn broadcast_metadata_bundle( + shard_id: u16, + bundle_tx: &crossfire::MAsyncTx>, + bundle: ServerMetadataBundle, + peers: u16, + shutdown_flag: &Arc, + poll_interval: Duration, +) -> Result<(), ServerError> { + for _ in 0..peers { + let mut pending = bundle.clone(); + loop { + match bundle_tx.try_send(pending) { + Ok(()) => break, + Err(crossfire::TrySendError::Disconnected(_)) => { + // Every peer dropped its `bundle_rx` before recv. Shard + // 0 must not silently continue past handoff: it would + // bind listeners and commit consensus state for a + // cluster whose peers are gone. Propagate the abort so + // `shard_main` short-circuits before further side + // effects; `shutdown_flag` will flip via the normal + // teardown path. + return Err(ServerError::MetadataHandoffAborted { shard_id }); + } + Err(crossfire::TrySendError::Full(returned)) => { + if shutdown_flag.load(Ordering::Relaxed) { + return Err(ServerError::MetadataHandoffAborted { shard_id }); + } + pending = returned; + compio::time::sleep(poll_interval).await; + } + } + } + } + Ok(()) +} + +/// Peer side of [`BootstrapBarrier`]: tell shard 0 this shard finished +/// loading its on-disk partitions. Mirrors [`broadcast_metadata_bundle`]'s +/// `try_send`-or-shutdown poll loop so a sibling failure (which flips the +/// shutdown flag) drives this out instead of stranding it on a full +/// channel. The channel is sized to the peer count and each peer sends +/// exactly once, so `Full` is not expected; the branch only keeps the +/// loop interruptible. +pub(in crate::boot) async fn signal_bootstrap_complete( + shard_id: u16, + ready_tx: &crossfire::MAsyncTx>, + shutdown_flag: &Arc, + poll_interval: Duration, +) -> Result<(), ServerError> { + let mut pending = shard_id; + loop { + match ready_tx.try_send(pending) { + Ok(()) => return Ok(()), + Err(crossfire::TrySendError::Disconnected(_)) => { + // Shard 0 dropped its `ready_rx` before draining (it + // aborted before binding listeners). Propagate so this + // shard short-circuits; the shutdown flag flips via the + // normal teardown path. + return Err(ServerError::MetadataHandoffAborted { shard_id }); + } + Err(crossfire::TrySendError::Full(returned)) => { + if shutdown_flag.load(Ordering::Relaxed) { + return Err(ServerError::MetadataHandoffAborted { shard_id }); + } + pending = returned; + compio::time::sleep(poll_interval).await; + } + } + } +} + +/// Owner side of [`BootstrapBarrier`]: drain one ready signal per peer +/// before shard 0 binds listeners. Polls the shutdown flag so a peer that +/// dies mid-load (flipping the flag) aborts the wait instead of hanging on +/// a signal that will never arrive. A single shard (`peers == 0`) returns +/// immediately. +pub(in crate::boot) async fn await_bootstrap_complete( + ready_rx: &crossfire::MAsyncRx>, + peers: usize, + shutdown_flag: &Arc, + poll_interval: Duration, +) -> Result<(), ServerError> { + let mut remaining = peers; + while remaining > 0 { + match ready_rx.try_recv() { + Ok(_shard_id) => remaining -= 1, + Err(crossfire::TryRecvError::Disconnected) => { + return Err(ServerError::ShardBootstrapBarrierAborted { remaining }); + } + Err(crossfire::TryRecvError::Empty) => { + if shutdown_flag.load(Ordering::Relaxed) { + return Err(ServerError::ShardBootstrapBarrierAborted { remaining }); + } + compio::time::sleep(poll_interval).await; + } + } + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::shell::ServerMuxStateMachine; + + const TEST_POLL_INTERVAL: Duration = Duration::from_millis(50); + + #[compio::test] + async fn broadcast_metadata_bundle_returns_immediately_with_no_peers() { + // Single-shard deployment: shard 0 has no peers to fan out to, + // so the handoff must complete without ever calling `send`. + let (bundle_tx, _bundle_rx) = crossfire::mpmc::bounded_async::(0); + let flag = Arc::new(AtomicBool::new(false)); + let mux = ServerMuxStateMachine::default(); + broadcast_metadata_bundle( + 0, + &bundle_tx, + mux.factory_bundle(), + 0, + &flag, + TEST_POLL_INTERVAL, + ) + .await + .expect("zero peers must not block shard 0"); + } + + #[compio::test] + async fn metadata_bundle_round_trips_through_channel() { + // End-to-end: shard 0 mints a bundle, a peer receives it on + // another runtime, and `from_factory_bundle` constructs a + // reader-mode mux that observes shard 0's writes via the same + // LeftRight pair. + let peers = 1u16; + let (bundle_tx, bundle_rx) = + crossfire::mpmc::bounded_async::(usize::from(peers)); + let flag = Arc::new(AtomicBool::new(false)); + + let owner = ServerMuxStateMachine::default(); + let bundle = owner.factory_bundle(); + broadcast_metadata_bundle(0, &bundle_tx, bundle, peers, &flag, TEST_POLL_INTERVAL) + .await + .expect("broadcast must succeed with one peer drained"); + + let received = await_metadata_bundle(1, &bundle_rx, &flag, TEST_POLL_INTERVAL) + .await + .expect("peer must receive the broadcast bundle"); + let _peer_mux = ServerMuxStateMachine::from_factory_bundle(received); + } + + #[compio::test] + async fn broadcast_metadata_bundle_aborts_when_peers_drop_rx() { + // Shard 0 drives handoff but every peer's `bundle_rx` was dropped + // before recv. Silently returning Ok would commit listener binds + // and consensus init for a cluster whose peers are gone; the + // broadcast must surface the disconnect so `shard_main` aborts. + let (bundle_tx, bundle_rx) = crossfire::mpmc::bounded_async::(0); + drop(bundle_rx); + let flag = Arc::new(AtomicBool::new(false)); + let mux = ServerMuxStateMachine::default(); + + let err = broadcast_metadata_bundle( + 0, + &bundle_tx, + mux.factory_bundle(), + 3, + &flag, + TEST_POLL_INTERVAL, + ) + .await + .expect_err("dropped rx must surface as MetadataHandoffAborted"); + assert!( + matches!(err, ServerError::MetadataHandoffAborted { shard_id: 0 }), + "expected MetadataHandoffAborted, got {err:?}" + ); + } + + #[compio::test] + async fn await_metadata_bundle_aborts_when_owner_drops_without_sending() { + let (bundle_tx, bundle_rx) = crossfire::mpmc::bounded_async::(1); + let flag = Arc::new(AtomicBool::new(false)); + + // Shard 0 dies before broadcasting; the peer must observe the + // disconnect and abort instead of hanging forever. + drop(bundle_tx); + + let err = await_metadata_bundle(1, &bundle_rx, &flag, TEST_POLL_INTERVAL) + .await + .expect_err("a peer whose owner never sends must abort"); + assert!( + matches!(err, ServerError::MetadataHandoffAborted { shard_id: 1 }), + "expected MetadataHandoffAborted, got {err:?}" + ); + } + + #[compio::test] + async fn await_metadata_bundle_aborts_on_shutdown_flag() { + // compio 0.19 `JoinHandle` yields `Result`; the + // `ResumeUnwind` impl re-raises a task panic and maps cancellation + // to `None`. + use compio::runtime::ResumeUnwind; + + let (_bundle_tx, bundle_rx) = crossfire::mpmc::bounded_async::(1); + let flag = Arc::new(AtomicBool::new(false)); + + let waiter = compio::runtime::spawn({ + let flag = Arc::clone(&flag); + async move { await_metadata_bundle(1, &bundle_rx, &flag, TEST_POLL_INTERVAL).await } + }); + + // Owner has not sent yet, but shutdown was requested; the peer + // must exit via the flag poll instead of hanging. + compio::time::sleep(TEST_POLL_INTERVAL / 2).await; + flag.store(true, Ordering::Relaxed); + + let err = waiter + .await + .resume_unwind() + .expect("waiter task was cancelled") + .expect_err("shutdown flag must abort the bundle wait"); + assert!( + matches!(err, ServerError::MetadataHandoffAborted { shard_id: 1 }), + "expected MetadataHandoffAborted on shutdown, got {err:?}" + ); + } + + #[compio::test] + async fn await_bootstrap_complete_returns_immediately_for_single_shard() { + // A single-shard server has no peers to wait on; the owner barrier + // must not block when `peers == 0`. + let (_ready_tx, ready_rx) = crossfire::mpmc::bounded_async::(1); + let flag = Arc::new(AtomicBool::new(false)); + await_bootstrap_complete(&ready_rx, 0, &flag, TEST_POLL_INTERVAL) + .await + .expect("single-shard server must not block on the barrier"); + } + + #[compio::test] + async fn await_bootstrap_complete_drains_every_peer_signal() { + // Two peers report load-complete; shard 0 drains both, then proceeds + // to bind listeners. + let (ready_tx, ready_rx) = crossfire::mpmc::bounded_async::(2); + let flag = Arc::new(AtomicBool::new(false)); + signal_bootstrap_complete(1, &ready_tx, &flag, TEST_POLL_INTERVAL) + .await + .expect("peer 1 must signal load-complete"); + signal_bootstrap_complete(2, &ready_tx, &flag, TEST_POLL_INTERVAL) + .await + .expect("peer 2 must signal load-complete"); + await_bootstrap_complete(&ready_rx, 2, &flag, TEST_POLL_INTERVAL) + .await + .expect("owner must drain both peer signals"); + } + + #[compio::test] + async fn await_bootstrap_complete_aborts_on_shutdown_flag() { + use compio::runtime::ResumeUnwind; + + // `_ready_tx` is held so the channel is not disconnected: the owner + // must exit via the shutdown flag, not a dropped sender. + let (_ready_tx, ready_rx) = crossfire::mpmc::bounded_async::(1); + let flag = Arc::new(AtomicBool::new(false)); + + let owner = compio::runtime::spawn({ + let flag = Arc::clone(&flag); + async move { await_bootstrap_complete(&ready_rx, 1, &flag, TEST_POLL_INTERVAL).await } + }); + + // The peer never signals, but a sibling failure flips the flag; the + // owner must abort instead of hanging before listeners. + compio::time::sleep(TEST_POLL_INTERVAL / 2).await; + flag.store(true, Ordering::Relaxed); + + let err = owner + .await + .resume_unwind() + .expect("owner task was cancelled") + .expect_err("shutdown flag must abort the barrier wait"); + assert!( + matches!( + err, + ServerError::ShardBootstrapBarrierAborted { remaining: 1 } + ), + "expected ShardBootstrapBarrierAborted, got {err:?}" + ); + } + + #[compio::test] + async fn signal_bootstrap_complete_aborts_when_owner_drops_rx() { + // Shard 0 aborted before draining and dropped its receiver; a peer's + // signal must surface the disconnect instead of stranding. + let (ready_tx, ready_rx) = crossfire::mpmc::bounded_async::(1); + let flag = Arc::new(AtomicBool::new(false)); + drop(ready_rx); + + let err = signal_bootstrap_complete(2, &ready_tx, &flag, TEST_POLL_INTERVAL) + .await + .expect_err("dropped rx must surface as an abort"); + assert!( + matches!(err, ServerError::MetadataHandoffAborted { shard_id: 2 }), + "expected MetadataHandoffAborted, got {err:?}" + ); + } +} diff --git a/core/server/src/boot/listeners.rs b/core/server/src/boot/listeners.rs new file mode 100644 index 0000000000..222b5d7eb8 --- /dev/null +++ b/core/server/src/boot/listeners.rs @@ -0,0 +1,636 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Shard 0 listener start-up and the accept-fn factories. + +use crate::boot::credentials::{ + load_quic_server_credentials, load_tcp_tls_server_credentials, load_wss_server_credentials, +}; +use crate::boot::topology::{ + TcpTopology, client_listeners, derived_address_misses_listener, derived_bind_ip, + wildcard_listener_under_loopback_address, +}; +use crate::cluster_meta::{ClusterRoster, self_advertised_address}; +use crate::config_writer::{BoundAddresses, write_current_config}; +use crate::http; +use crate::server_error::ServerError; +use crate::shell::ServerShard; +use configs::cluster::TransportPorts; +use configs::server::ServerConfig; +use message_bus::client_listener::{self, RequestHandler}; +use message_bus::installer::conn_info::{ClientConnMeta, ClientTransportKind}; +use message_bus::replica::io as replica_io; +use message_bus::replica::listener::{self as replica_listener}; +use message_bus::transports::quic::server_config_with_cert; +use message_bus::{ + AcceptedClientFn, AcceptedQuicClientFn, AcceptedReplicaFn, AcceptedTlsClientFn, + AcceptedWsClientFn, AcceptedWssClientFn, DialedReplicaFn, IggyMessageBus, + MAX_INFLIGHT_REPLICA_HANDSHAKES, connector, installer, +}; +use shard::metrics::ShardMetrics; +use std::net::SocketAddr; +use std::rc::Rc; +use std::sync::Arc; +use tracing::{error, info, warn}; + +pub(in crate::boot) struct LocalClientAcceptFns { + tcp: AcceptedClientFn, + ws: AcceptedWsClientFn, + quic: AcceptedQuicClientFn, + tcp_tls: AcceptedTlsClientFn, + wss: AcceptedWssClientFn, +} + +#[allow(clippy::too_many_arguments)] +pub(in crate::boot) async fn start_tcp_runtime( + shard: &Rc, + config: &ServerConfig, + topology: &TcpTopology, + roster: Rc, + accepted_replica: AcceptedReplicaFn, + dialed_replica: DialedReplicaFn, + accepted_clients: LocalClientAcceptFns, + shard_metrics_all: &[ShardMetrics], +) -> Result<(), ServerError> { + // HTTP is served over TCP but sits outside the replica_io / manual client + // reactor, so it binds on its own. Config first, so a bad `[http.*]` + // section fails boot before any listener accepts. Socket only after the + // replica peer dial below has returned: against a peer that drops SYNs + // that dial blocks for the kernel retry budget, and a port listening with + // no serve loop behind it would pass TCP readiness probes the whole time. + // Served last, once the roster and `current_config.toml` know the port, so + // no request is answered before they do. Shard-0 gating comes from the + // sole caller of this function. + let http = topology + .http_listen_addr + .map(|addr| http::prepare(addr, &config.http, &config.cluster)) + .transpose()?; + + let mut bound = if config.tcp.enabled && !config.tcp.tls.enabled { + start_via_replica_io( + shard, + config, + topology, + accepted_replica, + dialed_replica, + accepted_clients, + ) + .await? + } else { + start_manual_runtime( + shard, + config, + topology, + accepted_replica, + dialed_replica, + accepted_clients, + ) + .await? + }; + + // Cluster metadata carries one host for all four transports, so a listener + // the derived host does not reach is unreachable at it. Only a derived + // address is judged, and never against the listener it was derived from. + let declared = config.node.advertised_address.as_deref(); + let self_advertised = self_advertised_address(declared, derived_bind_ip(topology, config)); + // A roster entry answers this per node in cluster mode, so the derived + // address is never served and none of these listeners are judged against it. + let listeners = client_listeners(topology, config); + if !config.cluster.enabled + && let Some(derived_from) = listeners + .iter() + .find_map(|(key, listen_addr)| listen_addr.map(|_| *key)) + { + for (key, listen_addr) in listeners { + let Some(listen_addr) = listen_addr.filter(|_| key != derived_from) else { + continue; + }; + if derived_address_misses_listener(declared, &self_advertised, listen_addr) { + warn!( + "{key} binds {listen_addr} but cluster metadata publishes {self_advertised}, \ + derived from {derived_from}; a client reading that metadata would not reach \ + this listener. Set node.advertised_address to the address clients dial." + ); + } else if wildcard_listener_under_loopback_address( + declared, + &self_advertised, + listen_addr, + ) { + warn!( + "{key} binds the wildcard {listen_addr} but cluster metadata publishes the \ + loopback {self_advertised}, derived from {derived_from}; a client reaching \ + this listener from another host is told an address that points back at \ + itself. Set node.advertised_address to the address clients dial." + ); + } + } + } + + let http = http.map(http::PreparedHttp::bind).transpose()?; + bound.http = http.as_ref().map(|http| http.bound_addr); + roster.bound_ports.publish(TransportPorts::from(&bound)); + write_current_config(config, Some(topology.self_replica_id), &bound).await?; + + if let Some(http) = http { + http::start( + http, + shard, + &config.http, + config.metadata.clients_table_max, + config.personal_access_token.max_tokens_per_user, + Arc::clone(&config.system), + roster, + shard_metrics_all, + )?; + } + + Ok(()) +} + +// ws/wss bindings intentionally mirror the transport names (same convention as +// `replica_io::start_on_shard_zero`). +#[allow(clippy::similar_names)] +async fn start_via_replica_io( + shard: &Rc, + config: &ServerConfig, + topology: &TcpTopology, + accepted_replica: AcceptedReplicaFn, + dialed_replica: DialedReplicaFn, + accepted_clients: LocalClientAcceptFns, +) -> Result { + let replica_addr = topology + .replica_listen_addr + .expect("topology must include replica listener address"); + let quic_credentials = topology + .quic_listen_addr + .is_some() + .then(|| load_quic_server_credentials(config)) + .transpose()?; + let tcp_tls_credentials = topology + .tcp_tls_listen_addr + .is_some() + .then(|| load_tcp_tls_server_credentials(config)) + .transpose()?; + // `websocket.tls.enabled` upgrades the websocket address to a WSS + // listener; the plain-WS listener must NOT also bind it (one port, one + // handshake kind -- a plain upgrade parser fed a TLS ClientHello rejects + // every connection with an httparse error). + let wss_enabled = config.websocket.tls.enabled; + let ws_listen_addr = (!wss_enabled).then_some(topology.ws_listen_addr).flatten(); + let wss_listen_addr = wss_enabled.then_some(topology.ws_listen_addr).flatten(); + let wss_credentials = wss_listen_addr + .is_some() + .then(|| load_wss_server_credentials(config)) + .transpose()?; + + let LocalClientAcceptFns { + tcp, + ws, + quic, + tcp_tls, + wss, + } = accepted_clients; + + let bound = replica_io::start_on_shard_zero( + &shard.bus, + replica_addr, + topology.client_listen_addr, + ws_listen_addr, + topology.quic_listen_addr, + quic_credentials, + topology.tcp_tls_listen_addr, + tcp_tls_credentials, + wss_listen_addr, + wss_credentials, + topology.self_replica_id, + topology.peers.clone(), + accepted_replica, + dialed_replica, + tcp, + ws_listen_addr.map(|_| ws), + topology.quic_listen_addr.map(|_| quic), + topology.tcp_tls_listen_addr.map(|_| tcp_tls), + wss_listen_addr.map(|_| wss), + shard.bus.config().reconnect_period, + ) + .await + .map_err(|source| { + error!( + replica_addr = %replica_addr, + client_addr = %topology.client_listen_addr, + error = %source, + "failed to start server listeners via replica_io" + ); + source + })?; + // `start_on_shard_zero` answers `None` only on a non-zero shard, and the + // sole caller of this function is shard 0's listener block. + let bound = bound.ok_or(ServerError::ListenersOffShardZero { shard_id: shard.id })?; + + if config.cluster.enabled { + info!( + shard = shard.id, + replica = %bound.replica, + tcp = %bound.client, + tcp_tls = ?bound.tcp_tls, + ws = ?bound.ws, + quic = ?bound.quic, + "server listeners started" + ); + } else { + info!( + shard = shard.id, + tcp = %bound.client, + tcp_tls = ?bound.tcp_tls, + ws = ?bound.ws, + quic = ?bound.quic, + "server client listeners started" + ); + } + + Ok(BoundAddresses { + tcp: Some(bound.client), + tcp_tls: bound.tcp_tls, + quic: bound.quic, + // The WSS listener occupies the configured websocket address slot. + websocket: bound.wss.or(bound.ws), + http: None, + replica: config.cluster.enabled.then_some(bound.replica), + }) +} + +async fn start_manual_runtime( + shard: &Rc, + config: &ServerConfig, + topology: &TcpTopology, + accepted_replica: AcceptedReplicaFn, + dialed_replica: DialedReplicaFn, + accepted_clients: LocalClientAcceptFns, +) -> Result { + let bound_replica = if config.cluster.enabled { + let replica_addr = topology + .replica_listen_addr + .expect("cluster-enabled topology must include replica listener address"); + let (replica_listener, bound_addr) = + replica_listener::bind(replica_addr) + .await + .map_err(|source| { + error!( + replica_addr = %replica_addr, + error = %source, + "failed to bind replica listener" + ); + source + })?; + let token = shard.bus.token(); + let replica_handle = compio::runtime::spawn(async move { + replica_listener::run(replica_listener, token, accepted_replica).await; + }); + shard.bus.track_background(replica_handle); + connector::start( + &shard.bus, + topology.self_replica_id, + topology.peers.clone(), + dialed_replica, + shard.bus.config().reconnect_period, + ) + .await; + Some(bound_addr) + } else { + None + }; + + let mut bound = start_client_listeners(shard, config, topology, &accepted_clients)?; + bound.replica = bound_replica; + + if config.cluster.enabled { + info!( + shard = shard.id, + replica = ?bound.replica, + tcp = ?bound.tcp, + tcp_tls = ?bound.tcp_tls, + ws = ?bound.websocket, + quic = ?bound.quic, + "server listeners started" + ); + } else { + info!( + shard = shard.id, + tcp = ?bound.tcp, + tcp_tls = ?bound.tcp_tls, + ws = ?bound.websocket, + quic = ?bound.quic, + "server client listeners started" + ); + } + + Ok(bound) +} + +/// Replica delegation callbacks for shard 0's listener and connector. +/// +/// Inbound: acquire a slot in the shard-0-global in-flight handshake cap +/// (drop the connection when full), then blind-delegate the raw fd +/// through the coordinator's round-robin. The fd lands on the target +/// shard's inbox as a [`shard::LifecycleFrame::ReplicaInboundSetup`] +/// frame; the owning shard runs the acceptor handshake and acks the +/// slot back. A failed delegation releases the slot immediately. +/// +/// Outbound: delegate the dialed fd as +/// [`shard::LifecycleFrame::ReplicaOutboundSetup`] and mark the peer +/// dial-pending so the reconnect sweep skips it until the owning +/// shard's handshake outcome arrives (or the entry expires). +pub(in crate::boot) fn make_replica_delegation_fns( + coord: Rc, + bus: &Rc, +) -> (AcceptedReplicaFn, DialedReplicaFn) { + let inbound_bus = Rc::clone(bus); + let inbound_coord = Rc::clone(&coord); + let accepted: AcceptedReplicaFn = Rc::new(move |stream| { + let Some(slot) = inbound_bus.try_acquire_replica_handshake_slot() else { + warn!( + cap = MAX_INFLIGHT_REPLICA_HANDSHAKES, + "replica handshake in-flight cap reached; dropping inbound" + ); + return; + }; + match inbound_coord.delegate_replica_inbound(stream, slot) { + Ok(target) => { + info!(slot, target, "inbound replica connection delegated"); + } + Err(error) => { + inbound_bus.release_replica_handshake_slot(slot); + warn!( + error = ?error, + "delegate_replica_inbound failed; dropping inbound replica connection" + ); + } + } + }); + + let outbound_bus = Rc::clone(bus); + let dialed: DialedReplicaFn = + Rc::new( + move |stream, peer_id| match coord.delegate_replica_outbound(stream, peer_id) { + Ok(target) => { + outbound_bus.mark_dial_pending(peer_id); + info!(peer_id, target, "outbound replica connection delegated"); + } + Err(error) => { + warn!( + peer_id, + error = ?error, + "delegate_replica_outbound failed; dropping dialed replica connection" + ); + } + }, + ); + + (accepted, dialed) +} + +/// Shard-0 client accept callbacks. TCP and WS clients are delegated via +/// the coordinator (round-robin to peer shards); QUIC and TCP-TLS install +/// locally on shard 0 because their per-connection state is not portable +/// across shards (`compio_quic` endpoint binds one UDP socket; rustls TLS +/// state ties to the post-handshake reactor). +// ws/wss bindings intentionally mirror the transport names (same convention as +// `replica_io::start_on_shard_zero`). +#[allow(clippy::similar_names)] +pub(in crate::boot) fn make_shard_zero_client_accept_fns( + coord: Rc, + bus: &Rc, + on_request: RequestHandler, +) -> LocalClientAcceptFns { + let quic_bus = Rc::clone(bus); + let tcp_tls_bus = Rc::clone(bus); + let wss_bus = Rc::clone(bus); + let quic_request = on_request.clone(); + let wss_request = on_request.clone(); + let tcp_tls_request = on_request; + + let tcp_coord = Rc::clone(&coord); + let tcp = Rc::new(move |stream| match tcp_coord.delegate_client(stream) { + Ok(client_id) => info!(client_id, "TCP client delegated"), + Err(error) => warn!(error = ?error, "delegate_client failed; dropping TCP client"), + }); + + let ws_coord = Rc::clone(&coord); + let ws = Rc::new(move |stream| match ws_coord.delegate_ws_client(stream) { + Ok(client_id) => info!(client_id, "WS client delegated"), + Err(error) => warn!(error = ?error, "delegate_ws_client failed; dropping WS client"), + }); + + // QUIC and TCP-TLS terminate locally on shard 0 but mint their client + // ids through the coordinator's `client_seq`, the same counter the + // delegated TCP/WS path uses. A separate counter here would let a + // shard-0-local id collide with a delegated id that round-robined to + // shard 0 (both encode target shard 0) in shard 0's connection + // registry. + let quic_coord = Rc::clone(&coord); + let quic = Rc::new(move |accepted: message_bus::AcceptedQuicConn| { + let meta = mint_client_meta(&quic_coord, accepted.peer_addr(), ClientTransportKind::Quic); + installer::install_client_quic(&quic_bus, meta, accepted, quic_request.clone()); + }); + + let tcp_tls_coord = Rc::clone(&coord); + let tcp_tls = Rc::new(move |stream, tls_config| { + let Some(meta) = + client_meta_from_stream(&stream, &tcp_tls_coord, ClientTransportKind::TcpTls) + else { + return; + }; + installer::install_client_tcp_tls( + &tcp_tls_bus, + meta, + stream, + tls_config, + tcp_tls_request.clone(), + ); + }); + + // WSS terminates locally on shard 0 like TCP-TLS (rustls state is not + // serialisable across the delegate path), minting ids through the same + // coordinator counter. + let wss_coord = coord; + let wss = Rc::new(move |stream, tls_config| { + let Some(meta) = client_meta_from_stream(&stream, &wss_coord, ClientTransportKind::Wss) + else { + return; + }; + installer::install_client_wss(&wss_bus, meta, stream, tls_config, wss_request.clone()); + }); + + LocalClientAcceptFns { + tcp, + ws, + quic, + tcp_tls, + wss, + } +} + +fn client_meta_from_stream( + stream: &compio::net::TcpStream, + coord: &shard::coordinator::ShardZeroCoordinator, + transport: ClientTransportKind, +) -> Option { + let peer_addr = match stream.peer_addr() { + Ok(peer_addr) => peer_addr, + Err(error) => { + warn!(error = %error, "dropping accepted client with unknown peer address"); + return None; + } + }; + Some(mint_client_meta(coord, peer_addr, transport)) +} + +fn mint_client_meta( + coord: &shard::coordinator::ShardZeroCoordinator, + peer_addr: SocketAddr, + transport: ClientTransportKind, +) -> ClientConnMeta { + ClientConnMeta::new(coord.mint_shard_zero_client_id(), peer_addr, transport) +} + +fn start_client_listeners( + shard: &Rc, + config: &ServerConfig, + topology: &TcpTopology, + accepted_clients: &LocalClientAcceptFns, +) -> Result { + let mut bound = BoundAddresses::default(); + + if config.tcp.enabled && !config.tcp.tls.enabled { + let (listener, bound_addr) = client_listener::tcp::bind(topology.client_listen_addr) + .map_err(|source| { + error!( + addr = %topology.client_listen_addr, + error = %source, + "failed to bind TCP client listener" + ); + source + })?; + let token = shard.bus.token(); + let accepted_client = accepted_clients.tcp.clone(); + let client_handle = compio::runtime::spawn(async move { + client_listener::tcp::run(listener, token, accepted_client).await; + }); + shard.bus.track_background(client_handle); + bound.tcp = Some(bound_addr); + } + + if let Some(ws_addr) = topology.ws_listen_addr { + bound.websocket = Some(start_websocket_listener( + shard, + config, + ws_addr, + accepted_clients, + )?); + } + + if let Some(quic_addr) = topology.quic_listen_addr { + let credentials = load_quic_server_credentials(config)?; + let server_config = server_config_with_cert( + credentials.cert_chain, + credentials.key_der, + &shard.bus.config().quic, + ) + .map_err(|e| { + let source = + iggy_common::IggyError::IoError(format!("QUIC server config build failed: {e}")); + error!(addr = %quic_addr, error = %source, "failed to build QUIC server config"); + source + })?; + let (endpoint, bound_addr) = client_listener::quic::bind(quic_addr, server_config) + .map_err(|source| { + error!(addr = %quic_addr, error = %source, "failed to bind QUIC listener"); + source + })?; + let token = shard.bus.token(); + let handshake_grace = shard.bus.config().handshake_grace; + let accepted_quic = accepted_clients.quic.clone(); + let quic_handle = compio::runtime::spawn(async move { + client_listener::quic::run(endpoint, token, accepted_quic, handshake_grace).await; + }); + shard.bus.track_background(quic_handle); + bound.quic = Some(bound_addr); + } + + if config.tcp.enabled && config.tcp.tls.enabled { + let credentials = load_tcp_tls_server_credentials(config)?; + let (listener, tls_config, bound_addr) = + client_listener::tcp_tls::bind(topology.client_listen_addr, credentials).map_err( + |source| { + error!( + addr = %topology.client_listen_addr, + error = %source, + "failed to bind TCP TLS listener" + ); + source + }, + )?; + let token = shard.bus.token(); + let accepted_tls = accepted_clients.tcp_tls.clone(); + let tls_handle = compio::runtime::spawn(async move { + client_listener::tcp_tls::run(listener, tls_config, token, accepted_tls).await; + }); + shard.bus.track_background(tls_handle); + bound.tcp_tls = Some(bound_addr); + } + + Ok(bound) +} + +/// Bind the websocket client listener on `ws_addr`: WSS when +/// `websocket.tls.enabled` (the plain-WS accept loop must not also bind the +/// port -- a plain upgrade parser fed a TLS `ClientHello` rejects every +/// connection with an httparse error), plain WS otherwise. +fn start_websocket_listener( + shard: &Rc, + config: &ServerConfig, + ws_addr: SocketAddr, + accepted_clients: &LocalClientAcceptFns, +) -> Result { + if config.websocket.tls.enabled { + let credentials = load_wss_server_credentials(config)?; + let (listener, tls_config, bound_addr) = client_listener::wss::bind(ws_addr, credentials) + .map_err(|source| { + error!(addr = %ws_addr, error = %source, "failed to bind WSS listener"); + source + })?; + let token = shard.bus.token(); + let accepted_wss = accepted_clients.wss.clone(); + let wss_handle = compio::runtime::spawn(async move { + client_listener::wss::run(listener, tls_config, token, accepted_wss).await; + }); + shard.bus.track_background(wss_handle); + Ok(bound_addr) + } else { + let (listener, bound_addr) = client_listener::ws::bind(ws_addr).map_err(|source| { + error!(addr = %ws_addr, error = %source, "failed to bind websocket listener"); + source + })?; + let token = shard.bus.token(); + let accepted_ws = accepted_clients.ws.clone(); + let ws_handle = compio::runtime::spawn(async move { + client_listener::ws::run(listener, token, accepted_ws).await; + }); + shard.bus.track_background(ws_handle); + Ok(bound_addr) + } +} diff --git a/core/server/src/boot/mod.rs b/core/server/src/boot/mod.rs new file mode 100644 index 0000000000..b8a5ed9a3c --- /dev/null +++ b/core/server/src/boot/mod.rs @@ -0,0 +1,1038 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Process entry and the per-shard boot narrative. +//! +//! `load_config` -> `prepare_runtime_dirs` -> [`bootstrap`] spawn one OS thread +//! per shard; each runs `shard_main`, the ordered sequence every boot +//! invariant hangs off (pump before listeners, barrier before bind, config +//! write after bind, systemd notify points). The narrative stays whole here; +//! the leaves hold the support it calls into. + +mod credentials; +mod handoff; +mod listeners; +mod recovery; +#[cfg(feature = "systemd")] +pub mod systemd; +mod threads; +mod topology; + +pub use credentials::apply_default_root_credentials; +pub use threads::ShardHandles; + +use crate::boot::credentials::{ + ensure_default_root_user, load_replica_auth, load_replica_tls_ctx, + validate_root_credentials_env, +}; +use crate::boot::handoff::{ + BootstrapBarrier, MetadataHandoff, await_bootstrap_complete, await_metadata_bundle, + broadcast_metadata_bundle, signal_bootstrap_complete, +}; +use crate::boot::listeners::{ + make_replica_delegation_fns, make_shard_zero_client_accept_fns, start_tcp_runtime, +}; +use crate::boot::recovery::{ + RecoveredOwnerState, build_shard_for_thread, restore_metadata_consensus, +}; +use crate::boot::threads::{ + StopSignals, await_pump_drain, join_partial_shard_survivors, resolve_shard_assignments, + run_shard_thread, spawn_shutdown_watchdog, validate_sharding_runtime_knobs, +}; +use crate::boot::topology::{RosterCells, resolve_tcp_topology}; +use crate::dispatch::partition::make_partition_read_handler; +use crate::dispatch::session_ops::warm_dummy_password_hash; +use crate::dispatch::submit::make_metadata_submit_handler; +use crate::dispatch::{ + make_client_request_handler, make_deferred_client_request_handler, + make_deferred_replica_message_handler, make_list_clients_handler, +}; +use crate::server_error::ServerError; +use crate::session_manager::SessionManager; +use crate::shell::{ + ServerMetadata, ServerMetadataBundle, ServerMuxStateMachine, ShellBus, ShellHandlers, + ShellShardHandle, +}; +use configs::server::{ServerConfig, ServerSystemConfig}; +use consensus::{MetadataHandle, PartitionsHandle}; +use iggy_binary_protocol::{Operation, PrepareHeader}; +use journal::superblock::SuperblockStore; +use journal::{Journal, JournalHandle}; +use message_bus::replica::handshake::ReplicaHandshakeCtx; +use message_bus::transports::tls::install_default_crypto_provider; +use message_bus::{IggyMessageBus, ReplicaOwnerTable}; +use metadata::ReplicaIdentity; +use metadata::impls::metadata::StreamsFrontend; +use metadata::impls::recovery::recover; +use server_common::Message; +use server_common::bootstrap::create_directories; +use server_common::fs_utils::remove_dir_all; +use server_common::log::{Logging, LoggingSettings, TelemetrySettings}; +use shard::metrics::{ShardMetrics, frame_drop_reason, frame_drop_variant}; +use shard::{ + LifecycleFrame, Receiver as ShardReceiver, ShardFrame, TaggedSender, channel, + shard_mesh_channels, +}; +use std::cell::RefCell; +use std::path::{Path, PathBuf}; +use std::rc::Rc; +use std::sync::Arc; +use std::sync::atomic::{AtomicBool, Ordering}; +use std::thread; +use tracing::{error, info, warn}; + +/// Build the deferred dispatch handlers for `shard_handle` against `bus`. +/// +/// They share one fresh [`SessionManager`]. The caller must set the weak +/// self-reference in `shard_handle` once the shard is built, so the +/// handlers can upgrade it per frame. +pub fn wire_shell_handlers( + bus: &B, + shard_handle: &ShellShardHandle, + system_config: Arc, + max_tokens_per_user: u32, +) -> ShellHandlers +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let sessions = Rc::new(RefCell::new(SessionManager::new())); + ShellHandlers { + on_replica_message: make_deferred_replica_message_handler(shard_handle), + on_client_request: make_deferred_client_request_handler( + bus, + shard_handle, + &sessions, + system_config, + max_tokens_per_user, + ), + on_metadata_submit: make_metadata_submit_handler(shard_handle), + on_list_clients: make_list_clients_handler(&sessions), + on_partition_read: make_partition_read_handler(shard_handle), + sessions, + } +} + +/// Load the server configuration from the active config provider. +/// +/// # Errors +/// +/// Returns an error if the configuration cannot be read or parsed. +pub async fn load_config() -> Result { + ServerConfig::load().await.map_err(ServerError::Config) +} + +/// Prepare the on-disk layout the server boots from and complete late +/// logging init. +/// +/// `fresh` wipes the system path first: `late_init` opens a rolling +/// appender under `{system_path}/logs` and `create_directories` +/// materialises exactly what the wipe is meant to remove, so both have to +/// run after it. +/// +/// # Errors +/// +/// Returns an error if the wipe, directory preparation, or logging setup +/// fails. +pub async fn prepare_runtime_dirs( + config: &ServerConfig, + logging: &mut Logging, + fresh: bool, +) -> Result<(), ServerError> { + if fresh { + wipe_system_path(config).await?; + } + create_directories(&config.system).await.map_err(|source| { + error!( + system_path = %config.system.get_system_path(), + error = %source, + "failed to prepare server directories" + ); + source + })?; + logging + .late_init( + config.system.get_system_path(), + &LoggingSettings::from(&config.system.logging), + &TelemetrySettings::from(&config.telemetry), + ) + .map_err(ServerError::Logging)?; + + Ok(()) +} + +/// Delete the configured system path so the server boots on empty state. +async fn wipe_system_path(config: &ServerConfig) -> Result<(), ServerError> { + let path = config.system.get_system_path(); + // `system.path` is relative by default and IGGY_SYSTEM_PATH-overridable, + // so report what is actually about to be deleted, not what was configured. + let resolved = std::path::absolute(&path).unwrap_or_else(|_| PathBuf::from(&path)); + + if config.cluster.enabled { + warn!( + path = %resolved.display(), + "--fresh wipes only this replica, which then refills from the cluster by \ + state transfer; wiping a quorum at once destroys committed data, and a \ + service unit file carrying --fresh re-transfers everything on every restart" + ); + } + + if !Path::new(&path).exists() { + info!(path = %resolved.display(), "--fresh: system path does not exist, nothing to remove"); + return Ok(()); + } + + warn!(path = %resolved.display(), "--fresh: removing the system path, ALL local data will be deleted"); + // A half-removed directory is worse than no removal at all: the surviving + // superblock and snapshot no longer pair up, and boot would report the + // leftovers as a durability violation rather than as a failed wipe. + remove_dir_all(&path) + .await + .map_err(|source| ServerError::FreshWipeFailed { + path: resolved, + source, + }) +} + +/// Spawn the multi-shard `server` runtime. +/// +/// Resolves shard count + CPU affinities from +/// `system.sharding.cpu_allocation`, builds canonical-ordered +/// `(senders, inboxes)` channels, and spawns one OS thread per shard. +/// +/// Each thread pins itself (`nix::sched::sched_setaffinity` on Linux via +/// `ShardInfo::bind_cpu`), binds memory to its NUMA node when +/// configured, builds a fresh `compio::runtime::Runtime` (one +/// `io_uring` instance per shard), and runs `shard_main` inside it. +/// +/// Returns [`ShardHandles`] containing the cross-thread shutdown flag +/// and the per-shard `JoinHandle`s. The caller (`main.rs`) installs a +/// `ctrlc` handler that flips the flag, then `.join()`s every handle. +/// +/// # Errors +/// +/// Returns an error if shard allocation fails, the inbox capacity is +/// invalid, or any OS thread fails to spawn. Per-shard recovery / +/// listener / consensus failures surface through the per-thread `Result` +/// the caller observes on `.join()`. +/// +/// # Panics +/// +/// Panics if [`shard_mesh_channels`] returns an inbox slot already +/// consumed - a bootstrap programming error that would only fire if this +/// function were called twice with the same inboxes. +#[allow(clippy::too_many_lines)] +pub fn bootstrap( + config: ServerConfig, + current_replica_id: Option, +) -> Result { + // One process-wide rustls provider, installed before any shard thread + // exists. rustls is compiled with both `ring` and `aws-lc-rs`, so a + // `ServerConfig` / `ClientConfig` builder reached before this line panics + // ("could not determine process-level CryptoProvider") instead of picking + // one; every TLS surface (client listeners, the replica mesh, HTTP + // forwarding) resolves the default this call sets. message_bus keeps its + // own idempotent install in its TLS listeners for embedders that never + // run this bootstrap; after this line those are no-ops. + install_default_crypto_provider(); + validate_root_credentials_env(&config)?; + warm_dummy_password_hash(); + // The sync GetStats read path has no access to server config, so capture + // the data directory here for its disk-usage reporting. + crate::responses::init_stats_data_path(config.system.get_system_path().into()); + let (assignments, total_shards) = resolve_shard_assignments(&config.system.sharding)?; + let shards_count = assignments.len(); + + // Re-check the full valid range, not just the zero floor: a caller + // that built the config without running `ShardingConfig::validate` + // would otherwise OOM at boot allocating an oversized inbox channel, + // busy-loop every shutdown watchdog on a zero poll cadence, or wedge + // process exit on an unbounded drain budget. + let inbox_capacity = config.system.sharding.inbox_capacity; + let reply_inbox_capacity = config.system.sharding.reply_inbox_capacity; + validate_sharding_runtime_knobs(&config.system.sharding)?; + + let (senders, mut inboxes, mut reply_inboxes) = + shard_mesh_channels(total_shards, inbox_capacity, reply_inbox_capacity); + let shutdown_flag = Arc::new(AtomicBool::new(false)); + let config = Arc::new(config); + // One owner table per server process, Arc-cloned into every shard's bus so + // any shard's bus reads the same atomic slots that the owning + // shard's installer / disconnect path writes. + let owner_table = Arc::new(ReplicaOwnerTable::new()); + + // Single-shot bundle handoff (see `MetadataHandoff`): shard 0 sends + // one cloned `ServerMetadataBundle` per peer; each peer drains + // exactly one. Bounded to the peer count so shard 0's broadcast + // never blocks past a peer drain. A single-shard deployment (zero + // peers) still needs a non-zero capacity, so clamp up explicitly + // rather than relying on crossfire's internal cap=0 -> 1 promotion. + // If a peer dies before recv, shard 0's `send` eventually sees a + // disconnected channel; the cross-thread shutdown flag drives every + // waiter out of its recv loop if shard 0 panics before broadcasting. + let metadata_peers = shards_count.saturating_sub(1).max(1); + let (metadata_bundle_tx, metadata_bundle_rx) = + crossfire::mpmc::bounded_async::(metadata_peers); + + // Reverse barrier (see `BootstrapBarrier`): every peer sends one + // signal once it finishes loading its on-disk partitions; shard 0 + // drains them all before binding listeners. Bounded to the peer + // count so a sender never blocks (each peer sends exactly once). + let (ready_tx, ready_rx) = crossfire::mpmc::bounded_async::(metadata_peers); + + let mut shard_threads: Vec<(u16, thread::JoinHandle>)> = + Vec::with_capacity(shards_count); + let roster_cells = RosterCells::default(); + // Every shard's metric handles, minted before the threads spawn: each + // shard bumps its own entry, and shard 0's HTTP scrape endpoint registers + // the whole set (counters are Arc-backed, so cross-thread reads see the + // owning shard's bumps). + let shard_metrics_all: Vec = (0..shards_count) + .map(|_| ShardMetrics::for_shard()) + .collect(); + for (idx, assignment) in assignments.into_iter().enumerate() { + #[allow(clippy::cast_possible_truncation)] + let shard_id = idx as u16; + let inbox = inboxes[idx] + .take() + .expect("shard_mesh_channels populates every inbox slot exactly once"); + let reply_inbox = reply_inboxes[idx] + .take() + .expect("shard_mesh_channels populates every reply-inbox slot exactly once"); + let senders_for_shard = senders.clone(); + let config_for_shard = Arc::clone(&config); + let shutdown_flag_for_shard = Arc::clone(&shutdown_flag); + let owner_table_for_shard = Arc::clone(&owner_table); + let metadata_handoff_for_shard = if shard_id == 0 { + MetadataHandoff::Owner { + bundle_tx: metadata_bundle_tx.clone(), + } + } else { + MetadataHandoff::Waiter { + bundle_rx: metadata_bundle_rx.clone(), + } + }; + let barrier_for_shard = if shard_id == 0 { + BootstrapBarrier::Owner { + ready_rx: ready_rx.clone(), + } + } else { + BootstrapBarrier::Waiter { + ready_tx: ready_tx.clone(), + } + }; + + let roster_cells_for_shard = roster_cells.clone(); + let shard_metrics_for_shard = shard_metrics_all.clone(); + let handle = match thread::Builder::new() + .name(format!("shard-{shard_id}")) + .spawn(move || -> Result<(), ServerError> { + run_shard_thread( + shard_id, + total_shards, + current_replica_id, + assignment, + senders_for_shard, + inbox, + reply_inbox, + config_for_shard, + shutdown_flag_for_shard, + metadata_handoff_for_shard, + barrier_for_shard, + owner_table_for_shard, + roster_cells_for_shard, + shard_metrics_for_shard, + ) + }) { + Ok(handle) => handle, + Err(source) => { + // Signal every shard already spawned before propagating, so + // their watchdog loops drive `bus.shutdown(...)` and the + // process can exit instead of hanging on stuck OS threads. + shutdown_flag.store(true, Ordering::Relaxed); + // Drop bootstrap's own channel clones before joining + // survivors. Otherwise a peer waiting on `bundle_rx.recv` + // would never observe the sender side disconnecting and + // would hang until the shutdown watchdog kicks the bus. + drop(metadata_bundle_tx); + drop(metadata_bundle_rx); + drop(ready_tx); + drop(ready_rx); + join_partial_shard_survivors( + shard_threads, + config.system.sharding.shutdown_join_timeout.get_duration(), + ); + return Err(ServerError::ShardSpawnFailed { shard_id, source }); + } + }; + shard_threads.push((shard_id, handle)); + } + + // Drop bootstrap's own channel clones now that every shard owns its + // half. Keeping them on bootstrap's stack would deadlock a peer + // whose `bundle_rx.recv` only completes once every sender + // disconnects. + drop(metadata_bundle_tx); + drop(metadata_bundle_rx); + drop(ready_tx); + drop(ready_rx); + + info!( + shards_count, + "server bootstrap dispatched; awaiting shard runtimes" + ); + + Ok(ShardHandles { + shutdown_flag, + shard_threads, + join_timeout: config.system.sharding.shutdown_join_timeout.get_duration(), + }) +} + +/// Per-shard async lifecycle. Builds the bus, recovers metadata, +/// constructs the `IggyShard` for this shard's slice of partitions, +/// wires listeners on shard 0, and runs the message pump until +/// shutdown. +#[allow(clippy::too_many_arguments, clippy::too_many_lines)] +async fn shard_main( + shard_id: u16, + total_shards: u16, + replica_id: Option, + senders: Vec, + inbox: ShardReceiver, + reply_inbox: ShardReceiver, + config: &ServerConfig, + shutdown_flag: Arc, + metadata_handoff: MetadataHandoff, + barrier: BootstrapBarrier, + owner_table: Arc, + roster_cells: RosterCells, + shard_metrics_all: Vec, +) -> Result<(), ServerError> { + let topology = resolve_tcp_topology(config, replica_id)?; + let bus = Rc::new(IggyMessageBus::with_config_and_owner_table( + shard_id, + config, + owner_table, + )); + // Every shard can own a delegated replica connection, so every + // shard's bus needs the handshake identity (the handshake itself + // runs on the owning shard, not on shard 0). + bus.set_replica_handshake_ctx(ReplicaHandshakeCtx { + cluster_id: topology.cluster_id, + self_id: topology.self_replica_id, + replica_count: topology.replica_count, + auth: load_replica_auth(config).map(Rc::new), + tls: load_replica_tls_ctx(config, &topology)?.map(Rc::new), + }); + + let drain_timeout = config.system.sharding.shutdown_drain_timeout.get_duration(); + let poll_interval = config.system.sharding.shutdown_poll_interval.get_duration(); + + let shutdown_flag_for_handoff = Arc::clone(&shutdown_flag); + let mut shutdown_watchdog = Some(spawn_shutdown_watchdog( + Rc::clone(&bus), + shutdown_flag, + drain_timeout, + poll_interval, + )); + + // Metadata bootstrap is single-writer: shard 0 owns the WAL and the + // only `WriteHandle`-bearing `MuxStateMachine`. Peer shards receive + // a `ReadHandleFactory` bundle on the inter-thread channel and + // rebuild a reader-mode `MuxStateMachine` on their own runtime - no + // WAL access, no replay. Writes still funnel through shard 0's + // metadata VSR; per-commit `publish()` (in `WriteCell::apply`) + // bounds reader staleness to one op. + let data_dir = Path::new(&config.system.path); + let (mux_stm, owner_state) = match metadata_handoff { + MetadataHandoff::Owner { bundle_tx } => { + // Root is created locally at boot (never journaled), so replay + // must start from the same baseline or every WAL-created user + // shifts one slab id and root is lost after the first restart. + let recovered = recover::( + data_dir, + ReplicaIdentity { + cluster: topology.cluster_id, + replica_id: topology.self_replica_id, + replica_count: topology.replica_count, + }, + config.metadata.journal_slots, + config.metadata.clients_table_max, + |mux_stm| { + ensure_default_root_user(mux_stm); + }, + |mux_stm, client, stamp| { + mux_stm + .streams() + .remove_consumer_group_member(client, stamp); + }, + ) + .await + .map_err(ServerError::MetadataRecovery)?; + ensure_default_root_user(&recovered.mux_stm); + // The factory bundle hands every peer a read handle over the + // same `Inner`, so `Arc` (and the parent + // `Arc`) is shared across all shards. Zero the + // snapshot totals here, once, before any peer can observe the + // bundle. Per-shard `load_partition` deltas in + // `build_shard_for_thread` then race only against other + // atomic adds, never against a concurrent `swap(0)` that + // would mistake an in-flight delta for the snapshot total + // and decrement the parent `StreamStats` by it. + let () = recovered.mux_stm.streams().read(|inner| { + for (_, stream) in &inner.items { + for (_, topic) in &stream.topics { + topic.stats.zero_out_all(); + } + } + }); + broadcast_metadata_bundle( + shard_id, + &bundle_tx, + recovered.mux_stm.factory_bundle(), + total_shards.saturating_sub(1), + &shutdown_flag_for_handoff, + poll_interval, + ) + .await?; + ( + recovered.mux_stm, + Some(RecoveredOwnerState { + journal: recovered.journal, + snapshot: recovered.snapshot, + last_applied_op: recovered.last_applied_op, + last_journaled_op: recovered.last_journaled_op, + client_table: recovered.client_table, + superblock: recovered.superblock, + recovered_state: recovered.recovered_state, + snapshot_checkpoint: recovered.snapshot_checkpoint, + }), + ) + } + MetadataHandoff::Waiter { bundle_rx } => { + let bundle = await_metadata_bundle( + shard_id, + &bundle_rx, + &shutdown_flag_for_handoff, + poll_interval, + ) + .await?; + (ServerMuxStateMachine::from_factory_bundle(bundle), None) + } + }; + + // Metadata consensus + journal + snapshot live only on shard 0. + // `IggyShard::tick_metadata` short-circuits when `consensus.is_none()`, + // so peer shards have no caller that reads `journal` or `snapshot`. + let ( + metadata_consensus, + journal_for_metadata, + snapshot_for_metadata, + superblock_for_metadata, + checkpoint_seed, + recovered_client_table, + ) = if let Some(owner) = owner_state { + // `recover()` already opened the superblock, read `recovered_state`, and + // verified the on-disk snapshot against its checkpoint pairing BEFORE decoding + // it. Reuse that superblock rather than re-opening it, which would fork the + // ping-pong sequence counter. Consensus recovers its true (view, log_view) + // from `recovered_state` instead of inferring a stale view from the WAL. + let consensus = restore_metadata_consensus(&owner, &topology, config, Rc::clone(&bus)); + let superblock = Rc::new(owner.superblock); + ( + Some(consensus), + Some(owner.journal), + owner.snapshot, + Some(superblock), + owner.snapshot_checkpoint, + Some(owner.client_table), + ) + } else { + (None, None, None, None, (0, 0), None) + }; + let metadata = ServerMetadata::new( + metadata_consensus, + journal_for_metadata, + snapshot_for_metadata, + superblock_for_metadata, + mux_stm, + Some(PathBuf::from(&config.system.path)), + ); + // Size the VSR client table before listeners bind and any client registers. + // Must precede the recovered-table install below: the setter rebuilds the + // table from scratch, so running it afterwards would drop every resumed + // session (and trip its empty-table assert). + metadata.set_clients_table_max(config.metadata.clients_table_max); + // Reinstall the sessions recovery restored from the checkpoint and the WAL + // suffix, so a rebooted node dedups retries and admits continuations from + // clients that kept their identity across the restart (IGGY-137). Recovery + // sized this table from the same config value, so the install preserves the + // configured cap. + if let Some(client_table) = recovered_client_table { + // Refusal (a client registered before this ran) keeps the live table + // and is logged by the callee; boot continues either way. + let _ = metadata.install_client_table(client_table); + } + // Seed the coordinator's last-checkpoint pairing so the first post-boot + // view-change superblock write records the real (checkpoint_op, checksum) + // instead of (0, 0). No-op on peer shards, which have no coordinator. + metadata.seed_checkpoint_ref(checkpoint_seed.0, checkpoint_seed.1); + // Keep the forced-checkpoint margin >= the configured prepare-queue + // depth: ops already pipelined while a checkpoint runs append into that + // margin (config validation keeps journal_slots >= 4x this). + metadata.set_checkpoint_margin(config.metadata.checkpoint_margin()); + + let shard_metrics = shard_metrics_all[usize::from(shard_id)].clone(); + // Notifier install deferred until after tick handler wires below. + let senders_for_notifier = senders.clone(); + let metrics_for_notifier = shard_metrics.clone(); + // Heap-pin like `run_shard_thread` pins `shard_main`: the builder future + // carries the whole shard construction state machine and outgrew clippy's + // `large_futures` cap; one allocation per shard startup. + let (shard, sessions) = Box::pin(build_shard_for_thread( + shard_id, + total_shards, + config, + &topology, + metadata, + Rc::clone(&bus), + senders, + inbox, + reply_inbox, + shard_metrics, + &roster_cells, + )) + .await?; + + // Shard 0 owns the metadata consensus; publish its view so every shard's + // cluster-metadata read (and the SDK's leader discovery) marks the live + // primary. Detached: dies with this shard's runtime at process exit. + if shard_id == 0 { + let publisher_shard = Rc::clone(&shard); + let publisher_view = Arc::clone(&roster_cells.metadata_view); + compio::runtime::spawn(async move { + loop { + if let Some(consensus) = publisher_shard.plane.metadata().consensus.as_ref() { + // While this replica declines its recovered view's + // primaryship, that view must not reach the roster: the + // delegated shards would compute a leader that never + // heartbeats. Publish "unknown" until the election + // resolves the role. + let published = if consensus.has_ceded_primaryship() + && consensus.primary_index(consensus.view()) == consensus.replica() + { + crate::cluster_meta::METADATA_VIEW_UNKNOWN + } else { + u64::from(consensus.view()) + }; + publisher_view.store(published, Ordering::Relaxed); + } + compio::time::sleep(std::time::Duration::from_millis(100)).await; + } + }) + .detach(); + } + + info!( + shard = shard_id, + partitions = shard.plane.partitions().len(), + "server shard initialized" + ); + + // Re-check the cross-thread shutdown flag here, *before* spawning the + // message pump: it keeps the bus' `background_tasks` vec empty on the + // shutdown path, and shard 0 would otherwise still open TCP/QUIC/WS + // listeners for a server that is already tearing down, briefly + // accepting connections that immediately get torn by the watchdog. + // + // The flag is set, so the watchdog is (about to be) driving + // `bus.shutdown()`; await it so the runtime does not drop mid-drain. + if shutdown_flag_for_handoff.load(Ordering::Relaxed) { + if let Some(watchdog) = shutdown_watchdog.take() { + let _ = watchdog.await; + } + return Ok(()); + } + + // Tick handler must install before the notifier so early commits + // do not broadcast ticks whose handler slot is still `None`. + let (reconcile_wake_tx, reconcile_wake_rx) = channel::<()>(1); + let (reconcile_stop_tx, reconcile_stop_rx) = channel::<()>(1); + crate::partition_reconciler::install_tick_handler(&shard, reconcile_wake_tx); + + // Only shard 0 commits metadata. + if shard_id == 0 { + let notifier = make_metadata_commit_notifier(senders_for_notifier, metrics_for_notifier); + shard.plane.metadata().set_commit_notifier(Some(notifier)); + } else { + drop(senders_for_notifier); + drop(metrics_for_notifier); + } + + // The pump task also drives the consensus timer tick (heartbeats, prepare + // retransmit, view-change timeouts) as a select! arm, serialized with frame + // processing - see `run_message_pump`. + let (stop_tx, stop_rx) = channel(1); + let pump_shard = Rc::clone(&shard); + // Owned and awaited by shard_main at exit, NOT `track_background`: the + // background drain runs inside `bus.shutdown()`, which the Ctrl-C path + // never drives (the watchdog stands down when the token fires), so a + // tracked pump would be cancelled by runtime teardown mid final-flush + // and every graceful shutdown would silently drop the committed journal + // tail that had not hit a flush threshold yet. + let pump_shutdown_flag = Arc::clone(&shutdown_flag_for_handoff); + let mut pump_handle = Some(compio::runtime::spawn(async move { + // The pump itself flips the shared flag when a commit fault stops it, + // BEFORE its final flush, so a flush stalling on the failed device + // still reaches the watchdog and the bounded drain. Every sibling + // shard's watchdog drives its own graceful stop off the same flag; + // this shard's watchdog is what fires the token `shard_main` is + // parked on. The store below backstops the one fault the pump can + // only observe after that flip: a partition fenced by the final + // flush itself. + let fatal = pump_shard + .run_message_pump(stop_rx, Arc::clone(&pump_shutdown_flag)) + .await; + if fatal.is_some() { + pump_shutdown_flag.store(true, Ordering::Relaxed); + } + fatal + })); + + let reconciler_ctx = Rc::new(crate::partition_reconciler::ReconcilerCtx::new( + Rc::clone(&shard), + total_shards, + Rc::new(config.clone()), + topology.cluster_id, + topology.self_replica_id, + topology.replica_count, + )); + let reconcile_periodic = config + .system + .sharding + .reconcile_periodic_interval + .get_duration(); + let reconciler_handle = compio::runtime::spawn({ + let ctx = Rc::clone(&reconciler_ctx); + async move { + crate::partition_reconciler::run_reconciler( + ctx, + reconcile_wake_rx, + reconcile_stop_rx, + reconcile_periodic, + ) + .await; + } + }); + bus.track_background(reconciler_handle); + + // Per-shard heartbeat verifier: evicts connections that stop pinging, + // releasing their consumer-group membership. Gated on config so a + // deployment without heartbeats never reaps live sessions. + let heartbeat_stop_tx = if config.heartbeat.enabled { + let (hb_stop_tx, hb_stop_rx) = channel::<()>(1); + let hb_shard = Rc::clone(&shard); + let hb_sessions = Rc::clone(&sessions); + let hb_interval = config.heartbeat.interval.get_duration(); + let hb_handle = compio::runtime::spawn(async move { + crate::dispatch::session_ops::run_heartbeat_verifier( + hb_shard, + hb_sessions, + hb_interval, + hb_stop_rx, + ) + .await; + }); + bus.track_background(hb_handle); + Some(hb_stop_tx) + } else { + None + }; + // Expired-PAT cleaner: shard 0 only (it owns the metadata consensus + // group) and only when enabled. Each pass no-ops unless this node is + // the caught-up metadata primary, so the delete is proposed once and + // replicated to every replica. + let pat_cleaner_stop = if shard_id == 0 && config.personal_access_token.cleaner.enabled { + let (cleaner_stop_tx, cleaner_stop_rx) = channel(1); + let cleaner_shard = Rc::clone(&shard); + let interval = config.personal_access_token.cleaner.interval.get_duration(); + let cleaner_handle = compio::runtime::spawn(async move { + crate::personal_access_token_cleaner::run_pat_cleaner( + cleaner_shard, + cleaner_stop_rx, + interval, + ) + .await; + }); + bus.track_background(cleaner_handle); + Some(cleaner_stop_tx) + } else { + None + }; + + // Segment cleaner: runs on every shard (each replica trims its own log, + // primary and backup alike). Local and unreplicated; gated by the shared + // data-maintenance config. + let segment_cleaner_stop = if config.data_maintenance.messages.cleaner_enabled { + let (stop_tx, stop_rx) = channel(1); + let cleaner_shard = Rc::clone(&shard); + let interval = config.data_maintenance.messages.interval.get_duration(); + let cleaner_handle = compio::runtime::spawn(async move { + crate::segment_cleaner::run_segment_cleaner(cleaner_shard, stop_rx, interval).await; + }); + bus.track_background(cleaner_handle); + Some(stop_tx) + } else { + None + }; + let stop_signals = StopSignals { + pump: stop_tx, + reconciler: reconcile_stop_tx, + heartbeat: heartbeat_stop_tx, + pat_cleaner: pat_cleaner_stop, + segment_cleaner: segment_cleaner_stop, + }; + + // One keep-alive per process, so shard 0 owns it. Started before the + // listeners bind: systemd counts `WatchdogSec=` from unit start, not from + // `READY=1`, so a slow recovery must not look like a hang. + #[cfg(feature = "systemd")] + if shard_id == 0 { + systemd::spawn_watchdog(&bus); + } + + // Listener fence (see `BootstrapBarrier`). Peers still scan live + // shared metadata and load their on-disk partitions in + // `build_shard_for_thread`; the factory-bundle handoff only proves + // they *received* the bundle, not that they finished loading. Shard + // 0 must not accept client traffic until every peer's load scan is + // done, otherwise a partition created by the first client surfaces + // in a still-running scan with no segment dir on disk and aborts the + // node with `CannotReadPartitions`. By this point every shard has + // also spawned its pump + reconciler, so a partition created after + // the fence takes the runtime reconciler path on its owning shard. + match barrier { + BootstrapBarrier::Owner { ready_rx } => { + await_bootstrap_complete( + &ready_rx, + usize::from(total_shards.saturating_sub(1)), + &shutdown_flag_for_handoff, + poll_interval, + ) + .await?; + } + BootstrapBarrier::Waiter { ready_tx } => { + signal_bootstrap_complete( + shard_id, + &ready_tx, + &shutdown_flag_for_handoff, + poll_interval, + ) + .await?; + } + } + + // Listeners (replica + every client transport) bind on shard 0 only. + // Shard 0's coordinator round-robins inbound TCP/WS connections to + // peer shards via fd-transfer. QUIC and TCP-TLS clients terminate + // locally on shard 0 (their per-connection state is non-portable - + // see `LifecycleFrame::ClientWsConnectionSetup` rustdoc). + if shard_id == 0 { + let coord = shard + .coordinator() + .expect("shard 0 always has a coordinator attached by the builder"); + // Reseed the client-id minter above every recovered entry before any + // listener accepts. The counter is per process; the table it must not + // collide with was rebuilt from the previous boot's WAL. Keyed by view + // so a later promotion refolds the table (the minting path calls the + // same method, see `HttpInner::register_session_once`). + let boot_view = shard + .plane + .metadata() + .consensus + .as_ref() + .map_or(0, consensus::VsrConsensus::view); + coord.seed_client_sequence( + boot_view, + shard.plane.metadata().client_table.borrow().client_ids(), + ); + let on_client_request = make_client_request_handler( + &shard, + &sessions, + Arc::clone(&config.system), + config.personal_access_token.max_tokens_per_user, + ); + let (accepted_replica, dialed_replica) = + make_replica_delegation_fns(Rc::clone(&coord), &bus); + let accepted_client = make_shard_zero_client_accept_fns(coord, &bus, on_client_request); + let roster = sessions.borrow().cluster_roster(); + + if let Err(error) = start_tcp_runtime( + &shard, + config, + &topology, + roster, + accepted_replica, + dialed_replica, + accepted_client, + &shard_metrics_all, + ) + .await + { + stop_signals.fire(); + // The bind failure is the primary fault; the drain verdict only + // matters for the log it emits. + let _ = await_pump_drain(pump_handle.take(), config, shard_id).await; + // Neither the flag nor the bus token has fired yet on this path, + // so the watchdog is still idle-looping; awaiting it would hang. + // Detach and let `run_shard_thread`'s unwind flip the flag. + if let Some(watchdog) = shutdown_watchdog.take() { + watchdog.detach(); + } + return Err(error); + } + + // Every enabled client transport is bound and accepting by here, so + // this is the first point at which a unit ordered after us may dial. + #[cfg(feature = "systemd")] + systemd::notify_ready(); + } + + bus.token().wait().await; + #[cfg(feature = "systemd")] + if shard_id == 0 { + systemd::notify_stopping(); + } + stop_signals.fire(); + + // Await the watchdog even when the drain verdict is an error: the token + // has fired, so it either stands down within one poll interval or is + // mid-`bus.shutdown()`, and dropping it there truncates in-flight + // `ClientForwardFailed` replies. + let pump_verdict = await_pump_drain(pump_handle.take(), config, shard_id).await; + if let Some(watchdog) = shutdown_watchdog.take() { + let _ = watchdog.await; + } + pump_verdict?; + + info!(shard = shard_id, "server shard exited cleanly"); + Ok(()) +} + +/// Build the closure that broadcasts a +/// [`LifecycleFrame::MetadataCommitTick`] to every shard's inbox after a +/// partition-shaped metadata operation commits on shard 0. +/// +/// The receiver-side partition reconciliation loop listens for these +/// wake-ups; coalescing is intentional, so `Full` is recorded as a metric +/// and dropped (the periodic tick recovers). Installed via +/// [`metadata::IggyMetadata::set_commit_notifier`] on shard 0 only, the +/// sole writer of the metadata state machine. +fn make_metadata_commit_notifier( + senders: Vec, + metrics: ShardMetrics, +) -> metadata::CommitNotifier { + Rc::new(move |operation: Operation| { + if !operation_triggers_partition_reconcile(operation) { + return; + } + for sender in &senders { + let frame = ShardFrame::lifecycle(LifecycleFrame::MetadataCommitTick); + match sender.try_send(frame) { + Ok(()) => {} + Err(crossfire::TrySendError::Full(_)) => { + metrics.record_frame_drop( + frame_drop_variant::METADATA_COMMIT_TICK, + frame_drop_reason::FULL, + ); + } + Err(crossfire::TrySendError::Disconnected(_)) => { + metrics.record_frame_drop( + frame_drop_variant::METADATA_COMMIT_TICK, + frame_drop_reason::DISCONNECTED, + ); + } + } + } + }) +} + +/// Filter at the broadcast site, keeping unrelated ops off the SDK reply +/// path. Any new partition-shape op must be added here. +/// +/// The bare `CreateTopic` / `CreatePartitions` arms are unreachable: the +/// leader's prepare-builder in `IggyMetadata` rewrites both into their +/// `*WithAssignments` form, stamping each partition's `consensus_group_id` +/// before journaling, so a committed prepare only ever carries the +/// assignment-bearing variant. Kept as defense-in-depth against a future +/// commit path that emits a bare op. +/// +/// "Partition-shape" is not only the partition SET: the purge and truncate +/// ops leave the set intact but advance per-partition state (purge +/// generation, delete watermark) that only the reconciler enforces on disk. +/// Omitting them defers the on-disk effect to the periodic safety tick, +/// stretching a purge's client-visible tail to a full +/// `reconcile_periodic_interval`. `DeleteSegments` is absent by design: the +/// leader rewrites it into `TruncatePartition` before journaling, so no +/// commit ever carries it. +const fn operation_triggers_partition_reconcile(op: Operation) -> bool { + matches!( + op, + Operation::CreateTopic + | Operation::CreateTopicWithAssignments + | Operation::CreatePartitions + | Operation::CreatePartitionsWithAssignments + | Operation::DeleteTopic + | Operation::DeleteStream + | Operation::DeletePartitions + | Operation::PurgeStream + | Operation::PurgeTopic + | Operation::TruncatePartition + ) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn reconciler_driven_ops_broadcast_a_commit_tick() { + // These commit without touching the partition set, so nothing else + // signals the reconciler: `reconcile_partition_purges` and + // `reconcile_segment_truncations` are the only code that turns them + // into on-disk effect, and they run only when a pass runs. Dropping + // one from the filter silently downgrades it to the periodic tick. + for op in [ + Operation::PurgeStream, + Operation::PurgeTopic, + Operation::TruncatePartition, + ] { + assert!( + operation_triggers_partition_reconcile(op), + "{op:?} is enforced by the reconciler and must wake it on commit" + ); + } + assert!( + !operation_triggers_partition_reconcile(Operation::CreateUser), + "ops with no partition-shape effect must stay off the broadcast" + ); + } +} diff --git a/core/server/src/boot/recovery.rs b/core/server/src/boot/recovery.rs new file mode 100644 index 0000000000..943c73c043 --- /dev/null +++ b/core/server/src/boot/recovery.rs @@ -0,0 +1,1388 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Shard construction and the partition recovery it drives. + +use crate::boot::topology::{RosterCells, TcpTopology, build_cluster_roster}; +use crate::boot::wire_shell_handlers; +use crate::partition_helpers::{ + build_partition_fresh, configure_consumer_offsets, ensure_initial_segment, + open_partition_superblock, +}; +use crate::segment_recovery::{RecoveredSegment, load_persisted_segments}; +use crate::server_error::{PartitionRecoveryRefusal, ServerError}; +use crate::session_manager::SessionManager; +use crate::shell::{ + ServerMetadata, ServerShard, ShellHandlers, consensus_timers, repair_retry_ticks, +}; +use configs::server::ServerConfig; +use consensus::{ + ClientTable, JoinMode, LocalPipeline, PipelineEntry, Sequencer, VsrConsensus, VsrRestore, + VsrState, +}; +use iggy_common::{ + Aes256GcmEncryptor, EncryptorKind, IggyByteSize, IggyError, PartitionStats, TopicRuntimeOptions, +}; +use journal::Journal; +use journal::prepare_journal::PrepareJournal; +use journal::superblock::PingPongSuperblock; +use message_bus::IggyMessageBus; +use metadata::ReplicaIdentity; +use metadata::impls::metadata::{IggySnapshot, StreamsFrontend}; +use metadata::stm::snapshot::Snapshot; +use metadata::stm::stream::Partition; +use partitions::{ + IggyIndexWriter, IggyPartition, IggyPartitions, MessagesWriter, PartitionsConfig, +}; +use server_common::sharding::{IggyNamespace, PartitionLocation, ShardId}; +use shard::builder::IggyShardBuilder; +use shard::metrics::ShardMetrics; +use shard::shards_table::{PapayaShardsTable, ShardsTable, calculate_shard_assignment}; +use shard::{ + CoordinatorConfig, PartitionConsensusConfig, Receiver as ShardReceiver, ShardFrame, + ShardIdentity, TaggedSender, +}; +use std::cell::RefCell; +use std::path::PathBuf; +use std::rc::Rc; +use std::sync::Arc; +use std::sync::atomic::Ordering; +use std::time::Duration; +use tracing::{error, info, warn}; + +#[allow(clippy::too_many_arguments, clippy::too_many_lines)] +pub(in crate::boot) async fn build_shard_for_thread( + shard_id: u16, + total_shards: u16, + config: &ServerConfig, + topology: &TcpTopology, + metadata: ServerMetadata, + bus: Rc, + senders: Vec, + inbox: ShardReceiver, + reply_inbox: ShardReceiver, + metrics: ShardMetrics, + roster_cells: &RosterCells, +) -> Result<(Rc, Rc>), ServerError> { + let shard_local_id = ShardId::new(shard_id); + let total_partitions = metadata.mux_stm.streams().read(|inner| { + inner + .items + .iter() + .map(|(_, stream)| { + stream + .topics + .iter() + .map(|(_, topic)| topic.partitions.len()) + .sum::() + }) + .sum::() + }); + + // IggyPartitions holds only the partitions owned by this shard + // (see the filter below at insert time), so the server-wide total + // is an N-fold overshoot. `ceil(total / shards) * 2` is a coarse + // upper bound that absorbs hash skew without paying the full + // multiplier. PapayaShardsTable below stays sized to the server-wide + // total because every shard routes every namespace. + let owned_partitions_capacity = total_partitions + .div_ceil(usize::from(total_shards).max(1)) + .saturating_mul(2); + // At-rest encryption: built once per shard from the shared config; the + // ingestion path encrypts on the primary and the poll reply decrypts. + // A bad key fails the boot rather than silently serving plaintext. + let encryptor = if config.system.encryption.enabled { + let aes = Aes256GcmEncryptor::from_base64_key(&config.system.encryption.key) + .map_err(|error| ServerError::Iggy(Box::new(error)))?; + Some(Arc::new(EncryptorKind::Aes256Gcm(aes))) + } else { + None + }; + let partitions = IggyPartitions::with_capacity( + shard_local_id, + PartitionsConfig { + messages_required_to_save: iggy_common::DEFAULT_MESSAGES_REQUIRED_TO_SAVE, + size_of_messages_required_to_save: IggyByteSize::from( + iggy_common::DEFAULT_SIZE_OF_MESSAGES_REQUIRED_TO_SAVE, + ), + enforce_fsync: iggy_common::DEFAULT_ENFORCE_FSYNC, + validate_checksum: config.system.partition.validate_checksum, + segment_size: IggyByteSize::from(iggy_common::DEFAULT_SEGMENT_SIZE), + preallocate_segments: iggy_common::DEFAULT_PREALLOCATE_SEGMENTS, + encryptor, + path_layout: partitions::PartitionPathLayout { + streams_root: config.system.get_streams_path(), + topics_dir: config.system.topic.path.clone(), + partitions_dir: config.system.partition.path.clone(), + }, + }, + owned_partitions_capacity, + ); + let shards_table = PapayaShardsTable::with_capacity(total_partitions); + + // Stream-filter inside the `read()` closure: only partitions owned by + // this shard need the heavy (`Arc` + `Partition`) clones + // for the async `load_partition` below. Non-owning entries are pushed + // straight into `shards_table` here, so no Vec scales with the + // server-wide partition count. + let owned = metadata.mux_stm.streams().read(|inner| { + let mut owned = Vec::with_capacity(owned_partitions_capacity); + for (_, stream) in &inner.items { + for (topic_id, topic) in &stream.topics { + for partition in &topic.partitions { + let namespace = IggyNamespace::new(stream.id, topic_id, partition.id); + let owning_shard = + calculate_shard_assignment(&namespace, u32::from(total_shards)); + if owning_shard == shard_id { + // Shared per-partition stats from the registry: the + // same `Arc` backs every shard's `get_topic` reply. + let stats = inner.stats_registry.partition( + stream.id, + topic_id, + partition.id, + topic.stats.clone(), + ); + owned.push(( + stream.id, + topic_id, + stats, + partition.clone(), + TopicRuntimeOptions::from_resource_options(&topic.options), + )); + } else { + shards_table.insert( + namespace, + PartitionLocation::new( + ShardId::new(owning_shard), + partition.created_revision, + ), + ); + } + } + } + } + owned + }); + + // Snapshot totals were zeroed once on shard 0 before the factory + // bundle was broadcast (see `MetadataHandoff::Owner`). All shards + // here only add their per-partition deltas, so the shared + // `Arc` atomics race only against other atomic adds. + for (stream_id, topic_id, partition_stats, partition_metadata, topic_runtime) in owned { + let namespace = IggyNamespace::new(stream_id, topic_id, partition_metadata.id); + let partition = match load_partition( + config, + namespace, + Arc::clone(&partition_stats), + &partition_metadata, + topic_runtime, + topology.cluster_id, + topology.self_replica_id, + topology.replica_count, + Rc::clone(&bus), + ) + .await + { + Ok(partition) => partition, + // ONE damaged local chain must not take the node down. The shapes + // this refuses are structural -- what a failed state-transfer + // quarantine leaves behind, or damage the recovery walk proved + // inside a segment. What follows depends on whether a peer can + // restore the data. With peers, the segment files are fenced + // aside (keeping the superblock so the group cannot re-enter + // view 0), the group is materialised fresh, and the ordinary + // rejoin path (repair, then state transfer on a refused floor) + // refills it. Single-replica, only a chain-shape refusal whose + // planned chain provably holds ZERO recoverable bytes still + // fences and rebuilds: nothing servable is at stake, so an empty + // rebuild hides no loss. The verdict variant alone is not that + // evidence -- a hole and an orphan empty segment both fire over + // fully populated chains -- which is why the gate reads the byte + // total the refusal carries. Every other refusal tombstones, + // leaving its files exactly where they are: a rebuilt empty + // partition answers polls exactly like a healthy empty one and + // hides the loss, while an unrouted namespace is a failure an + // operator can see. + Err(ServerError::PartitionRecoveryRefused { dir, reason, .. }) => { + let partition_dir = dir.to_string_lossy().into_owned(); + let rebuild_for_rejoin = topology.replica_count > 1 + || matches!( + reason, + PartitionRecoveryRefusal::Hole { + recoverable_bytes: 0, + .. + } | PartitionRecoveryRefusal::EmptyNonTailSegment { + recoverable_bytes: 0, + .. + } + ); + error!( + stream_id, + topic_id, + partition_id = partition_metadata.id, + partition_dir, + %reason, + "refusing the recovered segment chain" + ); + // A pass-A refusal folded nothing into the stats (recovery + // counts only accepted chains), but the hydrate-reopen refusal + // arrives after a fully counted load, so clear them either way. + partition_stats.zero_out_all(); + if !rebuild_for_rejoin { + // No quarantine here, mirroring the superblock arm below: + // a tombstone is only durable if its cause is. Fencing the + // chain aside would leave the next boot zero segments to + // walk, so it would re-seed from the surviving superblock, + // plant a fresh segment, and serve the partition empty + // with no refusal logged. Left at their real paths, the + // same files re-derive this verdict (and this log line) + // every boot, and the reconciler's tombstone gate keeps + // the namespace away from a fresh build, whose + // initial-segment open would truncate the oldest refused + // segment in place. The one refusal whose cause is NOT + // durable is `StorageSizeMismatch`: it fires from the + // reopen right after recovery truncated the same file, so + // the next boot re-walks the already-truncated bytes and, + // unless the length diverges again, accepts the chain + // instead of re-tombstoning -- acceptable for an + // assertion that the filesystem lied about a length. + // `%reason` repeated on purpose: this is the line an + // operator greps to enumerate dark partitions, so it has + // to carry the verdict on its own. + error!( + stream_id, + topic_id, + partition_id = partition_metadata.id, + partition_dir, + %reason, + "no peer replica holds this partition's data; leaving the refused \ + segment files in place and tombstoning it instead of serving it \ + empty" + ); + partitions.tombstone(namespace); + continue; + } + match partitions::state_transfer::quarantine_segment_files(&partition_dir).await { + Ok(fenced_dir) => error!( + stream_id, + topic_id, + partition_id = partition_metadata.id, + fenced_dir, + "quarantined the refused segment files; they are kept for inspection" + ), + Err(error) => { + // NOT rebuilt: `build_partition_fresh` reaches + // `ensure_initial_segment`, which opens segment 0 with + // `file_exists = false` and TRUNCATES whatever the + // failed quarantine left behind. The likeliest failures + // (suffix cap exhausted, `create_dir_all`) move zero + // files, so rebuilding would destroy the oldest segment + // on the first attempt while the higher-offset survivors + // keep refusing every boot -- a loop that never + // terminates and eats the chain one segment at a time. + // Tombstone instead: the namespace stays unmaterialised + // and unrouted, the reconciler backs off, and an + // operator still has every byte. + error!( + stream_id, + topic_id, + partition_id = partition_metadata.id, + partition_dir, + %error, + "failed to quarantine the refused segment files; leaving this \ + partition tombstoned rather than rebuilding over them" + ); + partitions.tombstone(namespace); + continue; + } + } + build_partition_fresh( + config, + namespace, + partition_stats, + partition_metadata.created_revision, + topic_runtime, + topology.cluster_id, + topology.self_replica_id, + topology.replica_count, + partition_metadata.created_view, + Rc::clone(&bus), + ) + .await? + } + // An untrustworthy superblock fences ONE group, not the node. The + // segment files stay exactly where they are -- unlike a refused + // chain, the data on disk is not the thing in doubt -- so there is + // nothing to quarantine and nothing to rebuild: rebuilding fresh + // would hand this replica a view-0 identity while a record it + // cannot read says otherwise. Tombstoned, the namespace stays + // unmaterialised and unrouted, the reconciler backs off, and an + // operator has every byte plus a message naming the directory. + Err( + error @ (ServerError::PartitionSuperblockIo { .. } + | ServerError::PartitionSuperblockVersionUnknown { .. } + | ServerError::PartitionSuperblockUnverifiable { .. } + | ServerError::PartitionSuperblockUndecodable { .. } + | ServerError::PartitionSuperblockIdentityMismatch { .. }), + ) => { + error!( + stream_id, + topic_id, + partition_id = partition_metadata.id, + %error, + "cannot trust this partition's durable consensus state; tombstoning the \ + partition and continuing to boot the rest of the shard" + ); + partition_stats.zero_out_all(); + partitions.tombstone(namespace); + continue; + } + Err(error) => return Err(error), + }; + partitions.insert(namespace, partition); + shards_table.insert( + namespace, + PartitionLocation::new(ShardId::new(shard_id), partition_metadata.created_revision), + ); + } + + let shard_handle = Rc::new(RefCell::new(None)); + // Same wiring path as the simulator's shell mode: one per-shard + // SessionManager shared by the client-request handler (binds sessions) + // and the get_clients handler (reads them). It also carries this shard's + // cluster roster for the GetClusterMetadata read. + let ShellHandlers { + on_replica_message, + on_client_request, + on_metadata_submit, + on_list_clients, + on_partition_read, + sessions, + } = wire_shell_handlers( + &bus, + &shard_handle, + Arc::clone(&config.system), + config.personal_access_token.max_tokens_per_user, + ); + sessions + .borrow_mut() + .set_cluster_roster(Rc::new(build_cluster_roster( + shard_id, + config, + topology, + roster_cells, + )?)); + let shard_name = format!("server-shard-{shard_id}"); + let built = IggyShardBuilder::new( + ShardIdentity::new(shard_id, shard_name), + Rc::clone(&bus), + on_replica_message, + on_client_request, + on_metadata_submit, + on_list_clients, + on_partition_read, + metadata, + partitions, + senders, + inbox, + reply_inbox, + shards_table, + PartitionConsensusConfig::new( + topology.cluster_id, + shard::ReplicaTopology::new(topology.self_replica_id, topology.replica_count), + Rc::clone(&bus), + ), + CoordinatorConfig { + skip_shard_zero_for_replicas: config.cluster.coordinator.skip_shard_zero_for_replicas, + skip_shard_zero_for_clients: config.cluster.coordinator.skip_shard_zero_for_clients, + }, + metrics, + ) + .build() + .map_err(ServerError::ShardConstruction)?; + + let shard = Rc::new(built.shard); + // Repair pacing is shared by both planes' repair loops, so it is a + // per-shard tunable set once here rather than per consensus group. + shard.set_repair_retry_ticks(repair_retry_ticks(config)); + shard.set_superblock_wedged_fatal_failures(superblock_wedged_fatal_failures(config)); + shard.set_served_segment_cache_bytes_max( + config + .partition + .transfer_served_cache_bytes_max + .as_bytes_u64(), + ); + shard.set_partition_artifact_len_max( + config.partition.transfer_artifact_bytes_max.as_bytes_u64(), + ); + shard.set_repair_chunk_max(config.cluster.repair_chunk_max as u64); + // Bounds a served state-transfer chunk. A frame above the bus ceiling is + // rejected by the RECEIVING transport, which tears the replica connection + // down rather than dropping one message. + shard.set_bus_max_message_size( + usize::try_from(config.message_bus.max_message_size.as_bytes_u64()).unwrap_or(usize::MAX), + ); + *shard_handle.borrow_mut() = Some(Rc::downgrade(&shard)); + Ok((shard, sessions)) +} + +// Pin the configs-crate default literals (duplicated there to avoid a +// build-time edge onto the runtime crates) against the runtime constants, +// mirroring the message_bus IOV_MAX pin. A drift on either side fails this +// crate's build until both are reconciled. +const _: () = assert!( + configs::metadata::DEFAULT_METADATA_PREPARE_QUEUE_DEPTH + == consensus::PIPELINE_PREPARE_QUEUE_MAX +); +const _: () = assert!( + configs::metadata::DEFAULT_METADATA_JOURNAL_SLOTS + == journal::prepare_journal::DEFAULT_SLOT_COUNT +); +const _: () = assert!( + configs::partition::DEFAULT_PARTITION_PREPARE_QUEUE_DEPTH + == consensus::PIPELINE_PREPARE_QUEUE_MAX +); +const _: () = + assert!(configs::metadata::DEFAULT_METADATA_CLIENTS_TABLE_MAX == consensus::CLIENTS_TABLE_MAX); +const _: () = + assert!(configs::cluster::DEFAULT_VIEW_PROBE_ATTEMPTS_MAX == consensus::PROBE_ATTEMPTS_MAX); +const _: () = + assert!(configs::partition::DEFAULT_EVICTED_RING_CAPACITY == partitions::EVICTED_RING_CAPACITY); +const _: () = assert!( + configs::partition::DEFAULT_EVICTED_RING_BYTES_MAX == partitions::EVICTED_RING_BYTES_MAX +); +const _: () = assert!( + configs::partition::DEFAULT_TRANSFER_ARTIFACT_BYTES_MAX + == shard::PARTITION_ARTIFACT_LEN_DEFAULT +); +const _: () = assert!( + configs::partition::DEFAULT_TRANSFER_SERVED_CACHE_BYTES_MAX + == shard::SERVED_SEGMENT_CACHE_BYTES_DEFAULT +); +const _: () = assert!(configs::cluster::DEFAULT_REPAIR_CHUNK_MAX as u64 == shard::REPAIR_CHUNK_MAX); +const _: () = assert!( + configs::cluster::STATE_CHUNK_HEADER_LEN + == size_of::() as u64 +); +// Both prepare-queue ceilings are pinned by the view-change wire, not by memory: a +// `DoViewChange` carries the sender's suffix spanning `commit..=op` with one nack +// bit and one present bit per entry, each bitset a single `u128`. The depth bounds +// `op - commit`, so a depth at or above `DVC_HEADERS_MAX` produces entries the new +// primary can neither adopt nor prove dead. Strictly less than, because the head op +// needs the reserved slot. +const _: () = + assert!(configs::metadata::MAX_METADATA_PREPARE_QUEUE_DEPTH < consensus::DVC_HEADERS_MAX); +const _: () = + assert!(configs::partition::MAX_PARTITION_PREPARE_QUEUE_DEPTH < consensus::DVC_HEADERS_MAX); +// `DVC_HEADERS_MAX` is a bare literal in both the wire crate, which sizes the +// bitsets, and the consensus crate, which cannot depend on it the other way around. +// Same u128, so a drift lets one side address entries the other cannot. +const _: () = + assert!(consensus::DVC_HEADERS_MAX == iggy_binary_protocol::consensus::DVC_HEADERS_MAX); +const _: () = assert!(consensus::DVC_HEADERS_MAX == u128::BITS as usize); + +/// `[cluster] superblock_wedged_fatal_timeout` as a consecutive-failure count. +/// Retries pin at the backoff cap after warmup, so the window divided by +/// [`journal::superblock::SUPERBLOCK_RETRY_BACKOFF_MAX_MICROS`] bounds how +/// long a wedged replica may limp before it fail-stops. Zero stays zero +/// (fail-stop disabled). +fn superblock_wedged_fatal_failures(config: &ServerConfig) -> u64 { + superblock_window_to_failures( + config + .cluster + .superblock_wedged_fatal_timeout + .get_duration(), + ) +} + +fn superblock_window_to_failures(window: Duration) -> u64 { + if window.is_zero() { + return 0; + } + let cap_micros = u128::from(journal::superblock::SUPERBLOCK_RETRY_BACKOFF_MAX_MICROS); + u64::try_from((window.as_micros() / cap_micros).max(1)).unwrap_or(u64::MAX) +} + +/// Floor for the post-restart read-recovery deadline (see +/// [`recovery_barrier_deadline`]). At and below the 5s default heartbeat the +/// worst-case recovery is dominated by the heartbeat-independent term - the +/// `ViewChangeStatus` backstop plus election ceremony and suffix recommit, +/// empirically ~7s - so the scaled value must never fall under this or a +/// fast-heartbeat cluster would 503 legitimate reads mid-recovery. The backstop +/// is the configurable `[cluster] view_change_status_timeout`; raising it past +/// its 5s default is why `recovery_barrier_deadline` scales that knob in too +/// rather than leaning on this floor to cover it. +const RECOVERY_BARRIER_DEADLINE_FLOOR: Duration = Duration::from_secs(15); + +/// Safety factor applied to each scaled term of the recovery deadline: a slower +/// heartbeat stretches election and suffix recommit proportionally, and a wider +/// status backstop stretches the ceremony it bounds. 3x reproduces the +/// empirically chosen 15s margin at the shared 5s default (3 x 5s = 15s) and +/// holds that factor as either knob grows. +const RECOVERY_BARRIER_MULTIPLIER: u32 = 3; + +/// How long the post-restart read path waits for the recovered WAL suffix to +/// re-commit before failing loud (retryable 503): the largest of the fixed +/// floor, a `[cluster] heartbeat_timeout`-scaled window, and a +/// `[cluster] view_change_status_timeout`-scaled window. Both knobs feed it +/// because either, raised far past its default, stretches worst-case recovery +/// past the fixed floor; see `await_recovery_barrier` for the read-side wait. +fn recovery_barrier_deadline(heartbeat: Duration, view_change_status: Duration) -> Duration { + // saturating: neither timeout has a config ceiling, plain `*` panics + heartbeat + .saturating_mul(RECOVERY_BARRIER_MULTIPLIER) + .max(view_change_status.saturating_mul(RECOVERY_BARRIER_MULTIPLIER)) + .max(RECOVERY_BARRIER_DEADLINE_FLOOR) +} + +/// Shard 0's half of a metadata recovery: everything [`metadata::impls::recovery::recover`] produced except the +/// state machine, which every shard receives through the factory bundle. +/// +/// Named rather than a positional tuple: the fields are same-typed `Option`s and +/// `(u64, u128)` pairs that a reorder would silently rebind, and one of them decides +/// what view the replica boots into. +pub(in crate::boot) struct RecoveredOwnerState { + pub(in crate::boot) journal: PrepareJournal, + pub(in crate::boot) snapshot: Option, + pub(in crate::boot) last_applied_op: Option, + pub(in crate::boot) last_journaled_op: Option, + pub(in crate::boot) client_table: ClientTable, + pub(in crate::boot) superblock: PingPongSuperblock, + pub(in crate::boot) recovered_state: Option, + pub(in crate::boot) snapshot_checkpoint: (u64, u128), +} + +/// Rebuild metadata consensus from what recovery read off this replica's own disk. +/// +/// Takes the recovery result, topology and config whole rather than the dozen-plus +/// scalars it needs from them: most were `u64` tick counts, where a misordered +/// argument type-checks and mistunes a timeout silently. +pub(in crate::boot) fn restore_metadata_consensus( + owner: &RecoveredOwnerState, + topology: &TcpTopology, + config: &ServerConfig, + bus: Rc, +) -> VsrConsensus> { + let journal = &owner.journal; + let replica_count = topology.replica_count; + let recovered_state = owner.recovered_state; + let snapshot_floor = owner + .snapshot + .as_ref() + .map_or(0, IggySnapshot::sequence_number); + let commit_watermark = owner.last_applied_op.unwrap_or(snapshot_floor); + let restored_op = owner.last_journaled_op.unwrap_or(snapshot_floor); + let recovery_deadline = recovery_barrier_deadline( + config.cluster.heartbeat_timeout.get_duration(), + config.cluster.view_change_status_timeout.get_duration(), + ); + let prepare_queue_depth = config.metadata.prepare_queue_depth; + + let last_header = journal + .last_op() + .and_then(|op| usize::try_from(op).ok()) + .and_then(|op| journal.header(op).map(|header| *header)); + // On a RESTART in a cluster, rejoin as a quorum-invisible backup and + // probe for the current view (`RequestStartView`): the view's primary + // answers with a `StartView`, the replica adopts it as a backup, and + // journal repair fills any WAL gap. A probing replica never resumes + // primaryship -- if this replica IS the current primary-by-index, its + // probe makes the backups elect past it. + // The probe re-broadcasts on its timeout, so it needs no live mesh at + // boot. A FRESH boot keeps the plain init: the cluster needs its view-0 + // primary to exist, and a single-replica cluster has no peer to ask. + // + // Prior life is EITHER a non-empty WAL or a recovered superblock. A view + // change persists without touching the WAL, so a replica that changed + // view before its first metadata write comes back with a non-zero view + // and an empty journal; gating on the WAL alone would `init()` it into + // `Status::Normal` as primary for a view the cluster may have moved past, + // with `ceded_primaryship` false and no probe to correct it. + // + // The rejoin also awaits a state transfer: snapshot-shaped metadata state + // (snapshot + client table) is replaced from the live primary the probe + // finds, then journal repair fills the tail. If the probe exhausts + // instead -- full-cluster bootstrap, nobody live to fetch from -- the + // election fallback clears the stage and this local recovery stands. + let join = if replica_count > 1 && (restored_op > 0 || recovered_state.is_some()) { + JoinMode::ProbeAsBackup { + await_state_transfer: true, + } + } else { + JoinMode::Init + }; + let timers = consensus_timers(config); + let consensus = VsrConsensus::restored( + topology.cluster_id, + topology.self_replica_id, + replica_count, + server_common::sharding::METADATA_GROUP, + bus, + // Request queue keeps the stock 2x ratio over the prepare queue + // (32 -> 64 at defaults): buffered requests are cheap relative to + // in-flight prepares and drain as prepares commit. + LocalPipeline::with_capacities(prepare_queue_depth, prepare_queue_depth * 2), + VsrRestore { + timers: &timers, + // View and log_view come from the durable superblock when present. + // A present but unreadable superblock already refused boot in + // `recover()`, so no durable record means genuinely absent: a + // fresh node, or one that took writes but never checkpointed or + // changed view. There, inferring the view from the last WAL + // prepare is safe, since the persist-before-send gate guarantees + // this replica never externalized a view beyond what a re-probe + // re-derives, and it re-probes as a backup. + durable_view: recovered_state.map(|state| (state.view, state.log_view)), + view_fallback: last_header.map(|header| header.view), + // Metadata, not a partition group: it has a journal to infer from + // and no second plane to line up with. + seed_view: None, + // Fresh random incarnation each boot, so a StartView addressed to + // a previous incarnation still in flight is ignored + // (`handle_start_view` guard). `| 1` guarantees the non-zero the + // guard treats as set. The deterministic simulator overrides this + // with a seed-derived value bumped per restart. + incarnation: Some(rand::random::() | 1), + join, + }, + ); + consensus.sequencer().set_sequence(restored_op); + // A SOLO replica's durable journal head IS its commit point: quorum is + // 1-of-1, so an entry commits the instant it is durable, and the acks + // the cluster ceremony below would wait on cannot topologically exist. + // The embedded watermark is structurally one op stale (the commit point + // is only ever written down inside the NEXT entry), so trusting it solo + // manufactures an "uncommitted" suffix that provably committed and + // wedges the recovery barrier forever. + let commit_watermark = if replica_count == 1 { + restored_op + } else { + commit_watermark + }; + // The commit point is restored from the WAL's embedded watermark (each + // journaled prepare carries the primary's commit at send time), NOT from + // the journal head: journaled does not imply committed, and claiming + // commit for the un-quorum'd tail both risks split-brain on a later view + // change and starves the tail of re-replication (it would live in no + // pipeline). The suffix `(commit_watermark, restored_op]` is re-pipelined + // below when this replica is the recovered view's primary. + // + // TODO(hubcio): the watermark is a lower bound (the last entry stamps + // the commit point as of its send). Persisting an explicit (view, + // commit_op) watermark on the commit path would tighten recovery and + // allow refusing boot on an excessive gap; a backup that recovered a + // LONGER tail than the cluster's primary still needs uncommitted-suffix + // truncation when conflicting ops arrive (message repair milestone). + consensus.restore_commit_state(commit_watermark, commit_watermark); + if let Some(header) = last_header { + consensus.set_last_prepare_checksum(header.checksum); + consensus.observe_prepare_timestamp(header.timestamp); + } + + // The WAL's tail past the watermark is prepared-but-not-provably-committed + // state. Until the cluster confirms it (re-pipelined below on a resumed + // primary; via StartView adoption + the local commit walk on a rejoined + // backup), serving reads would show pre-restart state that clients already + // saw acked -- gate them on the barrier regardless of role. If the suffix + // never re-commits cluster-wide, the read path fails loud with a retryable + // 503 once the paired deadline expires (`await_recovery_barrier`). + if commit_watermark < restored_op { + consensus.set_recovery_barrier(restored_op); + consensus.set_recovery_deadline(recovery_deadline); + } + + // Re-pipeline the prepared-but-uncommitted suffix so the primary's + // retransmit machinery re-replicates it and quorum can (re-)commit it. + // A backup's suffix stays journal-only: the primary's traffic either + // confirms it (re-forward + re-ack path) or supersedes it. + if consensus.is_primary() + && !consensus.has_ceded_primaryship() + && commit_watermark < restored_op + { + info!( + commit_watermark, + restored_op, "re-pipelining recovered uncommitted metadata suffix" + ); + consensus.with_pipeline_mut(|pipeline| { + #[allow(clippy::cast_possible_truncation)] + for op in (commit_watermark + 1)..=restored_op { + let Some(header) = journal.header(op as usize) else { + warn!( + op, + "recovered journal suffix has a gap; stopping re-pipeline" + ); + break; + }; + let mut entry = PipelineEntry::new(*header); + entry.add_ack(topology.self_replica_id); + pipeline.push(entry); + } + }); + // These went in through `Pipeline::push`, not `push_prepare_entry`, and `init` + // no longer arms the timer: without this the recovered suffix sits in the + // pipeline with nothing driving its retransmit. + consensus.sync_prepare_timeout(); + } + + consensus +} + +/// Recover this partition's persisted segment chain, stamping each segment +/// with the topic's effective segment size (the per-topic value when the +/// topic was created with one, else the shard-wide configured size). +/// +/// The topic's effective `enforce_fsync` goes in for the same reason: it is +/// what tells recovery whether a durable index entry the log cannot back is a +/// benign torn index or previously durable data the log lost. +async fn recover_partition_segments( + config: &ServerConfig, + namespace: IggyNamespace, + runtime_options: TopicRuntimeOptions, + stats: &PartitionStats, +) -> Result, ServerError> { + let stream_id = namespace.stream_id(); + let topic_id = namespace.topic_id(); + let partition_id = namespace.partition_id(); + let segment_size = runtime_options + .segment_size + .unwrap_or_else(|| IggyByteSize::from(iggy_common::DEFAULT_SEGMENT_SIZE)); + let enforce_fsync = runtime_options + .enforce_fsync + .unwrap_or(iggy_common::DEFAULT_ENFORCE_FSYNC); + load_persisted_segments( + config, + stream_id, + topic_id, + partition_id, + segment_size, + enforce_fsync, + stats, + ) + .await + .map_err(|source| { + error!( + stream_id, + topic_id, + partition_id, + error = %source, + "failed to load partition log during server bootstrap" + ); + source + }) +} + +#[allow(clippy::too_many_arguments)] +async fn load_partition( + config: &ServerConfig, + namespace: IggyNamespace, + stats: Arc, + partition_metadata: &Partition, + runtime_options: TopicRuntimeOptions, + cluster_id: u128, + self_replica_id: u8, + replica_count: u8, + bus: Rc, +) -> Result>, ServerError> { + let stream_id = namespace.stream_id(); + let topic_id = namespace.topic_id(); + let partition_id = namespace.partition_id(); + // (view, log_view) come from the group's durable superblock when present; + // a present but unverifiable record already refused boot inside + // `open_partition_superblock`. + let partition_dir = config + .system + .get_partition_path(stream_id, topic_id, partition_id); + let (superblock, recovered_state) = open_partition_superblock( + &partition_dir, + ReplicaIdentity { + cluster: cluster_id, + replica_id: self_replica_id, + replica_count, + }, + ) + .await?; + + // A recovered partition lost its journal state with the process: the + // partition journal is in-memory and segments carry no op numbers, so + // this replica cannot know the group's (op, commit) even when the + // superblock restored its view. In a cluster it boots as a + // quorum-invisible backup and probes for the current view + // (`RequestStartView`): the view's primary answers with a `StartView`, + // journal repair fills the rejoin window, and the commit floor settles + // at the serving peer's retention point. The probe re-broadcasts on its + // timeout, so it needs no live mesh at boot. Single-replica groups + // have no peer to ask and keep the plain init. + let join = if replica_count > 1 { + JoinMode::ProbeAsBackup { + await_state_transfer: false, + } + } else { + JoinMode::Init + }; + // Request queue holds 2x the prepare depth (buffered requests drain as + // prepares commit); depth is the per-partition `[partition]` knob. + let prepare_queue_depth = config.partition.prepare_queue_depth; + let timers = consensus_timers(config); + let consensus = VsrConsensus::restored( + cluster_id, + self_replica_id, + replica_count, + namespace.inner(), + bus, + LocalPipeline::with_capacities(prepare_queue_depth, prepare_queue_depth * 2), + VsrRestore { + timers: &timers, + durable_view: recovered_state + .as_ref() + .map(|state| (state.view, state.log_view)), + view_fallback: None, + seed_view: None, + incarnation: None, + join, + }, + ); + + // No prepare-timestamp floor is restored here: the partition consensus + // journal is non-durable today, so there is no persisted head to observe + // (unlike `restore_metadata_consensus`, which observes its restored head). + // When PartitionJournal becomes durable (the milestone named in the + // multi-shard wiring commit body), observe the restored head and the max + // recovered message timestamp here, or an NTP rewind across a restart could + // regress persisted `base_timestamp`. + + let recovered_segments = + recover_partition_segments(config, namespace, runtime_options, &stats).await?; + + let mut partition = IggyPartition::new(stats.clone(), consensus); + partition.set_runtime_options(runtime_options); + partition.set_superblock(superblock, recovered_state.as_ref()); + // Recovered partitions honor the same config-surfaced ring ceilings as the + // fresh-create path (build_partition_fresh). Retention is already off for + // single-replica groups, so this only sizes the multi-replica ring. + partition.log.journal().inner.set_ring_caps( + config.partition.evicted_ring_capacity, + config.partition.evicted_ring_bytes_max.as_bytes_u64(), + ); + partition.set_partition_dir(partition_dir.clone()); + // Before the hydrate: the durable record is keyed by incarnation, so a + // `purge.gen` left behind by a previous life of this namespace reads 0. + partition.set_created_revision(partition_metadata.created_revision); + partition.hydrate_applied_purge_generation().await?; + hydrate_partition_log( + &mut partition, + &partition_dir, + stream_id, + topic_id, + partition_id, + recovered_segments, + ) + .await?; + + let sized_end = partition + .log + .segments() + .iter() + .filter(|segment| segment.size > IggyByteSize::default()) + .map(|segment| segment.end_offset) + .max(); + // An empty chain whose segment is named for a nonzero offset is the + // shape a state-transfer install (or its converge) plants at the group + // frontier after the origin GC'd everything: the file name carries the + // frontier, and re-minting offsets from 0 here would fork this + // replica's batch stamps from the rest of the group after a restart. + let empty_frontier = partition + .log + .segments() + .iter() + .map(|segment| segment.start_offset) + .max() + .filter(|&start| sized_end.is_none() && start > 0); + let current_offset = sized_end.or_else(|| empty_frontier.map(|start| start - 1)); + partition.created_at = partition_metadata.created_at; + partition.recovered_durable_offset = sized_end; + // The OFFSET COUNTER is restored from that file name (above), but the + // `installed_frontier` CLAIM deliberately is not: the claim says "everything + // below me is represented here", and `converge_to_empty_after_failed_install` + // refuses to make it when staged segments were dropped -- yet a converge + // plants exactly the same empty `{frontier:020}.log` a legitimate empty + // install does, so boot provably cannot tell them apart. Re-deriving it here + // would hand the refused claim back: the repair floor stand-in would accept a + // commit floor over ops this replica holds zero bytes for, and the replica + // would pass the serve gate and offer that emptiness onward, making a peer + // unlink its own chain. Leaving it `None` costs one spurious full + // re-transfer on the legitimate empty-install restart; a false caught-up + // claim is not recoverable. A durable home for the frontier (the partition + // superblock already reserves a field) is what would settle it properly. + let counter = current_offset.unwrap_or(0); + partition.offset.store(counter, Ordering::Release); + partition.dirty_offset.store(counter, Ordering::Relaxed); + partition.should_increment_offset = current_offset.is_some(); + // The durable frontier is a LOWER BOUND on top of what the segments proved: + // it is the only carrier left when the segments that named the frontier are + // gone (an all-GC'd origin's install, a crash inside the swap window), and + // taking the max means real recovered data always wins. + partition.restore_offset_frontier(recovered_state.as_ref()); + let current_offset = partition.offset.load(Ordering::Acquire); + + configure_consumer_offsets(&mut partition, config, namespace, current_offset)?; + ensure_initial_segment(&mut partition, config, stream_id, topic_id, partition_id).await?; + + Ok(partition) +} + +/// Reopen writers over a recovered segment chain. +/// +/// Takes no `&ServerConfig`: every knob it needs is the partition's own +/// resolved topic option now, which is the whole point of the per-topic move. +async fn hydrate_partition_log( + partition: &mut IggyPartition>, + partition_dir: &str, + stream_id: usize, + topic_id: usize, + partition_id: usize, + recovered_segments: Vec, +) -> Result<(), ServerError> { + // The partition's own resolved knobs, not the shard-wide config: a topic + // created with `enforce_fsync` or a per-topic `segment_size` must get them + // on the writers reopened over its recovered chain too, or a restart would + // silently drop back to the node defaults. + let runtime = partition.runtime_options(); + let enforce_fsync = runtime + .enforce_fsync + .unwrap_or(iggy_common::DEFAULT_ENFORCE_FSYNC); + let segment_size = runtime + .segment_size + .unwrap_or_else(|| IggyByteSize::from(iggy_common::DEFAULT_SEGMENT_SIZE)); + let preallocate_segments = runtime + .preallocate_segments + .unwrap_or(iggy_common::DEFAULT_PREALLOCATE_SEGMENTS); + for RecoveredSegment { segment, storage } in recovered_segments { + partition + .log + .add_persisted_segment(segment, storage, None, None); + } + + if let Some(active_index) = partition.log.segments().len().checked_sub(1) { + let storage = &partition.log.storages()[active_index]; + if let ( + Some(messages_reader), + Some(index_reader), + Some(storage_messages_writer), + Some(storage_index_writer), + ) = ( + storage.messages_reader.as_ref(), + storage.index_reader.as_ref(), + storage.messages_writer.as_ref(), + storage.index_writer.as_ref(), + ) { + let index_path = index_reader.path(); + let start_offset = partition.log.segments()[active_index].start_offset; + // Share the storage's size counters: they are the write cursors. + // A private counter would let the append position diverge from the + // segment bookkeeping that index entries and poll bounds rely on. + let messages_size_counter = storage_messages_writer.size_counter(); + let index_size_counter = storage_index_writer.size_counter(); + partition.log.messages_writers_mut()[active_index] = Some(Rc::new( + MessagesWriter::new( + &messages_reader.path(), + messages_size_counter, + enforce_fsync, + true, + preallocate_segments.then_some(segment_size), + ) + .await + .map_err(|source| { + error!( + stream_id, + topic_id, + partition_id, + path = %messages_reader.path(), + error = %source, + "failed to initialize persisted messages writer" + ); + hydrate_reopen_error( + source, + partition_dir, + stream_id, + topic_id, + partition_id, + start_offset, + ) + })?, + )); + partition.log.index_writers_mut()[active_index] = Some(Rc::new( + IggyIndexWriter::new(&index_path, index_size_counter, enforce_fsync, true) + .await + .map_err(|source| { + error!( + stream_id, + topic_id, + partition_id, + path = %index_path, + error = %source, + "failed to initialize persisted sparse index writer" + ); + hydrate_reopen_error( + source, + partition_dir, + stream_id, + topic_id, + partition_id, + start_offset, + ) + })?, + )); + } + } + + Ok(()) +} + +/// Routes a hydrate-reopen writer failure. The seed-vs-stat divergence guard +/// (`SegmentSizeMismatchAtOpen`) is a post-condition assertion on recovery's +/// own truncation: pass C truncates every file to its recovered size before +/// storage and writers reopen it, so the guard can only fire if the +/// filesystem lied about a length or a change broke that truncate-then-open +/// contract. Kept as defense-in-depth and routed as a structural refusal +/// because a retried boot cannot help. Every other failure here (open, stat, +/// sync) is transient I/O and stays node-fatal: a retried boot can still +/// serve the partition, while fencing would quarantine healthy data (and at +/// `replica_count = 1` tombstone the partition outright). +fn hydrate_reopen_error( + source: IggyError, + partition_dir: &str, + stream_id: usize, + topic_id: usize, + partition_id: usize, + start_offset: u64, +) -> ServerError { + match source { + IggyError::SegmentSizeMismatchAtOpen(on_disk_bytes, expected_bytes) => { + ServerError::PartitionRecoveryRefused { + dir: PathBuf::from(partition_dir), + stream_id, + topic_id, + partition_id, + reason: PartitionRecoveryRefusal::StorageSizeMismatch { + start_offset, + on_disk_bytes, + expected_bytes, + }, + } + } + transient => transient.into(), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn superblock_fatal_window_converts_to_capped_backoff_retries() { + assert_eq!( + superblock_window_to_failures(Duration::ZERO), + 0, + "zero window must stay the disabled sentinel" + ); + assert_eq!( + superblock_window_to_failures(Duration::from_mins(2)), + 120, + "past warmup one retry rides each 1s backoff cap" + ); + assert_eq!( + superblock_window_to_failures(Duration::from_micros(500)), + 1, + "a sub-cap window still needs one failure to fire" + ); + } + + #[test] + fn default_cluster_heartbeat_timeout_matches_consensus_constant() { + // The config default lives in core/server/config.toml (a string, + // so no static assert can pin it); keep it in lockstep with the + // built-in the simulator and un-configured replicas run on. + let config_default = configs::cluster::ClusterConfig::default() + .heartbeat_timeout + .get_duration() + .as_millis(); + let built_in = u128::from(consensus::TimeoutManager::NORMAL_HEARTBEAT_TICKS) + * shard::CONSENSUS_TICK_INTERVAL.as_millis(); + assert_eq!( + config_default, built_in, + "[cluster] heartbeat_timeout default drifted from \ + TimeoutManager::NORMAL_HEARTBEAT_TICKS" + ); + } + + #[test] + fn recovery_barrier_deadline_holds_the_floor_for_small_heartbeats() { + // Below the 5s default the heartbeat-independent recovery term (~7s of + // ViewChangeStatus backstop plus ceremony) dominates, so the floor + // governs however small the heartbeat is; 3 x 5s lands exactly on it. + // A default-sized status backstop stays on the floor, not above it. + assert_eq!( + recovery_barrier_deadline(Duration::from_secs(1), Duration::from_secs(5)), + RECOVERY_BARRIER_DEADLINE_FLOOR + ); + assert_eq!( + recovery_barrier_deadline(Duration::from_secs(5), Duration::from_secs(5)), + RECOVERY_BARRIER_DEADLINE_FLOOR + ); + } + + #[test] + fn recovery_barrier_deadline_scales_past_the_floor_for_large_heartbeats() { + // Once 3 x heartbeat clears the floor the scaled window governs, so a + // slow-heartbeat cluster is not failed 503 before its longer recovery + // can finish. A default-sized status backstop stays under it. + assert_eq!( + recovery_barrier_deadline(Duration::from_secs(10), Duration::from_secs(5)), + Duration::from_secs(30) + ); + assert_eq!( + recovery_barrier_deadline(Duration::from_secs(15), Duration::from_secs(5)), + Duration::from_secs(45) + ); + } + + #[test] + fn recovery_barrier_deadline_scales_with_the_status_backstop() { + // A raised view-change status backstop stretches worst-case recovery + // even when the heartbeat stays fast, so the deadline must track it or + // post-restart reads 503 before a slow election settles. + assert_eq!( + recovery_barrier_deadline(Duration::from_secs(1), Duration::from_secs(10)), + Duration::from_secs(30) + ); + } + + #[test] + fn recovery_barrier_deadline_at_config_defaults_matches_the_floor() { + // Folding the status term in must not move the stock deadline: at the + // shared 5s defaults each scaled term lands exactly on the 15s floor, + // so an un-tuned cluster keeps its pre-existing recovery window. + let cluster = configs::cluster::ClusterConfig::default(); + assert_eq!( + recovery_barrier_deadline( + cluster.heartbeat_timeout.get_duration(), + cluster.view_change_status_timeout.get_duration(), + ), + RECOVERY_BARRIER_DEADLINE_FLOOR + ); + } + + #[test] + fn recovery_barrier_deadline_saturates_instead_of_panicking() { + // Neither timeout has a config ceiling, so both multiplies must + // saturate rather than abort boot on an absurd parseable value. + assert_eq!( + recovery_barrier_deadline(Duration::MAX, Duration::from_secs(5)), + Duration::MAX + ); + assert_eq!( + recovery_barrier_deadline(Duration::from_secs(5), Duration::MAX), + Duration::MAX + ); + } + + #[test] + fn default_commit_broadcast_interval_matches_consensus_constant() { + // The config default lives in core/server/config.toml (a string, + // so no static assert can pin it); keep it in lockstep with the + // built-in the simulator and un-configured replicas run on. + let config_default = configs::cluster::ClusterConfig::default() + .commit_broadcast_interval + .get_duration() + .as_millis(); + let built_in = u128::from(consensus::TimeoutManager::COMMIT_MESSAGE_TICKS) + * shard::CONSENSUS_TICK_INTERVAL.as_millis(); + assert_eq!( + config_default, built_in, + "[cluster] commit_broadcast_interval default drifted from \ + TimeoutManager::COMMIT_MESSAGE_TICKS" + ); + } + + #[test] + fn default_prepare_retransmit_interval_matches_consensus_constant() { + // The config default lives in core/server/config.toml (a string, + // so no static assert can pin it); keep it in lockstep with the + // built-in the simulator and un-configured replicas run on. + let config_default = configs::cluster::ClusterConfig::default() + .prepare_retransmit_interval + .get_duration() + .as_millis(); + let built_in = u128::from(consensus::TimeoutManager::PREPARE_TICKS) + * shard::CONSENSUS_TICK_INTERVAL.as_millis(); + assert_eq!( + config_default, built_in, + "[cluster] prepare_retransmit_interval default drifted from \ + TimeoutManager::PREPARE_TICKS" + ); + } + + #[test] + fn default_partition_prepare_queue_depth_matches_consensus_constant() { + // The config default lives in core/server/config.toml and flows + // through PartitionConfig::default(); keep the embedded value in + // lockstep with the pipeline depth LocalPipeline::new() (the simulator + // and tests) runs on, so a default deployment is byte-identical. + let config_default = configs::partition::PartitionConfig::default().prepare_queue_depth; + assert_eq!( + config_default, + consensus::PIPELINE_PREPARE_QUEUE_MAX, + "[partition] prepare_queue_depth default drifted from \ + consensus::PIPELINE_PREPARE_QUEUE_MAX" + ); + } + + #[test] + fn default_view_change_retransmit_interval_matches_consensus_constant() { + // The config default lives in core/server/config.toml (a string, so + // no static assert can pin it). One knob drives both view-change + // retransmit timers, which are equal by design, so pin it against both. + let config_default = configs::cluster::ClusterConfig::default() + .view_change_retransmit_interval + .get_duration() + .as_millis(); + let start_view_change = + u128::from(consensus::TimeoutManager::START_VIEW_CHANGE_MESSAGE_TICKS) + * shard::CONSENSUS_TICK_INTERVAL.as_millis(); + let do_view_change = u128::from(consensus::TimeoutManager::DO_VIEW_CHANGE_MESSAGE_TICKS) + * shard::CONSENSUS_TICK_INTERVAL.as_millis(); + assert_eq!( + config_default, start_view_change, + "[cluster] view_change_retransmit_interval default drifted from \ + TimeoutManager::START_VIEW_CHANGE_MESSAGE_TICKS" + ); + assert_eq!( + config_default, do_view_change, + "[cluster] view_change_retransmit_interval default drifted from \ + TimeoutManager::DO_VIEW_CHANGE_MESSAGE_TICKS" + ); + } + + #[test] + fn default_view_change_status_timeout_matches_consensus_constant() { + // The config default lives in core/server/config.toml (a string, so + // no static assert can pin it); keep it in lockstep with the built-in + // the simulator and un-configured replicas run on. + let config_default = configs::cluster::ClusterConfig::default() + .view_change_status_timeout + .get_duration() + .as_millis(); + let built_in = u128::from(consensus::TimeoutManager::VIEW_CHANGE_STATUS_TICKS) + * shard::CONSENSUS_TICK_INTERVAL.as_millis(); + assert_eq!( + config_default, built_in, + "[cluster] view_change_status_timeout default drifted from \ + TimeoutManager::VIEW_CHANGE_STATUS_TICKS" + ); + } + + #[test] + fn default_request_start_view_retransmit_interval_matches_consensus_constant() { + // The config default lives in core/server/config.toml (a string, so + // no static assert can pin it); keep it in lockstep with the built-in + // the simulator and un-configured replicas run on. + let config_default = configs::cluster::ClusterConfig::default() + .request_start_view_retransmit_interval + .get_duration() + .as_millis(); + let built_in = u128::from(consensus::TimeoutManager::REQUEST_START_VIEW_MESSAGE_TICKS) + * shard::CONSENSUS_TICK_INTERVAL.as_millis(); + assert_eq!( + config_default, built_in, + "[cluster] request_start_view_retransmit_interval default drifted from \ + TimeoutManager::REQUEST_START_VIEW_MESSAGE_TICKS" + ); + } + + #[test] + fn default_view_probe_attempts_max_matches_consensus_constant() { + // Belt and suspenders with the static assert above: that pins the + // duplicated configs-crate literal, this pins the shipped config.toml + // value the simulator and un-configured replicas run on. + let config_default = configs::cluster::ClusterConfig::default().view_probe_attempts_max; + assert_eq!( + config_default, + consensus::PROBE_ATTEMPTS_MAX, + "[cluster] view_probe_attempts_max default drifted from \ + consensus::PROBE_ATTEMPTS_MAX" + ); + } + + #[test] + fn default_repair_retry_interval_matches_partitions_constant() { + // The config default lives in core/server/config.toml (a string, so + // no static assert can pin it); keep it in lockstep with the built-in + // the simulator and un-configured replicas run on. + let config_default = configs::cluster::ClusterConfig::default() + .repair_retry_interval + .get_duration() + .as_millis(); + let built_in = + u128::from(partitions::REPAIR_RETRY_TICKS) * shard::CONSENSUS_TICK_INTERVAL.as_millis(); + assert_eq!( + config_default, built_in, + "[cluster] repair_retry_interval default drifted from \ + partitions::REPAIR_RETRY_TICKS" + ); + } + + #[test] + fn default_repair_chunk_max_matches_shard_constant() { + // Belt and suspenders with the static assert above: that pins the + // duplicated configs-crate literal, this pins the shipped config.toml + // value the simulator and un-configured replicas run on. + let config_default = configs::cluster::ClusterConfig::default().repair_chunk_max; + assert_eq!( + config_default as u64, + shard::REPAIR_CHUNK_MAX, + "[cluster] repair_chunk_max default drifted from shard::REPAIR_CHUNK_MAX" + ); + } + + #[test] + fn default_evicted_ring_capacity_matches_partitions_constant() { + // Belt and suspenders with the static assert above; this pins the + // shipped config.toml value. + let config_default = configs::partition::PartitionConfig::default().evicted_ring_capacity; + assert_eq!( + config_default, + partitions::EVICTED_RING_CAPACITY, + "[partition] evicted_ring_capacity default drifted from \ + partitions::EVICTED_RING_CAPACITY" + ); + } + + #[test] + fn default_evicted_ring_bytes_max_matches_partitions_constant() { + // Belt and suspenders with the static assert above; this pins the + // shipped config.toml value. + let config_default = configs::partition::PartitionConfig::default() + .evicted_ring_bytes_max + .as_bytes_u64(); + assert_eq!( + config_default, + partitions::EVICTED_RING_BYTES_MAX, + "[partition] evicted_ring_bytes_max default drifted from \ + partitions::EVICTED_RING_BYTES_MAX" + ); + } +} diff --git a/core/server/src/systemd.rs b/core/server/src/boot/systemd.rs similarity index 100% rename from core/server/src/systemd.rs rename to core/server/src/boot/systemd.rs diff --git a/core/server/src/boot/threads.rs b/core/server/src/boot/threads.rs new file mode 100644 index 0000000000..95de1d5b20 --- /dev/null +++ b/core/server/src/boot/threads.rs @@ -0,0 +1,708 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Shard OS threads: thread entry, pin, join, and the shutdown plumbing. + +use crate::boot::handoff::{BootstrapBarrier, MetadataHandoff}; +use crate::boot::shard_main; +use crate::boot::topology::RosterCells; +use crate::server_error::{ServerError, ShardJoinFailure, ShardJoinFailureKind}; +use crate::shard_allocator::{ShardAllocator, ShardInfo}; +use compio::runtime::ResumeUnwind; +use configs::server::ServerConfig; +use configs::sharding::{ + INBOX_CAPACITY_MAX, SHUTDOWN_DRAIN_TIMEOUT_MAX, SHUTDOWN_POLL_INTERVAL_MAX, +}; +use message_bus::{IggyMessageBus, ReplicaOwnerTable}; +use partitions::FatalCommit; +use server_common::executor::create_shard_executor; +use shard::metrics::ShardMetrics; +use shard::{Receiver as ShardReceiver, Sender, ShardFrame, TaggedSender}; +use std::rc::Rc; +use std::sync::Arc; +use std::sync::atomic::{AtomicBool, Ordering}; +use std::time::{Duration, Instant}; +use std::{panic, thread}; +use tracing::{error, info, warn}; + +/// Result of a multi-shard bootstrap. +/// +/// Carries the cross-thread shutdown flag and one OS-thread `JoinHandle` +/// per shard. The caller flips the flag via [`Self::install_ctrlc_handler`] +/// and then drains every shard via [`Self::join_all`], bounded by +/// `join_timeout` (`system.sharding.shutdown_join_timeout`). +pub struct ShardHandles { + pub(in crate::boot) shutdown_flag: Arc, + pub(in crate::boot) shard_threads: Vec<(u16, thread::JoinHandle>)>, + pub(in crate::boot) join_timeout: Duration, +} + +impl ShardHandles { + /// Install a SIGINT/Ctrl-C handler that flips the shutdown flag on + /// the first signal. A second signal is logged but otherwise + /// ignored so an in-flight WAL fsync or replica drain runs to + /// completion. + /// + /// # Errors + /// + /// Returns the underlying `ctrlc::Error` if the handler cannot be + /// installed (typically because another handler already owns the + /// signal). + pub fn install_ctrlc_handler(&self) -> Result<(), ctrlc::Error> { + let flag = Arc::clone(&self.shutdown_flag); + ctrlc::set_handler(move || { + if flag.swap(true, Ordering::Relaxed) { + // Second Ctrl-C: leave the shutdown machinery to drain. + // Refusing to abort here keeps the WAL fsync / replica + // drain from being interrupted mid-frame. + warn!("second Ctrl-C ignored; server is already shutting down"); + } else { + info!("Ctrl-C received; signalling server shutdown"); + } + }) + } + + /// Drain every shard thread. This is the main thread's park for the + /// server's whole lifetime, so shards are awaited WITHOUT any time + /// bound while the server runs; the `shutdown_join_timeout` clock + /// only starts once the cross-thread shutdown flag flips (Ctrl-C or + /// a shard failure). Each shard's outcome is logged (`info` on clean + /// exit, `error` on Err, panic, or wedge). If any shard failed, + /// returns every failure together as + /// [`ServerError::ShardJoinFailures`] so the operator sees the + /// full set rather than just the first. + /// + /// A shard whose thread is still running when the post-shutdown + /// deadline passes is abandoned (its `JoinHandle` dropped, the OS + /// thread left to die with the process) and reported as + /// [`ShardJoinFailureKind::Wedged`]: a wedged pump or listener must + /// not block process exit forever. + /// + /// # Errors + /// + /// Returns [`ServerError::ShardJoinFailures`] if any shard + /// returned a `Result::Err`, panicked, or wedged past the deadline. + /// The variant carries every per-shard failure in shard-id order so + /// the caller does not need to read the trace log to discover + /// late-failing shards. + pub fn join_all(self) -> Result<(), ServerError> { + let mut failures: Vec = Vec::new(); + // Armed on the first poll that observes the shutdown flag, shared + // across all shards: one budget covers the whole drain, not one + // budget per shard. + let mut deadline: Option = None; + // Shards run thread-per-core with compio's blocking fallback pool + // disabled, so an io_uring opcode the kernel lacks aborts every shard + // with the same panic. Surface the actionable diagnostic once. + let mut io_uring_diagnostic_shown = false; + for (shard_id, handle) in self.shard_threads { + let Some(joined) = join_until_shutdown_deadline( + handle, + &self.shutdown_flag, + self.join_timeout, + &mut deadline, + ) else { + error!( + shard_id, + waited = ?self.join_timeout, + "shard thread still running at the shutdown join deadline; abandoning it" + ); + failures.push(ShardJoinFailure { + shard_id, + kind: ShardJoinFailureKind::Wedged { + waited: self.join_timeout, + }, + }); + continue; + }; + match joined { + Ok(Ok(())) => { + info!(shard_id, "shard thread exited cleanly"); + } + Ok(Err(error)) => { + error!(shard_id, error = %error, "shard thread returned error"); + failures.push(ShardJoinFailure { + shard_id, + kind: ShardJoinFailureKind::Error(Box::new(error)), + }); + } + Err(panic_payload) => { + let message = panic_payload_to_string(&*panic_payload); + error!(shard_id, message = %message, "shard thread panicked"); + if !io_uring_diagnostic_shown + && message + .contains(server_common::diagnostics::ASYNCIFY_POOL_DISABLED_PANIC_MSG) + { + server_common::diagnostics::print_incomplete_io_uring_ops_info(); + io_uring_diagnostic_shown = true; + } + failures.push(ShardJoinFailure { + shard_id, + kind: ShardJoinFailureKind::Panic { message }, + }); + } + } + } + if failures.is_empty() { + Ok(()) + } else { + Err(ServerError::ShardJoinFailures { failures }) + } + } +} + +/// Poll cadence for the bounded shard joins. Coarse enough to cost +/// nothing during a normal drain, fine enough that exit latency past +/// the last shard's return stays imperceptible. +const JOIN_POLL_INTERVAL: Duration = Duration::from_millis(25); + +/// Join `handle`, waiting indefinitely while the server runs. The +/// `join_timeout` clock starts only when `shutdown_flag` is observed set +/// (arming the caller-shared `deadline` once, so all shards drain under +/// ONE budget); a running server parked here for hours must never be +/// mistaken for a wedged shard. `None` means the thread was still +/// running at the post-shutdown deadline and the handle was dropped +/// (the OS thread keeps running detached; process exit reaps it). +/// `JoinHandle` has no timed join, so this polls `is_finished` at +/// [`JOIN_POLL_INTERVAL`]; the closing `join()` on a finished thread +/// returns immediately. +fn join_until_shutdown_deadline( + handle: thread::JoinHandle>, + shutdown_flag: &AtomicBool, + join_timeout: Duration, + deadline: &mut Option, +) -> Option>> { + while !handle.is_finished() { + if deadline.is_none() && shutdown_flag.load(Ordering::Relaxed) { + *deadline = Some(Instant::now() + join_timeout); + } + if let Some(deadline) = deadline + && Instant::now() >= *deadline + { + return None; + } + thread::sleep(JOIN_POLL_INTERVAL); + } + Some(handle.join()) +} + +/// Best-effort extraction of the panic message from a +/// `Box` returned by `JoinHandle::join`. Tries the two +/// payload shapes the standard library guarantees (`&'static str` and +/// `String`) and falls back to a placeholder so the panic still surfaces +/// in the error chain. +fn panic_payload_to_string(payload: &(dyn std::any::Any + Send)) -> String { + if let Some(s) = payload.downcast_ref::<&'static str>() { + return (*s).to_string(); + } + if let Some(s) = payload.downcast_ref::() { + return s.clone(); + } + "".to_string() +} + +/// Joins survivor shard threads after a partial-spawn failure, bounded +/// by the same `shutdown_join_timeout` budget as the normal exit path. +/// +/// Polls every survivor's `is_finished` in one loop instead of spawning +/// per-survivor joiner threads: the likely OS state on this path is +/// `pthread_create` EAGAIN (the parent spawn just failed with it), so +/// nothing here may create threads, and polling drains all survivors in +/// parallel anyway. A survivor still running at the deadline is +/// abandoned with an error log so the failed bootstrap can surface its +/// spawn error instead of hanging on a wedged shard. +pub(in crate::boot) fn join_partial_shard_survivors( + shard_threads: Vec<(u16, thread::JoinHandle>)>, + join_timeout: Duration, +) { + let deadline = Instant::now() + join_timeout; + let mut remaining = shard_threads; + loop { + let mut still_running = Vec::with_capacity(remaining.len()); + for (shard_id, survivor) in remaining { + if survivor.is_finished() { + let _ = survivor.join(); + info!(shard_id, "survivor shard thread drained"); + } else { + still_running.push((shard_id, survivor)); + } + } + remaining = still_running; + if remaining.is_empty() || Instant::now() >= deadline { + break; + } + thread::sleep(JOIN_POLL_INTERVAL); + } + for (shard_id, _survivor) in remaining { + error!( + shard_id, + waited = ?join_timeout, + "survivor shard thread still running at the shutdown join deadline; abandoning it" + ); + } +} + +/// Flips the cross-thread shutdown flag on `Drop` unless disarmed. +/// +/// A shard thread that exits via an error `?` or a panic unwind would +/// otherwise leave sibling shards parked forever on `bus.token().wait()`: +/// their watchdogs never observe the flag and the bus has no +/// `Drop`-triggered shutdown. Arming this for the whole thread body makes +/// every non-clean exit drive sibling-shard teardown. Disarmed only on a +/// clean `Ok(())`. +struct ShutdownOnDrop { + flag: Arc, + armed: bool, +} + +impl ShutdownOnDrop { + const fn new(flag: Arc) -> Self { + Self { flag, armed: true } + } + + const fn disarm(&mut self) { + self.armed = false; + } +} + +impl Drop for ShutdownOnDrop { + fn drop(&mut self) { + if self.armed { + self.flag.store(true, Ordering::Relaxed); + } + } +} + +/// Resolve the operator's `cpu_allocation` into concrete shard +/// assignments plus the checked `u16` shard count. +/// +/// Shard ids index `ReplicaOwnerTable` slots as `u16`. `OWNER_NONE` +/// (`u16::MAX`) is reserved as the empty-slot sentinel, so a server +/// configured with `u16::MAX` shards would mint a shard id that +/// collides with the sentinel and an owner-table lookup could never +/// tell that shard apart from an unowned slot. Reject at boot so the +/// invariant is held by the type system, not by hoping the operator +/// never configures 65535 cores worth of shards. +pub(in crate::boot) fn resolve_shard_assignments( + sharding: &configs::sharding::ShardingConfig, +) -> Result<(Vec, u16), ServerError> { + let allocator = ShardAllocator::new(&sharding.cpu_allocation, sharding.pin_cores) + .map_err(ServerError::ShardAllocator)?; + let assignments = allocator + .to_shard_assignments() + .map_err(ServerError::ShardAllocator)?; + if assignments.is_empty() { + return Err(ServerError::ShardsCountZero); + } + match u16::try_from(assignments.len()) { + Ok(count) if count < message_bus::OWNER_NONE => Ok((assignments, count)), + _ => Err(ServerError::ShardsCountOverflow { + count: assignments.len(), + }), + } +} + +/// Re-validate the runtime sharding knobs that the per-shard runtime +/// consumes directly. Mirrors `ShardingConfig::validate` so a caller +/// that built the config without running it (e.g. tests, embedded +/// usage) cannot OOM at boot or wedge process exit with an out-of-range +/// value. +pub(in crate::boot) fn validate_sharding_runtime_knobs( + sharding: &configs::sharding::ShardingConfig, +) -> Result<(), ServerError> { + let inbox_capacity = sharding.inbox_capacity; + if inbox_capacity == 0 || inbox_capacity > INBOX_CAPACITY_MAX { + return Err(ServerError::InvalidInboxCapacity { + value: inbox_capacity, + max: INBOX_CAPACITY_MAX, + }); + } + let reply_inbox_capacity = sharding.reply_inbox_capacity; + if reply_inbox_capacity == 0 || reply_inbox_capacity > INBOX_CAPACITY_MAX { + return Err(ServerError::InvalidReplyInboxCapacity { + value: reply_inbox_capacity, + max: INBOX_CAPACITY_MAX, + }); + } + let drain_timeout = sharding.shutdown_drain_timeout.get_duration(); + if drain_timeout.is_zero() || drain_timeout > SHUTDOWN_DRAIN_TIMEOUT_MAX { + return Err(ServerError::InvalidShutdownDrainTimeout { + value: drain_timeout, + max: SHUTDOWN_DRAIN_TIMEOUT_MAX, + }); + } + let poll_interval = sharding.shutdown_poll_interval.get_duration(); + if poll_interval.is_zero() || poll_interval > SHUTDOWN_POLL_INTERVAL_MAX { + return Err(ServerError::InvalidShutdownPollInterval { + value: poll_interval, + max: SHUTDOWN_POLL_INTERVAL_MAX, + }); + } + // Ordering: a poll cadence coarser than the drain budget makes the + // cross-thread shutdown flag effectively unobservable during teardown. + if poll_interval > drain_timeout { + return Err(ServerError::ShutdownPollExceedsDrain { + poll: poll_interval, + drain: drain_timeout, + }); + } + Ok(()) +} + +/// Per-shard OS thread entry. Pins CPU + memory, builds the compio +/// runtime, and `block_on`s `shard_main`. +#[allow(clippy::needless_pass_by_value, clippy::too_many_arguments)] +pub(in crate::boot) fn run_shard_thread( + shard_id: u16, + total_shards: u16, + replica_id: Option, + assignment: ShardInfo, + senders: Vec, + inbox: ShardReceiver, + reply_inbox: ShardReceiver, + config: Arc, + shutdown_flag: Arc, + metadata_handoff: MetadataHandoff, + barrier: BootstrapBarrier, + owner_table: Arc, + roster_cells: RosterCells, + shard_metrics_all: Vec, +) -> Result<(), ServerError> { + // Armed for the whole thread body: a post-spawn error `?` or a panic + // unwind here must flip `shutdown_flag` so sibling watchdogs drive + // their bus shutdown instead of parking forever on `bus.token().wait()`. + let mut shutdown_guard = ShutdownOnDrop::new(Arc::clone(&shutdown_flag)); + + assignment + .bind_cpu() + .map_err(|source| ServerError::CpuAffinityFailed { shard_id, source })?; + assignment + .bind_memory() + .map_err(|source| ServerError::MemoryAffinityFailed { shard_id, source })?; + + // `enrich_runtime_create_error` folds the io_uring remediation (raise + // `ulimit -l`, unblock seccomp, kernel-flag floor) into the error, so the + // guidance survives into the shard-join failure report instead of only + // stderr. Multi-shard boxes exhaust RLIMIT_MEMLOCK on per-shard rings + // before the bootstrap runtime does, so this path needs it most. + let runtime = create_shard_executor().map_err(|source| { + let source = server_common::diagnostics::enrich_runtime_create_error(source); + ServerError::ShardRuntimeCreateFailed { shard_id, source } + })?; + + let result = runtime.block_on(async move { + // `shard_main`'s future grows past clippy's `large_futures` cap + // (it ferries the metadata handoff, bus, builders, and inflight + // I/O in one state machine). Heap-pin it so the top-level + // `block_on` future stays small; one allocation per startup buys + // the stack budget back. + Box::pin(shard_main( + shard_id, + total_shards, + replica_id, + senders, + inbox, + reply_inbox, + &config, + shutdown_flag, + metadata_handoff, + barrier, + owner_table, + roster_cells, + shard_metrics_all, + )) + .await + }); + + if result.is_ok() { + shutdown_guard.disarm(); + } + result +} + +/// Await the message pump's completion before the shard returns: its +/// post-loop work includes the final flush of every committed journal to +/// segment storage, and returning first drops the compio runtime, which +/// cancels that flush at its next await point. +/// +/// `Err` means the pump was already dead (a panic, or an exit outside the +/// stop protocol), so its final flush never ran and the shard must not +/// report a clean exit. The verdict is the inner `JoinError`; the timeout +/// wrapper alone cannot see it, and a shard that swallows it prints +/// "exited cleanly" over a corpse. +pub(in crate::boot) async fn await_pump_drain( + pump_handle: Option>>, + config: &ServerConfig, + shard_id: u16, +) -> Result<(), ServerError> { + let Some(pump_handle) = pump_handle else { + return Ok(()); + }; + let drain_budget = config.system.sharding.shutdown_drain_timeout.get_duration(); + let Ok(join_result) = compio::time::timeout(drain_budget, pump_handle).await else { + error!( + shard = shard_id, + timeout = ?drain_budget, + "message pump did not drain within the shutdown budget; \ + committed journal tail may not have flushed" + ); + return Err(ServerError::ShardPumpDrainTimedOut { + shard_id, + timeout: drain_budget, + }); + }; + // `JoinError` renders a panic as the bare "Task has panicked" and the + // type is not re-exported, so the payload -- the only part with + // diagnostic value -- is lifted by re-raising into an immediate catch. + // The panic hook already ran when the task died; `resume_unwind` does + // not run it again, so nothing is printed twice and the message finally + // reaches the tracing sink too. + let reason = match panic::catch_unwind(panic::AssertUnwindSafe(|| join_result.resume_unwind())) + { + Ok(Some(None)) => return Ok(()), + // The pump drained and flushed; it just has nothing left to serve. + // Fail the shard so the process exits non-zero: a node that stopped + // because it could not persist a cluster-committed op must not look + // to an orchestrator like a clean shutdown. + Ok(Some(Some(fault))) => { + error!( + shard = shard_id, + namespace_raw = fault.namespace_raw, + op = fault.op, + operation = ?fault.operation, + "message pump stopped on a partition commit fault; \ + the server is shutting down" + ); + return Err(ServerError::ShardFatal { + shard_id, + namespace_raw: fault.namespace_raw, + op: fault.op, + }); + } + Ok(None) => "task was cancelled".to_string(), + Err(payload) => payload + .downcast_ref::<&str>() + .map(|message| (*message).to_string()) + .or_else(|| payload.downcast_ref::().cloned()) + .map_or_else( + || "task panicked".to_string(), + |message| format!("task panicked: {message}"), + ), + }; + error!( + shard = shard_id, + "message pump died instead of draining ({reason}); \ + committed journal tail may not have flushed" + ); + Err(ServerError::ShardPumpDied { shard_id, reason }) +} + +/// Spawn a per-shard polling task that watches the cross-thread shutdown +/// flag and triggers this shard's bus shutdown on transition. The flag +/// is the only Send signal we have; the bus' shutdown machinery is +/// `!Send` (`Rc>` + per-shard `async_channel`), so it must be +/// triggered from within the runtime that owns the bus. +/// +/// The caller owns the returned handle and must await it on the exit paths +/// where shutdown is in progress (flag set or bus token triggered): +/// dropping it there cancels the watchdog mid-`bus.shutdown()`, truncating +/// in-flight `ClientForwardFailed` replies (terminal per `SendError` docs). +/// It cannot go through `bus.track_background` instead: the watchdog itself +/// drives `bus.shutdown()`, and the bg-drain loop in `shutdown()` would +/// re-enter awaiting the watchdog's own pending shutdown call +/// (self-deadlock). The await is bounded: once the token fires the loop +/// stands down within one poll interval, and the shutdown call itself is +/// capped by `drain_timeout`. +#[allow(clippy::needless_pass_by_value)] +pub(in crate::boot) fn spawn_shutdown_watchdog( + bus: Rc, + shutdown_flag: Arc, + drain_timeout: Duration, + poll_interval: Duration, +) -> compio::runtime::JoinHandle<()> { + let bus_for_task = Rc::clone(&bus); + let bus_token = bus.token(); + compio::runtime::spawn(async move { + loop { + if shutdown_flag.load(Ordering::Relaxed) { + break; + } + if bus_token.is_triggered() { + // Bus shutdown was driven from elsewhere (e.g. internal + // failure path). Watchdog has nothing left to do. + return; + } + compio::time::sleep(poll_interval).await; + } + let _ = bus_for_task.shutdown(drain_timeout).await; + }) +} + +/// Stop senders of the background loops `shard_main` spawns, fired together +/// on `shard_main`'s bind-failure and normal-shutdown exits. +pub(in crate::boot) struct StopSignals { + pub(in crate::boot) pump: Sender<()>, + pub(in crate::boot) reconciler: Sender<()>, + pub(in crate::boot) heartbeat: Option>, + pub(in crate::boot) pat_cleaner: Option>, + pub(in crate::boot) segment_cleaner: Option>, +} + +impl StopSignals { + /// Best-effort: a loop that already exited has dropped its receiver. + pub(in crate::boot) fn fire(&self) { + let _ = self.pump.try_send(()); + let _ = self.reconciler.try_send(()); + for stop in [&self.heartbeat, &self.pat_cleaner, &self.segment_cleaner] + .into_iter() + .flatten() + { + let _ = stop.try_send(()); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn shutdown_on_drop_armed_flips_flag() { + let flag = Arc::new(AtomicBool::new(false)); + drop(ShutdownOnDrop::new(Arc::clone(&flag))); + assert!( + flag.load(Ordering::Relaxed), + "an armed guard must flip the flag on drop (covers the error `?` \ + and panic-unwind exit paths of run_shard_thread)" + ); + } + + #[test] + fn shutdown_on_drop_disarmed_leaves_flag() { + let flag = Arc::new(AtomicBool::new(false)); + let mut guard = ShutdownOnDrop::new(Arc::clone(&flag)); + guard.disarm(); + drop(guard); + assert!( + !flag.load(Ordering::Relaxed), + "a disarmed guard must not flip the flag (clean `Ok(())` exit)" + ); + } + + #[compio::test] + async fn pump_drain_timeout_is_not_reported_as_clean() { + let mut config = ServerConfig::default(); + let timeout = Duration::from_millis(1); + Arc::get_mut(&mut config.system) + .expect("a fresh ServerConfig owns its system config") + .sharding + .shutdown_drain_timeout = iggy_common::IggyDuration::new(timeout); + let pump = compio::runtime::spawn(std::future::pending::>()); + + let error = await_pump_drain(Some(pump), &config, 7) + .await + .expect_err("a live pump past the drain budget is not a clean exit"); + assert!(matches!( + error, + ServerError::ShardPumpDrainTimedOut { + shard_id: 7, + timeout: actual, + } if actual == timeout + )); + } + + #[compio::test] + async fn pump_stopped_by_a_commit_fault_is_not_reported_as_clean() { + // The pump drained and flushed, so the join succeeds. Reporting that + // as a clean exit would hand an orchestrator exit code 0 for a node + // that stopped because it could not persist a cluster-committed op. + let config = ServerConfig::default(); + let fault = FatalCommit { + namespace_raw: 42, + op: 7, + operation: iggy_binary_protocol::Operation::SendMessages, + }; + let pump = compio::runtime::spawn(async move { Some(fault) }); + + let error = await_pump_drain(Some(pump), &config, 3) + .await + .expect_err("a pump that stopped on a commit fault is not a clean exit"); + assert!(matches!( + error, + ServerError::ShardFatal { + shard_id: 3, + namespace_raw: 42, + op: 7, + } + )); + } + + /// Regression: the shutdown-join deadline must arm at SHUTDOWN, not + /// at boot. The original bound measured from `join_all` entry, so any + /// healthy server outliving `shutdown_join_timeout` (30s default) was + /// abandoned as "wedged" and the process exited - every BDD run died + /// at t+30s while the test container was still compiling. + #[test] + fn join_waits_unbounded_while_the_server_runs() { + let shutdown_flag = AtomicBool::new(false); + // Thread outlives a deliberately tiny join budget; with the flag + // clear the budget must never even arm. + let handle = thread::spawn(|| -> Result<(), ServerError> { + thread::sleep(Duration::from_millis(300)); + Ok(()) + }); + let mut deadline = None; + let joined = join_until_shutdown_deadline( + handle, + &shutdown_flag, + Duration::from_millis(20), + &mut deadline, + ); + assert!( + matches!(joined, Some(Ok(Ok(())))), + "a running server must be awaited indefinitely, not abandoned as wedged" + ); + assert!( + deadline.is_none(), + "the join deadline must not arm before the shutdown flag flips" + ); + } + + #[test] + fn join_abandons_a_wedged_shard_after_the_shutdown_deadline() { + let shutdown_flag = AtomicBool::new(true); + // Never finishes: stands in for a wedged pump. The thread leaks + // into the test process, which exits right after. + let handle = thread::spawn(|| -> Result<(), ServerError> { + loop { + thread::sleep(Duration::from_secs(1)); + } + }); + let mut deadline = None; + let joined = join_until_shutdown_deadline( + handle, + &shutdown_flag, + Duration::from_millis(100), + &mut deadline, + ); + assert!( + joined.is_none(), + "a shard still running past the post-shutdown budget must be abandoned" + ); + assert!(deadline.is_some(), "the deadline arms once the flag is set"); + } +} diff --git a/core/server/src/boot/topology.rs b/core/server/src/boot/topology.rs new file mode 100644 index 0000000000..dc673b596b --- /dev/null +++ b/core/server/src/boot/topology.rs @@ -0,0 +1,729 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Listener topology and the cluster roster, resolved from config. + +use crate::cluster_meta::{ + BoundPorts, ClusterRoster, METADATA_VIEW_UNKNOWN, resolved_roster_nodes, + self_advertised_address, +}; +use crate::server_error::ServerError; +use configs::server::ServerConfig; +use message_bus::replica::auth; +use std::net::{IpAddr, SocketAddr}; +use std::sync::Arc; +use std::sync::atomic::AtomicU64; +use tracing::warn; + +const SHARD_REPLICA_ID: u8 = 0; + +pub(in crate::boot) struct TcpTopology { + /// Domain-separation cluster id derived from `cluster.name`; threaded to + /// every consensus instance and the replica handshake so frames agree. + pub(in crate::boot) cluster_id: u128, + pub(in crate::boot) self_replica_id: u8, + pub(in crate::boot) replica_count: u8, + pub(in crate::boot) client_listen_addr: SocketAddr, + pub(in crate::boot) replica_listen_addr: Option, + pub(in crate::boot) ws_listen_addr: Option, + pub(in crate::boot) quic_listen_addr: Option, + pub(in crate::boot) http_listen_addr: Option, + pub(in crate::boot) tcp_tls_listen_addr: Option, + pub(in crate::boot) peers: Vec<(u8, SocketAddr)>, +} + +/// Process-wide cells behind every shard's [`ClusterRoster`], written by +/// shard 0 and read by each shard's cluster-metadata reply: the metadata-group +/// view its consensus publishes, so leader marking works off-shard, and the +/// client ports its listeners bound, so a configured `:0` port reports the +/// one the OS picked on every transport. +#[derive(Clone)] +pub(in crate::boot) struct RosterCells { + pub(in crate::boot) metadata_view: Arc, + pub(in crate::boot) bound_ports: Arc, +} + +/// Unpublished: an unknown view and no bound ports. +impl Default for RosterCells { + fn default() -> Self { + Self { + metadata_view: Arc::new(AtomicU64::new(METADATA_VIEW_UNKNOWN)), + bound_ports: Arc::default(), + } + } +} + +/// Copy the configured cluster roster plus this node's own client ports into +/// the shared [`ClusterRoster`] so the binary `GetClusterMetadata` read serves +/// the real topology. `self_advertised` and `configured_ports` back only the +/// cluster-disabled self-synthesis. The latter carries the configured listener +/// ports from the resolved topology until shard 0 publishes the bound ones +/// through `bound_ports` (a `:0` wildcard resolves once its listener is up). +/// The self address resolves through [`self_advertised_address`], which boot +/// validation has already guaranteed names somewhere a client can dial. +pub(in crate::boot) fn build_cluster_roster( + shard_id: u16, + config: &ServerConfig, + topology: &TcpTopology, + cells: &RosterCells, +) -> Result { + let declared = config.node.advertised_address.as_deref(); + let self_advertised = self_advertised_address(declared, derived_bind_ip(topology, config)); + // The roster answers this per node, so a value here would be read by + // nobody. Silence would leave the operator believing it took effect. + // Every shard builds its own roster off the same config, so keep the + // operator-facing explanation to one line per process. + if declared.is_some() && config.cluster.enabled && shard_id == 0 { + warn!( + "node.advertised_address is set but cluster.enabled is true, so it is ignored; \ + the client-facing address of each node comes from its cluster.nodes entry" + ); + } + Ok(ClusterRoster { + enabled: config.cluster.enabled, + name: config.cluster.name.clone(), + nodes: resolved_roster_nodes(&config.cluster).map_err(ServerError::Config)?, + self_advertised, + configured_ports: configs::cluster::TransportPorts { + tcp: config + .tcp + .enabled + .then(|| topology.client_listen_addr.port()), + quic: topology.quic_listen_addr.map(|addr| addr.port()), + http: topology.http_listen_addr.map(|addr| addr.port()), + websocket: topology.ws_listen_addr.map(|addr| addr.port()), + tcp_replica: None, + }, + bound_ports: Arc::clone(&cells.bound_ports), + metadata_view: Arc::clone(&cells.metadata_view), + }) +} + +pub(in crate::boot) fn resolve_tcp_topology( + config: &ServerConfig, + current_replica_id: Option, +) -> Result { + let default_client_addr = parse_socket_addr("tcp.address", &config.tcp.address)?; + let default_ws_addr = resolve_optional_listener_addr( + config.websocket.enabled, + "websocket.address", + &config.websocket.address, + )?; + let default_quic_addr = + resolve_optional_listener_addr(config.quic.enabled, "quic.address", &config.quic.address)?; + let default_http_addr = + resolve_optional_listener_addr(config.http.enabled, "http.address", &config.http.address)?; + if !config.cluster.enabled { + if let Some(replica_id) = current_replica_id + && replica_id != SHARD_REPLICA_ID + { + return Err(ServerError::ReplicaIdRequiresCluster { + supplied: replica_id, + default: SHARD_REPLICA_ID, + }); + } + return Ok(TcpTopology { + cluster_id: auth::cluster_domain_id(&config.cluster.name), + // Keep parity with the current server binary and the integration + // harness: `--replica-id 0` may be passed unconditionally in + // single-node mode; any other id is rejected above so the WAL + // cannot commit under an identity that will later disagree with + // a cluster.nodes[] entry. + self_replica_id: SHARD_REPLICA_ID, + replica_count: 1, + client_listen_addr: default_client_addr, + replica_listen_addr: Some(SocketAddr::new(default_client_addr.ip(), 0)), + ws_listen_addr: default_ws_addr, + quic_listen_addr: default_quic_addr, + http_listen_addr: default_http_addr, + tcp_tls_listen_addr: config.tcp.tls.enabled.then_some(default_client_addr), + peers: Vec::new(), + }); + } + + let self_replica_id = current_replica_id.ok_or(ServerError::MissingReplicaId)?; + + let self_node = config + .cluster + .nodes + .iter() + .find(|node| node.replica_id == self_replica_id) + .ok_or(ServerError::ClusterNodeNotFound { + replica_id: self_replica_id, + })?; + let replica_count = u8::try_from(config.cluster.nodes.len()).map_err(|_| { + ServerError::ClusterReplicaCountTooLarge { + count: config.cluster.nodes.len(), + } + })?; + let ClusterClientAddrs { + client: client_listen_addr, + ws: ws_listen_addr, + quic: quic_listen_addr, + http: http_listen_addr, + } = resolve_cluster_client_addrs( + self_node, + default_client_addr, + default_ws_addr, + default_quic_addr, + default_http_addr, + )?; + let replica_port = self_node + .ports + .tcp_replica + .ok_or(ServerError::ClusterPortMissing { + transport: "tcp_replica", + replica_id: self_node.replica_id, + })?; + let replica_listen_addr = Some(socket_addr_from_parts( + "cluster.nodes[*].ports.tcp_replica", + &self_node.ip, + replica_port, + )?); + let peers = resolve_cluster_replica_peers(&config.cluster.nodes, self_replica_id)?; + + Ok(TcpTopology { + cluster_id: auth::cluster_domain_id(&config.cluster.name), + self_replica_id, + replica_count, + client_listen_addr, + replica_listen_addr, + ws_listen_addr, + quic_listen_addr, + http_listen_addr, + tcp_tls_listen_addr: config.tcp.tls.enabled.then_some(client_listen_addr), + peers, + }) +} + +fn resolve_optional_listener_addr( + enabled: bool, + context: &'static str, + address: &str, +) -> Result, ServerError> { + if enabled { + return Ok(Some(parse_socket_addr(context, address)?)); + } + Ok(None) +} + +/// Client-facing listener addresses resolved for this cluster node. Each port +/// comes from the node's roster entry; there is no fallback to the top-level +/// listener port, an enabled transport without a roster port refuses to boot. +/// Every transport keeps the bind interface from its own `address` config: the +/// roster ip is advertised, not bound. +struct ClusterClientAddrs { + client: SocketAddr, + ws: Option, + quic: Option, + http: Option, +} + +fn resolve_cluster_client_addrs( + self_node: &configs::cluster::ClusterNodeConfig, + default_tcp_addr: SocketAddr, + default_ws_addr: Option, + default_quic_addr: Option, + default_http_addr: Option, +) -> Result { + let client_port = self_node.ports.tcp.ok_or(ServerError::ClusterPortMissing { + transport: "tcp", + replica_id: self_node.replica_id, + })?; + let client = + merge_roster_port_with_bind_ip("tcp", &self_node.ip, default_tcp_addr, client_port); + let ws = resolve_cluster_optional_addr(self_node, "websocket", default_ws_addr, |ports| { + ports.websocket + })?; + let quic = + resolve_cluster_optional_addr(self_node, "quic", default_quic_addr, |ports| ports.quic)?; + let http = + resolve_cluster_optional_addr(self_node, "http", default_http_addr, |ports| ports.http)?; + Ok(ClusterClientAddrs { + client, + ws, + quic, + http, + }) +} + +fn resolve_cluster_optional_addr( + self_node: &configs::cluster::ClusterNodeConfig, + transport: &'static str, + default_addr: Option, + port_selector: impl Fn(&configs::cluster::TransportPorts) -> Option, +) -> Result, ServerError> { + let Some(default_addr) = default_addr else { + return Ok(None); + }; + // No fallback to the top-level port: two same-host nodes leaving the same + // transport port unset would race for one socket. Either the roster is + // explicit or the server refuses to boot. + let port = port_selector(&self_node.ports).ok_or(ServerError::ClusterPortMissing { + transport, + replica_id: self_node.replica_id, + })?; + Ok(Some(merge_roster_port_with_bind_ip( + transport, + &self_node.ip, + default_addr, + port, + ))) +} + +/// Combine the roster-supplied `port` with the bind interface the transport's +/// own `address` config asked for. +/// +/// The roster ip is what the cluster advertises (metadata, follower-to-primary +/// HTTP forwarding targets); the transport's own `address` decides the bind +/// interface. Merging keeps a loopback-only `127.0.0.1` private and a +/// `0.0.0.0` wide in cluster mode instead of silently rebinding to the roster +/// interface, which would strand every co-located dialer (sidecars, health +/// probes, on-host consumers) on `ECONNREFUSED`. +fn merge_roster_port_with_bind_ip( + transport: &'static str, + roster_ip: &str, + bind_addr: SocketAddr, + port: u16, +) -> SocketAddr { + let listen_addr = SocketAddr::new(bind_addr.ip(), port); + if roster_ip_unreachable_from_bind_addr(roster_ip, listen_addr) { + warn!( + "{transport} listener binds {listen_addr} but the roster advertises {roster_ip}:{port}; \ + peers and clients dialing the advertised endpoint may not reach this node" + ); + } + listen_addr +} + +/// The client-facing listeners paired with the config key naming their bind +/// address, in the order [`ServerConfig::client_listeners`] derives the +/// published client-facing address from them. `None` marks a listener that is +/// switched off and therefore binds nothing. +pub(in crate::boot) fn client_listeners( + topology: &TcpTopology, + config: &ServerConfig, +) -> [(&'static str, Option); 4] { + [ + ( + "tcp.address", + config.tcp.enabled.then_some(topology.client_listen_addr), + ), + ("websocket.address", topology.ws_listen_addr), + ("quic.address", topology.quic_listen_addr), + ("http.address", topology.http_listen_addr), + ] +} + +/// The bind interface the published client-facing address names when none is +/// declared: the first enabled listener's, the same one boot validation gated +/// its wildcard refusal on. With every client listener off nothing dials this +/// node, so the tcp bind address stands in for an answer no client reads. +pub(in crate::boot) fn derived_bind_ip(topology: &TcpTopology, config: &ServerConfig) -> IpAddr { + client_listeners(topology, config) + .into_iter() + .find_map(|(_, listen_addr)| listen_addr) + .unwrap_or(topology.client_listen_addr) + .ip() +} + +/// Whether the address cluster metadata publishes for this node misses one of +/// its own listeners. Only a derived address is judged: it names one +/// listener's bind interface, so a listener on a different one is unreachable +/// at the published address. A declared `node.advertised_address` is +/// deliberate (NAT, a public name) and says nothing about which local +/// interface serves a transport, so it stays quiet. +pub(in crate::boot) fn derived_address_misses_listener( + declared: Option<&str>, + self_advertised: &str, + listen_addr: SocketAddr, +) -> bool { + declared.is_none() && roster_ip_unreachable_from_bind_addr(self_advertised, listen_addr) +} + +/// Whether a dialer aiming at the advertised roster ip misses `listen_addr`. An +/// unspecified bind covers every interface, and a roster ip that parses as +/// neither IPv4 nor IPv6 (a DNS name, say) can resolve to the bound interface, +/// so both cases stay quiet. Both sides reduce to the canonical form first, so +/// the v4-mapped wildcard (`[::ffff:0.0.0.0]`, which a dual-stack host binds as +/// `0.0.0.0`) stays quiet as well and `10.0.0.5` matches `::ffff:10.0.0.5`. +fn roster_ip_unreachable_from_bind_addr(roster_ip: &str, listen_addr: SocketAddr) -> bool { + let bind_ip = listen_addr.ip().to_canonical(); + !bind_ip.is_unspecified() + && roster_ip + .parse::() + .is_ok_and(|parsed| parsed.to_canonical() != bind_ip) +} + +pub(in crate::boot) fn wildcard_listener_under_loopback_address( + declared: Option<&str>, + self_advertised: &str, + listen_addr: SocketAddr, +) -> bool { + declared.is_none() + && listen_addr.ip().to_canonical().is_unspecified() + && self_advertised + .parse::() + .is_ok_and(|address| address.to_canonical().is_loopback()) +} + +fn resolve_cluster_replica_peers( + nodes: &[configs::cluster::ClusterNodeConfig], + self_replica_id: u8, +) -> Result, ServerError> { + let mut peers = Vec::with_capacity(nodes.len().saturating_sub(1)); + for node in nodes { + if node.replica_id == self_replica_id { + continue; + } + let replica_port = node + .ports + .tcp_replica + .ok_or(ServerError::ClusterPortMissing { + transport: "tcp_replica", + replica_id: node.replica_id, + })?; + peers.push(( + node.replica_id, + socket_addr_from_parts("cluster.nodes[*].ports.tcp_replica", &node.ip, replica_port)?, + )); + } + Ok(peers) +} + +fn parse_socket_addr(context: &'static str, address: &str) -> Result { + address + .parse() + .map_err(|source| ServerError::SocketAddressParse { + context, + address: address.to_string(), + source, + }) +} + +fn socket_addr_from_parts( + context: &'static str, + host: &str, + port: u16, +) -> Result { + let ip = host + .parse::() + .map_err(|source| ServerError::SocketAddressParse { + context, + address: format!("{host}:{port}"), + source, + })?; + Ok(SocketAddr::new(ip, port)) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn cluster_node(ip: &str, http: Option) -> configs::cluster::ClusterNodeConfig { + cluster_node_with_ports(ip, Some(18070), http) + } + + fn cluster_node_with_ports( + ip: &str, + tcp: Option, + http: Option, + ) -> configs::cluster::ClusterNodeConfig { + configs::cluster::ClusterNodeConfig { + name: "node".to_owned(), + ip: ip.to_owned(), + advertised_address: None, + advertised_addresses: Vec::new(), + replica_id: 0, + ports: configs::cluster::TransportPorts { + tcp, + http, + ..Default::default() + }, + } + } + + fn addr(value: &str) -> SocketAddr { + value.parse().expect("valid socket address literal") + } + + #[test] + fn cluster_http_addr_takes_port_from_roster() { + // A byte-identical top-level [http].address is shared across nodes on + // one host; the per-node roster port is the only port source so each + // node binds a distinct HTTP socket. + let node = cluster_node("127.0.0.1", Some(18090)); + let addrs = resolve_cluster_client_addrs( + &node, + addr("127.0.0.1:8090"), + None, + None, + Some(addr("127.0.0.1:3000")), + ) + .expect("cluster address resolution must succeed"); + assert_eq!(addrs.http, Some(addr("127.0.0.1:18090"))); + } + + #[test] + fn cluster_http_addr_merges_config_ip_with_roster_port() { + // Docker/Helm bind `0.0.0.0` and probe loopback; the roster ip is + // only the advertised address. Cluster mode must keep the configured + // interface and take just the port from the roster. + let node = cluster_node("10.0.0.5", Some(18090)); + let addrs = resolve_cluster_client_addrs( + &node, + addr("0.0.0.0:8090"), + None, + None, + Some(addr("0.0.0.0:3000")), + ) + .expect("cluster address resolution must succeed"); + assert_eq!(addrs.http, Some(addr("0.0.0.0:18090"))); + } + + #[test] + fn cluster_http_addr_requires_roster_port_for_enabled_transport() { + // No fallback to the top-level port: a silent default could collide + // with another same-host node, so a missing roster port for an + // enabled transport must refuse to boot. + let node = cluster_node("10.0.0.5", None); + let result = resolve_cluster_client_addrs( + &node, + addr("127.0.0.1:8090"), + None, + None, + Some(addr("127.0.0.1:3000")), + ); + assert!(matches!( + result, + Err(ServerError::ClusterPortMissing { + transport: "http", + replica_id: 0, + }) + )); + } + + #[test] + fn cluster_http_addr_is_none_when_http_disabled() { + // http.enabled = false collapses default_http_addr to None; no roster + // port can revive a listener the operator turned off. + let node = cluster_node("127.0.0.1", Some(18090)); + let addrs = resolve_cluster_client_addrs(&node, addr("127.0.0.1:8090"), None, None, None) + .expect("cluster address resolution must succeed"); + assert_eq!(addrs.http, None); + } + + #[test] + fn cluster_tcp_addr_takes_port_from_roster() { + // Same rule as the other transports: the roster owns the port so + // same-host nodes sharing one [tcp].address still bind distinct + // sockets. + let node = cluster_node("127.0.0.1", None); + let addrs = resolve_cluster_client_addrs(&node, addr("127.0.0.1:8090"), None, None, None) + .expect("cluster address resolution must succeed"); + assert_eq!(addrs.client, addr("127.0.0.1:18070")); + } + + #[test] + fn cluster_tcp_addr_merges_config_ip_with_roster_port() { + // The roster ip is advertised, not bound. Binding it directly would + // strand every co-located dialer (sidecars, health probes, on-host + // consumers) that reaches this node over loopback. + let node = cluster_node("10.0.0.5", None); + let addrs = resolve_cluster_client_addrs(&node, addr("0.0.0.0:8090"), None, None, None) + .expect("cluster address resolution must succeed"); + assert_eq!(addrs.client, addr("0.0.0.0:18070")); + } + + #[test] + fn cluster_tcp_addr_requires_roster_port() { + // tcp is always enabled in cluster mode, so a roster entry without a + // tcp port refuses to boot rather than falling back to [tcp].address. + let node = cluster_node_with_ports("10.0.0.5", None, None); + let result = resolve_cluster_client_addrs(&node, addr("127.0.0.1:8090"), None, None, None); + assert!(matches!( + result, + Err(ServerError::ClusterPortMissing { + transport: "tcp", + replica_id: 0, + }) + )); + } + + #[test] + fn cluster_tcp_addr_keeps_loopback_bind_and_warns_on_roster_mismatch() { + // A loopback [tcp].address under a routable roster ip is honoured + // as configured; remote peers cannot reach it, so the mismatch is + // warned about instead of silently rebinding. + let node = cluster_node("10.0.0.5", None); + let addrs = resolve_cluster_client_addrs(&node, addr("127.0.0.1:8090"), None, None, None) + .expect("cluster address resolution must succeed"); + assert_eq!(addrs.client, addr("127.0.0.1:18070")); + assert!(roster_ip_unreachable_from_bind_addr(&node.ip, addrs.client)); + } + + #[test] + fn roster_mismatch_warning_is_silent_for_wildcard_and_hostname_rosters() { + // A wildcard bind covers the roster interface, and a DNS roster entry + // can resolve to the bound one; neither is a misconfiguration. + assert!(!roster_ip_unreachable_from_bind_addr( + "10.0.0.5", + addr("0.0.0.0:18070") + )); + assert!(!roster_ip_unreachable_from_bind_addr( + "node-1.example.com", + addr("127.0.0.1:18070") + )); + assert!(!roster_ip_unreachable_from_bind_addr( + "10.0.0.5", + addr("10.0.0.5:18070") + )); + } + + #[test] + fn derived_address_warns_only_when_it_misses_a_listener() { + // Derived from a loopback tcp.address while another transport serves + // an external interface: metadata would publish an address no client + // reaches. Every non-TCP listener carries the same exposure, since one + // host is published for all four. + for listener in ["10.0.0.5:3000", "10.0.0.5:8080", "10.0.0.5:8092"] { + assert!( + derived_address_misses_listener(None, "127.0.0.1", addr(listener)), + "{listener} is not reachable at 127.0.0.1" + ); + } + // Same interface, and a wildcard bind that covers any of them. + assert!(!derived_address_misses_listener( + None, + "10.0.0.5", + addr("10.0.0.5:3000") + )); + assert!(!derived_address_misses_listener( + None, + "127.0.0.1", + addr("0.0.0.0:3000") + )); + // A declared address is deliberate and unrelated to local interfaces. + assert!(!derived_address_misses_listener( + Some("broker-1.example.com"), + "broker-1.example.com", + addr("10.0.0.5:3000") + )); + } + + #[test] + fn wildcard_listener_warns_only_under_a_derived_loopback_address() { + // Metadata says 127.0.0.1 while this listener takes connections from + // anywhere: whoever arrives from another host is told to dial itself. + assert!(wildcard_listener_under_loopback_address( + None, + "127.0.0.1", + addr("0.0.0.0:3000") + )); + assert!(wildcard_listener_under_loopback_address( + None, + "127.0.0.1", + addr("[::]:3000") + )); + // A published address that is reachable from elsewhere is what the + // wildcard listener wants, so there is nothing to say. + assert!(!wildcard_listener_under_loopback_address( + None, + "10.0.0.5", + addr("0.0.0.0:3000") + )); + // A concrete bind is the other warning's business, not this one's. + assert!(!wildcard_listener_under_loopback_address( + None, + "127.0.0.1", + addr("10.0.0.5:3000") + )); + // A declared address is deliberate; a loopback one is a local setup. + assert!(!wildcard_listener_under_loopback_address( + Some("127.0.0.1"), + "127.0.0.1", + addr("0.0.0.0:3000") + )); + } + + #[test] + fn derived_bind_ip_follows_the_first_enabled_listener() { + let mut config: ServerConfig = + toml::from_str(include_str!("../../config.toml")).expect("shipped config deserializes"); + let topology = |ws: Option<&str>, http: Option<&str>| TcpTopology { + cluster_id: 0, + self_replica_id: 0, + replica_count: 1, + client_listen_addr: addr("127.0.0.1:8090"), + replica_listen_addr: None, + ws_listen_addr: ws.map(addr), + quic_listen_addr: None, + http_listen_addr: http.map(addr), + tcp_tls_listen_addr: None, + peers: Vec::new(), + }; + let expected = |ip: &str| ip.parse::().unwrap(); + + assert_eq!( + derived_bind_ip( + &topology(Some("10.0.0.5:8092"), Some("10.0.0.6:3000")), + &config + ), + expected("127.0.0.1") + ); + + config.tcp.enabled = false; + assert_eq!( + derived_bind_ip( + &topology(Some("10.0.0.5:8092"), Some("10.0.0.6:3000")), + &config + ), + expected("10.0.0.5"), + "websocket is next in line once tcp is off" + ); + assert_eq!( + derived_bind_ip(&topology(None, Some("10.0.0.6:3000")), &config), + expected("10.0.0.6"), + "with websocket and quic off too, http answers" + ); + assert_eq!( + derived_bind_ip(&topology(None, None), &config), + expected("127.0.0.1"), + "with every client listener off the value reaches no client anyway" + ); + } + + #[test] + fn roster_mismatch_warning_is_silent_for_v4_mapped_binds() { + // `[::ffff:0.0.0.0]` is the v4 wildcard and `::ffff:10.0.0.5` is + // `10.0.0.5`, so neither reaches the dialer any differently than the + // plain spelling the case above covers. + assert!(!roster_ip_unreachable_from_bind_addr( + "10.0.0.5", + addr("[::ffff:0.0.0.0]:18070") + )); + assert!(!roster_ip_unreachable_from_bind_addr( + "10.0.0.5", + addr("[::ffff:10.0.0.5]:18070") + )); + // A genuine mismatch still warns through the mapped spelling. + assert!(roster_ip_unreachable_from_bind_addr( + "10.0.0.5", + addr("[::ffff:127.0.0.1]:18070") + )); + } +} diff --git a/core/server/src/bootstrap.rs b/core/server/src/bootstrap.rs deleted file mode 100644 index 5903aebafc..0000000000 --- a/core/server/src/bootstrap.rs +++ /dev/null @@ -1,5146 +0,0 @@ -// Licensed to the Apache Software Foundation (ASF) under one -// or more contributor license agreements. See the NOTICE file -// distributed with this work for additional information -// regarding copyright ownership. The ASF licenses this file -// to you under the Apache License, Version 2.0 (the -// "License"); you may not use this file except in compliance -// with the License. You may obtain a copy of the License at -// -// http://www.apache.org/licenses/LICENSE-2.0 -// -// Unless required by applicable law or agreed to in writing, -// software distributed under the License is distributed on an -// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY -// KIND, either express or implied. See the License for the -// specific language governing permissions and limitations -// under the License. - -use crate::cluster_meta::{ClusterRoster, resolved_roster_nodes, self_advertised_address}; -use crate::config_writer::write_current_config; -use crate::dispatch::partition::make_partition_read_handler; -use crate::dispatch::session_ops::warm_dummy_password_hash; -use crate::dispatch::submit::make_metadata_submit_handler; -use crate::dispatch::{ - make_client_request_handler, make_deferred_client_request_handler, - make_deferred_replica_message_handler, make_list_clients_handler, -}; -use crate::http; -use crate::partition_helpers::{ - build_partition_fresh, configure_consumer_offsets, ensure_initial_segment, - open_partition_superblock, -}; -use crate::segment_recovery::{RecoveredSegment, load_persisted_segments}; -use crate::server_error::{ - PartitionRecoveryRefusal, ServerError, ShardJoinFailure, ShardJoinFailureKind, -}; -use crate::session_manager::SessionManager; -use crate::shard_allocator::{ShardAllocator, ShardInfo}; -use crate::shell::{ - ServerMetadata, ServerMetadataBundle, ServerMuxStateMachine, ServerShard, ShellBus, - ShellHandlers, ShellShardHandle, consensus_timers, repair_retry_ticks, -}; -use compio::runtime::ResumeUnwind; -use configs::server::{ServerConfig, ServerSystemConfig}; -use configs::sharding::{ - INBOX_CAPACITY_MAX, SHUTDOWN_DRAIN_TIMEOUT_MAX, SHUTDOWN_POLL_INTERVAL_MAX, -}; -use consensus::{ - ClientTable, JoinMode, LocalPipeline, MetadataHandle, PartitionsHandle, PipelineEntry, - Sequencer, VsrConsensus, VsrRestore, -}; -// `try_send` / `try_recv` resolve through these traits on `MAsyncTx` / -// `MAsyncRx`; the metadata-handoff loops below depend on the -// non-blocking variants for cancel-safe shutdown polling. -use consensus::VsrState; -use crossfire::{AsyncRxTrait, AsyncTxTrait}; -use iggy_binary_protocol::{Operation, PrepareHeader}; -use iggy_common::defaults::{ - DEFAULT_ROOT_PASSWORD, DEFAULT_ROOT_USERNAME, MAX_PASSWORD_LENGTH, MAX_USERNAME_LENGTH, - MIN_PASSWORD_LENGTH, MIN_USERNAME_LENGTH, -}; -use iggy_common::{ - Aes256GcmEncryptor, EncryptorKind, IggyByteSize, IggyError, PartitionStats, TopicRuntimeOptions, -}; -use journal::prepare_journal::PrepareJournal; -use journal::superblock::{PingPongSuperblock, SuperblockStore}; -use journal::{Journal, JournalHandle}; -use message_bus::client_listener::{self, RequestHandler}; -use message_bus::installer; -use message_bus::installer::conn_info::{ClientConnMeta, ClientTransportKind}; -use message_bus::replica::auth::{self, ReplicaAuth}; -use message_bus::replica::handshake::{ReplicaHandshakeCtx, ReplicaTlsCtx}; -use message_bus::replica::io as replica_io; -use message_bus::replica::listener::{self as replica_listener}; -use message_bus::transports::quic::server_config_with_cert; -use message_bus::transports::tls::{ - AcceptAnyServerCert, REPLICA_ALPN, TlsServerCredentials, install_default_crypto_provider, - load_ca_pem, load_pem, self_signed_for_loopback, -}; -use message_bus::{ - AcceptedClientFn, AcceptedQuicClientFn, AcceptedReplicaFn, AcceptedTlsClientFn, - AcceptedWsClientFn, AcceptedWssClientFn, ConnectionInstaller, DialedReplicaFn, IggyMessageBus, - MAX_INFLIGHT_REPLICA_HANDSHAKES, MessageBus, ReplicaOwnerTable, connector, -}; -use metadata::ReplicaIdentity; -use metadata::impls::metadata::{IggySnapshot, StreamsFrontend}; -use metadata::impls::recovery::recover; -use metadata::stm::snapshot::Snapshot; -use metadata::stm::stream::Partition; -use partitions::{ - FatalCommit, IggyIndexWriter, IggyPartition, IggyPartitions, MessagesWriter, PartitionsConfig, -}; -use rustls::pki_types::ServerName; -use server_common::Message; -use server_common::bootstrap::create_directories; -use server_common::crypto; -use server_common::executor::create_shard_executor; -use server_common::fs_utils::remove_dir_all; -use server_common::log::{Logging, LoggingSettings, TelemetrySettings}; -use server_common::sharding::{IggyNamespace, PartitionLocation, ShardId}; -use shard::builder::IggyShardBuilder; -use shard::metrics::{ShardMetrics, frame_drop_reason, frame_drop_variant}; -use shard::shards_table::{PapayaShardsTable, ShardsTable, calculate_shard_assignment}; -use shard::{ - CoordinatorConfig, LifecycleFrame, PartitionConsensusConfig, Receiver as ShardReceiver, - ShardFrame, ShardIdentity, TaggedSender, channel, shard_mesh_channels, -}; -use std::cell::RefCell; -use std::collections::HashMap; -use std::env; -use std::net::{IpAddr, SocketAddr}; -use std::panic; -use std::path::{Path, PathBuf}; -use std::rc::Rc; -use std::sync::Arc; -use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; -use std::thread; -use std::time::{Duration, Instant}; -use tracing::{error, info, warn}; - -const SHARD_REPLICA_ID: u8 = 0; - -pub const IGGY_ROOT_USERNAME_ENV: &str = "IGGY_ROOT_USERNAME"; -pub const IGGY_ROOT_PASSWORD_ENV: &str = "IGGY_ROOT_PASSWORD"; - -/// Build the deferred dispatch handlers for `shard_handle` against `bus`. -/// -/// They share one fresh [`SessionManager`]. The caller must set the weak -/// self-reference in `shard_handle` once the shard is built, so the -/// handlers can upgrade it per frame. -pub fn wire_shell_handlers( - bus: &B, - shard_handle: &ShellShardHandle, - system_config: Arc, - max_tokens_per_user: u32, -) -> ShellHandlers -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let sessions = Rc::new(RefCell::new(SessionManager::new())); - ShellHandlers { - on_replica_message: make_deferred_replica_message_handler(shard_handle), - on_client_request: make_deferred_client_request_handler( - bus, - shard_handle, - &sessions, - system_config, - max_tokens_per_user, - ), - on_metadata_submit: make_metadata_submit_handler(shard_handle), - on_list_clients: make_list_clients_handler(&sessions), - on_partition_read: make_partition_read_handler(shard_handle), - sessions, - } -} - -/// Result of a multi-shard bootstrap. -/// -/// Carries the cross-thread shutdown flag and one OS-thread `JoinHandle` -/// per shard. The caller flips the flag via [`Self::install_ctrlc_handler`] -/// and then drains every shard via [`Self::join_all`], bounded by -/// `join_timeout` (`system.sharding.shutdown_join_timeout`). -pub struct ShardHandles { - shutdown_flag: Arc, - shard_threads: Vec<(u16, thread::JoinHandle>)>, - join_timeout: Duration, -} - -impl ShardHandles { - /// Install a SIGINT/Ctrl-C handler that flips the shutdown flag on - /// the first signal. A second signal is logged but otherwise - /// ignored so an in-flight WAL fsync or replica drain runs to - /// completion. - /// - /// # Errors - /// - /// Returns the underlying `ctrlc::Error` if the handler cannot be - /// installed (typically because another handler already owns the - /// signal). - pub fn install_ctrlc_handler(&self) -> Result<(), ctrlc::Error> { - let flag = Arc::clone(&self.shutdown_flag); - ctrlc::set_handler(move || { - if flag.swap(true, Ordering::Relaxed) { - // Second Ctrl-C: leave the shutdown machinery to drain. - // Refusing to abort here keeps the WAL fsync / replica - // drain from being interrupted mid-frame. - warn!("second Ctrl-C ignored; server is already shutting down"); - } else { - info!("Ctrl-C received; signalling server shutdown"); - } - }) - } - - /// Drain every shard thread. This is the main thread's park for the - /// server's whole lifetime, so shards are awaited WITHOUT any time - /// bound while the server runs; the `shutdown_join_timeout` clock - /// only starts once the cross-thread shutdown flag flips (Ctrl-C or - /// a shard failure). Each shard's outcome is logged (`info` on clean - /// exit, `error` on Err, panic, or wedge). If any shard failed, - /// returns every failure together as - /// [`ServerError::ShardJoinFailures`] so the operator sees the - /// full set rather than just the first. - /// - /// A shard whose thread is still running when the post-shutdown - /// deadline passes is abandoned (its `JoinHandle` dropped, the OS - /// thread left to die with the process) and reported as - /// [`ShardJoinFailureKind::Wedged`]: a wedged pump or listener must - /// not block process exit forever. - /// - /// # Errors - /// - /// Returns [`ServerError::ShardJoinFailures`] if any shard - /// returned a `Result::Err`, panicked, or wedged past the deadline. - /// The variant carries every per-shard failure in shard-id order so - /// the caller does not need to read the trace log to discover - /// late-failing shards. - pub fn join_all(self) -> Result<(), ServerError> { - let mut failures: Vec = Vec::new(); - // Armed on the first poll that observes the shutdown flag, shared - // across all shards: one budget covers the whole drain, not one - // budget per shard. - let mut deadline: Option = None; - // Shards run thread-per-core with compio's blocking fallback pool - // disabled, so an io_uring opcode the kernel lacks aborts every shard - // with the same panic. Surface the actionable diagnostic once. - let mut io_uring_diagnostic_shown = false; - for (shard_id, handle) in self.shard_threads { - let Some(joined) = join_until_shutdown_deadline( - handle, - &self.shutdown_flag, - self.join_timeout, - &mut deadline, - ) else { - error!( - shard_id, - waited = ?self.join_timeout, - "shard thread still running at the shutdown join deadline; abandoning it" - ); - failures.push(ShardJoinFailure { - shard_id, - kind: ShardJoinFailureKind::Wedged { - waited: self.join_timeout, - }, - }); - continue; - }; - match joined { - Ok(Ok(())) => { - info!(shard_id, "shard thread exited cleanly"); - } - Ok(Err(error)) => { - error!(shard_id, error = %error, "shard thread returned error"); - failures.push(ShardJoinFailure { - shard_id, - kind: ShardJoinFailureKind::Error(Box::new(error)), - }); - } - Err(panic_payload) => { - let message = panic_payload_to_string(&*panic_payload); - error!(shard_id, message = %message, "shard thread panicked"); - if !io_uring_diagnostic_shown - && message - .contains(server_common::diagnostics::ASYNCIFY_POOL_DISABLED_PANIC_MSG) - { - server_common::diagnostics::print_incomplete_io_uring_ops_info(); - io_uring_diagnostic_shown = true; - } - failures.push(ShardJoinFailure { - shard_id, - kind: ShardJoinFailureKind::Panic { message }, - }); - } - } - } - if failures.is_empty() { - Ok(()) - } else { - Err(ServerError::ShardJoinFailures { failures }) - } - } -} - -/// Poll cadence for the bounded shard joins. Coarse enough to cost -/// nothing during a normal drain, fine enough that exit latency past -/// the last shard's return stays imperceptible. -const JOIN_POLL_INTERVAL: Duration = Duration::from_millis(25); - -/// Join `handle`, waiting indefinitely while the server runs. The -/// `join_timeout` clock starts only when `shutdown_flag` is observed set -/// (arming the caller-shared `deadline` once, so all shards drain under -/// ONE budget); a running server parked here for hours must never be -/// mistaken for a wedged shard. `None` means the thread was still -/// running at the post-shutdown deadline and the handle was dropped -/// (the OS thread keeps running detached; process exit reaps it). -/// `JoinHandle` has no timed join, so this polls `is_finished` at -/// [`JOIN_POLL_INTERVAL`]; the closing `join()` on a finished thread -/// returns immediately. -fn join_until_shutdown_deadline( - handle: thread::JoinHandle>, - shutdown_flag: &AtomicBool, - join_timeout: Duration, - deadline: &mut Option, -) -> Option>> { - while !handle.is_finished() { - if deadline.is_none() && shutdown_flag.load(Ordering::Relaxed) { - *deadline = Some(Instant::now() + join_timeout); - } - if let Some(deadline) = deadline - && Instant::now() >= *deadline - { - return None; - } - thread::sleep(JOIN_POLL_INTERVAL); - } - Some(handle.join()) -} - -/// Best-effort extraction of the panic message from a -/// `Box` returned by `JoinHandle::join`. Tries the two -/// payload shapes the standard library guarantees (`&'static str` and -/// `String`) and falls back to a placeholder so the panic still surfaces -/// in the error chain. -fn panic_payload_to_string(payload: &(dyn std::any::Any + Send)) -> String { - if let Some(s) = payload.downcast_ref::<&'static str>() { - return (*s).to_string(); - } - if let Some(s) = payload.downcast_ref::() { - return s.clone(); - } - "".to_string() -} - -/// Joins survivor shard threads after a partial-spawn failure, bounded -/// by the same `shutdown_join_timeout` budget as the normal exit path. -/// -/// Polls every survivor's `is_finished` in one loop instead of spawning -/// per-survivor joiner threads: the likely OS state on this path is -/// `pthread_create` EAGAIN (the parent spawn just failed with it), so -/// nothing here may create threads, and polling drains all survivors in -/// parallel anyway. A survivor still running at the deadline is -/// abandoned with an error log so the failed bootstrap can surface its -/// spawn error instead of hanging on a wedged shard. -fn join_partial_shard_survivors( - shard_threads: Vec<(u16, thread::JoinHandle>)>, - join_timeout: Duration, -) { - let deadline = Instant::now() + join_timeout; - let mut remaining = shard_threads; - loop { - let mut still_running = Vec::with_capacity(remaining.len()); - for (shard_id, survivor) in remaining { - if survivor.is_finished() { - let _ = survivor.join(); - info!(shard_id, "survivor shard thread drained"); - } else { - still_running.push((shard_id, survivor)); - } - } - remaining = still_running; - if remaining.is_empty() || Instant::now() >= deadline { - break; - } - thread::sleep(JOIN_POLL_INTERVAL); - } - for (shard_id, _survivor) in remaining { - error!( - shard_id, - waited = ?join_timeout, - "survivor shard thread still running at the shutdown join deadline; abandoning it" - ); - } -} - -/// Flips the cross-thread shutdown flag on `Drop` unless disarmed. -/// -/// A shard thread that exits via an error `?` or a panic unwind would -/// otherwise leave sibling shards parked forever on `bus.token().wait()`: -/// their watchdogs never observe the flag and the bus has no -/// `Drop`-triggered shutdown. Arming this for the whole thread body makes -/// every non-clean exit drive sibling-shard teardown. Disarmed only on a -/// clean `Ok(())`. -struct ShutdownOnDrop { - flag: Arc, - armed: bool, -} - -impl ShutdownOnDrop { - const fn new(flag: Arc) -> Self { - Self { flag, armed: true } - } - - const fn disarm(&mut self) { - self.armed = false; - } -} - -impl Drop for ShutdownOnDrop { - fn drop(&mut self) { - if self.armed { - self.flag.store(true, Ordering::Relaxed); - } - } -} - -/// Shard-local end of the metadata bundle handoff. -/// -/// Shard 0 owns the WAL writer and runs `recover()` to build the only -/// `WriteHandle`-bearing [`ServerMuxStateMachine`]. It then mints a -/// [`ServerMetadataBundle`] (a tuple of `Send + Sync` -/// `ReadHandleFactory`s) and pushes one clone per peer onto `bundle_tx`. -/// Every other shard receives the bundle and rebuilds a reader-mode -/// `MuxStateMachine` on its own runtime - no WAL access, no replay, no -/// `RecoverySync` two-phase fence. The old phase-2 WAL fence is gone -/// because peers no longer scan the WAL. They do still scan live shared -/// metadata to load their on-disk partitions, so a separate listener -/// fence is still required - see [`BootstrapBarrier`]. -/// -/// The channel is bounded to the peer count so shard 0's `send` never -/// blocks beyond a peer drain. A peer that dies before recv drops its -/// `bundle_rx`, so shard 0's `send` eventually sees a disconnected -/// channel; the cross-thread shutdown flag drives every waiter out of -/// its `recv` loop if shard 0 panics before broadcasting. -enum MetadataHandoff { - Owner { - bundle_tx: crossfire::MAsyncTx>, - }, - Waiter { - bundle_rx: crossfire::MAsyncRx>, - }, -} - -/// Reverse handshake to [`MetadataHandoff`]: gates shard 0's client -/// listeners until every peer has loaded its on-disk partitions. -/// -/// Peers build their owned-partition set from live shared metadata and -/// load each segment from disk in `build_shard_for_thread`. If shard 0 -/// opened listeners the instant `broadcast_metadata_bundle` returned -/// (peers have only *received* the bundle, not *loaded* partitions), a -/// client could create a partition before a peer's load scan finished. -/// That freshly committed partition would surface in the peer's scan -/// with no segment dir on disk yet, and `load_partition`'s `walk_dir` -/// would fail with `CannotReadPartitions`, aborting the whole node. A -/// partition created after boot must take the runtime reconciler path -/// (which creates its dir), never the bootstrap load path. -/// -/// Shard 0 (`Owner`) drains one signal per peer before binding -/// listeners; each peer (`Waiter`) sends one once its load completes. -/// The cross-thread shutdown flag drives both sides out of their poll -/// loop if any shard dies mid-boot. -enum BootstrapBarrier { - Owner { - ready_rx: crossfire::MAsyncRx>, - }, - Waiter { - ready_tx: crossfire::MAsyncTx>, - }, -} - -struct TcpTopology { - /// Domain-separation cluster id derived from `cluster.name`; threaded to - /// every consensus instance and the replica handshake so frames agree. - cluster_id: u128, - self_replica_id: u8, - replica_count: u8, - client_listen_addr: SocketAddr, - replica_listen_addr: Option, - ws_listen_addr: Option, - quic_listen_addr: Option, - http_listen_addr: Option, - tcp_tls_listen_addr: Option, - peers: Vec<(u8, SocketAddr)>, -} - -struct LocalClientAcceptFns { - tcp: AcceptedClientFn, - ws: AcceptedWsClientFn, - quic: AcceptedQuicClientFn, - tcp_tls: AcceptedTlsClientFn, - wss: AcceptedWssClientFn, -} - -#[derive(Default)] -struct BoundClientListeners { - tcp: Option, - tcp_tls: Option, - ws: Option, - quic: Option, -} - -/// Load the server configuration from the active config provider. -/// -/// # Errors -/// -/// Returns an error if the configuration cannot be read or parsed. -pub async fn load_config() -> Result { - ServerConfig::load().await.map_err(ServerError::Config) -} - -/// Prepare the on-disk layout the server boots from and complete late -/// logging init. -/// -/// `fresh` wipes the system path first: `late_init` opens a rolling -/// appender under `{system_path}/logs` and `create_directories` -/// materialises exactly what the wipe is meant to remove, so both have to -/// run after it. -/// -/// # Errors -/// -/// Returns an error if the wipe, directory preparation, or logging setup -/// fails. -pub async fn prepare_runtime_dirs( - config: &ServerConfig, - logging: &mut Logging, - fresh: bool, -) -> Result<(), ServerError> { - if fresh { - wipe_system_path(config).await?; - } - create_directories(&config.system).await.map_err(|source| { - error!( - system_path = %config.system.get_system_path(), - error = %source, - "failed to prepare server directories" - ); - source - })?; - logging - .late_init( - config.system.get_system_path(), - &LoggingSettings::from(&config.system.logging), - &TelemetrySettings::from(&config.telemetry), - ) - .map_err(ServerError::Logging)?; - - Ok(()) -} - -/// Delete the configured system path so the server boots on empty state. -async fn wipe_system_path(config: &ServerConfig) -> Result<(), ServerError> { - let path = config.system.get_system_path(); - // `system.path` is relative by default and IGGY_SYSTEM_PATH-overridable, - // so report what is actually about to be deleted, not what was configured. - let resolved = std::path::absolute(&path).unwrap_or_else(|_| PathBuf::from(&path)); - - if config.cluster.enabled { - warn!( - path = %resolved.display(), - "--fresh wipes only this replica, which then refills from the cluster by \ - state transfer; wiping a quorum at once destroys committed data, and a \ - service unit file carrying --fresh re-transfers everything on every restart" - ); - } - - if !Path::new(&path).exists() { - info!(path = %resolved.display(), "--fresh: system path does not exist, nothing to remove"); - return Ok(()); - } - - warn!(path = %resolved.display(), "--fresh: removing the system path, ALL local data will be deleted"); - // A half-removed directory is worse than no removal at all: the surviving - // superblock and snapshot no longer pair up, and boot would report the - // leftovers as a durability violation rather than as a failed wipe. - remove_dir_all(&path) - .await - .map_err(|source| ServerError::FreshWipeFailed { - path: resolved, - source, - }) -} - -/// Resolve the operator's `cpu_allocation` into concrete shard -/// assignments plus the checked `u16` shard count. -/// -/// Shard ids index `ReplicaOwnerTable` slots as `u16`. `OWNER_NONE` -/// (`u16::MAX`) is reserved as the empty-slot sentinel, so a server -/// configured with `u16::MAX` shards would mint a shard id that -/// collides with the sentinel and an owner-table lookup could never -/// tell that shard apart from an unowned slot. Reject at boot so the -/// invariant is held by the type system, not by hoping the operator -/// never configures 65535 cores worth of shards. -fn resolve_shard_assignments( - sharding: &configs::sharding::ShardingConfig, -) -> Result<(Vec, u16), ServerError> { - let allocator = ShardAllocator::new(&sharding.cpu_allocation, sharding.pin_cores) - .map_err(ServerError::ShardAllocator)?; - let assignments = allocator - .to_shard_assignments() - .map_err(ServerError::ShardAllocator)?; - if assignments.is_empty() { - return Err(ServerError::ShardsCountZero); - } - match u16::try_from(assignments.len()) { - Ok(count) if count < message_bus::OWNER_NONE => Ok((assignments, count)), - _ => Err(ServerError::ShardsCountOverflow { - count: assignments.len(), - }), - } -} - -/// Re-validate the runtime sharding knobs that the per-shard runtime -/// consumes directly. Mirrors `ShardingConfig::validate` so a caller -/// that built the config without running it (e.g. tests, embedded -/// usage) cannot OOM at boot or wedge process exit with an out-of-range -/// value. -fn validate_sharding_runtime_knobs( - sharding: &configs::sharding::ShardingConfig, -) -> Result<(), ServerError> { - let inbox_capacity = sharding.inbox_capacity; - if inbox_capacity == 0 || inbox_capacity > INBOX_CAPACITY_MAX { - return Err(ServerError::InvalidInboxCapacity { - value: inbox_capacity, - max: INBOX_CAPACITY_MAX, - }); - } - let reply_inbox_capacity = sharding.reply_inbox_capacity; - if reply_inbox_capacity == 0 || reply_inbox_capacity > INBOX_CAPACITY_MAX { - return Err(ServerError::InvalidReplyInboxCapacity { - value: reply_inbox_capacity, - max: INBOX_CAPACITY_MAX, - }); - } - let drain_timeout = sharding.shutdown_drain_timeout.get_duration(); - if drain_timeout.is_zero() || drain_timeout > SHUTDOWN_DRAIN_TIMEOUT_MAX { - return Err(ServerError::InvalidShutdownDrainTimeout { - value: drain_timeout, - max: SHUTDOWN_DRAIN_TIMEOUT_MAX, - }); - } - let poll_interval = sharding.shutdown_poll_interval.get_duration(); - if poll_interval.is_zero() || poll_interval > SHUTDOWN_POLL_INTERVAL_MAX { - return Err(ServerError::InvalidShutdownPollInterval { - value: poll_interval, - max: SHUTDOWN_POLL_INTERVAL_MAX, - }); - } - // Ordering: a poll cadence coarser than the drain budget makes the - // cross-thread shutdown flag effectively unobservable during teardown. - if poll_interval > drain_timeout { - return Err(ServerError::ShutdownPollExceedsDrain { - poll: poll_interval, - drain: drain_timeout, - }); - } - Ok(()) -} - -/// Spawn the multi-shard `server` runtime. -/// -/// Resolves shard count + CPU affinities from -/// `system.sharding.cpu_allocation`, builds canonical-ordered -/// `(senders, inboxes)` channels, and spawns one OS thread per shard. -/// -/// Each thread pins itself (`nix::sched::sched_setaffinity` on Linux via -/// `ShardInfo::bind_cpu`), binds memory to its NUMA node when -/// configured, builds a fresh `compio::runtime::Runtime` (one -/// `io_uring` instance per shard), and runs `shard_main` inside it. -/// -/// Returns [`ShardHandles`] containing the cross-thread shutdown flag -/// and the per-shard `JoinHandle`s. The caller (`main.rs`) installs a -/// `ctrlc` handler that flips the flag, then `.join()`s every handle. -/// -/// # Errors -/// -/// Returns an error if shard allocation fails, the inbox capacity is -/// invalid, or any OS thread fails to spawn. Per-shard recovery / -/// listener / consensus failures surface through the per-thread `Result` -/// the caller observes on `.join()`. -/// -/// # Panics -/// -/// Panics if [`shard_mesh_channels`] returns an inbox slot already -/// consumed - a bootstrap programming error that would only fire if this -/// function were called twice with the same inboxes. -#[allow(clippy::too_many_lines)] -pub fn bootstrap( - config: ServerConfig, - current_replica_id: Option, -) -> Result { - validate_root_credentials_env(&config)?; - warm_dummy_password_hash(); - // The sync GetStats read path has no access to server config, so capture - // the data directory here for its disk-usage reporting. - crate::responses::init_stats_data_path(config.system.get_system_path().into()); - let (assignments, total_shards) = resolve_shard_assignments(&config.system.sharding)?; - let shards_count = assignments.len(); - - // Re-check the full valid range, not just the zero floor: a caller - // that built the config without running `ShardingConfig::validate` - // would otherwise OOM at boot allocating an oversized inbox channel, - // busy-loop every shutdown watchdog on a zero poll cadence, or wedge - // process exit on an unbounded drain budget. - let inbox_capacity = config.system.sharding.inbox_capacity; - let reply_inbox_capacity = config.system.sharding.reply_inbox_capacity; - validate_sharding_runtime_knobs(&config.system.sharding)?; - - let (senders, mut inboxes, mut reply_inboxes) = - shard_mesh_channels(total_shards, inbox_capacity, reply_inbox_capacity); - let shutdown_flag = Arc::new(AtomicBool::new(false)); - let config = Arc::new(config); - // One owner table per server process, Arc-cloned into every shard's bus so - // any shard's bus reads the same atomic slots that the owning - // shard's installer / disconnect path writes. - let owner_table = Arc::new(ReplicaOwnerTable::new()); - - // Single-shot bundle handoff (see `MetadataHandoff`): shard 0 sends - // one cloned `ServerMetadataBundle` per peer; each peer drains - // exactly one. Bounded to the peer count so shard 0's broadcast - // never blocks past a peer drain. A single-shard deployment (zero - // peers) still needs a non-zero capacity, so clamp up explicitly - // rather than relying on crossfire's internal cap=0 -> 1 promotion. - // If a peer dies before recv, shard 0's `send` eventually sees a - // disconnected channel; the cross-thread shutdown flag drives every - // waiter out of its recv loop if shard 0 panics before broadcasting. - let metadata_peers = shards_count.saturating_sub(1).max(1); - let (metadata_bundle_tx, metadata_bundle_rx) = - crossfire::mpmc::bounded_async::(metadata_peers); - - // Reverse barrier (see `BootstrapBarrier`): every peer sends one - // signal once it finishes loading its on-disk partitions; shard 0 - // drains them all before binding listeners. Bounded to the peer - // count so a sender never blocks (each peer sends exactly once). - let (ready_tx, ready_rx) = crossfire::mpmc::bounded_async::(metadata_peers); - - let mut shard_threads: Vec<(u16, thread::JoinHandle>)> = - Vec::with_capacity(shards_count); - // Shared metadata-group view: written by shard 0's publisher task, read by - // every shard's cluster-metadata roster so leader marking works off-shard. - let metadata_view = Arc::new(AtomicU64::new(crate::cluster_meta::METADATA_VIEW_UNKNOWN)); - // Every shard's metric handles, minted before the threads spawn: each - // shard bumps its own entry, and shard 0's HTTP scrape endpoint registers - // the whole set (counters are Arc-backed, so cross-thread reads see the - // owning shard's bumps). - let shard_metrics_all: Vec = (0..shards_count) - .map(|_| ShardMetrics::for_shard()) - .collect(); - for (idx, assignment) in assignments.into_iter().enumerate() { - #[allow(clippy::cast_possible_truncation)] - let shard_id = idx as u16; - let inbox = inboxes[idx] - .take() - .expect("shard_mesh_channels populates every inbox slot exactly once"); - let reply_inbox = reply_inboxes[idx] - .take() - .expect("shard_mesh_channels populates every reply-inbox slot exactly once"); - let senders_for_shard = senders.clone(); - let config_for_shard = Arc::clone(&config); - let shutdown_flag_for_shard = Arc::clone(&shutdown_flag); - let owner_table_for_shard = Arc::clone(&owner_table); - let metadata_handoff_for_shard = if shard_id == 0 { - MetadataHandoff::Owner { - bundle_tx: metadata_bundle_tx.clone(), - } - } else { - MetadataHandoff::Waiter { - bundle_rx: metadata_bundle_rx.clone(), - } - }; - let barrier_for_shard = if shard_id == 0 { - BootstrapBarrier::Owner { - ready_rx: ready_rx.clone(), - } - } else { - BootstrapBarrier::Waiter { - ready_tx: ready_tx.clone(), - } - }; - - let metadata_view_for_shard = Arc::clone(&metadata_view); - let shard_metrics_for_shard = shard_metrics_all.clone(); - let handle = match thread::Builder::new() - .name(format!("shard-{shard_id}")) - .spawn(move || -> Result<(), ServerError> { - run_shard_thread( - shard_id, - total_shards, - current_replica_id, - assignment, - senders_for_shard, - inbox, - reply_inbox, - config_for_shard, - shutdown_flag_for_shard, - metadata_handoff_for_shard, - barrier_for_shard, - owner_table_for_shard, - metadata_view_for_shard, - shard_metrics_for_shard, - ) - }) { - Ok(handle) => handle, - Err(source) => { - // Signal every shard already spawned before propagating, so - // their watchdog loops drive `bus.shutdown(...)` and the - // process can exit instead of hanging on stuck OS threads. - shutdown_flag.store(true, Ordering::Relaxed); - // Drop bootstrap's own channel clones before joining - // survivors. Otherwise a peer waiting on `bundle_rx.recv` - // would never observe the sender side disconnecting and - // would hang until the shutdown watchdog kicks the bus. - drop(metadata_bundle_tx); - drop(metadata_bundle_rx); - drop(ready_tx); - drop(ready_rx); - join_partial_shard_survivors( - shard_threads, - config.system.sharding.shutdown_join_timeout.get_duration(), - ); - return Err(ServerError::ShardSpawnFailed { shard_id, source }); - } - }; - shard_threads.push((shard_id, handle)); - } - - // Drop bootstrap's own channel clones now that every shard owns its - // half. Keeping them on bootstrap's stack would deadlock a peer - // whose `bundle_rx.recv` only completes once every sender - // disconnects. - drop(metadata_bundle_tx); - drop(metadata_bundle_rx); - drop(ready_tx); - drop(ready_rx); - - info!( - shards_count, - "server bootstrap dispatched; awaiting shard runtimes" - ); - - Ok(ShardHandles { - shutdown_flag, - shard_threads, - join_timeout: config.system.sharding.shutdown_join_timeout.get_duration(), - }) -} - -/// Per-shard OS thread entry. Pins CPU + memory, builds the compio -/// runtime, and `block_on`s `shard_main`. -#[allow(clippy::needless_pass_by_value, clippy::too_many_arguments)] -fn run_shard_thread( - shard_id: u16, - total_shards: u16, - replica_id: Option, - assignment: ShardInfo, - senders: Vec, - inbox: ShardReceiver, - reply_inbox: ShardReceiver, - config: Arc, - shutdown_flag: Arc, - metadata_handoff: MetadataHandoff, - barrier: BootstrapBarrier, - owner_table: Arc, - metadata_view: Arc, - shard_metrics_all: Vec, -) -> Result<(), ServerError> { - // Armed for the whole thread body: a post-spawn error `?` or a panic - // unwind here must flip `shutdown_flag` so sibling watchdogs drive - // their bus shutdown instead of parking forever on `bus.token().wait()`. - let mut shutdown_guard = ShutdownOnDrop::new(Arc::clone(&shutdown_flag)); - - assignment - .bind_cpu() - .map_err(|source| ServerError::CpuAffinityFailed { shard_id, source })?; - assignment - .bind_memory() - .map_err(|source| ServerError::MemoryAffinityFailed { shard_id, source })?; - - // `enrich_runtime_create_error` folds the io_uring remediation (raise - // `ulimit -l`, unblock seccomp, kernel-flag floor) into the error, so the - // guidance survives into the shard-join failure report instead of only - // stderr. Multi-shard boxes exhaust RLIMIT_MEMLOCK on per-shard rings - // before the bootstrap runtime does, so this path needs it most. - let runtime = create_shard_executor().map_err(|source| { - let source = server_common::diagnostics::enrich_runtime_create_error(source); - ServerError::ShardRuntimeCreateFailed { shard_id, source } - })?; - - let result = runtime.block_on(async move { - // `shard_main`'s future grows past clippy's `large_futures` cap - // (it ferries the metadata handoff, bus, builders, and inflight - // I/O in one state machine). Heap-pin it so the top-level - // `block_on` future stays small; one allocation per startup buys - // the stack budget back. - Box::pin(shard_main( - shard_id, - total_shards, - replica_id, - senders, - inbox, - reply_inbox, - &config, - shutdown_flag, - metadata_handoff, - barrier, - owner_table, - metadata_view, - shard_metrics_all, - )) - .await - }); - - if result.is_ok() { - shutdown_guard.disarm(); - } - result -} - -/// Per-shard async lifecycle. Builds the bus, recovers metadata, -/// constructs the `IggyShard` for this shard's slice of partitions, -/// wires listeners on shard 0, and runs the message pump until -/// shutdown. -#[allow(clippy::too_many_arguments, clippy::too_many_lines)] -async fn shard_main( - shard_id: u16, - total_shards: u16, - replica_id: Option, - senders: Vec, - inbox: ShardReceiver, - reply_inbox: ShardReceiver, - config: &ServerConfig, - shutdown_flag: Arc, - metadata_handoff: MetadataHandoff, - barrier: BootstrapBarrier, - owner_table: Arc, - metadata_view: Arc, - shard_metrics_all: Vec, -) -> Result<(), ServerError> { - let topology = resolve_tcp_topology(config, replica_id)?; - let bus = Rc::new(IggyMessageBus::with_config_and_owner_table( - shard_id, - config, - owner_table, - )); - // Every shard can own a delegated replica connection, so every - // shard's bus needs the handshake identity (the handshake itself - // runs on the owning shard, not on shard 0). - bus.set_replica_handshake_ctx(ReplicaHandshakeCtx { - cluster_id: topology.cluster_id, - self_id: topology.self_replica_id, - replica_count: topology.replica_count, - auth: load_replica_auth(config).map(Rc::new), - tls: load_replica_tls_ctx(config, &topology)?.map(Rc::new), - }); - - let drain_timeout = config.system.sharding.shutdown_drain_timeout.get_duration(); - let poll_interval = config.system.sharding.shutdown_poll_interval.get_duration(); - - let shutdown_flag_for_handoff = Arc::clone(&shutdown_flag); - let mut shutdown_watchdog = Some(spawn_shutdown_watchdog( - Rc::clone(&bus), - shutdown_flag, - drain_timeout, - poll_interval, - )); - - // Metadata bootstrap is single-writer: shard 0 owns the WAL and the - // only `WriteHandle`-bearing `MuxStateMachine`. Peer shards receive - // a `ReadHandleFactory` bundle on the inter-thread channel and - // rebuild a reader-mode `MuxStateMachine` on their own runtime - no - // WAL access, no replay. Writes still funnel through shard 0's - // metadata VSR; per-commit `publish()` (in `WriteCell::apply`) - // bounds reader staleness to one op. - let data_dir = Path::new(&config.system.path); - let (mux_stm, owner_state) = match metadata_handoff { - MetadataHandoff::Owner { bundle_tx } => { - // Root is created locally at boot (never journaled), so replay - // must start from the same baseline or every WAL-created user - // shifts one slab id and root is lost after the first restart. - let recovered = recover::( - data_dir, - ReplicaIdentity { - cluster: topology.cluster_id, - replica_id: topology.self_replica_id, - replica_count: topology.replica_count, - }, - config.metadata.journal_slots, - config.metadata.clients_table_max, - |mux_stm| { - ensure_default_root_user(mux_stm); - }, - |mux_stm, client, stamp| { - mux_stm - .streams() - .remove_consumer_group_member(client, stamp); - }, - ) - .await - .map_err(ServerError::MetadataRecovery)?; - ensure_default_root_user(&recovered.mux_stm); - // The factory bundle hands every peer a read handle over the - // same `Inner`, so `Arc` (and the parent - // `Arc`) is shared across all shards. Zero the - // snapshot totals here, once, before any peer can observe the - // bundle. Per-shard `load_partition` deltas in - // `build_shard_for_thread` then race only against other - // atomic adds, never against a concurrent `swap(0)` that - // would mistake an in-flight delta for the snapshot total - // and decrement the parent `StreamStats` by it. - let () = recovered.mux_stm.streams().read(|inner| { - for (_, stream) in &inner.items { - for (_, topic) in &stream.topics { - topic.stats.zero_out_all(); - } - } - }); - broadcast_metadata_bundle( - shard_id, - &bundle_tx, - recovered.mux_stm.factory_bundle(), - total_shards.saturating_sub(1), - &shutdown_flag_for_handoff, - poll_interval, - ) - .await?; - ( - recovered.mux_stm, - Some(RecoveredOwnerState { - journal: recovered.journal, - snapshot: recovered.snapshot, - last_applied_op: recovered.last_applied_op, - last_journaled_op: recovered.last_journaled_op, - client_table: recovered.client_table, - superblock: recovered.superblock, - recovered_state: recovered.recovered_state, - snapshot_checkpoint: recovered.snapshot_checkpoint, - }), - ) - } - MetadataHandoff::Waiter { bundle_rx } => { - let bundle = await_metadata_bundle( - shard_id, - &bundle_rx, - &shutdown_flag_for_handoff, - poll_interval, - ) - .await?; - (ServerMuxStateMachine::from_factory_bundle(bundle), None) - } - }; - - // Metadata consensus + journal + snapshot live only on shard 0. - // `IggyShard::tick_metadata` short-circuits when `consensus.is_none()`, - // so peer shards have no caller that reads `journal` or `snapshot`. - let ( - metadata_consensus, - journal_for_metadata, - snapshot_for_metadata, - superblock_for_metadata, - checkpoint_seed, - recovered_client_table, - ) = if let Some(owner) = owner_state { - // `recover()` already opened the superblock, read `recovered_state`, and - // verified the on-disk snapshot against its checkpoint pairing BEFORE decoding - // it. Reuse that superblock rather than re-opening it, which would fork the - // ping-pong sequence counter. Consensus recovers its true (view, log_view) - // from `recovered_state` instead of inferring a stale view from the WAL. - let consensus = restore_metadata_consensus(&owner, &topology, config, Rc::clone(&bus)); - let superblock = Rc::new(owner.superblock); - ( - Some(consensus), - Some(owner.journal), - owner.snapshot, - Some(superblock), - owner.snapshot_checkpoint, - Some(owner.client_table), - ) - } else { - (None, None, None, None, (0, 0), None) - }; - let metadata = ServerMetadata::new( - metadata_consensus, - journal_for_metadata, - snapshot_for_metadata, - superblock_for_metadata, - mux_stm, - Some(PathBuf::from(&config.system.path)), - ); - // Size the VSR client table before listeners bind and any client registers. - // Must precede the recovered-table install below: the setter rebuilds the - // table from scratch, so running it afterwards would drop every resumed - // session (and trip its empty-table assert). - metadata.set_clients_table_max(config.metadata.clients_table_max); - // Reinstall the sessions recovery restored from the checkpoint and the WAL - // suffix, so a rebooted node dedups retries and admits continuations from - // clients that kept their identity across the restart (IGGY-137). Recovery - // sized this table from the same config value, so the install preserves the - // configured cap. - if let Some(client_table) = recovered_client_table { - // Refusal (a client registered before this ran) keeps the live table - // and is logged by the callee; boot continues either way. - let _ = metadata.install_client_table(client_table); - } - // Seed the coordinator's last-checkpoint pairing so the first post-boot - // view-change superblock write records the real (checkpoint_op, checksum) - // instead of (0, 0). No-op on peer shards, which have no coordinator. - metadata.seed_checkpoint_ref(checkpoint_seed.0, checkpoint_seed.1); - // Keep the forced-checkpoint margin >= the configured prepare-queue - // depth: ops already pipelined while a checkpoint runs append into that - // margin (config validation keeps journal_slots >= 4x this). - metadata.set_checkpoint_margin(config.metadata.checkpoint_margin()); - - let shard_metrics = shard_metrics_all[usize::from(shard_id)].clone(); - // Notifier install deferred until after tick handler wires below. - let senders_for_notifier = senders.clone(); - let metrics_for_notifier = shard_metrics.clone(); - // Heap-pin like `shard_main` above: the builder future carries the whole - // shard construction state machine and outgrew clippy's `large_futures` - // cap; one allocation per shard startup. - let (shard, sessions) = Box::pin(build_shard_for_thread( - shard_id, - total_shards, - config, - &topology, - metadata, - Rc::clone(&bus), - senders, - inbox, - reply_inbox, - shard_metrics, - Arc::clone(&metadata_view), - )) - .await?; - - // Shard 0 owns the metadata consensus; publish its view so every shard's - // cluster-metadata read (and the SDK's leader discovery) marks the live - // primary. Detached: dies with this shard's runtime at process exit. - if shard_id == 0 { - let publisher_shard = Rc::clone(&shard); - let publisher_view = Arc::clone(&metadata_view); - compio::runtime::spawn(async move { - loop { - if let Some(consensus) = publisher_shard.plane.metadata().consensus.as_ref() { - // While this replica declines its recovered view's - // primaryship, that view must not reach the roster: the - // delegated shards would compute a leader that never - // heartbeats. Publish "unknown" until the election - // resolves the role. - let published = if consensus.has_ceded_primaryship() - && consensus.primary_index(consensus.view()) == consensus.replica() - { - crate::cluster_meta::METADATA_VIEW_UNKNOWN - } else { - u64::from(consensus.view()) - }; - publisher_view.store(published, Ordering::Relaxed); - } - compio::time::sleep(std::time::Duration::from_millis(100)).await; - } - }) - .detach(); - } - - info!( - shard = shard_id, - partitions = shard.plane.partitions().len(), - "server shard initialized" - ); - - // Re-check the cross-thread shutdown flag here, *before* spawning the - // message pump: it keeps the bus' `background_tasks` vec empty on the - // shutdown path, and shard 0 would otherwise still open TCP/QUIC/WS - // listeners for a server that is already tearing down, briefly - // accepting connections that immediately get torn by the watchdog. - // - // The flag is set, so the watchdog is (about to be) driving - // `bus.shutdown()`; await it so the runtime does not drop mid-drain. - if shutdown_flag_for_handoff.load(Ordering::Relaxed) { - if let Some(watchdog) = shutdown_watchdog.take() { - let _ = watchdog.await; - } - return Ok(()); - } - - // Tick handler must install before the notifier so early commits - // do not broadcast ticks whose handler slot is still `None`. - let (reconcile_wake_tx, reconcile_wake_rx) = channel::<()>(1); - let (reconcile_stop_tx, reconcile_stop_rx) = channel::<()>(1); - crate::partition_reconciler::install_tick_handler(&shard, reconcile_wake_tx); - - // Only shard 0 commits metadata. - if shard_id == 0 { - let notifier = make_metadata_commit_notifier(senders_for_notifier, metrics_for_notifier); - shard.plane.metadata().set_commit_notifier(Some(notifier)); - } else { - drop(senders_for_notifier); - drop(metrics_for_notifier); - } - - // The pump task also drives the consensus timer tick (heartbeats, prepare - // retransmit, view-change timeouts) as a select! arm, serialized with frame - // processing - see `run_message_pump`. - let (stop_tx, stop_rx) = channel(1); - let pump_shard = Rc::clone(&shard); - // Owned and awaited by shard_main at exit, NOT `track_background`: the - // background drain runs inside `bus.shutdown()`, which the Ctrl-C path - // never drives (the watchdog stands down when the token fires), so a - // tracked pump would be cancelled by runtime teardown mid final-flush - // and every graceful shutdown would silently drop the committed journal - // tail that had not hit a flush threshold yet. - let pump_shutdown_flag = Arc::clone(&shutdown_flag_for_handoff); - let mut pump_handle = Some(compio::runtime::spawn(async move { - // The pump itself flips the shared flag when a commit fault stops it, - // BEFORE its final flush, so a flush stalling on the failed device - // still reaches the watchdog and the bounded drain. Every sibling - // shard's watchdog drives its own graceful stop off the same flag; - // this shard's watchdog is what fires the token `shard_main` is - // parked on. The store below backstops the one fault the pump can - // only observe after that flip: a partition fenced by the final - // flush itself. - let fatal = pump_shard - .run_message_pump(stop_rx, Arc::clone(&pump_shutdown_flag)) - .await; - if fatal.is_some() { - pump_shutdown_flag.store(true, Ordering::Relaxed); - } - fatal - })); - - let reconciler_ctx = Rc::new(crate::partition_reconciler::ReconcilerCtx::new( - Rc::clone(&shard), - total_shards, - Rc::new(config.clone()), - topology.cluster_id, - topology.self_replica_id, - topology.replica_count, - )); - let reconcile_periodic = config - .system - .sharding - .reconcile_periodic_interval - .get_duration(); - let reconciler_handle = compio::runtime::spawn({ - let ctx = Rc::clone(&reconciler_ctx); - async move { - crate::partition_reconciler::run_reconciler( - ctx, - reconcile_wake_rx, - reconcile_stop_rx, - reconcile_periodic, - ) - .await; - } - }); - bus.track_background(reconciler_handle); - - // Per-shard heartbeat verifier: evicts connections that stop pinging, - // releasing their consumer-group membership. Gated on config so a - // deployment without heartbeats never reaps live sessions. - let heartbeat_stop_tx = if config.heartbeat.enabled { - let (hb_stop_tx, hb_stop_rx) = channel::<()>(1); - let hb_shard = Rc::clone(&shard); - let hb_sessions = Rc::clone(&sessions); - let hb_interval = config.heartbeat.interval.get_duration(); - let hb_handle = compio::runtime::spawn(async move { - crate::dispatch::session_ops::run_heartbeat_verifier( - hb_shard, - hb_sessions, - hb_interval, - hb_stop_rx, - ) - .await; - }); - bus.track_background(hb_handle); - Some(hb_stop_tx) - } else { - None - }; - // Expired-PAT cleaner: shard 0 only (it owns the metadata consensus - // group) and only when enabled. Each pass no-ops unless this node is - // the caught-up metadata primary, so the delete is proposed once and - // replicated to every replica. - let pat_cleaner_stop = if shard_id == 0 && config.personal_access_token.cleaner.enabled { - let (cleaner_stop_tx, cleaner_stop_rx) = channel(1); - let cleaner_shard = Rc::clone(&shard); - let interval = config.personal_access_token.cleaner.interval.get_duration(); - let cleaner_handle = compio::runtime::spawn(async move { - crate::personal_access_token_cleaner::run_pat_cleaner( - cleaner_shard, - cleaner_stop_rx, - interval, - ) - .await; - }); - bus.track_background(cleaner_handle); - Some(cleaner_stop_tx) - } else { - None - }; - - // Segment cleaner: runs on every shard (each replica trims its own log, - // primary and backup alike). Local and unreplicated; gated by the shared - // data-maintenance config. - let segment_cleaner_stop = if config.data_maintenance.messages.cleaner_enabled { - let (stop_tx, stop_rx) = channel(1); - let cleaner_shard = Rc::clone(&shard); - let interval = config.data_maintenance.messages.interval.get_duration(); - let cleaner_handle = compio::runtime::spawn(async move { - crate::segment_cleaner::run_segment_cleaner(cleaner_shard, stop_rx, interval).await; - }); - bus.track_background(cleaner_handle); - Some(stop_tx) - } else { - None - }; - - // One keep-alive per process, so shard 0 owns it. Started before the - // listeners bind: systemd counts `WatchdogSec=` from unit start, not from - // `READY=1`, so a slow recovery must not look like a hang. - #[cfg(feature = "systemd")] - if shard_id == 0 { - crate::systemd::spawn_watchdog(&bus); - } - - // Listener fence (see `BootstrapBarrier`). Peers still scan live - // shared metadata and load their on-disk partitions in - // `build_shard_for_thread`; the factory-bundle handoff only proves - // they *received* the bundle, not that they finished loading. Shard - // 0 must not accept client traffic until every peer's load scan is - // done, otherwise a partition created by the first client surfaces - // in a still-running scan with no segment dir on disk and aborts the - // node with `CannotReadPartitions`. By this point every shard has - // also spawned its pump + reconciler, so a partition created after - // the fence takes the runtime reconciler path on its owning shard. - match barrier { - BootstrapBarrier::Owner { ready_rx } => { - await_bootstrap_complete( - &ready_rx, - usize::from(total_shards.saturating_sub(1)), - &shutdown_flag_for_handoff, - poll_interval, - ) - .await?; - } - BootstrapBarrier::Waiter { ready_tx } => { - signal_bootstrap_complete( - shard_id, - &ready_tx, - &shutdown_flag_for_handoff, - poll_interval, - ) - .await?; - } - } - - // Listeners (replica + every client transport) bind on shard 0 only. - // Shard 0's coordinator round-robins inbound TCP/WS connections to - // peer shards via fd-transfer. QUIC and TCP-TLS clients terminate - // locally on shard 0 (their per-connection state is non-portable - - // see `LifecycleFrame::ClientWsConnectionSetup` rustdoc). - if shard_id == 0 { - let coord = shard - .coordinator() - .expect("shard 0 always has a coordinator attached by the builder"); - // Reseed the client-id minter above every recovered entry before any - // listener accepts. The counter is per process; the table it must not - // collide with was rebuilt from the previous boot's WAL. Keyed by view - // so a later promotion refolds the table (the minting path calls the - // same method, see `HttpInner::register_session_once`). - let boot_view = shard - .plane - .metadata() - .consensus - .as_ref() - .map_or(0, consensus::VsrConsensus::view); - coord.seed_client_sequence( - boot_view, - shard.plane.metadata().client_table.borrow().client_ids(), - ); - let on_client_request = make_client_request_handler( - &shard, - &sessions, - Arc::clone(&config.system), - config.personal_access_token.max_tokens_per_user, - ); - let (accepted_replica, dialed_replica) = - make_replica_delegation_fns(Rc::clone(&coord), &bus); - let accepted_client = make_shard_zero_client_accept_fns(coord, &bus, on_client_request); - - if let Err(error) = start_tcp_runtime( - &shard, - config, - &topology, - accepted_replica, - dialed_replica, - accepted_client, - &shard_metrics_all, - ) - .await - { - let _ = stop_tx.try_send(()); - let _ = reconcile_stop_tx.try_send(()); - if let Some(tx) = &heartbeat_stop_tx { - let _ = tx.try_send(()); - } - if let Some(cleaner_stop_tx) = &pat_cleaner_stop { - let _ = cleaner_stop_tx.try_send(()); - } - if let Some(tx) = &segment_cleaner_stop { - let _ = tx.try_send(()); - } - // The bind failure is the primary fault; the drain verdict only - // matters for the log it emits. - let _ = await_pump_drain(pump_handle.take(), config, shard_id).await; - // Neither the flag nor the bus token has fired yet on this path, - // so the watchdog is still idle-looping; awaiting it would hang. - // Detach and let `run_shard_thread`'s unwind flip the flag. - if let Some(watchdog) = shutdown_watchdog.take() { - watchdog.detach(); - } - return Err(error); - } - - // Every enabled client transport is bound and accepting by here, so - // this is the first point at which a unit ordered after us may dial. - #[cfg(feature = "systemd")] - crate::systemd::notify_ready(); - } - - bus.token().wait().await; - #[cfg(feature = "systemd")] - if shard_id == 0 { - crate::systemd::notify_stopping(); - } - let _ = stop_tx.try_send(()); - let _ = reconcile_stop_tx.try_send(()); - if let Some(tx) = &heartbeat_stop_tx { - let _ = tx.try_send(()); - } - if let Some(cleaner_stop_tx) = &pat_cleaner_stop { - let _ = cleaner_stop_tx.try_send(()); - } - if let Some(tx) = &segment_cleaner_stop { - let _ = tx.try_send(()); - } - - // Await the watchdog even when the drain verdict is an error: the token - // has fired, so it either stands down within one poll interval or is - // mid-`bus.shutdown()`, and dropping it there truncates in-flight - // `ClientForwardFailed` replies. - let pump_verdict = await_pump_drain(pump_handle.take(), config, shard_id).await; - if let Some(watchdog) = shutdown_watchdog.take() { - let _ = watchdog.await; - } - pump_verdict?; - - info!(shard = shard_id, "server shard exited cleanly"); - Ok(()) -} - -/// Await the message pump's completion before the shard returns: its -/// post-loop work includes the final flush of every committed journal to -/// segment storage, and returning first drops the compio runtime, which -/// cancels that flush at its next await point. -/// -/// `Err` means the pump was already dead (a panic, or an exit outside the -/// stop protocol), so its final flush never ran and the shard must not -/// report a clean exit. The verdict is the inner `JoinError`; the timeout -/// wrapper alone cannot see it, and a shard that swallows it prints -/// "exited cleanly" over a corpse. -async fn await_pump_drain( - pump_handle: Option>>, - config: &ServerConfig, - shard_id: u16, -) -> Result<(), ServerError> { - let Some(pump_handle) = pump_handle else { - return Ok(()); - }; - let drain_budget = config.system.sharding.shutdown_drain_timeout.get_duration(); - let Ok(join_result) = compio::time::timeout(drain_budget, pump_handle).await else { - error!( - shard = shard_id, - timeout = ?drain_budget, - "message pump did not drain within the shutdown budget; \ - committed journal tail may not have flushed" - ); - return Err(ServerError::ShardPumpDrainTimedOut { - shard_id, - timeout: drain_budget, - }); - }; - // `JoinError` renders a panic as the bare "Task has panicked" and the - // type is not re-exported, so the payload -- the only part with - // diagnostic value -- is lifted by re-raising into an immediate catch. - // The panic hook already ran when the task died; `resume_unwind` does - // not run it again, so nothing is printed twice and the message finally - // reaches the tracing sink too. - let reason = match panic::catch_unwind(panic::AssertUnwindSafe(|| join_result.resume_unwind())) - { - Ok(Some(None)) => return Ok(()), - // The pump drained and flushed; it just has nothing left to serve. - // Fail the shard so the process exits non-zero: a node that stopped - // because it could not persist a cluster-committed op must not look - // to an orchestrator like a clean shutdown. - Ok(Some(Some(fault))) => { - error!( - shard = shard_id, - namespace_raw = fault.namespace_raw, - op = fault.op, - operation = ?fault.operation, - "message pump stopped on a partition commit fault; \ - the server is shutting down" - ); - return Err(ServerError::ShardFatal { - shard_id, - namespace_raw: fault.namespace_raw, - op: fault.op, - }); - } - Ok(None) => "task was cancelled".to_string(), - Err(payload) => payload - .downcast_ref::<&str>() - .map(|message| (*message).to_string()) - .or_else(|| payload.downcast_ref::().cloned()) - .map_or_else( - || "task panicked".to_string(), - |message| format!("task panicked: {message}"), - ), - }; - error!( - shard = shard_id, - "message pump died instead of draining ({reason}); \ - committed journal tail may not have flushed" - ); - Err(ServerError::ShardPumpDied { shard_id, reason }) -} - -/// Block until shard 0 broadcasts the metadata factory bundle, or the -/// cross-thread shutdown flag flips. Polled in a `poll_interval` loop -/// so a shard 0 that panics before it broadcasts cannot strand peer -/// shards: the shutdown path flips the flag, every waiter observes it -/// on the next tick, and the server tears down instead of hanging. -/// -/// Uses `try_recv` + sleep rather than `timeout(recv())`. Crossfire 3.x -/// documents `recv()` as cancellation-safe (no leak/deadlock) but does -/// not guarantee atomicity for the dropped future's result; `try_recv` -/// keeps each tick fully synchronous and side-effect-free, so the -/// shutdown poll cadence cannot ambiguously consume a bundle. -async fn await_metadata_bundle( - shard_id: u16, - bundle_rx: &crossfire::MAsyncRx>, - shutdown_flag: &Arc, - poll_interval: Duration, -) -> Result { - loop { - match bundle_rx.try_recv() { - Ok(bundle) => return Ok(bundle), - Err(crossfire::TryRecvError::Disconnected) => { - return Err(ServerError::MetadataHandoffAborted { shard_id }); - } - Err(crossfire::TryRecvError::Empty) => { - if shutdown_flag.load(Ordering::Relaxed) { - return Err(ServerError::MetadataHandoffAborted { shard_id }); - } - compio::time::sleep(poll_interval).await; - } - } - } -} - -/// Push `peers` cloned bundles onto `bundle_tx`, polling each send in a -/// `poll_interval` loop so the cross-thread shutdown flag can interrupt -/// a stalled handoff. Symmetric to [`await_metadata_bundle`]: shutdown -/// observed mid-handshake aborts cleanly rather than stalling on a -/// `send` future that can no longer make progress. -/// -/// Uses `try_send` + sleep rather than `timeout(send())`. Crossfire 3.x -/// documents `send()` as cancellation-safe in the leak/deadlock sense -/// but explicitly warns the true result is unknown when `SendFuture` is -/// dropped on cancellation. For a retry loop that re-clones on every -/// tick that would risk publishing the same bundle twice, stuffing the -/// bounded channel past `peers` and stranding a follow-up `send`. -/// `try_send` returns the bundle back inside `TrySendError::Full`, so -/// the loop reuses it instead of re-cloning when the channel is full. -async fn broadcast_metadata_bundle( - shard_id: u16, - bundle_tx: &crossfire::MAsyncTx>, - bundle: ServerMetadataBundle, - peers: u16, - shutdown_flag: &Arc, - poll_interval: Duration, -) -> Result<(), ServerError> { - for _ in 0..peers { - let mut pending = bundle.clone(); - loop { - match bundle_tx.try_send(pending) { - Ok(()) => break, - Err(crossfire::TrySendError::Disconnected(_)) => { - // Every peer dropped its `bundle_rx` before recv. Shard - // 0 must not silently continue past handoff: it would - // bind listeners and commit consensus state for a - // cluster whose peers are gone. Propagate the abort so - // `shard_main` short-circuits before further side - // effects; `shutdown_flag` will flip via the normal - // teardown path. - return Err(ServerError::MetadataHandoffAborted { shard_id }); - } - Err(crossfire::TrySendError::Full(returned)) => { - if shutdown_flag.load(Ordering::Relaxed) { - return Err(ServerError::MetadataHandoffAborted { shard_id }); - } - pending = returned; - compio::time::sleep(poll_interval).await; - } - } - } - } - Ok(()) -} - -/// Peer side of [`BootstrapBarrier`]: tell shard 0 this shard finished -/// loading its on-disk partitions. Mirrors [`broadcast_metadata_bundle`]'s -/// `try_send`-or-shutdown poll loop so a sibling failure (which flips the -/// shutdown flag) drives this out instead of stranding it on a full -/// channel. The channel is sized to the peer count and each peer sends -/// exactly once, so `Full` is not expected; the branch only keeps the -/// loop interruptible. -async fn signal_bootstrap_complete( - shard_id: u16, - ready_tx: &crossfire::MAsyncTx>, - shutdown_flag: &Arc, - poll_interval: Duration, -) -> Result<(), ServerError> { - let mut pending = shard_id; - loop { - match ready_tx.try_send(pending) { - Ok(()) => return Ok(()), - Err(crossfire::TrySendError::Disconnected(_)) => { - // Shard 0 dropped its `ready_rx` before draining (it - // aborted before binding listeners). Propagate so this - // shard short-circuits; the shutdown flag flips via the - // normal teardown path. - return Err(ServerError::MetadataHandoffAborted { shard_id }); - } - Err(crossfire::TrySendError::Full(returned)) => { - if shutdown_flag.load(Ordering::Relaxed) { - return Err(ServerError::MetadataHandoffAborted { shard_id }); - } - pending = returned; - compio::time::sleep(poll_interval).await; - } - } - } -} - -/// Owner side of [`BootstrapBarrier`]: drain one ready signal per peer -/// before shard 0 binds listeners. Polls the shutdown flag so a peer that -/// dies mid-load (flipping the flag) aborts the wait instead of hanging on -/// a signal that will never arrive. A single shard (`peers == 0`) returns -/// immediately. -async fn await_bootstrap_complete( - ready_rx: &crossfire::MAsyncRx>, - peers: usize, - shutdown_flag: &Arc, - poll_interval: Duration, -) -> Result<(), ServerError> { - let mut remaining = peers; - while remaining > 0 { - match ready_rx.try_recv() { - Ok(_shard_id) => remaining -= 1, - Err(crossfire::TryRecvError::Disconnected) => { - return Err(ServerError::ShardBootstrapBarrierAborted { remaining }); - } - Err(crossfire::TryRecvError::Empty) => { - if shutdown_flag.load(Ordering::Relaxed) { - return Err(ServerError::ShardBootstrapBarrierAborted { remaining }); - } - compio::time::sleep(poll_interval).await; - } - } - } - Ok(()) -} - -/// Spawn a per-shard polling task that watches the cross-thread shutdown -/// flag and triggers this shard's bus shutdown on transition. The flag -/// is the only Send signal we have; the bus' shutdown machinery is -/// `!Send` (`Rc>` + per-shard `async_channel`), so it must be -/// triggered from within the runtime that owns the bus. -/// -/// The caller owns the returned handle and must await it on the exit paths -/// where shutdown is in progress (flag set or bus token triggered): -/// dropping it there cancels the watchdog mid-`bus.shutdown()`, truncating -/// in-flight `ClientForwardFailed` replies (terminal per `SendError` docs). -/// It cannot go through `bus.track_background` instead: the watchdog itself -/// drives `bus.shutdown()`, and the bg-drain loop in `shutdown()` would -/// re-enter awaiting the watchdog's own pending shutdown call -/// (self-deadlock). The await is bounded: once the token fires the loop -/// stands down within one poll interval, and the shutdown call itself is -/// capped by `drain_timeout`. -#[allow(clippy::needless_pass_by_value)] -fn spawn_shutdown_watchdog( - bus: Rc, - shutdown_flag: Arc, - drain_timeout: Duration, - poll_interval: Duration, -) -> compio::runtime::JoinHandle<()> { - let bus_for_task = Rc::clone(&bus); - let bus_token = bus.token(); - compio::runtime::spawn(async move { - loop { - if shutdown_flag.load(Ordering::Relaxed) { - break; - } - if bus_token.is_triggered() { - // Bus shutdown was driven from elsewhere (e.g. internal - // failure path). Watchdog has nothing left to do. - return; - } - compio::time::sleep(poll_interval).await; - } - let _ = bus_for_task.shutdown(drain_timeout).await; - }) -} - -/// Copy the configured cluster roster plus this node's own client ports into -/// the shared [`ClusterRoster`] so the binary `GetClusterMetadata` read serves -/// the real topology. `self_*` back only the cluster-disabled self-synthesis -/// and carry the requested listener ports from the resolved topology, not the -/// bound ones (a `:0` wildcard is reported as 0). The self address resolves -/// through [`self_advertised_address`], which boot validation has already -/// guaranteed names somewhere a client can dial. -fn build_cluster_roster( - shard_id: u16, - config: &ServerConfig, - topology: &TcpTopology, - metadata_view: Arc, -) -> Result { - let declared = config.node.advertised_address.as_deref(); - let self_advertised = self_advertised_address(declared, derived_bind_ip(topology, config)); - // The roster answers this per node, so a value here would be read by - // nobody. Silence would leave the operator believing it took effect. - // Every shard builds its own roster off the same config, so keep the - // operator-facing explanation to one line per process. - if declared.is_some() && config.cluster.enabled && shard_id == 0 { - warn!( - "node.advertised_address is set but cluster.enabled is true, so it is ignored; \ - the client-facing address of each node comes from its cluster.nodes entry" - ); - } - Ok(ClusterRoster { - enabled: config.cluster.enabled, - name: config.cluster.name.clone(), - nodes: resolved_roster_nodes(&config.cluster).map_err(ServerError::Config)?, - self_advertised, - self_ports: configs::cluster::TransportPorts { - tcp: config - .tcp - .enabled - .then(|| topology.client_listen_addr.port()), - quic: topology.quic_listen_addr.map(|addr| addr.port()), - http: topology.http_listen_addr.map(|addr| addr.port()), - websocket: topology.ws_listen_addr.map(|addr| addr.port()), - tcp_replica: None, - }, - metadata_view, - }) -} - -#[allow(clippy::too_many_arguments, clippy::too_many_lines)] -async fn build_shard_for_thread( - shard_id: u16, - total_shards: u16, - config: &ServerConfig, - topology: &TcpTopology, - metadata: ServerMetadata, - bus: Rc, - senders: Vec, - inbox: ShardReceiver, - reply_inbox: ShardReceiver, - metrics: ShardMetrics, - metadata_view: Arc, -) -> Result<(Rc, Rc>), ServerError> { - let shard_local_id = ShardId::new(shard_id); - let total_partitions = metadata.mux_stm.streams().read(|inner| { - inner - .items - .iter() - .map(|(_, stream)| { - stream - .topics - .iter() - .map(|(_, topic)| topic.partitions.len()) - .sum::() - }) - .sum::() - }); - - // IggyPartitions holds only the partitions owned by this shard - // (see the filter below at insert time), so the server-wide total - // is an N-fold overshoot. `ceil(total / shards) * 2` is a coarse - // upper bound that absorbs hash skew without paying the full - // multiplier. PapayaShardsTable below stays sized to the server-wide - // total because every shard routes every namespace. - let owned_partitions_capacity = total_partitions - .div_ceil(usize::from(total_shards).max(1)) - .saturating_mul(2); - // At-rest encryption: built once per shard from the shared config; the - // ingestion path encrypts on the primary and the poll reply decrypts. - // A bad key fails the boot rather than silently serving plaintext. - let encryptor = if config.system.encryption.enabled { - let aes = Aes256GcmEncryptor::from_base64_key(&config.system.encryption.key) - .map_err(|error| ServerError::Iggy(Box::new(error)))?; - Some(Arc::new(EncryptorKind::Aes256Gcm(aes))) - } else { - None - }; - let partitions = IggyPartitions::with_capacity( - shard_local_id, - PartitionsConfig { - messages_required_to_save: iggy_common::DEFAULT_MESSAGES_REQUIRED_TO_SAVE, - size_of_messages_required_to_save: IggyByteSize::from( - iggy_common::DEFAULT_SIZE_OF_MESSAGES_REQUIRED_TO_SAVE, - ), - enforce_fsync: iggy_common::DEFAULT_ENFORCE_FSYNC, - validate_checksum: config.system.partition.validate_checksum, - segment_size: IggyByteSize::from(iggy_common::DEFAULT_SEGMENT_SIZE), - preallocate_segments: iggy_common::DEFAULT_PREALLOCATE_SEGMENTS, - encryptor, - path_layout: partitions::PartitionPathLayout { - streams_root: config.system.get_streams_path(), - topics_dir: config.system.topic.path.clone(), - partitions_dir: config.system.partition.path.clone(), - }, - }, - owned_partitions_capacity, - ); - let shards_table = PapayaShardsTable::with_capacity(total_partitions); - - // Stream-filter inside the `read()` closure: only partitions owned by - // this shard need the heavy (`Arc` + `Partition`) clones - // for the async `load_partition` below. Non-owning entries are pushed - // straight into `shards_table` here, so no Vec scales with the - // server-wide partition count. - let owned = metadata.mux_stm.streams().read(|inner| { - let mut owned = Vec::with_capacity(owned_partitions_capacity); - for (_, stream) in &inner.items { - for (topic_id, topic) in &stream.topics { - for partition in &topic.partitions { - let namespace = IggyNamespace::new(stream.id, topic_id, partition.id); - let owning_shard = - calculate_shard_assignment(&namespace, u32::from(total_shards)); - if owning_shard == shard_id { - // Shared per-partition stats from the registry: the - // same `Arc` backs every shard's `get_topic` reply. - let stats = inner.stats_registry.partition( - stream.id, - topic_id, - partition.id, - topic.stats.clone(), - ); - owned.push(( - stream.id, - topic_id, - stats, - partition.clone(), - TopicRuntimeOptions::from_resource_options(&topic.options), - )); - } else { - shards_table.insert( - namespace, - PartitionLocation::new( - ShardId::new(owning_shard), - partition.created_revision, - ), - ); - } - } - } - } - owned - }); - - // Snapshot totals were zeroed once on shard 0 before the factory - // bundle was broadcast (see `MetadataHandoff::Owner`). All shards - // here only add their per-partition deltas, so the shared - // `Arc` atomics race only against other atomic adds. - for (stream_id, topic_id, partition_stats, partition_metadata, topic_runtime) in owned { - let namespace = IggyNamespace::new(stream_id, topic_id, partition_metadata.id); - let partition = match load_partition( - config, - namespace, - Arc::clone(&partition_stats), - &partition_metadata, - topic_runtime, - topology.cluster_id, - topology.self_replica_id, - topology.replica_count, - Rc::clone(&bus), - ) - .await - { - Ok(partition) => partition, - // ONE damaged local chain must not take the node down. The shapes - // this refuses are structural -- what a failed state-transfer - // quarantine leaves behind, or damage the recovery walk proved - // inside a segment. What follows depends on whether a peer can - // restore the data. With peers, the segment files are fenced - // aside (keeping the superblock so the group cannot re-enter - // view 0), the group is materialised fresh, and the ordinary - // rejoin path (repair, then state transfer on a refused floor) - // refills it. Single-replica, only a chain-shape refusal whose - // planned chain provably holds ZERO recoverable bytes still - // fences and rebuilds: nothing servable is at stake, so an empty - // rebuild hides no loss. The verdict variant alone is not that - // evidence -- a hole and an orphan empty segment both fire over - // fully populated chains -- which is why the gate reads the byte - // total the refusal carries. Every other refusal tombstones, - // leaving its files exactly where they are: a rebuilt empty - // partition answers polls exactly like a healthy empty one and - // hides the loss, while an unrouted namespace is a failure an - // operator can see. - Err(ServerError::PartitionRecoveryRefused { dir, reason, .. }) => { - let partition_dir = dir.to_string_lossy().into_owned(); - let rebuild_for_rejoin = topology.replica_count > 1 - || matches!( - reason, - PartitionRecoveryRefusal::Hole { - recoverable_bytes: 0, - .. - } | PartitionRecoveryRefusal::EmptyNonTailSegment { - recoverable_bytes: 0, - .. - } - ); - error!( - stream_id, - topic_id, - partition_id = partition_metadata.id, - partition_dir, - %reason, - "refusing the recovered segment chain" - ); - // A pass-A refusal folded nothing into the stats (recovery - // counts only accepted chains), but the hydrate-reopen refusal - // arrives after a fully counted load, so clear them either way. - partition_stats.zero_out_all(); - if !rebuild_for_rejoin { - // No quarantine here, mirroring the superblock arm below: - // a tombstone is only durable if its cause is. Fencing the - // chain aside would leave the next boot zero segments to - // walk, so it would re-seed from the surviving superblock, - // plant a fresh segment, and serve the partition empty - // with no refusal logged. Left at their real paths, the - // same files re-derive this verdict (and this log line) - // every boot, and the reconciler's tombstone gate keeps - // the namespace away from a fresh build, whose - // initial-segment open would truncate the oldest refused - // segment in place. The one refusal whose cause is NOT - // durable is `StorageSizeMismatch`: it fires from the - // reopen right after recovery truncated the same file, so - // the next boot re-walks the already-truncated bytes and, - // unless the length diverges again, accepts the chain - // instead of re-tombstoning -- acceptable for an - // assertion that the filesystem lied about a length. - // `%reason` repeated on purpose: this is the line an - // operator greps to enumerate dark partitions, so it has - // to carry the verdict on its own. - error!( - stream_id, - topic_id, - partition_id = partition_metadata.id, - partition_dir, - %reason, - "no peer replica holds this partition's data; leaving the refused \ - segment files in place and tombstoning it instead of serving it \ - empty" - ); - partitions.tombstone(namespace); - continue; - } - match partitions::state_transfer::quarantine_segment_files(&partition_dir).await { - Ok(fenced_dir) => error!( - stream_id, - topic_id, - partition_id = partition_metadata.id, - fenced_dir, - "quarantined the refused segment files; they are kept for inspection" - ), - Err(error) => { - // NOT rebuilt: `build_partition_fresh` reaches - // `ensure_initial_segment`, which opens segment 0 with - // `file_exists = false` and TRUNCATES whatever the - // failed quarantine left behind. The likeliest failures - // (suffix cap exhausted, `create_dir_all`) move zero - // files, so rebuilding would destroy the oldest segment - // on the first attempt while the higher-offset survivors - // keep refusing every boot -- a loop that never - // terminates and eats the chain one segment at a time. - // Tombstone instead: the namespace stays unmaterialised - // and unrouted, the reconciler backs off, and an - // operator still has every byte. - error!( - stream_id, - topic_id, - partition_id = partition_metadata.id, - partition_dir, - %error, - "failed to quarantine the refused segment files; leaving this \ - partition tombstoned rather than rebuilding over them" - ); - partitions.tombstone(namespace); - continue; - } - } - build_partition_fresh( - config, - namespace, - partition_stats, - partition_metadata.created_revision, - topic_runtime, - topology.cluster_id, - topology.self_replica_id, - topology.replica_count, - partition_metadata.created_view, - Rc::clone(&bus), - ) - .await? - } - // An untrustworthy superblock fences ONE group, not the node. The - // segment files stay exactly where they are -- unlike a refused - // chain, the data on disk is not the thing in doubt -- so there is - // nothing to quarantine and nothing to rebuild: rebuilding fresh - // would hand this replica a view-0 identity while a record it - // cannot read says otherwise. Tombstoned, the namespace stays - // unmaterialised and unrouted, the reconciler backs off, and an - // operator has every byte plus a message naming the directory. - Err( - error @ (ServerError::PartitionSuperblockIo { .. } - | ServerError::PartitionSuperblockVersionUnknown { .. } - | ServerError::PartitionSuperblockUnverifiable { .. } - | ServerError::PartitionSuperblockUndecodable { .. } - | ServerError::PartitionSuperblockIdentityMismatch { .. }), - ) => { - error!( - stream_id, - topic_id, - partition_id = partition_metadata.id, - %error, - "cannot trust this partition's durable consensus state; tombstoning the \ - partition and continuing to boot the rest of the shard" - ); - partition_stats.zero_out_all(); - partitions.tombstone(namespace); - continue; - } - Err(error) => return Err(error), - }; - partitions.insert(namespace, partition); - shards_table.insert( - namespace, - PartitionLocation::new(ShardId::new(shard_id), partition_metadata.created_revision), - ); - } - - let shard_handle = Rc::new(RefCell::new(None)); - // Same wiring path as the simulator's shell mode: one per-shard - // SessionManager shared by the client-request handler (binds sessions) - // and the get_clients handler (reads them). It also carries this shard's - // cluster roster for the GetClusterMetadata read. - let ShellHandlers { - on_replica_message, - on_client_request, - on_metadata_submit, - on_list_clients, - on_partition_read, - sessions, - } = wire_shell_handlers( - &bus, - &shard_handle, - Arc::clone(&config.system), - config.personal_access_token.max_tokens_per_user, - ); - sessions - .borrow_mut() - .set_cluster_roster(Rc::new(build_cluster_roster( - shard_id, - config, - topology, - metadata_view, - )?)); - let shard_name = format!("server-shard-{shard_id}"); - let built = IggyShardBuilder::new( - ShardIdentity::new(shard_id, shard_name), - Rc::clone(&bus), - on_replica_message, - on_client_request, - on_metadata_submit, - on_list_clients, - on_partition_read, - metadata, - partitions, - senders, - inbox, - reply_inbox, - shards_table, - PartitionConsensusConfig::new( - topology.cluster_id, - shard::ReplicaTopology::new(topology.self_replica_id, topology.replica_count), - Rc::clone(&bus), - ), - CoordinatorConfig { - skip_shard_zero_for_replicas: config.cluster.coordinator.skip_shard_zero_for_replicas, - skip_shard_zero_for_clients: config.cluster.coordinator.skip_shard_zero_for_clients, - }, - metrics, - ) - .build() - .map_err(ServerError::ShardConstruction)?; - - let shard = Rc::new(built.shard); - // Repair pacing is shared by both planes' repair loops, so it is a - // per-shard tunable set once here rather than per consensus group. - shard.set_repair_retry_ticks(repair_retry_ticks(config)); - shard.set_superblock_wedged_fatal_failures(superblock_wedged_fatal_failures(config)); - shard.set_served_segment_cache_bytes_max( - config - .partition - .transfer_served_cache_bytes_max - .as_bytes_u64(), - ); - shard.set_partition_artifact_len_max( - config.partition.transfer_artifact_bytes_max.as_bytes_u64(), - ); - shard.set_repair_chunk_max(config.cluster.repair_chunk_max as u64); - // Bounds a served state-transfer chunk. A frame above the bus ceiling is - // rejected by the RECEIVING transport, which tears the replica connection - // down rather than dropping one message. - shard.set_bus_max_message_size( - usize::try_from(config.message_bus.max_message_size.as_bytes_u64()).unwrap_or(usize::MAX), - ); - *shard_handle.borrow_mut() = Some(Rc::downgrade(&shard)); - Ok((shard, sessions)) -} - -// Pin the configs-crate default literals (duplicated there to avoid a -// build-time edge onto the runtime crates) against the runtime constants, -// mirroring the message_bus IOV_MAX pin. A drift on either side fails this -// crate's build until both are reconciled. -const _: () = assert!( - configs::metadata::DEFAULT_METADATA_PREPARE_QUEUE_DEPTH - == consensus::PIPELINE_PREPARE_QUEUE_MAX -); -const _: () = assert!( - configs::metadata::DEFAULT_METADATA_JOURNAL_SLOTS - == journal::prepare_journal::DEFAULT_SLOT_COUNT -); -const _: () = assert!( - configs::partition::DEFAULT_PARTITION_PREPARE_QUEUE_DEPTH - == consensus::PIPELINE_PREPARE_QUEUE_MAX -); -const _: () = - assert!(configs::metadata::DEFAULT_METADATA_CLIENTS_TABLE_MAX == consensus::CLIENTS_TABLE_MAX); -const _: () = - assert!(configs::cluster::DEFAULT_VIEW_PROBE_ATTEMPTS_MAX == consensus::PROBE_ATTEMPTS_MAX); -const _: () = - assert!(configs::partition::DEFAULT_EVICTED_RING_CAPACITY == partitions::EVICTED_RING_CAPACITY); -const _: () = assert!( - configs::partition::DEFAULT_EVICTED_RING_BYTES_MAX == partitions::EVICTED_RING_BYTES_MAX -); -const _: () = assert!( - configs::partition::DEFAULT_TRANSFER_ARTIFACT_BYTES_MAX - == shard::PARTITION_ARTIFACT_LEN_DEFAULT -); -const _: () = assert!( - configs::partition::DEFAULT_TRANSFER_SERVED_CACHE_BYTES_MAX - == shard::SERVED_SEGMENT_CACHE_BYTES_DEFAULT -); -const _: () = assert!(configs::cluster::DEFAULT_REPAIR_CHUNK_MAX as u64 == shard::REPAIR_CHUNK_MAX); -const _: () = assert!( - configs::cluster::STATE_CHUNK_HEADER_LEN - == size_of::() as u64 -); -// Both prepare-queue ceilings are pinned by the view-change wire, not by memory: a -// `DoViewChange` carries the sender's suffix spanning `commit..=op` with one nack -// bit and one present bit per entry, each bitset a single `u128`. The depth bounds -// `op - commit`, so a depth at or above `DVC_HEADERS_MAX` produces entries the new -// primary can neither adopt nor prove dead. Strictly less than, because the head op -// needs the reserved slot. -const _: () = - assert!(configs::metadata::MAX_METADATA_PREPARE_QUEUE_DEPTH < consensus::DVC_HEADERS_MAX); -const _: () = - assert!(configs::partition::MAX_PARTITION_PREPARE_QUEUE_DEPTH < consensus::DVC_HEADERS_MAX); -// `DVC_HEADERS_MAX` is a bare literal in both the wire crate, which sizes the -// bitsets, and the consensus crate, which cannot depend on it the other way around. -// Same u128, so a drift lets one side address entries the other cannot. -const _: () = - assert!(consensus::DVC_HEADERS_MAX == iggy_binary_protocol::consensus::DVC_HEADERS_MAX); -const _: () = assert!(consensus::DVC_HEADERS_MAX == u128::BITS as usize); -/// `[cluster] superblock_wedged_fatal_timeout` as a consecutive-failure count. -/// Retries pin at the backoff cap after warmup, so the window divided by -/// [`journal::superblock::SUPERBLOCK_RETRY_BACKOFF_MAX_MICROS`] bounds how -/// long a wedged replica may limp before it fail-stops. Zero stays zero -/// (fail-stop disabled). -pub(crate) fn superblock_wedged_fatal_failures(config: &ServerConfig) -> u64 { - superblock_window_to_failures( - config - .cluster - .superblock_wedged_fatal_timeout - .get_duration(), - ) -} - -fn superblock_window_to_failures(window: Duration) -> u64 { - if window.is_zero() { - return 0; - } - let cap_micros = u128::from(journal::superblock::SUPERBLOCK_RETRY_BACKOFF_MAX_MICROS); - u64::try_from((window.as_micros() / cap_micros).max(1)).unwrap_or(u64::MAX) -} - -/// Floor for the post-restart read-recovery deadline (see -/// [`recovery_barrier_deadline`]). At and below the 5s default heartbeat the -/// worst-case recovery is dominated by the heartbeat-independent term - the -/// `ViewChangeStatus` backstop plus election ceremony and suffix recommit, -/// empirically ~7s - so the scaled value must never fall under this or a -/// fast-heartbeat cluster would 503 legitimate reads mid-recovery. The backstop -/// is the configurable `[cluster] view_change_status_timeout`; raising it past -/// its 5s default is why `recovery_barrier_deadline` scales that knob in too -/// rather than leaning on this floor to cover it. -const RECOVERY_BARRIER_DEADLINE_FLOOR: Duration = Duration::from_secs(15); - -/// Safety factor applied to each scaled term of the recovery deadline: a slower -/// heartbeat stretches election and suffix recommit proportionally, and a wider -/// status backstop stretches the ceremony it bounds. 3x reproduces the -/// empirically chosen 15s margin at the shared 5s default (3 x 5s = 15s) and -/// holds that factor as either knob grows. -const RECOVERY_BARRIER_MULTIPLIER: u32 = 3; - -/// How long the post-restart read path waits for the recovered WAL suffix to -/// re-commit before failing loud (retryable 503): the largest of the fixed -/// floor, a `[cluster] heartbeat_timeout`-scaled window, and a -/// `[cluster] view_change_status_timeout`-scaled window. Both knobs feed it -/// because either, raised far past its default, stretches worst-case recovery -/// past the fixed floor; see `await_recovery_barrier` for the read-side wait. -pub(crate) fn recovery_barrier_deadline( - heartbeat: Duration, - view_change_status: Duration, -) -> Duration { - // saturating: neither timeout has a config ceiling, plain `*` panics - heartbeat - .saturating_mul(RECOVERY_BARRIER_MULTIPLIER) - .max(view_change_status.saturating_mul(RECOVERY_BARRIER_MULTIPLIER)) - .max(RECOVERY_BARRIER_DEADLINE_FLOOR) -} - -/// Shard 0's half of a metadata recovery: everything [`recover`] produced except the -/// state machine, which every shard receives through the factory bundle. -/// -/// Named rather than a positional tuple: the fields are same-typed `Option`s and -/// `(u64, u128)` pairs that a reorder would silently rebind, and one of them decides -/// what view the replica boots into. -struct RecoveredOwnerState { - journal: PrepareJournal, - snapshot: Option, - last_applied_op: Option, - last_journaled_op: Option, - client_table: ClientTable, - superblock: PingPongSuperblock, - recovered_state: Option, - snapshot_checkpoint: (u64, u128), -} - -/// Rebuild metadata consensus from what recovery read off this replica's own disk. -/// -/// Takes the recovery result, topology and config whole rather than the dozen-plus -/// scalars it needs from them: most were `u64` tick counts, where a misordered -/// argument type-checks and mistunes a timeout silently. -fn restore_metadata_consensus( - owner: &RecoveredOwnerState, - topology: &TcpTopology, - config: &ServerConfig, - bus: Rc, -) -> VsrConsensus> { - let journal = &owner.journal; - let replica_count = topology.replica_count; - let recovered_state = owner.recovered_state; - let snapshot_floor = owner - .snapshot - .as_ref() - .map_or(0, IggySnapshot::sequence_number); - let commit_watermark = owner.last_applied_op.unwrap_or(snapshot_floor); - let restored_op = owner.last_journaled_op.unwrap_or(snapshot_floor); - let recovery_deadline = recovery_barrier_deadline( - config.cluster.heartbeat_timeout.get_duration(), - config.cluster.view_change_status_timeout.get_duration(), - ); - let prepare_queue_depth = config.metadata.prepare_queue_depth; - - let last_header = journal - .last_op() - .and_then(|op| usize::try_from(op).ok()) - .and_then(|op| journal.header(op).map(|header| *header)); - // On a RESTART in a cluster, rejoin as a quorum-invisible backup and - // probe for the current view (`RequestStartView`): the view's primary - // answers with a `StartView`, the replica adopts it as a backup, and - // journal repair fills any WAL gap. A probing replica never resumes - // primaryship -- if this replica IS the current primary-by-index, its - // probe makes the backups elect past it. - // The probe re-broadcasts on its timeout, so it needs no live mesh at - // boot. A FRESH boot keeps the plain init: the cluster needs its view-0 - // primary to exist, and a single-replica cluster has no peer to ask. - // - // Prior life is EITHER a non-empty WAL or a recovered superblock. A view - // change persists without touching the WAL, so a replica that changed - // view before its first metadata write comes back with a non-zero view - // and an empty journal; gating on the WAL alone would `init()` it into - // `Status::Normal` as primary for a view the cluster may have moved past, - // with `ceded_primaryship` false and no probe to correct it. - // - // The rejoin also awaits a state transfer: snapshot-shaped metadata state - // (snapshot + client table) is replaced from the live primary the probe - // finds, then journal repair fills the tail. If the probe exhausts - // instead -- full-cluster bootstrap, nobody live to fetch from -- the - // election fallback clears the stage and this local recovery stands. - let join = if replica_count > 1 && (restored_op > 0 || recovered_state.is_some()) { - JoinMode::ProbeAsBackup { - await_state_transfer: true, - } - } else { - JoinMode::Init - }; - let timers = consensus_timers(config); - let consensus = VsrConsensus::restored( - topology.cluster_id, - topology.self_replica_id, - replica_count, - server_common::sharding::METADATA_GROUP, - bus, - // Request queue keeps the stock 2x ratio over the prepare queue - // (32 -> 64 at defaults): buffered requests are cheap relative to - // in-flight prepares and drain as prepares commit. - LocalPipeline::with_capacities(prepare_queue_depth, prepare_queue_depth * 2), - VsrRestore { - timers: &timers, - // View and log_view come from the durable superblock when present. - // A present but unreadable superblock already refused boot in - // `recover()`, so no durable record means genuinely absent: a - // fresh node, or one that took writes but never checkpointed or - // changed view. There, inferring the view from the last WAL - // prepare is safe, since the persist-before-send gate guarantees - // this replica never externalized a view beyond what a re-probe - // re-derives, and it re-probes as a backup. - durable_view: recovered_state.map(|state| (state.view, state.log_view)), - view_fallback: last_header.map(|header| header.view), - // Metadata, not a partition group: it has a journal to infer from - // and no second plane to line up with. - seed_view: None, - // Fresh random incarnation each boot, so a StartView addressed to - // a previous incarnation still in flight is ignored - // (`handle_start_view` guard). `| 1` guarantees the non-zero the - // guard treats as set. The deterministic simulator overrides this - // with a seed-derived value bumped per restart. - incarnation: Some(rand::random::() | 1), - join, - }, - ); - consensus.sequencer().set_sequence(restored_op); - // A SOLO replica's durable journal head IS its commit point: quorum is - // 1-of-1, so an entry commits the instant it is durable, and the acks - // the cluster ceremony below would wait on cannot topologically exist. - // The embedded watermark is structurally one op stale (the commit point - // is only ever written down inside the NEXT entry), so trusting it solo - // manufactures an "uncommitted" suffix that provably committed and - // wedges the recovery barrier forever. - let commit_watermark = if replica_count == 1 { - restored_op - } else { - commit_watermark - }; - // The commit point is restored from the WAL's embedded watermark (each - // journaled prepare carries the primary's commit at send time), NOT from - // the journal head: journaled does not imply committed, and claiming - // commit for the un-quorum'd tail both risks split-brain on a later view - // change and starves the tail of re-replication (it would live in no - // pipeline). The suffix `(commit_watermark, restored_op]` is re-pipelined - // below when this replica is the recovered view's primary. - // - // TODO(hubcio): the watermark is a lower bound (the last entry stamps - // the commit point as of its send). Persisting an explicit (view, - // commit_op) watermark on the commit path would tighten recovery and - // allow refusing boot on an excessive gap; a backup that recovered a - // LONGER tail than the cluster's primary still needs uncommitted-suffix - // truncation when conflicting ops arrive (message repair milestone). - consensus.restore_commit_state(commit_watermark, commit_watermark); - if let Some(header) = last_header { - consensus.set_last_prepare_checksum(header.checksum); - consensus.observe_prepare_timestamp(header.timestamp); - } - - // The WAL's tail past the watermark is prepared-but-not-provably-committed - // state. Until the cluster confirms it (re-pipelined below on a resumed - // primary; via StartView adoption + the local commit walk on a rejoined - // backup), serving reads would show pre-restart state that clients already - // saw acked -- gate them on the barrier regardless of role. If the suffix - // never re-commits cluster-wide, the read path fails loud with a retryable - // 503 once the paired deadline expires (`await_recovery_barrier`). - if commit_watermark < restored_op { - consensus.set_recovery_barrier(restored_op); - consensus.set_recovery_deadline(recovery_deadline); - } - - // Re-pipeline the prepared-but-uncommitted suffix so the primary's - // retransmit machinery re-replicates it and quorum can (re-)commit it. - // A backup's suffix stays journal-only: the primary's traffic either - // confirms it (re-forward + re-ack path) or supersedes it. - if consensus.is_primary() - && !consensus.has_ceded_primaryship() - && commit_watermark < restored_op - { - info!( - commit_watermark, - restored_op, "re-pipelining recovered uncommitted metadata suffix" - ); - consensus.with_pipeline_mut(|pipeline| { - #[allow(clippy::cast_possible_truncation)] - for op in (commit_watermark + 1)..=restored_op { - let Some(header) = journal.header(op as usize) else { - warn!( - op, - "recovered journal suffix has a gap; stopping re-pipeline" - ); - break; - }; - let mut entry = PipelineEntry::new(*header); - entry.add_ack(topology.self_replica_id); - pipeline.push(entry); - } - }); - // These went in through `Pipeline::push`, not `push_prepare_entry`, and `init` - // no longer arms the timer: without this the recovered suffix sits in the - // pipeline with nothing driving its retransmit. - consensus.sync_prepare_timeout(); - } - - consensus -} - -/// Recover this partition's persisted segment chain, stamping each segment -/// with the topic's effective segment size (the per-topic value when the -/// topic was created with one, else the shard-wide configured size). -/// -/// The topic's effective `enforce_fsync` goes in for the same reason: it is -/// what tells recovery whether a durable index entry the log cannot back is a -/// benign torn index or previously durable data the log lost. -async fn recover_partition_segments( - config: &ServerConfig, - namespace: IggyNamespace, - runtime_options: TopicRuntimeOptions, - stats: &PartitionStats, -) -> Result, ServerError> { - let stream_id = namespace.stream_id(); - let topic_id = namespace.topic_id(); - let partition_id = namespace.partition_id(); - let segment_size = runtime_options - .segment_size - .unwrap_or_else(|| IggyByteSize::from(iggy_common::DEFAULT_SEGMENT_SIZE)); - let enforce_fsync = runtime_options - .enforce_fsync - .unwrap_or(iggy_common::DEFAULT_ENFORCE_FSYNC); - load_persisted_segments( - config, - stream_id, - topic_id, - partition_id, - segment_size, - enforce_fsync, - stats, - ) - .await - .map_err(|source| { - error!( - stream_id, - topic_id, - partition_id, - error = %source, - "failed to load partition log during server bootstrap" - ); - source - }) -} - -#[allow(clippy::too_many_arguments)] -async fn load_partition( - config: &ServerConfig, - namespace: IggyNamespace, - stats: Arc, - partition_metadata: &Partition, - runtime_options: TopicRuntimeOptions, - cluster_id: u128, - self_replica_id: u8, - replica_count: u8, - bus: Rc, -) -> Result>, ServerError> { - let stream_id = namespace.stream_id(); - let topic_id = namespace.topic_id(); - let partition_id = namespace.partition_id(); - // (view, log_view) come from the group's durable superblock when present; - // a present but unverifiable record already refused boot inside - // `open_partition_superblock`. - let partition_dir = config - .system - .get_partition_path(stream_id, topic_id, partition_id); - let (superblock, recovered_state) = open_partition_superblock( - &partition_dir, - ReplicaIdentity { - cluster: cluster_id, - replica_id: self_replica_id, - replica_count, - }, - ) - .await?; - - // A recovered partition lost its journal state with the process: the - // partition journal is in-memory and segments carry no op numbers, so - // this replica cannot know the group's (op, commit) even when the - // superblock restored its view. In a cluster it boots as a - // quorum-invisible backup and probes for the current view - // (`RequestStartView`): the view's primary answers with a `StartView`, - // journal repair fills the rejoin window, and the commit floor settles - // at the serving peer's retention point. The probe re-broadcasts on its - // timeout, so it needs no live mesh at boot. Single-replica groups - // have no peer to ask and keep the plain init. - let join = if replica_count > 1 { - JoinMode::ProbeAsBackup { - await_state_transfer: false, - } - } else { - JoinMode::Init - }; - // Request queue holds 2x the prepare depth (buffered requests drain as - // prepares commit); depth is the per-partition `[partition]` knob. - let prepare_queue_depth = config.partition.prepare_queue_depth; - let timers = consensus_timers(config); - let consensus = VsrConsensus::restored( - cluster_id, - self_replica_id, - replica_count, - namespace.inner(), - bus, - LocalPipeline::with_capacities(prepare_queue_depth, prepare_queue_depth * 2), - VsrRestore { - timers: &timers, - durable_view: recovered_state - .as_ref() - .map(|state| (state.view, state.log_view)), - view_fallback: None, - seed_view: None, - incarnation: None, - join, - }, - ); - - // No prepare-timestamp floor is restored here: the partition consensus - // journal is non-durable today, so there is no persisted head to observe - // (unlike `restore_metadata_consensus`, which observes its restored head). - // When PartitionJournal becomes durable (the milestone named in the - // multi-shard wiring commit body), observe the restored head and the max - // recovered message timestamp here, or an NTP rewind across a restart could - // regress persisted `base_timestamp`. - - let recovered_segments = - recover_partition_segments(config, namespace, runtime_options, &stats).await?; - - let mut partition = IggyPartition::new(stats.clone(), consensus); - partition.set_runtime_options(runtime_options); - partition.set_superblock(superblock, recovered_state.as_ref()); - // Recovered partitions honor the same config-surfaced ring ceilings as the - // fresh-create path (build_partition_fresh). Retention is already off for - // single-replica groups, so this only sizes the multi-replica ring. - partition.log.journal().inner.set_ring_caps( - config.partition.evicted_ring_capacity, - config.partition.evicted_ring_bytes_max.as_bytes_u64(), - ); - partition.set_partition_dir(partition_dir.clone()); - // Before the hydrate: the durable record is keyed by incarnation, so a - // `purge.gen` left behind by a previous life of this namespace reads 0. - partition.set_created_revision(partition_metadata.created_revision); - partition.hydrate_applied_purge_generation().await?; - hydrate_partition_log( - &mut partition, - &partition_dir, - stream_id, - topic_id, - partition_id, - recovered_segments, - ) - .await?; - - let sized_end = partition - .log - .segments() - .iter() - .filter(|segment| segment.size > IggyByteSize::default()) - .map(|segment| segment.end_offset) - .max(); - // An empty chain whose segment is named for a nonzero offset is the - // shape a state-transfer install (or its converge) plants at the group - // frontier after the origin GC'd everything: the file name carries the - // frontier, and re-minting offsets from 0 here would fork this - // replica's batch stamps from the rest of the group after a restart. - let empty_frontier = partition - .log - .segments() - .iter() - .map(|segment| segment.start_offset) - .max() - .filter(|&start| sized_end.is_none() && start > 0); - let current_offset = sized_end.or_else(|| empty_frontier.map(|start| start - 1)); - partition.created_at = partition_metadata.created_at; - partition.recovered_durable_offset = sized_end; - // The OFFSET COUNTER is restored from that file name (above), but the - // `installed_frontier` CLAIM deliberately is not: the claim says "everything - // below me is represented here", and `converge_to_empty_after_failed_install` - // refuses to make it when staged segments were dropped -- yet a converge - // plants exactly the same empty `{frontier:020}.log` a legitimate empty - // install does, so boot provably cannot tell them apart. Re-deriving it here - // would hand the refused claim back: the repair floor stand-in would accept a - // commit floor over ops this replica holds zero bytes for, and the replica - // would pass the serve gate and offer that emptiness onward, making a peer - // unlink its own chain. Leaving it `None` costs one spurious full - // re-transfer on the legitimate empty-install restart; a false caught-up - // claim is not recoverable. A durable home for the frontier (the partition - // superblock already reserves a field) is what would settle it properly. - let counter = current_offset.unwrap_or(0); - partition.offset.store(counter, Ordering::Release); - partition.dirty_offset.store(counter, Ordering::Relaxed); - partition.should_increment_offset = current_offset.is_some(); - // The durable frontier is a LOWER BOUND on top of what the segments proved: - // it is the only carrier left when the segments that named the frontier are - // gone (an all-GC'd origin's install, a crash inside the swap window), and - // taking the max means real recovered data always wins. - partition.restore_offset_frontier(recovered_state.as_ref()); - let current_offset = partition.offset.load(Ordering::Acquire); - - configure_consumer_offsets(&mut partition, config, namespace, current_offset)?; - ensure_initial_segment(&mut partition, config, stream_id, topic_id, partition_id).await?; - - Ok(partition) -} - -/// Reopen writers over a recovered segment chain. -/// -/// Takes no `&ServerConfig`: every knob it needs is the partition's own -/// resolved topic option now, which is the whole point of the per-topic move. -async fn hydrate_partition_log( - partition: &mut IggyPartition>, - partition_dir: &str, - stream_id: usize, - topic_id: usize, - partition_id: usize, - recovered_segments: Vec, -) -> Result<(), ServerError> { - // The partition's own resolved knobs, not the shard-wide config: a topic - // created with `enforce_fsync` or a per-topic `segment_size` must get them - // on the writers reopened over its recovered chain too, or a restart would - // silently drop back to the node defaults. - let runtime = partition.runtime_options(); - let enforce_fsync = runtime - .enforce_fsync - .unwrap_or(iggy_common::DEFAULT_ENFORCE_FSYNC); - let segment_size = runtime - .segment_size - .unwrap_or_else(|| IggyByteSize::from(iggy_common::DEFAULT_SEGMENT_SIZE)); - let preallocate_segments = runtime - .preallocate_segments - .unwrap_or(iggy_common::DEFAULT_PREALLOCATE_SEGMENTS); - for RecoveredSegment { segment, storage } in recovered_segments { - partition - .log - .add_persisted_segment(segment, storage, None, None); - } - - if let Some(active_index) = partition.log.segments().len().checked_sub(1) { - let storage = &partition.log.storages()[active_index]; - if let ( - Some(messages_reader), - Some(index_reader), - Some(storage_messages_writer), - Some(storage_index_writer), - ) = ( - storage.messages_reader.as_ref(), - storage.index_reader.as_ref(), - storage.messages_writer.as_ref(), - storage.index_writer.as_ref(), - ) { - let index_path = index_reader.path(); - let start_offset = partition.log.segments()[active_index].start_offset; - // Share the storage's size counters: they are the write cursors. - // A private counter would let the append position diverge from the - // segment bookkeeping that index entries and poll bounds rely on. - let messages_size_counter = storage_messages_writer.size_counter(); - let index_size_counter = storage_index_writer.size_counter(); - partition.log.messages_writers_mut()[active_index] = Some(Rc::new( - MessagesWriter::new( - &messages_reader.path(), - messages_size_counter, - enforce_fsync, - true, - preallocate_segments.then_some(segment_size), - ) - .await - .map_err(|source| { - error!( - stream_id, - topic_id, - partition_id, - path = %messages_reader.path(), - error = %source, - "failed to initialize persisted messages writer" - ); - hydrate_reopen_error( - source, - partition_dir, - stream_id, - topic_id, - partition_id, - start_offset, - ) - })?, - )); - partition.log.index_writers_mut()[active_index] = Some(Rc::new( - IggyIndexWriter::new(&index_path, index_size_counter, enforce_fsync, true) - .await - .map_err(|source| { - error!( - stream_id, - topic_id, - partition_id, - path = %index_path, - error = %source, - "failed to initialize persisted sparse index writer" - ); - hydrate_reopen_error( - source, - partition_dir, - stream_id, - topic_id, - partition_id, - start_offset, - ) - })?, - )); - } - } - - Ok(()) -} - -/// Routes a hydrate-reopen writer failure. The seed-vs-stat divergence guard -/// (`SegmentSizeMismatchAtOpen`) is a post-condition assertion on recovery's -/// own truncation: pass C truncates every file to its recovered size before -/// storage and writers reopen it, so the guard can only fire if the -/// filesystem lied about a length or a change broke that truncate-then-open -/// contract. Kept as defense-in-depth and routed as a structural refusal -/// because a retried boot cannot help. Every other failure here (open, stat, -/// sync) is transient I/O and stays node-fatal: a retried boot can still -/// serve the partition, while fencing would quarantine healthy data (and at -/// `replica_count = 1` tombstone the partition outright). -fn hydrate_reopen_error( - source: IggyError, - partition_dir: &str, - stream_id: usize, - topic_id: usize, - partition_id: usize, - start_offset: u64, -) -> ServerError { - match source { - IggyError::SegmentSizeMismatchAtOpen(on_disk_bytes, expected_bytes) => { - ServerError::PartitionRecoveryRefused { - dir: PathBuf::from(partition_dir), - stream_id, - topic_id, - partition_id, - reason: PartitionRecoveryRefusal::StorageSizeMismatch { - start_offset, - on_disk_bytes, - expected_bytes, - }, - } - } - transient => transient.into(), - } -} - -fn resolve_tcp_topology( - config: &ServerConfig, - current_replica_id: Option, -) -> Result { - let default_client_addr = parse_socket_addr("tcp.address", &config.tcp.address)?; - let default_ws_addr = resolve_optional_listener_addr( - config.websocket.enabled, - "websocket.address", - &config.websocket.address, - )?; - let default_quic_addr = - resolve_optional_listener_addr(config.quic.enabled, "quic.address", &config.quic.address)?; - let default_http_addr = - resolve_optional_listener_addr(config.http.enabled, "http.address", &config.http.address)?; - if !config.cluster.enabled { - if let Some(replica_id) = current_replica_id - && replica_id != SHARD_REPLICA_ID - { - return Err(ServerError::ReplicaIdRequiresCluster { - supplied: replica_id, - default: SHARD_REPLICA_ID, - }); - } - return Ok(TcpTopology { - cluster_id: auth::cluster_domain_id(&config.cluster.name), - // Keep parity with the current server binary and the integration - // harness: `--replica-id 0` may be passed unconditionally in - // single-node mode; any other id is rejected above so the WAL - // cannot commit under an identity that will later disagree with - // a cluster.nodes[] entry. - self_replica_id: SHARD_REPLICA_ID, - replica_count: 1, - client_listen_addr: default_client_addr, - replica_listen_addr: Some(SocketAddr::new(default_client_addr.ip(), 0)), - ws_listen_addr: default_ws_addr, - quic_listen_addr: default_quic_addr, - http_listen_addr: default_http_addr, - tcp_tls_listen_addr: config.tcp.tls.enabled.then_some(default_client_addr), - peers: Vec::new(), - }); - } - - let self_replica_id = current_replica_id.ok_or(ServerError::MissingReplicaId)?; - - let self_node = config - .cluster - .nodes - .iter() - .find(|node| node.replica_id == self_replica_id) - .ok_or(ServerError::ClusterNodeNotFound { - replica_id: self_replica_id, - })?; - let replica_count = u8::try_from(config.cluster.nodes.len()).map_err(|_| { - ServerError::ClusterReplicaCountTooLarge { - count: config.cluster.nodes.len(), - } - })?; - let ClusterClientAddrs { - client: client_listen_addr, - ws: ws_listen_addr, - quic: quic_listen_addr, - http: http_listen_addr, - } = resolve_cluster_client_addrs( - self_node, - default_client_addr, - default_ws_addr, - default_quic_addr, - default_http_addr, - )?; - let replica_port = self_node - .ports - .tcp_replica - .ok_or(ServerError::ClusterPortMissing { - transport: "tcp_replica", - replica_id: self_node.replica_id, - })?; - let replica_listen_addr = Some(socket_addr_from_parts( - "cluster.nodes[*].ports.tcp_replica", - &self_node.ip, - replica_port, - )?); - let peers = resolve_cluster_replica_peers(&config.cluster.nodes, self_replica_id)?; - - Ok(TcpTopology { - cluster_id: auth::cluster_domain_id(&config.cluster.name), - self_replica_id, - replica_count, - client_listen_addr, - replica_listen_addr, - ws_listen_addr, - quic_listen_addr, - http_listen_addr, - tcp_tls_listen_addr: config.tcp.tls.enabled.then_some(client_listen_addr), - peers, - }) -} - -fn resolve_optional_listener_addr( - enabled: bool, - context: &'static str, - address: &str, -) -> Result, ServerError> { - if enabled { - return Ok(Some(parse_socket_addr(context, address)?)); - } - Ok(None) -} - -/// Client-facing listener addresses resolved for this cluster node. Each port -/// comes from the node's roster entry; there is no fallback to the top-level -/// listener port, an enabled transport without a roster port refuses to boot. -/// Every transport keeps the bind interface from its own `address` config: the -/// roster ip is advertised, not bound. -struct ClusterClientAddrs { - client: SocketAddr, - ws: Option, - quic: Option, - http: Option, -} - -fn resolve_cluster_client_addrs( - self_node: &configs::cluster::ClusterNodeConfig, - default_tcp_addr: SocketAddr, - default_ws_addr: Option, - default_quic_addr: Option, - default_http_addr: Option, -) -> Result { - let client_port = self_node.ports.tcp.ok_or(ServerError::ClusterPortMissing { - transport: "tcp", - replica_id: self_node.replica_id, - })?; - let client = - merge_roster_port_with_bind_ip("tcp", &self_node.ip, default_tcp_addr, client_port); - let ws = resolve_cluster_optional_addr(self_node, "websocket", default_ws_addr, |ports| { - ports.websocket - })?; - let quic = - resolve_cluster_optional_addr(self_node, "quic", default_quic_addr, |ports| ports.quic)?; - let http = - resolve_cluster_optional_addr(self_node, "http", default_http_addr, |ports| ports.http)?; - Ok(ClusterClientAddrs { - client, - ws, - quic, - http, - }) -} - -fn resolve_cluster_optional_addr( - self_node: &configs::cluster::ClusterNodeConfig, - transport: &'static str, - default_addr: Option, - port_selector: impl Fn(&configs::cluster::TransportPorts) -> Option, -) -> Result, ServerError> { - let Some(default_addr) = default_addr else { - return Ok(None); - }; - // No fallback to the top-level port: two same-host nodes leaving the same - // transport port unset would race for one socket. Either the roster is - // explicit or the server refuses to boot. - let port = port_selector(&self_node.ports).ok_or(ServerError::ClusterPortMissing { - transport, - replica_id: self_node.replica_id, - })?; - Ok(Some(merge_roster_port_with_bind_ip( - transport, - &self_node.ip, - default_addr, - port, - ))) -} - -/// Combine the roster-supplied `port` with the bind interface the transport's -/// own `address` config asked for. -/// -/// The roster ip is what the cluster advertises (metadata, follower-to-primary -/// HTTP forwarding targets); the transport's own `address` decides the bind -/// interface. Merging keeps a loopback-only `127.0.0.1` private and a -/// `0.0.0.0` wide in cluster mode instead of silently rebinding to the roster -/// interface, which would strand every co-located dialer (sidecars, health -/// probes, on-host consumers) on `ECONNREFUSED`. -fn merge_roster_port_with_bind_ip( - transport: &'static str, - roster_ip: &str, - bind_addr: SocketAddr, - port: u16, -) -> SocketAddr { - let listen_addr = SocketAddr::new(bind_addr.ip(), port); - if roster_ip_unreachable_from_bind_addr(roster_ip, listen_addr) { - warn!( - "{transport} listener binds {listen_addr} but the roster advertises {roster_ip}:{port}; \ - peers and clients dialing the advertised endpoint may not reach this node" - ); - } - listen_addr -} - -/// The client-facing listeners paired with the config key naming their bind -/// address, in the order [`ServerConfig::client_listeners`] derives the -/// published client-facing address from them. `None` marks a listener that is -/// switched off and therefore binds nothing. -fn client_listeners( - topology: &TcpTopology, - config: &ServerConfig, -) -> [(&'static str, Option); 4] { - [ - ( - "tcp.address", - config.tcp.enabled.then_some(topology.client_listen_addr), - ), - ("websocket.address", topology.ws_listen_addr), - ("quic.address", topology.quic_listen_addr), - ("http.address", topology.http_listen_addr), - ] -} - -/// The bind interface the published client-facing address names when none is -/// declared: the first enabled listener's, the same one boot validation gated -/// its wildcard refusal on. With every client listener off nothing dials this -/// node, so the tcp bind address stands in for an answer no client reads. -fn derived_bind_ip(topology: &TcpTopology, config: &ServerConfig) -> IpAddr { - client_listeners(topology, config) - .into_iter() - .find_map(|(_, listen_addr)| listen_addr) - .unwrap_or(topology.client_listen_addr) - .ip() -} - -/// Whether the address cluster metadata publishes for this node misses one of -/// its own listeners. Only a derived address is judged: it names one -/// listener's bind interface, so a listener on a different one is unreachable -/// at the published address. A declared `node.advertised_address` is -/// deliberate (NAT, a public name) and says nothing about which local -/// interface serves a transport, so it stays quiet. -fn derived_address_misses_listener( - declared: Option<&str>, - self_advertised: &str, - listen_addr: SocketAddr, -) -> bool { - declared.is_none() && roster_ip_unreachable_from_bind_addr(self_advertised, listen_addr) -} - -/// Whether a dialer aiming at the advertised roster ip misses `listen_addr`. An -/// unspecified bind covers every interface, and a roster ip that parses as -/// neither IPv4 nor IPv6 (a DNS name, say) can resolve to the bound interface, -/// so both cases stay quiet. Both sides reduce to the canonical form first, so -/// the v4-mapped wildcard (`[::ffff:0.0.0.0]`, which a dual-stack host binds as -/// `0.0.0.0`) stays quiet as well and `10.0.0.5` matches `::ffff:10.0.0.5`. -fn roster_ip_unreachable_from_bind_addr(roster_ip: &str, listen_addr: SocketAddr) -> bool { - let bind_ip = listen_addr.ip().to_canonical(); - !bind_ip.is_unspecified() - && roster_ip - .parse::() - .is_ok_and(|parsed| parsed.to_canonical() != bind_ip) -} - -fn wildcard_listener_under_loopback_address( - declared: Option<&str>, - self_advertised: &str, - listen_addr: SocketAddr, -) -> bool { - declared.is_none() - && listen_addr.ip().to_canonical().is_unspecified() - && self_advertised - .parse::() - .is_ok_and(|address| address.to_canonical().is_loopback()) -} - -fn resolve_cluster_replica_peers( - nodes: &[configs::cluster::ClusterNodeConfig], - self_replica_id: u8, -) -> Result, ServerError> { - let mut peers = Vec::with_capacity(nodes.len().saturating_sub(1)); - for node in nodes { - if node.replica_id == self_replica_id { - continue; - } - let replica_port = node - .ports - .tcp_replica - .ok_or(ServerError::ClusterPortMissing { - transport: "tcp_replica", - replica_id: node.replica_id, - })?; - peers.push(( - node.replica_id, - socket_addr_from_parts("cluster.nodes[*].ports.tcp_replica", &node.ip, replica_port)?, - )); - } - Ok(peers) -} - -async fn start_tcp_runtime( - shard: &Rc, - config: &ServerConfig, - topology: &TcpTopology, - accepted_replica: AcceptedReplicaFn, - dialed_replica: DialedReplicaFn, - accepted_clients: LocalClientAcceptFns, - shard_metrics_all: &[ShardMetrics], -) -> Result<(), ServerError> { - if config.tcp.enabled && !config.tcp.tls.enabled { - start_via_replica_io( - shard, - config, - topology, - accepted_replica, - dialed_replica, - accepted_clients, - ) - .await?; - } else { - start_manual_runtime( - shard, - config, - topology, - accepted_replica, - dialed_replica, - accepted_clients, - ) - .await?; - } - - // Cluster metadata carries one host for all four transports, so a listener - // the derived host does not reach is unreachable at it. Only a derived - // address is judged, and never against the listener it was derived from. - let declared = config.node.advertised_address.as_deref(); - let self_advertised = self_advertised_address(declared, derived_bind_ip(topology, config)); - // A roster entry answers this per node in cluster mode, so the derived - // address is never served and none of these listeners are judged against it. - let listeners = client_listeners(topology, config); - if !config.cluster.enabled - && let Some(derived_from) = listeners - .iter() - .find_map(|(key, listen_addr)| listen_addr.map(|_| *key)) - { - for (key, listen_addr) in listeners { - let Some(listen_addr) = listen_addr.filter(|_| key != derived_from) else { - continue; - }; - if derived_address_misses_listener(declared, &self_advertised, listen_addr) { - warn!( - "{key} binds {listen_addr} but cluster metadata publishes {self_advertised}, \ - derived from {derived_from}; a client reading that metadata would not reach \ - this listener. Set node.advertised_address to the address clients dial." - ); - } else if wildcard_listener_under_loopback_address( - declared, - &self_advertised, - listen_addr, - ) { - warn!( - "{key} binds the wildcard {listen_addr} but cluster metadata publishes the \ - loopback {self_advertised}, derived from {derived_from}; a client reaching \ - this listener from another host is told an address that points back at \ - itself. Set node.advertised_address to the address clients dial." - ); - } - } - } - - // HTTP is served over TCP but sits outside the replica_io / manual client - // reactor, so it binds independently. Shard-0 gating comes from the sole - // caller of this function. - if let Some(http_addr) = topology.http_listen_addr { - let self_ports = configs::cluster::TransportPorts { - tcp: config - .tcp - .enabled - .then(|| topology.client_listen_addr.port()), - quic: topology.quic_listen_addr.map(|addr| addr.port()), - websocket: topology.ws_listen_addr.map(|addr| addr.port()), - ..Default::default() - }; - http::start( - shard, - http_addr, - &config.http, - config.metadata.clients_table_max, - config.personal_access_token.max_tokens_per_user, - &config.cluster, - Arc::clone(&config.system), - &self_advertised, - &self_ports, - shard_metrics_all, - )?; - } - - Ok(()) -} - -// ws/wss bindings intentionally mirror the transport names (same convention as -// `replica_io::start_on_shard_zero`). -#[allow(clippy::similar_names)] -async fn start_via_replica_io( - shard: &Rc, - config: &ServerConfig, - topology: &TcpTopology, - accepted_replica: AcceptedReplicaFn, - dialed_replica: DialedReplicaFn, - accepted_clients: LocalClientAcceptFns, -) -> Result<(), ServerError> { - let replica_addr = topology - .replica_listen_addr - .expect("topology must include replica listener address"); - let quic_credentials = topology - .quic_listen_addr - .is_some() - .then(|| load_quic_server_credentials(config)) - .transpose()?; - let tcp_tls_credentials = topology - .tcp_tls_listen_addr - .is_some() - .then(|| load_tcp_tls_server_credentials(config)) - .transpose()?; - // `websocket.tls.enabled` upgrades the websocket address to a WSS - // listener; the plain-WS listener must NOT also bind it (one port, one - // handshake kind -- a plain upgrade parser fed a TLS ClientHello rejects - // every connection with an httparse error). - let wss_enabled = config.websocket.tls.enabled; - let ws_listen_addr = (!wss_enabled).then_some(topology.ws_listen_addr).flatten(); - let wss_listen_addr = wss_enabled.then_some(topology.ws_listen_addr).flatten(); - let wss_credentials = wss_listen_addr - .is_some() - .then(|| load_wss_server_credentials(config)) - .transpose()?; - - let LocalClientAcceptFns { - tcp, - ws, - quic, - tcp_tls, - wss, - } = accepted_clients; - - let bound = replica_io::start_on_shard_zero( - &shard.bus, - replica_addr, - topology.client_listen_addr, - ws_listen_addr, - topology.quic_listen_addr, - quic_credentials, - topology.tcp_tls_listen_addr, - tcp_tls_credentials, - wss_listen_addr, - wss_credentials, - topology.self_replica_id, - topology.peers.clone(), - accepted_replica, - dialed_replica, - tcp, - ws_listen_addr.map(|_| ws), - topology.quic_listen_addr.map(|_| quic), - topology.tcp_tls_listen_addr.map(|_| tcp_tls), - wss_listen_addr.map(|_| wss), - shard.bus.config().reconnect_period, - ) - .await - .map_err(|source| { - error!( - replica_addr = %replica_addr, - client_addr = %topology.client_listen_addr, - error = %source, - "failed to start server listeners via replica_io" - ); - source - })?; - let Some(bound) = bound else { - return Ok(()); - }; - - write_current_config( - config, - Some(topology.self_replica_id), - Some(bound.client), - config.cluster.enabled.then_some(bound.replica), - bound.tcp_tls, - bound.quic, - // The WSS listener occupies the configured websocket address slot. - bound.wss.or(bound.ws), - ) - .await?; - if config.cluster.enabled { - info!( - shard = shard.id, - replica = %bound.replica, - tcp = %bound.client, - tcp_tls = ?bound.tcp_tls, - ws = ?bound.ws, - quic = ?bound.quic, - "server listeners started" - ); - } else { - info!( - shard = shard.id, - tcp = %bound.client, - tcp_tls = ?bound.tcp_tls, - ws = ?bound.ws, - quic = ?bound.quic, - "server client listeners started" - ); - } - - Ok(()) -} - -async fn start_manual_runtime( - shard: &Rc, - config: &ServerConfig, - topology: &TcpTopology, - accepted_replica: AcceptedReplicaFn, - dialed_replica: DialedReplicaFn, - accepted_clients: LocalClientAcceptFns, -) -> Result<(), ServerError> { - let bound_replica = if config.cluster.enabled { - let replica_addr = topology - .replica_listen_addr - .expect("cluster-enabled topology must include replica listener address"); - let (replica_listener, bound_addr) = - replica_listener::bind(replica_addr) - .await - .map_err(|source| { - error!( - replica_addr = %replica_addr, - error = %source, - "failed to bind replica listener" - ); - source - })?; - let token = shard.bus.token(); - let replica_handle = compio::runtime::spawn(async move { - replica_listener::run(replica_listener, token, accepted_replica).await; - }); - shard.bus.track_background(replica_handle); - connector::start( - &shard.bus, - topology.self_replica_id, - topology.peers.clone(), - dialed_replica, - shard.bus.config().reconnect_period, - ) - .await; - Some(bound_addr) - } else { - None - }; - - let bound_clients = start_client_listeners(shard, config, topology, &accepted_clients)?; - write_current_config( - config, - Some(topology.self_replica_id), - bound_clients.tcp, - bound_replica, - bound_clients.tcp_tls, - bound_clients.quic, - bound_clients.ws, - ) - .await?; - - if config.cluster.enabled { - info!( - shard = shard.id, - replica = ?bound_replica, - tcp = ?bound_clients.tcp, - tcp_tls = ?bound_clients.tcp_tls, - ws = ?bound_clients.ws, - quic = ?bound_clients.quic, - "server listeners started" - ); - } else { - info!( - shard = shard.id, - tcp = ?bound_clients.tcp, - tcp_tls = ?bound_clients.tcp_tls, - ws = ?bound_clients.ws, - quic = ?bound_clients.quic, - "server client listeners started" - ); - } - - Ok(()) -} - -fn ensure_default_root_user(mux_stm: &ServerMuxStateMachine) { - if !mux_stm.users().read(|users| users.items.is_empty()) { - return; - } - - let (username, password_hash) = create_root_credentials(); - mux_stm.users().ensure_root_user(&username, &password_hash); -} - -/// Apply `--with-default-root-credentials`. -/// -/// Fills in whichever of [`IGGY_ROOT_USERNAME_ENV`] / -/// [`IGGY_ROOT_PASSWORD_ENV`] the operator did not export, so the flag is -/// exactly the sugar for setting both by hand and the environment keeps -/// winning over it. -/// -/// # Safety -/// -/// Mutates the process environment, so the caller must still be -/// single-threaded. -pub unsafe fn apply_default_root_credentials(enabled: bool) { - if !enabled { - return; - } - - let username_set = env::var(IGGY_ROOT_USERNAME_ENV).is_ok(); - let password_set = env::var(IGGY_ROOT_PASSWORD_ENV).is_ok(); - if username_set && password_set { - warn!( - "--with-default-root-credentials ignored: {IGGY_ROOT_USERNAME_ENV} and \ - {IGGY_ROOT_PASSWORD_ENV} are already set" - ); - return; - } - - // SAFETY: single-threaded caller, per this function's contract. - unsafe { - if !username_set { - env::set_var(IGGY_ROOT_USERNAME_ENV, DEFAULT_ROOT_USERNAME); - } - if !password_set { - env::set_var(IGGY_ROOT_PASSWORD_ENV, DEFAULT_ROOT_PASSWORD); - } - } - warn!( - "--with-default-root-credentials: a newly created root user will use the \ - well-known development credentials; INSECURE outside development" - ); -} - -/// Resolve the root user credentials from `IGGY_ROOT_USERNAME` / -/// `IGGY_ROOT_PASSWORD`, falling back to the default username with a -/// generated password. -/// -/// Returns `(username, password_hash)`; the plaintext password never -/// leaves this function. -fn create_root_credentials() -> (String, String) { - if let Some((username, password)) = root_credentials_from_env() { - info!("Using the custom root user credentials."); - return (username, crypto::hash_password(&password)); - } - - info!("Using the default root user credentials..."); - let password = crypto::generate_secret(20..40); - // Through tracing, not stdout: this is the only time the operator can read - // the password, so it has to reach the log file too. - warn!("Generated root user password: {password}"); - ( - DEFAULT_ROOT_USERNAME.to_string(), - crypto::hash_password(&password), - ) -} - -/// The credentials the operator supplied, `None` when neither variable is -/// set. A half-set pair never reaches here: [`validate_root_credentials`] -/// rejects it at boot. -fn root_credentials_from_env() -> Option<(String, String)> { - match ( - env::var(IGGY_ROOT_USERNAME_ENV), - env::var(IGGY_ROOT_PASSWORD_ENV), - ) { - (Ok(username), Ok(password)) => Some((username, password)), - _ => None, - } -} - -/// Reject root-credential misconfiguration before any shard thread exists. -/// -/// Shard 0 seeds the root user from inside `recover`'s baseline closure, -/// which cannot fail, so every operator-facing check has to run here or it -/// would have to panic a shard thread instead. -fn validate_root_credentials_env(config: &ServerConfig) -> Result<(), ServerError> { - // `recover` creates the metadata directory, so its absence is what tells a - // first cluster boot (root must come out identical on every replica, hence - // explicit credentials) apart from a restart that recovers the root user it - // already stored. `--fresh` has already wiped by this point, so a wiped - // replica is correctly treated as a first boot. - let fresh_cluster = config.cluster.enabled - && !Path::new(&config.system.path) - .join(metadata::impls::METADATA_DIR) - .exists(); - - validate_root_credentials( - fresh_cluster, - env::var(IGGY_ROOT_USERNAME_ENV).ok().as_deref(), - env::var(IGGY_ROOT_PASSWORD_ENV).ok().as_deref(), - ) -} - -fn validate_root_credentials( - explicit_required: bool, - username: Option<&str>, - password: Option<&str>, -) -> Result<(), ServerError> { - match (username, password) { - (Some(username), Some(password)) => { - validate_credential_length( - IGGY_ROOT_USERNAME_ENV, - username, - MIN_USERNAME_LENGTH, - MAX_USERNAME_LENGTH, - )?; - validate_credential_length( - IGGY_ROOT_PASSWORD_ENV, - password, - MIN_PASSWORD_LENGTH, - MAX_PASSWORD_LENGTH, - ) - } - (Some(_), None) => Err(ServerError::RootCredentialsIncomplete { - provided_env: IGGY_ROOT_USERNAME_ENV, - missing_env: IGGY_ROOT_PASSWORD_ENV, - }), - (None, Some(_)) => Err(ServerError::RootCredentialsIncomplete { - provided_env: IGGY_ROOT_PASSWORD_ENV, - missing_env: IGGY_ROOT_USERNAME_ENV, - }), - (None, None) if explicit_required => Err(ServerError::ClusterRootCredentialsRequired { - username_env: IGGY_ROOT_USERNAME_ENV, - password_env: IGGY_ROOT_PASSWORD_ENV, - }), - (None, None) => Ok(()), - } -} - -fn validate_credential_length( - env_name: &'static str, - value: &str, - min: usize, - max: usize, -) -> Result<(), ServerError> { - if (min..=max).contains(&value.len()) { - Ok(()) - } else { - Err(ServerError::RootCredentialLength { - env_name, - length: value.len(), - min, - max, - }) - } -} - -/// Replica delegation callbacks for shard 0's listener and connector. -/// -/// Inbound: acquire a slot in the shard-0-global in-flight handshake cap -/// (drop the connection when full), then blind-delegate the raw fd -/// through the coordinator's round-robin. The fd lands on the target -/// shard's inbox as a [`shard::LifecycleFrame::ReplicaInboundSetup`] -/// frame; the owning shard runs the acceptor handshake and acks the -/// slot back. A failed delegation releases the slot immediately. -/// -/// Outbound: delegate the dialed fd as -/// [`shard::LifecycleFrame::ReplicaOutboundSetup`] and mark the peer -/// dial-pending so the reconnect sweep skips it until the owning -/// shard's handshake outcome arrives (or the entry expires). -fn make_replica_delegation_fns( - coord: Rc, - bus: &Rc, -) -> (AcceptedReplicaFn, DialedReplicaFn) { - let inbound_bus = Rc::clone(bus); - let inbound_coord = Rc::clone(&coord); - let accepted: AcceptedReplicaFn = Rc::new(move |stream| { - let Some(slot) = inbound_bus.try_acquire_replica_handshake_slot() else { - warn!( - cap = MAX_INFLIGHT_REPLICA_HANDSHAKES, - "replica handshake in-flight cap reached; dropping inbound" - ); - return; - }; - match inbound_coord.delegate_replica_inbound(stream, slot) { - Ok(target) => { - info!(slot, target, "inbound replica connection delegated"); - } - Err(error) => { - inbound_bus.release_replica_handshake_slot(slot); - warn!( - error = ?error, - "delegate_replica_inbound failed; dropping inbound replica connection" - ); - } - } - }); - - let outbound_bus = Rc::clone(bus); - let dialed: DialedReplicaFn = - Rc::new( - move |stream, peer_id| match coord.delegate_replica_outbound(stream, peer_id) { - Ok(target) => { - outbound_bus.mark_dial_pending(peer_id); - info!(peer_id, target, "outbound replica connection delegated"); - } - Err(error) => { - warn!( - peer_id, - error = ?error, - "delegate_replica_outbound failed; dropping dialed replica connection" - ); - } - }, - ); - - (accepted, dialed) -} - -/// Shard-0 client accept callbacks. TCP and WS clients are delegated via -/// the coordinator (round-robin to peer shards); QUIC and TCP-TLS install -/// locally on shard 0 because their per-connection state is not portable -/// across shards (`compio_quic` endpoint binds one UDP socket; rustls TLS -/// state ties to the post-handshake reactor). -// ws/wss bindings intentionally mirror the transport names (same convention as -// `replica_io::start_on_shard_zero`). -#[allow(clippy::similar_names)] -fn make_shard_zero_client_accept_fns( - coord: Rc, - bus: &Rc, - on_request: RequestHandler, -) -> LocalClientAcceptFns { - let quic_bus = Rc::clone(bus); - let tcp_tls_bus = Rc::clone(bus); - let wss_bus = Rc::clone(bus); - let quic_request = on_request.clone(); - let wss_request = on_request.clone(); - let tcp_tls_request = on_request; - - let tcp_coord = Rc::clone(&coord); - let tcp = Rc::new(move |stream| match tcp_coord.delegate_client(stream) { - Ok(client_id) => info!(client_id, "TCP client delegated"), - Err(error) => warn!(error = ?error, "delegate_client failed; dropping TCP client"), - }); - - let ws_coord = Rc::clone(&coord); - let ws = Rc::new(move |stream| match ws_coord.delegate_ws_client(stream) { - Ok(client_id) => info!(client_id, "WS client delegated"), - Err(error) => warn!(error = ?error, "delegate_ws_client failed; dropping WS client"), - }); - - // QUIC and TCP-TLS terminate locally on shard 0 but mint their client - // ids through the coordinator's `client_seq`, the same counter the - // delegated TCP/WS path uses. A separate counter here would let a - // shard-0-local id collide with a delegated id that round-robined to - // shard 0 (both encode target shard 0) in shard 0's connection - // registry. - let quic_coord = Rc::clone(&coord); - let quic = Rc::new(move |accepted: message_bus::AcceptedQuicConn| { - let meta = mint_client_meta(&quic_coord, accepted.peer_addr(), ClientTransportKind::Quic); - installer::install_client_quic(&quic_bus, meta, accepted, quic_request.clone()); - }); - - let tcp_tls_coord = Rc::clone(&coord); - let tcp_tls = Rc::new(move |stream, tls_config| { - let Some(meta) = - client_meta_from_stream(&stream, &tcp_tls_coord, ClientTransportKind::TcpTls) - else { - return; - }; - installer::install_client_tcp_tls( - &tcp_tls_bus, - meta, - stream, - tls_config, - tcp_tls_request.clone(), - ); - }); - - // WSS terminates locally on shard 0 like TCP-TLS (rustls state is not - // serialisable across the delegate path), minting ids through the same - // coordinator counter. - let wss_coord = coord; - let wss = Rc::new(move |stream, tls_config| { - let Some(meta) = client_meta_from_stream(&stream, &wss_coord, ClientTransportKind::Wss) - else { - return; - }; - installer::install_client_wss(&wss_bus, meta, stream, tls_config, wss_request.clone()); - }); - - LocalClientAcceptFns { - tcp, - ws, - quic, - tcp_tls, - wss, - } -} - -fn client_meta_from_stream( - stream: &compio::net::TcpStream, - coord: &shard::coordinator::ShardZeroCoordinator, - transport: ClientTransportKind, -) -> Option { - let peer_addr = match stream.peer_addr() { - Ok(peer_addr) => peer_addr, - Err(error) => { - warn!(error = %error, "dropping accepted client with unknown peer address"); - return None; - } - }; - Some(mint_client_meta(coord, peer_addr, transport)) -} - -fn mint_client_meta( - coord: &shard::coordinator::ShardZeroCoordinator, - peer_addr: SocketAddr, - transport: ClientTransportKind, -) -> ClientConnMeta { - ClientConnMeta::new(coord.mint_shard_zero_client_id(), peer_addr, transport) -} - -fn start_client_listeners( - shard: &Rc, - config: &ServerConfig, - topology: &TcpTopology, - accepted_clients: &LocalClientAcceptFns, -) -> Result { - let mut bound = BoundClientListeners::default(); - - if config.tcp.enabled && !config.tcp.tls.enabled { - let (listener, bound_addr) = client_listener::tcp::bind(topology.client_listen_addr) - .map_err(|source| { - error!( - addr = %topology.client_listen_addr, - error = %source, - "failed to bind TCP client listener" - ); - source - })?; - let token = shard.bus.token(); - let accepted_client = accepted_clients.tcp.clone(); - let client_handle = compio::runtime::spawn(async move { - client_listener::tcp::run(listener, token, accepted_client).await; - }); - shard.bus.track_background(client_handle); - bound.tcp = Some(bound_addr); - } - - if let Some(ws_addr) = topology.ws_listen_addr { - bound.ws = Some(start_websocket_listener( - shard, - config, - ws_addr, - accepted_clients, - )?); - } - - if let Some(quic_addr) = topology.quic_listen_addr { - install_default_crypto_provider(); - let credentials = load_quic_server_credentials(config)?; - let server_config = server_config_with_cert( - credentials.cert_chain, - credentials.key_der, - &shard.bus.config().quic, - ) - .map_err(|e| { - let source = - iggy_common::IggyError::IoError(format!("QUIC server config build failed: {e}")); - error!(addr = %quic_addr, error = %source, "failed to build QUIC server config"); - source - })?; - let (endpoint, bound_addr) = client_listener::quic::bind(quic_addr, server_config) - .map_err(|source| { - error!(addr = %quic_addr, error = %source, "failed to bind QUIC listener"); - source - })?; - let token = shard.bus.token(); - let handshake_grace = shard.bus.config().handshake_grace; - let accepted_quic = accepted_clients.quic.clone(); - let quic_handle = compio::runtime::spawn(async move { - client_listener::quic::run(endpoint, token, accepted_quic, handshake_grace).await; - }); - shard.bus.track_background(quic_handle); - bound.quic = Some(bound_addr); - } - - if config.tcp.enabled && config.tcp.tls.enabled { - let credentials = load_tcp_tls_server_credentials(config)?; - let (listener, tls_config, bound_addr) = - client_listener::tcp_tls::bind(topology.client_listen_addr, credentials).map_err( - |source| { - error!( - addr = %topology.client_listen_addr, - error = %source, - "failed to bind TCP TLS listener" - ); - source - }, - )?; - let token = shard.bus.token(); - let accepted_tls = accepted_clients.tcp_tls.clone(); - let tls_handle = compio::runtime::spawn(async move { - client_listener::tcp_tls::run(listener, tls_config, token, accepted_tls).await; - }); - shard.bus.track_background(tls_handle); - bound.tcp_tls = Some(bound_addr); - } - - Ok(bound) -} - -/// Build the replica auth context from cluster config. Returns `None` when the -/// cluster or replica auth is disabled, keeping the handshake in legacy mode. -/// Only the derived MAC keys are carried onward in [`ReplicaAuth`]; the raw -/// secrets (masked in config logs via `config_env(secret)`) are read here only -/// to derive them. A non-empty `previous_shared_secret` opens the verify-only -/// rotation acceptance window (see the [`ReplicaAuth`] rustdoc for the rolling -/// rotation procedure). `ClusterConfig::validate` guarantees a non-empty -/// secret whenever both `cluster.enabled` and `cluster.auth.enabled` are set -/// (validate early-returns `Ok` while `cluster.enabled` is false). -fn load_replica_auth(config: &ServerConfig) -> Option { - if !config.cluster.enabled || !config.cluster.auth.enabled { - return None; - } - let auth = ReplicaAuth::new(config.cluster.auth.shared_secret.as_bytes()); - let previous_shared_secret = &config.cluster.auth.previous_shared_secret; - if previous_shared_secret.is_empty() { - return Some(auth); - } - Some(auth.with_previous_secret(previous_shared_secret.as_bytes())) -} - -/// Build the replica TLS context from cluster config. Returns `None` when -/// the cluster or replica TLS is disabled. Every shard calls this once at -/// boot: CA mode re-reads the same PEM files per shard; self-signed mode -/// mints a per-shard throwaway certificate. Neither mode carries client -/// certificates, so TLS authenticates the acceptor only; peer -/// authentication comes from the PSK handshake (`ClusterConfig::validate` -/// enforces `cluster.auth.enabled` whenever `cluster.tls.enabled`). -/// -/// Both rustls configs are TLS 1.3 only with the [`REPLICA_ALPN`] -/// protocol pinned. The dialer's SNI / certificate-verify name for each -/// peer is the roster entry's `ip` field (a hostname or IP literal, the -/// same string the connector dials). -fn load_replica_tls_ctx( - config: &ServerConfig, - topology: &TcpTopology, -) -> Result, ServerError> { - let tls = &config.cluster.tls; - if !config.cluster.enabled || !tls.enabled { - return Ok(None); - } - install_default_crypto_provider(); - let credential_error = |source: std::io::Error| ServerError::ListenerCredentials { - transport: "cluster.tls", - source, - }; - - let credentials = if tls.self_signed { - warn_ignored_certificate_files("cluster.tls", &tls.cert_file, &tls.key_file); - let san = config - .cluster - .nodes - .iter() - .find(|node| node.replica_id == topology.self_replica_id) - .map(|node| node.ip.as_str()) - .ok_or_else(|| { - credential_error(std::io::Error::other(format!( - "replica id {} not present in cluster.nodes", - topology.self_replica_id - ))) - })?; - let (cert_chain, key_der) = server_common::generate_self_signed_certificate(san) - .map_err(|error| credential_error(std::io::Error::other(error.to_string())))?; - TlsServerCredentials { - cert_chain, - key_der, - } - } else { - load_pem(Path::new(&tls.cert_file), Path::new(&tls.key_file)).map_err(credential_error)? - }; - - let mut server = - rustls::ServerConfig::builder_with_protocol_versions(&[&rustls::version::TLS13]) - .with_no_client_auth() - .with_single_cert(credentials.cert_chain, credentials.key_der) - .map_err(|error| { - credential_error(std::io::Error::other(format!( - "replica TLS server config rejected credentials: {error}" - ))) - })?; - server.alpn_protocols = vec![REPLICA_ALPN.to_vec()]; - - let client_builder = - rustls::ClientConfig::builder_with_protocol_versions(&[&rustls::version::TLS13]); - let mut client = if tls.self_signed { - client_builder - .dangerous() - .with_custom_certificate_verifier(Arc::new(AcceptAnyServerCert)) - .with_no_client_auth() - } else { - let roots = load_ca_pem(Path::new(&tls.ca_file)).map_err(credential_error)?; - client_builder - .with_root_certificates(Arc::new(roots)) - .with_no_client_auth() - }; - client.alpn_protocols = vec![REPLICA_ALPN.to_vec()]; - - // Keyed by replica id, never by roster position: sparse ids (dynamic - // replica join) would make a positional lookup verify against another - // peer's SNI name. - let peer_names = config - .cluster - .nodes - .iter() - .map(|node| { - let name = ServerName::try_from(node.ip.clone()).map_err(|error| { - credential_error(std::io::Error::new( - std::io::ErrorKind::InvalidInput, - format!( - "cluster node '{}' ip '{}' is not a valid TLS server name: {error}", - node.name, node.ip - ), - )) - })?; - Ok((node.replica_id, name)) - }) - .collect::, ServerError>>()?; - - Ok(Some(ReplicaTlsCtx { - server: Arc::new(server), - client: Arc::new(client), - peer_names, - })) -} - -fn load_tcp_tls_server_credentials( - config: &ServerConfig, -) -> Result { - let tls = &config.tcp.tls; - if ephemeral_certificate("tcp.tls", tls.self_signed, &tls.cert_file) { - return Ok(self_signed_for_loopback()); - } - - load_pem(Path::new(&tls.cert_file), Path::new(&tls.key_file)).map_err(|source| { - ServerError::ListenerCredentials { - transport: "tcp.tls", - source, - } - }) -} - -/// Bind the websocket client listener on `ws_addr`: WSS when -/// `websocket.tls.enabled` (the plain-WS accept loop must not also bind the -/// port -- a plain upgrade parser fed a TLS `ClientHello` rejects every -/// connection with an httparse error), plain WS otherwise. -fn start_websocket_listener( - shard: &Rc, - config: &ServerConfig, - ws_addr: SocketAddr, - accepted_clients: &LocalClientAcceptFns, -) -> Result { - if config.websocket.tls.enabled { - let credentials = load_wss_server_credentials(config)?; - let (listener, tls_config, bound_addr) = client_listener::wss::bind(ws_addr, credentials) - .map_err(|source| { - error!(addr = %ws_addr, error = %source, "failed to bind WSS listener"); - source - })?; - let token = shard.bus.token(); - let accepted_wss = accepted_clients.wss.clone(); - let wss_handle = compio::runtime::spawn(async move { - client_listener::wss::run(listener, tls_config, token, accepted_wss).await; - }); - shard.bus.track_background(wss_handle); - Ok(bound_addr) - } else { - let (listener, bound_addr) = client_listener::ws::bind(ws_addr).map_err(|source| { - error!(addr = %ws_addr, error = %source, "failed to bind websocket listener"); - source - })?; - let token = shard.bus.token(); - let accepted_ws = accepted_clients.ws.clone(); - let ws_handle = compio::runtime::spawn(async move { - client_listener::ws::run(listener, token, accepted_ws).await; - }); - shard.bus.track_background(ws_handle); - Ok(bound_addr) - } -} - -fn load_wss_server_credentials(config: &ServerConfig) -> Result { - let tls = &config.websocket.tls; - if ephemeral_certificate("websocket.tls", tls.self_signed, &tls.cert_file) { - return Ok(self_signed_for_loopback()); - } - - load_pem(Path::new(&tls.cert_file), Path::new(&tls.key_file)).map_err(|source| { - ServerError::ListenerCredentials { - transport: "websocket.tls", - source, - } - }) -} - -fn load_quic_server_credentials( - config: &ServerConfig, -) -> Result { - let certificate = &config.quic.certificate; - if certificate.self_signed { - warn_ignored_certificate_files( - "quic.certificate", - &certificate.cert_file, - &certificate.key_file, - ); - let (cert_chain, key_der) = server_common::generate_self_signed_certificate("localhost") - .map_err(|error| ServerError::ListenerCredentials { - transport: "quic", - source: std::io::Error::other(error.to_string()), - })?; - return Ok(replica_io::QuicServerCredentials { - cert_chain, - key_der, - }); - } - - let credentials = load_pem( - Path::new(&certificate.cert_file), - Path::new(&certificate.key_file), - ) - .map_err(|source| ServerError::ListenerCredentials { - transport: "quic", - source, - })?; - Ok(replica_io::QuicServerCredentials { - cert_chain: credentials.cert_chain, - key_der: credentials.key_der, - }) -} - -/// Client-listener certificate precedence: `self_signed = true` mints an -/// ephemeral loopback certificate only while `cert_file` is absent from disk. -/// An existing PEM pair wins, so a deployment that lays certificates down -/// serves them without also having to unset the flag - the contract every -/// SDK test lane relies on when it points the server at `core/certs/`. -fn ephemeral_certificate(section: &str, self_signed: bool, cert_file: &str) -> bool { - if !self_signed { - return false; - } - if Path::new(cert_file).exists() { - info!( - "{section}.self_signed = true but cert_file = {cert_file} exists on disk; loading it - remove the file or clear the path to serve an ephemeral certificate" - ); - return false; - } - true -} - -/// `self_signed = true` never reads the PEM pair (cluster and QUIC keep the -/// flag authoritative: their generated certificates carry non-loopback SANs), -/// so a cert path resolving on disk looks active to an operator who never -/// asked for it. -fn warn_ignored_certificate_files(section: &str, cert_file: &str, key_file: &str) { - let found: Vec = [("cert_file", cert_file), ("key_file", key_file)] - .into_iter() - .filter(|(_, path)| Path::new(path).exists()) - .map(|(field, path)| format!("{field} = {path}")) - .collect(); - if found.is_empty() { - return; - } - - warn!( - "{section}.self_signed = true, ignoring certificate files found on disk ({}); set {section}.self_signed = false to load them", - found.join(", ") - ); -} - -fn parse_socket_addr(context: &'static str, address: &str) -> Result { - address - .parse() - .map_err(|source| ServerError::SocketAddressParse { - context, - address: address.to_string(), - source, - }) -} - -fn socket_addr_from_parts( - context: &'static str, - host: &str, - port: u16, -) -> Result { - let ip = host - .parse::() - .map_err(|source| ServerError::SocketAddressParse { - context, - address: format!("{host}:{port}"), - source, - })?; - Ok(SocketAddr::new(ip, port)) -} - -/// Build the closure that broadcasts a -/// [`LifecycleFrame::MetadataCommitTick`] to every shard's inbox after a -/// partition-shaped metadata operation commits on shard 0. -/// -/// The receiver-side partition reconciliation loop listens for these -/// wake-ups; coalescing is intentional, so `Full` is recorded as a metric -/// and dropped (the periodic tick recovers). Installed via -/// [`metadata::IggyMetadata::set_commit_notifier`] on shard 0 only, the -/// sole writer of the metadata state machine. -fn make_metadata_commit_notifier( - senders: Vec, - metrics: ShardMetrics, -) -> metadata::CommitNotifier { - Rc::new(move |operation: Operation| { - if !operation_triggers_partition_reconcile(operation) { - return; - } - for sender in &senders { - let frame = ShardFrame::lifecycle(LifecycleFrame::MetadataCommitTick); - match sender.try_send(frame) { - Ok(()) => {} - Err(crossfire::TrySendError::Full(_)) => { - metrics.record_frame_drop( - frame_drop_variant::METADATA_COMMIT_TICK, - frame_drop_reason::FULL, - ); - } - Err(crossfire::TrySendError::Disconnected(_)) => { - metrics.record_frame_drop( - frame_drop_variant::METADATA_COMMIT_TICK, - frame_drop_reason::DISCONNECTED, - ); - } - } - } - }) -} - -/// Filter at the broadcast site, keeping unrelated ops off the SDK reply -/// path. Any new partition-shape op must be added here. -/// -/// The bare `CreateTopic` / `CreatePartitions` arms are unreachable: the -/// leader's prepare-builder in `IggyMetadata` rewrites both into their -/// `*WithAssignments` form, stamping each partition's `consensus_group_id` -/// before journaling, so a committed prepare only ever carries the -/// assignment-bearing variant. Kept as defense-in-depth against a future -/// commit path that emits a bare op. -/// -/// "Partition-shape" is not only the partition SET: the purge and truncate -/// ops leave the set intact but advance per-partition state (purge -/// generation, delete watermark) that only the reconciler enforces on disk. -/// Omitting them defers the on-disk effect to the periodic safety tick, -/// stretching a purge's client-visible tail to a full -/// `reconcile_periodic_interval`. `DeleteSegments` is absent by design: the -/// leader rewrites it into `TruncatePartition` before journaling, so no -/// commit ever carries it. -const fn operation_triggers_partition_reconcile(op: Operation) -> bool { - matches!( - op, - Operation::CreateTopic - | Operation::CreateTopicWithAssignments - | Operation::CreatePartitions - | Operation::CreatePartitionsWithAssignments - | Operation::DeleteTopic - | Operation::DeleteStream - | Operation::DeletePartitions - | Operation::PurgeStream - | Operation::PurgeTopic - | Operation::TruncatePartition - ) -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn superblock_fatal_window_converts_to_capped_backoff_retries() { - assert_eq!( - superblock_window_to_failures(Duration::ZERO), - 0, - "zero window must stay the disabled sentinel" - ); - assert_eq!( - superblock_window_to_failures(Duration::from_mins(2)), - 120, - "past warmup one retry rides each 1s backoff cap" - ); - assert_eq!( - superblock_window_to_failures(Duration::from_micros(500)), - 1, - "a sub-cap window still needs one failure to fire" - ); - } - - #[test] - fn fresh_cluster_bootstrap_requires_explicit_root_credentials() { - assert!(matches!( - validate_root_credentials(true, None, None), - Err(ServerError::ClusterRootCredentialsRequired { - username_env: IGGY_ROOT_USERNAME_ENV, - password_env: IGGY_ROOT_PASSWORD_ENV, - }) - )); - validate_root_credentials(true, Some("root"), Some("secret")) - .expect("both credentials supplied must satisfy the fresh-cluster guard"); - } - - #[test] - fn single_node_bootstrap_generates_root_credentials_when_unset() { - validate_root_credentials(false, None, None) - .expect("a single node mints its own root password"); - } - - #[test] - fn half_set_root_credentials_are_rejected_in_both_directions() { - assert!(matches!( - validate_root_credentials(false, Some("root"), None), - Err(ServerError::RootCredentialsIncomplete { - provided_env: IGGY_ROOT_USERNAME_ENV, - missing_env: IGGY_ROOT_PASSWORD_ENV, - }) - )); - assert!(matches!( - validate_root_credentials(false, None, Some("secret")), - Err(ServerError::RootCredentialsIncomplete { - provided_env: IGGY_ROOT_PASSWORD_ENV, - missing_env: IGGY_ROOT_USERNAME_ENV, - }) - )); - } - - #[test] - fn out_of_range_root_credentials_are_rejected() { - assert!(matches!( - validate_root_credentials(false, Some(""), Some("secret")), - Err(ServerError::RootCredentialLength { - env_name: IGGY_ROOT_USERNAME_ENV, - length: 0, - .. - }) - )); - let too_long = "x".repeat(MAX_PASSWORD_LENGTH + 1); - assert!(matches!( - validate_root_credentials(false, Some("root"), Some(&too_long)), - Err(ServerError::RootCredentialLength { - env_name: IGGY_ROOT_PASSWORD_ENV, - .. - }) - )); - } - - #[test] - fn default_cluster_heartbeat_timeout_matches_consensus_constant() { - // The config default lives in core/server/config.toml (a string, - // so no static assert can pin it); keep it in lockstep with the - // built-in the simulator and un-configured replicas run on. - let config_default = configs::cluster::ClusterConfig::default() - .heartbeat_timeout - .get_duration() - .as_millis(); - let built_in = u128::from(consensus::TimeoutManager::NORMAL_HEARTBEAT_TICKS) - * shard::CONSENSUS_TICK_INTERVAL.as_millis(); - assert_eq!( - config_default, built_in, - "[cluster] heartbeat_timeout default drifted from \ - TimeoutManager::NORMAL_HEARTBEAT_TICKS" - ); - } - - #[test] - fn reconciler_driven_ops_broadcast_a_commit_tick() { - // These commit without touching the partition set, so nothing else - // signals the reconciler: `reconcile_partition_purges` and - // `reconcile_segment_truncations` are the only code that turns them - // into on-disk effect, and they run only when a pass runs. Dropping - // one from the filter silently downgrades it to the periodic tick. - for op in [ - Operation::PurgeStream, - Operation::PurgeTopic, - Operation::TruncatePartition, - ] { - assert!( - operation_triggers_partition_reconcile(op), - "{op:?} is enforced by the reconciler and must wake it on commit" - ); - } - assert!( - !operation_triggers_partition_reconcile(Operation::CreateUser), - "ops with no partition-shape effect must stay off the broadcast" - ); - } - - #[test] - fn recovery_barrier_deadline_holds_the_floor_for_small_heartbeats() { - // Below the 5s default the heartbeat-independent recovery term (~7s of - // ViewChangeStatus backstop plus ceremony) dominates, so the floor - // governs however small the heartbeat is; 3 x 5s lands exactly on it. - // A default-sized status backstop stays on the floor, not above it. - assert_eq!( - recovery_barrier_deadline(Duration::from_secs(1), Duration::from_secs(5)), - RECOVERY_BARRIER_DEADLINE_FLOOR - ); - assert_eq!( - recovery_barrier_deadline(Duration::from_secs(5), Duration::from_secs(5)), - RECOVERY_BARRIER_DEADLINE_FLOOR - ); - } - - #[test] - fn recovery_barrier_deadline_scales_past_the_floor_for_large_heartbeats() { - // Once 3 x heartbeat clears the floor the scaled window governs, so a - // slow-heartbeat cluster is not failed 503 before its longer recovery - // can finish. A default-sized status backstop stays under it. - assert_eq!( - recovery_barrier_deadline(Duration::from_secs(10), Duration::from_secs(5)), - Duration::from_secs(30) - ); - assert_eq!( - recovery_barrier_deadline(Duration::from_secs(15), Duration::from_secs(5)), - Duration::from_secs(45) - ); - } - - #[test] - fn recovery_barrier_deadline_scales_with_the_status_backstop() { - // A raised view-change status backstop stretches worst-case recovery - // even when the heartbeat stays fast, so the deadline must track it or - // post-restart reads 503 before a slow election settles. - assert_eq!( - recovery_barrier_deadline(Duration::from_secs(1), Duration::from_secs(10)), - Duration::from_secs(30) - ); - } - - #[test] - fn recovery_barrier_deadline_at_config_defaults_matches_the_floor() { - // Folding the status term in must not move the stock deadline: at the - // shared 5s defaults each scaled term lands exactly on the 15s floor, - // so an un-tuned cluster keeps its pre-existing recovery window. - let cluster = configs::cluster::ClusterConfig::default(); - assert_eq!( - recovery_barrier_deadline( - cluster.heartbeat_timeout.get_duration(), - cluster.view_change_status_timeout.get_duration(), - ), - RECOVERY_BARRIER_DEADLINE_FLOOR - ); - } - - #[test] - fn recovery_barrier_deadline_saturates_instead_of_panicking() { - // Neither timeout has a config ceiling, so both multiplies must - // saturate rather than abort boot on an absurd parseable value. - assert_eq!( - recovery_barrier_deadline(Duration::MAX, Duration::from_secs(5)), - Duration::MAX - ); - assert_eq!( - recovery_barrier_deadline(Duration::from_secs(5), Duration::MAX), - Duration::MAX - ); - } - - #[test] - fn default_commit_broadcast_interval_matches_consensus_constant() { - // The config default lives in core/server/config.toml (a string, - // so no static assert can pin it); keep it in lockstep with the - // built-in the simulator and un-configured replicas run on. - let config_default = configs::cluster::ClusterConfig::default() - .commit_broadcast_interval - .get_duration() - .as_millis(); - let built_in = u128::from(consensus::TimeoutManager::COMMIT_MESSAGE_TICKS) - * shard::CONSENSUS_TICK_INTERVAL.as_millis(); - assert_eq!( - config_default, built_in, - "[cluster] commit_broadcast_interval default drifted from \ - TimeoutManager::COMMIT_MESSAGE_TICKS" - ); - } - - #[test] - fn default_prepare_retransmit_interval_matches_consensus_constant() { - // The config default lives in core/server/config.toml (a string, - // so no static assert can pin it); keep it in lockstep with the - // built-in the simulator and un-configured replicas run on. - let config_default = configs::cluster::ClusterConfig::default() - .prepare_retransmit_interval - .get_duration() - .as_millis(); - let built_in = u128::from(consensus::TimeoutManager::PREPARE_TICKS) - * shard::CONSENSUS_TICK_INTERVAL.as_millis(); - assert_eq!( - config_default, built_in, - "[cluster] prepare_retransmit_interval default drifted from \ - TimeoutManager::PREPARE_TICKS" - ); - } - - #[test] - fn default_partition_prepare_queue_depth_matches_consensus_constant() { - // The config default lives in core/server/config.toml and flows - // through PartitionConfig::default(); keep the embedded value in - // lockstep with the pipeline depth LocalPipeline::new() (the simulator - // and tests) runs on, so a default deployment is byte-identical. - let config_default = configs::partition::PartitionConfig::default().prepare_queue_depth; - assert_eq!( - config_default, - consensus::PIPELINE_PREPARE_QUEUE_MAX, - "[partition] prepare_queue_depth default drifted from \ - consensus::PIPELINE_PREPARE_QUEUE_MAX" - ); - } - - #[test] - fn default_view_change_retransmit_interval_matches_consensus_constant() { - // The config default lives in core/server/config.toml (a string, so - // no static assert can pin it). One knob drives both view-change - // retransmit timers, which are equal by design, so pin it against both. - let config_default = configs::cluster::ClusterConfig::default() - .view_change_retransmit_interval - .get_duration() - .as_millis(); - let start_view_change = - u128::from(consensus::TimeoutManager::START_VIEW_CHANGE_MESSAGE_TICKS) - * shard::CONSENSUS_TICK_INTERVAL.as_millis(); - let do_view_change = u128::from(consensus::TimeoutManager::DO_VIEW_CHANGE_MESSAGE_TICKS) - * shard::CONSENSUS_TICK_INTERVAL.as_millis(); - assert_eq!( - config_default, start_view_change, - "[cluster] view_change_retransmit_interval default drifted from \ - TimeoutManager::START_VIEW_CHANGE_MESSAGE_TICKS" - ); - assert_eq!( - config_default, do_view_change, - "[cluster] view_change_retransmit_interval default drifted from \ - TimeoutManager::DO_VIEW_CHANGE_MESSAGE_TICKS" - ); - } - - #[test] - fn default_view_change_status_timeout_matches_consensus_constant() { - // The config default lives in core/server/config.toml (a string, so - // no static assert can pin it); keep it in lockstep with the built-in - // the simulator and un-configured replicas run on. - let config_default = configs::cluster::ClusterConfig::default() - .view_change_status_timeout - .get_duration() - .as_millis(); - let built_in = u128::from(consensus::TimeoutManager::VIEW_CHANGE_STATUS_TICKS) - * shard::CONSENSUS_TICK_INTERVAL.as_millis(); - assert_eq!( - config_default, built_in, - "[cluster] view_change_status_timeout default drifted from \ - TimeoutManager::VIEW_CHANGE_STATUS_TICKS" - ); - } - - #[test] - fn default_request_start_view_retransmit_interval_matches_consensus_constant() { - // The config default lives in core/server/config.toml (a string, so - // no static assert can pin it); keep it in lockstep with the built-in - // the simulator and un-configured replicas run on. - let config_default = configs::cluster::ClusterConfig::default() - .request_start_view_retransmit_interval - .get_duration() - .as_millis(); - let built_in = u128::from(consensus::TimeoutManager::REQUEST_START_VIEW_MESSAGE_TICKS) - * shard::CONSENSUS_TICK_INTERVAL.as_millis(); - assert_eq!( - config_default, built_in, - "[cluster] request_start_view_retransmit_interval default drifted from \ - TimeoutManager::REQUEST_START_VIEW_MESSAGE_TICKS" - ); - } - - #[test] - fn default_view_probe_attempts_max_matches_consensus_constant() { - // Belt and suspenders with the static assert above: that pins the - // duplicated configs-crate literal, this pins the shipped config.toml - // value the simulator and un-configured replicas run on. - let config_default = configs::cluster::ClusterConfig::default().view_probe_attempts_max; - assert_eq!( - config_default, - consensus::PROBE_ATTEMPTS_MAX, - "[cluster] view_probe_attempts_max default drifted from \ - consensus::PROBE_ATTEMPTS_MAX" - ); - } - - #[test] - fn default_repair_retry_interval_matches_partitions_constant() { - // The config default lives in core/server/config.toml (a string, so - // no static assert can pin it); keep it in lockstep with the built-in - // the simulator and un-configured replicas run on. - let config_default = configs::cluster::ClusterConfig::default() - .repair_retry_interval - .get_duration() - .as_millis(); - let built_in = - u128::from(partitions::REPAIR_RETRY_TICKS) * shard::CONSENSUS_TICK_INTERVAL.as_millis(); - assert_eq!( - config_default, built_in, - "[cluster] repair_retry_interval default drifted from \ - partitions::REPAIR_RETRY_TICKS" - ); - } - - #[test] - fn default_repair_chunk_max_matches_shard_constant() { - // Belt and suspenders with the static assert above: that pins the - // duplicated configs-crate literal, this pins the shipped config.toml - // value the simulator and un-configured replicas run on. - let config_default = configs::cluster::ClusterConfig::default().repair_chunk_max; - assert_eq!( - config_default as u64, - shard::REPAIR_CHUNK_MAX, - "[cluster] repair_chunk_max default drifted from shard::REPAIR_CHUNK_MAX" - ); - } - - #[test] - fn default_evicted_ring_capacity_matches_partitions_constant() { - // Belt and suspenders with the static assert above; this pins the - // shipped config.toml value. - let config_default = configs::partition::PartitionConfig::default().evicted_ring_capacity; - assert_eq!( - config_default, - partitions::EVICTED_RING_CAPACITY, - "[partition] evicted_ring_capacity default drifted from \ - partitions::EVICTED_RING_CAPACITY" - ); - } - - #[test] - fn default_evicted_ring_bytes_max_matches_partitions_constant() { - // Belt and suspenders with the static assert above; this pins the - // shipped config.toml value. - let config_default = configs::partition::PartitionConfig::default() - .evicted_ring_bytes_max - .as_bytes_u64(); - assert_eq!( - config_default, - partitions::EVICTED_RING_BYTES_MAX, - "[partition] evicted_ring_bytes_max default drifted from \ - partitions::EVICTED_RING_BYTES_MAX" - ); - } - - #[test] - fn shutdown_on_drop_armed_flips_flag() { - let flag = Arc::new(AtomicBool::new(false)); - drop(ShutdownOnDrop::new(Arc::clone(&flag))); - assert!( - flag.load(Ordering::Relaxed), - "an armed guard must flip the flag on drop (covers the error `?` \ - and panic-unwind exit paths of run_shard_thread)" - ); - } - - #[test] - fn shutdown_on_drop_disarmed_leaves_flag() { - let flag = Arc::new(AtomicBool::new(false)); - let mut guard = ShutdownOnDrop::new(Arc::clone(&flag)); - guard.disarm(); - drop(guard); - assert!( - !flag.load(Ordering::Relaxed), - "a disarmed guard must not flip the flag (clean `Ok(())` exit)" - ); - } - - const TEST_POLL_INTERVAL: Duration = Duration::from_millis(50); - - #[compio::test] - async fn broadcast_metadata_bundle_returns_immediately_with_no_peers() { - // Single-shard deployment: shard 0 has no peers to fan out to, - // so the handoff must complete without ever calling `send`. - let (bundle_tx, _bundle_rx) = crossfire::mpmc::bounded_async::(0); - let flag = Arc::new(AtomicBool::new(false)); - let mux = ServerMuxStateMachine::default(); - broadcast_metadata_bundle( - 0, - &bundle_tx, - mux.factory_bundle(), - 0, - &flag, - TEST_POLL_INTERVAL, - ) - .await - .expect("zero peers must not block shard 0"); - } - - #[compio::test] - async fn metadata_bundle_round_trips_through_channel() { - // End-to-end: shard 0 mints a bundle, a peer receives it on - // another runtime, and `from_factory_bundle` constructs a - // reader-mode mux that observes shard 0's writes via the same - // LeftRight pair. - let peers = 1u16; - let (bundle_tx, bundle_rx) = - crossfire::mpmc::bounded_async::(usize::from(peers)); - let flag = Arc::new(AtomicBool::new(false)); - - let owner = ServerMuxStateMachine::default(); - let bundle = owner.factory_bundle(); - broadcast_metadata_bundle(0, &bundle_tx, bundle, peers, &flag, TEST_POLL_INTERVAL) - .await - .expect("broadcast must succeed with one peer drained"); - - let received = await_metadata_bundle(1, &bundle_rx, &flag, TEST_POLL_INTERVAL) - .await - .expect("peer must receive the broadcast bundle"); - let _peer_mux = ServerMuxStateMachine::from_factory_bundle(received); - } - - #[compio::test] - async fn broadcast_metadata_bundle_aborts_when_peers_drop_rx() { - // Shard 0 drives handoff but every peer's `bundle_rx` was dropped - // before recv. Silently returning Ok would commit listener binds - // and consensus init for a cluster whose peers are gone; the - // broadcast must surface the disconnect so `shard_main` aborts. - let (bundle_tx, bundle_rx) = crossfire::mpmc::bounded_async::(0); - drop(bundle_rx); - let flag = Arc::new(AtomicBool::new(false)); - let mux = ServerMuxStateMachine::default(); - - let err = broadcast_metadata_bundle( - 0, - &bundle_tx, - mux.factory_bundle(), - 3, - &flag, - TEST_POLL_INTERVAL, - ) - .await - .expect_err("dropped rx must surface as MetadataHandoffAborted"); - assert!( - matches!(err, ServerError::MetadataHandoffAborted { shard_id: 0 }), - "expected MetadataHandoffAborted, got {err:?}" - ); - } - - #[compio::test] - async fn await_metadata_bundle_aborts_when_owner_drops_without_sending() { - let (bundle_tx, bundle_rx) = crossfire::mpmc::bounded_async::(1); - let flag = Arc::new(AtomicBool::new(false)); - - // Shard 0 dies before broadcasting; the peer must observe the - // disconnect and abort instead of hanging forever. - drop(bundle_tx); - - let err = await_metadata_bundle(1, &bundle_rx, &flag, TEST_POLL_INTERVAL) - .await - .expect_err("a peer whose owner never sends must abort"); - assert!( - matches!(err, ServerError::MetadataHandoffAborted { shard_id: 1 }), - "expected MetadataHandoffAborted, got {err:?}" - ); - } - - #[compio::test] - async fn await_metadata_bundle_aborts_on_shutdown_flag() { - // compio 0.19 `JoinHandle` yields `Result`; the - // `ResumeUnwind` impl re-raises a task panic and maps cancellation - // to `None`. - use compio::runtime::ResumeUnwind; - - let (_bundle_tx, bundle_rx) = crossfire::mpmc::bounded_async::(1); - let flag = Arc::new(AtomicBool::new(false)); - - let waiter = compio::runtime::spawn({ - let flag = Arc::clone(&flag); - async move { await_metadata_bundle(1, &bundle_rx, &flag, TEST_POLL_INTERVAL).await } - }); - - // Owner has not sent yet, but shutdown was requested; the peer - // must exit via the flag poll instead of hanging. - compio::time::sleep(TEST_POLL_INTERVAL / 2).await; - flag.store(true, Ordering::Relaxed); - - let err = waiter - .await - .resume_unwind() - .expect("waiter task was cancelled") - .expect_err("shutdown flag must abort the bundle wait"); - assert!( - matches!(err, ServerError::MetadataHandoffAborted { shard_id: 1 }), - "expected MetadataHandoffAborted on shutdown, got {err:?}" - ); - } - - #[compio::test] - async fn await_bootstrap_complete_returns_immediately_for_single_shard() { - // A single-shard server has no peers to wait on; the owner barrier - // must not block when `peers == 0`. - let (_ready_tx, ready_rx) = crossfire::mpmc::bounded_async::(1); - let flag = Arc::new(AtomicBool::new(false)); - await_bootstrap_complete(&ready_rx, 0, &flag, TEST_POLL_INTERVAL) - .await - .expect("single-shard server must not block on the barrier"); - } - - #[compio::test] - async fn await_bootstrap_complete_drains_every_peer_signal() { - // Two peers report load-complete; shard 0 drains both, then proceeds - // to bind listeners. - let (ready_tx, ready_rx) = crossfire::mpmc::bounded_async::(2); - let flag = Arc::new(AtomicBool::new(false)); - signal_bootstrap_complete(1, &ready_tx, &flag, TEST_POLL_INTERVAL) - .await - .expect("peer 1 must signal load-complete"); - signal_bootstrap_complete(2, &ready_tx, &flag, TEST_POLL_INTERVAL) - .await - .expect("peer 2 must signal load-complete"); - await_bootstrap_complete(&ready_rx, 2, &flag, TEST_POLL_INTERVAL) - .await - .expect("owner must drain both peer signals"); - } - - #[compio::test] - async fn await_bootstrap_complete_aborts_on_shutdown_flag() { - use compio::runtime::ResumeUnwind; - - // `_ready_tx` is held so the channel is not disconnected: the owner - // must exit via the shutdown flag, not a dropped sender. - let (_ready_tx, ready_rx) = crossfire::mpmc::bounded_async::(1); - let flag = Arc::new(AtomicBool::new(false)); - - let owner = compio::runtime::spawn({ - let flag = Arc::clone(&flag); - async move { await_bootstrap_complete(&ready_rx, 1, &flag, TEST_POLL_INTERVAL).await } - }); - - // The peer never signals, but a sibling failure flips the flag; the - // owner must abort instead of hanging before listeners. - compio::time::sleep(TEST_POLL_INTERVAL / 2).await; - flag.store(true, Ordering::Relaxed); - - let err = owner - .await - .resume_unwind() - .expect("owner task was cancelled") - .expect_err("shutdown flag must abort the barrier wait"); - assert!( - matches!( - err, - ServerError::ShardBootstrapBarrierAborted { remaining: 1 } - ), - "expected ShardBootstrapBarrierAborted, got {err:?}" - ); - } - - #[compio::test] - async fn pump_drain_timeout_is_not_reported_as_clean() { - let mut config = ServerConfig::default(); - let timeout = Duration::from_millis(1); - Arc::get_mut(&mut config.system) - .expect("a fresh ServerConfig owns its system config") - .sharding - .shutdown_drain_timeout = iggy_common::IggyDuration::new(timeout); - let pump = compio::runtime::spawn(std::future::pending::>()); - - let error = await_pump_drain(Some(pump), &config, 7) - .await - .expect_err("a live pump past the drain budget is not a clean exit"); - assert!(matches!( - error, - ServerError::ShardPumpDrainTimedOut { - shard_id: 7, - timeout: actual, - } if actual == timeout - )); - } - - #[compio::test] - async fn pump_stopped_by_a_commit_fault_is_not_reported_as_clean() { - // The pump drained and flushed, so the join succeeds. Reporting that - // as a clean exit would hand an orchestrator exit code 0 for a node - // that stopped because it could not persist a cluster-committed op. - let config = ServerConfig::default(); - let fault = FatalCommit { - namespace_raw: 42, - op: 7, - operation: iggy_binary_protocol::Operation::SendMessages, - }; - let pump = compio::runtime::spawn(async move { Some(fault) }); - - let error = await_pump_drain(Some(pump), &config, 3) - .await - .expect_err("a pump that stopped on a commit fault is not a clean exit"); - assert!(matches!( - error, - ServerError::ShardFatal { - shard_id: 3, - namespace_raw: 42, - op: 7, - } - )); - } - - #[compio::test] - async fn signal_bootstrap_complete_aborts_when_owner_drops_rx() { - // Shard 0 aborted before draining and dropped its receiver; a peer's - // signal must surface the disconnect instead of stranding. - let (ready_tx, ready_rx) = crossfire::mpmc::bounded_async::(1); - let flag = Arc::new(AtomicBool::new(false)); - drop(ready_rx); - - let err = signal_bootstrap_complete(2, &ready_tx, &flag, TEST_POLL_INTERVAL) - .await - .expect_err("dropped rx must surface as an abort"); - assert!( - matches!(err, ServerError::MetadataHandoffAborted { shard_id: 2 }), - "expected MetadataHandoffAborted, got {err:?}" - ); - } - - fn cluster_node(ip: &str, http: Option) -> configs::cluster::ClusterNodeConfig { - cluster_node_with_ports(ip, Some(18070), http) - } - - fn cluster_node_with_ports( - ip: &str, - tcp: Option, - http: Option, - ) -> configs::cluster::ClusterNodeConfig { - configs::cluster::ClusterNodeConfig { - name: "node".to_owned(), - ip: ip.to_owned(), - advertised_address: None, - advertised_addresses: Vec::new(), - replica_id: 0, - ports: configs::cluster::TransportPorts { - tcp, - http, - ..Default::default() - }, - } - } - - fn addr(value: &str) -> SocketAddr { - value.parse().expect("valid socket address literal") - } - - #[test] - fn cluster_http_addr_takes_port_from_roster() { - // A byte-identical top-level [http].address is shared across nodes on - // one host; the per-node roster port is the only port source so each - // node binds a distinct HTTP socket. - let node = cluster_node("127.0.0.1", Some(18090)); - let addrs = resolve_cluster_client_addrs( - &node, - addr("127.0.0.1:8090"), - None, - None, - Some(addr("127.0.0.1:3000")), - ) - .expect("cluster address resolution must succeed"); - assert_eq!(addrs.http, Some(addr("127.0.0.1:18090"))); - } - - #[test] - fn cluster_http_addr_merges_config_ip_with_roster_port() { - // Docker/Helm bind `0.0.0.0` and probe loopback; the roster ip is - // only the advertised address. Cluster mode must keep the configured - // interface and take just the port from the roster. - let node = cluster_node("10.0.0.5", Some(18090)); - let addrs = resolve_cluster_client_addrs( - &node, - addr("0.0.0.0:8090"), - None, - None, - Some(addr("0.0.0.0:3000")), - ) - .expect("cluster address resolution must succeed"); - assert_eq!(addrs.http, Some(addr("0.0.0.0:18090"))); - } - - #[test] - fn cluster_http_addr_requires_roster_port_for_enabled_transport() { - // No fallback to the top-level port: a silent default could collide - // with another same-host node, so a missing roster port for an - // enabled transport must refuse to boot. - let node = cluster_node("10.0.0.5", None); - let result = resolve_cluster_client_addrs( - &node, - addr("127.0.0.1:8090"), - None, - None, - Some(addr("127.0.0.1:3000")), - ); - assert!(matches!( - result, - Err(ServerError::ClusterPortMissing { - transport: "http", - replica_id: 0, - }) - )); - } - - #[test] - fn cluster_http_addr_is_none_when_http_disabled() { - // http.enabled = false collapses default_http_addr to None; no roster - // port can revive a listener the operator turned off. - let node = cluster_node("127.0.0.1", Some(18090)); - let addrs = resolve_cluster_client_addrs(&node, addr("127.0.0.1:8090"), None, None, None) - .expect("cluster address resolution must succeed"); - assert_eq!(addrs.http, None); - } - - /// Regression: the shutdown-join deadline must arm at SHUTDOWN, not - /// at boot. The original bound measured from `join_all` entry, so any - /// healthy server outliving `shutdown_join_timeout` (30s default) was - /// abandoned as "wedged" and the process exited - every BDD run died - /// at t+30s while the test container was still compiling. - #[test] - fn join_waits_unbounded_while_the_server_runs() { - let shutdown_flag = AtomicBool::new(false); - // Thread outlives a deliberately tiny join budget; with the flag - // clear the budget must never even arm. - let handle = thread::spawn(|| -> Result<(), ServerError> { - thread::sleep(Duration::from_millis(300)); - Ok(()) - }); - let mut deadline = None; - let joined = join_until_shutdown_deadline( - handle, - &shutdown_flag, - Duration::from_millis(20), - &mut deadline, - ); - assert!( - matches!(joined, Some(Ok(Ok(())))), - "a running server must be awaited indefinitely, not abandoned as wedged" - ); - assert!( - deadline.is_none(), - "the join deadline must not arm before the shutdown flag flips" - ); - } - - #[test] - fn join_abandons_a_wedged_shard_after_the_shutdown_deadline() { - let shutdown_flag = AtomicBool::new(true); - // Never finishes: stands in for a wedged pump. The thread leaks - // into the test process, which exits right after. - let handle = thread::spawn(|| -> Result<(), ServerError> { - loop { - thread::sleep(Duration::from_secs(1)); - } - }); - let mut deadline = None; - let joined = join_until_shutdown_deadline( - handle, - &shutdown_flag, - Duration::from_millis(100), - &mut deadline, - ); - assert!( - joined.is_none(), - "a shard still running past the post-shutdown budget must be abandoned" - ); - assert!(deadline.is_some(), "the deadline arms once the flag is set"); - } - - #[test] - fn cluster_tcp_addr_takes_port_from_roster() { - // Same rule as the other transports: the roster owns the port so - // same-host nodes sharing one [tcp].address still bind distinct - // sockets. - let node = cluster_node("127.0.0.1", None); - let addrs = resolve_cluster_client_addrs(&node, addr("127.0.0.1:8090"), None, None, None) - .expect("cluster address resolution must succeed"); - assert_eq!(addrs.client, addr("127.0.0.1:18070")); - } - - #[test] - fn cluster_tcp_addr_merges_config_ip_with_roster_port() { - // The roster ip is advertised, not bound. Binding it directly would - // strand every co-located dialer (sidecars, health probes, on-host - // consumers) that reaches this node over loopback. - let node = cluster_node("10.0.0.5", None); - let addrs = resolve_cluster_client_addrs(&node, addr("0.0.0.0:8090"), None, None, None) - .expect("cluster address resolution must succeed"); - assert_eq!(addrs.client, addr("0.0.0.0:18070")); - } - - #[test] - fn cluster_tcp_addr_requires_roster_port() { - // tcp is always enabled in cluster mode, so a roster entry without a - // tcp port refuses to boot rather than falling back to [tcp].address. - let node = cluster_node_with_ports("10.0.0.5", None, None); - let result = resolve_cluster_client_addrs(&node, addr("127.0.0.1:8090"), None, None, None); - assert!(matches!( - result, - Err(ServerError::ClusterPortMissing { - transport: "tcp", - replica_id: 0, - }) - )); - } - - #[test] - fn cluster_tcp_addr_keeps_loopback_bind_and_warns_on_roster_mismatch() { - // A loopback [tcp].address under a routable roster ip is honoured - // as configured; remote peers cannot reach it, so the mismatch is - // warned about instead of silently rebinding. - let node = cluster_node("10.0.0.5", None); - let addrs = resolve_cluster_client_addrs(&node, addr("127.0.0.1:8090"), None, None, None) - .expect("cluster address resolution must succeed"); - assert_eq!(addrs.client, addr("127.0.0.1:18070")); - assert!(roster_ip_unreachable_from_bind_addr(&node.ip, addrs.client)); - } - - #[test] - fn roster_mismatch_warning_is_silent_for_wildcard_and_hostname_rosters() { - // A wildcard bind covers the roster interface, and a DNS roster entry - // can resolve to the bound one; neither is a misconfiguration. - assert!(!roster_ip_unreachable_from_bind_addr( - "10.0.0.5", - addr("0.0.0.0:18070") - )); - assert!(!roster_ip_unreachable_from_bind_addr( - "node-1.example.com", - addr("127.0.0.1:18070") - )); - assert!(!roster_ip_unreachable_from_bind_addr( - "10.0.0.5", - addr("10.0.0.5:18070") - )); - } - - #[test] - fn derived_address_warns_only_when_it_misses_a_listener() { - // Derived from a loopback tcp.address while another transport serves - // an external interface: metadata would publish an address no client - // reaches. Every non-TCP listener carries the same exposure, since one - // host is published for all four. - for listener in ["10.0.0.5:3000", "10.0.0.5:8080", "10.0.0.5:8092"] { - assert!( - derived_address_misses_listener(None, "127.0.0.1", addr(listener)), - "{listener} is not reachable at 127.0.0.1" - ); - } - // Same interface, and a wildcard bind that covers any of them. - assert!(!derived_address_misses_listener( - None, - "10.0.0.5", - addr("10.0.0.5:3000") - )); - assert!(!derived_address_misses_listener( - None, - "127.0.0.1", - addr("0.0.0.0:3000") - )); - // A declared address is deliberate and unrelated to local interfaces. - assert!(!derived_address_misses_listener( - Some("broker-1.example.com"), - "broker-1.example.com", - addr("10.0.0.5:3000") - )); - } - - #[test] - fn wildcard_listener_warns_only_under_a_derived_loopback_address() { - // Metadata says 127.0.0.1 while this listener takes connections from - // anywhere: whoever arrives from another host is told to dial itself. - assert!(wildcard_listener_under_loopback_address( - None, - "127.0.0.1", - addr("0.0.0.0:3000") - )); - assert!(wildcard_listener_under_loopback_address( - None, - "127.0.0.1", - addr("[::]:3000") - )); - // A published address that is reachable from elsewhere is what the - // wildcard listener wants, so there is nothing to say. - assert!(!wildcard_listener_under_loopback_address( - None, - "10.0.0.5", - addr("0.0.0.0:3000") - )); - // A concrete bind is the other warning's business, not this one's. - assert!(!wildcard_listener_under_loopback_address( - None, - "127.0.0.1", - addr("10.0.0.5:3000") - )); - // A declared address is deliberate; a loopback one is a local setup. - assert!(!wildcard_listener_under_loopback_address( - Some("127.0.0.1"), - "127.0.0.1", - addr("0.0.0.0:3000") - )); - } - - #[test] - fn derived_bind_ip_follows_the_first_enabled_listener() { - let mut config: ServerConfig = - toml::from_str(include_str!("../config.toml")).expect("shipped config deserializes"); - let topology = |ws: Option<&str>, http: Option<&str>| TcpTopology { - cluster_id: 0, - self_replica_id: 0, - replica_count: 1, - client_listen_addr: addr("127.0.0.1:8090"), - replica_listen_addr: None, - ws_listen_addr: ws.map(addr), - quic_listen_addr: None, - http_listen_addr: http.map(addr), - tcp_tls_listen_addr: None, - peers: Vec::new(), - }; - let expected = |ip: &str| ip.parse::().unwrap(); - - assert_eq!( - derived_bind_ip( - &topology(Some("10.0.0.5:8092"), Some("10.0.0.6:3000")), - &config - ), - expected("127.0.0.1") - ); - - config.tcp.enabled = false; - assert_eq!( - derived_bind_ip( - &topology(Some("10.0.0.5:8092"), Some("10.0.0.6:3000")), - &config - ), - expected("10.0.0.5"), - "websocket is next in line once tcp is off" - ); - assert_eq!( - derived_bind_ip(&topology(None, Some("10.0.0.6:3000")), &config), - expected("10.0.0.6"), - "with websocket and quic off too, http answers" - ); - assert_eq!( - derived_bind_ip(&topology(None, None), &config), - expected("127.0.0.1"), - "with every client listener off the value reaches no client anyway" - ); - } - - #[test] - fn roster_mismatch_warning_is_silent_for_v4_mapped_binds() { - // `[::ffff:0.0.0.0]` is the v4 wildcard and `::ffff:10.0.0.5` is - // `10.0.0.5`, so neither reaches the dialer any differently than the - // plain spelling the case above covers. - assert!(!roster_ip_unreachable_from_bind_addr( - "10.0.0.5", - addr("[::ffff:0.0.0.0]:18070") - )); - assert!(!roster_ip_unreachable_from_bind_addr( - "10.0.0.5", - addr("[::ffff:10.0.0.5]:18070") - )); - // A genuine mismatch still warns through the mapped spelling. - assert!(roster_ip_unreachable_from_bind_addr( - "10.0.0.5", - addr("[::ffff:127.0.0.1]:18070") - )); - } -} diff --git a/core/server/src/cluster_meta.rs b/core/server/src/cluster_meta.rs index 6fd0624534..41b8df97a6 100644 --- a/core/server/src/cluster_meta.rs +++ b/core/server/src/cluster_meta.rs @@ -26,7 +26,8 @@ //! each caller derives it from its own on-shard view (`None` when the serving //! shard has no consensus - peer shards - in which case no node is marked //! leader, but the full roster is still returned). The self-synthesized single -//! node is the cluster-disabled fallback, shared by both callers. +//! node is the cluster-disabled fallback, shared by both callers, and the one +//! place this node's own bound ports are reported. use configs::ConfigurationError; use configs::cluster::{AdvertisedAddress, ClusterConfig, ResolvedClusterNode, TransportPorts}; @@ -34,8 +35,8 @@ use iggy_common::{ ClusterMetadata, ClusterNode, ClusterNodeRole, ClusterNodeStatus, TransportEndpoints, }; use std::net::IpAddr; -use std::sync::Arc; use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::{Arc, OnceLock}; /// Node name reported for the synthesized self node when no roster applies. const SELF_NODE_NAME: &str = "iggy-node"; @@ -90,11 +91,12 @@ pub fn resolved_roster_nodes( .collect() } -/// Config-derived cluster topology reported by cluster-metadata reads. +/// Cluster topology reported by cluster-metadata reads. /// -/// Copied out of `ClusterConfig` at listener/shard start so both handlers stay -/// synchronous and never borrow live config. `self_*` describe this node and -/// back the cluster-disabled self-synthesis only. +/// Copied out of `ClusterConfig` at shard start so both handlers stay +/// synchronous and never borrow live config. `self_advertised` and the two +/// port sets describe this node and back the cluster-disabled self-synthesis +/// only. pub struct ClusterRoster { pub enabled: bool, pub name: String, @@ -104,9 +106,14 @@ pub struct ClusterRoster { /// This node's own client-facing address, reported for the synthesized /// self node (see [`self_advertised_address`]). pub self_advertised: String, - /// This node's own client ports for the same self node (`None` = transport - /// disabled). - pub self_ports: TransportPorts, + /// This node's own client ports for the same self node as configured + /// (`None` = transport disabled), reported until the bound ones are + /// published. + pub configured_ports: TransportPorts, + /// The same ports as the OS bound them, published by shard 0 once its + /// listeners are up, so a configured `:0` port resolves for the + /// self-synthesized node; cluster mode serves `cluster.nodes[*].ports`. + pub bound_ports: Arc, /// Metadata-group view, published by shard 0 (the only shard holding the /// consensus instance) so every shard's cluster-metadata read marks the /// current leader. `u64::MAX` until shard 0 first publishes. @@ -116,6 +123,20 @@ pub struct ClusterRoster { /// Sentinel for "shard 0 has not published a view yet". pub const METADATA_VIEW_UNKNOWN: u64 = u64::MAX; +/// Client ports as bound by shard 0's listeners. One cell per process, shared +/// by every shard's roster: blank from bootstrap until the last listener has +/// bound, then filled once, before the HTTP listener serves. A binary read +/// that lands in the bind window answers the configured ports instead. +#[derive(Default)] +pub struct BoundPorts(OnceLock); + +impl BoundPorts { + pub fn publish(&self, ports: TransportPorts) { + let published = self.0.set(ports); + debug_assert!(published.is_ok(), "listeners bind once per process"); + } +} + impl ClusterRoster { /// A cluster-disabled roster with no self address. The pre-bootstrap /// placeholder a [`crate::session_manager::SessionManager`] holds until @@ -127,11 +148,18 @@ impl ClusterRoster { name: String::new(), nodes: Vec::new(), self_advertised: String::new(), - self_ports: TransportPorts::default(), + configured_ports: TransportPorts::default(), + bound_ports: Arc::default(), metadata_view: Arc::new(AtomicU64::new(METADATA_VIEW_UNKNOWN)), } } + /// This node's own client ports: the bound ones once shard 0 has + /// published them, the configured ones before. + fn self_ports(&self) -> &TransportPorts { + self.bound_ports.0.get().unwrap_or(&self.configured_ports) + } + /// The current metadata primary's REPLICA ID, from the shard-0-published /// view; `None` until the first publish or with no roster. /// @@ -192,7 +220,7 @@ impl ClusterRoster { nodes: vec![ClusterNode { name: SELF_NODE_NAME.to_owned(), ip: self.self_advertised.clone(), - endpoints: ports_to_endpoints(&self.self_ports), + endpoints: ports_to_endpoints(self.self_ports()), role: ClusterNodeRole::Leader, status: ClusterNodeStatus::Healthy, }], @@ -250,7 +278,8 @@ mod tests { name: "test-cluster".to_owned(), nodes: vec![ResolvedClusterNode::try_from(node).expect("valid roster node")], self_advertised: "127.0.0.1".to_owned(), - self_ports: TransportPorts::default(), + configured_ports: TransportPorts::default(), + bound_ports: Arc::default(), metadata_view: Arc::new(AtomicU64::new(METADATA_VIEW_UNKNOWN)), } } @@ -408,10 +437,11 @@ mod tests { name: String::new(), nodes: Vec::new(), self_advertised: "broker-1.example.com".to_owned(), - self_ports: TransportPorts { + configured_ports: TransportPorts { tcp: Some(8090), ..TransportPorts::default() }, + bound_ports: Arc::default(), metadata_view: Arc::new(AtomicU64::new(METADATA_VIEW_UNKNOWN)), }; @@ -423,4 +453,37 @@ mod tests { assert_eq!(metadata.nodes[0].role, ClusterNodeRole::Leader); assert_eq!(metadata.nodes[0].status, ClusterNodeStatus::Healthy); } + + #[test] + fn self_metadata_reports_bound_ports_once_published() { + let roster = ClusterRoster { + enabled: false, + name: String::new(), + nodes: Vec::new(), + self_advertised: "127.0.0.1".to_owned(), + configured_ports: TransportPorts { + tcp: Some(0), + http: Some(0), + ..TransportPorts::default() + }, + bound_ports: Arc::default(), + metadata_view: Arc::new(AtomicU64::new(METADATA_VIEW_UNKNOWN)), + }; + assert_eq!( + roster.cluster_metadata(None, None).nodes[0].endpoints.tcp, + 0 + ); + + roster.bound_ports.publish(TransportPorts { + tcp: Some(45001), + http: Some(45002), + ..TransportPorts::default() + }); + + let endpoints = roster.cluster_metadata(None, None).nodes[0] + .endpoints + .clone(); + assert_eq!(endpoints.tcp, 45001); + assert_eq!(endpoints.http, 45002); + } } diff --git a/core/server/src/config_writer.rs b/core/server/src/config_writer.rs index 92d4fb2864..6c91a0383f 100644 --- a/core/server/src/config_writer.rs +++ b/core/server/src/config_writer.rs @@ -18,9 +18,44 @@ use crate::server_error::ServerError; use compio::fs::OpenOptions; use compio::io::AsyncWriteAtExt; +use configs::cluster::TransportPorts; use configs::server::ServerConfig; use std::net::SocketAddr; +/// Every listener the server bound, as the OS reports it (`None` = not bound +/// on this node). +#[derive(Default)] +pub struct BoundAddresses { + pub tcp: Option, + pub tcp_tls: Option, + pub quic: Option, + pub websocket: Option, + pub http: Option, + pub replica: Option, +} + +impl BoundAddresses { + /// The listener a client's `tcp.address` reaches: the TLS one occupies the + /// configured tcp slot, as the WSS one occupies the websocket slot. + pub fn client_tcp(&self) -> Option { + self.tcp_tls.or(self.tcp) + } +} + +/// The bound listeners as the roster ports the cluster metadata and +/// `current_config.toml` publish. +impl From<&BoundAddresses> for TransportPorts { + fn from(bound: &BoundAddresses) -> Self { + Self { + tcp: bound.client_tcp().map(|addr| addr.port()), + quic: bound.quic.map(|addr| addr.port()), + http: bound.http.map(|addr| addr.port()), + websocket: bound.websocket.map(|addr| addr.port()), + tcp_replica: bound.replica.map(|addr| addr.port()), + } + } +} + /// Write the runtime `current_config.toml` file with the effective bound ports. /// /// # Errors @@ -30,26 +65,25 @@ use std::net::SocketAddr; pub async fn write_current_config( config: &ServerConfig, current_replica_id: Option, - bound_tcp: Option, - bound_replica: Option, - bound_tcp_tls: Option, - bound_quic: Option, - bound_websocket: Option, + bound: &BoundAddresses, ) -> Result<(), ServerError> { let mut current_config = config.clone(); - if let Some(bound_client_tcp) = bound_tcp_tls.or(bound_tcp) { - // Keep parity with the current server binary: integration harnesses - // read `tcp.address` from `runtime/current_config.toml` to discover - // the actual port chosen by the OS when binding to port 0. + if let Some(bound_client_tcp) = bound.client_tcp() { + // Integration harnesses read the `*.address` fields from + // `runtime/current_config.toml` to discover the actual port chosen by + // the OS when binding to port 0. current_config.tcp.address = bound_client_tcp.to_string(); } - if let Some(bound_quic) = bound_quic { + if let Some(bound_quic) = bound.quic { current_config.quic.address = bound_quic.to_string(); } - if let Some(bound_websocket) = bound_websocket { + if let Some(bound_websocket) = bound.websocket { current_config.websocket.address = bound_websocket.to_string(); } + if let Some(bound_http) = bound.http { + current_config.http.address = bound_http.to_string(); + } if current_config.cluster.enabled && let Some(replica_id) = current_replica_id @@ -60,18 +94,13 @@ pub async fn write_current_config( .iter_mut() .find(|node| node.replica_id == replica_id) .ok_or(ServerError::ClusterNodeNotFound { replica_id })?; - if let Some(bound_client_tcp) = bound_tcp_tls.or(bound_tcp) { - node.ports.tcp = Some(bound_client_tcp.port()); - } - if let Some(bound_replica) = bound_replica { - node.ports.tcp_replica = Some(bound_replica.port()); - } - if let Some(bound_quic) = bound_quic { - node.ports.quic = Some(bound_quic.port()); - } - if let Some(bound_websocket) = bound_websocket { - node.ports.websocket = Some(bound_websocket.port()); - } + // A transport this node did not bind keeps its configured port. + let bound_ports = TransportPorts::from(bound); + node.ports.tcp = bound_ports.tcp.or(node.ports.tcp); + node.ports.tcp_replica = bound_ports.tcp_replica.or(node.ports.tcp_replica); + node.ports.quic = bound_ports.quic.or(node.ports.quic); + node.ports.websocket = bound_ports.websocket.or(node.ports.websocket); + node.ports.http = bound_ports.http.or(node.ports.http); } let runtime_path = current_config.system.get_runtime_path(); diff --git a/core/server/src/dispatch/mod.rs b/core/server/src/dispatch/mod.rs index 0385fdebac..8f69e7886e 100644 --- a/core/server/src/dispatch/mod.rs +++ b/core/server/src/dispatch/mod.rs @@ -1279,7 +1279,8 @@ mod tests { }) .to_vec(), self_advertised: "127.0.0.1".to_owned(), - self_ports: TransportPorts::default(), + configured_ports: TransportPorts::default(), + bound_ports: Arc::default(), metadata_view: Arc::new(std::sync::atomic::AtomicU64::new( crate::cluster_meta::METADATA_VIEW_UNKNOWN, )), diff --git a/core/server/src/http.rs b/core/server/src/http.rs index 25a91eff50..cbbb515ad6 100644 --- a/core/server/src/http.rs +++ b/core/server/src/http.rs @@ -47,7 +47,6 @@ use std::net::SocketAddr; use std::rc::Rc; use std::str::FromStr; use std::sync::Arc; -use std::sync::atomic::AtomicU64; use axum::Router; use axum::extract::connect_info::Connected; @@ -57,7 +56,7 @@ use axum::middleware::{Next, from_fn, from_fn_with_state}; use axum::response::Response; use axum::routing::{delete, get, post, put}; use compio::net::TcpListener; -use configs::cluster::{ClusterConfig, TransportPorts, http_forwarding_key_material}; +use configs::cluster::{ClusterConfig, http_forwarding_key_material}; use configs::http::{HttpConfig, HttpCorsConfig}; use configs::server::ServerSystemConfig; use iggy_common::IggyError; @@ -66,7 +65,7 @@ use send_wrapper::SendWrapper; use tower_http::cors::{AllowOrigin, CorsLayer}; use tracing::{error, info, warn}; -use crate::cluster_meta::{ClusterRoster, resolved_roster_nodes}; +use crate::cluster_meta::ClusterRoster; use crate::http::handlers::{ change_password, create_cg, create_partitions, create_pat, create_stream, create_topic, create_user, delete_cg, delete_consumer_offset, delete_partitions, delete_pat, delete_segments, @@ -79,36 +78,47 @@ use crate::http::handlers::{ }; use crate::http::jwt::JwtManager; use crate::http::session::RegistrationBarrier; -use crate::http::state::{HttpInner, HttpState, insert_view_header}; +use crate::http::state::{ForwardState, HttpInner, HttpState, insert_view_header}; use crate::server_error::ServerError; use crate::shell::ServerShard; -/// Bind the shard-0 HTTP listener and spawn the `cyper-axum` serve loop as a -/// background task on shard 0's compio runtime. Serves HTTPS when -/// `http.tls.enabled` (a TLS accept pump feeds handshaken streams to the -/// serve loop, see [`mod@tls`]), plain HTTP otherwise. -/// -/// The caller gates this to shard 0 and to `http.enabled`; the listener stops -/// when the bus shutdown token fires. +/// The shard-0 HTTP listener before its socket exists: the address to bind +/// and the config products the serve loop needs. Built first at boot, so a +/// bad `[http.*]` section fails before any listener accepts; [`Self::bind`] +/// opens the socket. +pub struct PreparedHttp { + addr: SocketAddr, + jwt: JwtManager, + forward: ForwardState, + cors: Option, + metrics_endpoint: Option, + max_request_size: usize, +} + +/// The shard-0 HTTP listener between bind and serve, held apart so the bound +/// port can be published (cluster roster, `current_config.toml`) before any +/// request is answered. +pub struct BoundHttp { + listener: TcpListener, + pub bound_addr: SocketAddr, + prepared: PreparedHttp, +} + +/// Validate the HTTP config and build what the serve loop needs, without +/// opening the socket; [`PreparedHttp::bind`] binds it and [`start`] serves +/// it. The caller gates this to shard 0 and to `http.enabled`. /// /// # Errors /// /// Returns [`ServerError`] if the JWT manager cannot be built from /// `http_config.jwt`, the `[http.cors]` config is invalid, the `[http.tls]` -/// credentials cannot be loaded, or the listener cannot bind to `addr`. -#[allow(clippy::too_many_arguments)] -pub fn start( - shard: &Rc, +/// forwarding credentials cannot be loaded, or the `[http.metrics]` endpoint +/// is not a route path. +pub fn prepare( addr: SocketAddr, http_config: &HttpConfig, - clients_table_max: usize, - max_tokens_per_user: u32, cluster: &ClusterConfig, - system_config: Arc, - self_advertised: &str, - self_ports: &TransportPorts, - shard_metrics_all: &[shard::metrics::ShardMetrics], -) -> Result<(), ServerError> { +) -> Result { // In cluster mode with no configured JWT secret the signing key derives // from the cluster PSK, so a bearer minted on any node verifies on every // node - the invariant follower-to-primary forwarding depends on. @@ -131,8 +141,8 @@ pub fn start( usize::try_from(http_config.max_request_size.as_bytes_u64()).unwrap_or(usize::MAX); let forward = forward::build_forward_state(&http_config.tls, max_request_size, forwarding_active)?; - // Validated before bind so a bad [http.cors] fails boot before the socket - // opens and the "started" log prints. + // Validated here, before the socket opens, so a bad [http.cors] fails boot + // before the "started" log prints. let cors = http_config .cors .enabled @@ -141,37 +151,89 @@ pub fn start( // Same early-fail rule for the scrape path: axum panics on a route // without a leading '/', so reject it as a config error instead. let metrics_endpoint = metrics::validated_endpoint(&http_config.metrics)?; - let (listener, bound_addr) = client_listener::tcp::bind(addr)?; + Ok(PreparedHttp { + addr, + jwt, + forward, + cors, + metrics_endpoint, + max_request_size, + }) +} +impl PreparedHttp { + /// Open the socket. Kept apart from [`prepare`] so the caller can order it + /// after any boot step that may block, and the port never sits listening + /// with nothing serving it. + /// + /// # Errors + /// + /// Returns [`ServerError`] if the listener cannot bind to the prepared + /// address. + pub fn bind(self) -> Result { + let (listener, bound_addr) = client_listener::tcp::bind(self.addr).map_err(|source| { + error!(addr = %self.addr, error = %source, "failed to bind HTTP listener"); + source + })?; + Ok(BoundHttp { + listener, + bound_addr, + prepared: self, + }) + } +} + +/// Assemble the router over a bound listener and spawn the `cyper-axum` +/// serve loop as a background task on shard 0's compio runtime. Serves HTTPS +/// when `http.tls.enabled` (a TLS accept pump feeds handshaken streams to the +/// serve loop, see [`mod@tls`]), plain HTTP otherwise; the listener stops +/// when the bus shutdown token fires. +/// +/// `roster` is shard 0's own [`ClusterRoster`], so the HTTP and binary +/// cluster-metadata reads serve one truth. +/// +/// # Errors +/// +/// Returns [`ServerError`] if the `[http.tls]` server credentials cannot be +/// loaded. +#[allow(clippy::too_many_arguments)] +pub fn start( + bound: BoundHttp, + shard: &Rc, + http_config: &HttpConfig, + clients_table_max: usize, + max_tokens_per_user: u32, + system_config: Arc, + roster: Rc, + shard_metrics_all: &[shard::metrics::ShardMetrics], +) -> Result<(), ServerError> { + let BoundHttp { + listener, + bound_addr, + prepared: + PreparedHttp { + jwt, + forward, + cors, + metrics_endpoint, + max_request_size, + .. + }, + } = bound; let state: HttpState = SendWrapper::new(Rc::new(HttpInner { shard: Rc::clone(shard), jwt, system_config, sessions: RefCell::new(HashMap::new()), registrations: RegistrationBarrier::default(), - roster: ClusterRoster { - enabled: cluster.enabled, - name: cluster.name.clone(), - nodes: resolved_roster_nodes(cluster).map_err(ServerError::Config)?, - self_advertised: self_advertised.to_owned(), - // The self node reports the live bound HTTP port; the other client - // ports arrive resolved from the caller. - self_ports: TransportPorts { - http: Some(bound_addr.port()), - ..self_ports.clone() - }, - // The HTTP listener is shard-0-only, where the live consensus - // handle supplies the leader; the published-view fallback is - // never consulted here. - metadata_view: Arc::new(AtomicU64::new(crate::cluster_meta::METADATA_VIEW_UNKNOWN)), - }, + roster, max_http_sessions: crate::http::session::max_http_sessions(clients_table_max), max_tokens_per_user, in_flight_writes: Cell::new(0), forward, metrics: metrics::HttpMetrics::init(shard_metrics_all), })); - let router = router( + let app = router( state, max_request_size, cors, @@ -189,7 +251,7 @@ pub fn start( ); shard.bus.track_background(pump); info!(address = %bound_addr, "server HTTPS listener started"); - let handle = compio::runtime::spawn(tls::serve(connections, router, shard.bus.token())); + let handle = compio::runtime::spawn(tls::serve(connections, app, shard.bus.token())); shard.bus.track_background(handle); } else { info!(address = %bound_addr, "server HTTP listener started"); @@ -197,7 +259,7 @@ pub fn start( let handle = compio::runtime::spawn(async move { if let Err(error) = cyper_axum::serve( listener, - router.into_make_service_with_connect_info::(), + app.into_make_service_with_connect_info::(), ) .with_graceful_shutdown(async move { shutdown.wait().await }) .await diff --git a/core/server/src/http/error.rs b/core/server/src/http/error.rs index db3edcc5ac..e369b5cfa3 100644 --- a/core/server/src/http/error.rs +++ b/core/server/src/http/error.rs @@ -617,7 +617,8 @@ mod tests { .map(|node| ResolvedClusterNode::try_from(node).expect("valid roster node")) .collect(), self_advertised: "127.0.0.1".to_owned(), - self_ports: TransportPorts::default(), + configured_ports: TransportPorts::default(), + bound_ports: std::sync::Arc::default(), metadata_view: std::sync::Arc::new(std::sync::atomic::AtomicU64::new( crate::cluster_meta::METADATA_VIEW_UNKNOWN, )), diff --git a/core/server/src/http/forward.rs b/core/server/src/http/forward.rs index b0ead9be58..fdfa2ea873 100644 --- a/core/server/src/http/forward.rs +++ b/core/server/src/http/forward.rs @@ -68,7 +68,7 @@ use configs::http::HttpTlsConfig; use consensus::MetadataHandle; use futures::StreamExt; use iggy_common::IggyError; -use message_bus::transports::tls::{install_default_crypto_provider, load_pem}; +use message_bus::transports::tls::load_pem; use rustls::client::danger::{HandshakeSignatureValid, ServerCertVerified, ServerCertVerifier}; use rustls::crypto::{CryptoProvider, WebPkiSupportedAlgorithms}; use rustls::pki_types::{CertificateDer, ServerName, UnixTime}; @@ -156,13 +156,17 @@ pub(in crate::http) fn build_forward_state( body_limit: usize, active: bool, ) -> Result { - // Unconditional: cyper's rustls connector resolves the process default - // provider even when the client only ever dials plain http. - install_default_crypto_provider(); let builder = cyper::Client::builder() // The retry loop re-resolves the primary from the local roster; a // followed `Location` would let the peer steer the bearer anywhere. .redirect(cyper::redirect::Policy::none()); + // `bootstrap()` installs the process-level provider before any shard + // thread exists; both `ClientConfig` builders below panic without one, so + // fail the boot instead. A unit test building this state installs it + // itself, as http/tls.rs does. + let provider = CryptoProvider::get_default().ok_or_else(|| ServerError::HttpForwardClient { + reason: "no process-level rustls CryptoProvider installed".to_string(), + })?; let (builder, scheme) = if tls.enabled { let credentials = load_pem(Path::new(&tls.cert_file), Path::new(&tls.key_file)).map_err(|source| { @@ -178,10 +182,7 @@ pub(in crate::http) fn build_forward_state( reason: "TLS certificate chain is empty".to_string(), } })?; - let algorithms = CryptoProvider::get_default() - // Installed above; absence is unreachable. - .expect("default crypto provider installed") - .signature_verification_algorithms; + let algorithms = provider.signature_verification_algorithms; let config = rustls::ClientConfig::builder() .dangerous() .with_custom_certificate_verifier(Arc::new(PinnedCertVerifier { pinned, algorithms })) @@ -726,7 +727,8 @@ mod tests { .map(|node| ResolvedClusterNode::try_from(node).expect("valid roster node")) .collect(), self_advertised: "127.0.0.1".to_owned(), - self_ports: TransportPorts::default(), + configured_ports: TransportPorts::default(), + bound_ports: Arc::default(), metadata_view: std::sync::Arc::new(std::sync::atomic::AtomicU64::new( crate::cluster_meta::METADATA_VIEW_UNKNOWN, )), diff --git a/core/server/src/http/state.rs b/core/server/src/http/state.rs index 5f61f82897..73b2c6cb7b 100644 --- a/core/server/src/http/state.rs +++ b/core/server/src/http/state.rs @@ -102,7 +102,7 @@ pub(in crate::http) struct HttpInner { /// Per-key registration barrier: prevents a thundering herd of first /// requests for one credential from each running its own `Register`. pub(in crate::http) registrations: RegistrationBarrier, - pub(in crate::http) roster: ClusterRoster, + pub(in crate::http) roster: Rc, /// Cap on live per-credential sessions: half the configured `[metadata] /// clients_table_max`, so HTTP sessions cannot crowd the TCP/QUIC/WS virtual /// clients out of the shared VSR client table. Read by `resolve_session` diff --git a/core/server/src/http/tls.rs b/core/server/src/http/tls.rs index a7b8510955..73f4108b8c 100644 --- a/core/server/src/http/tls.rs +++ b/core/server/src/http/tls.rs @@ -52,9 +52,7 @@ use hyper::rt::Executor; use hyper_util::server::conn::auto::Builder; use hyper_util::service::TowerToHyperService; use message_bus::ShutdownToken; -use message_bus::transports::tls::{ - TlsServerCredentials, install_default_crypto_provider, load_pem, -}; +use message_bus::transports::tls::{TlsServerCredentials, load_pem}; use tower_http::add_extension::AddExtension; use tracing::{debug, error}; @@ -200,7 +198,6 @@ impl + 'static> Executor for LocalExecutor { fn build_server_config( credentials: TlsServerCredentials, ) -> Result, ServerError> { - install_default_crypto_provider(); let mut config = rustls::ServerConfig::builder() .with_no_client_auth() .with_single_cert(credentials.cert_chain, credentials.key_der) @@ -272,10 +269,13 @@ fn spawn_handshake( #[cfg(test)] mod tests { use super::*; - use message_bus::transports::tls::self_signed_for_loopback; + use message_bus::transports::tls::{install_default_crypto_provider, self_signed_for_loopback}; #[test] fn server_config_advertises_h2_and_http11_alpn() { + // The server installs the provider once at bootstrap; a unit test + // has no bootstrap. + install_default_crypto_provider(); let config = build_server_config(self_signed_for_loopback()).expect("self-signed cert builds"); assert_eq!( diff --git a/core/server/src/lib.rs b/core/server/src/lib.rs index 6f0c5ad296..1b1036f0ee 100644 --- a/core/server/src/lib.rs +++ b/core/server/src/lib.rs @@ -30,17 +30,15 @@ static GLOBAL: MiMalloc = MiMalloc; pub const VERSION: &str = env!("CARGO_PKG_VERSION"); pub const SEMANTIC_VERSION: SemanticVersion = SemanticVersion::parse_const(VERSION); -// Visibility rule: `pub` = named external consumer. main.rs consumes -// `bootstrap`, `server_error`, and `systemd`; the simulator consumes `shell`, -// `bootstrap::wire_shell_handlers`, and (through `ShellHandlers.sessions`) +// Visibility rule: `pub` = named external consumer. main.rs consumes `boot` +// (including `boot::systemd`) and `server_error`; the simulator consumes +// `shell`, `boot::wire_shell_handlers`, and (through `ShellHandlers.sessions`) // `session_manager`. Everything else is crate-internal. // boot: process entry, shard threads, recovery orchestration. -pub mod bootstrap; +pub mod boot; pub(crate) mod config_writer; pub(crate) mod shard_allocator; -#[cfg(feature = "systemd")] -pub mod systemd; // spine: the request path - shell vocabulary, dispatch funnel, per-domain ops. pub(crate) mod consumer_group; diff --git a/core/server/src/main.rs b/core/server/src/main.rs index 866f46cb11..5706a32601 100644 --- a/core/server/src/main.rs +++ b/core/server/src/main.rs @@ -22,9 +22,7 @@ mod args; use args::Args; use clap::Parser; use configs::server::ServerConfig; -use server::bootstrap::{ - apply_default_root_credentials, bootstrap, load_config, prepare_runtime_dirs, -}; +use server::boot::{apply_default_root_credentials, bootstrap, load_config, prepare_runtime_dirs}; use server::server_error::ServerError; use server_common::log::Logging; use system_stats::capture_allowed_cpus; @@ -98,7 +96,7 @@ fn main() -> Result<(), ServerError> { let joined = shards.join_all(); #[cfg(feature = "systemd")] if let Err(error) = &joined { - server::systemd::notify_shutdown_failure(error); + server::boot::systemd::notify_shutdown_failure(error); } joined?; info!("server shutdown complete"); diff --git a/core/server/src/partition_helpers.rs b/core/server/src/partition_helpers.rs index 91610720b7..f6ab9d16a2 100644 --- a/core/server/src/partition_helpers.rs +++ b/core/server/src/partition_helpers.rs @@ -15,7 +15,7 @@ // specific language governing permissions and limitations // under the License. -//! Helpers shared between the recovery path in [`crate::bootstrap`] and +//! Helpers shared between the recovery path in [`crate::boot`] and //! the runtime partition reconciliation loop. //! //! Recovery hydrates an [`IggyPartition`] from on-disk state; the diff --git a/core/server/src/server_error.rs b/core/server/src/server_error.rs index 53e84e46d6..12438e7100 100644 --- a/core/server/src/server_error.rs +++ b/core/server/src/server_error.rs @@ -187,7 +187,7 @@ pub enum ServerError { // // The Display text deliberately claims nothing about what happens to the // refused files: disposition (quarantine into `.fenced.N` vs tombstone - // with files left in place) is decided by the `bootstrap.rs` arms that + // with files left in place) is decided by the `boot/recovery.rs` arms that // catch this error, and only they log it -- a claim here would render // beside theirs and contradict one branch or the other. #[error( @@ -220,6 +220,8 @@ pub enum ServerError { }, #[error("cluster enabled but no node is configured for replica {replica_id}")] ClusterNodeNotFound { replica_id: u8 }, + #[error("server listeners start on shard 0 only, not on shard {shard_id}")] + ListenersOffShardZero { shard_id: u16 }, #[error("cluster node count {count} exceeds supported u8 replica count")] ClusterReplicaCountTooLarge { count: usize }, #[error("cluster mode requires --replica-id to identify the current node")] @@ -547,7 +549,7 @@ impl std::fmt::Display for PartitionRecoveryRefusal { } } -/// Per-shard outcome captured by [`crate::bootstrap::ShardHandles::join_all`] +/// Per-shard outcome captured by [`crate::boot::ShardHandles::join_all`] /// when a shard either returned `Err` or panicked. /// /// Bundled into [`ServerError::ShardJoinFailures`] so the operator sees diff --git a/core/server/src/shell.rs b/core/server/src/shell.rs index bc05b40a55..661c34b9ce 100644 --- a/core/server/src/shell.rs +++ b/core/server/src/shell.rs @@ -21,7 +21,7 @@ //! [`ShellBus`] bound, the [`ShellHandlers`] slot struct, and the //! `[cluster]` timer-to-tick translation every consensus group boots with. //! Everything here is type- and config-level; construction (wiring the -//! handlers against a live bus) stays in [`crate::bootstrap`]. +//! handlers against a live bus) stays in [`crate::boot`]. use crate::session_manager::SessionManager; use configs::server::ServerConfig; @@ -82,7 +82,7 @@ impl ShellBus for B {} /// [`SessionManager`] the request-plane pair shares. /// /// Both production (`build_shard_for_thread`) and the simulator's shell -/// mode construct these through [`crate::bootstrap::wire_shell_handlers`], +/// mode construct these through [`crate::boot::wire_shell_handlers`], /// so the request plane is wired one way. The simulator's shell-off fast /// path uses [`ShellHandlers::noop`] instead. pub struct ShellHandlers { diff --git a/core/simulator/src/replica.rs b/core/simulator/src/replica.rs index 8e57773eab..869dc61857 100644 --- a/core/simulator/src/replica.rs +++ b/core/simulator/src/replica.rs @@ -32,7 +32,7 @@ use metadata::stm::stream::{Streams, StreamsInner}; use metadata::stm::user::{Users, UsersInner}; use metadata::{IggyMetadata, apply_committed_prepare}; use partitions::{IggyPartitions, PartitionPathLayout, PartitionsConfig}; -use server::bootstrap::wire_shell_handlers; +use server::boot::wire_shell_handlers; use server::shell::{ShellHandlers, ShellShardHandle}; use server_common::crypto; use server_common::sharding::{METADATA_GROUP, ShardId}; From ad9e5ddce23ad5570b8b432c0f02ab19c9982ef0 Mon Sep 17 00:00:00 2001 From: Hubert Gruszecki Date: Wed, 2 Sep 2026 11:46:23 +0200 Subject: [PATCH 042/182] test(integration): catch on-disk storage format breaks pre-merge (#4003) --- .config/nextest.toml | 6 + .github/actions/rust/pre-merge/action.yml | 33 +- .../utils/setup-rust-with-cache/action.yml | 7 +- .github/config/components.yml | 37 + .github/workflows/_test.yml | 52 +- .github/workflows/pre-merge.yml | 2 + core/integration/src/harness/handle/server.rs | 11 + core/integration/tests/data_integrity/mod.rs | 5 + .../tests/data_integrity/storage_compat.rs | 1473 +++++++++++++++++ scripts/ci/storage-compat.sh | 270 +++ 10 files changed, 1886 insertions(+), 10 deletions(-) create mode 100644 core/integration/tests/data_integrity/storage_compat.rs create mode 100755 scripts/ci/storage-compat.sh diff --git a/.config/nextest.toml b/.config/nextest.toml index 3070df8bdc..1d98894330 100644 --- a/.config/nextest.toml +++ b/.config/nextest.toml @@ -15,6 +15,12 @@ # specific language governing permissions and limitations # under the License. +# Floor for the flags CI and scripts/ci/storage-compat.sh pass: --no-tests +# (0.9.75), --run-ignored only (0.9.76) and --ignore-default-filter (0.9.77). +# nextest refuses to run below it, and setup-rust-with-cache replaces a +# cache-restored binary that fails `cargo nextest show-config version`. +nextest-version = { required = "0.9.77" } + [[profile.default.overrides]] # This is a solution (or actually a workaround) for the problem that nextest does not support # #[serial] macro which shall enforce sequential execution of the test case. diff --git a/.github/actions/rust/pre-merge/action.yml b/.github/actions/rust/pre-merge/action.yml index 97efa4d3af..204811ebc9 100644 --- a/.github/actions/rust/pre-merge/action.yml +++ b/.github/actions/rust/pre-merge/action.yml @@ -20,7 +20,7 @@ description: Rust pre-merge testing and linting github iggy actions inputs: task: - description: "Task to run (check, check-msrv, fmt, clippy, sort, machete, doctest, verify-publish, test-1, test-2, test-3, miri)" + description: "Task to run (check, check-msrv, fmt, clippy, sort, machete, doctest, verify-publish, test-1, test-2, test-3, test-storage-compat, miri)" required: true component: description: "Component name (for context)" @@ -84,19 +84,22 @@ runs: # a subset of crates changed. # Safety: cargo check/clippy run on the full workspace separately, catching all # compilation errors. This only scopes test BUILD and EXECUTION. + # test-storage-compat is excluded from the DAG and coverage setup below: + # storage-compat.sh builds two explicit binaries and drives its own nextest + # run, and reads none of the /tmp plan files or the llvm-cov environment. - name: Fetch base branch for DAG analysis - if: startsWith(inputs.task, 'test-') + if: startsWith(inputs.task, 'test-') && inputs.task != 'test-storage-compat' run: git fetch origin master --depth=1 2>/dev/null || true shell: bash - name: Install cargo-rail - if: startsWith(inputs.task, 'test-') + if: startsWith(inputs.task, 'test-') && inputs.task != 'test-storage-compat' uses: taiki-e/install-action@v2 with: tool: cargo-rail - name: Compute affected crates (cargo-rail) - if: startsWith(inputs.task, 'test-') + if: startsWith(inputs.task, 'test-') && inputs.task != 'test-storage-compat' run: | METADATA_JSON=$(cargo metadata --format-version 1 --no-deps 2>/dev/null || echo "{}") TOTAL_CRATES=$(echo "$METADATA_JSON" | jq '.workspace_members | length' 2>/dev/null || echo "?") @@ -202,18 +205,25 @@ runs: shell: bash - name: Install cargo-llvm-cov - if: startsWith(inputs.task, 'test-') + if: startsWith(inputs.task, 'test-') && inputs.task != 'test-storage-compat' uses: taiki-e/install-action@v2 with: tool: cargo-llvm-cov - name: Build and test with coverage - if: startsWith(inputs.task, 'test-') + # test-storage-compat drives its own nextest invocation from + # scripts/ci/storage-compat.sh. Without this exclusion it would also run + # here with an empty PARTITION_FLAG, i.e. a fourth *unpartitioned* copy + # of the whole suite on top of test-1/2/3. + if: startsWith(inputs.task, 'test-') && inputs.task != 'test-storage-compat' run: | # Parse partition index from task name (test-1 -> hash:1/3, test-2 -> hash:2/3, ...). # TEST_PARTITIONS must match the number of test-N tasks in # .github/config/components.yml. Cluster bootstrap makes each test # CPU-heavy, so partitions stay small. + # The regex is numeric-only on purpose: non-numeric test-* tasks + # (test-storage-compat) opt out of partitioning rather than producing + # an out-of-range hash:N/3. TEST_PARTITIONS=3 TASK="${{ inputs.task }}" PARTITION_FLAG="" @@ -383,6 +393,17 @@ runs: ls -la codecov.json shell: bash + # On-disk format backwards compatibility. The script builds an iggy-server + # from the master tip under the checked-out merge ref (origin/master off a + # pull_request run) and one from HEAD, seeds a data directory with the + # former and reads it back with the latter. It does its own shallow fetch + # of the baseline commit and exits non-zero if it cannot resolve one, so + # the default depth-1 checkout is sufficient here. + - name: Storage format backwards compatibility + if: inputs.task == 'test-storage-compat' + run: ./scripts/ci/storage-compat.sh + shell: bash + # Miri (UB detector) on the unsafe-heavy crates that don't pull tokio / # compio. Pinned nightly so MIRIFLAGS behavior is stable across CI runs; # bump the date quarterly. Tree-borrows is the future-default aliasing diff --git a/.github/actions/utils/setup-rust-with-cache/action.yml b/.github/actions/utils/setup-rust-with-cache/action.yml index 82867a968e..3ca1dffc02 100644 --- a/.github/actions/utils/setup-rust-with-cache/action.yml +++ b/.github/actions/utils/setup-rust-with-cache/action.yml @@ -125,9 +125,12 @@ runs: - name: Install cargo-nextest if: runner.os == 'Linux' && inputs.install-nextest == 'true' run: | - if command -v cargo-nextest &> /dev/null; then + # A binary restored from the cargo cache can predate the floor in + # .config/nextest.toml. `show-config version` exits non-zero below it, + # and on a nextest too old to know the subcommand at all, so both fall + # through to a fresh install instead of aborting the lane later. + if command -v cargo-nextest &> /dev/null && cargo nextest show-config version; then echo "cargo-nextest already installed" - cargo nextest --version exit 0 fi diff --git a/.github/config/components.yml b/.github/config/components.yml index 082aec83b5..d63cdd54a7 100644 --- a/.github/config/components.yml +++ b/.github/config/components.yml @@ -135,6 +135,43 @@ components: - "core/message_bus/**" - "core/partitions/**" + # On-disk format backwards compatibility. Boots a server built from the + # master tip the PR merges onto, seeds a data directory, then swaps in a + # HEAD-built binary and asserts it reads everything back. Paths are the + # crates that own the on-disk representation plus the boot path that reads + # it back; a change to any of them can silently break rollforward for + # existing deployments, which no other check covers. Matching is literal + # glob with no crate-graph expansion, so every crate has to be spelled out: + # `core/server/**` does not cover `core/server_common/**`. + # `breaking:storage` on the PR skips it (see .github/workflows/_test.yml). + rust-storage-compat: + depends_on: + - "rust-workspace" + - "rust-configs" # partition/metadata on-disk schema knobs and their defaults + - "ci-infrastructure" # action.yml/workflow edits should re-run the check + paths: + - "core/server/**" + - "core/server_common/**" # per-message segment record + segment readers/writers + - "core/binary_protocol/**" + - "core/common/**" + - "core/consensus/**" # VsrState is the superblock payload + - "core/journal/**" + - "core/metadata/**" + - "core/partitions/**" + - "core/shard/**" # drives the shutdown flush and boot-time partition recovery + - "scripts/ci/storage-compat.sh" + # The check itself, so weakening it re-runs it. Scoped to the test, the + # module wiring that keeps it compiled in, the crate manifest and the + # harness it drives rather than `core/integration/**`, which would fire + # this lane on every unrelated integration test change. + - "core/integration/Cargo.toml" + - "core/integration/tests/mod.rs" + - "core/integration/tests/data_integrity/mod.rs" + - "core/integration/tests/data_integrity/storage_compat.rs" + - "core/integration/src/harness/**" + tasks: + - "test-storage-compat" + # Standalone simulation tool, does NOT affect server binary or foreign SDKs. # Split from rust-cluster to avoid triggering SDK tests on simulator-only changes. rust-simulator: diff --git a/.github/workflows/_test.yml b/.github/workflows/_test.yml index eb0483efb7..82a70444ba 100644 --- a/.github/workflows/_test.yml +++ b/.github/workflows/_test.yml @@ -33,6 +33,7 @@ on: permissions: contents: read + issues: read # listLabelsOnIssue, for the breaking:storage escape hatch security-events: write pull-requests: write @@ -55,15 +56,62 @@ jobs: run: echo "No changes detected, skipping tests" # Rust + # `breaking:storage` is the documented escape hatch for a deliberate + # on-disk format change, which the compat check cannot pass by + # construction. The label has to be read live: pre-merge.yml does not + # subscribe to `labeled` (pr-triage-apply.yml writes S-* labels, so every + # triage write would re-run the whole gate), and a re-run replays the + # webhook-frozen `pull_request.labels` from the last push. The operator + # flow is "check fails, apply label, re-run", which the frozen payload + # cannot see. Paginated so a PR with more than 100 labels still resolves. + - name: Resolve breaking:storage escape hatch + id: storage_hatch + if: >- + startsWith(inputs.component, 'rust') && + inputs.task == 'test-storage-compat' && + github.event.pull_request.number + uses: actions/github-script@v9 + with: + script: | + let skip = false; + try { + const labels = await github.paginate( + github.rest.issues.listLabelsOnIssue, + { + owner: context.repo.owner, + repo: context.repo.repo, + issue_number: context.payload.pull_request.number, + per_page: 100, + }, + ); + skip = labels.some(l => l.name === 'breaking:storage'); + } catch (e) { + // Fail toward running. A rate limit or a narrowed token must not + // wave an on-disk format break through as a green check. + core.warning(`listLabelsOnIssue failed (${e.status ?? e.code ?? 'network'}), running the check`); + } + core.info(`breaking:storage escape hatch: skip=${skip}`); + core.setOutput('skip', String(skip)); + - name: Run Rust task - if: startsWith(inputs.component, 'rust') + # Skipping the step (not the job) keeps the leg green so the pre-merge + # status roll-up stays reportable instead of pending. + if: >- + startsWith(inputs.component, 'rust') && + steps.storage_hatch.outputs.skip != 'true' uses: ./.github/actions/rust/pre-merge with: task: ${{ inputs.task }} component: ${{ inputs.component }} - name: Upload coverage to Codecov - if: startsWith(inputs.component, 'rust') && startsWith(inputs.task, 'test-') + # test-storage-compat runs its own nextest invocation without llvm-cov, + # so it emits no codecov.json. Uploading an empty report under the + # `rust` flag would read as a coverage drop on the PR. + if: >- + startsWith(inputs.component, 'rust') && + startsWith(inputs.task, 'test-') && + inputs.task != 'test-storage-compat' uses: codecov/codecov-action@v7.0.0 with: token: ${{ secrets.CODECOV_TOKEN }} diff --git a/.github/workflows/pre-merge.yml b/.github/workflows/pre-merge.yml index edc9452f58..aae32a00c6 100644 --- a/.github/workflows/pre-merge.yml +++ b/.github/workflows/pre-merge.yml @@ -31,6 +31,8 @@ concurrency: permissions: contents: read + # listLabelsOnIssue in _test.yml, for the breaking:storage escape hatch + issues: read security-events: write pull-requests: write diff --git a/core/integration/src/harness/handle/server.rs b/core/integration/src/harness/handle/server.rs index da6b75af33..bac9f6cb20 100644 --- a/core/integration/src/harness/handle/server.rs +++ b/core/integration/src/harness/handle/server.rs @@ -726,6 +726,17 @@ impl ServerHandle { self.test_transport = Some(transport); } + /// Point the next `start()` at a different server binary. + /// + /// `start()` re-reads `config.executable_path` on every call, so a test + /// can boot one build, then restart the SAME data directory under another + /// one. `None` restores the cargo-built binary of the crate under test, + /// which is why this takes an explicit `Option` rather than + /// `impl Into`. + pub fn set_executable_path(&mut self, path: Option) { + self.config.executable_path = path; + } + /// Configure MCP server for this iggy server. pub fn set_mcp_config(&mut self, config: McpConfig) { self.mcp = Some(McpHandle::with_server_id( diff --git a/core/integration/tests/data_integrity/mod.rs b/core/integration/tests/data_integrity/mod.rs index 2418ef39b5..3386feccb0 100644 --- a/core/integration/tests/data_integrity/mod.rs +++ b/core/integration/tests/data_integrity/mod.rs @@ -39,3 +39,8 @@ mod verify_cluster_replica_data_identical; // Auto-commit offset replication is inherently a multi-node (VSR) property: the // backup only holds the offset if the poll's auto-commit rode consensus. mod verify_auto_commit_offset_replicates; + +// On-disk format compatibility across a binary swap. `#[ignore]`d: it needs a +// baseline `iggy-server` built from the merge base, which only +// `scripts/ci/storage-compat.sh` provides. +mod storage_compat; diff --git a/core/integration/tests/data_integrity/storage_compat.rs b/core/integration/tests/data_integrity/storage_compat.rs new file mode 100644 index 0000000000..5c7ce66471 --- /dev/null +++ b/core/integration/tests/data_integrity/storage_compat.rs @@ -0,0 +1,1473 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! On-disk format backwards compatibility. +//! +//! A server built from the master tip the change merges onto (the BASELINE) +//! writes a rich data directory. The same directory is then restarted under +//! the build under test, and everything must read back unchanged: +//! +//! - A checkpointed metadata snapshot holding topics with every option key, +//! partitions with a deletion watermark and a purge generation, consumer +//! groups, users with permissions and personal access tokens, all created +//! BEFORE the checkpoint fires, plus an uncheckpointed WAL tail holding one +//! more of each, created after it. The snapshot and the WAL encode the same +//! types differently, and either can break on its own. +//! - Sealed and active segments in a non-zero partition, holding messages +//! with explicit ids, typed user headers and payload lengths that vary +//! inside every batch, and a partial segment deletion. +//! - Individual and group consumer offsets, and a purge generation with +//! messages appended past it. +//! +//! The swap works because `ServerHandle::start` re-reads +//! `config.executable_path` on every call while the data directory, the +//! ports and the harness paths all survive a `stop`. +//! +//! # Why every assertion here is about DATA +//! +//! Three failure modes reach exit code 0 with a server that boots and serves: +//! +//! - A partition whose durable state cannot be read is TOMBSTONED and boot +//! continues. Only the log says so. +//! - A topic loses a single storage option, per key, with no log line at +//! all: `TopicCreateOptions::parse_committed` deliberately discards +//! per-entry parse errors so one unreadable key cannot drop the rest. +//! - A segment misparse can truncate a torn tail while the messages a test +//! happens to poll still read back fine, because segment batch headers +//! carry no magic and no version. +//! +//! So the test asserts the tombstone marker is absent, re-reads every seeded +//! option value AND its provenance flag, compares every stored message field +//! against a read taken from the baseline, and byte-compares the segment +//! files across the swap. It stops producing MID-segment on purpose: boot +//! unseals the last segment, so a chain that ends on a rotation boundary +//! would only ever hand that path an empty file. +//! +//! # Structural false positive to avoid +//! +//! The harness forwards every parent `IGGY_*` variable to the server child, +//! and the server treats an unknown `IGGY_*` name as a `debug_assert`. A pull +//! request that adds a config leaf AND a test that sets it therefore makes +//! the BASELINE binary die on startup, which looks like a compatibility +//! break. Nothing here detects that: `resolve_config_paths` validates the +//! name against the catalog of the build under test, which is the wrong side +//! of the swap, so it only proves the name exists on HEAD. Keeping every +//! override to a name that already exists on the merge base stays a rule the +//! author has to follow by hand. +//! +//! `IGGY_TEST_VERBOSE` makes the harness inherit the server's stdout instead +//! of capturing it, which would make the tombstone check vacuous. The +//! graceful-shutdown assertion below is taken on the same log and fails +//! loudly in that case rather than passing on an empty file. + +use bytes::Bytes; +use iggy::prelude::*; +use integration::harness::{ + TestHarness, TestServerConfig, USER_PASSWORD, disk, resolve_config_paths, +}; +use serial_test::parallel; +use std::collections::{BTreeMap, HashMap}; +use std::fs; +use std::path::{Path, PathBuf}; +use std::str::FromStr; +use std::time::{Duration, Instant}; +use tokio::time::sleep; + +/// Absolute path to the baseline `iggy-server` binary. See +/// [`baseline_server_binary`] for why a relative value is refused. +/// +/// NOT `IGGY_`-prefixed on purpose: the harness forwards every parent +/// `IGGY_*` variable to the server child, where an unknown name trips a +/// `debug_assert` and kills the debug build this test runs against. +const BASELINE_SERVER_ENV: &str = "COMPAT_BASELINE_SERVER"; + +/// `metadata.journal_slots` floor for `prepare_queue_depth = 32` +/// (`4 * max(64, 32)`). With the 64-op checkpoint margin a checkpoint fires +/// once the journal holds 192 committed ops. +const JOURNAL_SLOTS: &str = "256"; + +/// Committed metadata ops between forced checkpoints at [`JOURNAL_SLOTS`]. +const CHECKPOINT_EVERY: u32 = 192; + +/// Stream creates that drive the metadata plane past its first checkpoint. +/// Everything seeded before them lands in the snapshot. +const SEED_STREAMS: u32 = 250; + +/// Stream creates issued after the checkpoint, so recovery must fold a +/// snapshot AND replay a WAL suffix on top of it. +const WAL_TAIL_STREAMS: u32 = 5; + +const DATA_STREAM: &str = "compat-data"; +const DATA_TOPIC: &str = "data"; +/// Created after the checkpoint, so its options only exist as a WAL record. +const WAL_TAIL_TOPIC: &str = "tail"; +const PURGE_STREAM: &str = "compat-purge"; +const PURGE_TOPIC: &str = "purged"; +const OFFSET_CONSUMER: &str = "compat-offset-consumer"; +const READBACK_CONSUMER: &str = "compat-readback"; +const OFFSET_GROUP: &str = "compat-offset-group"; +const TRANSIENT_GROUP: &str = "compat-transient-group"; +const COMPAT_USER: &str = "compat-user"; +const COMPAT_PAT: &str = "compat-pat"; +const WAL_TAIL_USER: &str = "compat-tail-user"; +const WAL_TAIL_PAT: &str = "compat-tail-pat"; + +/// Partitions per seeded topic. Non-default (the default is 1), so the count +/// is itself a recovered value, and more than one so [`SEEDED_PARTITION`] +/// can be non-zero. +const PARTITIONS_COUNT: u32 = 2; + +/// The partition every message, offset and purge check targets. Non-zero on +/// purpose: the segment batch header carries the partition id, and a field +/// that decoded as 0 is indistinguishable from a correct one when the only +/// seeded partition is 0. Partition 0 stays empty. +const SEEDED_PARTITION: u32 = 1; + +/// Exactly `MIN_TOPIC_SEGMENT_SIZE`, and a 512-byte multiple, so it passes +/// create-time validation while keeping the produced data small. +const SEGMENT_SIZE_BYTES: u64 = 1024 * 1024; + +/// Byte flush threshold. Together with `messages_required_to_save = 1` it +/// forces every committed message onto a segment file, instead of leaving it +/// in the in-memory journal until the shutdown flush. +const FLUSH_SIZE_BYTES: u64 = 4096; + +/// Seeded `message_expiry`, far enough out that nothing expires mid-run. +/// +/// The value matters only in that it is NOT the default (`NeverExpire`): a key +/// lost across the swap is re-derived at its default, so seeding the default +/// would make the assertion pass either way. +const MESSAGE_EXPIRY_SECS: u64 = 30 * 24 * 60 * 60; +const WAL_TAIL_MESSAGE_EXPIRY_SECS: u64 = 7 * 24 * 60 * 60; + +/// Seeded `max_topic_size`. Non-default for the same reason (the default is +/// `Unlimited`), and far above the few MiB this test produces so retention +/// never trims a segment. +const MAX_TOPIC_SIZE_BYTES: u64 = 512 * 1024 * 1024; +const WAL_TAIL_MAX_TOPIC_SIZE_BYTES: u64 = 256 * 1024 * 1024; + +/// Base payload length; [`payload_for`] adds [`PAYLOAD_STEP`] per position +/// inside a batch. +const PAYLOAD_SIZE: usize = 16 * 1024; +const PAYLOAD_STEP: usize = 64; + +/// Messages a segment holds before the append that crosses +/// [`SEGMENT_SIZE_BYTES`] seals it. One [`SEND_BATCH`] costs 272304 bytes on +/// disk: a 256-byte batch header, then per message a 48-byte frame header, +/// the 91 bytes of [`user_headers_for`] and the payload, whose lengths sum +/// to 269824 over the 16 positions of [`payload_for`]. Three batches stay +/// under 1 MiB, the fourth crosses it. +const MESSAGES_PER_SEGMENT: u64 = 4 * SEND_BATCH; + +/// Sealed segments the seed leaves behind, before the deletion in step 3. +const SEALED_SEGMENTS: u64 = 5; + +/// Messages left in the ACTIVE segment once production stops. +/// +/// Non-zero on purpose. Boot unseals the last segment, and a run that stopped +/// on a segment boundary reopens an EMPTY file, so neither the unseal nor a +/// torn-tail truncation of a partially written segment is ever exercised. +/// Not a multiple of [`SEND_BATCH`] either, so the final batch is short. +const ACTIVE_SEGMENT_MESSAGES: u64 = 10; + +const PRODUCED_MESSAGES: u64 = SEALED_SEGMENTS * MESSAGES_PER_SEGMENT + ACTIVE_SEGMENT_MESSAGES; + +const SEND_BATCH: u64 = 16; + +/// Segment logs (the sealed ones plus the active one) the seed waits for +/// before deleting one, so a sealed segment is still left behind. +const MIN_SEGMENTS_BEFORE_DELETE: usize = 4; + +/// Messages produced into the topic that is purged afterwards. +const PURGED_MESSAGES: u64 = 8; + +/// Messages appended to the purged topic AFTER the purge, at generation 1. +/// +/// Without them the purge check is vacuous. An unreadable `purge.gen` reads +/// as 0 on purpose (the absent-or-torn sentinel), the reconciler then +/// re-purges the partition, and an empty topic passes a `messages_count == 0` +/// assertion whether or not the generation decoded. Only messages that must +/// SURVIVE the swap can tell the two apart. +const POST_PURGE_MESSAGES: u64 = 8; + +/// Seeded personal access token expiry. Non-default (the default is +/// `NeverExpire`, which reads back as `None`) so a dropped or misread field +/// cannot pass as the default. +const PAT_EXPIRY_SECS: u64 = 7 * 24 * 60 * 60; +const WAL_TAIL_PAT_EXPIRY_SECS: u64 = 3 * 24 * 60 * 60; + +/// Contiguous messages re-polled after the swap. +const READBACK_COUNT: u32 = 16; + +/// Bound on every [`wait_until`] probe loop. +/// +/// Kept small because the whole test has to finish inside nextest's 300s hard +/// kill (`slow-timeout` x `terminate-after` in `.config/nextest.toml`, and the +/// driving script deliberately runs without `--profile ci`). Two boots at 60s +/// plus two stops at 5s plus seven of these waits is 200s. A SIGKILL past +/// that budget would take the named `wait_until` message with it, losing the +/// diagnostic in exactly the run that needed it. At a 200ms +/// [`POLL_INTERVAL`] this is still ~50 probes per wait. +const SETTLE_TIMEOUT: Duration = Duration::from_secs(10); +const POLL_INTERVAL: Duration = Duration::from_millis(200); + +/// Substring shared by all three tombstone log lines in +/// `core/server/src/bootstrap.rs`. Matching the full sentence of any one of +/// them would miss the others, and a segment layout change lands on the +/// first arm, not the consensus-state one. +/// +/// Being shorter than a whole word is what makes it match every arm, and also +/// what makes it collide with [`TOMBSTONE_FIELD`]. Strip that before matching +/// rather than lengthening this. +const TOMBSTONE_MARKER: &str = "tombston"; + +/// The one tracing FIELD in the server named for the tombstone state +/// (`partition_reconciler.rs`, the parked-frame discard). Fields render as +/// `key=value`, so a shard that is merely not the namespace's owner logs +/// `tombstoned=false`, which contains [`TOMBSTONE_MARKER`] on a HEALTHY run. +/// +/// Removed from the log before the absence check. Only this exact field form +/// goes: every real arm reads `tombstoning it`, `tombstoned rather than` or +/// `tombstoning the`, and none of them emits a `tombstoned` field, so their +/// detection is untouched. +/// +/// The collision is dormant at the default `info` level, but the line is +/// `debug!`, and two independent variables reach the child: the harness +/// forwards every `IGGY_*` name, and `RUST_LOG` is inherited by any process +/// and OUTRANKS the configured level. Filtering the text keeps the check +/// working under both instead of refusing to run. +const TOMBSTONE_FIELD: &str = "tombstoned="; + +/// Logged by `iggy-server`'s `main` once shutdown ran to completion. A stop +/// that escalated to `SIGKILL` leaves crash-recovery state behind, which +/// would make the next boot exercise crash recovery and compatibility at the +/// same time. +const GRACEFUL_SHUTDOWN_MARKER: &str = "server shutdown complete"; + +#[tokio::test] +#[parallel] +#[ignore = "needs a baseline iggy-server built from master; run via scripts/ci/storage-compat.sh"] +async fn should_read_back_a_data_directory_written_by_the_baseline_server() { + let baseline = baseline_server_binary(); + let envs = resolve_config_paths(&HashMap::from([( + "metadata.journal_slots".to_string(), + JOURNAL_SLOTS.to_string(), + )])) + .expect("metadata.journal_slots resolves against the live config catalog"); + + let mut harness = TestHarness::builder() + .cluster_nodes(1) + .server( + TestServerConfig::builder() + .executable_path(baseline.clone()) + .extra_envs(envs) + .build(), + ) + .build() + .unwrap(); + harness.start().await.unwrap(); + + let data_path = harness.server().data_path(); + let client = harness.tcp_root_client().await.unwrap(); + + // 1. The topic carrying the segment chain. The flush thresholds are load + // bearing: without them committed messages stay in the journal and the + // partition directory looks empty until shutdown. + let data_stream_details = client.create_stream(DATA_STREAM).await.unwrap(); + let data_stream = Identifier::numeric(data_stream_details.id).unwrap(); + let data_topic_details = client + .create_topic(&data_stream, DATA_TOPIC, &data_topic_options()) + .await + .unwrap(); + let data_topic = Identifier::numeric(data_topic_details.id).unwrap(); + let partition = partition_dir( + &data_path, + data_stream_details.id, + data_topic_details.id, + SEEDED_PARTITION, + ); + + // 2. Produce past [`SEALED_SEGMENTS`] rotations and then stop MID-segment. + // A segment rolls on the append that crosses the target, so the + // crossing batch lands whole and the tail batch stays in the active one. + let mut produced = 0u64; + while produced < PRODUCED_MESSAGES { + let batch_end = (produced + SEND_BATCH).min(PRODUCED_MESSAGES); + let mut batch: Vec = (produced..batch_end) + .map(|offset| seeded_message(offset, payload_for(offset))) + .collect(); + client + .send_messages( + &data_stream, + &data_topic, + &Partitioning::partition_id(SEEDED_PARTITION), + &mut batch, + ) + .await + .unwrap_or_else(|error| panic!("send messages {produced}..{batch_end}: {error}")); + produced = batch_end; + } + + wait_until( + "the produced messages to seal enough segments", + async || { + let logs = segment_logs(&partition); + if logs.len() >= MIN_SEGMENTS_BEFORE_DELETE { + Ok(()) + } else { + Err(format!( + "{} holds {} segment log(s) ({logs:?}), need {MIN_SEGMENTS_BEFORE_DELETE}", + partition.display(), + logs.len() + )) + } + }, + ) + .await; + let before_delete = segment_logs(&partition); + + // 3. Delete the oldest sealed segment, BEFORE any consumer offset exists: + // a committed offset is a retention barrier the removal stops at. The + // ack only records the truncation watermark; the reconciler removes the + // files on a later commit-driven pass. + client + .delete_segments(&data_stream, &data_topic, SEEDED_PARTITION, 1) + .await + .unwrap(); + let deleted_segment = before_delete[0].clone(); + wait_until("the deleted segment to leave disk", async || { + let logs = segment_logs(&partition); + if logs.contains(&deleted_segment) { + Err(format!( + "{deleted_segment} is still present in {} ({logs:?})", + partition.display() + )) + } else { + Ok(()) + } + }) + .await; + + let retained = segment_logs(&partition); + assert!( + retained.len() >= 2, + "the seed must keep at least one sealed segment beside the active one, got {retained:?}" + ); + let first_retained_offset = base_offset_of(&retained[0]); + assert!( + first_retained_offset > 0, + "deleting the oldest segment must move the partition's first offset off zero, \ + segments before={before_delete:?} after={retained:?}" + ); + + // 4. Both offset flavours, so `offsets/consumers/` and `offsets/groups/` + // are populated. + let offset_consumer = Consumer::new(Identifier::named(OFFSET_CONSUMER).unwrap()); + client + .store_consumer_offset( + &offset_consumer, + &data_stream, + &data_topic, + Some(SEEDED_PARTITION), + first_retained_offset, + ) + .await + .unwrap(); + + client + .create_consumer_group(&data_stream, &data_topic, OFFSET_GROUP) + .await + .unwrap(); + // Storing a GROUP offset is a partition op gated on membership: a caller + // that only created the group does not own the partition at the current + // generation, so the namespace resolve fails and the server replies + // ResourceNotFound with an empty body. + client + .join_consumer_group( + &data_stream, + &data_topic, + &Identifier::named(OFFSET_GROUP).unwrap(), + ) + .await + .unwrap(); + let offset_group = Consumer::group(Identifier::named(OFFSET_GROUP).unwrap()); + client + .store_consumer_offset( + &offset_group, + &data_stream, + &data_topic, + Some(SEEDED_PARTITION), + first_retained_offset, + ) + .await + .unwrap(); + + for kind in ["consumers", "groups"] { + let dir = partition.join("offsets").join(kind); + wait_until(&format!("the {kind} offset to reach disk"), async || { + let count = fs::read_dir(&dir) + .map(|entries| entries.flatten().count()) + .unwrap_or(0); + if count > 0 { + Ok(()) + } else { + Err(format!("{} holds no offset file", dir.display())) + } + }) + .await; + } + + // 5. A purge on a SEPARATE topic: it resets the partition and empties its + // segment chain, so it must not touch the one above. + let purge_stream_details = client.create_stream(PURGE_STREAM).await.unwrap(); + let purge_stream = Identifier::numeric(purge_stream_details.id).unwrap(); + let purge_topic_details = client + .create_topic(&purge_stream, PURGE_TOPIC, &purge_topic_options()) + .await + .unwrap(); + let purge_topic = Identifier::numeric(purge_topic_details.id).unwrap(); + let mut purged_batch: Vec = (0..PURGED_MESSAGES) + .map(|index| seeded_message(index, Bytes::from(format!("compat-purged-{index}")))) + .collect(); + client + .send_messages( + &purge_stream, + &purge_topic, + &Partitioning::partition_id(SEEDED_PARTITION), + &mut purged_batch, + ) + .await + .unwrap(); + client + .purge_topic(&purge_stream, &purge_topic) + .await + .unwrap(); + + let purge_generation = partition_dir( + &data_path, + purge_stream_details.id, + purge_topic_details.id, + SEEDED_PARTITION, + ) + .join("purge.gen"); + wait_until("the purge generation to reach disk", async || { + if purge_generation.is_file() { + Ok(()) + } else { + Err(format!("{} does not exist", purge_generation.display())) + } + }) + .await; + + // Appended once the generation is durable, so they sit at offset 0 of the + // reset chain and only survive a boot that decodes `purge.gen`. + let mut post_purge_batch: Vec = (0..POST_PURGE_MESSAGES) + .map(|index| seeded_message(index, post_purge_payload(index))) + .collect(); + client + .send_messages( + &purge_stream, + &purge_topic, + &Partitioning::partition_id(SEEDED_PARTITION), + &mut post_purge_batch, + ) + .await + .unwrap(); + + // 6. A user with permissions at every level, and a personal access token. + let permissions = seeded_permissions(data_stream_details.id, data_topic_details.id); + client + .create_user( + COMPAT_USER, + USER_PASSWORD, + UserStatus::Active, + Some(permissions.clone()), + ) + .await + .unwrap(); + let seeded_user = client + .get_user(&Identifier::named(COMPAT_USER).unwrap()) + .await + .unwrap() + .expect("the baseline lists the user it just created"); + assert_eq!( + seeded_user.permissions, + Some(permissions.clone()), + "the baseline must report the permissions exactly as seeded, or a post-swap mismatch \ + could not be attributed to the swap" + ); + let pat = create_token(&client, COMPAT_PAT, PAT_EXPIRY_SECS).await; + + // 7. Drive the metadata plane past its first forced checkpoint. Everything + // above is now encoded in `snapshot.bin`, which the second boot must + // fold in as its recovery floor; everything below only exists as WAL + // records replayed on top of it. + for index in 0..SEED_STREAMS { + client + .create_stream(&format!("compat-seed-{index}")) + .await + .unwrap_or_else(|error| panic!("create seed stream {index}: {error}")); + } + + let snapshot_path = data_path.join("metadata").join("snapshot.bin"); + wait_until( + "the metadata checkpoint to persist a snapshot", + async || match fs::metadata(&snapshot_path).map(|meta| meta.len()) { + Ok(len) if len > 0 => Ok(()), + other => Err(format!( + "{SEED_STREAMS} committed stream creates must cross the {CHECKPOINT_EVERY}-op \ + checkpoint and persist a non-empty snapshot at {}, got {other:?}", + snapshot_path.display() + )), + }, + ) + .await; + + // 8. One more of each rich type past the checkpoint, so the WAL encodings + // are exercised as well: a topic with every option key, a user with + // permissions, a token, a group whose membership churned, and streams. + let tail_topic_details = client + .create_topic(&data_stream, WAL_TAIL_TOPIC, &wal_tail_topic_options()) + .await + .unwrap(); + let tail_permissions = seeded_permissions(data_stream_details.id, tail_topic_details.id); + client + .create_user( + WAL_TAIL_USER, + USER_PASSWORD, + UserStatus::Active, + Some(tail_permissions.clone()), + ) + .await + .unwrap(); + let tail_pat = create_token(&client, WAL_TAIL_PAT, WAL_TAIL_PAT_EXPIRY_SECS).await; + client + .create_consumer_group(&data_stream, &data_topic, TRANSIENT_GROUP) + .await + .unwrap(); + let transient_group = Identifier::named(TRANSIENT_GROUP).unwrap(); + client + .join_consumer_group(&data_stream, &data_topic, &transient_group) + .await + .unwrap(); + client + .leave_consumer_group(&data_stream, &data_topic, &transient_group) + .await + .unwrap(); + + for index in 0..WAL_TAIL_STREAMS { + client + .create_stream(&format!("compat-tail-{index}")) + .await + .unwrap_or_else(|error| panic!("create tail stream {index}: {error}")); + } + + // 9. What the baseline serves, captured for a field-by-field comparison + // after the swap. Checked against the seed first, so a post-swap + // mismatch can be attributed to the swap. + let expected_streams = stream_catalog(&client).await; + assert_eq!( + expected_streams.len(), + (SEED_STREAMS + WAL_TAIL_STREAMS + 2) as usize, + "the seed must have created every stream before the swap" + ); + let readback = Consumer::new(Identifier::named(READBACK_CONSUMER).unwrap()); + let last_offset = PRODUCED_MESSAGES - 1; + let retained_before = poll_window( + &client, + &data_stream, + &data_topic, + &readback, + first_retained_offset, + READBACK_COUNT, + ) + .await; + assert_seeded_messages( + "the retained window", + &retained_before, + first_retained_offset, + READBACK_COUNT as usize, + payload_for, + ); + let tail_before = poll_window( + &client, + &data_stream, + &data_topic, + &readback, + last_offset, + 1, + ) + .await; + assert_seeded_messages( + "the last produced message", + &tail_before, + last_offset, + 1, + payload_for, + ); + let post_purge_before = poll_window( + &client, + &purge_stream, + &purge_topic, + &readback, + 0, + READBACK_COUNT, + ) + .await; + assert_seeded_messages( + "the post-purge messages", + &post_purge_before, + 0, + POST_PURGE_MESSAGES as usize, + post_purge_payload, + ); + + drop(client); + harness.stop().await.unwrap(); + assert_graceful_shutdown(&harness, "the baseline server"); + + let baseline_files = disk::collect_comparable_files(&data_path, false); + assert!( + baseline_files + .keys() + .any(|rel| rel.starts_with("streams/") && rel.ends_with(".log")), + "the baseline wrote no segment .log under {}, so the byte comparison would be vacuous \ + (found: {:?})", + data_path.display(), + baseline_files.keys().collect::>() + ); + + // The active segment must be PARTIALLY filled: anything at or above the + // target is a sealed one. Boot unseals the last segment, so an empty + // active segment leaves both the reopen and any torn-tail truncation the + // comparison below would catch unreachable. + let active_segment = retained.last().expect("the retained chain is non-empty"); + let active_relative = partition + .strip_prefix(&data_path) + .expect("the partition directory sits under the data directory") + .join(active_segment) + .to_string_lossy() + .replace('\\', "/"); + let active_bytes = baseline_files + .get(&active_relative) + .unwrap_or_else(|| panic!("the baseline wrote no {active_relative}")) + .len() as u64; + assert!( + active_bytes > 0 && active_bytes < SEGMENT_SIZE_BYTES, + "{active_segment} holds {active_bytes} byte(s), so it is not a partially filled active \ + segment. Production must stop between two rotations: PRODUCED_MESSAGES \ + ({PRODUCED_MESSAGES}) must not be a multiple of MESSAGES_PER_SEGMENT \ + ({MESSAGES_PER_SEGMENT}), and the on-disk batch cost the latter is derived from must \ + still hold. Segments: {retained:?}" + ); + + // The swap: `None` selects the cargo-built binary of the crate under test. + harness.server_mut().set_executable_path(None); + harness.restart_server().await.unwrap(); + + // `tcp_root_client` hands out a client the harness does not own, so the + // reconnect loop inside `restart_server` never touched it. Take a fresh one. + let client = harness.tcp_root_client().await.unwrap(); + + // Ids AND names: later reads resolve by numeric id, so a corrupted name + // would pass a bare count. + assert_eq!( + stream_catalog(&client).await, + expected_streams, + "streams written by the baseline must all recover with their ids and names, both the \ + checkpointed prefix and the WAL tail" + ); + + assert_topic_recovered(&client, &data_stream, DATA_TOPIC, &data_topic_options()).await; + assert_topic_recovered( + &client, + &data_stream, + WAL_TAIL_TOPIC, + &wal_tail_topic_options(), + ) + .await; + assert_topic_recovered(&client, &purge_stream, PURGE_TOPIC, &purge_topic_options()).await; + + assert_eq!( + segment_logs(&partition), + retained, + "the segment chain must recover exactly as the baseline left it" + ); + + assert_messages_identical( + "the retained window", + &retained_before, + &poll_window( + &client, + &data_stream, + &data_topic, + &readback, + first_retained_offset, + READBACK_COUNT, + ) + .await, + ); + assert_messages_identical( + "the last produced message", + &tail_before, + &poll_window( + &client, + &data_stream, + &data_topic, + &readback, + last_offset, + 1, + ) + .await, + ); + + let stored = client + .get_consumer_offset( + &offset_consumer, + &data_stream, + &data_topic, + Some(SEEDED_PARTITION), + ) + .await + .unwrap() + .expect("the individual consumer offset survives the swap"); + assert_eq!( + stored.stored_offset, first_retained_offset, + "individual consumer offset" + ); + let stored_group = client + .get_consumer_offset( + &offset_group, + &data_stream, + &data_topic, + Some(SEEDED_PARTITION), + ) + .await + .unwrap() + .expect("the group consumer offset survives the swap"); + assert_eq!( + stored_group.stored_offset, first_retained_offset, + "group consumer offset" + ); + assert_eq!( + disk::read_replicated_consumer_offset(&data_path), + Some(first_retained_offset), + "the on-disk offset record must decode to the offset the baseline stored" + ); + + let purged = client + .get_topic(&purge_stream, &purge_topic) + .await + .unwrap() + .expect("the purged topic survives the swap"); + assert_eq!( + purged.messages_count, POST_PURGE_MESSAGES, + "the purged topic must hold exactly the messages appended after the purge: fewer means \ + `purge.gen` read back as 0 and the partition was purged again, more means the purge \ + itself was lost" + ); + assert_messages_identical( + "the post-purge messages", + &post_purge_before, + &poll_window( + &client, + &purge_stream, + &purge_topic, + &readback, + 0, + READBACK_COUNT, + ) + .await, + ); + + // A generation that decoded ABOVE the committed one passes both checks + // above, nothing was re-purged, while parking the partition past every + // purge the topic will ever commit: the reconciler stages a reset only + // for `committed > applied`. So a purge issued under the build under test + // must still take effect. + client + .purge_topic(&purge_stream, &purge_topic) + .await + .unwrap(); + wait_until("the post-swap purge to empty the topic", async || { + let topic = client + .get_topic(&purge_stream, &purge_topic) + .await + .map_err(|error| error.to_string())? + .ok_or_else(|| "the purged topic is gone".to_string())?; + if topic.messages_count == 0 { + Ok(()) + } else { + Err(format!( + "messages_count is still {}: the reconciler staged no reset, so the recovered \ + generation must sit at or above the newly committed one", + topic.messages_count + )) + } + }) + .await; + let after_purge = poll_window( + &client, + &purge_stream, + &purge_topic, + &readback, + 0, + READBACK_COUNT, + ) + .await; + assert!( + after_purge.is_empty(), + "offset 0 of the purged topic must be empty after the post-swap purge, got {} message(s)", + after_purge.len() + ); + + assert_user_recovered(&harness, &client, COMPAT_USER, &permissions).await; + assert_user_recovered(&harness, &client, WAL_TAIL_USER, &tail_permissions).await; + assert_token_recovered(&harness, &client, COMPAT_PAT, &pat).await; + assert_token_recovered(&harness, &client, WAL_TAIL_PAT, &tail_pat).await; + + let groups = client + .get_consumer_groups(&data_stream, &data_topic) + .await + .unwrap(); + let group_names: Vec<&str> = groups.iter().map(|group| group.name.as_str()).collect(); + assert!( + group_names.contains(&OFFSET_GROUP) && group_names.contains(&TRANSIENT_GROUP), + "both consumer groups must survive the swap, got {group_names:?}" + ); + + drop(client); + harness.stop().await.unwrap(); + + // Taken after the readback so the non-blocking stdout appender has had the + // whole run to drain, and after a positive marker so an uncaptured log + // fails loudly instead of passing the absence check vacuously. + assert_graceful_shutdown(&harness, "the server under test"); + // `stdout_plain` returns the same ANSI-stripped text `stdout_contains` + // matches against, so filtering it costs no matching power. + let stdout = harness.server().stdout_plain(); + assert!( + !stdout + .replace(TOMBSTONE_FIELD, "") + .contains(TOMBSTONE_MARKER), + "the server under test tombstoned a partition rather than reading the baseline's \ + durable state; boot still exits 0, so only the log says so. Server stdout:\n{stdout}" + ); + + // The purged topic was rewritten by the post-swap purge on purpose, so + // only the rest of the tree is held to byte identity. + assert_segments_identical( + &baseline_files, + &disk::collect_comparable_files(&data_path, false), + &topic_prefix(purge_stream_details.id, purge_topic_details.id), + ); +} + +/// Resolve the baseline binary to an absolute, existing path. +/// +/// The absolute requirement is not cosmetic. `ServerHandle::start` only spawns +/// the configured path directly when it has more than one component or already +/// exists; a bare name that does not exist falls through to +/// `Command::cargo_bin`, which resolves the binary of the crate under test. So +/// a single-component value would boot the HEAD build while the test reported +/// it as the baseline, comparing master against master and passing forever. +/// The same silent-green family as running a stale binary. +fn baseline_server_binary() -> String { + let raw = std::env::var(BASELINE_SERVER_ENV).unwrap_or_else(|_| { + panic!( + "{BASELINE_SERVER_ENV} is unset. It must be an ABSOLUTE path to the iggy-server \ + binary built from master. Without it this test would boot the build under test \ + twice and prove nothing, so it refuses to run. Use scripts/ci/storage-compat.sh, \ + which builds the baseline and exports the variable." + ) + }); + + let path = PathBuf::from(&raw); + assert!( + path.is_absolute(), + "{BASELINE_SERVER_ENV}={raw:?} is relative. It must be an ABSOLUTE path: a bare binary \ + name that does not exist on disk is silently resolved as the build under test, which \ + would compare master against master and report green forever." + ); + let resolved = fs::canonicalize(&path).unwrap_or_else(|error| { + panic!( + "{BASELINE_SERVER_ENV}={raw:?} does not resolve: {error}. It must point at an \ + iggy-server binary built from master." + ) + }); + assert!( + resolved.is_file(), + "{BASELINE_SERVER_ENV}={raw:?} resolves to {}, which is not a file.", + resolved.display() + ); + resolved.display().to_string() +} + +/// Options of the topic that carries the segment chain. Every key but +/// `compression_algorithm` is sent, so that one must come back derived while +/// the rest come back explicit. +fn data_topic_options() -> TopicCreateOptions { + TopicCreateOptions { + partitions_count: Some(PARTITIONS_COUNT), + message_expiry: Some(IggyExpiry::ExpireDuration(IggyDuration::new_from_secs( + MESSAGE_EXPIRY_SECS, + ))), + max_topic_size: Some(MaxTopicSize::Custom(IggyByteSize::from( + MAX_TOPIC_SIZE_BYTES, + ))), + segment_size: Some(IggyByteSize::from(SEGMENT_SIZE_BYTES)), + enforce_fsync: Some(true), + messages_required_to_save: Some(1), + size_of_messages_required_to_save: Some(IggyByteSize::from(FLUSH_SIZE_BYTES)), + // Left at the default, so only its provenance flag can tell a + // surviving key from a re-derived one. Turning it on would reserve + // the whole segment up front, which changes the shape of the active + // segment the byte comparison certifies. + preallocate_segments: Some(false), + ..TopicCreateOptions::default() + } +} + +/// Options of the topic created after the checkpoint. Different values from +/// [`data_topic_options`] where a key has a usable one, so a record attributed +/// to the wrong topic cannot pass, and a different unsent key +/// (`enforce_fsync`) for the provenance check. +fn wal_tail_topic_options() -> TopicCreateOptions { + TopicCreateOptions { + partitions_count: Some(PARTITIONS_COUNT), + compression_algorithm: Some(CompressionAlgorithm::Gzip), + message_expiry: Some(IggyExpiry::ExpireDuration(IggyDuration::new_from_secs( + WAL_TAIL_MESSAGE_EXPIRY_SECS, + ))), + max_topic_size: Some(MaxTopicSize::Custom(IggyByteSize::from( + WAL_TAIL_MAX_TOPIC_SIZE_BYTES, + ))), + segment_size: Some(IggyByteSize::from(2 * SEGMENT_SIZE_BYTES)), + messages_required_to_save: Some(1), + size_of_messages_required_to_save: Some(IggyByteSize::from(2 * FLUSH_SIZE_BYTES)), + preallocate_segments: Some(false), + ..TopicCreateOptions::default() + } +} + +/// Options of the topic that is purged. `compression_algorithm` is stored +/// topic metadata only (the server compresses nothing), so Gzip is just a +/// non-default value that must round-trip. +fn purge_topic_options() -> TopicCreateOptions { + TopicCreateOptions { + partitions_count: Some(PARTITIONS_COUNT), + compression_algorithm: Some(CompressionAlgorithm::Gzip), + messages_required_to_save: Some(1), + ..TopicCreateOptions::default() + } +} + +/// Deterministic message for `offset`: an explicit id, typed user headers and +/// the given payload, so a readback can name exactly which message diverged. +fn seeded_message(offset: u64, payload: Bytes) -> IggyMessage { + IggyMessage::builder() + .id(message_id_for(offset)) + .payload(payload) + .user_headers(user_headers_for(offset)) + .build() + .unwrap() +} + +/// Both 64-bit halves non-zero and unequal, so a swapped, truncated or zeroed +/// half cannot pass. The builder's default id is 0. +fn message_id_for(offset: u64) -> u128 { + ((u128::from(offset) + 1) << 64) | u128::from(u64::MAX - offset) +} + +/// One value of each of four kinds. Fixed width, so every message costs the +/// same header bytes and the segment math in [`MESSAGES_PER_SEGMENT`] holds. +fn user_headers_for(offset: u64) -> BTreeMap { + BTreeMap::from([ + ( + HeaderKey::try_from("offset").unwrap(), + HeaderValue::from(offset), + ), + ( + HeaderKey::try_from("origin").unwrap(), + HeaderValue::try_from(format!("compat-{offset:06}")).unwrap(), + ), + ( + HeaderKey::try_from("even").unwrap(), + HeaderValue::from(offset.is_multiple_of(2)), + ), + ( + HeaderKey::try_from("ratio").unwrap(), + HeaderValue::from(1.0 / (offset as f64 + 1.0)), + ), + ]) +} + +/// Prefixed with its own offset, then padded to a length that varies with +/// the position inside the batch, so a decoder that reused one message's +/// `payload_length` for the next cannot pass. The pattern repeats every +/// [`SEND_BATCH`], so every batch costs the same on disk. +fn payload_for(offset: u64) -> Bytes { + let mut bytes = format!("compat-message-{offset:06}").into_bytes(); + bytes.resize( + PAYLOAD_SIZE + (offset % SEND_BATCH) as usize * PAYLOAD_STEP, + b'.', + ); + Bytes::from(bytes) +} + +fn post_purge_payload(index: u64) -> Bytes { + Bytes::from(format!("compat-post-purge-{index}")) +} + +/// Alternating bits at every level. All-true (the harness default) reads back +/// identical under any field permutation, and a level left `None` cannot tell +/// a dropped map from one never seeded. +fn seeded_permissions(stream_id: u32, topic_id: u32) -> Permissions { + Permissions { + global: GlobalPermissions { + manage_servers: false, + read_servers: true, + manage_users: false, + read_users: true, + manage_streams: false, + read_streams: true, + manage_topics: false, + read_topics: true, + poll_messages: false, + send_messages: true, + }, + streams: Some(BTreeMap::from([( + stream_id as usize, + StreamPermissions { + manage_stream: true, + read_stream: false, + manage_topics: true, + read_topics: false, + poll_messages: true, + send_messages: false, + topics: Some(BTreeMap::from([( + topic_id as usize, + TopicPermissions { + manage_topic: false, + read_topic: true, + poll_messages: false, + send_messages: true, + }, + )])), + }, + )])), + } +} + +/// A personal access token as the baseline handed it out: the raw token a +/// later login presents, and the stored expiry a listing must reproduce. +struct SeededToken { + raw: String, + expiry_at: IggyTimestamp, +} + +async fn create_token(client: &IggyClient, name: &str, expiry_secs: u64) -> SeededToken { + let raw = client + .create_personal_access_token( + name, + PersonalAccessTokenExpiry::ExpireDuration(IggyDuration::new_from_secs(expiry_secs)), + ) + .await + .unwrap() + .token; + let expiry_at = client + .get_personal_access_tokens() + .await + .unwrap() + .into_iter() + .find(|token| token.name == name) + .and_then(|token| token.expiry_at) + .unwrap_or_else(|| panic!("the baseline lists {name} with its expiry")); + SeededToken { raw, expiry_at } +} + +/// Stream id to name, the identity the metadata plane must recover. +async fn stream_catalog(client: &impl StreamClient) -> BTreeMap { + client + .get_streams() + .await + .unwrap() + .into_iter() + .map(|stream| (stream.id, stream.name)) + .collect() +} + +/// `count` messages from `offset` in [`SEEDED_PARTITION`], without committing +/// an offset: that would add a retention barrier and a stored record of its +/// own. +async fn poll_window( + client: &IggyClient, + stream: &Identifier, + topic: &Identifier, + consumer: &Consumer, + offset: u64, + count: u32, +) -> Vec { + let polled = client + .poll_messages( + stream, + topic, + Some(SEEDED_PARTITION), + consumer, + &PollingStrategy::offset(offset), + count, + false, + ) + .await + .unwrap_or_else(|error| panic!("poll {count} message(s) from offset {offset}: {error}")); + assert_eq!( + polled.partition_id, SEEDED_PARTITION, + "the poll must be served from the seeded partition" + ); + polled.messages +} + +/// The baseline's own readback of what it was asked to store. Verified before +/// the swap, so the captured messages are a trustworthy oracle for the +/// field-by-field comparison after it. +fn assert_seeded_messages( + what: &str, + messages: &[IggyMessage], + first_offset: u64, + count: usize, + payload: fn(u64) -> Bytes, +) { + assert_eq!( + messages.len(), + count, + "{what}: the baseline must serve every seeded message from offset {first_offset}" + ); + for (index, message) in messages.iter().enumerate() { + let offset = first_offset + index as u64; + assert_eq!(message.header.offset, offset, "{what}: offset"); + assert_eq!( + message.header.id, + message_id_for(offset), + "{what}: id of the message at offset {offset}" + ); + assert_eq!( + message.user_headers_map().unwrap(), + Some(user_headers_for(offset)), + "{what}: user headers of the message at offset {offset}" + ); + assert!( + message.payload == payload(offset), + "{what}: the payload of the message at offset {offset} ({} bytes) is not the seeded one", + message.payload.len() + ); + } +} + +/// Every stored field. The header carries the id, both timestamps, the +/// checksum and the lengths, and a payload comparison alone would pass a +/// misread of any of them. +fn assert_messages_identical(what: &str, baseline: &[IggyMessage], current: &[IggyMessage]) { + assert_eq!( + baseline.len(), + current.len(), + "{what}: message count after the swap" + ); + for (before, after) in baseline.iter().zip(current) { + let offset = before.header.offset; + assert_eq!( + after.header, before.header, + "{what}: header of the message at offset {offset} (post-swap left, baseline right)" + ); + assert_eq!( + after.user_headers, before.user_headers, + "{what}: user headers of the message at offset {offset} (post-swap left, baseline \ + right)" + ); + assert!( + after.payload == before.payload, + "{what}: the payload of the message at offset {offset} differs after the swap ({} vs \ + {} bytes)", + after.payload.len(), + before.payload.len() + ); + } +} + +/// Per-key option degradation is silent, so every seeded value is re-read, +/// and with it the provenance flag: admission refills a key it cannot read +/// from the built-in default and marks it derived, which the value check +/// alone cannot catch for a key seeded at its default (`preallocate_segments` +/// has no usable non-default value here). The unsent key must come back +/// derived, or provenance stopped round-tripping and the explicit checks +/// prove nothing. +async fn assert_topic_recovered( + client: &IggyClient, + stream: &Identifier, + name: &str, + seed: &TopicCreateOptions, +) { + let topic = client + .get_topic(stream, &Identifier::named(name).unwrap()) + .await + .unwrap() + .unwrap_or_else(|| panic!("the topic {name} must survive the swap")); + + // Masked to the keys the seed sent: `from_resource_options` fills the + // rest from the derived defaults, which the seed leaves `None`. + let recovered = TopicCreateOptions::from_resource_options(&topic.options); + let recovered = TopicCreateOptions { + partitions_count: seed.partitions_count.map(|_| topic.partitions_count), + compression_algorithm: seed + .compression_algorithm + .map(|_| topic.compression_algorithm), + message_expiry: seed.message_expiry.map(|_| topic.message_expiry), + max_topic_size: seed.max_topic_size.map(|_| topic.max_topic_size), + segment_size: seed.segment_size.and(recovered.segment_size), + enforce_fsync: seed.enforce_fsync.and(recovered.enforce_fsync), + messages_required_to_save: seed + .messages_required_to_save + .and(recovered.messages_required_to_save), + size_of_messages_required_to_save: seed + .size_of_messages_required_to_save + .and(recovered.size_of_messages_required_to_save), + preallocate_segments: seed + .preallocate_segments + .and(recovered.preallocate_segments), + raw: BTreeMap::new(), + }; + assert_eq!( + &recovered, seed, + "{name}: every seeded value must read back as the baseline stored it (recovered left, \ + seed right); a lost key silently reverts to the shard-wide default with no log line" + ); + + for (key, sent) in [ + ( + topic_option_keys::COMPRESSION_ALGORITHM, + seed.compression_algorithm.is_some(), + ), + ( + topic_option_keys::MESSAGE_EXPIRY, + seed.message_expiry.is_some(), + ), + ( + topic_option_keys::MAX_TOPIC_SIZE, + seed.max_topic_size.is_some(), + ), + (topic_option_keys::SEGMENT_SIZE, seed.segment_size.is_some()), + ( + topic_option_keys::ENFORCE_FSYNC, + seed.enforce_fsync.is_some(), + ), + ( + topic_option_keys::MESSAGES_REQUIRED_TO_SAVE, + seed.messages_required_to_save.is_some(), + ), + ( + topic_option_keys::SIZE_OF_MESSAGES_REQUIRED_TO_SAVE, + seed.size_of_messages_required_to_save.is_some(), + ), + ( + topic_option_keys::PREALLOCATE_SEGMENTS, + seed.preallocate_segments.is_some(), + ), + ] { + let option = topic + .options + .get(&HeaderKey::from_str(key).unwrap()) + .unwrap_or_else(|| panic!("{name}: {key} is GONE from the topic's options")); + assert_eq!( + option.explicit, sent, + "{name}: {key} provenance. A seeded key that came back DERIVED was dropped and \ + refilled from the built-in default; an unsent key that came back EXPLICIT means \ + provenance no longer distinguishes a surviving option from a re-derived one" + ); + } +} + +/// Listing proves the metadata; the password hash is a separate field, and a +/// user that lists fine but cannot log in is still locked out. The login uses +/// a fresh, unauthenticated connection, so the root session vouches for +/// nothing. +async fn assert_user_recovered( + harness: &TestHarness, + client: &IggyClient, + name: &str, + permissions: &Permissions, +) { + let user = client + .get_user(&Identifier::named(name).unwrap()) + .await + .unwrap() + .unwrap_or_else(|| panic!("the user {name} must survive the swap")); + assert_eq!(user.status, UserStatus::Active, "{name}: status"); + assert_eq!( + user.permissions.as_ref(), + Some(permissions), + "{name}: permissions must survive the swap bit for bit at the global, stream and topic \ + level" + ); + let by_password = harness.tcp_new_client().await.unwrap(); + by_password + .login_user(name, USER_PASSWORD) + .await + .unwrap_or_else(|error| { + panic!("{name}: the seeded password must still authenticate after the swap: {error}") + }); +} + +/// Same split as [`assert_user_recovered`]: the token hash is not in the +/// listing, so only a login proves it. +async fn assert_token_recovered( + harness: &TestHarness, + client: &IggyClient, + name: &str, + seeded: &SeededToken, +) { + let tokens = client.get_personal_access_tokens().await.unwrap(); + let token = tokens + .iter() + .find(|token| token.name == name) + .unwrap_or_else(|| { + panic!( + "the personal access token {name} must survive the swap, got {:?}", + tokens.iter().map(|token| &token.name).collect::>() + ) + }); + assert_eq!(token.expiry_at, Some(seeded.expiry_at), "{name}: expiry"); + let by_token = harness.tcp_new_client().await.unwrap(); + by_token + .login_with_personal_access_token(&seeded.raw) + .await + .unwrap_or_else(|error| { + panic!( + "{name}: the seeded personal access token must still authenticate after the \ + swap: {error}" + ) + }); +} + +/// A topic directory relative to the data directory, in the key form of +/// [`disk::collect_comparable_files`]. +fn topic_prefix(stream_id: u32, topic_id: u32) -> String { + format!("streams/{stream_id}/topics/{topic_id}/") +} + +fn partition_dir(data_path: &Path, stream_id: u32, topic_id: u32, partition_id: u32) -> PathBuf { + data_path + .join(topic_prefix(stream_id, topic_id)) + .join("partitions") + .join(partition_id.to_string()) +} + +/// Segment log file names in a partition directory, in offset order (they are +/// zero-padded base offsets, so lexical order is offset order). +fn segment_logs(partition: &Path) -> Vec { + let mut names: Vec = fs::read_dir(partition) + .map(|entries| { + entries + .flatten() + .map(|entry| entry.file_name().to_string_lossy().into_owned()) + .filter(|name| name.ends_with(".log")) + .collect() + }) + .unwrap_or_default(); + names.sort(); + names +} + +fn base_offset_of(segment_log: &str) -> u64 { + segment_log + .trim_end_matches(".log") + .parse() + .unwrap_or_else(|error| { + panic!("segment {segment_log} is not named for a base offset: {error}") + }) +} + +fn assert_graceful_shutdown(harness: &TestHarness, who: &str) { + assert!( + harness.server().stdout_contains(GRACEFUL_SHUTDOWN_MARKER), + "{who} did not log {GRACEFUL_SHUTDOWN_MARKER:?}. Either the stop escalated to SIGKILL, \ + leaving crash-recovery state that confounds the compatibility verdict, or stdout was \ + not captured at all (IGGY_TEST_VERBOSE makes the harness inherit it, which would also \ + make the tombstone check vacuous). Server stdout:\n{}", + harness.server().stdout_plain() + ); +} + +/// Byte-compare the segment files across the binary swap, except under +/// `rewritten`, the one directory the build under test was told to change. +/// +/// Segment batch headers carry no magic and no version, and recovery is +/// allowed to truncate a torn tail, so a misparse can shorten a `.log` while +/// every message a test happens to poll still reads back correctly. +fn assert_segments_identical( + baseline: &BTreeMap>, + current: &BTreeMap>, + rewritten: &str, +) { + assert!( + baseline.keys().any(|rel| rel.starts_with(rewritten)), + "`{rewritten}` matches nothing the baseline wrote, so the exclusion would hide a typo \ + rather than the post-swap purge" + ); + let mut problems = Vec::new(); + for (rel, baseline_bytes) in baseline + .iter() + .filter(|(rel, _)| !rel.starts_with(rewritten)) + { + match current.get(rel) { + None => problems.push(format!( + "`{rel}` was written by the baseline but is GONE after the swap" + )), + Some(bytes) if bytes != baseline_bytes => { + problems.push(disk::describe_mismatch(rel, 0, baseline_bytes, 1, bytes)); + } + Some(_) => {} + } + } + for rel in current.keys().filter(|rel| !rel.starts_with(rewritten)) { + if !baseline.contains_key(rel) { + problems.push(format!("`{rel}` appeared only after the swap")); + } + } + assert!( + problems.is_empty(), + "the build under test did not preserve the baseline's segment bytes \ + ({} issue(s); node 0 = baseline, node 1 = post-swap):\n{}", + problems.len(), + problems.join("\n") + ); +} + +/// Poll `probe` until it succeeds, panicking with its last message on timeout. +async fn wait_until(what: &str, mut probe: impl AsyncFnMut() -> Result<(), String>) { + let deadline = Instant::now() + SETTLE_TIMEOUT; + loop { + match probe().await { + Ok(()) => return, + Err(last) => { + assert!( + Instant::now() < deadline, + "timed out after {SETTLE_TIMEOUT:?} waiting for {what}: {last}" + ); + sleep(POLL_INTERVAL).await; + } + } + } +} diff --git a/scripts/ci/storage-compat.sh b/scripts/ci/storage-compat.sh new file mode 100755 index 0000000000..b4c3a72a23 --- /dev/null +++ b/scripts/ci/storage-compat.sh @@ -0,0 +1,270 @@ +#!/usr/bin/env bash +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +set -euo pipefail + +# shellcheck source-path=SCRIPTDIR +source "$(dirname "${BASH_SOURCE[0]}")/lib/init.sh" + +# storage-compat.sh -- Prove HEAD still reads a data directory written by master. +# +# Builds two iggy-server binaries and hands both to one integration test: +# baseline built from --baseline-ref, copied aside. Default: on a +# pull_request run, the master tip under the checked-out merge ref; +# anywhere else, origin/master. +# HEAD built from the working tree, left at target/debug/iggy-server +# The test boots the baseline, seeds a data directory, swaps the binary to HEAD +# and restarts against that same directory. +# +# Both binaries MUST be built here. `core/integration` has no dependency on the +# `server` package; the harness only LOCATES a binary, through +# `assert_cmd::Command::cargo_bin`, which falls back to whatever file happens to +# sit at target/debug/iggy-server (assert_cmd 2.2.2 `legacy_cargo_bin`). So +# `cargo nextest run -p integration` compiles no server at all, and a lane that +# does not build both halves compares a stale binary against itself and reports +# green forever. +# +# Exit codes: 0 = compatible, non-zero = build failure or format regression. + +# Empty selects the default described above once HEAD is known. +BASELINE_REF="" +REBUILD_BASELINE=0 + +# The single #[ignore]d test this script exists to drive. The integration crate +# is one binary built from tests/mod.rs, so tests are selected by module path. +TEST_FILTER="data_integrity::storage_compat" + +while [[ $# -gt 0 ]]; do + case "$1" in + --baseline-ref) + BASELINE_REF="${2:-}" + if [ -z "${BASELINE_REF}" ]; then + echo "--baseline-ref requires a git ref" + exit 1 + fi + shift 2 + ;; + --rebuild-baseline) + REBUILD_BASELINE=1 + shift + ;; + --help|-h) + echo "Usage: $0 [--baseline-ref ] [--rebuild-baseline]" + echo "" + echo "Options:" + echo " --baseline-ref Ref to build the baseline server from (default: the merge ref's" + echo " first parent on a pull_request run, origin/master otherwise)" + echo " --rebuild-baseline Rebuild the baseline even when its binary is already on disk" + exit 0 + ;; + *) + echo "Unknown option: $1" + echo "Use --help for usage information" + exit 1 + ;; + esac +done + +REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" + +# Absolute before any cd: cargo resolves a relative CARGO_TARGET_DIR against +# its own CWD, and the baseline below builds from a worktree elsewhere on +# disk, so a relative value would scatter the two builds and the lookup of +# either binary across three directories. Exported so every cargo call here, +# nextest included, lands in the same place. +TARGET_DIR="${CARGO_TARGET_DIR:-${REPO_ROOT}/target}" +mkdir -p "${TARGET_DIR}" +TARGET_DIR="$(cd "${TARGET_DIR}" && pwd)" +export CARGO_TARGET_DIR="${TARGET_DIR}" + +cd "${REPO_ROOT}" + +if ! command -v cargo-nextest &>/dev/null; then + echo "cargo-nextest is not installed" + echo "" + echo "Install with:" + echo " cargo install cargo-nextest --locked" + exit 1 +fi + +HEAD_SERVER="${TARGET_DIR}/debug/iggy-server" + +WORKTREE_DIR="" + +# Every command is guarded: under `set -e` a failure inside an EXIT trap would +# replace the script's real exit status with the trap's. +cleanup() { + if [ -n "${WORKTREE_DIR}" ]; then + git worktree remove --force "${WORKTREE_DIR}" 2>/dev/null || rm -rf "${WORKTREE_DIR}" || true + fi + git worktree prune 2>/dev/null || true +} +trap cleanup EXIT + +HEAD_SHA="$(git rev-parse --verify HEAD)" + +# On a pull_request run actions/checkout leaves HEAD at the synthetic +# refs/pull/N/merge commit, whose first parent is the exact master tip GitHub +# merged the PR onto: an ancestor of HEAD by construction. Live origin/master +# can already be newer by the time this step runs, and a baseline HEAD does +# not contain would blame the PR for master's own changes. Read from the raw +# object: in the depth-1 checkout the parent is a shallow boundary, so +# `HEAD^1` does not resolve. Two parents required, so a checkout pinned to the +# PR head instead of the merge ref falls through rather than picking the PR's +# own parent. +if [ -z "${BASELINE_REF}" ]; then + if [[ "${GITHUB_REF:-}" =~ ^refs/pull/[0-9]+/merge$ ]] \ + && [ "$(git cat-file -p HEAD | grep -c '^parent ')" -eq 2 ]; then + BASELINE_REF="$(git cat-file -p HEAD | awk '/^parent /{print $2; exit}')" + echo "Pull request run: baseline is the master tip under the merge ref" + else + BASELINE_REF="origin/master" + fi +fi + +# The remote side of a fetch takes a branch or tag name (or a reachable SHA), +# never a remote-tracking name, so drop the remote prefix before asking origin. +# +# --depth=1 only where the clone is already shallow (the CI checkout). On a +# full clone that flag does not save anything, it WRITES a shallow boundary at +# the fetched tip and grafts every older commit off the developer's history, +# for every worktree sharing the repository. +FETCH_DEPTH=() +if [ "$(git rev-parse --is-shallow-repository)" = "true" ]; then + FETCH_DEPTH=(--depth=1) +fi +FETCHED=0 +if git fetch --no-tags "${FETCH_DEPTH[@]}" origin "${BASELINE_REF#origin/}" 2>/dev/null; then + FETCHED=1 +fi + +BASELINE_SHA="" +if [ "${FETCHED}" -eq 1 ]; then + # FETCH_HEAD in preference to the named ref: actions/checkout narrows + # remote.origin.fetch on a shallow clone, so refs/remotes/origin/master can + # stay absent or stale straight through a successful fetch. + BASELINE_SHA="$(git rev-parse --verify --quiet "FETCH_HEAD^{commit}" || true)" +fi +if [ -z "${BASELINE_SHA}" ]; then + BASELINE_SHA="$(git rev-parse --verify --quiet "${BASELINE_REF}^{commit}" || true)" +fi +if [ -z "${BASELINE_SHA}" ]; then + echo "Could not resolve baseline ref '${BASELINE_REF}' locally or from origin" + exit 1 +fi + +echo "Baseline: ${BASELINE_SHA} (${BASELINE_REF})" +echo "HEAD: ${HEAD_SHA}" +if [ "${BASELINE_SHA}" = "${HEAD_SHA}" ]; then + echo "WARNING: baseline and HEAD are the same commit, this run proves nothing" +fi +# Only decidable with history: a shallow clone answers "no" for every pair. +if [ "$(git rev-parse --is-shallow-repository)" = "false" ] \ + && ! git merge-base --is-ancestor "${BASELINE_SHA}" "${HEAD_SHA}"; then + echo "WARNING: baseline is not an ancestor of HEAD; a failure below may be master's change, not HEAD's" +fi + +# Keyed by baseline commit, which also pins that commit's Cargo.lock and +# rust-toolchain.toml. Only a developer machine ever hits this: hosted runners +# start empty and nothing publishes a baseline binary, so CI rebuilds it on +# every run. +BASELINE_SERVER="${TARGET_DIR}/storage-compat/${BASELINE_SHA}/iggy-server" + +if [ "${REBUILD_BASELINE}" -eq 1 ]; then + rm -f "${BASELINE_SERVER}" +fi + +if [ -x "${BASELINE_SERVER}" ]; then + echo "Reusing baseline server at ${BASELINE_SERVER}" +else + # Outside the repo on purpose: an in-tree worktree gets swept up by the + # repo-wide find(1) in the lint scripts and by cargo's workspace globs. + # A fresh mktemp path per run cannot collide with an aborted run; the only + # residue such a run leaves is an admin entry under .git/worktrees, which + # prune clears. + git worktree prune + WORKTREE_DIR="$(mktemp -d "${TMPDIR:-/tmp}/iggy-storage-compat.XXXXXX")" + git worktree add --detach "${WORKTREE_DIR}" "${BASELINE_SHA}" + + # Symmetric to the guard before the HEAD build below: a HEAD binary left by an + # earlier run makes cargo skip the uplift, and the cp further down would then + # capture that HEAD binary as the baseline, comparing HEAD against itself. + rm -f "${HEAD_SERVER}" + + echo "Building baseline iggy-server from ${BASELINE_SHA}..." + # Built from the worktree as CWD so the baseline's own rust-toolchain.toml + # applies, into the shared CARGO_TARGET_DIR so the registry graph compiles + # once. + # + # Debug profile on both sides, and no --all-features: release would compile + # debug_assert! out of the baseline while HEAD still panics on it, and + # --all-features turns on the server's `disable-mimalloc`, so the two halves + # would differ in ways the storage format never changed. + ( + cd "${WORKTREE_DIR}" + cargo build --locked -p server --bin iggy-server + ) + + if [ ! -x "${HEAD_SERVER}" ]; then + echo "Baseline build did not produce ${HEAD_SERVER}" + exit 1 + fi + + # Copy before HEAD builds: both trees uplift to the same target/debug path. + mkdir -p "$(dirname "${BASELINE_SERVER}")" + cp "${HEAD_SERVER}" "${BASELINE_SERVER}" +fi + +# Delete the uplifted binary before building HEAD. Cargo skips re-uplifting when +# the destination already looks current, and the baseline build just refreshed +# it, so on a second run the BASELINE binary could survive at +# target/debug/iggy-server and the test would compare master against master. +rm -f "${HEAD_SERVER}" + +echo "Building HEAD iggy-server from ${HEAD_SHA}..." +cargo build --locked -p server --bin iggy-server + +if [ ! -x "${HEAD_SERVER}" ]; then + echo "HEAD build did not produce ${HEAD_SERVER}" + exit 1 +fi + +# Absolute path: ServerHandle only treats this value as a literal path when it +# has more than one component, otherwise it falls back to a cargo_bin lookup. +export COMPAT_BASELINE_SERVER="${BASELINE_SERVER}" + +echo "Running storage compatibility test..." +# --run-ignored only: the test is #[ignore]d, so the normal lanes skip it. +# --no-tests=fail: a filter matching nothing has to be an error here, unlike the +# main lane's --no-tests=warn, which would report a typo'd filter as green. +# --retries 0: a read-back that only fails sometimes is exactly the signal this +# check exists for, so no retry may hide it. Explicit rather than "not +# --profile ci": a profile default or NEXTEST_RETRIES in the environment would +# otherwise apply. +# --ignore-default-filter: nextest intersects -E with the profile's +# default-filter, so one that excluded this module would turn the run into a +# --no-tests=fail abort instead of running the test. +# The version floor for these flags lives in .config/nextest.toml. +cargo nextest run --locked -p integration \ + --run-ignored only \ + --no-tests=fail \ + --retries 0 \ + --ignore-default-filter \ + -E "test(${TEST_FILTER})" + +echo "Storage format compatible: ${BASELINE_SHA} -> ${HEAD_SHA}" From 343ef06ebe279c59af9c47fefb9f8c3f0af4ca4c Mon Sep 17 00:00:00 2001 From: Hubert Gruszecki Date: Wed, 2 Sep 2026 15:57:32 +0200 Subject: [PATCH 043/182] feat(server,deps,ci): update deps, fix CI scoping, panic exit, crash recovery (#4033) --- .github/actions/rust/pre-merge/action.yml | 44 +- .github/config/hawkeye.version | 2 +- .github/workflows/post-merge.yml | 4 +- Cargo.lock | 1688 +++++++++-------- Cargo.toml | 144 +- bdd/python/uv.lock | 4 +- core/bench/report/src/prints.rs | 6 +- .../src/actors/consumer/benchmark_consumer.rs | 6 +- .../src/actors/producer/benchmark_producer.rs | 6 +- .../benchmark_producing_consumer.rs | 6 +- .../src/commands/binary_client/get_client.rs | 2 +- .../get_consumer_group.rs | 2 +- .../types/permissions/permissions_global.rs | 6 +- core/connectors/sdk/src/decoders/avro.rs | 18 +- core/connectors/sdk/src/encoders/avro.rs | 16 +- .../sdk/src/transforms/avro_convert.rs | 6 +- core/integration/Cargo.toml | 2 +- core/integration/src/harness/handle/server.rs | 21 +- .../tests/cluster/crash_durability.rs | 7 +- core/partitions/src/iggy_partition.rs | 6 +- core/server/src/boot/mod.rs | 9 +- core/server/src/boot/recovery.rs | 534 +----- core/server/src/boot/threads.rs | 113 +- core/server/src/partition_helpers.rs | 590 +++++- core/server/src/partition_reconciler.rs | 84 +- core/server/src/server_error.rs | 7 + core/server_common/src/crypto.rs | 19 +- core/shard/Cargo.toml | 16 +- core/simulator/Cargo.toml | 24 +- examples/python/uv.lock | 4 +- foreign/cpp/Cargo.toml | 4 +- foreign/cpp/MODULE.bazel | 6 +- foreign/cpp/MODULE.bazel.lock | 44 +- foreign/php/Cargo.toml | 6 +- foreign/php/composer.json | 2 +- foreign/php/phpunit.xml.dist | 2 +- foreign/python/Cargo.toml | 4 +- foreign/python/pylock.toml | 838 ++++---- foreign/python/pyproject.toml | 6 +- foreign/python/uv.lock | 404 ++-- gateways/kafka/Cargo.toml | 2 +- gateways/kafka/tools/kafka-tool/Cargo.toml | 6 +- gateways/kafka/tools/kafka-tool/src/main.rs | 1 + 43 files changed, 2601 insertions(+), 2120 deletions(-) diff --git a/.github/actions/rust/pre-merge/action.yml b/.github/actions/rust/pre-merge/action.yml index 204811ebc9..ab23317bef 100644 --- a/.github/actions/rust/pre-merge/action.yml +++ b/.github/actions/rust/pre-merge/action.yml @@ -92,11 +92,14 @@ runs: run: git fetch origin master --depth=1 2>/dev/null || true shell: bash + # Pinned: 0.24 dropped `-f json` and 0.25 retired `[change-detection]` in + # rail.toml. Either failure falls back to the full test suite on every PR, + # so bump on purpose together with `cargo rail config migrate`. - name: Install cargo-rail if: startsWith(inputs.task, 'test-') && inputs.task != 'test-storage-compat' uses: taiki-e/install-action@v2 with: - tool: cargo-rail + tool: cargo-rail@0.23.0 - name: Compute affected crates (cargo-rail) if: startsWith(inputs.task, 'test-') && inputs.task != 'test-storage-compat' @@ -293,32 +296,45 @@ runs: # api_handler_tests, version_firewall_tests, and server_e2e_tests need gitignored # wire fixtures. Generate when iggy-gateway-kafka is in the DAG test scope, or (on - # a full-workspace run) when gateways/** changed vs origin/master — avoid building + # a full-workspace run) when gateways/** changed vs the PR base — avoid building # kafka-message-gen / kafka-protocol when the gateway was not touched. NEEDS_KAFKA_FIXTURES=false if grep -q 'package(iggy-gateway-kafka)' <<< "$NEXTEST_FILTER"; then NEEDS_KAFKA_FIXTURES=true elif [[ -z "$NEXTEST_FILTER" ]]; then + # HEAD on a pull_request run is the synthetic merge commit; its first parent is + # the base tip GitHub merged against. The depth-1 checkout holds that parent only + # as a shallow boundary: `HEAD^1` does not resolve and `origin/master...HEAD` has + # no merge base (so it failed on every full-workspace lane). Read the parent from + # the raw object and fetch that one commit before diffing. A HEAD with one parent + # (workflow_dispatch on a branch tip) has no PR base, so it takes the fail-safe. + # # `2>&1` into a variable, not `2>/dev/null` piped straight to grep: the old form - # silenced *any* git failure (e.g. a shallow checkout with no merge-base history for - # origin/master) into empty output, which `grep -q` then reads identically to "diff - # succeeded, no gateways/ changes" - so a broken diff and a clean diff produced the - # same "skip fixtures" outcome. Confirmed in CI run 32256598833: gateways/-only PR, - # full-workspace lane, fixtures silently skipped, KAFKA_FIXTURES_REQUIRED never set - - # every fixture-backed test quietly passed by skipping all its assertions instead of - # running them (0.005s for a test meant to decode 9 real fixtures). - if DIFF_OUTPUT=$(git diff --name-only origin/master...HEAD 2>&1); then + # silenced *any* git failure into empty output, which `grep -q` then reads + # identically to "diff succeeded, no gateways/ changes" - so a broken diff and a + # clean diff produced the same "skip fixtures" outcome. Confirmed in CI run + # 32256598833: gateways/-only PR, full-workspace lane, fixtures silently skipped, + # KAFKA_FIXTURES_REQUIRED never set - every fixture-backed test quietly passed by + # skipping all its assertions instead of running them (0.005s for a test meant to + # decode 9 real fixtures). + BASE_SHA="" + if [[ "$(git cat-file -p HEAD | grep -c '^parent ')" -eq 2 ]]; then + BASE_SHA=$(git cat-file -p HEAD | awk '/^parent /{print $2; exit}') + fi + if [[ -n "$BASE_SHA" ]] \ + && DIFF_OUTPUT=$(git fetch -q --no-tags --depth=1 origin "$BASE_SHA" 2>&1 \ + && git diff --name-only "$BASE_SHA" HEAD 2>&1); then if grep -qE '^gateways/' <<< "$DIFF_OUTPUT"; then NEEDS_KAFKA_FIXTURES=true else echo "::notice::Skipping Kafka wire fixtures (full suite, gateways/** unchanged)" fi else - # Diff itself failed - NOT the same as "no gateways/ changes". Generate - # unconditionally as the fail-safe default; this is the case the bug above collapsed - # into a silent skip. + # No PR base, or fetch/diff failed - NOT the same as "no gateways/ changes". + # Generate unconditionally as the fail-safe default; this is the case the bug + # above collapsed into a silent skip. NEEDS_KAFKA_FIXTURES=true - echo "::warning::git diff origin/master...HEAD failed (${DIFF_OUTPUT}); generating Kafka fixtures unconditionally as a fail-safe" + echo "::warning::git diff against the PR base failed (base: ${BASE_SHA:-none, HEAD is not a merge commit}; ${DIFF_OUTPUT:-}); generating Kafka fixtures unconditionally as a fail-safe" fi fi if [[ "$NEEDS_KAFKA_FIXTURES" == true ]]; then diff --git a/.github/config/hawkeye.version b/.github/config/hawkeye.version index 66ce77b7ea..9fe9ff9d99 100644 --- a/.github/config/hawkeye.version +++ b/.github/config/hawkeye.version @@ -1 +1 @@ -7.0.0 +7.0.1 diff --git a/.github/workflows/post-merge.yml b/.github/workflows/post-merge.yml index 8d6c76007e..6b0d3fdb4f 100644 --- a/.github/workflows/post-merge.yml +++ b/.github/workflows/post-merge.yml @@ -64,10 +64,12 @@ jobs: # Metadata-only `cargo rail plan` (no compile), so it runs on the runner's # preinstalled cargo (version pinned by rust-toolchain.toml) and skips the # heavyweight build-cache restore. + # Pinned for the same reason as .github/actions/rust/pre-merge/action.yml: + # 0.24 dropped the `-f json` edge-affected-images.sh relies on. - name: Install cargo-rail uses: taiki-e/install-action@v2.86.7 with: - tool: cargo-rail + tool: cargo-rail@0.23.0 - name: Check all components id: check diff --git a/Cargo.lock b/Cargo.lock index be2a7673f4..56ecb71dd9 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -4,9 +4,9 @@ version = 4 [[package]] name = "actix-codec" -version = "0.5.2" +version = "0.5.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5f7b0a21988c1bf877cf4759ef5ddaac04c1c9fe808c9142ecb78ba97d97a28a" +checksum = "31404e1443b7b7bcaa311c1af456775cc3f668e34573f1503d7ce4844327629e" dependencies = [ "bitflags 2.13.1", "bytes", @@ -21,9 +21,9 @@ dependencies = [ [[package]] name = "actix-cors" -version = "0.7.1" +version = "0.7.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "daa239b93927be1ff123eebada5a3ff23e89f0124ccb8609234e5103d5a5ae6d" +checksum = "2aff07ada3254fc02618cb7850da91dceb23b5dbda53c6676ccbb28ba504f150" dependencies = [ "actix-utils", "actix-web", @@ -36,9 +36,9 @@ dependencies = [ [[package]] name = "actix-files" -version = "0.6.10" +version = "0.7.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "df8c4f30e3272d7c345f88ae0aac3848507ef5ba871f9cc2a41c8085a0f0523b" +checksum = "ed87c765e9e5be096cc29b43d04b932209892ec871d05fe255f0ae1ccfed9a06" dependencies = [ "actix-http", "actix-service", @@ -59,9 +59,9 @@ dependencies = [ [[package]] name = "actix-http" -version = "3.13.1" +version = "3.13.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "48e2faa3e7418ed780cca54829d32782a4008a077230f67457caa063415e99c2" +checksum = "11004b0e9b44b4eb3d15e0c3132b96fb178c7e50a74758b2f17bb9cc9a7fb4f6" dependencies = [ "actix-codec", "actix-rt", @@ -123,9 +123,9 @@ dependencies = [ [[package]] name = "actix-rt" -version = "2.11.0" +version = "2.13.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92589714878ca59a7626ea19734f0e07a6a875197eec751bb5d3f99e64998c63" +checksum = "6a16bf2f19c2ad84842bdfe6f3665620e93197d5c607889bfa3aaac45e762fd1" dependencies = [ "futures-core", "tokio", @@ -133,9 +133,9 @@ dependencies = [ [[package]] name = "actix-server" -version = "2.6.0" +version = "2.9.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a65064ea4a457eaf07f2fba30b4c695bf43b721790e9530d26cb6f9019ff7502" +checksum = "5d44ae8a6516f4ac7bfc7b61aabcd286104e96b4b24c747ce220832a016056d9" dependencies = [ "actix-rt", "actix-service", @@ -143,7 +143,7 @@ dependencies = [ "futures-core", "futures-util", "mio", - "socket2 0.5.10", + "socket2", "tokio", "tracing", ] @@ -170,9 +170,9 @@ dependencies = [ [[package]] name = "actix-web" -version = "4.14.0" +version = "4.15.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "df09e2d9239703dd64056359c920c7f3fba6535ec61a0059e0f44e095ffe02b4" +checksum = "bbacab3593b6b4f7be815076fc52d60a83c873426824675417e2abdd229e2e36" dependencies = [ "actix-codec", "actix-http", @@ -205,7 +205,7 @@ dependencies = [ "serde_json", "serde_urlencoded", "smallvec", - "socket2 0.6.5", + "socket2", "time", "tracing", "url", @@ -235,7 +235,7 @@ version = "0.5.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d122413f284cf2d62fb1b7db97e02edb8cda96d769b16e443a4f6195e35662b0" dependencies = [ - "crypto-common 0.1.7", + "crypto-common 0.1.6", "generic-array", ] @@ -262,13 +262,13 @@ dependencies = [ [[package]] name = "aes" -version = "0.9.1" +version = "0.9.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f1fc76eaeac4c9164506c466d4ffdd8ec9d0c5bf57ee97177c4d8eceb3a0e138" +checksum = "35f0f96ce78e38c3dc6d8948aa8163d06385be74000f3c7a95bf1eef35d3ea32" dependencies = [ "cipher 0.5.2", "cpubits", - "cpufeatures 0.3.0", + "cpufeatures 0.3.1", ] [[package]] @@ -287,16 +287,16 @@ dependencies = [ [[package]] name = "aes-gcm" -version = "0.11.0" +version = "0.11.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fdf011db2e21ce0d575593d749db5554b47fed37aff429e4dc50bc91ac93a028" +checksum = "7f2b8006a0c83f52b62ba44a97b58bf76fe2f70a329e588f67f89691d93d498f" dependencies = [ "aead 0.6.1", - "aes 0.9.1", + "aes 0.9.3", "cipher 0.5.2", "ctr 0.10.1", + "ctutils", "ghash 0.6.0", - "subtle", ] [[package]] @@ -327,13 +327,19 @@ dependencies = [ [[package]] name = "aho-corasick" -version = "1.1.4" +version = "1.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ddd31a130427c27518df266943a5308ed92d4b226cc639f5a8f1002816174301" +checksum = "c982642fa9e8606056828ee9a8505737230110bb1099153c79efe865c59d12ba" dependencies = [ "memchr", ] +[[package]] +name = "aliasable" +version = "0.1.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "250f629c0161ad8107cf89319e990051fae62832fd343083bea452d93e2205fd" + [[package]] name = "aligned" version = "0.4.3" @@ -375,9 +381,9 @@ checksum = "683d7910e743518b0e34f1186f92494becacb047c7b6bf616c96772180fef923" [[package]] name = "android_system_properties" -version = "0.1.5" +version = "0.1.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "819e7219dbd41043ac279b19830f2efc897156490d7fd6ea916720117ee66311" +checksum = "ae221649c9976a6f6c56ae1facf410f3ddb33cc661c4b7b61020a912d4237fbc" dependencies = [ "libc", ] @@ -449,7 +455,7 @@ dependencies = [ "crc32fast", "digest 0.10.7", "log", - "miniz_oxide", + "miniz_oxide 0.8.9", "num-bigint", "quad-rand", "rand 0.9.5", @@ -460,16 +466,40 @@ dependencies = [ "snap", "strum 0.27.2", "strum_macros 0.27.2", - "thiserror 2.0.19", + "thiserror 2.0.20", "uuid", "zstd", ] +[[package]] +name = "apache-avro" +version = "0.22.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "312c1ea69e5fe9966e0029fb95aca8790100b85aff4f0d3b00a9337c74069a9c" +dependencies = [ + "bigdecimal", + "bon", + "digest 0.11.3", + "log", + "miniz_oxide 0.9.1", + "num-bigint", + "ouroboros", + "quad-rand", + "rand 0.10.2", + "regex-lite", + "serde", + "serde_bytes", + "serde_json", + "strum 0.28.0", + "thiserror 2.0.20", + "uuid", +] + [[package]] name = "apple-native-keyring-store" -version = "1.0.1" +version = "1.0.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "797f94b6a53d7d10b56dc18290e0d40a2158352f108bb4ff32350825081a9f29" +checksum = "2b350bfd03649e07aa05c0a81b3e15934374e585c98204a57e20b9d49f49bb9a" dependencies = [ "keyring-core", "log", @@ -478,9 +508,9 @@ dependencies = [ [[package]] name = "ar_archive_writer" -version = "0.5.2" +version = "0.5.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4087686b4b0a3427190bae57a1d9a478dbb2d40c5dc1bd6e2b6d797913bdd348" +checksum = "73cd58deff2140a0a8eae87e417bd01db68a33e148aa93d1e8cd837e55e312b6" dependencies = [ "object", ] @@ -513,13 +543,13 @@ dependencies = [ [[package]] name = "argon2" -version = "0.5.3" +version = "0.6.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3c3610892ee6e0cbce8ae2700349fcf8f98adb0dbfbee85aec3c9179d29cc072" +checksum = "134c52ddac6d63c576bef8168db10c83c49c26444ecbc68060fef078925a901c" dependencies = [ "base64ct", "blake2", - "cpufeatures 0.2.17", + "cpufeatures 0.3.1", "password-hash", ] @@ -684,7 +714,7 @@ dependencies = [ "arrow-select", "chrono", "half", - "indexmap 2.14.0", + "indexmap 2.14.1", "itoa", "lexical-core", "memchr", @@ -790,7 +820,7 @@ dependencies = [ "nom 7.1.3", "num-traits", "rusticata-macros", - "thiserror 2.0.19", + "thiserror 2.0.20", "time", ] @@ -884,9 +914,9 @@ dependencies = [ [[package]] name = "async-compression" -version = "0.4.42" +version = "0.4.43" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e79b3f8a79cccc2898f31920fc69f304859b3bd567490f75ebf51ae1c792a9ac" +checksum = "3976abdc8fe7d1133d43d304afd42abdf5bc3e1319d263d223bde07b5efc4be8" dependencies = [ "compression-codecs", "compression-core", @@ -1065,13 +1095,13 @@ checksum = "8b75356056920673b02621b35afd0f7dda9306d03c79a30f5c56c44cf256e3de" [[package]] name = "async-trait" -version = "0.1.91" +version = "0.1.92" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ae36dc4177970ef04fde5178d3e2429882def40e57a451f919c098f72baa6cec" +checksum = "82f6aeea286b8eb4dd3431a1be1b59d290ace00f5bfd8e2a159bc2a05e2c1667" dependencies = [ "proc-macro2", "quote", - "syn 3.0.2", + "syn 3.0.4", ] [[package]] @@ -1092,15 +1122,16 @@ dependencies = [ [[package]] name = "async_zip" -version = "0.0.18" +version = "0.0.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0d8c50d65ce1b0e0cb65a785ff615f78860d7754290647d3b983208daa4f85e6" +checksum = "fb7f5f40e1eb30949a266fc900d37fd3267c7baf50c3705ac09c5d8ced5def63" dependencies = [ "async-compression", + "binrw", "crc32fast", "futures-lite", "pin-project", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tokio-util", ] @@ -1145,7 +1176,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "16e2cdb6d5ed835199484bb92bb8b3edd526effe995c61732580439c1a67e2e9" dependencies = [ "base64 0.22.1", - "http 1.4.2", + "http 1.5.0", "log", "rustls", "serde", @@ -1184,7 +1215,7 @@ dependencies = [ "num-traits", "pastey 0.1.1", "rayon", - "thiserror 2.0.19", + "thiserror 2.0.20", "v_frame", "y4m", ] @@ -1214,9 +1245,9 @@ dependencies = [ [[package]] name = "aws-config" -version = "1.9.0" +version = "1.11.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "47712fde1909402600ccfbb26e47d482d2e58bb9e9e603d9f17e67cc435a6319" +checksum = "a767267da9e2c2e189b2f9df8b5657e850ecf5352644734ba130d4a57095cf1b" dependencies = [ "aws-credential-types", "aws-runtime", @@ -1234,7 +1265,7 @@ dependencies = [ "bytes", "fastrand", "hex", - "http 1.4.2", + "http 1.5.0", "sha1 0.10.7", "time", "tokio", @@ -1267,16 +1298,16 @@ dependencies = [ "quick-xml 0.38.4", "rust-ini", "serde", - "thiserror 2.0.19", + "thiserror 2.0.20", "time", "url", ] [[package]] name = "aws-lc-rs" -version = "1.17.3" +version = "1.18.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "00bdb5da18dac48ca2cc7cd4a98e533e8635a58e2361d13a1a4ee3888e0d72f1" +checksum = "b281d307588d634de920874890732659e2e7672f72b5e10e81badc1a8a83621e" dependencies = [ "aws-lc-sys", "untrusted 0.7.1", @@ -1285,9 +1316,9 @@ dependencies = [ [[package]] name = "aws-lc-sys" -version = "0.43.0" +version = "0.45.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "43103168cc76fe62678a375e722fc9cb3a0146159ac5828bc4f0dfd755c2224c" +checksum = "9bff6c3b54fad79a2e60b8102caf565819711497c1f5f092f49508e2f5c31b27" dependencies = [ "cc", "cmake", @@ -1302,14 +1333,14 @@ version = "0.28.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "838b36c8dc927b6db1b6c6b8f5d05865f2213550b9e83bf92fa99ed6525472c0" dependencies = [ - "thiserror 2.0.19", + "thiserror 2.0.20", ] [[package]] name = "aws-runtime" -version = "1.8.1" +version = "1.9.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7816e98ee912159f45d307e5ee6bfea4a335a55aee15f7f3e32f81a6f3000f1d" +checksum = "c9007227e10b5fed2f3e0a2beff489211e2b5604c400b7a9d5d81ca9d64c24bb" dependencies = [ "aws-credential-types", "aws-sigv4", @@ -1322,7 +1353,7 @@ dependencies = [ "bytes", "bytes-utils", "fastrand", - "http 1.4.2", + "http 1.5.0", "http-body 1.1.0", "percent-encoding", "pin-project-lite", @@ -1332,9 +1363,9 @@ dependencies = [ [[package]] name = "aws-sdk-dynamodb" -version = "1.117.0" +version = "1.123.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0f2b4adcf6592e06ebb2c78235ae09cee178263a5b7fc20ae5ea07ca67067bb6" +checksum = "dcd10d92058d5da989a7838fd09d70e9457390a81ba672abf5685099f3562bce" dependencies = [ "arc-swap", "aws-credential-types", @@ -1351,16 +1382,17 @@ dependencies = [ "bytes", "fastrand", "http 0.2.12", - "http 1.4.2", + "http 1.5.0", "regex-lite", "tracing", + "url", ] [[package]] name = "aws-sdk-sso" -version = "1.103.0" +version = "1.108.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0469f435f645ad2162cfb463b15bde37115966ee3acf2d87fb4871ee309b8401" +checksum = "c15301b04372832947916607983b114b3374b9db0be058a00fb7513800de1f05" dependencies = [ "arc-swap", "aws-credential-types", @@ -1377,16 +1409,16 @@ dependencies = [ "bytes", "fastrand", "http 0.2.12", - "http 1.4.2", + "http 1.5.0", "regex-lite", "tracing", ] [[package]] name = "aws-sdk-ssooidc" -version = "1.105.0" +version = "1.110.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "085faefb253f770655e162b9304321e62a1e71adf7f019ee1f4454228a377b3a" +checksum = "72cc2c205cb27108183cf1856333f7d584c2ba0f505421b4209ca5828f9ea899" dependencies = [ "arc-swap", "aws-credential-types", @@ -1403,16 +1435,16 @@ dependencies = [ "bytes", "fastrand", "http 0.2.12", - "http 1.4.2", + "http 1.5.0", "regex-lite", "tracing", ] [[package]] name = "aws-sdk-sts" -version = "1.108.0" +version = "1.113.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3c72b08911d8128dd360fe1b22a9fec0fa8b552dde8ec828dcf20ef5ec974e9f" +checksum = "68182ecb449f7537db0f4d5d25917789cf41e32074a9fe47b6a0b847fe1d2032" dependencies = [ "arc-swap", "aws-credential-types", @@ -1430,7 +1462,7 @@ dependencies = [ "aws-types", "fastrand", "http 0.2.12", - "http 1.4.2", + "http 1.5.0", "regex-lite", "tracing", ] @@ -1450,7 +1482,7 @@ dependencies = [ "hex", "hmac 0.13.0", "http 0.2.12", - "http 1.4.2", + "http 1.5.0", "percent-encoding", "sha2 0.11.0", "time", @@ -1480,7 +1512,7 @@ dependencies = [ "bytes-utils", "futures-core", "futures-util", - "http 1.4.2", + "http 1.5.0", "http-body 1.1.0", "http-body-util", "percent-encoding", @@ -1491,15 +1523,15 @@ dependencies = [ [[package]] name = "aws-smithy-http-client" -version = "1.2.0" +version = "1.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "635d23afda0a6ab48d666c4d447c4873e8d1e83518a2be2093122397e50b838e" +checksum = "ebfd138fac0337cee7516c352757ea73b9f2266e57d0bcb5bc70e9547e45aef1" dependencies = [ "aws-smithy-async", "aws-smithy-runtime-api", "aws-smithy-types", - "h2 0.4.15", - "http 1.4.2", + "h2 0.4.19", + "http 1.5.0", "hyper", "hyper-rustls", "hyper-util", @@ -1535,19 +1567,22 @@ dependencies = [ [[package]] name = "aws-smithy-query" -version = "0.61.1" +version = "0.62.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dd22a6ba36e3f113cb8d5b3d1fe0ed31c76ee608ef63322d753bb8d2c9479e77" +checksum = "512346c7212ab7436df2d77a16d976a468ae44a418835511d2a69269810aaf62" dependencies = [ + "aws-smithy-runtime-api", + "aws-smithy-schema", "aws-smithy-types", + "aws-smithy-xml", "urlencoding", ] [[package]] name = "aws-smithy-runtime" -version = "1.12.0" +version = "1.14.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bea94a9ff8464016338c851e24b472d7131c388c88898a502e781815b2ee6045" +checksum = "b82e438d30e02a825d363bd639a9efaed68a8089d86101054b0081e7e0d3e606" dependencies = [ "aws-smithy-async", "aws-smithy-http", @@ -1559,7 +1594,7 @@ dependencies = [ "bytes", "fastrand", "http 0.2.12", - "http 1.4.2", + "http 1.5.0", "http-body 0.4.6", "http-body 1.1.0", "http-body-util", @@ -1571,16 +1606,16 @@ dependencies = [ [[package]] name = "aws-smithy-runtime-api" -version = "1.13.0" +version = "1.15.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "22ed1ebe6e0a95ea84570225f5a8208dec4b8f77e61a9b0d6f51773fcb4612f0" +checksum = "954c563ce84507722d2679f07a35d21b9c6466b3872d513020d0281fc8112ac9" dependencies = [ "aws-smithy-async", "aws-smithy-runtime-api-macros", "aws-smithy-types", "bytes", "http 0.2.12", - "http 1.4.2", + "http 1.5.0", "pin-project-lite", "tokio", "tracing", @@ -1606,21 +1641,21 @@ checksum = "7d56e0a4e53127a632224e43633b0fe045fa9e1e3cfc68b9830f1115e103f910" dependencies = [ "aws-smithy-runtime-api", "aws-smithy-types", - "http 1.4.2", + "http 1.5.0", ] [[package]] name = "aws-smithy-types" -version = "1.6.1" +version = "1.6.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d6dc683efb34b9e755675b37fedbe0103141e5b6df7bdc9eb6967756a8c167d8" +checksum = "fce83ce9abbb198d25bc7131e468d0f9fe1257125e58c39f3f9fc9f5098c9647" dependencies = [ "base64-simd", "bytes", "bytes-utils", "futures-core", "http 0.2.12", - "http 1.4.2", + "http 1.5.0", "http-body 0.4.6", "http-body 1.1.0", "http-body-util", @@ -1637,9 +1672,9 @@ dependencies = [ [[package]] name = "aws-smithy-xml" -version = "0.61.1" +version = "0.62.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ea3f68eec3607f02acd24067969ce2abc6ba16aa7d5ce59ca450ed2fb5f78957" +checksum = "ce84f71c72fee2cbbadde6e7d082f5fb466e3a84733855295fa7aafd1b31b7d8" dependencies = [ "aws-smithy-runtime-api", "aws-smithy-schema", @@ -1649,9 +1684,9 @@ dependencies = [ [[package]] name = "aws-types" -version = "1.4.0" +version = "1.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e957a6c6dbce82b7a91f44231c09273159703769f447cbe85e854dfe9cf67f86" +checksum = "eec1cd5469f328c782dc3e33d4153cf118a54e33cbb3356d60d16f89883e1f94" dependencies = [ "aws-credential-types", "aws-smithy-async", @@ -1673,7 +1708,7 @@ dependencies = [ "bytes", "form_urlencoded", "futures-util", - "http 1.4.2", + "http 1.5.0", "http-body 1.1.0", "http-body-util", "hyper", @@ -1704,7 +1739,7 @@ checksum = "08c78f31d7b1291f7ee735c1c6780ccde7785daae9a9206026862dab7d8792d1" dependencies = [ "bytes", "futures-core", - "http 1.4.2", + "http 1.5.0", "http-body 1.1.0", "http-body-util", "mime", @@ -1736,7 +1771,7 @@ dependencies = [ "bytes", "either", "fs-err", - "http 1.4.2", + "http 1.5.0", "http-body 1.1.0", "hyper", "hyper-util", @@ -1837,7 +1872,7 @@ dependencies = [ "gloo 0.12.0", "js-sys", "serde_json", - "thiserror 2.0.19", + "thiserror 2.0.20", "uuid", "wasm-bindgen", "web-sys", @@ -1928,7 +1963,7 @@ dependencies = [ "clang-sys", "itertools 0.13.0", "log", - "prettyplease", + "prettyplease 0.2.37", "proc-macro2", "quote", "regex", @@ -1937,6 +1972,30 @@ dependencies = [ "syn 2.0.119", ] +[[package]] +name = "binrw" +version = "0.15.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6ad120d555272286c1017d25165ab8bd74806f13fc85b258484ec7e4ce75458f" +dependencies = [ + "array-init", + "binrw_derive", + "bytemuck", +] + +[[package]] +name = "binrw_derive" +version = "0.15.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6df92e0e9baae4dc82c7bad7715ca40c0a5c71539057bf2ea04a5c29c980410b" +dependencies = [ + "either", + "owo-colors", + "proc-macro2", + "quote", + "syn 2.0.119", +] + [[package]] name = "bit-set" version = "0.8.0" @@ -2024,25 +2083,24 @@ dependencies = [ [[package]] name = "blake2" -version = "0.10.6" +version = "0.11.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "46502ad458c9a52b69d4d4d32775c788b7a1b85e8bc9d482d92250fc0e3f8efe" +checksum = "5b5d4d889834ee8ecfc0f8426ad30faf7cdcb10f741a8e6d7224d95325479f6f" dependencies = [ - "digest 0.10.7", + "digest 0.11.3", ] [[package]] name = "blake3" -version = "1.8.5" +version = "1.8.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0aa83c34e62843d924f905e0f5c866eb1dd6545fc4d719e803d9ba6030371fce" +checksum = "6d9e454fc11f76977dc803893aff6304ed33d6a26efae8696573bea74baa27ae" dependencies = [ - "arrayref", "arrayvec", "cc", "cfg-if", "constant_time_eq", - "cpufeatures 0.3.0", + "cpufeatures 0.3.1", ] [[package]] @@ -2065,11 +2123,11 @@ dependencies = [ [[package]] name = "block-padding" -version = "0.3.3" +version = "0.4.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a8894febbff9f758034a5b8e12d87918f56dfc64a8e1fe757d65e29041538d93" +checksum = "710f1dd022ef4e93f8a438b4ba958de7f64308434fa6a87104481645cc30068b" dependencies = [ - "generic-array", + "hybrid-array", ] [[package]] @@ -2083,9 +2141,9 @@ dependencies = [ [[package]] name = "blocking" -version = "1.6.2" +version = "1.7.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e83f8d02be6967315521be875afa792a316e28d57b5a2d401897e2a7921b7f21" +checksum = "a70e4329df6cb94385eed412ec92375c3cdd8a6e502493d1229b6414e4036dfa" dependencies = [ "async-channel", "async-task", @@ -2120,7 +2178,7 @@ dependencies = [ "futures-util", "hex", "home", - "http 1.4.2", + "http 1.5.0", "http-body-util", "hyper", "hyper-named-pipe", @@ -2138,7 +2196,7 @@ dependencies = [ "serde_derive", "serde_json", "serde_urlencoded", - "thiserror 2.0.19", + "thiserror 2.0.20", "time", "tokio", "tokio-stream", @@ -2180,34 +2238,32 @@ dependencies = [ [[package]] name = "bon" -version = "3.9.3" +version = "3.10.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a602c73c7b0148ec6d12af6fd5cc7a46e2eacc8878271a999abac56eed12f561" +checksum = "9e3fac94a66da67200398458a25412bcc3f9b6443b5119a6cad9cf3ccfcd8cc6" dependencies = [ "bon-macros", - "rustversion", ] [[package]] name = "bon-macros" -version = "3.9.3" +version = "3.10.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6dee98b0db6a962de883bf5d20362dee4d7ca0d12fe39a7c6c73c844e1cd7c1f" +checksum = "d4654961ad0494e4774c5c60b4cb4cd0ae9b9d92d039d901638b1dba97ebebf5" dependencies = [ - "darling 0.23.0", + "darling 0.24.1", "ident_case", - "prettyplease", + "prettyplease 0.3.0", "proc-macro2", "quote", - "rustversion", - "syn 2.0.119", + "syn 3.0.4", ] [[package]] name = "borsh" -version = "1.8.0" +version = "1.8.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a88b7ea17d208c4193f2c1e6de3c35fe71f98c96982d5ced308bdcc749ff6e1f" +checksum = "553c5d846a6ba5150c65e3b1b8ec073bcf1abc20f9b7220de384a4443ea4e20a" dependencies = [ "borsh-derive", "bytes", @@ -2216,15 +2272,15 @@ dependencies = [ [[package]] name = "borsh-derive" -version = "1.8.0" +version = "1.8.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d8f347189c62a579b8cd5f80714efa178f52e461dc2e6d701d264f5ff22e566c" +checksum = "12cdfe656708a01f89b451a7d36466e6fe6c414de0aa18fc54f864f6f9ca9f56" dependencies = [ "once_cell", "proc-macro-crate 3.3.0", "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.4", ] [[package]] @@ -2269,7 +2325,7 @@ dependencies = [ "getrandom 0.2.17", "getrandom 0.3.4", "hex", - "indexmap 2.14.0", + "indexmap 2.14.1", "js-sys", "once_cell", "rand 0.9.5", @@ -2282,9 +2338,9 @@ dependencies = [ [[package]] name = "bstr" -version = "1.13.0" +version = "1.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1f7dc094d718f2e1c1559ad110e27eeaae14a5465d3d56dd6dbd793079fbd530" +checksum = "6bb31b46c14244e20ee9984b11bf5c992b91fb6939fea616e3512c8baecdbe5f" dependencies = [ "memchr", "regex-automata", @@ -2315,7 +2371,7 @@ dependencies = [ "chrono", "crc", "futures", - "indexmap 2.14.0", + "indexmap 2.14.1", "itertools 0.14.0", "object_store", "parquet", @@ -2327,7 +2383,7 @@ dependencies = [ "serde", "serde_json", "strum 0.27.2", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tracing", "tracing-subscriber", @@ -2354,7 +2410,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "4a813de7f2bbedb7dce265b64f1cf5908ebe4d56281ece8d847e98113788b9b0" dependencies = [ "rust_decimal", - "schemars 1.2.1", + "schemars 1.2.2", "serde", "utf8-width", ] @@ -2398,13 +2454,13 @@ dependencies = [ [[package]] name = "bytemuck_derive" -version = "1.11.0" +version = "1.12.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f65693059b6b9c588b9f62fed1cedbf0a8b805631457ea162d68f0de186f3de5" +checksum = "fc0e56a716f1e132ff6bf4bdac1c944a3fcdc1cae65f70a4a2a1ac3b401d2d1f" dependencies = [ "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.4", ] [[package]] @@ -2455,9 +2511,9 @@ dependencies = [ [[package]] name = "camino" -version = "1.2.4" +version = "1.2.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5f2d30e4173c4026932d51d31d6b0613b1fd3014bf3f9f8943d4ba139c437ba0" +checksum = "bb1307f12aa967b5a58416e87b3653360e0fd614a016b6e970db08fecbb1b80d" dependencies = [ "serde_core", ] @@ -2503,7 +2559,7 @@ dependencies = [ "semver", "serde", "serde_json", - "thiserror 2.0.19", + "thiserror 2.0.20", ] [[package]] @@ -2517,18 +2573,18 @@ dependencies = [ [[package]] name = "cbc" -version = "0.1.2" +version = "0.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "26b52a9543ae338f279b96b0b9fed9c8093744685043739079ce85cd58f289a6" +checksum = "ce2dc9ee5f88d11e0beb842c88b33c8a5cf0d1329c4b19494af42b07dbfe8896" dependencies = [ - "cipher 0.4.4", + "cipher 0.5.2", ] [[package]] name = "cc" -version = "1.3.0" +version = "1.4.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c89588d05638b5b4594a3348a2d6c20277e43a7f5c5202b05cc56888475a47b8" +checksum = "0ad534f4357a5264cce5019c989cf66a4f0dc4e0d1b1d15f8aacec0ff7360273" dependencies = [ "find-msvc-tools", "jobserver", @@ -2564,7 +2620,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "65c35e4b699c7e15ccbe7ee35c005e4fc0a278d22238a2857e6ce2dadeda1b06" dependencies = [ "cfg-if", - "cpufeatures 0.3.0", + "cpufeatures 0.3.1", "rand_core 0.10.1", ] @@ -2630,7 +2686,7 @@ version = "0.4.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "773f3b9af64447d2ce9850330c473515014aa235e6a783b02db81ff39e4a3dad" dependencies = [ - "crypto-common 0.1.7", + "crypto-common 0.1.6", "inout 0.1.4", ] @@ -2647,9 +2703,9 @@ dependencies = [ [[package]] name = "clang-sys" -version = "1.8.1" +version = "1.9.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0b023947811758c97c59bf9d1c188fd619ad4718dcaa767947df1cadb14f39f4" +checksum = "157a8ba7b480713b56f4c09fd13fc3e0a22a5dfab8097ba61cbc5feef950788a" dependencies = [ "glob", "libc", @@ -2658,9 +2714,9 @@ dependencies = [ [[package]] name = "clap" -version = "4.6.3" +version = "4.6.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0fb99565819980999fb7b4a1796046a5c949e6d4ff132cf5fadf5a641e20d776" +checksum = "473c7e07f409a8d772161724aa8db6a765a2532a70f9667eeb7b49d3d02fbdca" dependencies = [ "clap_builder", "clap_derive", @@ -2668,9 +2724,9 @@ dependencies = [ [[package]] name = "clap_builder" -version = "4.6.2" +version = "4.6.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f09628afdcc538b57f3c6341e9c8e9970f18e4a481690a64974d7023bd33548b" +checksum = "7b48fea5a88e9ae728a2dcbedbfc0e730f7d60da42e1cb049a83c9fb8b789889" dependencies = [ "anstream", "anstyle", @@ -2681,23 +2737,23 @@ dependencies = [ [[package]] name = "clap_complete" -version = "4.6.7" +version = "4.6.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "db8b397918185f0161ff3d6fcaa9e4bfc09b8367caf6e1d4a2848e5477ed027b" +checksum = "3be2ad0423bdbbb0e25bc89add796f3559706d4a95e1bc98e4d9662a957b6a19" dependencies = [ "clap", ] [[package]] name = "clap_derive" -version = "4.6.3" +version = "4.6.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32f2392eae7f16557a3d727ef3a12e57b2b2ca6f98566a5f4fb41ffe305df077" +checksum = "d012d2b9d65aca7f18f4d9878a045bc17899bba951561ba5ec3c2ba1eed9a061" dependencies = [ - "heck", + "heck 0.5.0", "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.4", ] [[package]] @@ -2734,7 +2790,7 @@ version = "0.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0fa961b519f0b462e3a3b4a34b64d119eeaca1d59af726fe450bbba07a9fc0a1" dependencies = [ - "thiserror 2.0.19", + "thiserror 2.0.20", ] [[package]] @@ -2760,9 +2816,9 @@ dependencies = [ [[package]] name = "combine" -version = "4.6.7" +version = "4.6.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ba5a308b75df32fe02788e748662718f03fde005016435c444eea572398219fd" +checksum = "cfc320937d09e6de266b31b9afb480f197d7a861be86be7cb2ea7e5d1bfffc5e" dependencies = [ "bytes", "memchr", @@ -2770,9 +2826,9 @@ dependencies = [ [[package]] name = "comfy-table" -version = "7.2.2" +version = "8.0.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "958c5d6ecf1f214b4c2bbbbf6ab9523a864bd136dcf71a7e8904799acfe1ad47" +checksum = "136c8c4c3823846e8ba6d4bda011b4e5d5827b8bb0ac26c31f9ab31abe9f54f2" dependencies = [ "unicode-segmentation", "unicode-width 0.2.2", @@ -2793,9 +2849,9 @@ dependencies = [ [[package]] name = "compio" -version = "0.19.1" +version = "0.19.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e5cf9e29d3c2d2a37078198795631b61d2b5c9cab68191e31526fe342d742135" +checksum = "c19d81c636b0f47aa50041d25b02d9d8289a3319727a8051209f71bea10e89d6" dependencies = [ "compio-buf", "compio-driver", @@ -2823,9 +2879,9 @@ dependencies = [ [[package]] name = "compio-driver" -version = "0.12.4" +version = "0.12.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ce1f07e864382cbb51d601416dc41cb1dc99ad3a01c6745cece55e431ca547fe" +checksum = "293e8086a35f52b5002402937cf4e69b5e414917e511c29c5e7ba2cebe6ef7b3" dependencies = [ "bitflags 2.13.1", "cfg_aliases", @@ -2844,7 +2900,7 @@ dependencies = [ "polling", "rustix 1.1.4", "smallvec", - "socket2 0.6.5", + "socket2", "synchrony", "thin-cell", "windows-sys 0.61.2", @@ -2852,9 +2908,9 @@ dependencies = [ [[package]] name = "compio-executor" -version = "0.1.3" +version = "0.1.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2402a55af5e31454cd77d4b6044e2e6b8c30aa037655585ebd6505b97680468c" +checksum = "94961c5908b02bbb082e046968b8ee3d18995845a5c261376f0b0b5c95c87c2b" dependencies = [ "compio-log", "compio-send-wrapper", @@ -2865,9 +2921,9 @@ dependencies = [ [[package]] name = "compio-fs" -version = "0.12.0" +version = "0.12.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "934c384c7c1dca1d68540bd0ef16186a8e7f33fab330e54652b45ee80f051ee4" +checksum = "8a374c5a03bdf7ecb42894a49c3a40c22f743f8d7dc39cc21a5dd3258d8219c7" dependencies = [ "cfg_aliases", "compio-buf", @@ -2901,31 +2957,31 @@ dependencies = [ [[package]] name = "compio-log" -version = "0.2.0" +version = "0.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ef39fff6341af7ab6c27fae9a3887e1ec618320d43f3b9c924c7a644f693a859" +checksum = "9dd867f29de59e4eff577dfeee3744cb0a264249f3faaf11945e8be79a3c3c62" dependencies = [ "tracing", ] [[package]] name = "compio-macros" -version = "0.2.0" +version = "0.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a9ac573b3bb60fffabf47c2ce01c61536fb5eb2168d66e0375e85603e7d51614" +checksum = "dc53390a911fe6b317abe66c0583723539477f81a940e124b40d462fb976674a" dependencies = [ - "darling 0.23.0", + "darling 0.24.1", "proc-macro-crate 3.3.0", "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.4", ] [[package]] name = "compio-net" -version = "0.12.2" +version = "0.12.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e940998522d9f78342b7ab5c663aefe888a504b74a7f933dcd058cdde26df392" +checksum = "31737ec57865d2b46284def673cd21fe478c592ca05c1e040e10d457aac6c2a8" dependencies = [ "compio-buf", "compio-driver", @@ -2936,7 +2992,7 @@ dependencies = [ "libc", "once_cell", "pin-project-lite", - "socket2 0.6.5", + "socket2", "synchrony", "widestring", "windows-sys 0.61.2", @@ -2944,9 +3000,9 @@ dependencies = [ [[package]] name = "compio-quic" -version = "0.8.0" +version = "0.8.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "edd50cb1b2c2c8276728871bc3bf3c12618c551cdd550123d564aa49b059aea5" +checksum = "65b1501ba168bfd2cf63ffc120d81a015a1d78cc853bbd9b4f2a7cb85cd2ab9a" dependencies = [ "cfg_aliases", "compio-buf", @@ -2962,15 +3018,15 @@ dependencies = [ "rustls", "rustls-platform-verifier", "synchrony", - "thiserror 2.0.19", + "thiserror 2.0.20", "windows-sys 0.61.2", ] [[package]] name = "compio-runtime" -version = "0.12.3" +version = "0.12.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3919cf83e9e531fdceda551085cdf05d570de98436401c89f121d3fae827521" +checksum = "20406f0a4e69cdd0c0a58c4a7cf51bcc965a543fb5da2fe931e970354e291ded" dependencies = [ "compio-buf", "compio-driver", @@ -3067,7 +3123,7 @@ dependencies = [ "figment", "iggy_common", "ipnet", - "jsonwebtoken", + "jsonwebtoken 11.0.0", "serde", "serde_json", "serde_with", @@ -3080,10 +3136,10 @@ dependencies = [ name = "configs_derive" version = "0.1.0" dependencies = [ - "darling 0.23.0", + "darling 0.24.1", "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.4", ] [[package]] @@ -3282,9 +3338,9 @@ dependencies = [ [[package]] name = "cpufeatures" -version = "0.3.0" +version = "0.3.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8b2a41393f66f16b0823bb79094d54ac5fbd34ab292ddafb9a0456ac9f87d201" +checksum = "5ca28b0ae3115b884660db4118d803791fd6756b6e88f39c0f3f7859060d7566" dependencies = [ "libc", ] @@ -3315,9 +3371,9 @@ dependencies = [ [[package]] name = "crc32fast" -version = "1.5.0" +version = "1.5.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9481c1c90cbf2ac953f07c8d4a58aa3945c425b7185c9154d67a65e4230da511" +checksum = "8498c871161e1742aaa9d52551b2d6ebdd4c3d45a3be423e3728f33b955be550" dependencies = [ "cfg-if", ] @@ -3417,9 +3473,9 @@ dependencies = [ [[package]] name = "crypto-common" -version = "0.1.7" +version = "0.1.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a" +checksum = "1bfb12502f3fc46cca1bb51ac28df9d618d813cdc3d2f25b9fe775a34af26bb3" dependencies = [ "generic-array", "rand_core 0.6.4", @@ -3471,11 +3527,11 @@ dependencies = [ [[package]] name = "ctor" -version = "1.0.9" +version = "1.0.13" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a394189d59f9befacce833f337f7b1eca5e9a91221bcdd4d28e0114d96e597b3" +checksum = "914a755b7c2d4af2bdcff7ce1739e2db9a1b81a9b07123d8015786ae03c0980d" dependencies = [ - "link-section 0.19.0", + "link-section 0.19.3", "linktime-proc-macro", ] @@ -3621,7 +3677,7 @@ dependencies = [ "cyper-core", "encoding_rs", "futures-util", - "http 1.4.2", + "http 1.5.0", "http-body-util", "hyper", "hyper-util", @@ -3631,7 +3687,7 @@ dependencies = [ "serde", "serde_urlencoded", "synchrony", - "thiserror 2.0.19", + "thiserror 2.0.20", "tower-service", "url", ] @@ -3651,7 +3707,7 @@ dependencies = [ "hyper", "hyper-util", "send_wrapper", - "socket2 0.6.5", + "socket2", "tokio", "tower", "tower-service", @@ -3699,6 +3755,16 @@ dependencies = [ "darling_macro 0.23.0", ] +[[package]] +name = "darling" +version = "0.24.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ed17f5901b6630b993ca003def43f2f8ef4014fc13b047b57aad617ff32bc2ec" +dependencies = [ + "darling_core 0.24.1", + "darling_macro 0.24.1", +] + [[package]] name = "darling_core" version = "0.20.11" @@ -3739,6 +3805,19 @@ dependencies = [ "syn 2.0.119", ] +[[package]] +name = "darling_core" +version = "0.24.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6837e2cf7485aaae18f86181d2f0e9a7ed297a025e220aeabf63fdebd3a2ddff" +dependencies = [ + "ident_case", + "proc-macro2", + "quote", + "strsim", + "syn 3.0.4", +] + [[package]] name = "darling_macro" version = "0.20.11" @@ -3772,6 +3851,17 @@ dependencies = [ "syn 2.0.119", ] +[[package]] +name = "darling_macro" +version = "0.24.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2ac7135c3ef02b2f7833bbeb1be5ba7f966dcde8a87c6b87f65a778d71a02785" +dependencies = [ + "darling_core 0.24.1", + "quote", + "syn 3.0.4", +] + [[package]] name = "dashmap" version = "6.2.1" @@ -3788,9 +3878,9 @@ dependencies = [ [[package]] name = "data-encoding" -version = "2.11.0" +version = "2.11.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a4ae5f15dda3c708c0ade84bfee31ccab44a3da4f88015ed22f63732abe300c8" +checksum = "4583a4551df46e2792f82ceeac45e850d2e2d5debba0b91f102385cda5b11f06" [[package]] name = "data-url" @@ -3854,7 +3944,7 @@ version = "1.0.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "10d60334b3b2e7c9d91ef8150abfb6fa4c1c39ebbcf4a81c2e346aad939fee3e" dependencies = [ - "thiserror 2.0.19", + "thiserror 2.0.20", ] [[package]] @@ -3890,7 +3980,7 @@ dependencies = [ "futures", "object_store", "regex", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tracing", "typed-builder 0.23.2", @@ -3907,7 +3997,7 @@ dependencies = [ "bytes", "deltalake-core", "object_store", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "url", ] @@ -3940,7 +4030,7 @@ dependencies = [ "either", "futures", "humantime", - "indexmap 2.14.0", + "indexmap 2.14.1", "itertools 0.14.0", "num_cpus", "object_store", @@ -3955,7 +4045,7 @@ dependencies = [ "serde_json", "sqlparser 0.61.0", "strum 0.27.2", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tracing", "url", @@ -3987,7 +4077,7 @@ dependencies = [ "deltalake-core", "futures", "object_store", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tracing", "url", @@ -4013,7 +4103,7 @@ dependencies = [ "deno_path_util", "deno_unsync", "futures", - "indexmap 2.14.0", + "indexmap 2.14.1", "libc", "parking_lot", "percent-encoding", @@ -4024,7 +4114,7 @@ dependencies = [ "smallvec", "sourcemap", "static_assertions", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "url", "v8", @@ -4068,7 +4158,7 @@ version = "0.227.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1bab1eaf578a8cc0ae6fb933e91dc3388b41df22e5974d5891c17ba66b3a0bbb" dependencies = [ - "indexmap 2.14.0", + "indexmap 2.14.1", "proc-macro-rules", "proc-macro2", "quote", @@ -4076,7 +4166,7 @@ dependencies = [ "strum 0.27.2", "strum_macros 0.27.2", "syn 2.0.119", - "thiserror 2.0.19", + "thiserror 2.0.20", ] [[package]] @@ -4088,7 +4178,7 @@ dependencies = [ "deno_error", "percent-encoding", "sys_traits", - "thiserror 2.0.19", + "thiserror 2.0.20", "url", ] @@ -4238,7 +4328,7 @@ checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" dependencies = [ "block-buffer 0.10.4", "const-oid 0.9.6", - "crypto-common 0.1.7", + "crypto-common 0.1.6", "subtle", ] @@ -4300,13 +4390,13 @@ dependencies = [ [[package]] name = "displaydoc" -version = "0.2.6" +version = "0.2.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ac70aa55017e108007fbaf5aa0f54b021c98f92ff8af59d42eda9da96e3dd4f" +checksum = "c6232dd377dcc64799954cbd3a9bb882e9cdc1308ccd87b1c098f1fb2eaf82a8" dependencies = [ "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.4", ] [[package]] @@ -4317,25 +4407,24 @@ checksum = "aeda16ab4059c5fd2a83f2b9c9e9c981327b18aa8e3b313f7e6563799d4f093e" [[package]] name = "dlopen2" -version = "0.8.2" +version = "0.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5e2c5bd4158e66d1e215c49b837e11d62f3267b30c92f1d171c4d3105e3dc4d4" +checksum = "60768c353d76ba6dd1b6332a5dbbdd3bc2dddadc2935f4cc6a9d0f17ca8ecb6a" dependencies = [ "dlopen2_derive", "libc", - "once_cell", - "winapi", + "windows-sys 0.61.2", ] [[package]] name = "dlopen2_derive" -version = "0.4.3" +version = "0.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0fbbb781877580993a8707ec48672673ec7b81eeba04cfd2310bd28c08e47c8f" +checksum = "1aacdd87ba97c2f83ff479e881e547f776b0a7c42921be22d4391f32b6284050" dependencies = [ "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.4", ] [[package]] @@ -4387,9 +4476,9 @@ dependencies = [ [[package]] name = "dtor" -version = "1.0.5" +version = "1.0.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6d738e43aa64edab57c983d56de890d65fea7dc05605490c74451ce721dfd84b" +checksum = "cd3f6c362b2bbd7063090886afd78978f5dbf1f11da7efcc40f81d32c943fac1" dependencies = [ "linktime-proc-macro", ] @@ -4452,9 +4541,9 @@ dependencies = [ [[package]] name = "either" -version = "1.16.0" +version = "1.18.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "91622ff5e7162018101f2fea40d6ebf4a78bbe5a49736a2020649edf9693679e" +checksum = "252afb9ae5eaa683babdc6a068b3f5726eb19e05070c731f9b2a23a7c3e8ed34" dependencies = [ "serde", ] @@ -4688,11 +4777,10 @@ dependencies = [ [[package]] name = "event-listener" -version = "5.4.1" +version = "5.4.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e13b66accf52311f30a0db42147dadea9850cb48cd070028831ae5f5d4b856ab" +checksum = "5a23add41df1562121a9393cb065eab5146a1242410f23a644851e90cfd669d2" dependencies = [ - "concurrent-queue", "parking", "pin-project-lite", ] @@ -4726,7 +4814,7 @@ dependencies = [ "bit_field", "half", "lebe", - "miniz_oxide", + "miniz_oxide 0.8.9", "num-complex", "pulp", "rayon-core", @@ -4840,9 +4928,9 @@ dependencies = [ [[package]] name = "file-operation" -version = "0.8.28" +version = "0.8.32" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "069e9a5a2347a89fcf2d63241ecaee5aef21063df00ae22d7bde905950b7d962" +checksum = "f33d2bf3c4a174097353ba4c9a2374c308dbe4102622caeb705fa2f21308a73f" dependencies = [ "tokio", ] @@ -4859,9 +4947,9 @@ dependencies = [ [[package]] name = "find-msvc-tools" -version = "0.1.9" +version = "0.1.11" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5baebc0774151f905a1a2cc41989300b1e6fbb29aff0ceffa1064fdd3088d582" +checksum = "d45db016d36b838f563236e9193d0ee6ce38f3f68b6c94e914b4929c96bbb890" [[package]] name = "flatbuffers" @@ -4875,12 +4963,12 @@ dependencies = [ [[package]] name = "flate2" -version = "1.1.9" +version = "1.1.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "843fba2746e448b37e26a819579957415c8cef339bf08564fe8b7ddbd959573c" +checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb" dependencies = [ "crc32fast", - "miniz_oxide", + "miniz_oxide 0.9.1", "zlib-rs", ] @@ -5017,9 +5105,9 @@ checksum = "e6d5a32815ae3f33302d95fdcb2ce17862f8c65363dcfd29360480ba1001fc9c" [[package]] name = "futures" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a88cf1f829d945f548cf8fec32c61b1f202b6d93b45848602fc02af4b12ad218" +checksum = "9a31d2a3fbaaeb2af2368bbdd904aa8e812d3c04a1ee10d3171f52d556e5d0a3" dependencies = [ "futures-channel", "futures-core", @@ -5032,9 +5120,9 @@ dependencies = [ [[package]] name = "futures-channel" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "262590f4fe6afeb0bc83be1daa64e52657fe185690a958af7f3ad0e92085c5ae" +checksum = "b1f9e3d69d39e4862ffed03ed071a76f9a13ba1d9109d355b0f0aa6b15e393c4" dependencies = [ "futures-core", "futures-sink", @@ -5042,15 +5130,15 @@ dependencies = [ [[package]] name = "futures-core" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2cd50c473c80f6d7c3670a752354b8e569b1a7cbfdc0419ec88e5edad85e0dc7" +checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e" [[package]] name = "futures-executor" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6754879cc9f2c66f88c6e5c35344bb0bdb0708b0352b1201815667c7eabc7458" +checksum = "031b47cf1a3c6cc8bc2fc76cd437f521619387907d469316e7c0bc278f1f5432" dependencies = [ "futures-core", "futures-task", @@ -5070,9 +5158,9 @@ dependencies = [ [[package]] name = "futures-io" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4577ecaa3c4f96589d473f679a71b596316f6641bc350038b962a5daf0085d7a" +checksum = "53c0fa8157de1303bfffdaa1cc2a673bfffb60102f76b0ef4441659124373fed" [[package]] name = "futures-lite" @@ -5089,13 +5177,13 @@ dependencies = [ [[package]] name = "futures-macro" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2d6d3cde68c518367be28956066ddfef33813991b77a55005a69dae04bf3b10b" +checksum = "9fb9654ba8355388abeb8dcb4fc62f511300867002afc858860463bdd9fe0c44" dependencies = [ "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.4", ] [[package]] @@ -5111,15 +5199,15 @@ dependencies = [ [[package]] name = "futures-sink" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e34418ac499d6305c2fb5ad0ed2f6ac998c5f8ca209b4510f7f94242c647e307" +checksum = "1944426bf7d03f1d14f708785e4b33efd750b36d48a157b836b3efc15ede8e1d" [[package]] name = "futures-task" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b231ed28831efb4a61a08580c4bc233ec56bc009f4cd8f52da2c3cb97df0c109" +checksum = "cd417de3d1d015fc3bfd2b1ea46dfc7bab72ef86f1cc7cc9c78e728b34a6d1fd" [[package]] name = "futures-timer" @@ -5129,9 +5217,9 @@ checksum = "af43fadb8a98512d547e37b4e92e0ced13e205c061b87b4623eff01d918d6968" [[package]] name = "futures-util" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a77a90a256fce34da66415271e30f94ee91c57b04b8a2c042d9cf3220179deaa" +checksum = "0d50a92467f8ba5dd6e3ee5d4bd04d73ab2e4e1c44474a0674821dfce14b79bc" dependencies = [ "futures-channel", "futures-core", @@ -5161,9 +5249,9 @@ dependencies = [ [[package]] name = "generic-array" -version = "0.14.7" +version = "0.14.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a" +checksum = "4bb6743198531e02858aeaea5398fcc883e71851fcbcb5a2f773e2fb6cb1edf2" dependencies = [ "typenum", "version_check", @@ -5236,14 +5324,14 @@ version = "0.16.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9e2c0d8c632f8a251ce9a8198079b1022adc586ff4e3d33e18debd40eb463b31" dependencies = [ - "heck", + "heck 0.5.0", "peg", "quote", "serde", "serde_json", "syn 2.0.119", "textwrap", - "thiserror 2.0.19", + "thiserror 2.0.20", "typed-builder 0.23.2", ] @@ -5281,15 +5369,15 @@ dependencies = [ [[package]] name = "glob" -version = "0.3.3" +version = "0.3.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0cc23270f6e1808e30a928bdc84dea0b9b4136a8bc82338574f23baf47bbd280" +checksum = "e4eba85ea1d0a966a983acd07deee566e67395d2d96b6fb39e62b5a833f1eb0b" [[package]] name = "globset" -version = "0.4.19" +version = "0.4.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e47d37d2ae4464254884b60ab7071be2b876a9c35b696bd018ddcc76847309cd" +checksum = "07c34a9410465b45bd9787443bc7370f37735bad04b0f0cd57ff1a3186c98988" dependencies = [ "aho-corasick", "bstr", @@ -5467,7 +5555,7 @@ dependencies = [ "serde", "serde-wasm-bindgen", "serde_urlencoded", - "thiserror 2.0.19", + "thiserror 2.0.20", "wasm-bindgen", "web-sys", ] @@ -5503,12 +5591,12 @@ dependencies = [ "futures-core", "futures-sink", "gloo-utils 0.3.0", - "http 1.4.2", + "http 1.5.0", "js-sys", "pin-project", "serde", "serde_json", - "thiserror 2.0.19", + "thiserror 2.0.20", "wasm-bindgen", "wasm-bindgen-futures", "web-sys", @@ -5559,7 +5647,7 @@ dependencies = [ "js-sys", "serde", "serde_json", - "thiserror 2.0.19", + "thiserror 2.0.20", "wasm-bindgen", "web-sys", ] @@ -5644,7 +5732,7 @@ dependencies = [ "js-sys", "pinned", "serde", - "thiserror 2.0.19", + "thiserror 2.0.20", "wasm-bindgen", "wasm-bindgen-futures", "web-sys", @@ -5729,7 +5817,7 @@ dependencies = [ "futures-sink", "futures-util", "http 0.2.12", - "indexmap 2.14.0", + "indexmap 2.14.1", "slab", "tokio", "tokio-util", @@ -5738,17 +5826,17 @@ dependencies = [ [[package]] name = "h2" -version = "0.4.15" +version = "0.4.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6cb093c84e8bd9b188d4c4a8cb6579fc016968d14c99882163cd3ff402a4f155" +checksum = "ef8e5e5a340588f4452631496976cf8636d4a7ecf600239fdc27615d2530bc16" dependencies = [ "atomic-waker", "bytes", "fnv", "futures-core", "futures-sink", - "http 1.4.2", - "indexmap 2.14.0", + "http 1.5.0", + "indexmap 2.14.1", "slab", "tokio", "tokio-util", @@ -5779,9 +5867,9 @@ dependencies = [ [[package]] name = "handlebars" -version = "6.4.3" +version = "6.4.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4633d16a2350341713c379d6d06a4b9e1845329386026a49ce4fd09c2f3b16f6" +checksum = "75c54236f9045c8004a77942bebc52145b4844639db934a5c70fe08617fbe61a" dependencies = [ "derive_builder", "log", @@ -5790,7 +5878,7 @@ dependencies = [ "pest_derive", "serde", "serde_json", - "thiserror 2.0.19", + "thiserror 2.0.20", ] [[package]] @@ -5799,7 +5887,7 @@ version = "0.1.0" dependencies = [ "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.4", ] [[package]] @@ -5872,6 +5960,12 @@ dependencies = [ "stable_deref_trait", ] +[[package]] +name = "heck" +version = "0.4.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "95505c38b4572b2d910cecb0281560f54b440a19336cbbcb27bf6ce6adc6f5a8" + [[package]] name = "heck" version = "0.5.0" @@ -5880,9 +5974,9 @@ checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea" [[package]] name = "hermit-abi" -version = "0.5.2" +version = "0.5.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fc0fef456e4baa96da950455cd02c081ca953b141298e41db3fc7e36b1da849c" +checksum = "e17592d60ebacc7d5e169f4663c5f84f9161cc90328abcfe8456f41e4dfcb284" [[package]] name = "hex" @@ -5907,7 +6001,7 @@ dependencies = [ "ipnet", "jni", "rand 0.10.2", - "thiserror 2.0.19", + "thiserror 2.0.20", "tinyvec", "tokio", "tracing", @@ -5928,7 +6022,7 @@ dependencies = [ "prefix-trie", "rand 0.10.2", "ring", - "thiserror 2.0.19", + "thiserror 2.0.20", "tinyvec", "tracing", "url", @@ -5955,7 +6049,7 @@ dependencies = [ "resolv-conf", "smallvec", "system-configuration", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tracing", ] @@ -6029,9 +6123,9 @@ dependencies = [ [[package]] name = "http" -version = "1.4.2" +version = "1.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6970f50e31d6fc17d3fa27329444bfa74e196cf62e95052a3f6fee181dba6425" +checksum = "918d3568bebf352712bc2ef3d46a8bcf1a75b373be6539de198e9105cbbf9ce0" dependencies = [ "bytes", "itoa", @@ -6055,18 +6149,18 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ca2a8f2913ee65f60facd6a5905613afaa448497a0230cc41ce022d93290bc2c" dependencies = [ "bytes", - "http 1.4.2", + "http 1.5.0", ] [[package]] name = "http-body-util" -version = "0.1.4" +version = "0.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e9f41fd6a08e4d4ec69df65976da761afd5ad5e58a9d4acb46bd1c953a9e3ff2" +checksum = "23169fe34a5fbcdd3f3862e78fb9b6fccd5f02a6dc6f732547005d45631ce71c" dependencies = [ "bytes", "futures-core", - "http 1.4.2", + "http 1.5.0", "http-body 1.1.0", "pin-project-lite", ] @@ -6083,7 +6177,7 @@ version = "2.1.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0f056c8559e3757392c8d091e796416e4649d8e49e88b8d76df6c002f05027fd" dependencies = [ - "http 1.4.2", + "http 1.5.0", "serde", ] @@ -6124,7 +6218,7 @@ dependencies = [ "hwlocality-sys", "libc", "strum 0.28.0", - "thiserror 2.0.19", + "thiserror 2.0.20", "windows-sys 0.61.2", ] @@ -6147,25 +6241,25 @@ dependencies = [ [[package]] name = "hybrid-array" -version = "0.4.13" +version = "0.4.14" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "818356c5132c1fede50f837ca96afbe78ff42413047f4abb886217845e1b6c8c" +checksum = "707114b52a152fa7bdb290cd7cd5912d9467273b6d74e21b8d81aca1f8533f6b" dependencies = [ "typenum", ] [[package]] name = "hyper" -version = "1.11.0" +version = "1.11.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d22053281f852e11534f5198498373cbb59295120a20771d90f7ed1897490a72" +checksum = "27b501faa50e7a26c3d3560ca625132f4078a17771f4810baf70475ae48cbe43" dependencies = [ "atomic-waker", "bytes", "futures-channel", "futures-core", - "h2 0.4.15", - "http 1.4.2", + "h2 0.4.19", + "http 1.5.0", "http-body 1.1.0", "httparse", "httpdate", @@ -6196,7 +6290,7 @@ version = "0.27.9" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "33ca68d021ef39cf6463ab54c1d0f5daf03377b70561305bb89a8f83aab66e0f" dependencies = [ - "http 1.4.2", + "http 1.5.0", "hyper", "hyper-util", "log", @@ -6231,14 +6325,14 @@ dependencies = [ "bytes", "futures-channel", "futures-util", - "http 1.4.2", + "http 1.5.0", "http-body 1.1.0", "hyper", "ipnet", "libc", "percent-encoding", "pin-project-lite", - "socket2 0.6.5", + "socket2", "system-configuration", "tokio", "tower-service", @@ -6293,7 +6387,7 @@ checksum = "6e50fb53c7480f414911ab76b6e14c0042d54ad5c30f6ac2bf2f52a487a25716" dependencies = [ "aes-gcm 0.10.3", "anyhow", - "apache-avro", + "apache-avro 0.21.0", "array-init", "arrow-arith", "arrow-array", @@ -6350,7 +6444,7 @@ checksum = "d1500a26a9b18f286a319e914ce7290dddb7e2eef810edf535afc4ce0c8a6f45" dependencies = [ "async-trait", "chrono", - "http 1.4.2", + "http 1.5.0", "iceberg", "itertools 0.13.0", "reqwest 0.12.28", @@ -6384,9 +6478,9 @@ dependencies = [ [[package]] name = "icu_collections" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2984d1cd16c883d7935b9e07e44071dca8d917fd52ecc02c04d5fa0b5a3f191c" +checksum = "fa68d21081c4a05d5a901a1c62add574c77048b6a1c67be3b50ce0b60d4ca513" dependencies = [ "displaydoc", "potential_utf", @@ -6398,9 +6492,9 @@ dependencies = [ [[package]] name = "icu_locale_core" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92219b62b3e2b4d88ac5119f8904c10f8f61bf7e95b640d25ba3075e6cac2c29" +checksum = "d56e28588da92eee5c3201a6eff33fabdd49b62269c8938d4ff050ce4d900deb" dependencies = [ "displaydoc", "litemap", @@ -6411,9 +6505,9 @@ dependencies = [ [[package]] name = "icu_normalizer" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c56e5ee99d6e3d33bd91c5d85458b6005a22140021cc324cea84dd0e72cff3b4" +checksum = "12f9cf5f235641ed274641dd81c3f28d870e276763d0797aeeab72317b1c646f" dependencies = [ "icu_collections", "icu_normalizer_data", @@ -6425,16 +6519,17 @@ dependencies = [ [[package]] name = "icu_normalizer_data" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "da3be0ae77ea334f4da67c12f149704f19f81d1adf7c51cf482943e84a2bad38" +checksum = "1563da1ed3e0b3bf3d74c9b85917ac9c56464d2f57242270c09c9e752f8021a0" [[package]] name = "icu_properties" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bee3b67d0ea5c2cca5003417989af8996f8604e34fb9ddf96208a033901e70de" +checksum = "7e7ca276ad3145661a65914e6daf131ca5120cd3dcee8f8f3214b8875184a148" dependencies = [ + "displaydoc", "icu_collections", "icu_locale_core", "icu_properties_data", @@ -6445,15 +6540,15 @@ dependencies = [ [[package]] name = "icu_properties_data" -version = "2.2.0" +version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8e2bbb201e0c04f7b4b3e14382af113e17ba4f63e2c9d2ee626b720cbce54a14" +checksum = "e590f038c1464a96894fd6d10127e90a8be4509f56ff7ecef851b15cee0b7caa" [[package]] name = "icu_provider" -version = "2.2.0" +version = "2.3.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "139c4cf31c8b5f33d7e199446eff9c1e02decfc2f0eec2c8d71f65befa45b421" +checksum = "d27bbb9d3abbefac45d55f647c9de1d44aafcd1186eb91879afef17c396c3e73" dependencies = [ "displaydoc", "icu_locale_core", @@ -6579,7 +6674,7 @@ dependencies = [ "serde", "serde_json", "tempfile", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tracing", "tracing-subscriber", @@ -6612,9 +6707,9 @@ dependencies = [ "serde", "serde_json", "tempfile", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", - "toml 1.1.3+spec-1.1.0", + "toml 1.1.4+spec-1.1.0", "tracing", "tracing-appender", "tracing-subscriber", @@ -6667,10 +6762,10 @@ dependencies = [ "sysinfo 0.39.6", "system_stats", "tempfile", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", - "toml 1.1.3+spec-1.1.0", - "tower-http 0.7.0", + "toml 1.1.4+spec-1.1.0", + "tower-http 0.7.1", "tracing", "tracing-opentelemetry", "tracing-subscriber", @@ -6685,8 +6780,8 @@ dependencies = [ "bytes", "kafka-protocol", "libc", - "socket2 0.6.5", - "thiserror 2.0.19", + "socket2", + "thiserror 2.0.20", "tokio", "tokio-util", "tracing", @@ -6716,13 +6811,13 @@ dependencies = [ "sd-notify", "serde", "serde_json", - "socket2 0.6.5", + "socket2", "strum 0.28.0", "tempfile", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tokio-util", - "tower-http 0.7.0", + "tower-http 0.7.1", "tracing", "tracing-opentelemetry", "tracing-subscriber", @@ -6737,7 +6832,7 @@ dependencies = [ "bytes", "enumset", "secrecy", - "thiserror 2.0.19", + "thiserror 2.0.20", "twox-hash", ] @@ -6745,10 +6840,10 @@ dependencies = [ name = "iggy_common" version = "0.11.0-edge.6" dependencies = [ - "aes-gcm 0.11.0", + "aes-gcm 0.11.1", "async-broadcast", "async-trait", - "base64 0.22.1", + "base64 0.23.1", "blake3", "bon", "byte-unit", @@ -6766,7 +6861,7 @@ dependencies = [ "serde_json", "serde_with", "strum 0.28.0", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tracing", "tungstenite 0.30.0", @@ -6790,7 +6885,7 @@ dependencies = [ "serde_json", "simd-json", "tokio", - "toml 1.1.3+spec-1.1.0", + "toml 1.1.4+spec-1.1.0", "tracing", ] @@ -6816,7 +6911,7 @@ name = "iggy_connector_doris_sink" version = "0.2.0-edge.4" dependencies = [ "async-trait", - "base64 0.22.1", + "base64 0.23.1", "blake3", "bytes", "humantime", @@ -6836,7 +6931,7 @@ name = "iggy_connector_elasticsearch_sink" version = "0.5.0-edge.4" dependencies = [ "async-trait", - "base64 0.22.1", + "base64 0.23.1", "dashmap", "elasticsearch", "iggy_common", @@ -6872,7 +6967,7 @@ name = "iggy_connector_http_sink" version = "0.5.0-edge.4" dependencies = [ "async-trait", - "base64 0.22.1", + "base64 0.23.1", "bytes", "humantime", "iggy_connector_sdk", @@ -6885,7 +6980,7 @@ dependencies = [ "simd-json", "strum_macros 0.28.0", "tokio", - "toml 1.1.3+spec-1.1.0", + "toml 1.1.4+spec-1.1.0", "tracing", ] @@ -6916,7 +7011,7 @@ version = "0.5.0-edge.4" dependencies = [ "async-trait", "axum", - "base64 0.22.1", + "base64 0.23.1", "bytes", "iggy_common", "iggy_connector_sdk", @@ -6927,7 +7022,7 @@ dependencies = [ "serde_json", "simd-json", "tokio", - "toml 1.1.3+spec-1.1.0", + "toml 1.1.4+spec-1.1.0", "tracing", ] @@ -6938,7 +7033,7 @@ dependencies = [ "ahash 0.8.12", "async-trait", "axum", - "base64 0.22.1", + "base64 0.23.1", "chrono", "csv", "dashmap", @@ -6952,7 +7047,7 @@ dependencies = [ "serde_json", "simd-json", "tokio", - "toml 1.1.3+spec-1.1.0", + "toml 1.1.4+spec-1.1.0", "tracing", "uuid", ] @@ -6962,7 +7057,7 @@ name = "iggy_connector_meilisearch_sink" version = "0.5.0-edge.4" dependencies = [ "async-trait", - "base64 0.22.1", + "base64 0.23.1", "iggy_common", "iggy_connector_sdk", "meilisearch-sdk", @@ -7015,7 +7110,7 @@ name = "iggy_connector_postgres_source" version = "0.5.0-edge.4" dependencies = [ "async-trait", - "base64 0.22.1", + "base64 0.23.1", "chrono", "dashmap", "futures", @@ -7090,7 +7185,7 @@ name = "iggy_connector_s3_sink" version = "0.5.0-edge.4" dependencies = [ "async-trait", - "base64 0.22.1", + "base64 0.23.1", "byte-unit", "chrono", "dashmap", @@ -7111,12 +7206,12 @@ name = "iggy_connector_sdk" version = "0.4.0-edge.3" dependencies = [ "anyhow", - "apache-avro", + "apache-avro 0.22.0", "async-trait", - "base64 0.22.1", + "base64 0.23.1", "dashmap", "flatbuffers", - "http 1.4.2", + "http 1.5.0", "humantime", "iggy", "iggy_common", @@ -7135,7 +7230,7 @@ dependencies = [ "simd-json", "strum 0.28.0", "strum_macros 0.28.0", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tracing", "tracing-subscriber", @@ -7159,7 +7254,7 @@ name = "iggy_connector_surrealdb_sink" version = "0.5.0-edge.4" dependencies = [ "async-trait", - "base64 0.22.1", + "base64 0.23.1", "bytes", "iggy_common", "iggy_connector_sdk", @@ -7187,7 +7282,7 @@ dependencies = [ "rand 0.10.2", "serde", "serde_json", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tracing", "tracing-subscriber", @@ -7195,9 +7290,9 @@ dependencies = [ [[package]] name = "ignore" -version = "0.4.31" +version = "0.4.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7f8a7b8211e695a1d0cd91cace480d4d0bd57667ab10277cc412c5f7f4884f83" +checksum = "00b69833ed729dc5aa7d19541d96d6cf8e9137194207a04916d658e43168402f" dependencies = [ "crossbeam-deque", "globset", @@ -7229,7 +7324,7 @@ dependencies = [ "rayon", "rgb", "tiff", - "zune-core 0.5.1", + "zune-core 0.5.3", "zune-jpeg 0.5.15", ] @@ -7251,15 +7346,15 @@ checksum = "edcd27d72f2f071c64249075f42e205ff93c9a4c5f6c6da53e79ed9f9832c285" [[package]] name = "imgref" -version = "1.12.2" +version = "1.12.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "89194689a993ab15268672e99e7b0e19da2da3268ac682e8f02d29d4d1434cd7" +checksum = "6e44b0a4eaa4c82f441d50a963f2d5f05a787240aeee097597033e72accfd22f" [[package]] name = "impl-more" -version = "0.3.1" +version = "0.3.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "35a84fd5aa25fae5c0f4a33d9cac2ca017fc622cbd089be2229993514990f870" +checksum = "277ff51754a3f68f12f58446c5d006aa8baa4914ea273cce24a599cfaff33d4f" [[package]] name = "implicit-clone" @@ -7268,7 +7363,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1689b939ee35e3a075b0834b5672efd43aec8a6e81a1c6002b76a5ca2f211ae0" dependencies = [ "implicit-clone-derive", - "indexmap 2.14.0", + "indexmap 2.14.1", ] [[package]] @@ -7294,9 +7389,9 @@ dependencies = [ [[package]] name = "indexmap" -version = "2.14.0" +version = "2.14.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d466e9454f08e4a911e14806c24e16fba1b4c121d1ea474396f396069cf949d9" +checksum = "07aa2048142242915a31d35844fb311e0e53fcca590c3a0a40dcf1b841fa09eb" dependencies = [ "equivalent", "hashbrown 0.17.1", @@ -7318,9 +7413,9 @@ checksum = "c8fae54786f62fb2918dcfae3d568594e50eb9b5c25bf04371af6fe7516452fb" [[package]] name = "inotify" -version = "0.11.4" +version = "0.11.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "153be1941a183ec9ccd095ddbe17a8b8d435ef6c76e9e02451b933c3999af2c8" +checksum = "4cc00ea907cab49550b7da656f80ebb97be1b997d931fbcd28d39734e17ce592" dependencies = [ "bitflags 2.13.1", "inotify-sys", @@ -7342,7 +7437,6 @@ version = "0.1.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "879f10e63c20629ecabbb64a8010319738c66a5cd0c29b02d63d272b03751d01" dependencies = [ - "block-padding", "generic-array", ] @@ -7352,6 +7446,7 @@ version = "0.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "4250ce6452e92010fdf7268ccc5d14faa80bb12fc741938534c58f16804e03c7" dependencies = [ + "block-padding", "hybrid-array", ] @@ -7368,7 +7463,7 @@ dependencies = [ "arrow", "assert_cmd", "async-trait", - "base64 0.22.1", + "base64 0.23.1", "bon", "bytemuck", "bytes", @@ -7377,9 +7472,9 @@ dependencies = [ "configs", "configs_derive", "consensus", - "ctor 1.0.9", + "ctor 1.0.13", "deltalake", - "dtor 1.0.5", + "dtor 1.0.6", "figment", "futures", "harness_derive", @@ -7392,7 +7487,7 @@ dependencies = [ "iggy_connector_doris_sink", "iggy_connector_sdk", "journal", - "jsonwebtoken", + "jsonwebtoken 11.0.0", "keyring-core", "lazy_static", "libc", @@ -7420,7 +7515,7 @@ dependencies = [ "testcontainers-modules", "tokio", "tokio-postgres", - "toml 1.1.3+spec-1.1.0", + "toml 1.1.4+spec-1.1.0", "tracing", "tracing-subscriber", "twox-hash", @@ -7453,9 +7548,9 @@ dependencies = [ [[package]] name = "io-uring" -version = "0.7.13" +version = "0.7.14" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9080b15e63775b9a2ac7dca720f7050a8b955e092ea0f6020a4a80f69998cdc0" +checksum = "d64d8ca234d152948ceaede1f419b6a83983a5ecccaac05fb337a809c96d3aa6" dependencies = [ "bitflags 2.13.1", "cfg-if", @@ -7468,7 +7563,7 @@ version = "0.3.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "4d40460c0ce33d6ce4b0630ad68ff63d6661961c48b6dba35e5a4d81cfb48222" dependencies = [ - "socket2 0.6.5", + "socket2", "widestring", "windows-registry", "windows-result 0.4.1", @@ -7477,9 +7572,9 @@ dependencies = [ [[package]] name = "ipnet" -version = "2.12.0" +version = "2.12.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d98f6fed1fde3f8c21bc40a1abb88dd75e67924f9cffc3ef95607bad8017f8e2" +checksum = "6a756c3fac73139e83f14c2d742155dd2b78d3ee56597b419a0579b7bdd6dd78" dependencies = [ "serde", ] @@ -7525,9 +7620,9 @@ checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" [[package]] name = "jiff" -version = "0.2.34" +version = "0.2.35" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e184d09547b80eb7e20d141ba2fb1fbac843ca53f4cf1b31210adc4c1adc6e16" +checksum = "668b7183bd07af9a4885f5c35b0cc5c83c4607a913c16b7e17291832910d2dcc" dependencies = [ "defmt", "jiff-core", @@ -7553,9 +7648,9 @@ dependencies = [ [[package]] name = "jiff-static" -version = "0.2.34" +version = "0.2.35" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "323da076b7a6faf914dc677cb05a4b907742ff7375c8322c9e7f5061e5e0e9de" +checksum = "3a69dcb3a21cfb32ce1cd056169337ca284af0766dd766e7878819b251a49204" dependencies = [ "jiff-core", "proc-macro2", @@ -7590,7 +7685,7 @@ dependencies = [ "jni-sys", "log", "simd_cesu8", - "thiserror 2.0.19", + "thiserror 2.0.20", "walkdir", "windows-link 0.2.1", ] @@ -7653,9 +7748,9 @@ dependencies = [ [[package]] name = "js-sys" -version = "0.3.103" +version = "0.3.104" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "53b44bfcdb3f8d5837a46dae1ca9660a837176eee74a28b229bc626816589102" +checksum = "0e0c1080212aad755ea003d18543e8768dd432c48819efd73a7bf1e39b7a5a3a" dependencies = [ "cfg-if", "futures-util", @@ -7675,8 +7770,32 @@ dependencies = [ "js-sys", "p256", "p384", - "pem", - "rand 0.8.7", + "pem 3.0.6", + "rand 0.8.8", + "rsa", + "serde", + "serde_json", + "sha2 0.10.9", + "signature", + "simple_asn1", + "zeroize", +] + +[[package]] +name = "jsonwebtoken" +version = "11.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "881733cbc631fc9e472e24447ce32a64bedf2da498d6d8570b08edc87de71f65" +dependencies = [ + "base64 0.22.1", + "ed25519-dalek", + "getrandom 0.2.17", + "hmac 0.12.1", + "js-sys", + "p256", + "p384", + "pem 3.0.6", + "rand 0.8.8", "rsa", "serde", "serde_json", @@ -7705,7 +7824,7 @@ dependencies = [ "clap", "hex", "iggy-gateway-kafka", - "indexmap 2.14.0", + "indexmap 2.14.1", "kafka-protocol", "tokio", "tracing", @@ -7714,31 +7833,26 @@ dependencies = [ [[package]] name = "kafka-protocol" -version = "0.17.0" +version = "0.18.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "66292444a1cd4d430d450d472c30cba839d0724229aba2d79affffcf901516e2" +checksum = "099d5c2f1b40cd830cbf18ca4d2a0805f2875b811ca18f0372b2433e68fe2dda" dependencies = [ "anyhow", "bytes", "crc", "crc32c", - "flate2", - "indexmap 2.14.0", - "lz4", - "paste", - "snap", + "indexmap 2.14.1", "uuid", - "zstd", ] [[package]] name = "keccak" -version = "0.2.0" +version = "0.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9e24a010dd405bd7ed803e5253182815b41bf2e6a80cc3bfc066658e03a198aa" +checksum = "d8f198d1db720e4940b5a493201d199d9f24f568f8f746bd13706243a2f71598" dependencies = [ "cfg-if", - "cpufeatures 0.3.0", + "cpufeatures 0.3.1", ] [[package]] @@ -7752,9 +7866,9 @@ dependencies = [ [[package]] name = "kqueue" -version = "1.2.0" +version = "1.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "273c0752728918e0ac4976f2b275b6fefb9ecd400585dec929419f3844cd87b5" +checksum = "8d763e5b24120b4ddf50de6c92308156765aabfbbccebf401da7cff2d70a41ea" dependencies = [ "kqueue-sys", "libc", @@ -7789,9 +7903,9 @@ checksum = "d4345964bb142484797b161f473a503a434de77149dd8c7427788c6e13379388" [[package]] name = "lazy-regex" -version = "3.6.0" +version = "3.6.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6bae91019476d3ec7147de9aa291cadb6d870abf2f3015d2da73a90325ac1496" +checksum = "4994ba703f78b083e2f7946dac9251abd83fd43a0365f030e99b69be5b4b9ef9" dependencies = [ "lazy-regex-proc_macros", "once_cell", @@ -7800,9 +7914,9 @@ dependencies = [ [[package]] name = "lazy-regex-proc_macros" -version = "3.6.0" +version = "3.6.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4de9c1e1439d8b7b3061b2d209809f447ca33241733d9a3c01eabf2dc8d94358" +checksum = "fd97232314824e6dbef1918a871bb93f51070455e3715bf26e19a6d01aa977a0" dependencies = [ "proc-macro2", "quote", @@ -7901,9 +8015,9 @@ checksum = "34b357333733e8260735ba5894eb928c02ecc69c78715f01a8019e7fa7f2db4c" [[package]] name = "libc" -version = "0.2.188" +version = "0.2.189" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "22053b6a34f84abc97f9129e61334f40174659a1b9bd18c970b83db6a9a6348b" +checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" [[package]] name = "libfuzzer-sys" @@ -7917,9 +8031,9 @@ dependencies = [ [[package]] name = "libgit2-sys" -version = "0.18.5+1.9.4" +version = "0.18.8+1.9.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "005d6ae6eac1912906073e069f7db60b1fa98e052a68227824afe3e3a1c59ca2" +checksum = "7f7c568b25d7489bc3fb2988ed69ab111d2944d2f5fec3d5c987fe545ea97b50" dependencies = [ "cc", "libc", @@ -7939,18 +8053,18 @@ dependencies = [ [[package]] name = "liblzma" -version = "0.4.7" +version = "0.4.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "45aec2360b3933207e27908049d8e4df4e476b58180afb1e56b2a4fb72efe4ba" +checksum = "2fe0a34ca854fd4f20c07f696fc8675aec78f87d88d29f5e10257a7490a1b2e1" dependencies = [ "liblzma-sys", ] [[package]] name = "liblzma-sys" -version = "0.4.7" +version = "0.4.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a046c7f353ba30f810545151e04f63545833803f5b86ee3ddf1517247fe560a5" +checksum = "a0dad045e4b1b7b170be4b60b54b780cafb4490165461bac7d1cf7b703f61d5f" dependencies = [ "cc", "libc", @@ -7974,9 +8088,9 @@ dependencies = [ [[package]] name = "libredox" -version = "0.1.18" +version = "0.1.23" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c943259e342f1e06ff2da7a83eabdfe7f92ce10262688dbf1895ff0b3e6e4652" +checksum = "8d8f1ea3f21fd3405dcaf6c9b5c1630af9afc422d9073ea39c5f6d6c772e08ed" dependencies = [ "libc", ] @@ -8011,9 +8125,9 @@ checksum = "b685d66585d646efe09fec763d796c291049c8b6bf84e04954bffc8748341f0d" [[package]] name = "link-section" -version = "0.19.0" +version = "0.19.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e333fe507b738576d6da5bb3f1a7d7a1c80307ed9ef31624c057d844c19c93e9" +checksum = "39c29a617ce3df32c08497bdc1ab6e2376e0b17948ac166a2fbe5977c5954cd9" [[package]] name = "linked-hash-map" @@ -8023,9 +8137,9 @@ checksum = "0717cef1bc8b636c6e1c1bbdefc09e6322da8a9321966e8928ef80d20f7f770f" [[package]] name = "linktime-proc-macro" -version = "0.2.0" +version = "0.2.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8c7b0a3383c2a1002d11349c92c85a666a5fb679e96c79d782cf0dbe557fd6ee" +checksum = "7e57c38c1e860fd37c604281cdfb1dd2216977fd76a50f85ba2f388ef3219616" [[package]] name = "linux-raw-sys" @@ -8041,9 +8155,9 @@ checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53" [[package]] name = "litemap" -version = "0.8.2" +version = "0.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92daf443525c4cce67b150400bc2316076100ce0b3686209eb8cf3c31612e6f0" +checksum = "47d9d19d1d6efa0109d2f65ff4c85cddd50bd572e5a00127ab10987290bcefae" [[package]] name = "local-channel" @@ -8058,9 +8172,9 @@ dependencies = [ [[package]] name = "local-event" -version = "0.1.2" +version = "0.1.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "76dda8459b10a8960dfae91c1c77316fc15e7caf94f9f18794126512a8dc77e5" +checksum = "23ab4b951e96ffb2da6e25cec3d038d0e2930f4c406d1ec194b5120ae41ec6d3" [[package]] name = "local-waker" @@ -8079,9 +8193,9 @@ dependencies = [ [[package]] name = "log" -version = "0.4.33" +version = "0.4.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad" +checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" [[package]] name = "logos" @@ -8180,25 +8294,6 @@ version = "0.1.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "112b39cec0b298b6c1999fee3e31427f74f676e4cb9879ed1a121b43661a4154" -[[package]] -name = "lz4" -version = "1.28.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a20b523e860d03443e98350ceaac5e71c6ba89aea7d960769ec3ce37f4de5af4" -dependencies = [ - "lz4-sys", -] - -[[package]] -name = "lz4-sys" -version = "1.11.1+lz4-1.10.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6bd8c0d6c6ed0cd30b3652886bb8711dc4bb01d637a68105a3d5158039b418e6" -dependencies = [ - "cc", - "libc", -] - [[package]] name = "lz4_flex" version = "0.13.1" @@ -8364,14 +8459,14 @@ dependencies = [ "futures-io", "futures-util", "iso8601", - "jsonwebtoken", + "jsonwebtoken 10.4.0", "log", "meilisearch-index-setting-macro", "pin-project-lite", "reqwest 0.12.28", "serde", "serde_json", - "thiserror 2.0.19", + "thiserror 2.0.20", "time", "tokio", "uuid", @@ -8429,9 +8524,9 @@ dependencies = [ "rustls-pemfile", "scopeguard", "server_common", - "socket2 0.6.5", + "socket2", "tempfile", - "thiserror 2.0.19", + "thiserror 2.0.20", "tracing", "zeroize", ] @@ -8532,6 +8627,16 @@ dependencies = [ "simd-adler32", ] +[[package]] +name = "miniz_oxide" +version = "0.9.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b63fbc4a50860e98e7b2aa7804ded1db5cbc3aff9193adaff57a6931bf7c4b4c" +dependencies = [ + "adler2", + "simd-adler32", +] + [[package]] name = "mio" version = "1.2.2" @@ -8578,9 +8683,9 @@ checksum = "21909324aa58f5c284d91cac514c6d081210901dc372d7c8ea6a9d7e0406097a" [[package]] name = "moka" -version = "0.12.15" +version = "0.12.16" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "957228ad12042ee839f93c8f257b62b4c0ab5eaae1d4fa60de53b27c9d7c5046" +checksum = "4293f18e7567a1caf3c584855554377025c65e0aa445344d04171f5ad63d19b9" dependencies = [ "async-lock", "crossbeam-channel", @@ -8616,9 +8721,9 @@ checksum = "851fac73f7fe22f6a3ab87f720ce509cae7c9fd08e7dd27866cc232dee07ccf4" [[package]] name = "mongodb" -version = "3.8.0" +version = "3.8.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b814038f367d212f55de0a630cb35102a9b8ca23785a86955d62c0087c93846d" +checksum = "d220eb9ba80bad420e1f9efc4e895fede7418f8779a8e38a3b750e57ff397720" dependencies = [ "base64 0.22.1", "bitflags 2.13.1", @@ -8647,11 +8752,11 @@ dependencies = [ "serde_with", "sha1 0.11.0", "sha2 0.11.0", - "socket2 0.6.5", + "socket2", "stringprep", "strsim", "take_mut", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tokio-rustls", "tokio-util", @@ -8662,9 +8767,9 @@ dependencies = [ [[package]] name = "mongodb-internal-macros" -version = "3.8.0" +version = "3.8.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f736d2fbc56e0011a341fbb9172bd822fda75c5f93b82fae1c7aab1e2613c810" +checksum = "e38ff3c46c59c2f4d9b26a86e6980dc23583e9d4b45aa579cef6584642093128" dependencies = [ "macro_magic", "proc-macro2", @@ -8842,7 +8947,7 @@ checksum = "c89e69e7e0f03bea5ef08013795c25018e101932225a656383bd384495ecc367" dependencies = [ "num-integer", "num-traits", - "rand 0.8.7", + "rand 0.8.8", "serde", ] @@ -8857,7 +8962,7 @@ dependencies = [ "num-integer", "num-iter", "num-traits", - "rand 0.8.7", + "rand 0.8.8", "smallvec", "zeroize", ] @@ -8891,9 +8996,9 @@ dependencies = [ [[package]] name = "num-integer" -version = "0.1.46" +version = "0.1.47" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7969661fd2958a5cb096e56c8e1ad0444ac2bbcd0061bd28660485a44879858f" +checksum = "7ce2d95d4b3734dc35aa2f45e1aa22cd416814592a4f9d9205e11affd5b8e10b" dependencies = [ "num-traits", ] @@ -8910,9 +9015,9 @@ dependencies = [ [[package]] name = "num-modular" -version = "0.6.4" +version = "0.6.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fc41a1374056e9672221567958a66c16be12d0e2c1b408761e14d901c237d5e0" +checksum = "bd8e500409e6cd603b03e477c26a6caecdc27ac58979a53e881c75eafc079f44" [[package]] name = "num-order" @@ -9031,9 +9136,9 @@ dependencies = [ [[package]] name = "object" -version = "0.37.3" +version = "0.39.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ff76201f031d8863c38aa7f905eca4f53abbfa15f609db4277d44cd8938f33fe" +checksum = "2e5a6c098c7a3b6547378093f5cc30bc54fd361ce711e05293a5cc589562739b" dependencies = [ "memchr", ] @@ -9052,7 +9157,7 @@ dependencies = [ "futures-channel", "futures-core", "futures-util", - "http 1.4.2", + "http 1.5.0", "http-body-util", "httparse", "humantime", @@ -9069,7 +9174,7 @@ dependencies = [ "serde", "serde_json", "serde_urlencoded", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tracing", "url", @@ -9080,9 +9185,9 @@ dependencies = [ [[package]] name = "octocrab" -version = "0.54.0" +version = "0.54.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "27be39870c558e1fbc5a5f8d4a29aa916268c1dbcb9d467f2a9bbab35598060e" +checksum = "432fad6f447c52acda76a7a0660f022b21f4c42f373bbdc29f04332ba19a89bf" dependencies = [ "arc-swap", "async-trait", @@ -9094,7 +9199,7 @@ dependencies = [ "futures", "futures-util", "getrandom 0.2.17", - "http 1.4.2", + "http 1.5.0", "http-body 1.1.0", "http-body-util", "http-serde", @@ -9102,9 +9207,10 @@ dependencies = [ "hyper-rustls", "hyper-timeout", "hyper-util", - "jsonwebtoken", + "jsonwebtoken 10.4.0", "percent-encoding", "pin-project", + "rustls", "secrecy", "serde", "serde_json", @@ -9156,7 +9262,7 @@ version = "0.57.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "96c9c85ce253ff87225e7669979d877a20c98a06604ec9d6dd5f4473e08f1ae1" dependencies = [ - "ctor 1.0.9", + "ctor 1.0.13", "opendal-core", "opendal-layer-concurrent-limit", "opendal-layer-logging", @@ -9176,7 +9282,7 @@ dependencies = [ "base64 0.22.1", "bytes", "futures", - "http 1.4.2", + "http 1.5.0", "http-body 1.1.0", "jiff", "log", @@ -9201,7 +9307,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0d6f81ba6960e3fae1882f253b114b21d7e444e1534f209c7737a79f6243eb6f" dependencies = [ "futures", - "http 1.4.2", + "http 1.5.0", "mea", "opendal-core", ] @@ -9260,7 +9366,7 @@ dependencies = [ "base64 0.22.1", "bytes", "crc32c", - "http 1.4.2", + "http 1.5.0", "log", "md-5 0.11.0", "opendal-core", @@ -9288,7 +9394,7 @@ dependencies = [ "futures-sink", "js-sys", "pin-project-lite", - "thiserror 2.0.19", + "thiserror 2.0.20", "tracing", ] @@ -9313,7 +9419,7 @@ checksum = "5683015d09e2df236ef005b17f6f196f0d5f6313c4fa43a7b6a53b52776e4331" dependencies = [ "async-trait", "bytes", - "http 1.4.2", + "http 1.5.0", "opentelemetry", "reqwest 0.13.4", ] @@ -9324,14 +9430,14 @@ version = "0.32.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9966929966d17620d7c316c643ba62631826e10021409357772d5eea84f62c35" dependencies = [ - "http 1.4.2", + "http 1.5.0", "opentelemetry", "opentelemetry-http", "opentelemetry-proto", "opentelemetry_sdk", "prost", "reqwest 0.13.4", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tonic", "tonic-types", @@ -9369,7 +9475,7 @@ dependencies = [ "percent-encoding", "portable-atomic", "rand 0.9.5", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tokio-stream", ] @@ -9418,12 +9524,42 @@ dependencies = [ "pin-project-lite", ] +[[package]] +name = "ouroboros" +version = "0.18.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1e0f050db9c44b97a94723127e6be766ac5c340c48f2c4bb3ffa11713744be59" +dependencies = [ + "aliasable", + "ouroboros_macro", + "static_assertions", +] + +[[package]] +name = "ouroboros_macro" +version = "0.18.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3c7028bdd3d43083f6d8d4d5187680d0d3560d54df4cc9d752005268b41e64d0" +dependencies = [ + "heck 0.4.1", + "proc-macro2", + "proc-macro2-diagnostics", + "quote", + "syn 2.0.119", +] + [[package]] name = "outref" version = "0.5.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1a80800c0488c3a21695ea981a54918fbb37abf04f4d0720c453632255e2ff0e" +[[package]] +name = "owo-colors" +version = "4.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "13c45bb4a6ae1280ec0803b1ef9d3455eb50f01efbbe1447ab020f1d54fba9d8" + [[package]] name = "p256" version = "0.13.2" @@ -9450,9 +9586,9 @@ dependencies = [ [[package]] name = "papaya" -version = "0.2.4" +version = "0.2.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "997ee03cd38c01469a7046643714f0ad28880bcb9e6679ff0666e24817ca19b7" +checksum = "da2442474a9404698c42509b8967f437249dbc7b50493e83020333d3943ec0ae" dependencies = [ "equivalent", "seize", @@ -9585,13 +9721,12 @@ dependencies = [ [[package]] name = "password-hash" -version = "0.5.0" +version = "0.6.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "346f04948ba92c43e8469c1ee6736c7563d71012b17d40745260fe106aac2166" +checksum = "aab41826031698d6ffcd9cff78ef56ef998e39dc7e5067cdfebe373842d4723b" dependencies = [ - "base64ct", - "rand_core 0.6.4", - "subtle", + "getrandom 0.4.3", + "phc", ] [[package]] @@ -9681,6 +9816,16 @@ dependencies = [ "serde_core", ] +[[package]] +name = "pem" +version = "4.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d354a98a3d1251555de99e8fdd8afda05573c31b82f59063a7b0a29b5527f120" +dependencies = [ + "base64 0.23.1", + "serde_core", +] + [[package]] name = "pem-rfc7468" version = "0.7.0" @@ -9704,9 +9849,9 @@ checksum = "3637c05577168127568a64e9dc5a6887da720efef07b3d9472d45f63ab191166" [[package]] name = "pest" -version = "2.8.7" +version = "2.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "47627dd7305c6a2d6c8c6bcd24c5a4c17dbbf425f4f9c5313e724b38fc9782e9" +checksum = "5a07a60cc7a4d00c91f95c685609d1d2f79050e6804b70ebedd7650f0b839bcf" dependencies = [ "memchr", "ucd-trie", @@ -9714,9 +9859,9 @@ dependencies = [ [[package]] name = "pest_derive" -version = "2.8.7" +version = "2.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4b4254325ecad416ab689e27ba51da03ba01a9632bc6e108f5fe7c3c4ad29d58" +checksum = "b3a83744a5c8455b8b3e0dc5031362780a347c878bdd11584d1a8984228cc88d" dependencies = [ "pest", "pest_generator", @@ -9724,9 +9869,9 @@ dependencies = [ [[package]] name = "pest_generator" -version = "2.8.7" +version = "2.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6c4c0e91ead7a8f7acecbca6f003fc2e8282b1dbe2dd9c9d2f16aba42995e0a7" +checksum = "e0cd3451aa3de60d4b9a1e736885e4dea6b31617598026f12256ad566d63304a" dependencies = [ "pest", "pest_meta", @@ -9737,9 +9882,9 @@ dependencies = [ [[package]] name = "pest_meta" -version = "2.8.7" +version = "2.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f9744bc48116fee06334924bb5f2bad41eed5e89bd26e29b0b799f9a3f82c210" +checksum = "e04d3a0849e241d7dfce834c83b1c5edc8622009e8dd51a12ba1927c32f05496" dependencies = [ "pest", ] @@ -9757,13 +9902,13 @@ dependencies = [ [[package]] name = "pgwire" -version = "0.40.4" +version = "0.40.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7981cfde34009be689a05a30c497ad5fbb552531d3d54230b3627264ff1bc384" +checksum = "8265901ede50d0879fe401c6fe282e7e4ff83ae10a48c1f8e4f89b92f7d5f604" dependencies = [ "async-trait", "aws-lc-rs", - "base64 0.22.1", + "base64 0.23.1", "bytes", "chrono", "derive-new", @@ -9781,13 +9926,24 @@ dependencies = [ "serde_json", "smol_str", "stringprep", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tokio-rustls", "tokio-util", "x509-certificate", ] +[[package]] +name = "phc" +version = "0.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "44dc769b75f93afdddd8c7fa12d685292ddeff1e66f7f0f3a234cf1818afe892" +dependencies = [ + "base64ct", + "ctutils", + "getrandom 0.4.3", +] + [[package]] name = "phf" version = "0.12.1" @@ -9908,9 +10064,9 @@ dependencies = [ [[package]] name = "pkg-config" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "19f132c84eca552bf34cab8ec81f1c1dcc229b811638f9d283dceabe58c5569e" +checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548" [[package]] name = "png" @@ -9922,7 +10078,7 @@ dependencies = [ "crc32fast", "fdeflate", "flate2", - "miniz_oxide", + "miniz_oxide 0.8.9", ] [[package]] @@ -9935,7 +10091,7 @@ dependencies = [ "crc32fast", "fdeflate", "flate2", - "miniz_oxide", + "miniz_oxide 0.8.9", ] [[package]] @@ -9977,15 +10133,15 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f0fa31d631f2b2cb2a544d0aa321ce847a94764d701ca2becc411138b93d49cd" dependencies = [ "cpubits", - "cpufeatures 0.3.0", + "cpufeatures 0.3.1", "universal-hash 0.6.1", ] [[package]] name = "portable-atomic" -version = "1.14.0" +version = "1.15.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3d20d5497ef88037a52ff98267d066e7f11fcc5e99bbfbd58a42336193aacec3" +checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85" [[package]] name = "portable-atomic-util" @@ -10044,9 +10200,9 @@ dependencies = [ [[package]] name = "potential_utf" -version = "0.1.5" +version = "0.1.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0103b1cef7ec0cf76490e969665504990193874ea05c85ff9bab8b911d0a0564" +checksum = "d83eb9bc6d8e5cf568e7a1101d60ee05e81ed50ea106026f3d18deeb046d7661" dependencies = [ "zerovec", ] @@ -10117,6 +10273,16 @@ dependencies = [ "syn 2.0.119", ] +[[package]] +name = "prettyplease" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2bfe0f4c752e450fc2faf62654f1c134747922825d5b04ca717b8874f41a40c0" +dependencies = [ + "proc-macro2", + "syn 3.0.4", +] + [[package]] name = "primeorder" version = "0.13.6" @@ -10257,9 +10423,9 @@ dependencies = [ [[package]] name = "prometheus-client" -version = "0.25.0" +version = "0.25.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ba70bf887030e45213b4a95c9b08d5a450b157f87c1d63661ed0847a12fa2aad" +checksum = "1784dc11a05f5a8f57c4a62915686be140f24f70301290b997e317241fa9ffc2" dependencies = [ "dtoa", "itoa", @@ -10269,13 +10435,13 @@ dependencies = [ [[package]] name = "prometheus-client-derive-encode" -version = "0.5.0" +version = "0.5.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9adf1691c04c0a5ff46ff8f262b58beb07b0dbb61f96f9f54f6cbd82106ed87f" +checksum = "01e34894696ff94f64a20c2c373a6440903e9c2789a303d68ec6e6f953f890e4" dependencies = [ "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.4", ] [[package]] @@ -10334,7 +10500,7 @@ dependencies = [ "prost-reflect", "prost-types", "protox-parse", - "thiserror 2.0.19", + "thiserror 2.0.20", ] [[package]] @@ -10346,14 +10512,14 @@ dependencies = [ "logos 0.15.1", "miette", "prost-types", - "thiserror 2.0.19", + "thiserror 2.0.20", ] [[package]] name = "psm" -version = "0.1.31" +version = "0.1.32" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "645dbe486e346d9b5de3ef16ede18c26e6c70ad97418f4874b8b1889d6e761ea" +checksum = "4dcd034599e63b970727f70d79e02d62390a4a84f7c6b827c27c46d5ac3fa622" dependencies = [ "ar_archive_writer", "cc", @@ -10487,8 +10653,8 @@ dependencies = [ "quinn-udp", "rustc-hash", "rustls", - "socket2 0.6.5", - "thiserror 2.0.19", + "socket2", + "thiserror 2.0.20", "tokio", "tracing", "web-time", @@ -10496,9 +10662,9 @@ dependencies = [ [[package]] name = "quinn-proto" -version = "0.11.16" +version = "0.11.17" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2f4bfc015262b9df63c8845072ce59068853ff5872180c2ce2f13038b970e560" +checksum = "04759210543be93709136e28212294a659ef5001836ff4eab4d663e4529bba83" dependencies = [ "aws-lc-rs", "bytes", @@ -10513,7 +10679,7 @@ dependencies = [ "rustls-pki-types", "rustls-platform-verifier", "slab", - "thiserror 2.0.19", + "thiserror 2.0.20", "tinyvec", "tracing", "web-time", @@ -10528,7 +10694,7 @@ dependencies = [ "cfg_aliases", "libc", "once_cell", - "socket2 0.6.5", + "socket2", "tracing", "windows-sys 0.61.2", ] @@ -10562,9 +10728,9 @@ checksum = "dc33ff2d4973d518d823d61aa239014831e521c75da58e3df4840d3f47749d09" [[package]] name = "rand" -version = "0.8.7" +version = "0.8.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "22f6172bdec972074665ed81ed53b71da00bfc44b65a753cfde883ec4c702a1a" +checksum = "e058c7de0b26af77780c769414d6257830bb240f3c38477dbc2c16e5f54d6d4c" dependencies = [ "libc", "rand_chacha 0.3.1", @@ -10684,7 +10850,7 @@ dependencies = [ "rand 0.9.5", "rand_chacha 0.9.0", "simd_helpers", - "thiserror 2.0.19", + "thiserror 2.0.20", "v_frame", "wasm-bindgen", ] @@ -10735,11 +10901,11 @@ dependencies = [ [[package]] name = "rcgen" -version = "0.14.8" +version = "0.14.10" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "57f6d249aad744e274e682777a50283a225a32705394ee6d5fcc01efa25e4055" +checksum = "8774e05a7d0de114588e6a28fe7e71694b82614ed569d86d8b389dfbc98b8ad8" dependencies = [ - "pem", + "pem 4.0.0", "ring", "rustls-pki-types", "time", @@ -10790,27 +10956,27 @@ checksum = "a4e608c6638b9c18977b00b475ac1f28d14e84b27d8d42f70e0bf1e3dec127ac" dependencies = [ "getrandom 0.2.17", "libredox", - "thiserror 2.0.19", + "thiserror 2.0.20", ] [[package]] name = "ref-cast" -version = "1.0.26" +version = "1.0.27" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "216e8f773d7923bcba9ceb86a86c93cabb3903a11872fc3f138c49630e50b96d" +checksum = "7e440fb4e4b4147295338efb76001ab9e4efc0e5839df2c47fc5ac2381d365c3" dependencies = [ "ref-cast-impl", ] [[package]] name = "ref-cast-impl" -version = "1.0.26" +version = "1.0.27" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2c9283685feec7d69af75fb0e858d5e7378f33fe4fc699383b2916ab9273e03c" +checksum = "92ecd8964f8453721699a1ed72037b0db49ce2f5a5138486ee89bed6f67cdf3a" dependencies = [ "proc-macro2", "quote", - "syn 3.0.2", + "syn 3.0.4", ] [[package]] @@ -10827,9 +10993,9 @@ dependencies = [ [[package]] name = "regex-automata" -version = "0.4.16" +version = "0.4.18" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8fcfdb36bda0c880c5931cdc7a2bcdc8ba4556847b9d912bca70bc94708711ad" +checksum = "ad8553b9b26413251cbf30e620595c7a41b3887f03da04579c0e6b0d6a06b4b2" dependencies = [ "aho-corasick", "memchr", @@ -10866,7 +11032,7 @@ dependencies = [ "bytes", "form_urlencoded", "hex", - "http 1.4.2", + "http 1.5.0", "log", "percent-encoding", "quick-xml 0.41.0", @@ -10885,7 +11051,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ff250f0fd0b913fbd565e405acc553da0f13bde30bfb5403178c9d0313cdc15f" dependencies = [ "bytes", - "http 1.4.2", + "http 1.5.0", "log", "quick-xml 0.41.0", "reqsign-aws-core", @@ -10905,7 +11071,7 @@ dependencies = [ "futures", "hex", "hmac 0.13.0", - "http 1.4.2", + "http 1.5.0", "jiff", "log", "mea", @@ -10936,8 +11102,8 @@ dependencies = [ "bytes", "futures-core", "futures-util", - "h2 0.4.15", - "http 1.4.2", + "h2 0.4.19", + "http 1.5.0", "http-body 1.1.0", "http-body-util", "hyper", @@ -10981,8 +11147,8 @@ dependencies = [ "futures-channel", "futures-core", "futures-util", - "h2 0.4.15", - "http 1.4.2", + "h2 0.4.19", + "http 1.5.0", "http-body 1.1.0", "http-body-util", "hyper", @@ -11022,10 +11188,10 @@ checksum = "07bc3f1384cffa4f274dad2d4ddd73aed32fed8f786d96c6be8aa4e5fd3c3b58" dependencies = [ "anyhow", "async-trait", - "http 1.4.2", + "http 1.5.0", "reqwest 0.13.4", "serde", - "thiserror 2.0.19", + "thiserror 2.0.20", "tower-service", ] @@ -11039,12 +11205,12 @@ dependencies = [ "async-trait", "futures", "getrandom 0.2.17", - "http 1.4.2", + "http 1.5.0", "hyper", "reqwest 0.13.4", "reqwest-middleware", "retry-policies", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tracing", "wasmtimer", @@ -11059,7 +11225,7 @@ dependencies = [ "anyhow", "async-trait", "getrandom 0.2.17", - "http 1.4.2", + "http 1.5.0", "matchit", "reqwest 0.13.4", "reqwest-middleware", @@ -11168,28 +11334,29 @@ dependencies = [ [[package]] name = "rmcp" -version = "2.2.0" +version = "3.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "14db48ee17a9ba61810ab1a9c1beb7d06d8136ae39ac25a1137f10d357af01af" +checksum = "42b6914fac0be956fe704a38239c3f44a9f841d1b06a5713d2f638065593f5b5" dependencies = [ "async-trait", - "base64 0.22.1", + "base64 0.23.1", "bytes", "chrono", "futures", - "http 1.4.2", + "http 1.5.0", "http-body 1.1.0", "http-body-util", + "indexmap 2.14.1", "pastey 0.2.3", "pin-project-lite", "rand 0.10.2", "reqwest 0.13.4", "rmcp-macros", - "schemars 1.2.1", + "schemars 1.2.2", "serde", "serde_json", "sse-stream", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tokio-stream", "tokio-util", @@ -11200,15 +11367,15 @@ dependencies = [ [[package]] name = "rmcp-macros" -version = "2.2.0" +version = "3.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "783d787bf21813b285f13019adc49e11af501c658890c1e519f31f937c68b7e3" +checksum = "cdf1c49bd4d52014b94db0877410db273c2008f01628b0252a2e9460ad9b7fda" dependencies = [ - "darling 0.23.0", + "darling 0.24.1", "proc-macro2", "quote", "serde_json", - "syn 2.0.119", + "syn 3.0.4", ] [[package]] @@ -11232,9 +11399,9 @@ dependencies = [ [[package]] name = "roaring" -version = "0.11.4" +version = "0.11.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1dedc5658c6ecb3bdb5ef5f3295bb9253f42dcf3fd1402c03f6b1f7659c3c4a9" +checksum = "18bd8a37d17a58532776dcdf6041ce64929adca78e8489d5cacbafe99229d3e1" dependencies = [ "bytemuck", "byteorder", @@ -11341,7 +11508,7 @@ dependencies = [ "futures-util", "hex", "hmac 0.12.1", - "http 1.4.2", + "http 1.5.0", "log", "maybe-async", "md5", @@ -11354,7 +11521,7 @@ dependencies = [ "serde_json", "sha2 0.10.9", "sysinfo 0.37.2", - "thiserror 2.0.19", + "thiserror 2.0.20", "time", "tokio", "tokio-stream", @@ -11372,7 +11539,7 @@ dependencies = [ "bytes", "num-traits", "postgres-types", - "rand 0.8.7", + "rand 0.8.8", "rkyv", "serde", "serde_json", @@ -11441,9 +11608,9 @@ dependencies = [ [[package]] name = "rustls" -version = "0.23.42" +version = "0.23.43" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3c54fcab019b409d04215d3a17cb438fd7fbf192ee61461f20f4fe18704bc138" +checksum = "0283386ce02abc0151e1761d08802dfe86c173b0b494af5cbc086574e453da06" dependencies = [ "aws-lc-rs", "log", @@ -11478,9 +11645,9 @@ dependencies = [ [[package]] name = "rustls-pki-types" -version = "1.15.0" +version = "1.15.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "764899a24af3980067ee14bc143654f297b22eaebfe3c7b6b211920a5a59b046" +checksum = "2f4925028c7eb5d1fcdaf196971378ed9d2c1c4efc7dc5d011256f76c99c0a96" dependencies = [ "web-time", "zeroize", @@ -11515,9 +11682,9 @@ checksum = "f87165f0995f63a9fbeea62b64d10b4d9d8e78ec6d7d51fb2125fda7bb36788f" [[package]] name = "rustls-webpki" -version = "0.103.13" +version = "0.103.15" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "61c429a8649f110dddef65e2a5ad240f747e85f7758a6bccc7e5777bd33f756e" +checksum = "f3c3cf1d8b1e7d4927e2d154c3fcb02979afb9939629c62cd9048d4f07b60ac2" dependencies = [ "aws-lc-rs", "ring", @@ -11606,9 +11773,9 @@ dependencies = [ [[package]] name = "schemars" -version = "1.2.1" +version = "1.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a2b42f36aa1cd011945615b92222f6bf73c599a102a300334cd7f8dbeec726cc" +checksum = "687274d293b6cdc6e73e0fee520bf2049650090d7164f87672d212a3c530cf4a" dependencies = [ "chrono", "dyn-clone", @@ -11620,14 +11787,14 @@ dependencies = [ [[package]] name = "schemars_derive" -version = "1.2.1" +version = "1.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7d115b50f4aaeea07e79c1912f645c7513d81715d0420f8bc77a18c6260b307f" +checksum = "d98c67716b46af2f0b8cf752abc930f6f9aecfbf671ecfb531db8a31dbe4e2ba" dependencies = [ "proc-macro2", "quote", "serde_derive_internals", - "syn 2.0.119", + "syn 3.0.4", ] [[package]] @@ -11694,20 +11861,20 @@ dependencies = [ [[package]] name = "secret-service" -version = "5.1.0" +version = "5.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9a62d7f86047af0077255a29494136b9aaaf697c76ff70b8e49cded4e2623c14" +checksum = "5107b24b91445dd2aa449a258a1807b63240942157292354dc5bfdbeb8bc6db8" dependencies = [ - "aes 0.8.4", + "aes 0.9.3", "cbc", "futures-util", - "generic-array", - "getrandom 0.2.17", - "hkdf 0.12.4", + "getrandom 0.4.3", + "hkdf 0.13.0", + "hybrid-array", "num", "once_cell", "serde", - "sha2 0.10.9", + "sha2 0.11.0", "zbus", ] @@ -11826,18 +11993,18 @@ checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" dependencies = [ "proc-macro2", "quote", - "syn 3.0.2", + "syn 3.0.4", ] [[package]] name = "serde_derive_internals" -version = "0.29.1" +version = "0.30.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "18d26a20a969b9e3fdf2fc2d9f21eda6c40e2de84c9408bb5d3b05d499aae711" +checksum = "f852137cce035d6a4df67ccce505ff6b3e9fd3a10e3e52b24dc71e650bb1a9bd" dependencies = [ "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.4", ] [[package]] @@ -11846,7 +12013,7 @@ version = "1.0.151" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" dependencies = [ - "indexmap 2.14.0", + "indexmap 2.14.1", "itoa", "memchr", "serde", @@ -11873,7 +12040,7 @@ checksum = "8d3b1629de253c70a0508c3899572da79ca359fdab27c7920ff00406df418906" dependencies = [ "proc-macro2", "quote", - "syn 3.0.2", + "syn 3.0.4", ] [[package]] @@ -11916,24 +12083,25 @@ dependencies = [ "num-bigint", "serde", "smallvec", - "thiserror 2.0.19", + "thiserror 2.0.20", "v8", ] [[package]] name = "serde_with" -version = "3.21.0" +version = "3.22.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "76a5c54c7310e7b8b9577c286d7e399ddd876c3e12b3ed917a8aabc4b96e9e8c" +checksum = "ee78f1fbe43ac4a0e47aadb3dbd357b69eb0d3793e948624cd03dd2750ab1c0a" dependencies = [ "base64 0.22.1", "bs58", "chrono", "hex", "indexmap 1.9.3", - "indexmap 2.14.0", + "indexmap 2.14.1", + "jiff", "schemars 0.9.0", - "schemars 1.2.1", + "schemars 1.2.2", "serde_core", "serde_json", "serde_with_macros", @@ -11942,9 +12110,9 @@ dependencies = [ [[package]] name = "serde_with_macros" -version = "3.21.0" +version = "3.22.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "84d57bc0c8b9a17920c178daa6bb924850d54a9c97ab45194bb8c17ad66bb660" +checksum = "8705578779c2b6bd90d84d66eb2e206b708b1a4d7b9f17641b293545bf1c7e46" dependencies = [ "darling 0.23.0", "proc-macro2", @@ -11958,7 +12126,7 @@ version = "0.10.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7b4db627b98b36d4203a7b458cf3573730f2bb591b28871d916dfa9efabfd41f" dependencies = [ - "indexmap 2.14.0", + "indexmap 2.14.1", "itoa", "ryu", "serde", @@ -11967,9 +12135,9 @@ dependencies = [ [[package]] name = "serial_test" -version = "3.5.0" +version = "4.0.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "699f4197115b8a7e7ff19c9a315a4bd6fffec26cc4626ef45ecaea389e081c6d" +checksum = "a6df5ed973ad8d834e09f824f9e9f449af6b9a3745f78dec7cc752770bd3bf11" dependencies = [ "futures-executor", "futures-util", @@ -11981,13 +12149,13 @@ dependencies = [ [[package]] name = "serial_test_derive" -version = "3.5.0" +version = "4.0.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "94e153fc76e1c6a068703d6d29c508a0b15c061c4b7e43da59cc097bc342673c" +checksum = "a22144e767da4ddd8416dbf383700542ffd8a5dc493dfecedfe1fe3ad03c98ae" dependencies = [ "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.4", ] [[package]] @@ -12030,7 +12198,7 @@ dependencies = [ "iggy_binary_protocol", "iggy_common", "journal", - "jsonwebtoken", + "jsonwebtoken 11.0.0", "left-right", "message_bus", "metadata", @@ -12061,16 +12229,16 @@ dependencies = [ "serde_json", "server_common", "shard", - "socket2 0.6.5", + "socket2", "strum 0.28.0", - "syn 2.0.119", + "syn 3.0.4", "sysinfo 0.39.6", "system_stats", "tempfile", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", - "toml 1.1.3+spec-1.1.0", - "tower-http 0.7.0", + "toml 1.1.4+spec-1.1.0", + "tower-http 0.7.1", "tracing", "tracing-appender", "tracing-opentelemetry", @@ -12112,7 +12280,7 @@ dependencies = [ "serial_test", "smallvec", "tempfile", - "thiserror 2.0.19", + "thiserror 2.0.20", "tracing", "tracing-appender", "tracing-opentelemetry", @@ -12138,7 +12306,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "aacc4cc499359472b4abe1bf11d0b12e688af9a805fa5e3016f9a386dc2d0214" dependencies = [ "cfg-if", - "cpufeatures 0.3.0", + "cpufeatures 0.3.1", "digest 0.11.3", ] @@ -12160,7 +12328,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "446ba717509524cb3f22f17ecc096f10f4822d76ab5c0b9822c5f9c284e825f4" dependencies = [ "cfg-if", - "cpufeatures 0.3.0", + "cpufeatures 0.3.1", "digest 0.11.3", ] @@ -12193,7 +12361,7 @@ dependencies = [ "partitions", "prometheus-client", "server_common", - "thiserror 2.0.19", + "thiserror 2.0.20", "tracing", ] @@ -12246,9 +12414,9 @@ checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea" [[package]] name = "simd-json" -version = "0.17.3" +version = "0.18.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e32d7ab2678282d21e53374fbead7119b7eacbede73685dcaac472870a29a11c" +checksum = "6ebe141eb4a23ecc4d5de93fa5b139ac65dc07c38037a2fe9a8fb7c9d36bd832" dependencies = [ "halfbrown", "ref-cast", @@ -12291,7 +12459,7 @@ checksum = "0d585997b0ac10be3c5ee635f1bab02d512760d14b7c468801ac8a01d9ae5f1d" dependencies = [ "num-bigint", "num-traits", - "thiserror 2.0.19", + "thiserror 2.0.20", "time", ] @@ -12318,7 +12486,7 @@ dependencies = [ "futures", "iggy_binary_protocol", "iggy_common", - "indexmap 2.14.0", + "indexmap 2.14.1", "journal", "message_bus", "metadata", @@ -12358,9 +12526,9 @@ dependencies = [ [[package]] name = "smallvec" -version = "1.15.2" +version = "1.16.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8ed6a63f02c8539c91a8685a86f4099661ba3da017932f6ebbea6de3f0fa7c90" +checksum = "b9be42f50aa861c555654aa3a37f52f4b1074bacf4e48fe0ef7fa584e80f1f0f" dependencies = [ "serde", ] @@ -12407,7 +12575,7 @@ version = "0.8.9" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c1c97747dbf44bb1ca44a561ece23508e99cb592e862f22222dcf42f51d1e451" dependencies = [ - "heck", + "heck 0.5.0", "proc-macro2", "quote", "syn 2.0.119", @@ -12419,16 +12587,6 @@ version = "1.1.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "199905e6153d6405f9728fe44daace35f8f837bbf830bb6e85fbd5828709a886" -[[package]] -name = "socket2" -version = "0.5.10" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e22376abed350d73dd1cd119b57ffccad95b4e585a7cda43e286245ce23c0678" -dependencies = [ - "libc", - "windows-sys 0.52.0", -] - [[package]] name = "socket2" version = "0.6.5" @@ -12556,7 +12714,7 @@ dependencies = [ "futures-util", "hashbrown 0.16.1", "hashlink", - "indexmap 2.14.0", + "indexmap 2.14.1", "log", "memchr", "percent-encoding", @@ -12565,7 +12723,7 @@ dependencies = [ "serde_json", "sha2 0.10.9", "smallvec", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tokio-stream", "tracing", @@ -12596,7 +12754,7 @@ dependencies = [ "cfg-if", "dotenvy", "either", - "heck", + "heck 0.5.0", "hex", "proc-macro2", "quote", @@ -12608,7 +12766,7 @@ dependencies = [ "sqlx-postgres", "sqlx-sqlite", "syn 2.0.119", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "url", ] @@ -12636,7 +12794,7 @@ dependencies = [ "sha1 0.11.0", "sha2 0.11.0", "sqlx-core", - "thiserror 2.0.19", + "thiserror 2.0.20", "tracing", "uuid", ] @@ -12672,7 +12830,7 @@ dependencies = [ "smallvec", "sqlx-core", "stringprep", - "thiserror 2.0.19", + "thiserror 2.0.20", "tracing", "uuid", "whoami", @@ -12698,7 +12856,7 @@ dependencies = [ "percent-encoding", "serde", "sqlx-core", - "thiserror 2.0.19", + "thiserror 2.0.20", "tracing", "url", "uuid", @@ -12706,9 +12864,9 @@ dependencies = [ [[package]] name = "sse-stream" -version = "0.2.4" +version = "0.2.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "39f24a9b78c40b90817bbcd1821c74ddfd74916aadd29403d001532a9195532d" +checksum = "c123f296ade4ec4b8b0f6162116e6629f5146922ca5ab40ca9d3c2e73ab4761e" dependencies = [ "bytes", "futures-util", @@ -12725,9 +12883,9 @@ checksum = "6ce2be8dc25455e1f91df71bfa12ad37d7af1092ae736f3a6cd0e37bc7810596" [[package]] name = "stacker" -version = "0.1.24" +version = "0.1.25" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "640c8cdd92b6b12f5bcb1803ca3bbf5ab96e5e6b6b96b9ab77dabe9e880b3190" +checksum = "707f49d46706bacf8a2b00d51dace3f9de527c13eec3778f570c411f89e69967" dependencies = [ "cc", "cfg-if", @@ -12835,7 +12993,7 @@ version = "0.27.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7695ce3845ea4b33927c055a39dc438a45b059f7c1b3d91d38d10355fb8cbca7" dependencies = [ - "heck", + "heck 0.5.0", "proc-macro2", "quote", "syn 2.0.119", @@ -12847,7 +13005,7 @@ version = "0.28.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ab85eea0270ee17587ed4156089e10b9e6880ee688791d45a905f5b1ca36f664" dependencies = [ - "heck", + "heck 0.5.0", "proc-macro2", "quote", "syn 2.0.119", @@ -12899,9 +13057,9 @@ dependencies = [ [[package]] name = "syn" -version = "3.0.2" +version = "3.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a207d6d6a2b7fc470b80443726053f18a2481b7e1eee970597051596567987a3" +checksum = "e6275cddf4610d1775e6d1fe9469b2e77d0f39fd98fb7450901b821e0c53649f" dependencies = [ "proc-macro2", "quote", @@ -13158,7 +13316,7 @@ dependencies = [ "etcetera", "ferroid", "futures", - "http 1.4.2", + "http 1.5.0", "itertools 0.14.0", "log", "memchr", @@ -13168,7 +13326,7 @@ dependencies = [ "serde", "serde_json", "serde_with", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tokio-stream", "tokio-util", @@ -13215,11 +13373,11 @@ dependencies = [ [[package]] name = "thiserror" -version = "2.0.19" +version = "2.0.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "09a43598840e33d5b0331f38c5e30d13bb11c11210a4b58f0d9b18a5a5eefcd9" +checksum = "ec86235f5fcc2a73650310756d2ac5b138a5780bbbdfae3eeccec992c435ba4f" dependencies = [ - "thiserror-impl 2.0.19", + "thiserror-impl 2.0.20", ] [[package]] @@ -13235,13 +13393,13 @@ dependencies = [ [[package]] name = "thiserror-impl" -version = "2.0.19" +version = "2.0.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "43cbfe0cf76104d42a574802844187e84a305e531ed54455f11fbde0f10541cd" +checksum = "bc04cd3e1236dd4a98afca4569f2deb3f120e5422a4023be2cb683f8486292af" dependencies = [ "proc-macro2", "quote", - "syn 3.0.2", + "syn 3.0.4", ] [[package]] @@ -13280,9 +13438,9 @@ dependencies = [ [[package]] name = "time" -version = "0.3.54" +version = "0.3.55" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3e1d5e639ff6bab73cb6885cc7e7b1de96c3f32c68ec55f3952614bec1092244" +checksum = "cdb87b95ec50ddfa440816d227a17b2ccbdda963a316a727fda0fc4334f7d134" dependencies = [ "deranged", "libc", @@ -13347,9 +13505,9 @@ dependencies = [ [[package]] name = "tinystr" -version = "0.8.3" +version = "0.8.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c8323304221c2a851516f22236c5722a72eaa19749016521d6dff0824447d96d" +checksum = "b1e27c91459209c2986af3dcf603a5a74a4368754ce37414f59acc971167f643" dependencies = [ "displaydoc", "zerovec", @@ -13382,20 +13540,20 @@ dependencies = [ "parking_lot", "pin-project-lite", "signal-hook-registry", - "socket2 0.6.5", + "socket2", "tokio-macros", "windows-sys 0.61.2", ] [[package]] name = "tokio-macros" -version = "2.7.1" +version = "2.7.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6328af13490e73a9b4694030fafd93f8c8c6a9dede33e821c3fc63eddf8042ba" +checksum = "78773a2a397f451582ce068015985c33193cf6dea8b74d2a639fe457b2f07b0e" dependencies = [ "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.4", ] [[package]] @@ -13418,7 +13576,7 @@ dependencies = [ "postgres-protocol", "postgres-types", "rand 0.10.2", - "socket2 0.6.5", + "socket2", "tokio", "tokio-util", "whoami", @@ -13436,9 +13594,9 @@ dependencies = [ [[package]] name = "tokio-stream" -version = "0.1.18" +version = "0.1.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "32da49809aab5c3bc678af03902d4ccddea2a87d028d86392a4b1560c6906c70" +checksum = "a3d06f0b082ba57c26b79407372e57cf2a1e28124f78e9479fe80322cf53420b" dependencies = [ "futures-core", "pin-project-lite", @@ -13463,15 +13621,16 @@ dependencies = [ [[package]] name = "tokio-util" -version = "0.7.18" +version = "0.7.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9ae9cec805b01e8fc3fd2fe289f89149a9b66dd16786abd8b19cfa7b48cb0098" +checksum = "494815d09bf52b5548659851081238f0ca39ff638363907596da739561c62c52" dependencies = [ "bytes", "futures-core", "futures-io", "futures-sink", "futures-util", + "libc", "pin-project-lite", "tokio", ] @@ -13507,11 +13666,11 @@ dependencies = [ [[package]] name = "toml" -version = "1.1.3+spec-1.1.0" +version = "1.1.4+spec-1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "53c96ecdfa941c8fc4fcaed14f99ada8ebed502eef533015095a07e3301d4c3c" +checksum = "3aace63f4bbcdfc2c965b059de67119c89c4017a70d633be6c104910f67056f5" dependencies = [ - "indexmap 2.14.0", + "indexmap 2.14.1", "serde_core", "serde_spanned 1.1.1", "toml_datetime 1.1.1+spec-1.1.0", @@ -13544,7 +13703,7 @@ version = "0.19.15" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1b5bb770da30e5cbfde35a2d7b9b8a2c4b8ef89548a7a6aeab5c9a576e3e7421" dependencies = [ - "indexmap 2.14.0", + "indexmap 2.14.1", "toml_datetime 0.6.11", "winnow 0.5.40", ] @@ -13555,7 +13714,7 @@ version = "0.22.27" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "41fe8c660ae4257887cf66394862d21dbca4a6ddd26f04a3560410406a2f819a" dependencies = [ - "indexmap 2.14.0", + "indexmap 2.14.1", "serde", "serde_spanned 0.6.9", "toml_datetime 0.6.11", @@ -13565,9 +13724,9 @@ dependencies = [ [[package]] name = "toml_parser" -version = "1.1.2+spec-1.1.0" +version = "1.1.3+spec-1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a2abe9b86193656635d2411dc43050282ca48aa31c2451210f4202550afb7526" +checksum = "1d38ac1cf9b95face32296c0a3ede1fdc270627c9d9c02a7274dd6d960dc4d56" dependencies = [ "winnow 1.0.4", ] @@ -13594,8 +13753,8 @@ dependencies = [ "axum", "base64 0.22.1", "bytes", - "h2 0.4.15", - "http 1.4.2", + "h2 0.4.19", + "http 1.5.0", "http-body 1.1.0", "http-body-util", "hyper", @@ -13603,7 +13762,7 @@ dependencies = [ "hyper-util", "percent-encoding", "pin-project", - "socket2 0.6.5", + "socket2", "sync_wrapper", "tokio", "tokio-stream", @@ -13656,7 +13815,7 @@ checksum = "ebe5ef63511595f1344e2d5cfa636d973292adc0eec1f0ad45fae9f0851ab1d4" dependencies = [ "futures-core", "futures-util", - "indexmap 2.14.0", + "indexmap 2.14.1", "pin-project-lite", "slab", "sync_wrapper", @@ -13678,7 +13837,7 @@ dependencies = [ "bytes", "futures-core", "futures-util", - "http 1.4.2", + "http 1.5.0", "http-body 1.1.0", "http-body-util", "pin-project-lite", @@ -13693,13 +13852,13 @@ dependencies = [ [[package]] name = "tower-http" -version = "0.7.0" +version = "0.7.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b11f75e912b0c2be01b63d8cf8057b8c3f97cf34abb3d431a3a4c8675498e233" +checksum = "08a05a66a4fdd61cbbe0a1d755ffe0ca6aba159dd4820936a0ff8a8278245b9c" dependencies = [ "bitflags 2.13.1", "bytes", - "http 1.4.2", + "http 1.5.0", "http-body 1.1.0", "percent-encoding", "pin-project-lite", @@ -13740,7 +13899,7 @@ checksum = "050686193eb999b4bb3bc2acfa891a13da00f79734704c4b8b4ef1a10b368a3c" dependencies = [ "crossbeam-channel", "symlink", - "thiserror 2.0.19", + "thiserror 2.0.20", "time", "tracing-subscriber", ] @@ -13826,9 +13985,9 @@ dependencies = [ [[package]] name = "trait-variant" -version = "0.1.2" +version = "0.1.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "70977707304198400eb4835a78f6a9f928bf41bba420deb8fdb175cd965d77a7" +checksum = "b19a4867a870f6edc4c283f2b455804b1879c0baf0e642f26b03ed8ee262d9d3" dependencies = [ "proc-macro2", "quote", @@ -13858,14 +14017,14 @@ checksum = "6c01152af293afb9c7c2a57e4b559c5620b421f6d133261c60dd2d0cdb38e6b8" dependencies = [ "bytes", "data-encoding", - "http 1.4.2", + "http 1.5.0", "httparse", "log", "rand 0.9.5", "rustls", "rustls-pki-types", "sha1 0.10.7", - "thiserror 2.0.19", + "thiserror 2.0.20", ] [[package]] @@ -13876,21 +14035,21 @@ checksum = "e48ac77174b19c110a50ab2128b24215ac9cb40e0e12e093fb602d175c569d22" dependencies = [ "bytes", "data-encoding", - "http 1.4.2", + "http 1.5.0", "httparse", "log", "rand 0.10.2", "rustls", "rustls-pki-types", "sha1 0.11.0", - "thiserror 2.0.19", + "thiserror 2.0.20", ] [[package]] name = "twox-hash" -version = "2.1.3" +version = "2.1.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8464ec13c3691491391d9fce00f6416c9a48e46972f72d7865688be2080192c9" +checksum = "5283634e518fe9e82c7b20520bb4bc209009fd16c82077c802f8111ecbb0117a" dependencies = [ "rand 0.10.2", ] @@ -13994,7 +14153,7 @@ checksum = "f153acc4e99a5f2a5aefa09fb078be54e26271b2813f6041200b224c098d8328" dependencies = [ "proc-macro2", "quote", - "syn 3.0.2", + "syn 3.0.4", ] [[package]] @@ -14132,7 +14291,7 @@ version = "0.5.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "fc1de2c688dc15305988b563c3854064043356019f97a4b46276fe734c4f07ea" dependencies = [ - "crypto-common 0.1.7", + "crypto-common 0.1.6", "subtle", ] @@ -14166,11 +14325,11 @@ checksum = "8ecb6da28b8a351d773b68d5825ac39017e680750f980f3a1a85cd8dd28a47c1" [[package]] name = "ureq" -version = "3.3.0" +version = "3.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dea7109cdcd5864d4eeb1b58a1648dc9bf520360d7af16ec26d0a9354bafcfc0" +checksum = "972d7902c8735f2695410b8aed7df6ed12a47394aa1c8d7af49f0497b731a94d" dependencies = [ - "base64 0.22.1", + "base64 0.23.1", "flate2", "log", "percent-encoding", @@ -14183,12 +14342,12 @@ dependencies = [ [[package]] name = "ureq-proto" -version = "0.6.0" +version = "0.6.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e994ba84b0bd1b1b0cf92878b7ef898a5c1760108fe7b6010327e274917a808c" +checksum = "da5f78b09e6941e1a0f2e30e695e4b120377b54d5e0aec11b594bb57b3971613" dependencies = [ - "base64 0.22.1", - "http 1.4.2", + "base64 0.23.1", + "http 1.5.0", "httparse", "log", ] @@ -14265,9 +14424,9 @@ checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821" [[package]] name = "uuid" -version = "1.24.0" +version = "1.26.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bf3923a6f5c4c6382e0b653c4117f48d631ea17f38ed86e2a828e6f7412f5239" +checksum = "b5772d71c9be8a8a6ac2117d949c5b224c1b72241bb611d9a3012edcf8af7812" dependencies = [ "getrandom 0.4.3", "js-sys", @@ -14288,11 +14447,17 @@ dependencies = [ "fslock", "gzip-header", "home", - "miniz_oxide", + "miniz_oxide 0.8.9", "paste", "which", ] +[[package]] +name = "v_escape-base" +version = "0.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f1212fce830b75af194b578e55b3db9049f2c8c45f58d397fb25602fdb50fb3d" + [[package]] name = "v_frame" version = "0.3.9" @@ -14306,9 +14471,12 @@ dependencies = [ [[package]] name = "v_htmlescape" -version = "0.15.8" +version = "0.17.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4e8257fbc510f0a46eb602c10215901938b5c2a7d5e70fc11483b1d3c9b5b18c" +checksum = "befb3d53c9e3ec641417685896cbc8cc5bd264d6a2e190c56aaef1af24740d99" +dependencies = [ + "v_escape-base", +] [[package]] name = "validator" @@ -14486,9 +14654,9 @@ dependencies = [ [[package]] name = "wasm-bindgen" -version = "0.2.126" +version = "0.2.127" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4b067c0c11094aef6b7a801c1e34a26affafdf3d051dba08456b868789aaf9a4" +checksum = "1b70935747edd64d89de3efa29d73789b806c15798f8e7dca4d8ac356b50ce70" dependencies = [ "cfg-if", "once_cell", @@ -14500,9 +14668,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-futures" -version = "0.4.76" +version = "0.4.77" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c62df1340f32221cb9c54d6a27b030e3dba64361d4a95bed55f9aacb44da291d" +checksum = "6b7777d5cc23d0e91404e53ce2d5e8ec7acae3026b16233dba62cd3246457950" dependencies = [ "js-sys", "wasm-bindgen", @@ -14510,9 +14678,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-macro" -version = "0.2.126" +version = "0.2.127" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "167ce5e579f6bcf889c4f7175a8a5a585de84e8ff93976ce393efa5f2837aab1" +checksum = "77775f8f3f7217702089053b94958f8f54061a3f663417df76e19cbdcca29bc1" dependencies = [ "quote", "wasm-bindgen-macro-support", @@ -14520,9 +14688,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-macro-support" -version = "0.2.126" +version = "0.2.127" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f3997c7839262f4ef12cf90b818d6340c18e80f263f1a94bf157d0ec4420380e" +checksum = "e11d33f857dc2fb11b8bc75aee111aa9cbeb12cd9f25efd3d4c2a3dd4e235284" dependencies = [ "bumpalo", "proc-macro2", @@ -14533,9 +14701,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-shared" -version = "0.2.126" +version = "0.2.127" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dc1b4cb0cc549fcf58d7dfc081778139b3d283a081644e833e84682ad71cea24" +checksum = "7ef64dbcc55df09c7e5a46182d181c2cfa3e925f3da937ea764728b4bbb9dcbf" dependencies = [ "unicode-ident", ] @@ -14573,7 +14741,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e51cf5f08b357e64cd7642ab4bbeb11aecab9e15520692129624fb9908b8df2c" dependencies = [ "deno_error", - "thiserror 2.0.19", + "thiserror 2.0.20", ] [[package]] @@ -14592,9 +14760,9 @@ dependencies = [ [[package]] name = "web-sys" -version = "0.3.103" +version = "0.3.104" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8622dcb61c0bcc9fffa6938bed81210af2da9a7e4a1a834b2e37a59b6dfb6141" +checksum = "c435338968042f4f59a557f690a253676d47ce13ceb55d70100e7facf6620a30" dependencies = [ "js-sys", "wasm-bindgen", @@ -14658,9 +14826,9 @@ dependencies = [ [[package]] name = "whoami" -version = "2.1.2" +version = "2.1.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "998767ef88740d1f5b0682a9c53c24431453923962269c2db68ee43788c5a40d" +checksum = "626c4bac6755d76ffc12cb01b2eac751db1996b9e0041de9aa02c8c211ddc82c" dependencies = [ "libc", "libredox", @@ -15137,7 +15305,7 @@ dependencies = [ "base64 0.22.1", "deadpool", "futures", - "http 1.4.2", + "http 1.5.0", "http-body-util", "hyper", "hyper-util", @@ -15158,9 +15326,9 @@ checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" [[package]] name = "writeable" -version = "0.6.3" +version = "0.6.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1ffae5123b2d3fc086436f8834ae3ab053a283cfac8fe0a0b8eaae044768a4c4" +checksum = "3ad82d2a33cdc9674dc7465672f271e096168fcdbe0f799d9e6db8c5892679dc" [[package]] name = "wyz" @@ -15182,11 +15350,11 @@ dependencies = [ "chrono", "der", "hex", - "pem", + "pem 3.0.6", "ring", "signature", "spki", - "thiserror 2.0.19", + "thiserror 2.0.20", "zeroize", ] @@ -15204,7 +15372,7 @@ dependencies = [ "oid-registry", "ring", "rusticata-macros", - "thiserror 2.0.19", + "thiserror 2.0.20", "time", ] @@ -15273,12 +15441,12 @@ dependencies = [ "futures", "gloo 0.11.0", "implicit-clone", - "indexmap 2.14.0", + "indexmap 2.14.1", "js-sys", "rustversion", "serde", "slab", - "thiserror 2.0.19", + "thiserror 2.0.20", "tokio", "tokise", "tracing", @@ -15294,7 +15462,7 @@ version = "0.23.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c08b883d84a035f57519d057f65a2a07ae25f1884ad485e1a9a523c9d880c1ad" dependencies = [ - "prettyplease", + "prettyplease 0.2.37", "proc-macro-error", "proc-macro2", "quote", @@ -15363,9 +15531,9 @@ checksum = "c6e61e59a957b7ccee15d2049f86e8bfd6f66968fcd88f018950662d9b86e675" [[package]] name = "zbus" -version = "5.18.0" +version = "5.19.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fe18fb60dc696039e738717b76eaea21e7a4489bbb1885020b43c94236d7e98a" +checksum = "5db4be7c075cb421e4b7ee645541604239bd243ba7c357511f4ff3a74b555907" dependencies = [ "async-broadcast", "async-executor", @@ -15398,9 +15566,9 @@ dependencies = [ [[package]] name = "zbus-secret-service-keyring-store" -version = "1.0.0" +version = "1.0.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4ccede190ba363386a24e8021c7f3848393976609ec9f5d1f8c6c09ef37075b4" +checksum = "74801d001b9e7729adb4f1825b67b398185fed424749aa3d8bacf70417137d9a" dependencies = [ "keyring-core", "secret-service", @@ -15409,14 +15577,14 @@ dependencies = [ [[package]] name = "zbus_macros" -version = "5.18.0" +version = "5.19.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fe96480bed92df2b442a1a30df364e12d08eed03aeb061f2b8dc6afb2be91119" +checksum = "2990635d09ade6df1868f72f8cac69a876a90981e8bd3c40b1be413f8dc88f40" dependencies = [ "proc-macro-crate 3.3.0", "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.4", "zbus_names", "zvariant", "zvariant_utils", @@ -15433,20 +15601,29 @@ dependencies = [ "zvariant", ] +[[package]] +name = "zcheapstr" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d1afec51604565183aeb5c54c20aeab286120d4e4460f7f76e3e8bb8c0d99473" +dependencies = [ + "serde", +] + [[package]] name = "zerocopy" -version = "0.8.55" +version = "0.8.56" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b5a105cd7b140f6eeec8acff2ea38135d3cab283ada58540f629fe51e46696eb" +checksum = "556764e583adb45a9f8d413c2a147fa7e8d821e48e12b14fd560b607998b75eb" dependencies = [ "zerocopy-derive", ] [[package]] name = "zerocopy-derive" -version = "0.8.55" +version = "0.8.56" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0fe976fb70c78cd64cccfe3a6fc142244e8a77b70959b30faf9d0ac37ee228eb" +checksum = "f2ab42fc20575779bd240faa45f94a74256f755c0fa9e89f0ede20d91d0cdfc1" dependencies = [ "proc-macro2", "quote", @@ -15496,9 +15673,9 @@ dependencies = [ [[package]] name = "zerotrie" -version = "0.2.4" +version = "0.2.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0f9152d31db0792fa83f70fb2f83148effb5c1f5b8c7686c3459e361d9bc20bf" +checksum = "4ea269c3bd32f0a32c321907a2ae912ba6f4649bb0fc764a15627e99a7095a3f" dependencies = [ "displaydoc", "yoke", @@ -15507,9 +15684,9 @@ dependencies = [ [[package]] name = "zerovec" -version = "0.11.6" +version = "0.11.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "90f911cbc359ab6af17377d242225f4d75119aec87ea711a880987b18cd7b239" +checksum = "bb0464e17806c1d976d5cba29399c7f08e516e279e2ba493f63123b5fca67dd8" dependencies = [ "yoke", "zerofrom", @@ -15518,13 +15695,13 @@ dependencies = [ [[package]] name = "zerovec-derive" -version = "0.11.3" +version = "0.11.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "625dc425cab0dca6dc3c3319506e6593dcb08a9f387ea3b284dbd52a92c40555" +checksum = "34df6fc39dbd26ddc9c10e6a2984476e13acce22e64e4487636ef494369225da" dependencies = [ "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.4", ] [[package]] @@ -15547,7 +15724,7 @@ checksum = "2d04a6b5381502aa6087c94c669499eb1602eb9c5e8198e534de571f7154809b" dependencies = [ "crc32fast", "flate2", - "indexmap 2.14.0", + "indexmap 2.14.1", "memchr", "typed-path", "zopfli", @@ -15555,9 +15732,9 @@ dependencies = [ [[package]] name = "zlib-rs" -version = "0.6.6" +version = "0.6.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b142a20ec14a91d5bc708c1dc21b080c550113d8aa77afa29635673a65dd02c5" +checksum = "34b31d188d9d685a4f9c7b46d6e36631b07058d2cfe190267adce54dc230bf12" [[package]] name = "zmij" @@ -15613,9 +15790,9 @@ checksum = "3f423a2c17029964870cfaabb1f13dfab7d092a62a29a89264f4d36990ca414a" [[package]] name = "zune-core" -version = "0.5.1" +version = "0.5.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cb8a0807f7c01457d0379ba880ba6322660448ddebc890ce29bb64da71fb40f9" +checksum = "d56377fd46368984a170bc5aac5567e52ca5da874caa60bea39fcbca78fb658b" [[package]] name = "zune-inflate" @@ -15641,45 +15818,46 @@ version = "0.5.15" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "27bc9d5b815bc103f142aa054f561d9187d191692ec7c2d1e2b4737f8dbd7296" dependencies = [ - "zune-core 0.5.1", + "zune-core 0.5.3", ] [[package]] name = "zvariant" -version = "5.13.1" +version = "5.15.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bee2a0bcd2a907786a456fff45aaaaf54c9ba5f50b71ae9ec1a4edd200c94911" +checksum = "c1d34c27cc6cdd1f458427519dd6b8612f7b7e3f7b9a0b2355d041dda9869147" dependencies = [ "endi", "enumflags2", "serde", "winnow 1.0.4", + "zcheapstr", "zvariant_derive", "zvariant_utils", ] [[package]] name = "zvariant_derive" -version = "5.13.1" +version = "5.15.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "38a708216a18780796770bfe3f4739c7c83a3e8f789b755534bbbc06e4e23e12" +checksum = "864155e69b4352db0c7f374917bf45d1e0c8d17659c8b3dbf9795f3673f8c497" dependencies = [ "proc-macro-crate 3.3.0", "proc-macro2", "quote", - "syn 2.0.119", + "syn 3.0.4", "zvariant_utils", ] [[package]] name = "zvariant_utils" -version = "3.5.0" +version = "4.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "90cb9383f9b45290407a1258b202d3f8f01db719eb60b4e4055c6375af4fc7c7" +checksum = "bad0294361a320b694a328460dc73add56c306150f5cb6bfafc44446120008a3" dependencies = [ "proc-macro2", "quote", "serde", - "syn 2.0.119", + "syn 3.0.4", "winnow 1.0.4", ] diff --git a/Cargo.toml b/Cargo.toml index fd6424b3b2..af7a697ca9 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -81,20 +81,17 @@ rust-version = "1.95" warnings = "deny" [workspace.dependencies] -actix-cors = "0.7.1" -actix-files = "0.6.10" -actix-web = "4.14.0" -aes-gcm = "0.11.0" +actix-cors = "0.7.2" +actix-files = "0.7.0" +actix-web = "4.15.0" +aes-gcm = "0.11.1" ahash = { version = "0.8.12", features = ["serde"] } aligned-vec = "0.6.4" anyhow = "1.0.104" -apache-avro = "0.21.0" -apple-native-keyring-store = { version = "1.0.1", features = ["keychain"] } -# "std" is load-bearing: it cascades to rand_core/getrandom, which the -# crypto module's `OsRng` import needs even when no sibling crate in the -# build graph happens to enable it via feature unification. -argon2 = { version = "0.5.3", features = ["std"] } -# Pinned to 57 because iceberg 0.9.1 still requires arrow/parquet 57. +apache-avro = "0.22.0" +apple-native-keyring-store = { version = "1.0.2", features = ["keychain"] } +argon2 = "0.6.0" +# arrow 58: iceberg 0.10.1 and deltalake 0.32.4 still require it. arrow = "58.4.0" arrow-array = "58.4.0" arrow-json = "58.4.0" @@ -102,31 +99,31 @@ assert_cmd = "2.2.2" async-broadcast = "0.7.2" async-channel = "2.5.0" async-dropper = { version = "0.3.1", features = ["tokio", "simple"] } -async-trait = "0.1.91" -async_zip = { version = "0.0.18", features = ["tokio", "lzma", "bzip2", "xz", "deflate", "zstd"] } +async-trait = "0.1.92" +async_zip = { version = "0.0.19", features = ["tokio", "lzma", "bzip2", "xz", "deflate", "zstd"] } axum = { version = "0.8.9", features = ["macros"] } axum-server = { version = "0.8.0", features = ["tls-rustls"] } -base64 = "0.22.1" +base64 = "0.23.1" bench-dashboard-frontend = { path = "core/bench/dashboard/frontend" } bench-dashboard-server = { path = "core/bench/dashboard/server" } bench-dashboard-shared = { path = "core/bench/dashboard/shared" } bench-report = { path = "core/bench/report" } bench-runner = { path = "core/bench/runner" } bit-set = "0.11.1" -blake3 = "1.8.5" -bon = "3.9.3" +blake3 = "1.8.7" +bon = "3.10.0" byte-unit = { version = "5.2.5", default-features = false, features = ["serde", "byte", "std"] } bytemuck = { version = "1.25", features = ["derive", "min_const_generics"] } bytes = "1.12.1" cfg_aliases = "0.2.2" charming = "0.6.0" chrono = { version = "0.4.45", features = ["serde"] } -clap = { version = "4.6.3", features = ["derive", "wrap_help"] } -clap_complete = "4.6.7" +clap = { version = "4.6.6", features = ["derive", "wrap_help"] } +clap_complete = "4.6.9" clock = { path = "core/clock" } colored = "3.1.1" -comfy-table = { version = "7.2.2", default-features = false } -compio = { version = "=0.19.1", features = [ +comfy-table = { version = "8.0.0", default-features = false } +compio = { version = "0.19.2", features = [ "runtime", "macros", "io-uring", @@ -140,9 +137,9 @@ compio = { version = "=0.19.1", features = [ "fs", ] } compio-buf = "0.8.3" -compio-driver = "0.12.4" -compio-quic = "=0.8.0" -compio-ws = "=0.4.0" +compio-driver = "0.12.5" +compio-quic = "0.8.2" +compio-ws = "0.4.0" configs = { path = "core/configs", version = "0.1.0" } configs_derive = { path = "core/configs_derive", version = "0.1.0" } consensus = { path = "core/consensus" } @@ -151,13 +148,13 @@ cpu_allocation = { path = "core/cpu_allocation" } crossbeam = "0.8.4" crossfire = "3.1.19" csv = "1.4.0" -ctor = "1.0.9" +ctor = "1.0.13" ctrlc = { version = "3.5", features = ["termination"] } cucumber = "0.23" cyper = { version = "0.9.0", features = ["rustls", "stream"], default-features = false } cyper-axum = { version = "0.9.0" } cyper-core = { version = "0.9.0", default-features = false } -darling = "0.23" +darling = "0.24" dashmap = "6.2.1" deltalake = { version = "0.32.4", features = ["azure", "gcs", "s3"] } derive-new = "0.7.0" @@ -165,9 +162,9 @@ derive_builder = "0.20.2" derive_more = { version = "2.1.1", features = ["full"] } dircpy = "0.3.20" dirs = "6.0.0" -dlopen2 = "0.8.2" +dlopen2 = "0.9.0" dotenvy = "0.15.7" -dtor = "1.0.5" +dtor = "1.0.6" elasticsearch = { version = "9.1.0-alpha.1", features = ["rustls-tls"], default-features = false } enumset = "1.1" env_logger = "0.11.11" @@ -175,19 +172,15 @@ err_trail = { version = "0.11.1", features = ["tracing"] } error_set = "0.9.2" figlet-rs = "1.0.0" figment = { version = "0.10.19", features = ["toml", "env"] } -file-operation = "0.8.28" +file-operation = "0.8.32" flatbuffers = "25.12.19" flume = "0.12.0" fs2 = "0.4.3" -futures = "0.3.33" -futures-core = { version = "0.3.33", default-features = false } -# Pinned to the minor compio-tls 0.10.0 resolves to: the replica plane -# handshakes with futures-rustls directly (TLS exporter access) and hands -# the stream to compio-tls via its From impls, so both crates must agree -# on the futures-rustls types. Cargo unifies within 0.26.x; drift to 0.27 -# fails compilation loudly (see the tripwire test in message_bus). +futures = "0.3.34" +futures-core = { version = "0.3.34", default-features = false } +# Keep on the minor compio-tls 0.10 uses; the message_bus tripwire test catches drift. futures-rustls = "0.26.0" -futures-util = "0.3.33" +futures-util = "0.3.34" getrandom = { version = "0.4", features = ["wasm_js"] } git2 = { version = "0.21.0", default-features = false, features = ["vendored-libgit2"] } gloo = "0.12" @@ -196,31 +189,34 @@ harness_derive = { path = "core/harness_derive" } hash32 = "1.0.0" hex = "0.4.3" hostname = "0.4.2" -http = "1.4.2" +http = "1.5.0" human-repr = "1.1.0" humantime = "2.4.0" hwlocality = "1.0.0-alpha.12" -hyper = "1.11.0" +hyper = "1.11.1" hyper-util = { version = "0.1.20", features = ["server-auto", "service"] } iceberg = "0.10.1" iceberg-catalog-rest = "0.10.1" iceberg-storage-opendal = "0.10.1" iggy = { path = "core/sdk", version = "0.11.0-edge.6" } iggy-cli = { path = "core/cli", version = "0.14.0-edge.6" } +iggy-gateway-kafka = { path = "gateways/kafka" } iggy_binary_protocol = { path = "core/binary_protocol", version = "0.11.0-edge.6" } iggy_common = { path = "core/common", version = "0.11.0-edge.6" } +iggy_connector_doris_sink = { path = "core/connectors/sinks/doris_sink" } iggy_connector_sdk = { path = "core/connectors/sdk", version = "0.4.0-edge.3" } -indexmap = "2.14.0" +indexmap = "2.14.1" integration = { path = "core/integration" } -ipnet = "2.12.0" +ipnet = "2.12.1" journal = { path = "core/journal" } js-sys = "0.3" -jsonwebtoken = { version = "10.4.0", features = ["rust_crypto"] } +jsonwebtoken = { version = "11.0.0", features = ["rust_crypto"] } +kafka-protocol = { version = "0.18.0", default-features = false, features = ["broker"] } keyring-core = "1.0.0" lazy_static = "1.5.0" left-right = "0.11" -libc = "0.2.188" -log = "0.4.33" +libc = "0.2.189" +log = "0.4.34" lz4_flex = "0.14.0" meilisearch-sdk = { version = "0.33.0", default-features = false, features = [ "reqwest", @@ -232,11 +228,11 @@ metadata = { path = "core/metadata" } mimalloc = "0.1" mime_guess = "2.0" mockall = "0.15.0" -mongodb = { version = "3.8.0", features = ["rustls-tls"] } +mongodb = { version = "3.8.2", features = ["rustls-tls"] } nix = { version = "0.31.3", features = ["feature", "fs", "resource", "sched"] } nonzero_lit = "0.1.2" notify = "8.2.0" -octocrab = "0.54.0" +octocrab = "0.54.1" opentelemetry = { version = "0.32.0", features = ["trace", "logs"] } opentelemetry-appender-tracing = { version = "0.32.0", features = ["log"] } opentelemetry-otlp = { version = "0.32.0", features = [ @@ -255,16 +251,16 @@ opentelemetry_sdk = { version = "0.32.1", features = [ "experimental_logs_batch_log_processor_with_async_runtime", "experimental_trace_batch_span_processor_with_async_runtime", ] } -papaya = "0.2.4" +papaya = "0.2.5" parquet = "58.4.0" partitions = { path = "core/partitions" } passterm = "2.0.6" paste = "1.0" -pgwire = "0.40.4" +pgwire = "0.40.7" postcard = { version = "1.1.3", features = ["alloc"] } predicates = "3.1.4" proc-macro2 = "1" -prometheus-client = "0.25.0" +prometheus-client = "0.25.1" prost = "0.14.4" prost-types = "0.14.4" protox = "0.9.1" @@ -274,7 +270,7 @@ quote = "1" rand = "0.10.2" rand_xoshiro = "0.8.1" rayon = "1.12.0" -rcgen = "0.14.8" +rcgen = "0.14.10" regex = "1.13.1" reqwest = { version = "0.13.4", default-features = false, features = ["json", "rustls"] } reqwest-middleware = { version = "0.5.2", features = ["json", "query"] } @@ -282,13 +278,13 @@ reqwest-retry = "0.9.1" reqwest-tracing = "0.7.1" ring = "0.17.14" ringbuffer = "0.16.0" -rmcp = "2.2.0" +rmcp = "3.2.0" rmp = "0.8.15" rmp-serde = "1.3.1" rolling-file = "0.2.0" rust-embed = "8.12.0" rust-s3 = { version = "0.37.2", default-features = false, features = ["tokio-rustls-tls", "tags"] } -rustls = { version = "0.23.42", features = ["ring"] } +rustls = { version = "0.23.43", features = ["ring"] } rustls-pemfile = "2.2.0" scopeguard = "1.2.0" sd-notify = "0.5" @@ -296,23 +292,21 @@ secrecy = { version = "0.10", features = ["serde"] } send_wrapper = "0.6.0" serde = { version = "1.0.229", features = ["derive", "rc"] } serde_json = "1.0.151" -serde_with = { version = "3.21.0", features = ["base64", "macros"] } +serde_with = { version = "3.22.0", features = ["base64", "macros"] } serde_yaml_ng = "0.10.0" -serial_test = "3.5.0" -server = { path = "core/server" } +serial_test = "4.0.1" +server = { path = "core/server", default-features = false } server_common = { path = "core/server_common" } shard = { path = "core/shard" } -simd-json = { version = "0.17.3", features = ["serde_impl"] } -smallvec = "1.15" +simd-json = { version = "0.18.1", features = ["serde_impl"] } +smallvec = "1.16" socket2 = "0.6.5" sqlparser = { version = "0.62.0", features = ["visitor"] } sqlx = { version = "0.9.0", features = [ "runtime-tokio", "tls-rustls", "postgres", - # "mysql": Doris exposes a MySQL-wire frontend; the Doris sink's - # integration-test fixture talks to it over this driver. Cargo unifies - # features across the workspace, so it is declared once here. + # "mysql": the Doris integration fixture talks to Doris over its MySQL frontend. "mysql", "chrono", "uuid", @@ -322,23 +316,23 @@ static-toml = "1.3.0" strum = { version = "0.28.0", features = ["derive"] } strum_macros = "0.28.0" subtle = "2.6.1" -# Pinned to 2 because darling 0.23 still emits syn 2 types. -syn = { version = "2", features = ["full", "extra-traits"] } +syn = { version = "3", features = ["full", "extra-traits"] } sysinfo = "0.39.6" system_stats = { path = "core/system_stats" } tempfile = "3.27.0" terminal_size = { version = "0.4.4" } test-case = "3.3.1" +# Pinned to 0.27 because testcontainers-modules 0.15.0 still requires it. testcontainers = { version = "0.27.3", features = ["reusable-containers"] } testcontainers-modules = { version = "0.15.0", features = ["postgres", "http_wait"] } -thiserror = "2.0.19" +thiserror = "2.0.20" tokio = { version = "1.53.1", features = ["full"] } tokio-postgres = "0.7.18" tokio-rustls = "0.26.4" tokio-tungstenite = { version = "0.30", features = ["rustls-tls-webpki-roots"] } -tokio-util = { version = "0.7.18", features = ["compat"] } -toml = "1.1.3" -tower-http = { version = "0.7.0", features = ["add-extension", "cors", "trace"] } +tokio-util = { version = "0.7.19", features = ["compat"] } +toml = "1.1.4" +tower-http = { version = "0.7.1", features = ["add-extension", "cors", "trace"] } tracing = "0.1.44" tracing-appender = "0.2.5" tracing-opentelemetry = "0.33.0" @@ -347,13 +341,13 @@ tracing-subscriber = { version = "0.3.23", default-features = false, features = "env-filter", "ansi", ] } -trait-variant = "0.1.2" +trait-variant = "0.1.3" tungstenite = "0.30.0" -twox-hash = { version = "2.1.3", features = ["xxhash32"] } +twox-hash = { version = "2.1.4", features = ["xxhash32"] } ulid = "3.0.0" -ureq = "3.3" +ureq = "3.4" url = "2.5.8" -uuid = { version = "1.24.0", features = ["v4", "v7", "fast-rng", "serde", "zerocopy"] } +uuid = { version = "1.26.0", features = ["v4", "v7", "fast-rng", "serde", "zerocopy"] } vergen-git2 = { version = "10.0.1", features = ["build", "cargo", "rustc", "si"] } walkdir = "2.5.0" wasm-bindgen = "0.2" @@ -371,7 +365,7 @@ windows-native-keyring-store = "1.1.0" wiremock = "0.6" yew = { version = "0.23", features = ["csr"] } yew-router = "0.20" -zbus-secret-service-keyring-store = { version = "1.0.0", features = ["rt-async-io-crypto-rust"] } +zbus-secret-service-keyring-store = { version = "1.0.1", features = ["rt-async-io-crypto-rust"] } zeroize = "1.9.0" zip = { version = "8.6.0", default-features = false, features = ["deflate"] } @@ -379,19 +373,15 @@ zip = { version = "8.6.0", default-features = false, features = ["deflate"] } lto = true codegen-units = 1 -# Selectively optimize CPU-intensive dependencies in dev profile: -# - argon2: password hashing -# - twox-hash: checksum calculation -# - rand_chacha: random number generation -# - iggy_common: wrapper functions for above -# This is faster than global opt-level=1 which adds significant compile time. +# CPU-heavy crates that dominate dev-profile test time: password hashing, +# checksums, RNG. Cheaper than raising opt-level for the whole graph. [profile.dev.package.argon2] opt-level = 3 [profile.dev.package.twox-hash] opt-level = 3 -[profile.dev.package.rand_chacha] +[profile.dev.package.chacha20] opt-level = 3 [profile.dev.package.iggy_common] diff --git a/bdd/python/uv.lock b/bdd/python/uv.lock index 9198d276fa..b2510b94a0 100644 --- a/bdd/python/uv.lock +++ b/bdd/python/uv.lock @@ -13,8 +13,8 @@ source = { directory = "../../foreign/python" } [package.metadata] requires-dist = [ - { name = "maturin", marker = "extra == 'all'", specifier = ">=1.14.1,<2.0" }, - { name = "maturin", marker = "extra == 'dev'", specifier = ">=1.14.1,<2.0" }, + { name = "maturin", marker = "extra == 'all'", specifier = ">=1.15.0,<2.0" }, + { name = "maturin", marker = "extra == 'dev'", specifier = ">=1.15.0,<2.0" }, { name = "pyrefly", marker = "extra == 'all'", specifier = ">=1.2.0" }, { name = "pyrefly", marker = "extra == 'dev'", specifier = ">=1.2.0" }, { name = "pytest", marker = "extra == 'all'", specifier = ">=9.1.1,<10.0" }, diff --git a/core/bench/report/src/prints.rs b/core/bench/report/src/prints.rs index 508fb41389..e3d697341a 100644 --- a/core/bench/report/src/prints.rs +++ b/core/bench/report/src/prints.rs @@ -204,7 +204,7 @@ impl BenchmarkGroupMetrics { let mut summary_table = Table::new(); summary_table - .load_preset(UTF8_FULL) + .load_style(UTF8_FULL) .set_content_arrangement(ContentArrangement::Dynamic); summary_table.add_row(vec![ @@ -229,7 +229,7 @@ impl BenchmarkGroupMetrics { let mut latency_table = Table::new(); latency_table - .load_preset(UTF8_FULL) + .load_style(UTF8_FULL) .set_content_arrangement(ContentArrangement::Dynamic); latency_table.add_row(vec![ @@ -265,7 +265,7 @@ impl BenchmarkGroupMetrics { let mut table = Table::new(); table - .load_preset(UTF8_FULL) + .load_style(UTF8_FULL) .set_content_arrangement(ContentArrangement::Dynamic) .set_width(60); diff --git a/core/bench/src/actors/consumer/benchmark_consumer.rs b/core/bench/src/actors/consumer/benchmark_consumer.rs index ed0ff57ba9..6e2bdc739b 100644 --- a/core/bench/src/actors/consumer/benchmark_consumer.rs +++ b/core/bench/src/actors/consumer/benchmark_consumer.rs @@ -246,7 +246,7 @@ impl BenchmarkConsumer { ) { let mut summary_table = Table::new(); summary_table - .load_preset(UTF8_FULL) + .load_style(UTF8_FULL) .set_content_arrangement(ContentArrangement::Dynamic); summary_table.add_row(vec![ @@ -269,7 +269,7 @@ impl BenchmarkConsumer { let mut latency_table = Table::new(); latency_table - .load_preset(UTF8_FULL) + .load_style(UTF8_FULL) .set_content_arrangement(ContentArrangement::Dynamic); latency_table.add_row(vec![ @@ -303,7 +303,7 @@ impl BenchmarkConsumer { ) { let mut table = Table::new(); table - .load_preset(UTF8_FULL) + .load_style(UTF8_FULL) .set_content_arrangement(ContentArrangement::Dynamic) .set_width(60); diff --git a/core/bench/src/actors/producer/benchmark_producer.rs b/core/bench/src/actors/producer/benchmark_producer.rs index 6541762ded..c51c45c5f4 100644 --- a/core/bench/src/actors/producer/benchmark_producer.rs +++ b/core/bench/src/actors/producer/benchmark_producer.rs @@ -256,7 +256,7 @@ impl BenchmarkProducer

{ ) { let mut summary_table = Table::new(); summary_table - .load_preset(UTF8_FULL) + .load_style(UTF8_FULL) .set_content_arrangement(ContentArrangement::Dynamic); summary_table.add_row(vec![ @@ -279,7 +279,7 @@ impl BenchmarkProducer

{ let mut latency_table = Table::new(); latency_table - .load_preset(UTF8_FULL) + .load_style(UTF8_FULL) .set_content_arrangement(ContentArrangement::Dynamic); latency_table.add_row(vec![ @@ -313,7 +313,7 @@ impl BenchmarkProducer

{ ) { let mut table = Table::new(); table - .load_preset(UTF8_FULL) + .load_style(UTF8_FULL) .set_content_arrangement(ContentArrangement::Dynamic) .set_width(60); diff --git a/core/bench/src/actors/producing_consumer/benchmark_producing_consumer.rs b/core/bench/src/actors/producing_consumer/benchmark_producing_consumer.rs index 9382461de4..ee0539470a 100644 --- a/core/bench/src/actors/producing_consumer/benchmark_producing_consumer.rs +++ b/core/bench/src/actors/producing_consumer/benchmark_producing_consumer.rs @@ -333,7 +333,7 @@ where ) { let mut summary_table = Table::new(); summary_table - .load_preset(UTF8_FULL) + .load_style(UTF8_FULL) .set_content_arrangement(ContentArrangement::Dynamic); summary_table.add_row(vec![ @@ -356,7 +356,7 @@ where let mut latency_table = Table::new(); latency_table - .load_preset(UTF8_FULL) + .load_style(UTF8_FULL) .set_content_arrangement(ContentArrangement::Dynamic); latency_table.add_row(vec![ @@ -390,7 +390,7 @@ where ) { let mut table = Table::new(); table - .load_preset(UTF8_FULL) + .load_style(UTF8_FULL) .set_content_arrangement(ContentArrangement::Dynamic) .set_width(60); diff --git a/core/cli/src/commands/binary_client/get_client.rs b/core/cli/src/commands/binary_client/get_client.rs index 20b340da43..e65d6313ce 100644 --- a/core/cli/src/commands/binary_client/get_client.rs +++ b/core/cli/src/commands/binary_client/get_client.rs @@ -73,7 +73,7 @@ impl CliCommand for GetClientCmd { if client_details.consumer_groups_count > 0 { let mut consumer_groups = Table::new(); - consumer_groups.load_preset(ASCII_NO_BORDERS); + consumer_groups.load_style(ASCII_NO_BORDERS); consumer_groups.set_header(vec!["Stream ID", "Topic ID", "Consumer Group ID"]); for consumer_group in client_details.consumer_groups { consumer_groups.add_row(vec![ diff --git a/core/cli/src/commands/binary_consumer_groups/get_consumer_group.rs b/core/cli/src/commands/binary_consumer_groups/get_consumer_group.rs index 83b21033e1..39f70376a0 100644 --- a/core/cli/src/commands/binary_consumer_groups/get_consumer_group.rs +++ b/core/cli/src/commands/binary_consumer_groups/get_consumer_group.rs @@ -84,7 +84,7 @@ impl CliCommand for GetConsumerGroupCmd { if consumer_group.members_count > 0 { let mut members_table = Table::new(); - members_table.load_preset(ASCII_NO_BORDERS); + members_table.load_style(ASCII_NO_BORDERS); members_table.set_header(vec!["Member id", "Partitions count", "Partitions"]); for member in consumer_group.members { members_table.add_row(vec![ diff --git a/core/common/src/types/permissions/permissions_global.rs b/core/common/src/types/permissions/permissions_global.rs index 6d58cf4e0d..c9ae88b0e0 100644 --- a/core/common/src/types/permissions/permissions_global.rs +++ b/core/common/src/types/permissions/permissions_global.rs @@ -219,7 +219,7 @@ impl From for Table { fn from(value: GlobalPermissions) -> Self { let mut table = Self::new(); - table.load_preset(ASCII_NO_BORDERS); + table.load_style(ASCII_NO_BORDERS); table.set_header(vec!["Permission", "Value"]); table.add_row(vec![ "Manage Servers", @@ -264,7 +264,7 @@ impl From<&TopicPermissions> for Table { fn from(value: &TopicPermissions) -> Self { let mut table = Self::new(); - table.load_preset(ASCII_NO_BORDERS); + table.load_style(ASCII_NO_BORDERS); table.set_header(vec!["Permission", "Value"]); table.add_row(vec![ "Manage Topic", @@ -288,7 +288,7 @@ impl From<&StreamPermissions> for Table { fn from(value: &StreamPermissions) -> Self { let mut table = Self::new(); - table.load_preset(ASCII_NO_BORDERS); + table.load_style(ASCII_NO_BORDERS); table.set_header(vec!["Permission", "Value"]); table.add_row(vec![ "Manage Stream", diff --git a/core/connectors/sdk/src/decoders/avro.rs b/core/connectors/sdk/src/decoders/avro.rs index 089a9735e9..7a0e519c81 100644 --- a/core/connectors/sdk/src/decoders/avro.rs +++ b/core/connectors/sdk/src/decoders/avro.rs @@ -17,6 +17,7 @@ use crate::{Error, Payload, Schema, StreamDecoder}; use apache_avro::Schema as AvroSchema; +use apache_avro::reader::datum::GenericDatumReader; use serde::{Deserialize, Serialize}; use std::collections::HashMap; use std::path::PathBuf; @@ -134,10 +135,13 @@ impl AvroStreamDecoder { })?; let mut reader = payload; - let avro_value = apache_avro::from_avro_datum(schema, &mut reader, None).map_err(|e| { - error!("Failed to decode Avro datum: {}", e); - Error::CannotDecode(Schema::Avro) - })?; + let avro_value = GenericDatumReader::builder(schema) + .build() + .and_then(|datum_reader| datum_reader.read_value(&mut reader)) + .map_err(|e| { + error!("Failed to decode Avro datum: {}", e); + Error::CannotDecode(Schema::Avro) + })?; if !reader.is_empty() { error!( @@ -250,7 +254,11 @@ mod tests { ("name".to_string(), AvroValue::String("Alice".to_string())), ("age".to_string(), AvroValue::Int(30)), ]); - apache_avro::to_avro_datum(&schema, record).unwrap() + apache_avro::writer::datum::GenericDatumWriter::builder(&schema) + .build() + .unwrap() + .write_value_to_vec(record) + .unwrap() } #[test] diff --git a/core/connectors/sdk/src/encoders/avro.rs b/core/connectors/sdk/src/encoders/avro.rs index 103a3fb499..1d67013907 100644 --- a/core/connectors/sdk/src/encoders/avro.rs +++ b/core/connectors/sdk/src/encoders/avro.rs @@ -17,6 +17,7 @@ use crate::{Error, Payload, Schema, StreamEncoder, convert::owned_value_to_serde_json}; use apache_avro::Schema as AvroSchema; +use apache_avro::writer::datum::GenericDatumWriter; use base64::Engine; use serde::{Deserialize, Serialize}; use std::collections::HashMap; @@ -158,10 +159,13 @@ impl AvroStreamEncoder { let serde_value = owned_value_to_serde_json(&json_value); let avro_value = Self::serde_json_to_avro_value(serde_value, schema)?; - apache_avro::to_avro_datum(schema, avro_value).map_err(|e| { - error!("Failed to encode Avro datum: {}", e); - Error::Serialization(format!("Avro encoding failed: {e}")) - }) + GenericDatumWriter::builder(schema) + .build() + .and_then(|datum_writer| datum_writer.write_value_to_vec(avro_value)) + .map_err(|e| { + error!("Failed to encode Avro datum: {}", e); + Error::Serialization(format!("Avro encoding failed: {e}")) + }) } fn serde_json_to_avro_value( @@ -264,7 +268,9 @@ impl AvroStreamEncoder { } (value, schema) => { - let avro_val: AvroValue = value.into(); + let avro_val = AvroValue::try_from(value).map_err(|e| { + Error::Serialization(format!("JSON value cannot be converted to Avro: {e}")) + })?; if avro_val.validate(schema) { Ok(avro_val) } else { diff --git a/core/connectors/sdk/src/transforms/avro_convert.rs b/core/connectors/sdk/src/transforms/avro_convert.rs index 6f400022d5..6ac1819ca1 100644 --- a/core/connectors/sdk/src/transforms/avro_convert.rs +++ b/core/connectors/sdk/src/transforms/avro_convert.rs @@ -226,7 +226,11 @@ mod tests { ("name".to_string(), AvroValue::String("Alice".to_string())), ("age".to_string(), AvroValue::Int(30)), ]); - apache_avro::to_avro_datum(&schema, record).unwrap() + apache_avro::writer::datum::GenericDatumWriter::builder(&schema) + .build() + .unwrap() + .write_value_to_vec(record) + .unwrap() } #[test] diff --git a/core/integration/Cargo.toml b/core/integration/Cargo.toml index 2a4acaa090..f53fc13697 100644 --- a/core/integration/Cargo.toml +++ b/core/integration/Cargo.toml @@ -60,7 +60,7 @@ iggy_binary_protocol = { workspace = true } iggy_common = { workspace = true } # Path-dep only so the Doris integration test can reuse the connector's pure # `build_label` function — keeping the test and production label format in lock-step. -iggy_connector_doris_sink = { path = "../connectors/sinks/doris_sink" } +iggy_connector_doris_sink = { workspace = true } iggy_connector_sdk = { workspace = true, features = ["api"] } # Locates and decodes the on-disk superblock slot files in the recovery test # (`SLOT_FILE_NAMES`, `decode_slots`). diff --git a/core/integration/src/harness/handle/server.rs b/core/integration/src/harness/handle/server.rs index bac9f6cb20..1326c6cb57 100644 --- a/core/integration/src/harness/handle/server.rs +++ b/core/integration/src/harness/handle/server.rs @@ -1137,6 +1137,16 @@ impl ServerHandle { } Ok(()) } + + /// Names this node in the log dumps. A cluster failure prints every + /// node's logs back to back and the server never logs its own identity, + /// so without this the reader cannot tell which replica wrote what. + fn log_label(&self) -> String { + match self.addrs.tcp { + Some(tcp) => format!("Iggy server replica {} (tcp {tcp})", self.server_id), + None => format!("Iggy server replica {}", self.server_id), + } + } } impl Drop for ServerHandle { @@ -1145,11 +1155,12 @@ impl Drop for ServerHandle { // without waiting, so the freed slot could be reused while this server // still holds its ports. let _ = self.stop(); + let label = self.log_label(); if let Some(report) = super::common::stderr_panic_report(&self.stderr_path) { if std::thread::panicking() { // Ahead of the full dump, which buries these lines under the // complete stdout of every node. - eprintln!("Iggy server panicked:\n{report}"); + eprintln!("{label} panicked:\n{report}"); } else { // A dead task leaves the process alive and the test green; // failing here is the only thing that surfaces it. The panic @@ -1157,12 +1168,12 @@ impl Drop for ServerHandle { // print this node's logs first. let (stdout, stderr) = super::common::collect_logs(&self.stdout_path, &self.stderr_path); - eprintln!("Iggy server stdout:\n{stdout}"); - eprintln!("Iggy server stderr:\n{stderr}"); - panic!("Iggy server panicked:\n{report}"); + eprintln!("{label} stdout:\n{stdout}"); + eprintln!("{label} stderr:\n{stderr}"); + panic!("{label} panicked:\n{report}"); } } - super::common::dump_logs_on_panic("Iggy server", &self.stdout_path, &self.stderr_path); + super::common::dump_logs_on_panic(&label, &self.stdout_path, &self.stderr_path); } } diff --git a/core/integration/tests/cluster/crash_durability.rs b/core/integration/tests/cluster/crash_durability.rs index 251454d4c1..95bb23ee7a 100644 --- a/core/integration/tests/cluster/crash_durability.rs +++ b/core/integration/tests/cluster/crash_durability.rs @@ -501,8 +501,13 @@ async fn given_eager_flush_topic_when_the_whole_cluster_is_killed_should_recover let acked = produce_acked(&client, "flushed", 30).await; let payloads: Vec = acked.iter().map(|(_, payload)| payload.clone()).collect(); wait_until_payloads_installed(harness, &payloads, FLUSH_INSTALL_TIMEOUT).await; - drop(client); + // The producer's connection stays open across the kill on purpose. + // Closing it first replicates a Logout, and journaling that op is what + // tells a backup the topic create before it committed: a backup recovers + // its commit point from the watermark stamped on the NEXT entry. With the + // create as the last journaled op, the backups boot without the partition + // and must recover it from disk once the metadata election re-commits it. harness.kill_cluster().expect("SIGKILL every node"); harness .restart_cluster() diff --git a/core/partitions/src/iggy_partition.rs b/core/partitions/src/iggy_partition.rs index b08a4e2db5..1af433af23 100644 --- a/core/partitions/src/iggy_partition.rs +++ b/core/partitions/src/iggy_partition.rs @@ -8553,8 +8553,12 @@ mod purge_floor_tests { .expect("purge partition"); assert_eq!(partition.applied_purge_generation(), 4); + // Boxed: all four rebuilt partitions live across awaits. Five inline + // partitions make this future large enough that unoptimized builds, + // which keep a copy of it in every compio `block_on` wrapper frame, + // overflow the 2 MiB test thread stack. let rebuild = |created_revision: u64| { - let mut rebuilt = test_partition(); + let mut rebuilt = Box::new(test_partition()); rebuilt.set_partition_dir(dir.to_string_lossy().into_owned()); rebuilt.set_created_revision(created_revision); rebuilt diff --git a/core/server/src/boot/mod.rs b/core/server/src/boot/mod.rs index b8a5ed9a3c..841b3e3ab0 100644 --- a/core/server/src/boot/mod.rs +++ b/core/server/src/boot/mod.rs @@ -50,8 +50,9 @@ use crate::boot::recovery::{ RecoveredOwnerState, build_shard_for_thread, restore_metadata_consensus, }; use crate::boot::threads::{ - StopSignals, await_pump_drain, join_partial_shard_survivors, resolve_shard_assignments, - run_shard_thread, spawn_shutdown_watchdog, validate_sharding_runtime_knobs, + StopSignals, await_pump_drain, install_panic_hook, join_partial_shard_survivors, + resolve_shard_assignments, run_shard_thread, spawn_shutdown_watchdog, + validate_sharding_runtime_knobs, }; use crate::boot::topology::{RosterCells, resolve_tcp_topology}; use crate::dispatch::partition::make_partition_read_handler; @@ -272,6 +273,9 @@ pub fn bootstrap( let (senders, mut inboxes, mut reply_inboxes) = shard_mesh_channels(total_shards, inbox_capacity, reply_inbox_capacity); let shutdown_flag = Arc::new(AtomicBool::new(false)); + // Before the first shard thread exists, so no panic on a shard, in the + // thread body or in a task compio's `spawn` would swallow, escapes it. + let first_panic = install_panic_hook(Arc::clone(&shutdown_flag)); let config = Arc::new(config); // One owner table per server process, Arc-cloned into every shard's bus so // any shard's bus reads the same atomic slots that the owning @@ -403,6 +407,7 @@ pub fn bootstrap( shutdown_flag, shard_threads, join_timeout: config.system.sharding.shutdown_join_timeout.get_duration(), + first_panic, }) } diff --git a/core/server/src/boot/recovery.rs b/core/server/src/boot/recovery.rs index 943c73c043..0115b5c609 100644 --- a/core/server/src/boot/recovery.rs +++ b/core/server/src/boot/recovery.rs @@ -19,12 +19,8 @@ use crate::boot::topology::{RosterCells, TcpTopology, build_cluster_roster}; use crate::boot::wire_shell_handlers; -use crate::partition_helpers::{ - build_partition_fresh, configure_consumer_offsets, ensure_initial_segment, - open_partition_superblock, -}; -use crate::segment_recovery::{RecoveredSegment, load_persisted_segments}; -use crate::server_error::{PartitionRecoveryRefusal, ServerError}; +use crate::partition_helpers::load_partition_or_fence; +use crate::server_error::ServerError; use crate::session_manager::SessionManager; use crate::shell::{ ServerMetadata, ServerShard, ShellHandlers, consensus_timers, repair_retry_ticks, @@ -34,20 +30,14 @@ use consensus::{ ClientTable, JoinMode, LocalPipeline, PipelineEntry, Sequencer, VsrConsensus, VsrRestore, VsrState, }; -use iggy_common::{ - Aes256GcmEncryptor, EncryptorKind, IggyByteSize, IggyError, PartitionStats, TopicRuntimeOptions, -}; +use iggy_common::{Aes256GcmEncryptor, EncryptorKind, IggyByteSize, TopicRuntimeOptions}; use journal::Journal; use journal::prepare_journal::PrepareJournal; use journal::superblock::PingPongSuperblock; use message_bus::IggyMessageBus; -use metadata::ReplicaIdentity; use metadata::impls::metadata::{IggySnapshot, StreamsFrontend}; use metadata::stm::snapshot::Snapshot; -use metadata::stm::stream::Partition; -use partitions::{ - IggyIndexWriter, IggyPartition, IggyPartitions, MessagesWriter, PartitionsConfig, -}; +use partitions::{IggyPartitions, PartitionsConfig}; use server_common::sharding::{IggyNamespace, PartitionLocation, ShardId}; use shard::builder::IggyShardBuilder; use shard::metrics::ShardMetrics; @@ -57,12 +47,10 @@ use shard::{ ShardIdentity, TaggedSender, }; use std::cell::RefCell; -use std::path::PathBuf; use std::rc::Rc; use std::sync::Arc; -use std::sync::atomic::Ordering; use std::time::Duration; -use tracing::{error, info, warn}; +use tracing::{info, warn}; #[allow(clippy::too_many_arguments, clippy::too_many_lines)] pub(in crate::boot) async fn build_shard_for_thread( @@ -184,174 +172,21 @@ pub(in crate::boot) async fn build_shard_for_thread( // `Arc` atomics race only against other atomic adds. for (stream_id, topic_id, partition_stats, partition_metadata, topic_runtime) in owned { let namespace = IggyNamespace::new(stream_id, topic_id, partition_metadata.id); - let partition = match load_partition( + let Some(partition) = load_partition_or_fence( config, namespace, - Arc::clone(&partition_stats), + partition_stats, &partition_metadata, topic_runtime, topology.cluster_id, topology.self_replica_id, topology.replica_count, Rc::clone(&bus), + &partitions, ) - .await - { - Ok(partition) => partition, - // ONE damaged local chain must not take the node down. The shapes - // this refuses are structural -- what a failed state-transfer - // quarantine leaves behind, or damage the recovery walk proved - // inside a segment. What follows depends on whether a peer can - // restore the data. With peers, the segment files are fenced - // aside (keeping the superblock so the group cannot re-enter - // view 0), the group is materialised fresh, and the ordinary - // rejoin path (repair, then state transfer on a refused floor) - // refills it. Single-replica, only a chain-shape refusal whose - // planned chain provably holds ZERO recoverable bytes still - // fences and rebuilds: nothing servable is at stake, so an empty - // rebuild hides no loss. The verdict variant alone is not that - // evidence -- a hole and an orphan empty segment both fire over - // fully populated chains -- which is why the gate reads the byte - // total the refusal carries. Every other refusal tombstones, - // leaving its files exactly where they are: a rebuilt empty - // partition answers polls exactly like a healthy empty one and - // hides the loss, while an unrouted namespace is a failure an - // operator can see. - Err(ServerError::PartitionRecoveryRefused { dir, reason, .. }) => { - let partition_dir = dir.to_string_lossy().into_owned(); - let rebuild_for_rejoin = topology.replica_count > 1 - || matches!( - reason, - PartitionRecoveryRefusal::Hole { - recoverable_bytes: 0, - .. - } | PartitionRecoveryRefusal::EmptyNonTailSegment { - recoverable_bytes: 0, - .. - } - ); - error!( - stream_id, - topic_id, - partition_id = partition_metadata.id, - partition_dir, - %reason, - "refusing the recovered segment chain" - ); - // A pass-A refusal folded nothing into the stats (recovery - // counts only accepted chains), but the hydrate-reopen refusal - // arrives after a fully counted load, so clear them either way. - partition_stats.zero_out_all(); - if !rebuild_for_rejoin { - // No quarantine here, mirroring the superblock arm below: - // a tombstone is only durable if its cause is. Fencing the - // chain aside would leave the next boot zero segments to - // walk, so it would re-seed from the surviving superblock, - // plant a fresh segment, and serve the partition empty - // with no refusal logged. Left at their real paths, the - // same files re-derive this verdict (and this log line) - // every boot, and the reconciler's tombstone gate keeps - // the namespace away from a fresh build, whose - // initial-segment open would truncate the oldest refused - // segment in place. The one refusal whose cause is NOT - // durable is `StorageSizeMismatch`: it fires from the - // reopen right after recovery truncated the same file, so - // the next boot re-walks the already-truncated bytes and, - // unless the length diverges again, accepts the chain - // instead of re-tombstoning -- acceptable for an - // assertion that the filesystem lied about a length. - // `%reason` repeated on purpose: this is the line an - // operator greps to enumerate dark partitions, so it has - // to carry the verdict on its own. - error!( - stream_id, - topic_id, - partition_id = partition_metadata.id, - partition_dir, - %reason, - "no peer replica holds this partition's data; leaving the refused \ - segment files in place and tombstoning it instead of serving it \ - empty" - ); - partitions.tombstone(namespace); - continue; - } - match partitions::state_transfer::quarantine_segment_files(&partition_dir).await { - Ok(fenced_dir) => error!( - stream_id, - topic_id, - partition_id = partition_metadata.id, - fenced_dir, - "quarantined the refused segment files; they are kept for inspection" - ), - Err(error) => { - // NOT rebuilt: `build_partition_fresh` reaches - // `ensure_initial_segment`, which opens segment 0 with - // `file_exists = false` and TRUNCATES whatever the - // failed quarantine left behind. The likeliest failures - // (suffix cap exhausted, `create_dir_all`) move zero - // files, so rebuilding would destroy the oldest segment - // on the first attempt while the higher-offset survivors - // keep refusing every boot -- a loop that never - // terminates and eats the chain one segment at a time. - // Tombstone instead: the namespace stays unmaterialised - // and unrouted, the reconciler backs off, and an - // operator still has every byte. - error!( - stream_id, - topic_id, - partition_id = partition_metadata.id, - partition_dir, - %error, - "failed to quarantine the refused segment files; leaving this \ - partition tombstoned rather than rebuilding over them" - ); - partitions.tombstone(namespace); - continue; - } - } - build_partition_fresh( - config, - namespace, - partition_stats, - partition_metadata.created_revision, - topic_runtime, - topology.cluster_id, - topology.self_replica_id, - topology.replica_count, - partition_metadata.created_view, - Rc::clone(&bus), - ) - .await? - } - // An untrustworthy superblock fences ONE group, not the node. The - // segment files stay exactly where they are -- unlike a refused - // chain, the data on disk is not the thing in doubt -- so there is - // nothing to quarantine and nothing to rebuild: rebuilding fresh - // would hand this replica a view-0 identity while a record it - // cannot read says otherwise. Tombstoned, the namespace stays - // unmaterialised and unrouted, the reconciler backs off, and an - // operator has every byte plus a message naming the directory. - Err( - error @ (ServerError::PartitionSuperblockIo { .. } - | ServerError::PartitionSuperblockVersionUnknown { .. } - | ServerError::PartitionSuperblockUnverifiable { .. } - | ServerError::PartitionSuperblockUndecodable { .. } - | ServerError::PartitionSuperblockIdentityMismatch { .. }), - ) => { - error!( - stream_id, - topic_id, - partition_id = partition_metadata.id, - %error, - "cannot trust this partition's durable consensus state; tombstoning the \ - partition and continuing to boot the rest of the shard" - ); - partition_stats.zero_out_all(); - partitions.tombstone(namespace); - continue; - } - Err(error) => return Err(error), + .await? + else { + continue; }; partitions.insert(namespace, partition); shards_table.insert( @@ -741,353 +576,6 @@ pub(in crate::boot) fn restore_metadata_consensus( consensus } -/// Recover this partition's persisted segment chain, stamping each segment -/// with the topic's effective segment size (the per-topic value when the -/// topic was created with one, else the shard-wide configured size). -/// -/// The topic's effective `enforce_fsync` goes in for the same reason: it is -/// what tells recovery whether a durable index entry the log cannot back is a -/// benign torn index or previously durable data the log lost. -async fn recover_partition_segments( - config: &ServerConfig, - namespace: IggyNamespace, - runtime_options: TopicRuntimeOptions, - stats: &PartitionStats, -) -> Result, ServerError> { - let stream_id = namespace.stream_id(); - let topic_id = namespace.topic_id(); - let partition_id = namespace.partition_id(); - let segment_size = runtime_options - .segment_size - .unwrap_or_else(|| IggyByteSize::from(iggy_common::DEFAULT_SEGMENT_SIZE)); - let enforce_fsync = runtime_options - .enforce_fsync - .unwrap_or(iggy_common::DEFAULT_ENFORCE_FSYNC); - load_persisted_segments( - config, - stream_id, - topic_id, - partition_id, - segment_size, - enforce_fsync, - stats, - ) - .await - .map_err(|source| { - error!( - stream_id, - topic_id, - partition_id, - error = %source, - "failed to load partition log during server bootstrap" - ); - source - }) -} - -#[allow(clippy::too_many_arguments)] -async fn load_partition( - config: &ServerConfig, - namespace: IggyNamespace, - stats: Arc, - partition_metadata: &Partition, - runtime_options: TopicRuntimeOptions, - cluster_id: u128, - self_replica_id: u8, - replica_count: u8, - bus: Rc, -) -> Result>, ServerError> { - let stream_id = namespace.stream_id(); - let topic_id = namespace.topic_id(); - let partition_id = namespace.partition_id(); - // (view, log_view) come from the group's durable superblock when present; - // a present but unverifiable record already refused boot inside - // `open_partition_superblock`. - let partition_dir = config - .system - .get_partition_path(stream_id, topic_id, partition_id); - let (superblock, recovered_state) = open_partition_superblock( - &partition_dir, - ReplicaIdentity { - cluster: cluster_id, - replica_id: self_replica_id, - replica_count, - }, - ) - .await?; - - // A recovered partition lost its journal state with the process: the - // partition journal is in-memory and segments carry no op numbers, so - // this replica cannot know the group's (op, commit) even when the - // superblock restored its view. In a cluster it boots as a - // quorum-invisible backup and probes for the current view - // (`RequestStartView`): the view's primary answers with a `StartView`, - // journal repair fills the rejoin window, and the commit floor settles - // at the serving peer's retention point. The probe re-broadcasts on its - // timeout, so it needs no live mesh at boot. Single-replica groups - // have no peer to ask and keep the plain init. - let join = if replica_count > 1 { - JoinMode::ProbeAsBackup { - await_state_transfer: false, - } - } else { - JoinMode::Init - }; - // Request queue holds 2x the prepare depth (buffered requests drain as - // prepares commit); depth is the per-partition `[partition]` knob. - let prepare_queue_depth = config.partition.prepare_queue_depth; - let timers = consensus_timers(config); - let consensus = VsrConsensus::restored( - cluster_id, - self_replica_id, - replica_count, - namespace.inner(), - bus, - LocalPipeline::with_capacities(prepare_queue_depth, prepare_queue_depth * 2), - VsrRestore { - timers: &timers, - durable_view: recovered_state - .as_ref() - .map(|state| (state.view, state.log_view)), - view_fallback: None, - seed_view: None, - incarnation: None, - join, - }, - ); - - // No prepare-timestamp floor is restored here: the partition consensus - // journal is non-durable today, so there is no persisted head to observe - // (unlike `restore_metadata_consensus`, which observes its restored head). - // When PartitionJournal becomes durable (the milestone named in the - // multi-shard wiring commit body), observe the restored head and the max - // recovered message timestamp here, or an NTP rewind across a restart could - // regress persisted `base_timestamp`. - - let recovered_segments = - recover_partition_segments(config, namespace, runtime_options, &stats).await?; - - let mut partition = IggyPartition::new(stats.clone(), consensus); - partition.set_runtime_options(runtime_options); - partition.set_superblock(superblock, recovered_state.as_ref()); - // Recovered partitions honor the same config-surfaced ring ceilings as the - // fresh-create path (build_partition_fresh). Retention is already off for - // single-replica groups, so this only sizes the multi-replica ring. - partition.log.journal().inner.set_ring_caps( - config.partition.evicted_ring_capacity, - config.partition.evicted_ring_bytes_max.as_bytes_u64(), - ); - partition.set_partition_dir(partition_dir.clone()); - // Before the hydrate: the durable record is keyed by incarnation, so a - // `purge.gen` left behind by a previous life of this namespace reads 0. - partition.set_created_revision(partition_metadata.created_revision); - partition.hydrate_applied_purge_generation().await?; - hydrate_partition_log( - &mut partition, - &partition_dir, - stream_id, - topic_id, - partition_id, - recovered_segments, - ) - .await?; - - let sized_end = partition - .log - .segments() - .iter() - .filter(|segment| segment.size > IggyByteSize::default()) - .map(|segment| segment.end_offset) - .max(); - // An empty chain whose segment is named for a nonzero offset is the - // shape a state-transfer install (or its converge) plants at the group - // frontier after the origin GC'd everything: the file name carries the - // frontier, and re-minting offsets from 0 here would fork this - // replica's batch stamps from the rest of the group after a restart. - let empty_frontier = partition - .log - .segments() - .iter() - .map(|segment| segment.start_offset) - .max() - .filter(|&start| sized_end.is_none() && start > 0); - let current_offset = sized_end.or_else(|| empty_frontier.map(|start| start - 1)); - partition.created_at = partition_metadata.created_at; - partition.recovered_durable_offset = sized_end; - // The OFFSET COUNTER is restored from that file name (above), but the - // `installed_frontier` CLAIM deliberately is not: the claim says "everything - // below me is represented here", and `converge_to_empty_after_failed_install` - // refuses to make it when staged segments were dropped -- yet a converge - // plants exactly the same empty `{frontier:020}.log` a legitimate empty - // install does, so boot provably cannot tell them apart. Re-deriving it here - // would hand the refused claim back: the repair floor stand-in would accept a - // commit floor over ops this replica holds zero bytes for, and the replica - // would pass the serve gate and offer that emptiness onward, making a peer - // unlink its own chain. Leaving it `None` costs one spurious full - // re-transfer on the legitimate empty-install restart; a false caught-up - // claim is not recoverable. A durable home for the frontier (the partition - // superblock already reserves a field) is what would settle it properly. - let counter = current_offset.unwrap_or(0); - partition.offset.store(counter, Ordering::Release); - partition.dirty_offset.store(counter, Ordering::Relaxed); - partition.should_increment_offset = current_offset.is_some(); - // The durable frontier is a LOWER BOUND on top of what the segments proved: - // it is the only carrier left when the segments that named the frontier are - // gone (an all-GC'd origin's install, a crash inside the swap window), and - // taking the max means real recovered data always wins. - partition.restore_offset_frontier(recovered_state.as_ref()); - let current_offset = partition.offset.load(Ordering::Acquire); - - configure_consumer_offsets(&mut partition, config, namespace, current_offset)?; - ensure_initial_segment(&mut partition, config, stream_id, topic_id, partition_id).await?; - - Ok(partition) -} - -/// Reopen writers over a recovered segment chain. -/// -/// Takes no `&ServerConfig`: every knob it needs is the partition's own -/// resolved topic option now, which is the whole point of the per-topic move. -async fn hydrate_partition_log( - partition: &mut IggyPartition>, - partition_dir: &str, - stream_id: usize, - topic_id: usize, - partition_id: usize, - recovered_segments: Vec, -) -> Result<(), ServerError> { - // The partition's own resolved knobs, not the shard-wide config: a topic - // created with `enforce_fsync` or a per-topic `segment_size` must get them - // on the writers reopened over its recovered chain too, or a restart would - // silently drop back to the node defaults. - let runtime = partition.runtime_options(); - let enforce_fsync = runtime - .enforce_fsync - .unwrap_or(iggy_common::DEFAULT_ENFORCE_FSYNC); - let segment_size = runtime - .segment_size - .unwrap_or_else(|| IggyByteSize::from(iggy_common::DEFAULT_SEGMENT_SIZE)); - let preallocate_segments = runtime - .preallocate_segments - .unwrap_or(iggy_common::DEFAULT_PREALLOCATE_SEGMENTS); - for RecoveredSegment { segment, storage } in recovered_segments { - partition - .log - .add_persisted_segment(segment, storage, None, None); - } - - if let Some(active_index) = partition.log.segments().len().checked_sub(1) { - let storage = &partition.log.storages()[active_index]; - if let ( - Some(messages_reader), - Some(index_reader), - Some(storage_messages_writer), - Some(storage_index_writer), - ) = ( - storage.messages_reader.as_ref(), - storage.index_reader.as_ref(), - storage.messages_writer.as_ref(), - storage.index_writer.as_ref(), - ) { - let index_path = index_reader.path(); - let start_offset = partition.log.segments()[active_index].start_offset; - // Share the storage's size counters: they are the write cursors. - // A private counter would let the append position diverge from the - // segment bookkeeping that index entries and poll bounds rely on. - let messages_size_counter = storage_messages_writer.size_counter(); - let index_size_counter = storage_index_writer.size_counter(); - partition.log.messages_writers_mut()[active_index] = Some(Rc::new( - MessagesWriter::new( - &messages_reader.path(), - messages_size_counter, - enforce_fsync, - true, - preallocate_segments.then_some(segment_size), - ) - .await - .map_err(|source| { - error!( - stream_id, - topic_id, - partition_id, - path = %messages_reader.path(), - error = %source, - "failed to initialize persisted messages writer" - ); - hydrate_reopen_error( - source, - partition_dir, - stream_id, - topic_id, - partition_id, - start_offset, - ) - })?, - )); - partition.log.index_writers_mut()[active_index] = Some(Rc::new( - IggyIndexWriter::new(&index_path, index_size_counter, enforce_fsync, true) - .await - .map_err(|source| { - error!( - stream_id, - topic_id, - partition_id, - path = %index_path, - error = %source, - "failed to initialize persisted sparse index writer" - ); - hydrate_reopen_error( - source, - partition_dir, - stream_id, - topic_id, - partition_id, - start_offset, - ) - })?, - )); - } - } - - Ok(()) -} - -/// Routes a hydrate-reopen writer failure. The seed-vs-stat divergence guard -/// (`SegmentSizeMismatchAtOpen`) is a post-condition assertion on recovery's -/// own truncation: pass C truncates every file to its recovered size before -/// storage and writers reopen it, so the guard can only fire if the -/// filesystem lied about a length or a change broke that truncate-then-open -/// contract. Kept as defense-in-depth and routed as a structural refusal -/// because a retried boot cannot help. Every other failure here (open, stat, -/// sync) is transient I/O and stays node-fatal: a retried boot can still -/// serve the partition, while fencing would quarantine healthy data (and at -/// `replica_count = 1` tombstone the partition outright). -fn hydrate_reopen_error( - source: IggyError, - partition_dir: &str, - stream_id: usize, - topic_id: usize, - partition_id: usize, - start_offset: u64, -) -> ServerError { - match source { - IggyError::SegmentSizeMismatchAtOpen(on_disk_bytes, expected_bytes) => { - ServerError::PartitionRecoveryRefused { - dir: PathBuf::from(partition_dir), - stream_id, - topic_id, - partition_id, - reason: PartitionRecoveryRefusal::StorageSizeMismatch { - start_offset, - on_disk_bytes, - expected_bytes, - }, - } - } - transient => transient.into(), - } -} - #[cfg(test)] mod tests { use super::*; diff --git a/core/server/src/boot/threads.rs b/core/server/src/boot/threads.rs index 95de1d5b20..2dc1efd729 100644 --- a/core/server/src/boot/threads.rs +++ b/core/server/src/boot/threads.rs @@ -32,23 +32,26 @@ use partitions::FatalCommit; use server_common::executor::create_shard_executor; use shard::metrics::ShardMetrics; use shard::{Receiver as ShardReceiver, Sender, ShardFrame, TaggedSender}; +use std::backtrace::Backtrace; use std::rc::Rc; -use std::sync::Arc; use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::{Arc, OnceLock}; use std::time::{Duration, Instant}; use std::{panic, thread}; use tracing::{error, info, warn}; /// Result of a multi-shard bootstrap. /// -/// Carries the cross-thread shutdown flag and one OS-thread `JoinHandle` -/// per shard. The caller flips the flag via [`Self::install_ctrlc_handler`] -/// and then drains every shard via [`Self::join_all`], bounded by -/// `join_timeout` (`system.sharding.shutdown_join_timeout`). +/// Carries the cross-thread shutdown flag, one OS-thread `JoinHandle` +/// per shard, and the first panic `install_panic_hook` recorded. The +/// caller flips the flag via [`Self::install_ctrlc_handler`] and then +/// drains every shard via [`Self::join_all`], bounded by `join_timeout` +/// (`system.sharding.shutdown_join_timeout`). pub struct ShardHandles { pub(in crate::boot) shutdown_flag: Arc, pub(in crate::boot) shard_threads: Vec<(u16, thread::JoinHandle>)>, pub(in crate::boot) join_timeout: Duration, + pub(in crate::boot) first_panic: Arc>, } impl ShardHandles { @@ -98,7 +101,9 @@ impl ShardHandles { /// returned a `Result::Err`, panicked, or wedged past the deadline. /// The variant carries every per-shard failure in shard-id order so /// the caller does not need to read the trace log to discover - /// late-failing shards. + /// late-failing shards. Returns [`ServerError::Panicked`] when every + /// thread exited `Ok` but the panic hook recorded a panic: a task + /// compio's `spawn` caught, which no thread result can carry. pub fn join_all(self) -> Result<(), ServerError> { let mut failures: Vec = Vec::new(); // Armed on the first poll that observes the shutdown flag, shared @@ -157,14 +162,55 @@ impl ShardHandles { } } } - if failures.is_empty() { - Ok(()) - } else { - Err(ServerError::ShardJoinFailures { failures }) + if !failures.is_empty() { + return Err(ServerError::ShardJoinFailures { failures }); } + self.first_panic.get().map_or_else( + || Ok(()), + |description| { + Err(ServerError::Panicked { + description: description.clone(), + }) + }, + ) } } +/// Install the process-wide panic hook and return the slot it records +/// the first panic into. +/// +/// The hook runs on the panicking thread before the unwind, so it sees +/// every panic on a shard: the thread body, whose unwind +/// [`run_shard_thread`] already turns into a join failure, and the tasks +/// compio's `spawn` catches, which nothing observes while the server +/// runs (a dead listener or connection task leaves every thread exiting +/// `Ok`). It logs the panic with its backtrace, records the first one for +/// [`ShardHandles::join_all`] to fail the exit on, and flips the shutdown +/// flag so every shard drains instead of the half-alive state one dead +/// task leaves behind. The previous hook still runs after it, so stderr +/// keeps the standard panic line. +pub(in crate::boot) fn install_panic_hook(shutdown_flag: Arc) -> Arc> { + let first_panic = Arc::new(OnceLock::new()); + let recorded = Arc::clone(&first_panic); + let previous_hook = panic::take_hook(); + panic::set_hook(Box::new(move |info| { + let current = thread::current(); + let thread_name = current.name().unwrap_or(""); + let location = info + .location() + .map_or_else(|| "".to_string(), ToString::to_string); + let message = panic_payload_to_string(info.payload()); + let backtrace = Backtrace::force_capture(); + error!(thread = thread_name, location = %location, backtrace = %backtrace, "{message}"); + let _ = recorded.set(format!( + "thread '{thread_name}' panicked at {location}: {message}" + )); + shutdown_flag.store(true, Ordering::Relaxed); + previous_hook(info); + })); + first_panic +} + /// Poll cadence for the bounded shard joins. Coarse enough to cost /// nothing during a normal drain, fine enough that exit latency past /// the last shard's return stays imperceptible. @@ -682,6 +728,53 @@ mod tests { ); } + #[test] + fn join_all_fails_the_exit_on_a_panic_no_thread_surfaced() { + // A listener or connection task panic leaves every shard thread + // exiting Ok; only the hook's record can keep that from reading + // as a clean shutdown to an orchestrator. + let first_panic = Arc::new(OnceLock::new()); + first_panic + .set("thread 'shard-0' panicked at listener.rs:1:1: boom".to_string()) + .expect("a fresh slot accepts the first record"); + let handle = thread::spawn(|| -> Result<(), ServerError> { Ok(()) }); + let handles = ShardHandles { + shutdown_flag: Arc::new(AtomicBool::new(true)), + shard_threads: vec![(0, handle)], + join_timeout: Duration::from_secs(1), + first_panic, + }; + let error = handles + .join_all() + .expect_err("a recorded panic must fail the exit even when every thread exited Ok"); + assert!( + matches!(&error, ServerError::Panicked { description } if description.contains("boom")), + "unexpected error: {error}" + ); + } + + #[test] + fn panic_hook_records_the_panic_and_flips_the_shutdown_flag() { + let shutdown_flag = Arc::new(AtomicBool::new(false)); + let first_panic = install_panic_hook(Arc::clone(&shutdown_flag)); + let joined = thread::Builder::new() + .name("shard-7".to_string()) + .spawn(|| panic!("injected task panic")) + .expect("spawn") + .join(); + assert!(joined.is_err(), "the thread must have panicked"); + assert!( + shutdown_flag.load(Ordering::Relaxed), + "a panic anywhere must drive the whole server down" + ); + let description = first_panic.get().expect("the hook records the first panic"); + assert!( + description.starts_with("thread 'shard-7' panicked at ") + && description.ends_with(": injected task panic"), + "unexpected record: {description}" + ); + } + #[test] fn join_abandons_a_wedged_shard_after_the_shutdown_deadline() { let shutdown_flag = AtomicBool::new(true); diff --git a/core/server/src/partition_helpers.rs b/core/server/src/partition_helpers.rs index f6ab9d16a2..ad4471da98 100644 --- a/core/server/src/partition_helpers.rs +++ b/core/server/src/partition_helpers.rs @@ -15,21 +15,28 @@ // specific language governing permissions and limitations // under the License. -//! Helpers shared between the recovery path in [`crate::boot`] and -//! the runtime partition reconciliation loop. +//! Partition materialisation shared by the boot path and the runtime +//! reconciliation loop. //! -//! Recovery hydrates an [`IggyPartition`] from on-disk state; the -//! reconciler builds one from scratch when a committed -//! `CreateTopic` / `CreatePartitions` metadata event has no matching -//! local partition yet. The two paths share namespace-bounds validation, +//! [`load_partition_or_fence`] hydrates an [`IggyPartition`] from its +//! on-disk state and rules on a segment chain recovery refused; +//! [`build_partition_fresh`] materialises one for a namespace that has no +//! directory yet. Both sit on the same namespace-bounds validation, //! consumer-offset configuration, and initial-segment provisioning. +//! +//! Boot runs the loader over every owned namespace. The reconciler picks +//! whichever builder the partition directory calls for when a committed +//! `CreateTopic` / `CreatePartitions` event has no matching local +//! partition yet. use crate::offset_recovery::{load_consumer_group_offsets, load_consumer_offsets}; -use crate::server_error::ServerError; +use crate::segment_recovery::{RecoveredSegment, load_persisted_segments}; +use crate::server_error::{PartitionRecoveryRefusal, ServerError}; +use crate::shell::consensus_timers; use compio::fs::create_dir_all; use configs::server::ServerConfig; use consensus::{ - FreshGroupStart, LocalPipeline, VsrConsensus, VsrRestore, VsrState, fresh_group_start, + FreshGroupStart, JoinMode, LocalPipeline, VsrConsensus, VsrRestore, VsrState, fresh_group_start, }; use iggy_common::{ ConsumerGroupOffsets, ConsumerOffsets, IggyByteSize, IggyError, IggyTimestamp, PartitionStats, @@ -37,8 +44,9 @@ use iggy_common::{ }; use journal::superblock::{PingPongSuperblock, SuperblockContents}; use message_bus::IggyMessageBus; +use metadata::stm::stream::Partition; use metadata::{IdentityField, ReplicaIdentity}; -use partitions::{IggyIndexWriter, IggyPartition, MessagesWriter, Segment}; +use partitions::{IggyIndexWriter, IggyPartition, IggyPartitions, MessagesWriter, Segment}; use server_common::SegmentStorage; use server_common::fs_utils::remove_dir_all; use server_common::sharding::IggyNamespace; @@ -501,13 +509,558 @@ pub async fn open_partition_superblock( Ok((Rc::new(superblock), recovered_state)) } +/// Recover an owned partition from its on-disk state. +/// +/// Shared by boot and the reconciler so a partition this replica committed +/// before a crash but re-learns only after restart (its WAL watermark trails +/// the commit by one op) is hydrated from its segments like any other, not +/// rebuilt over them. `Ok(None)` means the namespace was tombstoned here; the +/// arms below say when. Errors are transient I/O, left to the caller. +#[allow(clippy::too_many_arguments, clippy::too_many_lines)] +pub async fn load_partition_or_fence( + config: &ServerConfig, + namespace: IggyNamespace, + partition_stats: Arc, + partition_metadata: &Partition, + topic_runtime: TopicRuntimeOptions, + cluster_id: u128, + self_replica_id: u8, + replica_count: u8, + bus: Rc, + partitions: &IggyPartitions, PingPongSuperblock>, +) -> Result>>, ServerError> { + let stream_id = namespace.stream_id(); + let topic_id = namespace.topic_id(); + // Heap-pinned: the loader's and the rebuilder's futures side by side + // outgrow clippy's `large_futures` cap, and this runs once per partition. + match Box::pin(load_partition( + config, + namespace, + Arc::clone(&partition_stats), + partition_metadata, + topic_runtime, + cluster_id, + self_replica_id, + replica_count, + Rc::clone(&bus), + )) + .await + { + Ok(partition) => Ok(Some(partition)), + // ONE damaged local chain must not take the node down. The shapes + // this refuses are structural -- what a failed state-transfer + // quarantine leaves behind, or damage the recovery walk proved + // inside a segment. What follows depends on whether a peer can + // restore the data. With peers, the segment files are fenced + // aside (keeping the superblock so the group cannot re-enter + // view 0), the group is materialised fresh, and the ordinary + // rejoin path (repair, then state transfer on a refused floor) + // refills it. Single-replica, only a chain-shape refusal whose + // planned chain provably holds ZERO recoverable bytes still + // fences and rebuilds: nothing servable is at stake, so an empty + // rebuild hides no loss. The verdict variant alone is not that + // evidence -- a hole and an orphan empty segment both fire over + // fully populated chains -- which is why the gate reads the byte + // total the refusal carries. Every other refusal tombstones, + // leaving its files exactly where they are: a rebuilt empty + // partition answers polls exactly like a healthy empty one and + // hides the loss, while an unrouted namespace is a failure an + // operator can see. + Err(ServerError::PartitionRecoveryRefused { dir, reason, .. }) => { + let partition_dir = dir.to_string_lossy().into_owned(); + let rebuild_for_rejoin = replica_count > 1 + || matches!( + reason, + PartitionRecoveryRefusal::Hole { + recoverable_bytes: 0, + .. + } | PartitionRecoveryRefusal::EmptyNonTailSegment { + recoverable_bytes: 0, + .. + } + ); + error!( + stream_id, + topic_id, + partition_id = partition_metadata.id, + partition_dir, + %reason, + "refusing the recovered segment chain" + ); + // A pass-A refusal folded nothing into the stats (recovery + // counts only accepted chains), but the hydrate-reopen refusal + // arrives after a fully counted load, so clear them either way. + partition_stats.zero_out_all(); + if !rebuild_for_rejoin { + // No quarantine here, mirroring the superblock arm below: + // a tombstone is only durable if its cause is. Fencing the + // chain aside would leave the next boot zero segments to + // walk, so it would re-seed from the surviving superblock, + // plant a fresh segment, and serve the partition empty + // with no refusal logged. Left at their real paths, the + // same files re-derive this verdict (and this log line) + // every boot, and the reconciler's tombstone gate keeps + // the namespace away from a fresh build, whose + // initial-segment open would truncate the oldest refused + // segment in place. The one refusal whose cause is NOT + // durable is `StorageSizeMismatch`: it fires from the + // reopen right after recovery truncated the same file, so + // the next boot re-walks the already-truncated bytes and, + // unless the length diverges again, accepts the chain + // instead of re-tombstoning -- acceptable for an + // assertion that the filesystem lied about a length. + // `%reason` repeated on purpose: this is the line an + // operator greps to enumerate dark partitions, so it has + // to carry the verdict on its own. + error!( + stream_id, + topic_id, + partition_id = partition_metadata.id, + partition_dir, + %reason, + "no peer replica holds this partition's data; leaving the refused \ + segment files in place and tombstoning it instead of serving it \ + empty" + ); + partitions.tombstone(namespace); + return Ok(None); + } + match partitions::state_transfer::quarantine_segment_files(&partition_dir).await { + Ok(fenced_dir) => error!( + stream_id, + topic_id, + partition_id = partition_metadata.id, + fenced_dir, + "quarantined the refused segment files; they are kept for inspection" + ), + Err(error) => { + // NOT rebuilt: `build_partition_fresh` reaches + // `ensure_initial_segment`, which opens segment 0 with + // `file_exists = false` and TRUNCATES whatever the + // failed quarantine left behind. The likeliest failures + // (suffix cap exhausted, `create_dir_all`) move zero + // files, so rebuilding would destroy the oldest segment + // on the first attempt while the higher-offset survivors + // keep refusing every boot -- a loop that never + // terminates and eats the chain one segment at a time. + // Tombstone instead: the namespace stays unmaterialised + // and unrouted, the reconciler backs off, and an + // operator still has every byte. + error!( + stream_id, + topic_id, + partition_id = partition_metadata.id, + partition_dir, + %error, + "failed to quarantine the refused segment files; leaving this \ + partition tombstoned rather than rebuilding over them" + ); + partitions.tombstone(namespace); + return Ok(None); + } + } + Box::pin(build_partition_fresh( + config, + namespace, + partition_stats, + partition_metadata.created_revision, + topic_runtime, + cluster_id, + self_replica_id, + replica_count, + partition_metadata.created_view, + Rc::clone(&bus), + )) + .await + .map(Some) + } + // An untrustworthy superblock fences ONE group, not the node. The + // segment files stay exactly where they are -- unlike a refused + // chain, the data on disk is not the thing in doubt -- so there is + // nothing to quarantine and nothing to rebuild: rebuilding fresh + // would hand this replica a view-0 identity while a record it + // cannot read says otherwise. Tombstoned, the namespace stays + // unmaterialised and unrouted, the reconciler backs off, and an + // operator has every byte plus a message naming the directory. + Err( + error @ (ServerError::PartitionSuperblockIo { .. } + | ServerError::PartitionSuperblockVersionUnknown { .. } + | ServerError::PartitionSuperblockUnverifiable { .. } + | ServerError::PartitionSuperblockUndecodable { .. } + | ServerError::PartitionSuperblockIdentityMismatch { .. }), + ) => { + error!( + stream_id, + topic_id, + partition_id = partition_metadata.id, + %error, + "cannot trust this partition's durable consensus state; tombstoning the \ + partition instead of serving it" + ); + partition_stats.zero_out_all(); + partitions.tombstone(namespace); + Ok(None) + } + Err(error) => Err(error), + } +} + +#[allow(clippy::too_many_arguments)] +async fn load_partition( + config: &ServerConfig, + namespace: IggyNamespace, + stats: Arc, + partition_metadata: &Partition, + runtime_options: TopicRuntimeOptions, + cluster_id: u128, + self_replica_id: u8, + replica_count: u8, + bus: Rc, +) -> Result>, ServerError> { + let stream_id = namespace.stream_id(); + let topic_id = namespace.topic_id(); + let partition_id = namespace.partition_id(); + // (view, log_view) come from the group's durable superblock when present; + // a present but unverifiable record already refused boot inside + // `open_partition_superblock`. + let partition_dir = config + .system + .get_partition_path(stream_id, topic_id, partition_id); + let (superblock, recovered_state) = open_partition_superblock( + &partition_dir, + ReplicaIdentity { + cluster: cluster_id, + replica_id: self_replica_id, + replica_count, + }, + ) + .await?; + + // A recovered partition lost its journal state with the process: the + // partition journal is in-memory and segments carry no op numbers, so + // this replica cannot know the group's (op, commit) even when the + // superblock restored its view. In a cluster it boots as a + // quorum-invisible backup and probes for the current view + // (`RequestStartView`): the view's primary answers with a `StartView`, + // journal repair fills the rejoin window, and the commit floor settles + // at the serving peer's retention point. The probe re-broadcasts on its + // timeout, so it needs no live mesh at boot. Single-replica groups + // have no peer to ask and keep the plain init. + let join = if replica_count > 1 { + JoinMode::ProbeAsBackup { + await_state_transfer: false, + } + } else { + JoinMode::Init + }; + // Request queue holds 2x the prepare depth (buffered requests drain as + // prepares commit); depth is the per-partition `[partition]` knob. + let prepare_queue_depth = config.partition.prepare_queue_depth; + let timers = consensus_timers(config); + let consensus = VsrConsensus::restored( + cluster_id, + self_replica_id, + replica_count, + namespace.inner(), + bus, + LocalPipeline::with_capacities(prepare_queue_depth, prepare_queue_depth * 2), + VsrRestore { + timers: &timers, + durable_view: recovered_state + .as_ref() + .map(|state| (state.view, state.log_view)), + view_fallback: None, + seed_view: None, + incarnation: None, + join, + }, + ); + + // No prepare-timestamp floor is restored here: the partition consensus + // journal is non-durable today, so there is no persisted head to observe + // (unlike `restore_metadata_consensus`, which observes its restored head). + // When PartitionJournal becomes durable (the milestone named in the + // multi-shard wiring commit body), observe the restored head and the max + // recovered message timestamp here, or an NTP rewind across a restart could + // regress persisted `base_timestamp`. + + let recovered_segments = + recover_partition_segments(config, namespace, runtime_options, &stats).await?; + + let mut partition = IggyPartition::new(stats.clone(), consensus); + partition.set_runtime_options(runtime_options); + partition.set_superblock(superblock, recovered_state.as_ref()); + // Recovered partitions honor the same config-surfaced ring ceilings as the + // fresh-create path (build_partition_fresh). Retention is already off for + // single-replica groups, so this only sizes the multi-replica ring. + partition.log.journal().inner.set_ring_caps( + config.partition.evicted_ring_capacity, + config.partition.evicted_ring_bytes_max.as_bytes_u64(), + ); + partition.set_partition_dir(partition_dir.clone()); + // Before the hydrate: the durable record is keyed by incarnation, so a + // `purge.gen` left behind by a previous life of this namespace reads 0. + partition.set_created_revision(partition_metadata.created_revision); + partition.hydrate_applied_purge_generation().await?; + hydrate_partition_log( + &mut partition, + &partition_dir, + stream_id, + topic_id, + partition_id, + recovered_segments, + ) + .await?; + + let sized_end = partition + .log + .segments() + .iter() + .filter(|segment| segment.size > IggyByteSize::default()) + .map(|segment| segment.end_offset) + .max(); + // An empty chain whose segment is named for a nonzero offset is the + // shape a state-transfer install (or its converge) plants at the group + // frontier after the origin GC'd everything: the file name carries the + // frontier, and re-minting offsets from 0 here would fork this + // replica's batch stamps from the rest of the group after a restart. + let empty_frontier = partition + .log + .segments() + .iter() + .map(|segment| segment.start_offset) + .max() + .filter(|&start| sized_end.is_none() && start > 0); + let current_offset = sized_end.or_else(|| empty_frontier.map(|start| start - 1)); + partition.created_at = partition_metadata.created_at; + partition.recovered_durable_offset = sized_end; + // The OFFSET COUNTER is restored from that file name (above), but the + // `installed_frontier` CLAIM deliberately is not: the claim says "everything + // below me is represented here", and `converge_to_empty_after_failed_install` + // refuses to make it when staged segments were dropped -- yet a converge + // plants exactly the same empty `{frontier:020}.log` a legitimate empty + // install does, so boot provably cannot tell them apart. Re-deriving it here + // would hand the refused claim back: the repair floor stand-in would accept a + // commit floor over ops this replica holds zero bytes for, and the replica + // would pass the serve gate and offer that emptiness onward, making a peer + // unlink its own chain. Leaving it `None` costs one spurious full + // re-transfer on the legitimate empty-install restart; a false caught-up + // claim is not recoverable. A durable home for the frontier (the partition + // superblock already reserves a field) is what would settle it properly. + let counter = current_offset.unwrap_or(0); + partition.offset.store(counter, Ordering::Release); + partition.dirty_offset.store(counter, Ordering::Relaxed); + partition.should_increment_offset = current_offset.is_some(); + // The durable frontier is a LOWER BOUND on top of what the segments proved: + // it is the only carrier left when the segments that named the frontier are + // gone (an all-GC'd origin's install, a crash inside the swap window), and + // taking the max means real recovered data always wins. + partition.restore_offset_frontier(recovered_state.as_ref()); + let current_offset = partition.offset.load(Ordering::Acquire); + + configure_consumer_offsets(&mut partition, config, namespace, current_offset)?; + ensure_initial_segment(&mut partition, config, stream_id, topic_id, partition_id).await?; + + Ok(partition) +} + +/// Recover this partition's persisted segment chain, stamping each segment +/// with the topic's effective segment size (the per-topic value when the +/// topic was created with one, else the shard-wide configured size). +/// +/// The topic's effective `enforce_fsync` goes in for the same reason: it is +/// what tells recovery whether a durable index entry the log cannot back is a +/// benign torn index or previously durable data the log lost. +async fn recover_partition_segments( + config: &ServerConfig, + namespace: IggyNamespace, + runtime_options: TopicRuntimeOptions, + stats: &PartitionStats, +) -> Result, ServerError> { + let stream_id = namespace.stream_id(); + let topic_id = namespace.topic_id(); + let partition_id = namespace.partition_id(); + let segment_size = runtime_options + .segment_size + .unwrap_or_else(|| IggyByteSize::from(iggy_common::DEFAULT_SEGMENT_SIZE)); + let enforce_fsync = runtime_options + .enforce_fsync + .unwrap_or(iggy_common::DEFAULT_ENFORCE_FSYNC); + load_persisted_segments( + config, + stream_id, + topic_id, + partition_id, + segment_size, + enforce_fsync, + stats, + ) + .await + .map_err(|source| { + error!( + stream_id, + topic_id, + partition_id, + error = %source, + "failed to load partition log during server bootstrap" + ); + source + }) +} + +/// Reopen writers over a recovered segment chain. +/// +/// Takes no `&ServerConfig`: every knob it needs is the partition's own +/// resolved topic option now, which is the whole point of the per-topic move. +async fn hydrate_partition_log( + partition: &mut IggyPartition>, + partition_dir: &str, + stream_id: usize, + topic_id: usize, + partition_id: usize, + recovered_segments: Vec, +) -> Result<(), ServerError> { + // The partition's own resolved knobs, not the shard-wide config: a topic + // created with `enforce_fsync` or a per-topic `segment_size` must get them + // on the writers reopened over its recovered chain too, or a restart would + // silently drop back to the node defaults. + let runtime = partition.runtime_options(); + let enforce_fsync = runtime + .enforce_fsync + .unwrap_or(iggy_common::DEFAULT_ENFORCE_FSYNC); + let segment_size = runtime + .segment_size + .unwrap_or_else(|| IggyByteSize::from(iggy_common::DEFAULT_SEGMENT_SIZE)); + let preallocate_segments = runtime + .preallocate_segments + .unwrap_or(iggy_common::DEFAULT_PREALLOCATE_SEGMENTS); + for RecoveredSegment { segment, storage } in recovered_segments { + partition + .log + .add_persisted_segment(segment, storage, None, None); + } + + if let Some(active_index) = partition.log.segments().len().checked_sub(1) { + let storage = &partition.log.storages()[active_index]; + if let ( + Some(messages_reader), + Some(index_reader), + Some(storage_messages_writer), + Some(storage_index_writer), + ) = ( + storage.messages_reader.as_ref(), + storage.index_reader.as_ref(), + storage.messages_writer.as_ref(), + storage.index_writer.as_ref(), + ) { + let index_path = index_reader.path(); + let start_offset = partition.log.segments()[active_index].start_offset; + // Share the storage's size counters: they are the write cursors. + // A private counter would let the append position diverge from the + // segment bookkeeping that index entries and poll bounds rely on. + let messages_size_counter = storage_messages_writer.size_counter(); + let index_size_counter = storage_index_writer.size_counter(); + partition.log.messages_writers_mut()[active_index] = Some(Rc::new( + MessagesWriter::new( + &messages_reader.path(), + messages_size_counter, + enforce_fsync, + true, + preallocate_segments.then_some(segment_size), + ) + .await + .map_err(|source| { + error!( + stream_id, + topic_id, + partition_id, + path = %messages_reader.path(), + error = %source, + "failed to initialize persisted messages writer" + ); + hydrate_reopen_error( + source, + partition_dir, + stream_id, + topic_id, + partition_id, + start_offset, + ) + })?, + )); + partition.log.index_writers_mut()[active_index] = Some(Rc::new( + IggyIndexWriter::new(&index_path, index_size_counter, enforce_fsync, true) + .await + .map_err(|source| { + error!( + stream_id, + topic_id, + partition_id, + path = %index_path, + error = %source, + "failed to initialize persisted sparse index writer" + ); + hydrate_reopen_error( + source, + partition_dir, + stream_id, + topic_id, + partition_id, + start_offset, + ) + })?, + )); + } + } + + Ok(()) +} + +/// Routes a hydrate-reopen writer failure. The seed-vs-stat divergence guard +/// (`SegmentSizeMismatchAtOpen`) is a post-condition assertion on recovery's +/// own truncation: pass C truncates every file to its recovered size before +/// storage and writers reopen it, so the guard can only fire if the +/// filesystem lied about a length or a change broke that truncate-then-open +/// contract. Kept as defense-in-depth and routed as a structural refusal +/// because a retried boot cannot help. Every other failure here (open, stat, +/// sync) is transient I/O and stays node-fatal: a retried boot can still +/// serve the partition, while fencing would quarantine healthy data (and at +/// `replica_count = 1` tombstone the partition outright). +fn hydrate_reopen_error( + source: IggyError, + partition_dir: &str, + stream_id: usize, + topic_id: usize, + partition_id: usize, + start_offset: u64, +) -> ServerError { + match source { + IggyError::SegmentSizeMismatchAtOpen(on_disk_bytes, expected_bytes) => { + ServerError::PartitionRecoveryRefused { + dir: PathBuf::from(partition_dir), + stream_id, + topic_id, + partition_id, + reason: PartitionRecoveryRefusal::StorageSizeMismatch { + start_offset, + on_disk_bytes, + expected_bytes, + }, + } + } + transient => transient.into(), + } +} + /// Materialise a brand-new [`IggyPartition`] for a namespace that has no on-disk state yet. /// -/// Counterpart to bootstrap's `load_partition`, which hydrates from -/// on-disk state during recovery; this builder is the runtime path -/// invoked by the reconciliation loop when a committed -/// `CreateTopic` / `CreatePartitions` metadata event names a partition -/// the local shard has not yet materialised. +/// Counterpart to [`load_partition_or_fence`], which hydrates from +/// on-disk state; this builder is the runtime path invoked by the +/// reconciliation loop when a committed `CreateTopic` / +/// `CreatePartitions` metadata event names a partition the local shard +/// has not yet materialised and has no directory for. A directory +/// already on disk is routed through the loader instead, so a prior +/// life's segments are hydrated rather than built over. /// /// Steps performed (all idempotent on retry after a partial failure): /// 1. Create directory hierarchy on disk. @@ -590,9 +1143,12 @@ pub async fn build_partition_fresh( ) .await?; - // A partition directory that already holds segment bytes is a RESTART - // materialization, not a fresh create: this replica's group state died - // with the process, so claiming view-0 primaryship would heartbeat + // A partition directory that already exists here is a rebuild over a + // fenced chain (`load_partition_or_fence` quarantined the refused segment + // files and kept the superblock; every other prior life is hydrated by + // that loader before reaching this builder), not a fresh create: this + // replica's group state died with the process, so claiming view-0 + // primaryship would heartbeat // commit_min=0 at peers that hold the committed log (racing their // election). Join as a quorum-invisible backup and probe for the // current view instead; journal repair re-materializes the data from a diff --git a/core/server/src/partition_reconciler.rs b/core/server/src/partition_reconciler.rs index 0a7fd5e625..4db62a2b7a 100644 --- a/core/server/src/partition_reconciler.rs +++ b/core/server/src/partition_reconciler.rs @@ -146,7 +146,9 @@ //! discriminator, like `checkpoint_id` on every prepare //! -- `PrepareHeader.reserved` has room, but it is a `#[repr(C)]` wire change. -use crate::partition_helpers::{build_partition_fresh, delete_partitions_from_disk}; +use crate::partition_helpers::{ + build_partition_fresh, delete_partitions_from_disk, load_partition_or_fence, +}; use crate::shell::ServerShard; use ahash::{AHashMap, AHashSet}; use configs::server::ServerConfig; @@ -155,6 +157,7 @@ use futures::FutureExt; use iggy_common::{ConsumerGroupId, IggyTimestamp}; use message_bus::MessageBus; use metadata::impls::metadata::StreamsFrontend; +use metadata::stm::stream::Partition; use partitions::delete_persisted_offset; use server_common::sharding::{IggyNamespace, ShardId}; use shard::MetadataSubmit; @@ -649,21 +652,52 @@ async fn reconcile_additions( continue; }; - match build_partition_fresh( - ctx.config.as_ref(), - ns, - partition_stats, - epoch, - topic_runtime, - ctx.cluster_id, - ctx.self_replica_id, - ctx.replica_count, - created_view, - Rc::clone(&ctx.shard.bus), - ) - .await - { - Ok(partition) => { + // A directory already on disk is a prior life of this namespace. The + // usual shape: this replica applied the create before a crash, but its + // WAL watermark trails the commit by one op, so boot saw no such + // partition and the post-election re-commit lands here. Recover it the + // way boot does, segments included. Building fresh opens segment 0 + // with truncation and throws away flushed data that no peer may still + // hold once the whole cluster restarted. + let partition_dir = + ctx.config + .system + .get_partition_path(ns.stream_id(), ns.topic_id(), ns.partition_id()); + let built = if std::fs::metadata(&partition_dir).is_ok() { + let Some(partition_metadata) = fetch_partition_metadata(ctx, ns) else { + continue; + }; + load_partition_or_fence( + ctx.config.as_ref(), + ns, + partition_stats, + &partition_metadata, + topic_runtime, + ctx.cluster_id, + ctx.self_replica_id, + ctx.replica_count, + Rc::clone(&ctx.shard.bus), + partitions, + ) + .await + } else { + build_partition_fresh( + ctx.config.as_ref(), + ns, + partition_stats, + epoch, + topic_runtime, + ctx.cluster_id, + ctx.self_replica_id, + ctx.replica_count, + created_view, + Rc::clone(&ctx.shard.bus), + ) + .await + .map(Some) + }; + match built { + Ok(Some(partition)) => { ctx.shard.enqueue_reconcile_op(ReconcileOp::InsertOwned { namespace: ns, partition: Box::new(partition), @@ -672,6 +706,9 @@ async fn reconcile_additions( ctx.record_success(ns, FailureCause::Add); counters.materialised += 1; } + // Tombstoned by the loader over damaged files; the gate above keeps + // every later pass from rebuilding over them. + Ok(None) => {} Err(err) => { ctx.record_failure(ns, FailureCause::Add, now); ctx.shard.metrics().record_partition_reconcile_failure(); @@ -1110,6 +1147,21 @@ fn fetch_partition_stats( }) } +/// The committed [`Partition`] record for `ns`, which the disk loader needs +/// (`created_at`, `created_revision`, `created_view`). `None` if the topic or +/// partition vanished between the target snapshot and this read. +fn fetch_partition_metadata(ctx: &ReconcilerCtx, ns: IggyNamespace) -> Option { + ctx.shard.plane.metadata().mux_stm.streams().read(|inner| { + let stream = inner.items.get(ns.stream_id())?; + let topic = stream.topics.get(ns.topic_id())?; + topic + .partitions + .iter() + .find(|partition| partition.id == ns.partition_id()) + .cloned() + }) +} + /// `true` when this shard's routing row for `ns` already records `epoch`. A row /// carrying any other epoch (or none) is stale and must be rewritten, since the /// namespace is byte-identical across incarnations. diff --git a/core/server/src/server_error.rs b/core/server/src/server_error.rs index 12438e7100..7f0f86c5e3 100644 --- a/core/server/src/server_error.rs +++ b/core/server/src/server_error.rs @@ -292,6 +292,13 @@ pub enum ServerError { ShardConstruction(#[source] ShardCtorError), #[error("{} shard thread(s) failed: {}", failures.len(), format_shard_failures(failures))] ShardJoinFailures { failures: Vec }, + /// A panic no shard thread could surface: compio's `spawn` catches task + /// panics, so a dead listener or connection task leaves every thread + /// exiting `Ok`. The panic hook records the first one and the join path + /// fails the exit on it, so an orchestrator does not read the shutdown + /// as clean. + #[error("server shut down after a panic: {description}")] + Panicked { description: String }, } /// Why a partition's recovered segments cannot be served. diff --git a/core/server_common/src/crypto.rs b/core/server_common/src/crypto.rs index 773e9b1302..66dfdb5278 100644 --- a/core/server_common/src/crypto.rs +++ b/core/server_common/src/crypto.rs @@ -15,10 +15,7 @@ // specific language governing permissions and limitations // under the License. -use argon2::{ - Argon2, - password_hash::{PasswordHash, PasswordHasher, PasswordVerifier, SaltString, rand_core::OsRng}, -}; +use argon2::{Argon2, PasswordHasher, PasswordVerifier, password_hash}; use rand::{RngExt, distr::Alphanumeric}; use std::ops::Range; @@ -31,7 +28,7 @@ use std::ops::Range; /// simulator's deterministic hashing is the additive [`hash_with_fixed_salt`]. #[must_use] pub fn hash_password(password: &str) -> String { - hash_with_salt(password, &SaltString::generate(&mut OsRng)) + hash_with_salt(password, &password_hash::generate_salt()) } /// SIMULATOR / TEST ONLY. Argon2-hash with a fixed salt so the hash is a pure @@ -45,13 +42,12 @@ pub fn hash_password(password: &str) -> String { pub fn hash_with_fixed_salt(password: &str) -> String { // Fixed 16-byte salt; the value is irrelevant beyond being constant. const SIMULATOR_SALT: [u8; 16] = *b"iggy-sim-salt-16"; - let salt = SaltString::encode_b64(&SIMULATOR_SALT).expect("fixed sim salt encodes"); - hash_with_salt(password, &salt) + hash_with_salt(password, &SIMULATOR_SALT) } -fn hash_with_salt(password: &str, salt: &SaltString) -> String { +fn hash_with_salt(password: &str, salt: &[u8]) -> String { Argon2::default() - .hash_password(password.as_bytes(), salt) + .hash_password_with_salt(password.as_bytes(), salt) .expect("Password hashing failed") .to_string() } @@ -62,11 +58,8 @@ fn hash_with_salt(password: &str, salt: &SaltString) -> String { /// not take down the pump on every login. Not a timing oracle: the password /// input cannot influence whether the stored hash parses. pub fn verify_password(password: &str, hash: &str) -> bool { - let Ok(hash) = PasswordHash::new(hash) else { - return false; - }; Argon2::default() - .verify_password(password.as_bytes(), &hash) + .verify_password(password.as_bytes(), hash) .is_ok() } diff --git a/core/shard/Cargo.toml b/core/shard/Cargo.toml index 96a72c9117..1ee8ae7135 100644 --- a/core/shard/Cargo.toml +++ b/core/shard/Cargo.toml @@ -34,19 +34,19 @@ simulator = ["partitions/simulator"] [dependencies] compio = { workspace = true } -consensus = { path = "../consensus" } +consensus = { workspace = true } crossfire = { workspace = true } futures = { workspace = true } hash32 = { workspace = true } -iggy_binary_protocol = { path = "../binary_protocol" } -iggy_common = { path = "../common" } -journal = { path = "../journal" } -message_bus = { path = "../message_bus" } -metadata = { path = "../metadata" } +iggy_binary_protocol = { workspace = true } +iggy_common = { workspace = true } +journal = { workspace = true } +message_bus = { workspace = true } +metadata = { workspace = true } papaya = { workspace = true } -partitions = { path = "../partitions" } +partitions = { workspace = true } prometheus-client = { workspace = true } -server_common = { path = "../server_common" } +server_common = { workspace = true } thiserror = { workspace = true } tracing = { workspace = true } diff --git a/core/simulator/Cargo.toml b/core/simulator/Cargo.toml index d24918b527..0e46f73fc6 100644 --- a/core/simulator/Cargo.toml +++ b/core/simulator/Cargo.toml @@ -28,24 +28,24 @@ bytes = { workspace = true } clap = { workspace = true } clock = { workspace = true } configs = { workspace = true } -consensus = { path = "../consensus" } +consensus = { workspace = true } enumset = { workspace = true } futures = { workspace = true } -iggy_binary_protocol = { path = "../binary_protocol" } -iggy_common = { path = "../common" } +iggy_binary_protocol = { workspace = true } +iggy_common = { workspace = true } indexmap = { workspace = true } -journal = { path = "../journal" } -message_bus = { path = "../message_bus" } -metadata = { path = "../metadata", features = ["simulator"] } -partitions = { path = "../partitions", features = ["simulator"] } +journal = { workspace = true } +message_bus = { workspace = true } +metadata = { workspace = true, features = ["simulator"] } +partitions = { workspace = true, features = ["simulator"] } rand = { workspace = true } rand_xoshiro = { workspace = true } secrecy = { workspace = true } -# `default-features = false` drops the mimalloc global allocator and the -# web-embed feature; the sim only needs the dispatch/bootstrap library. -server = { path = "../server", default-features = false } -server_common = { path = "../server_common", features = ["simulator"] } -shard = { path = "../shard", features = ["simulator"] } +# The workspace entry disables default features: the sim only needs the +# dispatch/bootstrap library, not the mimalloc allocator or the web embed. +server = { workspace = true } +server_common = { workspace = true, features = ["simulator"] } +shard = { workspace = true, features = ["simulator"] } strum = { workspace = true } tracing = { workspace = true } tracing-subscriber = { workspace = true } diff --git a/examples/python/uv.lock b/examples/python/uv.lock index eee0abbdd9..118a278a92 100644 --- a/examples/python/uv.lock +++ b/examples/python/uv.lock @@ -13,8 +13,8 @@ source = { directory = "../../foreign/python" } [package.metadata] requires-dist = [ - { name = "maturin", marker = "extra == 'all'", specifier = ">=1.14.1,<2.0" }, - { name = "maturin", marker = "extra == 'dev'", specifier = ">=1.14.1,<2.0" }, + { name = "maturin", marker = "extra == 'all'", specifier = ">=1.15.0,<2.0" }, + { name = "maturin", marker = "extra == 'dev'", specifier = ">=1.15.0,<2.0" }, { name = "pyrefly", marker = "extra == 'all'", specifier = ">=1.2.0" }, { name = "pyrefly", marker = "extra == 'dev'", specifier = ">=1.2.0" }, { name = "pytest", marker = "extra == 'all'", specifier = ">=9.1.1,<10.0" }, diff --git a/foreign/cpp/Cargo.toml b/foreign/cpp/Cargo.toml index 0325327300..bb4802daee 100644 --- a/foreign/cpp/Cargo.toml +++ b/foreign/cpp/Cargo.toml @@ -28,7 +28,7 @@ crate-type = ["staticlib"] [dependencies] bytes = "1.12.1" -cxx = "1.0.198" +cxx = "1.0.199" iggy = { path = "../../core/sdk" } iggy_binary_protocol = { path = "../../core/binary_protocol" } iggy_common = { path = "../../core/common" } @@ -37,4 +37,4 @@ iggy_common = { path = "../../core/common" } tokio = { version = "1.53.1", features = ["rt-multi-thread", "macros", "time", "net", "io-util"] } [build-dependencies] -cxx-build = "1.0.198" +cxx-build = "1.0.199" diff --git a/foreign/cpp/MODULE.bazel b/foreign/cpp/MODULE.bazel index f274abe1de..ce679c28c1 100644 --- a/foreign/cpp/MODULE.bazel +++ b/foreign/cpp/MODULE.bazel @@ -22,13 +22,13 @@ module( bazel_dep(name = "rules_cc", version = "0.2.22") bazel_dep(name = "platforms", version = "1.1.0") -bazel_dep(name = "googletest", version = "1.18.0") +bazel_dep(name = "googletest", version = "1.18.0.bcr.1") bazel_dep(name = "cucumber-cpp", version = "0.8.0.bcr.1") -bazel_dep(name = "rules_rust", version = "0.73.0") +bazel_dep(name = "rules_rust", version = "0.74.0") rust_host_tools = use_extension("@rules_rust//rust:extensions.bzl", "rust_host_tools") rust_host_tools.host_tools( name = "rs_host_tools", - version = "1.97.1", + version = "1.98.0", ) use_repo(rust_host_tools, "rs_host_tools") diff --git a/foreign/cpp/MODULE.bazel.lock b/foreign/cpp/MODULE.bazel.lock index 0ac4a0878a..55a016bf6e 100644 --- a/foreign/cpp/MODULE.bazel.lock +++ b/foreign/cpp/MODULE.bazel.lock @@ -75,8 +75,8 @@ "https://bcr.bazel.build/modules/googletest/1.15.2/MODULE.bazel": "6de1edc1d26cafb0ea1a6ab3f4d4192d91a312fd2d360b63adaa213cd00b2108", "https://bcr.bazel.build/modules/googletest/1.17.0.bcr.2/MODULE.bazel": "827f54f492a3ce549c940106d73de332c2b30cebd0c20c0bc5d786aba7f116cb", "https://bcr.bazel.build/modules/googletest/1.17.0/MODULE.bazel": "dbec758171594a705933a29fcf69293d2468c49ec1f2ebca65c36f504d72df46", - "https://bcr.bazel.build/modules/googletest/1.18.0/MODULE.bazel": "00f855225d5746ca0d59965b09762e278b7259a6dfb7840d1b034ab60e72921e", - "https://bcr.bazel.build/modules/googletest/1.18.0/source.json": "cbf8951f03dcff3750c0265b431bb6245e91cf327a2c7284e896acefd1b87b68", + "https://bcr.bazel.build/modules/googletest/1.18.0.bcr.1/MODULE.bazel": "b489ffe6790bb6a3544b64c7604cbbea8a5529574cd4697dac227472656cea2b", + "https://bcr.bazel.build/modules/googletest/1.18.0.bcr.1/source.json": "963014628876b76748ab3babb368f53ff4afe6ee327bf9fc1e893e62757b6ac4", "https://bcr.bazel.build/modules/jsoncpp/1.9.5/MODULE.bazel": "31271aedc59e815656f5736f282bb7509a97c7ecb43e927ac1a37966e0578075", "https://bcr.bazel.build/modules/jsoncpp/1.9.6/MODULE.bazel": "2f8d20d3b7d54143213c4dfc3d98225c42de7d666011528dc8fe91591e2e17b0", "https://bcr.bazel.build/modules/jsoncpp/1.9.6/source.json": "a04756d367a2126c3541682864ecec52f92cdee80a35735a3cb249ce015ca000", @@ -199,8 +199,8 @@ "https://bcr.bazel.build/modules/rules_python/1.6.3/MODULE.bazel": "a7b80c42cb3de5ee2a5fa1abc119684593704fcd2fec83165ebe615dec76574f", "https://bcr.bazel.build/modules/rules_python/1.7.0/MODULE.bazel": "d01f995ecd137abf30238ad9ce97f8fc3ac57289c8b24bd0bf53324d937a14f8", "https://bcr.bazel.build/modules/rules_python/1.7.0/source.json": "028a084b65dcf8f4dc4f82f8778dbe65df133f234b316828a82e060d81bdce32", - "https://bcr.bazel.build/modules/rules_rust/0.73.0/MODULE.bazel": "25e3b077128612754c4add1b4c90d20a6be06566b623dee6e32038d0e8f93062", - "https://bcr.bazel.build/modules/rules_rust/0.73.0/source.json": "8eeb3d9ba7c57916b63887a651e8f84c2f68b7243af9e712d728c2a0b7882255", + "https://bcr.bazel.build/modules/rules_rust/0.74.0/MODULE.bazel": "6e6fa04d7c8202f9cd63205ec9f96da39382a4ac971ee1d7a900fe8e1f68be17", + "https://bcr.bazel.build/modules/rules_rust/0.74.0/source.json": "04eac31b4536eb23dd8e8e8ca7e079d3c1c4ac99934d38aa54bbae36917c7507", "https://bcr.bazel.build/modules/rules_shell/0.2.0/MODULE.bazel": "fda8a652ab3c7d8fee214de05e7a9916d8b28082234e8d2c0094505c5268ed3c", "https://bcr.bazel.build/modules/rules_shell/0.3.0/MODULE.bazel": "de4402cd12f4cc8fda2354fce179fdb068c0b9ca1ec2d2b17b3e21b24c1a937b", "https://bcr.bazel.build/modules/rules_shell/0.6.1/MODULE.bazel": "72e76b0eea4e81611ef5452aa82b3da34caca0c8b7b5c0c9584338aa93bae26b", @@ -229,26 +229,6 @@ }, "selectedYankedVersions": {}, "moduleExtensions": { - "@@pybind11_bazel+//:internal_configure.bzl%internal_configure_extension": { - "general": { - "bzlTransitiveDigest": "lUqTnA8vmoqQ3SKxclekEGQgFg1l4LQoUFH2ENIty5A=", - "usagesDigest": "tVQNvLoXMWAbiK39am3yovKGpwINdftfn7RpDyN+JZc=", - "recordedInputs": [ - "REPO_MAPPING:pybind11_bazel+,bazel_tools bazel_tools" - ], - "generatedRepoSpecs": { - "pybind11": { - "repoRuleId": "@@bazel_tools//tools/build_defs/repo:http.bzl%http_archive", - "attributes": { - "build_file": "@@pybind11_bazel+//:pybind11-BUILD.bazel", - "strip_prefix": "pybind11-2.13.6", - "url": "https://github.com/pybind/pybind11/archive/refs/tags/v2.13.6.tar.gz", - "integrity": "sha256-4Iy4f0dz2pf6e18DXeh2OrxlbYfVdz5i9toFh9Hw7CA=" - } - } - } - } - }, "@@rules_kotlin+//src/main/starlark/core/repositories:bzlmod_setup.bzl%rules_kotlin_extensions": { "general": { "bzlTransitiveDigest": "+Kp6j204mBZ3mxlIDDR0gBoP45BZ4jYRhRAcB8sU0qc=", @@ -506,17 +486,20 @@ }, "@@rules_rust+//crate_universe/private:internal_extensions.bzl%cu_nr": { "general": { - "bzlTransitiveDigest": "VGHJuEwR5ycRmFFD/5oLU3gxx+/rnoFRQozsFk9btzo=", - "usagesDigest": "ZmL90WEq2B6/NJ8rtHAqdnDPn+/9xG/GWR5K4UU4tyo=", + "bzlTransitiveDigest": "qQd7ktlamEwHDmeHLMSsEyM7+9Y2dOsIhs3Y/RXmCDY=", + "usagesDigest": "McimNXAztVHDzspa1Kng94T+YeyY/44zx9FkNqIMn2s=", "recordedInputs": [ + "REPO_MAPPING:apple_support+,bazel_skylib bazel_skylib+", "REPO_MAPPING:bazel_features+,bazel_features_globals bazel_features++version_extension+bazel_features_globals", "REPO_MAPPING:bazel_features+,bazel_features_version bazel_features++version_extension+bazel_features_version", + "REPO_MAPPING:rules_cc+,bazel_features bazel_features+", "REPO_MAPPING:rules_cc+,bazel_skylib bazel_skylib+", "REPO_MAPPING:rules_cc+,bazel_tools bazel_tools", "REPO_MAPPING:rules_cc+,cc_compatibility_proxy rules_cc++compatibility_proxy+cc_compatibility_proxy", "REPO_MAPPING:rules_cc+,platforms platforms", "REPO_MAPPING:rules_cc+,rules_cc rules_cc+", "REPO_MAPPING:rules_cc++compatibility_proxy+cc_compatibility_proxy,rules_cc rules_cc+", + "REPO_MAPPING:rules_rust+,apple_support apple_support+", "REPO_MAPPING:rules_rust+,bazel_features bazel_features+", "REPO_MAPPING:rules_rust+,bazel_skylib bazel_skylib+", "REPO_MAPPING:rules_rust+,bazel_tools bazel_tools", @@ -539,6 +522,7 @@ "@@rules_rust+//crate_universe:src/cli/splice.rs", "@@rules_rust+//crate_universe:src/cli/vendor.rs", "@@rules_rust+//crate_universe:src/config.rs", + "@@rules_rust+//crate_universe:src/config/label_injection.rs", "@@rules_rust+//crate_universe:src/context.rs", "@@rules_rust+//crate_universe:src/context/crate_context.rs", "@@rules_rust+//crate_universe:src/context/platforms.rs", @@ -554,13 +538,13 @@ "@@rules_rust+//crate_universe:src/metadata/metadata_annotation.rs", "@@rules_rust+//crate_universe:src/rendering.rs", "@@rules_rust+//crate_universe:src/rendering/template_engine.rs", + "@@rules_rust+//crate_universe:src/rendering/templates/defs_bzl_shim.j2", "@@rules_rust+//crate_universe:src/rendering/templates/module_bzl.j2", "@@rules_rust+//crate_universe:src/rendering/templates/partials/header.j2", "@@rules_rust+//crate_universe:src/rendering/templates/partials/module/aliases_map.j2", "@@rules_rust+//crate_universe:src/rendering/templates/partials/module/deps_map.j2", "@@rules_rust+//crate_universe:src/rendering/templates/partials/module/repo_git.j2", "@@rules_rust+//crate_universe:src/rendering/templates/partials/module/repo_http.j2", - "@@rules_rust+//crate_universe:src/rendering/templates/vendor_module.j2", "@@rules_rust+//crate_universe:src/rendering/verbatim/alias_rules.bzl", "@@rules_rust+//crate_universe:src/select.rs", "@@rules_rust+//crate_universe:src/splicing.rs", @@ -585,11 +569,11 @@ "binary": "cargo-bazel", "cargo_lockfile": "@@rules_rust+//crate_universe:Cargo.lock", "cargo_toml": "@@rules_rust+//crate_universe:Cargo.toml", - "version": "1.95.0", - "timeout": 900, + "version": "1.98.0", "rust_toolchain_cargo_template": "@rust_host_tools//:bin/{tool}", "rust_toolchain_rustc_template": "@rust_host_tools//:bin/{tool}", - "compressed_windows_toolchain_names": false + "compressed_windows_toolchain_names": false, + "timeout": 900 } } }, diff --git a/foreign/php/Cargo.toml b/foreign/php/Cargo.toml index 58502591d8..fad3fcdba6 100644 --- a/foreign/php/Cargo.toml +++ b/foreign/php/Cargo.toml @@ -29,11 +29,11 @@ repository = "https://github.com/apache/iggy" crate-type = ["cdylib"] [dependencies] -bytes = "1.11.1" -futures = "0.3.32" +bytes = "1.12.1" +futures = "0.3.34" ext-php-rs = "=0.15.14" iggy = { path = "../../core/sdk" } -tokio = "1.52.3" +tokio = "1.53.1" [profile.release] strip = "debuginfo" diff --git a/foreign/php/composer.json b/foreign/php/composer.json index 7433a0bd81..73b9a8d546 100644 --- a/foreign/php/composer.json +++ b/foreign/php/composer.json @@ -30,7 +30,7 @@ "php": ">=8.3" }, "require-dev": { - "phpunit/phpunit": "^10.5" + "phpunit/phpunit": "^12.5" }, "php-ext": { "extension-name": "iggy_php", diff --git a/foreign/php/phpunit.xml.dist b/foreign/php/phpunit.xml.dist index aef6443c5c..8cf59e2cca 100644 --- a/foreign/php/phpunit.xml.dist +++ b/foreign/php/phpunit.xml.dist @@ -19,7 +19,7 @@ under the License. --> diff --git a/foreign/python/Cargo.toml b/foreign/python/Cargo.toml index ee723b4888..f3965a16e3 100644 --- a/foreign/python/Cargo.toml +++ b/foreign/python/Cargo.toml @@ -36,10 +36,10 @@ doc = false [dependencies] bytes = "1.12.1" -futures = "0.3.33" +futures = "0.3.34" iggy = { path = "../../core/sdk", version = "0.11.0-edge.6" } paste = "1" -pyo3 = "0.29.0" +pyo3 = "0.29.2" pyo3-async-runtimes = { version = "0.29.0", features = [ "attributes", "tokio-runtime", diff --git a/foreign/python/pylock.toml b/foreign/python/pylock.toml index 4c01b652b7..37ced59acf 100644 --- a/foreign/python/pylock.toml +++ b/foreign/python/pylock.toml @@ -37,116 +37,174 @@ wheels = [ [[packages]] name = "certifi" -version = "2026.4.22" +version = "2026.7.22" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/25/ee/6caf7a40c36a1220410afe15a1cc64993a1f864871f698c0f93acb72842a/certifi-2026.4.22.tar.gz", upload-time = 2026-04-22T11:26:11Z, size = 137077, hashes = { sha256 = "8d455352a37b71bf76a79caa83a3d6c25afee4a385d632127b6afb3963f1c580" } } +sdist = { url = "https://files.pythonhosted.org/packages/a3/c2/24167ea9858356b47a87a50d39908bfdb72ceeefe0041586e704e5376b3a/certifi-2026.7.22.tar.gz", upload-time = 2026-07-22T03:35:12Z, size = 138112, hashes = { sha256 = "741e2c3b351ddf169a738da9f2c048608ff7f2c5cc02f1ebc6b118bb090d5d55" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/22/30/7cd8fdcdfbc5b869528b079bfb76dcdf6056b1a2097a662e5e8c04f42965/certifi-2026.4.22-py3-none-any.whl", upload-time = 2026-04-22T11:26:09Z, size = 135707, hashes = { sha256 = "3cb2210c8f88ba2318d29b0388d1023c8492ff72ecdde4ebdaddbb13a31b1c4a" } }, + { url = "https://files.pythonhosted.org/packages/0b/a7/71ac2cff56fec219ed242bb11b8efb69fcc4bec75db06fb7bfe35de520e6/certifi-2026.7.22-py3-none-any.whl", upload-time = 2026-07-22T03:35:11Z, size = 136983, hashes = { sha256 = "62f22742b58a1a33014a2b6b706588a8d7e2a88ae7bd1a6ebe8c992928483775" } }, ] [[packages]] name = "charset-normalizer" -version = "3.4.7" +version = "3.5.1" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/e7/a1/67fe25fac3c7642725500a3f6cfe5821ad557c3abb11c9d20d12c7008d3e/charset_normalizer-3.4.7.tar.gz", upload-time = 2026-04-02T09:28:39Z, size = 144271, hashes = { sha256 = "ae89db9e5f98a11a4bf50407d4363e7b09b31e55bc117b4f7d80aab97ba009e5" } } +sdist = { url = "https://files.pythonhosted.org/packages/e5/3f/143b048436775b0f76ac3eec145c019e8173ccc2885c8f20319b996d5e83/charset_normalizer-3.5.1.tar.gz", upload-time = 2026-08-15T08:20:44Z, size = 171764, hashes = { sha256 = "6117b84ea48435e5356dc737f5121485c30920ba43375fa7b434fd753df0eac3" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/26/08/0f303cb0b529e456bb116f2d50565a482694fbb94340bf56d44677e7ed03/charset_normalizer-3.4.7-cp310-cp310-macosx_10_9_universal2.whl", upload-time = 2026-04-02T09:25:40Z, size = 315182, hashes = { sha256 = "cdd68a1fb318e290a2077696b7eb7a21a49163c455979c639bf5a5dcdc46617d" } }, - { url = "https://files.pythonhosted.org/packages/24/47/b192933e94b546f1b1fe4df9cc1f84fcdbf2359f8d1081d46dd029b50207/charset_normalizer-3.4.7-cp310-cp310-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-04-02T09:25:42Z, size = 209329, hashes = { sha256 = "e17b8d5d6a8c47c85e68ca8379def1303fd360c3e22093a807cd34a71cd082b8" } }, - { url = "https://files.pythonhosted.org/packages/c2/b4/01fa81c5ca6141024d89a8fc15968002b71da7f825dd14113207113fabbd/charset_normalizer-3.4.7-cp310-cp310-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-04-02T09:25:44Z, size = 231230, hashes = { sha256 = "511ef87c8aec0783e08ac18565a16d435372bc1ac25a91e6ac7f5ef2b0bff790" } }, - { url = "https://files.pythonhosted.org/packages/20/f7/7b991776844dfa058017e600e6e55ff01984a063290ca5622c0b63162f68/charset_normalizer-3.4.7-cp310-cp310-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", upload-time = 2026-04-02T09:25:45Z, size = 225890, hashes = { sha256 = "007d05ec7321d12a40227aae9e2bc6dca73f3cb21058999a1df9e193555a9dcc" } }, - { url = "https://files.pythonhosted.org/packages/20/e7/bed0024a0f4ab0c8a9c64d4445f39b30c99bd1acd228291959e3de664247/charset_normalizer-3.4.7-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", upload-time = 2026-04-02T09:25:46Z, size = 216930, hashes = { sha256 = "cf29836da5119f3c8a8a70667b0ef5fdca3bb12f80fd06487cfa575b3909b393" } }, - { url = "https://files.pythonhosted.org/packages/e2/ab/b18f0ab31cdd7b3ddb8bb76c4a414aeb8160c9810fdf1bc62f269a539d87/charset_normalizer-3.4.7-cp310-cp310-manylinux_2_31_armv7l.whl", upload-time = 2026-04-02T09:25:48Z, size = 202109, hashes = { sha256 = "12d8baf840cc7889b37c7c770f478adea7adce3dcb3944d02ec87508e2dcf153" } }, - { url = "https://files.pythonhosted.org/packages/82/e5/7e9440768a06dfb3075936490cb82dbf0ee20a133bf0dd8551fa096914ec/charset_normalizer-3.4.7-cp310-cp310-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-04-02T09:25:49Z, size = 214684, hashes = { sha256 = "d560742f3c0d62afaccf9f41fe485ed69bd7661a241f86a3ef0f0fb8b1a397af" } }, - { url = "https://files.pythonhosted.org/packages/71/94/8c61d8da9f062fdf457c80acfa25060ec22bf1d34bbeaca4350f13bcfd07/charset_normalizer-3.4.7-cp310-cp310-musllinux_1_2_aarch64.whl", upload-time = 2026-04-02T09:25:50Z, size = 212785, hashes = { sha256 = "b14b2d9dac08e28bb8046a1a0434b1750eb221c8f5b87a68f4fa11a6f97b5e34" } }, - { url = "https://files.pythonhosted.org/packages/66/cd/6e9889c648e72c0ab2e5967528bb83508f354d706637bc7097190c874e13/charset_normalizer-3.4.7-cp310-cp310-musllinux_1_2_armv7l.whl", upload-time = 2026-04-02T09:25:51Z, size = 203055, hashes = { sha256 = "bc17a677b21b3502a21f66a8cc64f5bfad4df8a0b8434d661666f8ce90ac3af1" } }, - { url = "https://files.pythonhosted.org/packages/92/2e/7a951d6a08aefb7eb8e1b54cdfb580b1365afdd9dd484dc4bee9e5d8f258/charset_normalizer-3.4.7-cp310-cp310-musllinux_1_2_ppc64le.whl", upload-time = 2026-04-02T09:25:53Z, size = 232502, hashes = { sha256 = "750e02e074872a3fad7f233b47734166440af3cdea0add3e95163110816d6752" } }, - { url = "https://files.pythonhosted.org/packages/58/d5/abcf2d83bf8e0a1286df55cd0dc1d49af0da4282aa77e986df343e7de124/charset_normalizer-3.4.7-cp310-cp310-musllinux_1_2_riscv64.whl", upload-time = 2026-04-02T09:25:54Z, size = 214295, hashes = { sha256 = "4e5163c14bffd570ef2affbfdd77bba66383890797df43dc8b4cc7d6f500bf53" } }, - { url = "https://files.pythonhosted.org/packages/47/3a/7d4cd7ed54be99973a0dc176032cba5cb1f258082c31fa6df35cff46acfc/charset_normalizer-3.4.7-cp310-cp310-musllinux_1_2_s390x.whl", upload-time = 2026-04-02T09:25:55Z, size = 227145, hashes = { sha256 = "6ed74185b2db44f41ef35fd1617c5888e59792da9bbc9190d6c7300617182616" } }, - { url = "https://files.pythonhosted.org/packages/1d/98/3a45bf8247889cf28262ebd3d0872edff11565b2a1e3064ccb132db3fbb0/charset_normalizer-3.4.7-cp310-cp310-musllinux_1_2_x86_64.whl", upload-time = 2026-04-02T09:25:57Z, size = 218884, hashes = { sha256 = "94e1885b270625a9a828c9793b4d52a64445299baa1fea5a173bf1d3dd9a1a5a" } }, - { url = "https://files.pythonhosted.org/packages/ad/80/2e8b7f8915ed5c9ef13aa828d82738e33888c485b65ebf744d615040c7ea/charset_normalizer-3.4.7-cp310-cp310-win32.whl", upload-time = 2026-04-02T09:25:58Z, size = 148343, hashes = { sha256 = "6785f414ae0f3c733c437e0f3929197934f526d19dfaa75e18fdb4f94c6fb374" } }, - { url = "https://files.pythonhosted.org/packages/35/1b/3b8c8c77184af465ee9ad88b5aea46ea6b2e1f7b9dc9502891e37af21e30/charset_normalizer-3.4.7-cp310-cp310-win_amd64.whl", upload-time = 2026-04-02T09:25:59Z, size = 159174, hashes = { sha256 = "6696b7688f54f5af4462118f0bfa7c1621eeb87154f77fa04b9295ce7a8f2943" } }, - { url = "https://files.pythonhosted.org/packages/be/c1/feb40dca40dbb21e0a908801782d9288c64fc8d8e562c2098e9994c8c21b/charset_normalizer-3.4.7-cp310-cp310-win_arm64.whl", upload-time = 2026-04-02T09:26:00Z, size = 147805, hashes = { sha256 = "66671f93accb62ed07da56613636f3641f1a12c13046ce91ffc923721f23c008" } }, - { url = "https://files.pythonhosted.org/packages/c2/d7/b5b7020a0565c2e9fa8c09f4b5fa6232feb326b8c20081ccded47ea368fd/charset_normalizer-3.4.7-cp311-cp311-macosx_10_9_universal2.whl", upload-time = 2026-04-02T09:26:02Z, size = 309705, hashes = { sha256 = "7641bb8895e77f921102f72833904dcd9901df5d6d72a2ab8f31d04b7e51e4e7" } }, - { url = "https://files.pythonhosted.org/packages/5a/53/58c29116c340e5456724ecd2fff4196d236b98f3da97b404bc5e51ac3493/charset_normalizer-3.4.7-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-04-02T09:26:03Z, size = 206419, hashes = { sha256 = "202389074300232baeb53ae2569a60901f7efadd4245cf3a3bf0617d60b439d7" } }, - { url = "https://files.pythonhosted.org/packages/b2/02/e8146dc6591a37a00e5144c63f29fb7c97a734ea8a111190783c0e60ab63/charset_normalizer-3.4.7-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-04-02T09:26:04Z, size = 227901, hashes = { sha256 = "30b8d1d8c52a48c2c5690e152c169b673487a2a58de1ec7393196753063fcd5e" } }, - { url = "https://files.pythonhosted.org/packages/fb/73/77486c4cd58f1267bf17db420e930c9afa1b3be3fe8c8b8ebbebc9624359/charset_normalizer-3.4.7-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", upload-time = 2026-04-02T09:26:06Z, size = 222742, hashes = { sha256 = "532bc9bf33a68613fd7d65e4b1c71a6a38d7d42604ecf239c77392e9b4e8998c" } }, - { url = "https://files.pythonhosted.org/packages/a1/fa/f74eb381a7d94ded44739e9d94de18dc5edc9c17fb8c11f0a6890696c0a9/charset_normalizer-3.4.7-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", upload-time = 2026-04-02T09:26:08Z, size = 214061, hashes = { sha256 = "2fe249cb4651fd12605b7288b24751d8bfd46d35f12a20b1ba33dea122e690df" } }, - { url = "https://files.pythonhosted.org/packages/dc/92/42bd3cefcf7687253fb86694b45f37b733c97f59af3724f356fa92b8c344/charset_normalizer-3.4.7-cp311-cp311-manylinux_2_31_armv7l.whl", upload-time = 2026-04-02T09:26:09Z, size = 199239, hashes = { sha256 = "65bcd23054beab4d166035cabbc868a09c1a49d1efe458fe8e4361215df40265" } }, - { url = "https://files.pythonhosted.org/packages/4c/3d/069e7184e2aa3b3cddc700e3dd267413dc259854adc3380421c805c6a17d/charset_normalizer-3.4.7-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-04-02T09:26:10Z, size = 210173, hashes = { sha256 = "08e721811161356f97b4059a9ba7bafb23ea5ee2255402c42881c214e173c6b4" } }, - { url = "https://files.pythonhosted.org/packages/62/51/9d56feb5f2e7074c46f93e0ebdbe61f0848ee246e2f0d89f8e20b89ebb8f/charset_normalizer-3.4.7-cp311-cp311-musllinux_1_2_aarch64.whl", upload-time = 2026-04-02T09:26:12Z, size = 209841, hashes = { sha256 = "e060d01aec0a910bdccb8be71faf34e7799ce36950f8294c8bf612cba65a2c9e" } }, - { url = "https://files.pythonhosted.org/packages/d2/59/893d8f99cc4c837dda1fe2f1139079703deb9f321aabcb032355de13b6c7/charset_normalizer-3.4.7-cp311-cp311-musllinux_1_2_armv7l.whl", upload-time = 2026-04-02T09:26:13Z, size = 200304, hashes = { sha256 = "38c0109396c4cfc574d502df99742a45c72c08eff0a36158b6f04000043dbf38" } }, - { url = "https://files.pythonhosted.org/packages/7d/1d/ee6f3be3464247578d1ed5c46de545ccc3d3ff933695395c402c21fa6b77/charset_normalizer-3.4.7-cp311-cp311-musllinux_1_2_ppc64le.whl", upload-time = 2026-04-02T09:26:14Z, size = 229455, hashes = { sha256 = "1c2a768fdd44ee4a9339a9b0b130049139b8ce3c01d2ce09f67f5a68048d477c" } }, - { url = "https://files.pythonhosted.org/packages/54/bb/8fb0a946296ea96a488928bdce8ef99023998c48e4713af533e9bb98ef07/charset_normalizer-3.4.7-cp311-cp311-musllinux_1_2_riscv64.whl", upload-time = 2026-04-02T09:26:16Z, size = 210036, hashes = { sha256 = "1a87ca9d5df6fe460483d9a5bbf2b18f620cbed41b432e2bddb686228282d10b" } }, - { url = "https://files.pythonhosted.org/packages/9a/bc/015b2387f913749f82afd4fcba07846d05b6d784dd16123cb66860e0237d/charset_normalizer-3.4.7-cp311-cp311-musllinux_1_2_s390x.whl", upload-time = 2026-04-02T09:26:17Z, size = 224739, hashes = { sha256 = "d635aab80466bc95771bb78d5370e74d36d1fe31467b6b29b8b57b2a3cd7d22c" } }, - { url = "https://files.pythonhosted.org/packages/17/ab/63133691f56baae417493cba6b7c641571a2130eb7bceba6773367ab9ec5/charset_normalizer-3.4.7-cp311-cp311-musllinux_1_2_x86_64.whl", upload-time = 2026-04-02T09:26:18Z, size = 216277, hashes = { sha256 = "ae196f021b5e7c78e918242d217db021ed2a6ace2bc6ae94c0fc596221c7f58d" } }, - { url = "https://files.pythonhosted.org/packages/06/6d/3be70e827977f20db77c12a97e6a9f973631a45b8d186c084527e53e77a4/charset_normalizer-3.4.7-cp311-cp311-win32.whl", upload-time = 2026-04-02T09:26:20Z, size = 147819, hashes = { sha256 = "adb2597b428735679446b46c8badf467b4ca5f5056aae4d51a19f9570301b1ad" } }, - { url = "https://files.pythonhosted.org/packages/20/d9/5f67790f06b735d7c7637171bbfd89882ad67201891b7275e51116ed8207/charset_normalizer-3.4.7-cp311-cp311-win_amd64.whl", upload-time = 2026-04-02T09:26:21Z, size = 159281, hashes = { sha256 = "8e385e4267ab76874ae30db04c627faaaf0b509e1ccc11a95b3fc3e83f855c00" } }, - { url = "https://files.pythonhosted.org/packages/ca/83/6413f36c5a34afead88ce6f66684d943d91f233d76dd083798f9602b75ae/charset_normalizer-3.4.7-cp311-cp311-win_arm64.whl", upload-time = 2026-04-02T09:26:22Z, size = 147843, hashes = { sha256 = "d4a48e5b3c2a489fae013b7589308a40146ee081f6f509e047e0e096084ceca1" } }, - { url = "https://files.pythonhosted.org/packages/0c/eb/4fc8d0a7110eb5fc9cc161723a34a8a6c200ce3b4fbf681bc86feee22308/charset_normalizer-3.4.7-cp312-cp312-macosx_10_13_universal2.whl", upload-time = 2026-04-02T09:26:24Z, size = 311328, hashes = { sha256 = "eca9705049ad3c7345d574e3510665cb2cf844c2f2dcfe675332677f081cbd46" } }, - { url = "https://files.pythonhosted.org/packages/f8/e3/0fadc706008ac9d7b9b5be6dc767c05f9d3e5df51744ce4cc9605de7b9f4/charset_normalizer-3.4.7-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-04-02T09:26:25Z, size = 208061, hashes = { sha256 = "6178f72c5508bfc5fd446a5905e698c6212932f25bcdd4b47a757a50605a90e2" } }, - { url = "https://files.pythonhosted.org/packages/42/f0/3dd1045c47f4a4604df85ec18ad093912ae1344ac706993aff91d38773a2/charset_normalizer-3.4.7-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-04-02T09:26:26Z, size = 229031, hashes = { sha256 = "e1421b502d83040e6d7fb2fb18dff63957f720da3d77b2fbd3187ceb63755d7b" } }, - { url = "https://files.pythonhosted.org/packages/dc/67/675a46eb016118a2fbde5a277a5d15f4f69d5f3f5f338e5ee2f8948fcf43/charset_normalizer-3.4.7-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", upload-time = 2026-04-02T09:26:28Z, size = 225239, hashes = { sha256 = "edac0f1ab77644605be2cbba52e6b7f630731fc42b34cb0f634be1a6eface56a" } }, - { url = "https://files.pythonhosted.org/packages/4b/f8/d0118a2f5f23b02cd166fa385c60f9b0d4f9194f574e2b31cef350ad7223/charset_normalizer-3.4.7-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", upload-time = 2026-04-02T09:26:29Z, size = 216589, hashes = { sha256 = "5649fd1c7bade02f320a462fdefd0b4bd3ce036065836d4f42e0de958038e116" } }, - { url = "https://files.pythonhosted.org/packages/b1/f1/6d2b0b261b6c4ceef0fcb0d17a01cc5bc53586c2d4796fa04b5c540bc13d/charset_normalizer-3.4.7-cp312-cp312-manylinux_2_31_armv7l.whl", upload-time = 2026-04-02T09:26:30Z, size = 202733, hashes = { sha256 = "203104ed3e428044fd943bc4bf45fa73c0730391f9621e37fe39ecf477b128cb" } }, - { url = "https://files.pythonhosted.org/packages/6f/c0/7b1f943f7e87cc3db9626ba17807d042c38645f0a1d4415c7a14afb5591f/charset_normalizer-3.4.7-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-04-02T09:26:31Z, size = 212652, hashes = { sha256 = "298930cec56029e05497a76988377cbd7457ba864beeea92ad7e844fe74cd1f1" } }, - { url = "https://files.pythonhosted.org/packages/38/dd/5a9ab159fe45c6e72079398f277b7d2b523e7f716acc489726115a910097/charset_normalizer-3.4.7-cp312-cp312-musllinux_1_2_aarch64.whl", upload-time = 2026-04-02T09:26:33Z, size = 211229, hashes = { sha256 = "708838739abf24b2ceb208d0e22403dd018faeef86ddac04319a62ae884c4f15" } }, - { url = "https://files.pythonhosted.org/packages/d5/ff/531a1cad5ca855d1c1a8b69cb71abfd6d85c0291580146fda7c82857caa1/charset_normalizer-3.4.7-cp312-cp312-musllinux_1_2_armv7l.whl", upload-time = 2026-04-02T09:26:34Z, size = 203552, hashes = { sha256 = "0f7eb884681e3938906ed0434f20c63046eacd0111c4ba96f27b76084cd679f5" } }, - { url = "https://files.pythonhosted.org/packages/c1/4c/a5fb52d528a8ca41f7598cb619409ece30a169fbdf9cdce592e53b46c3a6/charset_normalizer-3.4.7-cp312-cp312-musllinux_1_2_ppc64le.whl", upload-time = 2026-04-02T09:26:36Z, size = 230806, hashes = { sha256 = "4dc1e73c36828f982bfe79fadf5919923f8a6f4df2860804db9a98c48824ce8d" } }, - { url = "https://files.pythonhosted.org/packages/59/7a/071feed8124111a32b316b33ae4de83d36923039ef8cf48120266844285b/charset_normalizer-3.4.7-cp312-cp312-musllinux_1_2_riscv64.whl", upload-time = 2026-04-02T09:26:37Z, size = 212316, hashes = { sha256 = "aed52fea0513bac0ccde438c188c8a471c4e0f457c2dd20cdbf6ea7a450046c7" } }, - { url = "https://files.pythonhosted.org/packages/fd/35/f7dba3994312d7ba508e041eaac39a36b120f32d4c8662b8814dab876431/charset_normalizer-3.4.7-cp312-cp312-musllinux_1_2_s390x.whl", upload-time = 2026-04-02T09:26:38Z, size = 227274, hashes = { sha256 = "fea24543955a6a729c45a73fe90e08c743f0b3334bbf3201e6c4bc1b0c7fa464" } }, - { url = "https://files.pythonhosted.org/packages/8a/2d/a572df5c9204ab7688ec1edc895a73ebded3b023bb07364710b05dd1c9be/charset_normalizer-3.4.7-cp312-cp312-musllinux_1_2_x86_64.whl", upload-time = 2026-04-02T09:26:40Z, size = 218468, hashes = { sha256 = "bb6d88045545b26da47aa879dd4a89a71d1dce0f0e549b1abcb31dfe4a8eac49" } }, - { url = "https://files.pythonhosted.org/packages/86/eb/890922a8b03a568ca2f336c36585a4713c55d4d67bf0f0c78924be6315ca/charset_normalizer-3.4.7-cp312-cp312-win32.whl", upload-time = 2026-04-02T09:26:41Z, size = 148460, hashes = { sha256 = "2257141f39fe65a3fdf38aeccae4b953e5f3b3324f4ff0daf9f15b8518666a2c" } }, - { url = "https://files.pythonhosted.org/packages/35/d9/0e7dffa06c5ab081f75b1b786f0aefc88365825dfcd0ac544bdb7b2b6853/charset_normalizer-3.4.7-cp312-cp312-win_amd64.whl", upload-time = 2026-04-02T09:26:42Z, size = 159330, hashes = { sha256 = "5ed6ab538499c8644b8a3e18debabcd7ce684f3fa91cf867521a7a0279cab2d6" } }, - { url = "https://files.pythonhosted.org/packages/9e/5d/481bcc2a7c88ea6b0878c299547843b2521ccbc40980cb406267088bc701/charset_normalizer-3.4.7-cp312-cp312-win_arm64.whl", upload-time = 2026-04-02T09:26:44Z, size = 147828, hashes = { sha256 = "56be790f86bfb2c98fb742ce566dfb4816e5a83384616ab59c49e0604d49c51d" } }, - { url = "https://files.pythonhosted.org/packages/c1/3b/66777e39d3ae1ddc77ee606be4ec6d8cbd4c801f65e5a1b6f2b11b8346dd/charset_normalizer-3.4.7-cp313-cp313-macosx_10_13_universal2.whl", upload-time = 2026-04-02T09:26:45Z, size = 309627, hashes = { sha256 = "f496c9c3cc02230093d8330875c4c3cdfc3b73612a5fd921c65d39cbcef08063" } }, - { url = "https://files.pythonhosted.org/packages/2e/4e/b7f84e617b4854ade48a1b7915c8ccfadeba444d2a18c291f696e37f0d3b/charset_normalizer-3.4.7-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-04-02T09:26:46Z, size = 207008, hashes = { sha256 = "0ea948db76d31190bf08bd371623927ee1339d5f2a0b4b1b4a4439a65298703c" } }, - { url = "https://files.pythonhosted.org/packages/c4/bb/ec73c0257c9e11b268f018f068f5d00aa0ef8c8b09f7753ebd5f2880e248/charset_normalizer-3.4.7-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-04-02T09:26:48Z, size = 228303, hashes = { sha256 = "a277ab8928b9f299723bc1a2dabb1265911b1a76341f90a510368ca44ad9ab66" } }, - { url = "https://files.pythonhosted.org/packages/85/fb/32d1f5033484494619f701e719429c69b766bfc4dbc61aa9e9c8c166528b/charset_normalizer-3.4.7-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", upload-time = 2026-04-02T09:26:49Z, size = 224282, hashes = { sha256 = "3bec022aec2c514d9cf199522a802bd007cd588ab17ab2525f20f9c34d067c18" } }, - { url = "https://files.pythonhosted.org/packages/fa/07/330e3a0dda4c404d6da83b327270906e9654a24f6c546dc886a0eb0ffb23/charset_normalizer-3.4.7-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", upload-time = 2026-04-02T09:26:50Z, size = 215595, hashes = { sha256 = "e044c39e41b92c845bc815e5ae4230804e8e7bc29e399b0437d64222d92809dd" } }, - { url = "https://files.pythonhosted.org/packages/e3/7c/fc890655786e423f02556e0216d4b8c6bcb6bdfa890160dc66bf52dee468/charset_normalizer-3.4.7-cp313-cp313-manylinux_2_31_armv7l.whl", upload-time = 2026-04-02T09:26:52Z, size = 201986, hashes = { sha256 = "f495a1652cf3fbab2eb0639776dad966c2fb874d79d87ca07f9d5f059b8bd215" } }, - { url = "https://files.pythonhosted.org/packages/d8/97/bfb18b3db2aed3b90cf54dc292ad79fdd5ad65c4eae454099475cbeadd0d/charset_normalizer-3.4.7-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-04-02T09:26:53Z, size = 211711, hashes = { sha256 = "e712b419df8ba5e42b226c510472b37bd57b38e897d3eca5e8cfd410a29fa859" } }, - { url = "https://files.pythonhosted.org/packages/6f/a5/a581c13798546a7fd557c82614a5c65a13df2157e9ad6373166d2a3e645d/charset_normalizer-3.4.7-cp313-cp313-musllinux_1_2_aarch64.whl", upload-time = 2026-04-02T09:26:54Z, size = 210036, hashes = { sha256 = "7804338df6fcc08105c7745f1502ba68d900f45fd770d5bdd5288ddccb8a42d8" } }, - { url = "https://files.pythonhosted.org/packages/8c/bf/b3ab5bcb478e4193d517644b0fb2bf5497fbceeaa7a1bc0f4d5b50953861/charset_normalizer-3.4.7-cp313-cp313-musllinux_1_2_armv7l.whl", upload-time = 2026-04-02T09:26:56Z, size = 202998, hashes = { sha256 = "481551899c856c704d58119b5025793fa6730adda3571971af568f66d2424bb5" } }, - { url = "https://files.pythonhosted.org/packages/e7/4e/23efd79b65d314fa320ec6017b4b5834d5c12a58ba4610aa353af2e2f577/charset_normalizer-3.4.7-cp313-cp313-musllinux_1_2_ppc64le.whl", upload-time = 2026-04-02T09:26:57Z, size = 230056, hashes = { sha256 = "f59099f9b66f0d7145115e6f80dd8b1d847176df89b234a5a6b3f00437aa0832" } }, - { url = "https://files.pythonhosted.org/packages/b9/9f/1e1941bc3f0e01df116e68dc37a55c4d249df5e6fa77f008841aef68264f/charset_normalizer-3.4.7-cp313-cp313-musllinux_1_2_riscv64.whl", upload-time = 2026-04-02T09:26:58Z, size = 211537, hashes = { sha256 = "f59ad4c0e8f6bba240a9bb85504faa1ab438237199d4cce5f622761507b8f6a6" } }, - { url = "https://files.pythonhosted.org/packages/80/0f/088cbb3020d44428964a6c97fe1edfb1b9550396bf6d278330281e8b709c/charset_normalizer-3.4.7-cp313-cp313-musllinux_1_2_s390x.whl", upload-time = 2026-04-02T09:27:00Z, size = 226176, hashes = { sha256 = "3dedcc22d73ec993f42055eff4fcfed9318d1eeb9a6606c55892a26964964e48" } }, - { url = "https://files.pythonhosted.org/packages/6a/9f/130394f9bbe06f4f63e22641d32fc9b202b7e251c9aef4db044324dac493/charset_normalizer-3.4.7-cp313-cp313-musllinux_1_2_x86_64.whl", upload-time = 2026-04-02T09:27:02Z, size = 217723, hashes = { sha256 = "64f02c6841d7d83f832cd97ccf8eb8a906d06eb95d5276069175c696b024b60a" } }, - { url = "https://files.pythonhosted.org/packages/73/55/c469897448a06e49f8fa03f6caae97074fde823f432a98f979cc42b90e69/charset_normalizer-3.4.7-cp313-cp313-win32.whl", upload-time = 2026-04-02T09:27:03Z, size = 148085, hashes = { sha256 = "4042d5c8f957e15221d423ba781e85d553722fc4113f523f2feb7b188cc34c5e" } }, - { url = "https://files.pythonhosted.org/packages/5d/78/1b74c5bbb3f99b77a1715c91b3e0b5bdb6fe302d95ace4f5b1bec37b0167/charset_normalizer-3.4.7-cp313-cp313-win_amd64.whl", upload-time = 2026-04-02T09:27:04Z, size = 158819, hashes = { sha256 = "3946fa46a0cf3e4c8cb1cc52f56bb536310d34f25f01ca9b6c16afa767dab110" } }, - { url = "https://files.pythonhosted.org/packages/68/86/46bd42279d323deb8687c4a5a811fd548cb7d1de10cf6535d099877a9a9f/charset_normalizer-3.4.7-cp313-cp313-win_arm64.whl", upload-time = 2026-04-02T09:27:05Z, size = 147915, hashes = { sha256 = "80d04837f55fc81da168b98de4f4b797ef007fc8a79ab71c6ec9bc4dd662b15b" } }, - { url = "https://files.pythonhosted.org/packages/97/c8/c67cb8c70e19ef1960b97b22ed2a1567711de46c4ddf19799923adc836c2/charset_normalizer-3.4.7-cp314-cp314-macosx_10_15_universal2.whl", upload-time = 2026-04-02T09:27:07Z, size = 309234, hashes = { sha256 = "c36c333c39be2dbca264d7803333c896ab8fa7d4d6f0ab7edb7dfd7aea6e98c0" } }, - { url = "https://files.pythonhosted.org/packages/99/85/c091fdee33f20de70d6c8b522743b6f831a2f1cd3ff86de4c6a827c48a76/charset_normalizer-3.4.7-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-04-02T09:27:08Z, size = 208042, hashes = { sha256 = "1c2aed2e5e41f24ea8ef1590b8e848a79b56f3a5564a65ceec43c9d692dc7d8a" } }, - { url = "https://files.pythonhosted.org/packages/87/1c/ab2ce611b984d2fd5d86a5a8a19c1ae26acac6bad967da4967562c75114d/charset_normalizer-3.4.7-cp314-cp314-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-04-02T09:27:09Z, size = 228706, hashes = { sha256 = "54523e136b8948060c0fa0bc7b1b50c32c186f2fceee897a495406bb6e311d2b" } }, - { url = "https://files.pythonhosted.org/packages/a8/29/2b1d2cb00bf085f59d29eb773ce58ec2d325430f8c216804a0a5cd83cbca/charset_normalizer-3.4.7-cp314-cp314-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", upload-time = 2026-04-02T09:27:11Z, size = 224727, hashes = { sha256 = "715479b9a2802ecac752a3b0efa2b0b60285cf962ee38414211abdfccc233b41" } }, - { url = "https://files.pythonhosted.org/packages/47/5c/032c2d5a07fe4d4855fea851209cca2b6f03ebeb6d4e3afdb3358386a684/charset_normalizer-3.4.7-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", upload-time = 2026-04-02T09:27:12Z, size = 215882, hashes = { sha256 = "bd6c2a1c7573c64738d716488d2cdd3c00e340e4835707d8fdb8dc1a66ef164e" } }, - { url = "https://files.pythonhosted.org/packages/2c/c2/356065d5a8b78ed04499cae5f339f091946a6a74f91e03476c33f0ab7100/charset_normalizer-3.4.7-cp314-cp314-manylinux_2_31_armv7l.whl", upload-time = 2026-04-02T09:27:13Z, size = 200860, hashes = { sha256 = "c45e9440fb78f8ddabcf714b68f936737a121355bf59f3907f4e17721b9d1aae" } }, - { url = "https://files.pythonhosted.org/packages/0c/cd/a32a84217ced5039f53b29f460962abb2d4420def55afabe45b1c3c7483d/charset_normalizer-3.4.7-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-04-02T09:27:15Z, size = 211564, hashes = { sha256 = "3534e7dcbdcf757da6b85a0bbf5b6868786d5982dd959b065e65481644817a18" } }, - { url = "https://files.pythonhosted.org/packages/44/86/58e6f13ce26cc3b8f4a36b94a0f22ae2f00a72534520f4ae6857c4b81f89/charset_normalizer-3.4.7-cp314-cp314-musllinux_1_2_aarch64.whl", upload-time = 2026-04-02T09:27:16Z, size = 211276, hashes = { sha256 = "e8ac484bf18ce6975760921bb6148041faa8fef0547200386ea0b52b5d27bf7b" } }, - { url = "https://files.pythonhosted.org/packages/8f/fe/d17c32dc72e17e155e06883efa84514ca375f8a528ba2546bee73fc4df81/charset_normalizer-3.4.7-cp314-cp314-musllinux_1_2_armv7l.whl", upload-time = 2026-04-02T09:27:18Z, size = 201238, hashes = { sha256 = "a5fe03b42827c13cdccd08e6c0247b6a6d4b5e3cdc53fd1749f5896adcdc2356" } }, - { url = "https://files.pythonhosted.org/packages/6a/29/f33daa50b06525a237451cdb6c69da366c381a3dadcd833fa5676bc468b3/charset_normalizer-3.4.7-cp314-cp314-musllinux_1_2_ppc64le.whl", upload-time = 2026-04-02T09:27:19Z, size = 230189, hashes = { sha256 = "2d6eb928e13016cea4f1f21d1e10c1cebd5a421bc57ddf5b1142ae3f86824fab" } }, - { url = "https://files.pythonhosted.org/packages/b6/6e/52c84015394a6a0bdcd435210a7e944c5f94ea1055f5cc5d56c5fe368e7b/charset_normalizer-3.4.7-cp314-cp314-musllinux_1_2_riscv64.whl", upload-time = 2026-04-02T09:27:20Z, size = 211352, hashes = { sha256 = "e74327fb75de8986940def6e8dee4f127cc9752bee7355bb323cc5b2659b6d46" } }, - { url = "https://files.pythonhosted.org/packages/8c/d7/4353be581b373033fb9198bf1da3cf8f09c1082561e8e922aa7b39bf9fe8/charset_normalizer-3.4.7-cp314-cp314-musllinux_1_2_s390x.whl", upload-time = 2026-04-02T09:27:22Z, size = 227024, hashes = { sha256 = "d6038d37043bced98a66e68d3aa2b6a35505dc01328cd65217cefe82f25def44" } }, - { url = "https://files.pythonhosted.org/packages/30/45/99d18aa925bd1740098ccd3060e238e21115fffbfdcb8f3ece837d0ace6c/charset_normalizer-3.4.7-cp314-cp314-musllinux_1_2_x86_64.whl", upload-time = 2026-04-02T09:27:23Z, size = 217869, hashes = { sha256 = "7579e913a5339fb8fa133f6bbcfd8e6749696206cf05acdbdca71a1b436d8e72" } }, - { url = "https://files.pythonhosted.org/packages/5c/05/5ee478aa53f4bb7996482153d4bfe1b89e0f087f0ab6b294fcf92d595873/charset_normalizer-3.4.7-cp314-cp314-win32.whl", upload-time = 2026-04-02T09:27:25Z, size = 148541, hashes = { sha256 = "5b77459df20e08151cd6f8b9ef8ef1f961ef73d85c21a555c7eed5b79410ec10" } }, - { url = "https://files.pythonhosted.org/packages/48/77/72dcb0921b2ce86420b2d79d454c7022bf5be40202a2a07906b9f2a35c97/charset_normalizer-3.4.7-cp314-cp314-win_amd64.whl", upload-time = 2026-04-02T09:27:26Z, size = 159634, hashes = { sha256 = "92a0a01ead5e668468e952e4238cccd7c537364eb7d851ab144ab6627dbbe12f" } }, - { url = "https://files.pythonhosted.org/packages/c6/a3/c2369911cd72f02386e4e340770f6e158c7980267da16af8f668217abaa0/charset_normalizer-3.4.7-cp314-cp314-win_arm64.whl", upload-time = 2026-04-02T09:27:28Z, size = 148384, hashes = { sha256 = "67f6279d125ca0046a7fd386d01b311c6363844deac3e5b069b514ba3e63c246" } }, - { url = "https://files.pythonhosted.org/packages/94/09/7e8a7f73d24dba1f0035fbbf014d2c36828fc1bf9c88f84093e57d315935/charset_normalizer-3.4.7-cp314-cp314t-macosx_10_15_universal2.whl", upload-time = 2026-04-02T09:27:29Z, size = 330133, hashes = { sha256 = "effc3f449787117233702311a1b7d8f59cba9ced946ba727bdc329ec69028e24" } }, - { url = "https://files.pythonhosted.org/packages/8d/da/96975ddb11f8e977f706f45cddd8540fd8242f71ecdb5d18a80723dcf62c/charset_normalizer-3.4.7-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-04-02T09:27:30Z, size = 216257, hashes = { sha256 = "fbccdc05410c9ee21bbf16a35f4c1d16123dcdeb8a1d38f33654fa21d0234f79" } }, - { url = "https://files.pythonhosted.org/packages/e5/e8/1d63bf8ef2d388e95c64b2098f45f84758f6d102a087552da1485912637b/charset_normalizer-3.4.7-cp314-cp314t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-04-02T09:27:32Z, size = 234851, hashes = { sha256 = "733784b6d6def852c814bce5f318d25da2ee65dd4839a0718641c696e09a2960" } }, - { url = "https://files.pythonhosted.org/packages/9b/40/e5ff04233e70da2681fa43969ad6f66ca5611d7e669be0246c4c7aaf6dc8/charset_normalizer-3.4.7-cp314-cp314t-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", upload-time = 2026-04-02T09:27:34Z, size = 233393, hashes = { sha256 = "a89c23ef8d2c6b27fd200a42aa4ac72786e7c60d40efdc76e6011260b6e949c4" } }, - { url = "https://files.pythonhosted.org/packages/be/c1/06c6c49d5a5450f76899992f1ee40b41d076aee9279b49cf9974d2f313d5/charset_normalizer-3.4.7-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", upload-time = 2026-04-02T09:27:35Z, size = 223251, hashes = { sha256 = "6c114670c45346afedc0d947faf3c7f701051d2518b943679c8ff88befe14f8e" } }, - { url = "https://files.pythonhosted.org/packages/2b/9f/f2ff16fb050946169e3e1f82134d107e5d4ae72647ec8a1b1446c148480f/charset_normalizer-3.4.7-cp314-cp314t-manylinux_2_31_armv7l.whl", upload-time = 2026-04-02T09:27:36Z, size = 206609, hashes = { sha256 = "a180c5e59792af262bf263b21a3c49353f25945d8d9f70628e73de370d55e1e1" } }, - { url = "https://files.pythonhosted.org/packages/69/d5/a527c0cd8d64d2eab7459784fb4169a0ac76e5a6fc5237337982fd61347e/charset_normalizer-3.4.7-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-04-02T09:27:38Z, size = 220014, hashes = { sha256 = "3c9a494bc5ec77d43cea229c4f6db1e4d8fe7e1bbffa8b6f0f0032430ff8ab44" } }, - { url = "https://files.pythonhosted.org/packages/7e/80/8a7b8104a3e203074dc9aa2c613d4b726c0e136bad1cc734594b02867972/charset_normalizer-3.4.7-cp314-cp314t-musllinux_1_2_aarch64.whl", upload-time = 2026-04-02T09:27:39Z, size = 218979, hashes = { sha256 = "8d828b6667a32a728a1ad1d93957cdf37489c57b97ae6c4de2860fa749b8fc1e" } }, - { url = "https://files.pythonhosted.org/packages/02/9a/b759b503d507f375b2b5c153e4d2ee0a75aa215b7f2489cf314f4541f2c0/charset_normalizer-3.4.7-cp314-cp314t-musllinux_1_2_armv7l.whl", upload-time = 2026-04-02T09:27:40Z, size = 209238, hashes = { sha256 = "cf1493cd8607bec4d8a7b9b004e699fcf8f9103a9284cc94962cb73d20f9d4a3" } }, - { url = "https://files.pythonhosted.org/packages/c2/4e/0f3f5d47b86bdb79256e7290b26ac847a2832d9a4033f7eb2cd4bcf4bb5b/charset_normalizer-3.4.7-cp314-cp314t-musllinux_1_2_ppc64le.whl", upload-time = 2026-04-02T09:27:42Z, size = 236110, hashes = { sha256 = "0c96c3b819b5c3e9e165495db84d41914d6894d55181d2d108cc1a69bfc9cce0" } }, - { url = "https://files.pythonhosted.org/packages/96/23/bce28734eb3ed2c91dcf93abeb8a5cf393a7b2749725030bb630e554fdd8/charset_normalizer-3.4.7-cp314-cp314t-musllinux_1_2_riscv64.whl", upload-time = 2026-04-02T09:27:43Z, size = 219824, hashes = { sha256 = "752a45dc4a6934060b3b0dab47e04edc3326575f82be64bc4fc293914566503e" } }, - { url = "https://files.pythonhosted.org/packages/2c/6f/6e897c6984cc4d41af319b077f2f600fc8214eb2fe2d6bcb79141b882400/charset_normalizer-3.4.7-cp314-cp314t-musllinux_1_2_s390x.whl", upload-time = 2026-04-02T09:27:45Z, size = 233103, hashes = { sha256 = "8778f0c7a52e56f75d12dae53ae320fae900a8b9b4164b981b9c5ce059cd1fcb" } }, - { url = "https://files.pythonhosted.org/packages/76/22/ef7bd0fe480a0ae9b656189ec00744b60933f68b4f42a7bb06589f6f576a/charset_normalizer-3.4.7-cp314-cp314t-musllinux_1_2_x86_64.whl", upload-time = 2026-04-02T09:27:46Z, size = 225194, hashes = { sha256 = "ce3412fbe1e31eb81ea42f4169ed94861c56e643189e1e75f0041f3fe7020abe" } }, - { url = "https://files.pythonhosted.org/packages/c5/a7/0e0ab3e0b5bc1219bd80a6a0d4d72ca74d9250cb2382b7c699c147e06017/charset_normalizer-3.4.7-cp314-cp314t-win32.whl", upload-time = 2026-04-02T09:27:48Z, size = 159827, hashes = { sha256 = "c03a41a8784091e67a39648f70c5f97b5b6a37f216896d44d2cdcb82615339a0" } }, - { url = "https://files.pythonhosted.org/packages/7a/1d/29d32e0fb40864b1f878c7f5a0b343ae676c6e2b271a2d55cc3a152391da/charset_normalizer-3.4.7-cp314-cp314t-win_amd64.whl", upload-time = 2026-04-02T09:27:49Z, size = 174168, hashes = { sha256 = "03853ed82eeebbce3c2abfdbc98c96dc205f32a79627688ac9a27370ea61a49c" } }, - { url = "https://files.pythonhosted.org/packages/de/32/d92444ad05c7a6e41fb2036749777c163baf7a0301a040cb672d6b2b1ae9/charset_normalizer-3.4.7-cp314-cp314t-win_arm64.whl", upload-time = 2026-04-02T09:27:51Z, size = 153018, hashes = { sha256 = "c35abb8bfff0185efac5878da64c45dafd2b37fb0383add1be155a763c1f083d" } }, - { url = "https://files.pythonhosted.org/packages/db/8f/61959034484a4a7c527811f4721e75d02d653a35afb0b6054474d8185d4c/charset_normalizer-3.4.7-py3-none-any.whl", upload-time = 2026-04-02T09:28:37Z, size = 61958, hashes = { sha256 = "3dce51d0f5e7951f8bb4900c257dad282f49190fdbebecd4ba99bcc41fef404d" } }, + { url = "https://files.pythonhosted.org/packages/71/aa/554e2614f38fc34c58ff1d0911ae8535ad2516440d5482d76fe59f1088b0/charset_normalizer-3.5.1-cp310-cp310-macosx_10_9_universal2.whl", upload-time = 2026-08-15T08:16:22Z, size = 369072, hashes = { sha256 = "d1ee1e296209fdce05b81b663250eefa02213a2da7b41bf26f7829b8ba3545aa" } }, + { url = "https://files.pythonhosted.org/packages/03/6d/439231dfc3ccfa6f8c06477b7da2219cbd41a2de3d49084df8ec7b5100f2/charset_normalizer-3.5.1-cp310-cp310-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-15T08:16:24Z, size = 251142, hashes = { sha256 = "e9fbdce1e47394b09bc9f26ab117dfc8d6491977a11d86f592bb42c779db2fda" } }, + { url = "https://files.pythonhosted.org/packages/55/53/7d819bd23a00ef45039146fa2cce1daa2f0771e758c5653ee1f6edac91ed/charset_normalizer-3.5.1-cp310-cp310-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", upload-time = 2026-08-15T08:16:26Z, size = 240714, hashes = { sha256 = "00668ebb0609751758682eb0b5857e7c35b9f00e84dfdef062e103244ec94d45" } }, + { url = "https://files.pythonhosted.org/packages/b2/2c/45847198c16f4b38090cc7423b2b6a9008e438704d8ab413211832498d31/charset_normalizer-3.5.1-cp310-cp310-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-15T08:16:27Z, size = 279637, hashes = { sha256 = "ba2f37ee79e6338845261a3c5b1784e5d1acdff2c0785b284f1b633033d136ab" } }, + { url = "https://files.pythonhosted.org/packages/69/2b/d8be3523ddf9f0b0f3e56d1359034aa10653a4d11564c697f802b4775766/charset_normalizer-3.5.1-cp310-cp310-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", upload-time = 2026-08-15T08:16:29Z, size = 276543, hashes = { sha256 = "ce854f5f478050ade5a238731c4ca985a7d3b3cb53ff600a9b5c3b689b5f0a7a" } }, + { url = "https://files.pythonhosted.org/packages/32/cd/4f564b8f132de25db594efc706897069f016790cea63a5669c9df2675f64/charset_normalizer-3.5.1-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", upload-time = 2026-08-15T08:16:30Z, size = 261644, hashes = { sha256 = "96eefc178f8636b9c760c5829345307fd81cfae9ab1e80997dbddeb0f54ee9a3" } }, + { url = "https://files.pythonhosted.org/packages/f5/e3/38b975422534a608f98c360e79c2f07c763d66dd4272300d45fb1fee54b0/charset_normalizer-3.5.1-cp310-cp310-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-15T08:16:32Z, size = 259609, hashes = { sha256 = "366ec70f5547c640d3ce1985722490f23faf4eb5216a7eeba78277490e78dacb" } }, + { url = "https://files.pythonhosted.org/packages/87/bd/fbc24d825c66f1c74f6ccdea3742c3d8354a4888e86d1315a197fee69061/charset_normalizer-3.5.1-cp310-cp310-musllinux_1_2_aarch64.whl", upload-time = 2026-08-15T08:16:33Z, size = 252457, hashes = { sha256 = "950f23cb393f85543777b0433f082cddd25b51ab398eac7971146495679efe5f" } }, + { url = "https://files.pythonhosted.org/packages/b9/2d/918d0e98a0e679469ed05bb2d90c2088b4d315bb612969d8499f76fb5210/charset_normalizer-3.5.1-cp310-cp310-musllinux_1_2_armv7l.whl", upload-time = 2026-08-15T08:16:35Z, size = 242240, hashes = { sha256 = "c1dcc36dcb96abc02236e182d17e0f71430152a6c2c7447421da2d2dc144edea" } }, + { url = "https://files.pythonhosted.org/packages/20/c8/c36f6e0b2dfec351bd38cbc05362697e58bcd073d7dbd95154290c9714ce/charset_normalizer-3.5.1-cp310-cp310-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-15T08:16:36Z, size = 280308, hashes = { sha256 = "07ffd07412fc5d5e84cd8952acf9ff7e4ed7a708e69d1bada19d8ba91711353f" } }, + { url = "https://files.pythonhosted.org/packages/ca/7b/311b3e02e8c4092400c449c850a760d8c45d900983c83a70cc07208c551d/charset_normalizer-3.5.1-cp310-cp310-musllinux_1_2_riscv64.whl", upload-time = 2026-08-15T08:16:38Z, size = 258679, hashes = { sha256 = "f5542f9b941279d82d41eb0aa9f98eba36fe4df5c7086c651df7944935b37182" } }, + { url = "https://files.pythonhosted.org/packages/b9/90/082cc45599c392f28c036a497f49e0634041a785fc3849c80ccf396d096f/charset_normalizer-3.5.1-cp310-cp310-musllinux_1_2_s390x.whl", upload-time = 2026-08-15T08:16:39Z, size = 277221, hashes = { sha256 = "a545775cfe815855ea32d7c27731d79da358ef2055b4a25830231b1622dd18aa" } }, + { url = "https://files.pythonhosted.org/packages/58/ad/b9aecf38d805cbcf84fa94f14c5d972a16561e20296a11dc799a5dcf3763/charset_normalizer-3.5.1-cp310-cp310-musllinux_1_2_x86_64.whl", upload-time = 2026-08-15T08:16:40Z, size = 263799, hashes = { sha256 = "494b70049a4d69aec6e8137c13af4cf8db8c9f9820a1392ac293b0dd2987a818" } }, + { url = "https://files.pythonhosted.org/packages/b7/23/b38a20598d5a825f85d9d7636860e56ff0db1479f86497a6e485aa9326f7/charset_normalizer-3.5.1-cp310-cp310-win32.whl", upload-time = 2026-08-15T08:16:42Z, size = 182037, hashes = { sha256 = "94fbf1c0c6cc0d3d5e50f9a9313a8cdca90dd696d34b381cd1704f8c9e939f20" } }, + { url = "https://files.pythonhosted.org/packages/d2/21/83fffb77864408b8bf0fe1ca603926401d6f8775a8e150b39aacc9958f8a/charset_normalizer-3.5.1-cp310-cp310-win_amd64.whl", upload-time = 2026-08-15T08:16:43Z, size = 206030, hashes = { sha256 = "be47f99644b208bff7766314013f9acf57b056b04191d570d68ad14022cf5b1d" } }, + { url = "https://files.pythonhosted.org/packages/86/2e/b93135b5034b1157fb29554b0d06d4844ce62282f0e0a14036f93d7ee2e7/charset_normalizer-3.5.1-cp310-cp310-win_arm64.whl", upload-time = 2026-08-15T08:16:45Z, size = 185092, hashes = { sha256 = "a6d095662e73e74f0a49988e0593373e243e3a52e27bfeea0a859e88acf4a0f5" } }, + { url = "https://files.pythonhosted.org/packages/6a/b6/034f6802e9c3f6418966cfabb7db8c9252cc2429c5098f41cc43af804149/charset_normalizer-3.5.1-cp311-cp311-macosx_10_9_universal2.whl", upload-time = 2026-08-15T08:16:46Z, size = 363585, hashes = { sha256 = "eda059b6bc8bc0812d626fd91a7ce01bf583df0a61296eff390fd94141a34e30" } }, + { url = "https://files.pythonhosted.org/packages/d5/fa/6a7e2a7c4b5451912b8c417732df79574354443592a88d616de03da66ae5/charset_normalizer-3.5.1-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-15T08:16:48Z, size = 251189, hashes = { sha256 = "aa2bb0b37202dca27175591f761108b5d34096ade1191ffe4808bdf6b1571488" } }, + { url = "https://files.pythonhosted.org/packages/a4/c8/ab42b07cfd82e919f427fcfaa7c41abae8242833ad1aad66d42bae40b669/charset_normalizer-3.5.1-cp311-cp311-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", upload-time = 2026-08-15T08:16:49Z, size = 239724, hashes = { sha256 = "0b2b1b3fa5670c127b246df1d0c059defd41f689a868a3b9d79df9b1cac42d22" } }, + { url = "https://files.pythonhosted.org/packages/e7/80/b9348b5d3041209f98b4cdad7655766369233f1d533f4f4f7558e9717bec/charset_normalizer-3.5.1-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-15T08:16:51Z, size = 280078, hashes = { sha256 = "6e5e4d73d588ca5ed09df1b7dcd1b203d1df3c542e3f50d126c947d432b10731" } }, + { url = "https://files.pythonhosted.org/packages/82/38/083a24028304bc85bb9e376fed801178423dcbb67495f73b6ea0624e1894/charset_normalizer-3.5.1-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", upload-time = 2026-08-15T08:16:52Z, size = 276650, hashes = { sha256 = "b54e7e13267d49ffbfe68e25b3cbd774dab38fa37238f71265e91b36146eb21c" } }, + { url = "https://files.pythonhosted.org/packages/0d/35/731ac04aa0a097fc1c97f0994c375bdb230c6c96619db794208fe664e9ce/charset_normalizer-3.5.1-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", upload-time = 2026-08-15T08:16:54Z, size = 262325, hashes = { sha256 = "c7b742bf31c88566b4bb6335a7f393bb322e580b6bb98df7bd0c25e6e3519ce8" } }, + { url = "https://files.pythonhosted.org/packages/f5/28/c2028e7021fb89c6e56868ed0e387b8e9aa811abdd2ab3208d6578d2c930/charset_normalizer-3.5.1-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-15T08:16:55Z, size = 261140, hashes = { sha256 = "6ba32c4d2abf1d2fe7cf27d280f4cca5664233b0f885549c7761719eb977f486" } }, + { url = "https://files.pythonhosted.org/packages/28/f0/0c0ceec6d98b7daa62e361e418135d59685811d79ba11529aad5cdf15e84/charset_normalizer-3.5.1-cp311-cp311-musllinux_1_2_aarch64.whl", upload-time = 2026-08-15T08:16:57Z, size = 252791, hashes = { sha256 = "0722590aabf9dc6a6c0343d523c05458fa2b5047dbe6302fd526bb570600753f" } }, + { url = "https://files.pythonhosted.org/packages/f0/3e/48f4cd187b1c33189d86039e9cbe4f92c05454175504b44ff81806d4d1bf/charset_normalizer-3.5.1-cp311-cp311-musllinux_1_2_armv7l.whl", upload-time = 2026-08-15T08:16:58Z, size = 240730, hashes = { sha256 = "aa1099b956fb795e686d073568f6dc002a0bb89765ea6d5b055dd7d9bf1b116c" } }, + { url = "https://files.pythonhosted.org/packages/42/85/f9e22af69af67c54cce42be9455d9c81294f918b4ccc454db01f66efcac2/charset_normalizer-3.5.1-cp311-cp311-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-15T08:16:59Z, size = 280791, hashes = { sha256 = "bd6c173f04743d483881bffa1478d5a4624475b8cd1d2194956a75548e191c18" } }, + { url = "https://files.pythonhosted.org/packages/fd/4c/9044135f42127630b6fa742feb51256353f6ab87a78f2fdd1de3de955a7f/charset_normalizer-3.5.1-cp311-cp311-musllinux_1_2_riscv64.whl", upload-time = 2026-08-15T08:17:01Z, size = 259598, hashes = { sha256 = "f298e218441525d3794428b4c8b8fb8662c6d3ea79925d4807ee6b9a96a3bca5" } }, + { url = "https://files.pythonhosted.org/packages/ba/ed/1dd7cfebb4e75812934c49ca3b79757d11948053f7937ab7070c151f3c55/charset_normalizer-3.5.1-cp311-cp311-musllinux_1_2_s390x.whl", upload-time = 2026-08-15T08:17:02Z, size = 278217, hashes = { sha256 = "6e2912d4babbc65196ac13c2f53468dc57fb8b9c25ef913e8c59ddf7c6dc0e1b" } }, + { url = "https://files.pythonhosted.org/packages/bf/eb/239c84503cc9e3ba6eb34686a24bc66e84f3924efdd7e38e751a19f6bc10/charset_normalizer-3.5.1-cp311-cp311-musllinux_1_2_x86_64.whl", upload-time = 2026-08-15T08:17:04Z, size = 263417, hashes = { sha256 = "3d27167433c0d5f18dc850f07d0b3816221984fecdc405d6c157a6f0b8f8e9e6" } }, + { url = "https://files.pythonhosted.org/packages/37/ab/4e4510e1e288478e2c8333131d1c1382382ba8cd2165053c79e39d1da961/charset_normalizer-3.5.1-cp311-cp311-win32.whl", upload-time = 2026-08-15T08:17:05Z, size = 181774, hashes = { sha256 = "ac00177c4831ffa650f8609e4bdddd5fe09c03b1c0c47acece7e6ea20421598b" } }, + { url = "https://files.pythonhosted.org/packages/e3/57/32f0ccea59e8612057c61d6fd22ef2cb63cca93c9fe594094919696ac170/charset_normalizer-3.5.1-cp311-cp311-win_amd64.whl", upload-time = 2026-08-15T08:17:07Z, size = 206653, hashes = { sha256 = "f9b1e28d0e8dbfa858abdba91d6b547beaf2df1a59bec6da6faae7b96a4991a9" } }, + { url = "https://files.pythonhosted.org/packages/17/d4/b65c433fc521e58b5f54293982a5e51c05cb5f2dd3f1c7a6acb65b75324e/charset_normalizer-3.5.1-cp311-cp311-win_arm64.whl", upload-time = 2026-08-15T08:17:08Z, size = 185630, hashes = { sha256 = "ae31a1a1db2ee6cc2942fccaf695c934bc7f3db9f2133a3fef1f367cf1a4ab10" } }, + { url = "https://files.pythonhosted.org/packages/30/27/78873dc8b6a56357517b74b6bb9568b80450e7bb4f6ef7e3fa9d22aa0bd7/charset_normalizer-3.5.1-cp312-cp312-macosx_10_13_universal2.whl", upload-time = 2026-08-15T08:17:10Z, size = 344456, hashes = { sha256 = "5b6d1386bf0096d26d3a863dc0a487a5b4eb9aa93cf5ba69683d29dde6b9d60f" } }, + { url = "https://files.pythonhosted.org/packages/9a/4c/be49ada26b1f0232d57aa89bbebf997a5cc2332a5616b6eca26ff680044d/charset_normalizer-3.5.1-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-15T08:17:11Z, size = 238530, hashes = { sha256 = "4582c27e8c889d64811987b5967fbd3ae0c823fe1fd933b543d55ac20bb475fa" } }, + { url = "https://files.pythonhosted.org/packages/76/84/6f1290fa07ae6978d3960caa3eb1b8019bf9284ab7c2297b00c099ef4250/charset_normalizer-3.5.1-cp312-cp312-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", upload-time = 2026-08-15T08:17:12Z, size = 230200, hashes = { sha256 = "1d1c7a53a6c2103925cdd6d7229f8c567379f211c869793df679f2e9f738c369" } }, + { url = "https://files.pythonhosted.org/packages/e7/a0/47b18adeed31c8f16ba9700f32c1b18594cfa09f47eb672a488c273c22bf/charset_normalizer-3.5.1-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-15T08:17:14Z, size = 262222, hashes = { sha256 = "e6621fb2a4988d6e53eedc455e5903e2679f3967b8acb3d639f1b63c14a2e893" } }, + { url = "https://files.pythonhosted.org/packages/38/fe/341861ac118dae06f3ec0eb487488af52128f2ef2faf0b11003944d22259/charset_normalizer-3.5.1-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", upload-time = 2026-08-15T08:17:16Z, size = 258951, hashes = { sha256 = "7c0c10730342b0c9b35dd1d619beb8214e520bd96a1f870f452680b238aab3e0" } }, + { url = "https://files.pythonhosted.org/packages/6f/89/bb5108dc6c3651dca963f2b0a3ba19bbcb370c94e1b6d3e0e844a58e6dca/charset_normalizer-3.5.1-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", upload-time = 2026-08-15T08:17:17Z, size = 248801, hashes = { sha256 = "b9af956078716df40d985fb0dfeb2c2120c5ca92ba4ff4b388acfd01cdc14d08" } }, + { url = "https://files.pythonhosted.org/packages/b1/ba/ef83ae3aca816393decfa3530976f38a79812d707b80b580ac33b83f9877/charset_normalizer-3.5.1-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-15T08:17:19Z, size = 244070, hashes = { sha256 = "f9f8405c2c758532c74fed975dbee57be1f31a6e865c031870c79a6ed3212ada" } }, + { url = "https://files.pythonhosted.org/packages/f6/0b/c5292a2462d69b7378ea89793bbb5b2b6fcf6f7dd6d1667f9619094ad553/charset_normalizer-3.5.1-cp312-cp312-musllinux_1_2_aarch64.whl", upload-time = 2026-08-15T08:17:20Z, size = 240110, hashes = { sha256 = "96fef3e886d6a9874b14f27fc193fbdc69d5d8035783d86aa4e1cea594e695f9" } }, + { url = "https://files.pythonhosted.org/packages/46/22/111e5be3b740d5c2a5bfcedb3d237b6591e5c2e82ae9d6ffcb121fe0909c/charset_normalizer-3.5.1-cp312-cp312-musllinux_1_2_armv7l.whl", upload-time = 2026-08-15T08:17:21Z, size = 232836, hashes = { sha256 = "5d8531a6569d025f68e2321e7638fb7978f23db58e5f69f56913837aae03816e" } }, + { url = "https://files.pythonhosted.org/packages/f9/d2/d2aad6fe0dbb44b194bf3becb60f5a0ac48446ade999a47fe7bb41eb09a7/charset_normalizer-3.5.1-cp312-cp312-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-15T08:17:23Z, size = 262712, hashes = { sha256 = "aae2ee51122d3ae968a3837d97dc24a0aeebb0dea23694422cd172bd30017cd6" } }, + { url = "https://files.pythonhosted.org/packages/35/5a/337e4663a5eae6de99db940ee8066d4145caafb61327db62deda15313cce/charset_normalizer-3.5.1-cp312-cp312-musllinux_1_2_riscv64.whl", upload-time = 2026-08-15T08:17:25Z, size = 242977, hashes = { sha256 = "7235dc28fc6dd9d832ac7c7bce95367dedb85929f17368a0c2bee1e080b9acbf" } }, + { url = "https://files.pythonhosted.org/packages/ca/85/f82f8a92e31c7519410e2e1afdc630f28ec47490ce2c09a11c1a43cbb459/charset_normalizer-3.5.1-cp312-cp312-musllinux_1_2_s390x.whl", upload-time = 2026-08-15T08:17:26Z, size = 260207, hashes = { sha256 = "4abdc5f9ad448c1ecbfae2974b820535d6bc6e7eef63babbab3d81cf46968c71" } }, + { url = "https://files.pythonhosted.org/packages/b7/52/643d11ffd60e9ac2fd1fb87e167a19285b9eefeff4a40e63c87cbfbeab36/charset_normalizer-3.5.1-cp312-cp312-musllinux_1_2_x86_64.whl", upload-time = 2026-08-15T08:17:27Z, size = 250562, hashes = { sha256 = "ba501e667c17d8411f98e67a022d9604ef179aff0e459b7e292c796837c13573" } }, + { url = "https://files.pythonhosted.org/packages/62/16/46556278c2168d12df9da7fede5dc6fc70e60301b26a82bbeec238c9cfe3/charset_normalizer-3.5.1-cp312-cp312-win32.whl", upload-time = 2026-08-15T08:17:29Z, size = 178507, hashes = { sha256 = "cfa1c0cc3a8f9f53f1243a5a99ac36fd003880199383b37672e86ddda9cb07e2" } }, + { url = "https://files.pythonhosted.org/packages/9d/7a/4c6c298171e6b3e745633180ff59350fc0ca0db1ffd28df1e369e0579f71/charset_normalizer-3.5.1-cp312-cp312-win_amd64.whl", upload-time = 2026-08-15T08:17:30Z, size = 200551, hashes = { sha256 = "3617ac3cfd8b9888f145ad89dd6e692285834b0201c6074a5eeaad3fd4d668c2" } }, + { url = "https://files.pythonhosted.org/packages/cd/d7/eb95a042f0dd22e304b0b6472b154f3546a1a039a9ee89ccb2a7f61591fc/charset_normalizer-3.5.1-cp312-cp312-win_arm64.whl", upload-time = 2026-08-15T08:17:32Z, size = 180700, hashes = { sha256 = "88e85ab89cb822c1e635f51d6d32e488f94e002e70e2f492bdb8b945543f345a" } }, + { url = "https://files.pythonhosted.org/packages/bc/61/2cb6ad133dbbb449fa2d37ccae973232f4827e799af258d15e589a3d1e9e/charset_normalizer-3.5.1-cp313-cp313-android_24_arm64_v8a.whl", upload-time = 2026-08-15T08:17:33Z, size = 211584, hashes = { sha256 = "4f298bdadb8f0b9e5672877f647d1be9373ef5320c9e2f049795e26cad28b6a9" } }, + { url = "https://files.pythonhosted.org/packages/18/57/a305c968be1ca13f3dd1b32f445877e97addf55d80b65c7cb35fac82b777/charset_normalizer-3.5.1-cp313-cp313-android_24_x86_64.whl", upload-time = 2026-08-15T08:17:35Z, size = 223359, hashes = { sha256 = "88ca277405c2d3b71c4e1c2ee0e7966e807bcba86a69d11e19ba199d18ae4491" } }, + { url = "https://files.pythonhosted.org/packages/09/0a/d3646670292ce8d8f8cc11ac067d44885e697a5591f57a9221128da5e7b3/charset_normalizer-3.5.1-cp313-cp313-ios_13_0_arm64_iphoneos.whl", upload-time = 2026-08-15T08:17:36Z, size = 194464, hashes = { sha256 = "9362dd90aa7dab48c0054a21187791ccf05473f7dba5d92b8033ae62164675e7" } }, + { url = "https://files.pythonhosted.org/packages/de/93/d51ec556e01042fed6f993ea859311bc7917b466684182fbbceb6ca24762/charset_normalizer-3.5.1-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", upload-time = 2026-08-15T08:17:37Z, size = 197676, hashes = { sha256 = "977cdbd483a9cff38179bea4fd754289a6f2195c7abd414aba85410b3e66cc5e" } }, + { url = "https://files.pythonhosted.org/packages/a4/a0/562247944386f7d4ef94467e84876600cc1e0f1b93239aaa9213d2bc3cbd/charset_normalizer-3.5.1-cp313-cp313-macosx_10_13_universal2.whl", upload-time = 2026-08-15T08:17:39Z, size = 340473, hashes = { sha256 = "e90251c0c7bdd54a100a0dce3c07b7e637278c93af29dbf78ebb89a58c4bac7d" } }, + { url = "https://files.pythonhosted.org/packages/31/e7/1d994be1b93d41e9502b8b0460eaa88a1dd8df335df415db87d6c3e91ab2/charset_normalizer-3.5.1-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-15T08:17:40Z, size = 240156, hashes = { sha256 = "94d78ecec2605a8d0398b0f365d5f12a63248438516f5dac536a5eff7337df4a" } }, + { url = "https://files.pythonhosted.org/packages/09/53/27923ce5cc6cbccb832037b27dca98882d9c53e9b69e866bbbef4aae7fc8/charset_normalizer-3.5.1-cp313-cp313-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", upload-time = 2026-08-15T08:17:42Z, size = 228246, hashes = { sha256 = "d59b75732e9b6f27388e10c14b0259cc5f2e48c78627d185e6a177b58ad3cffe" } }, + { url = "https://files.pythonhosted.org/packages/ce/48/5a97e84d63af1d55c07439cb80e56d99a8efb4295700eb4e18c0d1615d2c/charset_normalizer-3.5.1-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-15T08:17:43Z, size = 263660, hashes = { sha256 = "0d929fc574b4d6fd9e7c0f5c2ede8716a41911923aa7fa5fce38e0818aa4a1ac" } }, + { url = "https://files.pythonhosted.org/packages/7a/c2/071575791dcc88316c0a9a65ce38897a82e4cfe4a325f0f7fe1b1ac47bcf/charset_normalizer-3.5.1-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", upload-time = 2026-08-15T08:17:45Z, size = 260354, hashes = { sha256 = "394fea06235c8543390050ed5f529187074b029fb027213f6c46ac11ab5d950e" } }, + { url = "https://files.pythonhosted.org/packages/fb/af/63240b0c0248c075c2535a1f1bd992821d8251b9f173abc13329661d09e4/charset_normalizer-3.5.1-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", upload-time = 2026-08-15T08:17:46Z, size = 250638, hashes = { sha256 = "62b55f6722735a6c472f88361cde6640608773d9443cebdbb51abf436a1fcdd3" } }, + { url = "https://files.pythonhosted.org/packages/4d/66/70dfad64f15be09c15ccfee81330a7e515895dbe296dd23114e9a231268a/charset_normalizer-3.5.1-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-15T08:17:47Z, size = 244583, hashes = { sha256 = "fa48b1b63d639f9483e0633e092f5851e2348c352f1f9bb6c8182f87884ef876" } }, + { url = "https://files.pythonhosted.org/packages/c0/24/ef36367d38b9ddd4bccbf72888c342e8de1f5ae506fa0b2dcf970e2732a1/charset_normalizer-3.5.1-cp313-cp313-musllinux_1_2_aarch64.whl", upload-time = 2026-08-15T08:17:49Z, size = 242038, hashes = { sha256 = "c71fb0d56c920c269cd3e2e3fe7c610e3f1fdb21a6ce60efa6430ff63676cea6" } }, + { url = "https://files.pythonhosted.org/packages/db/ab/55e683ba0fff2e43adafc10daa3001eac90fdaa419a97227d5a7067eedde/charset_normalizer-3.5.1-cp313-cp313-musllinux_1_2_armv7l.whl", upload-time = 2026-08-15T08:17:50Z, size = 233677, hashes = { sha256 = "485a0d363cafefcd2538a73c7c838daa2035f09b2c9f9b5e3133f80c6aeb84c2" } }, + { url = "https://files.pythonhosted.org/packages/bd/67/0f40eaf8d1b6e7cf15e82382a2965efaca787fc1c2794b7021d37aaf5036/charset_normalizer-3.5.1-cp313-cp313-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-15T08:17:52Z, size = 264491, hashes = { sha256 = "5c0ea61a470e070686aa30892fed79e297d2c8d0ab46b8bcdf027d38c51da591" } }, + { url = "https://files.pythonhosted.org/packages/5c/64/12b4c2a11ee8df4fcc518c78b0d93e3a92bd3d5253d1617ce74ff0e8c7ef/charset_normalizer-3.5.1-cp313-cp313-musllinux_1_2_riscv64.whl", upload-time = 2026-08-15T08:17:54Z, size = 245196, hashes = { sha256 = "90b7481fb62fbe172c558bc6fd1c4c98d82004a54a7551f20e11ac9bf0b8708c" } }, + { url = "https://files.pythonhosted.org/packages/37/2e/651d910af6d0fba325eee1cda37ec5443462ed25360e666c144166eb6091/charset_normalizer-3.5.1-cp313-cp313-musllinux_1_2_s390x.whl", upload-time = 2026-08-15T08:17:55Z, size = 261660, hashes = { sha256 = "35fe081843b35aad20ffeccec3eeffbe637b15d14f3fb22cc1b59cd8ec17e93c" } }, + { url = "https://files.pythonhosted.org/packages/90/c6/b09e05e6db7f64338e0dc067c79577b1138da86c1e38369096851d96be88/charset_normalizer-3.5.1-cp313-cp313-musllinux_1_2_x86_64.whl", upload-time = 2026-08-15T08:17:57Z, size = 252618, hashes = { sha256 = "fd0350afdc3aabd5576f60ea109228bd5538139713c7b094c5cd27c73a98bc6f" } }, + { url = "https://files.pythonhosted.org/packages/76/4e/362d4f9fdcdf5556fb2aa3ce7d4a58ebce03ed1ff03aa1d9aca8d02f13f3/charset_normalizer-3.5.1-cp313-cp313-pyemscripten_2025_0_wasm32.whl", upload-time = 2026-08-15T08:17:58Z, size = 140362, hashes = { sha256 = "9d9a0dc7cbe9bec24c3f767c9122c41fe5a1bc43f47cd099d00d393e09769de4" } }, + { url = "https://files.pythonhosted.org/packages/b4/d4/703be739b26acce318bd29eb3b25b7209e1b1f527f9eae3d1f1f01fdde2b/charset_normalizer-3.5.1-cp313-cp313-win32.whl", upload-time = 2026-08-15T08:18:00Z, size = 177755, hashes = { sha256 = "d63600d620ad0064c3a748b950ac5ea38a80190e5498532efefa4b7b3f1da1f3" } }, + { url = "https://files.pythonhosted.org/packages/8a/33/56d97ade41c8db611e727168c52ae46c9224c362ec28d4b65d7e9869e8da/charset_normalizer-3.5.1-cp313-cp313-win_amd64.whl", upload-time = 2026-08-15T08:18:01Z, size = 199295, hashes = { sha256 = "aea996a6aba25260827c9ea511d1addfde2da9eb686ac961838509086188b7e6" } }, + { url = "https://files.pythonhosted.org/packages/5b/75/5b20dd1e6573a01a08158fe104104fa2c8abf941745596954185726cd46c/charset_normalizer-3.5.1-cp313-cp313-win_arm64.whl", upload-time = 2026-08-15T08:18:02Z, size = 179856, hashes = { sha256 = "fd0a274c0e5f9a21565cd9d3dd749b61f96b7aa1e20a93aa1ba4029518f2e5c0" } }, + { url = "https://files.pythonhosted.org/packages/29/cd/2b812ce5e888f1ce69a5350281e58aab07ae64a958ecae8912f30865718e/charset_normalizer-3.5.1-cp314-cp314-android_24_arm64_v8a.whl", upload-time = 2026-08-15T08:18:04Z, size = 212318, hashes = { sha256 = "774d157f112367ff4abd29019f38f023c24e00e56edc7829c20e358a5a913ad8" } }, + { url = "https://files.pythonhosted.org/packages/9e/4a/a6ee107430768a5334e6d63f31f148a04a1a491ef161a1ac9415a73f2fa8/charset_normalizer-3.5.1-cp314-cp314-android_24_x86_64.whl", upload-time = 2026-08-15T08:18:05Z, size = 224897, hashes = { sha256 = "26422d45fd13551cf564c58932f7d72b4f58b93b0fcf18c35ba6be12b46bb102" } }, + { url = "https://files.pythonhosted.org/packages/c3/d9/35ae3f64f29d0179c35c3baefe575904df2913dde519129c7f75995a2b1d/charset_normalizer-3.5.1-cp314-cp314-ios_13_0_arm64_iphoneos.whl", upload-time = 2026-08-15T08:18:07Z, size = 194848, hashes = { sha256 = "09a7bba9f739468c8e78c36a75c33768e53cb1959fc638f510454c14683f00d5" } }, + { url = "https://files.pythonhosted.org/packages/74/76/f2fc7380f056cc273a53af37f50d08ad54b2c59f61078f31432edcf1c2bd/charset_normalizer-3.5.1-cp314-cp314-ios_13_0_arm64_iphonesimulator.whl", upload-time = 2026-08-15T08:18:08Z, size = 198163, hashes = { sha256 = "4c9548dc78002099910abaebc0a72ac58b7d30931869e0351c09b507dff4ece3" } }, + { url = "https://files.pythonhosted.org/packages/e9/40/095ce62fa078483cccc1fa2b36e6bc9580b85422a20ee9f925341c50e44f/charset_normalizer-3.5.1-cp314-cp314-macosx_10_15_universal2.whl", upload-time = 2026-08-15T08:18:10Z, size = 341823, hashes = { sha256 = "c428c6c31eb5f4277d7f8eccaf767fbd548ddd5ce3c8b4f4cbbfab3d96b5904c" } }, + { url = "https://files.pythonhosted.org/packages/f1/5a/0e58b1c04a1596e0256f407274a92d5fb2ee21324409d1fab1da48a65b5b/charset_normalizer-3.5.1-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-15T08:18:11Z, size = 242458, hashes = { sha256 = "2f06b7eae9dbe77fe1d644ca244dad508de8d302870a43f3c559b521270938a0" } }, + { url = "https://files.pythonhosted.org/packages/22/95/b4618ce912e6db0b1aae89ba788e38e8a7eba0f3025cc66e8c0699f977b2/charset_normalizer-3.5.1-cp314-cp314-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", upload-time = 2026-08-15T08:18:13Z, size = 226717, hashes = { sha256 = "6b7430cf5728e68f6c462254009a6ef4086e1bea43cf2f57aa9c55fb4f50ff96" } }, + { url = "https://files.pythonhosted.org/packages/8a/76/c681192bbda3d55356db5dadd64381d5202b37c6b598fcda5282e88b5d3d/charset_normalizer-3.5.1-cp314-cp314-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-15T08:18:14Z, size = 266111, hashes = { sha256 = "ab743e9bc90c1f73552ec33e10e3331315acd2c397b36065b591b0181de533cc" } }, + { url = "https://files.pythonhosted.org/packages/88/be/55127bfca72c0cff6c022488d140d7c5b04c771e3b72e9bdb4836d54979d/charset_normalizer-3.5.1-cp314-cp314-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", upload-time = 2026-08-15T08:18:16Z, size = 263128, hashes = { sha256 = "f6f7deae3feb4edfa2efaf7c574fe88cbf055038a6abdb40188e4fff66d5699f" } }, + { url = "https://files.pythonhosted.org/packages/e0/91/39c3af510b0aa32bbda03374259200f28430febfd1bf5e511fe765282ce5/charset_normalizer-3.5.1-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", upload-time = 2026-08-15T08:18:18Z, size = 251240, hashes = { sha256 = "15f024313246a4ed976c60f440bb8d257815513a681d212ff74fd46f7d715a90" } }, + { url = "https://files.pythonhosted.org/packages/1c/a5/cbe418bbc6ecdfc3e05a0116002897c4b403a5e838d697e64c78e9f0190d/charset_normalizer-3.5.1-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-15T08:18:19Z, size = 245282, hashes = { sha256 = "823f82903d189af463d7df250ef1f7f696f3cee08cc8d91deb565e8d425f6506" } }, + { url = "https://files.pythonhosted.org/packages/cc/a4/689bb42e8e7cd492f3cb64907c6bc00ad247ec9a3628cd3f8eed126e8ae1/charset_normalizer-3.5.1-cp314-cp314-musllinux_1_2_aarch64.whl", upload-time = 2026-08-15T08:18:21Z, size = 244597, hashes = { sha256 = "01e93745f7f219b703b60ba7afead36cfc4242782be5af484673fc500df12da5" } }, + { url = "https://files.pythonhosted.org/packages/c1/ce/9962938e179cf9f699d3f1e7b3114b5d7642dee6a893745229f9dd04f274/charset_normalizer-3.5.1-cp314-cp314-musllinux_1_2_armv7l.whl", upload-time = 2026-08-15T08:18:22Z, size = 231376, hashes = { sha256 = "329fc3ccb63ad22d867d84c2adea759a64079a37ba4a343433b02c7a2816871e" } }, + { url = "https://files.pythonhosted.org/packages/85/54/46000450ada53bd9eac5429a2c8c54cd2d9b39c0c255f229aea9af0948a5/charset_normalizer-3.5.1-cp314-cp314-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-15T08:18:24Z, size = 266715, hashes = { sha256 = "bb57753e36e4855b8ca375069482250a6246372331a3e4f3407eaebb007443f5" } }, + { url = "https://files.pythonhosted.org/packages/3d/bb/618749d70f792b44252a777bf89bfb86823b9bbc1ea13fe8ce759b07f38a/charset_normalizer-3.5.1-cp314-cp314-musllinux_1_2_riscv64.whl", upload-time = 2026-08-15T08:18:25Z, size = 245848, hashes = { sha256 = "fce8cbd4997efeb450bd298b54f755dcdff18d496f7a5ddbb4867c6d7c88fdc3" } }, + { url = "https://files.pythonhosted.org/packages/7e/3f/ffb64458527c7668031d5eb095d978de561958dc9f5b53f8e488a533e603/charset_normalizer-3.5.1-cp314-cp314-musllinux_1_2_s390x.whl", upload-time = 2026-08-15T08:18:27Z, size = 264521, hashes = { sha256 = "6c9cdde8becb25a7fde49924511aa2644d6f8081cc8df8e9452724303348d8e3" } }, + { url = "https://files.pythonhosted.org/packages/4f/ab/74a55fd803916a35ac461daf002708191aac19b546b80dc8cabfedc63d98/charset_normalizer-3.5.1-cp314-cp314-musllinux_1_2_x86_64.whl", upload-time = 2026-08-15T08:18:28Z, size = 253054, hashes = { sha256 = "9ac4444d8d4fd4c4bd08bf451ed3167aa9e7ec6cdb41b648794f1d1103652e36" } }, + { url = "https://files.pythonhosted.org/packages/a0/2a/6a9034b7d3c60b17499afb482df5878bf9fa20b50cc3887d5ef017a833db/charset_normalizer-3.5.1-cp314-cp314-pyemscripten_2026_0_wasm32.whl", upload-time = 2026-08-15T08:18:30Z, size = 140580, hashes = { sha256 = "f03ac127268b43ef4fe9e6ab6794a6794b49485a0cc0c1db79876d2f33f75bc7" } }, + { url = "https://files.pythonhosted.org/packages/f3/46/1d362e1a00d035d66b9869e1281eee115907f7e390a16a07824ab5737360/charset_normalizer-3.5.1-cp314-cp314-win32.whl", upload-time = 2026-08-15T08:18:31Z, size = 180325, hashes = { sha256 = "1f5883d77fd409a261abb5dc8ccbe335720d798b1de4abb3b1d47ccbbc76b53b" } }, + { url = "https://files.pythonhosted.org/packages/7a/7c/4938c329b6a9d446f6a59aa2092ff7118f274209b5ed0e26893d1d30a63c/charset_normalizer-3.5.1-cp314-cp314-win_amd64.whl", upload-time = 2026-08-15T08:18:33Z, size = 204175, hashes = { sha256 = "c658c50ac0c98cd755a2dd50b7977d3bca7df401dcc47fbdfa87db53ef7d4e8b" } }, + { url = "https://files.pythonhosted.org/packages/ac/33/eeb384dbd8dec570661354592f4f2e1b2fcc92585624d146a000caf53841/charset_normalizer-3.5.1-cp314-cp314-win_arm64.whl", upload-time = 2026-08-15T08:18:34Z, size = 184123, hashes = { sha256 = "4bea7f8ebe90bbd7f0e4a2de42ca6924ba23e3e76418c408ff82f1d46fabd687" } }, + { url = "https://files.pythonhosted.org/packages/1c/6c/c73fa9d5a85f6ab05395de61c5f6984e0a9ff40bb5ff888d46dff02526c6/charset_normalizer-3.5.1-cp314-cp314t-macosx_10_15_universal2.whl", upload-time = 2026-08-15T08:18:36Z, size = 381682, hashes = { sha256 = "fbc597639158fd7c14d55e808718848319540f51b0e6746e3eefa59723a4a348" } }, + { url = "https://files.pythonhosted.org/packages/30/c7/63565f860921457feba93bae6c86fb7746deb4cffeed2f375cb845318146/charset_normalizer-3.5.1-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-15T08:18:37Z, size = 240826, hashes = { sha256 = "e71c909f353863b2b89c83de2ebed71ea6d0df8a6ef65a128193c5e650766bef" } }, + { url = "https://files.pythonhosted.org/packages/06/ae/7ae8807410dfa33f8e6f1715740adeaafa8a816cc4cb33508f54b1f7c896/charset_normalizer-3.5.1-cp314-cp314t-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", upload-time = 2026-08-15T08:18:39Z, size = 227861, hashes = { sha256 = "7ac76cf9afd34929d76eb7fcb63be476a4853d8a96f0dcf2d0db68a0cbdf9885" } }, + { url = "https://files.pythonhosted.org/packages/e9/a3/887c1642f0da26000b0e0652d91071113c0e72cea33952e225cf589f49a9/charset_normalizer-3.5.1-cp314-cp314t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-15T08:18:40Z, size = 260758, hashes = { sha256 = "a3a370082ce34d0612f421e15fe011c53bb1feff21a26d06ad4fb244dab5a375" } }, + { url = "https://files.pythonhosted.org/packages/3e/11/e6f5b9a3d0e55b0ef7505cd3765cdd48f22db89994c947b316f52f801fd8/charset_normalizer-3.5.1-cp314-cp314t-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", upload-time = 2026-08-15T08:18:42Z, size = 259950, hashes = { sha256 = "256dd4d85d9e4dc595e2bc983c980e73f62ddeb3165c58b4c3dfe78c5c8548c1" } }, + { url = "https://files.pythonhosted.org/packages/1b/ee/e4e10a94d51cd1ee638aa7e00b65399e6b2a4e8376ab6d2eac9f95586671/charset_normalizer-3.5.1-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", upload-time = 2026-08-15T08:18:43Z, size = 249329, hashes = { sha256 = "58d4aa13a59c969dbfdf9e6a9560e242cbfd9e8a8f50c2747714df1a423adf65" } }, + { url = "https://files.pythonhosted.org/packages/c4/25/d5f4198819e6059735a84e8d0bfb72dc33976da67b97adcd3fb5a5e07ec6/charset_normalizer-3.5.1-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-15T08:18:45Z, size = 243137, hashes = { sha256 = "0c6dfb5ca6723eeed15aa8e564a014d69fcb8812f94eef11fe3631e0508199f5" } }, + { url = "https://files.pythonhosted.org/packages/a5/e9/e925ca7569cf9fb9701fd82503fee73eea5268fdb856bdd64947092d3daa/charset_normalizer-3.5.1-cp314-cp314t-musllinux_1_2_aarch64.whl", upload-time = 2026-08-15T08:18:46Z, size = 242820, hashes = { sha256 = "c010f5581d9c612804cc59fcf7b524b707fbcb72828551237ab545bb5c7034af" } }, + { url = "https://files.pythonhosted.org/packages/34/17/672c251a888ed2aebcdd2fe830ad0104e25ff83c43f5c4f9c15e9fc6853c/charset_normalizer-3.5.1-cp314-cp314t-musllinux_1_2_armv7l.whl", upload-time = 2026-08-15T08:18:48Z, size = 230504, hashes = { sha256 = "52ec005752a56ae79547a05c0139ca2501a0c866390b6115008456b9f0e7cde1" } }, + { url = "https://files.pythonhosted.org/packages/3f/fc/f6a85abebd42ce4da2f1db0aa56cc6a0df1995e318b3875d14401b8381d1/charset_normalizer-3.5.1-cp314-cp314t-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-15T08:18:49Z, size = 263087, hashes = { sha256 = "2bced4061f000f7187254a02ad3433ae17eaf991747ceea2f478422590a5bba9" } }, + { url = "https://files.pythonhosted.org/packages/98/66/7c42677e739ba66746b297e2046918d793078094dc239e1e72768cffccc6/charset_normalizer-3.5.1-cp314-cp314t-musllinux_1_2_riscv64.whl", upload-time = 2026-08-15T08:18:51Z, size = 243269, hashes = { sha256 = "9eea3ab2597a5e65fe65296e2d6a84570845a6b55532d90333d740d48bbc850a" } }, + { url = "https://files.pythonhosted.org/packages/de/d8/a50b79237f417af10f8c2a501ce8d1ca87829a22e69117891ca4ba20a69e/charset_normalizer-3.5.1-cp314-cp314t-musllinux_1_2_s390x.whl", upload-time = 2026-08-15T08:18:53Z, size = 258766, hashes = { sha256 = "496846868fea80e479324862fa877f02411f2fd0f83b79ccee2607aa68b2a032" } }, + { url = "https://files.pythonhosted.org/packages/2e/1d/0fc91aeaeb3c83b748f532399ce67cf84604b48297405d740000f7a9e786/charset_normalizer-3.5.1-cp314-cp314t-musllinux_1_2_x86_64.whl", upload-time = 2026-08-15T08:18:54Z, size = 250814, hashes = { sha256 = "85d5855daafc240cc045c026d7a15fd198a09b0fc8ff6f5ecbb5297b509cb11e" } }, + { url = "https://files.pythonhosted.org/packages/ae/10/3d8c777cf9024615295aa1b808324ad5b4a77855869c00824bad74ffaf8a/charset_normalizer-3.5.1-cp314-cp314t-win32.whl", upload-time = 2026-08-15T08:18:56Z, size = 191074, hashes = { sha256 = "58d3e12c88e0950bca850ae1f7c256055c097639c2edb9eb123af9807d8b15e4" } }, + { url = "https://files.pythonhosted.org/packages/4d/81/ae557d3c44d1a1d688696d60563413a0866a91b7ebc50f20df838be3d8c8/charset_normalizer-3.5.1-cp314-cp314t-win_amd64.whl", upload-time = 2026-08-15T08:18:57Z, size = 216476, hashes = { sha256 = "acaf604462bf330b0d07e7a07c1d6e4adac79e5fb13e9c5140590542cafacc00" } }, + { url = "https://files.pythonhosted.org/packages/27/e9/61c01fb8b804692569c036b3fc50495814502dcf13a60649c6055390b02c/charset_normalizer-3.5.1-cp314-cp314t-win_arm64.whl", upload-time = 2026-08-15T08:18:59Z, size = 194115, hashes = { sha256 = "fdb8a068947befafba9952162645dc2fecaeb400e64584829ed5e9b2fbe21a7f" } }, + { url = "https://files.pythonhosted.org/packages/4a/4e/8544831ef59d8f27ce92c80871380fdacc8076a8a56ed62f82e54f991333/charset_normalizer-3.5.1-cp315-cp315-macosx_10_15_universal2.whl", upload-time = 2026-08-15T08:19:01Z, size = 342048, hashes = { sha256 = "9085f87b0e38a2b92b8923059b4e8789fe40d9279712d15dcc670048d77079af" } }, + { url = "https://files.pythonhosted.org/packages/7f/a6/e3b46852424246065355644f4fb6dbccc0239a42a2eee27ecfc8957f0bcd/charset_normalizer-3.5.1-cp315-cp315-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-15T08:19:02Z, size = 242997, hashes = { sha256 = "2679de311c7946dde5d3b6f44941844133ff5c7cb86099c0061ab1e8901c20a8" } }, + { url = "https://files.pythonhosted.org/packages/03/3b/0cc9a26777334ab2f2e3089b948bbf4e4fe72ea70b897715ef6415043ec8/charset_normalizer-3.5.1-cp315-cp315-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", upload-time = 2026-08-15T08:19:03Z, size = 237014, hashes = { sha256 = "baf3775a2635e5a11fbd5e4e64ee69c7e86875d224a5c72aca4c141064589a90" } }, + { url = "https://files.pythonhosted.org/packages/8c/c2/027335f0aa337a2a2e121bac1ad88c4f02ba6053ea0926802784f3db11af/charset_normalizer-3.5.1-cp315-cp315-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-15T08:19:05Z, size = 266174, hashes = { sha256 = "8ac8c94b6539074e0f40899301273ac8402b9b3e01c7b7ba269ff30340aaaf20" } }, + { url = "https://files.pythonhosted.org/packages/86/d3/e367787febe4e74769dec0f406f2c3c8d1b955fce5aee1fd0f94e8367a45/charset_normalizer-3.5.1-cp315-cp315-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", upload-time = 2026-08-15T08:19:07Z, size = 263361, hashes = { sha256 = "8fe532b3c966d1fb794e0698e4589d0444017ae77fc0b31edea13c0e35bcc449" } }, + { url = "https://files.pythonhosted.org/packages/af/3d/391b193eb9f3e84b02f9314088c386debdc0debee843535aaea2e2c6715d/charset_normalizer-3.5.1-cp315-cp315-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", upload-time = 2026-08-15T08:19:08Z, size = 252143, hashes = { sha256 = "5c84bec0ab5ae0c64bfe73a7d2adcb5ce73b467523fc27fd6a28ab2aa6cbe35a" } }, + { url = "https://files.pythonhosted.org/packages/2e/57/de221f1745a90d418199761967e2776bfe2c275a1194220985e8c1d37833/charset_normalizer-3.5.1-cp315-cp315-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-15T08:19:10Z, size = 252086, hashes = { sha256 = "854066be00447fa8de2ccbbe893e2ffc4b123ef16d897af794c1e18bd4a714b0" } }, + { url = "https://files.pythonhosted.org/packages/c8/e3/d119f86a01f9331e8186175f24873b1d74a7ee9e2e4b4d68f9947dae5afd/charset_normalizer-3.5.1-cp315-cp315-musllinux_1_2_aarch64.whl", upload-time = 2026-08-15T08:19:11Z, size = 245231, hashes = { sha256 = "21b82d8082f6f5e7f456ef0bd16323d08de1266efbfeb476e64b2a91d1471a4e" } }, + { url = "https://files.pythonhosted.org/packages/26/de/d8e48c135ae480879539cdb179c8d3b50c7879497d75dd899b5763b69cee/charset_normalizer-3.5.1-cp315-cp315-musllinux_1_2_armv7l.whl", upload-time = 2026-08-15T08:19:13Z, size = 241546, hashes = { sha256 = "838648accb3a7fd9803fd45c87bce8509648eb0c11bc34e216141300977244f2" } }, + { url = "https://files.pythonhosted.org/packages/67/c4/217755fd1abc50d326c252922cd642002758095a81ff45010337b8b3ef65/charset_normalizer-3.5.1-cp315-cp315-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-15T08:19:14Z, size = 267033, hashes = { sha256 = "195ce897c6153c0700078142cf8efe3e6454ca4cf4357499e4078dfd83396626" } }, + { url = "https://files.pythonhosted.org/packages/b8/d7/34d8e404e358d2adcc5a228c2134643af00104c8fb0bf525f3688d756f05/charset_normalizer-3.5.1-cp315-cp315-musllinux_1_2_riscv64.whl", upload-time = 2026-08-15T08:19:16Z, size = 252045, hashes = { sha256 = "978eab16f55b4ab2c2a745be9a0a840bf8f09a7f227d9c76eb30214d078865a5" } }, + { url = "https://files.pythonhosted.org/packages/5e/fa/40414471acf0aa0692ca77305aa00e434fcd8288f0941c93c30e9a5f8f2f/charset_normalizer-3.5.1-cp315-cp315-musllinux_1_2_s390x.whl", upload-time = 2026-08-15T08:19:18Z, size = 264866, hashes = { sha256 = "cc0329df4caaceb950d2f580b5ac716a377f7059624a0bafaeaf8a218c6ed774" } }, + { url = "https://files.pythonhosted.org/packages/32/90/fcc850bae791abd2e0c041847f13e270aa08692a79f3e00de6d2dce1cb50/charset_normalizer-3.5.1-cp315-cp315-musllinux_1_2_x86_64.whl", upload-time = 2026-08-15T08:19:19Z, size = 253932, hashes = { sha256 = "687c9ca3035544b113bea2055e180af96fb63c0c476e22a9180f51925186e7b7" } }, + { url = "https://files.pythonhosted.org/packages/af/af/53afe99068b3c10b4cbae592a52ef72a7c92c0188440e83ee3a078fd8f75/charset_normalizer-3.5.1-cp315-cp315-win32.whl", upload-time = 2026-08-15T08:19:21Z, size = 180320, hashes = { sha256 = "706bfd38730a5ac7a365793269a00f4e988178cec121391f4248d84ad8c972e9" } }, + { url = "https://files.pythonhosted.org/packages/c9/bc/f46a132041b29e4a8779ed712d3df1bf112e94ca8de58b66d7ec2c0cf8b9/charset_normalizer-3.5.1-cp315-cp315-win_amd64.whl", upload-time = 2026-08-15T08:19:23Z, size = 204174, hashes = { sha256 = "92caef967d287a407085d61176fce4012b1dd62daed4eb6d5ceb26d3d2538712" } }, + { url = "https://files.pythonhosted.org/packages/a1/5d/9ed554480eda8e447b673648628fdc29574d23dbad01fe11837adedd1cae/charset_normalizer-3.5.1-cp315-cp315-win_arm64.whl", upload-time = 2026-08-15T08:19:24Z, size = 184126, hashes = { sha256 = "5fc45d653ea8c9a20479167e11d4a0f8cb2fa3470737ab6f9c827532313187b7" } }, + { url = "https://files.pythonhosted.org/packages/3b/32/9b8929bf384061ee1fe5d9c27c6f9776d3d824039ad4e14c88ec00c7808e/charset_normalizer-3.5.1-cp315-cp315t-macosx_10_15_universal2.whl", upload-time = 2026-08-15T08:19:26Z, size = 381441, hashes = { sha256 = "59171c6e45bf07d0d5cab3b0bf81d945035530f6873398b3b531c31184d46663" } }, + { url = "https://files.pythonhosted.org/packages/96/10/e9aa7923d3ddac652c99a1c5f7be494e737e151566a44abe018daf757f2c/charset_normalizer-3.5.1-cp315-cp315t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-15T08:19:27Z, size = 241742, hashes = { sha256 = "9dbdd9205662134957cf0c324f639bdc5031c0ca056e2369e238db75187c0f11" } }, + { url = "https://files.pythonhosted.org/packages/28/53/a2d249ebddf47b889a100c0bdcb61a2f9dbb8bc24ef325cc062e4f476877/charset_normalizer-3.5.1-cp315-cp315t-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", upload-time = 2026-08-15T08:19:29Z, size = 235298, hashes = { sha256 = "e4b018dc5a0eee4676e38fe84a47a427816c590b93b55d9025274ec4d6ffc2dc" } }, + { url = "https://files.pythonhosted.org/packages/7d/07/469f78af590f7d5cd48e20d8dbfa3d66deeff9ba37768c04d886b5afd45c/charset_normalizer-3.5.1-cp315-cp315t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-15T08:19:30Z, size = 262500, hashes = { sha256 = "ced3fdd71aaa83ce593746c2edb42b7a59cb4c19c8b5c407781c72e493aae55a" } }, + { url = "https://files.pythonhosted.org/packages/55/66/3bb56a47f7dcba014055b1a1d33c6f08bbe9c1e74dba154cfa25f90ae885/charset_normalizer-3.5.1-cp315-cp315t-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", upload-time = 2026-08-15T08:19:32Z, size = 258888, hashes = { sha256 = "19a3dd5aa73cef1c99687c4fc57db016a9c17104ae1185da88ba566a5d3bebe4" } }, + { url = "https://files.pythonhosted.org/packages/ff/c1/2adc2800903fb013210349313b710a5376856578d9e33e6b9a1d8b36714a/charset_normalizer-3.5.1-cp315-cp315t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", upload-time = 2026-08-15T08:19:33Z, size = 250243, hashes = { sha256 = "cc5d36d96478aa9c60654bd932525bf32964c62a7281eafdf16d85003a8d6004" } }, + { url = "https://files.pythonhosted.org/packages/95/b5/a18d0dd1157ab655cc2cb14a545f4a4784bbad70ab3502412e36097502d9/charset_normalizer-3.5.1-cp315-cp315t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-15T08:19:35Z, size = 249871, hashes = { sha256 = "04368edf83514385ffc3e1cfd4546e595f4f1272dd23ba437a93a9cc3741d47b" } }, + { url = "https://files.pythonhosted.org/packages/ad/c3/525f508cd1e58d0450ac55ed40ac75bc3a97482c59def5278456a5fbf03c/charset_normalizer-3.5.1-cp315-cp315t-musllinux_1_2_aarch64.whl", upload-time = 2026-08-15T08:19:36Z, size = 243580, hashes = { sha256 = "9b5db6052055d34d41230fb78d7c439c23dc536a9896f6cb039e8dd92cfc1263" } }, + { url = "https://files.pythonhosted.org/packages/7c/c1/49a91fe7e97c8140094ca5c64161ab623a70d9f636bf834eace14048acb5/charset_normalizer-3.5.1-cp315-cp315t-musllinux_1_2_armv7l.whl", upload-time = 2026-08-15T08:19:38Z, size = 239807, hashes = { sha256 = "252d099029bcbea642f2a06c4ed5046bdf8b5a8150b64afa5e027e88b106e5ee" } }, + { url = "https://files.pythonhosted.org/packages/d3/58/56a48c296601274c4689b864a8e2dfb209b81dfcb39472753ce95eea662b/charset_normalizer-3.5.1-cp315-cp315t-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-15T08:19:39Z, size = 264083, hashes = { sha256 = "6199d5606e2bbf2b096cf64d03f8b6790c91081d5ac866b8e7bb6422738cc60c" } }, + { url = "https://files.pythonhosted.org/packages/10/4c/dc48409274a1817ff349711d26c62aa0c597df865d4d69ef79160c859193/charset_normalizer-3.5.1-cp315-cp315t-musllinux_1_2_riscv64.whl", upload-time = 2026-08-15T08:19:41Z, size = 250317, hashes = { sha256 = "77efcff2b23071c349402ac1066667a3d011f62398d81408c9b88ad991747c9e" } }, + { url = "https://files.pythonhosted.org/packages/81/58/d325912115caec62d6bdd77bbab5e0b7da5d234a9f20affdffcbcb530d0b/charset_normalizer-3.5.1-cp315-cp315t-musllinux_1_2_s390x.whl", upload-time = 2026-08-15T08:19:43Z, size = 258173, hashes = { sha256 = "a5cbd90ecf0fc62e64726917ad083b73001f0563657a87ec3c0b504e277dc90d" } }, + { url = "https://files.pythonhosted.org/packages/34/f7/b13b1ccae2c8ec63980d13be1890eb73f8aeabbfce02a24aabc0908788f5/charset_normalizer-3.5.1-cp315-cp315t-musllinux_1_2_x86_64.whl", upload-time = 2026-08-15T08:19:44Z, size = 251960, hashes = { sha256 = "4d26f14f041e83dd8edfd61f4cd4fa7285d31798b5bf1f28e70c367ba6c41d61" } }, + { url = "https://files.pythonhosted.org/packages/1e/25/ed3f9919c5aef8cc818be1f972f565f7610d7b2076b8ebb98839516ffc3c/charset_normalizer-3.5.1-cp315-cp315t-win32.whl", upload-time = 2026-08-15T08:19:46Z, size = 191186, hashes = { sha256 = "ac13b004224fb341e1e25a1ed5e19d32f57cdb2a403e01f003b46f051a550f6f" } }, + { url = "https://files.pythonhosted.org/packages/69/d5/43c2b3e9d8267092b913eb8b0603f0f71993c395632886bd37a7223f96cf/charset_normalizer-3.5.1-cp315-cp315t-win_amd64.whl", upload-time = 2026-08-15T08:19:47Z, size = 215947, hashes = { sha256 = "35aea775dc2bd5f54cd84a1cd2696cc3207c479cb9cf0bd346f0d343e4300ddb" } }, + { url = "https://files.pythonhosted.org/packages/a8/76/9aad3e9c8865e5e0efa9a7f6f81c37a67635a985145ecd44528a81e088ee/charset_normalizer-3.5.1-cp315-cp315t-win_arm64.whl", upload-time = 2026-08-15T08:19:49Z, size = 193909, hashes = { sha256 = "fb78f6e7fcd8ad785d28cd577168bc1aaee827b25bb8755638f694794ea98f0a" } }, + { url = "https://files.pythonhosted.org/packages/5b/97/fb4e82231aba271ffd775a1b4993b0defc4e3059f286ae41d9433409fe85/charset_normalizer-3.5.1-cp37-abi3-macosx_10_9_universal2.whl", upload-time = 2026-08-15T08:19:50Z, size = 331467, hashes = { sha256 = "41876ee62a3dddf48ff1121ad8f0798032aa03f2fd35f21f34a4cab14f18d8d2" } }, + { url = "https://files.pythonhosted.org/packages/9f/2f/fe3f187327aac18e2d54e9d2b08e15d27bf9b642d9e51c219f130fc34d1a/charset_normalizer-3.5.1-cp37-abi3-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-08-15T08:19:52Z, size = 253057, hashes = { sha256 = "a6dac12ff6b846103483683f60c5f8fee205121adc58ffd87e90a90a3af69e99" } }, + { url = "https://files.pythonhosted.org/packages/d7/c7/9e48cee5c161fe24da823b61bf381921d77cb994a0a4de148e95018c1984/charset_normalizer-3.5.1-cp37-abi3-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-15T08:19:54Z, size = 240930, hashes = { sha256 = "cee5dd7c6fb5dd52a0fe2a740f9bc6e3593f5f8b1788bde49de02086f30182b2" } }, + { url = "https://files.pythonhosted.org/packages/49/e0/716601f3cc69be7b198951150c75ead1ece33c3c8036ff6ffa46029659a0/charset_normalizer-3.5.1-cp37-abi3-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", upload-time = 2026-08-15T08:19:55Z, size = 230822, hashes = { sha256 = "343fb4f2821043bd87095f7b08a1a181febc8e36ac64212143bbfd0a0e1bc235" } }, + { url = "https://files.pythonhosted.org/packages/d3/05/71bfc5caa0abcc45aea1f6a4d50ac68e59605ddc7666fe8494f4cd229665/charset_normalizer-3.5.1-cp37-abi3-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-15T08:19:57Z, size = 260037, hashes = { sha256 = "ae4a097991662cd4fff0ddc74e0fe7874f82e00042fa0ea00855645ed0c79598" } }, + { url = "https://files.pythonhosted.org/packages/c3/92/de7e32ed05341e7a9c4c877c318418197b7f2d66a3b68d561bf2ac57ca3e/charset_normalizer-3.5.1-cp37-abi3-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", upload-time = 2026-08-15T08:19:59Z, size = 255097, hashes = { sha256 = "4b599739b93b2cbeded49645ae3c8d1405c29ddfbceac1545c87a3f9580a9e96" } }, + { url = "https://files.pythonhosted.org/packages/f5/7b/ade0a122600319dfa0b1000ab0f9731c94a817904cf3c5de408c73a4ede7/charset_normalizer-3.5.1-cp37-abi3-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-15T08:20:00Z, size = 250166, hashes = { sha256 = "b39b69b347e5e47a3b5b8cfc005c68c1ba347474e3960236c4944a8ecd174962" } }, + { url = "https://files.pythonhosted.org/packages/75/9c/019fbb9f4834491a160951349b1a3714439376f66e5f7cf18b4f18f0c7aa/charset_normalizer-3.5.1-cp37-abi3-musllinux_1_2_aarch64.whl", upload-time = 2026-08-15T08:20:02Z, size = 241821, hashes = { sha256 = "a2028475ba855475b8b4d3cfeb4994269c967aea8b9892dfba907f4263a863a3" } }, + { url = "https://files.pythonhosted.org/packages/2b/b8/11d4840bfc99330cc7fbcc2681ee5a044553a6e77655508d8f9b2bff7b34/charset_normalizer-3.5.1-cp37-abi3-musllinux_1_2_armv7l.whl", upload-time = 2026-08-15T08:20:04Z, size = 232529, hashes = { sha256 = "36047af20e17097c3bb9476c2b7655f2f7aa51322c0ba58c07695bedf755a950" } }, + { url = "https://files.pythonhosted.org/packages/18/96/2b3a21492d9f65171ac75d872f5018260013d00bfa0ff70ec9f179148cbd/charset_normalizer-3.5.1-cp37-abi3-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-15T08:20:05Z, size = 260348, hashes = { sha256 = "4c4fb141a727957c93edfe5c32a26ceb6b5f6461d67146e2d39f51e16170bea8" } }, + { url = "https://files.pythonhosted.org/packages/d6/aa/a69a2028e8bd052476c245460ab19d7de595de084dd968f2d75cd50c3e25/charset_normalizer-3.5.1-cp37-abi3-musllinux_1_2_riscv64.whl", upload-time = 2026-08-15T08:20:07Z, size = 247234, hashes = { sha256 = "2f293479cce755c75f1697e87c409b7ae4c555c7dfecb6e988ad13abba943031" } }, + { url = "https://files.pythonhosted.org/packages/35/8a/3d130aeabcaf3d2466af76b7b141c08d9e89c9016ab4b7cdd0f7dc2d1c62/charset_normalizer-3.5.1-cp37-abi3-musllinux_1_2_s390x.whl", upload-time = 2026-08-15T08:20:09Z, size = 256917, hashes = { sha256 = "3588e376b3ea2eea84976f67273d679f229e24c66dce7b82ae45aef04ff6e072" } }, + { url = "https://files.pythonhosted.org/packages/80/c2/a7379b840292d0c1ab9fbd17d1f3967aa81794dc95bc74be8999d7fedcf7/charset_normalizer-3.5.1-cp37-abi3-musllinux_1_2_x86_64.whl", upload-time = 2026-08-15T08:20:10Z, size = 254846, hashes = { sha256 = "e199fb99720074809a7720f1c0b4d919eea8b87e88713e0f8f602f7bef543d9d" } }, + { url = "https://files.pythonhosted.org/packages/01/65/d43b714731bb2f40d4053dfa00ecfc1c5a301f8e3316c5db3a09af59fe94/charset_normalizer-3.5.1-cp37-abi3-win32.whl", upload-time = 2026-08-15T08:20:12Z, size = 174216, hashes = { sha256 = "dd732602a7009217f658d5863d12d79d373a4de0eebc111094bcdd3bb8e0a6cc" } }, + { url = "https://files.pythonhosted.org/packages/35/4f/b911ed898b26a09789eba9c9200c999aff6c61b4bafaf4838e56d1a1e1a3/charset_normalizer-3.5.1-cp37-abi3-win_amd64.whl", upload-time = 2026-08-15T08:20:13Z, size = 199764, hashes = { sha256 = "70055ff39b97c99e7ae40ea3e393fb62aa2e44dbd9b29f8d14f42fb0025c3959" } }, + { url = "https://files.pythonhosted.org/packages/f0/a7/920baf467bfd9bf689f3b318340f37aee4572a71f162bd8db51da55ba4fa/charset_normalizer-3.5.1-cp37-abi3-win_arm64.whl", upload-time = 2026-08-15T08:20:15Z, size = 287318, hashes = { sha256 = "87e4f41d375c0b9be2fb5251aee4b8a689169e134535aed81bf085c3b647451e" } }, + { url = "https://files.pythonhosted.org/packages/cc/61/d01fc49b8dea277640b55a9e15960dbca9fdc8c9fde18e572d39c59f4019/charset_normalizer-3.5.1-py3-none-any.whl", upload-time = 2026-08-15T08:20:43Z, size = 68658, hashes = { sha256 = "6df0ec430f9a831772c23ca5a224cba36517a58a84bb32c32bb59a9fa67c47f6" } }, ] [[packages]] @@ -161,124 +219,139 @@ wheels = [ [[packages]] name = "coverage" -version = "7.14.0" +version = "7.15.4" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/23/7f/d0720730a397a999ffc0fd3f5bebef347338e3a47b727da66fbb228e2ff2/coverage-7.14.0.tar.gz", upload-time = 2026-05-10T18:02:31Z, size = 919489, hashes = { sha256 = "057a6af2f160a85384cde4ab36f0d2777bae1057bae255f95413cdd382aa5c74" } } +sdist = { url = "https://files.pythonhosted.org/packages/be/c3/4f2195f512fb172aa425a8803a874b2baa9ba7f80ff7b6080998761fc701/coverage-7.15.4.tar.gz", upload-time = 2026-08-06T13:50:24Z, size = 936952, hashes = { sha256 = "0548198fff07ccf4faf469520bce1c2eceb1ce3e62891921138dec10907f9d00" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/59/9d/7c83ef51c3eb495f10010094e661833588b7709946da634c8b66520b97c7/coverage-7.14.0-cp310-cp310-macosx_10_9_x86_64.whl", upload-time = 2026-05-10T17:59:23Z, size = 219668, hashes = { sha256 = "84c32d90bf4537f0e7b4dec9aaa9a938fb8205136b9d2ecf4d7629d5262dc075" } }, - { url = "https://files.pythonhosted.org/packages/24/34/898546aefbd28f0af131201d0dc852c9e976f817bd7d5bfb8dc4e02863bb/coverage-7.14.0-cp310-cp310-macosx_11_0_arm64.whl", upload-time = 2026-05-10T17:59:26Z, size = 220192, hashes = { sha256 = "7c843572c605ab51cfdb5c6b5f2586e2a8467c0d28eca4bdef4ec70c5fecbd82" } }, - { url = "https://files.pythonhosted.org/packages/df/4a/b457c88aca72b0df13a98167ebd5d947135ccd9881ea88ce6a570e13aa9b/coverage-7.14.0-cp310-cp310-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-05-10T17:59:27Z, size = 246932, hashes = { sha256 = "0c451757d3fa2603354fdc789b5e58a0e327a117c370a40e3476ba4eabab228c" } }, - { url = "https://files.pythonhosted.org/packages/b5/d9/92600e89486fd074c50f0117422b2c9592c3e144e2f25bd5ac0bc62bc7a0/coverage-7.14.0-cp310-cp310-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-05-10T17:59:29Z, size = 248762, hashes = { sha256 = "3fd43f0616e765ab78d069cf8358def7363957a45cee446d65c502dcfeea7893" } }, - { url = "https://files.pythonhosted.org/packages/0d/e1/9ea1eb9c311da7f15853559dc1d9d82bef88ecd3e59fbeb51f16bc2ffa91/coverage-7.14.0-cp310-cp310-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-05-10T17:59:31Z, size = 250625, hashes = { sha256 = "731e535b1498b27d13594a0527a79b0510867b0ad891532be41cb883f2128e20" } }, - { url = "https://files.pythonhosted.org/packages/a5/03/57afca1b8106f8549a5329139315041fe166d6099bd9381346b9430dfbd1/coverage-7.14.0-cp310-cp310-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-05-10T17:59:32Z, size = 252539, hashes = { sha256 = "c7492f2d493b976941c7ca050f273cbda2f43c381124f7586a3e3c16d1804fec" } }, - { url = "https://files.pythonhosted.org/packages/57/5e/2e9fc63c9928119c1dbae02222be51407d3e7ebac5811ebbda4af3557795/coverage-7.14.0-cp310-cp310-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-05-10T17:59:34Z, size = 247636, hashes = { sha256 = "dc38367eaa2abb1b766ac333142bce7655335a73537f5c8b75aaa89c2b987757" } }, - { url = "https://files.pythonhosted.org/packages/f0/e2/0b7898cda21041cc67546e19b80ba66cbbb47cbece52a76a5904de6a3aaf/coverage-7.14.0-cp310-cp310-musllinux_1_2_aarch64.whl", upload-time = 2026-05-10T17:59:36Z, size = 248666, hashes = { sha256 = "0a951308cde22cf77f953955a754d04dccb57fe3bb8e345d685778ed9fc1632a" } }, - { url = "https://files.pythonhosted.org/packages/d6/e3/d33662a2fdaef23229c15921f39c84ec38441f3069ba26e134ed402c833b/coverage-7.14.0-cp310-cp310-musllinux_1_2_i686.whl", upload-time = 2026-05-10T17:59:38Z, size = 246670, hashes = { sha256 = "fab3877e4ebb06bd9d4d4d00ee53309ee5478e66873c66a382272e3ee33eb7ea" } }, - { url = "https://files.pythonhosted.org/packages/99/b2/533942c3bfbf6770b5c32d7f2ff029fe013dba31f3fe8b45cabbb250365e/coverage-7.14.0-cp310-cp310-musllinux_1_2_ppc64le.whl", upload-time = 2026-05-10T17:59:39Z, size = 250484, hashes = { sha256 = "b812eb847b19876ebf33fb6c4f11819af05ab6050b0bfa1bc53412ae81779adb" } }, - { url = "https://files.pythonhosted.org/packages/d8/00/15acbad83a96de13c73831486c7627bfed73dfaec53b04e4a6315edf3fd8/coverage-7.14.0-cp310-cp310-musllinux_1_2_riscv64.whl", upload-time = 2026-05-10T17:59:41Z, size = 246942, hashes = { sha256 = "d9c8ef6ed820c433de075657d72dda1f89a2984955e58b8a75feb3f184250218" } }, - { url = "https://files.pythonhosted.org/packages/70/db/cef0228de493f2c740c760a9057a61d00c6849480073b70a75b87c7d4bab/coverage-7.14.0-cp310-cp310-musllinux_1_2_x86_64.whl", upload-time = 2026-05-10T17:59:43Z, size = 247544, hashes = { sha256 = "d128b1bba9361fbaaf6a19e179e6cfd6a9103ce0c0555876f72780acc93efd85" } }, - { url = "https://files.pythonhosted.org/packages/77/a0/d9ef8e148f3025c2ae8401d77cda1502b6d2a4d8102603a8af31460aedb6/coverage-7.14.0-cp310-cp310-win32.whl", upload-time = 2026-05-10T17:59:44Z, size = 222285, hashes = { sha256 = "65f267ca1370726ec2c1aa38bbe4df9a71a740f22878d2d4bf59d71a4cd8d323" } }, - { url = "https://files.pythonhosted.org/packages/85/c0/30c454c7d3cf47b2805d4e06f12443f5eece8a5d030d3b0350e7b74ecb49/coverage-7.14.0-cp310-cp310-win_amd64.whl", upload-time = 2026-05-10T17:59:46Z, size = 223215, hashes = { sha256 = "b34ece8065914f938ed7f2c5872bb865336977a52919149846eac3744327267a" } }, - { url = "https://files.pythonhosted.org/packages/fc/e4/649c8d4f7f1709b6dbfc474358aa1bba02f67bcd52e2fec291a5014006cd/coverage-7.14.0-cp311-cp311-macosx_10_9_x86_64.whl", upload-time = 2026-05-10T17:59:48Z, size = 219795, hashes = { sha256 = "6a78e2a9d9c5e3b8d4ab9b9d28c985ea66fced0a7d7c2aec1f216e03a2011480" } }, - { url = "https://files.pythonhosted.org/packages/7f/8d/46692d24b3f395d4cbf17bfcc57136b4f2f9c0c0df864b0bddfc1d71a014/coverage-7.14.0-cp311-cp311-macosx_11_0_arm64.whl", upload-time = 2026-05-10T17:59:49Z, size = 220299, hashes = { sha256 = "a1816c505187592dcd1c5a5f226601a549f70365fbd00930ac88b0c225b76bb4" } }, - { url = "https://files.pythonhosted.org/packages/12/c2/a40f5cb295bbcbb697a76947a56081c494c61950366294ee426ffe261099/coverage-7.14.0-cp311-cp311-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-05-10T17:59:51Z, size = 250721, hashes = { sha256 = "d8e1762f0e9cbc26ec315471e7b47855218e833cd5a032d706fbf43845d878c7" } }, - { url = "https://files.pythonhosted.org/packages/fd/35/202235eb5c3c14c212462cd91d61b7386bf8fc44bc7a77f4742d2a69174b/coverage-7.14.0-cp311-cp311-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-05-10T17:59:53Z, size = 252633, hashes = { sha256 = "9336e23e8bb3a3925398261385e2a1533957d3e760e91070dcb0e98bfa514eed" } }, - { url = "https://files.pythonhosted.org/packages/bb/80/5f596e8995785124ee191c42535664c5e62c65995b66f4ca21e28ae04c81/coverage-7.14.0-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-05-10T17:59:55Z, size = 254743, hashes = { sha256 = "9cd1169b2230f9cbe9c638ba38022ed7a2b1e641cc07f7cea0365e4be2a74980" } }, - { url = "https://files.pythonhosted.org/packages/1e/6d/0d178825be2350f0adb27984d0aa7cf84bbdab201f6fb926b535d23a8f5f/coverage-7.14.0-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-05-10T17:59:56Z, size = 256700, hashes = { sha256 = "d1bb3543b58fea74d2cd1abc4054cc927e4724687cb4560cd2ed88d2c7d820c0" } }, - { url = "https://files.pythonhosted.org/packages/19/5b/9e549c2f6e9dfea472adadba06c294e64735dabc2dd19015fac082095013/coverage-7.14.0-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-05-10T17:59:57Z, size = 250854, hashes = { sha256 = "a93bac2cb577ef60074999ed56d8a1535894398e2ed920d4185c3ec0c8864742" } }, - { url = "https://files.pythonhosted.org/packages/3d/1c/b94f9f5f36396021ee2f62c5834b12e6a3d31f0bed5d6fc6d1c3caec087c/coverage-7.14.0-cp311-cp311-musllinux_1_2_aarch64.whl", upload-time = 2026-05-10T17:59:59Z, size = 252433, hashes = { sha256 = "5904abf7e18cddc463219b17552229650c6b79e061d31a1059283051169cf7d5" } }, - { url = "https://files.pythonhosted.org/packages/b5/cb/d192cd8e1345eccabc32016f2d39072ecd10cb4f4b983ed8d0ebdeaf00dc/coverage-7.14.0-cp311-cp311-musllinux_1_2_i686.whl", upload-time = 2026-05-10T18:00:01Z, size = 250494, hashes = { sha256 = "741f57cddc9004a8c81b084660215f33a6b597dbe62c31386b983ee26310e327" } }, - { url = "https://files.pythonhosted.org/packages/53/c5/aac9f460a41d835dbddef1d377f105f6ac2311d0f3c1588e9f51046d8813/coverage-7.14.0-cp311-cp311-musllinux_1_2_ppc64le.whl", upload-time = 2026-05-10T18:00:03Z, size = 254261, hashes = { sha256 = "664123feb0929d7affc135717dbd70d61d98688a08ab1e5ba464739620c6252d" } }, - { url = "https://files.pythonhosted.org/packages/23/aa/7af7c0081980a9cb3d289c5a435a4b7657dcecbd128e25c580e6a50389b5/coverage-7.14.0-cp311-cp311-musllinux_1_2_riscv64.whl", upload-time = 2026-05-10T18:00:05Z, size = 250216, hashes = { sha256 = "c83d2399a51bbec8429266905d33616f04bc5726b1138c35844d5fcd896b2e20" } }, - { url = "https://files.pythonhosted.org/packages/35/60/a4257538ce2f6b978aeb51870d6c4208c510928a03db7e0339bb625dccb7/coverage-7.14.0-cp311-cp311-musllinux_1_2_x86_64.whl", upload-time = 2026-05-10T18:00:06Z, size = 251125, hashes = { sha256 = "bcb2e855b87321259a037429288ae85216d191c74de3e79bf57cd2bc0761992c" } }, - { url = "https://files.pythonhosted.org/packages/a1/ab/f91af47642ec1aa53490e835a95847168d9c77fc39aa58527604c051e145/coverage-7.14.0-cp311-cp311-win32.whl", upload-time = 2026-05-10T18:00:08Z, size = 222300, hashes = { sha256 = "731dc15b385ac52289743d476245b61e1a2927e803bef655b52bc3b2a75a21f3" } }, - { url = "https://files.pythonhosted.org/packages/f0/f0/a71ddbd874431e7a7cd96071f0c331cfbbad07704833c765d24ffbab8a67/coverage-7.14.0-cp311-cp311-win_amd64.whl", upload-time = 2026-05-10T18:00:10Z, size = 223241, hashes = { sha256 = "bfb0ed8ec5d25e93face268115d7964db9df8b9aae8edcde9ec6b16c726a7cc1" } }, - { url = "https://files.pythonhosted.org/packages/d8/6e/d9d312a5151a96cd110efee32efc3fc97b01ebd86203fe618ccb29cf4c92/coverage-7.14.0-cp311-cp311-win_arm64.whl", upload-time = 2026-05-10T18:00:12Z, size = 221908, hashes = { sha256 = "7ebb1c6df9f78046a1b1e0a89674cd4bf73b7c648914eebcf976a57fd99a5627" } }, - { url = "https://files.pythonhosted.org/packages/09/1e/2f996b2c8415cbb6f54b0f5ec1ee850c96d7911961afb4fc05f4a89d8c58/coverage-7.14.0-cp312-cp312-macosx_10_13_x86_64.whl", upload-time = 2026-05-10T18:00:13Z, size = 219967, hashes = { sha256 = "7ffd19fc8aed057fd686a17a4935eef5f9859d69208f96310e893e64b9b6ccf5" } }, - { url = "https://files.pythonhosted.org/packages/34/23/35c7aea1274aef7525bdd2dc92f710bdde6d11652239d71d1ec450067939/coverage-7.14.0-cp312-cp312-macosx_11_0_arm64.whl", upload-time = 2026-05-10T18:00:15Z, size = 220329, hashes = { sha256 = "829994cfe1aeb773ca27bf246d4badc1e764893e3bfb98fff820fcecd1ca4662" } }, - { url = "https://files.pythonhosted.org/packages/75/cf/a8f4b43a16e194b0261257ad28ded5853ec052570afef4a84e1d81189f3b/coverage-7.14.0-cp312-cp312-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-05-10T18:00:17Z, size = 251839, hashes = { sha256 = "b4f07cf7edcb7ec39431a5074d7ea83b29a9f71fcfc494f0f40af4e65180420f" } }, - { url = "https://files.pythonhosted.org/packages/69/ff/6699e7b71e60d3049eb2bdcbc95ee3f35707b2b0e48f32e9e63d3ce30c08/coverage-7.14.0-cp312-cp312-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-05-10T18:00:18Z, size = 254576, hashes = { sha256 = "ca3d9cf2c32b521bd9518385608787fa86f38daf993695307531822c3430ed67" } }, - { url = "https://files.pythonhosted.org/packages/22/ec/c936d495fcd67f48f03a9c4ad3297ff80d1f222a5df3980f15b34c186c21/coverage-7.14.0-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-05-10T18:00:20Z, size = 255690, hashes = { sha256 = "92af52828e7f29d827346b0294e5a0853fa206db77db0395b282918d41e28db9" } }, - { url = "https://files.pythonhosted.org/packages/5c/42/5af63f636cc62a4a2b1b3ba9146f6ee6f53a35a50d5cefc54d5670f60999/coverage-7.14.0-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-05-10T18:00:22Z, size = 257949, hashes = { sha256 = "7b2bb6c9d7e769360d0f20a0f219603fd64f0c8f97de17ab25853261602be0fb" } }, - { url = "https://files.pythonhosted.org/packages/26/d3/a225317bd2012132a27e1176d51660b826f99bb975876463c44ea0d7ee5a/coverage-7.14.0-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-05-10T18:00:24Z, size = 252242, hashes = { sha256 = "1c9ed6ef99f88fb8c14aa8e2bf8eb0fe55fa2edfea68f8675d78741df1a5ac0e" } }, - { url = "https://files.pythonhosted.org/packages/f1/7f/9e65495298c3ea414742998539c37d048b5e81cc818fb1828cc6b51d10bf/coverage-7.14.0-cp312-cp312-musllinux_1_2_aarch64.whl", upload-time = 2026-05-10T18:00:25Z, size = 253608, hashes = { sha256 = "8231ade007f37959fbf58acc677f26b922c02eda6f0428ea307da0fd39681bf3" } }, - { url = "https://files.pythonhosted.org/packages/94/46/1522b524a35bdad22b2b8c4f9d32d0a104b524726ec380b2db68db1746f5/coverage-7.14.0-cp312-cp312-musllinux_1_2_i686.whl", upload-time = 2026-05-10T18:00:27Z, size = 251753, hashes = { sha256 = "d8b013632cc1ce1d09dbe4f32667b4d320ec2f54fc326ebeffcd0b0bcc2bb6c4" } }, - { url = "https://files.pythonhosted.org/packages/f3/e9/cdf00d38817742c541ade405e115a3f7bf36e6f2a8b99d4f209861b85a2d/coverage-7.14.0-cp312-cp312-musllinux_1_2_ppc64le.whl", upload-time = 2026-05-10T18:00:29Z, size = 255823, hashes = { sha256 = "1733198802d71ec4c524f322e2867ee05c62e9e75df86bdca545407a221827d1" } }, - { url = "https://files.pythonhosted.org/packages/38/fc/5e7877cf5f902d08a17ff1c532511476d87e1bea355bd5028cb97f902e79/coverage-7.14.0-cp312-cp312-musllinux_1_2_riscv64.whl", upload-time = 2026-05-10T18:00:30Z, size = 251323, hashes = { sha256 = "72a305291fa8ee01332f1aaf38b348ca34097f6aa0b0ef627eef2837e57bbba5" } }, - { url = "https://files.pythonhosted.org/packages/18/9d/50f05a72dff8487464fdd4178dda5daed642a060e60afb644e3d45123559/coverage-7.14.0-cp312-cp312-musllinux_1_2_x86_64.whl", upload-time = 2026-05-10T18:00:32Z, size = 253197, hashes = { sha256 = "fcaba850dd317c65423a9d63d88f9573c53b00354d6dd95724576cc98a131595" } }, - { url = "https://files.pythonhosted.org/packages/00/3f/6f61ffe6439df266c3cf60f5c99cfaa21103d0210d706a42fc6c30683ff8/coverage-7.14.0-cp312-cp312-win32.whl", upload-time = 2026-05-10T18:00:33Z, size = 222515, hashes = { sha256 = "5ac83957a80d0701310e96d8bec68cdcf4f90a7674b7d13f15a344315b41ab27" } }, - { url = "https://files.pythonhosted.org/packages/85/19/93853133df2cb371083285ef6a93982a0173e7a233b0f61373ba9fd30eb2/coverage-7.14.0-cp312-cp312-win_amd64.whl", upload-time = 2026-05-10T18:00:35Z, size = 223324, hashes = { sha256 = "70390b0da32cb90b501953716302906e8bcce087cb283e70d8c97729f22e92b2" } }, - { url = "https://files.pythonhosted.org/packages/74/18/9f7fe62f659f24b7a82a0be56bf94c1bd0a89e0ae7ab4c668f6e82404294/coverage-7.14.0-cp312-cp312-win_arm64.whl", upload-time = 2026-05-10T18:00:37Z, size = 221944, hashes = { sha256 = "91b993743d959b8be85b4abf9d5478216a69329c321efe5be0433c1a841d691d" } }, - { url = "https://files.pythonhosted.org/packages/6b/76/b7c66ee3c66e1b0f9d894c8125983aa0c03fb2336f2fd16559f9c966157f/coverage-7.14.0-cp313-cp313-macosx_10_13_x86_64.whl", upload-time = 2026-05-10T18:00:38Z, size = 219990, hashes = { sha256 = "f2bbb8254370eb4c628ff3d6fa8a7f74ddc40565394d4f7ab791d1fe568e37ef" } }, - { url = "https://files.pythonhosted.org/packages/b3/af/e567cbad5ba69c013a50146dfa886dc7193361fda77521f51274ff620e1b/coverage-7.14.0-cp313-cp313-macosx_11_0_arm64.whl", upload-time = 2026-05-10T18:00:40Z, size = 220365, hashes = { sha256 = "23b81107f46d3f21d0cbce30664fcec0f5d9f585638a67081750f99738f6bf66" } }, - { url = "https://files.pythonhosted.org/packages/44/6f/9ad575d505b4d805b254febc8a5b338a2efe278f8786e56ff1cb8413f9c3/coverage-7.14.0-cp313-cp313-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-05-10T18:00:42Z, size = 251363, hashes = { sha256 = "22a7e06a5f11a757cdfe79018e9095f9f69ae283c5cd8123774c788deec8717b" } }, - { url = "https://files.pythonhosted.org/packages/6f/5f/b5370068b2f57787454592ed7dcd1002f0f1703b7db1fa30f6a325a4ca6e/coverage-7.14.0-cp313-cp313-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-05-10T18:00:44Z, size = 253961, hashes = { sha256 = "9d1aa57a1dc8e05bdc42e81c5d671d849577aeedf279f4c449d6d286f9ed88ca" } }, - { url = "https://files.pythonhosted.org/packages/29/1e/51adf17738976e8f2b85ddef7b7aa12a0838b056c92f175941d8862767c1/coverage-7.14.0-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-05-10T18:00:45Z, size = 255193, hashes = { sha256 = "90c1a51bcfddf645b3bb7ec333d9e94393a8e94f55642380fa8a9a5a9e636cb7" } }, - { url = "https://files.pythonhosted.org/packages/9e/7b/5bfd7ac1df3b881c2ac7a5cbc99c7609e6296c402f5ef587cd81c6f355b3/coverage-7.14.0-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-05-10T18:00:47Z, size = 257326, hashes = { sha256 = "a841fae2fadcae4f438d43b6ccc4aac2ad609f47cdb6cfdce60cbb3fe5ca7bc2" } }, - { url = "https://files.pythonhosted.org/packages/7d/38/1d37d316b174fad3843a1d76dbdfe4398771c9ecd0515935dd9ece9cd627/coverage-7.14.0-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-05-10T18:00:49Z, size = 251582, hashes = { sha256 = "c79d2319cabef1fe8e86df73371126931550804738f78ad7d31e3aad85a67367" } }, - { url = "https://files.pythonhosted.org/packages/34/46/746704f95980ba220214e1a41e18cec5aea80a898eaa53c51bf2d645ff36/coverage-7.14.0-cp313-cp313-musllinux_1_2_aarch64.whl", upload-time = 2026-05-10T18:00:51Z, size = 253325, hashes = { sha256 = "1b23b0c6f0b1db6ad769b7050c8b641c0bf215ded26c1816955b17b7f26edfa9" } }, - { url = "https://files.pythonhosted.org/packages/e1/b9/bbe87206d9687b192352f893797825b5f5b15ecd3aa9c68fbff0c074d77b/coverage-7.14.0-cp313-cp313-musllinux_1_2_i686.whl", upload-time = 2026-05-10T18:00:52Z, size = 251291, hashes = { sha256 = "55d3089079ce181a4566b1065ab28d2575eb76d8ac8f81f4fcda2bf037fee087" } }, - { url = "https://files.pythonhosted.org/packages/46/57/b8cdb12ac0d73ef0243218bd5e22c9df8f92edab8018213a86aec67c5324/coverage-7.14.0-cp313-cp313-musllinux_1_2_ppc64le.whl", upload-time = 2026-05-10T18:00:54Z, size = 255448, hashes = { sha256 = "49c005cba1e2f9677fb2845dcdf9a2e72a52a17d63e8231aaaae35d9f50215ef" } }, - { url = "https://files.pythonhosted.org/packages/1f/d4/5002019538b2036ce3c84340f54d2fd5100d55b0a6b0894eee56128d03c7/coverage-7.14.0-cp313-cp313-musllinux_1_2_riscv64.whl", upload-time = 2026-05-10T18:00:56Z, size = 251110, hashes = { sha256 = "9117377b823daa28aa8635fbb08cda1cd6be3d7143257345459559aeef852d52" } }, - { url = "https://files.pythonhosted.org/packages/37/53/20c5009477660f084e6ed60bc02a91894b8e234e617e86ecfd9aaf78e27b/coverage-7.14.0-cp313-cp313-musllinux_1_2_x86_64.whl", upload-time = 2026-05-10T18:00:57Z, size = 252885, hashes = { sha256 = "7b79d646cf46d5cf9a9f40281d4441df5849e445726e369006d2b117710b33fe" } }, - { url = "https://files.pythonhosted.org/packages/ae/ab/3cf6427ac9c1f1db747dbb1ce71dde47984876d4c2cfd018a3fef0a78d4d/coverage-7.14.0-cp313-cp313-win32.whl", upload-time = 2026-05-10T18:00:59Z, size = 222539, hashes = { sha256 = "fb609b3658479e33f9516d46f1a89dbb9b6c261366e3a11844a96ec487533dae" } }, - { url = "https://files.pythonhosted.org/packages/8f/b8/9228523e80321c2cb4880d1f589bc0171f2f71432c35118ad04dc01decce/coverage-7.14.0-cp313-cp313-win_amd64.whl", upload-time = 2026-05-10T18:01:01Z, size = 223344, hashes = { sha256 = "0773d8329cf32b6fd222e4b52622c61fe8d503eb966cfc8d3c3c10c96266d50e" } }, - { url = "https://files.pythonhosted.org/packages/a3/99/118daa192f95e3a6cb2740100fbf8797cda1734b4134ef0b5d501a7fa8f3/coverage-7.14.0-cp313-cp313-win_arm64.whl", upload-time = 2026-05-10T18:01:03Z, size = 221966, hashes = { sha256 = "b4e26a0f1b696faf283bffe5b8569e44e336c582439df5d53281ab89ee0cba96" } }, - { url = "https://files.pythonhosted.org/packages/e6/f1/a46cc0c013be170216253184a32366d7cbdb9252feaec866b05c2d12a894/coverage-7.14.0-cp313-cp313t-macosx_10_13_x86_64.whl", upload-time = 2026-05-10T18:01:05Z, size = 220679, hashes = { sha256 = "953f521ca9445300397e65fda3dca58b2dbd68fee983777420b57ac3c77e9f90" } }, - { url = "https://files.pythonhosted.org/packages/64/8c/9c30a3d311a34177fa432995be7fbfc64477d8bac5630bd38055b1c9b424/coverage-7.14.0-cp313-cp313t-macosx_11_0_arm64.whl", upload-time = 2026-05-10T18:01:07Z, size = 221033, hashes = { sha256 = "98af83fd65ae24b1fdd03aaead967a9f523bcd2f1aab2d4f3ffda65bb568a6f1" } }, - { url = "https://files.pythonhosted.org/packages/9a/cd/3fb5e06c3badefd0c1b47e2044fdca67f8220a4ec2e7fcfb476aa0a67c6c/coverage-7.14.0-cp313-cp313t-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-05-10T18:01:08Z, size = 262333, hashes = { sha256 = "668b92e6958c4db7cf92e81caac328dfbbdbb215db2850ad28f0cbe1eea0bfbd" } }, - { url = "https://files.pythonhosted.org/packages/a8/e6/fbc322325c7294d3e22c1ad6b79e45d0806b25228c8e5842aed6d8169aa7/coverage-7.14.0-cp313-cp313t-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-05-10T18:01:10Z, size = 264410, hashes = { sha256 = "9fbd898551762dea00d3fef2b1c4f99afd2c6a3ff952ea07d60a9bd5ed4f34bc" } }, - { url = "https://files.pythonhosted.org/packages/08/92/c497b264bec1673c47cc77e26f760fcda4654cabf1f39546d1a23a3b8c35/coverage-7.14.0-cp313-cp313t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-05-10T18:01:12Z, size = 266836, hashes = { sha256 = "68af363c07ecd8d4b7d4043d85cb376d7d227eceb54e5323ee45da73dbd3e426" } }, - { url = "https://files.pythonhosted.org/packages/78/fc/045da320987f401af5d2815d351e8aa799aec859f60e29f445e3089eeedb/coverage-7.14.0-cp313-cp313t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-05-10T18:01:13Z, size = 267974, hashes = { sha256 = "6e57054a583da8ac55edf24117ea4c9133032cfc4cf72aa2d48c1e5d4b52f899" } }, - { url = "https://files.pythonhosted.org/packages/1b/ae/227b1e379497fb7a4fc3286e620f80c8a1e7cec66d45695a01639eb1af65/coverage-7.14.0-cp313-cp313t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-05-10T18:01:15Z, size = 261578, hashes = { sha256 = "cc3499459bbcdd51a65b64c35ab7ed2764eaf3cba826e0df3f1d7fe2e102b70b" } }, - { url = "https://files.pythonhosted.org/packages/a0/f5/3570342900f2acea31d33ff1590c5d8bac1a8e1a2e1c6d34a5d5e61de681/coverage-7.14.0-cp313-cp313t-musllinux_1_2_aarch64.whl", upload-time = 2026-05-10T18:01:17Z, size = 264394, hashes = { sha256 = "45899ec2138a4346ed34d601dedf5076fb74edf2d1dd9dc76a78e82397edee90" } }, - { url = "https://files.pythonhosted.org/packages/16/29/de1bbc01c935b28f89b1dc3db85b011c055e843a8e5e3b83141c3f80af7f/coverage-7.14.0-cp313-cp313t-musllinux_1_2_i686.whl", upload-time = 2026-05-10T18:01:19Z, size = 262022, hashes = { sha256 = "8767486808c436f05b23ab98eb963fb29185e32a9357a166971685cb3459900f" } }, - { url = "https://files.pythonhosted.org/packages/35/95/f53890b0bf2fc10ab168e05d38869215e73ca24c4cb521c3bb0eb62fe16b/coverage-7.14.0-cp313-cp313t-musllinux_1_2_ppc64le.whl", upload-time = 2026-05-10T18:01:21Z, size = 265732, hashes = { sha256 = "a3b5ddfd6aa7ddad53ee3edb231e88a2151507a43229b7d71b953916deca127d" } }, - { url = "https://files.pythonhosted.org/packages/ed/ea/c919e259081dd2bdf0e43b87209709ba7ec2e4117c2a7f5185379c43463c/coverage-7.14.0-cp313-cp313t-musllinux_1_2_riscv64.whl", upload-time = 2026-05-10T18:01:23Z, size = 260921, hashes = { sha256 = "63df0fe568e698e1045792399f8ab6da3a6c2dce3182813fb92afa2641087b47" } }, - { url = "https://files.pythonhosted.org/packages/1a/2c/c2831889705a81dc5d1c6ca12e4d8e9b95dfc146d153488a6c0ea685d28e/coverage-7.14.0-cp313-cp313t-musllinux_1_2_x86_64.whl", upload-time = 2026-05-10T18:01:25Z, size = 263109, hashes = { sha256 = "827d6397dbd95144939b18f89edf31f63e1f99633e8d5f32f22ba8bdda567477" } }, - { url = "https://files.pythonhosted.org/packages/5a/a9/2fcae5003cac3d63fe344d2166243c2756935f48420863c5272b240d550b/coverage-7.14.0-cp313-cp313t-win32.whl", upload-time = 2026-05-10T18:01:27Z, size = 223212, hashes = { sha256 = "7bf43e000d24012599b879791cff41589af90674722421ef11b11a5431920bab" } }, - { url = "https://files.pythonhosted.org/packages/3f/bb/18e94d7b14b9b398164197114a587a04ab7c9fdbe1d237eef57311c5e883/coverage-7.14.0-cp313-cp313t-win_amd64.whl", upload-time = 2026-05-10T18:01:29Z, size = 224272, hashes = { sha256 = "3f5549365af25d770e06b1f8f5682d9a5637d06eb494db91c6fa75d3950cc917" } }, - { url = "https://files.pythonhosted.org/packages/db/56/4f14fad782b035c81c4ffd09159e7103d42bb1d93ac8496d04b90a11b7da/coverage-7.14.0-cp313-cp313t-win_arm64.whl", upload-time = 2026-05-10T18:01:31Z, size = 222530, hashes = { sha256 = "6d160217ec6fe890f16ad3a9531761589443749e448f91986c972714fad361c8" } }, - { url = "https://files.pythonhosted.org/packages/1c/18/b9a6586d73992807c26f9a5f274131be3d76b56b18a82b9392e2a25d2e45/coverage-7.14.0-cp314-cp314-macosx_10_15_x86_64.whl", upload-time = 2026-05-10T18:01:33Z, size = 220036, hashes = { sha256 = "9aed9fa983514ca032790f3fe0d1c0e42ca7e16b42432af1706b50a9a46bef5d" } }, - { url = "https://files.pythonhosted.org/packages/f3/9b/4165a1d56ddc302a0e2d518fd9d412a4fd0b57562618c78c5f21c57194f5/coverage-7.14.0-cp314-cp314-macosx_11_0_arm64.whl", upload-time = 2026-05-10T18:01:34Z, size = 220368, hashes = { sha256 = "ba3b8390db29296dbbf49e91b6fe08f990743a90c8f447ba4c2ffc29670dfa63" } }, - { url = "https://files.pythonhosted.org/packages/69/aa/c12e52a5ba148d9995229d557e3be6e554fe469addc0e9241b2f0956d8ea/coverage-7.14.0-cp314-cp314-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-05-10T18:01:36Z, size = 251417, hashes = { sha256 = "3a5d8e876dfa2f102e970b183863d6dedd023d3c0eeca1fe7a9787bc5f28b212" } }, - { url = "https://files.pythonhosted.org/packages/d7/51/ec641c26e6dca1b25a7d2035ba6ecb7c884ef1a100a9e42fbe4ce4405139/coverage-7.14.0-cp314-cp314-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-05-10T18:01:38Z, size = 253924, hashes = { sha256 = "5ebb8f4614a3787d567e610bbfdf96a4798dd69a1afb1bd8ad228d4111fe6ff3" } }, - { url = "https://files.pythonhosted.org/packages/33/c4/59c3de0bd1b538824173fd518fed51c1ce740ca5ed68e74545983f4053a9/coverage-7.14.0-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-05-10T18:01:40Z, size = 255269, hashes = { sha256 = "6b9bf47223dd8db3d4c4b2e443b02bace480d428f0822c3f991600448a176c97" } }, - { url = "https://files.pythonhosted.org/packages/7b/a9/36dfa153a62040296f6e7febfdb20a5720622f6ef5a81a41e8237b9a5344/coverage-7.14.0-cp314-cp314-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-05-10T18:01:42Z, size = 257583, hashes = { sha256 = "3485a836550b303d006d57cc06e3d5afaabc642c77050b7c985a97b13e3776b8" } }, - { url = "https://files.pythonhosted.org/packages/26/7b/cc2c048d4114d9ab1c2409e9ee365e5ae10736df6dffcfc9444effa6c708/coverage-7.14.0-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-05-10T18:01:44Z, size = 251434, hashes = { sha256 = "3e7e88110bae996d199d1693ca8ec3fd52441d426401ae963437598667b4c5eb" } }, - { url = "https://files.pythonhosted.org/packages/ee/df/6770eaa576e604575e9a78055313250faef5faa84bd6f71a39fece519c43/coverage-7.14.0-cp314-cp314-musllinux_1_2_aarch64.whl", upload-time = 2026-05-10T18:01:46Z, size = 253280, hashes = { sha256 = "15228a6800ce7bdf1b74800595e56db7138cecb338fdbf044806e10dcf182dfe" } }, - { url = "https://files.pythonhosted.org/packages/ad/9e/1c0264514a3f98259a6d64765a397b2c8373e3ba59ee722a4802d3ec0c61/coverage-7.14.0-cp314-cp314-musllinux_1_2_i686.whl", upload-time = 2026-05-10T18:01:48Z, size = 251241, hashes = { sha256 = "9d26ac7f5398bafc5b57421ad994e8a4749e8a7a0e62d05ec7d53014d5963bfa" } }, - { url = "https://files.pythonhosted.org/packages/64/16/4efdf3e3c4079cdbf0ece56a2fea872df9e8a3e15a13a0af4400e1075944/coverage-7.14.0-cp314-cp314-musllinux_1_2_ppc64le.whl", upload-time = 2026-05-10T18:01:50Z, size = 255516, hashes = { sha256 = "2fb73254ff43c911c967a899e1359bc5049b4b115d6e8fbdde4937d0a2246cd5" } }, - { url = "https://files.pythonhosted.org/packages/93/69/b1de96346603881b3d1bc8d6447c83200e1c9700ffbaff926ba01ff5724c/coverage-7.14.0-cp314-cp314-musllinux_1_2_riscv64.whl", upload-time = 2026-05-10T18:01:52Z, size = 251059, hashes = { sha256 = "454a380af72c6adada298ed270d38c7a391288198dbfb8467f786f588751a90c" } }, - { url = "https://files.pythonhosted.org/packages/a4/66/2881853e0363a5e0a724d1103e53650795367471b6afb234f8b49e713bc6/coverage-7.14.0-cp314-cp314-musllinux_1_2_x86_64.whl", upload-time = 2026-05-10T18:01:54Z, size = 252716, hashes = { sha256 = "65c86fb646d2bd2972e96bd1a8b45817ed907cee68655d6295fe7ec031d04cca" } }, - { url = "https://files.pythonhosted.org/packages/55/5c/0d3305d002c41dcde873dbe456491e663dc55152ca526b630b5c47efd62f/coverage-7.14.0-cp314-cp314-win32.whl", upload-time = 2026-05-10T18:01:56Z, size = 222788, hashes = { sha256 = "6a6516b02a6101398e19a3f44820f69bab2590697f7def4331f668b14adaf828" } }, - { url = "https://files.pythonhosted.org/packages/f9/58/6e1b8f52fdc3184b47dc5037f5070d83a3d11042db1594b02d2a44d786c8/coverage-7.14.0-cp314-cp314-win_amd64.whl", upload-time = 2026-05-10T18:01:58Z, size = 223600, hashes = { sha256 = "45e0f79d8351fa76e256716df91eab12890d32678b9590df7ae1042e4bd4cf5d" } }, - { url = "https://files.pythonhosted.org/packages/00/70/a18c408e674bc26281cadaedc7351f929bd2094e191e4b15271c30b084cc/coverage-7.14.0-cp314-cp314-win_arm64.whl", upload-time = 2026-05-10T18:02:00Z, size = 222168, hashes = { sha256 = "4b899594a8b2d81e5cc064a0d7f9cac2081fed91049456cae7676787e41549c9" } }, - { url = "https://files.pythonhosted.org/packages/3d/89/2681f071d238b62aff8dfc2ab44fc24cfdb38d1c01f391a80522ff5d3a16/coverage-7.14.0-cp314-cp314t-macosx_10_15_x86_64.whl", upload-time = 2026-05-10T18:02:02Z, size = 220766, hashes = { sha256 = "f580f8c80acd94ac72e863efe2cab791d8c38d153e0b463b92dfa000d5c84cd1" } }, - { url = "https://files.pythonhosted.org/packages/bd/c7/c987babafd9207ffa1995e1ef1f9b26762cf4963aa768a66b6f0501e4616/coverage-7.14.0-cp314-cp314t-macosx_11_0_arm64.whl", upload-time = 2026-05-10T18:02:04Z, size = 221035, hashes = { sha256 = "a2bd259c442cd43c49b30fbafc51776eb19ea396faf159d26a83e6a0a5f13b0c" } }, - { url = "https://files.pythonhosted.org/packages/5a/e9/d6a5ac3b333088143d6fc877d398a9a674dc03124a2f776e131f03864823/coverage-7.14.0-cp314-cp314t-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-05-10T18:02:05Z, size = 262405, hashes = { sha256 = "a706b908dfa85538863504c624b237a3cc34232bf403c057414ebfdb3b4d9f84" } }, - { url = "https://files.pythonhosted.org/packages/38/b1/e70838d29a7c08e22d44398a46db90815bbcbf28de06992bd9210d1a8d8e/coverage-7.14.0-cp314-cp314t-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-05-10T18:02:07Z, size = 264530, hashes = { sha256 = "7333cd944ee4393b9b3d3c1b598c936d4fc8d70573a4c7dacfec5590dd50e436" } }, - { url = "https://files.pythonhosted.org/packages/6b/73/5c31ef97763288d03d9995152b96d5475b527c63d91c84b01caea894b83a/coverage-7.14.0-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-05-10T18:02:09Z, size = 266932, hashes = { sha256 = "0f162bc9a15b82d947b02651b0c7e1609d6f7a8735ca330cfadec8481dd97d5a" } }, - { url = "https://files.pythonhosted.org/packages/e1/76/dd56d80f29c5f05b4d76f7e7c6d47cafacae017189c75c5759d24f9ff0cc/coverage-7.14.0-cp314-cp314t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-05-10T18:02:11Z, size = 268062, hashes = { sha256 = "362cb78e01a5dc82009d88004cf60f2e6b6d6fcbfdec05b05af73b0abf40118f" } }, - { url = "https://files.pythonhosted.org/packages/6e/c7/27ba85cd5b95614f159ff93ebff1901584a8d192e2e5e24c4943a7453f59/coverage-7.14.0-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-05-10T18:02:13Z, size = 261504, hashes = { sha256 = "acebd068fca5512c3a6fde9c045f901613478781a73f0e82b307b214daef23fb" } }, - { url = "https://files.pythonhosted.org/packages/13/2e/e8149f60ab5d5684c6eee881bdf34b127115cddbb958b196768dd9d63473/coverage-7.14.0-cp314-cp314t-musllinux_1_2_aarch64.whl", upload-time = 2026-05-10T18:02:15Z, size = 264398, hashes = { sha256 = "29fe3da551dface75deb2ccbf87b6b66e2e7ef38f6d89050b428be94afff3490" } }, - { url = "https://files.pythonhosted.org/packages/d9/7f/1261b025285323225f4b4abffa5a643649dfd67e25ddca7ebcbdea3b7cb3/coverage-7.14.0-cp314-cp314t-musllinux_1_2_i686.whl", upload-time = 2026-05-10T18:02:16Z, size = 262000, hashes = { sha256 = "b4cc4fce8672fffcb09b0eafc167b396b3ba53c4a7230f54b7aaffbf6c835fa9" } }, - { url = "https://files.pythonhosted.org/packages/d3/dc/829c54f60b9d08389439c00f813c752781c496fc5788c78d8006db4b4f2b/coverage-7.14.0-cp314-cp314t-musllinux_1_2_ppc64le.whl", upload-time = 2026-05-10T18:02:18Z, size = 265732, hashes = { sha256 = "5d4a51aad8ba8bdcd2b8bd8f03d4aca19693fa2327a3470e4718a25b03481020" } }, - { url = "https://files.pythonhosted.org/packages/ed/b0/70bd1419941652fa062689cba9c3eeafb8f5e6fbb890bce41c3bdda5dbd6/coverage-7.14.0-cp314-cp314t-musllinux_1_2_riscv64.whl", upload-time = 2026-05-10T18:02:20Z, size = 260847, hashes = { sha256 = "9f323af3e1e4f68b60b7b247e37b8515563a61375518fa59de1af48ba28a3db6" } }, - { url = "https://files.pythonhosted.org/packages/f2/73/be40b2390656c654d35ea0015ea7ba3d945769cf80790ad5e0bb2d56d2ba/coverage-7.14.0-cp314-cp314t-musllinux_1_2_x86_64.whl", upload-time = 2026-05-10T18:02:22Z, size = 263166, hashes = { sha256 = "1a0abc7342ea9711c469dd8b821c6c311e6bc6aac1442e5fbd6b27fae0a8f3db" } }, - { url = "https://files.pythonhosted.org/packages/29/55/4a643f712fcf7cf2881f8ec1e0ccb7b164aff3108f69b51801246c8799f2/coverage-7.14.0-cp314-cp314t-win32.whl", upload-time = 2026-05-10T18:02:24Z, size = 223573, hashes = { sha256 = "a9f864ef57b7172e2db87a096642dd51e179e085ab6b2c371c29e885f65c8fb2" } }, - { url = "https://files.pythonhosted.org/packages/27/96/3acae5da0953be042c0b4dea6d6789d2f080701c77b88e44d5bd41b9219b/coverage-7.14.0-cp314-cp314t-win_amd64.whl", upload-time = 2026-05-10T18:02:25Z, size = 224680, hashes = { sha256 = "29943e552fdc08e082eb51400fb2f58e118a83b5542bd06531214e084399b644" } }, - { url = "https://files.pythonhosted.org/packages/93/3d/6ab5d2dd8325d838737c6f8d83d62eb6230e0d70b87b51b57bbfd08fa767/coverage-7.14.0-cp314-cp314t-win_arm64.whl", upload-time = 2026-05-10T18:02:27Z, size = 222703, hashes = { sha256 = "742a73ea621953b012f2c4c2219b512180dd84489acf5b1596b0aafc55b9100b" } }, - { url = "https://files.pythonhosted.org/packages/61/e8/cb8e80d6f9f55b99588625062822bf946cf03ed06315df4bd8397f5632a1/coverage-7.14.0-py3-none-any.whl", upload-time = 2026-05-10T18:02:29Z, size = 211764, hashes = { sha256 = "8de5b61163aee3d05c8a2beab6f47913df7981dad1baf82c414d99158c286ab1" } }, + { url = "https://files.pythonhosted.org/packages/30/70/b052a519a584663a7bd052841a2debe11c8309ec49a7786340003f9c0a02/coverage-7.15.4-cp310-cp310-macosx_10_9_x86_64.whl", upload-time = 2026-08-06T13:46:55Z, size = 222245, hashes = { sha256 = "d0be6daac4cce6b8c8dc65886bae1b082ddbca4da8e5cbb5e15166acf253e264" } }, + { url = "https://files.pythonhosted.org/packages/67/39/892fa511aba3d1c3c8f49509a0ff5c71eab9f9f88d08e1a38da395821660/coverage-7.15.4-cp310-cp310-macosx_11_0_arm64.whl", upload-time = 2026-08-06T13:46:57Z, size = 222762, hashes = { sha256 = "b24e078eabcd6a9caa8b0713f9bc1eeb310bcc960a29d45a3b4fcd4b16d5b11d" } }, + { url = "https://files.pythonhosted.org/packages/9f/95/b2c724ce1e64bc23cb5b1d7eeffa9548dc3d811f7a6297b2d01607f4e062/coverage-7.15.4-cp310-cp310-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-08-06T13:46:59Z, size = 249498, hashes = { sha256 = "cfe20cc8cf8821d4fe54f89106cbf06aa27f37b5bbe3535568065a81539b4150" } }, + { url = "https://files.pythonhosted.org/packages/0b/4f/b1973f67a1382af65b572a31ed692f8e490a6ad707191eab59148376832a/coverage-7.15.4-cp310-cp310-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-08-06T13:47:00Z, size = 251328, hashes = { sha256 = "83cf06cdd687677742caff1a9134833b7a8b75f111519d2cb0e0ba1b9a851e15" } }, + { url = "https://files.pythonhosted.org/packages/a2/09/03efa6722a132abcac91b32a60b64b240dd707c189c64eee697e48992c96/coverage-7.15.4-cp310-cp310-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-06T13:47:01Z, size = 253194, hashes = { sha256 = "8fa4de68e2a752468ff14b4e15db7def689a71be759e826a31ccecbef69c5fd0" } }, + { url = "https://files.pythonhosted.org/packages/45/63/8299201d9c80fb65551ce99c966cab83d706ec4066ac999bef08201346de/coverage-7.15.4-cp310-cp310-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-06T13:47:03Z, size = 255106, hashes = { sha256 = "4dff9daa47d83120c3ec38ce921214242944a832aa04e903e50b5b7ebac8972d" } }, + { url = "https://files.pythonhosted.org/packages/ee/16/26fd8a691eb8d9a230128685f6d23309d7402cb030aa553001788c8c50fc/coverage-7.15.4-cp310-cp310-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-06T13:47:04Z, size = 250177, hashes = { sha256 = "a093fd37229918976f602aa07aa59e0973cde82186f220c8e197f721f5be0ce4" } }, + { url = "https://files.pythonhosted.org/packages/ad/ef/3c7556f33783a0a566e01443ca62bd8eb2cdfe22d271efdc02e08beb5654/coverage-7.15.4-cp310-cp310-musllinux_1_2_aarch64.whl", upload-time = 2026-08-06T13:47:06Z, size = 251234, hashes = { sha256 = "317db01a2cb02552fd67e2b1cca77a4b528a2a277176c5e0bf2cecbb639d3f54" } }, + { url = "https://files.pythonhosted.org/packages/29/49/640a34043edac950738f36a3567832db5731d4cb2ed84b59cdb89c6bccbf/coverage-7.15.4-cp310-cp310-musllinux_1_2_i686.whl", upload-time = 2026-08-06T13:47:07Z, size = 249237, hashes = { sha256 = "8ee3838dcb656602c3b51e16aed9bfb0822f8d8d6d1c5966d32ec8c104be8e20" } }, + { url = "https://files.pythonhosted.org/packages/48/f5/e80f212669dd1be954ff844f883ef11a437ef4fd0089c6e0effc7b66b15d/coverage-7.15.4-cp310-cp310-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-06T13:47:08Z, size = 253050, hashes = { sha256 = "425920379052ff1fe465268f3361d35804a241bbdd5a1b592c8cb60df4c52325" } }, + { url = "https://files.pythonhosted.org/packages/c7/e9/e5da0fe39f7fde1bca9edc09c60921bb5fdba4cec7db5bbad41ddfd8c230/coverage-7.15.4-cp310-cp310-musllinux_1_2_riscv64.whl", upload-time = 2026-08-06T13:47:10Z, size = 249508, hashes = { sha256 = "69bb2400abef928e365ea7d4d9925169ada78ed2295546780002d4b65de3df88" } }, + { url = "https://files.pythonhosted.org/packages/7d/38/41bf25774a0c8bba6b467f917cb1c9a0a2605e02dc93aad489fc7050ed59/coverage-7.15.4-cp310-cp310-musllinux_1_2_x86_64.whl", upload-time = 2026-08-06T13:47:11Z, size = 250110, hashes = { sha256 = "81661f82d302484e3119e7c80c519c02fa9bcc2a6b339baf67d67bc89c580f04" } }, + { url = "https://files.pythonhosted.org/packages/89/6e/26f2e54b79acc29d179ee4272922625aedb69198c4eb61f7ff4f098f3c78/coverage-7.15.4-cp310-cp310-win32.whl", upload-time = 2026-08-06T13:47:12Z, size = 224294, hashes = { sha256 = "cb476b2e828ecb71cb6b6a928d23fd20a7ddb501188022dae1c37499149cc338" } }, + { url = "https://files.pythonhosted.org/packages/7b/06/9a318fc3ae040d4d6cb2d86101c6aa963fab20899a5c58666adf52cde0ca/coverage-7.15.4-cp310-cp310-win_amd64.whl", upload-time = 2026-08-06T13:47:14Z, size = 224919, hashes = { sha256 = "3fc2130bf37df31852a8384f12601563a45a0024bccc6624f38355cba7a8d360" } }, + { url = "https://files.pythonhosted.org/packages/2a/66/edcec7d7a0b524aa8923e22925fde6fe50ce005a113dca13ae1581455c4c/coverage-7.15.4-cp311-cp311-macosx_10_9_x86_64.whl", upload-time = 2026-08-06T13:47:15Z, size = 222367, hashes = { sha256 = "bbac5abad70df71019988f83f26ac7092ff2642975def4429e98dc7585ef3490" } }, + { url = "https://files.pythonhosted.org/packages/e6/c6/ab8de429e2e8548faf58ec7e1674a4ce00414b4113942d3fe87109cf0f68/coverage-7.15.4-cp311-cp311-macosx_11_0_arm64.whl", upload-time = 2026-08-06T13:47:16Z, size = 222874, hashes = { sha256 = "357a173465c7ce028d07a95cc2b63b5bf59f50ecdd5ad75c5cbb78ada984048e" } }, + { url = "https://files.pythonhosted.org/packages/be/c4/3b7b49587e8a6b9af79b3eb468d443d6042b6d65b47aa26586846a0d6566/coverage-7.15.4-cp311-cp311-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-08-06T13:47:18Z, size = 253287, hashes = { sha256 = "21b803935e2efc3acebe9697197a294fccf5dc4e5382bd6369542ff7a7d2a1d7" } }, + { url = "https://files.pythonhosted.org/packages/fb/65/ec03b743a2a229c72cc1eff3e57be9d3564e9c6b4d5aba2d70744a3fc0d8/coverage-7.15.4-cp311-cp311-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-08-06T13:47:19Z, size = 255199, hashes = { sha256 = "7a2b580774a4786c1053157c0165e04476e03ff293993d7c148eee784a94bae6" } }, + { url = "https://files.pythonhosted.org/packages/41/4b/5163729e4b6582d61975cfd3ccab45b4ec53e21cf156d9941cb025188468/coverage-7.15.4-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-06T13:47:21Z, size = 257308, hashes = { sha256 = "a9464451c4efffe8d47ace5a540b10b0dc10e879066290f8600872b7f54a419d" } }, + { url = "https://files.pythonhosted.org/packages/86/08/2167a0f08fb87d702fa423a48578a32865464b7c9e1db3911ad7812ab414/coverage-7.15.4-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-06T13:47:22Z, size = 259268, hashes = { sha256 = "de602f34123c2f4af1c1869c6dbbbd60da6d5983bf01937367295d135cccbfce" } }, + { url = "https://files.pythonhosted.org/packages/1e/e5/68eebae3053dbd48508edea559c21b23fbdf3460784f91370c83a86a6acd/coverage-7.15.4-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-06T13:47:23Z, size = 253392, hashes = { sha256 = "6879ded16a27f3eeca19b900c147e81616e7054db451471a611b2755ee5249f7" } }, + { url = "https://files.pythonhosted.org/packages/1a/46/fd4ced40a2b691c774e515c9b69500bfa64c7960b67fcee4b2f6fad97fc3/coverage-7.15.4-cp311-cp311-musllinux_1_2_aarch64.whl", upload-time = 2026-08-06T13:47:25Z, size = 255001, hashes = { sha256 = "986be58c3ab54aae8d3496a6225eea74f760fdbe739b38bd442c7e8d133aa53b" } }, + { url = "https://files.pythonhosted.org/packages/53/25/ae2e5fa710bb6957a9aadeb9e3598d3b3e4af6587ce857ad42e8639a3f30/coverage-7.15.4-cp311-cp311-musllinux_1_2_i686.whl", upload-time = 2026-08-06T13:47:26Z, size = 253061, hashes = { sha256 = "c6103639613fe6c1e989082948419bc77a2d26b6c825c99d7fad25f7d3d87afc" } }, + { url = "https://files.pythonhosted.org/packages/d7/31/67ddc0365db2c6e93ac8580bc4bbc50f65273262f973f63ebcdbc15c0495/coverage-7.15.4-cp311-cp311-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-06T13:47:28Z, size = 256831, hashes = { sha256 = "d3af93dddb5659276c63bc16ac6466ac2033a70ca816097bbc06345b8ccdf571" } }, + { url = "https://files.pythonhosted.org/packages/f6/78/82b8fd18f57fb13f12d98fe874995bb2c4f9f17be8aff762c426323fdb96/coverage-7.15.4-cp311-cp311-musllinux_1_2_riscv64.whl", upload-time = 2026-08-06T13:47:29Z, size = 252781, hashes = { sha256 = "b10075e5421d04265766a6d1dac809bbeb8a946fbb23c8f82c227409b2190719" } }, + { url = "https://files.pythonhosted.org/packages/0a/eb/6c74ef4dd12b252e573c49bdef9e2ac265bf3dbb79b8d7feb3266e084e9e/coverage-7.15.4-cp311-cp311-musllinux_1_2_x86_64.whl", upload-time = 2026-08-06T13:47:31Z, size = 253692, hashes = { sha256 = "a67a9f78b2942d87ba8ce3059c642164d2aedd65337377fb52fe9803656bc5c7" } }, + { url = "https://files.pythonhosted.org/packages/5a/66/eb9aed1c3fd2d36ee00eb173f434b14fa607fc056739c9a89ff4244010ea/coverage-7.15.4-cp311-cp311-win32.whl", upload-time = 2026-08-06T13:47:32Z, size = 224461, hashes = { sha256 = "69484d1aca26e322e1c3ce03f09341e84524ababad2d7202161738d83cc9f82e" } }, + { url = "https://files.pythonhosted.org/packages/e2/6d/81fa4161dfb3ed9d74e40d58647eff83a56b7612e78352581280fce2f477/coverage-7.15.4-cp311-cp311-win_amd64.whl", upload-time = 2026-08-06T13:47:34Z, size = 224937, hashes = { sha256 = "63fd6fcd1dd6e158f7eb78606e72933b3f6d01e7b747f99c6c12d764307a0fdc" } }, + { url = "https://files.pythonhosted.org/packages/5b/c1/d8dacf683c6cad3cf85ce68fd3774a6774ec402128822fdfaed920f11e6a/coverage-7.15.4-cp311-cp311-win_arm64.whl", upload-time = 2026-08-06T13:47:36Z, size = 224479, hashes = { sha256 = "ea82116c9893fa89e929b7f197ee5a1950a76e91cc5c85ba503fc02379d04890" } }, + { url = "https://files.pythonhosted.org/packages/1d/48/bc8d4ba7b37551a767bd863f15b3f80182b271c2f55975356f5f7dbe94c2/coverage-7.15.4-cp312-cp312-macosx_10_13_x86_64.whl", upload-time = 2026-08-06T13:47:37Z, size = 222543, hashes = { sha256 = "d4fedd1f7f428f9fe83b1ead5e7cc87a43427be31aadafbac3ac0636dc7abb22" } }, + { url = "https://files.pythonhosted.org/packages/20/dd/88d6f83f1fffc974a3691a34a97951c5b12df7512a6782c5963883cbc058/coverage-7.15.4-cp312-cp312-macosx_11_0_arm64.whl", upload-time = 2026-08-06T13:47:38Z, size = 222905, hashes = { sha256 = "37e2f0cdf58e2e1fed4e4d5a8f8786ae2f7eb80b478016876667dc4a01d60a97" } }, + { url = "https://files.pythonhosted.org/packages/bd/5c/54ee0d4748585bb0acab9891cd8d92f2d3593165b4e59fc9de113bfb3140/coverage-7.15.4-cp312-cp312-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-08-06T13:47:40Z, size = 254407, hashes = { sha256 = "fb55d0e70bb15f2e81477613627286581414693d74ac7963c93a790dd453ca9d" } }, + { url = "https://files.pythonhosted.org/packages/8c/3f/f0642a372f494bd0d7dad3b497083b910194a5f1c88be2c94fef707c3b59/coverage-7.15.4-cp312-cp312-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-08-06T13:47:41Z, size = 257145, hashes = { sha256 = "899b9da30f3c6c336566e3707495bb23e8302d39d862f01fa78c48b99b9437e2" } }, + { url = "https://files.pythonhosted.org/packages/71/17/8b46d0ed68251016002ec972c8fc0119961a765d0984cafb8bf317c43758/coverage-7.15.4-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-06T13:47:43Z, size = 258257, hashes = { sha256 = "d15715e8c46552827e5e4f30a35575a2dbcad14454cf3284c54483946bd16931" } }, + { url = "https://files.pythonhosted.org/packages/30/b8/8498a0e72d0adbe15477dd07463d2b3bb2c9f6a4815e8589e50939e2c3ae/coverage-7.15.4-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-06T13:47:45Z, size = 260517, hashes = { sha256 = "002a438859f7b430bc99afeaf01a6d187dad1d0dc907b64cdeffc632a5db8fd8" } }, + { url = "https://files.pythonhosted.org/packages/41/e1/7dce19c3bdb1e3dd63e769508216500edad81bd5f69a26d724e32aceaf78/coverage-7.15.4-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-06T13:47:46Z, size = 254785, hashes = { sha256 = "e4193a04b518f7968f3099755f5509ee7cccc6dc2b92a6b14841934d22e222c9" } }, + { url = "https://files.pythonhosted.org/packages/dd/b1/e1494703c675a2561723cd9b89f45c9168782c31280c611b1f767851e57c/coverage-7.15.4-cp312-cp312-musllinux_1_2_aarch64.whl", upload-time = 2026-08-06T13:47:48Z, size = 256176, hashes = { sha256 = "e98dcc55d572b38e69d117da7e8e8efb8500f1f5eaf81ecd460a63220790b839" } }, + { url = "https://files.pythonhosted.org/packages/73/76/a5629d270fb638a43a4b10466f51e2f49d532c1aa4da2913cbbb150bbe0a/coverage-7.15.4-cp312-cp312-musllinux_1_2_i686.whl", upload-time = 2026-08-06T13:47:49Z, size = 254321, hashes = { sha256 = "af6c538498ce66c10d3fd541c2a8d5b03da5850355add34e6cba564210cb9e72" } }, + { url = "https://files.pythonhosted.org/packages/ff/4f/9c44447218435d5766b911534f9d798144a5560f85e9a54ebe5f3f5d19f9/coverage-7.15.4-cp312-cp312-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-06T13:47:51Z, size = 258390, hashes = { sha256 = "1d10025d96ea89fc2f73714dbc4cbd433fe012c1ac9e23f895d7728b238b6e52" } }, + { url = "https://files.pythonhosted.org/packages/de/36/c1e127616fb3fa18a9ff71e76c417f2fd7424332a4870015ac224ef4c039/coverage-7.15.4-cp312-cp312-musllinux_1_2_riscv64.whl", upload-time = 2026-08-06T13:47:52Z, size = 253894, hashes = { sha256 = "d802e1947603162ded419bff83ac7489820355d2b856dfb09206574e3a37ac0c" } }, + { url = "https://files.pythonhosted.org/packages/e9/b9/fdb92c8ae7a8bb9b850cc253b7b3b9c8526f68130002048b5671cd510d09/coverage-7.15.4-cp312-cp312-musllinux_1_2_x86_64.whl", upload-time = 2026-08-06T13:47:54Z, size = 255763, hashes = { sha256 = "c2de40895718f91951b86712b4c5b694acaf9a0a49be13874896f599a1eed3f4" } }, + { url = "https://files.pythonhosted.org/packages/6f/c0/a7d51b2587c7bdb76e71b0896d2565bf7d60436b5122fc83e511adb1f7cd/coverage-7.15.4-cp312-cp312-win32.whl", upload-time = 2026-08-06T13:47:56Z, size = 224597, hashes = { sha256 = "5c3431b2161279b7db5c2a1aa58ae02e5cb8c3c42d93a5094be3f5537bd5b11b" } }, + { url = "https://files.pythonhosted.org/packages/49/b9/5c5f80cc55f5acaaca6dee677626bfcec8c87204a7809b438b08e84f4571/coverage-7.15.4-cp312-cp312-win_amd64.whl", upload-time = 2026-08-06T13:47:57Z, size = 225135, hashes = { sha256 = "6befeab5fb2b51c958ca4ac6c5d141a1e8240f4f76e46350f1911963deda49cd" } }, + { url = "https://files.pythonhosted.org/packages/47/e4/2a4561f89ff6bf7c925c287d0f2cce8bdf139c3a33735c87e3203401cf94/coverage-7.15.4-cp312-cp312-win_arm64.whl", upload-time = 2026-08-06T13:47:58Z, size = 224515, hashes = { sha256 = "67bc345491ab55b837277d76f5775d057e8c7f1ac44d890d8c2c82adde258c6f" } }, + { url = "https://files.pythonhosted.org/packages/f1/84/651a9310859673aaa3b3203f1aa1641ca60fcf2494683e1c9474c7172780/coverage-7.15.4-cp313-cp313-macosx_10_13_x86_64.whl", upload-time = 2026-08-06T13:48:00Z, size = 222565, hashes = { sha256 = "c705b28feb2775dc82a25f1d473a370bc37ff93f5177f4e29ce2425f560f6921" } }, + { url = "https://files.pythonhosted.org/packages/82/f9/4dcf700137e8af550670f4d74d1b63828ce93e1e2b05e5f10710eb2ea987/coverage-7.15.4-cp313-cp313-macosx_11_0_arm64.whl", upload-time = 2026-08-06T13:48:02Z, size = 222936, hashes = { sha256 = "3ff205ab5e3ecc670f6a4dd19d9cbf12ede53dd41cfc1e15716ec961ea6d314e" } }, + { url = "https://files.pythonhosted.org/packages/07/4a/612ff1e780b3fbfd637486f542f84adc5503873d8b5d279dec1ffeef9414/coverage-7.15.4-cp313-cp313-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-08-06T13:48:04Z, size = 253926, hashes = { sha256 = "5172326e861a38b48b48befca15e0f477a26b283337a33a739c8fed229934e36" } }, + { url = "https://files.pythonhosted.org/packages/b0/04/d1cff1c2ead4708a6a79c01d3736b6a25bd38a36678398f72a8dd33dfad9/coverage-7.15.4-cp313-cp313-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-08-06T13:48:05Z, size = 256523, hashes = { sha256 = "12b59c90084e3234fb11184886bf4a40f4f16a8c8f867be2e087b81f8e8868d4" } }, + { url = "https://files.pythonhosted.org/packages/b9/80/d34e13fb4b293cbdb9665838cf5522077b8ad14ef947550631a4bced36a5/coverage-7.15.4-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-06T13:48:08Z, size = 257759, hashes = { sha256 = "349062d66f00b40fa2c1c222438bad25fabf755631b5d82937fe985c8008615c" } }, + { url = "https://files.pythonhosted.org/packages/0f/e7/2c5fe7636fdb0732fe0f09f308a5b066864078b7fc61f6678e8478554f2e/coverage-7.15.4-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-06T13:48:09Z, size = 259890, hashes = { sha256 = "4256ced708e598e05209bc1a8ab4074e04a51dba4c62fb45926a229af675ace7" } }, + { url = "https://files.pythonhosted.org/packages/92/28/9689f0858dfff59c2ea688938ab9fa2925631235df67126a42b6c5c70ae1/coverage-7.15.4-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-06T13:48:11Z, size = 254121, hashes = { sha256 = "d80f974b20782d9612c8b4c9beeca867074c7cf4079d1419843fa25a26428b25" } }, + { url = "https://files.pythonhosted.org/packages/f9/e2/785077c230c157243eb5aa9a26c3be260ecd02001bead54a3cada3df8e03/coverage-7.15.4-cp313-cp313-musllinux_1_2_aarch64.whl", upload-time = 2026-08-06T13:48:13Z, size = 255891, hashes = { sha256 = "2e179f19bfe1d31f8eeeaa12990194d761c4f62f0759661000bca6cd8729f40b" } }, + { url = "https://files.pythonhosted.org/packages/d4/90/e20371b17b40f912f21305c2db2f30efa3de306f7320fc916804872c85a4/coverage-7.15.4-cp313-cp313-musllinux_1_2_i686.whl", upload-time = 2026-08-06T13:48:14Z, size = 253859, hashes = { sha256 = "8bc16bb47b7679670eceff71d78bfb7d6e5b143f6c2cd117487ec7c75e0d4b78" } }, + { url = "https://files.pythonhosted.org/packages/05/49/25371987ee459a5f67c0427fb75c74f9358e65f2c71fe75bf41c1b6c5fcb/coverage-7.15.4-cp313-cp313-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-06T13:48:16Z, size = 258011, hashes = { sha256 = "1cd685005cd2c4200adfc14cf39a603b9320efab3f18a8f7f156d20c9cc3345f" } }, + { url = "https://files.pythonhosted.org/packages/30/6e/32e67467f6154bf4f1c4f63b05acc5097cba4237d45bbeeea446b52e8ac1/coverage-7.15.4-cp313-cp313-musllinux_1_2_riscv64.whl", upload-time = 2026-08-06T13:48:18Z, size = 253676, hashes = { sha256 = "337399ad2c93b3acd2a937627dae8b3e86b66707cd3d3e856347999aadf1ef8d" } }, + { url = "https://files.pythonhosted.org/packages/03/c1/8b24192e89286399765155251f99ee9f070a9d637109018ac23d99b99f6f/coverage-7.15.4-cp313-cp313-musllinux_1_2_x86_64.whl", upload-time = 2026-08-06T13:48:20Z, size = 255453, hashes = { sha256 = "96e257121228ec5cd2bb919276e94ac11074471bc37d68dbae0e8308cce15fff" } }, + { url = "https://files.pythonhosted.org/packages/16/6f/8b41ebdf67c87854e17c035336a90f1cfbad0c14c2a584301be6ff148718/coverage-7.15.4-cp313-cp313-win32.whl", upload-time = 2026-08-06T13:48:21Z, size = 224605, hashes = { sha256 = "c65a9e0dfc6143491879da4e13b5e30f8be192055de508d737fb14601edbd22c" } }, + { url = "https://files.pythonhosted.org/packages/e0/e2/2946c7f0b42b152ecb21ff1bdad72e3d301e790c0c487e4a86e8c9f69347/coverage-7.15.4-cp313-cp313-win_amd64.whl", upload-time = 2026-08-06T13:48:23Z, size = 225148, hashes = { sha256 = "2ff8f5e9b8f7a94f0c11c45631eee103dbcb7d63274edd12c56efe1be690b3b4" } }, + { url = "https://files.pythonhosted.org/packages/9e/83/3f4a69957f48ae7a0aba76c34743f88963d607b19e03f3f8e66f91cae0f9/coverage-7.15.4-cp313-cp313-win_arm64.whl", upload-time = 2026-08-06T13:48:25Z, size = 224536, hashes = { sha256 = "6e0a8a5083b096487d6cfced94cdd514d8f5db6f113610fb36c0620edb1028cf" } }, + { url = "https://files.pythonhosted.org/packages/ea/ac/748cf29eeb2d6be34a3176ce26a4f49e38085ee08e8935f05f6f26ed7e0f/coverage-7.15.4-cp314-cp314-macosx_10_15_x86_64.whl", upload-time = 2026-08-06T13:48:26Z, size = 222608, hashes = { sha256 = "770e9325ab5ea6d56f77e59b29ecfe0ac20b57a82a601876f90494a4dda0386f" } }, + { url = "https://files.pythonhosted.org/packages/0b/02/1abbf5c984677b0aa439cdacaccbf38d248939d8ef8fe1cc7a50d73edb77/coverage-7.15.4-cp314-cp314-macosx_11_0_arm64.whl", upload-time = 2026-08-06T13:48:28Z, size = 222940, hashes = { sha256 = "d12b33a3a50a1676b7784dc8d00a0c6d66a9f2add4b85a041c19b6a7e53ef23c" } }, + { url = "https://files.pythonhosted.org/packages/eb/e1/ff8f9f53d9fcf586125b55d0b1f04ec1c14955fee41e83d5814bee141bb5/coverage-7.15.4-cp314-cp314-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-08-06T13:48:29Z, size = 253985, hashes = { sha256 = "5669c8378ebde86f5def7a25d29586631b58acc27ffde04399f678f3dfc6e082" } }, + { url = "https://files.pythonhosted.org/packages/a1/26/595759762e514e81be1d7d01ed03444303bcd152226a6529998d253f9201/coverage-7.15.4-cp314-cp314-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-08-06T13:48:31Z, size = 256492, hashes = { sha256 = "ff97a14362eef486483ed44042ca2027ea257df6ff768e62358ee0c9776925ac" } }, + { url = "https://files.pythonhosted.org/packages/24/68/b79aabac54d482be23b5fcdd4f4662bff24a78edc4ee29201726929936d5/coverage-7.15.4-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-06T13:48:33Z, size = 257837, hashes = { sha256 = "5a325e815318638aed1655d9c06e6d7c2d3d46c09231ce988070428a8762d734" } }, + { url = "https://files.pythonhosted.org/packages/09/0f/bf7f297885a5bf6fd71e5782404e0ff059ca09e8711ceb3a08544abde45a/coverage-7.15.4-cp314-cp314-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-06T13:48:34Z, size = 260152, hashes = { sha256 = "474223409d88eb20d2d6a0d37ea60e8647a65a90cc008dc1f0410af5f64f1e0d" } }, + { url = "https://files.pythonhosted.org/packages/fd/f1/296744e854ff8368542343457414380465e9ceefb9192342feb9d3bc461d/coverage-7.15.4-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-06T13:48:36Z, size = 253978, hashes = { sha256 = "7f2f62ae3cd189dd2e13aece758c57b3eecbd27be070dbd4cbd10936049e5dbf" } }, + { url = "https://files.pythonhosted.org/packages/55/b0/bbdb2e9057493e66220a2e149ca2d301ba0e3a58a83bd6b90de9826d16f3/coverage-7.15.4-cp314-cp314-musllinux_1_2_aarch64.whl", upload-time = 2026-08-06T13:48:38Z, size = 255846, hashes = { sha256 = "39ece820e29e0a2ba34b3ecb3be83c27e997eed8926f2ba6fe7ce7a0bda5843b" } }, + { url = "https://files.pythonhosted.org/packages/96/e4/38015b2b6d21258713bd17e76b59d033b191efb5703589cffd037dfbca20/coverage-7.15.4-cp314-cp314-musllinux_1_2_i686.whl", upload-time = 2026-08-06T13:48:39Z, size = 253808, hashes = { sha256 = "f21b56dcace11dfe013014201f577dcd592b2a9b72182d930361b47cf6f73f25" } }, + { url = "https://files.pythonhosted.org/packages/0b/64/0d515c1e60ee6fbfd1a0e79c07cd87d388a233b7adc37758735677203808/coverage-7.15.4-cp314-cp314-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-06T13:48:41Z, size = 258081, hashes = { sha256 = "93a3a0b662abcc10c73a47cbc72cd60f63618d6989fb2d1286e50eacd974f303" } }, + { url = "https://files.pythonhosted.org/packages/91/71/04d9e7a3642146c6351338aef4ef85ab11dbbb54744c13245caba1aad1c0/coverage-7.15.4-cp314-cp314-musllinux_1_2_riscv64.whl", upload-time = 2026-08-06T13:48:43Z, size = 253624, hashes = { sha256 = "141fae2cabf5569b782c10afc4c850ce10f618c13f8db54765cba99cc839da1f" } }, + { url = "https://files.pythonhosted.org/packages/b4/a7/6c28b74c81ebff66987b0e2522ba5cffa3e90b0c33cb6a2eb264d4ee8cf1/coverage-7.15.4-cp314-cp314-musllinux_1_2_x86_64.whl", upload-time = 2026-08-06T13:48:45Z, size = 255280, hashes = { sha256 = "81294c7e6ab30c5f74c0353b11b2fd6320e72d9bee6ac73b357caa8b916323a5" } }, + { url = "https://files.pythonhosted.org/packages/52/af/bc19996a7014b98d7bbb0f0939453c67074af65784a3aa16a789a07381fa/coverage-7.15.4-cp314-cp314-win32.whl", upload-time = 2026-08-06T13:48:47Z, size = 224768, hashes = { sha256 = "7bbd7d6418e0dab31a206af5203bd43ae36edb8e7fba1940b055d3e9249290d7" } }, + { url = "https://files.pythonhosted.org/packages/ee/90/219484e476d6e101ba0a444852579e05f5b75c37c611a42ed1190f73ef62/coverage-7.15.4-cp314-cp314-win_amd64.whl", upload-time = 2026-08-06T13:48:49Z, size = 225259, hashes = { sha256 = "f0204ed122758782970526057093f448051a39db9d810d4e344bb87a3546f425" } }, + { url = "https://files.pythonhosted.org/packages/b7/66/fa77daf4e383e5f776dac62c2409b6af81910ae6fe326bd5170dba74cc63/coverage-7.15.4-cp314-cp314-win_arm64.whl", upload-time = 2026-08-06T13:48:51Z, size = 224684, hashes = { sha256 = "9e71e7bc71c686a123347ae47a0de33a175e797a85bb57b791492adf4eec8ed8" } }, + { url = "https://files.pythonhosted.org/packages/58/5b/f03bf0ce362bbf3f785fa5219620d00778d4ac6fc9e407734828e9c672f6/coverage-7.15.4-cp314-cp314t-macosx_10_15_x86_64.whl", upload-time = 2026-08-06T13:48:52Z, size = 223338, hashes = { sha256 = "7c922735321eef3f87c280a3d39afff6b646723a2880b862cda4ac7a093b8aa8" } }, + { url = "https://files.pythonhosted.org/packages/0f/76/e77d0ae22501831cc9f92193e8a957a5caa1dd177f90a6d1d9b106242d92/coverage-7.15.4-cp314-cp314t-macosx_11_0_arm64.whl", upload-time = 2026-08-06T13:48:54Z, size = 223609, hashes = { sha256 = "f41c17c4668a655ce96d090d8d5ffdc24ef64b5a02f9753884d08483e8a4a41a" } }, + { url = "https://files.pythonhosted.org/packages/82/1a/b1f089da8d38ac612fa2dd6dc7f4a1a7657d12f3e261d2996edd3a838d0b/coverage-7.15.4-cp314-cp314t-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-08-06T13:48:56Z, size = 264970, hashes = { sha256 = "46822e9b6ff1c6a72b518c162c44a8f45a61a1d609c51084bf5b16c023c5037b" } }, + { url = "https://files.pythonhosted.org/packages/bf/31/e66d98d6e9c7fcc88470f1e234eaf6b1950dc0dfbf797f7282c1c861da24/coverage-7.15.4-cp314-cp314t-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-08-06T13:48:58Z, size = 267088, hashes = { sha256 = "3d6f4955b73b5445271379a59e3792b0d978f42d4a01e0cf7a67d9c33a3bb0a5" } }, + { url = "https://files.pythonhosted.org/packages/59/a1/ae94eb2c541add426378408379f233591e069040b1e2cdb33df9498a0682/coverage-7.15.4-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-06T13:49:00Z, size = 269508, hashes = { sha256 = "3fc9e047706fb4a9abb54f719d3aa643e80e5bb3818182c40aee01ac0f0247ba" } }, + { url = "https://files.pythonhosted.org/packages/9c/c7/88a10694a1c6a213569766aba9f25847b28155d4ac731b13226db216356d/coverage-7.15.4-cp314-cp314t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-06T13:49:02Z, size = 270629, hashes = { sha256 = "05e491d4f3165d62d4f5c8fd48dfeabf2ae8f42cbbd484319af33ea851b78982" } }, + { url = "https://files.pythonhosted.org/packages/b3/34/d8b8232e5e55169933b59aabcef2fedfa4b9d8897361bb80fcbda146505f/coverage-7.15.4-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-06T13:49:04Z, size = 264043, hashes = { sha256 = "226c66e80ec0598d3b9b4874123df167ccca342aca8714f77cac6829688ee09c" } }, + { url = "https://files.pythonhosted.org/packages/7e/35/58b009dbf8c471c7224716478b9fed4a7e1af15320e1ed41660978504663/coverage-7.15.4-cp314-cp314t-musllinux_1_2_aarch64.whl", upload-time = 2026-08-06T13:49:05Z, size = 266963, hashes = { sha256 = "ac41cc14bebda0dbfb0628036b7f75706935c95bcc07fefe9a0f93614aa60a57" } }, + { url = "https://files.pythonhosted.org/packages/62/aa/57fbda1b42c892968273c56b6ee9dc0f1310850859230a507bc7873b1f65/coverage-7.15.4-cp314-cp314t-musllinux_1_2_i686.whl", upload-time = 2026-08-06T13:49:07Z, size = 264569, hashes = { sha256 = "8af623e5cd92080acddd02b38f2f406a2c3a0893c38950b211890361448fbf26" } }, + { url = "https://files.pythonhosted.org/packages/98/8a/360e6e7f24d477b7e889703af0afa878d15b6d4d8d2a822b2835c169a879/coverage-7.15.4-cp314-cp314t-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-06T13:49:09Z, size = 268299, hashes = { sha256 = "07545711d4f0f32852a18f18ad11f76f0109909d09e78b9008b4cfc67e829429" } }, + { url = "https://files.pythonhosted.org/packages/4e/89/6f701261aee21b6b5fa8f7872229406dc917e125069448292223bf213606/coverage-7.15.4-cp314-cp314t-musllinux_1_2_riscv64.whl", upload-time = 2026-08-06T13:49:11Z, size = 263413, hashes = { sha256 = "a0865421cfdc53654b342d515e5a233187590882d20b95752150e53f65460017" } }, + { url = "https://files.pythonhosted.org/packages/3f/0f/6f04036edc260ed425af83e834f627fad48941ce97b50bfe6edd8b6fa623/coverage-7.15.4-cp314-cp314t-musllinux_1_2_x86_64.whl", upload-time = 2026-08-06T13:49:13Z, size = 265725, hashes = { sha256 = "460115e32ee40566476db5048f9bec1e842c127ad8e6f8be745aad3ac9cbc839" } }, + { url = "https://files.pythonhosted.org/packages/c4/ce/d19b5d4d5c49a7bfb925fd74310fee7d28bc99520ac3367ccbc54e662518/coverage-7.15.4-cp314-cp314t-win32.whl", upload-time = 2026-08-06T13:49:15Z, size = 225079, hashes = { sha256 = "cbde877ef9dd7baf272b9bfef2b8a25edd45d9170fc326951dd20eb480335e85" } }, + { url = "https://files.pythonhosted.org/packages/26/bb/7aa1b3b173faee0679037ca950bbbe1247273656697994d8d13f80f8d4b4/coverage-7.15.4-cp314-cp314t-win_amd64.whl", upload-time = 2026-08-06T13:49:17Z, size = 225911, hashes = { sha256 = "3da9e92d1c551fd7563833e9ade686efb0c4b7363ab7681a94283958c950bf5e" } }, + { url = "https://files.pythonhosted.org/packages/81/1c/4ea9e47426d80038d9222db3c4534cb6021a74b237d3ff97ffd33b6600dd/coverage-7.15.4-cp314-cp314t-win_arm64.whl", upload-time = 2026-08-06T13:49:19Z, size = 225219, hashes = { sha256 = "3a54f5a0d85050c73a38f6793090ee83974531e67fe5e57a1da9bee11398aa5e" } }, + { url = "https://files.pythonhosted.org/packages/2b/c4/dc5d2ac8f9142e7ec7de66e7bf0591db29d78955a040bd915870d9c0e657/coverage-7.15.4-cp315-cp315-macosx_10_15_x86_64.whl", upload-time = 2026-08-06T13:49:21Z, size = 222604, hashes = { sha256 = "2c9872e4d9dc5d3cf616bf4b382f5a00359305a5be666a3dd0b5cdb4e49597f9" } }, + { url = "https://files.pythonhosted.org/packages/70/39/33e63df81fe2ee100897451841c821467635923e58e37c6bd4b46dd8106c/coverage-7.15.4-cp315-cp315-macosx_11_0_arm64.whl", upload-time = 2026-08-06T13:49:23Z, size = 222944, hashes = { sha256 = "e101dbb4b9b72f0cddd8cdc8c9c5b47f456766f5e0ac82dbfb75e5c55409b78a" } }, + { url = "https://files.pythonhosted.org/packages/99/1f/ef3ffb5557febc75a0d97aa459d0266d7d741110265121cc6d8539343d44/coverage-7.15.4-cp315-cp315-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-08-06T13:49:25Z, size = 254050, hashes = { sha256 = "7d1abebdb047729e852b9c77a00497dfbeb11eb3a117e037d7dbc3ac8e5f5c54" } }, + { url = "https://files.pythonhosted.org/packages/6f/f5/1f0f6f77698c3601ca0ae7431e34b24c62ca2f06fecb23b73ed1f651d2be/coverage-7.15.4-cp315-cp315-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-08-06T13:49:26Z, size = 256967, hashes = { sha256 = "d28a4a899354d0ea6214cc59b4fa19eefbce1b9ff1688ab579acf49e894bd3fb" } }, + { url = "https://files.pythonhosted.org/packages/03/7a/2ed9bed79925f4367c83c77f66a89e5ca7229c288d2d19ad5f36d1ca0070/coverage-7.15.4-cp315-cp315-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-06T13:49:28Z, size = 258587, hashes = { sha256 = "ffb3c2aacea411cc7e1d27712490c11108e2de1d39019ae32915493a59a8b9ed" } }, + { url = "https://files.pythonhosted.org/packages/45/8c/fa34044f71b7cc4ecb6da9c2408770959b0591fa9b5fb6fb6bca38f94298/coverage-7.15.4-cp315-cp315-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-06T13:49:30Z, size = 260785, hashes = { sha256 = "a9447978a92f405d301123cfd39ff49895490efb769a758fe2734c7f631bf8ce" } }, + { url = "https://files.pythonhosted.org/packages/4f/54/d5727ce36b4524a7394ab9f5f1df378e1f23affcdab01037dc8655185cc7/coverage-7.15.4-cp315-cp315-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-06T13:49:32Z, size = 254545, hashes = { sha256 = "050467a7983b8e2fe7dd41a78bb30c3e7f8c0b8cafda14b1c46f8b5e3cf2dd3c" } }, + { url = "https://files.pythonhosted.org/packages/dc/e6/6e3783e576719590194bdffb6dd6d85490801785b7c331e35a245d8cb8b5/coverage-7.15.4-cp315-cp315-musllinux_1_2_aarch64.whl", upload-time = 2026-08-06T13:49:34Z, size = 256682, hashes = { sha256 = "d003b7a5708ddad5c206c79607a6b92abb6fc13c57d99d8a4468cc03a2941ced" } }, + { url = "https://files.pythonhosted.org/packages/dc/f2/bacdbde18b69ed2de424fcf64d9fb0a4913753d4f0eca8bae9daad69f4bd/coverage-7.15.4-cp315-cp315-musllinux_1_2_i686.whl", upload-time = 2026-08-06T13:49:36Z, size = 254560, hashes = { sha256 = "c38efe30fd74e5c19e9433f11fb1f5dc9c6522770971b7c6145bbaa413dc8800" } }, + { url = "https://files.pythonhosted.org/packages/6c/a3/1fb927196e3477c1b48831169ab58ba08f451ba87ae311ff1de68b26a616/coverage-7.15.4-cp315-cp315-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-06T13:49:38Z, size = 258792, hashes = { sha256 = "1f4f826d70f772ab8b0c052329580d7fe8b8abd191e4ce0c8f81aec6614665d3" } }, + { url = "https://files.pythonhosted.org/packages/41/58/30d4c149c69053de0edfe325614c1d28d508f62b1783e0e4a234d2e49136/coverage-7.15.4-cp315-cp315-musllinux_1_2_riscv64.whl", upload-time = 2026-08-06T13:49:39Z, size = 253968, hashes = { sha256 = "4a4bf917c9953f57c957be31c1cd504e3bd2f34d4a352b9d391a3025336f6768" } }, + { url = "https://files.pythonhosted.org/packages/89/e4/77f639371b918aad30dda4051f95404b43578f7f2e2f87ba73e02ed1ff37/coverage-7.15.4-cp315-cp315-musllinux_1_2_x86_64.whl", upload-time = 2026-08-06T13:49:41Z, size = 255893, hashes = { sha256 = "1c9bf40ebef178a45192c75c4964760bb261b0e6ad725da5fc4c93f674f19753" } }, + { url = "https://files.pythonhosted.org/packages/5c/62/13be29b3ddab35f14c87967a4820a05106d2a3eccb4fa4ff550bf30b75e0/coverage-7.15.4-cp315-cp315-win32.whl", upload-time = 2026-08-06T13:49:44Z, size = 224768, hashes = { sha256 = "43619d04c3671792d2c4706ae8bf45e265dc87bbd4078189ef8b847ea1e74be2" } }, + { url = "https://files.pythonhosted.org/packages/a1/70/af0c6be0f964af6954f6b74bc109b0dbca02824696d2520fb17fe1ab06e3/coverage-7.15.4-cp315-cp315-win_amd64.whl", upload-time = 2026-08-06T13:49:45Z, size = 225242, hashes = { sha256 = "be619439dbcd31a2eab10b32de9fff62c26ed4bab69dc32b8363fdaaa0882809" } }, + { url = "https://files.pythonhosted.org/packages/4f/2d/f3bd3aab899fc9efc18b53133ee68f5f98574ef480649b23e12962226387/coverage-7.15.4-cp315-cp315-win_arm64.whl", upload-time = 2026-08-06T13:49:48Z, size = 224674, hashes = { sha256 = "def597967dafc2e8d97c9097ea453c464e0bb8ed38f193a43070f10dc623bb6d" } }, + { url = "https://files.pythonhosted.org/packages/f5/ca/f69251cd63eabc6438321aea22148754cce758a26bde07dd490e3fe7cfc5/coverage-7.15.4-cp315-cp315t-macosx_10_15_x86_64.whl", upload-time = 2026-08-06T13:49:50Z, size = 223333, hashes = { sha256 = "c7dbc748ac8a1e3e59a2b28bea47675e6e778081dbbf081bde0d75def2fcbe1d" } }, + { url = "https://files.pythonhosted.org/packages/a7/a7/037b53b2885b0d8447064432491a4d5a1014cd9f97a594d53acd0c04541a/coverage-7.15.4-cp315-cp315t-macosx_11_0_arm64.whl", upload-time = 2026-08-06T13:49:52Z, size = 223630, hashes = { sha256 = "2413074a5ecbb61a01a7888fc72db0ca324d13588c5b38bc0dd8564cdcdfea26" } }, + { url = "https://files.pythonhosted.org/packages/80/4f/152b8a4779ae90da11bb24f7467df8a59f0be48a5c52acb856325ca48289/coverage-7.15.4-cp315-cp315t-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-08-06T13:49:54Z, size = 264489, hashes = { sha256 = "4e6f6f632b7b2f714bf7a1346e8f97b650ee71f3c298aaad42a2ab60f0f07645" } }, + { url = "https://files.pythonhosted.org/packages/10/2d/84b4b9e0e1dd6528a51920ff7031f35b789382e467a28ec6a5a578cb8812/coverage-7.15.4-cp315-cp315t-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-08-06T13:49:56Z, size = 267567, hashes = { sha256 = "8df457da2249d3c75ca2e5e835d59c725abfe92d27fdff6cd99eed85b51d5e9a" } }, + { url = "https://files.pythonhosted.org/packages/53/fc/ba01cc25299f9f8a2c8b02d3b28c53f3543d9fbfbe4e74fa2760b48f163e/coverage-7.15.4-cp315-cp315t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-06T13:49:58Z, size = 270123, hashes = { sha256 = "050f66a08805acb5b8a23c6d4a517b1ecf82c08e81ed0e4bd727df065e5c6624" } }, + { url = "https://files.pythonhosted.org/packages/cf/d0/db2647cbf40b14f8c308f94ff7bf89c06d564e59f396906edf50086ec788/coverage-7.15.4-cp315-cp315t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-06T13:50:00Z, size = 271107, hashes = { sha256 = "1587fb771d1ccceef708fdde1e5af8c7ed24b486b61d13a321acb7d8145390aa" } }, + { url = "https://files.pythonhosted.org/packages/70/ff/4d2d17924552c458bb4f77dd631f0e3bc92fbbdf2d2d916cd4b33bbfd5b1/coverage-7.15.4-cp315-cp315t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-06T13:50:03Z, size = 264955, hashes = { sha256 = "8b4f1c3a69ca580f3fbd6b2046915f536d7f586874f25c1bb23add2a3c88d50f" } }, + { url = "https://files.pythonhosted.org/packages/ee/de/dc010c7a3691f396d93bbc26bfcafa1c2a3a351cd520470f15faf5795bd5/coverage-7.15.4-cp315-cp315t-musllinux_1_2_aarch64.whl", upload-time = 2026-08-06T13:50:05Z, size = 267949, hashes = { sha256 = "ffb58d7eff5b7f6ecc6fa21d6288ab7f968a212cb67d682c269c09b9eba3b66f" } }, + { url = "https://files.pythonhosted.org/packages/78/ea/dc96a11375e83c045c2f7c61fb6918277cfe9401db7c0f7b1d111a84b2e5/coverage-7.15.4-cp315-cp315t-musllinux_1_2_i686.whl", upload-time = 2026-08-06T13:50:07Z, size = 264421, hashes = { sha256 = "d9df165544774574ee004b953023d1bebada1894a80b1052a43d798b0f676e67" } }, + { url = "https://files.pythonhosted.org/packages/c8/86/b77131a0f9503ce461cd577076147d7a9040f0c5dda772686f729e2cc9cb/coverage-7.15.4-cp315-cp315t-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-06T13:50:09Z, size = 269121, hashes = { sha256 = "f9de0a24a4079b53e523b5c5e2c5945ec251ab486652659955187cf255a259bc" } }, + { url = "https://files.pythonhosted.org/packages/24/24/944bc35007862955e7ebf05754e645419dcf5d7526c52735cfa2715e8ebf/coverage-7.15.4-cp315-cp315t-musllinux_1_2_riscv64.whl", upload-time = 2026-08-06T13:50:11Z, size = 264565, hashes = { sha256 = "150089274bdc9f940628552cb92844e0223c987f1902ab8efe9f45a2ec758d88" } }, + { url = "https://files.pythonhosted.org/packages/c7/cc/a3bb9f93e7e740659163e2ea584f8196ddcd2c456a5dbe15f6c50105fec1/coverage-7.15.4-cp315-cp315t-musllinux_1_2_x86_64.whl", upload-time = 2026-08-06T13:50:13Z, size = 266522, hashes = { sha256 = "a58a94fed5da6997d258e8f7668c1e195fbd04a691d781b7558f1e468f9e68bc" } }, + { url = "https://files.pythonhosted.org/packages/49/dd/e0e40f3560d878d888c580698ff5ad1179f5e1c3ac949684ef66b41a3817/coverage-7.15.4-cp315-cp315t-win32.whl", upload-time = 2026-08-06T13:50:15Z, size = 225068, hashes = { sha256 = "ebd5a6d8466ff30836572f3ba2cae8a5e8f85029b1c6d5e2ed338dc472a5166a" } }, + { url = "https://files.pythonhosted.org/packages/c6/7e/37732ea80eebc30e976e4cdab15c190bc42d96959a42e38ddf6f8c60468f/coverage-7.15.4-cp315-cp315t-win_amd64.whl", upload-time = 2026-08-06T13:50:17Z, size = 225895, hashes = { sha256 = "288bde2a2d7ab6b6c2d7252fcde8b524387f2d970bdba9658fc6f8bbcaef0f9b" } }, + { url = "https://files.pythonhosted.org/packages/c6/08/1e00f7923eaaba45fb3d51dd794125fc766304b1df264f3a9c6557bfb30e/coverage-7.15.4-cp315-cp315t-win_arm64.whl", upload-time = 2026-08-06T13:50:19Z, size = 225213, hashes = { sha256 = "68be5e1de60ff13c9095bbec0e5a7fa45b33b101752215b91345ea1f61c4a278" } }, + { url = "https://files.pythonhosted.org/packages/b4/d9/e70c286c979378f061d8266e279b686ab0b0b688e1fe0af864684f23a77d/coverage-7.15.4-py3-none-any.whl", upload-time = 2026-08-06T13:50:22Z, size = 214332, hashes = { sha256 = "964730a1e9de9c0cf11be6a1a3c79ce419c34882842abd256086ba4698705e84" } }, ] [[packages]] name = "docker" -version = "7.1.0" +version = "7.2.0" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/91/9b/4a2ea29aeba62471211598dac5d96825bb49348fa07e906ea930394a83ce/docker-7.1.0.tar.gz", upload-time = 2024-05-23T11:13:57Z, size = 117834, hashes = { sha256 = "ad8c70e6e3f8926cb8a92619b832b4ea5299e2831c14284663184e200546fa6c" } } +sdist = { url = "https://files.pythonhosted.org/packages/88/7f/731ff914b0255d3d065f45fd4e626d4b8c95dbcbaada049f337a6ac16410/docker-7.2.0.tar.gz", upload-time = 2026-07-09T14:53:46Z, size = 118731, hashes = { sha256 = "cebb93773d334f778e023a7ee352a8d6e13ab1bd3b863a4d4a59dec897df43ac" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/e3/26/57c6fb270950d476074c087527a558ccb6f4436657314bfb6cdf484114c4/docker-7.1.0-py3-none-any.whl", upload-time = 2024-05-23T11:13:55Z, size = 147774, hashes = { sha256 = "c96b93b7f0a746f9e77d325bcfb87422a3d8bd4f03136ae8a85b37f1898d5fc0" } }, + { url = "https://files.pythonhosted.org/packages/75/23/529140fe1aab80fc6992f93a706deec709140a6397439139a054e1515c45/docker-7.2.0-py3-none-any.whl", upload-time = 2026-07-09T14:53:45Z, size = 148775, hashes = { sha256 = "a3f45fdeb9165e2d25d9a1d02ddf3bc70fb572cf5ebbf9b58558c22caf29b71f" } }, ] [[packages]] @@ -302,11 +375,11 @@ wheels = [ [[packages]] name = "idna" -version = "3.15" +version = "3.19" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/82/77/7b3966d0b9d1d31a36ddf1746926a11dface89a83409bf1483f0237aa758/idna-3.15.tar.gz", upload-time = 2026-05-12T22:45:57Z, size = 199245, hashes = { sha256 = "ca962446ea538f7092a95e057da437618e886f4d349216d2b1e294abfdb65fdc" } } +sdist = { url = "https://files.pythonhosted.org/packages/5f/f7/abb373e5757eaec4b922b92f97ec8d6d7e057cf06778247604fbc4e7c3f3/idna-3.19.tar.gz", upload-time = 2026-08-18T05:14:24Z, size = 215237, hashes = { sha256 = "5e0811a4383b21dc5838069f801c4fb62113b7447663d2530d2bd6e77b49bf15" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/d2/23/408243171aa9aaba178d3e2559159c24c1171a641aa83b67bdd3394ead8e/idna-3.15-py3-none-any.whl", upload-time = 2026-05-12T22:45:55Z, size = 72340, hashes = { sha256 = "048adeaf8c2d788c40fee287673ccaa74c24ffd8dcf09ffa555a2fbb59f10ac8" } }, + { url = "https://files.pythonhosted.org/packages/57/b0/0e52c878c53f245edd3a11020f20979b3f490f245af532c7cae3027754b5/idna-3.19-py3-none-any.whl", upload-time = 2026-08-18T05:14:22Z, size = 68550, hashes = { sha256 = "815e7be7a7806d54abb586dc943addc79e8b2ee16915059658cbeff4b1b43bf4" } }, ] [[packages]] @@ -320,32 +393,32 @@ wheels = [ [[packages]] name = "maturin" -version = "1.13.3" +version = "1.15.0" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/9c/1c/612d23d33ec21b9ae7ece7b3f0dd5f9dfd57b4009e9d2938165869ebd6ae/maturin-1.13.3.tar.gz", upload-time = 2026-05-11T07:43:39Z, size = 357934, hashes = { sha256 = "771e1e9e71a278e56db01552e0d1acfd1464259f9575b6e72842f893cd299079" } } +sdist = { url = "https://files.pythonhosted.org/packages/b9/c8/22e5e21b2679c9bce6415ca578034ca2cc9316be0642ae21e051a2d5198c/maturin-1.15.0.tar.gz", upload-time = 2026-08-24T12:11:22Z, size = 385504, hashes = { sha256 = "94b26cc8e8aba61a5f2099715fe640e18c5f678e9a500408b38761263954228a" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/71/66/18c2aaac0b2a5dea9f1db5984ce83b905ad205cfc7c02d0091e707c0c2e7/maturin-1.13.3-py3-none-linux_armv6l.whl", upload-time = 2026-05-11T07:43:10Z, size = 10190971, hashes = { sha256 = "3cc13929ca82aefa4adbf0f2c35419369796213c6fb0eb24e914945f50ef5d8c" } }, - { url = "https://files.pythonhosted.org/packages/bc/71/26a988d092e4fd6a9523d46d44400a46cad7cdf3fd206ce702240c748aee/maturin-1.13.3-py3-none-macosx_10_12_x86_64.macosx_11_0_arm64.macosx_10_12_universal2.whl", upload-time = 2026-05-11T07:43:36Z, size = 19716714, hashes = { sha256 = "53b08bd075649ce96513ad9abf241a43cb685ed6e9e7790f8dbc2d66e95d8323" } }, - { url = "https://files.pythonhosted.org/packages/82/5c/f3fd0e184255d9fc7e272c62af3dfa84c617b2577ef83af9ce615f5279cc/maturin-1.13.3-py3-none-macosx_10_12_x86_64.whl", upload-time = 2026-05-11T07:43:07Z, size = 10194726, hashes = { sha256 = "4cd478e6e4c56251e48ed079b8efd55b30bc5c09cf695a1bdafaeb582ee735a0" } }, - { url = "https://files.pythonhosted.org/packages/a9/e1/f4edb69fb647b77c4769a9bfd4d6fb62961e653d164bc277ecdffac3ab61/maturin-1.13.3-py3-none-manylinux_2_12_i686.manylinux2010_i686.musllinux_1_1_i686.whl", upload-time = 2026-05-11T07:43:40Z, size = 10172781, hashes = { sha256 = "a2675e25f313034ae6f57388cf14818f87d8961c4a96795287f3e155f59beb11" } }, - { url = "https://files.pythonhosted.org/packages/c7/7d/a1be934690cdcc3c6609769ceaad322ab7501c2ee5bafcac1b14d609e403/maturin-1.13.3-py3-none-manylinux_2_12_x86_64.manylinux2010_x86_64.musllinux_1_1_x86_64.whl", upload-time = 2026-05-11T07:43:13Z, size = 10682670, hashes = { sha256 = "4667ef609ab446c1b5e0bfe4f9fb99699ab6d8548433f8d1a684256e0b67217f" } }, - { url = "https://files.pythonhosted.org/packages/18/f5/372ae19b72ce8f6e37e5864ae4dc5b252ee9fce0619ccc3aa366aa3a7f97/maturin-1.13.3-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.musllinux_1_1_aarch64.whl", upload-time = 2026-05-11T07:43:21Z, size = 10060363, hashes = { sha256 = "3db93337ed97e60ffc878aa8b493cd7ae44d3a5e1a37256db3a4491f57565018" } }, - { url = "https://files.pythonhosted.org/packages/cb/5b/c68340cca09368af0df80965dfabed4234205a492a93da00793c7b9aae20/maturin-1.13.3-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.musllinux_1_1_armv7l.whl", upload-time = 2026-05-11T07:43:33Z, size = 10017551, hashes = { sha256 = "1cc0a110b224ca90406b668a3e3c1f5a515062e59e26292f6dbaf5fd4909c6f3" } }, - { url = "https://files.pythonhosted.org/packages/28/1e/f90fb2b000bad9e6d850cd5afb88b2f1e2a279cfb4de02ea40078484690e/maturin-1.13.3-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.musllinux_1_1_ppc64le.whl", upload-time = 2026-05-11T07:43:26Z, size = 13301712, hashes = { sha256 = "c00ea6428dea17bf616fe93770837634454b28c2de1a876e42ef8036c616079a" } }, - { url = "https://files.pythonhosted.org/packages/be/58/1670f68a8f04ccd7b90df11047bd9a046585310e84e1967cc9849cd1c5a3/maturin-1.13.3-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", upload-time = 2026-05-11T07:43:16Z, size = 10946765, hashes = { sha256 = "49fd6ab08da28098ccf37afca24cdba72376ba9c1eedf9dd25ff82ed771961ff" } }, - { url = "https://files.pythonhosted.org/packages/4b/ac/00c955c2ef134817b1a7bdaa76b0309e9c5291eb17d9ff88069eecd08bc2/maturin-1.13.3-py3-none-manylinux_2_31_riscv64.musllinux_1_1_riscv64.whl", upload-time = 2026-05-11T07:43:18Z, size = 10388661, hashes = { sha256 = "b6741d7bf4af97da937528fd1e523c6ab54f53d9a21870fa735d6e67fd88e273" } }, - { url = "https://files.pythonhosted.org/packages/97/c6/cbf8a51dde19c19aeba0d9b075095a2effb9b31fd312b1aae3ac79f8aea2/maturin-1.13.3-py3-none-win32.whl", upload-time = 2026-05-11T07:43:23Z, size = 8901838, hashes = { sha256 = "0ef257e692cc756c87af5bea95ddfe7d3ac49d3376a7a87f728d63f06e7b6f8b" } }, - { url = "https://files.pythonhosted.org/packages/a1/ff/c6a50a59dc8313097d43ac5f4d74df6a500c8cb62b0dc9e054f53e203a48/maturin-1.13.3-py3-none-win_amd64.whl", upload-time = 2026-05-11T07:43:29Z, size = 10340801, hashes = { sha256 = "def4a435ea9d2ee93b18ba579dc8c9cf898889a66f312cd379b5e374ec3e3ad6" } }, - { url = "https://files.pythonhosted.org/packages/6c/93/e32e79333f0902ba292b996f504f5f06be59587f7d02ab8d5ed1e3066445/maturin-1.13.3-py3-none-win_arm64.whl", upload-time = 2026-05-11T07:43:31Z, size = 9706562, hashes = { sha256 = "2389fe92d017cea9d94e521fa0175314a4c52f79a1057b901fbc9f8686ef7d0b" } }, + { url = "https://files.pythonhosted.org/packages/14/69/5c01b461044eb1f45ddcce006706eb88110c793cdb11c7ae0b5e08492e94/maturin-1.15.0-py3-none-linux_armv6l.whl", upload-time = 2026-08-24T12:10:53Z, size = 10206220, hashes = { sha256 = "6bf6dc62e22d4dcfd5a51244ff0d58975fa4979c48209fe84159617648956d82" } }, + { url = "https://files.pythonhosted.org/packages/eb/1f/2b431554e11687cdb1077e0cdadcc118c53f611086b3af00c8545a67c6a5/maturin-1.15.0-py3-none-macosx_10_12_x86_64.macosx_11_0_arm64.macosx_10_12_universal2.whl", upload-time = 2026-08-24T12:10:56Z, size = 19416513, hashes = { sha256 = "cd35772633f489841132bc8e71d6fc7f842df30b9c05cd5cdf1ee1ddcb744cc7" } }, + { url = "https://files.pythonhosted.org/packages/51/36/e23a21cb34a648b711036b9b2fe1d4f3f4ee24f8db54215d73f1a9a3a3ec/maturin-1.15.0-py3-none-macosx_10_12_x86_64.whl", upload-time = 2026-08-24T12:10:58Z, size = 10014962, hashes = { sha256 = "c40b4eae7bf5ef1f4b1af8d623fe4105016f93578fb15b764e741d08ec3b92dd" } }, + { url = "https://files.pythonhosted.org/packages/8c/80/33b15cb2d8f30f12c807955e8f2fd775027692904e30ec0784744ce8cd83/maturin-1.15.0-py3-none-manylinux_2_12_i686.manylinux2010_i686.musllinux_1_1_i686.whl", upload-time = 2026-08-24T12:11:00Z, size = 10196223, hashes = { sha256 = "7eb066372f541f8eb4909c79c5d9bd0b9e8125980bdf1ec9e8aba23c6c8d6c55" } }, + { url = "https://files.pythonhosted.org/packages/fe/91/b495e19e2f5c503b540452b2039115e7b2363867e8c5ad4179eb752fa92c/maturin-1.15.0-py3-none-manylinux_2_12_x86_64.manylinux2010_x86_64.musllinux_1_1_x86_64.whl", upload-time = 2026-08-24T12:11:02Z, size = 10541186, hashes = { sha256 = "653020a63525bb224e5ab0adf02e17a2e08bc86dbea7fc1399c9a56d7529b99e" } }, + { url = "https://files.pythonhosted.org/packages/5d/2b/2abff58037188d852b124871b1f0d720e1c2bfb3d4f1b03d87c52cd66488/maturin-1.15.0-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.musllinux_1_1_aarch64.whl", upload-time = 2026-08-24T12:11:05Z, size = 10083468, hashes = { sha256 = "0ebf9767892725083138e671c34482c660317a2f3d6a29fc0e0f34e9d8c99136" } }, + { url = "https://files.pythonhosted.org/packages/3f/07/b7e9f8be99a6627849e81ac7b6694876bce8f50a92995fe17e3cf2610f0a/maturin-1.15.0-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.musllinux_1_1_armv7l.whl", upload-time = 2026-08-24T12:11:07Z, size = 10047786, hashes = { sha256 = "7ab7eebffd7b8debca2265985de4eaeb332141276d24b9560b5ad484d4b3add1" } }, + { url = "https://files.pythonhosted.org/packages/ca/ab/167e3cb7accee11b507dbe53e0e87aeccb376d44ae66284c96ee4df3a9fd/maturin-1.15.0-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.musllinux_1_1_ppc64le.whl", upload-time = 2026-08-24T12:11:09Z, size = 13315332, hashes = { sha256 = "126e12e618b4db42f68c779a56d41f82a390145ba36ac3f621d057eb34f5ad9d" } }, + { url = "https://files.pythonhosted.org/packages/14/4d/801379f646cbc6b00998e5289b0630a886be3a4ee4c75b6bc9b87478a7f1/maturin-1.15.0-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", upload-time = 2026-08-24T12:11:11Z, size = 10807183, hashes = { sha256 = "4f9d33e6c3f9615c8caceecbbbd440f8eb25a3ddeb687077682cd5eca2e9ae15" } }, + { url = "https://files.pythonhosted.org/packages/89/27/2e612e1cbd1580e9e94d4722c227b5180dca27b32b955a34b79918aa1292/maturin-1.15.0-py3-none-manylinux_2_31_riscv64.musllinux_1_1_riscv64.whl", upload-time = 2026-08-24T12:11:14Z, size = 10413274, hashes = { sha256 = "bf29beddd0c6708f112db51d5275fc28b28b9e9c9c5faae387eaef662918b176" } }, + { url = "https://files.pythonhosted.org/packages/70/d8/202a7b4d75a51f20f84ec9ce3b7345b12b822207072164e2b1c6ef665125/maturin-1.15.0-py3-none-win32.whl", upload-time = 2026-08-24T12:11:16Z, size = 8928744, hashes = { sha256 = "da649988be98e87e009e51b1bf0d301b6a301bc0cecbdd60d40d8ba60748d1ca" } }, + { url = "https://files.pythonhosted.org/packages/40/dc/4e90da594986ba78dd3bc8a5921ecdcdb11085b22b02a412caab3b225601/maturin-1.15.0-py3-none-win_amd64.whl", upload-time = 2026-08-24T12:11:18Z, size = 10335085, hashes = { sha256 = "552c2be4afd43fe8d5c9f3ec8d4c4756d973b8dcbe94c14084390301f50243e1" } }, + { url = "https://files.pythonhosted.org/packages/8b/10/15d4314edf130955edf2dc237aa393a8a7c10f2b9b57b89fa2f61f915659/maturin-1.15.0-py3-none-win_arm64.whl", upload-time = 2026-08-24T12:11:20Z, size = 9713795, hashes = { sha256 = "c7dc0c66c78d3debdd9c5aa807e861fbcbf07f3505d34b125df74c03986b0f48" } }, ] [[packages]] name = "packaging" -version = "26.2" +version = "26.3" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/d7/f1/e7a6dd94a8d4a5626c03e4e99c87f241ba9e350cd9e6d75123f992427270/packaging-26.2.tar.gz", upload-time = 2026-04-24T20:15:23Z, size = 228134, hashes = { sha256 = "ff452ff5a3e828ce110190feff1178bb1f2ea2281fa2075aadb987c2fb221661" } } +sdist = { url = "https://files.pythonhosted.org/packages/7d/fa/3944b40b07da9ce895c0e6303a5ab7d53da063554f534556b134a54d6093/packaging-26.3.tar.gz", upload-time = 2026-08-04T18:15:28Z, size = 313412, hashes = { sha256 = "94edc256424af38762eb31306eed28beb9f0efc50a8837492c9d6fd6004aed79" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/df/b2/87e62e8c3e2f4b32e5fe99e0b86d576da1312593b39f47d8ceef365e95ed/packaging-26.2-py3-none-any.whl", upload-time = 2026-04-24T20:15:22Z, size = 100195, hashes = { sha256 = "5fc45236b9446107ff2415ce77c807cee2862cb6fac22b8a73826d0693b0980e" } }, + { url = "https://files.pythonhosted.org/packages/63/34/ba1c580383c9eada3711951fef0795c80b829a078d72188184bcab9dd527/packaging-26.3-py3-none-any.whl", upload-time = 2026-08-04T18:15:27Z, size = 129956, hashes = { sha256 = "d7193f7c8e4e93f444fde0262bf90af30e16fa0ad0ad44cb553c87339b23cd1c" } }, ] [[packages]] @@ -359,55 +432,57 @@ wheels = [ [[packages]] name = "pygments" -version = "2.20.0" +version = "2.21.0" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/c3/b2/bc9c9196916376152d655522fdcebac55e66de6603a76a02bca1b6414f6c/pygments-2.20.0.tar.gz", upload-time = 2026-03-29T13:29:33Z, size = 4955991, hashes = { sha256 = "6757cd03768053ff99f3039c1a36d6c0aa0b263438fcab17520b30a303a82b5f" } } +sdist = { url = "https://files.pythonhosted.org/packages/49/2e/ced460408999b33da6b31b0021b0f37d329e202d4169aeb164493778f25b/pygments-2.21.0.tar.gz", upload-time = 2026-08-17T08:02:48Z, size = 5005329, hashes = { sha256 = "610ca751c9bc2492b38eb9a38a7fbc93edbbb2d7182edaf34e66ae493dee5c8c" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/f4/7e/a72dd26f3b0f4f2bf1dd8923c85f7ceb43172af56d63c7383eb62b332364/pygments-2.20.0-py3-none-any.whl", upload-time = 2026-03-29T13:29:30Z, size = 1231151, hashes = { sha256 = "81a9e26dd42fd28a23a2d169d86d7ac03b46e2f8b59ed4698fb4785f946d0176" } }, + { url = "https://files.pythonhosted.org/packages/71/46/17f022dd3e953bf20a04a028a21ec746d942f8d2af30fa0f124fa0e6a684/pygments-2.21.0-py3-none-any.whl", upload-time = 2026-08-17T08:02:44Z, size = 1250147, hashes = { sha256 = "2363c69b61c4a97c838da3b130dcd6468f4848992b21a82f2a63ec34377137d9" } }, ] [[packages]] name = "pyrefly" -version = "1.0.0" +version = "1.2.0" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/9f/3a/9045b0097ac58979c7c30a4fa0e673db942d4adbc7b6d439bd54ae58c441/pyrefly-1.0.0.tar.gz", upload-time = 2026-05-12T20:12:46Z, size = 5677995, hashes = { sha256 = "5c2b810ffcebd84be71de5df1223651edee951653a66935c6f091e957c452455" } } +sdist = { url = "https://files.pythonhosted.org/packages/89/01/a86e9f24722b095c3f88e3616132b75a21b0df53804bdc6a45314dd4d93c/pyrefly-1.2.0.tar.gz", upload-time = 2026-08-01T02:56:27Z, size = 6243654, hashes = { sha256 = "5485f960fc2481617068c918335c39ab1507ef90b6b5bd35bf57726e60e73185" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/f4/c6/90788819bac9c61dd7bacba53b79f3c12d47ccbe5e51b3d6d89f2387e1d2/pyrefly-1.0.0-py3-none-macosx_10_12_x86_64.whl", upload-time = 2026-05-12T20:12:20Z, size = 13122950, hashes = { sha256 = "e355a0908555348ed4b9585ef25c76ff566673e345c866c325f1633f44d890b6" } }, - { url = "https://files.pythonhosted.org/packages/82/91/a3cf2a1e87d336eaa804a1e6fc93266faf6dc2a97eecdbc7eae289628022/pyrefly-1.0.0-py3-none-macosx_11_0_arm64.whl", upload-time = 2026-05-12T20:12:23Z, size = 12599494, hashes = { sha256 = "a7038efc3a40f8294edee339895633cf22db268c0d434cdbcbefc34f78a9ecc3" } }, - { url = "https://files.pythonhosted.org/packages/cd/ab/74d1e11e737e99b1c003ecc5d7d2e846c4ea1f328966bfdbbd0ac63fad0a/pyrefly-1.0.0-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", upload-time = 2026-05-12T20:12:25Z, size = 12995507, hashes = { sha256 = "da331ca515ed1c08791da2b5f664cf9c1294c48fd802133262e7d5d51e0f4416" } }, - { url = "https://files.pythonhosted.org/packages/7c/ac/2df0899f8464c97e5d995f994c97c5cb5b0f58610432aa90d26d924e1db5/pyrefly-1.0.0-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", upload-time = 2026-05-12T20:12:29Z, size = 13947693, hashes = { sha256 = "c74219d8f3e63cdaa5501a0b21d1c9d37011820f9606728d0ed06f09ae86a878" } }, - { url = "https://files.pythonhosted.org/packages/6b/3e/b247c24321e36f04b7d51f9ccf3df93e5009e4b29939524b36ec2e17dc2a/pyrefly-1.0.0-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", upload-time = 2026-05-12T20:12:31Z, size = 13925803, hashes = { sha256 = "c0d05543b1bb6ee6d64149eb5d6b2fb15aa72d3962d6a97abca0afaca8b0c131" } }, - { url = "https://files.pythonhosted.org/packages/61/16/cfa2d61a4aa1e1f7bca48bb37acd01c6a09db4864b16a54f9587092765ff/pyrefly-1.0.0-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", upload-time = 2026-05-12T20:12:35Z, size = 13470398, hashes = { sha256 = "1382d5b1fcdb49a4de9f34d112d2bddf290a78ff93ee8149492ad5f1077ddffc" } }, - { url = "https://files.pythonhosted.org/packages/cb/2b/6372c7dddb326223e24a46b17efd0d4bd7b4fe22c821e523157577eed2d2/pyrefly-1.0.0-py3-none-win32.whl", upload-time = 2026-05-12T20:12:38Z, size = 12222643, hashes = { sha256 = "aa8b5d0e47080e3202a2547b39f7a5a61d2c781c712b3b67884f745ca2c759d2" } }, - { url = "https://files.pythonhosted.org/packages/be/ad/1d23be700b6b2ddaeb362360c7145917a8edbbf7240ae428d40541772fce/pyrefly-1.0.0-py3-none-win_amd64.whl", upload-time = 2026-05-12T20:12:41Z, size = 13146369, hashes = { sha256 = "c8abcb0f2082e83c890375128f9cff4aa4d3f210b85eea7b3046c1ae764e77f5" } }, - { url = "https://files.pythonhosted.org/packages/8c/38/16589134f3012fd097a10dcc85771555f1a5fb76e04b682597180743af30/pyrefly-1.0.0-py3-none-win_arm64.whl", upload-time = 2026-05-12T20:12:43Z, size = 12538326, hashes = { sha256 = "d150fa9e40e8392832be81c3bcfc0497c146674ce4d0f8e04e1ec29e775ffb8c" } }, + { url = "https://files.pythonhosted.org/packages/7d/9d/3c0ef1d4843987b22f996ed381ec9cf5a3b1273e29804db276252e4c95eb/pyrefly-1.2.0-py3-none-macosx_10_12_x86_64.whl", upload-time = 2026-08-01T02:56:02Z, size = 14026305, hashes = { sha256 = "7f46d983ac49ddd2b043694960a01dc6a19a5cfd8eec609d6bd9c42866f91b4e" } }, + { url = "https://files.pythonhosted.org/packages/0a/06/03bbb78fbea54cdc65b626619f3597d5611aca4fdef11e72a4e8360e7e63/pyrefly-1.2.0-py3-none-macosx_11_0_arm64.whl", upload-time = 2026-08-01T02:56:04Z, size = 13463880, hashes = { sha256 = "756f669b5555090f5c1a4fef30db1785fabe657764f7e4e6dc88994dfb8ca82d" } }, + { url = "https://files.pythonhosted.org/packages/13/5a/7d8bc00a38e93bbc9c3e7bd14d305f7948717e667c9bcddeab9dd42fd255/pyrefly-1.2.0-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", upload-time = 2026-08-01T02:56:07Z, size = 13907329, hashes = { sha256 = "e3465812ce5ef4781fb592edbf2724547296f0a3124be115d73c7e8b2401862d" } }, + { url = "https://files.pythonhosted.org/packages/be/94/9e08b4bf799d0b8f36b55a2783c7ba5f51730cf0632a85a67b5b5ed876cd/pyrefly-1.2.0-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", upload-time = 2026-08-01T02:56:09Z, size = 15039020, hashes = { sha256 = "5de7b2ad2bba5c8055181681a84b74143eac2234a48ba5d1b7ed7e7a722b02bd" } }, + { url = "https://files.pythonhosted.org/packages/5b/bd/bca5fd0c80f4daf8ee6903a29df9f3de1feb05ff0946b8f35ec8c5096b13/pyrefly-1.2.0-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", upload-time = 2026-08-01T02:56:11Z, size = 14986199, hashes = { sha256 = "25822ea9505f589ea8a725e4268b475132fb89e038fbf092e446510443ac142a" } }, + { url = "https://files.pythonhosted.org/packages/97/f7/f07087f3d185ad2eced0c56cef89ca5474dfb4ff25f146cd50a861c97553/pyrefly-1.2.0-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", upload-time = 2026-08-01T02:56:14Z, size = 14393715, hashes = { sha256 = "90efe75e17491ef5d636e10469e9278d7d0256b3b4c5e1f4750069bf3ae0f5d1" } }, + { url = "https://files.pythonhosted.org/packages/d3/70/0d142c320e284b9e3ce35e9b1e58b8ce2ee1f578f2a7234bc30e5022b94f/pyrefly-1.2.0-py3-none-musllinux_1_2_aarch64.whl", upload-time = 2026-08-01T02:56:16Z, size = 13933008, hashes = { sha256 = "368aaf7eee4f511ddc0f8e564cf14e01ab2f10b0db9105c6d5b153bf498d07bf" } }, + { url = "https://files.pythonhosted.org/packages/5d/e8/e84f11b6e1f63fd453ad3654213b9a0f6f4de8cef6b58038eef2d0d5955d/pyrefly-1.2.0-py3-none-musllinux_1_2_x86_64.whl", upload-time = 2026-08-01T02:56:18Z, size = 14431827, hashes = { sha256 = "d52d5da7bc65fb7675fbaa80eda879d4f8787c494f04cac21603330d3abbdbbe" } }, + { url = "https://files.pythonhosted.org/packages/0f/06/810d31380f66c75e1c0779a408d3b16117b1b368b57894f6aa66bef21686/pyrefly-1.2.0-py3-none-win32.whl", upload-time = 2026-08-01T02:56:20Z, size = 13229447, hashes = { sha256 = "8c90751de8506d938e8f802659c74cf35bd7a0036510ee6c634a38eebb280bfa" } }, + { url = "https://files.pythonhosted.org/packages/ed/98/4dafa3c7a1caed2dc8cc708dde09ba27963c7736508f55b626fff3024113/pyrefly-1.2.0-py3-none-win_amd64.whl", upload-time = 2026-08-01T02:56:23Z, size = 14087387, hashes = { sha256 = "8a8964c224ccc4882730130955815de21ff443c1ac3f0b90685b19bf63848170" } }, + { url = "https://files.pythonhosted.org/packages/1b/1c/df3cb0a2e5591660ded7a1836cd2f29dc48c91adb1c0a3a700a96f6d09e1/pyrefly-1.2.0-py3-none-win_arm64.whl", upload-time = 2026-08-01T02:56:25Z, size = 13430873, hashes = { sha256 = "3a90bb8df39dfbac74b1f3b2e9d7c526b8f80568884c3944d955023a73ebf61e" } }, ] [[packages]] name = "pytest" -version = "9.0.3" +version = "9.1.1" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/7d/0d/549bd94f1a0a402dc8cf64563a117c0f3765662e2e668477624baeec44d5/pytest-9.0.3.tar.gz", upload-time = 2026-04-07T17:16:18Z, size = 1572165, hashes = { sha256 = "b86ada508af81d19edeb213c681b1d48246c1a91d304c6c81a427674c17eb91c" } } +sdist = { url = "https://files.pythonhosted.org/packages/e4/47/b9efed96c114afcfa3c9d3fe98a76a1d14c74a9e266d397cf6eb64be5e01/pytest-9.1.1.tar.gz", upload-time = 2026-06-19T10:58:32Z, size = 1636369, hashes = { sha256 = "1088fbde8f2b49d95a549a195707afa7a76a3ce9bcadc26b6d71f0ffda5fe313" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/d4/24/a372aaf5c9b7208e7112038812994107bc65a84cd00e0354a88c2c77a617/pytest-9.0.3-py3-none-any.whl", upload-time = 2026-04-07T17:16:16Z, size = 375249, hashes = { sha256 = "2c5efc453d45394fdd706ade797c0a81091eccd1d6e4bccfcd476e2b8e0ab5d9" } }, + { url = "https://files.pythonhosted.org/packages/24/25/1de2678b631f5a49215c6c96fff41ba892b0a34df68d6d80292b1b48aa7f/pytest-9.1.1-py3-none-any.whl", upload-time = 2026-06-19T10:58:31Z, size = 386536, hashes = { sha256 = "37a86b45efb9a47a61a36449063e8e18d0cab3161329fc099eb21783169c4f0c" } }, ] [[packages]] name = "pytest-asyncio" -version = "1.3.0" +version = "1.4.0" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/90/2c/8af215c0f776415f3590cac4f9086ccefd6fd463befeae41cd4d3f193e5a/pytest_asyncio-1.3.0.tar.gz", upload-time = 2025-11-10T16:07:47Z, size = 50087, hashes = { sha256 = "d7f52f36d231b80ee124cd216ffb19369aa168fc10095013c6b014a34d3ee9e5" } } +sdist = { url = "https://files.pythonhosted.org/packages/43/7c/d36d04db312ecf4298932ef77e6e4a9e8ad017906e24e34f0b0c361a2473/pytest_asyncio-1.4.0.tar.gz", upload-time = 2026-05-26T09:56:04Z, size = 58514, hashes = { sha256 = "c6c0d2259945122819f171a32ecea2c349ead889ee28176caaf492143424be42" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/e5/35/f8b19922b6a25bc0880171a2f1a003eaeb93657475193ab516fd87cac9da/pytest_asyncio-1.3.0-py3-none-any.whl", upload-time = 2025-11-10T16:07:45Z, size = 15075, hashes = { sha256 = "611e26147c7f77640e6d0a92a38ed17c3e9848063698d5c93d5aa7aa11cebff5" } }, + { url = "https://files.pythonhosted.org/packages/03/e2/08a497ef684b88559c9cc5f4ad53a37e7b99e727094a86d6ea32536d5d3c/pytest_asyncio-1.4.0-py3-none-any.whl", upload-time = 2026-05-26T09:56:02Z, size = 16930, hashes = { sha256 = "933ca923a23075a87fb7070c0ec272a6848489824d887c85c812670932835aa1" } }, ] [[packages]] name = "pytest-cov" -version = "6.3.0" +version = "7.1.0" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/30/4c/f883ab8f0daad69f47efdf95f55a66b51a8b939c430dadce0611508d9e99/pytest_cov-6.3.0.tar.gz", upload-time = 2025-09-06T15:40:14Z, size = 70398, hashes = { sha256 = "35c580e7800f87ce892e687461166e1ac2bcb8fb9e13aea79032518d6e503ff2" } } +sdist = { url = "https://files.pythonhosted.org/packages/b1/51/a849f96e117386044471c8ec2bd6cfebacda285da9525c9106aeb28da671/pytest_cov-7.1.0.tar.gz", upload-time = 2026-03-21T20:11:16Z, size = 55592, hashes = { sha256 = "30674f2b5f6351aa09702a9c8c364f6a01c27aae0c1366ae8016160d1efc56b2" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/80/b4/bb7263e12aade3842b938bc5c6958cae79c5ee18992f9b9349019579da0f/pytest_cov-6.3.0-py3-none-any.whl", upload-time = 2025-09-06T15:40:12Z, size = 25115, hashes = { sha256 = "440db28156d2468cafc0415b4f8e50856a0d11faefa38f30906048fe490f1749" } }, + { url = "https://files.pythonhosted.org/packages/9d/7a/d968e294073affff457b041c2be9868a40c1c71f4a35fcc1e45e5493067b/pytest_cov-7.1.0-py3-none-any.whl", upload-time = 2026-03-21T20:11:14Z, size = 22876, hashes = { sha256 = "a0461110b7865f9a271aa1b51e516c9a95de9d696734a2f71e3e78f46e1d4678" } }, ] [[packages]] @@ -430,34 +505,37 @@ wheels = [ [[packages]] name = "python-dotenv" -version = "1.2.2" +version = "1.2.3" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/82/ed/0301aeeac3e5353ef3d94b6ec08bbcabd04a72018415dcb29e588514bba8/python_dotenv-1.2.2.tar.gz", upload-time = 2026-03-01T16:00:26Z, size = 50135, hashes = { sha256 = "2c371a91fbd7ba082c2c1dc1f8bf89ca22564a087c2c287cd9b662adde799cf3" } } +sdist = { url = "https://files.pythonhosted.org/packages/6a/53/ed9d74092561d4b01a2ef1349d52cdbc135e526c245f366b089cfca6de49/python_dotenv-1.2.3.tar.gz", upload-time = 2026-08-16T16:54:54Z, size = 58945, hashes = { sha256 = "a20a594dabeaa385725aa239d5244871c143ecb356add8a20fcf23773a6c3a35" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/0b/d7/1959b9648791274998a9c3526f6d0ec8fd2233e4d4acce81bbae76b44b2a/python_dotenv-1.2.2-py3-none-any.whl", upload-time = 2026-03-01T16:00:25Z, size = 22101, hashes = { sha256 = "1d8214789a24de455a8b8bd8ae6fe3c6b69a5e3d64aa8a8e5d68e694bbcb285a" } }, + { url = "https://files.pythonhosted.org/packages/0d/17/c5c6b53ddc18f297992099b3d9ec16c855c0ccc83263a21fe4d1c625ec6c/python_dotenv-1.2.3-py3-none-any.whl", upload-time = 2026-08-16T16:54:52Z, size = 22780, hashes = { sha256 = "904552145e8bfed22162c09dab1c2b9b54fefa7b23ba780f4f26ca0316b0f0d9" } }, ] [[packages]] name = "pywin32" -version = "311" +version = "312" marker = "sys_platform == 'win32'" index = "https://pypi.org/simple" wheels = [ - { url = "https://files.pythonhosted.org/packages/7b/40/44efbb0dfbd33aca6a6483191dae0716070ed99e2ecb0c53683f400a0b4f/pywin32-311-cp310-cp310-win32.whl", upload-time = 2025-07-14T20:13:05Z, size = 8760432, hashes = { sha256 = "d03ff496d2a0cd4a5893504789d4a15399133fe82517455e78bad62efbb7f0a3" } }, - { url = "https://files.pythonhosted.org/packages/5e/bf/360243b1e953bd254a82f12653974be395ba880e7ec23e3731d9f73921cc/pywin32-311-cp310-cp310-win_amd64.whl", upload-time = 2025-07-14T20:13:07Z, size = 9590103, hashes = { sha256 = "797c2772017851984b97180b0bebe4b620bb86328e8a884bb626156295a63b3b" } }, - { url = "https://files.pythonhosted.org/packages/57/38/d290720e6f138086fb3d5ffe0b6caa019a791dd57866940c82e4eeaf2012/pywin32-311-cp310-cp310-win_arm64.whl", upload-time = 2025-07-14T20:13:11Z, size = 8778557, hashes = { sha256 = "0502d1facf1fed4839a9a51ccbcc63d952cf318f78ffc00a7e78528ac27d7a2b" } }, - { url = "https://files.pythonhosted.org/packages/7c/af/449a6a91e5d6db51420875c54f6aff7c97a86a3b13a0b4f1a5c13b988de3/pywin32-311-cp311-cp311-win32.whl", upload-time = 2025-07-14T20:13:13Z, size = 8697031, hashes = { sha256 = "184eb5e436dea364dcd3d2316d577d625c0351bf237c4e9a5fabbcfa5a58b151" } }, - { url = "https://files.pythonhosted.org/packages/51/8f/9bb81dd5bb77d22243d33c8397f09377056d5c687aa6d4042bea7fbf8364/pywin32-311-cp311-cp311-win_amd64.whl", upload-time = 2025-07-14T20:13:15Z, size = 9508308, hashes = { sha256 = "3ce80b34b22b17ccbd937a6e78e7225d80c52f5ab9940fe0506a1a16f3dab503" } }, - { url = "https://files.pythonhosted.org/packages/44/7b/9c2ab54f74a138c491aba1b1cd0795ba61f144c711daea84a88b63dc0f6c/pywin32-311-cp311-cp311-win_arm64.whl", upload-time = 2025-07-14T20:13:16Z, size = 8703930, hashes = { sha256 = "a733f1388e1a842abb67ffa8e7aad0e70ac519e09b0f6a784e65a136ec7cefd2" } }, - { url = "https://files.pythonhosted.org/packages/e7/ab/01ea1943d4eba0f850c3c61e78e8dd59757ff815ff3ccd0a84de5f541f42/pywin32-311-cp312-cp312-win32.whl", upload-time = 2025-07-14T20:13:20Z, size = 8706543, hashes = { sha256 = "750ec6e621af2b948540032557b10a2d43b0cee2ae9758c54154d711cc852d31" } }, - { url = "https://files.pythonhosted.org/packages/d1/a8/a0e8d07d4d051ec7502cd58b291ec98dcc0c3fff027caad0470b72cfcc2f/pywin32-311-cp312-cp312-win_amd64.whl", upload-time = 2025-07-14T20:13:22Z, size = 9495040, hashes = { sha256 = "b8c095edad5c211ff31c05223658e71bf7116daa0ecf3ad85f3201ea3190d067" } }, - { url = "https://files.pythonhosted.org/packages/ba/3a/2ae996277b4b50f17d61f0603efd8253cb2d79cc7ae159468007b586396d/pywin32-311-cp312-cp312-win_arm64.whl", upload-time = 2025-07-14T20:13:24Z, size = 8710102, hashes = { sha256 = "e286f46a9a39c4a18b319c28f59b61de793654af2f395c102b4f819e584b5852" } }, - { url = "https://files.pythonhosted.org/packages/a5/be/3fd5de0979fcb3994bfee0d65ed8ca9506a8a1260651b86174f6a86f52b3/pywin32-311-cp313-cp313-win32.whl", upload-time = 2025-07-14T20:13:26Z, size = 8705700, hashes = { sha256 = "f95ba5a847cba10dd8c4d8fefa9f2a6cf283b8b88ed6178fa8a6c1ab16054d0d" } }, - { url = "https://files.pythonhosted.org/packages/e3/28/e0a1909523c6890208295a29e05c2adb2126364e289826c0a8bc7297bd5c/pywin32-311-cp313-cp313-win_amd64.whl", upload-time = 2025-07-14T20:13:28Z, size = 9494700, hashes = { sha256 = "718a38f7e5b058e76aee1c56ddd06908116d35147e133427e59a3983f703a20d" } }, - { url = "https://files.pythonhosted.org/packages/04/bf/90339ac0f55726dce7d794e6d79a18a91265bdf3aa70b6b9ca52f35e022a/pywin32-311-cp313-cp313-win_arm64.whl", upload-time = 2025-07-14T20:13:30Z, size = 8709318, hashes = { sha256 = "7b4075d959648406202d92a2310cb990fea19b535c7f4a78d3f5e10b926eeb8a" } }, - { url = "https://files.pythonhosted.org/packages/c9/31/097f2e132c4f16d99a22bfb777e0fd88bd8e1c634304e102f313af69ace5/pywin32-311-cp314-cp314-win32.whl", upload-time = 2025-07-14T20:13:32Z, size = 8840714, hashes = { sha256 = "b7a2c10b93f8986666d0c803ee19b5990885872a7de910fc460f9b0c2fbf92ee" } }, - { url = "https://files.pythonhosted.org/packages/90/4b/07c77d8ba0e01349358082713400435347df8426208171ce297da32c313d/pywin32-311-cp314-cp314-win_amd64.whl", upload-time = 2025-07-14T20:13:34Z, size = 9656800, hashes = { sha256 = "3aca44c046bd2ed8c90de9cb8427f581c479e594e99b5c0bb19b29c10fd6cb87" } }, - { url = "https://files.pythonhosted.org/packages/c0/d2/21af5c535501a7233e734b8af901574572da66fcc254cb35d0609c9080dd/pywin32-311-cp314-cp314-win_arm64.whl", upload-time = 2025-07-14T20:13:36Z, size = 8932540, hashes = { sha256 = "a508e2d9025764a8270f93111a970e1d0fbfc33f4153b388bb649b7eec4f9b42" } }, + { url = "https://files.pythonhosted.org/packages/fe/1b/9cfdeac80ee45bebbbcb31f1b7b99a0d81a1c72de48d837be984e0e88b1d/pywin32-312-cp310-cp310-win32.whl", upload-time = 2026-06-04T07:49:14Z, size = 6361387, hashes = { sha256 = "772235332b5d1024c696f11cea1ae4be7930f0a8b894bb43db14e3f435f1ff7e" } }, + { url = "https://files.pythonhosted.org/packages/33/b1/7afc96d041d982c27bc2df6f853d43f01fd273e3d39d04be3647ddeb533d/pywin32-312-cp310-cp310-win_amd64.whl", upload-time = 2026-06-04T07:49:16Z, size = 6926780, hashes = { sha256 = "5dbc35d2b5320dc07f25fa31269cfb767471002b17de5eb067d03da68c7cb2db" } }, + { url = "https://files.pythonhosted.org/packages/ce/3a/4140da9ad54108e517f4a16b2d83da3033e08662144623e1239587cb7db6/pywin32-312-cp310-cp310-win_arm64.whl", upload-time = 2026-06-04T07:49:18Z, size = 4307203, hashes = { sha256 = "3020656e34f1cf7faeb7bccd2b84653a607c6ff0c55ada85e6487d61716deabd" } }, + { url = "https://files.pythonhosted.org/packages/1f/f5/10a6e845a00fc5e7afd0a988b744f403d4d57162a28d160a093c4d9322f0/pywin32-312-cp311-cp311-win32.whl", upload-time = 2026-06-04T07:49:21Z, size = 6362659, hashes = { sha256 = "17948aeadbdb091f0ced6ef0841620794e68327b94ee415571c1203594b7215c" } }, + { url = "https://files.pythonhosted.org/packages/35/c4/dcd2d62b5944b6d5db53413a5899016ccd57ffcb7278f3f81655d25d2027/pywin32-312-cp311-cp311-win_amd64.whl", upload-time = 2026-06-04T07:49:23Z, size = 6928825, hashes = { sha256 = "d11417d84412f859b722fad0841b3614459ed0047f7542d8362e77884f6b6e8a" } }, + { url = "https://files.pythonhosted.org/packages/b7/56/3cbb433fe4501cdba2eb9040f56a4e1a8243faa4186b25295564d1a7a79d/pywin32-312-cp311-cp311-win_arm64.whl", upload-time = 2026-06-04T07:49:26Z, size = 6721875, hashes = { sha256 = "b2200a054ca6d6625c4842fc56a4976a4b47f96b73dbe5538c3f813a80359f47" } }, + { url = "https://files.pythonhosted.org/packages/83/ff/32aa7d2ed0ab12b323aaa64f9b75e6ad4f8fd09f9ccfc28c79414d46838d/pywin32-312-cp312-cp312-win32.whl", upload-time = 2026-06-04T07:49:28Z, size = 6371877, hashes = { sha256 = "dab4f65ac9c4e48400a2a0530c46c3c579cd5905ecd11b80692373915269208b" } }, + { url = "https://files.pythonhosted.org/packages/03/d9/77040d3b43df3f3be32ea289433d660d2727f5ba327bc73be835127d9d60/pywin32-312-cp312-cp312-win_amd64.whl", upload-time = 2026-06-04T07:49:31Z, size = 6914841, hashes = { sha256 = "b457f6d628a47e8a7346ce22acb7e1a46a4a78b52e1d17e1af56871bd19a93bc" } }, + { url = "https://files.pythonhosted.org/packages/e3/cc/7b1ec671775756020a0ee7f4feeaf3c568f0ab86bd3900088cf986937a92/pywin32-312-cp312-cp312-win_arm64.whl", upload-time = 2026-06-04T07:49:34Z, size = 6727901, hashes = { sha256 = "6017c58e12f6809fbb0555b75df144c2922a9ffd18e4b9b5afa863b6c1a9d950" } }, + { url = "https://files.pythonhosted.org/packages/2d/41/12fbfd7f36ed2146d8bc9de96c2741296bf0d490b98508496cff322e274c/pywin32-312-cp313-cp313-win32.whl", upload-time = 2026-06-04T07:49:36Z, size = 6370184, hashes = { sha256 = "7a27df850933d16a8eabfbaeb73d52b273e2da667f80d70b01a89d1f6828d02c" } }, + { url = "https://files.pythonhosted.org/packages/ba/db/36a78e3403099d31d9746d13fdcde5accc43c1155f375a34d15983a479a7/pywin32-312-cp313-cp313-win_amd64.whl", upload-time = 2026-06-04T07:49:38Z, size = 6914298, hashes = { sha256 = "c53e878d15a1c44788082bfe712a905433473aa38f86375b7cf8b45e3acbaaf9" } }, + { url = "https://files.pythonhosted.org/packages/84/37/c1697194092b76de9ed47ca124323f02c57ffc8a45c06f88a3d5acaf01eb/pywin32-312-cp313-cp313-win_arm64.whl", upload-time = 2026-06-04T07:49:41Z, size = 6727640, hashes = { sha256 = "59aba5d5940842075343a5ddc6b11f1cdf0d1567fe745290359dfbcc7c2eb831" } }, + { url = "https://files.pythonhosted.org/packages/fc/2b/1f3cded5822fd49c02f40544cbb5f58c7cfd6b1694869fd476cb6170ee97/pywin32-312-cp314-cp314-win32.whl", upload-time = 2026-06-04T07:49:43Z, size = 6468928, hashes = { sha256 = "a77a90fbb6881238d2ca9c6fd797b25817f3768fe78d214a90137ff055a75f5b" } }, + { url = "https://files.pythonhosted.org/packages/21/82/3bf86d2e2808902013132e1ce905a7da0da53790f3836c64bf44d55e24f3/pywin32-312-cp314-cp314-win_amd64.whl", upload-time = 2026-06-04T07:49:45Z, size = 7024157, hashes = { sha256 = "a4dd3a848290ef724347b19f301045831d8e802fa4464f491b98b1e0a081432e" } }, + { url = "https://files.pythonhosted.org/packages/a4/0e/73f6d6800b4f27655abd9e9f6aaeaefcddb2b946e4674efa2bab184a7f7b/pywin32-312-cp314-cp314-win_arm64.whl", upload-time = 2026-06-04T07:49:47Z, size = 6839598, hashes = { sha256 = "9fce94568364e0155e6dfb781ac5d95903be8baf28670632beab1b523f300daa" } }, + { url = "https://files.pythonhosted.org/packages/eb/61/caa39686032d2ebdd04ff0ab5cbe163126c0066d98e00c9018646e42393b/pywin32-312-cp315-cp315-win32.whl", upload-time = 2026-06-04T07:49:50Z, size = 6471159, hashes = { sha256 = "5c1fbe4a937a73ae9297384a3da38518cbc694c68ad8a809b2e19acd350f03ed" } }, + { url = "https://files.pythonhosted.org/packages/0f/cd/7e1de64a4a6f69c04214169657ccab0d93a670ea50e35eb8f489d7378249/pywin32-312-cp315-cp315-win_amd64.whl", upload-time = 2026-06-04T07:49:54Z, size = 7025293, hashes = { sha256 = "c2f03a0f73f804a13c2735b99392b0cd426bb4f2c4d0178e5ac966a0f21618d5" } }, + { url = "https://files.pythonhosted.org/packages/23/ed/4532e9388e65fa16b46776ef47ad631a64eda1631884488af707666350ed/pywin32-312-cp315-cp315-win_arm64.whl", upload-time = 2026-06-04T07:49:57Z, size = 6840337, hashes = { sha256 = "a8597d28f267b39074aef51fa593530082b39cbe5a074226096857b1fed2dfb9" } }, ] [[packages]] @@ -471,36 +549,36 @@ wheels = [ [[packages]] name = "ruff" -version = "0.15.13" +version = "0.16.4" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/24/21/a7d5c126d5b557715ef81098f3db2fe20f622a039ff2e626af28d674ab80/ruff-0.15.13.tar.gz", upload-time = 2026-05-14T13:44:37Z, size = 4678180, hashes = { sha256 = "f9d89f17f7ba7fb2ed42921f0df75da797a9a5d71bc39049e2c687cf2baf44b7" } } +sdist = { url = "https://files.pythonhosted.org/packages/00/8f/d8074b1f25e003164087a8bfe79a0f1a3945135764dbb6aaab04103dcaf9/ruff-0.16.4.tar.gz", upload-time = 2026-08-20T17:43:59Z, size = 4899731, hashes = { sha256 = "13171aa9d9af2240ee3504e639de73122c67e74036de5ba2e1d01422cd17e3dc" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/c6/61/11d458dc6ac22504fd8e237b29dfd40504c7fbbcc8930402cfe51a8e63ed/ruff-0.15.13-py3-none-linux_armv6l.whl", upload-time = 2026-05-14T13:44:18Z, size = 10738279, hashes = { sha256 = "444b580fc72fd6887e650acd3e575e18cdc79dbcf42fb4030b491057921f61f8" } }, - { url = "https://files.pythonhosted.org/packages/86/ca/caa871ee7be718c45256fada4e16a218ee3e33f0c4a46b729a60a24912e6/ruff-0.15.13-py3-none-macosx_10_12_x86_64.whl", upload-time = 2026-05-14T13:44:06Z, size = 11124798, hashes = { sha256 = "6590d009e7cb7ebf36f83dbdd44a3fa48a0994ff6f1cdc1b08006abe58f98dc7" } }, - { url = "https://files.pythonhosted.org/packages/d3/19/43f5f2e568dddde567fc41f8471f9432c09563e19d3e617a48cfa52f8f0a/ruff-0.15.13-py3-none-macosx_11_0_arm64.whl", upload-time = 2026-05-14T13:44:04Z, size = 10460761, hashes = { sha256 = "1c26d2f66163deeb6e08d8b39fbbe983ce3c71cea06a6d7591cfd1421793c629" } }, - { url = "https://files.pythonhosted.org/packages/99/df/cf938cd6de3003178f03ad7c1ea2a6c099468c03a35037985070b37e76be/ruff-0.15.13-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", upload-time = 2026-05-14T13:44:25Z, size = 10804451, hashes = { sha256 = "9dbd6f94b434f896308e4d57fb7bfde0d02b99f7a64b3bdab0fdfa6a864203a5" } }, - { url = "https://files.pythonhosted.org/packages/c7/7d/5d0973129b154ded2225729169d7068f26b467760b146493fde138415f23/ruff-0.15.13-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", upload-time = 2026-05-14T13:44:08Z, size = 10534285, hashes = { sha256 = "bf3259f3be4d181bda591da5db2571aed6853c6a048157756448020bc6c5cd22" } }, - { url = "https://files.pythonhosted.org/packages/1f/e3/6b999bbc66cd51e5f073842bc2a3995e99c5e0e72e16b15e7261f7abf57a/ruff-0.15.13-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", upload-time = 2026-05-14T13:44:11Z, size = 11312063, hashes = { sha256 = "ae9c17e5eb4430c154e76abc25d79a318190f5a997f38fb6b114416c5319ffc9" } }, - { url = "https://files.pythonhosted.org/packages/af/5a/642639e9f5db04f1e97fbd6e091c6fd20725bdf072fb114d00eefb9e6eb8/ruff-0.15.13-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", upload-time = 2026-05-14T13:44:01Z, size = 12183079, hashes = { sha256 = "2e2e39bff6c341f4b577a21b801326fab0b11847f48fcaa83f00a113c9b3cb55" } }, - { url = "https://files.pythonhosted.org/packages/19/4c/7585735f6b53b0f12de13618b2f7d250a844f018822efc899df2e7b8295f/ruff-0.15.13-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", upload-time = 2026-05-14T13:43:59Z, size = 11440833, hashes = { sha256 = "e8d9a8e08013542e94d3220bc5b62cc3e5ef87c5f74bff367d3fac14fab013e6" } }, - { url = "https://files.pythonhosted.org/packages/e8/31/bf1a0803d077e679cfeee5f2f67290a0fa79c7385b5d9a8c17b9db2c48f0/ruff-0.15.13-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", upload-time = 2026-05-14T13:44:27Z, size = 11434486, hashes = { sha256 = "cc411dfebe5eebe55ce041c6ae080eb7668955e866daa2fbb16692a784f1c4ca" } }, - { url = "https://files.pythonhosted.org/packages/e1/4e/62c9b999875d4f14db80f277c030578f5e249c9852d65b7ac7ad0b43c041/ruff-0.15.13-py3-none-manylinux_2_31_riscv64.whl", upload-time = 2026-05-14T13:44:13Z, size = 11385189, hashes = { sha256 = "768494eb08b9cee54e2fd27969966f74db5a57f6eaa7a90fcb3306af34dfc4bd" } }, - { url = "https://files.pythonhosted.org/packages/fc/89/7e959047a104df3eb12863447c110140191fc5b6c4f379ea2e803fcdb0e4/ruff-0.15.13-py3-none-musllinux_1_2_aarch64.whl", upload-time = 2026-05-14T13:43:56Z, size = 10781380, hashes = { sha256 = "fb75f9a3a7e42ffe117d734494e6c5e5cb3565d66e12612cb63d0e572a41a5b6" } }, - { url = "https://files.pythonhosted.org/packages/ff/52/5fd18f3b88cab63e88aa11516b3b4e1e5f720e5c330f8dbe5c26210f41f8/ruff-0.15.13-py3-none-musllinux_1_2_armv7l.whl", upload-time = 2026-05-14T13:44:20Z, size = 10540605, hashes = { sha256 = "8cb74dd33bb2f6613faf7fc03b660053b5ac4f80e706d5788c6335e2a8048d51" } }, - { url = "https://files.pythonhosted.org/packages/e8/e0/9e35f338990d3e41a82875ff7053ffe97541dae81c9d02143177f381d572/ruff-0.15.13-py3-none-musllinux_1_2_i686.whl", upload-time = 2026-05-14T13:44:16Z, size = 11036554, hashes = { sha256 = "7ef823f817fcd191dc934e984be9cf4094f808effa16f2542ad8e821ba02bbf2" } }, - { url = "https://files.pythonhosted.org/packages/c2/13/070fb048c24080fba188f66371e2a92785be257ad02242066dc7255ac6e9/ruff-0.15.13-py3-none-musllinux_1_2_x86_64.whl", upload-time = 2026-05-14T13:44:22Z, size = 11528133, hashes = { sha256 = "f345a13937bd7f09f6f5d19fa0721b0c103e00e7f62bc67089a8e5e037719e0b" } }, - { url = "https://files.pythonhosted.org/packages/6b/8c/b1e1666aef7fc6555094d73ae6cd981701781ae85b97ceefc0eebd0b4668/ruff-0.15.13-py3-none-win32.whl", upload-time = 2026-05-14T13:44:35Z, size = 10721455, hashes = { sha256 = "4044f94208b3b05ba0fc4a4abd0558cf4d6459bd18325eead7fd8cc66f909b41" } }, - { url = "https://files.pythonhosted.org/packages/ab/a6/870a3e8a50590bb92be184ad928c2922f088b00d9dc5c5ec7b924ee08c22/ruff-0.15.13-py3-none-win_amd64.whl", upload-time = 2026-05-14T13:44:30Z, size = 11900409, hashes = { sha256 = "7064884d442b7d477b4e7473d12da7f08851d2b1982763c5d3f388a19468a1a4" } }, - { url = "https://files.pythonhosted.org/packages/9b/36/9c015cd052fca743dae8cb2aeb16b551444787467db42ceab0fc968865af/ruff-0.15.13-py3-none-win_arm64.whl", upload-time = 2026-05-14T13:44:33Z, size = 11179336, hashes = { sha256 = "2471da9bd1068c8c064b5fd9c0c4b6dddffd6369cb1cd68b29993b1709ff1b21" } }, + { url = "https://files.pythonhosted.org/packages/ff/80/779895ef584e089d22f2c6df0d0e99a65ec2df0805f1fffd439415b8c1f0/ruff-0.16.4-py3-none-linux_armv6l.whl", upload-time = 2026-08-20T17:43:16Z, size = 10006909, hashes = { sha256 = "df4075f71ddac40b9934af60c3ec8a53047dd5a5fdc43224e6e4e8e9a27cb6f7" } }, + { url = "https://files.pythonhosted.org/packages/a9/e6/f553199b5e8927a05cb5c422d921fd0656b29ab976e91c44802107c6b0da/ruff-0.16.4-py3-none-macosx_10_12_x86_64.whl", upload-time = 2026-08-20T17:43:19Z, size = 10240201, hashes = { sha256 = "0c95538517af68004306b0fb3214ff2f2af67a65092aee77cd9eb86db6656604" } }, + { url = "https://files.pythonhosted.org/packages/1c/70/4a6dc4bb34da4dee35e30f09bbd1bfbdd26f33b62fb9b8df31f08a199cd2/ruff-0.16.4-py3-none-macosx_11_0_arm64.whl", upload-time = 2026-08-20T17:43:21Z, size = 9835122, hashes = { sha256 = "963f83df8e69e575b64d67dd447ebbc917db41a14bf38d4593a4183e7aaa8255" } }, + { url = "https://files.pythonhosted.org/packages/24/12/c6e22d686372c15bcb7af99831f1a1be96df696491babf4f24e4f942c527/ruff-0.16.4-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", upload-time = 2026-08-20T17:43:24Z, size = 9977162, hashes = { sha256 = "32a5057c7ff3f6e6480a48fccfb3a412a690f48a3d03ac5cf08177d6c2da3ade" } }, + { url = "https://files.pythonhosted.org/packages/46/49/72b10ec912f5ab5854992eaf7aa7cd36729b6937d9dc4e0fb41b3bf428ec/ruff-0.16.4-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", upload-time = 2026-08-20T17:43:26Z, size = 9829789, hashes = { sha256 = "b3dce8d9b0c57c265b91885a66a567d8ea1372e8eb4e250fa8e5e3f579e99cff" } }, + { url = "https://files.pythonhosted.org/packages/fa/80/0f30e32e7f6ee26edc39075502db9d368d788a44a79b55f763eb4ab03796/ruff-0.16.4-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", upload-time = 2026-08-20T17:43:29Z, size = 10527949, hashes = { sha256 = "7dc651db49283c69f8e72c834eec4fe5573e4c646856aebece0ce385dceb2a80" } }, + { url = "https://files.pythonhosted.org/packages/52/3d/86e8ad3542169e56cac3859a343afdb9df2ad54d35a59ce1e67baee83421/ruff-0.16.4-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", upload-time = 2026-08-20T17:43:31Z, size = 11333695, hashes = { sha256 = "3817b87dbcabc92f13b05019257c5b89b5b4d51b5fb20f56fb5235ceb723cd07" } }, + { url = "https://files.pythonhosted.org/packages/d0/16/481c29b380c20a0054a8261066665e1b3488e23636c49d0a43e75975b9bb/ruff-0.16.4-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", upload-time = 2026-08-20T17:43:34Z, size = 10727741, hashes = { sha256 = "e9fce1499134b2c8c68e5166f95705a5812062bb93aacc5f9873bb1a27084bc7" } }, + { url = "https://files.pythonhosted.org/packages/5e/b6/56bc0b8cf45b54b28b3a5e6381c8945d51b5b18adf659454c32295209a31/ruff-0.16.4-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", upload-time = 2026-08-20T17:43:37Z, size = 10286522, hashes = { sha256 = "f2d812e482f5a7e02eee26cd73d2a37ebbdf47d795ea63ba1b89110ae93e9fb3" } }, + { url = "https://files.pythonhosted.org/packages/e8/8b/b345b4fb110f2fbe2bd31eabd271e5e8b3b7e4ee6c0e02f2dc6be78db000/ruff-0.16.4-py3-none-manylinux_2_31_riscv64.whl", upload-time = 2026-08-20T17:43:39Z, size = 10584182, hashes = { sha256 = "6baaf984aa7976edf93d3b627fe2d1d22ee94bbca05fa6f90fc76d73924e3454" } }, + { url = "https://files.pythonhosted.org/packages/29/e5/827b34041c35f58774a9681a4213994c164fc987800f4dddabcf451da0bf/ruff-0.16.4-py3-none-musllinux_1_2_aarch64.whl", upload-time = 2026-08-20T17:43:42Z, size = 10134195, hashes = { sha256 = "bdfcf0b28662eb890372d50f92c283bb94e67e7635ed93c7fd533970acff7b2b" } }, + { url = "https://files.pythonhosted.org/packages/0f/10/d0bffcdd6729b87afc82ba0ef377173356a7dc8e972f5179968cf2fdf98c/ruff-0.16.4-py3-none-musllinux_1_2_armv7l.whl", upload-time = 2026-08-20T17:43:44Z, size = 9825821, hashes = { sha256 = "b66b02cb9b04f537643cadf5768e5f98dc461890d530cb67113d71c8c76e605d" } }, + { url = "https://files.pythonhosted.org/packages/f5/32/0db2a863b796ca62d83e92a07a3ccf00921b14db02059347576a2fda3d4b/ruff-0.16.4-py3-none-musllinux_1_2_i686.whl", upload-time = 2026-08-20T17:43:46Z, size = 10267658, hashes = { sha256 = "8528bf9a4b291a60bf02ea453511e8ce6215bd2b982ee80405b66b008b6c30a0" } }, + { url = "https://files.pythonhosted.org/packages/b2/a0/fbdeb59e48c6261f523e56c8f12e9c08fbe693786595cc7e3959207a9232/ruff-0.16.4-py3-none-musllinux_1_2_x86_64.whl", upload-time = 2026-08-20T17:43:49Z, size = 10697071, hashes = { sha256 = "fbd85d2875fdd67e833213a651f613bbf25303abf6aa822a5121f4531195678d" } }, + { url = "https://files.pythonhosted.org/packages/aa/28/0c6dd865859c6d17bc8ccc34cb72b0e02d6c7eb25e8a1e22b5bea681e2c0/ruff-0.16.4-py3-none-win32.whl", upload-time = 2026-08-20T17:43:52Z, size = 10021687, hashes = { sha256 = "312769988007aaeb8e189b443ccdd03c0e6374489e053467be6d96518ebff76e" } }, + { url = "https://files.pythonhosted.org/packages/a3/03/e724450f621698117f9aa6dd241c94d0274ae96781378dc86745ae29f0e7/ruff-0.16.4-py3-none-win_amd64.whl", upload-time = 2026-08-20T17:43:54Z, size = 10567657, hashes = { sha256 = "05d9d27a18c4bcbefada602480ec9e01e0bc949d432e0ced5df77edac195919c" } }, + { url = "https://files.pythonhosted.org/packages/0e/fe/da8b9e1347696bb22120b77280ec5ce25d500ca5cb39d5ad6e5c18de19c1/ruff-0.16.4-py3-none-win_arm64.whl", upload-time = 2026-08-20T17:43:57Z, size = 10451579, hashes = { sha256 = "a3a61621c9b6f6a89573e938a080e648f1695baa3f58570a3a707bc51ff65a21" } }, ] [[packages]] name = "testcontainers" -version = "4.14.2" +version = "4.15.0" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/ca/ac/a597c3a0e02b26cbed6dd07df68be1e57684766fd1c381dee9b170a99690/testcontainers-4.14.2.tar.gz", upload-time = 2026-03-18T05:19:16Z, size = 166841, hashes = { sha256 = "1340ccf16fe3acd9389a6c9e1d9ab21d9fe99a8afdf8165f89c3e69c1967d239" } } +sdist = { url = "https://files.pythonhosted.org/packages/4b/13/2cc466bddf26d0085f30a2b2bd56b7f8708b54a54db833eec97c5c69129b/testcontainers-4.15.0.tar.gz", upload-time = 2026-07-24T23:08:01Z, size = 95340, hashes = { sha256 = "085cde086337632e19002719460b7b80bbab2bdd51bb3ea04f77d0de96504706" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/13/2d/26b8b30067d94339afee62c3edc9b803a6eb9332f521ba77d8aaab5de873/testcontainers-4.14.2-py3-none-any.whl", upload-time = 2026-03-18T05:19:15Z, size = 125712, hashes = { sha256 = "0d0522c3cd8f8d9627cda41f7a6b51b639fa57bdc492923c045117933c668d68" } }, + { url = "https://files.pythonhosted.org/packages/00/7e/424aac8b355597835deb333e757a0e94b5ccf38ad00f07fe6ed1f4e17c88/testcontainers-4.15.0-py3-none-any.whl", upload-time = 2026-07-24T23:08:00Z, size = 160771, hashes = { sha256 = "8796c14e76604031ad39cf0ed3b8e9806283a1fbf5270965c2b1c594caa31b74" } }, ] [[packages]] @@ -560,11 +638,11 @@ wheels = [ [[packages]] name = "typing-extensions" -version = "4.15.0" +version = "4.16.0" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/72/94/1a15dd82efb362ac84269196e94cf00f187f7ed21c242792a923cdb1c61f/typing_extensions-4.15.0.tar.gz", upload-time = 2025-08-25T13:49:26Z, size = 109391, hashes = { sha256 = "0cea48d173cc12fa28ecabc3b837ea3cf6f38c6d1136f85cbaaf598984861466" } } +sdist = { url = "https://files.pythonhosted.org/packages/f6/cc/6253133b5bb138fc3306cebfbda2c520f545d36b5be2c7255cc528bb45d6/typing_extensions-4.16.0.tar.gz", upload-time = 2026-07-02T08:40:05Z, size = 113555, hashes = { sha256 = "dc983d19a509c94dba722ee6abd33940f7c05a89e243c47e907eb4db6f1a43e5" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/18/67/36e9267722cc04a6b9f15c7f3441c2363321a3ea07da7ae0c0707beb2a9c/typing_extensions-4.15.0-py3-none-any.whl", upload-time = 2025-08-25T13:49:24Z, size = 44614, hashes = { sha256 = "f0fa19c6845758ab08074a0cfa8b7aecb71c999ca73d62883bc25cc018c4e548" } }, + { url = "https://files.pythonhosted.org/packages/49/d3/b8441a820a491ddfc024b0b0cf0393375b75ea13866d9c66727e54c2fc80/typing_extensions-4.16.0-py3-none-any.whl", upload-time = 2026-07-02T08:40:04Z, size = 45571, hashes = { sha256 = "481caa481374e813c1b176ada14e97f1f67a4539ce9cfeb3f350d78d6370c2e8" } }, ] [[packages]] @@ -578,86 +656,86 @@ wheels = [ [[packages]] name = "wrapt" -version = "2.1.2" +version = "2.3.0" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/2e/64/925f213fdcbb9baeb1530449ac71a4d57fc361c053d06bf78d0c5c7cd80c/wrapt-2.1.2.tar.gz", upload-time = 2026-03-06T02:53:25Z, size = 81678, hashes = { sha256 = "3996a67eecc2c68fd47b4e3c564405a5777367adfd9b8abb58387b63ee83b21e" } } +sdist = { url = "https://files.pythonhosted.org/packages/2b/b0/c1f5a970721f06b85c0cd5142e0ff8fe067708abd779b0c4f4be7d61d09f/wrapt-2.3.0.tar.gz", upload-time = 2026-07-28T06:06:14Z, size = 131509, hashes = { sha256 = "681a2d0eefd721998f90642762b8e75c2159ec531b20ad5e437245ea7b06a107" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/da/d2/387594fb592d027366645f3d7cc9b4d7ca7be93845fbaba6d835a912ef3c/wrapt-2.1.2-cp310-cp310-macosx_10_9_x86_64.whl", upload-time = 2026-03-06T02:52:40Z, size = 60669, hashes = { sha256 = "4b7a86d99a14f76facb269dc148590c01aaf47584071809a70da30555228158c" } }, - { url = "https://files.pythonhosted.org/packages/c9/18/3f373935bc5509e7ac444c8026a56762e50c1183e7061797437ca96c12ce/wrapt-2.1.2-cp310-cp310-macosx_11_0_arm64.whl", upload-time = 2026-03-06T02:54:21Z, size = 61603, hashes = { sha256 = "a819e39017f95bf7aede768f75915635aa8f671f2993c036991b8d3bfe8dbb6f" } }, - { url = "https://files.pythonhosted.org/packages/c2/7a/32758ca2853b07a887a4574b74e28843919103194bb47001a304e24af62f/wrapt-2.1.2-cp310-cp310-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-03-06T02:53:54Z, size = 113632, hashes = { sha256 = "5681123e60aed0e64c7d44f72bbf8b4ce45f79d81467e2c4c728629f5baf06eb" } }, - { url = "https://files.pythonhosted.org/packages/1d/d5/eeaa38f670d462e97d978b3b0d9ce06d5b91e54bebac6fbed867809216e7/wrapt-2.1.2-cp310-cp310-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-03-06T02:54:53Z, size = 115644, hashes = { sha256 = "2b8b28e97a44d21836259739ae76284e180b18abbb4dcfdff07a415cf1016c3e" } }, - { url = "https://files.pythonhosted.org/packages/e3/09/2a41506cb17affb0bdf9d5e2129c8c19e192b388c4c01d05e1b14db23c00/wrapt-2.1.2-cp310-cp310-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-03-06T02:54:43Z, size = 112016, hashes = { sha256 = "cef91c95a50596fcdc31397eb6955476f82ae8a3f5a8eabdc13611b60ee380ba" } }, - { url = "https://files.pythonhosted.org/packages/64/15/0e6c3f5e87caadc43db279724ee36979246d5194fa32fed489c73643ba59/wrapt-2.1.2-cp310-cp310-musllinux_1_2_aarch64.whl", upload-time = 2026-03-06T02:54:29Z, size = 114823, hashes = { sha256 = "dad63212b168de8569b1c512f4eac4b57f2c6934b30df32d6ee9534a79f1493f" } }, - { url = "https://files.pythonhosted.org/packages/56/b2/0ad17c8248f4e57bedf44938c26ec3ee194715f812d2dbbd9d7ff4be6c06/wrapt-2.1.2-cp310-cp310-musllinux_1_2_riscv64.whl", upload-time = 2026-03-06T02:54:02Z, size = 111244, hashes = { sha256 = "d307aa6888d5efab2c1cde09843d48c843990be13069003184b67d426d145394" } }, - { url = "https://files.pythonhosted.org/packages/ff/04/bcdba98c26f2c6522c7c09a726d5d9229120163493620205b2f76bd13c01/wrapt-2.1.2-cp310-cp310-musllinux_1_2_x86_64.whl", upload-time = 2026-03-06T02:54:12Z, size = 113307, hashes = { sha256 = "c87cf3f0c85e27b3ac7d9ad95da166bf8739ca215a8b171e8404a2d739897a45" } }, - { url = "https://files.pythonhosted.org/packages/0e/1b/5e2883c6bc14143924e465a6fc5a92d09eeabe35310842a481fb0581f832/wrapt-2.1.2-cp310-cp310-win32.whl", upload-time = 2026-03-06T02:54:26Z, size = 57986, hashes = { sha256 = "d1c5fea4f9fe3762e2b905fdd67df51e4be7a73b7674957af2d2ade71a5c075d" } }, - { url = "https://files.pythonhosted.org/packages/42/5a/4efc997bccadd3af5749c250b49412793bc41e13a83a486b2b54a33e240c/wrapt-2.1.2-cp310-cp310-win_amd64.whl", upload-time = 2026-03-06T02:54:18Z, size = 60336, hashes = { sha256 = "d8f7740e1af13dff2684e4d56fe604a7e04d6c94e737a60568d8d4238b9a0c71" } }, - { url = "https://files.pythonhosted.org/packages/c1/f5/a2bb833e20181b937e87c242645ed5d5aa9c373006b0467bfe1a35c727d0/wrapt-2.1.2-cp310-cp310-win_arm64.whl", upload-time = 2026-03-06T02:53:51Z, size = 58757, hashes = { sha256 = "1c6cc827c00dc839350155f316f1f8b4b0c370f52b6a19e782e2bda89600c7dc" } }, - { url = "https://files.pythonhosted.org/packages/c7/81/60c4471fce95afa5922ca09b88a25f03c93343f759aae0f31fb4412a85c7/wrapt-2.1.2-cp311-cp311-macosx_10_9_x86_64.whl", upload-time = 2026-03-06T02:52:58Z, size = 60666, hashes = { sha256 = "96159a0ee2b0277d44201c3b5be479a9979cf154e8c82fa5df49586a8e7679bb" } }, - { url = "https://files.pythonhosted.org/packages/6b/be/80e80e39e7cb90b006a0eaf11c73ac3a62bbfb3068469aec15cc0bc795de/wrapt-2.1.2-cp311-cp311-macosx_11_0_arm64.whl", upload-time = 2026-03-06T02:53:00Z, size = 61601, hashes = { sha256 = "98ba61833a77b747901e9012072f038795de7fc77849f1faa965464f3f87ff2d" } }, - { url = "https://files.pythonhosted.org/packages/b0/be/d7c88cd9293c859fc74b232abdc65a229bb953997995d6912fc85af18323/wrapt-2.1.2-cp311-cp311-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-03-06T02:52:44Z, size = 114057, hashes = { sha256 = "767c0dbbe76cae2a60dd2b235ac0c87c9cccf4898aef8062e57bead46b5f6894" } }, - { url = "https://files.pythonhosted.org/packages/ea/25/36c04602831a4d685d45a93b3abea61eca7fe35dab6c842d6f5d570ef94a/wrapt-2.1.2-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-03-06T02:54:56Z, size = 116099, hashes = { sha256 = "9c691a6bc752c0cc4711cc0c00896fcd0f116abc253609ef64ef930032821842" } }, - { url = "https://files.pythonhosted.org/packages/5c/4e/98a6eb417ef551dc277bec1253d5246b25003cf36fdf3913b65cb7657a56/wrapt-2.1.2-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-03-06T02:53:52Z, size = 112457, hashes = { sha256 = "f3b7d73012ea75aee5844de58c88f44cf62d0d62711e39da5a82824a7c4626a8" } }, - { url = "https://files.pythonhosted.org/packages/cb/a6/a6f7186a5297cad8ec53fd7578533b28f795fdf5372368c74bd7e6e9841c/wrapt-2.1.2-cp311-cp311-musllinux_1_2_aarch64.whl", upload-time = 2026-03-06T02:53:32Z, size = 115351, hashes = { sha256 = "577dff354e7acd9d411eaf4bfe76b724c89c89c8fc9b7e127ee28c5f7bcb25b6" } }, - { url = "https://files.pythonhosted.org/packages/97/6f/06e66189e721dbebd5cf20e138acc4d1150288ce118462f2fcbff92d38db/wrapt-2.1.2-cp311-cp311-musllinux_1_2_riscv64.whl", upload-time = 2026-03-06T02:53:08Z, size = 111748, hashes = { sha256 = "3d7b6fd105f8b24e5bd23ccf41cb1d1099796524bcc6f7fbb8fe576c44befbc9" } }, - { url = "https://files.pythonhosted.org/packages/ef/43/4808b86f499a51370fbdbdfa6cb91e9b9169e762716456471b619fca7a70/wrapt-2.1.2-cp311-cp311-musllinux_1_2_x86_64.whl", upload-time = 2026-03-06T02:53:02Z, size = 113783, hashes = { sha256 = "866abdbf4612e0b34764922ef8b1c5668867610a718d3053d59e24a5e5fcfc15" } }, - { url = "https://files.pythonhosted.org/packages/91/2c/a3f28b8fa7ac2cefa01cfcaca3471f9b0460608d012b693998cd61ef43df/wrapt-2.1.2-cp311-cp311-win32.whl", upload-time = 2026-03-06T02:53:27Z, size = 57977, hashes = { sha256 = "5a0a0a3a882393095573344075189eb2d566e0fd205a2b6414e9997b1b800a8b" } }, - { url = "https://files.pythonhosted.org/packages/3f/c3/2b1c7bd07a27b1db885a2fab469b707bdd35bddf30a113b4917a7e2139d2/wrapt-2.1.2-cp311-cp311-win_amd64.whl", upload-time = 2026-03-06T02:54:28Z, size = 60336, hashes = { sha256 = "64a07a71d2730ba56f11d1a4b91f7817dc79bc134c11516b75d1921a7c6fcda1" } }, - { url = "https://files.pythonhosted.org/packages/ec/5c/76ece7b401b088daa6503d6264dd80f9a727df3e6042802de9a223084ea2/wrapt-2.1.2-cp311-cp311-win_arm64.whl", upload-time = 2026-03-06T02:53:16Z, size = 58756, hashes = { sha256 = "b89f095fe98bc12107f82a9f7d570dc83a0870291aeb6b1d7a7d35575f55d98a" } }, - { url = "https://files.pythonhosted.org/packages/4c/b6/1db817582c49c7fcbb7df6809d0f515af29d7c2fbf57eb44c36e98fb1492/wrapt-2.1.2-cp312-cp312-macosx_10_13_x86_64.whl", upload-time = 2026-03-06T02:52:45Z, size = 61255, hashes = { sha256 = "ff2aad9c4cda28a8f0653fc2d487596458c2a3f475e56ba02909e950a9efa6a9" } }, - { url = "https://files.pythonhosted.org/packages/a2/16/9b02a6b99c09227c93cd4b73acc3678114154ec38da53043c0ddc1fba0dc/wrapt-2.1.2-cp312-cp312-macosx_11_0_arm64.whl", upload-time = 2026-03-06T02:53:48Z, size = 61848, hashes = { sha256 = "6433ea84e1cfacf32021d2a4ee909554ade7fd392caa6f7c13f1f4bf7b8e8748" } }, - { url = "https://files.pythonhosted.org/packages/af/aa/ead46a88f9ec3a432a4832dfedb84092fc35af2d0ba40cd04aea3889f247/wrapt-2.1.2-cp312-cp312-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-03-06T02:54:40Z, size = 121433, hashes = { sha256 = "c20b757c268d30d6215916a5fa8461048d023865d888e437fab451139cad6c8e" } }, - { url = "https://files.pythonhosted.org/packages/3a/9f/742c7c7cdf58b59085a1ee4b6c37b013f66ac33673a7ef4aaed5e992bc33/wrapt-2.1.2-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-03-06T02:53:26Z, size = 123013, hashes = { sha256 = "79847b83eb38e70d93dc392c7c5b587efe65b3e7afcc167aa8abd5d60e8761c8" } }, - { url = "https://files.pythonhosted.org/packages/e8/44/2c3dd45d53236b7ed7c646fcf212251dc19e48e599debd3926b52310fafb/wrapt-2.1.2-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-03-06T02:53:11Z, size = 117326, hashes = { sha256 = "f8fba1bae256186a83d1875b2b1f4e2d1242e8fac0f58ec0d7e41b26967b965c" } }, - { url = "https://files.pythonhosted.org/packages/74/e2/b17d66abc26bd96f89dec0ecd0ef03da4a1286e6ff793839ec431b9fae57/wrapt-2.1.2-cp312-cp312-musllinux_1_2_aarch64.whl", upload-time = 2026-03-06T02:54:09Z, size = 121444, hashes = { sha256 = "e3d3b35eedcf5f7d022291ecd7533321c4775f7b9cd0050a31a68499ba45757c" } }, - { url = "https://files.pythonhosted.org/packages/3c/62/e2977843fdf9f03daf1586a0ff49060b1b2fc7ff85a7ea82b6217c1ae36e/wrapt-2.1.2-cp312-cp312-musllinux_1_2_riscv64.whl", upload-time = 2026-03-06T02:54:03Z, size = 116237, hashes = { sha256 = "6f2c5390460de57fa9582bc8a1b7a6c86e1a41dfad74c5225fc07044c15cc8d1" } }, - { url = "https://files.pythonhosted.org/packages/88/dd/27fc67914e68d740bce512f11734aec08696e6b17641fef8867c00c949fc/wrapt-2.1.2-cp312-cp312-musllinux_1_2_x86_64.whl", upload-time = 2026-03-06T02:53:20Z, size = 120563, hashes = { sha256 = "7dfa9f2cf65d027b951d05c662cc99ee3bd01f6e4691ed39848a7a5fffc902b2" } }, - { url = "https://files.pythonhosted.org/packages/ec/9f/b750b3692ed2ef4705cb305bd68858e73010492b80e43d2a4faa5573cbe7/wrapt-2.1.2-cp312-cp312-win32.whl", upload-time = 2026-03-06T02:53:37Z, size = 58198, hashes = { sha256 = "eba8155747eb2cae4a0b913d9ebd12a1db4d860fc4c829d7578c7b989bd3f2f0" } }, - { url = "https://files.pythonhosted.org/packages/8e/b2/feecfe29f28483d888d76a48f03c4c4d8afea944dbee2b0cd3380f9df032/wrapt-2.1.2-cp312-cp312-win_amd64.whl", upload-time = 2026-03-06T02:52:47Z, size = 60441, hashes = { sha256 = "1c51c738d7d9faa0b3601708e7e2eda9bf779e1b601dce6c77411f2a1b324a63" } }, - { url = "https://files.pythonhosted.org/packages/44/e1/e328f605d6e208547ea9fd120804fcdec68536ac748987a68c47c606eea8/wrapt-2.1.2-cp312-cp312-win_arm64.whl", upload-time = 2026-03-06T02:53:22Z, size = 58836, hashes = { sha256 = "c8e46ae8e4032792eb2f677dbd0d557170a8e5524d22acc55199f43efedd39bf" } }, - { url = "https://files.pythonhosted.org/packages/4c/7a/d936840735c828b38d26a854e85d5338894cda544cb7a85a9d5b8b9c4df7/wrapt-2.1.2-cp313-cp313-macosx_10_13_x86_64.whl", upload-time = 2026-03-06T02:53:41Z, size = 61259, hashes = { sha256 = "787fd6f4d67befa6fe2abdffcbd3de2d82dfc6fb8a6d850407c53332709d030b" } }, - { url = "https://files.pythonhosted.org/packages/5e/88/9a9b9a90ac8ca11c2fdb6a286cb3a1fc7dd774c00ed70929a6434f6bc634/wrapt-2.1.2-cp313-cp313-macosx_11_0_arm64.whl", upload-time = 2026-03-06T02:52:48Z, size = 61851, hashes = { sha256 = "4bdf26e03e6d0da3f0e9422fd36bcebf7bc0eeb55fdf9c727a09abc6b9fe472e" } }, - { url = "https://files.pythonhosted.org/packages/03/a9/5b7d6a16fd6533fed2756900fc8fc923f678179aea62ada6d65c92718c00/wrapt-2.1.2-cp313-cp313-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-03-06T02:54:14Z, size = 121446, hashes = { sha256 = "bbac24d879aa22998e87f6b3f481a5216311e7d53c7db87f189a7a0266dafffb" } }, - { url = "https://files.pythonhosted.org/packages/45/bb/34c443690c847835cfe9f892be78c533d4f32366ad2888972c094a897e39/wrapt-2.1.2-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-03-06T02:54:10Z, size = 123056, hashes = { sha256 = "16997dfb9d67addc2e3f41b62a104341e80cac52f91110dece393923c0ebd5ca" } }, - { url = "https://files.pythonhosted.org/packages/93/b9/ff205f391cb708f67f41ea148545f2b53ff543a7ac293b30d178af4d2271/wrapt-2.1.2-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-03-06T02:53:03Z, size = 117359, hashes = { sha256 = "162e4e2ba7542da9027821cb6e7c5e068d64f9a10b5f15512ea28e954893a267" } }, - { url = "https://files.pythonhosted.org/packages/1f/3d/1ea04d7747825119c3c9a5e0874a40b33594ada92e5649347c457d982805/wrapt-2.1.2-cp313-cp313-musllinux_1_2_aarch64.whl", upload-time = 2026-03-06T02:53:45Z, size = 121479, hashes = { sha256 = "f29c827a8d9936ac320746747a016c4bc66ef639f5cd0d32df24f5eacbf9c69f" } }, - { url = "https://files.pythonhosted.org/packages/78/cc/ee3a011920c7a023b25e8df26f306b2484a531ab84ca5c96260a73de76c0/wrapt-2.1.2-cp313-cp313-musllinux_1_2_riscv64.whl", upload-time = 2026-03-06T02:54:46Z, size = 116271, hashes = { sha256 = "a9dd9813825f7ecb018c17fd147a01845eb330254dff86d3b5816f20f4d6aaf8" } }, - { url = "https://files.pythonhosted.org/packages/98/fd/e5ff7ded41b76d802cf1191288473e850d24ba2e39a6ec540f21ae3b57cb/wrapt-2.1.2-cp313-cp313-musllinux_1_2_x86_64.whl", upload-time = 2026-03-06T02:52:50Z, size = 120573, hashes = { sha256 = "6f8dbdd3719e534860d6a78526aafc220e0241f981367018c2875178cf83a413" } }, - { url = "https://files.pythonhosted.org/packages/47/c5/242cae3b5b080cd09bacef0591691ba1879739050cc7c801ff35c8886b66/wrapt-2.1.2-cp313-cp313-win32.whl", upload-time = 2026-03-06T02:53:47Z, size = 58205, hashes = { sha256 = "5c35b5d82b16a3bc6e0a04349b606a0582bc29f573786aebe98e0c159bc48db6" } }, - { url = "https://files.pythonhosted.org/packages/12/69/c358c61e7a50f290958809b3c61ebe8b3838ea3e070d7aac9814f95a0528/wrapt-2.1.2-cp313-cp313-win_amd64.whl", upload-time = 2026-03-06T02:53:30Z, size = 60452, hashes = { sha256 = "f8bc1c264d8d1cf5b3560a87bbdd31131573eb25f9f9447bb6252b8d4c44a3a1" } }, - { url = "https://files.pythonhosted.org/packages/8e/66/c8a6fcfe321295fd8c0ab1bd685b5a01462a9b3aa2f597254462fc2bc975/wrapt-2.1.2-cp313-cp313-win_arm64.whl", upload-time = 2026-03-06T02:52:52Z, size = 58842, hashes = { sha256 = "3beb22f674550d5634642c645aba4c72a2c66fb185ae1aebe1e955fae5a13baf" } }, - { url = "https://files.pythonhosted.org/packages/da/55/9c7052c349106e0b3f17ae8db4b23a691a963c334de7f9dbd60f8f74a831/wrapt-2.1.2-cp313-cp313t-macosx_10_13_x86_64.whl", upload-time = 2026-03-06T02:53:19Z, size = 63075, hashes = { sha256 = "0fc04bc8664a8bc4c8e00b37b5355cffca2535209fba1abb09ae2b7c76ddf82b" } }, - { url = "https://files.pythonhosted.org/packages/09/a8/ce7b4006f7218248dd71b7b2b732d0710845a0e49213b18faef64811ffef/wrapt-2.1.2-cp313-cp313t-macosx_11_0_arm64.whl", upload-time = 2026-03-06T02:54:33Z, size = 63719, hashes = { sha256 = "a9b9d50c9af998875a1482a038eb05755dfd6fe303a313f6a940bb53a83c3f18" } }, - { url = "https://files.pythonhosted.org/packages/e4/e5/2ca472e80b9e2b7a17f106bb8f9df1db11e62101652ce210f66935c6af67/wrapt-2.1.2-cp313-cp313t-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-03-06T02:52:42Z, size = 152643, hashes = { sha256 = "2d3ff4f0024dd224290c0eabf0240f1bfc1f26363431505fb1b0283d3b08f11d" } }, - { url = "https://files.pythonhosted.org/packages/36/42/30f0f2cefca9d9cbf6835f544d825064570203c3e70aa873d8ae12e23791/wrapt-2.1.2-cp313-cp313t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-03-06T02:54:25Z, size = 158805, hashes = { sha256 = "3278c471f4468ad544a691b31bb856374fbdefb7fee1a152153e64019379f015" } }, - { url = "https://files.pythonhosted.org/packages/bb/67/d08672f801f604889dcf58f1a0b424fe3808860ede9e03affc1876b295af/wrapt-2.1.2-cp313-cp313t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-03-06T02:53:57Z, size = 145990, hashes = { sha256 = "a8914c754d3134a3032601c6984db1c576e6abaf3fc68094bb8ab1379d75ff92" } }, - { url = "https://files.pythonhosted.org/packages/68/a7/fd371b02e73babec1de6ade596e8cd9691051058cfdadbfd62a5898f3295/wrapt-2.1.2-cp313-cp313t-musllinux_1_2_aarch64.whl", upload-time = 2026-03-06T02:54:55Z, size = 155670, hashes = { sha256 = "ff95d4264e55839be37bafe1536db2ab2de19da6b65f9244f01f332b5286cfbf" } }, - { url = "https://files.pythonhosted.org/packages/86/2d/9fe0095dfdb621009f40117dcebf41d7396c2c22dca6eac779f4c007b86c/wrapt-2.1.2-cp313-cp313t-musllinux_1_2_riscv64.whl", upload-time = 2026-03-06T02:54:24Z, size = 144357, hashes = { sha256 = "76405518ca4e1b76fbb1b9f686cff93aebae03920cc55ceeec48ff9f719c5f67" } }, - { url = "https://files.pythonhosted.org/packages/0e/b6/ec7b4a254abbe4cde9fa15c5d2cca4518f6b07d0f1b77d4ee9655e30280e/wrapt-2.1.2-cp313-cp313t-musllinux_1_2_x86_64.whl", upload-time = 2026-03-06T02:53:31Z, size = 150269, hashes = { sha256 = "c0be8b5a74c5824e9359b53e7e58bef71a729bacc82e16587db1c4ebc91f7c5a" } }, - { url = "https://files.pythonhosted.org/packages/6e/6b/2fabe8ebf148f4ee3c782aae86a795cc68ffe7d432ef550f234025ce0cfa/wrapt-2.1.2-cp313-cp313t-win32.whl", upload-time = 2026-03-06T02:54:15Z, size = 59894, hashes = { sha256 = "f01277d9a5fc1862f26f7626da9cf443bebc0abd2f303f41c5e995b15887dabd" } }, - { url = "https://files.pythonhosted.org/packages/ca/fb/9ba66fc2dedc936de5f8073c0217b5d4484e966d87723415cc8262c5d9c2/wrapt-2.1.2-cp313-cp313t-win_amd64.whl", upload-time = 2026-03-06T02:54:41Z, size = 63197, hashes = { sha256 = "84ce8f1c2104d2f6daa912b1b5b039f331febfeee74f8042ad4e04992bd95c8f" } }, - { url = "https://files.pythonhosted.org/packages/c0/1c/012d7423c95d0e337117723eb8ecf73c622ce15a97847e84cf3f8f26cd7e/wrapt-2.1.2-cp313-cp313t-win_arm64.whl", upload-time = 2026-03-06T02:54:48Z, size = 60363, hashes = { sha256 = "a93cd767e37faeddbe07d8fc4212d5cba660af59bdb0f6372c93faaa13e6e679" } }, - { url = "https://files.pythonhosted.org/packages/39/25/e7ea0b417db02bb796182a5316398a75792cd9a22528783d868755e1f669/wrapt-2.1.2-cp314-cp314-macosx_10_15_x86_64.whl", upload-time = 2026-03-06T02:53:55Z, size = 61418, hashes = { sha256 = "1370e516598854e5b4366e09ce81e08bfe94d42b0fd569b88ec46cc56d9164a9" } }, - { url = "https://files.pythonhosted.org/packages/ec/0f/fa539e2f6a770249907757eaeb9a5ff4deb41c026f8466c1c6d799088a9b/wrapt-2.1.2-cp314-cp314-macosx_11_0_arm64.whl", upload-time = 2026-03-06T02:52:53Z, size = 61914, hashes = { sha256 = "6de1a3851c27e0bd6a04ca993ea6f80fc53e6c742ee1601f486c08e9f9b900a9" } }, - { url = "https://files.pythonhosted.org/packages/53/37/02af1867f5b1441aaeda9c82deed061b7cd1372572ddcd717f6df90b5e93/wrapt-2.1.2-cp314-cp314-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-03-06T02:54:30Z, size = 120417, hashes = { sha256 = "de9f1a2bbc5ac7f6012ec24525bdd444765a2ff64b5985ac6e0692144838542e" } }, - { url = "https://files.pythonhosted.org/packages/c3/b7/0138a6238c8ba7476c77cf786a807f871672b37f37a422970342308276e7/wrapt-2.1.2-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-03-06T02:54:51Z, size = 122797, hashes = { sha256 = "970d57ed83fa040d8b20c52fe74a6ae7e3775ae8cff5efd6a81e06b19078484c" } }, - { url = "https://files.pythonhosted.org/packages/e1/ad/819ae558036d6a15b7ed290d5b14e209ca795dd4da9c58e50c067d5927b0/wrapt-2.1.2-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-03-06T02:54:37Z, size = 117350, hashes = { sha256 = "3969c56e4563c375861c8df14fa55146e81ac11c8db49ea6fb7f2ba58bc1ff9a" } }, - { url = "https://files.pythonhosted.org/packages/8b/2d/afc18dc57a4600a6e594f77a9ae09db54f55ba455440a54886694a84c71b/wrapt-2.1.2-cp314-cp314-musllinux_1_2_aarch64.whl", upload-time = 2026-03-06T02:54:35Z, size = 121223, hashes = { sha256 = "57d7c0c980abdc5f1d98b11a2aa3bb159790add80258c717fa49a99921456d90" } }, - { url = "https://files.pythonhosted.org/packages/b9/5b/5ec189b22205697bc56eb3b62aed87a1e0423e9c8285d0781c7a83170d15/wrapt-2.1.2-cp314-cp314-musllinux_1_2_riscv64.whl", upload-time = 2026-03-06T02:54:19Z, size = 116287, hashes = { sha256 = "776867878e83130c7a04237010463372e877c1c994d449ca6aaafeab6aab2586" } }, - { url = "https://files.pythonhosted.org/packages/f7/2d/f84939a7c9b5e6cdd8a8d0f6a26cabf36a0f7e468b967720e8b0cd2bdf69/wrapt-2.1.2-cp314-cp314-musllinux_1_2_x86_64.whl", upload-time = 2026-03-06T02:54:16Z, size = 119593, hashes = { sha256 = "fab036efe5464ec3291411fabb80a7a39e2dd80bae9bcbeeca5087fdfa891e19" } }, - { url = "https://files.pythonhosted.org/packages/0b/fe/ccd22a1263159c4ac811ab9374c061bcb4a702773f6e06e38de5f81a1bdc/wrapt-2.1.2-cp314-cp314-win32.whl", upload-time = 2026-03-06T02:53:06Z, size = 58631, hashes = { sha256 = "e6ed62c82ddf58d001096ae84ce7f833db97ae2263bff31c9b336ba8cfe3f508" } }, - { url = "https://files.pythonhosted.org/packages/65/0a/6bd83be7bff2e7efaac7b4ac9748da9d75a34634bbbbc8ad077d527146df/wrapt-2.1.2-cp314-cp314-win_amd64.whl", upload-time = 2026-03-06T02:53:50Z, size = 60875, hashes = { sha256 = "467e7c76315390331c67073073d00662015bb730c566820c9ca9b54e4d67fd04" } }, - { url = "https://files.pythonhosted.org/packages/6c/c0/0b3056397fe02ff80e5a5d72d627c11eb885d1ca78e71b1a5c1e8c7d45de/wrapt-2.1.2-cp314-cp314-win_arm64.whl", upload-time = 2026-03-06T02:53:59Z, size = 59164, hashes = { sha256 = "da1f00a557c66225d53b095a97eace0fc5349e3bfda28fa34ffae238978ee575" } }, - { url = "https://files.pythonhosted.org/packages/71/ed/5d89c798741993b2371396eb9d4634f009ff1ad8a6c78d366fe2883ea7a6/wrapt-2.1.2-cp314-cp314t-macosx_10_15_x86_64.whl", upload-time = 2026-03-06T02:52:54Z, size = 63163, hashes = { sha256 = "62503ffbc2d3a69891cf29beeaccdb4d5e0a126e2b6a851688d4777e01428dbb" } }, - { url = "https://files.pythonhosted.org/packages/c6/8c/05d277d182bf36b0a13d6bd393ed1dec3468a25b59d01fba2dd70fe4d6ae/wrapt-2.1.2-cp314-cp314t-macosx_11_0_arm64.whl", upload-time = 2026-03-06T02:52:56Z, size = 63723, hashes = { sha256 = "c7e6cd120ef837d5b6f860a6ea3745f8763805c418bb2f12eeb1fa6e25f22d22" } }, - { url = "https://files.pythonhosted.org/packages/f4/27/6c51ec1eff4413c57e72d6106bb8dec6f0c7cdba6503d78f0fa98767bcc9/wrapt-2.1.2-cp314-cp314t-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-03-06T02:53:23Z, size = 152652, hashes = { sha256 = "3769a77df8e756d65fbc050333f423c01ae012b4f6731aaf70cf2bef61b34596" } }, - { url = "https://files.pythonhosted.org/packages/db/4c/d7dd662d6963fc7335bfe29d512b02b71cdfa23eeca7ab3ac74a67505deb/wrapt-2.1.2-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-03-06T02:53:35Z, size = 158807, hashes = { sha256 = "a76d61a2e851996150ba0f80582dd92a870643fa481f3b3846f229de88caf044" } }, - { url = "https://files.pythonhosted.org/packages/b4/4d/1e5eea1a78d539d346765727422976676615814029522c76b87a95f6bcdd/wrapt-2.1.2-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-03-06T02:52:57Z, size = 146061, hashes = { sha256 = "6f97edc9842cf215312b75fe737ee7c8adda75a89979f8e11558dfff6343cc4b" } }, - { url = "https://files.pythonhosted.org/packages/89/bc/62cabea7695cd12a288023251eeefdcb8465056ddaab6227cb78a2de005b/wrapt-2.1.2-cp314-cp314t-musllinux_1_2_aarch64.whl", upload-time = 2026-03-06T02:53:39Z, size = 155667, hashes = { sha256 = "4006c351de6d5007aa33a551f600404ba44228a89e833d2fadc5caa5de8edfbf" } }, - { url = "https://files.pythonhosted.org/packages/e9/99/6f2888cd68588f24df3a76572c69c2de28287acb9e1972bf0c83ce97dbc1/wrapt-2.1.2-cp314-cp314t-musllinux_1_2_riscv64.whl", upload-time = 2026-03-06T02:54:22Z, size = 144392, hashes = { sha256 = "a9372fc3639a878c8e7d87e1556fa209091b0a66e912c611e3f833e2c4202be2" } }, - { url = "https://files.pythonhosted.org/packages/40/51/1dfc783a6c57971614c48e361a82ca3b6da9055879952587bc99fe1a7171/wrapt-2.1.2-cp314-cp314t-musllinux_1_2_x86_64.whl", upload-time = 2026-03-06T02:54:07Z, size = 150296, hashes = { sha256 = "3144b027ff30cbd2fca07c0a87e67011adb717eb5f5bd8496325c17e454257a3" } }, - { url = "https://files.pythonhosted.org/packages/6c/38/cbb8b933a0201076c1f64fc42883b0023002bdc14a4964219154e6ff3350/wrapt-2.1.2-cp314-cp314t-win32.whl", upload-time = 2026-03-06T02:54:00Z, size = 60539, hashes = { sha256 = "3b8d15e52e195813efe5db8cec156eebe339aaf84222f4f4f051a6c01f237ed7" } }, - { url = "https://files.pythonhosted.org/packages/82/dd/e5176e4b241c9f528402cebb238a36785a628179d7d8b71091154b3e4c9e/wrapt-2.1.2-cp314-cp314t-win_amd64.whl", upload-time = 2026-03-06T02:54:39Z, size = 63969, hashes = { sha256 = "08ffa54146a7559f5b8df4b289b46d963a8e74ed16ba3687f99896101a3990c5" } }, - { url = "https://files.pythonhosted.org/packages/5c/99/79f17046cf67e4a95b9987ea129632ba8bcec0bc81f3fb3d19bdb0bd60cd/wrapt-2.1.2-cp314-cp314t-win_arm64.whl", upload-time = 2026-03-06T02:53:14Z, size = 60554, hashes = { sha256 = "72aaa9d0d8e4ed0e2e98019cea47a21f823c9dd4b43c7b77bba6679ffcca6a00" } }, - { url = "https://files.pythonhosted.org/packages/1a/c7/8528ac2dfa2c1e6708f647df7ae144ead13f0a31146f43c7264b4942bf12/wrapt-2.1.2-py3-none-any.whl", upload-time = 2026-03-06T02:53:12Z, size = 43993, hashes = { sha256 = "b8fd6fa2b2c4e7621808f8c62e8317f4aae56e59721ad933bac5239d913cf0e8" } }, + { url = "https://files.pythonhosted.org/packages/40/31/5822ce37ca8820c2ed35a498c67c8b37960b9cee2ba437fd32849d0a234c/wrapt-2.3.0-cp310-cp310-macosx_10_9_x86_64.whl", upload-time = 2026-07-28T06:04:04Z, size = 81191, hashes = { sha256 = "0bb2797048db0956348cb3058c33bc4184614f13231389cfbccc16a5d32780a7" } }, + { url = "https://files.pythonhosted.org/packages/7a/5a/3c6117938be98754578ab83f5a40d7d0ea2cd2c487dc5cd6027ee7228229/wrapt-2.3.0-cp310-cp310-macosx_11_0_arm64.whl", upload-time = 2026-07-28T06:04:07Z, size = 82255, hashes = { sha256 = "ce9f398f868d2b3b27aa2ea4de79645ef9077aeeac8dfc2814b0d542c6a2b87f" } }, + { url = "https://files.pythonhosted.org/packages/a5/0f/94ae724c5087eb6054c0d63febd7094947dcf302fe058e2e0488102a872b/wrapt-2.3.0-cp310-cp310-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-07-28T06:04:08Z, size = 155228, hashes = { sha256 = "ad71df7a04dd3497e9302e81f4a7c91bd401ea0e15a9df9029527900f94bee43" } }, + { url = "https://files.pythonhosted.org/packages/6c/21/1f780bba935dcf697c0c59de9be3a559bbb8e31a53ca3f25422023738432/wrapt-2.3.0-cp310-cp310-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-07-28T06:04:09Z, size = 157073, hashes = { sha256 = "fc82c2ccc8e234c844f5303d9f2984b346dcdd53e94823ce8420d2c75b4b9023" } }, + { url = "https://files.pythonhosted.org/packages/73/31/6c7799d7b6431fcd7e1b83245fb45258a2d2c3a2187fbaecb83572a72d7a/wrapt-2.3.0-cp310-cp310-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-07-28T06:04:10Z, size = 151594, hashes = { sha256 = "a6e19531ae33c508cea7d84a7edfda01fa86e51b8d1a93a77712c55e6e469152" } }, + { url = "https://files.pythonhosted.org/packages/ce/17/42d670dbfafd49076c6eb2b7d67633d7e1c968e39bfb11a135acb6fac67b/wrapt-2.3.0-cp310-cp310-musllinux_1_2_aarch64.whl", upload-time = 2026-07-28T06:04:12Z, size = 156069, hashes = { sha256 = "df4ce31150bcd5d9f36f816aac3010ab4f4bf8672ac1d3b0ac7d539ec61c7c02" } }, + { url = "https://files.pythonhosted.org/packages/1f/d6/c66b4ba4eda49257c84d5c2df26118280f09ca7905aee20d0064db778d13/wrapt-2.3.0-cp310-cp310-musllinux_1_2_riscv64.whl", upload-time = 2026-07-28T06:04:13Z, size = 150930, hashes = { sha256 = "e2e692bc0d63f881cf7006730a56bd4e0c2fab5dc318466942805d692b166276" } }, + { url = "https://files.pythonhosted.org/packages/c0/f2/1a3b949c0322fb27396eafd1044328c1cb0400e0b32105d75a3cd03096e7/wrapt-2.3.0-cp310-cp310-musllinux_1_2_x86_64.whl", upload-time = 2026-07-28T06:04:14Z, size = 154525, hashes = { sha256 = "c8388ba7faf5dbf9ee106bb70d66f257629b1bd98091123e19e8a4553a319199" } }, + { url = "https://files.pythonhosted.org/packages/12/65/147563a3dfa6e830c857b93b530ebd8c0cd9d540e5914aec8f9b12880c02/wrapt-2.3.0-cp310-cp310-win32.whl", upload-time = 2026-07-28T06:04:16Z, size = 77879, hashes = { sha256 = "e045ff75d7d94900fc32896ed93c45ce2d2cac28c9dead582ff9a5a49d446e35" } }, + { url = "https://files.pythonhosted.org/packages/c4/eb/921405b4dc55d4f8be4c700ef120539fdd75d5fdb50d83bd257171ee18e0/wrapt-2.3.0-cp310-cp310-win_amd64.whl", upload-time = 2026-07-28T06:04:17Z, size = 80733, hashes = { sha256 = "b4fc96b159af0a3e0faa72475a69d66292bea72a5bed1e1aca1bffbddc3cb2b0" } }, + { url = "https://files.pythonhosted.org/packages/b6/13/75947450c5bb57795fa86384721cd52c5c4deb0879022f309501a8a85d44/wrapt-2.3.0-cp310-cp310-win_arm64.whl", upload-time = 2026-07-28T06:04:18Z, size = 80199, hashes = { sha256 = "1236fa25173ca964c97422470482e9011b9e3c7ed0d75798b40b3da3b0e0e760" } }, + { url = "https://files.pythonhosted.org/packages/00/b8/9182e4c618a847be0baccb68e4602b070d0fa22c782cf058f4bc66b32709/wrapt-2.3.0-cp311-cp311-macosx_10_9_x86_64.whl", upload-time = 2026-07-28T06:04:20Z, size = 81427, hashes = { sha256 = "5ab559e1b2551d23d54db2a0001c6d73bad022a254639561c5f6c382a9d6c2fe" } }, + { url = "https://files.pythonhosted.org/packages/84/ca/613cefd9c5977366b1587e61c0b428176d382e6d75b454084c5e58503042/wrapt-2.3.0-cp311-cp311-macosx_11_0_arm64.whl", upload-time = 2026-07-28T06:04:21Z, size = 82360, hashes = { sha256 = "bff9a671bc00709cab5a7f745c592b5671873449db0ee2a569af994f16b29a4d" } }, + { url = "https://files.pythonhosted.org/packages/71/71/4cd2151a236f44a6e2dd4ed8011838d7ba0be3d656c8bafdfc65a2ed1917/wrapt-2.3.0-cp311-cp311-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-07-28T06:04:22Z, size = 161700, hashes = { sha256 = "fc648a335d7e01adb3640b25f02fd0ea05886cf04d0af7f4ee902bc7b5e466e8" } }, + { url = "https://files.pythonhosted.org/packages/49/2c/bc508fee75eb2919ed69769800b09968e4aab16897f909a23f39c81e323f/wrapt-2.3.0-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-07-28T06:04:24Z, size = 162922, hashes = { sha256 = "d0077f3d65541925fa83002f967b22ad6550d24813ac64cb905f717194128d9c" } }, + { url = "https://files.pythonhosted.org/packages/4d/e5/04f34d38e66d857dfc2fc4088d60e70c0e422467822defa49b2b4a26e17b/wrapt-2.3.0-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-07-28T06:04:25Z, size = 156125, hashes = { sha256 = "9790ea25190a4e0fe4cdf4eeb868e9d75f8a024a70a5b6bf9c348a3a2b72e731" } }, + { url = "https://files.pythonhosted.org/packages/23/41/c35940ea1c423f129ebe4361db853bc80d4def6326242e1206fa15bf94f4/wrapt-2.3.0-cp311-cp311-musllinux_1_2_aarch64.whl", upload-time = 2026-07-28T06:04:27Z, size = 162039, hashes = { sha256 = "816877aa749253149f9ecfd2635d4d948ecfa338e1a0311d187b1acb1bb8a3eb" } }, + { url = "https://files.pythonhosted.org/packages/0e/60/9bda34c3d7d182aa703fe35339ae0ed4c4dad5e5c587f93890143e1f87fb/wrapt-2.3.0-cp311-cp311-musllinux_1_2_riscv64.whl", upload-time = 2026-07-28T06:04:28Z, size = 155110, hashes = { sha256 = "3d1c2c1b808600d2ea808e6360910a60ed5f409a4011655e10f9164ba0a414a6" } }, + { url = "https://files.pythonhosted.org/packages/e8/ba/60bfd9b1a751f4fcb2d603668fc272d651ccdd339a56acf8c40ad21a0293/wrapt-2.3.0-cp311-cp311-musllinux_1_2_x86_64.whl", upload-time = 2026-07-28T06:04:29Z, size = 161089, hashes = { sha256 = "5ba1e5e08ddc46130e9682b2c249f2d1dd39bda9106ed4bd401b7519f18f41bd" } }, + { url = "https://files.pythonhosted.org/packages/0f/32/2bd358c6f4f1305c813479d1e9ba746bebdd794f4a20107ab2b3ee0cbd45/wrapt-2.3.0-cp311-cp311-win32.whl", upload-time = 2026-07-28T06:04:31Z, size = 78030, hashes = { sha256 = "45c9279b373d15649dfa2c2077cb3408ea1a6d3125afbdab9d6b809a66f68e14" } }, + { url = "https://files.pythonhosted.org/packages/4a/62/ecc969b13b141fef89b888c9760821cb01a86ac8fc953911592c8e1e1522/wrapt-2.3.0-cp311-cp311-win_amd64.whl", upload-time = 2026-07-28T06:04:32Z, size = 80944, hashes = { sha256 = "195b1842b4122fb54e3cd3dd5b2b4aa49302a5a61da901df0481f5c97aedde84" } }, + { url = "https://files.pythonhosted.org/packages/c4/3d/9278ada8a2b3f24372b630361e84e9a7de7abc3784634860c26d1c37785a/wrapt-2.3.0-cp311-cp311-win_arm64.whl", upload-time = 2026-07-28T06:04:33Z, size = 80074, hashes = { sha256 = "6db604ef0c67bdb2042ecdfd7b7f037cf09733557ca42360d1018285634f7b98" } }, + { url = "https://files.pythonhosted.org/packages/5b/4a/d17a0fad1bf1c5f2c887ff71fef75654141b0880bff71d157d955b5bec3a/wrapt-2.3.0-cp312-cp312-macosx_10_13_x86_64.whl", upload-time = 2026-07-28T06:04:35Z, size = 82139, hashes = { sha256 = "0a45ffae742ce91a16e11cb6c7cd71e7f9994f3cbd283b962ab093f5c6dcf525" } }, + { url = "https://files.pythonhosted.org/packages/6e/55/51b92daaf6defb57f4dc56bdcce985400f75c6984a03ca5e78ccac717028/wrapt-2.3.0-cp312-cp312-macosx_11_0_arm64.whl", upload-time = 2026-07-28T06:04:36Z, size = 82723, hashes = { sha256 = "69e477046f2237ef0bc6547544ee73008dc764ca26eff44f09e976d221b34d5d" } }, + { url = "https://files.pythonhosted.org/packages/28/7f/cfd9bc4b1f5e424eeea83d0493e43f3b1b02707ce8e50c47945873982bd5/wrapt-2.3.0-cp312-cp312-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-07-28T06:04:37Z, size = 172381, hashes = { sha256 = "5d221a6e6ddd302b8397433184e96b59f259f50024b854db1c411a881586b6b8" } }, + { url = "https://files.pythonhosted.org/packages/cb/89/ff7814f6eb6856b479946117d1138a2fbb46cdb6b1f379db359056c69743/wrapt-2.3.0-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-07-28T06:04:38Z, size = 174120, hashes = { sha256 = "392158c9a7f2ab1b8699418bfc0fe6f83548788c418b27d7bf2019ad3405cebb" } }, + { url = "https://files.pythonhosted.org/packages/12/1e/8eded8615d39e3ce81f626937a3a87b280a2a86239a2bf14a4b4bb345034/wrapt-2.3.0-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-07-28T06:04:40Z, size = 163035, hashes = { sha256 = "e5301c35cf75655eb33498f2bd6ae8703ca19940e3167dc9cdf740c712a39c60" } }, + { url = "https://files.pythonhosted.org/packages/35/ea/a0af2d9da62897af2a055484920de05dade30d2ba2c0d65cbdea875d3d8b/wrapt-2.3.0-cp312-cp312-musllinux_1_2_aarch64.whl", upload-time = 2026-07-28T06:04:41Z, size = 171887, hashes = { sha256 = "418f54bb09d1762db02c7009b4051149893af3153a87f92d70356703c11eea02" } }, + { url = "https://files.pythonhosted.org/packages/7e/dd/63cd4c864c65ef4906df64bd2d378f4a62b54f28063f282dfb3bf93caead/wrapt-2.3.0-cp312-cp312-musllinux_1_2_riscv64.whl", upload-time = 2026-07-28T06:04:42Z, size = 161113, hashes = { sha256 = "1598becd30f8f2777d18564064eb4f4dbe1ab0e05a8f09786d0ef505ac782bf3" } }, + { url = "https://files.pythonhosted.org/packages/ca/ee/82f1fc9e431b5c2c5a6d201aa865dbeae3984c311c6d11a185f0c8367cf6/wrapt-2.3.0-cp312-cp312-musllinux_1_2_x86_64.whl", upload-time = 2026-07-28T06:04:44Z, size = 170530, hashes = { sha256 = "3da470536bf9645143323dd41b32db55c6f4304ad382094c1a1da8a92061e10d" } }, + { url = "https://files.pythonhosted.org/packages/37/a5/5dc590e863a419930d988f8b7ca3e75a6befcfb10b6003b3a152f3d5f732/wrapt-2.3.0-cp312-cp312-win32.whl", upload-time = 2026-07-28T06:04:45Z, size = 78323, hashes = { sha256 = "fb8e2e6704a1e0b1b989546c69e2688371ef4a07fa5f61bde3eb6211186f5ac1" } }, + { url = "https://files.pythonhosted.org/packages/51/f9/4a6925a07951df56394f7e6ebe14f69f1c5ef9d87aa63e0839acf15aa63a/wrapt-2.3.0-cp312-cp312-win_amd64.whl", upload-time = 2026-07-28T06:04:47Z, size = 81180, hashes = { sha256 = "cdc021cb0b62471d6aac7f2bd92f3b4658073775f9ee7fcd325c511129e7bcc8" } }, + { url = "https://files.pythonhosted.org/packages/a8/4f/8b5de0395b2a72216751d41c9861df6facaeb611b619d8810ed2b3b23eb2/wrapt-2.3.0-cp312-cp312-win_arm64.whl", upload-time = 2026-07-28T06:04:48Z, size = 80155, hashes = { sha256 = "67bfe2485f50368c3fcd2275fc1fd100e350d601e0058921a7c82678a465aeab" } }, + { url = "https://files.pythonhosted.org/packages/8e/6e/0f88a072483e76b881e3fdcd6b6ffb4a5791002514fe541e72b1b73c859a/wrapt-2.3.0-cp313-cp313-macosx_10_13_x86_64.whl", upload-time = 2026-07-28T06:04:49Z, size = 81960, hashes = { sha256 = "0d3fb71e65b001adfc42684522eeccd9c21d8ba679945abc993439567b66e59f" } }, + { url = "https://files.pythonhosted.org/packages/d7/ff/b7e2776e7c294075eb712cc9ef573d1b818f393006d09787262b8fc871c4/wrapt-2.3.0-cp313-cp313-macosx_11_0_arm64.whl", upload-time = 2026-07-28T06:04:50Z, size = 82435, hashes = { sha256 = "51a7a4181c1295774812271fbcd7c909df372bc25579d4ed9eb875caaf0ae86f" } }, + { url = "https://files.pythonhosted.org/packages/d8/90/343bb5d0f1f9669bc252a6073f085b4abf862511bd5c9c9eaec754341f1d/wrapt-2.3.0-cp313-cp313-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-07-28T06:04:52Z, size = 170350, hashes = { sha256 = "9045917809c63fdf7abe3a2ceaed3d670b8ee4500ddd9291192d30aeb34467c5" } }, + { url = "https://files.pythonhosted.org/packages/59/f8/13b79a392930bd0dd6b86cbfbfe1c40944110456e1dc6d809e5c46ece904/wrapt-2.3.0-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-07-28T06:04:53Z, size = 170022, hashes = { sha256 = "54ca1d5573f69b5fe1d74f1f65799c68015e82f685efec9fd8cfa40a094c44d0" } }, + { url = "https://files.pythonhosted.org/packages/b2/fc/4f1b6918f5290db959d6e0c07f77385d87cede29c39c9cf8f145e9c82954/wrapt-2.3.0-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-07-28T06:04:54Z, size = 161043, hashes = { sha256 = "242b60c21e30866e6a2fa606c612b47c553fa60c0eaeeeb7797fb842ac0ce609" } }, + { url = "https://files.pythonhosted.org/packages/01/e1/45d3cf74414780bdff6d0380467e003f6eb0f028b6c9403db868dbc7209c/wrapt-2.3.0-cp313-cp313-musllinux_1_2_aarch64.whl", upload-time = 2026-07-28T06:04:56Z, size = 168576, hashes = { sha256 = "e3f3d7ec0a51fbfe00d3aef047641ff2c58b25565b4717fc1f90e050be01cba8" } }, + { url = "https://files.pythonhosted.org/packages/f3/73/2fa58dd97f191c997755e2c6d569a68f0c433db4e4b36099bdd7227b6cac/wrapt-2.3.0-cp313-cp313-musllinux_1_2_riscv64.whl", upload-time = 2026-07-28T06:04:57Z, size = 159140, hashes = { sha256 = "261f53870cd4fb2bf38f9f972c56c728fd224cb7c65721307de59d9e7e6741ae" } }, + { url = "https://files.pythonhosted.org/packages/29/a8/08a56e2000a8816d449dcbad8c8b081697acbbd490821ceca0f9d8e8d20c/wrapt-2.3.0-cp313-cp313-musllinux_1_2_x86_64.whl", upload-time = 2026-07-28T06:04:59Z, size = 169263, hashes = { sha256 = "8159ec0b0cb7608175eb150de94c19e34f4d47ac655f5ca9baf45df6b688ffd3" } }, + { url = "https://files.pythonhosted.org/packages/9e/d4/354e1725e35a73b2af4fa70a3e024c7a5d1bf1802dfb862dcb668aae0253/wrapt-2.3.0-cp313-cp313-win32.whl", upload-time = 2026-07-28T06:05:00Z, size = 78241, hashes = { sha256 = "10461884b3014fbfc8eb7d09a93c5f246363e6711d9d881f95eb8c27fdef049f" } }, + { url = "https://files.pythonhosted.org/packages/6c/7e/34c87fa2174848dfee820322aaa318bab08913998ccecc8d2f57b4ad4639/wrapt-2.3.0-cp313-cp313-win_amd64.whl", upload-time = 2026-07-28T06:05:01Z, size = 81113, hashes = { sha256 = "ac870cc97b73bb00ac353329e9559a4bebc47c4c86792ed9b23b58c15b6ad838" } }, + { url = "https://files.pythonhosted.org/packages/11/86/fcc9a530579e008c9478bb565a6cdfbfd33536660f069c8b91a6607c5050/wrapt-2.3.0-cp313-cp313-win_arm64.whl", upload-time = 2026-07-28T06:05:03Z, size = 80182, hashes = { sha256 = "a65e8db2b4e90c2e7ade931086351c98ef420bf7a94ee08c95ac8a3cbbc43579" } }, + { url = "https://files.pythonhosted.org/packages/96/50/3864848b95b28ef73e17551fc8dccbff2628a834f52cf26a57f9c419fb83/wrapt-2.3.0-cp313-cp313t-macosx_10_13_x86_64.whl", upload-time = 2026-07-28T06:05:04Z, size = 83921, hashes = { sha256 = "fd1f2f557dd3491fe75905e578f4db967393d40d1a8f468edc4d40ac7f2d5944" } }, + { url = "https://files.pythonhosted.org/packages/3b/4c/3d1921a60c3e8c71c540ff136e6a47a1fbccf7f671e818394889f7871d9c/wrapt-2.3.0-cp313-cp313t-macosx_11_0_arm64.whl", upload-time = 2026-07-28T06:05:05Z, size = 84412, hashes = { sha256 = "9f5d2aec29dfc76c37e23897dee92766a3fd4f3bff3ae7fc9c6b4bf37d8c1360" } }, + { url = "https://files.pythonhosted.org/packages/fa/1a/4a796ff7adb26ada6d4b758c94d47a38320b085e7099afc088efbbcdb006/wrapt-2.3.0-cp313-cp313t-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-07-28T06:05:07Z, size = 207168, hashes = { sha256 = "646d20d413ffcd1b0a2f700076e2d0252d872dcb7754860a73e45a59ea883614" } }, + { url = "https://files.pythonhosted.org/packages/1d/3e/d7777776806c579b761bac2f91721dda9f04c7a1b380213c5935cc750ae6/wrapt-2.3.0-cp313-cp313t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-07-28T06:05:08Z, size = 214351, hashes = { sha256 = "379f670f45b7bb8993edd9f6fc36c6cc65edb81cffa0b504be34acb0303fff0a" } }, + { url = "https://files.pythonhosted.org/packages/63/27/2d64d394df7bf181955b3bb562bf33c4492fb4be113f53071106d43ad8b5/wrapt-2.3.0-cp313-cp313t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-07-28T06:05:10Z, size = 199020, hashes = { sha256 = "6208f302f110295d64b22a7ac96500c791bf492dce4366e622e4912b077c9687" } }, + { url = "https://files.pythonhosted.org/packages/3e/3d/fb31d3db7d9834d265fb1a27a2adf0ddf51557c67458c97b22439ad6ae3d/wrapt-2.3.0-cp313-cp313t-musllinux_1_2_aarch64.whl", upload-time = 2026-07-28T06:05:11Z, size = 209969, hashes = { sha256 = "ed635a9ca4f3a5a2b900c10c69e823373bc00ebc114b459383596d3487da3570" } }, + { url = "https://files.pythonhosted.org/packages/1f/d1/8724b5da582e62070dc9bf4d8bf1972f317297eefd7ba1f2b5c6393ccf6c/wrapt-2.3.0-cp313-cp313t-musllinux_1_2_riscv64.whl", upload-time = 2026-07-28T06:05:13Z, size = 196324, hashes = { sha256 = "e3b9eaa742ae7a0aaaaad4ca4b69469d757af2d6e6663ef1dadc47adec0aeb41" } }, + { url = "https://files.pythonhosted.org/packages/0d/5c/3d9ef411149543016ee6bcf3af707f787cebd946527452b94bf122e9b7b4/wrapt-2.3.0-cp313-cp313t-musllinux_1_2_x86_64.whl", upload-time = 2026-07-28T06:05:15Z, size = 202610, hashes = { sha256 = "d0f7284f88f4833705132d06d3b425a43095c2cbd07c58166aac3ab646ba12a4" } }, + { url = "https://files.pythonhosted.org/packages/13/9b/4fc042ceb757866dd4a5fc057b3b736f2b360d3703ce9f830d83dc9226e0/wrapt-2.3.0-cp313-cp313t-win32.whl", upload-time = 2026-07-28T06:05:16Z, size = 79178, hashes = { sha256 = "7ebb274aba688b043429eb1500ff8a76ce0cb8ac0812ca3e301f06247b8722b3" } }, + { url = "https://files.pythonhosted.org/packages/6b/ff/b94878f8eed809ca042685276bcea9f24e8c2ca7c9653bb80bbb920a68a5/wrapt-2.3.0-cp313-cp313t-win_amd64.whl", upload-time = 2026-07-28T06:05:18Z, size = 82634, hashes = { sha256 = "c4bded758ad6f03b965830944a2f0bc5b2eb3767fe5a7310134315d1a6610e98" } }, + { url = "https://files.pythonhosted.org/packages/80/fb/663e1de5332a71685a729754312d327d4cada767c36e1c5a2db4c8de49e6/wrapt-2.3.0-cp313-cp313t-win_arm64.whl", upload-time = 2026-07-28T06:05:19Z, size = 81387, hashes = { sha256 = "d2cc64539da63e39ffb9c7ede849b6e8ddaaf7b3876b5cfb04efd85a5f3f4eb6" } }, + { url = "https://files.pythonhosted.org/packages/58/10/b073beaea89bc0d3670a75ff51139430a54b6af7ba7796507730634536dd/wrapt-2.3.0-cp314-cp314-macosx_10_15_x86_64.whl", upload-time = 2026-07-28T06:05:21Z, size = 81978, hashes = { sha256 = "ea52a0d0f08c584943d5764be0e84efa912c8da23c23e1e285ff2f5641c18fcc" } }, + { url = "https://files.pythonhosted.org/packages/b3/31/0916d9cebf848ed3f1a0c1888faee421747df77331e4db2bc527a9a85988/wrapt-2.3.0-cp314-cp314-macosx_11_0_arm64.whl", upload-time = 2026-07-28T06:05:22Z, size = 82518, hashes = { sha256 = "fd85b0aa88efdb189d6ae2f35f4526943a8f091c38599c9c31478241c819e6a1" } }, + { url = "https://files.pythonhosted.org/packages/f5/73/31c1bf0f3384062751c2094dadb314916d70aa9b6bfd26d994b4a7b393fa/wrapt-2.3.0-cp314-cp314-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-07-28T06:05:23Z, size = 170187, hashes = { sha256 = "141ed6211286a9660d8d6702de598b43f0934b4f0eda16393f100a80f501d945" } }, + { url = "https://files.pythonhosted.org/packages/ed/25/fce087d54b79b8905f3c3c9dd5f454bbd8d8acb80b960c4a6aee5b4659b3/wrapt-2.3.0-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-07-28T06:05:25Z, size = 169288, hashes = { sha256 = "2e49885a62ec4ee854d1b9e6371fda6afd219917225752abf729a3f36d4df9a5" } }, + { url = "https://files.pythonhosted.org/packages/c7/30/0d09e6dddc6b7a7230ac77f50254b5980ab4fcd22976f72f8cc8a0404458/wrapt-2.3.0-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-07-28T06:05:27Z, size = 160932, hashes = { sha256 = "1d6159c9b2fefec02314e1332dbbbfaf960e369dfd26bcf7f8b258b5732065b3" } }, + { url = "https://files.pythonhosted.org/packages/2c/ca/0913af0d2ec0c43865d32d615f518fea66c13c5c930e489e9b0de248e9a8/wrapt-2.3.0-cp314-cp314-musllinux_1_2_aarch64.whl", upload-time = 2026-07-28T06:05:28Z, size = 169017, hashes = { sha256 = "24da48596326ef8e448cfa837b454f638713d3531262375f00e5a9681682fc07" } }, + { url = "https://files.pythonhosted.org/packages/c3/f2/3d1e47ea81b822210f5df1bf942fd90780a75c055243d569b664529dea88/wrapt-2.3.0-cp314-cp314-musllinux_1_2_riscv64.whl", upload-time = 2026-07-28T06:05:30Z, size = 159065, hashes = { sha256 = "cd3a2edf0427013736b8127955cec62608c56e53ea47e82812ea32059cda407f" } }, + { url = "https://files.pythonhosted.org/packages/43/a5/ef2066ced8e5fca204e2b361e9708e36555b40949c583d997ea3b590817d/wrapt-2.3.0-cp314-cp314-musllinux_1_2_x86_64.whl", upload-time = 2026-07-28T06:05:31Z, size = 168821, hashes = { sha256 = "4fa0df3bff4e7ce45759f33fd39335fe2f60477bb9ecf7b8aa41e7d07ee36a23" } }, + { url = "https://files.pythonhosted.org/packages/d5/e1/016104650d4e572fa91506eb396b3dd8efbccc9284fdc1c9479c3d21db28/wrapt-2.3.0-cp314-cp314-win32.whl", upload-time = 2026-07-28T06:05:33Z, size = 78700, hashes = { sha256 = "2935d5454b3f179a29b12cf390ee47246740ba2c3a7545b1b46ba31a5f2a4a0b" } }, + { url = "https://files.pythonhosted.org/packages/3d/97/6fdc20a9f2ca304748b3f0819cbf377d55260562777bf0b615431bc3c181/wrapt-2.3.0-cp314-cp314-win_amd64.whl", upload-time = 2026-07-28T06:05:34Z, size = 81422, hashes = { sha256 = "cc2cea812e5cb179a796b766747e7d3b21088760d8deb95676d482b8c8e6fa7d" } }, + { url = "https://files.pythonhosted.org/packages/5e/a4/9cbd53bf05746bea2c392af39cb052427a8ec95cbd494d930733d8f44681/wrapt-2.3.0-cp314-cp314-win_arm64.whl", upload-time = 2026-07-28T06:05:36Z, size = 80639, hashes = { sha256 = "22cc5c0a717bd4da87018ae0bffd4c19c6fb679d3ff357216ba566ab26c76cab" } }, + { url = "https://files.pythonhosted.org/packages/43/bb/6c5e4a0f66ea0d2b2dd267e8dd05a0014eea56840b3c8595d40b0a5d1f91/wrapt-2.3.0-cp314-cp314t-macosx_10_15_x86_64.whl", upload-time = 2026-07-28T06:05:37Z, size = 84030, hashes = { sha256 = "a6b5984cd65dd639546f0eb4b8eacf1c31cb2fe9fb5c27bffe240987cdb2cf84" } }, + { url = "https://files.pythonhosted.org/packages/6a/eb/a1aedf03283bc9cbf8a1783995ddc54e3c5a86878f19002d2c428494f4c5/wrapt-2.3.0-cp314-cp314t-macosx_11_0_arm64.whl", upload-time = 2026-07-28T06:05:39Z, size = 84419, hashes = { sha256 = "c88abcf53daef80e01a75c7530e727fa6e2c1888fe83e3dcdba4c96216a1f5c7" } }, + { url = "https://files.pythonhosted.org/packages/63/61/50d511c0dc5105563849e86daa3e16ac7feef699f79fb05af45ea70107d5/wrapt-2.3.0-cp314-cp314t-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-07-28T06:05:40Z, size = 207171, hashes = { sha256 = "85de890ff968196e92dd1ae73a9fb8970495e7650a457b1c9ef0ac3dd550bce2" } }, + { url = "https://files.pythonhosted.org/packages/3f/59/9b538cf7795217e810699d16bc88b96a830d9b5c403eb2ec2db6b5f2ae81/wrapt-2.3.0-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-07-28T06:05:42Z, size = 214329, hashes = { sha256 = "50f416b74d092bb9f41b424e90dd457f365f7ba4b11de62a23679769a21bd85c" } }, + { url = "https://files.pythonhosted.org/packages/b3/28/9935d62b1499e5c8b3d191e99ba4eb31ca237a0b699142011a837e9dc7ea/wrapt-2.3.0-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-07-28T06:05:43Z, size = 199079, hashes = { sha256 = "39febbee6d77301d31da6996b152ce52452da7c7ef72aba10c2fa976dff9c295" } }, + { url = "https://files.pythonhosted.org/packages/2b/01/4446b80fa2ffa47a3449b250d004ba1c1937f07f64a179608fec735df866/wrapt-2.3.0-cp314-cp314t-musllinux_1_2_aarch64.whl", upload-time = 2026-07-28T06:05:45Z, size = 209992, hashes = { sha256 = "93513bec052c6cd987f9f580c3df068c8bc4ebae6543736be3ca7ec5959cafcd" } }, + { url = "https://files.pythonhosted.org/packages/d4/07/56f26c9f9979586a021e8148747004aba4498f49458c90b0502969b904e1/wrapt-2.3.0-cp314-cp314t-musllinux_1_2_riscv64.whl", upload-time = 2026-07-28T06:05:47Z, size = 196334, hashes = { sha256 = "729126e667da34d251b8ebf8a45ef0c5ddadc21542b3d6e1abf4259ece6508df" } }, + { url = "https://files.pythonhosted.org/packages/8b/41/6d7bcc895b0f28b2250e10908f060687b9165429dcd7f22ddb3d4c031b74/wrapt-2.3.0-cp314-cp314t-musllinux_1_2_x86_64.whl", upload-time = 2026-07-28T06:05:49Z, size = 202644, hashes = { sha256 = "626b69db2021aa01671ec7bbc9740e558522bd44c18cf2ce69bf3d666a014109" } }, + { url = "https://files.pythonhosted.org/packages/cd/25/7860927edba06b758b8852a6f02e832be715563c67a6795d94350bc81099/wrapt-2.3.0-cp314-cp314t-win32.whl", upload-time = 2026-07-28T06:05:50Z, size = 79685, hashes = { sha256 = "629d73378082c00a8173031f9fb30a3ac6abbc894a5bfdfae71fabc60642d501" } }, + { url = "https://files.pythonhosted.org/packages/c4/0f/270bafe92fde3b069a39bc01e39ee79340895b335640df861d43d2a51885/wrapt-2.3.0-cp314-cp314t-win_amd64.whl", upload-time = 2026-07-28T06:05:52Z, size = 83104, hashes = { sha256 = "42869085687f0aefd57c0f636c3f9354f8ffb321a8ba9cb52d19beb796e561c5" } }, + { url = "https://files.pythonhosted.org/packages/55/b3/af176d79a8515a8a720eccdad9a96f6e31a30abf2865430c8c42adf2fd13/wrapt-2.3.0-cp314-cp314t-win_arm64.whl", upload-time = 2026-07-28T06:05:53Z, size = 81774, hashes = { sha256 = "b1e5aa486e269b00ed35e64771c7d0ab8096cfd2643405ca8cd60ebedc099a51" } }, + { url = "https://files.pythonhosted.org/packages/00/39/3daf9f47be208606586de4568ba6713db53ebc8fd7a575aea1fe57983b69/wrapt-2.3.0-py3-none-any.whl", upload-time = 2026-07-28T06:06:12Z, size = 61866, hashes = { sha256 = "d8c7ed08477429752b8c44991f40ad7838b18332a160698740a6bfbc10d998a2" } }, ] diff --git a/foreign/python/pyproject.toml b/foreign/python/pyproject.toml index d9623d4312..54d53ec901 100644 --- a/foreign/python/pyproject.toml +++ b/foreign/python/pyproject.toml @@ -16,7 +16,7 @@ # under the License. [build-system] -requires = ["maturin>=1.14.1,<2.0"] +requires = ["maturin>=1.15.0,<2.0"] build-backend = "maturin" [project] @@ -111,14 +111,14 @@ testing-docker = ["testcontainers>=4.15.0,<5.0"] # Development tools dev = [ - "maturin>=1.14.1,<2.0", + "maturin>=1.15.0,<2.0", "pyrefly>=1.2.0", "ruff>=0.1.0,<1.0", ] # All dependencies for full development setup all = [ - "maturin>=1.14.1,<2.0", + "maturin>=1.15.0,<2.0", "pyrefly>=1.2.0", "pytest>=9.1.1,<10.0", "pytest-asyncio>=0.24.0,<2.0", diff --git a/foreign/python/uv.lock b/foreign/python/uv.lock index 214d7023b5..d6f3eb5067 100644 --- a/foreign/python/uv.lock +++ b/foreign/python/uv.lock @@ -41,8 +41,8 @@ testing-docker = [ [package.metadata] requires-dist = [ - { name = "maturin", marker = "extra == 'all'", specifier = ">=1.14.1,<2.0" }, - { name = "maturin", marker = "extra == 'dev'", specifier = ">=1.14.1,<2.0" }, + { name = "maturin", marker = "extra == 'all'", specifier = ">=1.15.0,<2.0" }, + { name = "maturin", marker = "extra == 'dev'", specifier = ">=1.15.0,<2.0" }, { name = "pyrefly", marker = "extra == 'all'", specifier = ">=1.2.0" }, { name = "pyrefly", marker = "extra == 'dev'", specifier = ">=1.2.0" }, { name = "pytest", marker = "extra == 'all'", specifier = ">=9.1.1,<10.0" }, @@ -82,165 +82,165 @@ wheels = [ [[package]] name = "charset-normalizer" -version = "3.5.0" +version = "3.5.1" source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/cb/31/4971872b3ed8715346231fb6eb4da8fcba65a4143c189db151ee28a2812b/charset_normalizer-3.5.0.tar.gz", hash = "sha256:49bd5feb59b0bf3cbf6ebcf4352e371c95b9da9bacd4449f8b64d0ad2c10a26e", size = 169295, upload-time = "2026-08-12T14:35:31.624Z" } +sdist = { url = "https://files.pythonhosted.org/packages/e5/3f/143b048436775b0f76ac3eec145c019e8173ccc2885c8f20319b996d5e83/charset_normalizer-3.5.1.tar.gz", hash = "sha256:6117b84ea48435e5356dc737f5121485c30920ba43375fa7b434fd753df0eac3", size = 171764, upload-time = "2026-08-15T08:20:44.807Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/ca/8e/1e9a03657c2f12960c67b05e0550059787d4c8fc3fae822672f4fe7f40c5/charset_normalizer-3.5.0-cp310-cp310-macosx_10_9_universal2.whl", hash = "sha256:d2478bd3b2ead3962a484fb802891be40d10049fb74f83e09cb4463fad023fea", size = 349795, upload-time = "2026-08-12T14:31:47.546Z" }, - { url = "https://files.pythonhosted.org/packages/76/e5/a42c8884ec0321c56124a5b43c135efe75260dad4f70ea0ffba4dcac85ec/charset_normalizer-3.5.0-cp310-cp310-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:7cdded069549b5eae3d5d9bb6c2e5bb4fe83f9b81863e2a193cd747bf197aebb", size = 250677, upload-time = "2026-08-12T14:31:48.918Z" }, - { url = "https://files.pythonhosted.org/packages/5d/21/d94021b005d2e660fd13634eecfe4ad378b0d0bbb45134809aed42b94a99/charset_normalizer-3.5.0-cp310-cp310-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:aff38231e3171c578b2c449a01afa44e9ff40844597a32873da102394f63d28e", size = 239868, upload-time = "2026-08-12T14:31:50.107Z" }, - { url = "https://files.pythonhosted.org/packages/47/d0/8b9b07e41501bfcdfe380b826481033ca2eb5fa4b5507eb64dbc4cba9885/charset_normalizer-3.5.0-cp310-cp310-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:4346a693c08b1d0cfc0e3325bfb0ecd4322fb1a6904d68cf416f8da5e981b234", size = 280688, upload-time = "2026-08-12T14:31:51.182Z" }, - { url = "https://files.pythonhosted.org/packages/a8/29/b83b3469e596adff778bf3c8ca8f705822c1c3c0ada8477a0d00113bfa59/charset_normalizer-3.5.0-cp310-cp310-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:b787efadba00f5da6fe89513bfbe3852d52ca3a448fdec165765cb3b44a80248", size = 277133, upload-time = "2026-08-12T14:31:52.362Z" }, - { url = "https://files.pythonhosted.org/packages/0c/aa/a080c4facaab4ece837558649eb6e2c760867d1c0ea52220cbbc4e89dcaa/charset_normalizer-3.5.0-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:143792a43e06dc3b27fc891948406e251502dc19ff9216cd80182b79131be5c5", size = 261492, upload-time = "2026-08-12T14:31:53.628Z" }, - { url = "https://files.pythonhosted.org/packages/f9/24/e23bcc9d4bc897bd620bc39c6b6052574f5ef93b85b50b175dd427a7dc07/charset_normalizer-3.5.0-cp310-cp310-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:d74bcf1cdd8ac8267fb216473ce6b112efa07b163536288094541415084d131c", size = 259739, upload-time = "2026-08-12T14:31:54.86Z" }, - { url = "https://files.pythonhosted.org/packages/ef/85/f4c391c807f67c6483d02c02d28683e2fe229403728369c0cd33f61e4292/charset_normalizer-3.5.0-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:3bfbe543d957213fc9a3db4979a8e171b7aa7504c1d737029defdb03a6095a38", size = 252120, upload-time = "2026-08-12T14:31:56.227Z" }, - { url = "https://files.pythonhosted.org/packages/4e/cb/dcf5aa14e24270613fafa2f02b3979820c244610410e5508696ca96bd0d2/charset_normalizer-3.5.0-cp310-cp310-musllinux_1_2_armv7l.whl", hash = "sha256:3587d94b5c9f05c2dc4c3f3d47aba6375ff141a21adae3051d8d4d53e8a937c0", size = 241622, upload-time = "2026-08-12T14:31:57.54Z" }, - { url = "https://files.pythonhosted.org/packages/83/a9/decd1c1cc01ccbfce66845096e3945397e4ede354f31cc5cb28fc6ca601b/charset_normalizer-3.5.0-cp310-cp310-musllinux_1_2_ppc64le.whl", hash = "sha256:a7cb4cd266bd85613367fb85a30cfbf6fe6349919e87e18ca8dba584951bfb8a", size = 280745, upload-time = "2026-08-12T14:31:58.91Z" }, - { url = "https://files.pythonhosted.org/packages/8b/47/420ddefc04f089e4497b3bcbadea37586477713300c13204fb4a7c736071/charset_normalizer-3.5.0-cp310-cp310-musllinux_1_2_riscv64.whl", hash = "sha256:ffdd7ac514301d0a67f7c23b9f2b431ef909a3c3dd6c3766668d0a6f5900c94e", size = 258669, upload-time = "2026-08-12T14:32:00.069Z" }, - { url = "https://files.pythonhosted.org/packages/58/73/01d5173097a5f0a13417e77db9ba3aea24de54421344c8e0c4973bf8cf21/charset_normalizer-3.5.0-cp310-cp310-musllinux_1_2_s390x.whl", hash = "sha256:3684ebbdffd51329ac44245d1d227d90b965797aa1a8abd026568a1f6ae88811", size = 277344, upload-time = "2026-08-12T14:32:01.211Z" }, - { url = "https://files.pythonhosted.org/packages/41/4a/300ef2598da520744aca7c49370e48732cab1f04de5f784f0e4f0e9e1d11/charset_normalizer-3.5.0-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:8b8788f114845c01f2b520e0b91ea58d143276cfc0483aa943e815f7b9555c15", size = 263440, upload-time = "2026-08-12T14:32:02.421Z" }, - { url = "https://files.pythonhosted.org/packages/be/5e/78dd366df9ae4810039c0f799ec47d325a5cb76e92bb707ca9039b7c2306/charset_normalizer-3.5.0-cp310-cp310-win32.whl", hash = "sha256:9a1d9b13e5e394e13e3c316f0d910d100b17681ff59797f30da1dba032061296", size = 181680, upload-time = "2026-08-12T14:32:03.932Z" }, - { url = "https://files.pythonhosted.org/packages/67/d6/455982837da317ce47cafe35260f446f218f5843090b77a59ee69aa0ad94/charset_normalizer-3.5.0-cp310-cp310-win_amd64.whl", hash = "sha256:5a54587f93f2e289f8faf25b35c997d4cc75cf677485ac6f50c985715989f99c", size = 205665, upload-time = "2026-08-12T14:32:05.045Z" }, - { url = "https://files.pythonhosted.org/packages/e5/e5/5b285d50d4fc82111c18ea4fff8b5e1916a257b98dcd5cd0abf1c165473b/charset_normalizer-3.5.0-cp310-cp310-win_arm64.whl", hash = "sha256:38a395079f229a631dece74e24c69c1f612536dd51f345a7d6a98abe2d3e047a", size = 184409, upload-time = "2026-08-12T14:32:06.503Z" }, - { url = "https://files.pythonhosted.org/packages/50/42/71e4e3bfe59202feef062c68487f54c6adf501cfbe087ecd93e3cd597fea/charset_normalizer-3.5.0-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:e46a37ea7fcf9ae01d71b2e5ece19f1565987f3e308394b829197cbefc061f92", size = 349110, upload-time = "2026-08-12T14:32:07.832Z" }, - { url = "https://files.pythonhosted.org/packages/0f/dd/fd3386d0fbd358d3b5c7a2fa5bf312afe6159b04fafeb67d39fa971d7448/charset_normalizer-3.5.0-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:1cdfed4d7a59333c8220c67dd3be4e7a6c887b67453a64394022dcc919570add", size = 250773, upload-time = "2026-08-12T14:32:09.113Z" }, - { url = "https://files.pythonhosted.org/packages/17/ad/4901a66d6d3b17f1096725d7e50266132c16555aa6a70047fe1cf262b4b2/charset_normalizer-3.5.0-cp311-cp311-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:9491f594859b68052edebd69e05fb045055a713b57a67974e6c1553b4e503c39", size = 240229, upload-time = "2026-08-12T14:32:10.43Z" }, - { url = "https://files.pythonhosted.org/packages/43/b7/1d790e0425e0f4c99e9b89a3956a94a6d2d0c6f01b2a5eca93d8f082d5ac/charset_normalizer-3.5.0-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:420b19411959eec115063229536788e6b32d0a7fa907d6b940317919120d702d", size = 280757, upload-time = "2026-08-12T14:32:11.628Z" }, - { url = "https://files.pythonhosted.org/packages/44/bb/4b8c8086c67636e52d6354ad17697f54a00b40041ca53dc765737e21709b/charset_normalizer-3.5.0-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:a565303d118ea3b94a4b6c076bf568069726be414e43b06d58f7070b076ce11d", size = 276174, upload-time = "2026-08-12T14:32:12.808Z" }, - { url = "https://files.pythonhosted.org/packages/25/fa/690a11924c40c766d258d2b74d817cc5efe7bbcfceeedc5c0f35256d7524/charset_normalizer-3.5.0-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:815f143a91983ba3041bba066e492ae3c42de523fb1c699685a1abf3313b7d1b", size = 261808, upload-time = "2026-08-12T14:32:14.138Z" }, - { url = "https://files.pythonhosted.org/packages/53/2d/4a6945eff0c8f684e3f5b7b978644ab18ba9198da060dbaa1d9206bc6cc9/charset_normalizer-3.5.0-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:c5e981a5ac8641381efe6f0029467500661616a530d27bc6eedfe45f840599f8", size = 259922, upload-time = "2026-08-12T14:32:15.485Z" }, - { url = "https://files.pythonhosted.org/packages/60/4d/10ac7e07bbf7ea569effeb9524e32f345f7e643800b20d538fb4706eab4e/charset_normalizer-3.5.0-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:1a573e1e428f93908e79e04b349717f400e720f2f82285f0aaaf3ee0ff7f4c79", size = 252315, upload-time = "2026-08-12T14:32:16.764Z" }, - { url = "https://files.pythonhosted.org/packages/13/bf/1ddacf7aa7da12097f229a6a4e71a70200937a621b3617f6dc819fb99a66/charset_normalizer-3.5.0-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:32e6d56dd825205f81e5c45bcebb4df6a11fb2bbf4969a01ef156d6ced90c224", size = 240600, upload-time = "2026-08-12T14:32:17.956Z" }, - { url = "https://files.pythonhosted.org/packages/c2/88/293444d48c8d86fe51552262117134a9fc666d68cbbbc9b8c1b2b35a29be/charset_normalizer-3.5.0-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:30ae26a1adcd943690dcbbc47f28be762bae9e08ad7442b78c86b1c0dd5a626c", size = 280788, upload-time = "2026-08-12T14:32:19.1Z" }, - { url = "https://files.pythonhosted.org/packages/12/ed/0e34e40584f51eda38d4a5daf25fd8586366347efd4b2470dbf64710e778/charset_normalizer-3.5.0-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:5f51a19dc52197a20218b05ec5336d0c6b3b09935f838724722032c8d45dc91a", size = 258425, upload-time = "2026-08-12T14:32:20.207Z" }, - { url = "https://files.pythonhosted.org/packages/72/b2/c1b1c27f6f0ef35b21a8bdb592854bbf7f219629db3477f74c0b1380e0ed/charset_normalizer-3.5.0-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:6e44bc2780516b3df986d6fe33103c7080cd9dcd5576fe3cb4b0f64309c8f22b", size = 277437, upload-time = "2026-08-12T14:32:21.345Z" }, - { url = "https://files.pythonhosted.org/packages/b2/bd/a1b7d959a37847675adb2f5978f81703306dd78df4fefb82ee5a1cc5e37f/charset_normalizer-3.5.0-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:7faa47b56070b3dd6f4898ed28528843ab130d53266cb9948d9b1f3bb1a5c5e8", size = 262947, upload-time = "2026-08-12T14:32:22.613Z" }, - { url = "https://files.pythonhosted.org/packages/c1/fd/a76a2d7e639bfc6a9c869371e34f28231eaca7d7126ec19895aa38eedb51/charset_normalizer-3.5.0-cp311-cp311-win32.whl", hash = "sha256:830c04a49998b5ed58c8b642c65b7b26419397f52392a64121ba9fd0e95e7f9f", size = 181313, upload-time = "2026-08-12T14:32:23.736Z" }, - { url = "https://files.pythonhosted.org/packages/97/84/6fc03e802578df41a2ab9b6a1f26657fb92e32285a603ad5852a4a4f68c1/charset_normalizer-3.5.0-cp311-cp311-win_amd64.whl", hash = "sha256:8cb9b6892b53bd6d11fa4cde3dbee020b1f0b6656be1fbaa1ec0d4324a7839db", size = 206197, upload-time = "2026-08-12T14:32:24.938Z" }, - { url = "https://files.pythonhosted.org/packages/0a/27/7208360ff1901607359869fcc45ca0989d597f3e30777ba30c7254170587/charset_normalizer-3.5.0-cp311-cp311-win_arm64.whl", hash = "sha256:2403b489c103e9a18c835863fc6dd54361355c8291d4cafdb37492b683440b9b", size = 184925, upload-time = "2026-08-12T14:32:26.123Z" }, - { url = "https://files.pythonhosted.org/packages/6d/3c/045ea64ea5a550870dd8ab60b2242870328d53f17d2be593b4f9f3121474/charset_normalizer-3.5.0-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:98820e1ceb25c6df7a80c4fd8efa59cb121f99bc7c4c1693ad94a2caff5b311d", size = 343861, upload-time = "2026-08-12T14:32:27.254Z" }, - { url = "https://files.pythonhosted.org/packages/fc/1b/7502be709db899d5b4801509829188b3a5a10969411da9c846115a5f1b70/charset_normalizer-3.5.0-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:608553f476fca509537e804c4a71f5eb166ce63b75141f89c2c686ce1aa36956", size = 237550, upload-time = "2026-08-12T14:32:28.385Z" }, - { url = "https://files.pythonhosted.org/packages/ca/ce/66392661375148d9455c17bee25509a54e28c39969e34befa48ec8777936/charset_normalizer-3.5.0-cp312-cp312-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:6753de11eef42f1c321b26d682957d92c7f7bbce6530f34bbe0f9291dd37cc6f", size = 229673, upload-time = "2026-08-12T14:32:29.668Z" }, - { url = "https://files.pythonhosted.org/packages/6a/64/f58c32a8d4ecf55b82ee61ee9aa6a664d4afcd36c72feb4c926fd6fe9af8/charset_normalizer-3.5.0-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:0f76dc0a47f94cb9b69d86f01e477f4b0371ca70208b9ccea7e063c41eed9046", size = 260768, upload-time = "2026-08-12T14:32:30.891Z" }, - { url = "https://files.pythonhosted.org/packages/b1/ea/36c90e59a96386174377e855479ec154221ef001e96637e0b23be92489c4/charset_normalizer-3.5.0-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:c387c6bf91b4774e359a48a179e2872b8e8bf741e4fde06ba8d1665eb9a4760a", size = 257880, upload-time = "2026-08-12T14:32:32.175Z" }, - { url = "https://files.pythonhosted.org/packages/23/35/5b85772eb82528ef22ba29487ad544a7049dfd27f35b1a5a55dbc0843048/charset_normalizer-3.5.0-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:14f6904a3cf870abf044df3a8c4924ac6c8ef77e9896586fd37e73ae96cff2af", size = 247547, upload-time = "2026-08-12T14:32:33.495Z" }, - { url = "https://files.pythonhosted.org/packages/b6/14/ba11a99c2a22ab04c2d5383a700b378cb463a78ab15f36444cabc10cd671/charset_normalizer-3.5.0-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:0cce46dd29d73e135e8087b96eb62a4aca6d69391b7f97808c6588ebed3178f3", size = 243326, upload-time = "2026-08-12T14:32:34.794Z" }, - { url = "https://files.pythonhosted.org/packages/4c/4a/cadba3f2400b45aa1d62a4ae0298bf58a3b30b1158baf15c338c7ce5b601/charset_normalizer-3.5.0-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:b476cdb63df22da2b91837593380be3ddbe406f36c506c1c91d80e7196b66288", size = 238820, upload-time = "2026-08-12T14:32:35.984Z" }, - { url = "https://files.pythonhosted.org/packages/47/41/d5188b9342d75b72c2b05d3ee373f01a691397e770f001ee05e3b37925f5/charset_normalizer-3.5.0-cp312-cp312-musllinux_1_2_armv7l.whl", hash = "sha256:1f56ce84b317ef2a59d7d3461891c7597c79247d2192bb8114c68a1a1debfcc0", size = 231661, upload-time = "2026-08-12T14:32:37.191Z" }, - { url = "https://files.pythonhosted.org/packages/13/5f/df38fa972c4e945c3d8cee2bc4e610613af522fd359c7dc7a74c419f0278/charset_normalizer-3.5.0-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:9ce0f885239357379d92fd9a5fddbe20f0e30e0527c29ba69f8e99eeb1304a76", size = 261459, upload-time = "2026-08-12T14:32:38.369Z" }, - { url = "https://files.pythonhosted.org/packages/a5/d7/043ff7720067a3beee05523465ac9c1c846c68b7884930dd483f72ee5ab6/charset_normalizer-3.5.0-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:96ae7ab5d8155fde927aa0864fbc8ba3cc4fde6d41ab0c7cea9d6012b4978603", size = 242300, upload-time = "2026-08-12T14:32:39.505Z" }, - { url = "https://files.pythonhosted.org/packages/19/f6/33980b7b802a048e546a6d9ad2ea783a6cf6b10a86aaccd10db462d8b913/charset_normalizer-3.5.0-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:bf91921009025e96ce57a03ced6d14604fc3baf0530351638e9504a55da6fa3b", size = 259101, upload-time = "2026-08-12T14:32:41.02Z" }, - { url = "https://files.pythonhosted.org/packages/a6/b7/0d19bde844bff9165377c1da9ef3c4792a4c24bd49b5b7094d9e6f6ab58b/charset_normalizer-3.5.0-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:0b2e44e6d42d1a4ff78ccc219a93c5449105d10b16198d1aea581080df8073f9", size = 249246, upload-time = "2026-08-12T14:32:42.47Z" }, - { url = "https://files.pythonhosted.org/packages/1c/cc/2c34fdfaacdf0e96e880ef562cbf80a9b5f8ea97e0dd9e57ba348a9c65cf/charset_normalizer-3.5.0-cp312-cp312-win32.whl", hash = "sha256:deb99535e9bf0bea8e274c6413eb939a21be35a3f492678dba4d5b1f4d70f142", size = 178025, upload-time = "2026-08-12T14:32:43.909Z" }, - { url = "https://files.pythonhosted.org/packages/76/d0/c34dbd1df23bcbdc1b5d2f48256340d72fa747f1eb03924a9d2fa35ed85b/charset_normalizer-3.5.0-cp312-cp312-win_amd64.whl", hash = "sha256:e54dd1a66fa4bce0ccaf0db9dde336e49b3eec646dc4c1c0991279369d373a14", size = 200143, upload-time = "2026-08-12T14:32:45.254Z" }, - { url = "https://files.pythonhosted.org/packages/c0/ea/4e6bf1465d60c3d8f488d5bde140d0cb91cd23ccab0dc2895cc6c6982047/charset_normalizer-3.5.0-cp312-cp312-win_arm64.whl", hash = "sha256:b8ea208b304587d47931b36481342d20336e0d338ab052f8b4305926482598d6", size = 180046, upload-time = "2026-08-12T14:32:46.545Z" }, - { url = "https://files.pythonhosted.org/packages/1d/be/cc7b7b6fc41984902c0d31b06f5d9297e67705c1dae9352608e5540fad09/charset_normalizer-3.5.0-cp313-cp313-android_24_arm64_v8a.whl", hash = "sha256:5c23fa4f6eccdd601949cb00f3988c01d64e671d8faba356397971077022e144", size = 211050, upload-time = "2026-08-12T14:32:47.81Z" }, - { url = "https://files.pythonhosted.org/packages/d3/ae/e3ec8f17313609f43f7b323012fdb1ee37b83432277ca4eceba83e00366c/charset_normalizer-3.5.0-cp313-cp313-android_24_x86_64.whl", hash = "sha256:07f6f42b5a6325df35b458004fb5f9f29bf502d89287a33c7cdef3590e31de0f", size = 222768, upload-time = "2026-08-12T14:32:49.027Z" }, - { url = "https://files.pythonhosted.org/packages/c5/50/9f9c0d7ccc1512d49e27a0e7c12c58ec71dfe91698fa4326f058c33e1f1b/charset_normalizer-3.5.0-cp313-cp313-ios_13_0_arm64_iphoneos.whl", hash = "sha256:8efc3f1563ed431882dd0dc0411b5f8ace1b1b89074981deaf6bd8af77dbe1bc", size = 193907, upload-time = "2026-08-12T14:32:50.414Z" }, - { url = "https://files.pythonhosted.org/packages/fd/1d/cfe7b745ef7f4c3b7214581955b5a0869ba2ac551a58fc11036281ae167c/charset_normalizer-3.5.0-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:368eb2fc9482158b3a3386e8f01fa61f479c968e9a19ceab8f0188b86b312991", size = 197135, upload-time = "2026-08-12T14:32:51.605Z" }, - { url = "https://files.pythonhosted.org/packages/28/55/30fafdcfca9ba616bc394240545e4cd52f4f66dea43ded81b7d2d5274fde/charset_normalizer-3.5.0-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:826a295a039178479a325be1ae60eded1f0b10f7dda749df59e2440de8f61d64", size = 339892, upload-time = "2026-08-12T14:32:52.82Z" }, - { url = "https://files.pythonhosted.org/packages/f4/08/bdca5fc2bdc36ee443673dc7d12b23885a5a7b282bef85a1a4c3b325b40e/charset_normalizer-3.5.0-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:70ff1c16eb0eb5ee6bb12739292347f981a5ba764cc4df1bc2e69b0405d4ac3b", size = 239439, upload-time = "2026-08-12T14:32:54.058Z" }, - { url = "https://files.pythonhosted.org/packages/d1/9e/506c8d7a7722bba7c8cdd78c1b5ef23bda92bfbe0b3e28ea84673d519a0f/charset_normalizer-3.5.0-cp313-cp313-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:3cfdab178a4add5483e26a9bb1c16d8018ccf39b4be7a3aea6c3979e6828f2ee", size = 227896, upload-time = "2026-08-12T14:32:55.326Z" }, - { url = "https://files.pythonhosted.org/packages/0d/1a/dd828f2b1d6f4bf10821b9a74d866be05ffcdbfcddfc501d6fe6428762a7/charset_normalizer-3.5.0-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:6083d10a846218502d664375b9448508d9fa580bd834567423156c6abfbe899d", size = 262548, upload-time = "2026-08-12T14:32:56.484Z" }, - { url = "https://files.pythonhosted.org/packages/be/81/196d26f6bd78b93e0d451b69082a71027ceeddd4b0be9170b81bb038f824/charset_normalizer-3.5.0-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:0f211c21aa316cb6e2662e54a1194633a79d98a50a876addacfce7ba5b34b09f", size = 259986, upload-time = "2026-08-12T14:32:57.661Z" }, - { url = "https://files.pythonhosted.org/packages/3a/6a/5b964a1eb0f9075ecd45083eeb21aaec215334f98bac3d400302ea73875d/charset_normalizer-3.5.0-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:3b08ebf9488c7ff5eff038e48e6ea938178dfd9dcc8598b5ca941e4ae27b20be", size = 249853, upload-time = "2026-08-12T14:32:59.123Z" }, - { url = "https://files.pythonhosted.org/packages/c8/b4/aee3a9d82edd0e931091ef3e9f03e46491ae3590e96e998d0975dadbe17c/charset_normalizer-3.5.0-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:6c95450fce59f00c6d08eff6572ec2e736e5054c9450253afd5748f8416f2eb9", size = 244217, upload-time = "2026-08-12T14:33:00.341Z" }, - { url = "https://files.pythonhosted.org/packages/22/3e/33f72ca11c1b619b220fd9f35905ebd171cbd0e0470f2357e467b9e861ee/charset_normalizer-3.5.0-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:5780a29823e1d2bec69b7a104ead4195a43f3e97782efaedbf1f79a0157af715", size = 241307, upload-time = "2026-08-12T14:33:01.589Z" }, - { url = "https://files.pythonhosted.org/packages/78/27/6029dccba958621c7f3a65136f87c5512d712aef9e890f09512cc171bd03/charset_normalizer-3.5.0-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:054420b5db984971d886e5e4e2c37c760ae6682aedbd066687ff0949d9ed5f08", size = 233305, upload-time = "2026-08-12T14:33:02.866Z" }, - { url = "https://files.pythonhosted.org/packages/18/d7/691c967be459153fe9faf49bf78bc95639ef8bf6dd008f38cc6389a349eb/charset_normalizer-3.5.0-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:d016dc857136c726958102c3b8a3986acdc65ace6fbf12cfdc09cc4bfa2935b2", size = 263465, upload-time = "2026-08-12T14:33:04.166Z" }, - { url = "https://files.pythonhosted.org/packages/7c/2d/9202221be5c90b2a835924191e362690ed8dc8c7d6606100c2bd03fe0f8c/charset_normalizer-3.5.0-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:f2ce3d39fb4a9d674e6639dd5d3146b2e273475d2260f10163228d66fc04433d", size = 245060, upload-time = "2026-08-12T14:33:05.325Z" }, - { url = "https://files.pythonhosted.org/packages/b3/9a/298772fd0a0cbccadf36451a1cd7eef4b66a11e99b4a7f6fafc47cc62c75/charset_normalizer-3.5.0-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:a5613a3a82c974227bde18f03409e30c467f8065cb56d822e3eb83708a5f223d", size = 261091, upload-time = "2026-08-12T14:33:06.475Z" }, - { url = "https://files.pythonhosted.org/packages/82/3b/1a11fe66e555dbe2f5714ade6ba74fa29edc9155d9cf1001d4d6ed096aa7/charset_normalizer-3.5.0-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:fded2e82ff082e5d8e017e2ddcc1411bd8cb83b8585097fc401ef574f756b888", size = 251857, upload-time = "2026-08-12T14:33:07.784Z" }, - { url = "https://files.pythonhosted.org/packages/17/fc/73b817e8af3f1d25ec5cf458d405abba5a144cf9812238a61530f5eac186/charset_normalizer-3.5.0-cp313-cp313-pyemscripten_2025_0_wasm32.whl", hash = "sha256:8f006866047c6ec4b627ec144b1e0bbc7427cb31fd7c08d19897d0ac9032af3d", size = 139745, upload-time = "2026-08-12T14:33:09.123Z" }, - { url = "https://files.pythonhosted.org/packages/b3/fb/ddb66303c86f7dc5043a457dad9fa82b4d6d0cb97094f9bcdde21693fd58/charset_normalizer-3.5.0-cp313-cp313-win32.whl", hash = "sha256:196e270c4e80827b5072eed7d6aa661d133afada94fe366669f9609e718d305e", size = 177217, upload-time = "2026-08-12T14:33:10.305Z" }, - { url = "https://files.pythonhosted.org/packages/fb/88/6018cc8d76ea2b7cb02918f37e23e86c261d1a102713d7e88d2cfb8b211c/charset_normalizer-3.5.0-cp313-cp313-win_amd64.whl", hash = "sha256:72982d9958a42f8132bf2d6b90214ed66477295ef1188731f98ae3511c6eeb5a", size = 198896, upload-time = "2026-08-12T14:33:11.51Z" }, - { url = "https://files.pythonhosted.org/packages/20/2e/04c0bbfc8d9abf91959f7a3d207d45cbf63a8116984caae2381890019bb5/charset_normalizer-3.5.0-cp313-cp313-win_arm64.whl", hash = "sha256:0b373bab0b867b68b8eb249da9478cab9181a42993437cd2f5dba5fb0b4fbd1b", size = 179193, upload-time = "2026-08-12T14:33:12.726Z" }, - { url = "https://files.pythonhosted.org/packages/43/14/d098868dac5ff27e0258f548b1c74c6484be528384965d8fcf8fc6a4011d/charset_normalizer-3.5.0-cp314-cp314-android_24_arm64_v8a.whl", hash = "sha256:d95244906ed69d0f79f190893c65e336c15959003e21449256dc05c001b52ea2", size = 211664, upload-time = "2026-08-12T14:33:14.153Z" }, - { url = "https://files.pythonhosted.org/packages/e7/da/a944b32a46601ae5a4c3499e8d64ecd14fe82313f00da74dcdf00273a0b4/charset_normalizer-3.5.0-cp314-cp314-android_24_x86_64.whl", hash = "sha256:d788e2ded0c4c47efa4d73cfe59eaf975ee32f425219873d2cb3e3fbaa00f636", size = 224375, upload-time = "2026-08-12T14:33:15.472Z" }, - { url = "https://files.pythonhosted.org/packages/f7/db/eabb5996be2f529744755e7b2fc9396eff4a64961f034e7fd49d54b9afb2/charset_normalizer-3.5.0-cp314-cp314-ios_13_0_arm64_iphoneos.whl", hash = "sha256:f9f91d3e8382900f3a68fa0ce94294479de9cd2de6bc0c70acd0f0dfd511836b", size = 194364, upload-time = "2026-08-12T14:33:16.607Z" }, - { url = "https://files.pythonhosted.org/packages/78/65/4ad3c5be108930310d8003f5602861d5b89f728293b9f09c3a4837f7ba10/charset_normalizer-3.5.0-cp314-cp314-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:d54625cbf4e6b60bf0639728cb8b4cb541e340f6d7cafae5806051a40ddf4c45", size = 197643, upload-time = "2026-08-12T14:33:17.88Z" }, - { url = "https://files.pythonhosted.org/packages/3d/39/8fee3201b98d52289be60a775797d69be05a04fb6cfb48c1587dad33e649/charset_normalizer-3.5.0-cp314-cp314-macosx_10_15_universal2.whl", hash = "sha256:1f99a8c3a1da5d955edbad18208b3d627bdd54c48a6e739fa877bdca98c686d6", size = 341384, upload-time = "2026-08-12T14:33:19.239Z" }, - { url = "https://files.pythonhosted.org/packages/a7/dd/9e757101d1f76c35c0643684ba499ac3a181fb2b264c68174bf727d627e8/charset_normalizer-3.5.0-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:bf1e75dc07a3850b53d1e5f75e04d3ae12afe56284be7821771eaa2466350c73", size = 241637, upload-time = "2026-08-12T14:33:20.619Z" }, - { url = "https://files.pythonhosted.org/packages/eb/e4/7857023015400bc4aa0a82fbcca29fa2dc7ec25f971a130764cb2dc7a589/charset_normalizer-3.5.0-cp314-cp314-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:ac5a9cc079c67d75f4ddf343276031879eadbb333d1bb231cce297b8d7b9aae8", size = 226170, upload-time = "2026-08-12T14:33:21.773Z" }, - { url = "https://files.pythonhosted.org/packages/7d/ae/8b52935b304f7b6bbf33151ed2b75266b09aa4b6f8f04230d948885b2577/charset_normalizer-3.5.0-cp314-cp314-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:0dfe83c1b4d00abbf433998117a14f56a5c2bc68226c0d331709eed0d1ce539b", size = 265093, upload-time = "2026-08-12T14:33:22.999Z" }, - { url = "https://files.pythonhosted.org/packages/dc/78/6e838f6bb059f2c0afc60a4e7f294252f043c254656ad4114c50302cae4d/charset_normalizer-3.5.0-cp314-cp314-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:e4e8fa586df2208ef040684751345f10f503834a757c9a74ecd19c1a2f9b1ccd", size = 262789, upload-time = "2026-08-12T14:33:24.214Z" }, - { url = "https://files.pythonhosted.org/packages/c2/08/189b27e51fddc9d6b3695331da0e31792c1d88b953ad854e57f06e9b2cc8/charset_normalizer-3.5.0-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:82cc5835997ec78afe293a192e385099355770a7db94b2fb1239d36b32796f1c", size = 250580, upload-time = "2026-08-12T14:33:25.707Z" }, - { url = "https://files.pythonhosted.org/packages/ac/55/64854e99b25841f83e8e37d9df2f3d1f96f693439f80e5fabd542a7e47ab/charset_normalizer-3.5.0-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:19e52bda45086df8a4be4bb5910af6f5d9d3b538c78712c8ae09ef10b85bf458", size = 245008, upload-time = "2026-08-12T14:33:26.971Z" }, - { url = "https://files.pythonhosted.org/packages/4f/01/7720c904fa635d4260b4dced6029cf3d298c57b26741365d5a8d28c54043/charset_normalizer-3.5.0-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:3418edd0ecb72a0a3861cf72f31be0ad9b7fe338ce2b58fb5cc80b9aeb792700", size = 243892, upload-time = "2026-08-12T14:33:28.237Z" }, - { url = "https://files.pythonhosted.org/packages/70/50/7bfcb327631d4870c720872b548745f6ec8baa044d51c21b5d1d32ac4e3a/charset_normalizer-3.5.0-cp314-cp314-musllinux_1_2_armv7l.whl", hash = "sha256:125ee619611019471b177c70bc3e9d4cda9fad7e01d93523501d3b188df0193a", size = 230996, upload-time = "2026-08-12T14:33:29.511Z" }, - { url = "https://files.pythonhosted.org/packages/24/51/40c45d6d940c04005ed721aa54bdebf1ebb2930f8a2ae537e8d60484fb27/charset_normalizer-3.5.0-cp314-cp314-musllinux_1_2_ppc64le.whl", hash = "sha256:a3ad0e3da22852533858663848608f3f24c0d35e5cde415a4903476f2b4c88ec", size = 265834, upload-time = "2026-08-12T14:33:30.689Z" }, - { url = "https://files.pythonhosted.org/packages/eb/d4/ef7a227ef89d215b47f9df79c3966610b17faa13bb2f236989207a631622/charset_normalizer-3.5.0-cp314-cp314-musllinux_1_2_riscv64.whl", hash = "sha256:4ebebb410bc517e1d284c52a123e82704b21e4e7e26a21ebecf7439d0647b8a3", size = 245544, upload-time = "2026-08-12T14:33:31.859Z" }, - { url = "https://files.pythonhosted.org/packages/37/a9/a4ca9156964ded61c7718eba410ce11be2fd2b263fda4bcf08367b6578cd/charset_normalizer-3.5.0-cp314-cp314-musllinux_1_2_s390x.whl", hash = "sha256:91f9f7c151e772acebe489eaec96e96a2877202d7dd144e3f96b8676881715a0", size = 264110, upload-time = "2026-08-12T14:33:33.13Z" }, - { url = "https://files.pythonhosted.org/packages/38/6a/838364bb8702229c6e5f8b23f80ff0f052a12dfaf3113a12fd6acbe92a44/charset_normalizer-3.5.0-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:401ea6e7af9e7852ed818f64714b579c1935482049670847ca3bd7ba45dc63fb", size = 252303, upload-time = "2026-08-12T14:33:34.98Z" }, - { url = "https://files.pythonhosted.org/packages/9c/5f/d88032edce951f499a2321cf7ae0d35a043c74be12bc22d81084cc7afbcc/charset_normalizer-3.5.0-cp314-cp314-pyemscripten_2026_0_wasm32.whl", hash = "sha256:f7496aed56b06325a1ad419c5bf23c6dd042558e874f71dd1b958f3e255f3053", size = 139964, upload-time = "2026-08-12T14:33:36.195Z" }, - { url = "https://files.pythonhosted.org/packages/37/ae/1c4a46b6b00d1c34d2ee355ef99ad6173674166800d1af0f05f85028d513/charset_normalizer-3.5.0-cp314-cp314-win32.whl", hash = "sha256:606a86c1c3196f3738de39a67a7490bbd61cb31c0e0436070bd0c6a48170b38e", size = 179790, upload-time = "2026-08-12T14:33:37.356Z" }, - { url = "https://files.pythonhosted.org/packages/01/51/f94dcf34fa8eba48c1fb89b6490a5f1426e19488fe5f38aac6c648c99057/charset_normalizer-3.5.0-cp314-cp314-win_amd64.whl", hash = "sha256:ec6c464cf45867f66a2273e2214d9199a8fbad5cb95ca0fd45f6a2fe1d9d2cf4", size = 203723, upload-time = "2026-08-12T14:33:38.639Z" }, - { url = "https://files.pythonhosted.org/packages/9f/ba/91d386870b5d9e4b0d8c4034f63877cc2e47b99c81ef05f3e6d42bf9a53f/charset_normalizer-3.5.0-cp314-cp314-win_arm64.whl", hash = "sha256:dc28949de1bb5f7f30a46f15d74ce7ac5aaa63e03c5de04d68f571c7423af834", size = 183423, upload-time = "2026-08-12T14:33:39.899Z" }, - { url = "https://files.pythonhosted.org/packages/f1/c9/534ecb17b7fb95f9052c4a44cf316316a27d4a8f73e8475ff55e778dcdd7/charset_normalizer-3.5.0-cp314-cp314t-macosx_10_15_universal2.whl", hash = "sha256:68b7e84ae8239a94f8d2c8f3f3a3a81bcde54805ec8f42a34de927d155688ec6", size = 368967, upload-time = "2026-08-12T14:33:41.093Z" }, - { url = "https://files.pythonhosted.org/packages/3c/b2/ad7c3242d7fe55cd55126c22c65cb1b49779782cdf8932fd01d12232d86a/charset_normalizer-3.5.0-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:58ca5dc0a0ef99f2801ec0574214c978e9574055bc783830bbb6e7433218609f", size = 239478, upload-time = "2026-08-12T14:33:42.428Z" }, - { url = "https://files.pythonhosted.org/packages/7e/62/77f0b850048e430fc350ec58876b0c020f5c8d0d3956fd1a4d6ae2fa292f/charset_normalizer-3.5.0-cp314-cp314t-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:6c57af4084c10cb3286688d65e4c654190ff5edcbc2411d08cdca0a8a44c59a1", size = 227036, upload-time = "2026-08-12T14:33:43.635Z" }, - { url = "https://files.pythonhosted.org/packages/f0/ba/47d951e1a51dddbaad0a1410baf49fb1d897ceb00281568f1183b79bce9a/charset_normalizer-3.5.0-cp314-cp314t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:f1619a3cc174a7e3963dd34348e6fceb6e50db0ddeb0031bd7c73a58286454fa", size = 260772, upload-time = "2026-08-12T14:33:44.96Z" }, - { url = "https://files.pythonhosted.org/packages/b0/61/8c7ff4c81b2a88271126acf4b83ab3e31f6d63868b0f01d331eaa0f9cb67/charset_normalizer-3.5.0-cp314-cp314t-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:1328cc57dd4372be1265f68232cee890e087416e3e6e93e6ffb32c2bad4d36a4", size = 259273, upload-time = "2026-08-12T14:33:46.185Z" }, - { url = "https://files.pythonhosted.org/packages/e1/fd/36129689be08dc287b951306946657ff70d76e287dd57018861f86d0e474/charset_normalizer-3.5.0-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:06f4fb62a9139bef056b8b2da6773c94c2f259f90e4b8e53b166f3d0372d7cf6", size = 248086, upload-time = "2026-08-12T14:33:47.54Z" }, - { url = "https://files.pythonhosted.org/packages/cf/f8/bcae67f994c8fd31dda445e5ebf84045823c31443fe46f0e9ee6aca99aa0/charset_normalizer-3.5.0-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:478650a70a750d75d5add401606c77f77069c32e4ba2c9131dc6cee566962ca0", size = 242671, upload-time = "2026-08-12T14:33:48.746Z" }, - { url = "https://files.pythonhosted.org/packages/61/92/0472cdad1061c2f0e4d3aee29973eb6e81bb8fe256ff2860cf115b15f1c9/charset_normalizer-3.5.0-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:48920bf6fe83eb2226756ac623fa54940487154eb18f80889d5735cf234965c0", size = 241311, upload-time = "2026-08-12T14:33:50.152Z" }, - { url = "https://files.pythonhosted.org/packages/f2/89/04a03de5d27c77c624d9fcf6287073754bd438df1b58cb7d030c57c2824d/charset_normalizer-3.5.0-cp314-cp314t-musllinux_1_2_armv7l.whl", hash = "sha256:168a0cb536b5123a77bc42ecf5e0bf6f923d0d9ae43c42a14eb0677c19ac6c19", size = 229898, upload-time = "2026-08-12T14:33:51.523Z" }, - { url = "https://files.pythonhosted.org/packages/42/a2/639c4278adcb7ed1f4db608dd9ac19b6774fa2285a96b1c0bdb9c124ccbd/charset_normalizer-3.5.0-cp314-cp314t-musllinux_1_2_ppc64le.whl", hash = "sha256:f278e131afa96a3622cef9211c406ea2ad1b68eb06f8837cd443684a40e0ae50", size = 262852, upload-time = "2026-08-12T14:33:52.924Z" }, - { url = "https://files.pythonhosted.org/packages/9d/95/02e34c97bedfd0c5574efb9179c850591acc7f967ba039ed8dd29d332b73/charset_normalizer-3.5.0-cp314-cp314t-musllinux_1_2_riscv64.whl", hash = "sha256:d6100f877d2ed95f0856a3fde25334153add94bf2224c43f45f88e7039262aaa", size = 242913, upload-time = "2026-08-12T14:33:54.18Z" }, - { url = "https://files.pythonhosted.org/packages/a3/64/0946aeab6462dad9f160a50dfb4704d3f58a5ee708f085abc2105fbbff0c/charset_normalizer-3.5.0-cp314-cp314t-musllinux_1_2_s390x.whl", hash = "sha256:4c440122e1ea68b1f8b44a631ebf49c39180f6869b1da22d76e8a724208ec6e9", size = 257938, upload-time = "2026-08-12T14:33:55.802Z" }, - { url = "https://files.pythonhosted.org/packages/79/77/36787d41ead124746506a4425c729f4f17c68280af8a6a5baa0a598cae86/charset_normalizer-3.5.0-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:e5f834965c2fe589837bac1002e07e25734ff70381903ccd95b3d649e22bfa40", size = 249467, upload-time = "2026-08-12T14:33:57.081Z" }, - { url = "https://files.pythonhosted.org/packages/65/10/d9f6c5589cd24198d4ce6cd2948191c18e657272f433e5a00d258d9f5c22/charset_normalizer-3.5.0-cp314-cp314t-win32.whl", hash = "sha256:076cf9d3f3c7e410295c09d96355cf3b1bcae74990034d80e4371e20fe1ba4c6", size = 190624, upload-time = "2026-08-12T14:33:58.449Z" }, - { url = "https://files.pythonhosted.org/packages/6c/81/43e0584a802051a22c725795ebe1df78263abc7de858eef6cdc9b36637e9/charset_normalizer-3.5.0-cp314-cp314t-win_amd64.whl", hash = "sha256:3288a560dc3114d5d2ebe309b1ef43f8af355eafe25856832415c2a8196c9db3", size = 215902, upload-time = "2026-08-12T14:33:59.753Z" }, - { url = "https://files.pythonhosted.org/packages/30/f3/af6a1160fef0eac4510d035241e11eccf78e5350e4cd4de79e79fe02a5e5/charset_normalizer-3.5.0-cp314-cp314t-win_arm64.whl", hash = "sha256:a284c36b9c6616bf0a8aa4aabba668a0c75ba65ccf40a79868aeaa69ad996897", size = 193452, upload-time = "2026-08-12T14:34:01.017Z" }, - { url = "https://files.pythonhosted.org/packages/42/a4/dee470afb7a55c4f78b6fef37306c51fed17ebf94dbe530798c91d394350/charset_normalizer-3.5.0-cp315-cp315-macosx_10_15_universal2.whl", hash = "sha256:c38d1e9bc2073b0984d2099ea647fd7f6c0d8f83a1e14e0cd32926f16e4c44ce", size = 341595, upload-time = "2026-08-12T14:34:02.4Z" }, - { url = "https://files.pythonhosted.org/packages/6a/32/9c3126dc429c6d9d7f79c52681a7c4453ed20a26267c9a8275d7ab620aba/charset_normalizer-3.5.0-cp315-cp315-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:c9f45186390aee4d1f26f723c615b67df346766c3b16df000d84d6e374f06757", size = 242177, upload-time = "2026-08-12T14:34:03.741Z" }, - { url = "https://files.pythonhosted.org/packages/4b/9d/5b616a887301ff4cc0916b39ba44257390d3da80deeed6e8b6f2f26b14a8/charset_normalizer-3.5.0-cp315-cp315-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:f0fde5e5100c735b2274ab898f0742a5dcde492796296cfbe7e0ad6a4cd1a396", size = 236730, upload-time = "2026-08-12T14:34:04.991Z" }, - { url = "https://files.pythonhosted.org/packages/be/b4/d6d3e70be93ebe5fabef65e4c7ac113e1d1705cbaeb5fb72467e713aca17/charset_normalizer-3.5.0-cp315-cp315-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:3d00e18e7bbf47e332ab63903d18bae31efc701b1d8cca0382b97784a621fc44", size = 265158, upload-time = "2026-08-12T14:34:06.235Z" }, - { url = "https://files.pythonhosted.org/packages/30/e7/3f1fafa87e2643257474f9c4eec609f2193a61d907dce7dd4f3f2390ebd5/charset_normalizer-3.5.0-cp315-cp315-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:5b81980668800dd1c69faad8aea6e85a8cee0e13bcd3bba7671695ff16260293", size = 262931, upload-time = "2026-08-12T14:34:07.511Z" }, - { url = "https://files.pythonhosted.org/packages/0a/df/ebeb224a949d91829e5e114c6b64372a3c792b00762a9e951ce416f3a32d/charset_normalizer-3.5.0-cp315-cp315-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:9bb3e0d1345b9c0fe73673ea656375f38a78ec679c2edeae0c24800f04798a85", size = 251388, upload-time = "2026-08-12T14:34:08.949Z" }, - { url = "https://files.pythonhosted.org/packages/a0/64/9a6ce2e7acc5cf1b4636f78f82e89ff581e06a0216a40678b28bd4d832c4/charset_normalizer-3.5.0-cp315-cp315-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:2401f7671242e921e604f609d429f6b282ea4ca787a6ffd22ed7372011ddb9d1", size = 251821, upload-time = "2026-08-12T14:34:10.138Z" }, - { url = "https://files.pythonhosted.org/packages/f1/b1/6e69b8056f615e5ccff6b91ca16db2d47922251f016821a300c115267fef/charset_normalizer-3.5.0-cp315-cp315-musllinux_1_2_aarch64.whl", hash = "sha256:96720f2aeed3434bc48f4d52fbad64ecc820cfed88915d664780ed9ba09ede78", size = 244507, upload-time = "2026-08-12T14:34:11.488Z" }, - { url = "https://files.pythonhosted.org/packages/0f/34/02c15d6a0aa6b934dcdc136b111da63ae857b9fd51cf5505b0736337c2eb/charset_normalizer-3.5.0-cp315-cp315-musllinux_1_2_armv7l.whl", hash = "sha256:c455829625df983f716cbaecbba77f2d1dc2e0e0ed1638c059cece15a279344b", size = 240951, upload-time = "2026-08-12T14:34:12.991Z" }, - { url = "https://files.pythonhosted.org/packages/ae/15/0fe893d3e1c7d111280bd6c4bd4c1e431487a1124a1bcbce78dfeda3a3a8/charset_normalizer-3.5.0-cp315-cp315-musllinux_1_2_ppc64le.whl", hash = "sha256:c41b067eddcfa5ee6b1169c287605be7fb6b0ea22bba6474c5bb978a668def4f", size = 266162, upload-time = "2026-08-12T14:34:14.232Z" }, - { url = "https://files.pythonhosted.org/packages/55/ea/eca03527307670f5d102c295671a800c404ca958cf94fefd10fc963a72f0/charset_normalizer-3.5.0-cp315-cp315-musllinux_1_2_riscv64.whl", hash = "sha256:e31786a947b136329bfdc458c82c06d4ec539b4a4436b7da4df4aafc9902ee80", size = 251835, upload-time = "2026-08-12T14:34:15.48Z" }, - { url = "https://files.pythonhosted.org/packages/03/a8/fee5633081e595fe9e191df6f215106c791ad596eddf5e41e39b8ea0f2e2/charset_normalizer-3.5.0-cp315-cp315-musllinux_1_2_s390x.whl", hash = "sha256:c5c6d47a865147e0ae3322ce92e7fb52ba3169d94b447deda56897ea2aa6fac9", size = 264314, upload-time = "2026-08-12T14:34:16.679Z" }, - { url = "https://files.pythonhosted.org/packages/cf/fb/17f47ae6ca35b562fb6e6f4b05f7aec6034217353eb4a23aaa3566dc7340/charset_normalizer-3.5.0-cp315-cp315-musllinux_1_2_x86_64.whl", hash = "sha256:c75191e3c8052045179646cb40e280800a4e0bdfda34d9c949c2f268d44e80e4", size = 253194, upload-time = "2026-08-12T14:34:17.975Z" }, - { url = "https://files.pythonhosted.org/packages/59/88/f2b0f7ebb92493e925889ff29239b3b0073ffafd91230dbfc69e5cf9389c/charset_normalizer-3.5.0-cp315-cp315-win32.whl", hash = "sha256:83b62410bd36bb1178a7d563e2ee0cf21eb1c980c912ab99c2c78f06227f1731", size = 179800, upload-time = "2026-08-12T14:34:19.409Z" }, - { url = "https://files.pythonhosted.org/packages/e2/f0/afb5bfdea52fd943b1960403847a276b8e900c6e4cd6a38752321b4eda64/charset_normalizer-3.5.0-cp315-cp315-win_amd64.whl", hash = "sha256:e3b9eaa99a6d8c9ace4cd303915947ef55088d4cd87c6676874f98c5c03aa040", size = 203726, upload-time = "2026-08-12T14:34:20.656Z" }, - { url = "https://files.pythonhosted.org/packages/fc/71/219783eb691aa2ec879c0e521afdfe2b826f9678eed51b9c039d03e0db2b/charset_normalizer-3.5.0-cp315-cp315-win_arm64.whl", hash = "sha256:fec352b793cdc183cc9e7e0b6c10fd7bff38ec54ba44cc43599b9b56f7f3db2e", size = 183428, upload-time = "2026-08-12T14:34:21.975Z" }, - { url = "https://files.pythonhosted.org/packages/d1/d0/14aef3b9f80f2593c039d897e89034635b9eb0eb44b6ce5173bbd79ff338/charset_normalizer-3.5.0-cp315-cp315t-macosx_10_15_universal2.whl", hash = "sha256:c9bde7a960720c8b8e1b5ef7afaa0c9a2f3b55c44abd635b2b29dd066b298e3a", size = 368728, upload-time = "2026-08-12T14:34:23.221Z" }, - { url = "https://files.pythonhosted.org/packages/10/fc/b249466ddbbeffa448b6597631e9091d1f01b5132ff8e7a0e21a6eb72b63/charset_normalizer-3.5.0-cp315-cp315t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:f8cd1283a9fe6c2065c807e9d5da81afe5e1e004caef39adc0d8ae86dd883698", size = 240925, upload-time = "2026-08-12T14:34:24.504Z" }, - { url = "https://files.pythonhosted.org/packages/2b/b9/c17e72aaa1b3e1ca6c184e8025cf138ed492d01a54f85286ff7d31253a4b/charset_normalizer-3.5.0-cp315-cp315t-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:1c010dd86d3f4c4433c9634d33ce8147393b270dfa54f217f965540b8ae8e075", size = 234932, upload-time = "2026-08-12T14:34:25.822Z" }, - { url = "https://files.pythonhosted.org/packages/18/d7/f84ef0966bbe216f71029e34e7fa425a16b1682e2a40265e679dedf2b655/charset_normalizer-3.5.0-cp315-cp315t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:5e68229977b2dea28e7061c0c0630a23f2f9f6e9c6fb38d77d3d6dbfe3768b74", size = 261733, upload-time = "2026-08-12T14:34:27.112Z" }, - { url = "https://files.pythonhosted.org/packages/3b/73/3e887fa0781a395339355ed934ab6561ceb5bb52574160f070224039c630/charset_normalizer-3.5.0-cp315-cp315t-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:a60773eb5fda796e6e6f76b9c152d270fe59f9788a51a6ff8ba44082d8548ae4", size = 258460, upload-time = "2026-08-12T14:34:28.431Z" }, - { url = "https://files.pythonhosted.org/packages/b3/81/52ebd9849bf9e35d0b21fff115cb6543162a8e1f2f564e8f87121a336b8c/charset_normalizer-3.5.0-cp315-cp315t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:d9419f44e568f7fafcdc0b3b5c766a2364e705a9b34fb8a56b431e0d1f3f4258", size = 249894, upload-time = "2026-08-12T14:34:29.698Z" }, - { url = "https://files.pythonhosted.org/packages/85/f3/9366492b8a5fe0187de282e001d61345740cf79eb4a5f20181d769be02b5/charset_normalizer-3.5.0-cp315-cp315t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:75e243abbb528c1a774390ed71e3f868a9f37b1373442e4bbadd401cfc505ff4", size = 249540, upload-time = "2026-08-12T14:34:30.96Z" }, - { url = "https://files.pythonhosted.org/packages/0b/af/28bb5e5dbd3e67cb9196a62781ac2b6d79492f4fc7a069b6ca7d6d6c8d58/charset_normalizer-3.5.0-cp315-cp315t-musllinux_1_2_aarch64.whl", hash = "sha256:54c963ce6404e52255b737e8a06d356fc762d59096ae566203a67cf2b7d050f2", size = 242734, upload-time = "2026-08-12T14:34:32.481Z" }, - { url = "https://files.pythonhosted.org/packages/26/d6/7ccfa62b53b40fc06b2d3504825aa400764740bd10cf248fdc4272441b93/charset_normalizer-3.5.0-cp315-cp315t-musllinux_1_2_armv7l.whl", hash = "sha256:63ea0cc840c66670183578c2630d138c0e944aeadfc33f25173ee240f5db780d", size = 239580, upload-time = "2026-08-12T14:34:33.916Z" }, - { url = "https://files.pythonhosted.org/packages/0b/82/71c0c9b046697b8da66b3acefa8d5f92d00a9ef433ad7c3522b971d0369a/charset_normalizer-3.5.0-cp315-cp315t-musllinux_1_2_ppc64le.whl", hash = "sha256:6c06875a1d4a7537bef70f659b55c6b55b9a47ec3ba8f2db610350c2d9915e6e", size = 263281, upload-time = "2026-08-12T14:34:35.333Z" }, - { url = "https://files.pythonhosted.org/packages/d6/01/d027583c869f40ba980c1c76994adbd522c360a6327e72beb44d7c267385/charset_normalizer-3.5.0-cp315-cp315t-musllinux_1_2_riscv64.whl", hash = "sha256:4253da1b4456b633651a8d59eb1dc7a8a8fa38241014dd7c217b353e547ae394", size = 250027, upload-time = "2026-08-12T14:34:36.991Z" }, - { url = "https://files.pythonhosted.org/packages/2a/e9/6475d739e0ec8bb1236e06263dc3affaffdf947d8114ad27024932f325da/charset_normalizer-3.5.0-cp315-cp315t-musllinux_1_2_s390x.whl", hash = "sha256:f044cb1cf44012184715f46584658993b5fee9344d71c4b0c455a17a299730c0", size = 257547, upload-time = "2026-08-12T14:34:38.277Z" }, - { url = "https://files.pythonhosted.org/packages/a5/60/d1f502fcaa048a2aca3ab80bfef8407659c131e4f1792fa805fec14b4960/charset_normalizer-3.5.0-cp315-cp315t-musllinux_1_2_x86_64.whl", hash = "sha256:a17864853f7c518ae7d4b368af98f427f9396805476af40af8698560f09d7d97", size = 251718, upload-time = "2026-08-12T14:34:39.562Z" }, - { url = "https://files.pythonhosted.org/packages/c3/69/76343dcf4381a698807ff8a20d89f66bbdd9f6222b0b17740f77ab764335/charset_normalizer-3.5.0-cp315-cp315t-win32.whl", hash = "sha256:9e726478d7a213847860219d74665a6892a643ac93b8f76580f6cf9ed39996b7", size = 190757, upload-time = "2026-08-12T14:34:41.053Z" }, - { url = "https://files.pythonhosted.org/packages/e1/ea/d18147626a1667cc773c42104ab155a4ca5d6d4d174b7a35e01062213ea5/charset_normalizer-3.5.0-cp315-cp315t-win_amd64.whl", hash = "sha256:d7229a99120c6c2792d96f4857c2648ce5530e93667a2c2388c5ef69a6b84775", size = 215431, upload-time = "2026-08-12T14:34:42.501Z" }, - { url = "https://files.pythonhosted.org/packages/47/21/4869598aae0872d94faa5933918a4fe37ab2c5af9d095786e241f9506fed/charset_normalizer-3.5.0-cp315-cp315t-win_arm64.whl", hash = "sha256:527e28a5e751d9e11369b9c5f9ab35c748eb9c109101920c7deb40d6eadf8d03", size = 193205, upload-time = "2026-08-12T14:34:43.781Z" }, - { url = "https://files.pythonhosted.org/packages/5b/f3/7b523d807cb5e73562ef8acf21d39cdb9d704955327362c781bc3478a73d/charset_normalizer-3.5.0-cp37-abi3-macosx_10_9_universal2.whl", hash = "sha256:5a4ee37248dfac25107c758bda99d545ce73e60b44d2dd39e4a2bb9f2831e9f5", size = 330840, upload-time = "2026-08-12T14:34:45.06Z" }, - { url = "https://files.pythonhosted.org/packages/f0/de/fc68978fe78ca97063c96d764e41ff92ca639948f319271e0ff450e577a2/charset_normalizer-3.5.0-cp37-abi3-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:a864bdcacd8bff58bb4845304e031f821a3ec64b2b7259f2d409cd49c9e59ca3", size = 251862, upload-time = "2026-08-12T14:34:46.58Z" }, - { url = "https://files.pythonhosted.org/packages/a9/cb/82b41a0ab7fb1a88065f1d78ad32696ad88ea3fe8e25b8189d08833938de/charset_normalizer-3.5.0-cp37-abi3-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:84b736e3b391601bc47b86da381c749c0f894e9191aaca9f31f30c2632206df3", size = 239484, upload-time = "2026-08-12T14:34:47.869Z" }, - { url = "https://files.pythonhosted.org/packages/e8/0c/19608b631f4538f908098d4a2d56a8f79a665e27cc58e9d90479761a9227/charset_normalizer-3.5.0-cp37-abi3-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:6abb1f356fb865baeb6ebc3fadd843e9a96fbf49b9adcca55037f3cceccb7438", size = 230602, upload-time = "2026-08-12T14:34:49.265Z" }, - { url = "https://files.pythonhosted.org/packages/29/db/f648eb30e14eba301aed61e11672156f137905c1bdbb530151abe8065943/charset_normalizer-3.5.0-cp37-abi3-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:1d366548d2ee28a8cfdcc4296363978cc644a728333be9824d2de4652e83df0a", size = 259208, upload-time = "2026-08-12T14:34:50.632Z" }, - { url = "https://files.pythonhosted.org/packages/d3/e0/ed2c8bdbac484d69614d6993143aeb6cb0f4dd1561c883402517b623c8ef/charset_normalizer-3.5.0-cp37-abi3-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:d672f329ae504ee240eb39b6effb3318aa8e7e8924c0ce8eee5760b3fad98539", size = 253659, upload-time = "2026-08-12T14:34:52.11Z" }, - { url = "https://files.pythonhosted.org/packages/32/08/b4907cb9ec5b521d9d024ced13611240b86ef065c2eb15b3ad2334dc9940/charset_normalizer-3.5.0-cp37-abi3-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:c54036a518748b6c02e666f6d46c3817561998fb904c3be25b56fb4fe3dc5706", size = 248821, upload-time = "2026-08-12T14:34:53.399Z" }, - { url = "https://files.pythonhosted.org/packages/12/b2/e2d1abcfbc05822f0030869efb4e9f8a3658e13b4821796d4b62da917327/charset_normalizer-3.5.0-cp37-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:22a1889f1c9b752c63c36758a0c2145458e3cadb20fced7a0790002e9dd12b26", size = 240271, upload-time = "2026-08-12T14:34:55.09Z" }, - { url = "https://files.pythonhosted.org/packages/dc/f9/4ba127ad610542fa3eabfa41c45bf12d357860a815b3566374ec0188e213/charset_normalizer-3.5.0-cp37-abi3-musllinux_1_2_armv7l.whl", hash = "sha256:b7eb3eab5c646d3de7dcb14a7c9caebace5249c5767da39e1761cb1576e521a3", size = 232155, upload-time = "2026-08-12T14:34:56.543Z" }, - { url = "https://files.pythonhosted.org/packages/01/68/40613182366d00bd6dbd5f6c84a926cbd120960e038a8269e9ae7d782762/charset_normalizer-3.5.0-cp37-abi3-musllinux_1_2_ppc64le.whl", hash = "sha256:2080aa129a28267984cdc902898993d788c995c384e285d0d19199f56760d52e", size = 259674, upload-time = "2026-08-12T14:34:57.815Z" }, - { url = "https://files.pythonhosted.org/packages/0a/53/4574a14fa4c9de4a6c9f31725354bfa40b67f653e6d594ce1654f9a41b32/charset_normalizer-3.5.0-cp37-abi3-musllinux_1_2_riscv64.whl", hash = "sha256:fd68c825548a611158230e2f9222e210ceb2e3391995c0aa5865cbdf3ab4bd49", size = 246122, upload-time = "2026-08-12T14:34:59.337Z" }, - { url = "https://files.pythonhosted.org/packages/5a/02/bd8030d13d92c058ca7b2b9615bbb3169569e144db64d65c149cd45abf5e/charset_normalizer-3.5.0-cp37-abi3-musllinux_1_2_s390x.whl", hash = "sha256:d8a9316f4da85e937242642b537c6d55d7e9287dd38e5634732f8233932aff45", size = 255221, upload-time = "2026-08-12T14:35:00.71Z" }, - { url = "https://files.pythonhosted.org/packages/77/9d/10ecd3bcbe2666b3d4d4026c97b48f73990682815db516052a1e8f4a31c5/charset_normalizer-3.5.0-cp37-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:d90254c8f609338c53ec180fcd4c4f9c16502e238e3fc88ca7fd4c2f38d445b8", size = 253450, upload-time = "2026-08-12T14:35:02.217Z" }, - { url = "https://files.pythonhosted.org/packages/e4/0f/d044c4872c0938a84f87b5027a698c0e61bacfc5c3551a4e749ca9b7bc5c/charset_normalizer-3.5.0-cp37-abi3-win32.whl", hash = "sha256:8b3e9e29b8b07cc461b9ce7768db7693a93979d0dadf22046f6f3555ded2f516", size = 173594, upload-time = "2026-08-12T14:35:03.845Z" }, - { url = "https://files.pythonhosted.org/packages/10/6b/6046773901f1944b9a89436351529811ee958afc7b774563be9d74a6f0c3/charset_normalizer-3.5.0-cp37-abi3-win_amd64.whl", hash = "sha256:0c8953d9d1617794cfc40d81179571c9ba3805dd029623a15c93f1fb70e60a74", size = 198959, upload-time = "2026-08-12T14:35:05.187Z" }, - { url = "https://files.pythonhosted.org/packages/ab/a6/b57708ac92aefc8e8389d51d5178129b81f03196da61ee2c23e687b8178a/charset_normalizer-3.5.0-cp37-abi3-win_arm64.whl", hash = "sha256:562d24ca7797c1af8852994950c2e623a907b201fc4b0ed29e92af173d3828ca", size = 267055, upload-time = "2026-08-12T14:35:06.533Z" }, - { url = "https://files.pythonhosted.org/packages/22/c7/754d09943a616937df61e4ba367c409ded2a987e872972098d51a6fcf73b/charset_normalizer-3.5.0-py3-none-any.whl", hash = "sha256:993dfcbe75a85a3784abb5084f2c41b915767c90546fcc92803cffa28611baea", size = 67943, upload-time = "2026-08-12T14:35:30.363Z" }, + { url = "https://files.pythonhosted.org/packages/71/aa/554e2614f38fc34c58ff1d0911ae8535ad2516440d5482d76fe59f1088b0/charset_normalizer-3.5.1-cp310-cp310-macosx_10_9_universal2.whl", hash = "sha256:d1ee1e296209fdce05b81b663250eefa02213a2da7b41bf26f7829b8ba3545aa", size = 369072, upload-time = "2026-08-15T08:16:22.964Z" }, + { url = "https://files.pythonhosted.org/packages/03/6d/439231dfc3ccfa6f8c06477b7da2219cbd41a2de3d49084df8ec7b5100f2/charset_normalizer-3.5.1-cp310-cp310-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:e9fbdce1e47394b09bc9f26ab117dfc8d6491977a11d86f592bb42c779db2fda", size = 251142, upload-time = "2026-08-15T08:16:24.81Z" }, + { url = "https://files.pythonhosted.org/packages/55/53/7d819bd23a00ef45039146fa2cce1daa2f0771e758c5653ee1f6edac91ed/charset_normalizer-3.5.1-cp310-cp310-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:00668ebb0609751758682eb0b5857e7c35b9f00e84dfdef062e103244ec94d45", size = 240714, upload-time = "2026-08-15T08:16:26.392Z" }, + { url = "https://files.pythonhosted.org/packages/b2/2c/45847198c16f4b38090cc7423b2b6a9008e438704d8ab413211832498d31/charset_normalizer-3.5.1-cp310-cp310-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:ba2f37ee79e6338845261a3c5b1784e5d1acdff2c0785b284f1b633033d136ab", size = 279637, upload-time = "2026-08-15T08:16:27.961Z" }, + { url = "https://files.pythonhosted.org/packages/69/2b/d8be3523ddf9f0b0f3e56d1359034aa10653a4d11564c697f802b4775766/charset_normalizer-3.5.1-cp310-cp310-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:ce854f5f478050ade5a238731c4ca985a7d3b3cb53ff600a9b5c3b689b5f0a7a", size = 276543, upload-time = "2026-08-15T08:16:29.399Z" }, + { url = "https://files.pythonhosted.org/packages/32/cd/4f564b8f132de25db594efc706897069f016790cea63a5669c9df2675f64/charset_normalizer-3.5.1-cp310-cp310-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:96eefc178f8636b9c760c5829345307fd81cfae9ab1e80997dbddeb0f54ee9a3", size = 261644, upload-time = "2026-08-15T08:16:30.722Z" }, + { url = "https://files.pythonhosted.org/packages/f5/e3/38b975422534a608f98c360e79c2f07c763d66dd4272300d45fb1fee54b0/charset_normalizer-3.5.1-cp310-cp310-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:366ec70f5547c640d3ce1985722490f23faf4eb5216a7eeba78277490e78dacb", size = 259609, upload-time = "2026-08-15T08:16:32.248Z" }, + { url = "https://files.pythonhosted.org/packages/87/bd/fbc24d825c66f1c74f6ccdea3742c3d8354a4888e86d1315a197fee69061/charset_normalizer-3.5.1-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:950f23cb393f85543777b0433f082cddd25b51ab398eac7971146495679efe5f", size = 252457, upload-time = "2026-08-15T08:16:33.849Z" }, + { url = "https://files.pythonhosted.org/packages/b9/2d/918d0e98a0e679469ed05bb2d90c2088b4d315bb612969d8499f76fb5210/charset_normalizer-3.5.1-cp310-cp310-musllinux_1_2_armv7l.whl", hash = "sha256:c1dcc36dcb96abc02236e182d17e0f71430152a6c2c7447421da2d2dc144edea", size = 242240, upload-time = "2026-08-15T08:16:35.396Z" }, + { url = "https://files.pythonhosted.org/packages/20/c8/c36f6e0b2dfec351bd38cbc05362697e58bcd073d7dbd95154290c9714ce/charset_normalizer-3.5.1-cp310-cp310-musllinux_1_2_ppc64le.whl", hash = "sha256:07ffd07412fc5d5e84cd8952acf9ff7e4ed7a708e69d1bada19d8ba91711353f", size = 280308, upload-time = "2026-08-15T08:16:36.825Z" }, + { url = "https://files.pythonhosted.org/packages/ca/7b/311b3e02e8c4092400c449c850a760d8c45d900983c83a70cc07208c551d/charset_normalizer-3.5.1-cp310-cp310-musllinux_1_2_riscv64.whl", hash = "sha256:f5542f9b941279d82d41eb0aa9f98eba36fe4df5c7086c651df7944935b37182", size = 258679, upload-time = "2026-08-15T08:16:38.22Z" }, + { url = "https://files.pythonhosted.org/packages/b9/90/082cc45599c392f28c036a497f49e0634041a785fc3849c80ccf396d096f/charset_normalizer-3.5.1-cp310-cp310-musllinux_1_2_s390x.whl", hash = "sha256:a545775cfe815855ea32d7c27731d79da358ef2055b4a25830231b1622dd18aa", size = 277221, upload-time = "2026-08-15T08:16:39.62Z" }, + { url = "https://files.pythonhosted.org/packages/58/ad/b9aecf38d805cbcf84fa94f14c5d972a16561e20296a11dc799a5dcf3763/charset_normalizer-3.5.1-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:494b70049a4d69aec6e8137c13af4cf8db8c9f9820a1392ac293b0dd2987a818", size = 263799, upload-time = "2026-08-15T08:16:40.885Z" }, + { url = "https://files.pythonhosted.org/packages/b7/23/b38a20598d5a825f85d9d7636860e56ff0db1479f86497a6e485aa9326f7/charset_normalizer-3.5.1-cp310-cp310-win32.whl", hash = "sha256:94fbf1c0c6cc0d3d5e50f9a9313a8cdca90dd696d34b381cd1704f8c9e939f20", size = 182037, upload-time = "2026-08-15T08:16:42.198Z" }, + { url = "https://files.pythonhosted.org/packages/d2/21/83fffb77864408b8bf0fe1ca603926401d6f8775a8e150b39aacc9958f8a/charset_normalizer-3.5.1-cp310-cp310-win_amd64.whl", hash = "sha256:be47f99644b208bff7766314013f9acf57b056b04191d570d68ad14022cf5b1d", size = 206030, upload-time = "2026-08-15T08:16:43.787Z" }, + { url = "https://files.pythonhosted.org/packages/86/2e/b93135b5034b1157fb29554b0d06d4844ce62282f0e0a14036f93d7ee2e7/charset_normalizer-3.5.1-cp310-cp310-win_arm64.whl", hash = "sha256:a6d095662e73e74f0a49988e0593373e243e3a52e27bfeea0a859e88acf4a0f5", size = 185092, upload-time = "2026-08-15T08:16:45.177Z" }, + { url = "https://files.pythonhosted.org/packages/6a/b6/034f6802e9c3f6418966cfabb7db8c9252cc2429c5098f41cc43af804149/charset_normalizer-3.5.1-cp311-cp311-macosx_10_9_universal2.whl", hash = "sha256:eda059b6bc8bc0812d626fd91a7ce01bf583df0a61296eff390fd94141a34e30", size = 363585, upload-time = "2026-08-15T08:16:46.646Z" }, + { url = "https://files.pythonhosted.org/packages/d5/fa/6a7e2a7c4b5451912b8c417732df79574354443592a88d616de03da66ae5/charset_normalizer-3.5.1-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:aa2bb0b37202dca27175591f761108b5d34096ade1191ffe4808bdf6b1571488", size = 251189, upload-time = "2026-08-15T08:16:48.287Z" }, + { url = "https://files.pythonhosted.org/packages/a4/c8/ab42b07cfd82e919f427fcfaa7c41abae8242833ad1aad66d42bae40b669/charset_normalizer-3.5.1-cp311-cp311-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:0b2b1b3fa5670c127b246df1d0c059defd41f689a868a3b9d79df9b1cac42d22", size = 239724, upload-time = "2026-08-15T08:16:49.67Z" }, + { url = "https://files.pythonhosted.org/packages/e7/80/b9348b5d3041209f98b4cdad7655766369233f1d533f4f4f7558e9717bec/charset_normalizer-3.5.1-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:6e5e4d73d588ca5ed09df1b7dcd1b203d1df3c542e3f50d126c947d432b10731", size = 280078, upload-time = "2026-08-15T08:16:51.228Z" }, + { url = "https://files.pythonhosted.org/packages/82/38/083a24028304bc85bb9e376fed801178423dcbb67495f73b6ea0624e1894/charset_normalizer-3.5.1-cp311-cp311-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:b54e7e13267d49ffbfe68e25b3cbd774dab38fa37238f71265e91b36146eb21c", size = 276650, upload-time = "2026-08-15T08:16:52.625Z" }, + { url = "https://files.pythonhosted.org/packages/0d/35/731ac04aa0a097fc1c97f0994c375bdb230c6c96619db794208fe664e9ce/charset_normalizer-3.5.1-cp311-cp311-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:c7b742bf31c88566b4bb6335a7f393bb322e580b6bb98df7bd0c25e6e3519ce8", size = 262325, upload-time = "2026-08-15T08:16:54.085Z" }, + { url = "https://files.pythonhosted.org/packages/f5/28/c2028e7021fb89c6e56868ed0e387b8e9aa811abdd2ab3208d6578d2c930/charset_normalizer-3.5.1-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:6ba32c4d2abf1d2fe7cf27d280f4cca5664233b0f885549c7761719eb977f486", size = 261140, upload-time = "2026-08-15T08:16:55.604Z" }, + { url = "https://files.pythonhosted.org/packages/28/f0/0c0ceec6d98b7daa62e361e418135d59685811d79ba11529aad5cdf15e84/charset_normalizer-3.5.1-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:0722590aabf9dc6a6c0343d523c05458fa2b5047dbe6302fd526bb570600753f", size = 252791, upload-time = "2026-08-15T08:16:57.103Z" }, + { url = "https://files.pythonhosted.org/packages/f0/3e/48f4cd187b1c33189d86039e9cbe4f92c05454175504b44ff81806d4d1bf/charset_normalizer-3.5.1-cp311-cp311-musllinux_1_2_armv7l.whl", hash = "sha256:aa1099b956fb795e686d073568f6dc002a0bb89765ea6d5b055dd7d9bf1b116c", size = 240730, upload-time = "2026-08-15T08:16:58.418Z" }, + { url = "https://files.pythonhosted.org/packages/42/85/f9e22af69af67c54cce42be9455d9c81294f918b4ccc454db01f66efcac2/charset_normalizer-3.5.1-cp311-cp311-musllinux_1_2_ppc64le.whl", hash = "sha256:bd6c173f04743d483881bffa1478d5a4624475b8cd1d2194956a75548e191c18", size = 280791, upload-time = "2026-08-15T08:16:59.918Z" }, + { url = "https://files.pythonhosted.org/packages/fd/4c/9044135f42127630b6fa742feb51256353f6ab87a78f2fdd1de3de955a7f/charset_normalizer-3.5.1-cp311-cp311-musllinux_1_2_riscv64.whl", hash = "sha256:f298e218441525d3794428b4c8b8fb8662c6d3ea79925d4807ee6b9a96a3bca5", size = 259598, upload-time = "2026-08-15T08:17:01.421Z" }, + { url = "https://files.pythonhosted.org/packages/ba/ed/1dd7cfebb4e75812934c49ca3b79757d11948053f7937ab7070c151f3c55/charset_normalizer-3.5.1-cp311-cp311-musllinux_1_2_s390x.whl", hash = "sha256:6e2912d4babbc65196ac13c2f53468dc57fb8b9c25ef913e8c59ddf7c6dc0e1b", size = 278217, upload-time = "2026-08-15T08:17:02.782Z" }, + { url = "https://files.pythonhosted.org/packages/bf/eb/239c84503cc9e3ba6eb34686a24bc66e84f3924efdd7e38e751a19f6bc10/charset_normalizer-3.5.1-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:3d27167433c0d5f18dc850f07d0b3816221984fecdc405d6c157a6f0b8f8e9e6", size = 263417, upload-time = "2026-08-15T08:17:04.216Z" }, + { url = "https://files.pythonhosted.org/packages/37/ab/4e4510e1e288478e2c8333131d1c1382382ba8cd2165053c79e39d1da961/charset_normalizer-3.5.1-cp311-cp311-win32.whl", hash = "sha256:ac00177c4831ffa650f8609e4bdddd5fe09c03b1c0c47acece7e6ea20421598b", size = 181774, upload-time = "2026-08-15T08:17:05.58Z" }, + { url = "https://files.pythonhosted.org/packages/e3/57/32f0ccea59e8612057c61d6fd22ef2cb63cca93c9fe594094919696ac170/charset_normalizer-3.5.1-cp311-cp311-win_amd64.whl", hash = "sha256:f9b1e28d0e8dbfa858abdba91d6b547beaf2df1a59bec6da6faae7b96a4991a9", size = 206653, upload-time = "2026-08-15T08:17:07.075Z" }, + { url = "https://files.pythonhosted.org/packages/17/d4/b65c433fc521e58b5f54293982a5e51c05cb5f2dd3f1c7a6acb65b75324e/charset_normalizer-3.5.1-cp311-cp311-win_arm64.whl", hash = "sha256:ae31a1a1db2ee6cc2942fccaf695c934bc7f3db9f2133a3fef1f367cf1a4ab10", size = 185630, upload-time = "2026-08-15T08:17:08.502Z" }, + { url = "https://files.pythonhosted.org/packages/30/27/78873dc8b6a56357517b74b6bb9568b80450e7bb4f6ef7e3fa9d22aa0bd7/charset_normalizer-3.5.1-cp312-cp312-macosx_10_13_universal2.whl", hash = "sha256:5b6d1386bf0096d26d3a863dc0a487a5b4eb9aa93cf5ba69683d29dde6b9d60f", size = 344456, upload-time = "2026-08-15T08:17:10.072Z" }, + { url = "https://files.pythonhosted.org/packages/9a/4c/be49ada26b1f0232d57aa89bbebf997a5cc2332a5616b6eca26ff680044d/charset_normalizer-3.5.1-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:4582c27e8c889d64811987b5967fbd3ae0c823fe1fd933b543d55ac20bb475fa", size = 238530, upload-time = "2026-08-15T08:17:11.563Z" }, + { url = "https://files.pythonhosted.org/packages/76/84/6f1290fa07ae6978d3960caa3eb1b8019bf9284ab7c2297b00c099ef4250/charset_normalizer-3.5.1-cp312-cp312-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:1d1c7a53a6c2103925cdd6d7229f8c567379f211c869793df679f2e9f738c369", size = 230200, upload-time = "2026-08-15T08:17:12.919Z" }, + { url = "https://files.pythonhosted.org/packages/e7/a0/47b18adeed31c8f16ba9700f32c1b18594cfa09f47eb672a488c273c22bf/charset_normalizer-3.5.1-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:e6621fb2a4988d6e53eedc455e5903e2679f3967b8acb3d639f1b63c14a2e893", size = 262222, upload-time = "2026-08-15T08:17:14.571Z" }, + { url = "https://files.pythonhosted.org/packages/38/fe/341861ac118dae06f3ec0eb487488af52128f2ef2faf0b11003944d22259/charset_normalizer-3.5.1-cp312-cp312-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:7c0c10730342b0c9b35dd1d619beb8214e520bd96a1f870f452680b238aab3e0", size = 258951, upload-time = "2026-08-15T08:17:16.158Z" }, + { url = "https://files.pythonhosted.org/packages/6f/89/bb5108dc6c3651dca963f2b0a3ba19bbcb370c94e1b6d3e0e844a58e6dca/charset_normalizer-3.5.1-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:b9af956078716df40d985fb0dfeb2c2120c5ca92ba4ff4b388acfd01cdc14d08", size = 248801, upload-time = "2026-08-15T08:17:17.683Z" }, + { url = "https://files.pythonhosted.org/packages/b1/ba/ef83ae3aca816393decfa3530976f38a79812d707b80b580ac33b83f9877/charset_normalizer-3.5.1-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:f9f8405c2c758532c74fed975dbee57be1f31a6e865c031870c79a6ed3212ada", size = 244070, upload-time = "2026-08-15T08:17:19.191Z" }, + { url = "https://files.pythonhosted.org/packages/f6/0b/c5292a2462d69b7378ea89793bbb5b2b6fcf6f7dd6d1667f9619094ad553/charset_normalizer-3.5.1-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:96fef3e886d6a9874b14f27fc193fbdc69d5d8035783d86aa4e1cea594e695f9", size = 240110, upload-time = "2026-08-15T08:17:20.547Z" }, + { url = "https://files.pythonhosted.org/packages/46/22/111e5be3b740d5c2a5bfcedb3d237b6591e5c2e82ae9d6ffcb121fe0909c/charset_normalizer-3.5.1-cp312-cp312-musllinux_1_2_armv7l.whl", hash = "sha256:5d8531a6569d025f68e2321e7638fb7978f23db58e5f69f56913837aae03816e", size = 232836, upload-time = "2026-08-15T08:17:21.895Z" }, + { url = "https://files.pythonhosted.org/packages/f9/d2/d2aad6fe0dbb44b194bf3becb60f5a0ac48446ade999a47fe7bb41eb09a7/charset_normalizer-3.5.1-cp312-cp312-musllinux_1_2_ppc64le.whl", hash = "sha256:aae2ee51122d3ae968a3837d97dc24a0aeebb0dea23694422cd172bd30017cd6", size = 262712, upload-time = "2026-08-15T08:17:23.727Z" }, + { url = "https://files.pythonhosted.org/packages/35/5a/337e4663a5eae6de99db940ee8066d4145caafb61327db62deda15313cce/charset_normalizer-3.5.1-cp312-cp312-musllinux_1_2_riscv64.whl", hash = "sha256:7235dc28fc6dd9d832ac7c7bce95367dedb85929f17368a0c2bee1e080b9acbf", size = 242977, upload-time = "2026-08-15T08:17:25.157Z" }, + { url = "https://files.pythonhosted.org/packages/ca/85/f82f8a92e31c7519410e2e1afdc630f28ec47490ce2c09a11c1a43cbb459/charset_normalizer-3.5.1-cp312-cp312-musllinux_1_2_s390x.whl", hash = "sha256:4abdc5f9ad448c1ecbfae2974b820535d6bc6e7eef63babbab3d81cf46968c71", size = 260207, upload-time = "2026-08-15T08:17:26.602Z" }, + { url = "https://files.pythonhosted.org/packages/b7/52/643d11ffd60e9ac2fd1fb87e167a19285b9eefeff4a40e63c87cbfbeab36/charset_normalizer-3.5.1-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:ba501e667c17d8411f98e67a022d9604ef179aff0e459b7e292c796837c13573", size = 250562, upload-time = "2026-08-15T08:17:27.971Z" }, + { url = "https://files.pythonhosted.org/packages/62/16/46556278c2168d12df9da7fede5dc6fc70e60301b26a82bbeec238c9cfe3/charset_normalizer-3.5.1-cp312-cp312-win32.whl", hash = "sha256:cfa1c0cc3a8f9f53f1243a5a99ac36fd003880199383b37672e86ddda9cb07e2", size = 178507, upload-time = "2026-08-15T08:17:29.277Z" }, + { url = "https://files.pythonhosted.org/packages/9d/7a/4c6c298171e6b3e745633180ff59350fc0ca0db1ffd28df1e369e0579f71/charset_normalizer-3.5.1-cp312-cp312-win_amd64.whl", hash = "sha256:3617ac3cfd8b9888f145ad89dd6e692285834b0201c6074a5eeaad3fd4d668c2", size = 200551, upload-time = "2026-08-15T08:17:30.668Z" }, + { url = "https://files.pythonhosted.org/packages/cd/d7/eb95a042f0dd22e304b0b6472b154f3546a1a039a9ee89ccb2a7f61591fc/charset_normalizer-3.5.1-cp312-cp312-win_arm64.whl", hash = "sha256:88e85ab89cb822c1e635f51d6d32e488f94e002e70e2f492bdb8b945543f345a", size = 180700, upload-time = "2026-08-15T08:17:32.028Z" }, + { url = "https://files.pythonhosted.org/packages/bc/61/2cb6ad133dbbb449fa2d37ccae973232f4827e799af258d15e589a3d1e9e/charset_normalizer-3.5.1-cp313-cp313-android_24_arm64_v8a.whl", hash = "sha256:4f298bdadb8f0b9e5672877f647d1be9373ef5320c9e2f049795e26cad28b6a9", size = 211584, upload-time = "2026-08-15T08:17:33.597Z" }, + { url = "https://files.pythonhosted.org/packages/18/57/a305c968be1ca13f3dd1b32f445877e97addf55d80b65c7cb35fac82b777/charset_normalizer-3.5.1-cp313-cp313-android_24_x86_64.whl", hash = "sha256:88ca277405c2d3b71c4e1c2ee0e7966e807bcba86a69d11e19ba199d18ae4491", size = 223359, upload-time = "2026-08-15T08:17:35.022Z" }, + { url = "https://files.pythonhosted.org/packages/09/0a/d3646670292ce8d8f8cc11ac067d44885e697a5591f57a9221128da5e7b3/charset_normalizer-3.5.1-cp313-cp313-ios_13_0_arm64_iphoneos.whl", hash = "sha256:9362dd90aa7dab48c0054a21187791ccf05473f7dba5d92b8033ae62164675e7", size = 194464, upload-time = "2026-08-15T08:17:36.452Z" }, + { url = "https://files.pythonhosted.org/packages/de/93/d51ec556e01042fed6f993ea859311bc7917b466684182fbbceb6ca24762/charset_normalizer-3.5.1-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:977cdbd483a9cff38179bea4fd754289a6f2195c7abd414aba85410b3e66cc5e", size = 197676, upload-time = "2026-08-15T08:17:37.819Z" }, + { url = "https://files.pythonhosted.org/packages/a4/a0/562247944386f7d4ef94467e84876600cc1e0f1b93239aaa9213d2bc3cbd/charset_normalizer-3.5.1-cp313-cp313-macosx_10_13_universal2.whl", hash = "sha256:e90251c0c7bdd54a100a0dce3c07b7e637278c93af29dbf78ebb89a58c4bac7d", size = 340473, upload-time = "2026-08-15T08:17:39.303Z" }, + { url = "https://files.pythonhosted.org/packages/31/e7/1d994be1b93d41e9502b8b0460eaa88a1dd8df335df415db87d6c3e91ab2/charset_normalizer-3.5.1-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:94d78ecec2605a8d0398b0f365d5f12a63248438516f5dac536a5eff7337df4a", size = 240156, upload-time = "2026-08-15T08:17:40.66Z" }, + { url = "https://files.pythonhosted.org/packages/09/53/27923ce5cc6cbccb832037b27dca98882d9c53e9b69e866bbbef4aae7fc8/charset_normalizer-3.5.1-cp313-cp313-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:d59b75732e9b6f27388e10c14b0259cc5f2e48c78627d185e6a177b58ad3cffe", size = 228246, upload-time = "2026-08-15T08:17:42.003Z" }, + { url = "https://files.pythonhosted.org/packages/ce/48/5a97e84d63af1d55c07439cb80e56d99a8efb4295700eb4e18c0d1615d2c/charset_normalizer-3.5.1-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:0d929fc574b4d6fd9e7c0f5c2ede8716a41911923aa7fa5fce38e0818aa4a1ac", size = 263660, upload-time = "2026-08-15T08:17:43.627Z" }, + { url = "https://files.pythonhosted.org/packages/7a/c2/071575791dcc88316c0a9a65ce38897a82e4cfe4a325f0f7fe1b1ac47bcf/charset_normalizer-3.5.1-cp313-cp313-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:394fea06235c8543390050ed5f529187074b029fb027213f6c46ac11ab5d950e", size = 260354, upload-time = "2026-08-15T08:17:45.094Z" }, + { url = "https://files.pythonhosted.org/packages/fb/af/63240b0c0248c075c2535a1f1bd992821d8251b9f173abc13329661d09e4/charset_normalizer-3.5.1-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:62b55f6722735a6c472f88361cde6640608773d9443cebdbb51abf436a1fcdd3", size = 250638, upload-time = "2026-08-15T08:17:46.496Z" }, + { url = "https://files.pythonhosted.org/packages/4d/66/70dfad64f15be09c15ccfee81330a7e515895dbe296dd23114e9a231268a/charset_normalizer-3.5.1-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:fa48b1b63d639f9483e0633e092f5851e2348c352f1f9bb6c8182f87884ef876", size = 244583, upload-time = "2026-08-15T08:17:47.963Z" }, + { url = "https://files.pythonhosted.org/packages/c0/24/ef36367d38b9ddd4bccbf72888c342e8de1f5ae506fa0b2dcf970e2732a1/charset_normalizer-3.5.1-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:c71fb0d56c920c269cd3e2e3fe7c610e3f1fdb21a6ce60efa6430ff63676cea6", size = 242038, upload-time = "2026-08-15T08:17:49.481Z" }, + { url = "https://files.pythonhosted.org/packages/db/ab/55e683ba0fff2e43adafc10daa3001eac90fdaa419a97227d5a7067eedde/charset_normalizer-3.5.1-cp313-cp313-musllinux_1_2_armv7l.whl", hash = "sha256:485a0d363cafefcd2538a73c7c838daa2035f09b2c9f9b5e3133f80c6aeb84c2", size = 233677, upload-time = "2026-08-15T08:17:50.845Z" }, + { url = "https://files.pythonhosted.org/packages/bd/67/0f40eaf8d1b6e7cf15e82382a2965efaca787fc1c2794b7021d37aaf5036/charset_normalizer-3.5.1-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:5c0ea61a470e070686aa30892fed79e297d2c8d0ab46b8bcdf027d38c51da591", size = 264491, upload-time = "2026-08-15T08:17:52.61Z" }, + { url = "https://files.pythonhosted.org/packages/5c/64/12b4c2a11ee8df4fcc518c78b0d93e3a92bd3d5253d1617ce74ff0e8c7ef/charset_normalizer-3.5.1-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:90b7481fb62fbe172c558bc6fd1c4c98d82004a54a7551f20e11ac9bf0b8708c", size = 245196, upload-time = "2026-08-15T08:17:54.023Z" }, + { url = "https://files.pythonhosted.org/packages/37/2e/651d910af6d0fba325eee1cda37ec5443462ed25360e666c144166eb6091/charset_normalizer-3.5.1-cp313-cp313-musllinux_1_2_s390x.whl", hash = "sha256:35fe081843b35aad20ffeccec3eeffbe637b15d14f3fb22cc1b59cd8ec17e93c", size = 261660, upload-time = "2026-08-15T08:17:55.491Z" }, + { url = "https://files.pythonhosted.org/packages/90/c6/b09e05e6db7f64338e0dc067c79577b1138da86c1e38369096851d96be88/charset_normalizer-3.5.1-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:fd0350afdc3aabd5576f60ea109228bd5538139713c7b094c5cd27c73a98bc6f", size = 252618, upload-time = "2026-08-15T08:17:57.025Z" }, + { url = "https://files.pythonhosted.org/packages/76/4e/362d4f9fdcdf5556fb2aa3ce7d4a58ebce03ed1ff03aa1d9aca8d02f13f3/charset_normalizer-3.5.1-cp313-cp313-pyemscripten_2025_0_wasm32.whl", hash = "sha256:9d9a0dc7cbe9bec24c3f767c9122c41fe5a1bc43f47cd099d00d393e09769de4", size = 140362, upload-time = "2026-08-15T08:17:58.425Z" }, + { url = "https://files.pythonhosted.org/packages/b4/d4/703be739b26acce318bd29eb3b25b7209e1b1f527f9eae3d1f1f01fdde2b/charset_normalizer-3.5.1-cp313-cp313-win32.whl", hash = "sha256:d63600d620ad0064c3a748b950ac5ea38a80190e5498532efefa4b7b3f1da1f3", size = 177755, upload-time = "2026-08-15T08:18:00.037Z" }, + { url = "https://files.pythonhosted.org/packages/8a/33/56d97ade41c8db611e727168c52ae46c9224c362ec28d4b65d7e9869e8da/charset_normalizer-3.5.1-cp313-cp313-win_amd64.whl", hash = "sha256:aea996a6aba25260827c9ea511d1addfde2da9eb686ac961838509086188b7e6", size = 199295, upload-time = "2026-08-15T08:18:01.506Z" }, + { url = "https://files.pythonhosted.org/packages/5b/75/5b20dd1e6573a01a08158fe104104fa2c8abf941745596954185726cd46c/charset_normalizer-3.5.1-cp313-cp313-win_arm64.whl", hash = "sha256:fd0a274c0e5f9a21565cd9d3dd749b61f96b7aa1e20a93aa1ba4029518f2e5c0", size = 179856, upload-time = "2026-08-15T08:18:02.929Z" }, + { url = "https://files.pythonhosted.org/packages/29/cd/2b812ce5e888f1ce69a5350281e58aab07ae64a958ecae8912f30865718e/charset_normalizer-3.5.1-cp314-cp314-android_24_arm64_v8a.whl", hash = "sha256:774d157f112367ff4abd29019f38f023c24e00e56edc7829c20e358a5a913ad8", size = 212318, upload-time = "2026-08-15T08:18:04.403Z" }, + { url = "https://files.pythonhosted.org/packages/9e/4a/a6ee107430768a5334e6d63f31f148a04a1a491ef161a1ac9415a73f2fa8/charset_normalizer-3.5.1-cp314-cp314-android_24_x86_64.whl", hash = "sha256:26422d45fd13551cf564c58932f7d72b4f58b93b0fcf18c35ba6be12b46bb102", size = 224897, upload-time = "2026-08-15T08:18:05.997Z" }, + { url = "https://files.pythonhosted.org/packages/c3/d9/35ae3f64f29d0179c35c3baefe575904df2913dde519129c7f75995a2b1d/charset_normalizer-3.5.1-cp314-cp314-ios_13_0_arm64_iphoneos.whl", hash = "sha256:09a7bba9f739468c8e78c36a75c33768e53cb1959fc638f510454c14683f00d5", size = 194848, upload-time = "2026-08-15T08:18:07.397Z" }, + { url = "https://files.pythonhosted.org/packages/74/76/f2fc7380f056cc273a53af37f50d08ad54b2c59f61078f31432edcf1c2bd/charset_normalizer-3.5.1-cp314-cp314-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:4c9548dc78002099910abaebc0a72ac58b7d30931869e0351c09b507dff4ece3", size = 198163, upload-time = "2026-08-15T08:18:08.989Z" }, + { url = "https://files.pythonhosted.org/packages/e9/40/095ce62fa078483cccc1fa2b36e6bc9580b85422a20ee9f925341c50e44f/charset_normalizer-3.5.1-cp314-cp314-macosx_10_15_universal2.whl", hash = "sha256:c428c6c31eb5f4277d7f8eccaf767fbd548ddd5ce3c8b4f4cbbfab3d96b5904c", size = 341823, upload-time = "2026-08-15T08:18:10.458Z" }, + { url = "https://files.pythonhosted.org/packages/f1/5a/0e58b1c04a1596e0256f407274a92d5fb2ee21324409d1fab1da48a65b5b/charset_normalizer-3.5.1-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:2f06b7eae9dbe77fe1d644ca244dad508de8d302870a43f3c559b521270938a0", size = 242458, upload-time = "2026-08-15T08:18:11.989Z" }, + { url = "https://files.pythonhosted.org/packages/22/95/b4618ce912e6db0b1aae89ba788e38e8a7eba0f3025cc66e8c0699f977b2/charset_normalizer-3.5.1-cp314-cp314-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:6b7430cf5728e68f6c462254009a6ef4086e1bea43cf2f57aa9c55fb4f50ff96", size = 226717, upload-time = "2026-08-15T08:18:13.401Z" }, + { url = "https://files.pythonhosted.org/packages/8a/76/c681192bbda3d55356db5dadd64381d5202b37c6b598fcda5282e88b5d3d/charset_normalizer-3.5.1-cp314-cp314-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:ab743e9bc90c1f73552ec33e10e3331315acd2c397b36065b591b0181de533cc", size = 266111, upload-time = "2026-08-15T08:18:14.961Z" }, + { url = "https://files.pythonhosted.org/packages/88/be/55127bfca72c0cff6c022488d140d7c5b04c771e3b72e9bdb4836d54979d/charset_normalizer-3.5.1-cp314-cp314-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:f6f7deae3feb4edfa2efaf7c574fe88cbf055038a6abdb40188e4fff66d5699f", size = 263128, upload-time = "2026-08-15T08:18:16.515Z" }, + { url = "https://files.pythonhosted.org/packages/e0/91/39c3af510b0aa32bbda03374259200f28430febfd1bf5e511fe765282ce5/charset_normalizer-3.5.1-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:15f024313246a4ed976c60f440bb8d257815513a681d212ff74fd46f7d715a90", size = 251240, upload-time = "2026-08-15T08:18:18.127Z" }, + { url = "https://files.pythonhosted.org/packages/1c/a5/cbe418bbc6ecdfc3e05a0116002897c4b403a5e838d697e64c78e9f0190d/charset_normalizer-3.5.1-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:823f82903d189af463d7df250ef1f7f696f3cee08cc8d91deb565e8d425f6506", size = 245282, upload-time = "2026-08-15T08:18:19.625Z" }, + { url = "https://files.pythonhosted.org/packages/cc/a4/689bb42e8e7cd492f3cb64907c6bc00ad247ec9a3628cd3f8eed126e8ae1/charset_normalizer-3.5.1-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:01e93745f7f219b703b60ba7afead36cfc4242782be5af484673fc500df12da5", size = 244597, upload-time = "2026-08-15T08:18:21.121Z" }, + { url = "https://files.pythonhosted.org/packages/c1/ce/9962938e179cf9f699d3f1e7b3114b5d7642dee6a893745229f9dd04f274/charset_normalizer-3.5.1-cp314-cp314-musllinux_1_2_armv7l.whl", hash = "sha256:329fc3ccb63ad22d867d84c2adea759a64079a37ba4a343433b02c7a2816871e", size = 231376, upload-time = "2026-08-15T08:18:22.57Z" }, + { url = "https://files.pythonhosted.org/packages/85/54/46000450ada53bd9eac5429a2c8c54cd2d9b39c0c255f229aea9af0948a5/charset_normalizer-3.5.1-cp314-cp314-musllinux_1_2_ppc64le.whl", hash = "sha256:bb57753e36e4855b8ca375069482250a6246372331a3e4f3407eaebb007443f5", size = 266715, upload-time = "2026-08-15T08:18:24.235Z" }, + { url = "https://files.pythonhosted.org/packages/3d/bb/618749d70f792b44252a777bf89bfb86823b9bbc1ea13fe8ce759b07f38a/charset_normalizer-3.5.1-cp314-cp314-musllinux_1_2_riscv64.whl", hash = "sha256:fce8cbd4997efeb450bd298b54f755dcdff18d496f7a5ddbb4867c6d7c88fdc3", size = 245848, upload-time = "2026-08-15T08:18:25.726Z" }, + { url = "https://files.pythonhosted.org/packages/7e/3f/ffb64458527c7668031d5eb095d978de561958dc9f5b53f8e488a533e603/charset_normalizer-3.5.1-cp314-cp314-musllinux_1_2_s390x.whl", hash = "sha256:6c9cdde8becb25a7fde49924511aa2644d6f8081cc8df8e9452724303348d8e3", size = 264521, upload-time = "2026-08-15T08:18:27.193Z" }, + { url = "https://files.pythonhosted.org/packages/4f/ab/74a55fd803916a35ac461daf002708191aac19b546b80dc8cabfedc63d98/charset_normalizer-3.5.1-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:9ac4444d8d4fd4c4bd08bf451ed3167aa9e7ec6cdb41b648794f1d1103652e36", size = 253054, upload-time = "2026-08-15T08:18:28.568Z" }, + { url = "https://files.pythonhosted.org/packages/a0/2a/6a9034b7d3c60b17499afb482df5878bf9fa20b50cc3887d5ef017a833db/charset_normalizer-3.5.1-cp314-cp314-pyemscripten_2026_0_wasm32.whl", hash = "sha256:f03ac127268b43ef4fe9e6ab6794a6794b49485a0cc0c1db79876d2f33f75bc7", size = 140580, upload-time = "2026-08-15T08:18:30.214Z" }, + { url = "https://files.pythonhosted.org/packages/f3/46/1d362e1a00d035d66b9869e1281eee115907f7e390a16a07824ab5737360/charset_normalizer-3.5.1-cp314-cp314-win32.whl", hash = "sha256:1f5883d77fd409a261abb5dc8ccbe335720d798b1de4abb3b1d47ccbbc76b53b", size = 180325, upload-time = "2026-08-15T08:18:31.877Z" }, + { url = "https://files.pythonhosted.org/packages/7a/7c/4938c329b6a9d446f6a59aa2092ff7118f274209b5ed0e26893d1d30a63c/charset_normalizer-3.5.1-cp314-cp314-win_amd64.whl", hash = "sha256:c658c50ac0c98cd755a2dd50b7977d3bca7df401dcc47fbdfa87db53ef7d4e8b", size = 204175, upload-time = "2026-08-15T08:18:33.466Z" }, + { url = "https://files.pythonhosted.org/packages/ac/33/eeb384dbd8dec570661354592f4f2e1b2fcc92585624d146a000caf53841/charset_normalizer-3.5.1-cp314-cp314-win_arm64.whl", hash = "sha256:4bea7f8ebe90bbd7f0e4a2de42ca6924ba23e3e76418c408ff82f1d46fabd687", size = 184123, upload-time = "2026-08-15T08:18:34.913Z" }, + { url = "https://files.pythonhosted.org/packages/1c/6c/c73fa9d5a85f6ab05395de61c5f6984e0a9ff40bb5ff888d46dff02526c6/charset_normalizer-3.5.1-cp314-cp314t-macosx_10_15_universal2.whl", hash = "sha256:fbc597639158fd7c14d55e808718848319540f51b0e6746e3eefa59723a4a348", size = 381682, upload-time = "2026-08-15T08:18:36.349Z" }, + { url = "https://files.pythonhosted.org/packages/30/c7/63565f860921457feba93bae6c86fb7746deb4cffeed2f375cb845318146/charset_normalizer-3.5.1-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:e71c909f353863b2b89c83de2ebed71ea6d0df8a6ef65a128193c5e650766bef", size = 240826, upload-time = "2026-08-15T08:18:37.887Z" }, + { url = "https://files.pythonhosted.org/packages/06/ae/7ae8807410dfa33f8e6f1715740adeaafa8a816cc4cb33508f54b1f7c896/charset_normalizer-3.5.1-cp314-cp314t-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:7ac76cf9afd34929d76eb7fcb63be476a4853d8a96f0dcf2d0db68a0cbdf9885", size = 227861, upload-time = "2026-08-15T08:18:39.315Z" }, + { url = "https://files.pythonhosted.org/packages/e9/a3/887c1642f0da26000b0e0652d91071113c0e72cea33952e225cf589f49a9/charset_normalizer-3.5.1-cp314-cp314t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:a3a370082ce34d0612f421e15fe011c53bb1feff21a26d06ad4fb244dab5a375", size = 260758, upload-time = "2026-08-15T08:18:40.88Z" }, + { url = "https://files.pythonhosted.org/packages/3e/11/e6f5b9a3d0e55b0ef7505cd3765cdd48f22db89994c947b316f52f801fd8/charset_normalizer-3.5.1-cp314-cp314t-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:256dd4d85d9e4dc595e2bc983c980e73f62ddeb3165c58b4c3dfe78c5c8548c1", size = 259950, upload-time = "2026-08-15T08:18:42.351Z" }, + { url = "https://files.pythonhosted.org/packages/1b/ee/e4e10a94d51cd1ee638aa7e00b65399e6b2a4e8376ab6d2eac9f95586671/charset_normalizer-3.5.1-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:58d4aa13a59c969dbfdf9e6a9560e242cbfd9e8a8f50c2747714df1a423adf65", size = 249329, upload-time = "2026-08-15T08:18:43.914Z" }, + { url = "https://files.pythonhosted.org/packages/c4/25/d5f4198819e6059735a84e8d0bfb72dc33976da67b97adcd3fb5a5e07ec6/charset_normalizer-3.5.1-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:0c6dfb5ca6723eeed15aa8e564a014d69fcb8812f94eef11fe3631e0508199f5", size = 243137, upload-time = "2026-08-15T08:18:45.368Z" }, + { url = "https://files.pythonhosted.org/packages/a5/e9/e925ca7569cf9fb9701fd82503fee73eea5268fdb856bdd64947092d3daa/charset_normalizer-3.5.1-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:c010f5581d9c612804cc59fcf7b524b707fbcb72828551237ab545bb5c7034af", size = 242820, upload-time = "2026-08-15T08:18:46.842Z" }, + { url = "https://files.pythonhosted.org/packages/34/17/672c251a888ed2aebcdd2fe830ad0104e25ff83c43f5c4f9c15e9fc6853c/charset_normalizer-3.5.1-cp314-cp314t-musllinux_1_2_armv7l.whl", hash = "sha256:52ec005752a56ae79547a05c0139ca2501a0c866390b6115008456b9f0e7cde1", size = 230504, upload-time = "2026-08-15T08:18:48.353Z" }, + { url = "https://files.pythonhosted.org/packages/3f/fc/f6a85abebd42ce4da2f1db0aa56cc6a0df1995e318b3875d14401b8381d1/charset_normalizer-3.5.1-cp314-cp314t-musllinux_1_2_ppc64le.whl", hash = "sha256:2bced4061f000f7187254a02ad3433ae17eaf991747ceea2f478422590a5bba9", size = 263087, upload-time = "2026-08-15T08:18:49.859Z" }, + { url = "https://files.pythonhosted.org/packages/98/66/7c42677e739ba66746b297e2046918d793078094dc239e1e72768cffccc6/charset_normalizer-3.5.1-cp314-cp314t-musllinux_1_2_riscv64.whl", hash = "sha256:9eea3ab2597a5e65fe65296e2d6a84570845a6b55532d90333d740d48bbc850a", size = 243269, upload-time = "2026-08-15T08:18:51.601Z" }, + { url = "https://files.pythonhosted.org/packages/de/d8/a50b79237f417af10f8c2a501ce8d1ca87829a22e69117891ca4ba20a69e/charset_normalizer-3.5.1-cp314-cp314t-musllinux_1_2_s390x.whl", hash = "sha256:496846868fea80e479324862fa877f02411f2fd0f83b79ccee2607aa68b2a032", size = 258766, upload-time = "2026-08-15T08:18:53.23Z" }, + { url = "https://files.pythonhosted.org/packages/2e/1d/0fc91aeaeb3c83b748f532399ce67cf84604b48297405d740000f7a9e786/charset_normalizer-3.5.1-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:85d5855daafc240cc045c026d7a15fd198a09b0fc8ff6f5ecbb5297b509cb11e", size = 250814, upload-time = "2026-08-15T08:18:54.768Z" }, + { url = "https://files.pythonhosted.org/packages/ae/10/3d8c777cf9024615295aa1b808324ad5b4a77855869c00824bad74ffaf8a/charset_normalizer-3.5.1-cp314-cp314t-win32.whl", hash = "sha256:58d3e12c88e0950bca850ae1f7c256055c097639c2edb9eb123af9807d8b15e4", size = 191074, upload-time = "2026-08-15T08:18:56.305Z" }, + { url = "https://files.pythonhosted.org/packages/4d/81/ae557d3c44d1a1d688696d60563413a0866a91b7ebc50f20df838be3d8c8/charset_normalizer-3.5.1-cp314-cp314t-win_amd64.whl", hash = "sha256:acaf604462bf330b0d07e7a07c1d6e4adac79e5fb13e9c5140590542cafacc00", size = 216476, upload-time = "2026-08-15T08:18:57.889Z" }, + { url = "https://files.pythonhosted.org/packages/27/e9/61c01fb8b804692569c036b3fc50495814502dcf13a60649c6055390b02c/charset_normalizer-3.5.1-cp314-cp314t-win_arm64.whl", hash = "sha256:fdb8a068947befafba9952162645dc2fecaeb400e64584829ed5e9b2fbe21a7f", size = 194115, upload-time = "2026-08-15T08:18:59.418Z" }, + { url = "https://files.pythonhosted.org/packages/4a/4e/8544831ef59d8f27ce92c80871380fdacc8076a8a56ed62f82e54f991333/charset_normalizer-3.5.1-cp315-cp315-macosx_10_15_universal2.whl", hash = "sha256:9085f87b0e38a2b92b8923059b4e8789fe40d9279712d15dcc670048d77079af", size = 342048, upload-time = "2026-08-15T08:19:01.054Z" }, + { url = "https://files.pythonhosted.org/packages/7f/a6/e3b46852424246065355644f4fb6dbccc0239a42a2eee27ecfc8957f0bcd/charset_normalizer-3.5.1-cp315-cp315-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:2679de311c7946dde5d3b6f44941844133ff5c7cb86099c0061ab1e8901c20a8", size = 242997, upload-time = "2026-08-15T08:19:02.492Z" }, + { url = "https://files.pythonhosted.org/packages/03/3b/0cc9a26777334ab2f2e3089b948bbf4e4fe72ea70b897715ef6415043ec8/charset_normalizer-3.5.1-cp315-cp315-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:baf3775a2635e5a11fbd5e4e64ee69c7e86875d224a5c72aca4c141064589a90", size = 237014, upload-time = "2026-08-15T08:19:03.943Z" }, + { url = "https://files.pythonhosted.org/packages/8c/c2/027335f0aa337a2a2e121bac1ad88c4f02ba6053ea0926802784f3db11af/charset_normalizer-3.5.1-cp315-cp315-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:8ac8c94b6539074e0f40899301273ac8402b9b3e01c7b7ba269ff30340aaaf20", size = 266174, upload-time = "2026-08-15T08:19:05.598Z" }, + { url = "https://files.pythonhosted.org/packages/86/d3/e367787febe4e74769dec0f406f2c3c8d1b955fce5aee1fd0f94e8367a45/charset_normalizer-3.5.1-cp315-cp315-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:8fe532b3c966d1fb794e0698e4589d0444017ae77fc0b31edea13c0e35bcc449", size = 263361, upload-time = "2026-08-15T08:19:07.251Z" }, + { url = "https://files.pythonhosted.org/packages/af/3d/391b193eb9f3e84b02f9314088c386debdc0debee843535aaea2e2c6715d/charset_normalizer-3.5.1-cp315-cp315-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:5c84bec0ab5ae0c64bfe73a7d2adcb5ce73b467523fc27fd6a28ab2aa6cbe35a", size = 252143, upload-time = "2026-08-15T08:19:08.816Z" }, + { url = "https://files.pythonhosted.org/packages/2e/57/de221f1745a90d418199761967e2776bfe2c275a1194220985e8c1d37833/charset_normalizer-3.5.1-cp315-cp315-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:854066be00447fa8de2ccbbe893e2ffc4b123ef16d897af794c1e18bd4a714b0", size = 252086, upload-time = "2026-08-15T08:19:10.255Z" }, + { url = "https://files.pythonhosted.org/packages/c8/e3/d119f86a01f9331e8186175f24873b1d74a7ee9e2e4b4d68f9947dae5afd/charset_normalizer-3.5.1-cp315-cp315-musllinux_1_2_aarch64.whl", hash = "sha256:21b82d8082f6f5e7f456ef0bd16323d08de1266efbfeb476e64b2a91d1471a4e", size = 245231, upload-time = "2026-08-15T08:19:11.807Z" }, + { url = "https://files.pythonhosted.org/packages/26/de/d8e48c135ae480879539cdb179c8d3b50c7879497d75dd899b5763b69cee/charset_normalizer-3.5.1-cp315-cp315-musllinux_1_2_armv7l.whl", hash = "sha256:838648accb3a7fd9803fd45c87bce8509648eb0c11bc34e216141300977244f2", size = 241546, upload-time = "2026-08-15T08:19:13.416Z" }, + { url = "https://files.pythonhosted.org/packages/67/c4/217755fd1abc50d326c252922cd642002758095a81ff45010337b8b3ef65/charset_normalizer-3.5.1-cp315-cp315-musllinux_1_2_ppc64le.whl", hash = "sha256:195ce897c6153c0700078142cf8efe3e6454ca4cf4357499e4078dfd83396626", size = 267033, upload-time = "2026-08-15T08:19:14.981Z" }, + { url = "https://files.pythonhosted.org/packages/b8/d7/34d8e404e358d2adcc5a228c2134643af00104c8fb0bf525f3688d756f05/charset_normalizer-3.5.1-cp315-cp315-musllinux_1_2_riscv64.whl", hash = "sha256:978eab16f55b4ab2c2a745be9a0a840bf8f09a7f227d9c76eb30214d078865a5", size = 252045, upload-time = "2026-08-15T08:19:16.618Z" }, + { url = "https://files.pythonhosted.org/packages/5e/fa/40414471acf0aa0692ca77305aa00e434fcd8288f0941c93c30e9a5f8f2f/charset_normalizer-3.5.1-cp315-cp315-musllinux_1_2_s390x.whl", hash = "sha256:cc0329df4caaceb950d2f580b5ac716a377f7059624a0bafaeaf8a218c6ed774", size = 264866, upload-time = "2026-08-15T08:19:18.101Z" }, + { url = "https://files.pythonhosted.org/packages/32/90/fcc850bae791abd2e0c041847f13e270aa08692a79f3e00de6d2dce1cb50/charset_normalizer-3.5.1-cp315-cp315-musllinux_1_2_x86_64.whl", hash = "sha256:687c9ca3035544b113bea2055e180af96fb63c0c476e22a9180f51925186e7b7", size = 253932, upload-time = "2026-08-15T08:19:19.734Z" }, + { url = "https://files.pythonhosted.org/packages/af/af/53afe99068b3c10b4cbae592a52ef72a7c92c0188440e83ee3a078fd8f75/charset_normalizer-3.5.1-cp315-cp315-win32.whl", hash = "sha256:706bfd38730a5ac7a365793269a00f4e988178cec121391f4248d84ad8c972e9", size = 180320, upload-time = "2026-08-15T08:19:21.37Z" }, + { url = "https://files.pythonhosted.org/packages/c9/bc/f46a132041b29e4a8779ed712d3df1bf112e94ca8de58b66d7ec2c0cf8b9/charset_normalizer-3.5.1-cp315-cp315-win_amd64.whl", hash = "sha256:92caef967d287a407085d61176fce4012b1dd62daed4eb6d5ceb26d3d2538712", size = 204174, upload-time = "2026-08-15T08:19:23.088Z" }, + { url = "https://files.pythonhosted.org/packages/a1/5d/9ed554480eda8e447b673648628fdc29574d23dbad01fe11837adedd1cae/charset_normalizer-3.5.1-cp315-cp315-win_arm64.whl", hash = "sha256:5fc45d653ea8c9a20479167e11d4a0f8cb2fa3470737ab6f9c827532313187b7", size = 184126, upload-time = "2026-08-15T08:19:24.471Z" }, + { url = "https://files.pythonhosted.org/packages/3b/32/9b8929bf384061ee1fe5d9c27c6f9776d3d824039ad4e14c88ec00c7808e/charset_normalizer-3.5.1-cp315-cp315t-macosx_10_15_universal2.whl", hash = "sha256:59171c6e45bf07d0d5cab3b0bf81d945035530f6873398b3b531c31184d46663", size = 381441, upload-time = "2026-08-15T08:19:26.038Z" }, + { url = "https://files.pythonhosted.org/packages/96/10/e9aa7923d3ddac652c99a1c5f7be494e737e151566a44abe018daf757f2c/charset_normalizer-3.5.1-cp315-cp315t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:9dbdd9205662134957cf0c324f639bdc5031c0ca056e2369e238db75187c0f11", size = 241742, upload-time = "2026-08-15T08:19:27.532Z" }, + { url = "https://files.pythonhosted.org/packages/28/53/a2d249ebddf47b889a100c0bdcb61a2f9dbb8bc24ef325cc062e4f476877/charset_normalizer-3.5.1-cp315-cp315t-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:e4b018dc5a0eee4676e38fe84a47a427816c590b93b55d9025274ec4d6ffc2dc", size = 235298, upload-time = "2026-08-15T08:19:29.274Z" }, + { url = "https://files.pythonhosted.org/packages/7d/07/469f78af590f7d5cd48e20d8dbfa3d66deeff9ba37768c04d886b5afd45c/charset_normalizer-3.5.1-cp315-cp315t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:ced3fdd71aaa83ce593746c2edb42b7a59cb4c19c8b5c407781c72e493aae55a", size = 262500, upload-time = "2026-08-15T08:19:30.955Z" }, + { url = "https://files.pythonhosted.org/packages/55/66/3bb56a47f7dcba014055b1a1d33c6f08bbe9c1e74dba154cfa25f90ae885/charset_normalizer-3.5.1-cp315-cp315t-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:19a3dd5aa73cef1c99687c4fc57db016a9c17104ae1185da88ba566a5d3bebe4", size = 258888, upload-time = "2026-08-15T08:19:32.458Z" }, + { url = "https://files.pythonhosted.org/packages/ff/c1/2adc2800903fb013210349313b710a5376856578d9e33e6b9a1d8b36714a/charset_normalizer-3.5.1-cp315-cp315t-manylinux2014_x86_64.manylinux_2_17_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:cc5d36d96478aa9c60654bd932525bf32964c62a7281eafdf16d85003a8d6004", size = 250243, upload-time = "2026-08-15T08:19:33.94Z" }, + { url = "https://files.pythonhosted.org/packages/95/b5/a18d0dd1157ab655cc2cb14a545f4a4784bbad70ab3502412e36097502d9/charset_normalizer-3.5.1-cp315-cp315t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:04368edf83514385ffc3e1cfd4546e595f4f1272dd23ba437a93a9cc3741d47b", size = 249871, upload-time = "2026-08-15T08:19:35.413Z" }, + { url = "https://files.pythonhosted.org/packages/ad/c3/525f508cd1e58d0450ac55ed40ac75bc3a97482c59def5278456a5fbf03c/charset_normalizer-3.5.1-cp315-cp315t-musllinux_1_2_aarch64.whl", hash = "sha256:9b5db6052055d34d41230fb78d7c439c23dc536a9896f6cb039e8dd92cfc1263", size = 243580, upload-time = "2026-08-15T08:19:36.886Z" }, + { url = "https://files.pythonhosted.org/packages/7c/c1/49a91fe7e97c8140094ca5c64161ab623a70d9f636bf834eace14048acb5/charset_normalizer-3.5.1-cp315-cp315t-musllinux_1_2_armv7l.whl", hash = "sha256:252d099029bcbea642f2a06c4ed5046bdf8b5a8150b64afa5e027e88b106e5ee", size = 239807, upload-time = "2026-08-15T08:19:38.392Z" }, + { url = "https://files.pythonhosted.org/packages/d3/58/56a48c296601274c4689b864a8e2dfb209b81dfcb39472753ce95eea662b/charset_normalizer-3.5.1-cp315-cp315t-musllinux_1_2_ppc64le.whl", hash = "sha256:6199d5606e2bbf2b096cf64d03f8b6790c91081d5ac866b8e7bb6422738cc60c", size = 264083, upload-time = "2026-08-15T08:19:39.856Z" }, + { url = "https://files.pythonhosted.org/packages/10/4c/dc48409274a1817ff349711d26c62aa0c597df865d4d69ef79160c859193/charset_normalizer-3.5.1-cp315-cp315t-musllinux_1_2_riscv64.whl", hash = "sha256:77efcff2b23071c349402ac1066667a3d011f62398d81408c9b88ad991747c9e", size = 250317, upload-time = "2026-08-15T08:19:41.53Z" }, + { url = "https://files.pythonhosted.org/packages/81/58/d325912115caec62d6bdd77bbab5e0b7da5d234a9f20affdffcbcb530d0b/charset_normalizer-3.5.1-cp315-cp315t-musllinux_1_2_s390x.whl", hash = "sha256:a5cbd90ecf0fc62e64726917ad083b73001f0563657a87ec3c0b504e277dc90d", size = 258173, upload-time = "2026-08-15T08:19:43.07Z" }, + { url = "https://files.pythonhosted.org/packages/34/f7/b13b1ccae2c8ec63980d13be1890eb73f8aeabbfce02a24aabc0908788f5/charset_normalizer-3.5.1-cp315-cp315t-musllinux_1_2_x86_64.whl", hash = "sha256:4d26f14f041e83dd8edfd61f4cd4fa7285d31798b5bf1f28e70c367ba6c41d61", size = 251960, upload-time = "2026-08-15T08:19:44.587Z" }, + { url = "https://files.pythonhosted.org/packages/1e/25/ed3f9919c5aef8cc818be1f972f565f7610d7b2076b8ebb98839516ffc3c/charset_normalizer-3.5.1-cp315-cp315t-win32.whl", hash = "sha256:ac13b004224fb341e1e25a1ed5e19d32f57cdb2a403e01f003b46f051a550f6f", size = 191186, upload-time = "2026-08-15T08:19:46.293Z" }, + { url = "https://files.pythonhosted.org/packages/69/d5/43c2b3e9d8267092b913eb8b0603f0f71993c395632886bd37a7223f96cf/charset_normalizer-3.5.1-cp315-cp315t-win_amd64.whl", hash = "sha256:35aea775dc2bd5f54cd84a1cd2696cc3207c479cb9cf0bd346f0d343e4300ddb", size = 215947, upload-time = "2026-08-15T08:19:47.853Z" }, + { url = "https://files.pythonhosted.org/packages/a8/76/9aad3e9c8865e5e0efa9a7f6f81c37a67635a985145ecd44528a81e088ee/charset_normalizer-3.5.1-cp315-cp315t-win_arm64.whl", hash = "sha256:fb78f6e7fcd8ad785d28cd577168bc1aaee827b25bb8755638f694794ea98f0a", size = 193909, upload-time = "2026-08-15T08:19:49.383Z" }, + { url = "https://files.pythonhosted.org/packages/5b/97/fb4e82231aba271ffd775a1b4993b0defc4e3059f286ae41d9433409fe85/charset_normalizer-3.5.1-cp37-abi3-macosx_10_9_universal2.whl", hash = "sha256:41876ee62a3dddf48ff1121ad8f0798032aa03f2fd35f21f34a4cab14f18d8d2", size = 331467, upload-time = "2026-08-15T08:19:50.959Z" }, + { url = "https://files.pythonhosted.org/packages/9f/2f/fe3f187327aac18e2d54e9d2b08e15d27bf9b642d9e51c219f130fc34d1a/charset_normalizer-3.5.1-cp37-abi3-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", hash = "sha256:a6dac12ff6b846103483683f60c5f8fee205121adc58ffd87e90a90a3af69e99", size = 253057, upload-time = "2026-08-15T08:19:52.654Z" }, + { url = "https://files.pythonhosted.org/packages/d7/c7/9e48cee5c161fe24da823b61bf381921d77cb994a0a4de148e95018c1984/charset_normalizer-3.5.1-cp37-abi3-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:cee5dd7c6fb5dd52a0fe2a740f9bc6e3593f5f8b1788bde49de02086f30182b2", size = 240930, upload-time = "2026-08-15T08:19:54.163Z" }, + { url = "https://files.pythonhosted.org/packages/49/e0/716601f3cc69be7b198951150c75ead1ece33c3c8036ff6ffa46029659a0/charset_normalizer-3.5.1-cp37-abi3-manylinux2014_armv7l.manylinux_2_17_armv7l.manylinux_2_31_armv7l.whl", hash = "sha256:343fb4f2821043bd87095f7b08a1a181febc8e36ac64212143bbfd0a0e1bc235", size = 230822, upload-time = "2026-08-15T08:19:55.807Z" }, + { url = "https://files.pythonhosted.org/packages/d3/05/71bfc5caa0abcc45aea1f6a4d50ac68e59605ddc7666fe8494f4cd229665/charset_normalizer-3.5.1-cp37-abi3-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:ae4a097991662cd4fff0ddc74e0fe7874f82e00042fa0ea00855645ed0c79598", size = 260037, upload-time = "2026-08-15T08:19:57.312Z" }, + { url = "https://files.pythonhosted.org/packages/c3/92/de7e32ed05341e7a9c4c877c318418197b7f2d66a3b68d561bf2ac57ca3e/charset_normalizer-3.5.1-cp37-abi3-manylinux2014_s390x.manylinux_2_17_s390x.manylinux_2_28_s390x.whl", hash = "sha256:4b599739b93b2cbeded49645ae3c8d1405c29ddfbceac1545c87a3f9580a9e96", size = 255097, upload-time = "2026-08-15T08:19:59.056Z" }, + { url = "https://files.pythonhosted.org/packages/f5/7b/ade0a122600319dfa0b1000ab0f9731c94a817904cf3c5de408c73a4ede7/charset_normalizer-3.5.1-cp37-abi3-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:b39b69b347e5e47a3b5b8cfc005c68c1ba347474e3960236c4944a8ecd174962", size = 250166, upload-time = "2026-08-15T08:20:00.612Z" }, + { url = "https://files.pythonhosted.org/packages/75/9c/019fbb9f4834491a160951349b1a3714439376f66e5f7cf18b4f18f0c7aa/charset_normalizer-3.5.1-cp37-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:a2028475ba855475b8b4d3cfeb4994269c967aea8b9892dfba907f4263a863a3", size = 241821, upload-time = "2026-08-15T08:20:02.321Z" }, + { url = "https://files.pythonhosted.org/packages/2b/b8/11d4840bfc99330cc7fbcc2681ee5a044553a6e77655508d8f9b2bff7b34/charset_normalizer-3.5.1-cp37-abi3-musllinux_1_2_armv7l.whl", hash = "sha256:36047af20e17097c3bb9476c2b7655f2f7aa51322c0ba58c07695bedf755a950", size = 232529, upload-time = "2026-08-15T08:20:04.008Z" }, + { url = "https://files.pythonhosted.org/packages/18/96/2b3a21492d9f65171ac75d872f5018260013d00bfa0ff70ec9f179148cbd/charset_normalizer-3.5.1-cp37-abi3-musllinux_1_2_ppc64le.whl", hash = "sha256:4c4fb141a727957c93edfe5c32a26ceb6b5f6461d67146e2d39f51e16170bea8", size = 260348, upload-time = "2026-08-15T08:20:05.877Z" }, + { url = "https://files.pythonhosted.org/packages/d6/aa/a69a2028e8bd052476c245460ab19d7de595de084dd968f2d75cd50c3e25/charset_normalizer-3.5.1-cp37-abi3-musllinux_1_2_riscv64.whl", hash = "sha256:2f293479cce755c75f1697e87c409b7ae4c555c7dfecb6e988ad13abba943031", size = 247234, upload-time = "2026-08-15T08:20:07.487Z" }, + { url = "https://files.pythonhosted.org/packages/35/8a/3d130aeabcaf3d2466af76b7b141c08d9e89c9016ab4b7cdd0f7dc2d1c62/charset_normalizer-3.5.1-cp37-abi3-musllinux_1_2_s390x.whl", hash = "sha256:3588e376b3ea2eea84976f67273d679f229e24c66dce7b82ae45aef04ff6e072", size = 256917, upload-time = "2026-08-15T08:20:09.142Z" }, + { url = "https://files.pythonhosted.org/packages/80/c2/a7379b840292d0c1ab9fbd17d1f3967aa81794dc95bc74be8999d7fedcf7/charset_normalizer-3.5.1-cp37-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:e199fb99720074809a7720f1c0b4d919eea8b87e88713e0f8f602f7bef543d9d", size = 254846, upload-time = "2026-08-15T08:20:10.727Z" }, + { url = "https://files.pythonhosted.org/packages/01/65/d43b714731bb2f40d4053dfa00ecfc1c5a301f8e3316c5db3a09af59fe94/charset_normalizer-3.5.1-cp37-abi3-win32.whl", hash = "sha256:dd732602a7009217f658d5863d12d79d373a4de0eebc111094bcdd3bb8e0a6cc", size = 174216, upload-time = "2026-08-15T08:20:12.334Z" }, + { url = "https://files.pythonhosted.org/packages/35/4f/b911ed898b26a09789eba9c9200c999aff6c61b4bafaf4838e56d1a1e1a3/charset_normalizer-3.5.1-cp37-abi3-win_amd64.whl", hash = "sha256:70055ff39b97c99e7ae40ea3e393fb62aa2e44dbd9b29f8d14f42fb0025c3959", size = 199764, upload-time = "2026-08-15T08:20:13.908Z" }, + { url = "https://files.pythonhosted.org/packages/f0/a7/920baf467bfd9bf689f3b318340f37aee4572a71f162bd8db51da55ba4fa/charset_normalizer-3.5.1-cp37-abi3-win_arm64.whl", hash = "sha256:87e4f41d375c0b9be2fb5251aee4b8a689169e134535aed81bf085c3b647451e", size = 287318, upload-time = "2026-08-15T08:20:15.551Z" }, + { url = "https://files.pythonhosted.org/packages/cc/61/d01fc49b8dea277640b55a9e15960dbca9fdc8c9fde18e572d39c59f4019/charset_normalizer-3.5.1-py3-none-any.whl", hash = "sha256:6df0ec430f9a831772c23ca5a224cba36517a58a84bb32c32bb59a9fa67c47f6", size = 68658, upload-time = "2026-08-15T08:20:43.306Z" }, ] [[package]] @@ -422,11 +422,11 @@ wheels = [ [[package]] name = "idna" -version = "3.18" +version = "3.19" source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/cd/63/9496c57188a2ee585e0f1db071d75089a11e98aa86eb99d9d7618fc1edce/idna-3.18.tar.gz", hash = "sha256:ffb385a7e039654cef1ab9ef32c6fafe283c0c0467bba1d9029738ce4a14a848", size = 196711, upload-time = "2026-06-02T14:34:07.794Z" } +sdist = { url = "https://files.pythonhosted.org/packages/5f/f7/abb373e5757eaec4b922b92f97ec8d6d7e057cf06778247604fbc4e7c3f3/idna-3.19.tar.gz", hash = "sha256:5e0811a4383b21dc5838069f801c4fb62113b7447663d2530d2bd6e77b49bf15", size = 215237, upload-time = "2026-08-18T05:14:24.27Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/1e/5e/d4e9f1a599fb8e573b7b87160658329fbf28d19eac2718f51fc3def3aa5a/idna-3.18-py3-none-any.whl", hash = "sha256:7f952cbe720b688055e3f87de14f5c3e5fdaa8bc3928985c4077ca689de849a2", size = 65455, upload-time = "2026-06-02T14:34:06.319Z" }, + { url = "https://files.pythonhosted.org/packages/57/b0/0e52c878c53f245edd3a11020f20979b3f490f245af532c7cae3027754b5/idna-3.19-py3-none-any.whl", hash = "sha256:815e7be7a7806d54abb586dc943addc79e8b2ee16915059658cbeff4b1b43bf4", size = 68550, upload-time = "2026-08-18T05:14:22.343Z" }, ] [[package]] @@ -440,26 +440,26 @@ wheels = [ [[package]] name = "maturin" -version = "1.14.1" +version = "1.15.0" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "tomli", marker = "python_full_version < '3.11'" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/e7/b3/addd877f871fb1860d46d3a4f206ecb10b946c85846805e6367631926fd3/maturin-1.14.1.tar.gz", hash = "sha256:9d6577a62cd08e0ceba7a0db06fb098e0c9b1b3429bad747a4f3a18215a1b3df", size = 369637, upload-time = "2026-06-19T05:19:49.774Z" } +sdist = { url = "https://files.pythonhosted.org/packages/b9/c8/22e5e21b2679c9bce6415ca578034ca2cc9316be0642ae21e051a2d5198c/maturin-1.15.0.tar.gz", hash = "sha256:94b26cc8e8aba61a5f2099715fe640e18c5f678e9a500408b38761263954228a", size = 385504, upload-time = "2026-08-24T12:11:22.665Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/f4/f0/97c5a5bd9c71653a066c0976a484eaaae50b9369557838a4176b7b0bdaa5/maturin-1.14.1-py3-none-linux_armv6l.whl", hash = "sha256:522292398945442cdafa9daeb2271b2340fbde57027b818f923f88eab04174f8", size = 10207496, upload-time = "2026-06-19T05:19:09.321Z" }, - { url = "https://files.pythonhosted.org/packages/fe/83/294bca639b0e052f1e2f65199b3db258780c7d4e31408b934c9c974a1379/maturin-1.14.1-py3-none-macosx_10_12_x86_64.macosx_11_0_arm64.macosx_10_12_universal2.whl", hash = "sha256:ffe5ad71f21d1e6603c4dd75f7fee34adf5ed5ebcebb692886549888ebb329ed", size = 19680113, upload-time = "2026-06-19T05:19:13.43Z" }, - { url = "https://files.pythonhosted.org/packages/43/b6/79c881410a3b1c187f7eb3d407aecae646c6a4433d630d72200359015e83/maturin-1.14.1-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:f3306078070c1508fd715b9116070cbcaff5959024272a9f1e6f5cb29768b86c", size = 10169205, upload-time = "2026-06-19T05:19:16.615Z" }, - { url = "https://files.pythonhosted.org/packages/93/9d/44b6f26dcb7f7a04c5501ac2dbb6ca1490150682baa525ca5860504f9eab/maturin-1.14.1-py3-none-manylinux_2_12_i686.manylinux2010_i686.musllinux_1_1_i686.whl", hash = "sha256:cd457cd88961156e26379e1155bd287cc0ec1c8b2f1582b0660fb31b87c8842d", size = 10188098, upload-time = "2026-06-19T05:19:19.736Z" }, - { url = "https://files.pythonhosted.org/packages/1a/bd/9c0d5d6983905ce2c9edaa073a7e89355a9cf7f396988e05d32f1c37785d/maturin-1.14.1-py3-none-manylinux_2_12_x86_64.manylinux2010_x86_64.musllinux_1_1_x86_64.whl", hash = "sha256:dfc54ae32e6fcb18302193ab9a30b0b25eefffba994ae13238974805533ef75e", size = 10627576, upload-time = "2026-06-19T05:19:22.713Z" }, - { url = "https://files.pythonhosted.org/packages/e5/33/b096412bd6a7cb399652b260666f901adf88a687181a6dbd6a3f89f0a94e/maturin-1.14.1-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.musllinux_1_1_aarch64.whl", hash = "sha256:a131d912b5267e640bc96d70f4914e10590aed64082ec9abacba7cea52004224", size = 10085181, upload-time = "2026-06-19T05:19:25.69Z" }, - { url = "https://files.pythonhosted.org/packages/56/8d/08c3bf469c38a23c9e6c877e338193001eb604d010fedc08341974e38528/maturin-1.14.1-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.musllinux_1_1_armv7l.whl", hash = "sha256:be18fc568fb76884c0205456336892a75105ec398e6b667cd777c6268bd06d69", size = 10026363, upload-time = "2026-06-19T05:19:28.904Z" }, - { url = "https://files.pythonhosted.org/packages/3a/a4/c4d1a92839f8745ab4aab988a7db884a79d6d710bd3b286fcf9316dece1a/maturin-1.14.1-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.musllinux_1_1_ppc64le.whl", hash = "sha256:994a0c8ba3ad8a92b3a9ee1b02645d200d610216b15cff5102b0fe65e8e08666", size = 13321347, upload-time = "2026-06-19T05:19:32.411Z" }, - { url = "https://files.pythonhosted.org/packages/b3/fa/170f04624d03fd07d2a8b1b67de83a127af93aef9eaa425839553347297b/maturin-1.14.1-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:be80866363e605d137991b491a741a84cde9ae350183c4c85f49690ca9aaaa65", size = 10877609, upload-time = "2026-06-19T05:19:35.448Z" }, - { url = "https://files.pythonhosted.org/packages/61/ad/1ae2e1d0ded282bf2c55ac13f0811d87deb425e200ae64a15785675dede9/maturin-1.14.1-py3-none-manylinux_2_31_riscv64.musllinux_1_1_riscv64.whl", hash = "sha256:5282dffd4b539d2be245f4e5b1a5ab6bc1033b58f4a4872f5833f9d43c954aa4", size = 10417316, upload-time = "2026-06-19T05:19:38.28Z" }, - { url = "https://files.pythonhosted.org/packages/fb/27/bf677183920718da49cd7982d6a3ffc440aad8919329f571d189f81b7bdf/maturin-1.14.1-py3-none-win32.whl", hash = "sha256:1a04de0a20188f95c721b5702eed18140bdcccb28c386797093eca3f62f4d4e0", size = 8931293, upload-time = "2026-06-19T05:19:41.183Z" }, - { url = "https://files.pythonhosted.org/packages/63/4b/585adeb9167b08d3cdff0032a938b0e72655c92003df4f52c3f696a1bcc2/maturin-1.14.1-py3-none-win_amd64.whl", hash = "sha256:3c9f94640ecc4895e94abaf834a0684430032c865b2748a36c12461fd9252fdd", size = 10314067, upload-time = "2026-06-19T05:19:44.389Z" }, - { url = "https://files.pythonhosted.org/packages/51/d4/dac8c0720ae246be1700afb6fbdbbea20fe35b13f6570b2f70faa005df77/maturin-1.14.1-py3-none-win_arm64.whl", hash = "sha256:15cea8fcb3ba47dd636f50092bb34baea8b04ac777392f23e6bf8a9a61efb894", size = 9718943, upload-time = "2026-06-19T05:19:47.49Z" }, + { url = "https://files.pythonhosted.org/packages/14/69/5c01b461044eb1f45ddcce006706eb88110c793cdb11c7ae0b5e08492e94/maturin-1.15.0-py3-none-linux_armv6l.whl", hash = "sha256:6bf6dc62e22d4dcfd5a51244ff0d58975fa4979c48209fe84159617648956d82", size = 10206220, upload-time = "2026-08-24T12:10:53.327Z" }, + { url = "https://files.pythonhosted.org/packages/eb/1f/2b431554e11687cdb1077e0cdadcc118c53f611086b3af00c8545a67c6a5/maturin-1.15.0-py3-none-macosx_10_12_x86_64.macosx_11_0_arm64.macosx_10_12_universal2.whl", hash = "sha256:cd35772633f489841132bc8e71d6fc7f842df30b9c05cd5cdf1ee1ddcb744cc7", size = 19416513, upload-time = "2026-08-24T12:10:56.126Z" }, + { url = "https://files.pythonhosted.org/packages/51/36/e23a21cb34a648b711036b9b2fe1d4f3f4ee24f8db54215d73f1a9a3a3ec/maturin-1.15.0-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:c40b4eae7bf5ef1f4b1af8d623fe4105016f93578fb15b764e741d08ec3b92dd", size = 10014962, upload-time = "2026-08-24T12:10:58.486Z" }, + { url = "https://files.pythonhosted.org/packages/8c/80/33b15cb2d8f30f12c807955e8f2fd775027692904e30ec0784744ce8cd83/maturin-1.15.0-py3-none-manylinux_2_12_i686.manylinux2010_i686.musllinux_1_1_i686.whl", hash = "sha256:7eb066372f541f8eb4909c79c5d9bd0b9e8125980bdf1ec9e8aba23c6c8d6c55", size = 10196223, upload-time = "2026-08-24T12:11:00.696Z" }, + { url = "https://files.pythonhosted.org/packages/fe/91/b495e19e2f5c503b540452b2039115e7b2363867e8c5ad4179eb752fa92c/maturin-1.15.0-py3-none-manylinux_2_12_x86_64.manylinux2010_x86_64.musllinux_1_1_x86_64.whl", hash = "sha256:653020a63525bb224e5ab0adf02e17a2e08bc86dbea7fc1399c9a56d7529b99e", size = 10541186, upload-time = "2026-08-24T12:11:02.857Z" }, + { url = "https://files.pythonhosted.org/packages/5d/2b/2abff58037188d852b124871b1f0d720e1c2bfb3d4f1b03d87c52cd66488/maturin-1.15.0-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.musllinux_1_1_aarch64.whl", hash = "sha256:0ebf9767892725083138e671c34482c660317a2f3d6a29fc0e0f34e9d8c99136", size = 10083468, upload-time = "2026-08-24T12:11:05.012Z" }, + { url = "https://files.pythonhosted.org/packages/3f/07/b7e9f8be99a6627849e81ac7b6694876bce8f50a92995fe17e3cf2610f0a/maturin-1.15.0-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.musllinux_1_1_armv7l.whl", hash = "sha256:7ab7eebffd7b8debca2265985de4eaeb332141276d24b9560b5ad484d4b3add1", size = 10047786, upload-time = "2026-08-24T12:11:07.042Z" }, + { url = "https://files.pythonhosted.org/packages/ca/ab/167e3cb7accee11b507dbe53e0e87aeccb376d44ae66284c96ee4df3a9fd/maturin-1.15.0-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.musllinux_1_1_ppc64le.whl", hash = "sha256:126e12e618b4db42f68c779a56d41f82a390145ba36ac3f621d057eb34f5ad9d", size = 13315332, upload-time = "2026-08-24T12:11:09.433Z" }, + { url = "https://files.pythonhosted.org/packages/14/4d/801379f646cbc6b00998e5289b0630a886be3a4ee4c75b6bc9b87478a7f1/maturin-1.15.0-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:4f9d33e6c3f9615c8caceecbbbd440f8eb25a3ddeb687077682cd5eca2e9ae15", size = 10807183, upload-time = "2026-08-24T12:11:11.73Z" }, + { url = "https://files.pythonhosted.org/packages/89/27/2e612e1cbd1580e9e94d4722c227b5180dca27b32b955a34b79918aa1292/maturin-1.15.0-py3-none-manylinux_2_31_riscv64.musllinux_1_1_riscv64.whl", hash = "sha256:bf29beddd0c6708f112db51d5275fc28b28b9e9c9c5faae387eaef662918b176", size = 10413274, upload-time = "2026-08-24T12:11:14.04Z" }, + { url = "https://files.pythonhosted.org/packages/70/d8/202a7b4d75a51f20f84ec9ce3b7345b12b822207072164e2b1c6ef665125/maturin-1.15.0-py3-none-win32.whl", hash = "sha256:da649988be98e87e009e51b1bf0d301b6a301bc0cecbdd60d40d8ba60748d1ca", size = 8928744, upload-time = "2026-08-24T12:11:16.269Z" }, + { url = "https://files.pythonhosted.org/packages/40/dc/4e90da594986ba78dd3bc8a5921ecdcdb11085b22b02a412caab3b225601/maturin-1.15.0-py3-none-win_amd64.whl", hash = "sha256:552c2be4afd43fe8d5c9f3ec8d4c4756d973b8dcbe94c14084390301f50243e1", size = 10335085, upload-time = "2026-08-24T12:11:18.326Z" }, + { url = "https://files.pythonhosted.org/packages/8b/10/15d4314edf130955edf2dc237aa393a8a7c10f2b9b57b89fa2f61f915659/maturin-1.15.0-py3-none-win_arm64.whl", hash = "sha256:c7dc0c66c78d3debdd9c5aa807e861fbcbf07f3505d34b125df74c03986b0f48", size = 9713795, upload-time = "2026-08-24T12:11:20.83Z" }, ] [[package]] @@ -482,11 +482,11 @@ wheels = [ [[package]] name = "pygments" -version = "2.20.0" +version = "2.21.0" source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/c3/b2/bc9c9196916376152d655522fdcebac55e66de6603a76a02bca1b6414f6c/pygments-2.20.0.tar.gz", hash = "sha256:6757cd03768053ff99f3039c1a36d6c0aa0b263438fcab17520b30a303a82b5f", size = 4955991, upload-time = "2026-03-29T13:29:33.898Z" } +sdist = { url = "https://files.pythonhosted.org/packages/49/2e/ced460408999b33da6b31b0021b0f37d329e202d4169aeb164493778f25b/pygments-2.21.0.tar.gz", hash = "sha256:610ca751c9bc2492b38eb9a38a7fbc93edbbb2d7182edaf34e66ae493dee5c8c", size = 5005329, upload-time = "2026-08-17T08:02:48.824Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/f4/7e/a72dd26f3b0f4f2bf1dd8923c85f7ceb43172af56d63c7383eb62b332364/pygments-2.20.0-py3-none-any.whl", hash = "sha256:81a9e26dd42fd28a23a2d169d86d7ac03b46e2f8b59ed4698fb4785f946d0176", size = 1231151, upload-time = "2026-03-29T13:29:30.038Z" }, + { url = "https://files.pythonhosted.org/packages/71/46/17f022dd3e953bf20a04a028a21ec746d942f8d2af30fa0f124fa0e6a684/pygments-2.21.0-py3-none-any.whl", hash = "sha256:2363c69b61c4a97c838da3b130dcd6468f4848992b21a82f2a63ec34377137d9", size = 1250147, upload-time = "2026-08-17T08:02:44.912Z" }, ] [[package]] @@ -581,11 +581,11 @@ wheels = [ [[package]] name = "python-dotenv" -version = "1.2.2" +version = "1.2.3" source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/82/ed/0301aeeac3e5353ef3d94b6ec08bbcabd04a72018415dcb29e588514bba8/python_dotenv-1.2.2.tar.gz", hash = "sha256:2c371a91fbd7ba082c2c1dc1f8bf89ca22564a087c2c287cd9b662adde799cf3", size = 50135, upload-time = "2026-03-01T16:00:26.196Z" } +sdist = { url = "https://files.pythonhosted.org/packages/6a/53/ed9d74092561d4b01a2ef1349d52cdbc135e526c245f366b089cfca6de49/python_dotenv-1.2.3.tar.gz", hash = "sha256:a20a594dabeaa385725aa239d5244871c143ecb356add8a20fcf23773a6c3a35", size = 58945, upload-time = "2026-08-16T16:54:54.067Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/0b/d7/1959b9648791274998a9c3526f6d0ec8fd2233e4d4acce81bbae76b44b2a/python_dotenv-1.2.2-py3-none-any.whl", hash = "sha256:1d8214789a24de455a8b8bd8ae6fe3c6b69a5e3d64aa8a8e5d68e694bbcb285a", size = 22101, upload-time = "2026-03-01T16:00:25.09Z" }, + { url = "https://files.pythonhosted.org/packages/0d/17/c5c6b53ddc18f297992099b3d9ec16c855c0ccc83263a21fe4d1c625ec6c/python_dotenv-1.2.3-py3-none-any.whl", hash = "sha256:904552145e8bfed22162c09dab1c2b9b54fefa7b23ba780f4f26ca0316b0f0d9", size = 22780, upload-time = "2026-08-16T16:54:52.473Z" }, ] [[package]] @@ -630,27 +630,27 @@ wheels = [ [[package]] name = "ruff" -version = "0.16.3" +version = "0.16.4" source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/61/b3/3213589383f8f1b3938781bd1278713f6d18621a14992b3e81fefb8a5ef9/ruff-0.16.3.tar.gz", hash = "sha256:e76d33a347661a84b5be6d043d0347fdc745dfdcf825a8f4fed64b5e26eebdf2", size = 4891904, upload-time = "2026-08-13T15:17:13.381Z" } +sdist = { url = "https://files.pythonhosted.org/packages/00/8f/d8074b1f25e003164087a8bfe79a0f1a3945135764dbb6aaab04103dcaf9/ruff-0.16.4.tar.gz", hash = "sha256:13171aa9d9af2240ee3504e639de73122c67e74036de5ba2e1d01422cd17e3dc", size = 4899731, upload-time = "2026-08-20T17:43:59.196Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/bf/96/493770daebd68c0a67f1549fdf519f53be51fc435186c0585bcc272fd76c/ruff-0.16.3-py3-none-linux_armv6l.whl", hash = "sha256:0c5710e247a58a4521e66e124ba9a74655b414f61ba3a2e9e3811e11098f48f7", size = 10902799, upload-time = "2026-08-13T15:16:27.382Z" }, - { url = "https://files.pythonhosted.org/packages/5e/e6/2becf3942fddc29a29b8df47691d456fb1085391a694f74d84513251418c/ruff-0.16.3-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:fe155130631a2471fd2e14a7a664a4dfbd7194b8229c3d7b2a40b21178639081", size = 11135539, upload-time = "2026-08-13T15:16:30.87Z" }, - { url = "https://files.pythonhosted.org/packages/3e/1e/4b8b72f0d006dbf19326aa99f9ca0ee2ff374187c4d301cf529a51aa06fe/ruff-0.16.3-py3-none-macosx_11_0_arm64.whl", hash = "sha256:e2ed719e14aa64d895c2ee922594a90a43c861a93f0575a95ff8c47cdbd13eb9", size = 10475095, upload-time = "2026-08-13T15:16:33.259Z" }, - { url = "https://files.pythonhosted.org/packages/92/32/2201fa49ba1f6c101ee321e83f051ac7a4b8d07b0ef6b4d3f2772b302275/ruff-0.16.3-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:9e0b1da805eb043654645d74d5de1e5ce2edc686e40790d2b86f56d71cc06a84", size = 10668771, upload-time = "2026-08-13T15:16:35.65Z" }, - { url = "https://files.pythonhosted.org/packages/c3/66/4afc5c8363bd04d45effce1b7c8713ca037d7a6740b7451a2403a6e3a972/ruff-0.16.3-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:a37bdea0bbe21780f590bf437d6412c8c4e1b6cd010f91a65c2c40c5e5f5f870", size = 10699568, upload-time = "2026-08-13T15:16:38.195Z" }, - { url = "https://files.pythonhosted.org/packages/53/fd/c67d246bf36bf1698551c56de39e95cd07f70e64433e0098e6267d77061b/ruff-0.16.3-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:09571e6d1288ed9be475207a3ac04ada404f1cd898104be0f6ab8d7df438575b", size = 11499365, upload-time = "2026-08-13T15:16:40.623Z" }, - { url = "https://files.pythonhosted.org/packages/67/0b/00ecbceb99a263af7b12f6f05ac3c92bc47b905e91adc3f207a836e3bc01/ruff-0.16.3-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:2c18c5a101eb540010638cc1ff3c84944d3adb3df62b8d98ca8f22ba484d3413", size = 12311728, upload-time = "2026-08-13T15:16:43.564Z" }, - { url = "https://files.pythonhosted.org/packages/54/b2/b7b3bb54f4d3f7db504e476ad4ab8de530dceebe2c061384b2757ee419e8/ruff-0.16.3-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:8457c44f15033c85ddbb77b15d451df9e24e4bd03b628396dd3610cedc3b8f82", size = 11699896, upload-time = "2026-08-13T15:16:46.209Z" }, - { url = "https://files.pythonhosted.org/packages/c7/30/4c468429ac195addc5ee1b717b6ab1b66632786737ca3b2ed3443fb0c26a/ruff-0.16.3-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:294b95c4ae0cda9388525c2047778aa758d6b8d4bb876fd4e9eaa3ebc92343eb", size = 11058736, upload-time = "2026-08-13T15:16:48.823Z" }, - { url = "https://files.pythonhosted.org/packages/43/67/7a113cdaddf24b64d7f75b1242a99d04c82fcef4f6921fdbb832beaffb5f/ruff-0.16.3-py3-none-manylinux_2_31_riscv64.whl", hash = "sha256:3d0c7c40c87c2a820509c31ba007968da6e1306468c067b2d82fbfdbcd0e8474", size = 11586911, upload-time = "2026-08-13T15:16:51.913Z" }, - { url = "https://files.pythonhosted.org/packages/f1/c1/2e66f24c0f3ead25a5e660111778685e505e5da353c82802bf49f0cbe7b9/ruff-0.16.3-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:9f738c0fdfa8eed0b2ce7fb27ee7258208a92a68d7949e62aa15164bc7b389da", size = 10954265, upload-time = "2026-08-13T15:16:54.763Z" }, - { url = "https://files.pythonhosted.org/packages/c2/ba/4cee23bf52cba9a058d3726de623624daf50ef9638868edd86f4126157f6/ruff-0.16.3-py3-none-musllinux_1_2_armv7l.whl", hash = "sha256:fb785f0be25abe69d320415cd4f833b59e17ba7613d9ba6a958023b6bceb0a50", size = 10709886, upload-time = "2026-08-13T15:16:57.339Z" }, - { url = "https://files.pythonhosted.org/packages/82/df/7da7194fa5d9dc0a285f7e6fa5a4722e7c63faac0b45b614ded9314363a1/ruff-0.16.3-py3-none-musllinux_1_2_i686.whl", hash = "sha256:c5536e3acfbf9563085aa2be7b13c629c3077e902afc5b941ac44024dbb9f506", size = 11210392, upload-time = "2026-08-13T15:17:00.171Z" }, - { url = "https://files.pythonhosted.org/packages/35/85/7795f6e817af050e7517bf3e7aa9b061cce70ef33d280aad902c956c1ecf/ruff-0.16.3-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:a2d85c02f9b8e165d85e6779184d38c4132de12603dab59c51c28e22584f9e4d", size = 11626910, upload-time = "2026-08-13T15:17:03.299Z" }, - { url = "https://files.pythonhosted.org/packages/78/9b/475b927cf27a5cbbda3c7bafb69ed6ff77e1d7923d5d85f17c2749d7ae32/ruff-0.16.3-py3-none-win32.whl", hash = "sha256:388cdf2166642bd9b13d52b5932d3170f34f8abed7e8d9a855f1d84b83645a0a", size = 10931415, upload-time = "2026-08-13T15:17:05.726Z" }, - { url = "https://files.pythonhosted.org/packages/b2/99/e2a2bfc4fbf0a1e8a916bc9ebe6fe6c58cc34c28e0ffc6ce281d572d1c2e/ruff-0.16.3-py3-none-win_amd64.whl", hash = "sha256:e80a7d69ca2a6d1c4d352ec91458cdca6e56c83cdbcabd93e4abe1e53591d948", size = 11445993, upload-time = "2026-08-13T15:17:08.353Z" }, - { url = "https://files.pythonhosted.org/packages/69/3e/4132e539aed78c148854d4997a2685b0ed4dc4e87110b59ce528564e184e/ruff-0.16.3-py3-none-win_arm64.whl", hash = "sha256:b8ca152da82c1acc1fa8d5874b15951935f0eef46f10e6954c83859011b6178a", size = 11399302, upload-time = "2026-08-13T15:17:10.908Z" }, + { url = "https://files.pythonhosted.org/packages/ff/80/779895ef584e089d22f2c6df0d0e99a65ec2df0805f1fffd439415b8c1f0/ruff-0.16.4-py3-none-linux_armv6l.whl", hash = "sha256:df4075f71ddac40b9934af60c3ec8a53047dd5a5fdc43224e6e4e8e9a27cb6f7", size = 10006909, upload-time = "2026-08-20T17:43:16.888Z" }, + { url = "https://files.pythonhosted.org/packages/a9/e6/f553199b5e8927a05cb5c422d921fd0656b29ab976e91c44802107c6b0da/ruff-0.16.4-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:0c95538517af68004306b0fb3214ff2f2af67a65092aee77cd9eb86db6656604", size = 10240201, upload-time = "2026-08-20T17:43:19.337Z" }, + { url = "https://files.pythonhosted.org/packages/1c/70/4a6dc4bb34da4dee35e30f09bbd1bfbdd26f33b62fb9b8df31f08a199cd2/ruff-0.16.4-py3-none-macosx_11_0_arm64.whl", hash = "sha256:963f83df8e69e575b64d67dd447ebbc917db41a14bf38d4593a4183e7aaa8255", size = 9835122, upload-time = "2026-08-20T17:43:21.708Z" }, + { url = "https://files.pythonhosted.org/packages/24/12/c6e22d686372c15bcb7af99831f1a1be96df696491babf4f24e4f942c527/ruff-0.16.4-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:32a5057c7ff3f6e6480a48fccfb3a412a690f48a3d03ac5cf08177d6c2da3ade", size = 9977162, upload-time = "2026-08-20T17:43:24.236Z" }, + { url = "https://files.pythonhosted.org/packages/46/49/72b10ec912f5ab5854992eaf7aa7cd36729b6937d9dc4e0fb41b3bf428ec/ruff-0.16.4-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:b3dce8d9b0c57c265b91885a66a567d8ea1372e8eb4e250fa8e5e3f579e99cff", size = 9829789, upload-time = "2026-08-20T17:43:26.966Z" }, + { url = "https://files.pythonhosted.org/packages/fa/80/0f30e32e7f6ee26edc39075502db9d368d788a44a79b55f763eb4ab03796/ruff-0.16.4-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:7dc651db49283c69f8e72c834eec4fe5573e4c646856aebece0ce385dceb2a80", size = 10527949, upload-time = "2026-08-20T17:43:29.384Z" }, + { url = "https://files.pythonhosted.org/packages/52/3d/86e8ad3542169e56cac3859a343afdb9df2ad54d35a59ce1e67baee83421/ruff-0.16.4-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:3817b87dbcabc92f13b05019257c5b89b5b4d51b5fb20f56fb5235ceb723cd07", size = 11333695, upload-time = "2026-08-20T17:43:31.872Z" }, + { url = "https://files.pythonhosted.org/packages/d0/16/481c29b380c20a0054a8261066665e1b3488e23636c49d0a43e75975b9bb/ruff-0.16.4-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:e9fce1499134b2c8c68e5166f95705a5812062bb93aacc5f9873bb1a27084bc7", size = 10727741, upload-time = "2026-08-20T17:43:34.596Z" }, + { url = "https://files.pythonhosted.org/packages/5e/b6/56bc0b8cf45b54b28b3a5e6381c8945d51b5b18adf659454c32295209a31/ruff-0.16.4-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:f2d812e482f5a7e02eee26cd73d2a37ebbdf47d795ea63ba1b89110ae93e9fb3", size = 10286522, upload-time = "2026-08-20T17:43:37.288Z" }, + { url = "https://files.pythonhosted.org/packages/e8/8b/b345b4fb110f2fbe2bd31eabd271e5e8b3b7e4ee6c0e02f2dc6be78db000/ruff-0.16.4-py3-none-manylinux_2_31_riscv64.whl", hash = "sha256:6baaf984aa7976edf93d3b627fe2d1d22ee94bbca05fa6f90fc76d73924e3454", size = 10584182, upload-time = "2026-08-20T17:43:39.984Z" }, + { url = "https://files.pythonhosted.org/packages/29/e5/827b34041c35f58774a9681a4213994c164fc987800f4dddabcf451da0bf/ruff-0.16.4-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:bdfcf0b28662eb890372d50f92c283bb94e67e7635ed93c7fd533970acff7b2b", size = 10134195, upload-time = "2026-08-20T17:43:42.351Z" }, + { url = "https://files.pythonhosted.org/packages/0f/10/d0bffcdd6729b87afc82ba0ef377173356a7dc8e972f5179968cf2fdf98c/ruff-0.16.4-py3-none-musllinux_1_2_armv7l.whl", hash = "sha256:b66b02cb9b04f537643cadf5768e5f98dc461890d530cb67113d71c8c76e605d", size = 9825821, upload-time = "2026-08-20T17:43:44.532Z" }, + { url = "https://files.pythonhosted.org/packages/f5/32/0db2a863b796ca62d83e92a07a3ccf00921b14db02059347576a2fda3d4b/ruff-0.16.4-py3-none-musllinux_1_2_i686.whl", hash = "sha256:8528bf9a4b291a60bf02ea453511e8ce6215bd2b982ee80405b66b008b6c30a0", size = 10267658, upload-time = "2026-08-20T17:43:46.989Z" }, + { url = "https://files.pythonhosted.org/packages/b2/a0/fbdeb59e48c6261f523e56c8f12e9c08fbe693786595cc7e3959207a9232/ruff-0.16.4-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:fbd85d2875fdd67e833213a651f613bbf25303abf6aa822a5121f4531195678d", size = 10697071, upload-time = "2026-08-20T17:43:49.891Z" }, + { url = "https://files.pythonhosted.org/packages/aa/28/0c6dd865859c6d17bc8ccc34cb72b0e02d6c7eb25e8a1e22b5bea681e2c0/ruff-0.16.4-py3-none-win32.whl", hash = "sha256:312769988007aaeb8e189b443ccdd03c0e6374489e053467be6d96518ebff76e", size = 10021687, upload-time = "2026-08-20T17:43:52.281Z" }, + { url = "https://files.pythonhosted.org/packages/a3/03/e724450f621698117f9aa6dd241c94d0274ae96781378dc86745ae29f0e7/ruff-0.16.4-py3-none-win_amd64.whl", hash = "sha256:05d9d27a18c4bcbefada602480ec9e01e0bc949d432e0ced5df77edac195919c", size = 10567657, upload-time = "2026-08-20T17:43:54.78Z" }, + { url = "https://files.pythonhosted.org/packages/0e/fe/da8b9e1347696bb22120b77280ec5ce25d500ca5cb39d5ad6e5c18de19c1/ruff-0.16.4-py3-none-win_arm64.whl", hash = "sha256:a3a61621c9b6f6a89573e938a080e648f1695baa3f58570a3a707bc51ff65a21", size = 10451579, upload-time = "2026-08-20T17:43:57.135Z" }, ] [[package]] diff --git a/gateways/kafka/Cargo.toml b/gateways/kafka/Cargo.toml index 648cd70385..4031a4d73a 100644 --- a/gateways/kafka/Cargo.toml +++ b/gateways/kafka/Cargo.toml @@ -37,7 +37,7 @@ bytes = { workspace = true } # Broker-role only: decodes requests and encodes responses. Default features also pull in # client-role codec paths and compression codecs (gzip/lz4/snappy/zstd) this gateway never # uses, since RecordBatch payloads stay opaque `Bytes` here. -kafka-protocol = { version = "0.17", default-features = false, features = ["broker"] } +kafka-protocol = { workspace = true } libc = { workspace = true } socket2 = { workspace = true } thiserror = { workspace = true } diff --git a/gateways/kafka/tools/kafka-tool/Cargo.toml b/gateways/kafka/tools/kafka-tool/Cargo.toml index 809704cb76..f9650df949 100644 --- a/gateways/kafka/tools/kafka-tool/Cargo.toml +++ b/gateways/kafka/tools/kafka-tool/Cargo.toml @@ -33,10 +33,10 @@ path = "src/main.rs" anyhow = { workspace = true } bytes = { workspace = true } clap = { workspace = true } -hex = "0.4" -iggy-gateway-kafka = { path = "../.." } +hex = { workspace = true } +iggy-gateway-kafka = { workspace = true } indexmap = { workspace = true } -kafka-protocol = "0.17" +kafka-protocol = { workspace = true, features = ["client"] } tokio = { workspace = true } tracing = { workspace = true } tracing-subscriber = { workspace = true } diff --git a/gateways/kafka/tools/kafka-tool/src/main.rs b/gateways/kafka/tools/kafka-tool/src/main.rs index d8ede17dd8..51421cbd6f 100644 --- a/gateways/kafka/tools/kafka-tool/src/main.rs +++ b/gateways/kafka/tools/kafka-tool/src/main.rs @@ -267,6 +267,7 @@ fn build_payload(api_key: i16, version: i16) -> Result { let rec = Record { transactional: false, control: false, + delete_horizon: false, partition_leader_epoch: 0, producer_id: -1, producer_epoch: -1, From e6ec869500c3b362bf7f4b85650b7fc6df9a5fa1 Mon Sep 17 00:00:00 2001 From: Grzegorz Koszyk <112548209+numinnex@users.noreply.github.com> Date: Wed, 2 Sep 2026 16:17:27 +0200 Subject: [PATCH 044/182] feat(helm): deploy a cluster with one release per node (#4035) --- helm/charts/iggy/Chart.yaml | 4 +- helm/charts/iggy/README.md | 211 ++++++++++++++-- helm/charts/iggy/README.md.gotmpl | 170 ++++++++++++- helm/charts/iggy/examples/cluster-3-node.yaml | 130 ++++++++++ helm/charts/iggy/templates/_helpers.tpl | 234 ++++++++++++++++++ helm/charts/iggy/templates/deployment.yaml | 75 +++++- helm/charts/iggy/templates/hpa.yaml | 61 ----- .../charts/iggy/templates/server-secrets.yaml | 44 ++++ helm/charts/iggy/templates/service.yaml | 6 +- helm/charts/iggy/values.yaml | 132 ++++++++-- scripts/ci/test-helm.sh | 72 +++++- 11 files changed, 1019 insertions(+), 120 deletions(-) create mode 100644 helm/charts/iggy/examples/cluster-3-node.yaml delete mode 100644 helm/charts/iggy/templates/hpa.yaml create mode 100644 helm/charts/iggy/templates/server-secrets.yaml diff --git a/helm/charts/iggy/Chart.yaml b/helm/charts/iggy/Chart.yaml index 08f86a8030..a521509243 100644 --- a/helm/charts/iggy/Chart.yaml +++ b/helm/charts/iggy/Chart.yaml @@ -20,8 +20,8 @@ apiVersion: v2 name: iggy description: A Helm chart for Apache Iggy server and web-ui type: application -version: 0.5.0 -appVersion: "0.7.0" +version: 0.6.0 +appVersion: "0.9.0-edge.6" sources: - https://github.com/apache/iggy keywords: diff --git a/helm/charts/iggy/README.md b/helm/charts/iggy/README.md index bfd38cb7ae..8026d99ab7 100644 --- a/helm/charts/iggy/README.md +++ b/helm/charts/iggy/README.md @@ -2,7 +2,7 @@ A Helm chart for Apache Iggy server and web-ui -![Version: 0.5.0](https://img.shields.io/badge/Version-0.5.0-informational?style=flat-square) ![Type: application](https://img.shields.io/badge/Type-application-informational?style=flat-square) ![AppVersion: 0.7.0](https://img.shields.io/badge/AppVersion-0.7.0-informational?style=flat-square) +![Version: 0.6.0](https://img.shields.io/badge/Version-0.6.0-informational?style=flat-square) ![Type: application](https://img.shields.io/badge/Type-application-informational?style=flat-square) ![AppVersion: 0.9.0-edge.6](https://img.shields.io/badge/AppVersion-0.9.0--edge.6-informational?style=flat-square) ## Prerequisites @@ -83,6 +83,156 @@ If Prometheus Operator is installed and you want monitoring, set `server.serviceMonitor.enabled=true` in `custom-values.yaml` or pass it on the command line with `--set server.serviceMonitor.enabled=true`. +## Cluster Mode + +The server runs a Viewstamped Replication cluster when `cluster.enabled` is set +and each node is told which roster entry it is. The chart models this as **one +release per node**: every release is handed the same roster and overrides only +its own identity. + +`helm/charts/iggy/examples/cluster-3-node.yaml` is a ready three-node roster. +Point the `ip` values at the nodes you will pin the releases to, then: + +```bash +for i in 0 1 2; do + helm upgrade --install "iggy-n$i" ./helm/charts/iggy \ + -f ./helm/charts/iggy/examples/cluster-3-node.yaml \ + --set "server.cluster.selfReplicaId=$i" \ + --set "server.nodeSelector.kubernetes\.io/hostname=node-$i" +done +``` + +The chart turns `server.cluster` into the server's `IGGY_CLUSTER_*` environment +variables and passes `--replica-id`, so no configuration file is mounted. The +values mirror the server's `[[cluster.nodes]]` config one for one. + +### Why hostNetwork and node pinning are required + +The server's consensus listener binds `cluster.nodes[*].ip` **verbatim**, unlike +the client listeners, which keep their own bind address and take only the port +from the roster. A pod therefore has to own the address its roster entry names. +Pod IPs are neither stable nor known before install, and a Service ClusterIP is +not an address a pod can bind, so the workable layout today is `hostNetwork: +true` with each release pinned to a node and its roster `ip` set to that node's +IP. Node IPs are known up front and survive a pod restart, which is what lets a +replica come back and rejoin. + +The cost is real: the server pod shares the node's network namespace and its +ports. Weigh it before running this in a shared cluster. + +If a release lands on a node whose IP is not the one in its roster entry, the +server refuses to start rather than misbehaving quietly: + +```text +Error: ShardJoinFailures { failures: [ShardJoinFailure { shard_id: 0, + kind: Error(Iggy(CannotBindToSocket("10.0.1.11:9090"))) }] } +``` + +Check the `nodeSelector` on that release against the `ip` in its roster entry. + +### Upgrades + +Host ports make a rolling update impossible: the replacement pod cannot bind +ports the outgoing pod still holds, so it stays `Pending` while the Deployment +waits for it to become ready. The chart therefore switches the server to the +`Recreate` strategy whenever `hostNetwork` is set, which takes the node down +for the length of the restart. Roll **one release at a time** and let the +cluster regain quorum before starting the next. + +Turning `hostNetwork` on for a release that already exists moves the Deployment +from `RollingUpdate` to `Recreate`. Under Helm 4's server-side apply that is +rejected, because the patch cannot drop the `rollingUpdate` block the API server +defaulted in: + +```text +Deployment.apps "iggy-n0" is invalid: spec.strategy.rollingUpdate: Forbidden: + may not be specified when strategy `type` is 'Recreate' +``` + +On Helm 4, run that one upgrade with `helm upgrade --server-side=false`, which +replaces the strategy in place. Helm 3 has no such flag, since it does not use +server-side apply. Set `server.strategy` explicitly if you want a different +strategy. + +### Roster rules + +* Every node runs the **identical** `server.cluster.nodes` list. Only + `selfReplicaId` differs between releases. +* `replicaId` values are unique and cover `0..N-1` for an `N`-node roster. +* `ports.tcpReplica` is required on every entry. In cluster mode the server + takes every listener port from the roster and will not fall back to defaults, + because two nodes on one host would otherwise race for the same socket. The + remaining ports default to `server.ports`. +* `server.cluster.name` is hashed into the on-disk cluster id on first boot. + Changing it later makes the server refuse to start against existing data. +* `ip` must be a literal IP. Hostnames are rejected. Use + `advertisedAddress` for the name clients dial, which does accept DNS. + +### Secrets + +Four settings have to carry the byte-identical value on every node, so they +belong in one Secret that all releases reference rather than in each release's +values. Create it in the release namespace before the first install: + +```bash +JWT="$(head -c 32 /dev/urandom | base64)" +kubectl create secret generic iggy-cluster-secrets \ + --from-literal=username=iggy \ + --from-literal=password="$(head -c 24 /dev/urandom | base64)" \ + --from-literal=clusterSharedSecret="$(head -c 32 /dev/urandom | base64)" \ + --from-literal=jwtEncodingSecret="$JWT" \ + --from-literal=jwtDecodingSecret="$JWT" \ + --from-literal=encryptionKey="$(head -c 32 /dev/urandom | base64)" +``` + +`examples/cluster-3-node.yaml` points `server.users.root`, `server.encryption`, +`server.jwt` and `server.cluster.auth` at that one Secret. All four also accept +an inline value, which the chart turns into a Secret of its own: +`-root-credentials` for the root username and password, +`-secrets` for the other three. That is convenient for a single node +and a poor fit for a cluster, since the value then lives in every release's +stored values. + +* **`server.users.root`** seeds the root user. Every node creates it locally + from its own `IGGY_ROOT_USERNAME` and `IGGY_ROOT_PASSWORD`, so a password that + differs on one release logs in on that node only, and a first cluster boot + refuses to start without both. +* **`server.cluster.auth`** makes every replica connection complete an + authenticated handshake or be rejected. The key is at least 32 bytes of + CSPRNG output. Turning it on or off is a coordinated restart of the whole + cluster: a node that authenticates cannot talk to one that does not. +* **`server.encryption`** encrypts message payloads and state commands at rest + with AES-256-GCM, under a 32-byte base64 key. A node holding a different key + cannot read what its peers wrote. +* **`server.jwt`** makes HTTP bearer tokens valid on every node and survive a + restart. With `cluster.auth` enabled the server already derives a cluster-wide + key from the replica PSK, so setting the JWT secrets is an alternative to that + rather than an addition; setting them anyway keeps tokens working if auth is + later turned off. + +None of the encryption, JWT and replica-auth values is written to the data +directory, and each is masked as `******` in the startup log. + +### Storage + +A replica cannot move between nodes without invalidating its roster entry, so +give each release storage that stays on its node, and size +`server.persistence` per node rather than for the cluster. A replica that starts +on an empty volume rejoins by state transfer, which is correct but re-reads the +whole dataset from its peers. `server.persistence.enabled: false` puts the +replica on `emptyDir` and forces exactly that on every pod replacement, so the +cluster example enables persistence and leaves `storageClass` for you to point +at a node-local provisioner. + +### What the chart refuses + +`server.replicaCount > 1` fails at render time. Scaling the server Deployment +produces N independent servers behind one Service, all writing the same PVC +subpath with no lock between them, which corrupts the data directory while +`helm --wait` still reports success. The chart ships no HorizontalPodAutoscaler +for the same reason. Cluster size is a roster decision, so add a node to +`server.cluster.nodes` and install another release instead. + ## Uninstallation ```bash @@ -148,7 +298,7 @@ If a previous local smoke install failed and left resources behind, reset the sm scripts/ci/test-helm.sh cleanup-smoke ``` -On Apple Silicon hosts, the released `apache/iggy:0.7.0` `arm64` image may still fail during the runtime smoke path in kind. If your Docker setup supports amd64 emulation well enough, you can try recreating the dedicated smoke cluster with: +On Apple Silicon hosts, the released `arm64` server image may still fail during the runtime smoke path in kind. If your Docker setup supports amd64 emulation well enough, you can try recreating the dedicated smoke cluster with: ```bash HELM_SMOKE_KIND_PLATFORM=linux/amd64 scripts/ci/setup-helm-smoke-cluster.sh @@ -219,12 +369,18 @@ Ensure the server binds to `0.0.0.0` instead of `127.0.0.1`. This is configured A wildcard bind says which interfaces accept connections, not where clients reach the pod, so the server also needs the address to publish in cluster -metadata. Server builds that carry the setting refuse to start without it; -older ones, including the `0.7.0` this chart pins by default, have no such -refusal and log the variable as unknown and ignored. Either way the chart -sets `IGGY_NODE_ADVERTISED_ADDRESS` to the in-cluster Service DNS name; -override it with `server.advertisedAddress` when clients arrive through a -LoadBalancer or an Ingress. +metadata. Server builds that carry the setting refuse to start without it. +`0.9.0-edge.6`, the image this chart pins, predates it: +that build logs `IGGY_NODE_ADVERTISED_ADDRESS` as an unknown variable and +publishes the bind address, so the chart's default stays inert until the pinned +image moves past it. Either way the chart sets `IGGY_NODE_ADVERTISED_ADDRESS` to +the in-cluster Service DNS name; override it with `server.advertisedAddress` +when clients arrive through a LoadBalancer or an Ingress. + +In cluster mode the server ignores the variable altogether and publishes each +node's roster `ip`, or its `advertisedAddress` when the entry carries one, so +the chart leaves the variable out there and refuses a render that sets +`server.advertisedAddress` alongside `server.cluster.enabled`. Declaring `IGGY_NODE_ADVERTISED_ADDRESS` in `server.env` yourself works too: the chart then leaves its own default out, so the variable is declared once. @@ -325,10 +481,6 @@ pre-commit install | Key | Type | Default | Description | |-----|------|---------|-------------| | additionalLabels | object | `{}` | Additional labels for all resources | -| autoscaling.enabled | bool | `false` | Enable horizontal pod autoscaling | -| autoscaling.maxReplicas | int | `100` | Maximum replicas for autoscaling | -| autoscaling.minReplicas | int | `1` | Minimum replicas for autoscaling | -| autoscaling.targetCPUUtilizationPercentage | int | `80` | Target CPU utilization for autoscaling | | fullnameOverride | string | `""` | Override full release name | | imagePullSecrets | list | `[]` | Image pull secrets for private registries | | nameOverride | string | `""` | Override chart name | @@ -336,19 +488,44 @@ pre-commit install | podSecurityContext | object | `{"seccompProfile":{"type":"Unconfined"}}` | Pod security context (server uses io_uring, requires unconfined seccomp) | | resources | object | `{}` | Resource limits and requests for server | | securityContext | object | `{"capabilities":{"add":["IPC_LOCK"]}}` | Container security context (server requires IPC_LOCK for io_uring) | -| server | object | `{"advertisedAddress":"","affinity":{},"enabled":true,"env":[{"name":"RUST_LOG","value":"info"},{"name":"IGGY_HTTP_ADDRESS","value":"0.0.0.0:3000"},{"name":"IGGY_TCP_ADDRESS","value":"0.0.0.0:8090"},{"name":"IGGY_QUIC_ADDRESS","value":"0.0.0.0:8080"},{"name":"IGGY_WEBSOCKET_ADDRESS","value":"0.0.0.0:8092"}],"image":{"pullPolicy":"Always","repository":"apache/iggy","tag":"0.7.0"},"ingress":{"annotations":{},"className":"","enabled":false,"hosts":[{"host":"chart-example.local","paths":[{"path":"/","pathType":"ImplementationSpecific"}]}],"tls":[]},"nodeSelector":{},"persistence":{"accessMode":"ReadWriteOnce","annotations":{},"enabled":false,"existingClaim":"","size":"8Gi","storageClass":""},"ports":{"http":3000,"quic":8080,"tcp":8090},"replicaCount":1,"service":{"port":3000,"type":"ClusterIP"},"serviceMonitor":{"additionalLabels":{},"authorization":{},"enabled":false,"honorLabels":false,"interval":"30s","namespace":"","path":"/metrics","scrapeTimeout":"10s"},"tolerations":[],"users":{"root":{"createSecret":true,"existingSecret":{"name":"","passwordKey":"password","usernameKey":"username"},"password":"changeit","username":"iggy"}}}` | Iggy server configuration | -| server.advertisedAddress | string | `""` | Client-facing address published in cluster metadata. Declaring `IGGY_NODE_ADVERTISED_ADDRESS` in `server.env` instead also works, but setting both is refused at render time. Empty falls back to the in-cluster Service DNS name. | +| server | object | `{"advertisedAddress":"","affinity":{},"cluster":{"auth":{"enabled":false,"existingSecret":{"name":"","previousSharedSecretKey":"clusterPreviousSharedSecret","sharedSecretKey":"clusterSharedSecret"},"previousSharedSecret":"","sharedSecret":""},"enabled":false,"name":"iggy-cluster","nodes":[],"requireHostNetwork":true,"selfReplicaId":0},"enabled":true,"encryption":{"enabled":false,"existingSecret":{"key":"encryptionKey","name":""},"key":""},"env":[{"name":"RUST_LOG","value":"info"},{"name":"IGGY_HTTP_ADDRESS","value":"0.0.0.0:3000"},{"name":"IGGY_TCP_ADDRESS","value":"0.0.0.0:8090"},{"name":"IGGY_QUIC_ADDRESS","value":"0.0.0.0:8080"},{"name":"IGGY_WEBSOCKET_ADDRESS","value":"0.0.0.0:8092"}],"extraArgs":[],"hostNetwork":false,"image":{"pullPolicy":"Always","repository":"apache/iggy","tag":""},"ingress":{"annotations":{},"className":"","enabled":false,"hosts":[{"host":"chart-example.local","paths":[{"path":"/","pathType":"ImplementationSpecific"}]}],"tls":[]},"jwt":{"decodingSecret":"","encodingSecret":"","existingSecret":{"decodingSecretKey":"jwtDecodingSecret","encodingSecretKey":"jwtEncodingSecret","name":""}},"nodeSelector":{},"persistence":{"accessMode":"ReadWriteOnce","annotations":{},"enabled":false,"existingClaim":"","size":"8Gi","storageClass":""},"ports":{"http":3000,"quic":8080,"tcp":8090,"websocket":8092},"replicaCount":1,"service":{"port":3000,"type":"ClusterIP"},"serviceMonitor":{"additionalLabels":{},"authorization":{},"enabled":false,"honorLabels":false,"interval":"30s","namespace":"","path":"/metrics","scrapeTimeout":"10s"},"strategy":{},"tolerations":[],"users":{"root":{"createSecret":true,"existingSecret":{"name":"","passwordKey":"password","usernameKey":"username"},"password":"changeit","username":"iggy"}}}` | Iggy server configuration | +| server.advertisedAddress | string | `""` | Client-facing address published in cluster metadata. Declaring `IGGY_NODE_ADVERTISED_ADDRESS` in `server.env` instead also works, but setting both is refused at render time. Empty falls back to the in-cluster Service DNS name. Ignored in cluster mode, where the address comes from the node's roster entry, so setting both is refused there too. | | server.affinity | object | `{}` | Affinity rules for server pods | +| server.cluster.auth | object | `{"enabled":false,"existingSecret":{"name":"","previousSharedSecretKey":"clusterPreviousSharedSecret","sharedSecretKey":"clusterSharedSecret"},"previousSharedSecret":"","sharedSecret":""}` | Replica-to-replica authentication on the consensus port. When enabled every peer must complete an authenticated handshake or be rejected, and a shared secret becomes mandatory. Enabling it on a running cluster is a coordinated-restart change, not a rolling one. | +| server.cluster.auth.enabled | bool | `false` | Require the authenticated replica handshake | +| server.cluster.auth.existingSecret.name | string | `""` | Name of an existing Secret holding the pre-shared keys | +| server.cluster.auth.existingSecret.previousSharedSecretKey | string | `"clusterPreviousSharedSecret"` | Key inside that Secret holding the retiring shared secret | +| server.cluster.auth.existingSecret.sharedSecretKey | string | `"clusterSharedSecret"` | Key inside that Secret holding the active shared secret | +| server.cluster.auth.previousSharedSecret | string | `""` | Retiring key, accepted for verification only while a rotation is in flight. Leave empty outside a rotation. | +| server.cluster.auth.sharedSecret | string | `""` | Cluster-wide pre-shared key, at least 32 bytes of CSPRNG output, byte-identical on every node. Ignored when `existingSecret.name` is set. | +| server.cluster.enabled | bool | `false` | Enable cluster (VSR consensus) mode. One Helm release per node: every release shares the same `nodes` roster and overrides only `selfReplicaId`. See the Cluster Mode section of the chart README. | +| server.cluster.name | string | `"iggy-cluster"` | Cluster name, byte-identical on every node. Hashed into the on-disk cluster id on first boot, so changing it later means starting from an empty data directory. | +| server.cluster.nodes | list | `[]` | Cluster roster, mirroring the server's `[[cluster.nodes]]` config. Every node runs the identical list. `ip` is the replica-plane address: this node's consensus listener binds it verbatim and every peer dials it verbatim, so it must be a literal IP that the pod itself owns. With `server.hostNetwork` that is the node IP. `ports.tcpReplica` is required on every entry; the remaining ports default to `server.ports`. | +| server.cluster.requireHostNetwork | bool | `true` | Refuse to render a cluster node without `server.hostNetwork`. The replica listener binds the roster `ip` verbatim, which no pod owns on the cluster network, so the pod would die at boot with `CannotBindToSocket`. Set this to false only when the roster `ip` is an address the pod itself holds. | +| server.cluster.selfReplicaId | int | `0` | Which `nodes` entry this release runs, matched against `replicaId`. | | server.enabled | bool | `true` | Enable the Iggy server deployment | +| server.encryption | object | `{"enabled":false,"existingSecret":{"key":"encryptionKey","name":""},"key":""}` | Server-side encryption of message payloads and state commands, using AES-256-GCM. Every node of a cluster must hold the identical key, or it cannot read data another node wrote. | +| server.encryption.enabled | bool | `false` | Enable encryption at rest | +| server.encryption.existingSecret.key | string | `"encryptionKey"` | Key inside that Secret | +| server.encryption.existingSecret.name | string | `""` | Name of an existing Secret holding the encryption key | +| server.encryption.key | string | `""` | 32-byte key, base64 encoded. Ignored when `existingSecret.name` is set. Prefer `existingSecret` outside development: a value here is stored in the Helm release and readable by anyone who can read it. | | server.env | list | `[{"name":"RUST_LOG","value":"info"},{"name":"IGGY_HTTP_ADDRESS","value":"0.0.0.0:3000"},{"name":"IGGY_TCP_ADDRESS","value":"0.0.0.0:8090"},{"name":"IGGY_QUIC_ADDRESS","value":"0.0.0.0:8080"},{"name":"IGGY_WEBSOCKET_ADDRESS","value":"0.0.0.0:8092"}]` | Environment variables for the server container | +| server.extraArgs | list | `[]` | Extra command-line arguments appended to the server entrypoint, e.g. `["--with-default-root-credentials"]` for a throwaway development install. `--replica-id` is not one of them: the chart passes it already whenever `cluster.enabled` is set. | +| server.hostNetwork | bool | `false` | Run the server pod in the host network namespace. Required for cluster mode, where the replica listener binds the roster IP verbatim. | | server.image.pullPolicy | string | `"Always"` | Image pull policy | | server.image.repository | string | `"apache/iggy"` | Server image repository | -| server.image.tag | string | `"0.7.0"` | Server image tag (overrides chart appVersion) | +| server.image.tag | string | `""` | Server image tag. Empty uses the chart appVersion. | | server.ingress.annotations | object | `{}` | Ingress annotations (controller-specific) | | server.ingress.className | string | `""` | Ingress class name (controller-neutral) | | server.ingress.enabled | bool | `false` | Enable ingress for the server | | server.ingress.hosts | list | `[{"host":"chart-example.local","paths":[{"path":"/","pathType":"ImplementationSpecific"}]}]` | Ingress hosts configuration | | server.ingress.tls | list | `[]` | Ingress TLS configuration | +| server.jwt | object | `{"decodingSecret":"","encodingSecret":"","existingSecret":{"decodingSecretKey":"jwtDecodingSecret","encodingSecretKey":"jwtEncodingSecret","name":""}}` | Secrets used to sign and validate HTTP bearer tokens. Left unset, each node generates a random secret on every start, which invalidates tokens across restarts and keeps them node-local. Setting the identical secret on every node makes bearers valid cluster-wide and activates follower to primary HTTP forwarding; `cluster.auth.enabled` derives the same thing from the replica PSK, so it is an alternative rather than an addition. | +| server.jwt.decodingSecret | string | `""` | Decoding secret. Ignored when `existingSecret.name` is set. | +| server.jwt.encodingSecret | string | `""` | Encoding secret. Ignored when `existingSecret.name` is set. | +| server.jwt.existingSecret.decodingSecretKey | string | `"jwtDecodingSecret"` | Key inside that Secret holding the decoding secret | +| server.jwt.existingSecret.encodingSecretKey | string | `"jwtEncodingSecret"` | Key inside that Secret holding the encoding secret | +| server.jwt.existingSecret.name | string | `""` | Name of an existing Secret holding the JWT secrets | | server.nodeSelector | object | `{}` | Node selector for server pods | | server.persistence.accessMode | string | `"ReadWriteOnce"` | PVC access mode | | server.persistence.annotations | object | `{}` | PVC annotations | @@ -357,8 +534,9 @@ pre-commit install | server.persistence.size | string | `"8Gi"` | PVC storage size | | server.persistence.storageClass | string | `""` | Storage class for PVC (empty uses default provisioner) | | server.ports.http | int | `3000` | HTTP API port | -| server.ports.quic | int | `8080` | QUIC protocol port | +| server.ports.quic | int | `8080` | QUIC protocol port (UDP) | | server.ports.tcp | int | `8090` | TCP protocol port | +| server.ports.websocket | int | `8092` | WebSocket protocol port | | server.replicaCount | int | `1` | Number of server replicas | | server.service.port | int | `3000` | Service port for the server | | server.service.type | string | `"ClusterIP"` | Service type for the server | @@ -370,6 +548,7 @@ pre-commit install | server.serviceMonitor.namespace | string | `""` | Namespace to deploy the ServiceMonitor | | server.serviceMonitor.path | string | `"/metrics"` | Path to scrape metrics from | | server.serviceMonitor.scrapeTimeout | string | `"10s"` | Timeout for scrape metrics request | +| server.strategy | object | `{}` | Deployment update strategy. Empty lets the chart choose: `Recreate` when `hostNetwork` is set, because a rolling update would wait forever for a replacement pod that cannot bind host ports the outgoing pod still holds, and the Kubernetes default otherwise. | | server.tolerations | list | `[]` | Tolerations for server pods | | server.users.root.createSecret | bool | `true` | Create a secret for the root user credentials | | server.users.root.existingSecret.name | string | `""` | Name of existing secret for root credentials | diff --git a/helm/charts/iggy/README.md.gotmpl b/helm/charts/iggy/README.md.gotmpl index a36d30adcd..ec1fb50d1f 100644 --- a/helm/charts/iggy/README.md.gotmpl +++ b/helm/charts/iggy/README.md.gotmpl @@ -101,6 +101,156 @@ If Prometheus Operator is installed and you want monitoring, set `server.serviceMonitor.enabled=true` in `custom-values.yaml` or pass it on the command line with `--set server.serviceMonitor.enabled=true`. +## Cluster Mode + +The server runs a Viewstamped Replication cluster when `cluster.enabled` is set +and each node is told which roster entry it is. The chart models this as **one +release per node**: every release is handed the same roster and overrides only +its own identity. + +`helm/charts/iggy/examples/cluster-3-node.yaml` is a ready three-node roster. +Point the `ip` values at the nodes you will pin the releases to, then: + +```bash +for i in 0 1 2; do + helm upgrade --install "iggy-n$i" ./helm/charts/iggy \ + -f ./helm/charts/iggy/examples/cluster-3-node.yaml \ + --set "server.cluster.selfReplicaId=$i" \ + --set "server.nodeSelector.kubernetes\.io/hostname=node-$i" +done +``` + +The chart turns `server.cluster` into the server's `IGGY_CLUSTER_*` environment +variables and passes `--replica-id`, so no configuration file is mounted. The +values mirror the server's `[[cluster.nodes]]` config one for one. + +### Why hostNetwork and node pinning are required + +The server's consensus listener binds `cluster.nodes[*].ip` **verbatim**, unlike +the client listeners, which keep their own bind address and take only the port +from the roster. A pod therefore has to own the address its roster entry names. +Pod IPs are neither stable nor known before install, and a Service ClusterIP is +not an address a pod can bind, so the workable layout today is `hostNetwork: +true` with each release pinned to a node and its roster `ip` set to that node's +IP. Node IPs are known up front and survive a pod restart, which is what lets a +replica come back and rejoin. + +The cost is real: the server pod shares the node's network namespace and its +ports. Weigh it before running this in a shared cluster. + +If a release lands on a node whose IP is not the one in its roster entry, the +server refuses to start rather than misbehaving quietly: + +```text +Error: ShardJoinFailures { failures: [ShardJoinFailure { shard_id: 0, + kind: Error(Iggy(CannotBindToSocket("10.0.1.11:9090"))) }] } +``` + +Check the `nodeSelector` on that release against the `ip` in its roster entry. + +### Upgrades + +Host ports make a rolling update impossible: the replacement pod cannot bind +ports the outgoing pod still holds, so it stays `Pending` while the Deployment +waits for it to become ready. The chart therefore switches the server to the +`Recreate` strategy whenever `hostNetwork` is set, which takes the node down +for the length of the restart. Roll **one release at a time** and let the +cluster regain quorum before starting the next. + +Turning `hostNetwork` on for a release that already exists moves the Deployment +from `RollingUpdate` to `Recreate`. Under Helm 4's server-side apply that is +rejected, because the patch cannot drop the `rollingUpdate` block the API server +defaulted in: + +```text +Deployment.apps "iggy-n0" is invalid: spec.strategy.rollingUpdate: Forbidden: + may not be specified when strategy `type` is 'Recreate' +``` + +On Helm 4, run that one upgrade with `helm upgrade --server-side=false`, which +replaces the strategy in place. Helm 3 has no such flag, since it does not use +server-side apply. Set `server.strategy` explicitly if you want a different +strategy. + +### Roster rules + +* Every node runs the **identical** `server.cluster.nodes` list. Only + `selfReplicaId` differs between releases. +* `replicaId` values are unique and cover `0..N-1` for an `N`-node roster. +* `ports.tcpReplica` is required on every entry. In cluster mode the server + takes every listener port from the roster and will not fall back to defaults, + because two nodes on one host would otherwise race for the same socket. The + remaining ports default to `server.ports`. +* `server.cluster.name` is hashed into the on-disk cluster id on first boot. + Changing it later makes the server refuse to start against existing data. +* `ip` must be a literal IP. Hostnames are rejected. Use + `advertisedAddress` for the name clients dial, which does accept DNS. + +### Secrets + +Four settings have to carry the byte-identical value on every node, so they +belong in one Secret that all releases reference rather than in each release's +values. Create it in the release namespace before the first install: + +```bash +JWT="$(head -c 32 /dev/urandom | base64)" +kubectl create secret generic iggy-cluster-secrets \ + --from-literal=username=iggy \ + --from-literal=password="$(head -c 24 /dev/urandom | base64)" \ + --from-literal=clusterSharedSecret="$(head -c 32 /dev/urandom | base64)" \ + --from-literal=jwtEncodingSecret="$JWT" \ + --from-literal=jwtDecodingSecret="$JWT" \ + --from-literal=encryptionKey="$(head -c 32 /dev/urandom | base64)" +``` + +`examples/cluster-3-node.yaml` points `server.users.root`, `server.encryption`, +`server.jwt` and `server.cluster.auth` at that one Secret. All four also accept +an inline value, which the chart turns into a Secret of its own: +`-root-credentials` for the root username and password, +`-secrets` for the other three. That is convenient for a single node +and a poor fit for a cluster, since the value then lives in every release's +stored values. + +* **`server.users.root`** seeds the root user. Every node creates it locally + from its own `IGGY_ROOT_USERNAME` and `IGGY_ROOT_PASSWORD`, so a password that + differs on one release logs in on that node only, and a first cluster boot + refuses to start without both. +* **`server.cluster.auth`** makes every replica connection complete an + authenticated handshake or be rejected. The key is at least 32 bytes of + CSPRNG output. Turning it on or off is a coordinated restart of the whole + cluster: a node that authenticates cannot talk to one that does not. +* **`server.encryption`** encrypts message payloads and state commands at rest + with AES-256-GCM, under a 32-byte base64 key. A node holding a different key + cannot read what its peers wrote. +* **`server.jwt`** makes HTTP bearer tokens valid on every node and survive a + restart. With `cluster.auth` enabled the server already derives a cluster-wide + key from the replica PSK, so setting the JWT secrets is an alternative to that + rather than an addition; setting them anyway keeps tokens working if auth is + later turned off. + +None of the encryption, JWT and replica-auth values is written to the data +directory, and each is masked as `******` in the startup log. + +### Storage + +A replica cannot move between nodes without invalidating its roster entry, so +give each release storage that stays on its node, and size +`server.persistence` per node rather than for the cluster. A replica that starts +on an empty volume rejoins by state transfer, which is correct but re-reads the +whole dataset from its peers. `server.persistence.enabled: false` puts the +replica on `emptyDir` and forces exactly that on every pod replacement, so the +cluster example enables persistence and leaves `storageClass` for you to point +at a node-local provisioner. + +### What the chart refuses + +`server.replicaCount > 1` fails at render time. Scaling the server Deployment +produces N independent servers behind one Service, all writing the same PVC +subpath with no lock between them, which corrupts the data directory while +`helm --wait` still reports success. The chart ships no HorizontalPodAutoscaler +for the same reason. Cluster size is a roster decision, so add a node to +`server.cluster.nodes` and install another release instead. + ## Uninstallation ```bash @@ -166,7 +316,7 @@ If a previous local smoke install failed and left resources behind, reset the sm scripts/ci/test-helm.sh cleanup-smoke ``` -On Apple Silicon hosts, the released `apache/iggy:0.7.0` `arm64` image may still fail during the runtime smoke path in kind. If your Docker setup supports amd64 emulation well enough, you can try recreating the dedicated smoke cluster with: +On Apple Silicon hosts, the released `arm64` server image may still fail during the runtime smoke path in kind. If your Docker setup supports amd64 emulation well enough, you can try recreating the dedicated smoke cluster with: ```bash HELM_SMOKE_KIND_PLATFORM=linux/amd64 scripts/ci/setup-helm-smoke-cluster.sh @@ -237,12 +387,18 @@ Ensure the server binds to `0.0.0.0` instead of `127.0.0.1`. This is configured A wildcard bind says which interfaces accept connections, not where clients reach the pod, so the server also needs the address to publish in cluster -metadata. Server builds that carry the setting refuse to start without it; -older ones, including the `0.7.0` this chart pins by default, have no such -refusal and log the variable as unknown and ignored. Either way the chart -sets `IGGY_NODE_ADVERTISED_ADDRESS` to the in-cluster Service DNS name; -override it with `server.advertisedAddress` when clients arrive through a -LoadBalancer or an Ingress. +metadata. Server builds that carry the setting refuse to start without it. +`{{ template "chart.appVersion" . }}`, the image this chart pins, predates it: +that build logs `IGGY_NODE_ADVERTISED_ADDRESS` as an unknown variable and +publishes the bind address, so the chart's default stays inert until the pinned +image moves past it. Either way the chart sets `IGGY_NODE_ADVERTISED_ADDRESS` to +the in-cluster Service DNS name; override it with `server.advertisedAddress` +when clients arrive through a LoadBalancer or an Ingress. + +In cluster mode the server ignores the variable altogether and publishes each +node's roster `ip`, or its `advertisedAddress` when the entry carries one, so +the chart leaves the variable out there and refuses a render that sets +`server.advertisedAddress` alongside `server.cluster.enabled`. Declaring `IGGY_NODE_ADVERTISED_ADDRESS` in `server.env` yourself works too: the chart then leaves its own default out, so the variable is declared once. diff --git a/helm/charts/iggy/examples/cluster-3-node.yaml b/helm/charts/iggy/examples/cluster-3-node.yaml new file mode 100644 index 0000000000..96a9329ede --- /dev/null +++ b/helm/charts/iggy/examples/cluster-3-node.yaml @@ -0,0 +1,130 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# Shared roster for a three-node Iggy cluster. Install one release per node, +# overriding only the identity: +# +# for i in 0 1 2; do +# helm upgrade --install "iggy-n$i" ./helm/charts/iggy \ +# -f ./helm/charts/iggy/examples/cluster-3-node.yaml \ +# --set "server.cluster.selfReplicaId=$i" \ +# --set "server.nodeSelector.kubernetes\.io/hostname=node-$i" +# done +# +# Replace the `ip` values with the IPs of the nodes you pin each release to. +# The server binds this address verbatim for replica traffic, so with +# hostNetwork it must be the node's own IP. +# +# Every secret below has to be byte-identical on every node, so they live in one +# Secret that all three releases reference rather than in this file. Create it +# once, in the release namespace, before the first install: +# +# JWT="$(head -c 32 /dev/urandom | base64)" +# kubectl create secret generic iggy-cluster-secrets \ +# --from-literal=username=iggy \ +# --from-literal=password="$(head -c 24 /dev/urandom | base64)" \ +# --from-literal=clusterSharedSecret="$(head -c 32 /dev/urandom | base64)" \ +# --from-literal=jwtEncodingSecret="$JWT" \ +# --from-literal=jwtDecodingSecret="$JWT" \ +# --from-literal=encryptionKey="$(head -c 32 /dev/urandom | base64)" +# +# The JWT encoding and decoding secrets are one HMAC key, which is why both +# entries carry the same value. Two different values make every token from +# /users/login fail with 401, and the server only warns about the mismatch. + +ui: + enabled: false + +server: + hostNetwork: true + + # The root user is created locally on each node out of that node's own + # IGGY_ROOT_* environment, so the credentials have to match across the + # releases the same way the keys below do. + users: + root: + createSecret: false + existingSecret: + name: iggy-cluster-secrets + + # A release is pinned to one node, so its storage has to stay on that node. + # On emptyDir every pod replacement wipes the replica, which then refetches + # the whole dataset from its peers. + persistence: + enabled: true + # Name a node-local class here (local-path, topolvm, a local volume + # provisioner). Empty takes the default provisioner, which may hand out + # storage that does not stay on the pinned node. + storageClass: "" + size: 8Gi + + # Encrypts message payloads and state commands at rest with AES-256-GCM. The + # key is cluster-wide: a node that holds a different one cannot read what its + # peers wrote. + encryption: + enabled: true + existingSecret: + name: iggy-cluster-secrets + + # Makes bearer tokens valid on every node and survive a restart. Redundant + # while cluster.auth is enabled, which derives the same key from the replica + # PSK, but set explicitly here so tokens keep working if auth is turned off. + jwt: + existingSecret: + name: iggy-cluster-secrets + + cluster: + enabled: true + name: iggy-cluster + + # Every replica connection completes an authenticated handshake or is + # rejected. Turning this on or off is a coordinated restart of the whole + # cluster, not a rolling one. + auth: + enabled: true + existingSecret: + name: iggy-cluster-secrets + nodes: + - name: iggy-node-0 + ip: 10.0.1.10 + replicaId: 0 + ports: + tcpReplica: 9090 + - name: iggy-node-1 + ip: 10.0.1.11 + replicaId: 1 + ports: + tcpReplica: 9090 + - name: iggy-node-2 + ip: 10.0.1.12 + replicaId: 2 + ports: + tcpReplica: 9090 + + env: + - name: RUST_LOG + value: info + # Bind interface only: in cluster mode every listener port comes from the + # roster entry for this node. + - name: IGGY_HTTP_ADDRESS + value: "0.0.0.0:3000" + - name: IGGY_TCP_ADDRESS + value: "0.0.0.0:8090" + - name: IGGY_QUIC_ADDRESS + value: "0.0.0.0:8080" + - name: IGGY_WEBSOCKET_ADDRESS + value: "0.0.0.0:8092" diff --git a/helm/charts/iggy/templates/_helpers.tpl b/helm/charts/iggy/templates/_helpers.tpl index 92f69780e0..f6943c8c4a 100644 --- a/helm/charts/iggy/templates/_helpers.tpl +++ b/helm/charts/iggy/templates/_helpers.tpl @@ -97,3 +97,237 @@ Create the name of the service account to use {{- default "default" .Values.serviceAccount.name }} {{- end }} {{- end }} + +{{/* +Validate the cluster roster and fail the render with an actionable message. +Every check here is one the server would otherwise only reject at boot, after +the pod is already scheduled. +*/}} +{{- define "iggy.validateCluster" -}} + {{- $cluster := .Values.server.cluster }} + {{- if not $cluster.nodes }} + {{- fail "server.cluster.enabled is true but server.cluster.nodes is empty. Every node runs the identical roster; see the Cluster Mode section of the chart README." }} + {{- end }} + {{- if and $cluster.requireHostNetwork (not .Values.server.hostNetwork) }} + {{- fail "server.cluster.enabled is true but server.hostNetwork is false. The replica listener binds the roster ip verbatim, which is not an address a pod owns on the cluster network, so the pod dies at boot with CannotBindToSocket. Set server.hostNetwork to true, or server.cluster.requireHostNetwork to false if the roster ip really is one this pod owns." }} + {{- end }} + {{- $root := .Values.server.users.root }} + {{- if and (not $root.createSecret) (not $root.existingSecret.name) }} + {{- fail "server.users.root has neither createSecret nor existingSecret.name, so the container receives no IGGY_ROOT_USERNAME or IGGY_ROOT_PASSWORD. A first cluster boot refuses to start without both, because every node creates root locally and the credentials have to come out identical on all of them." }} + {{- end }} + {{- $count := len $cluster.nodes }} + {{- $seen := dict }} + {{- range $cluster.nodes }} + {{- if not .name }} + {{- fail "every server.cluster.nodes entry needs a name" }} + {{- end }} + {{- if not .ip }} + {{- fail (printf "server.cluster.nodes entry %q has no ip. The replica listener binds this address verbatim, so it must be a literal IP the pod owns." .name) }} + {{- end }} + {{- $ip := toString .ip }} + {{- if not (regexMatch "^[0-9a-fA-F.:]+$" $ip) }} + {{- fail (printf "server.cluster.nodes entry %q has ip %q, which the server parses as an IP address and rejects. Use server.cluster.nodes[*].advertisedAddress for the hostname clients dial." .name $ip) }} + {{- end }} + {{- if or (eq $ip "0.0.0.0") (eq $ip "::") }} + {{- fail (printf "server.cluster.nodes entry %q has the wildcard ip %q. Every peer dials this address verbatim, so it has to name one interface." .name $ip) }} + {{- end }} + {{- if kindIs "invalid" .replicaId }} + {{- fail (printf "server.cluster.nodes entry %q has no replicaId" .name) }} + {{- end }} + {{- $id := int .replicaId }} + {{- if or (lt $id 0) (ge $id $count) }} + {{- fail (printf "server.cluster.nodes entry %q has replicaId %d, which is outside 0..%d for a %d-node roster" .name $id (sub $count 1) $count) }} + {{- end }} + {{- if hasKey $seen (printf "%d" $id) }} + {{- fail (printf "server.cluster.nodes has two entries with replicaId %d; ids must be unique" $id) }} + {{- end }} + {{- $_ := set $seen (printf "%d" $id) .name }} + {{- if not (and .ports .ports.tcpReplica) }} + {{- fail (printf "server.cluster.nodes entry %q has no ports.tcpReplica. In cluster mode the server takes every listener port from the roster and refuses to start without it." .name) }} + {{- end }} + {{- end }} + {{- if not (hasKey $seen (printf "%d" (int $cluster.selfReplicaId))) }} + {{- fail (printf "server.cluster.selfReplicaId is %d but no server.cluster.nodes entry declares that replicaId. Each release picks its own identity out of the shared roster." (int $cluster.selfReplicaId)) }} + {{- end }} +{{- end }} + +{{/* +The ports this release's server binds, as a JSON object. In cluster mode the +server takes every listener port from its own roster entry and never falls back +to the top-level ones, so a node whose entry names other ports has to reach the +container ports, the probes and the Service targets as well. `server.ports` +supplies the per-field default there and the whole answer outside cluster mode. +`tcpReplica` has no top-level default because the roster owns it: it is +mandatory on every entry and there is no replica listener outside cluster mode. +*/}} +{{- define "iggy.serverPorts" -}} + {{- $ports := .Values.server.ports }} + {{- $resolved := dict "http" $ports.http "quic" $ports.quic "tcp" $ports.tcp "websocket" $ports.websocket }} + {{- if .Values.server.cluster.enabled }} + {{- $selfReplicaId := int .Values.server.cluster.selfReplicaId }} + {{- range .Values.server.cluster.nodes }} + {{- if eq (int .replicaId) $selfReplicaId }} + {{- range $name, $port := (default (dict) .ports) }} + {{- if $port }} + {{- $_ := set $resolved $name $port }} + {{- end }} + {{- end }} + {{- end }} + {{- end }} + {{- end }} + {{- $resolved | toJson }} +{{- end }} + +{{/* +Render the roster as IGGY_CLUSTER_* environment variables. The server accepts +the whole cluster config this way, so the chart needs no config file mount. +*/}} +{{- define "iggy.clusterEnv" -}} + {{- $ports := .Values.server.ports }} +- name: IGGY_CLUSTER_ENABLED + value: "true" +- name: IGGY_CLUSTER_NAME + value: {{ .Values.server.cluster.name | quote }} + {{- range .Values.server.cluster.nodes }} + {{- $id := int .replicaId }} +- name: IGGY_CLUSTER_NODES_{{ $id }}_NAME + value: {{ .name | quote }} +- name: IGGY_CLUSTER_NODES_{{ $id }}_IP + value: {{ .ip | quote }} +- name: IGGY_CLUSTER_NODES_{{ $id }}_REPLICA_ID + value: {{ $id | quote }} + {{- if .advertisedAddress }} +- name: IGGY_CLUSTER_NODES_{{ $id }}_ADVERTISED_ADDRESS + value: {{ .advertisedAddress | quote }} + {{- end }} +- name: IGGY_CLUSTER_NODES_{{ $id }}_PORTS_TCP + value: {{ (default $ports.tcp (and .ports .ports.tcp)) | quote }} +- name: IGGY_CLUSTER_NODES_{{ $id }}_PORTS_QUIC + value: {{ (default $ports.quic (and .ports .ports.quic)) | quote }} +- name: IGGY_CLUSTER_NODES_{{ $id }}_PORTS_HTTP + value: {{ (default $ports.http (and .ports .ports.http)) | quote }} +- name: IGGY_CLUSTER_NODES_{{ $id }}_PORTS_WEBSOCKET + value: {{ (default $ports.websocket (and .ports .ports.websocket)) | quote }} +- name: IGGY_CLUSTER_NODES_{{ $id }}_PORTS_TCP_REPLICA + value: {{ .ports.tcpReplica | quote }} + {{- end }} +{{- end }} + +{{/* +Name of the Secret holding the cluster-wide secrets the chart generates from +inline values. Each of encryption, JWT and replica auth may instead point at a +Secret the operator made, which is the path a multi-node cluster wants: the +values have to be byte-identical on every node, so they belong in one object +every release references rather than in each release's values. +*/}} +{{- define "iggy.secretName" -}} + {{- printf "%s-secrets" (include "iggy.fullname" .) }} +{{- end }} + +{{/* +True when any secret has to be generated by the chart, i.e. an inline value is +set for a feature that is switched on and no existing Secret was named for it. +*/}} +{{- define "iggy.createsSecret" -}} + {{- $server := .Values.server }} + {{- $create := false }} + {{- if and $server.encryption.enabled (not $server.encryption.existingSecret.name) }} + {{- $create = true }} + {{- end }} + {{- if and (or $server.jwt.encodingSecret $server.jwt.decodingSecret) (not $server.jwt.existingSecret.name) }} + {{- $create = true }} + {{- end }} + {{- if and $server.cluster.auth.enabled (not $server.cluster.auth.existingSecret.name) }} + {{- $create = true }} + {{- end }} + {{- if $create }}true{{ end }} +{{- end }} + +{{/* +Validate the secret configuration. The server enforces all of this at boot, so +catching it during render only saves a scheduling round trip, but a cluster +whose PSK differs between nodes fails as a handshake rejection rather than as +anything that names the cause. +*/}} +{{- define "iggy.validateSecrets" -}} + {{- $server := .Values.server }} + {{- if $server.encryption.enabled }} + {{- if and (not $server.encryption.key) (not $server.encryption.existingSecret.name) }} + {{- fail "server.encryption.enabled is true but no key was given. Set server.encryption.key to a base64-encoded 32-byte key, or point server.encryption.existingSecret.name at a Secret holding one." }} + {{- end }} + {{- if and $server.encryption.key (not $server.encryption.existingSecret.name) }} + {{- if ne (len (b64dec $server.encryption.key)) 32 }} + {{- fail (printf "server.encryption.key decodes to %d bytes. AES-256-GCM takes a base64-encoded 32-byte key, and the server fails the boot with 'Invalid encryption key' on anything else." (len (b64dec $server.encryption.key))) }} + {{- end }} + {{- end }} + {{- end }} + {{- if $server.cluster.auth.enabled }} + {{- if not $server.cluster.enabled }} + {{- fail "server.cluster.auth.enabled is true but server.cluster.enabled is false. Replica authentication only applies to the consensus port, which exists in cluster mode." }} + {{- end }} + {{- if and (not $server.cluster.auth.sharedSecret) (not $server.cluster.auth.existingSecret.name) }} + {{- fail "server.cluster.auth.enabled is true but no shared secret was given. Set server.cluster.auth.sharedSecret, or point server.cluster.auth.existingSecret.name at a Secret holding one. Every node needs the byte-identical value." }} + {{- end }} + {{- if and $server.cluster.auth.sharedSecret (lt (len $server.cluster.auth.sharedSecret) 32) }} + {{- fail (printf "server.cluster.auth.sharedSecret is %d bytes; the server requires at least 32 bytes of CSPRNG output." (len $server.cluster.auth.sharedSecret)) }} + {{- end }} + {{- if $server.cluster.auth.previousSharedSecret }} + {{- if lt (len $server.cluster.auth.previousSharedSecret) 32 }} + {{- fail (printf "server.cluster.auth.previousSharedSecret is %d bytes; the server requires at least 32 bytes of CSPRNG output." (len $server.cluster.auth.previousSharedSecret)) }} + {{- end }} + {{- if eq $server.cluster.auth.previousSharedSecret $server.cluster.auth.sharedSecret }} + {{- fail "server.cluster.auth.previousSharedSecret equals server.cluster.auth.sharedSecret, which the server rejects as a no-op rotation window. Leave it empty outside a rotation." }} + {{- end }} + {{- end }} + {{- end }} +{{- end }} + +{{/* +Environment entries for the secret-backed settings, each sourced from whichever +Secret owns it. +*/}} +{{- define "iggy.secretEnv" -}} + {{- $server := .Values.server }} + {{- $generated := include "iggy.secretName" . }} + {{- if $server.encryption.enabled }} +- name: IGGY_SYSTEM_ENCRYPTION_ENABLED + value: "true" +- name: IGGY_SYSTEM_ENCRYPTION_KEY + valueFrom: + secretKeyRef: + name: {{ default $generated $server.encryption.existingSecret.name }} + key: {{ ternary $server.encryption.existingSecret.key "encryptionKey" (ne $server.encryption.existingSecret.name "") }} + {{- end }} + {{- if or $server.jwt.existingSecret.name $server.jwt.encodingSecret }} +- name: IGGY_HTTP_JWT_ENCODING_SECRET + valueFrom: + secretKeyRef: + name: {{ default $generated $server.jwt.existingSecret.name }} + key: {{ ternary $server.jwt.existingSecret.encodingSecretKey "jwtEncodingSecret" (ne $server.jwt.existingSecret.name "") }} + {{- end }} + {{- if or $server.jwt.existingSecret.name $server.jwt.decodingSecret }} +- name: IGGY_HTTP_JWT_DECODING_SECRET + valueFrom: + secretKeyRef: + name: {{ default $generated $server.jwt.existingSecret.name }} + key: {{ ternary $server.jwt.existingSecret.decodingSecretKey "jwtDecodingSecret" (ne $server.jwt.existingSecret.name "") }} + optional: true + {{- end }} + {{- if $server.cluster.auth.enabled }} +- name: IGGY_CLUSTER_AUTH_ENABLED + value: "true" +- name: IGGY_CLUSTER_AUTH_SHARED_SECRET + valueFrom: + secretKeyRef: + name: {{ default $generated $server.cluster.auth.existingSecret.name }} + key: {{ ternary $server.cluster.auth.existingSecret.sharedSecretKey "clusterSharedSecret" (ne $server.cluster.auth.existingSecret.name "") }} + {{- if or $server.cluster.auth.previousSharedSecret $server.cluster.auth.existingSecret.name }} +- name: IGGY_CLUSTER_AUTH_PREVIOUS_SHARED_SECRET + valueFrom: + secretKeyRef: + name: {{ default $generated $server.cluster.auth.existingSecret.name }} + key: {{ ternary $server.cluster.auth.existingSecret.previousSharedSecretKey "clusterPreviousSharedSecret" (ne $server.cluster.auth.existingSecret.name "") }} + optional: true + {{- end }} + {{- end }} +{{- end }} diff --git a/helm/charts/iggy/templates/deployment.yaml b/helm/charts/iggy/templates/deployment.yaml index 4e7b6751ef..c99ac9dc98 100644 --- a/helm/charts/iggy/templates/deployment.yaml +++ b/helm/charts/iggy/templates/deployment.yaml @@ -19,6 +19,14 @@ {{- if hasKey .Values.server "podSecurityContext" }} {{- fail "server.podSecurityContext has been moved to podSecurityContext (root level). Please update your values." }} {{- end }} + {{- if gt (int .Values.server.replicaCount) 1 }} + {{- fail "server.replicaCount > 1 is refused: the replicas share one PVC at one subPath and one advertised address, so they would run as independent servers writing the same data directory, and the server takes no lock against that. Run cluster mode instead: one release per node, each with server.cluster.enabled and its own server.cluster.selfReplicaId. See the Cluster Mode section of the chart README." }} + {{- end }} + {{- if .Values.server.cluster.enabled }} + {{- include "iggy.validateCluster" . }} + {{- end }} + {{- include "iggy.validateSecrets" . }} + {{- $ports := include "iggy.serverPorts" . | fromJson }} --- apiVersion: apps/v1 kind: Deployment @@ -27,17 +35,38 @@ metadata: labels: {{- include "iggy.labels" . | nindent 4 }} spec: - {{- if not .Values.autoscaling.enabled }} replicas: {{ .Values.server.replicaCount }} + {{- if .Values.server.strategy }} + strategy: + {{- toYaml .Values.server.strategy | nindent 4 }} + {{- else if .Values.server.hostNetwork }} + strategy: + type: Recreate {{- end }} selector: matchLabels: {{- include "iggy.selectorLabels" . | nindent 6 }} template: metadata: - {{- with .Values.podAnnotations }} + {{- $secretsChecksum := "" }} + {{- if (include "iggy.createsSecret" .) }} + {{- $secretsChecksum = (include (print $.Template.BasePath "/server-secrets.yaml") . | sha256sum) }} + {{- end }} + {{- $rootChecksum := "" }} + {{- if and .Values.server.users.root.createSecret (not .Values.server.users.root.existingSecret.name) }} + {{- $rootChecksum = (include (print $.Template.BasePath "/root-user-credentials.yaml") . | sha256sum) }} + {{- end }} + {{- if or $secretsChecksum $rootChecksum .Values.podAnnotations }} annotations: + {{- if $secretsChecksum }} + checksum/server-secrets: {{ $secretsChecksum }} + {{- end }} + {{- if $rootChecksum }} + checksum/root-user-credentials: {{ $rootChecksum }} + {{- end }} + {{- with .Values.podAnnotations }} {{- toYaml . | nindent 8 }} + {{- end }} {{- end }} labels: {{- include "iggy.labels" . | nindent 8 }} @@ -47,6 +76,11 @@ spec: {{- toYaml . | nindent 8 }} {{- end }} serviceAccountName: {{ include "iggy.serviceAccountName" . }} + enableServiceLinks: false + {{- if .Values.server.hostNetwork }} + hostNetwork: true + dnsPolicy: ClusterFirstWithHostNet + {{- end }} securityContext: {{- toYaml .Values.podSecurityContext | nindent 8 }} containers: @@ -55,16 +89,34 @@ spec: {{- toYaml .Values.securityContext | nindent 12 }} image: "{{ .Values.server.image.repository }}:{{ .Values.server.image.tag | default .Chart.AppVersion }}" imagePullPolicy: {{ .Values.server.image.pullPolicy }} + {{- if or .Values.server.cluster.enabled .Values.server.extraArgs }} + args: + {{- if .Values.server.cluster.enabled }} + - "--replica-id" + - {{ .Values.server.cluster.selfReplicaId | quote }} + {{- end }} + {{- range .Values.server.extraArgs }} + - {{ . | quote }} + {{- end }} + {{- end }} ports: - name: http - containerPort: {{ .Values.server.ports.http }} + containerPort: {{ int $ports.http }} protocol: TCP - name: tcp - containerPort: {{ .Values.server.ports.tcp }} + containerPort: {{ int $ports.tcp }} + protocol: TCP + - name: websocket + containerPort: {{ int $ports.websocket }} protocol: TCP - name: quic - containerPort: {{ .Values.server.ports.quic }} + containerPort: {{ int $ports.quic }} + protocol: UDP + {{- if .Values.server.cluster.enabled }} + - name: tcp-replica + containerPort: {{ int $ports.tcpReplica }} protocol: TCP + {{- end }} env: {{- if .Values.server.users.root.existingSecret.name }} - name: IGGY_ROOT_USERNAME @@ -98,13 +150,22 @@ spec: {{- $declaredInEnv = true }} {{- end }} {{- end }} + {{- if and .Values.server.cluster.enabled .Values.server.advertisedAddress }} + {{- fail "server.advertisedAddress is set together with server.cluster.enabled. In cluster mode the server ignores IGGY_NODE_ADVERTISED_ADDRESS and publishes the roster entry's ip, or its advertisedAddress when one is given, so set it on this node's server.cluster.nodes entry instead." }} + {{- end }} {{- if and $declaredInEnv .Values.server.advertisedAddress }} {{- fail "IGGY_NODE_ADVERTISED_ADDRESS is set in server.env and server.advertisedAddress is also set. The server.env entry wins and server.advertisedAddress is ignored. Please set only one." }} {{- end }} - {{- if not $declaredInEnv }} + {{- if and (not $declaredInEnv) (not .Values.server.cluster.enabled) }} - name: IGGY_NODE_ADVERTISED_ADDRESS value: {{ .Values.server.advertisedAddress | default (printf "%s.%s.svc.cluster.local" (include "iggy.fullname" .) .Release.Namespace) | quote }} {{- end }} + {{- if .Values.server.cluster.enabled }} + {{- include "iggy.clusterEnv" . | nindent 12 }} + {{- end }} + {{- with (include "iggy.secretEnv" . | trim) }} + {{- . | nindent 12 }} + {{- end }} {{- if .Values.server.env }} {{- range .Values.server.env }} - name: {{ .name }} @@ -173,7 +234,7 @@ spec: {{- toYaml . | nindent 8 }} {{- end }} labels: - {{- include "iggy-ui.labels" . | nindent 8 }}-ui + {{- include "iggy-ui.labels" . | nindent 8 }} spec: {{- with .Values.imagePullSecrets }} imagePullSecrets: diff --git a/helm/charts/iggy/templates/hpa.yaml b/helm/charts/iggy/templates/hpa.yaml deleted file mode 100644 index d1f6581294..0000000000 --- a/helm/charts/iggy/templates/hpa.yaml +++ /dev/null @@ -1,61 +0,0 @@ -# Licensed to the Apache Software Foundation (ASF) under one -# or more contributor license agreements. See the NOTICE file -# distributed with this work for additional information -# regarding copyright ownership. The ASF licenses this file -# to you under the Apache License, Version 2.0 (the -# "License"); you may not use this file except in compliance -# with the License. You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, -# software distributed under the License is distributed on an -# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY -# KIND, either express or implied. See the License for the -# specific language governing permissions and limitations -# under the License. - -{{ if .Values.autoscaling.enabled -}} - {{- if semverCompare ">=1.23-0" .Capabilities.KubeVersion.GitVersion -}} -apiVersion: autoscaling/v2 - {{- else -}} -apiVersion: autoscaling/v2beta2 - {{- end }} -kind: HorizontalPodAutoscaler -metadata: - name: {{ include "iggy.fullname" . }} - labels: - {{- include "iggy.labels" . | nindent 4 }} -spec: - scaleTargetRef: - apiVersion: apps/v1 - kind: Deployment - name: {{ include "iggy.fullname" . }} - minReplicas: {{ .Values.autoscaling.minReplicas }} - maxReplicas: {{ .Values.autoscaling.maxReplicas }} - metrics: - {{- if .Values.autoscaling.targetCPUUtilizationPercentage }} - - type: Resource - resource: - name: cpu - {{- if semverCompare ">=1.23-0" .Capabilities.KubeVersion.GitVersion }} - target: - type: Utilization - averageUtilization: {{ .Values.autoscaling.targetCPUUtilizationPercentage }} - {{- else }} - targetAverageUtilization: {{ .Values.autoscaling.targetCPUUtilizationPercentage }} - {{- end }} - {{- end }} - {{- if .Values.autoscaling.targetMemoryUtilizationPercentage }} - - type: Resource - resource: - name: memory - {{- if semverCompare ">=1.23-0" .Capabilities.KubeVersion.GitVersion }} - target: - type: Utilization - averageUtilization: {{ .Values.autoscaling.targetMemoryUtilizationPercentage }} - {{- else }} - targetAverageUtilization: {{ .Values.autoscaling.targetMemoryUtilizationPercentage }} - {{- end }} - {{- end }} -{{- end }} diff --git a/helm/charts/iggy/templates/server-secrets.yaml b/helm/charts/iggy/templates/server-secrets.yaml new file mode 100644 index 0000000000..e75e6e4985 --- /dev/null +++ b/helm/charts/iggy/templates/server-secrets.yaml @@ -0,0 +1,44 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +{{- if and .Values.server.enabled (include "iggy.createsSecret" .) }} +apiVersion: v1 +kind: Secret +metadata: + name: {{ include "iggy.secretName" . }} + labels: + {{- include "iggy.labels" . | nindent 4 }} +type: Opaque +stringData: + {{- if and .Values.server.encryption.enabled (not .Values.server.encryption.existingSecret.name) }} + encryptionKey: {{ .Values.server.encryption.key | quote }} + {{- end }} + {{- if not .Values.server.jwt.existingSecret.name }} + {{- with .Values.server.jwt.encodingSecret }} + jwtEncodingSecret: {{ . | quote }} + {{- end }} + {{- with .Values.server.jwt.decodingSecret }} + jwtDecodingSecret: {{ . | quote }} + {{- end }} + {{- end }} + {{- if and .Values.server.cluster.auth.enabled (not .Values.server.cluster.auth.existingSecret.name) }} + clusterSharedSecret: {{ .Values.server.cluster.auth.sharedSecret | quote }} + {{- with .Values.server.cluster.auth.previousSharedSecret }} + clusterPreviousSharedSecret: {{ . | quote }} + {{- end }} + {{- end }} +{{- end }} diff --git a/helm/charts/iggy/templates/service.yaml b/helm/charts/iggy/templates/service.yaml index 9820044247..4294e38288 100644 --- a/helm/charts/iggy/templates/service.yaml +++ b/helm/charts/iggy/templates/service.yaml @@ -32,11 +32,15 @@ spec: - name: quic port: {{ .Values.server.ports.quic }} targetPort: quic - protocol: TCP + protocol: UDP - name: tcp port: {{ .Values.server.ports.tcp }} targetPort: tcp protocol: TCP + - name: websocket + port: {{ .Values.server.ports.websocket }} + targetPort: websocket + protocol: TCP selector: {{- include "iggy.selectorLabels" . | nindent 4 }} {{- end }} diff --git a/helm/charts/iggy/values.yaml b/helm/charts/iggy/values.yaml index be97571569..ab1a5f728f 100644 --- a/helm/charts/iggy/values.yaml +++ b/helm/charts/iggy/values.yaml @@ -20,7 +20,8 @@ server: # -- Client-facing address published in cluster metadata. Declaring # `IGGY_NODE_ADVERTISED_ADDRESS` in `server.env` instead also works, but # setting both is refused at render time. Empty falls back to the in-cluster - # Service DNS name. + # Service DNS name. Ignored in cluster mode, where the address comes from the + # node's roster entry, so setting both is refused there too. advertisedAddress: "" # -- Enable the Iggy server deployment enabled: true @@ -31,15 +32,17 @@ server: repository: apache/iggy # -- Image pull policy pullPolicy: Always - # -- Server image tag (overrides chart appVersion) - tag: "0.7.0" + # -- Server image tag. Empty uses the chart appVersion. + tag: "" ports: # -- HTTP API port http: 3000 - # -- QUIC protocol port + # -- QUIC protocol port (UDP) quic: 8080 # -- TCP protocol port tcp: 8090 + # -- WebSocket protocol port + websocket: 8092 service: # -- Service type for the server @@ -100,6 +103,115 @@ server: # -- PVC storage size size: 8Gi + # -- Server-side encryption of message payloads and state commands, using + # AES-256-GCM. Every node of a cluster must hold the identical key, or it + # cannot read data another node wrote. + encryption: + # -- Enable encryption at rest + enabled: false + # -- 32-byte key, base64 encoded. Ignored when `existingSecret.name` is set. + # Prefer `existingSecret` outside development: a value here is stored in the + # Helm release and readable by anyone who can read it. + key: "" + existingSecret: + # -- Name of an existing Secret holding the encryption key + name: "" + # -- Key inside that Secret + key: encryptionKey + + # -- Secrets used to sign and validate HTTP bearer tokens. Left unset, each + # node generates a random secret on every start, which invalidates tokens + # across restarts and keeps them node-local. Setting the identical secret on + # every node makes bearers valid cluster-wide and activates follower to + # primary HTTP forwarding; `cluster.auth.enabled` derives the same thing from + # the replica PSK, so it is an alternative rather than an addition. + jwt: + # -- Encoding secret. Ignored when `existingSecret.name` is set. + encodingSecret: "" + # -- Decoding secret. Ignored when `existingSecret.name` is set. + decodingSecret: "" + existingSecret: + # -- Name of an existing Secret holding the JWT secrets + name: "" + # -- Key inside that Secret holding the encoding secret + encodingSecretKey: jwtEncodingSecret + # -- Key inside that Secret holding the decoding secret + decodingSecretKey: jwtDecodingSecret + + cluster: + # -- Enable cluster (VSR consensus) mode. One Helm release per node: every + # release shares the same `nodes` roster and overrides only `selfReplicaId`. + # See the Cluster Mode section of the chart README. + enabled: false + # -- Cluster name, byte-identical on every node. Hashed into the on-disk + # cluster id on first boot, so changing it later means starting from an + # empty data directory. + name: iggy-cluster + # -- Which `nodes` entry this release runs, matched against `replicaId`. + selfReplicaId: 0 + # -- Refuse to render a cluster node without `server.hostNetwork`. The + # replica listener binds the roster `ip` verbatim, which no pod owns on the + # cluster network, so the pod would die at boot with `CannotBindToSocket`. + # Set this to false only when the roster `ip` is an address the pod itself + # holds. + requireHostNetwork: true + # -- Replica-to-replica authentication on the consensus port. When enabled + # every peer must complete an authenticated handshake or be rejected, and a + # shared secret becomes mandatory. Enabling it on a running cluster is a + # coordinated-restart change, not a rolling one. + auth: + # -- Require the authenticated replica handshake + enabled: false + # -- Cluster-wide pre-shared key, at least 32 bytes of CSPRNG output, + # byte-identical on every node. Ignored when `existingSecret.name` is set. + sharedSecret: "" + # -- Retiring key, accepted for verification only while a rotation is in + # flight. Leave empty outside a rotation. + previousSharedSecret: "" + existingSecret: + # -- Name of an existing Secret holding the pre-shared keys + name: "" + # -- Key inside that Secret holding the active shared secret + sharedSecretKey: clusterSharedSecret + # -- Key inside that Secret holding the retiring shared secret + previousSharedSecretKey: clusterPreviousSharedSecret + + # -- Cluster roster, mirroring the server's `[[cluster.nodes]]` config. + # Every node runs the identical list. `ip` is the replica-plane address: + # this node's consensus listener binds it verbatim and every peer dials it + # verbatim, so it must be a literal IP that the pod itself owns. With + # `server.hostNetwork` that is the node IP. `ports.tcpReplica` is required + # on every entry; the remaining ports default to `server.ports`. + nodes: [] + # nodes: + # - name: iggy-node-0 + # ip: 10.0.1.5 + # replicaId: 0 + # # -- Optional client-facing address, a literal IP or a DNS hostname. + # advertisedAddress: "" + # ports: + # tcp: 8090 + # quic: 8080 + # http: 3000 + # websocket: 8092 + # tcpReplica: 9090 + + # -- Run the server pod in the host network namespace. Required for cluster + # mode, where the replica listener binds the roster IP verbatim. + hostNetwork: false + + # -- Deployment update strategy. Empty lets the chart choose: `Recreate` when + # `hostNetwork` is set, because a rolling update would wait forever for a + # replacement pod that cannot bind host ports the outgoing pod still holds, + # and the Kubernetes default otherwise. + strategy: {} + + # -- Extra command-line arguments appended to the server entrypoint, e.g. + # `["--with-default-root-credentials"]` for a throwaway development install. + # `--replica-id` is not one of them: the chart passes it already whenever + # `cluster.enabled` is set. + extraArgs: [] + # -- Environment variables for the server container env: - name: RUST_LOG @@ -232,15 +344,3 @@ securityContext: # -- Resource limits and requests for server resources: {} - -autoscaling: - # -- Enable horizontal pod autoscaling - enabled: false - # -- Minimum replicas for autoscaling - minReplicas: 1 - # -- Maximum replicas for autoscaling - maxReplicas: 100 - # -- Target CPU utilization for autoscaling - targetCPUUtilizationPercentage: 80 - # -- Target memory utilization for autoscaling (optional) - # targetMemoryUtilizationPercentage: 80 diff --git a/scripts/ci/test-helm.sh b/scripts/ci/test-helm.sh index 370b2f00b2..5315223143 100755 --- a/scripts/ci/test-helm.sh +++ b/scripts/ci/test-helm.sh @@ -53,6 +53,9 @@ HELM_SMOKE_GATEWAY_NAME="${HELM_SMOKE_GATEWAY_NAME:-iggy-smoke-gateway}" HELM_SMOKE_GATEWAY_PF_PORT="${HELM_SMOKE_GATEWAY_PF_PORT:-8080}" HELM_SMOKE_KIND_NAME="${HELM_SMOKE_KIND_NAME:-iggy-helm-smoke}" HELM_SMOKE_SERVER_CPU_ALLOCATION="${HELM_SMOKE_SERVER_CPU_ALLOCATION:-1}" +# A well-formed 32-byte AES-256-GCM key, base64 encoded. Render-only: it never +# reaches a running server. +HELM_TEST_ENCRYPTION_KEY="AAECAwQFBgcICQoLDA0ODxAREhMUFRYXGBkaGxwdHh8=" # HELM_SMOKE_GATEWAY_NAMESPACE and HELM_SMOKE_GATEWAY_NAME must match the # defaults in scripts/ci/setup-helm-smoke-cluster.sh - the HTTPRoute @@ -197,12 +200,10 @@ validate() { local chart_version local chart_app_version - local server_image_tag local ui_image_tag chart_version="$(extract_chart_field version)" chart_app_version="$(extract_chart_field appVersion)" - server_image_tag="$(extract_values_tag server)" ui_image_tag="$(extract_values_tag ui)" prepare_render_dir @@ -217,19 +218,16 @@ validate() { grep -q "helm.sh/chart: iggy-${chart_version}" "$HELM_RENDER_DIR/default.yaml" grep -q "helm.sh/chart: iggy-ui-${chart_version}" "$HELM_RENDER_DIR/default.yaml" grep -q "app.kubernetes.io/version: \"${chart_app_version}\"" "$HELM_RENDER_DIR/default.yaml" - grep -q "image: \"apache/iggy:${server_image_tag}\"" "$HELM_RENDER_DIR/default.yaml" + grep -q "image: \"apache/iggy:${chart_app_version}\"" "$HELM_RENDER_DIR/default.yaml" grep -q "image: \"apache/iggy-web-ui:${ui_image_tag}\"" "$HELM_RENDER_DIR/default.yaml" helm template iggy "$CHART_DIR" \ --set server.persistence.enabled=true \ - --set autoscaling.enabled=true \ - --set autoscaling.targetCPUUtilizationPercentage=80 \ --set server.ingress.enabled=true \ --set ui.ingress.enabled=true \ --set server.serviceMonitor.enabled=true \ > "$HELM_RENDER_DIR/all-features.yaml" grep -q '^kind: PersistentVolumeClaim$' "$HELM_RENDER_DIR/all-features.yaml" - grep -q '^kind: HorizontalPodAutoscaler$' "$HELM_RENDER_DIR/all-features.yaml" test "$(grep -c '^kind: Ingress$' "$HELM_RENDER_DIR/all-features.yaml")" -eq 2 test "$(extract_kind_names "$HELM_RENDER_DIR/all-features.yaml" Ingress | sort -u | wc -l | tr -d ' ')" -eq 2 extract_kind_names "$HELM_RENDER_DIR/all-features.yaml" Ingress | grep -qx 'iggy' @@ -239,14 +237,10 @@ validate() { helm template iggy "$CHART_DIR" \ --kube-version 1.18.0 \ --api-versions networking.k8s.io/v1beta1 \ - --api-versions autoscaling/v2beta2 \ - --set autoscaling.enabled=true \ - --set autoscaling.targetCPUUtilizationPercentage=80 \ --set server.ingress.enabled=true \ --set ui.ingress.enabled=true \ > "$HELM_RENDER_DIR/legacy-k8s-1.18.yaml" test "$(grep -c '^apiVersion: networking.k8s.io/v1beta1$' "$HELM_RENDER_DIR/legacy-k8s-1.18.yaml")" -eq 2 - grep -q '^apiVersion: autoscaling/v2beta2$' "$HELM_RENDER_DIR/legacy-k8s-1.18.yaml" helm template iggy "$CHART_DIR" --set ui.enabled=false > "$HELM_RENDER_DIR/server-only.yaml" test "$(grep -c '^kind: Deployment$' "$HELM_RENDER_DIR/server-only.yaml")" -eq 1 @@ -266,11 +260,69 @@ validate() { fi grep -q 'name: supersecret' "$HELM_RENDER_DIR/existing-secret.yaml" + helm template iggy "$CHART_DIR" \ + -f "$CHART_DIR/examples/cluster-3-node.yaml" \ + --set server.cluster.selfReplicaId=1 \ + > "$HELM_RENDER_DIR/cluster.yaml" + grep -q 'name: IGGY_CLUSTER_ENABLED' "$HELM_RENDER_DIR/cluster.yaml" + grep -q 'name: IGGY_CLUSTER_NODES_2_PORTS_TCP_REPLICA' "$HELM_RENDER_DIR/cluster.yaml" + grep -q -- '- "--replica-id"' "$HELM_RENDER_DIR/cluster.yaml" + grep -q '^ hostNetwork: true$' "$HELM_RENDER_DIR/cluster.yaml" + grep -q 'name: tcp-replica' "$HELM_RENDER_DIR/cluster.yaml" + + grep -q 'name: IGGY_CLUSTER_AUTH_SHARED_SECRET' "$HELM_RENDER_DIR/cluster.yaml" + grep -q 'name: IGGY_SYSTEM_ENCRYPTION_KEY' "$HELM_RENDER_DIR/cluster.yaml" + grep -q 'name: IGGY_HTTP_JWT_ENCODING_SECRET' "$HELM_RENDER_DIR/cluster.yaml" + if grep -qE '^ +(encryptionKey|clusterSharedSecret|jwtEncodingSecret):' "$HELM_RENDER_DIR/cluster.yaml"; then + echo "Error: cluster render inlined a secret value instead of referencing the existing Secret" >&2 + exit 1 + fi + + helm template iggy "$CHART_DIR" \ + --set server.encryption.enabled=true \ + --set-string server.encryption.key="$HELM_TEST_ENCRYPTION_KEY" \ + > "$HELM_RENDER_DIR/generated-secret.yaml" + grep -q "^ encryptionKey: \"${HELM_TEST_ENCRYPTION_KEY}\"$" "$HELM_RENDER_DIR/generated-secret.yaml" + grep -q '^ name: iggy-secrets$' "$HELM_RENDER_DIR/generated-secret.yaml" + grep -q '^ name: iggy-secrets$' "$HELM_RENDER_DIR/generated-secret.yaml" + grep -q '^ key: encryptionKey$' "$HELM_RENDER_DIR/generated-secret.yaml" + test "$(grep -c '^kind: Secret$' "$HELM_RENDER_DIR/generated-secret.yaml")" -eq 2 + + assert_render_rejected "server.replicaCount=3" --set server.replicaCount=3 + assert_render_rejected "encryption without a key" --set server.encryption.enabled=true + assert_render_rejected "an encryption key that is not 32 bytes" \ + --set server.encryption.enabled=true \ + --set-string server.encryption.key=bm90LTMyLWJ5dGVz + assert_render_rejected "replica auth without a secret" \ + -f "$CHART_DIR/examples/cluster-3-node.yaml" \ + --set server.cluster.auth.existingSecret.name="" \ + --set server.cluster.selfReplicaId=0 + assert_render_rejected "a shared secret under 32 bytes" \ + -f "$CHART_DIR/examples/cluster-3-node.yaml" \ + --set server.cluster.auth.existingSecret.name="" \ + --set-string server.cluster.auth.sharedSecret=tooshort \ + --set server.cluster.selfReplicaId=0 + assert_render_rejected "cluster without a roster" --set server.cluster.enabled=true + assert_render_rejected "selfReplicaId outside the roster" \ + -f "$CHART_DIR/examples/cluster-3-node.yaml" --set server.cluster.selfReplicaId=9 + validate_yamllint validate_helmfmt validate_helm_docs } +# Assert that `helm template` refuses a values combination the chart guards +# against. A guard that stops failing is a silent data-corruption regression, +# so the render succeeding is the error case here. +assert_render_rejected() { + local description="$1" + shift + if helm template iggy "$CHART_DIR" "$@" > /dev/null 2>&1; then + echo "Error: chart rendered ${description}, but that combination must be refused" >&2 + exit 1 + fi +} + # PID of the kubectl port-forward process started by smoke(). # Stored at script scope so the EXIT trap can kill it on any code path. HELM_SMOKE_GW_PF_PID="" From ca0b578a225bf4b9169e4ae78844f3f0423a4e99 Mon Sep 17 00:00:00 2001 From: Hubert Gruszecki Date: Wed, 2 Sep 2026 17:37:18 +0200 Subject: [PATCH 045/182] fix(sdk): correct consumer group polling and commits in IggyConsumer (#4034) --- .../src/actors/consumer/client/low_level.rs | 28 +- .../common/src/consumer_group_client_state.rs | 38 + core/common/src/lib.rs | 7 + .../src/traits/binary_impls/messages.rs | 43 +- core/common/src/traits/message_client.rs | 41 +- .../src/types/message/polled_messages.rs | 6 +- core/integration/tests/sdk/consumer_group.rs | 83 ++ .../tests/sdk/consumer_group_membership.rs | 276 +++++++ .../tests/sdk/consumer_shutdown.rs | 112 +++ core/integration/tests/sdk/mod.rs | 1 + .../client_wrappers/binary_message_client.rs | 42 +- core/sdk/src/clients/binary_message.rs | 26 +- core/sdk/src/clients/consumer.rs | 762 +++++++++++------- core/sdk/src/clients/consumer_builder.rs | 25 +- core/sdk/src/prelude.rs | 3 +- core/sdk/src/quic/quic_client.rs | 7 + core/sdk/src/tcp/tcp_client.rs | 7 + core/sdk/src/websocket/websocket_client.rs | 7 + foreign/php/README.md | 6 +- foreign/php/iggy-php.stubs.php | 3 + foreign/php/src/client.rs | 3 + foreign/python/apache_iggy.pyi | 2 + foreign/python/src/client.rs | 2 + 23 files changed, 1172 insertions(+), 358 deletions(-) create mode 100644 core/integration/tests/sdk/consumer_shutdown.rs diff --git a/core/bench/src/actors/consumer/client/low_level.rs b/core/bench/src/actors/consumer/client/low_level.rs index 7f47748959..964231f628 100644 --- a/core/bench/src/actors/consumer/client/low_level.rs +++ b/core/bench/src/actors/consumer/client/low_level.rs @@ -22,6 +22,7 @@ use crate::benchmarks::common::create_consumer; use crate::utils::ClientFactory; use crate::utils::{batch_total_size_bytes, batch_user_size_bytes}; use iggy::prelude::*; +use std::collections::HashMap; use std::sync::Arc; use std::time::Duration; use tokio::time::Instant; @@ -36,7 +37,9 @@ pub struct LowLevelConsumerClient { partition_id: Option, polling_strategy: PollingStrategy, auto_commit: bool, - offset: u64, + /// Where offset polling continues in each partition. A group member polls its partitions + /// round-robin, so one shared cursor would skip the start of every partition but the first. + next_offsets: HashMap, } impl LowLevelConsumerClient { @@ -51,7 +54,7 @@ impl LowLevelConsumerClient { partition_id: None, polling_strategy: PollingStrategy::next(), auto_commit: true, - offset: 0, + next_offsets: HashMap::new(), } } } @@ -62,14 +65,21 @@ impl ConsumerClient for LowLevelConsumerClient { let consumer = self.consumer.as_ref().expect("consumer not initialized"); let messages_to_receive = self.config.messages_per_batch.get(); + let polling_strategy = self.polling_strategy; + let next_offsets = &self.next_offsets; + let strategy_for = |partition_id: u32| { + next_offsets + .get(&partition_id) + .map_or(polling_strategy, |offset| PollingStrategy::offset(*offset)) + }; let before_poll = Instant::now(); let polled = client - .poll_messages( + .poll_messages_with_strategy_for( &self.stream_id, &self.topic_id, self.partition_id, consumer, - &self.polling_strategy, + &strategy_for, messages_to_receive, self.auto_commit, ) @@ -99,9 +109,11 @@ impl ConsumerClient for LowLevelConsumerClient { let user_bytes = batch_user_size_bytes(&polled); let total_bytes = batch_total_size_bytes(&polled); - self.offset += messages_count; - if self.polling_strategy.kind == PollingKind::Offset { - self.polling_strategy.value += messages_count; + if self.polling_strategy.kind == PollingKind::Offset + && let Some(last) = polled.messages.last() + { + self.next_offsets + .insert(polled.partition_id, last.header.offset + 1); } Ok(Some(BatchMetrics { @@ -137,7 +149,7 @@ impl BenchmarkInit for LowLevelConsumerClient { ) .await; let (polling_strategy, auto_commit) = match self.config.polling_kind { - PollingKind::Offset => (PollingStrategy::offset(self.offset), false), + PollingKind::Offset => (PollingStrategy::offset(0), false), PollingKind::Next => (PollingStrategy::next(), true), _ => panic!("Unsupported polling kind: {:?}", self.config.polling_kind), }; diff --git a/core/common/src/consumer_group_client_state.rs b/core/common/src/consumer_group_client_state.rs index 1e694cc700..d4331cb129 100644 --- a/core/common/src/consumer_group_client_state.rs +++ b/core/common/src/consumer_group_client_state.rs @@ -185,6 +185,24 @@ impl ConsumerGroupClientState { .cloned() .collect() } + + /// Drop what the consensus session owned. Membership is per connection + /// and the coordinator fences assignments by a generation it tracks per + /// session, so nothing synced under the old session holds once it is + /// reset. The balanced cursors and the partition counts stay: they belong + /// to a topic, not to a session, and clearing them would restart the + /// produce round-robin at partition 0 and cost a metadata round trip per + /// topic on every reconnect. + pub fn clear_session_scoped(&self) { + self.assignments + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .clear(); + self.joined_groups + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .clear(); + } } #[cfg(test)] @@ -242,4 +260,24 @@ mod tests { state.deregister_group("s|t|g"); assert!(!state.is_registered("s|t|g")); } + + #[test] + fn session_reset_drops_membership_and_keeps_topic_state() { + let state = ConsumerGroupClientState::new(); + let id = Identifier::named("g").unwrap(); + state.register_group("s|t|g".to_owned(), id.clone(), id.clone(), id); + state.set_assignment("s|t|g".to_owned(), 1, vec![0, 1]); + assert_eq!(state.next_balanced_partition("s|t", 3), 0); + state.set_partition_count("s|t".to_owned(), 3); + + state.clear_session_scoped(); + + assert!(state.registered_groups().is_empty()); + assert!(!state.is_registered("s|t|g")); + assert!(!state.has_assignment("s|t|g")); + // Topic state outlives the session: the produce cursor carries on and + // the partition count is still cached. + assert_eq!(state.next_balanced_partition("s|t", 3), 1); + assert_eq!(state.partition_count("s|t"), Some(3)); + } } diff --git a/core/common/src/lib.rs b/core/common/src/lib.rs index 89a9fc74be..e74eae292f 100644 --- a/core/common/src/lib.rs +++ b/core/common/src/lib.rs @@ -41,6 +41,13 @@ pub use consumer_group_client_state::ConsumerGroupClientState; /// a genuine end-of-partition empty poll, which echoes the real partition id. pub const RESYNC_REQUIRED_PARTITION_SENTINEL: u32 = u32::MAX; +/// Client-side `partition_id` of an empty poll reply for a consumer-group +/// member that currently holds no partitions. The server never sends it: the +/// transport fills it in when the synced assignment is empty, so a caller can +/// tell "nothing assigned" from a genuine empty poll, which echoes the real +/// partition id. Same value as the Go and Node SDKs use for that case. +pub const NO_ASSIGNED_PARTITION: u32 = u32::MAX - 1; + /// Frozen ceiling on the widest batch record any admission path can persist. /// Two knobs bound admission and both are validated against this at boot: /// `message_bus.max_message_size` caps every bus-framed wire message, and diff --git a/core/common/src/traits/binary_impls/messages.rs b/core/common/src/traits/binary_impls/messages.rs index bf720a1fe8..d8b12410d0 100644 --- a/core/common/src/traits/binary_impls/messages.rs +++ b/core/common/src/traits/binary_impls/messages.rs @@ -166,14 +166,15 @@ async fn resolve_partitioning( } /// Poll a consumer group: select one of the member's assigned partitions -/// (round-robin) and send an explicit-partition poll. A coordinator fence -/// rejection (stale assignment after a rebalance) triggers one re-sync + retry. +/// (round-robin), ask `strategy_for` where to read it from and send an +/// explicit-partition poll. A coordinator fence rejection (stale assignment +/// after a rebalance) triggers one re-sync + retry. async fn poll_group_messages( client: &B, stream_id: &Identifier, topic_id: &Identifier, consumer: &Consumer, - strategy: &PollingStrategy, + strategy_for: &(dyn Fn(u32) -> PollingStrategy + Send + Sync), count: u32, auto_commit: bool, ) -> Result { @@ -195,14 +196,19 @@ async fn poll_group_messages( topic_id.clone(), )); } - return Ok(PolledMessages::empty()); + return Ok(PolledMessages { + partition_id: crate::NO_ASSIGNED_PARTITION, + ..PolledMessages::empty() + }); }; + // Resolved per attempt: a fence retry can land on another partition. + let strategy = strategy_for(partition_id); let request = PollMessagesRequest { consumer: consumer_to_wire(consumer)?, stream_id: identifier_to_wire(stream_id)?, topic_id: identifier_to_wire(topic_id)?, partition_id: Some(partition_id), - strategy: polling_strategy_to_wire(strategy), + strategy: polling_strategy_to_wire(&strategy), count, auto_commit, }; @@ -283,6 +289,28 @@ impl MessageClient for B { strategy: &PollingStrategy, count: u32, auto_commit: bool, + ) -> Result { + self.poll_messages_with_strategy_for( + stream_id, + topic_id, + partition_id, + consumer, + &|_: u32| *strategy, + count, + auto_commit, + ) + .await + } + + async fn poll_messages_with_strategy_for( + &self, + stream_id: &Identifier, + topic_id: &Identifier, + partition_id: Option, + consumer: &Consumer, + strategy_for: &(dyn Fn(u32) -> PollingStrategy + Send + Sync), + count: u32, + auto_commit: bool, ) -> Result { fail_if_not_authenticated(self).await?; // VSR: a consumer-group poll without an explicit partition is resolved @@ -294,18 +322,19 @@ impl MessageClient for B { stream_id, topic_id, consumer, - strategy, + strategy_for, count, auto_commit, ) .await; } + let strategy = strategy_for(partition_id.unwrap_or(0)); let req = PollMessagesRequest { consumer: consumer_to_wire(consumer)?, stream_id: identifier_to_wire(stream_id)?, topic_id: identifier_to_wire(topic_id)?, partition_id, - strategy: polling_strategy_to_wire(strategy), + strategy: polling_strategy_to_wire(&strategy), count, auto_commit, }; diff --git a/core/common/src/traits/message_client.rs b/core/common/src/traits/message_client.rs index 0c5fa11e4b..18c6d2294a 100644 --- a/core/common/src/traits/message_client.rs +++ b/core/common/src/traits/message_client.rs @@ -16,8 +16,8 @@ // under the License. use crate::{ - Consumer, Identifier, IggyError, IggyMessage, Partitioning, PolledMessages, PollingStrategy, - SendMessagesResponse, + Consumer, ConsumerKind, Identifier, IggyError, IggyMessage, Partitioning, PolledMessages, + PollingStrategy, SendMessagesResponse, }; use async_trait::async_trait; @@ -29,6 +29,7 @@ pub trait MessageClient { /// Authentication is required, and the permission to poll the messages. /// /// Polling a consumer group the client is not (or no longer) a member of fails with `ConsumerGroupMemberNotFound` rather than returning an empty batch, so the caller can rejoin. + /// A member that holds no partitions gets an empty batch whose `partition_id` is [`NO_ASSIGNED_PARTITION`](crate::NO_ASSIGNED_PARTITION). #[allow(clippy::too_many_arguments)] async fn poll_messages( &self, @@ -41,6 +42,42 @@ pub trait MessageClient { auto_commit: bool, ) -> Result; + /// [`poll_messages`](Self::poll_messages) whose strategy is chosen once the partition is + /// known. A consumer-group poll without a partition picks one of the member's assigned + /// partitions first and then asks `strategy_for` for it, so a caller can continue every + /// partition from its own position. Any other poll has its partition up front and asks + /// `strategy_for` for that one, or for `0` when none was given, which is the partition the + /// server reads then. + /// + /// The default implementation is for transports that cannot pick a partition client-side: + /// a consumer-group poll without a partition fails with `FeatureUnavailable`. + #[allow(clippy::too_many_arguments)] + async fn poll_messages_with_strategy_for( + &self, + stream_id: &Identifier, + topic_id: &Identifier, + partition_id: Option, + consumer: &Consumer, + strategy_for: &(dyn Fn(u32) -> PollingStrategy + Send + Sync), + count: u32, + auto_commit: bool, + ) -> Result { + if consumer.kind == ConsumerKind::ConsumerGroup && partition_id.is_none() { + return Err(IggyError::FeatureUnavailable); + } + let strategy = strategy_for(partition_id.unwrap_or(0)); + self.poll_messages( + stream_id, + topic_id, + partition_id, + consumer, + &strategy, + count, + auto_commit, + ) + .await + } + /// Send messages using specified partitioning strategy to the given stream and topic by unique IDs or names. /// /// Authentication is required, and the permission to send the messages. diff --git a/core/common/src/types/message/polled_messages.rs b/core/common/src/types/message/polled_messages.rs index af23d4ec7f..d974fa4795 100644 --- a/core/common/src/types/message/polled_messages.rs +++ b/core/common/src/types/message/polled_messages.rs @@ -29,7 +29,11 @@ use tracing::error; /// - `messages`: the collection of messages. #[derive(Debug, Serialize, Deserialize)] pub struct PolledMessages { - /// The identifier of the partition. If it's '0', then there's no partition assigned to the consumer group member. + /// The identifier of the partition. An empty reply can carry a sentinel instead of a real + /// id: [`NO_ASSIGNED_PARTITION`](crate::NO_ASSIGNED_PARTITION) for a consumer-group member + /// that holds no partitions, or + /// [`RESYNC_REQUIRED_PARTITION_SENTINEL`](crate::RESYNC_REQUIRED_PARTITION_SENTINEL) when + /// the server fenced a stale group assignment. pub partition_id: u32, /// The current offset of the partition. pub current_offset: u64, diff --git a/core/integration/tests/sdk/consumer_group.rs b/core/integration/tests/sdk/consumer_group.rs index b670de3ee3..f233560da0 100644 --- a/core/integration/tests/sdk/consumer_group.rs +++ b/core/integration/tests/sdk/consumer_group.rs @@ -15,6 +15,7 @@ // specific language governing permissions and limitations // under the License. +use std::collections::HashMap; use std::str::FromStr; use std::time::Duration; @@ -29,6 +30,9 @@ const CONSUMER_GROUP_NAME: &str = "consumer-group-rejoin-group"; const CONSUMER_USERNAME: &str = "consumer-group-rejoin-user"; const CONSUMER_PASSWORD: &str = "password123"; const CONSUMER_REJOIN_TIMEOUT: Duration = Duration::from_secs(10); +const PARTITIONS_COUNT: u32 = 2; +const MESSAGES_PER_PARTITION: u32 = 5; +const READ_TIMEOUT: Duration = Duration::from_secs(10); // Pins a 60s server heartbeat because harness clients never ping on their own: // the SDK pinger is spawned by `IggyClient::connect`, which the harness builder @@ -169,3 +173,82 @@ async fn consumer_group_retries_rejoin_after_failure(harness: &TestHarness) { assert_eq!(group.members_count, 1); assert_eq!(group.members.len(), 1); } + +// A group member polls its partitions round-robin. Under a strategy other than `next()` the +// continuation must be kept per partition: one shared cursor would ask the second partition for +// the offset reached in the first one and skip its beginning. +#[iggy_harness( + test_client_transport = [Tcp, WebSocket, Quic], + server(heartbeat.enabled = true, heartbeat.interval = "60s") +)] +async fn given_offset_strategy_when_member_polls_two_partitions_should_read_each_from_its_start( + harness: &TestHarness, +) { + let root_client = harness + .root_client() + .await + .expect("Failed to get root client"); + let stream_id = Identifier::named(STREAM_NAME).unwrap(); + let topic_id = Identifier::named(TOPIC_NAME).unwrap(); + + root_client.create_stream(STREAM_NAME).await.unwrap(); + root_client + .create_topic( + &stream_id, + TOPIC_NAME, + &TopicCreateOptions { + partitions_count: Some(PARTITIONS_COUNT), + message_expiry: Some(IggyExpiry::NeverExpire), + ..TopicCreateOptions::default() + }, + ) + .await + .unwrap(); + for partition_id in 0..PARTITIONS_COUNT { + let mut messages: Vec = (0..MESSAGES_PER_PARTITION) + .map(|index| IggyMessage::from_str(&format!("{partition_id}-{index}")).unwrap()) + .collect(); + root_client + .send_messages( + &stream_id, + &topic_id, + &Partitioning::partition_id(partition_id), + &mut messages, + ) + .await + .unwrap(); + } + + let mut consumer = root_client + .consumer_group(CONSUMER_GROUP_NAME, STREAM_NAME, TOPIC_NAME) + .unwrap() + .polling_strategy(PollingStrategy::offset(0)) + .batch_length(MESSAGES_PER_PARTITION) + .auto_commit(AutoCommit::Disabled) + .build(); + consumer.init().await.unwrap(); + + let expected_total = (PARTITIONS_COUNT * MESSAGES_PER_PARTITION) as usize; + let mut offsets_by_partition: HashMap> = HashMap::new(); + for _ in 0..expected_total { + let received = timeout(READ_TIMEOUT, consumer.next()) + .await + .expect("every partition must be read from its start before the timeout") + .expect("consumer stream should remain open") + .expect("polling must not fail"); + offsets_by_partition + .entry(received.partition_id) + .or_default() + .push(received.message.header.offset); + } + consumer.shutdown().await.unwrap(); + + let expected_offsets: Vec = (0..u64::from(MESSAGES_PER_PARTITION)).collect(); + for partition_id in 0..PARTITIONS_COUNT { + assert_eq!( + offsets_by_partition.get(&partition_id), + Some(&expected_offsets), + "partition {partition_id} must be read from offset 0 without gaps" + ); + } +} diff --git a/core/integration/tests/sdk/consumer_group_membership.rs b/core/integration/tests/sdk/consumer_group_membership.rs index 429aea6493..a8e7f59e79 100644 --- a/core/integration/tests/sdk/consumer_group_membership.rs +++ b/core/integration/tests/sdk/consumer_group_membership.rs @@ -15,10 +15,14 @@ // specific language governing permissions and limitations // under the License. +use std::str::FromStr; +use std::sync::Arc; use std::time::Duration; use futures::StreamExt; +use iggy::prelude::locking::IggyRwLockFn; use iggy::prelude::*; +use iggy_common::{BinaryTransport, ConsumerGroupClientState}; use integration::iggy_harness; use tokio::time::timeout; @@ -156,6 +160,161 @@ async fn given_group_member_holds_no_partitions_when_group_deleted_should_surfac } } +// A member holding zero partitions gets an empty poll reply whose partition id +// is the `NO_ASSIGNED_PARTITION` sentinel, so a caller can tell it from an +// empty partition, which echoes its real id. +#[iggy_harness( + test_client_transport = [Tcp, WebSocket, Quic], + server(heartbeat.enabled = true, heartbeat.interval = "60s") +)] +async fn given_member_holds_no_partitions_when_polled_should_report_no_assigned_partition( + harness: &TestHarness, +) { + let stream_id = Identifier::named(STREAM_NAME).unwrap(); + let topic_id = Identifier::named(TOPIC_NAME).unwrap(); + let group_id = Identifier::named(CONSUMER_GROUP_NAME).unwrap(); + + let mut clients = harness + .root_clients(2) + .await + .expect("Failed to create root clients"); + let first_member = clients.remove(0); + let second_member = clients.remove(0); + + first_member.create_stream(STREAM_NAME).await.unwrap(); + first_member + .create_topic( + &stream_id, + TOPIC_NAME, + &TopicCreateOptions { + partitions_count: Some(1), + message_expiry: Some(IggyExpiry::NeverExpire), + ..TopicCreateOptions::default() + }, + ) + .await + .unwrap(); + first_member + .create_consumer_group(&stream_id, &topic_id, CONSUMER_GROUP_NAME) + .await + .unwrap(); + + // The first member to join keeps the topic's only partition. + first_member + .join_consumer_group(&stream_id, &topic_id, &group_id) + .await + .unwrap(); + second_member + .join_consumer_group(&stream_id, &topic_id, &group_id) + .await + .unwrap(); + + let consumer = Consumer::group(group_id.clone()); + let owner_poll = first_member + .poll_messages( + &stream_id, + &topic_id, + None, + &consumer, + &PollingStrategy::next(), + 1, + false, + ) + .await + .unwrap(); + assert!(owner_poll.messages.is_empty()); + assert_eq!( + owner_poll.partition_id, 0, + "an empty poll of an owned partition must echo its id" + ); + + let surplus_poll = second_member + .poll_messages( + &stream_id, + &topic_id, + None, + &consumer, + &PollingStrategy::next(), + 1, + false, + ) + .await + .unwrap(); + assert!(surplus_poll.messages.is_empty()); + assert_eq!( + surplus_poll.partition_id, NO_ASSIGNED_PARTITION, + "a member without partitions must report the sentinel" + ); +} + +// A member built with `do_not_auto_join_consumer_group()` relies on the caller for the join, so +// polling must not wait for a join the consumer itself never performs. +#[iggy_harness( + test_client_transport = [Tcp, WebSocket, Quic], + server(heartbeat.enabled = true, heartbeat.interval = "60s") +)] +async fn given_member_that_does_not_auto_join_when_joined_by_the_caller_should_receive_messages( + harness: &TestHarness, +) { + let stream_id = Identifier::named(STREAM_NAME).unwrap(); + let topic_id = Identifier::named(TOPIC_NAME).unwrap(); + let group_id = Identifier::named(CONSUMER_GROUP_NAME).unwrap(); + + let client = harness.new_client().await.expect("Failed to create client"); + client + .login_user(DEFAULT_ROOT_USERNAME, DEFAULT_ROOT_PASSWORD) + .await + .unwrap(); + client.create_stream(STREAM_NAME).await.unwrap(); + client + .create_topic( + &stream_id, + TOPIC_NAME, + &TopicCreateOptions { + partitions_count: Some(1), + message_expiry: Some(IggyExpiry::NeverExpire), + ..TopicCreateOptions::default() + }, + ) + .await + .unwrap(); + client + .create_consumer_group(&stream_id, &topic_id, CONSUMER_GROUP_NAME) + .await + .unwrap(); + let mut messages = vec![IggyMessage::from_str("message").unwrap()]; + client + .send_messages( + &stream_id, + &topic_id, + &Partitioning::partition_id(0), + &mut messages, + ) + .await + .unwrap(); + + // Membership is per connection, so the join goes through the client the consumer uses. + client + .join_consumer_group(&stream_id, &topic_id, &group_id) + .await + .unwrap(); + + let mut consumer = client + .consumer_group(CONSUMER_GROUP_NAME, STREAM_NAME, TOPIC_NAME) + .unwrap() + .batch_length(1) + .do_not_auto_join_consumer_group() + .build(); + consumer.init().await.unwrap(); + + let received = timeout(RESOLVE_TIMEOUT, consumer.next()) + .await + .expect("a member joined by the caller must poll instead of waiting for a join") + .expect("consumer stream should remain open") + .expect("the member must receive the message from its assigned partition"); + assert_eq!(received.message.payload, "message"); +} + // End-to-end wire pin for the consumer-group join/leave error ladder. The // metadata STM unit tests pin the committed result codes; this pins that // the server actually emits them over the wire, so a client observes the same @@ -252,3 +411,120 @@ fn assert_rejected(result: Result<(), IggyError>, expected_code: u32, context: & error.as_code(), ); } + +// Membership is per connection: a reconnect registers a new client identity +// that the coordinator knows as a member of nothing, so the transport must not +// carry the old session's membership and assignment into the new one. Carried +// over, the first poll would run against a stale assignment instead of +// reporting the missing membership straight away. +#[iggy_harness( + test_client_transport = [Tcp, WebSocket, Quic], + server(heartbeat.enabled = true, heartbeat.interval = "60s") +)] +async fn given_group_member_when_session_reset_should_forget_membership(harness: &TestHarness) { + let stream_id = Identifier::named(STREAM_NAME).unwrap(); + let topic_id = Identifier::named(TOPIC_NAME).unwrap(); + let group_id = Identifier::named(CONSUMER_GROUP_NAME).unwrap(); + + let client = harness.new_client().await.expect("Failed to create client"); + client + .login_user(DEFAULT_ROOT_USERNAME, DEFAULT_ROOT_PASSWORD) + .await + .unwrap(); + client.create_stream(STREAM_NAME).await.unwrap(); + client + .create_topic( + &stream_id, + TOPIC_NAME, + &TopicCreateOptions { + partitions_count: Some(1), + message_expiry: Some(IggyExpiry::NeverExpire), + ..TopicCreateOptions::default() + }, + ) + .await + .unwrap(); + client + .create_consumer_group(&stream_id, &topic_id, CONSUMER_GROUP_NAME) + .await + .unwrap(); + client + .join_consumer_group(&stream_id, &topic_id, &group_id) + .await + .unwrap(); + + let consumer = Consumer::group(group_id.clone()); + // The first group poll syncs the assignment, which is what registers the + // membership in the transport cache. + client + .poll_messages( + &stream_id, + &topic_id, + None, + &consumer, + &PollingStrategy::next(), + 1, + false, + ) + .await + .unwrap(); + + let state = consumer_group_state(&client).await; + // The key the join and the group poll build from the same identifiers. + let key = format!("{stream_id}|{topic_id}|{group_id}"); + assert!( + state.is_registered(&key), + "the sync must register the membership" + ); + assert!( + state.has_assignment(&key), + "the sole member must hold the topic's only partition" + ); + + client.disconnect().await.unwrap(); + // Reconnects through the transport rather than `IggyClient::connect`: the + // heartbeat the latter spawns re-syncs every registered group on its own, + // and this asserts on the reset alone. + client.client().read().await.connect().await.unwrap(); + client + .login_user(DEFAULT_ROOT_USERNAME, DEFAULT_ROOT_PASSWORD) + .await + .unwrap(); + + assert!( + state.registered_groups().is_empty(), + "a reconnected client must not carry the old session's membership" + ); + assert!(!state.is_registered(&key)); + assert!(!state.has_assignment(&key)); + + // The new identity never joined, and the poll must say so instead of + // running against the old assignment. + let poll = client + .poll_messages( + &stream_id, + &topic_id, + None, + &consumer, + &PollingStrategy::next(), + 1, + false, + ) + .await; + assert!( + matches!(poll, Err(IggyError::ConsumerGroupMemberNotFound(..))), + "expected ConsumerGroupMemberNotFound for a poll without a rejoin, got {poll:?}" + ); +} + +/// The consumer-group cache lives on the transport `IggyClient` wraps. +async fn consumer_group_state(client: &IggyClient) -> Arc { + match &*client.client().read().await { + ClientWrapper::Tcp(client) => client.consumer_group_state(), + ClientWrapper::Quic(client) => client.consumer_group_state(), + ClientWrapper::WebSocket(client) => client.consumer_group_state(), + ClientWrapper::Http(_) | ClientWrapper::Iggy(_) => { + panic!("the consumer-group cache is a binary-transport concern") + } + } +} diff --git a/core/integration/tests/sdk/consumer_shutdown.rs b/core/integration/tests/sdk/consumer_shutdown.rs new file mode 100644 index 0000000000..172a569858 --- /dev/null +++ b/core/integration/tests/sdk/consumer_shutdown.rs @@ -0,0 +1,112 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use std::str::FromStr; + +use futures::StreamExt; +use iggy::prelude::*; +use integration::iggy_harness; +use tokio::time::{Duration, Instant, timeout}; + +const STREAM_NAME: &str = "consumer-shutdown-stream"; +const TOPIC_NAME: &str = "consumer-shutdown-topic"; +const CONSUMER_NAME: &str = "consumer-shutdown-consumer"; +const PARTITION_ID: u32 = 0; +const POLL_TIMEOUT: Duration = Duration::from_secs(10); +const OFFSET_DRAIN_TIMEOUT: Duration = Duration::from_secs(5); + +// `AutoCommit::Disabled` leaves every commit to the caller, so the final flush of `shutdown()` +// must not run either: it would commit a message whose handler failed. +#[iggy_harness] +async fn given_disabled_auto_commit_when_shutdown_should_not_store_the_reading_position( + harness: &TestHarness, +) { + let client = harness.root_client().await.unwrap(); + let stream_id = Identifier::named(STREAM_NAME).unwrap(); + let topic_id = Identifier::named(TOPIC_NAME).unwrap(); + + client.create_stream(STREAM_NAME).await.unwrap(); + client + .create_topic( + &stream_id, + TOPIC_NAME, + &TopicCreateOptions { + partitions_count: Some(1), + message_expiry: Some(IggyExpiry::NeverExpire), + ..TopicCreateOptions::default() + }, + ) + .await + .unwrap(); + let mut messages = vec![ + IggyMessage::from_str("message_1").unwrap(), + IggyMessage::from_str("message_2").unwrap(), + ]; + client + .send_messages( + &stream_id, + &topic_id, + &Partitioning::partition_id(PARTITION_ID), + &mut messages, + ) + .await + .unwrap(); + + // Both messages must come in one batch: with nothing committed, `next()` serves the same + // batch again and the consumer's own filter drops it, so a second poll would stall. + let mut consumer = client + .consumer(CONSUMER_NAME, STREAM_NAME, TOPIC_NAME, PARTITION_ID) + .unwrap() + .auto_commit(AutoCommit::Disabled) + .batch_length(2) + .offset_drain_timeout(IggyDuration::from(OFFSET_DRAIN_TIMEOUT)) + .build(); + consumer.init().await.unwrap(); + + for expected_offset in [0, 1] { + let received = timeout(POLL_TIMEOUT, consumer.next()) + .await + .expect("Consumer should receive a message before timeout") + .expect("Consumer stream should remain open") + .expect("Consumer should poll a message"); + assert_eq!(received.message.header.offset, expected_offset); + } + assert_eq!(consumer.get_last_consumed_offset(PARTITION_ID), Some(1)); + + // The store task has to exit on the shutdown flag. One that never exits only costs the drain + // timeout and a warning, so the stored offset alone would not catch it. + let shutdown_started = Instant::now(); + consumer.shutdown().await.unwrap(); + assert!( + shutdown_started.elapsed() < OFFSET_DRAIN_TIMEOUT, + "shutdown() must not wait for the offset drain timeout" + ); + + let stored_offset = client + .get_consumer_offset( + &Consumer::new(Identifier::named(CONSUMER_NAME).unwrap()), + &stream_id, + &topic_id, + Some(PARTITION_ID), + ) + .await + .unwrap(); + assert!( + stored_offset.is_none(), + "nothing must be stored under AutoCommit::Disabled, got {stored_offset:?}" + ); +} diff --git a/core/integration/tests/sdk/mod.rs b/core/integration/tests/sdk/mod.rs index 065b257a76..0cec5f837c 100644 --- a/core/integration/tests/sdk/mod.rs +++ b/core/integration/tests/sdk/mod.rs @@ -18,6 +18,7 @@ mod consumer_group; mod consumer_group_membership; mod consumer_offset; +mod consumer_shutdown; mod disconnect_relogin; mod hello_world; mod http_refresh; diff --git a/core/sdk/src/client_wrappers/binary_message_client.rs b/core/sdk/src/client_wrappers/binary_message_client.rs index eb3eed94f9..3fb72563a0 100644 --- a/core/sdk/src/client_wrappers/binary_message_client.rs +++ b/core/sdk/src/client_wrappers/binary_message_client.rs @@ -34,16 +34,38 @@ impl MessageClient for ClientWrapper { strategy: &PollingStrategy, count: u32, auto_commit: bool, + ) -> Result { + self.poll_messages_with_strategy_for( + stream_id, + topic_id, + partition_id, + consumer, + &|_: u32| *strategy, + count, + auto_commit, + ) + .await + } + + async fn poll_messages_with_strategy_for( + &self, + stream_id: &Identifier, + topic_id: &Identifier, + partition_id: Option, + consumer: &Consumer, + strategy_for: &(dyn Fn(u32) -> PollingStrategy + Send + Sync), + count: u32, + auto_commit: bool, ) -> Result { match self { ClientWrapper::Iggy(client) => { client - .poll_messages( + .poll_messages_with_strategy_for( stream_id, topic_id, partition_id, consumer, - strategy, + strategy_for, count, auto_commit, ) @@ -51,12 +73,12 @@ impl MessageClient for ClientWrapper { } ClientWrapper::Http(client) => { client - .poll_messages( + .poll_messages_with_strategy_for( stream_id, topic_id, partition_id, consumer, - strategy, + strategy_for, count, auto_commit, ) @@ -64,12 +86,12 @@ impl MessageClient for ClientWrapper { } ClientWrapper::Tcp(client) => { client - .poll_messages( + .poll_messages_with_strategy_for( stream_id, topic_id, partition_id, consumer, - strategy, + strategy_for, count, auto_commit, ) @@ -77,12 +99,12 @@ impl MessageClient for ClientWrapper { } ClientWrapper::Quic(client) => { client - .poll_messages( + .poll_messages_with_strategy_for( stream_id, topic_id, partition_id, consumer, - strategy, + strategy_for, count, auto_commit, ) @@ -90,12 +112,12 @@ impl MessageClient for ClientWrapper { } ClientWrapper::WebSocket(client) => { client - .poll_messages( + .poll_messages_with_strategy_for( stream_id, topic_id, partition_id, consumer, - strategy, + strategy_for, count, auto_commit, ) diff --git a/core/sdk/src/clients/binary_message.rs b/core/sdk/src/clients/binary_message.rs index db722a0066..e83321bbd4 100644 --- a/core/sdk/src/clients/binary_message.rs +++ b/core/sdk/src/clients/binary_message.rs @@ -36,6 +36,28 @@ impl MessageClient for IggyClient { strategy: &PollingStrategy, count: u32, auto_commit: bool, + ) -> Result { + self.poll_messages_with_strategy_for( + stream_id, + topic_id, + partition_id, + consumer, + &|_: u32| *strategy, + count, + auto_commit, + ) + .await + } + + async fn poll_messages_with_strategy_for( + &self, + stream_id: &Identifier, + topic_id: &Identifier, + partition_id: Option, + consumer: &Consumer, + strategy_for: &(dyn Fn(u32) -> PollingStrategy + Send + Sync), + count: u32, + auto_commit: bool, ) -> Result { if count == 0 { return Err(IggyError::InvalidMessagesCount); @@ -45,12 +67,12 @@ impl MessageClient for IggyClient { .client .read() .await - .poll_messages( + .poll_messages_with_strategy_for( stream_id, topic_id, partition_id, consumer, - strategy, + strategy_for, count, auto_commit, ) diff --git a/core/sdk/src/clients/consumer.rs b/core/sdk/src/clients/consumer.rs index a7bf549836..1f38f2c76e 100644 --- a/core/sdk/src/clients/consumer.rs +++ b/core/sdk/src/clients/consumer.rs @@ -26,8 +26,8 @@ use iggy_common::{ }; use iggy_common::{ Consumer, ConsumerKind, DiagnosticEvent, EncryptorKind, IdKind, Identifier, IggyDuration, - IggyError, IggyMessage, IggyTimestamp, NonZeroIggyDuration, PolledMessages, PollingKind, - PollingStrategy, + IggyError, IggyMessage, IggyTimestamp, NO_ASSIGNED_PARTITION, NonZeroIggyDuration, + PolledMessages, PollingKind, PollingStrategy, }; use std::collections::VecDeque; use std::fmt::{self, Debug, Formatter}; @@ -400,7 +400,8 @@ unsafe impl Sync for IggyConsumer {} /// .store_offset(received.message.header.offset, Some(received.partition_id)) /// .await?; /// } -/// // No shutdown() here: it would commit the reading position, failed message included. +/// +/// consumer.shutdown().await?; /// # Ok(()) /// # } /// ``` @@ -420,13 +421,14 @@ unsafe impl Sync for IggyConsumer {} /// creating the group first if [`create_consumer_group_if_not_exists()`] is set (the default). /// It rejoins on its own after a reconnect and whenever the server reports that its membership /// is gone. -/// - Until the join has succeeded the consumer does not poll. It re-checks every -/// [`polling_retry_interval()`] and polls once joined. +/// - Such a member does not poll until it is in the group. A join that fails is yielded as +/// `Some(Err(..))` after [`polling_retry_interval()`], and the next call tries again. /// - Partitions are redistributed whenever members join or leave, so a member reads different /// partitions over time and messages from several partitions interleave in its stream. /// - More members than partitions leaves the surplus members without partitions. Such a member -/// still polls, one group sync round trip per attempt, so give it a [`poll_interval()`]. The -/// partition count of the topic is the ceiling on how far one group can be scaled out. +/// keeps asking the server for an assignment, parking for [`polling_retry_interval()`] between +/// attempts. The partition count of the topic is the ceiling on how far one group can be +/// scaled out. /// - The group shares one set of stored offsets, kept under the group name. Thus, /// a partition taken over by another member continues where the previous one /// committed. @@ -452,13 +454,12 @@ unsafe impl Sync for IggyConsumer {} /// | [`PollingStrategy::timestamp()`] | the first message at or after a given point in time | /// /// Only [`PollingStrategy::next()`] consults the offset stored on the server. -/// Use this if you want to resume where a previous run stopped. The other four are starting points -/// for the first request only. From the second request onwards, the consumer asks for whatever -/// follows the last message it handed over, and it keeps one such continuation point for all -/// partitions. That makes them fit for standalone consumers only: a group member polls a different -/// partition on every request, so a continuation point taken from one partition is applied to the -/// next, where it skips or repeats messages, and under the default [`auto_commit()`] the skipped -/// range is committed as read. +/// Use this if you want to resume where a previous run stopped. The other four are the starting +/// point for the first request to each partition. From then on the consumer asks that partition +/// for whatever follows the last message it handed over from it, and it keeps that position per +/// partition. A partition that moves to another member and back therefore continues from this +/// consumer's own position, not from where the other member got to, so a rebalance can repeat +/// messages under these strategies, which ignore the group's stored offsets by definition. /// /// [`StreamExt::next`] yields `None` once [`shutdown()`](Self::shutdown) has been called, and never /// otherwise: not when the topic is empty and not while the client is disconnected. A request that @@ -495,10 +496,10 @@ unsafe impl Sync for IggyConsumer {} /// /// | Setting | Commits | /// | --- | --- | -/// | [`AutoCommit::Disabled`] | never on its own, decide manually with [`store_offset()`](Self::store_offset). [`shutdown()`](Self::shutdown) still commits the reading position | +/// | [`AutoCommit::Disabled`] | never, not even on [`shutdown()`](Self::shutdown). Commit with [`store_offset()`](Self::store_offset) | /// | [`AutoCommit::Interval`] | on every tick, the reading position of every partition read so far | /// | [`AutoCommitWhen::PollingMessages`] | sends the commit with the poll request itself, before your code sees the batch | -/// | [`AutoCommitWhen::ConsumingEachMessage`] | queued just before every message is handed over to the calling code, at one round trip per message and without backpressure | +/// | [`AutoCommitWhen::ConsumingEachMessage`] | queued just before every message is handed over to the calling code. Commits queued faster than they are sent collapse into the latest one per partition | /// | [`AutoCommitWhen::ConsumingEveryNthMessage`] | queued just before a message whose offset divides by `n` is handed over | /// | [`AutoCommitWhen::ConsumingAllMessages`] | queued when the buffer of the current batch runs empty | /// | [`AutoCommitAfter`] variants | once the handler returned, `Ok` or `Err`, and only under [`IggyConsumerMessageExt::consume_messages`], see below | @@ -529,9 +530,8 @@ unsafe impl Sync for IggyConsumer {} /// and store the offset using [`Self::store_offset()`] after handling a message. Every other /// setting except the plain [`AutoCommit::After`] variants can commit a message before your /// handler is done with it, so a crash in the handler loses it. [`AutoCommit::IntervalOrAfter`] -/// still commits on its interval tick. [`shutdown()`](Self::shutdown) commits the reading -/// position under every setting, [`AutoCommit::Disabled`] included, so it also commits a message -/// whose handler failed. +/// still commits on its interval tick, and [`shutdown()`](Self::shutdown) commits the reading +/// position under every setting but [`AutoCommit::Disabled`], a failed message included. /// /// # Options and defaults /// @@ -540,18 +540,18 @@ unsafe impl Sync for IggyConsumer {} /// /// | Option | Default | Controls | /// | --- | --- | --- | -/// | [`stream()`], [`topic()`], [`partition()`] | the values passed to the entry point | what is read. [`partition()`] is for standalone consumers, on a group member it pins every poll to that partition instead of the server's assignment | +/// | [`stream()`], [`topic()`], [`partition()`] | the values passed to the entry point | what is read. [`partition()`] is for standalone consumers, a group member ignores it with a warning and reads the server's assignment | /// | [`batch_length()`] | 1000 | messages fetched per request | /// | [`poll_interval()`] | none | smallest gap between two requests | -/// | [`polling_strategy()`] | [`PollingStrategy::next()`] | where reading starts. Anything but [`PollingStrategy::next()`] is for standalone consumers only | -/// | [`auto_commit()`] | [`AutoCommit::IntervalOrWhen`], one second, [`AutoCommitWhen::PollingMessages`] | when offsets are committed. [`commit_failed_messages()`] is a synonym for [`AutoCommit::Disabled`] | +/// | [`polling_strategy()`] | [`PollingStrategy::next()`] | where reading each partition starts | +/// | [`auto_commit()`] | [`AutoCommit::IntervalOrWhen`], one second, [`AutoCommitWhen::PollingMessages`] | when offsets are committed | /// | [`allow_replay()`] | off | whether a message can be handed over again | -/// | [`auto_join_consumer_group()`] | on | joining the group during [`init()`](Self::init) and after a reconnect. A group member built with [`do_not_auto_join_consumer_group()`] never polls, since polling waits for the join | +/// | [`auto_join_consumer_group()`] | on | joining the group during [`init()`](Self::init) and again whenever the membership is lost. With [`do_not_auto_join_consumer_group()`] joining is up to the caller, and a poll without a membership fails with [`IggyError::ConsumerGroupMemberNotFound`] | /// | [`create_consumer_group_if_not_exists()`] | on | creating the group when it is missing | -/// | [`polling_retry_interval()`] | one second | wait between attempts while polling is blocked | +/// | [`polling_retry_interval()`] | one second | wait between attempts while polling is blocked or the member holds no partitions | /// | [`init_retries()`] | none, one second apart | retries when the stream or topic is missing at [`init()`](Self::init) | /// | [`offset_drain_timeout()`] | five seconds | how long [`shutdown()`](Self::shutdown) waits for pending commits | -/// | [`encryptor()`] | inherited from the client | decrypting payloads and user headers | +/// | [`encryptor()`] | inherited from the client | decrypting payloads and user headers, see [Encryption](#encryption) | /// /// The switches have inverse setters as well, such as [`without_poll_interval()`], /// [`without_encryptor()`], [`do_not_auto_join_consumer_group()`] and @@ -565,12 +565,12 @@ unsafe impl Sync for IggyConsumer {} /// client share unless one of them overrides it on its builder. Without an encryptor the consumer /// yields payloads as stored, encrypted or not. /// -/// A message that cannot be decrypted is yielded as an `Err` and the whole batch is dropped. What -/// happens next depends on [`auto_commit()`]. Under [`AutoCommitWhen::PollingMessages`] (the -/// default) the server committed the batch with the poll, so it is skipped for good. Under every -/// other setting the next request fetches the same batch and fails the same way until -/// [`store_offset()`](Self::store_offset) moves the offset past it. Pick a setting other than -/// [`AutoCommitWhen::PollingMessages`] if a batch that fails to decrypt must not be lost silently. +/// A message that cannot be decrypted is yielded as an `Err` and the whole batch is dropped. The +/// next request fetches the same batch and fails the same way until +/// [`store_offset()`](Self::store_offset) moves the offset past it. Under +/// [`AutoCommitWhen::PollingMessages`] the server would have committed the batch with the poll and +/// skipped it for good, so [`init()`](Self::init) rejects that setting, the default included, with +/// [`IggyError::InvalidConfiguration`] when the consumer has an encryptor. /// /// # Concurrency /// @@ -587,11 +587,11 @@ unsafe impl Sync for IggyConsumer {} /// # Shutting down /// /// Call [`shutdown()`](Self::shutdown) once done consuming. It drains the commit tasks, commits -/// the reading position of every partition, [`AutoCommit::Disabled`] included, and leaves the -/// consumer group. Dropping an `IggyConsumer` instead skips that final commit and the group -/// leave, so the server reassigns the member's partitions only once the connection is gone. -/// Commits already queued are still sent. Neither stops the lifecycle task, which runs until the -/// client shuts down. +/// the reading position of every partition unless [`auto_commit()`] is [`AutoCommit::Disabled`], +/// leaves the consumer group and stops the background tasks. Dropping an `IggyConsumer` instead +/// skips the final commit and the group leave, so the server reassigns the member's partitions +/// only once the connection is gone. Commits already queued are still sent and the background +/// tasks still stop. /// /// [`IggyClient`]: crate::prelude::IggyClient /// [`IggyClient::consumer()`]: crate::prelude::IggyClient::consumer @@ -604,7 +604,6 @@ unsafe impl Sync for IggyConsumer {} /// [`auto_join_consumer_group()`]: crate::prelude::IggyConsumerBuilder::auto_join_consumer_group /// [`batch_length()`]: crate::prelude::IggyConsumerBuilder::batch_length /// [`build()`]: crate::prelude::IggyConsumerBuilder::build -/// [`commit_failed_messages()`]: crate::prelude::IggyConsumerBuilder::commit_failed_messages /// [`create_consumer_group_if_not_exists()`]: crate::prelude::IggyConsumerBuilder::create_consumer_group_if_not_exists /// [`do_not_auto_join_consumer_group()`]: crate::prelude::IggyConsumerBuilder::do_not_auto_join_consumer_group /// [`do_not_create_consumer_group_if_not_exists()`]: crate::prelude::IggyConsumerBuilder::do_not_create_consumer_group_if_not_exists @@ -632,6 +631,9 @@ pub struct IggyConsumer { topic_id: Arc, partition_id: Option, polling_strategy: PollingStrategy, + /// The next offset to ask each partition for. Empty under [`PollingStrategy::next()`], which + /// leaves the continuation to the offset stored on the server. + next_offsets: Arc>, poll_interval_micros: u64, batch_length: u32, auto_commit: AutoCommit, @@ -643,10 +645,14 @@ pub struct IggyConsumer { poll_future: Option, buffered_messages: VecDeque, encryptor: Option>, - store_offset_sender: flume::Sender<(u32, u64)>, + /// The latest offset each message trigger asked to commit, per partition. The store task + /// drains it, so a burst of triggers costs one round trip per partition instead of one each. + pending_commits: Arc>, + store_offset_notify: Arc, store_offset_task: Option>, background_commit_task: Option>, background_commit_notify: Arc, + events_task: Option>, store_offset_after_each_message: bool, store_offset_after_all_messages: bool, store_after_every_nth_message: u64, @@ -680,8 +686,15 @@ impl IggyConsumer { allow_replay: bool, offset_drain_timeout: IggyDuration, ) -> Self { - let (store_offset_sender, _) = flume::unbounded(); let is_consumer_group = consumer.kind == ConsumerKind::ConsumerGroup; + let partition_id = if is_consumer_group && partition_id.is_some() { + warn!( + "Consumer group member: {consumer_name} ignores the partition set on the builder and reads the server's assignment" + ); + None + } else { + partition_id + }; let consumer = Arc::new(consumer); let stream_id = Arc::new(stream_id); let topic_id = Arc::new(topic_id); @@ -706,6 +719,7 @@ impl IggyConsumer { topic_id, partition_id, polling_strategy, + next_offsets: Arc::new(DashMap::new()), poll_interval_micros: polling_interval.map_or(0, |interval| interval.as_micros()), state, current_offsets: Arc::new(DashMap::new()), @@ -721,10 +735,12 @@ impl IggyConsumer { create_consumer_group_if_not_exists, buffered_messages: VecDeque::new(), encryptor, - store_offset_sender, + pending_commits: Arc::new(DashMap::new()), + store_offset_notify: Arc::new(Notify::new()), store_offset_task: None, background_commit_task: None, background_commit_notify: Arc::new(Notify::new()), + events_task: None, store_offset_after_each_message: matches!( auto_commit, AutoCommit::When(AutoCommitWhen::ConsumingEachMessage) @@ -772,11 +788,11 @@ impl IggyConsumer { &self.stream_id } - /// Returns the partition the most recent poll response came from. + /// Returns the partition the most recent poll response with messages came from, or `0` before + /// the first one. /// - /// This is `0` before the first response, and an empty response can report `0` as well. For a - /// consumer group the value changes over time, as the server hands different partitions to - /// this member. To commit for the partition a message came from, pass + /// For a consumer group the value changes over time, as the server hands different partitions + /// to this member. To commit for the partition a message came from, pass /// [`ReceivedMessage::partition_id`] to [`store_offset()`](Self::store_offset) instead. pub fn partition_id(&self) -> u32 { self.state.partition_id() @@ -874,14 +890,15 @@ impl IggyConsumer { /// # Lifecycle events /// /// Calling init spawns a background task that listens for lifecycle changes ([`DiagnosticEvent`]s) of the - /// client connection. It runs until the client shuts down and is not stopped by - /// [`shutdown()`](Self::shutdown). + /// client connection. It runs until [`shutdown()`](Self::shutdown) or until the client shuts + /// down. /// - [`DiagnosticEvent::Connected`]: a fresh connection has not joined anything yet. /// Polling resumes immediately only for a consumer that is not a group member. - /// - [`DiagnosticEvent::SignedIn`]: re-enables polling. A group member signing in after a - /// reconnect rejoins its group first and only polls once that succeeded. A failed rejoin is - /// logged and leaves polling disabled until the next reconnect or an explicit login. - /// - [`DiagnosticEvent::Disconnected`] and [`DiagnosticEvent::SignedOut`] disable polling. + /// - [`DiagnosticEvent::SignedIn`]: re-enables polling. A group member whose membership is + /// gone rejoins its group on the next poll, before the request goes out. A failed rejoin is + /// yielded as a poll error and tried again on the poll after. + /// - [`DiagnosticEvent::Disconnected`] and [`DiagnosticEvent::SignedOut`] disable polling and + /// forget the group membership. /// - [`DiagnosticEvent::Shutdown`] disables polling and terminates the background task listening /// for lifecycle changes. It does not flush in-flight commits; that only happens when /// [`shutdown()`](Self::shutdown) itself is called. @@ -896,7 +913,8 @@ impl IggyConsumer { /// [`AutoCommit::IntervalOrWhen`], [`AutoCommit::IntervalOrAfter`]). Every tick it stores the /// reading position of every partition read so far. /// - An offset store task, always. It sends the commits queued by the [`AutoCommitWhen`] and - /// [`AutoCommitAfter`] triggers one at a time and stays idle under [`AutoCommit::Disabled`]. + /// [`AutoCommitAfter`] triggers, keeping only the latest queued offset per partition, and + /// stays idle under [`AutoCommit::Disabled`]. /// /// Both skip an offset that is not ahead of this consumer's own record of what it stored /// ([`get_last_stored_offset()`](Self::get_last_stored_offset)). Only offset `0` is always sent. @@ -906,6 +924,10 @@ impl IggyConsumer { /// /// # Errors /// + /// - [`IggyError::InvalidConfiguration`] when the consumer has an encryptor and + /// [`auto_commit()`](crate::prelude::IggyConsumerBuilder::auto_commit) is + /// [`AutoCommitWhen::PollingMessages`], checked before anything is sent. See + /// [Encryption](IggyConsumer#encryption). /// - [`IggyError::StreamNameNotFound`] or [`IggyError::TopicNameNotFound`] when the /// stream or the topic still does not exist once the retries are exhausted. /// - [`IggyError::ConsumerGroupNameNotFound`] when the consumer group does not exist @@ -922,6 +944,13 @@ impl IggyConsumer { let topic_id = self.topic_id.clone(); let consumer_name = &self.consumer_name; + if self.encryptor.is_some() && self.auto_commit_after_polling { + error!( + "Consumer: {consumer_name} has an encryptor and auto-commit on polling. That commits a batch before it is decrypted, so a batch that fails to decrypt would be lost. Pick another auto-commit setting." + ); + return Err(IggyError::InvalidConfiguration); + } + info!( "Initializing consumer: {consumer_name} for stream: {stream_id}, topic: {topic_id}..." ); @@ -993,8 +1022,10 @@ impl IggyConsumer { } } - self.subscribe_events().await; - // No-op if either is_consumer_group or auto_join_consumer_group is false + // A retried init() after a failed join must not leave the earlier task behind. + if let Some(previous) = self.events_task.replace(self.subscribe_events().await) { + previous.abort(); + } self.init_consumer_group().await?; match self.auto_commit { @@ -1006,24 +1037,7 @@ impl IggyConsumer { _ => {} } - let state = self.state.clone(); - let (store_offset_sender, store_offset_receiver) = flume::unbounded(); - self.store_offset_sender = store_offset_sender; - - // Message-triggered commits from `poll_next` and `consume_messages` queue here and go out - // one at a time. The interval task above and the poll request's own `auto_commit` flag - // are the other commit paths. - self.store_offset_task = Some(tokio::spawn(async move { - while let Ok((partition_id, offset)) = store_offset_receiver.recv_async().await { - trace!( - "Received offset to store: {offset}, partition ID: {partition_id}, stream: {}, topic: {}", - state.stream_id, state.topic_id - ); - _ = state - .store_consumer_offset(partition_id, offset, false) - .await - } - })); + self.store_offset_task = Some(self.store_pending_commits_in_background()); self.initialized = true; info!( @@ -1062,12 +1076,47 @@ impl IggyConsumer { }) } + /// Sends the commits queued by the message triggers of `poll_next` and `consume_messages`. + /// The interval task and the poll request's own `auto_commit` flag are the other commit paths. + fn store_pending_commits_in_background(&self) -> JoinHandle<()> { + let state = self.state.clone(); + let pending_commits = self.pending_commits.clone(); + let shutdown = self.shutdown.clone(); + let notify = self.store_offset_notify.clone(); + tokio::spawn(async move { + loop { + notify.notified().await; + // Keys first, so no map guard is held across a round trip. An offset queued + // meanwhile stays in the map and the permit its trigger leaves wakes the next turn. + let partitions: Vec = + pending_commits.iter().map(|entry| *entry.key()).collect(); + for partition_id in partitions { + let Some((_, offset)) = pending_commits.remove(&partition_id) else { + continue; + }; + _ = state + .store_consumer_offset(partition_id, offset, false) + .await; + } + if shutdown.load(ORDERING) && pending_commits.is_empty() { + break; + } + } + }) + } + + /// Queues a commit for the store task. A later offset for the same partition replaces a + /// queued one that has not been sent yet. pub(crate) fn send_store_offset(&self, partition_id: u32, offset: u64) { - if let Err(error) = self.store_offset_sender.send((partition_id, offset)) { + if !self.initialized || self.shutdown.load(ORDERING) { error!( - "Failed to send offset to store: {error}, please verify if `init()` on IggyConsumer object has been called." + "Offset: {offset} for partition ID: {partition_id} was not queued for storing, consumer: {} is not initialized or has been shut down.", + self.consumer_name ); + return; } + self.pending_commits.insert(partition_id, offset); + self.store_offset_notify.notify_one(); } async fn init_consumer_group(&self) -> Result<(), IggyError> { @@ -1098,7 +1147,9 @@ impl IggyConsumer { .await } - async fn subscribe_events(&self) { + /// Keeps the polling flags in step with the connection. Joining the group again after a + /// reconnect is left to the poll path, which retries it and reports a failure as a poll error. + async fn subscribe_events(&self) -> JoinHandle<()> { trace!("Subscribing to diagnostic events"); let mut receiver; { @@ -1107,17 +1158,8 @@ impl IggyConsumer { } let is_consumer_group = self.is_consumer_group; - let can_join_consumer_group = is_consumer_group && self.auto_join_consumer_group; - let client = self.client.clone(); - let create_consumer_group_if_not_exists = self.create_consumer_group_if_not_exists; - let stream_id = self.stream_id.clone(); - let topic_id = self.topic_id.clone(); - let consumer = self.consumer.clone(); - let consumer_name = self.consumer_name.clone(); let can_poll = self.can_poll.clone(); let joined_consumer_group = self.joined_consumer_group.clone(); - let mut reconnected = false; - let mut disconnected = false; tokio::spawn(async move { while let Some(event) = receiver.next().await { @@ -1129,69 +1171,19 @@ impl IggyConsumer { can_poll.store(false, ORDERING); break; } - DiagnosticEvent::Connected => { trace!("Connected to the server"); joined_consumer_group.store(false, ORDERING); if !is_consumer_group { can_poll.store(true, ORDERING); } - if disconnected { - reconnected = true; - disconnected = false; - } } DiagnosticEvent::Disconnected => { - disconnected = true; - reconnected = false; joined_consumer_group.store(false, ORDERING); can_poll.store(false, ORDERING); warn!("Disconnected from the server"); } DiagnosticEvent::SignedIn => { - if !is_consumer_group { - can_poll.store(true, ORDERING); - continue; - } - - if !can_join_consumer_group { - can_poll.store(true, ORDERING); - trace!("Auto join consumer group is disabled"); - continue; - } - - if !reconnected { - can_poll.store(true, ORDERING); - continue; - } - - if joined_consumer_group.load(ORDERING) { - can_poll.store(true, ORDERING); - continue; - } - - info!( - "Rejoining consumer group: {consumer_name} for stream: {stream_id}, topic: {topic_id}..." - ); - if let Err(error) = Self::initialize_consumer_group( - client.clone(), - create_consumer_group_if_not_exists, - stream_id.clone(), - topic_id.clone(), - consumer.clone(), - &consumer_name, - joined_consumer_group.clone(), - ) - .await - { - error!( - "Failed to join consumer group: {consumer_name} for stream: {stream_id}, topic: {topic_id}. {error}" - ); - continue; - } - info!( - "Rejoined consumer group: {consumer_name} for stream: {stream_id}, topic: {topic_id}" - ); can_poll.store(true, ORDERING); } DiagnosticEvent::SignedOut => { @@ -1200,7 +1192,7 @@ impl IggyConsumer { } } } - }); + }) } fn create_poll_messages_future( @@ -1211,10 +1203,10 @@ impl IggyConsumer { let partition_id = self.partition_id; let consumer = self.consumer.clone(); let polling_strategy = self.polling_strategy; + let next_offsets = self.next_offsets.clone(); let client = self.client.clone(); let count = self.batch_length; let auto_commit_after_polling = self.auto_commit_after_polling; - let auto_commit_enabled = self.auto_commit != AutoCommit::Disabled; let interval = self.poll_interval_micros; let last_polled_at = self.last_polled_at.clone(); let can_poll = self.can_poll.clone(); @@ -1226,39 +1218,74 @@ impl IggyConsumer { let auto_join_consumer_group = self.auto_join_consumer_group; let create_consumer_group_if_not_exists = self.create_consumer_group_if_not_exists; let joined_consumer_group = self.joined_consumer_group.clone(); + let consumer_name = self.consumer_name.clone(); async move { if interval > 0 { Self::wait_before_polling(interval, last_polled_at.load(ORDERING)).await; } - while !can_poll.load(ORDERING) - || (is_consumer_group && !joined_consumer_group.load(ORDERING)) + while !can_poll.load(ORDERING) { + trace!("Cannot poll yet, waiting {retry_interval}..."); + sleep(retry_interval.get_duration()).await; + } + + // A member that joins on its own is in the group before it polls. One built with + // `do_not_auto_join_consumer_group()` polls right away and gets a missing + // membership reported as a poll error. + if is_consumer_group + && auto_join_consumer_group + && !joined_consumer_group.load(ORDERING) + && let Err(error) = Self::initialize_consumer_group( + client.clone(), + create_consumer_group_if_not_exists, + stream_id.clone(), + topic_id.clone(), + consumer.clone(), + &consumer_name, + joined_consumer_group.clone(), + ) + .await { - trace!( - "Cannot poll yet (can_poll={}, joined_cg={}), waiting {retry_interval}...", - can_poll.load(ORDERING), - joined_consumer_group.load(ORDERING) + error!( + "Failed to join consumer group: {consumer_name} for stream: {stream_id}, topic: {topic_id}. {error}" ); sleep(retry_interval.get_duration()).await; + return Err(error); } trace!("Sending poll messages request"); last_polled_at.store(IggyTimestamp::now().into(), ORDERING); + // The map guard is dropped inside `map_or`, and the only writer is `poll_next`, + // which runs after this future has returned, so the lookup cannot block. + let strategy_for = |partition: u32| { + next_offsets + .get(&partition) + .map_or(polling_strategy, |offset| PollingStrategy::offset(*offset)) + }; let polled_messages = client .read() .await - .poll_messages( + .poll_messages_with_strategy_for( &stream_id, &topic_id, partition_id, &consumer, - &polling_strategy, + &strategy_for, count, auto_commit_after_polling, ) .await; + if let Ok(polled) = &polled_messages + && polled.partition_id == NO_ASSIGNED_PARTITION + { + trace!( + "No partition assigned to consumer: {consumer_name}, waiting {retry_interval}..." + ); + sleep(retry_interval.get_duration()).await; + } + if let Ok(mut polled_messages) = polled_messages { if polled_messages.messages.is_empty() { return Ok(polled_messages); @@ -1280,8 +1307,9 @@ impl IggyConsumer { polled_messages .messages .retain(|message| message.header.offset > consumed_offset); + polled_messages.count = polled_messages.messages.len() as u32; if polled_messages.messages.is_empty() { - return Ok(PolledMessages::empty()); + return Ok(polled_messages); } } @@ -1306,44 +1334,6 @@ impl IggyConsumer { "Last consumed offset: {consumed_offset}, current offset: {}, stored offset: {stored_offset}, in partition ID: {partition_id}, topic: {topic_id}, stream: {stream_id}, consumer: {consumer}", polled_messages.current_offset ); - - if !allow_replay - && (has_consumed_offset && polled_messages.current_offset == consumed_offset) - { - trace!( - "No new messages to consume in partition ID: {partition_id}, topic: {topic_id}, stream: {stream_id}, consumer: {consumer}" - ); - if auto_commit_enabled && stored_offset < consumed_offset { - trace!( - "Auto-committing the offset: {consumed_offset} in partition ID: {partition_id}, topic: {topic_id}, stream: {stream_id}, consumer: {consumer}" - ); - client - .read() - .await - .store_consumer_offset( - &consumer, - &stream_id, - &topic_id, - Some(partition_id), - consumed_offset, - ) - .await?; - if let Some(stored_offset_entry) = last_stored_offset.get(&partition_id) { - stored_offset_entry.store(consumed_offset, ORDERING); - } else { - last_stored_offset - .insert(partition_id, AtomicU64::new(consumed_offset)); - } - } - - return Ok(PolledMessages { - messages: vec![], - current_offset: polled_messages.current_offset, - partition_id, - count: 0, - }); - } - return Ok(polled_messages); } @@ -1354,26 +1344,10 @@ impl IggyConsumer { && auto_join_consumer_group && matches!(&error, IggyError::ConsumerGroupMemberNotFound(..)) { - joined_consumer_group.store(false, ORDERING); - let consumer_name = consumer.id.as_string(); info!( - "Consumer group membership was revoked for consumer: {consumer_name}, stream: {stream_id}, topic: {topic_id}. Rejoining..." + "Consumer group membership was revoked for consumer: {consumer_name}, stream: {stream_id}, topic: {topic_id}. Rejoining on the next poll..." ); - if let Err(error) = Self::initialize_consumer_group( - client, - create_consumer_group_if_not_exists, - stream_id, - topic_id, - consumer, - &consumer_name, - joined_consumer_group.clone(), - ) - .await - { - // Allow the next poll to retry rejoining - joined_consumer_group.store(true, ORDERING); - return Err(error); - } + joined_consumer_group.store(false, ORDERING); return Ok(PolledMessages::empty()); } @@ -1565,14 +1539,13 @@ impl Stream for IggyConsumer { } } - // Popping above may have left the buffer empty. - // The next turn will therefore poll messages from the server. - // With `PollingStrategy` the user defines the starting point where to poll from. - // After that, each poll must read the next sequential offset. Hence, strategy is - // set to `PollingKind::Offset` and the next offset to read from is the last consumed message + 1. + // Popping above may have left the buffer empty, so the next turn polls the server. + // `polling_strategy` is only where reading a partition starts; from then on each + // poll continues after the last message handed over from that partition. if self.buffered_messages.is_empty() { if self.polling_strategy.kind != PollingKind::Next { - self.polling_strategy = PollingStrategy::offset(message.header.offset + 1); + self.next_offsets + .insert(partition_id, message.header.offset + 1); } if self.store_offset_after_all_messages { @@ -1604,95 +1577,98 @@ impl Stream for IggyConsumer { while let Some(future) = self.poll_future.as_mut() { match future.poll_unpin(cx) { - Poll::Ready(Ok(mut polled_messages)) => { - let partition_id = polled_messages.partition_id; + Poll::Ready(Ok(polled_messages)) => { + let PolledMessages { + partition_id, + current_offset, + messages, + .. + } = polled_messages; + let mut messages = VecDeque::from(messages); + let Some(mut first) = messages.pop_front() else { + self.poll_future = Some(Box::pin(self.create_poll_messages_future())); + continue; + }; + + // Only a response that carries messages names a partition; an empty one can + // carry a sentinel instead of a real id. self.state .current_partition_id .store(partition_id, ORDERING); - if polled_messages.messages.is_empty() { - self.poll_future = Some(Box::pin(self.create_poll_messages_future())); - } else { - if let Some(ref encryptor) = self.encryptor { - for message in &mut polled_messages.messages { - let offset = message.header.offset; - let payload = encryptor.decrypt(&message.payload); - if let Err(error) = payload { + + if let Some(ref encryptor) = self.encryptor { + for message in std::iter::once(&mut first).chain(messages.iter_mut()) { + let offset = message.header.offset; + let payload = encryptor.decrypt(&message.payload); + if let Err(error) = payload { + self.poll_future = None; + error!( + "Failed to decrypt the message payload at offset: {offset}, partition ID: {partition_id}", + ); + return Poll::Ready(Some(Err(error))); + } + + let payload = payload.unwrap(); + message.payload = Bytes::from(payload); + message.header.payload_length = message.payload.len() as u32; + + if let Some(ref user_headers) = message.user_headers { + let decrypted_headers = encryptor.decrypt(user_headers); + if let Err(error) = decrypted_headers { self.poll_future = None; error!( - "Failed to decrypt the message payload at offset: {offset}, partition ID: {partition_id}", + "Failed to decrypt the message user headers at offset: {offset}, partition ID: {partition_id}", ); return Poll::Ready(Some(Err(error))); } - - let payload = payload.unwrap(); - message.payload = Bytes::from(payload); - message.header.payload_length = message.payload.len() as u32; - - if let Some(ref user_headers) = message.user_headers { - let decrypted_headers = encryptor.decrypt(user_headers); - if let Err(error) = decrypted_headers { - self.poll_future = None; - error!( - "Failed to decrypt the message user headers at offset: {offset}, partition ID: {partition_id}", - ); - return Poll::Ready(Some(Err(error))); - } - let decrypted_headers = decrypted_headers.unwrap(); - message.header.user_headers_length = - decrypted_headers.len() as u32; - message.user_headers = Some(Bytes::from(decrypted_headers)); - } + let decrypted_headers = decrypted_headers.unwrap(); + message.header.user_headers_length = decrypted_headers.len() as u32; + message.user_headers = Some(Bytes::from(decrypted_headers)); } } + } - if let Some(current_offset_entry) = self.current_offsets.get(&partition_id) - { - current_offset_entry.store(polled_messages.current_offset, ORDERING); - } else { - self.current_offsets.insert( - partition_id, - AtomicU64::new(polled_messages.current_offset), - ); - } - - let message = polled_messages.messages.remove(0); - self.buffered_messages.extend(polled_messages.messages); + if let Some(current_offset_entry) = self.current_offsets.get(&partition_id) { + current_offset_entry.store(current_offset, ORDERING); + } else { + self.current_offsets + .insert(partition_id, AtomicU64::new(current_offset)); + } - if self.polling_strategy.kind != PollingKind::Next { - self.polling_strategy = - PollingStrategy::offset(message.header.offset + 1); - } + // A poll is only sent once the buffer has run empty, so nothing is overwritten. + self.buffered_messages = messages; - if let Some(last_consumed_offset_entry) = - self.state.last_consumed_offsets.get(&partition_id) - { - last_consumed_offset_entry.store(message.header.offset, ORDERING); - } else { - self.state - .last_consumed_offsets - .insert(partition_id, AtomicU64::new(message.header.offset)); - } + if self.polling_strategy.kind != PollingKind::Next { + self.next_offsets + .insert(partition_id, first.header.offset + 1); + } - if (self.store_after_every_nth_message > 0 - && message.header.offset % self.store_after_every_nth_message == 0) - || self.store_offset_after_each_message - || (self.store_offset_after_all_messages - && self.buffered_messages.is_empty()) - { - self.send_store_offset( - polled_messages.partition_id, - message.header.offset, - ); - } + if let Some(last_consumed_offset_entry) = + self.state.last_consumed_offsets.get(&partition_id) + { + last_consumed_offset_entry.store(first.header.offset, ORDERING); + } else { + self.state + .last_consumed_offsets + .insert(partition_id, AtomicU64::new(first.header.offset)); + } - // Drop future since it is [invalid after being ready](https://doc.rust-lang.org/std/future/trait.Future.html#panics) - self.poll_future = None; - return Poll::Ready(Some(Ok(ReceivedMessage::new( - message, - polled_messages.current_offset, - polled_messages.partition_id, - )))); + if (self.store_after_every_nth_message > 0 + && first.header.offset % self.store_after_every_nth_message == 0) + || self.store_offset_after_each_message + || (self.store_offset_after_all_messages + && self.buffered_messages.is_empty()) + { + self.send_store_offset(partition_id, first.header.offset); } + + // Drop future since it is [invalid after being ready](https://doc.rust-lang.org/std/future/trait.Future.html#panics) + self.poll_future = None; + return Poll::Ready(Some(Ok(ReceivedMessage::new( + first, + current_offset, + partition_id, + )))); } Poll::Ready(Err(err)) => { self.poll_future = None; @@ -1715,14 +1691,15 @@ impl IggyConsumer { /// commits in flight. The consumer waits for `offset_drain_timeout` on each in turn before /// forcing it to abort. /// - commit the reading position of every partition where it is ahead of this consumer's own - /// record of what it stored, under every [`AutoCommit`] setting, [`AutoCommit::Disabled`] - /// included. Under auto-commit-on-poll (the default) the poll already committed the whole - /// batch, so this store moves the server offset back to the last message handed over, and - /// the next run resumes right after it instead of after the last batch fetched. + /// record of what it stored, unless [`auto_commit()`] is [`AutoCommit::Disabled`]. Under + /// auto-commit-on-poll (the default) the poll already committed the whole batch, so this + /// store moves the server offset back to the last message handed over, and the next run + /// resumes right after it instead of after the last batch fetched. /// - leave the consumer group, if this consumer is a group member. This lets the server give its partitions to /// the remaining members immediately instead of waiting for the connection to time out. + /// - stop the task watching the connection lifecycle. /// - /// The lifecycle event task is not stopped. It runs until the client shuts down. + /// [`auto_commit()`]: crate::prelude::IggyConsumerBuilder::auto_commit /// /// # Errors /// @@ -1757,17 +1734,8 @@ impl IggyConsumer { ); } - // Drop the sending end of the store offset task to end the `recv_async()` loop in `init()`. - // Offsets in queue will still be committed. This prevents loading additional offsets into a channel - // that is not read anymore. - // Replace with a new (hanging) channel, since `store_offset_sender` is not optional. - let (closed_sender, _) = flume::bounded(0); - drop(std::mem::replace( - &mut self.store_offset_sender, - closed_sender, - )); - - // This task never sleeps, so no need to notify. + // Wakes the store task, which sends what is still queued and exits on the shutdown flag. + self.store_offset_notify.notify_one(); if let Some(mut task) = self.store_offset_task.take() && time::timeout(self.offset_drain_timeout.get_duration(), &mut task) .await @@ -1780,18 +1748,19 @@ impl IggyConsumer { ); } - for (partition_id, consumed_offset) in self.state.last_consumed_offsets() { - let stored_offset = self.state.get_last_stored_offset(partition_id).unwrap_or(0); - - if consumed_offset > stored_offset { - trace!( - "Flushing final offset: {consumed_offset} for partition: {partition_id}, stream: {}, topic: {}", - self.stream_id, self.topic_id - ); - let _ = self - .state - .store_consumer_offset(partition_id, consumed_offset, self.allow_replay) - .await; + if self.auto_commit != AutoCommit::Disabled { + for (partition_id, consumed_offset) in self.state.last_consumed_offsets() { + let stored_offset = self.state.get_last_stored_offset(partition_id).unwrap_or(0); + if consumed_offset > stored_offset { + trace!( + "Flushing final offset: {consumed_offset} for partition: {partition_id}, stream: {}, topic: {}", + self.stream_id, self.topic_id + ); + let _ = self + .state + .store_consumer_offset(partition_id, consumed_offset, self.allow_replay) + .await; + } } } @@ -1820,19 +1789,26 @@ impl IggyConsumer { } } + if let Some(task) = self.events_task.take() { + task.abort(); + } + info!("Consumer: {} has been shut down.", self.consumer_name); Ok(()) } } -/// Wakes the interval commit task so it exits. Commits already queued still go out. -/// -/// Nothing is flushed and the consumer group is not left. Await [`IggyConsumer::shutdown`] first, -/// see [Shutting down](IggyConsumer#shutting-down). +/// Stops the background tasks. Commits already queued still go out, nothing else is flushed and +/// the consumer group is not left. Await [`IggyConsumer::shutdown`] first, see +/// [Shutting down](IggyConsumer#shutting-down). impl Drop for IggyConsumer { fn drop(&mut self) { self.shutdown.store(true, ORDERING); self.background_commit_notify.notify_one(); + self.store_offset_notify.notify_one(); + if let Some(task) = self.events_task.take() { + task.abort(); + } trace!( "Consumer {} has been dropped, shutdown signal sent", self.consumer_name @@ -1846,9 +1822,14 @@ mod tests { use crate::client_wrappers::client_wrapper::ClientWrapper; use crate::clients::consumer_builder::IggyConsumerBuilder; use crate::tcp::tcp_client::TcpClient; + use iggy_common::Aes256GcmEncryptor; use iggy_common::locking::IggyRwLockFn; use std::str::FromStr; use std::task::Waker; + use tokio::time::timeout; + + const POLL_RETRY_INTERVAL: Duration = Duration::from_millis(10); + const POLL_TIMEOUT: Duration = Duration::from_secs(2); fn builder_for(consumer: Consumer) -> IggyConsumerBuilder { IggyConsumerBuilder::new( @@ -1921,6 +1902,165 @@ mod tests { assert!(consumer.poll_future.is_none()); } + #[test] + fn group_member_should_ignore_the_partition_set_on_the_builder() { + let consumer = builder_for(Consumer::group(Identifier::numeric(1).unwrap())) + .partition(Some(1)) + .build(); + + assert_eq!(consumer.partition_id, None); + } + + #[test] + fn standalone_consumer_should_keep_the_partition_set_on_the_builder() { + let consumer = builder().partition(Some(1)).build(); + + assert_eq!(consumer.partition_id, Some(1)); + } + + fn message_at(offset: u64) -> IggyMessage { + let mut message = IggyMessage::from_str("payload").unwrap(); + message.header.offset = offset; + message + } + + /// Hands over `messages` as one buffered batch read from `partition_id`. + fn hand_over_batch(consumer: &mut IggyConsumer, partition_id: u32, messages: Vec) { + consumer + .state + .current_partition_id + .store(partition_id, ORDERING); + consumer.buffered_messages = VecDeque::from(messages); + let mut context = Context::from_waker(Waker::noop()); + while !consumer.buffered_messages.is_empty() { + assert!(matches!( + Pin::new(&mut *consumer).poll_next(&mut context), + Poll::Ready(Some(Ok(_))) + )); + } + } + + fn next_offset(consumer: &IggyConsumer, partition_id: u32) -> Option { + consumer + .next_offsets + .get(&partition_id) + .map(|offset| *offset) + } + + #[test] + fn group_member_should_continue_each_partition_after_its_last_message() { + let mut consumer = builder_for(Consumer::group(Identifier::numeric(1).unwrap())) + .polling_strategy(PollingStrategy::first()) + .auto_commit(AutoCommit::Disabled) + .build(); + + hand_over_batch(&mut consumer, 3, vec![message_at(10), message_at(11)]); + assert_eq!(next_offset(&consumer, 3), Some(12)); + assert_eq!(consumer.polling_strategy, PollingStrategy::first()); + + hand_over_batch(&mut consumer, 4, vec![message_at(7)]); + assert_eq!(next_offset(&consumer, 4), Some(8)); + assert_eq!(next_offset(&consumer, 3), Some(12)); + } + + #[test] + fn next_strategy_should_leave_the_continuation_to_the_server() { + let mut consumer = builder_for(Consumer::group(Identifier::numeric(1).unwrap())) + .auto_commit(AutoCommit::Disabled) + .build(); + + hand_over_batch(&mut consumer, 3, vec![message_at(10), message_at(11)]); + + assert!(consumer.next_offsets.is_empty()); + } + + /// Polls once as a group member on a client that is not connected. The outcome must be an + /// error, never an endless wait for a join. + async fn poll_once_as_group_member( + builder: IggyConsumerBuilder, + ) -> Option> { + let mut consumer = builder + .polling_retry_interval(NonZeroIggyDuration::new(POLL_RETRY_INTERVAL).unwrap()) + .build(); + timeout(POLL_TIMEOUT, consumer.next()) + .await + .expect("a group member must poll or report an error instead of waiting for a join") + } + + #[tokio::test] + async fn group_member_without_auto_join_should_poll_instead_of_waiting_for_the_join() { + let builder = builder_for(Consumer::group(Identifier::numeric(1).unwrap())) + .do_not_auto_join_consumer_group(); + + assert!(matches!( + poll_once_as_group_member(builder).await, + Some(Err(_)) + )); + } + + #[tokio::test] + async fn group_member_should_report_a_failed_join_as_a_poll_error() { + let builder = builder_for(Consumer::group(Identifier::numeric(1).unwrap())) + .auto_join_consumer_group(); + + assert!(matches!( + poll_once_as_group_member(builder).await, + Some(Err(_)) + )); + } + + #[tokio::test] + async fn init_should_reject_an_encryptor_with_auto_commit_on_polling() { + let encryptor = Arc::new(EncryptorKind::Aes256Gcm( + Aes256GcmEncryptor::new(&[1; 32]).unwrap(), + )); + for auto_commit in [ + AutoCommit::When(AutoCommitWhen::PollingMessages), + AutoCommit::IntervalOrWhen( + NonZeroIggyDuration::ONE_SECOND, + AutoCommitWhen::PollingMessages, + ), + ] { + let mut consumer = builder() + .encryptor(encryptor.clone()) + .auto_commit(auto_commit) + .build(); + + assert!( + matches!(consumer.init().await, Err(IggyError::InvalidConfiguration)), + "{auto_commit:?} must be rejected with an encryptor" + ); + } + + let mut consumer = builder() + .encryptor(encryptor) + .auto_commit(AutoCommit::When(AutoCommitWhen::ConsumingEachMessage)) + .build(); + + assert!(!matches!( + consumer.init().await, + Err(IggyError::InvalidConfiguration) + )); + } + + #[test] + fn send_store_offset_should_keep_the_latest_offset_per_partition() { + let mut consumer = builder().build(); + consumer.initialized = true; + + consumer.send_store_offset(1, 5); + consumer.send_store_offset(1, 7); + consumer.send_store_offset(2, 3); + + let mut queued: Vec<(u32, u64)> = consumer + .pending_commits + .iter() + .map(|entry| (*entry.key(), *entry.value())) + .collect(); + queued.sort_unstable(); + assert_eq!(queued, vec![(1, 7), (2, 3)]); + } + #[tokio::test] async fn should_accept_every_auto_commit_mode() { for auto_commit in [ diff --git a/core/sdk/src/clients/consumer_builder.rs b/core/sdk/src/clients/consumer_builder.rs index 87b427d834..91cb9ac837 100644 --- a/core/sdk/src/clients/consumer_builder.rs +++ b/core/sdk/src/clients/consumer_builder.rs @@ -93,8 +93,8 @@ impl IggyConsumerBuilder { } /// Sets the partition to read. `None` lets a consumer group read its assigned partitions and - /// makes the server read partition `0` for a standalone consumer. `Some(n)` on a group member - /// pins every poll to that partition instead of the assignment. + /// makes the server read partition `0` for a standalone consumer. `Some(n)` is for standalone + /// consumers. A group member ignores it with a warning and reads its assignment. pub fn partition(self, partition: Option) -> Self { Self { partition, ..self } } @@ -123,15 +123,8 @@ impl IggyConsumerBuilder { } } - /// Same as [`auto_commit`](Self::auto_commit) with [`AutoCommit::Disabled`]. - pub fn commit_failed_messages(self) -> Self { - Self { - auto_commit: AutoCommit::Disabled, - ..self - } - } - - /// Automatically joins the consumer group if the consumer is a part of a consumer group. + /// Joins the consumer group during `init()` and again after the membership was lost, for + /// example after a reconnect. On by default. pub fn auto_join_consumer_group(self) -> Self { Self { auto_join_consumer_group: true, @@ -139,7 +132,9 @@ impl IggyConsumerBuilder { } } - /// Does not automatically join the consumer group if the consumer is a part of a consumer group. + /// Leaves joining the consumer group to the caller. The member polls as soon as `init()` + /// returns, and a poll without a membership fails with + /// [`IggyError::ConsumerGroupMemberNotFound`](iggy_common::IggyError::ConsumerGroupMemberNotFound). pub fn do_not_auto_join_consumer_group(self) -> Self { Self { auto_join_consumer_group: false, @@ -195,7 +190,9 @@ impl IggyConsumerBuilder { } } - /// Sets the polling retry interval in case of server disconnection. + /// Sets how long a poll waits before the next attempt while it is blocked: after a + /// disconnect, after a failed group join, or while the group member holds no partitions. + /// One second by default. pub fn polling_retry_interval(self, interval: NonZeroIggyDuration) -> Self { Self { polling_retry_interval: interval, @@ -233,7 +230,7 @@ impl IggyConsumerBuilder { /// Builds the consumer. /// - /// Note: After building the consumer, `init()` must be invoked before producing messages. + /// Note: After building the consumer, `init()` must be invoked before consuming messages. pub fn build(self) -> IggyConsumer { IggyConsumer::new( self.client, diff --git a/core/sdk/src/prelude.rs b/core/sdk/src/prelude.rs index 81f7d8c16d..72e2503d16 100644 --- a/core/sdk/src/prelude.rs +++ b/core/sdk/src/prelude.rs @@ -78,6 +78,7 @@ pub use iggy_common::{ IGGY_MESSAGE_HEADERS_LENGTH_OFFSET_RANGE, IGGY_MESSAGE_ID_OFFSET_RANGE, IGGY_MESSAGE_OFFSET_OFFSET_RANGE, IGGY_MESSAGE_ORIGIN_TIMESTAMP_OFFSET_RANGE, IGGY_MESSAGE_PAYLOAD_LENGTH_OFFSET_RANGE, IGGY_MESSAGE_TIMESTAMP_OFFSET_RANGE, INDEX_SIZE, - MAX_PAYLOAD_SIZE, MAX_USER_HEADERS_SIZE, SEC_IN_MICRO, + MAX_PAYLOAD_SIZE, MAX_USER_HEADERS_SIZE, NO_ASSIGNED_PARTITION, + RESYNC_REQUIRED_PARTITION_SENTINEL, SEC_IN_MICRO, defaults::{DEFAULT_ROOT_PASSWORD, DEFAULT_ROOT_USER_ID, DEFAULT_ROOT_USERNAME}, }; diff --git a/core/sdk/src/quic/quic_client.rs b/core/sdk/src/quic/quic_client.rs index e69e94af6c..5dcadc1d60 100644 --- a/core/sdk/src/quic/quic_client.rs +++ b/core/sdk/src/quic/quic_client.rs @@ -400,6 +400,12 @@ impl iggy_common::VsrSessionControl for QuicClient { } consensus_session.bind(session); + drop(consensus_session); + // Every fresh client identity passes through here, including one that + // replaces a session the transport never reset: a connection lost + // mid-request can leave the old session in place until this sign-in + // re-mints it. + self.consumer_group_state.clear_session_scoped(); Ok(()) } @@ -408,6 +414,7 @@ impl iggy_common::VsrSessionControl for QuicClient { .consensus_session .lock() .expect("consensus session mutex poisoned") = ConsensusSession::new(); + self.consumer_group_state.clear_session_scoped(); Ok(()) } diff --git a/core/sdk/src/tcp/tcp_client.rs b/core/sdk/src/tcp/tcp_client.rs index 23f70f7910..1546708330 100644 --- a/core/sdk/src/tcp/tcp_client.rs +++ b/core/sdk/src/tcp/tcp_client.rs @@ -344,6 +344,12 @@ impl iggy_common::VsrSessionControl for TcpClient { } consensus_session.bind(session); + drop(consensus_session); + // Every fresh client identity passes through here, including one that + // replaces a session the transport never reset: a connection lost + // mid-request can leave the old session in place until this sign-in + // re-mints it. + self.consumer_group_state.clear_session_scoped(); Ok(()) } @@ -352,6 +358,7 @@ impl iggy_common::VsrSessionControl for TcpClient { .consensus_session .lock() .expect("consensus session mutex poisoned") = ConsensusSession::new(); + self.consumer_group_state.clear_session_scoped(); Ok(()) } diff --git a/core/sdk/src/websocket/websocket_client.rs b/core/sdk/src/websocket/websocket_client.rs index f40bc9178b..2a22683e02 100644 --- a/core/sdk/src/websocket/websocket_client.rs +++ b/core/sdk/src/websocket/websocket_client.rs @@ -395,6 +395,12 @@ impl iggy_common::VsrSessionControl for WebSocketClient { } consensus_session.bind(session); + drop(consensus_session); + // Every fresh client identity passes through here, including one that + // replaces a session the transport never reset: a connection lost + // mid-request can leave the old session in place until this sign-in + // re-mints it. + self.consumer_group_state.clear_session_scoped(); Ok(()) } @@ -403,6 +409,7 @@ impl iggy_common::VsrSessionControl for WebSocketClient { .consensus_session .lock() .expect("consensus session mutex poisoned") = ConsensusSession::new(); + self.consumer_group_state.clear_session_scoped(); Ok(()) } diff --git a/foreign/php/README.md b/foreign/php/README.md index 795b3bebbf..15007bb901 100644 --- a/foreign/php/README.md +++ b/foreign/php/README.md @@ -117,7 +117,9 @@ foreach ($messages as $message) { } ``` -Consumer group callbacks require a finite message limit: +Consumer group callbacks require a finite message limit. The partition id +argument is ignored for a consumer group, since the member reads the partitions +the server assigns to it: ```php consumerGroup( 'php-consumer', $stream, $topic, - $partitionId, + null, \Iggy\PollingStrategy::next(), 10, \Iggy\AutoCommit::disabled(), diff --git a/foreign/php/iggy-php.stubs.php b/foreign/php/iggy-php.stubs.php index 211e6f120f..be8437cb0a 100644 --- a/foreign/php/iggy-php.stubs.php +++ b/foreign/php/iggy-php.stubs.php @@ -94,6 +94,9 @@ public function connect(): void {} /** * Creates and initializes a consumer group consumer. * + * `$partition_id` is ignored for a consumer group: the member reads the partitions + * the server assigns to it. + * * @param string $name * @param string $stream * @param string $topic diff --git a/foreign/php/src/client.rs b/foreign/php/src/client.rs index fee9752edf..23e5a61243 100644 --- a/foreign/php/src/client.rs +++ b/foreign/php/src/client.rs @@ -293,6 +293,9 @@ impl IggyClient { } /// Creates and initializes a consumer group consumer. + /// + /// `$partition_id` is ignored for a consumer group: the member reads the partitions + /// the server assigns to it. #[allow(clippy::too_many_arguments)] #[php(defaults( create_consumer_group_if_not_exists = true, diff --git a/foreign/python/apache_iggy.pyi b/foreign/python/apache_iggy.pyi index 6c3715b54b..c0f603e58d 100644 --- a/foreign/python/apache_iggy.pyi +++ b/foreign/python/apache_iggy.pyi @@ -1377,6 +1377,8 @@ class IggyClient: ) -> collections.abc.Awaitable[IggyConsumer]: r""" Creates a new consumer group consumer. + `partition_id` is ignored for a consumer group: the member reads the partitions + the server assigns to it. Returns the consumer or a RuntimeError on failure. Raises `ValueError` if `poll_interval`, `polling_retry_interval`, `init_retry_interval` or an `AutoCommit` interval is negative, or if any of those except `poll_interval` diff --git a/foreign/python/src/client.rs b/foreign/python/src/client.rs index f669a87484..10c5c776b0 100644 --- a/foreign/python/src/client.rs +++ b/foreign/python/src/client.rs @@ -1087,6 +1087,8 @@ impl IggyClient { } /// Creates a new consumer group consumer. + /// `partition_id` is ignored for a consumer group: the member reads the partitions + /// the server assigns to it. /// Returns the consumer or a RuntimeError on failure. Raises `ValueError` if /// `poll_interval`, `polling_retry_interval`, `init_retry_interval` or an /// `AutoCommit` interval is negative, or if any of those except `poll_interval` From c3df915ff907ec6ca439a10857e1a1ffebeb6091 Mon Sep 17 00:00:00 2001 From: Justin Mclean Date: Thu, 3 Sep 2026 17:26:34 +1000 Subject: [PATCH 046/182] docs: fix partition IDs and auto commit wording in the README quickstart (#4040) --- README.md | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/README.md b/README.md index 74690f17e1..1700f19c95 100644 --- a/README.md +++ b/README.md @@ -317,7 +317,7 @@ Get `dev` stream details: `cargo run --bin iggy -- -u -p stream get dev` -Create a topic named `sample` (numerical ID will be assigned by server automatically) for stream `dev`, with 2 partitions (IDs 1 and 2), no topic compression (`none`), and disabled message expiry (skipped optional parameter). Other compression values are reserved for future server-side support: +Create a topic named `sample` (numerical ID will be assigned by server automatically) for stream `dev`, with 2 partitions (IDs 0 and 1), no topic compression (`none`), and disabled message expiry (skipped optional parameter). Other compression values are reserved for future server-side support: `cargo run --bin iggy -- -u -p topic create dev sample 2 none` @@ -329,17 +329,17 @@ Get topic details for topic `sample` in stream `dev`: `cargo run --bin iggy -- -u -p topic get dev sample` -Send a message 'hello world' (message ID 1) to the stream `dev` to topic `sample` and partition 1: +Send the first message 'hello world' to the stream `dev` to topic `sample` and partition 0: -`cargo run --bin iggy -- -u -p message send --partition-id 1 dev sample "hello world"` +`cargo run --bin iggy -- -u -p message send --partition-id 0 dev sample "hello world"` -Send another message 'lorem ipsum' (message ID 2) to the same stream, topic and partition: +Send a second message 'lorem ipsum' to the same stream, topic and partition: -`cargo run --bin iggy -- -u -p message send --partition-id 1 dev sample "lorem ipsum"` +`cargo run --bin iggy -- -u -p message send --partition-id 0 dev sample "lorem ipsum"` -Poll messages by a regular consumer with ID 1 from the stream `dev` for topic `sample` and partition with ID 1, starting with offset 0, messages count 2, without auto commit (storing consumer offset on server): +Poll messages by a regular consumer with ID 1 from the stream `dev` for topic `sample` and partition with ID 0, starting with offset 0, messages count 2, with auto commit (storing consumer offset on server): -`cargo run --bin iggy -- -u -p message poll --consumer 1 --offset 0 --message-count 2 --auto-commit dev sample 1` +`cargo run --bin iggy -- -u -p message poll --consumer 1 --offset 0 --message-count 2 --auto-commit dev sample 0` Finally, restart the server to see it is able to load the persisted data. From d3028f9b244901420a5fec5895de58a55d77e605 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Thu, 3 Sep 2026 09:34:20 +0200 Subject: [PATCH 047/182] chore(deps): Bump fast-uri from 3.1.5 to 3.1.7 in /foreign/node (#4042) --- foreign/node/package-lock.json | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/foreign/node/package-lock.json b/foreign/node/package-lock.json index 31539e7159..967611bc66 100644 --- a/foreign/node/package-lock.json +++ b/foreign/node/package-lock.json @@ -10,7 +10,6 @@ "license": "Apache-2.0", "dependencies": { "@node-rs/xxhash": "1.7.7", - "@swc/core-win32-x64-msvc": "1.16.1", "debug": "4.4.3", "generic-pool": "3.9.0", "uuidv7": "1.2.1" @@ -2955,9 +2954,9 @@ "peer": true }, "node_modules/fast-uri": { - "version": "3.1.5", - "resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.5.tgz", - "integrity": "sha512-gHwA1O9LDIcKunMKhObS/HimwtehO1nPUECKAu5TpKgaO19fcWEl4bliWe1jWxVFvIXztJjjQ4L8XQ1EU9f7Jw==", + "version": "3.1.7", + "resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.7.tgz", + "integrity": "sha512-dOvZVzjdZdz7phd9v6jCbwxrBW3fK6n8Rc0CtdmM4bumzMnxywBYhuph6J819RRw/ku+rLbelwfMunktuzVVHg==", "dev": true, "funding": [ { From e77ee39d3cffd45f68b2bd7e9bb9ac77d8aec06f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=C5=81ukasz=20Zborek?= Date: Thu, 3 Sep 2026 09:44:49 +0200 Subject: [PATCH 048/182] refactor(csharp): update integer properties to unsigned types for consistency (#4027) --- .../BasicMessagingOperationsSteps.cs | 2 +- .../ClusterRedirectionTests.cs | 2 +- .../ConsumerGroupTests.cs | 9 +- .../FetchMessagesTests.cs | 4 +- .../HeartbeatTests.cs | 8 +- .../IggyPublisherTests.cs | 2 +- .../Iggy_SDK.Tests.Integration/OffsetTests.cs | 6 +- .../PersonalAccessTokenTests.cs | 2 +- .../StreamsTests.cs | 12 +- .../Iggy_SDK.Tests.Integration/SystemTests.cs | 29 +- .../Iggy_SDK.Tests.Integration/TopicsTests.cs | 4 - .../Iggy_SDK.Tests.Integration/UsersTests.cs | 14 +- .../Vsr/VsrMessagingTests.cs | 2 +- .../Consumers/IggyConsumer.Logging.cs | 5 + .../Iggy_SDK/Consumers/IggyConsumer.Rented.cs | 15 +- .../csharp/Iggy_SDK/Consumers/IggyConsumer.cs | 28 +- .../Iggy_SDK/Contracts/Auth/AuthResponse.cs | 2 +- .../Iggy_SDK/Contracts/Auth/Permissions.cs | 2 +- .../csharp/Iggy_SDK/Contracts/ClientInfo.cs | 4 +- .../csharp/Iggy_SDK/Contracts/ClusterNode.cs | 6 - .../Iggy_SDK/Contracts/ClusterNodeStatus.cs | 7 +- .../Iggy_SDK/Contracts/ConsumerGroupInfo.cs | 6 +- .../Contracts/ConsumerGroupMembers.cs | 4 +- .../Iggy_SDK/Contracts/MessageResponse.cs | 2 +- .../Iggy_SDK/Contracts/OffsetResponse.cs | 2 +- .../Iggy_SDK/Contracts/PartitionResponse.cs | 4 +- .../Iggy_SDK/Contracts/PolledMessages.cs | 9 +- .../Contracts/PolledMessagesRental.cs | 2 +- .../Contracts/RentedMessageResponse.cs | 2 +- .../Iggy_SDK/Contracts/StatsResponse.cs | 14 +- .../Iggy_SDK/Contracts/StreamPermissions.cs | 2 +- .../Iggy_SDK/Contracts/StreamResponse.cs | 2 +- .../Iggy_SDK/Contracts/Tcp/TcpContracts.cs | 225 +++----- .../csharp/Iggy_SDK/Enums/ClientTransport.cs | 49 ++ foreign/csharp/Iggy_SDK/Identifier.cs | 11 +- .../Iggy_SDK/IggyClient/IIggyConsumer.cs | 12 +- .../Implementations/HttpMessageStream.cs | 4 +- .../Implementations/TcpMessageStream.Vsr.cs | 24 +- foreign/csharp/Iggy_SDK/Iggy_SDK.csproj | 2 +- foreign/csharp/Iggy_SDK/Kinds/Consumer.cs | 22 + foreign/csharp/Iggy_SDK/Kinds/Partitioning.cs | 13 +- .../csharp/Iggy_SDK/Mappers/BinaryMapper.cs | 510 ++++++++++-------- .../Iggy_SDK/Vsr/ConsumerGroupClientState.cs | 25 +- .../NoAssignedPartitionBackoffTests.cs | 193 +++++++ .../MessageBatchGoldenVectorTests.cs | 2 +- .../ContractsTests/UserContractsTests.cs | 124 +++++ .../MapperTests/BinaryMapper.cs | 300 ++++++++++- .../MapperTests/HeaderEncryptionTests.cs | 56 +- .../OptionsBlockGoldenVectorTests.cs | 10 + .../IdentifiersByteSerializationTests.cs | 25 + .../Iggy_SDK_Tests/Utils/BinaryFactory.cs | 61 +-- .../Utils/Streams/StreamFactory.cs | 4 +- .../Utils/Users/PermissionsFactory.cs | 24 +- .../Utils/Users/UsersFactory.cs | 6 +- .../VsrTests/ConsumerGroupClientStateTests.cs | 14 +- .../VsrTests/EndpointFailoverTests.cs | 265 +-------- .../VsrTests/GroupPollingTests.cs | 150 ++++++ .../Iggy_SDK_Tests/VsrTests/MockNode.cs | 251 +++++++++ 58 files changed, 1759 insertions(+), 837 deletions(-) create mode 100644 foreign/csharp/Iggy_SDK/Enums/ClientTransport.cs create mode 100644 foreign/csharp/Iggy_SDK_Tests/ConsumerTests/NoAssignedPartitionBackoffTests.cs create mode 100644 foreign/csharp/Iggy_SDK_Tests/ContractsTests/UserContractsTests.cs create mode 100644 foreign/csharp/Iggy_SDK_Tests/VsrTests/GroupPollingTests.cs create mode 100644 foreign/csharp/Iggy_SDK_Tests/VsrTests/MockNode.cs diff --git a/foreign/csharp/Iggy_SDK.Tests.BDD/StepDefinitions/BasicMessagingOperationsSteps.cs b/foreign/csharp/Iggy_SDK.Tests.BDD/StepDefinitions/BasicMessagingOperationsSteps.cs index f0f44b539a..87e7c8a0ce 100644 --- a/foreign/csharp/Iggy_SDK.Tests.BDD/StepDefinitions/BasicMessagingOperationsSteps.cs +++ b/foreign/csharp/Iggy_SDK.Tests.BDD/StepDefinitions/BasicMessagingOperationsSteps.cs @@ -59,7 +59,7 @@ public async Task GivenIAmAuthenticatedAsTheRootUser() var loginResult = await _context.IggyClient.LoginUserAsync(TestEnvironment.RootUsername, TestEnvironment.RootPassword); loginResult.ShouldNotBeNull(); - loginResult.UserId.ShouldBe(0); + loginResult.UserId.ShouldBe(0u); } [Given(@"I have no streams in the system")] diff --git a/foreign/csharp/Iggy_SDK.Tests.Integration/ClusterRedirectionTests.cs b/foreign/csharp/Iggy_SDK.Tests.Integration/ClusterRedirectionTests.cs index 879b45fc02..f0c1c9343e 100644 --- a/foreign/csharp/Iggy_SDK.Tests.Integration/ClusterRedirectionTests.cs +++ b/foreign/csharp/Iggy_SDK.Tests.Integration/ClusterRedirectionTests.cs @@ -135,7 +135,7 @@ public async Task ConnectToFollowerWithPersonalAccessToken_Should_RedirectToLead } authResponse.ShouldNotBeNull(); - authResponse!.UserId.ShouldBeGreaterThanOrEqualTo(0); + authResponse!.UserId.ShouldNotBe(uint.MaxValue); var address = client.GetCurrentAddress(); address.ShouldNotBeNullOrEmpty(); diff --git a/foreign/csharp/Iggy_SDK.Tests.Integration/ConsumerGroupTests.cs b/foreign/csharp/Iggy_SDK.Tests.Integration/ConsumerGroupTests.cs index c6ed9ed56a..b0765fdcd9 100644 --- a/foreign/csharp/Iggy_SDK.Tests.Integration/ConsumerGroupTests.cs +++ b/foreign/csharp/Iggy_SDK.Tests.Integration/ConsumerGroupTests.cs @@ -54,7 +54,6 @@ public async Task CreateConsumerGroup_HappyPath_Should_CreateConsumerGroup_Succe Identifier.String(streamName), Identifier.String(TopicName), GroupName); consumerGroup.ShouldNotBeNull(); - consumerGroup.Id.ShouldBeGreaterThanOrEqualTo(0u); consumerGroup.PartitionsCount.ShouldBe(PartitionsCount); consumerGroup.MembersCount.ShouldBe(0u); consumerGroup.Name.ShouldBe(GroupName); @@ -129,8 +128,8 @@ await Should.NotThrowAsync(() => // Verify via GetMe that the client is now a member of the consumer group var me = await client.GetMeAsync(); me.ShouldNotBeNull(); - me.ConsumerGroupsCount.ShouldBe(1); - me.ConsumerGroups.ShouldContain(x => x.GroupId == (int)cg!.Id); + me.ConsumerGroupsCount.ShouldBe(1u); + me.ConsumerGroups.ShouldContain(x => x.GroupId == cg!.Id); } [Test] @@ -165,8 +164,8 @@ await Should.NotThrowAsync(() => // Verify via GetMe that the client is no longer a member of the consumer group var me = await client.GetMeAsync(); me.ShouldNotBeNull(); - me.ConsumerGroupsCount.ShouldBe(0); - me.ConsumerGroups.ShouldNotContain(x => x.GroupId == (int)cg.Id); + me.ConsumerGroupsCount.ShouldBe(0u); + me.ConsumerGroups.ShouldNotContain(x => x.GroupId == cg.Id); } [Test] diff --git a/foreign/csharp/Iggy_SDK.Tests.Integration/FetchMessagesTests.cs b/foreign/csharp/Iggy_SDK.Tests.Integration/FetchMessagesTests.cs index f9ae474a23..c4d1325ef3 100644 --- a/foreign/csharp/Iggy_SDK.Tests.Integration/FetchMessagesTests.cs +++ b/foreign/csharp/Iggy_SDK.Tests.Integration/FetchMessagesTests.cs @@ -78,7 +78,7 @@ public async Task PollMessages_WithNoHeaders_Should_PollMessages_Successfully(Pr }); response.Messages.Count.ShouldBe(10); - response.PartitionId.ShouldBe(0); + response.PartitionId.ShouldBe(0u); response.CurrentOffset.ShouldBe(19u); foreach (var responseMessage in response.Messages) @@ -129,7 +129,7 @@ public async Task PollMessages_WithHeaders_Should_PollMessages_Successfully(Prot var response = await client.PollMessagesAsync(headersMessageFetchRequest); response.Messages.Count.ShouldBe(10); - response.PartitionId.ShouldBe(0); + response.PartitionId.ShouldBe(0u); response.CurrentOffset.ShouldBe(19u); foreach (var responseMessage in response.Messages) { diff --git a/foreign/csharp/Iggy_SDK.Tests.Integration/HeartbeatTests.cs b/foreign/csharp/Iggy_SDK.Tests.Integration/HeartbeatTests.cs index 882c89f881..c6b7245793 100644 --- a/foreign/csharp/Iggy_SDK.Tests.Integration/HeartbeatTests.cs +++ b/foreign/csharp/Iggy_SDK.Tests.Integration/HeartbeatTests.cs @@ -56,7 +56,7 @@ public async Task IdleGroupMember_WithHeartbeat_Should_StayMember() var me = await client.GetMeAsync(); me.ShouldNotBeNull(); - me.ConsumerGroupsCount.ShouldBe(1); + me.ConsumerGroupsCount.ShouldBe(1u); } [Test] @@ -78,7 +78,7 @@ public async Task IdleGroupMember_WithSlowHeartbeat_Should_BeEvicted_And_Reconne // fresh, auto-logged-in session. var me = await client.GetMeAsync(); me.ShouldNotBeNull(); - me.ConsumerGroupsCount.ShouldBe(0); + me.ConsumerGroupsCount.ShouldBe(0u); } [Test] @@ -101,7 +101,7 @@ public async Task EvictedClient_WithPersonalAccessTokenAutoLogin_Should_Reconnec var me = await client.GetMeAsync(); me.ShouldNotBeNull(); - me.ConsumerGroupsCount.ShouldBe(0); + me.ConsumerGroupsCount.ShouldBe(0u); } ///

@@ -128,7 +128,7 @@ public async Task EvictedClient_WithoutAutoLogin_Should_ReestablishItsSession() // that belonged to it is gone. var me = await client.GetMeAsync(); me.ShouldNotBeNull(); - me.ConsumerGroupsCount.ShouldBe(0); + me.ConsumerGroupsCount.ShouldBe(0u); } private Task CreateClient(TimeSpan heartbeatInterval, bool autoLogin = true) diff --git a/foreign/csharp/Iggy_SDK.Tests.Integration/IggyPublisherTests.cs b/foreign/csharp/Iggy_SDK.Tests.Integration/IggyPublisherTests.cs index c50c1ba603..4d2bd92a58 100644 --- a/foreign/csharp/Iggy_SDK.Tests.Integration/IggyPublisherTests.cs +++ b/foreign/csharp/Iggy_SDK.Tests.Integration/IggyPublisherTests.cs @@ -409,7 +409,7 @@ public async Task SendMessages_ToMultiplePartitions_Should_DistributeMessages(Pr false); polledMessages.Messages.Count.ShouldBeGreaterThanOrEqualTo(10); - polledMessages.PartitionId.ShouldBe((int)partitionId); + polledMessages.PartitionId.ShouldBe(partitionId); } } diff --git a/foreign/csharp/Iggy_SDK.Tests.Integration/OffsetTests.cs b/foreign/csharp/Iggy_SDK.Tests.Integration/OffsetTests.cs index f96ced773f..03783b1fa2 100644 --- a/foreign/csharp/Iggy_SDK.Tests.Integration/OffsetTests.cs +++ b/foreign/csharp/Iggy_SDK.Tests.Integration/OffsetTests.cs @@ -81,7 +81,7 @@ await client.StoreOffsetAsync(Consumer.New("test-consumer"), Identifier.String(s offset.ShouldNotBeNull(); offset.StoredOffset.ShouldBe(SetOffset); - offset.PartitionId.ShouldBe(0); + offset.PartitionId.ShouldBe(0u); offset.CurrentOffset.ShouldBe(3u); } @@ -142,7 +142,7 @@ await client.StoreOffsetAsync(Consumer.Group("test_consumer_group"), Identifier. offset.ShouldNotBeNull(); offset.StoredOffset.ShouldBe(SetOffset); - offset.PartitionId.ShouldBe(0); + offset.PartitionId.ShouldBe(0u); offset.CurrentOffset.ShouldBe(3u); } @@ -175,7 +175,7 @@ await client.StoreOffsetAsync(Consumer.Group("test_consumer_group"), Identifier. offset.ShouldNotBeNull(); offset.StoredOffset.ShouldBe(SetOffset); - offset.PartitionId.ShouldBe(0); + offset.PartitionId.ShouldBe(0u); offset.CurrentOffset.ShouldBe(3u); } diff --git a/foreign/csharp/Iggy_SDK.Tests.Integration/PersonalAccessTokenTests.cs b/foreign/csharp/Iggy_SDK.Tests.Integration/PersonalAccessTokenTests.cs index b8ffc39b91..d321157c49 100644 --- a/foreign/csharp/Iggy_SDK.Tests.Integration/PersonalAccessTokenTests.cs +++ b/foreign/csharp/Iggy_SDK.Tests.Integration/PersonalAccessTokenTests.cs @@ -91,7 +91,7 @@ public async Task LoginWithPersonalAccessToken_Should_Be_Successfully(Protocol p var authResponse = await loginClient.LoginWithPersonalAccessTokenAsync(response!.Token); authResponse.ShouldNotBeNull(); - authResponse.UserId.ShouldBeGreaterThanOrEqualTo(0); + authResponse.UserId.ShouldNotBe(uint.MaxValue); } [Test] diff --git a/foreign/csharp/Iggy_SDK.Tests.Integration/StreamsTests.cs b/foreign/csharp/Iggy_SDK.Tests.Integration/StreamsTests.cs index 41d762bbf6..b1e1d12f0c 100644 --- a/foreign/csharp/Iggy_SDK.Tests.Integration/StreamsTests.cs +++ b/foreign/csharp/Iggy_SDK.Tests.Integration/StreamsTests.cs @@ -41,12 +41,11 @@ public async Task CreateStream_HappyPath_Should_CreateStream_Successfully(Protoc var response = await client.CreateStreamAsync(name); response.ShouldNotBeNull(); - response.Id.ShouldBeGreaterThanOrEqualTo(0u); response.Name.ShouldBe(name); response.Size.ShouldBe(0u); response.CreatedAt.UtcDateTime.ShouldBe(DateTimeOffset.UtcNow.UtcDateTime, TimeSpan.FromMinutes(1)); response.MessagesCount.ShouldBe(0u); - response.TopicsCount.ShouldBe(0); + response.TopicsCount.ShouldBe(0u); response.Topics.ShouldBeEmpty(); } @@ -98,7 +97,7 @@ public async Task GetStreamById_Should_ReturnValidResponse(Protocol protocol) response.Size.ShouldBe(0u); response.CreatedAt.UtcDateTime.ShouldBe(DateTimeOffset.UtcNow.UtcDateTime, TimeSpan.FromMinutes(1)); response.MessagesCount.ShouldBe(0u); - response.TopicsCount.ShouldBe(0); + response.TopicsCount.ShouldBe(0u); response.Topics.ShouldBeEmpty(); } @@ -119,7 +118,7 @@ public async Task GetStreams_ByStreamName_Should_ReturnValidResponse(Protocol pr response.Size.ShouldBe(0u); response.CreatedAt.UtcDateTime.ShouldBe(DateTimeOffset.UtcNow.UtcDateTime, TimeSpan.FromMinutes(1)); response.MessagesCount.ShouldBe(0u); - response.TopicsCount.ShouldBe(0); + response.TopicsCount.ShouldBe(0u); response.Topics.ShouldBeEmpty(); } @@ -157,12 +156,11 @@ await client.SendMessagesAsync(Identifier.String(streamName), var response = await client.GetStreamByIdAsync(Identifier.String(streamName)); response.ShouldNotBeNull(); - response.Id.ShouldBeGreaterThanOrEqualTo(0u); response.Name.ShouldBe(streamName); response.Size.ShouldBeGreaterThan(0u); response.CreatedAt.UtcDateTime.ShouldBe(DateTimeOffset.UtcNow.UtcDateTime, TimeSpan.FromMinutes(1)); response.MessagesCount.ShouldBe(7u); - response.TopicsCount.ShouldBe(2); + response.TopicsCount.ShouldBe(2u); response.Topics.Count().ShouldBe(2); var topic = response.Topics.First(x => x.Name == topicName1); @@ -222,7 +220,7 @@ await client.SendMessagesAsync(Identifier.String(streamName), purged => purged?.MessagesCount == 0, TimeSpan.FromSeconds(10)); stream.ShouldNotBeNull(); stream.MessagesCount.ShouldBe(0u); - stream.TopicsCount.ShouldBe(1); + stream.TopicsCount.ShouldBe(1u); } [Test] diff --git a/foreign/csharp/Iggy_SDK.Tests.Integration/SystemTests.cs b/foreign/csharp/Iggy_SDK.Tests.Integration/SystemTests.cs index dd2b26ab94..324b706300 100644 --- a/foreign/csharp/Iggy_SDK.Tests.Integration/SystemTests.cs +++ b/foreign/csharp/Iggy_SDK.Tests.Integration/SystemTests.cs @@ -45,7 +45,7 @@ public async Task GetClients_Should_Return_NonEmptyClientsList(Protocol protocol { c.ClientId.ShouldNotBe(0u); c.Address.ShouldNotBeNullOrEmpty(); - c.Transport.ShouldBe(Protocol.Tcp); + c.Transport.ShouldBe(ClientTransport.Tcp); } } @@ -63,10 +63,9 @@ public async Task GetClient_Should_Return_CorrectClient(Protocol protocol) response.ShouldNotBeNull(); response.ClientId.ShouldBe(clientInfo.ClientId); response.UserId.ShouldNotBeNull(); - response.UserId.Value.ShouldBeGreaterThanOrEqualTo(0u); response.Address.ShouldNotBeNullOrEmpty(); - response.Transport.ShouldBe(Protocol.Tcp); - response.ConsumerGroupsCount.ShouldBe(0); + response.Transport.ShouldBe(ClientTransport.Tcp); + response.ConsumerGroupsCount.ShouldBe(0u); response.ConsumerGroups.ShouldBeEmpty(); } @@ -82,7 +81,7 @@ public async Task GetMe_Tcp_Should_Return_MyClient(Protocol protocol) me.ClientId.ShouldNotBe(0u); me.UserId.ShouldBe(0u); me.Address.ShouldNotBeNullOrEmpty(); - me.Transport.ShouldBe(Protocol.Tcp); + me.Transport.ShouldBe(ClientTransport.Tcp); } [Test] @@ -117,8 +116,8 @@ await tcpClient.JoinConsumerGroupAsync(Identifier.String(streamName), var response = await client.GetClientByIdAsync(me!.ClientId); response.ShouldNotBeNull(); response.Address.ShouldNotBeNullOrEmpty(); - response.Transport.ShouldBe(Protocol.Tcp); - response.ConsumerGroupsCount.ShouldBe(1); + response.Transport.ShouldBe(ClientTransport.Tcp); + response.ConsumerGroupsCount.ShouldBe(1u); response.ConsumerGroups.ShouldNotBeEmpty(); response.ConsumerGroups.ShouldContain(x => x.GroupId == consumerGroup!.Id); response.ConsumerGroups.ShouldContain(x => x.StreamId == stream!.Id); @@ -141,21 +140,15 @@ await client.SendMessagesAsync(Identifier.String(streamName), var response = await client.GetStatsAsync(); response.ShouldNotBeNull(); - response.ProcessId.ShouldBeGreaterThanOrEqualTo(0); + response.ProcessId.ShouldNotBe(0u); response.CpuUsage.ShouldBeGreaterThanOrEqualTo(0); response.TotalCpuUsage.ShouldBeGreaterThanOrEqualTo(0); - response.MemoryUsage.ShouldBeGreaterThanOrEqualTo(0u); - response.TotalMemory.ShouldBeGreaterThanOrEqualTo(0u); response.AvailableMemory.ShouldNotBe(0u); - response.RunTime.ShouldBeGreaterThanOrEqualTo(0u); response.StartTime.ShouldBe(DateTimeOffset.UtcNow, TimeSpan.FromMinutes(5)); - response.ReadBytes.ShouldBeGreaterThanOrEqualTo(0u); - response.WrittenBytes.ShouldBeGreaterThanOrEqualTo(0u); - response.MessagesSizeBytes.ShouldBeGreaterThanOrEqualTo(0u); - response.StreamsCount.ShouldBeGreaterThanOrEqualTo(1); - response.TopicsCount.ShouldBeGreaterThanOrEqualTo(1); - response.PartitionsCount.ShouldBeGreaterThanOrEqualTo(1); - response.SegmentsCount.ShouldBeGreaterThanOrEqualTo(1); + response.StreamsCount.ShouldBeGreaterThanOrEqualTo(1u); + response.TopicsCount.ShouldBeGreaterThanOrEqualTo(1u); + response.PartitionsCount.ShouldBeGreaterThanOrEqualTo(1u); + response.SegmentsCount.ShouldBeGreaterThanOrEqualTo(1u); response.MessagesCount.ShouldBeGreaterThanOrEqualTo(1u); // iggy-server leaves the connected-client tally out of its stats reply, so ClientsCount goes unchecked. response.Hostname.ShouldNotBeNullOrEmpty(); diff --git a/foreign/csharp/Iggy_SDK.Tests.Integration/TopicsTests.cs b/foreign/csharp/Iggy_SDK.Tests.Integration/TopicsTests.cs index 050b3f9b87..dd0db67ccf 100644 --- a/foreign/csharp/Iggy_SDK.Tests.Integration/TopicsTests.cs +++ b/foreign/csharp/Iggy_SDK.Tests.Integration/TopicsTests.cs @@ -46,7 +46,6 @@ public async Task Create_NewTopic_Should_Return_Successfully(Protocol protocol) TimeSpan.FromMinutes(10), 2_000_000_000); response.ShouldNotBeNull(); - response.Id.ShouldBeGreaterThanOrEqualTo(0u); response.CreatedAt.UtcDateTime.ShouldBe(DateTimeOffset.UtcNow.UtcDateTime, TimeSpan.FromMinutes(1)); response.Name.ShouldBe("Test Topic"); response.CompressionAlgorithm.ShouldBe(CompressionAlgorithm.Gzip); @@ -86,7 +85,6 @@ await client.CreateTopicAsync(Identifier.String(streamName), "Get Topic", 2, var response = await client.GetTopicByIdAsync(Identifier.String(streamName), Identifier.Numeric(0)); response.ShouldNotBeNull(); - response.Id.ShouldBeGreaterThanOrEqualTo(0u); response.CreatedAt.UtcDateTime.ShouldBe(DateTimeOffset.UtcNow.UtcDateTime, TimeSpan.FromMinutes(1)); response.Name.ShouldBe("Get Topic"); response.CompressionAlgorithm.ShouldBe(CompressionAlgorithm.Gzip); @@ -113,7 +111,6 @@ await client.CreateTopicAsync(Identifier.String(streamName), "Name Topic", 2, Identifier.String("Name Topic")); response.ShouldNotBeNull(); - response.Id.ShouldBeGreaterThanOrEqualTo(0u); response.Name.ShouldBe("Name Topic"); response.CompressionAlgorithm.ShouldBe(CompressionAlgorithm.Gzip); response.Partitions!.Count().ShouldBe(2); @@ -169,7 +166,6 @@ await client.SendMessagesAsync(Identifier.String(streamName), Identifier.String("Parts Topic")); response.ShouldNotBeNull(); - response.Id.ShouldBeGreaterThanOrEqualTo(0u); response.Name.ShouldBe("Parts Topic"); response.Partitions!.Count().ShouldBe(3); response.Size.ShouldBeGreaterThan(0u); diff --git a/foreign/csharp/Iggy_SDK.Tests.Integration/UsersTests.cs b/foreign/csharp/Iggy_SDK.Tests.Integration/UsersTests.cs index 3c86131b29..85c74e0368 100644 --- a/foreign/csharp/Iggy_SDK.Tests.Integration/UsersTests.cs +++ b/foreign/csharp/Iggy_SDK.Tests.Integration/UsersTests.cs @@ -69,7 +69,6 @@ public async Task GetUser_WithoutPermissions_Should_ReturnValidResponse(Protocol var response = await client.GetUserAsync(Identifier.String(username)); response.ShouldNotBeNull(); - response.Id.ShouldBeGreaterThanOrEqualTo(0u); response.Username.ShouldBe(username); response.Status.ShouldBe(UserStatus.Active); response.CreatedAt.ShouldBeGreaterThan(0u); @@ -109,7 +108,6 @@ public async Task UpdateUser_Should_UpdateUser_Successfully(Protocol protocol) var user = await client.GetUserAsync(Identifier.String(newUsername)); user.ShouldNotBeNull(); - user.Id.ShouldBeGreaterThanOrEqualTo(0u); user.Username.ShouldBe(newUsername); user.Status.ShouldBe(UserStatus.Active); user.CreatedAt.ShouldBeGreaterThan(0u); @@ -144,7 +142,7 @@ public async Task UpdatePermissions_Should_UpdatePermissions_Successfully(Protoc user.Permissions.Global.ReadUsers.ShouldBeTrue(); user.Permissions.Global.SendMessages.ShouldBeTrue(); user.Permissions.Streams.ShouldNotBeNull(); - user.Permissions.Streams.ShouldContainKey(1); + user.Permissions.Streams.ShouldContainKey(1u); user.Permissions.Streams[1].ManageStream.ShouldBeTrue(); user.Permissions.Streams[1].ManageTopics.ShouldBeTrue(); user.Permissions.Streams[1].ReadStream.ShouldBeTrue(); @@ -152,7 +150,7 @@ public async Task UpdatePermissions_Should_UpdatePermissions_Successfully(Protoc user.Permissions.Streams[1].ReadTopics.ShouldBeTrue(); user.Permissions.Streams[1].PollMessages.ShouldBeTrue(); user.Permissions.Streams[1].Topics.ShouldNotBeNull(); - user.Permissions.Streams[1].Topics!.ShouldContainKey(1); + user.Permissions.Streams[1].Topics!.ShouldContainKey(1u); user.Permissions.Streams[1].Topics![1].ManageTopic.ShouldBeTrue(); user.Permissions.Streams[1].Topics![1].PollMessages.ShouldBeTrue(); user.Permissions.Streams[1].Topics![1].ReadTopic.ShouldBeTrue(); @@ -175,7 +173,7 @@ await Should.NotThrowAsync(client.ChangePasswordAsync(Identifier.String(username var loginClient = await Fixture.CreateClient(protocol, true); var loginResponse = await loginClient.LoginUserAsync(username, "new_password"); loginResponse.ShouldNotBeNull(); - loginResponse.UserId.ShouldBeGreaterThan(0); + loginResponse.UserId.ShouldBeGreaterThan(0u); } [Test] @@ -204,7 +202,7 @@ public async Task LoginUser_Should_LoginUser_Successfully(Protocol protocol) var response = await loginClient.LoginUserAsync(username, "login_password"); response.ShouldNotBeNull(); - response.UserId.ShouldBeGreaterThan(0); + response.UserId.ShouldBeGreaterThan(0u); switch (protocol) { case Protocol.Tcp: @@ -255,7 +253,7 @@ private static Permissions CreatePermissions() ReadUsers = true, SendMessages = true }, - Streams = new Dictionary + Streams = new Dictionary { { 1, new StreamPermissions @@ -266,7 +264,7 @@ private static Permissions CreatePermissions() SendMessages = true, ReadTopics = true, PollMessages = true, - Topics = new Dictionary + Topics = new Dictionary { { 1, new TopicPermissions diff --git a/foreign/csharp/Iggy_SDK.Tests.Integration/Vsr/VsrMessagingTests.cs b/foreign/csharp/Iggy_SDK.Tests.Integration/Vsr/VsrMessagingTests.cs index d1fdbd3e47..d6d9c5ecec 100644 --- a/foreign/csharp/Iggy_SDK.Tests.Integration/Vsr/VsrMessagingTests.cs +++ b/foreign/csharp/Iggy_SDK.Tests.Integration/Vsr/VsrMessagingTests.cs @@ -49,7 +49,7 @@ public async Task SendMessages_ToAnExplicitPartition_Should_PollBack_FromThatPar var polled = await PollAsync(client, streamName, 2); polled.Messages.Count.ShouldBe(5); - polled.PartitionId.ShouldBe(2); + polled.PartitionId.ShouldBe(2u); (await PollAsync(client, streamName, 1)).Messages.ShouldBeEmpty(); } diff --git a/foreign/csharp/Iggy_SDK/Consumers/IggyConsumer.Logging.cs b/foreign/csharp/Iggy_SDK/Consumers/IggyConsumer.Logging.cs index e853b2232d..013b4c6bc6 100644 --- a/foreign/csharp/Iggy_SDK/Consumers/IggyConsumer.Logging.cs +++ b/foreign/csharp/Iggy_SDK/Consumers/IggyConsumer.Logging.cs @@ -122,6 +122,11 @@ public partial class IggyConsumer Message = "Waiting for {Remaining} milliseconds before polling messages")] private partial void LogWaitingBeforePolling(long remaining); + [LoggerMessage(EventId = 203, + Level = LogLevel.Debug, + Message = "No partition assigned to this group member, backing off for {BackoffMs} milliseconds")] + private partial void LogNoPartitionAssignedBackingOff(int backoffMs); + [LoggerMessage(EventId = 301, Level = LogLevel.Warning, Message = "PartitionId is ignored when ConsumerType is ConsumerGroup")] diff --git a/foreign/csharp/Iggy_SDK/Consumers/IggyConsumer.Rented.cs b/foreign/csharp/Iggy_SDK/Consumers/IggyConsumer.Rented.cs index a545df5fbb..ac73a026f1 100644 --- a/foreign/csharp/Iggy_SDK/Consumers/IggyConsumer.Rented.cs +++ b/foreign/csharp/Iggy_SDK/Consumers/IggyConsumer.Rented.cs @@ -128,6 +128,13 @@ protected async Task PollRentedMessagesAsync(CancellationToken ct) if (rental.Messages.Count == 0) { + if (rental.PartitionId == PolledMessages.NoAssignedPartition) + { + LogNoPartitionAssignedBackingOff(NoAssignedPartitionBackoffMs); + await Task.Delay(NoAssignedPartitionBackoffMs, ct); + return; + } + if (_logger.IsEnabled(LogLevel.Debug)) { _logger.LogDebug("No messages received from poll for partition {PartitionId}", rental.PartitionId); @@ -136,8 +143,6 @@ protected async Task PollRentedMessagesAsync(CancellationToken ct) return; } - var partitionId = (uint)rental.PartitionId; - var hasLastOffset = _lastPolledOffset.TryGetValue(rental.PartitionId, out var lastPolledPartitionOffset); var currentOffset = 0ul; @@ -155,7 +160,7 @@ protected async Task PollRentedMessagesAsync(CancellationToken ct) batchHandle.Acquire(); try { - await PublishRentedAsync(batchHandle, message, partitionId, MessageStatus.Success, null, ct); + await PublishRentedAsync(batchHandle, message, rental.PartitionId, MessageStatus.Success, null, ct); } catch { @@ -177,13 +182,13 @@ protected async Task PollRentedMessagesAsync(CancellationToken ct) lastPolledPartitionOffset, rental.PartitionId); } - await StoreOffsetAsync(lastPolledPartitionOffset, partitionId, false, ct); + await StoreOffsetAsync(lastPolledPartitionOffset, rental.PartitionId, false, ct); } return; } - _lastPolledOffset.AddOrUpdate(rental.PartitionId, currentOffset, (_, _) => currentOffset); + _lastPolledOffset[rental.PartitionId] = currentOffset; if (_config.PollingStrategy.Kind == MessagePolling.Offset) { diff --git a/foreign/csharp/Iggy_SDK/Consumers/IggyConsumer.cs b/foreign/csharp/Iggy_SDK/Consumers/IggyConsumer.cs index 91c48a1d57..5d0bfcbe3a 100644 --- a/foreign/csharp/Iggy_SDK/Consumers/IggyConsumer.cs +++ b/foreign/csharp/Iggy_SDK/Consumers/IggyConsumer.cs @@ -18,6 +18,7 @@ using System.Collections.Concurrent; using System.Runtime.CompilerServices; using System.Threading.Channels; +using Apache.Iggy.Contracts; using Apache.Iggy.Enums; using Apache.Iggy.Exceptions; using Apache.Iggy.IggyClient; @@ -41,12 +42,18 @@ public partial class IggyConsumer : IAsyncDisposable /// private const int GroupRejoinRetryDelayMs = 1_000; + /// + /// Backoff after a group poll reported . Without it a + /// member holding zero partitions re-polls in a hot loop whenever PollingIntervalMs is zero. + /// + private const int NoAssignedPartitionBackoffMs = 100; + private readonly Channel _channel; private readonly IIggyClient _client; private readonly IggyConsumerConfig _config; private readonly SemaphoreSlim _connectionStateSemaphore = new(1, 1); private readonly EventAggregator _consumerErrorEvents; - private readonly ConcurrentDictionary _lastPolledOffset = new(); + private readonly ConcurrentDictionary _lastPolledOffset = new(); private readonly ILogger _logger; private readonly SemaphoreSlim _pollingSemaphore = new(1, 1); private readonly Channel _rentedChannel; @@ -235,7 +242,7 @@ public async Task StoreOffsetAsync(ulong offset, uint partitionId, bool resetLas if (resetLastPolled) { - _lastPolledOffset[(int)partitionId] = offset; + _lastPolledOffset[partitionId] = offset; } } @@ -409,6 +416,12 @@ private async Task PollMessagesAsync(CancellationToken ct) if (messages.Messages.Count == 0) { + if (messages.PartitionId == PolledMessages.NoAssignedPartition) + { + LogNoPartitionAssignedBackingOff(NoAssignedPartitionBackoffMs); + await Task.Delay(NoAssignedPartitionBackoffMs, ct); + } + return; } @@ -428,7 +441,7 @@ private async Task PollMessagesAsync(CancellationToken ct) { Message = message, CurrentOffset = message.Header.Offset, - PartitionId = (uint)messages.PartitionId, + PartitionId = messages.PartitionId, Status = MessageStatus.Success, Error = null }; @@ -444,20 +457,23 @@ private async Task PollMessagesAsync(CancellationToken ct) { _logger.LogDebug("No new messages found, committing offset {Offset} for partition {PartitionId}", lastPolledPartitionOffset, messages.PartitionId); - await StoreOffsetAsync(lastPolledPartitionOffset, (uint)messages.PartitionId, false, ct); + await StoreOffsetAsync(lastPolledPartitionOffset, messages.PartitionId, false, ct); } return; } - _lastPolledOffset.AddOrUpdate(messages.PartitionId, currentOffset, - (_, _) => currentOffset); + _lastPolledOffset[messages.PartitionId] = currentOffset; if (_config.PollingStrategy.Kind == MessagePolling.Offset) { _config.PollingStrategy = PollingStrategy.Offset(currentOffset + 1); } } + catch (OperationCanceledException) when (ct.IsCancellationRequested) + { + throw; + } catch (MessageDecryptionException ex) { LogFailedToDecryptMessage(ex, ex.Offset, ex.PartitionId); diff --git a/foreign/csharp/Iggy_SDK/Contracts/Auth/AuthResponse.cs b/foreign/csharp/Iggy_SDK/Contracts/Auth/AuthResponse.cs index 0cdc6d2b5e..d40a947f87 100644 --- a/foreign/csharp/Iggy_SDK/Contracts/Auth/AuthResponse.cs +++ b/foreign/csharp/Iggy_SDK/Contracts/Auth/AuthResponse.cs @@ -22,4 +22,4 @@ namespace Apache.Iggy.Contracts.Auth; ///
/// The unique identifier (numeric) of the user /// The optional tokens, used only by HTTP transport -public record AuthResponse(int UserId, TokenInfo? AccessToken); +public record AuthResponse(uint UserId, TokenInfo? AccessToken); diff --git a/foreign/csharp/Iggy_SDK/Contracts/Auth/Permissions.cs b/foreign/csharp/Iggy_SDK/Contracts/Auth/Permissions.cs index 9d11aa74a0..e611830366 100644 --- a/foreign/csharp/Iggy_SDK/Contracts/Auth/Permissions.cs +++ b/foreign/csharp/Iggy_SDK/Contracts/Auth/Permissions.cs @@ -30,5 +30,5 @@ public sealed class Permissions /// /// Permissions applied to specific streams. /// - public Dictionary? Streams { get; init; } + public Dictionary? Streams { get; init; } } diff --git a/foreign/csharp/Iggy_SDK/Contracts/ClientInfo.cs b/foreign/csharp/Iggy_SDK/Contracts/ClientInfo.cs index 902429cfe1..bbcb1326c1 100644 --- a/foreign/csharp/Iggy_SDK/Contracts/ClientInfo.cs +++ b/foreign/csharp/Iggy_SDK/Contracts/ClientInfo.cs @@ -42,12 +42,12 @@ public sealed class ClientResponse /// /// Transport protocol used by the client. /// - public required Protocol Transport { get; init; } + public required ClientTransport Transport { get; init; } /// /// Number of consumer groups the client is part of. /// - public required int ConsumerGroupsCount { get; init; } + public required uint ConsumerGroupsCount { get; init; } /// /// List of consumer groups the client is part of. diff --git a/foreign/csharp/Iggy_SDK/Contracts/ClusterNode.cs b/foreign/csharp/Iggy_SDK/Contracts/ClusterNode.cs index 6a9ee0c8fe..f4395082a6 100644 --- a/foreign/csharp/Iggy_SDK/Contracts/ClusterNode.cs +++ b/foreign/csharp/Iggy_SDK/Contracts/ClusterNode.cs @@ -46,10 +46,4 @@ public class ClusterNode /// Node status /// public required ClusterNodeStatus Status { get; set; } - - internal int GetSize() - { - // name length, name, ip length, ip, endpoints (4 * 2 bytes), role, status - return 4 + Name.Length + 4 + Ip.Length + 8 + 1 + 1; - } } diff --git a/foreign/csharp/Iggy_SDK/Contracts/ClusterNodeStatus.cs b/foreign/csharp/Iggy_SDK/Contracts/ClusterNodeStatus.cs index e930b0eb61..39e4d698af 100644 --- a/foreign/csharp/Iggy_SDK/Contracts/ClusterNodeStatus.cs +++ b/foreign/csharp/Iggy_SDK/Contracts/ClusterNodeStatus.cs @@ -45,5 +45,10 @@ public enum ClusterNodeStatus : byte /// /// Node is in maintenance mode /// - Maintenance = 4 + Maintenance = 4, + + /// + /// Node status could not be determined + /// + Unknown = 5 } diff --git a/foreign/csharp/Iggy_SDK/Contracts/ConsumerGroupInfo.cs b/foreign/csharp/Iggy_SDK/Contracts/ConsumerGroupInfo.cs index 5e55b7d42f..db81e988c1 100644 --- a/foreign/csharp/Iggy_SDK/Contracts/ConsumerGroupInfo.cs +++ b/foreign/csharp/Iggy_SDK/Contracts/ConsumerGroupInfo.cs @@ -25,15 +25,15 @@ public sealed class ConsumerGroupInfo /// /// Stream identifier. /// - public required int StreamId { get; init; } + public required uint StreamId { get; init; } /// /// Topic identifier. /// - public required int TopicId { get; init; } + public required uint TopicId { get; init; } /// /// Consumer group identifier. /// - public required int GroupId { get; init; } + public required uint GroupId { get; init; } } diff --git a/foreign/csharp/Iggy_SDK/Contracts/ConsumerGroupMembers.cs b/foreign/csharp/Iggy_SDK/Contracts/ConsumerGroupMembers.cs index 62a40a3afc..fabadc6154 100644 --- a/foreign/csharp/Iggy_SDK/Contracts/ConsumerGroupMembers.cs +++ b/foreign/csharp/Iggy_SDK/Contracts/ConsumerGroupMembers.cs @@ -30,10 +30,10 @@ public sealed class ConsumerGroupMember /// /// Number of partitions the consumer group member is consuming. /// - public required int PartitionsCount { get; init; } + public required uint PartitionsCount { get; init; } /// /// List of partition identifiers the consumer group member is consuming. /// - public required List Partitions { get; init; } + public required List Partitions { get; init; } } diff --git a/foreign/csharp/Iggy_SDK/Contracts/MessageResponse.cs b/foreign/csharp/Iggy_SDK/Contracts/MessageResponse.cs index 6d2b5bc74f..03b1472ee7 100644 --- a/foreign/csharp/Iggy_SDK/Contracts/MessageResponse.cs +++ b/foreign/csharp/Iggy_SDK/Contracts/MessageResponse.cs @@ -53,7 +53,7 @@ public Dictionary? UserHeaders if (!_userHeadersInitialized) { _userHeaders = _rawUserHeaders is { Length: > 0 } - ? BinaryMapper.TryMapHeaders(_rawUserHeaders) + ? BinaryMapper.MapHeaders(_rawUserHeaders) : null; _userHeadersInitialized = true; } diff --git a/foreign/csharp/Iggy_SDK/Contracts/OffsetResponse.cs b/foreign/csharp/Iggy_SDK/Contracts/OffsetResponse.cs index 6c0f5c3e48..ef27be094e 100644 --- a/foreign/csharp/Iggy_SDK/Contracts/OffsetResponse.cs +++ b/foreign/csharp/Iggy_SDK/Contracts/OffsetResponse.cs @@ -25,7 +25,7 @@ public sealed class OffsetResponse /// /// Partition identifier. /// - public required int PartitionId { get; init; } + public required uint PartitionId { get; init; } /// /// Current offset. diff --git a/foreign/csharp/Iggy_SDK/Contracts/PartitionResponse.cs b/foreign/csharp/Iggy_SDK/Contracts/PartitionResponse.cs index 6cb2cb7b03..16b6d65ecf 100644 --- a/foreign/csharp/Iggy_SDK/Contracts/PartitionResponse.cs +++ b/foreign/csharp/Iggy_SDK/Contracts/PartitionResponse.cs @@ -28,7 +28,7 @@ public sealed class PartitionResponse /// /// Partition identifier. /// - public required int Id { get; init; } + public required uint Id { get; init; } /// /// Number of messages in the partition. @@ -44,7 +44,7 @@ public sealed class PartitionResponse /// /// Number of segments in the partition. /// - public required int SegmentsCount { get; init; } + public required uint SegmentsCount { get; init; } /// /// Current offset of the partition. diff --git a/foreign/csharp/Iggy_SDK/Contracts/PolledMessages.cs b/foreign/csharp/Iggy_SDK/Contracts/PolledMessages.cs index 9d3f24f666..07e9131cf3 100644 --- a/foreign/csharp/Iggy_SDK/Contracts/PolledMessages.cs +++ b/foreign/csharp/Iggy_SDK/Contracts/PolledMessages.cs @@ -22,10 +22,17 @@ namespace Apache.Iggy.Contracts; /// public sealed class PolledMessages { + /// + /// Partition id an empty group poll reports when the member currently owns no partition of the group, + /// for example mid-rebalance or when the group has more members than partitions. Matches the Go and + /// Node SDKs. Consumers should back off before polling again rather than spin. + /// + public static readonly uint NoAssignedPartition = 0xFFFF_FFFE; + /// /// Partition identifier for the messages. /// - public required int PartitionId { get; init; } + public required uint PartitionId { get; init; } /// /// Current offset for the partition. diff --git a/foreign/csharp/Iggy_SDK/Contracts/PolledMessagesRental.cs b/foreign/csharp/Iggy_SDK/Contracts/PolledMessagesRental.cs index d2dae6aacd..37d47ef9b3 100644 --- a/foreign/csharp/Iggy_SDK/Contracts/PolledMessagesRental.cs +++ b/foreign/csharp/Iggy_SDK/Contracts/PolledMessagesRental.cs @@ -31,7 +31,7 @@ public sealed class PolledMessagesRental : IDisposable /// /// Partition identifier for the messages. /// - public required int PartitionId { get; init; } + public required uint PartitionId { get; init; } /// /// Current offset for the partition. diff --git a/foreign/csharp/Iggy_SDK/Contracts/RentedMessageResponse.cs b/foreign/csharp/Iggy_SDK/Contracts/RentedMessageResponse.cs index f4db2413af..7fc7f0c541 100644 --- a/foreign/csharp/Iggy_SDK/Contracts/RentedMessageResponse.cs +++ b/foreign/csharp/Iggy_SDK/Contracts/RentedMessageResponse.cs @@ -56,7 +56,7 @@ public Dictionary? UserHeaders { _userHeaders = RawUserHeaders.IsEmpty ? null - : BinaryMapper.TryMapHeaders(RawUserHeaders.Span); + : BinaryMapper.MapHeaders(RawUserHeaders.Span); _userHeadersInitialized = true; } diff --git a/foreign/csharp/Iggy_SDK/Contracts/StatsResponse.cs b/foreign/csharp/Iggy_SDK/Contracts/StatsResponse.cs index 8ca1fb9be3..772fbaa574 100644 --- a/foreign/csharp/Iggy_SDK/Contracts/StatsResponse.cs +++ b/foreign/csharp/Iggy_SDK/Contracts/StatsResponse.cs @@ -28,7 +28,7 @@ public sealed class StatsResponse /// /// Process identifier. /// - public required int ProcessId { get; init; } + public required uint ProcessId { get; init; } /// /// CPU usage of the process. @@ -90,22 +90,22 @@ public sealed class StatsResponse /// /// Total number of streams. /// - public required int StreamsCount { get; init; } + public required uint StreamsCount { get; init; } /// /// Total number of topics. /// - public required int TopicsCount { get; init; } + public required uint TopicsCount { get; init; } /// /// Total number of partitions. /// - public required int PartitionsCount { get; init; } + public required uint PartitionsCount { get; init; } /// /// Total number of segments. /// - public required int SegmentsCount { get; init; } + public required uint SegmentsCount { get; init; } /// /// Total number of messages. @@ -115,12 +115,12 @@ public sealed class StatsResponse /// /// Total number of connected clients. /// - public required int ClientsCount { get; init; } + public required uint ClientsCount { get; init; } /// /// Total number of consumer groups. /// - public required int ConsumerGroupsCount { get; init; } + public required uint ConsumerGroupsCount { get; init; } /// /// Hostname of the server. diff --git a/foreign/csharp/Iggy_SDK/Contracts/StreamPermissions.cs b/foreign/csharp/Iggy_SDK/Contracts/StreamPermissions.cs index 46ca6ce620..37011a7e9a 100644 --- a/foreign/csharp/Iggy_SDK/Contracts/StreamPermissions.cs +++ b/foreign/csharp/Iggy_SDK/Contracts/StreamPermissions.cs @@ -144,5 +144,5 @@ public sealed class StreamPermissions /// /// Permissions for topics in the stream. /// - public Dictionary? Topics { get; init; } + public Dictionary? Topics { get; init; } } diff --git a/foreign/csharp/Iggy_SDK/Contracts/StreamResponse.cs b/foreign/csharp/Iggy_SDK/Contracts/StreamResponse.cs index 39f46c795a..8555ea1128 100644 --- a/foreign/csharp/Iggy_SDK/Contracts/StreamResponse.cs +++ b/foreign/csharp/Iggy_SDK/Contracts/StreamResponse.cs @@ -56,7 +56,7 @@ public sealed class StreamResponse /// /// Number of topics in the stream. /// - public required int TopicsCount { get; init; } + public required uint TopicsCount { get; init; } /// /// List of topics in the stream. diff --git a/foreign/csharp/Iggy_SDK/Contracts/Tcp/TcpContracts.cs b/foreign/csharp/Iggy_SDK/Contracts/Tcp/TcpContracts.cs index b367cc23fc..74b3821d7a 100644 --- a/foreign/csharp/Iggy_SDK/Contracts/Tcp/TcpContracts.cs +++ b/foreign/csharp/Iggy_SDK/Contracts/Tcp/TcpContracts.cs @@ -148,26 +148,11 @@ internal static byte[] ChangePassword(Identifier userId, string currentPassword, internal static byte[] UpdatePermissions(Identifier userId, Permissions? permissions) { - var length = userId.Length + 2 + - (permissions is not null ? 1 + 4 + CalculatePermissionsSize(permissions) : 0); - Span bytes = stackalloc byte[length]; - bytes.WriteBytesFromIdentifier(userId); - var position = userId.Length + 2; - if (permissions is not null) - { - bytes[position++] = 1; - var permissionsBytes = GetBytesFromPermissions(permissions); - BinaryPrimitives.WriteInt32LittleEndian(bytes[position..(position + 4)], - permissionsBytes.Length); - position += 4; - permissionsBytes.CopyTo(bytes[position..(position + permissionsBytes.Length)]); - } - else - { - bytes[position++] = 0; - } - - return bytes.ToArray(); + var permissionsBytes = permissions is not null ? GetBytesFromPermissions(permissions) : []; + var bytes = new byte[userId.Length + 2 + 1 + (permissions is not null ? 4 + permissionsBytes.Length : 0)]; + bytes.AsSpan().WriteBytesFromIdentifier(userId); + WritePermissionsBlock(bytes.AsSpan(userId.Length + 2), permissions, permissionsBytes); + return bytes; } internal static byte[] UpdateUser(Identifier userId, string? userName, UserStatus? status) @@ -213,166 +198,98 @@ internal static byte[] CreateUser(string userName, string password, UserStatus s { var userNameLength = Encoding.UTF8.GetByteCount(userName); var passwordLength = Encoding.UTF8.GetByteCount(password); - var capacity = 3 + userNameLength + passwordLength - + (permissions is not null ? 1 + 4 + CalculatePermissionsSize(permissions) : 1); - - Span bytes = stackalloc byte[capacity]; + var permissionsBytes = permissions is not null ? GetBytesFromPermissions(permissions) : []; + var bytes = new byte[3 + userNameLength + passwordLength + 1 + + (permissions is not null ? 4 + permissionsBytes.Length : 0)]; var position = 0; bytes[position++] = (byte)userNameLength; - position += Encoding.UTF8.GetBytes(userName, bytes[position..(position + userNameLength)]); + position += Encoding.UTF8.GetBytes(userName, bytes.AsSpan(position, userNameLength)); bytes[position++] = (byte)passwordLength; - position += Encoding.UTF8.GetBytes(password, bytes[position..(position + passwordLength)]); + position += Encoding.UTF8.GetBytes(password, bytes.AsSpan(position, passwordLength)); bytes[position++] = (byte)status; - if (permissions is not null) - { - bytes[position++] = 1; - var permissionsBytes = GetBytesFromPermissions(permissions); - BinaryPrimitives.WriteInt32LittleEndian(bytes[position..(position + 4)], permissionsBytes.Length); - position += 4; - permissionsBytes.CopyTo(bytes[position..(position + permissionsBytes.Length)]); - } - else - { - bytes[position++] = 0; - } - - return bytes.ToArray(); + WritePermissionsBlock(bytes.AsSpan(position), permissions, permissionsBytes); + return bytes; } - private static byte[] GetBytesFromPermissions(Permissions data) + private static void WritePermissionsBlock(Span destination, Permissions? permissions, + byte[] permissionsBytes) { - var size = CalculatePermissionsSize(data); - Span bytes = stackalloc byte[size]; - - bytes[0] = data.Global.ManageServers ? (byte)1 : (byte)0; - bytes[1] = data.Global.ReadServers ? (byte)1 : (byte)0; - bytes[2] = data.Global.ManageUsers ? (byte)1 : (byte)0; - bytes[3] = data.Global.ReadUsers ? (byte)1 : (byte)0; - bytes[4] = data.Global.ManageStreams ? (byte)1 : (byte)0; - bytes[5] = data.Global.ReadStreams ? (byte)1 : (byte)0; - bytes[6] = data.Global.ManageTopics ? (byte)1 : (byte)0; - bytes[7] = data.Global.ReadTopics ? (byte)1 : (byte)0; - bytes[8] = data.Global.PollMessages ? (byte)1 : (byte)0; - bytes[9] = data.Global.SendMessages ? (byte)1 : (byte)0; - - - if (data.Streams is not null) + if (permissions is null) { - var streamsCount = data.Streams.Count; - var currentStream = 1; - bytes[10] = 1; - var position = 11; - foreach (var (streamId, stream) in data.Streams) - { - BinaryPrimitives.WriteInt32LittleEndian(bytes[position..(position + 4)], streamId); - position += 4; - - bytes[position] = stream.ManageStream ? (byte)1 : (byte)0; - bytes[position + 1] = stream.ReadStream ? (byte)1 : (byte)0; - bytes[position + 2] = stream.ManageTopics ? (byte)1 : (byte)0; - bytes[position + 3] = stream.ReadTopics ? (byte)1 : (byte)0; - bytes[position + 4] = stream.PollMessages ? (byte)1 : (byte)0; - bytes[position + 5] = stream.SendMessages ? (byte)1 : (byte)0; - position += 6; - - if (stream.Topics != null) - { - var topicsCount = stream.Topics.Count; - var currentTopic = 1; - bytes[position] = 1; - position += 1; - - foreach (var (topicId, topic) in stream.Topics) - { - BinaryPrimitives.WriteInt32LittleEndian(bytes[position..(position + 4)], topicId); - position += 4; - - bytes[position] = topic.ManageTopic ? (byte)1 : (byte)0; - bytes[position + 1] = topic.ReadTopic ? (byte)1 : (byte)0; - bytes[position + 2] = topic.PollMessages ? (byte)1 : (byte)0; - bytes[position + 3] = topic.SendMessages ? (byte)1 : (byte)0; - position += 4; - if (currentTopic < topicsCount) - { - currentTopic++; - bytes[position++] = 1; - } - else - { - bytes[position++] = 0; - } - } - } - else - { - bytes[position++] = 0; - } - - if (currentStream < streamsCount) - { - currentStream++; - bytes[position++] = 1; - } - else - { - bytes[position++] = 0; - } - } - } - else - { - bytes[0] = 0; + destination[0] = 0; + return; } - return bytes.ToArray(); + destination[0] = 1; + BinaryPrimitives.WriteInt32LittleEndian(destination[1..5], permissionsBytes.Length); + permissionsBytes.CopyTo(destination[5..]); } - private static int CalculatePermissionsSize(Permissions data) + private static byte[] GetBytesFromPermissions(Permissions data) { - var size = 10; - - if (data.Streams is not null) - { - size += 1; - foreach (var (_, stream) in data.Streams) + var writer = new ArrayBufferWriter(); + + WriteFlag(writer, data.Global.ManageServers); + WriteFlag(writer, data.Global.ReadServers); + WriteFlag(writer, data.Global.ManageUsers); + WriteFlag(writer, data.Global.ReadUsers); + WriteFlag(writer, data.Global.ManageStreams); + WriteFlag(writer, data.Global.ReadStreams); + WriteFlag(writer, data.Global.ManageTopics); + WriteFlag(writer, data.Global.ReadTopics); + WriteFlag(writer, data.Global.PollMessages); + WriteFlag(writer, data.Global.SendMessages); + + var hasStreams = data.Streams is { Count: > 0 }; + WriteFlag(writer, hasStreams); + if (!hasStreams) + { + return writer.WrittenSpan.ToArray(); + } + + var remainingStreams = data.Streams!.Count; + foreach (var (streamId, stream) in data.Streams) + { + BinaryPrimitives.WriteUInt32LittleEndian(writer.GetSpan(4), streamId); + writer.Advance(4); + WriteFlag(writer, stream.ManageStream); + WriteFlag(writer, stream.ReadStream); + WriteFlag(writer, stream.ManageTopics); + WriteFlag(writer, stream.ReadTopics); + WriteFlag(writer, stream.PollMessages); + WriteFlag(writer, stream.SendMessages); + + var hasTopics = stream.Topics is { Count: > 0 }; + WriteFlag(writer, hasTopics); + if (hasTopics) { - size += 4; - size += 6; - size += 1; - - if (stream.Topics is not null) + var remainingTopics = stream.Topics!.Count; + foreach (var (topicId, topic) in stream.Topics) { - size += 1; - size += stream.Topics.Count * 9; - } - else - { - size += 1; + BinaryPrimitives.WriteUInt32LittleEndian(writer.GetSpan(4), topicId); + writer.Advance(4); + WriteFlag(writer, topic.ManageTopic); + WriteFlag(writer, topic.ReadTopic); + WriteFlag(writer, topic.PollMessages); + WriteFlag(writer, topic.SendMessages); + WriteFlag(writer, --remainingTopics > 0); } } - } - else - { - size += 1; + + WriteFlag(writer, --remainingStreams > 0); } - return size; + return writer.WrittenSpan.ToArray(); } - public static byte[] FlushUnsavedBuffer(Identifier streamId, Identifier topicId, uint partitionId, bool fsync) + private static void WriteFlag(ArrayBufferWriter writer, bool value) { - var length = streamId.Length + 2 + topicId.Length + 2 + 4 + 1; - Span bytes = stackalloc byte[length]; - bytes.WriteBytesFromStreamAndTopicIdentifiers(streamId, topicId); - var position = streamId.Length + 2 + topicId.Length + 2; - BinaryPrimitives.WriteUInt32LittleEndian(bytes[position..(position + 4)], partitionId); - bytes[position + 4] = fsync ? (byte)1 : (byte)0; - - return bytes.ToArray(); + writer.GetSpan(1)[0] = value ? (byte)1 : (byte)0; + writer.Advance(1); } internal static void GetMessages(Span bytes, Consumer consumer, Identifier streamId, Identifier topicId, diff --git a/foreign/csharp/Iggy_SDK/Enums/ClientTransport.cs b/foreign/csharp/Iggy_SDK/Enums/ClientTransport.cs new file mode 100644 index 0000000000..9726e86fed --- /dev/null +++ b/foreign/csharp/Iggy_SDK/Enums/ClientTransport.cs @@ -0,0 +1,49 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +namespace Apache.Iggy.Enums; + +/// +/// Transport a connected client reached the server through, as reported by the server. +/// +public enum ClientTransport : byte +{ + /// + /// Transport this SDK does not recognize, or one newer than this SDK. + /// + Unknown = 0, + + /// + /// Custom binary protocol over TCP. + /// + Tcp = 1, + + /// + /// QUIC. + /// + Quic = 2, + + /// + /// HTTP REST. + /// + Http = 3, + + /// + /// WebSocket. + /// + WebSocket = 4 +} diff --git a/foreign/csharp/Iggy_SDK/Identifier.cs b/foreign/csharp/Iggy_SDK/Identifier.cs index 0414fac97e..8800fc542a 100644 --- a/foreign/csharp/Iggy_SDK/Identifier.cs +++ b/foreign/csharp/Iggy_SDK/Identifier.cs @@ -48,15 +48,8 @@ namespace Apache.Iggy; /// public static Identifier Numeric(int value) { - var bytes = new byte[4]; - BinaryPrimitives.WriteInt32LittleEndian(bytes, value); - - return new Identifier - { - Kind = IdKind.Numeric, - Length = 4, - Value = bytes - }; + ArgumentOutOfRangeException.ThrowIfNegative(value); + return Numeric((uint)value); } /// diff --git a/foreign/csharp/Iggy_SDK/IggyClient/IIggyConsumer.cs b/foreign/csharp/Iggy_SDK/IggyClient/IIggyConsumer.cs index 418fd5a93f..4e822dded9 100644 --- a/foreign/csharp/Iggy_SDK/IggyClient/IIggyConsumer.cs +++ b/foreign/csharp/Iggy_SDK/IggyClient/IIggyConsumer.cs @@ -41,7 +41,12 @@ public interface IIggyConsumer /// The maximum number of messages to retrieve. /// If true, automatically commit the offset after polling. /// The cancellation token to cancel the operation. - /// A task that represents the asynchronous operation and returns the polled messages. + /// + /// A task that represents the asynchronous operation and returns the polled messages. A consumer-group + /// poll whose member currently owns no partition returns an empty batch whose + /// is ; do + /// not key offsets on it, back off and poll again. + /// Task PollMessagesAsync(Identifier streamId, Identifier topicId, uint? partitionId, Consumer consumer, PollingStrategy pollingStrategy, uint count, bool autoCommit, CancellationToken token = default); @@ -52,7 +57,10 @@ Task PollMessagesAsync(Identifier streamId, Identifier topicId, /// /// /// The returned rental must be disposed when the caller is done reading the payload and raw header memory. - /// Payload and raw header slices are invalidated once the rental is disposed. + /// Payload and raw header slices are invalidated once the rental is disposed. A consumer-group poll whose + /// member currently owns no partition returns an empty batch whose + /// is ; + /// do not key offsets on it, back off and poll again. /// Task PollMessagesRentedAsync(Identifier streamId, Identifier topicId, uint? partitionId, Consumer consumer, PollingStrategy pollingStrategy, uint count, bool autoCommit, diff --git a/foreign/csharp/Iggy_SDK/IggyClient/Implementations/HttpMessageStream.cs b/foreign/csharp/Iggy_SDK/IggyClient/Implementations/HttpMessageStream.cs index 443e278b1e..8310f26c73 100644 --- a/foreign/csharp/Iggy_SDK/IggyClient/Implementations/HttpMessageStream.cs +++ b/foreign/csharp/Iggy_SDK/IggyClient/Implementations/HttpMessageStream.cs @@ -339,7 +339,7 @@ public async Task PollMessagesAsync(Identifier streamId, Identif if (MessageEncryptor is not null) { - DecryptMessages(pollMessages.Messages, (uint)pollMessages.PartitionId); + DecryptMessages(pollMessages.Messages, pollMessages.PartitionId); } return pollMessages; @@ -1003,7 +1003,7 @@ private async ValueTask ResolvePartitioningAsync(Identifier stream _ => throw new FeatureUnavailableException() }; - return Partitioning.PartitionId((int)partition); + return Partitioning.PartitionId(partition); } private void DecryptMessages(IReadOnlyList messages, uint partitionId) diff --git a/foreign/csharp/Iggy_SDK/IggyClient/Implementations/TcpMessageStream.Vsr.cs b/foreign/csharp/Iggy_SDK/IggyClient/Implementations/TcpMessageStream.Vsr.cs index d0c52f6542..56c37b5175 100644 --- a/foreign/csharp/Iggy_SDK/IggyClient/Implementations/TcpMessageStream.Vsr.cs +++ b/foreign/csharp/Iggy_SDK/IggyClient/Implementations/TcpMessageStream.Vsr.cs @@ -88,7 +88,19 @@ public sealed partial class TcpMessageStream : ISessionGenerationProvider /// RESYNC_REQUIRED_PARTITION_SENTINEL (u32::MAX). The reply header carries no status for an /// empty poll, so the sentinel is the only channel the coordinator has to ask for a re-sync. /// - private const int VsrResyncRequiredPartitionSentinel = -1; + private const uint VsrResyncRequiredPartitionSentinel = uint.MaxValue; + + /// + /// Shared empty poll result for a group member that currently owns no partition, carrying + /// so a consumer can back off instead of re-polling at + /// once. Same ownership rules as . + /// + private static readonly PolledMessagesRental NoAssignedPartitionPolledMessages = new(EmptyMemoryOwner.Instance) + { + PartitionId = PolledMessages.NoAssignedPartition, + CurrentOffset = 0, + Messages = [] + }; /// /// Shared empty poll result. An idle consumer loop returns one on every iteration, and the instance owns @@ -158,7 +170,7 @@ public sealed partial class TcpMessageStream : ISessionGenerationProvider response.ServerVersion, response.ServerProtocolVersion); SetConnectionState(ConnectionState.Authenticated); - var authResponse = new AuthResponse((int)response.UserId, null); + var authResponse = new AuthResponse(response.UserId, null); if (IsConnecting) { return authResponse; @@ -227,7 +239,7 @@ private async ValueTask ResolvePartitioningAsync(Identifier stream $"Partitioning kind {partitioning.Kind} cannot be resolved to a partition id.") }; - return Partitioning.PartitionId((int)partition); + return Partitioning.PartitionId(partition); } private async ValueTask TopicPartitionCountAsync(Identifier streamId, Identifier topicId, @@ -274,7 +286,7 @@ private async Task PollGroupMessagesRentedAsync(Identifier $"Client is not a member of consumer group {consumer.ConsumerId} on topic {topicId}."); } - return EmptyPolledMessages; + return NoAssignedPartitionPolledMessages; } PolledMessagesRental? rental = null; @@ -307,7 +319,9 @@ private async Task PollGroupMessagesRentedAsync(Identifier await SyncGroupAssignmentAsync(streamId, topicId, consumer.ConsumerId, token); } - return EmptyPolledMessages; + // Running out of attempts mid-rebalance is not a failure: report it like a member that owns nothing + // yet, so the consumer backs off and polls again. + return NoAssignedPartitionPolledMessages; } /// diff --git a/foreign/csharp/Iggy_SDK/Iggy_SDK.csproj b/foreign/csharp/Iggy_SDK/Iggy_SDK.csproj index baa0fe5801..3cacac12e1 100644 --- a/foreign/csharp/Iggy_SDK/Iggy_SDK.csproj +++ b/foreign/csharp/Iggy_SDK/Iggy_SDK.csproj @@ -26,7 +26,7 @@ under the License. net8.0;net10.0 Apache.Iggy Apache.Iggy - 0.9.0-edge.6 + 0.9.0-edge.7 true diff --git a/foreign/csharp/Iggy_SDK/Kinds/Consumer.cs b/foreign/csharp/Iggy_SDK/Kinds/Consumer.cs index 7905354ed4..c970458a5f 100644 --- a/foreign/csharp/Iggy_SDK/Kinds/Consumer.cs +++ b/foreign/csharp/Iggy_SDK/Kinds/Consumer.cs @@ -40,6 +40,17 @@ public readonly struct Consumer /// Identifier value /// Consumer instance public static Consumer New(int id) + { + ArgumentOutOfRangeException.ThrowIfNegative(id); + return New((uint)id); + } + + /// + /// Creates a new regular consumer identifier. + /// + /// Identifier value + /// Consumer instance + public static Consumer New(uint id) { return new Consumer { @@ -68,6 +79,17 @@ public static Consumer New(string id) /// Identifier value /// Consumer instance public static Consumer Group(int id) + { + ArgumentOutOfRangeException.ThrowIfNegative(id); + return Group((uint)id); + } + + /// + /// Creates a new consumer group identifier. + /// + /// Identifier value + /// Consumer instance + public static Consumer Group(uint id) { return new Consumer { diff --git a/foreign/csharp/Iggy_SDK/Kinds/Partitioning.cs b/foreign/csharp/Iggy_SDK/Kinds/Partitioning.cs index 101769a20d..9efbdd46ef 100644 --- a/foreign/csharp/Iggy_SDK/Kinds/Partitioning.cs +++ b/foreign/csharp/Iggy_SDK/Kinds/Partitioning.cs @@ -60,9 +60,20 @@ public static Partitioning None() /// Partition id /// Partitioning instance public static Partitioning PartitionId(int value) + { + ArgumentOutOfRangeException.ThrowIfNegative(value); + return PartitionId((uint)value); + } + + /// + /// Creates a partitioning strategy that use a specific partition id. + /// + /// Partition id + /// Partitioning instance + public static Partitioning PartitionId(uint value) { var bytes = new byte[4]; - BinaryPrimitives.WriteInt32LittleEndian(bytes, value); + BinaryPrimitives.WriteUInt32LittleEndian(bytes, value); return new Partitioning { diff --git a/foreign/csharp/Iggy_SDK/Mappers/BinaryMapper.cs b/foreign/csharp/Iggy_SDK/Mappers/BinaryMapper.cs index 8c6d44d6ef..c9afa5f501 100644 --- a/foreign/csharp/Iggy_SDK/Mappers/BinaryMapper.cs +++ b/foreign/csharp/Iggy_SDK/Mappers/BinaryMapper.cs @@ -33,6 +33,15 @@ namespace Apache.Iggy.Mappers; internal static class BinaryMapper { + private const int CONSUMER_GROUP_HEADER_SIZE = 13; + private const int MEMBER_HEADER_SIZE = 8; + private const int CONSUMER_GROUP_INFO_SIZE = 12; + private const int CACHE_METRICS_ENTRY_SIZE = 32; + private const int MIN_TOPIC_SIZE = 50 + 4 + 4; + private const int MIN_OPTION_SPEC_SIZE = 1 + 1 + 4 + 4; + private const int CLUSTER_NODE_TAIL_SIZE = 4 * 2 + 1 + 1; + private const int MIN_CLUSTER_NODE_SIZE = 4 + 4 + CLUSTER_NODE_TAIL_SIZE; + internal static RawPersonalAccessToken MapRawPersonalAccessToken(ReadOnlySpan payload) { var tokenLength = payload[0]; @@ -101,7 +110,7 @@ internal static UserResponse MapUser(ReadOnlySpan payload) var hasPermissions = payload[position]; if (hasPermissions == 1) { - var permissionLength = BinaryPrimitives.ReadInt32LittleEndian(payload[(position + 1)..(position + 5)]); + var permissionLength = ReadLength(payload, position + 1, "User permissions"); ReadOnlySpan permissionsPayload = payload[(position + 5)..(position + 5 + permissionLength)]; var permissions = MapPermissions(permissionsPayload); return new UserResponse @@ -128,7 +137,7 @@ internal static UserResponse MapUser(ReadOnlySpan payload) private static Permissions MapPermissions(ReadOnlySpan bytes) { - var streamMap = new Dictionary(); + var streamMap = new Dictionary(); var index = 0; var globalPermissions = new GlobalPermissions @@ -149,8 +158,8 @@ private static Permissions MapPermissions(ReadOnlySpan bytes) { while (true) { - var streamId = BinaryPrimitives.ReadInt32LittleEndian(bytes[index..(index + 4)]); - index += sizeof(int); + var streamId = BinaryPrimitives.ReadUInt32LittleEndian(bytes[index..(index + 4)]); + index += sizeof(uint); var manageStream = bytes[index++] == 1; var readStream = bytes[index++] == 1; @@ -158,14 +167,14 @@ private static Permissions MapPermissions(ReadOnlySpan bytes) var readTopics = bytes[index++] == 1; var pollMessagesStream = bytes[index++] == 1; var sendMessagesStream = bytes[index++] == 1; - var topicsMap = new Dictionary(); + var topicsMap = new Dictionary(); if (bytes[index++] == 1) { while (true) { - var topicId = BinaryPrimitives.ReadInt32LittleEndian(bytes[index..(index + 4)]); - index += sizeof(int); + var topicId = BinaryPrimitives.ReadUInt32LittleEndian(bytes[index..(index + 4)]); + index += sizeof(uint); var manageTopic = bytes[index++] == 1; var readTopic = bytes[index++] == 1; @@ -245,13 +254,14 @@ private static (UserResponse response, int position) MapToUserResponse(ReadOnlyS internal static ClientResponse MapClient(ReadOnlySpan payload) { var (response, position) = MapClientInfo(payload, 0); - var consumerGroups = new List(response.ConsumerGroupsCount); + var consumerGroups = new List(ValidatedCollectionSize(response.ConsumerGroupsCount, + payload.Length - position, CONSUMER_GROUP_INFO_SIZE, "Client consumer groups count")); for (var i = 0; i < response.ConsumerGroupsCount; i++) { - var streamId = BinaryPrimitives.ReadInt32LittleEndian(payload[position..(position + 4)]); - var topicId = BinaryPrimitives.ReadInt32LittleEndian(payload[(position + 4)..(position + 8)]); - var consumerGroupId = BinaryPrimitives.ReadInt32LittleEndian(payload[(position + 8)..(position + 12)]); + var streamId = BinaryPrimitives.ReadUInt32LittleEndian(payload[position..(position + 4)]); + var topicId = BinaryPrimitives.ReadUInt32LittleEndian(payload[(position + 4)..(position + 8)]); + var consumerGroupId = BinaryPrimitives.ReadUInt32LittleEndian(payload[(position + 8)..(position + 12)]); var consumerGroup = new ConsumerGroupInfo { @@ -295,38 +305,37 @@ internal static IReadOnlyList MapClients(ReadOnlySpan payl return response; } - private static (ClientResponse response, int position) MapClientInfo(ReadOnlySpan payload, int position) + private static (ClientResponse response, int readBytes) MapClientInfo(ReadOnlySpan payload, int position) { - int readBytes; + var start = position; var id = BinaryPrimitives.ReadUInt32LittleEndian(payload[position..(position + 4)]); var userId = BinaryPrimitives.ReadUInt32LittleEndian(payload[(position + 4)..(position + 8)]); - var transportByte = payload[position + 8]; - var transport = transportByte switch + var transport = payload[position + 8] switch { - 1 => "TCP", - 2 => "QUIC", - _ => "Unknown" + 1 => ClientTransport.Tcp, + 2 => ClientTransport.Quic, + 3 => ClientTransport.Http, + 4 => ClientTransport.WebSocket, + _ => ClientTransport.Unknown }; - var addressLength = BinaryPrimitives.ReadInt32LittleEndian(payload[(position + 9)..(position + 13)]); - var address = Encoding.UTF8.GetString(payload[(position + 13)..(position + 13 + addressLength)]); - readBytes = 4 + 1 + 4 + 4 + addressLength; - position += readBytes; - var consumerGroupsCount = BinaryPrimitives.ReadInt32LittleEndian(payload[position..(position + 4)]); - readBytes += 4; + position += 9; + var address = ReadString(payload, ref position, "Client address"); + var consumerGroupsCount = BinaryPrimitives.ReadUInt32LittleEndian(payload[position..(position + 4)]); + position += 4; return (new ClientResponse { ClientId = id, - UserId = userId, - Transport = Enum.Parse(transport, true), + UserId = userId == uint.MaxValue ? null : userId, + Transport = transport, Address = address, ConsumerGroupsCount = consumerGroupsCount - }, readBytes); + }, position - start); } internal static OffsetResponse MapOffsets(ReadOnlySpan payload) { - var partitionId = BinaryPrimitives.ReadInt32LittleEndian(payload[..4]); + var partitionId = BinaryPrimitives.ReadUInt32LittleEndian(payload[..4]); var currentOffset = BinaryPrimitives.ReadUInt64LittleEndian(payload[4..12]); var offset = BinaryPrimitives.ReadUInt64LittleEndian(payload[12..20]); @@ -343,7 +352,7 @@ internal static PolledMessagesRental MapRentedMessages(ReadOnlyMemory payl { ReadOnlySpan span = payload.Span; var length = payload.Length; - var partitionId = BinaryPrimitives.ReadInt32LittleEndian(span[..4]); + var partitionId = BinaryPrimitives.ReadUInt32LittleEndian(span[..4]); var currentOffset = BinaryPrimitives.ReadUInt64LittleEndian(span[4..12]); var messagesCount = BinaryPrimitives.ReadUInt32LittleEndian(span[12..16]); var position = 16; @@ -433,7 +442,7 @@ internal static PolledMessagesRental MapRentedMessages(ReadOnlyMemory payl } catch (Exception ex) { - throw new MessageDecryptionException(offset, (uint)partitionId, ex); + throw new MessageDecryptionException(offset, partitionId, ex); } } @@ -608,31 +617,10 @@ internal static Dictionary MapHeaders(ReadOnlySpan while (position < payload.Length) { - var keyKind = MapHeaderKind(payload[position]); - position++; - - var keyLength = BinaryPrimitives.ReadInt32LittleEndian(payload[position..(position + 4)]); - if (keyLength is 0 or > 255) - { - throw new ArgumentException("Key has incorrect size, must be between 1 and 255", nameof(keyLength)); - } - - position += 4; - var keyValue = payload[position..(position + keyLength)].ToArray(); - position += keyLength; - - var valueKind = MapHeaderKind(payload[position]); - position++; - - var valueLength = BinaryPrimitives.ReadInt32LittleEndian(payload[position..(position + 4)]); - if (valueLength is 0 or > 255) - { - throw new ArgumentException("Value has incorrect size, must be between 1 and 255", nameof(valueLength)); - } - - position += 4; - ReadOnlySpan value = payload[position..(position + valueLength)]; - position += valueLength; + var keyKind = ReadHeaderKind(payload, ref position, "key"); + var keyValue = ReadHeaderField(payload, ref position, keyKind, "key"); + var valueKind = ReadHeaderKind(payload, ref position, "value"); + var value = ReadHeaderField(payload, ref position, valueKind, "value"); headers[new HeaderKey { @@ -642,97 +630,61 @@ headers[new HeaderKey new HeaderValue { Kind = valueKind, - Value = value.ToArray() + Value = value }; } return headers; } - internal static Dictionary? TryMapHeaders(ReadOnlySpan payload) + private static HeaderKind ReadHeaderKind(ReadOnlySpan payload, ref int position, string field) { - if (payload.Length == 0 || payload[0] is 0 or > 15) + if (position >= payload.Length) { - return null; + throw new MalformedResponseException($"Header {field} kind at byte {position} is missing."); } - var headers = new Dictionary(); - var position = 0; - - while (position < payload.Length) + if (!TryMapHeaderKind(payload[position], out var kind)) { - if (!TryMapHeaderKind(payload[position], out var keyKind)) - { - return null; - } - - position++; - - if (position + 4 > payload.Length) - { - return null; - } - - var keyLength = BinaryPrimitives.ReadInt32LittleEndian(payload[position..(position + 4)]); - if (keyLength is <= 0 or > 255) - { - return null; - } - - position += 4; - if (position + keyLength > payload.Length) - { - return null; - } - - var keyValue = payload[position..(position + keyLength)].ToArray(); - position += keyLength; - - if (position >= payload.Length) - { - return null; - } - - if (!TryMapHeaderKind(payload[position], out var valueKind)) - { - return null; - } - - position++; + throw new MalformedResponseException( + $"Header {field} kind {payload[position]} at byte {position} is unknown."); + } - if (position + 4 > payload.Length) - { - return null; - } + position++; + return kind; + } - var valueLength = BinaryPrimitives.ReadInt32LittleEndian(payload[position..(position + 4)]); - if (valueLength is <= 0 or > 255) - { - return null; - } + private static byte[] ReadHeaderField(ReadOnlySpan payload, ref int position, HeaderKind kind, + string field) + { + if (position + 4 > payload.Length) + { + throw new MalformedResponseException($"Header {field} length at byte {position} is truncated."); + } - position += 4; - if (position + valueLength > payload.Length) - { - return null; - } + var length = BinaryPrimitives.ReadUInt32LittleEndian(payload[position..(position + 4)]); + if (length is 0 or > 255) + { + throw new MalformedResponseException( + $"Header {field} length {length} at byte {position} must be between 1 and 255."); + } - ReadOnlySpan value = payload[position..(position + valueLength)]; - position += valueLength; + if (!ValueLengthMatchesKind(kind, (int)length)) + { + throw new MalformedResponseException( + $"Header {field} of kind {kind} has {length} bytes, expected {HeaderKindWidth(kind)}."); + } - headers[new HeaderKey - { - Kind = keyKind, - Value = keyValue - }] = - new HeaderValue - { - Kind = valueKind, - Value = value.ToArray() - }; + position += 4; + if (position + length > payload.Length) + { + throw new MalformedResponseException( + $"Header {field} of {length} bytes at byte {position} exceeds the {payload.Length}-byte payload."); } - return headers; + var value = payload[position..(position + (int)length)].ToArray(); + position += (int)length; + return value; } internal static HeaderKind MapHeaderKind(byte value) @@ -758,6 +710,76 @@ internal static HeaderKind MapHeaderKind(byte value) }; } + /// + /// The typed accessors on slice the raw bytes by kind, so a value whose + /// length disagrees with its kind byte is rejected at parse time instead of throwing in the caller's + /// message handler. + /// + private static bool ValueLengthMatchesKind(HeaderKind kind, int length) + { + var width = HeaderKindWidth(kind); + return width == 0 || width == length; + } + + private static int HeaderKindWidth(HeaderKind kind) + { + return kind switch + { + HeaderKind.Bool or HeaderKind.Int8 or HeaderKind.Uint8 => 1, + HeaderKind.Int16 or HeaderKind.Uint16 => 2, + HeaderKind.Int32 or HeaderKind.Uint32 or HeaderKind.Float => 4, + HeaderKind.Int64 or HeaderKind.Uint64 or HeaderKind.Double => 8, + HeaderKind.Int128 or HeaderKind.Uint128 => 16, + _ => 0 + }; + } + + /// + /// The bytes left in the payload bound how many elements can exist, so a count above that is rejected + /// before the list is pre-sized instead of failing mid-loop. + /// + private static int ValidatedCollectionSize(uint count, int remaining, int minElementSize, string field) + { + if (count > Math.Max(remaining, 0) / minElementSize) + { + throw new MalformedResponseException( + $"{field} {count} exceeds remaining payload of {remaining} bytes."); + } + + return (int)count; + } + + /// + /// Length prefixes are u32 on the wire. Reading them signed would let 2^31 and above go negative and + /// surface as a slicing exception instead of . + /// + private static int ReadLength(ReadOnlySpan payload, int position, string field) + { + if (position + 4 > payload.Length) + { + throw new MalformedResponseException($"{field} length prefix at byte {position} is truncated."); + } + + var length = BinaryPrimitives.ReadUInt32LittleEndian(payload[position..(position + 4)]); + var remaining = payload.Length - position - 4; + if (length > remaining) + { + throw new MalformedResponseException( + $"{field} length {length} exceeds remaining payload of {remaining} bytes."); + } + + return (int)length; + } + + private static string ReadString(ReadOnlySpan payload, ref int position, string field) + { + var length = ReadLength(payload, position, field); + position += 4; + var value = Encoding.UTF8.GetString(payload[position..(position + length)]); + position += length; + return value; + } + private static bool TryMapHeaderKind(byte value, out HeaderKind kind) { if (value is >= 1 and <= 15) @@ -773,25 +795,15 @@ private static bool TryMapHeaderKind(byte value, out HeaderKind kind) private static Dictionary MapOptions(ReadOnlySpan payload, int position, out int readBytes) { - // Every length here is server-controlled. Read the block length as long - // so a value above int.MaxValue cannot wrap negative, and bound each - // entry against the block before slicing: an entry that overruns `end` - // would otherwise be accepted and silently consume the response bytes - // that follow the block. - var optionsLength = BinaryPrimitives.ReadUInt32LittleEndian(payload[position..(position + 4)]); - var available = (long)payload.Length - (position + 4); - if (optionsLength > available) - { - throw new MalformedResponseException( - $"Malformed options block at byte {position}: declared length {optionsLength} exceeds the " + - $"{available} bytes remaining in the payload."); - } - - readBytes = 4 + (int)optionsLength; + // Every length here is server-controlled. Bound each entry against the + // block before slicing: an entry that overruns `end` would otherwise be + // accepted and silently consume the response bytes that follow the block. + var optionsLength = ReadLength(payload, position, "Options block"); + readBytes = 4 + optionsLength; var options = new Dictionary(); var cursor = position + 4; - var end = cursor + (int)optionsLength; + var end = cursor + optionsLength; while (cursor < end) { var keyKindCode = ReadOptionByte(payload, ref cursor, end, position); @@ -809,6 +821,19 @@ private static Dictionary MapOptions(ReadOnlySpan continue; } + if (!ValueLengthMatchesKind(keyKind, key.Length)) + { + throw new MalformedResponseException( + $"Malformed options block at byte {position}: key of kind {keyKind} has {key.Length} bytes."); + } + + if (!ValueLengthMatchesKind(valueKind, value.Length)) + { + throw new MalformedResponseException( + $"Malformed options block at byte {position}: value of kind {valueKind} has {value.Length} " + + "bytes."); + } + options[new HeaderKey { Kind = keyKind, @@ -932,9 +957,8 @@ internal static StreamResponse MapStream(ReadOnlySpan payload) { var (stream, position) = MapToStream(payload, 0); - // Count-driven: topic elements carry variable-length options blocks, - // so "consume until the buffer ends" no longer delimits them. - List topics = new(stream.TopicsCount); + List topics = new(ValidatedCollectionSize(stream.TopicsCount, payload.Length - position, + MIN_TOPIC_SIZE, "Stream topics count")); for (var i = 0; i < stream.TopicsCount; i++) { var (topic, readBytes) = MapToTopic(payload, position); @@ -959,7 +983,7 @@ private static (StreamResponse stream, int readBytes) MapToStream(ReadOnlySpan MapTopics(ReadOnlySpan payload) { var topicsCount = BinaryPrimitives.ReadUInt32LittleEndian(payload[..4]); - List topics = new((int)topicsCount); var position = 4; + List topics = new(ValidatedCollectionSize(topicsCount, payload.Length - position, + MIN_TOPIC_SIZE, "Topics count")); for (var i = 0; i < topicsCount; i++) { @@ -1066,9 +1091,9 @@ private static (TopicResponse topic, int readBytes) MapToTopic(ReadOnlySpan payload, int position) { - var id = BinaryPrimitives.ReadInt32LittleEndian(payload[position..(position + 4)]); + var id = BinaryPrimitives.ReadUInt32LittleEndian(payload[position..(position + 4)]); var createdAt = BinaryPrimitives.ReadUInt64LittleEndian(payload[(position + 4)..(position + 12)]); - var segmentsCount = BinaryPrimitives.ReadInt32LittleEndian(payload[(position + 12)..(position + 16)]); + var segmentsCount = BinaryPrimitives.ReadUInt32LittleEndian(payload[(position + 12)..(position + 16)]); var currentOffset = BinaryPrimitives.ReadUInt64LittleEndian(payload[(position + 16)..(position + 24)]); var sizeBytes = BinaryPrimitives.ReadUInt64LittleEndian(payload[(position + 24)..(position + 32)]); var messagesCount = BinaryPrimitives.ReadUInt64LittleEndian(payload[(position + 32)..(position + 40)]); @@ -1103,7 +1128,7 @@ internal static List MapConsumerGroups(ReadOnlySpan internal static StatsResponse MapStats(ReadOnlySpan payload) { - var processId = BinaryPrimitives.ReadInt32LittleEndian(payload[..4]); + var processId = BinaryPrimitives.ReadUInt32LittleEndian(payload[..4]); var cpuUsage = BitConverter.ToSingle(payload[4..8]); var totalCpuUsage = BitConverter.ToSingle(payload[8..12]); var memoryUsage = BinaryPrimitives.ReadUInt64LittleEndian(payload[12..20]); @@ -1114,37 +1139,28 @@ internal static StatsResponse MapStats(ReadOnlySpan payload) var readBytes = BinaryPrimitives.ReadUInt64LittleEndian(payload[52..60]); var writtenBytes = BinaryPrimitives.ReadUInt64LittleEndian(payload[60..68]); var totalSizeBytes = BinaryPrimitives.ReadUInt64LittleEndian(payload[68..76]); - var streamsCount = BinaryPrimitives.ReadInt32LittleEndian(payload[76..80]); - var topicsCount = BinaryPrimitives.ReadInt32LittleEndian(payload[80..84]); - var partitionsCount = BinaryPrimitives.ReadInt32LittleEndian(payload[84..88]); - var segmentsCount = BinaryPrimitives.ReadInt32LittleEndian(payload[88..92]); + var streamsCount = BinaryPrimitives.ReadUInt32LittleEndian(payload[76..80]); + var topicsCount = BinaryPrimitives.ReadUInt32LittleEndian(payload[80..84]); + var partitionsCount = BinaryPrimitives.ReadUInt32LittleEndian(payload[84..88]); + var segmentsCount = BinaryPrimitives.ReadUInt32LittleEndian(payload[88..92]); var messagesCount = BinaryPrimitives.ReadUInt64LittleEndian(payload[92..100]); - var clientsCount = BinaryPrimitives.ReadInt32LittleEndian(payload[100..104]); - var consumerGroupsCount = BinaryPrimitives.ReadInt32LittleEndian(payload[104..108]); + var clientsCount = BinaryPrimitives.ReadUInt32LittleEndian(payload[100..104]); + var consumerGroupsCount = BinaryPrimitives.ReadUInt32LittleEndian(payload[104..108]); var position = 108; - var hostnameLength = BinaryPrimitives.ReadInt32LittleEndian(payload[position..(position + 4)]); - var hostname = Encoding.UTF8.GetString(payload[(position + 4)..(position + 4 + hostnameLength)]); - position += 4 + hostnameLength; - var osNameLength = BinaryPrimitives.ReadInt32LittleEndian(payload[position..(position + 4)]); - var osName = Encoding.UTF8.GetString(payload[(position + 4)..(position + 4 + osNameLength)]); - position += 4 + osNameLength; - var osVersionLength = BinaryPrimitives.ReadInt32LittleEndian(payload[position..(position + 4)]); - var osVersion = Encoding.UTF8.GetString(payload[(position + 4)..(position + 4 + osVersionLength)]); - position += 4 + osVersionLength; - var kernelVersionLength = BinaryPrimitives.ReadInt32LittleEndian(payload[position..(position + 4)]); - var kernelVersion = Encoding.UTF8.GetString(payload[(position + 4)..(position + 4 + kernelVersionLength)]); - position += 4 + kernelVersionLength; - var iggyVersionLength = BinaryPrimitives.ReadInt32LittleEndian(payload[position..(position + 4)]); - var iggyVersion = Encoding.UTF8.GetString(payload[(position + 4)..(position + 4 + iggyVersionLength)]); - position += 4 + iggyVersionLength; + var hostname = ReadString(payload, ref position, "Stats hostname"); + var osName = ReadString(payload, ref position, "Stats os name"); + var osVersion = ReadString(payload, ref position, "Stats os version"); + var kernelVersion = ReadString(payload, ref position, "Stats kernel version"); + var iggyVersion = ReadString(payload, ref position, "Stats iggy version"); var iggySemVersion = BinaryPrimitives.ReadUInt32LittleEndian(payload[position..(position + 4)]); position += 4; - var cacheMetricsLength = BinaryPrimitives.ReadInt32LittleEndian(payload[position..(position + 4)]); + var cacheMetricsLength = BinaryPrimitives.ReadUInt32LittleEndian(payload[position..(position + 4)]); position += 4; - var cacheMetricsList = new Dictionary(cacheMetricsLength); + var cacheMetricsList = new Dictionary(ValidatedCollectionSize( + cacheMetricsLength, payload.Length - position, CACHE_METRICS_ENTRY_SIZE, "Cache metrics count")); for (var i = 0; i < cacheMetricsLength; i++) { var cacheMetricsKey = new CacheMetricsKey @@ -1228,13 +1244,29 @@ internal static ConsumerGroupResponse MapConsumerGroup(ReadOnlySpan payloa private static (ConsumerGroupMember, int readBytes) MapToMember(ReadOnlySpan payload, int position) { + if (position + MEMBER_HEADER_SIZE > payload.Length) + { + throw new MalformedResponseException( + $"Malformed consumer group member at byte {position}: {payload.Length - position} bytes cannot " + + "hold a member header."); + } + var id = BinaryPrimitives.ReadUInt32LittleEndian(payload[position..(position + 4)]); - var partitionsCount = BinaryPrimitives.ReadInt32LittleEndian(payload[(position + 4)..(position + 8)]); - var partitions = new List(); + var partitionsCount = BinaryPrimitives.ReadUInt32LittleEndian(payload[(position + 4)..(position + 8)]); + + var readBytes = MEMBER_HEADER_SIZE + (long)partitionsCount * 4; + if (position + readBytes > payload.Length) + { + throw new MalformedResponseException( + $"Malformed consumer group member at byte {position}: partitions count {partitionsCount} does not " + + "fit the response."); + } + + var partitions = new List((int)partitionsCount); for (var i = 0; i < partitionsCount; i++) { - var partitionId - = BinaryPrimitives.ReadInt32LittleEndian(payload[(position + 8 + i * 4)..(position + 8 + (i + 1) * 4)]); + var partitionStart = position + MEMBER_HEADER_SIZE + i * 4; + var partitionId = BinaryPrimitives.ReadUInt32LittleEndian(payload[partitionStart..(partitionStart + 4)]); partitions.Add(partitionId); } @@ -1244,17 +1276,30 @@ var partitionId PartitionsCount = partitionsCount, Partitions = partitions }, - 8 + partitionsCount * 4); + (int)readBytes); } private static (ConsumerGroupResponse consumerGroup, int readBytes) MapToConsumerGroup(ReadOnlySpan payload, int position) { + if (position + CONSUMER_GROUP_HEADER_SIZE > payload.Length) + { + throw new MalformedResponseException( + $"Malformed consumer group at byte {position}: {payload.Length - position} bytes cannot hold a " + + "consumer group header."); + } + var id = BinaryPrimitives.ReadUInt32LittleEndian(payload[position..(position + 4)]); var partitionsCount = BinaryPrimitives.ReadUInt32LittleEndian(payload[(position + 4)..(position + 8)]); var membersCount = BinaryPrimitives.ReadUInt32LittleEndian(payload[(position + 8)..(position + 12)]); var nameLength = payload[position + 12]; - var name = Encoding.UTF8.GetString(payload[(position + 13)..(position + 13 + nameLength)]); + if (position + CONSUMER_GROUP_HEADER_SIZE + nameLength > payload.Length) + { + throw new MalformedResponseException( + $"Malformed consumer group at byte {position}: name length {nameLength} does not fit the response."); + } + + var name = Encoding.UTF8.GetString(payload[(position + CONSUMER_GROUP_HEADER_SIZE)..(position + CONSUMER_GROUP_HEADER_SIZE + nameLength)]); return (new ConsumerGroupResponse { @@ -1262,36 +1307,38 @@ private static (ConsumerGroupResponse consumerGroup, int readBytes) MapToConsume Name = name, MembersCount = membersCount, PartitionsCount = partitionsCount - }, 13 + name.Length); + }, CONSUMER_GROUP_HEADER_SIZE + nameLength); } internal static IReadOnlyList MapOptionSpecs(ReadOnlySpan payload) { var count = BinaryPrimitives.ReadUInt32LittleEndian(payload[..4]); var position = 4; - var specs = new List(); + var specs = new List(ValidatedCollectionSize(count, payload.Length - position, + MIN_OPTION_SPEC_SIZE, "Option specs count")); for (var i = 0; i < count; i++) { var keyLength = payload[position]; position += 1; - EnsureFits(payload, position, keyLength, "option key"); + if (position + keyLength > payload.Length) + { + throw new MalformedResponseException( + $"Malformed DescribeOptions response: option key of {keyLength} bytes at offset {position} " + + $"overruns the {payload.Length}-byte payload"); + } + var key = Encoding.UTF8.GetString(payload[position..(position + keyLength)]); position += keyLength; var kind = payload[position]; position += 1; - var defaultLength = (int)BinaryPrimitives.ReadUInt32LittleEndian(payload[position..(position + 4)]); + var defaultLength = ReadLength(payload, position, "Option default value"); position += 4; - EnsureFits(payload, position, defaultLength, "option default value"); var defaultValue = payload[position..(position + defaultLength)].ToArray(); position += defaultLength; - var descriptionLength = (int)BinaryPrimitives.ReadUInt32LittleEndian(payload[position..(position + 4)]); - position += 4; - EnsureFits(payload, position, descriptionLength, "option description"); - var description = Encoding.UTF8.GetString(payload[position..(position + descriptionLength)]); - position += descriptionLength; + var description = ReadString(payload, ref position, "Option description"); specs.Add(new OptionSpec { @@ -1305,31 +1352,23 @@ internal static IReadOnlyList MapOptionSpecs(ReadOnlySpan payl return specs; } - private static void EnsureFits(ReadOnlySpan payload, int position, int length, string what) + internal static ClusterMetadata MapClusterMetadata(ReadOnlySpan payload) { - if (position + length > payload.Length) + var position = 0; + var clusterName = ReadString(payload, ref position, "Cluster name"); + if (position + 4 > payload.Length) { - throw new InvalidOperationException( - $"Malformed DescribeOptions response: {what} of {length} bytes at offset {position} " + - $"overruns the {payload.Length}-byte payload"); + throw new MalformedResponseException($"Cluster nodes count at byte {position} is truncated."); } - } - - internal static ClusterMetadata MapClusterMetadata(ReadOnlySpan payload) - { - var nameLength = BinaryPrimitives.ReadUInt32LittleEndian(payload[..4]); - var clusterName = Encoding.UTF8.GetString(payload[4..(4 + (int)nameLength)]); - var position = 4 + (int)nameLength; var nodesCount = BinaryPrimitives.ReadUInt32LittleEndian(payload[position..(position + 4)]); position += 4; - var nodes = new ClusterNode[nodesCount]; - for (var i = 0; i < nodesCount; i++) + var nodes = new ClusterNode[ValidatedCollectionSize(nodesCount, payload.Length - position, + MIN_CLUSTER_NODE_SIZE, "Cluster nodes count")]; + for (var i = 0; i < nodes.Length; i++) { - var node = MapClusterNode(payload[position..]); - nodes[i] = node; - position += node.GetSize(); + nodes[i] = MapClusterNode(payload, ref position); } return new ClusterMetadata @@ -1339,37 +1378,36 @@ internal static ClusterMetadata MapClusterMetadata(ReadOnlySpan payload) }; } - private static ClusterNode MapClusterNode(ReadOnlySpan payload) + private static ClusterNode MapClusterNode(ReadOnlySpan payload, ref int position) { - var position = 0; - - // Read name - var nameLength = BinaryPrimitives.ReadUInt32LittleEndian(payload[position..(position + 4)]); - position += 4; - - var name = Encoding.UTF8.GetString(payload[position..(position + (int)nameLength)]); - position += (int)nameLength; - - // Read IP - var ipLength = BinaryPrimitives.ReadUInt32LittleEndian(payload[position..(position + 4)]); - position += 4; - - var ip = Encoding.UTF8.GetString(payload[position..(position + (int)ipLength)]); - position += (int)ipLength; + var name = ReadString(payload, ref position, "Cluster node name"); + var ip = ReadString(payload, ref position, "Cluster node ip"); + if (position + CLUSTER_NODE_TAIL_SIZE > payload.Length) + { + throw new MalformedResponseException( + $"Cluster node at byte {position}: {payload.Length - position} bytes cannot hold the ports, role and status."); + } - // Read transport endpoints (4 ports, each 2 bytes) var tcp = BinaryPrimitives.ReadUInt16LittleEndian(payload[position..(position + 2)]); - position += 2; - var quic = BinaryPrimitives.ReadUInt16LittleEndian(payload[position..(position + 2)]); - position += 2; - var http = BinaryPrimitives.ReadUInt16LittleEndian(payload[position..(position + 2)]); - position += 2; - var webSocket = BinaryPrimitives.ReadUInt16LittleEndian(payload[position..(position + 2)]); - position += 2; - - // Read role and status - var role = (ClusterNodeRole)payload[position++]; - var status = (ClusterNodeStatus)payload[position]; + var quic = BinaryPrimitives.ReadUInt16LittleEndian(payload[(position + 2)..(position + 4)]); + var http = BinaryPrimitives.ReadUInt16LittleEndian(payload[(position + 4)..(position + 6)]); + var webSocket = BinaryPrimitives.ReadUInt16LittleEndian(payload[(position + 6)..(position + 8)]); + var role = payload[position + 8] switch + { + 0 => ClusterNodeRole.Leader, + 1 => ClusterNodeRole.Follower, + var unknown => throw new MalformedResponseException($"Unknown cluster node role {unknown}.") + }; + var status = payload[position + 9] switch + { + 0 => ClusterNodeStatus.Healthy, + 1 => ClusterNodeStatus.Starting, + 2 => ClusterNodeStatus.Stopping, + 3 => ClusterNodeStatus.Unreachable, + 4 => ClusterNodeStatus.Maintenance, + _ => ClusterNodeStatus.Unknown + }; + position += 10; return new ClusterNode { diff --git a/foreign/csharp/Iggy_SDK/Vsr/ConsumerGroupClientState.cs b/foreign/csharp/Iggy_SDK/Vsr/ConsumerGroupClientState.cs index e594443ae9..4f65635796 100644 --- a/foreign/csharp/Iggy_SDK/Vsr/ConsumerGroupClientState.cs +++ b/foreign/csharp/Iggy_SDK/Vsr/ConsumerGroupClientState.cs @@ -38,12 +38,29 @@ internal sealed class ConsumerGroupClientState /// Nothing tells this client when another one resizes a topic, so the count expires on its own. private const long PartitionCountTtlMs = 30_000; - /// True when a non-empty assignment is cached for the group. + /// + /// How long a cached assignment is trusted before the next poll asks the coordinator again. A rebalance + /// that took a partition away shows up as a fenced poll long before this expires; the periodic re-sync + /// catches what fencing cannot, such as a member holding zero partitions being handed one, which no poll + /// of its own would ever reveal. Matches the Go SDK's assignmentRefreshInterval. + /// + internal static readonly long AssignmentRefreshMs = 5_000; + + /// + /// True when a fresh assignment is cached for the group, even one holding zero partitions. Treating an + /// empty assignment as missing would re-sync on every poll of a member that owns nothing; treating it as + /// fresh forever would leave that member polling nothing until an unrelated heartbeat refreshed it. + /// internal bool HasAssignment(GroupKey key) + { + return HasAssignment(key, Environment.TickCount64); + } + + internal bool HasAssignment(GroupKey key, long now) { lock (_gate) { - return _assignments.TryGetValue(key, out var assignment) && assignment.Partitions.Count > 0; + return _assignments.TryGetValue(key, out var assignment) && now < assignment.RefreshAt; } } @@ -68,6 +85,7 @@ internal void SetAssignment(GroupKey key, ulong generation, IReadOnlyList assignment.Generation = generation; assignment.Partitions = partitions; + assignment.RefreshAt = Environment.TickCount64 + AssignmentRefreshMs; } } @@ -179,7 +197,7 @@ internal void DeregisterGroup(GroupKey key) /// /// True when the last assignment sync saw this client as a member. A member mid-rebalance, or one holding /// zero partitions, is still registered, so this asks a different question than - /// . + /// . /// internal bool IsRegistered(GroupKey key) { @@ -222,6 +240,7 @@ private sealed class GroupAssignment internal IReadOnlyList Partitions { get; set; } = []; internal ulong Generation { get; set; } internal int Cursor { get; set; } + internal long RefreshAt { get; set; } } } diff --git a/foreign/csharp/Iggy_SDK_Tests/ConsumerTests/NoAssignedPartitionBackoffTests.cs b/foreign/csharp/Iggy_SDK_Tests/ConsumerTests/NoAssignedPartitionBackoffTests.cs new file mode 100644 index 0000000000..f2acad646e --- /dev/null +++ b/foreign/csharp/Iggy_SDK_Tests/ConsumerTests/NoAssignedPartitionBackoffTests.cs @@ -0,0 +1,193 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +using System.Diagnostics; +using Apache.Iggy.Consumers; +using Apache.Iggy.Contracts; +using Apache.Iggy.Enums; +using Apache.Iggy.IggyClient; +using Apache.Iggy.Kinds; +using Apache.Iggy.Messages; +using Apache.Iggy.Vsr; +using Microsoft.Extensions.Logging.Abstractions; +using Moq; + +namespace Apache.Iggy.Tests.ConsumerTests; + +/// +/// A group poll that reports must not be re-issued at once: +/// with PollingIntervalMs at zero that would spin against the coordinator until a partition arrives. +/// +public sealed class NoAssignedPartitionBackoffTests +{ + private const int MinimumGapMs = 80; + + [Fact] + public async Task given_no_assigned_partition_when_receiving_should_back_off_before_polling_again() + { + var polls = new List(); + var stopwatch = Stopwatch.StartNew(); + var mock = new Mock(MockBehavior.Loose); + mock.Setup(c => c.PollMessagesAsync(It.IsAny(), It.IsAny(), It.IsAny(), + It.IsAny(), It.IsAny(), It.IsAny(), It.IsAny(), + It.IsAny())) + .ReturnsAsync(() => + { + polls.Add(stopwatch.ElapsedMilliseconds); + + return polls.Count < 3 ? NoAssignment() : OneMessage(); + }); + var consumer = new IggyConsumer(mock.Object, BuildConfig(), NullLoggerFactory.Instance); + await consumer.InitAsync(TestContext.Current.CancellationToken); + + using var cts = new CancellationTokenSource(TimeSpan.FromSeconds(5)); + await using var messages = consumer.ReceiveAsync(cts.Token) + .GetAsyncEnumerator(TestContext.Current.CancellationToken); + Assert.True(await messages.MoveNextAsync()); + + Assert.True(polls.Count >= 3); + Assert.True(polls[1] - polls[0] >= MinimumGapMs, $"second poll came {polls[1] - polls[0]}ms after the first"); + Assert.True(polls[2] - polls[1] >= MinimumGapMs, $"third poll came {polls[2] - polls[1]}ms after the second"); + await consumer.DisposeAsync(); + } + + [Fact] + public async Task given_no_assigned_partition_when_receiving_rented_should_back_off_before_polling_again() + { + var polls = new List(); + var stopwatch = Stopwatch.StartNew(); + var mock = new Mock(MockBehavior.Loose); + mock.Setup(c => c.PollMessagesRentedAsync(It.IsAny(), It.IsAny(), + It.IsAny(), It.IsAny(), It.IsAny(), It.IsAny(), + It.IsAny(), It.IsAny())) + .ReturnsAsync(() => + { + polls.Add(stopwatch.ElapsedMilliseconds); + + return polls.Count < 3 ? NoAssignmentRental() : OneMessageRental(); + }); + var consumer = new IggyConsumer(mock.Object, BuildConfig(), NullLoggerFactory.Instance); + await consumer.InitAsync(TestContext.Current.CancellationToken); + + using var cts = new CancellationTokenSource(TimeSpan.FromSeconds(5)); + await using var messages = consumer.ReceiveRentedAsync(cts.Token) + .GetAsyncEnumerator(TestContext.Current.CancellationToken); + Assert.True(await messages.MoveNextAsync()); + messages.Current.Dispose(); + + Assert.True(polls.Count >= 3); + Assert.True(polls[1] - polls[0] >= MinimumGapMs, $"second poll came {polls[1] - polls[0]}ms after the first"); + Assert.True(polls[2] - polls[1] >= MinimumGapMs, $"third poll came {polls[2] - polls[1]}ms after the second"); + await consumer.DisposeAsync(); + } + + [Fact] + public async Task given_no_assigned_partition_when_cancelled_mid_backoff_should_not_publish_an_error() + { + var errors = 0; + var mock = new Mock(MockBehavior.Loose); + mock.Setup(c => c.PollMessagesAsync(It.IsAny(), It.IsAny(), It.IsAny(), + It.IsAny(), It.IsAny(), It.IsAny(), It.IsAny(), + It.IsAny())) + .ReturnsAsync(NoAssignment); + var consumer = new IggyConsumer(mock.Object, BuildConfig(), NullLoggerFactory.Instance); + consumer.SubscribeToErrorEvents(_ => + { + Interlocked.Increment(ref errors); + return Task.CompletedTask; + }); + await consumer.InitAsync(TestContext.Current.CancellationToken); + + using var cts = new CancellationTokenSource(TimeSpan.FromMilliseconds(30)); + await using var messages = consumer.ReceiveAsync(cts.Token) + .GetAsyncEnumerator(TestContext.Current.CancellationToken); + await Assert.ThrowsAnyAsync(async () => await messages.MoveNextAsync()); + + Assert.Equal(0, Volatile.Read(ref errors)); + await consumer.DisposeAsync(); + } + + private static IggyConsumerConfig BuildConfig() + { + return new IggyConsumerConfig + { + StreamId = Identifier.Numeric(1), + TopicId = Identifier.Numeric(1), + Consumer = Consumer.Group("group-1"), + PollingStrategy = PollingStrategy.Next(), + BatchSize = 10, + PartitionId = null, + AutoCommitMode = AutoCommitMode.Disabled, + AutoCommit = false, + PollingIntervalMs = 0 + }; + } + + private static PolledMessages NoAssignment() + { + return new PolledMessages + { + PartitionId = PolledMessages.NoAssignedPartition, + CurrentOffset = 0, + Messages = [] + }; + } + + private static PolledMessages OneMessage() + { + return new PolledMessages + { + PartitionId = 1, + CurrentOffset = 0, + Messages = + [ + new MessageResponse + { + Header = new MessageHeader + { + Offset = 0, + PayloadLength = 1 + }, + Payload = [1], + UserHeaders = null + } + ] + }; + } + + private static PolledMessagesRental NoAssignmentRental() + { + return new PolledMessagesRental(EmptyMemoryOwner.Instance) + { + PartitionId = PolledMessages.NoAssignedPartition, + CurrentOffset = 0, + Messages = [] + }; + } + + private static PolledMessagesRental OneMessageRental() + { + var owner = new RentedConsumerTests.TrackingMemoryOwner(16); + + return new PolledMessagesRental(owner) + { + PartitionId = 1, + CurrentOffset = 0, + Messages = RentedConsumerTests.BuildMessages(owner, 1) + }; + } +} diff --git a/foreign/csharp/Iggy_SDK_Tests/ContractsTests/MessageBatchGoldenVectorTests.cs b/foreign/csharp/Iggy_SDK_Tests/ContractsTests/MessageBatchGoldenVectorTests.cs index af2d80be91..5d7ef77694 100644 --- a/foreign/csharp/Iggy_SDK_Tests/ContractsTests/MessageBatchGoldenVectorTests.cs +++ b/foreign/csharp/Iggy_SDK_Tests/ContractsTests/MessageBatchGoldenVectorTests.cs @@ -141,7 +141,7 @@ public void MapRentedMessages_DecodesThePollGoldenVector() using var rental = Mappers.BinaryMapper.MapRentedMessages(pollBody, EmptyMemoryOwner.Instance); - Assert.Equal(3, rental.PartitionId); + Assert.Equal(3u, rental.PartitionId); Assert.Equal(101ul, rental.CurrentOffset); Assert.Equal(2, rental.Messages.Count); diff --git a/foreign/csharp/Iggy_SDK_Tests/ContractsTests/UserContractsTests.cs b/foreign/csharp/Iggy_SDK_Tests/ContractsTests/UserContractsTests.cs new file mode 100644 index 0000000000..7bd17fe1b0 --- /dev/null +++ b/foreign/csharp/Iggy_SDK_Tests/ContractsTests/UserContractsTests.cs @@ -0,0 +1,124 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +using System.Buffers.Binary; +using Apache.Iggy.Contracts; +using Apache.Iggy.Contracts.Auth; +using Apache.Iggy.Contracts.Tcp; + +namespace Apache.Iggy.Tests.ContractsTests; + +public sealed class UserContractsTests +{ + // [user id: kind 1 + length 1 + value 4][has permissions: 1][permissions length: 4][permissions] + private const int PermissionsOffset = 11; + + [Theory] + [InlineData(false)] + [InlineData(true)] + public void UpdatePermissions_WithGlobalOnlyPermissions_KeepsManageServersAndWritesNoStreams( + bool emptyStreamsDictionary) + { + var permissions = new Permissions + { + Global = new GlobalPermissions + { + ManageServers = true, + ReadServers = true, + ManageUsers = false, + ReadUsers = false, + ManageStreams = false, + ReadStreams = false, + ManageTopics = false, + ReadTopics = false, + PollMessages = false, + SendMessages = false + }, + Streams = emptyStreamsDictionary ? new Dictionary() : null + }; + + var bytes = TcpContracts.UpdatePermissions(Identifier.Numeric(1u), permissions); + + var permissionsLength = BinaryPrimitives.ReadInt32LittleEndian(bytes.AsSpan(PermissionsOffset - 4, 4)); + Assert.Equal(11, permissionsLength); + Assert.Equal(PermissionsOffset + 11, bytes.Length); + + var wire = bytes.AsSpan(PermissionsOffset, 11); + Assert.Equal(1, wire[0]); + Assert.Equal(1, wire[1]); + Assert.Equal(0, wire[10]); + } + + [Fact] + public void UpdatePermissions_WithNullPermissions_WritesHasPermissionsZero() + { + var bytes = TcpContracts.UpdatePermissions(Identifier.Numeric(1u), null); + + Assert.Equal(PermissionsOffset - 4, bytes.Length); + Assert.Equal(0, bytes[^1]); + } + + [Theory] + [InlineData(false)] + [InlineData(true)] + public void UpdatePermissions_WithStreamWithoutTopics_WritesHasTopicsZero(bool emptyTopicsDictionary) + { + var permissions = new Permissions + { + Global = new GlobalPermissions + { + ManageServers = false, + ReadServers = false, + ManageUsers = false, + ReadUsers = false, + ManageStreams = false, + ReadStreams = false, + ManageTopics = false, + ReadTopics = false, + PollMessages = false, + SendMessages = false + }, + Streams = new Dictionary + { + [7] = new StreamPermissions + { + ManageStream = false, + ReadStream = true, + ManageTopics = false, + ReadTopics = false, + PollMessages = false, + SendMessages = false, + Topics = emptyTopicsDictionary ? new Dictionary() : null + } + } + }; + + var bytes = TcpContracts.UpdatePermissions(Identifier.Numeric(1u), permissions); + + // [global: 10][has streams: 1][stream id: 4][stream flags: 6][has topics: 1][has next stream: 1] + const int permissionsLength = 10 + 1 + 4 + 6 + 1 + 1; + Assert.Equal(permissionsLength, BinaryPrimitives.ReadInt32LittleEndian(bytes.AsSpan(PermissionsOffset - 4, 4))); + Assert.Equal(PermissionsOffset + permissionsLength, bytes.Length); + + var wire = bytes.AsSpan(PermissionsOffset, permissionsLength); + Assert.Equal(1, wire[10]); + Assert.Equal(7u, BinaryPrimitives.ReadUInt32LittleEndian(wire.Slice(11, 4))); + Assert.Equal(1, wire[16]); + Assert.Equal(0, wire[21]); + Assert.Equal(0, wire[22]); + } +} diff --git a/foreign/csharp/Iggy_SDK_Tests/MapperTests/BinaryMapper.cs b/foreign/csharp/Iggy_SDK_Tests/MapperTests/BinaryMapper.cs index 93847e277c..b35afe3336 100644 --- a/foreign/csharp/Iggy_SDK_Tests/MapperTests/BinaryMapper.cs +++ b/foreign/csharp/Iggy_SDK_Tests/MapperTests/BinaryMapper.cs @@ -61,7 +61,7 @@ public void MapPersonalAccessTokens_ReturnsValidPersonalAccessTokenResponse() public void MapOffsets_ReturnsValidOffsetResponse() { // Arrange - var partitionId = Random.Shared.Next(1, 19); + var partitionId = (uint)Random.Shared.Next(1, 19); var currentOffset = (ulong)Random.Shared.Next(420, 69420); var storedOffset = (ulong)Random.Shared.Next(69, 420); var payload = BinaryFactory.CreateOffsetPayload(partitionId, currentOffset, storedOffset); @@ -153,7 +153,7 @@ public void MapStream_ReturnsValidStreamResponse() // Arrange var (id, _, sizeBytes, messagesCount, name, createdAt) = StreamFactory.CreateStreamsResponseFields(); // Topics are decoded count-driven, so the header count must match the appended topics. - var topicsCount = 1; + var topicsCount = 1u; var streamPayload = BinaryFactory.CreateStreamPayload(id, topicsCount, name, sizeBytes, messagesCount, createdAt); var (topicId1, partitionsCount1, topicName1, messageExpiry1, topicSizeBytes1, messagesCountTopic1, @@ -296,6 +296,22 @@ public void MapTopic_WithAnOptionOfAnUnknownKind_KeepsTheOtherEntries() Assert.Empty(response.DerivedOptions); } + [Fact] + public void MapTopic_WithAnOptionValueOfTheWrongWidth_Throws() + { + // Arrange: a Uint64 value carrying three bytes, which the Rust decoder rejects too. + const byte stringKind = 2; + const byte uint64Kind = 12; + var (topicId, partitionsCount, topicName, messageExpiry, sizeBytes, messagesCount, createdAt, + maxTopicSize) = TopicFactory.CreateTopicResponseFields(); + var options = BinaryFactory.CreateOptionEntry(stringKind, "segment_size", uint64Kind, [1, 2, 3]); + var topicPayload = BinaryFactory.CreateTopicPayload(topicId, partitionsCount, messageExpiry, topicName, + sizeBytes, messagesCount, createdAt, maxTopicSize, 1, options); + + // Act + Assert + Assert.Throws(() => Mappers.BinaryMapper.MapTopic(topicPayload)); + } + [Fact] public void MapOptionSpecs_ReturnsTheCatalogWithKindsAndDefaults() { @@ -337,7 +353,7 @@ public void MapOptionSpecs_RejectsAnEntryThatOverrunsThePayload() payload.Add(12); payload.AddRange(BitConverter.GetBytes(64u)); // claims 64 bytes that are not there - Assert.Throws(() => + Assert.Throws(() => Mappers.BinaryMapper.MapOptionSpecs(payload.ToArray())); } @@ -377,7 +393,7 @@ public void MapConsumerGroup_ReturnsValidConsumerGroupResponse() { // Arrange var (groupId, membersCount, partitionsCount, name) = ConsumerGroupFactory.CreateConsumerGroupResponseFields(); - List memberPartitions = Enumerable.Range(0, (int)partitionsCount).ToList(); + List memberPartitions = Enumerable.Range(0, (int)partitionsCount).Select(i => (uint)i).ToList(); var groupPayload = BinaryFactory.CreateGroupPayload(groupId, membersCount, partitionsCount, name, memberPartitions); @@ -394,6 +410,64 @@ var groupPayload Assert.Single(response.Members); } + [Fact] + public void MapConsumerGroup_NegativeMemberPartitionsCount_Throws() + { + var payload = new byte[21]; + BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(0, 4), 1); // group id + BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(4, 4), 3); // group partitions_count + BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(8, 4), 1); // members_count + payload[12] = 0; // name_len + BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(13, 4), 42); // member id + BinaryPrimitives.WriteInt32LittleEndian(payload.AsSpan(17, 4), -2); // member partitions_count + + Assert.Throws(() => Mappers.BinaryMapper.MapConsumerGroup(payload)); + } + + [Fact] + public void MapConsumerGroup_MemberPartitionsCountExceedsPayload_Throws() + { + var payload = BinaryFactory.CreateGroupPayload(1, 1, 3, "group", [0, 1, 2]); + BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(13 + "group".Length + 4, 4), uint.MaxValue); + + Assert.Throws(() => Mappers.BinaryMapper.MapConsumerGroup(payload)); + } + + [Fact] + public void MapConsumerGroup_TruncatedMemberHeader_Throws() + { + var payload = BinaryFactory.CreateGroupPayload(1, 1, 3, "group", [0, 1, 2]); + var truncated = payload.AsSpan(0, 13 + "group".Length + 5).ToArray(); + + Assert.Throws(() => Mappers.BinaryMapper.MapConsumerGroup(truncated)); + } + + [Fact] + public void MapConsumerGroup_NameLengthExceedsPayload_Throws() + { + var payload = BinaryFactory.CreateGroupPayload(1, 0, 3, "group"); + payload[12] = byte.MaxValue; + + Assert.Throws(() => Mappers.BinaryMapper.MapConsumerGroup(payload)); + } + + [Fact] + public void MapConsumerGroups_MultiByteName_KeepsWalkAligned() + { + var first = BinaryFactory.CreateGroupPayload(1, 2, 3, "grüppe"); + var second = BinaryFactory.CreateGroupPayload(2, 4, 5, "other"); + var combined = new byte[first.Length + second.Length]; + first.CopyTo(combined, 0); + second.CopyTo(combined, first.Length); + + var responses = Mappers.BinaryMapper.MapConsumerGroups(combined); + + Assert.Equal(2, responses.Count); + Assert.Equal("grüppe", responses[0].Name); + Assert.Equal(2u, responses[1].Id); + Assert.Equal("other", responses[1].Name); + } + [Fact] public void MapStats_ReturnsValidStatsResponse() { @@ -451,7 +525,7 @@ public void MapRentedMessages_WithEncryptor_DecryptsPayloadsAndHeadersIntoPooled using var rental = Mappers.BinaryMapper.MapRentedMessages(combined, EmptyMemoryOwner.Instance, encryptor); - Assert.Equal(7, rental.PartitionId); + Assert.Equal(7u, rental.PartitionId); Assert.Equal(101ul, rental.CurrentOffset); Assert.Equal(2, rental.Messages.Count); @@ -474,7 +548,7 @@ public void MapRentedMessages_WithEncryptor_DecryptsPayloadsAndHeadersIntoPooled } [Fact] - public void MapRentedMessages_WithEncryptor_NegativePayloadLength_ThrowsInsteadOfSpinning() + public void MapRentedMessages_WithEncryptor_NegativePayloadLength_Throws() { var encryptor = new AesMessageEncryptor(AesMessageEncryptor.GenerateKey()); @@ -615,4 +689,218 @@ private static byte[] BuildEncryptedFrame(AesMessageEncryptor encryptor, uint of return BinaryFactory.CreateMessageFrame(0, Guid.NewGuid(), offsetDelta, 0, cipherHeaders, cipherPayload); } + + [Fact] + public void MapClusterMetadata_OversizedNodesCount_Throws() + { + var payload = new byte[8]; + BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(0, 4), 0); // cluster name length + BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(4, 4), uint.MaxValue); // nodes count + + Assert.Throws(() => Mappers.BinaryMapper.MapClusterMetadata(payload)); + } + + [Fact] + public void MapClusterMetadata_MultiByteNodeName_KeepsWalkAligned() + { + var payload = CreateClusterMetadataPayload("cluster", + ("nödé-1", "10.0.0.1", 8090, 1, 1), + ("node-2", "10.0.0.2", 8091, 0, 1)); + + var metadata = Mappers.BinaryMapper.MapClusterMetadata(payload); + + Assert.Equal("cluster", metadata.Name); + Assert.Equal(2, metadata.Nodes.Length); + Assert.Equal("nödé-1", metadata.Nodes[0].Name); + Assert.Equal("10.0.0.1", metadata.Nodes[0].Ip); + Assert.Equal(8090, metadata.Nodes[0].Endpoints.Tcp); + Assert.Equal("node-2", metadata.Nodes[1].Name); + Assert.Equal("10.0.0.2", metadata.Nodes[1].Ip); + Assert.Equal(8091, metadata.Nodes[1].Endpoints.Tcp); + } + + [Fact] + public void MapClusterMetadata_TruncatedNode_Throws() + { + var payload = CreateClusterMetadataPayload("cluster", ("node-1", "10.0.0.1", 8090, 1, 1)); + var truncated = payload.AsSpan(0, payload.Length - 3).ToArray(); + + Assert.Throws(() => Mappers.BinaryMapper.MapClusterMetadata(truncated)); + } + + [Fact] + public void MapClusterMetadata_TruncatedInsideLengthPrefix_Throws() + { + var payload = CreateClusterMetadataPayload("cluster", ("node-1", "10.0.0.1", 8090, 1, 1)); + // cut inside the u32 length prefix of the node ip, right after the node name + var cut = 4 + "cluster".Length + 4 + 4 + "node-1".Length + 2; + var truncated = payload.AsSpan(0, cut).ToArray(); + + Assert.Throws(() => Mappers.BinaryMapper.MapClusterMetadata(truncated)); + } + + [Fact] + public void MapClusterMetadata_TruncatedBeforeNodesCount_Throws() + { + var payload = CreateClusterMetadataPayload("cluster"); + var truncated = payload.AsSpan(0, payload.Length - 2).ToArray(); + + Assert.Throws(() => Mappers.BinaryMapper.MapClusterMetadata(truncated)); + } + + [Fact] + public void MapClusterMetadata_NameLengthExceedsPayload_Throws() + { + var payload = CreateClusterMetadataPayload("cluster", ("node-1", "10.0.0.1", 8090, 1, 1)); + BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(0, 4), uint.MaxValue); + + Assert.Throws(() => Mappers.BinaryMapper.MapClusterMetadata(payload)); + } + + [Fact] + public void MapClient_OversizedConsumerGroupsCount_Throws() + { + var payload = new byte[17]; + BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(0, 4), 1); // client id + BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(4, 4), 1); // user id + payload[8] = 1; // transport tcp + BinaryPrimitives.WriteInt32LittleEndian(payload.AsSpan(9, 4), 0); // address length + BinaryPrimitives.WriteInt32LittleEndian(payload.AsSpan(13, 4), int.MaxValue); // consumer groups count + + Assert.Throws(() => Mappers.BinaryMapper.MapClient(payload)); + } + + [Theory] + [InlineData(0, ClientTransport.Unknown)] + [InlineData(1, ClientTransport.Tcp)] + [InlineData(2, ClientTransport.Quic)] + [InlineData(3, ClientTransport.Http)] + [InlineData(4, ClientTransport.WebSocket)] + [InlineData(9, ClientTransport.Unknown)] + public void MapClient_MapsEveryWireTransport(byte wire, ClientTransport expected) + { + var payload = CreateClientPayload(userId: 7, transport: wire); + + var response = Mappers.BinaryMapper.MapClient(payload); + + Assert.Equal(expected, response.Transport); + Assert.Equal(7u, response.UserId); + } + + [Fact] + public void MapClient_UnauthenticatedSentinel_YieldsNullUserId() + { + var payload = CreateClientPayload(userId: uint.MaxValue, transport: 1); + + var response = Mappers.BinaryMapper.MapClient(payload); + + Assert.Null(response.UserId); + } + + [Theory] + [InlineData(5)] + [InlineData(6)] + [InlineData(200)] + public void MapClusterMetadata_UnknownStatus_MapsToUnknown(byte status) + { + var payload = CreateClusterMetadataPayload("cluster", ("node", "127.0.0.1", 8090, 1, status)); + + var metadata = Mappers.BinaryMapper.MapClusterMetadata(payload); + + Assert.Equal(ClusterNodeStatus.Unknown, metadata.Nodes[0].Status); + } + + [Fact] + public void MapClusterMetadata_UnknownRole_Throws() + { + var payload = CreateClusterMetadataPayload("cluster", ("node", "127.0.0.1", 8090, 2, 0)); + + Assert.Throws(() => Mappers.BinaryMapper.MapClusterMetadata(payload)); + } + + [Fact] + public void MapClient_AddressLengthAboveInt32_ThrowsMalformed() + { + var payload = CreateClientPayload(userId: 7, transport: 1); + BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(9, 4), 0x8000_0000); + + Assert.Throws(() => Mappers.BinaryMapper.MapClient(payload)); + } + + private static byte[] CreateClientPayload(uint userId, byte transport) + { + var payload = new byte[17]; + BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(0, 4), 1); + BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(4, 4), userId); + payload[8] = transport; + BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(9, 4), 0); + BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(13, 4), 0); + return payload; + } + + [Fact] + public void MapTopics_OversizedTopicsCount_Throws() + { + var payload = new byte[4]; + BinaryPrimitives.WriteUInt32LittleEndian(payload, uint.MaxValue); + + Assert.Throws(() => Mappers.BinaryMapper.MapTopics(payload)); + } + + [Fact] + public void MapHeaders_KeyLengthDisagreesWithKind_Throws() + { + // key: kind Uint32 (11) with a single byte; value: string "v". + var bytes = new byte[] { 11, 1, 0, 0, 0, 65, 2, 1, 0, 0, 0, 118 }; + + Assert.Throws(() => Mappers.BinaryMapper.MapHeaders(bytes)); + } + + [Fact] + public void MapHeaders_FixedWidthValueOfMatchingLength_Parses() + { + var bytes = new byte[] { 2, 1, 0, 0, 0, 65, 12, 8, 0, 0, 0, 7, 0, 0, 0, 0, 0, 0, 0 }; + + Dictionary headers = Mappers.BinaryMapper.MapHeaders(bytes); + + var value = Assert.Single(headers).Value; + Assert.Equal(HeaderKind.Uint64, value.Kind); + Assert.Equal(7ul, value.ToUInt64()); + } + + [Fact] + public void MapHeaders_ValueLengthDisagreesWithKind_Throws() + { + var bytes = new byte[] { 2, 1, 0, 0, 0, 65, 6, 2, 0, 0, 0, 7, 0 }; + + Assert.Throws(() => Mappers.BinaryMapper.MapHeaders(bytes)); + } + + private static byte[] CreateClusterMetadataPayload(string clusterName, + params (string name, string ip, ushort tcp, byte role, byte status)[] nodes) + { + var buffer = new List(); + WriteString(buffer, clusterName); + buffer.AddRange(BitConverter.GetBytes((uint)nodes.Length)); + foreach (var (name, ip, tcp, role, status) in nodes) + { + WriteString(buffer, name); + WriteString(buffer, ip); + buffer.AddRange(BitConverter.GetBytes(tcp)); + buffer.AddRange(BitConverter.GetBytes((ushort)0)); // quic + buffer.AddRange(BitConverter.GetBytes((ushort)0)); // http + buffer.AddRange(BitConverter.GetBytes((ushort)0)); // websocket + buffer.Add(role); + buffer.Add(status); + } + + return buffer.ToArray(); + + static void WriteString(List buffer, string value) + { + var bytes = Encoding.UTF8.GetBytes(value); + buffer.AddRange(BitConverter.GetBytes((uint)bytes.Length)); + buffer.AddRange(bytes); + } + } } diff --git a/foreign/csharp/Iggy_SDK_Tests/MapperTests/HeaderEncryptionTests.cs b/foreign/csharp/Iggy_SDK_Tests/MapperTests/HeaderEncryptionTests.cs index 8d8a62e040..872d5b4d9f 100644 --- a/foreign/csharp/Iggy_SDK_Tests/MapperTests/HeaderEncryptionTests.cs +++ b/foreign/csharp/Iggy_SDK_Tests/MapperTests/HeaderEncryptionTests.cs @@ -17,6 +17,7 @@ using Apache.Iggy.Contracts.Tcp; using Apache.Iggy.Encryption; +using Apache.Iggy.Exceptions; using Apache.Iggy.Shared; namespace Apache.Iggy.Tests.MapperTests; @@ -44,8 +45,7 @@ public void Headers_should_survive_encrypt_decrypt_roundtrip() Assert.True(encrypted.Length > headerBytes.Length); // Encrypted bytes must not parse as valid headers - var parsed = Mappers.BinaryMapper.TryMapHeaders(encrypted); - Assert.Null(parsed); + Assert.Throws(() => Mappers.BinaryMapper.MapHeaders(encrypted)); var decrypted = encryptor.DecryptToArray(encrypted); Assert.Equal(headerBytes, decrypted); @@ -62,85 +62,85 @@ public void Headers_should_survive_encrypt_decrypt_roundtrip() } [Fact] - public void TryMapHeaders_returns_null_on_invalid_first_byte() + public void MapHeaders_throws_on_invalid_first_byte() { var random = new byte[64]; Random.Shared.NextBytes(random); random[0] = 0; - Assert.Null(Mappers.BinaryMapper.TryMapHeaders(random)); + Assert.Throws(() => Mappers.BinaryMapper.MapHeaders(random)); } [Fact] - public void TryMapHeaders_returns_null_on_first_byte_above_range() + public void MapHeaders_throws_on_first_byte_above_range() { - Assert.Null(Mappers.BinaryMapper.TryMapHeaders(new byte[] { 16, 0, 0, 0 })); + Assert.Throws(() => Mappers.BinaryMapper.MapHeaders(new byte[] { 16, 0, 0, 0 })); } [Fact] - public void TryMapHeaders_returns_null_on_empty_payload() + public void MapHeaders_returns_empty_on_empty_payload() { - Assert.Null(Mappers.BinaryMapper.TryMapHeaders([])); + Assert.Empty(Mappers.BinaryMapper.MapHeaders([])); } [Fact] - public void TryMapHeaders_returns_null_on_truncated_key_length() + public void MapHeaders_throws_on_truncated_key_length() { // Valid header kind byte but not enough bytes for key length - Assert.Null(Mappers.BinaryMapper.TryMapHeaders(new byte[] { 2 })); + Assert.Throws(() => Mappers.BinaryMapper.MapHeaders(new byte[] { 2 })); } [Fact] - public void TryMapHeaders_returns_null_on_zero_key_length() + public void MapHeaders_throws_on_zero_key_length() { // Valid kind, then key length = 0 (invalid) - Assert.Null(Mappers.BinaryMapper.TryMapHeaders(new byte[] { 2, 0, 0, 0, 0 })); + Assert.Throws(() => Mappers.BinaryMapper.MapHeaders(new byte[] { 2, 0, 0, 0, 0 })); } [Fact] - public void TryMapHeaders_returns_null_on_key_length_exceeding_payload() + public void MapHeaders_throws_on_key_length_exceeding_payload() { // Valid kind, key length = 100 but only a few bytes remain - Assert.Null(Mappers.BinaryMapper.TryMapHeaders(new byte[] { 2, 100, 0, 0, 0, 1, 2 })); + Assert.Throws(() => Mappers.BinaryMapper.MapHeaders(new byte[] { 2, 100, 0, 0, 0, 1, 2 })); } [Fact] - public void TryMapHeaders_returns_null_on_truncated_value_kind() + public void MapHeaders_throws_on_truncated_value_kind() { // Valid kind(2=String), key_len=1, key='A', then no value kind byte - Assert.Null(Mappers.BinaryMapper.TryMapHeaders(new byte[] { 2, 1, 0, 0, 0, 65 })); + Assert.Throws(() => Mappers.BinaryMapper.MapHeaders(new byte[] { 2, 1, 0, 0, 0, 65 })); } [Fact] - public void TryMapHeaders_returns_null_on_invalid_value_kind() + public void MapHeaders_throws_on_invalid_value_kind() { // Valid kind(2), key_len=1, key='A', invalid value kind=0 - Assert.Null(Mappers.BinaryMapper.TryMapHeaders(new byte[] { 2, 1, 0, 0, 0, 65, 0 })); + Assert.Throws(() => Mappers.BinaryMapper.MapHeaders(new byte[] { 2, 1, 0, 0, 0, 65, 0 })); } [Fact] - public void TryMapHeaders_returns_null_on_truncated_value_length() + public void MapHeaders_throws_on_truncated_value_length() { // Valid kind(2), key_len=1, key='A', valid value kind(2), then not enough for value length - Assert.Null(Mappers.BinaryMapper.TryMapHeaders(new byte[] { 2, 1, 0, 0, 0, 65, 2, 1 })); + Assert.Throws(() => Mappers.BinaryMapper.MapHeaders(new byte[] { 2, 1, 0, 0, 0, 65, 2, 1 })); } [Fact] - public void TryMapHeaders_returns_null_on_zero_value_length() + public void MapHeaders_throws_on_zero_value_length() { // Valid kind(2), key_len=1, key='A', valid value kind(2), value_len=0 (invalid) - Assert.Null(Mappers.BinaryMapper.TryMapHeaders(new byte[] { 2, 1, 0, 0, 0, 65, 2, 0, 0, 0, 0 })); + Assert.Throws(() => Mappers.BinaryMapper.MapHeaders(new byte[] { 2, 1, 0, 0, 0, 65, 2, 0, 0, 0, 0 })); } [Fact] - public void TryMapHeaders_returns_null_on_value_length_exceeding_payload() + public void MapHeaders_throws_on_value_length_exceeding_payload() { // Valid kind(2), key_len=1, key='A', valid value kind(2), value_len=100 but not enough bytes - Assert.Null(Mappers.BinaryMapper.TryMapHeaders(new byte[] { 2, 1, 0, 0, 0, 65, 2, 100, 0, 0, 0 })); + Assert.Throws(() => Mappers.BinaryMapper.MapHeaders(new byte[] { 2, 1, 0, 0, 0, 65, 2, 100, 0, 0, 0 })); } [Fact] - public void TryMapHeaders_returns_valid_headers_on_plaintext() + public void MapHeaders_returns_valid_headers_on_plaintext() { var headers = new Dictionary { @@ -148,14 +148,14 @@ public void TryMapHeaders_returns_valid_headers_on_plaintext() }; var bytes = HeadersToBytes(headers); - var result = Mappers.BinaryMapper.TryMapHeaders(bytes); + var result = Mappers.BinaryMapper.MapHeaders(bytes); Assert.NotNull(result); Assert.Single(result); } [Fact] - public void TryMapHeaders_returns_valid_for_all_header_kinds() + public void MapHeaders_returns_valid_for_all_header_kinds() { var headers = new Dictionary { @@ -167,7 +167,7 @@ public void TryMapHeaders_returns_valid_for_all_header_kinds() }; var bytes = HeadersToBytes(headers); - var result = Mappers.BinaryMapper.TryMapHeaders(bytes); + var result = Mappers.BinaryMapper.MapHeaders(bytes); Assert.NotNull(result); Assert.Equal(5, result.Count); diff --git a/foreign/csharp/Iggy_SDK_Tests/MapperTests/OptionsBlockGoldenVectorTests.cs b/foreign/csharp/Iggy_SDK_Tests/MapperTests/OptionsBlockGoldenVectorTests.cs index 62c375a453..fe4ed73040 100644 --- a/foreign/csharp/Iggy_SDK_Tests/MapperTests/OptionsBlockGoldenVectorTests.cs +++ b/foreign/csharp/Iggy_SDK_Tests/MapperTests/OptionsBlockGoldenVectorTests.cs @@ -18,6 +18,7 @@ using System.Buffers.Binary; using System.Text; using Apache.Iggy.Contracts.Tcp; +using Apache.Iggy.Exceptions; using Apache.Iggy.Headers; namespace Apache.Iggy.Tests.MapperTests; @@ -79,6 +80,15 @@ public void MapTopic_DecodesTheCrossSdkGoldenVector() Assert.Empty(topic.DerivedOptions!); } + [Fact] + public void MapTopic_WithOptionsLengthPrefixCutShort_ThrowsMalformedResponse() + { + var payload = TopicPayloadWithOptions(GoldenOptionsBlock); + var truncated = payload[..(50 + "topic".Length + 2)]; + + Assert.Throws(() => Mappers.BinaryMapper.MapTopic(truncated)); + } + /// /// A topic response carrying as its explicit block and an empty /// derived block, with no partitions after it. diff --git a/foreign/csharp/Iggy_SDK_Tests/UtilityTests/IdentifiersByteSerializationTests.cs b/foreign/csharp/Iggy_SDK_Tests/UtilityTests/IdentifiersByteSerializationTests.cs index 4d128c4175..60aa552eb8 100644 --- a/foreign/csharp/Iggy_SDK_Tests/UtilityTests/IdentifiersByteSerializationTests.cs +++ b/foreign/csharp/Iggy_SDK_Tests/UtilityTests/IdentifiersByteSerializationTests.cs @@ -45,4 +45,29 @@ public void KeyBytes_WithInvalidLength_ShouldThrowArgumentException() var val = Enumerable.Range(0, 500).Select(x => (byte)x).ToArray(); Assert.Throws(() => Partitioning.EntityIdBytes(val)); } + + [Fact] + public void NumericIdentifier_WithNegativeValue_ShouldThrow() + { + Assert.Throws(() => Identifier.Numeric(-1)); + } + + [Fact] + public void NumericIdentifier_IntAndUintOverloads_ProduceSameBytes() + { + Assert.Equal(Identifier.Numeric(42u).Value, Identifier.Numeric(42).Value); + } + + [Fact] + public void PartitionId_WithNegativeValue_ShouldThrow() + { + Assert.Throws(() => Partitioning.PartitionId(-1)); + } + + [Fact] + public void Consumer_WithNegativeId_ShouldThrow() + { + Assert.Throws(() => Consumer.New(-1)); + Assert.Throws(() => Consumer.Group(-1)); + } } diff --git a/foreign/csharp/Iggy_SDK_Tests/Utils/BinaryFactory.cs b/foreign/csharp/Iggy_SDK_Tests/Utils/BinaryFactory.cs index d0c6078f62..8255f349d4 100644 --- a/foreign/csharp/Iggy_SDK_Tests/Utils/BinaryFactory.cs +++ b/foreign/csharp/Iggy_SDK_Tests/Utils/BinaryFactory.cs @@ -33,10 +33,10 @@ internal static byte[] CreatePersonalAccessTokensPayload(string name, uint expir return result.ToArray(); } - internal static byte[] CreateOffsetPayload(int partitionId, ulong currentOffset, ulong offset) + internal static byte[] CreateOffsetPayload(uint partitionId, ulong currentOffset, ulong offset) { var payload = new byte[20]; - BinaryPrimitives.WriteInt32LittleEndian(payload, partitionId); + BinaryPrimitives.WriteUInt32LittleEndian(payload, partitionId); BinaryPrimitives.WriteUInt64LittleEndian(payload.AsSpan(4), currentOffset); BinaryPrimitives.WriteUInt64LittleEndian(payload.AsSpan(12), offset); return payload; @@ -88,7 +88,7 @@ internal static byte[] CreateBatchRecord(ulong baseOffset, ulong baseTimestamp, return record; } - internal static byte[] CreateStreamPayload(uint id, int topicsCount, string name, ulong sizeBytes, + internal static byte[] CreateStreamPayload(uint id, uint topicsCount, string name, ulong sizeBytes, ulong messagesCount, ulong createdAt) { var nameBytes = Encoding.UTF8.GetBytes(name); @@ -96,7 +96,7 @@ internal static byte[] CreateStreamPayload(uint id, int topicsCount, string name var payload = new byte[totalSize]; BinaryPrimitives.WriteUInt32LittleEndian(payload, id); BinaryPrimitives.WriteUInt64LittleEndian(payload.AsSpan(4), createdAt); - BinaryPrimitives.WriteInt32LittleEndian(payload.AsSpan(12), topicsCount); + BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(12), topicsCount); BinaryPrimitives.WriteUInt64LittleEndian(payload.AsSpan(16), sizeBytes); BinaryPrimitives.WriteUInt64LittleEndian(payload.AsSpan(24), messagesCount); payload[32] = (byte)nameBytes.Length; @@ -152,35 +152,24 @@ internal static byte[] CreateOptionEntry(byte keyKind, string key, byte valueKin return entry; } - internal static byte[] CreatePartitionPayload(int id, int segmentsCount, int currentOffset, ulong sizeBytes, - ulong messagesCount) - { - var payload = new byte[16]; - BinaryPrimitives.WriteInt32LittleEndian(payload, id); - BinaryPrimitives.WriteInt32LittleEndian(payload.AsSpan(4), segmentsCount); - BinaryPrimitives.WriteInt32LittleEndian(payload.AsSpan(8), currentOffset); - BinaryPrimitives.WriteUInt64LittleEndian(payload.AsSpan(12), sizeBytes); - BinaryPrimitives.WriteUInt64LittleEndian(payload.AsSpan(16), messagesCount); - return payload; - } - internal static byte[] CreateGroupPayload(uint id, uint membersCount, uint partitionsCount, string name, - List? partitionsOnMember = null) + List? partitionsOnMember = null) { - var payload = new byte[13 + name.Length + (partitionsOnMember?.Count * 4 + 8 ?? 0)]; + var nameBytes = Encoding.UTF8.GetBytes(name); + var payload = new byte[13 + nameBytes.Length + (partitionsOnMember?.Count * 4 + 8 ?? 0)]; BinaryPrimitives.WriteUInt32LittleEndian(payload, id); BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(4), partitionsCount); BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(8), membersCount); - payload[12] = (byte)name.Length; - var nameBytes = Encoding.UTF8.GetBytes(name); + payload[12] = (byte)nameBytes.Length; nameBytes.CopyTo(payload.AsSpan(13)); if (partitionsOnMember is not null) { - BinaryPrimitives.WriteInt32LittleEndian(payload.AsSpan(13 + name.Length), 30); - BinaryPrimitives.WriteInt32LittleEndian(payload.AsSpan(17 + name.Length), partitionsOnMember.Count); + var memberStart = 13 + nameBytes.Length; + BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(memberStart), 30); + BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(memberStart + 4), (uint)partitionsOnMember.Count); for (var i = 0; i < partitionsOnMember.Count; i++) { - BinaryPrimitives.WriteInt32LittleEndian(payload.AsSpan(21 + name.Length + i * 4), + BinaryPrimitives.WriteUInt32LittleEndian(payload.AsSpan(memberStart + 8 + i * 4), partitionsOnMember[i]); } } @@ -191,7 +180,7 @@ internal static byte[] CreateGroupPayload(uint id, uint membersCount, uint parti internal static byte[] CreateStatsPayload(StatsResponse stats) { var bytes = new byte[1024]; - BinaryPrimitives.WriteInt32LittleEndian(bytes.AsSpan(0, 4), stats.ProcessId); + BinaryPrimitives.WriteUInt32LittleEndian(bytes.AsSpan(0, 4), stats.ProcessId); BinaryPrimitives.WriteSingleLittleEndian(bytes.AsSpan(4, 4), stats.CpuUsage); BinaryPrimitives.WriteSingleLittleEndian(bytes.AsSpan(8, 8), stats.TotalCpuUsage); BinaryPrimitives.WriteUInt64LittleEndian(bytes.AsSpan(12, 8), stats.MemoryUsage); @@ -203,32 +192,32 @@ internal static byte[] CreateStatsPayload(StatsResponse stats) BinaryPrimitives.WriteUInt64LittleEndian(bytes.AsSpan(52, 8), stats.ReadBytes); BinaryPrimitives.WriteUInt64LittleEndian(bytes.AsSpan(60, 8), stats.WrittenBytes); BinaryPrimitives.WriteUInt64LittleEndian(bytes.AsSpan(68, 8), stats.MessagesSizeBytes); - BinaryPrimitives.WriteInt32LittleEndian(bytes.AsSpan(76, 4), stats.StreamsCount); - BinaryPrimitives.WriteInt32LittleEndian(bytes.AsSpan(80, 4), stats.TopicsCount); - BinaryPrimitives.WriteInt32LittleEndian(bytes.AsSpan(84, 4), stats.PartitionsCount); - BinaryPrimitives.WriteInt32LittleEndian(bytes.AsSpan(88, 4), stats.SegmentsCount); + BinaryPrimitives.WriteUInt32LittleEndian(bytes.AsSpan(76, 4), stats.StreamsCount); + BinaryPrimitives.WriteUInt32LittleEndian(bytes.AsSpan(80, 4), stats.TopicsCount); + BinaryPrimitives.WriteUInt32LittleEndian(bytes.AsSpan(84, 4), stats.PartitionsCount); + BinaryPrimitives.WriteUInt32LittleEndian(bytes.AsSpan(88, 4), stats.SegmentsCount); BinaryPrimitives.WriteUInt64LittleEndian(bytes.AsSpan(92, 8), stats.MessagesCount); - BinaryPrimitives.WriteInt32LittleEndian(bytes.AsSpan(100, 4), stats.ClientsCount); - BinaryPrimitives.WriteInt32LittleEndian(bytes.AsSpan(104, 4), stats.ConsumerGroupsCount); + BinaryPrimitives.WriteUInt32LittleEndian(bytes.AsSpan(100, 4), stats.ClientsCount); + BinaryPrimitives.WriteUInt32LittleEndian(bytes.AsSpan(104, 4), stats.ConsumerGroupsCount); // Convert string properties to bytes and set them in the byte array var hostnameBytes = Encoding.UTF8.GetBytes(stats.Hostname); - BinaryPrimitives.WriteInt32LittleEndian(bytes.AsSpan(108, 4), hostnameBytes.Length); + BinaryPrimitives.WriteUInt32LittleEndian(bytes.AsSpan(108, 4), (uint)hostnameBytes.Length); hostnameBytes.CopyTo(bytes, 112); var osNameBytes = Encoding.UTF8.GetBytes(stats.OsName); - BinaryPrimitives.WriteInt32LittleEndian(bytes.AsSpan(112 + hostnameBytes.Length, 4), osNameBytes.Length); + BinaryPrimitives.WriteUInt32LittleEndian(bytes.AsSpan(112 + hostnameBytes.Length, 4), (uint)osNameBytes.Length); osNameBytes.CopyTo(bytes, 116 + hostnameBytes.Length); var osVersionBytes = Encoding.UTF8.GetBytes(stats.OsVersion); - BinaryPrimitives.WriteInt32LittleEndian(bytes.AsSpan(116 + hostnameBytes.Length + osNameBytes.Length, 4), - osVersionBytes.Length); + BinaryPrimitives.WriteUInt32LittleEndian(bytes.AsSpan(116 + hostnameBytes.Length + osNameBytes.Length, 4), + (uint)osVersionBytes.Length); osVersionBytes.CopyTo(bytes, 120 + hostnameBytes.Length + osNameBytes.Length); var kernelVersionBytes = Encoding.UTF8.GetBytes(stats.KernelVersion); - BinaryPrimitives.WriteInt32LittleEndian( + BinaryPrimitives.WriteUInt32LittleEndian( bytes.AsSpan(120 + hostnameBytes.Length + osNameBytes.Length + osVersionBytes.Length, 4), - kernelVersionBytes.Length); + (uint)kernelVersionBytes.Length); kernelVersionBytes.CopyTo(bytes, 124 + hostnameBytes.Length + osNameBytes.Length + osVersionBytes.Length); return bytes; diff --git a/foreign/csharp/Iggy_SDK_Tests/Utils/Streams/StreamFactory.cs b/foreign/csharp/Iggy_SDK_Tests/Utils/Streams/StreamFactory.cs index 92f1b1d375..77728a2049 100644 --- a/foreign/csharp/Iggy_SDK_Tests/Utils/Streams/StreamFactory.cs +++ b/foreign/csharp/Iggy_SDK_Tests/Utils/Streams/StreamFactory.cs @@ -19,11 +19,11 @@ namespace Apache.Iggy.Tests.Utils.Streams; internal static class StreamFactory { - internal static (uint id, int topicsCount, ulong sizeBytes, ulong messagesCount, string name, ulong createdAt) + internal static (uint id, uint topicsCount, ulong sizeBytes, ulong messagesCount, string name, ulong createdAt) CreateStreamsResponseFields() { var id = (uint)Random.Shared.Next(1, 69); - var topicsCount = Random.Shared.Next(1, 69); + var topicsCount = (uint)Random.Shared.Next(1, 69); var sizeBytes = (ulong)Random.Shared.Next(69, 42069); var messageCount = (ulong)Random.Shared.Next(2, 3); var name = "Stream " + Random.Shared.Next(1, 4) + Utility.RandomString(3).ToLower(); diff --git a/foreign/csharp/Iggy_SDK_Tests/Utils/Users/PermissionsFactory.cs b/foreign/csharp/Iggy_SDK_Tests/Utils/Users/PermissionsFactory.cs index d6ab48097c..607487f03e 100644 --- a/foreign/csharp/Iggy_SDK_Tests/Utils/Users/PermissionsFactory.cs +++ b/foreign/csharp/Iggy_SDK_Tests/Utils/Users/PermissionsFactory.cs @@ -41,10 +41,10 @@ internal static Permissions CreatePermissions() PollMessages = Random.Shared.Next() % 2 == 0, SendMessages = Random.Shared.Next() % 2 == 0 }, - Streams = new Dictionary + Streams = new Dictionary { { - Random.Shared.Next(1, 30), new StreamPermissions + (uint)Random.Shared.Next(1, 30), new StreamPermissions { ManageStream = Random.Shared.Next() % 2 == 0, ReadStream = Random.Shared.Next() % 2 == 0, @@ -52,10 +52,10 @@ internal static Permissions CreatePermissions() ReadTopics = Random.Shared.Next() % 2 == 0, PollMessages = Random.Shared.Next() % 2 == 0, SendMessages = Random.Shared.Next() % 2 == 0, - Topics = new Dictionary + Topics = new Dictionary { { - Random.Shared.Next(1, 30), + (uint)Random.Shared.Next(1, 30), new TopicPermissions { ManageTopic = Random.Shared.Next() % 2 == 0, @@ -65,7 +65,7 @@ internal static Permissions CreatePermissions() } }, { - Random.Shared.Next(31, 69), + (uint)Random.Shared.Next(31, 69), new TopicPermissions { ManageTopic = Random.Shared.Next() % 2 == 0, @@ -78,7 +78,7 @@ internal static Permissions CreatePermissions() } }, { - Random.Shared.Next(31, 69), new StreamPermissions + (uint)Random.Shared.Next(31, 69), new StreamPermissions { ManageStream = Random.Shared.Next() % 2 == 0, ReadStream = Random.Shared.Next() % 2 == 0, @@ -95,7 +95,7 @@ internal static Permissions CreatePermissions() internal static Permissions PermissionsFromBytes(byte[] bytes) { - var streamMap = new Dictionary(); + var streamMap = new Dictionary(); var index = 0; var globalPermissions = new GlobalPermissions @@ -116,8 +116,8 @@ internal static Permissions PermissionsFromBytes(byte[] bytes) { while (true) { - var streamId = BinaryPrimitives.ReadInt32LittleEndian(bytes[index..(index + 4)]); - index += sizeof(int); + var streamId = BinaryPrimitives.ReadUInt32LittleEndian(bytes[index..(index + 4)]); + index += sizeof(uint); var manageStream = bytes[index++] == 1; var readStream = bytes[index++] == 1; @@ -125,14 +125,14 @@ internal static Permissions PermissionsFromBytes(byte[] bytes) var readTopics = bytes[index++] == 1; var pollMessagesStream = bytes[index++] == 1; var sendMessagesStream = bytes[index++] == 1; - var topicsMap = new Dictionary(); + var topicsMap = new Dictionary(); if (bytes[index++] == 1) { while (true) { - var topicId = BinaryPrimitives.ReadInt32LittleEndian(bytes[index..(index + 4)]); - index += sizeof(int); + var topicId = BinaryPrimitives.ReadUInt32LittleEndian(bytes[index..(index + 4)]); + index += sizeof(uint); var manageTopic = bytes[index++] == 1; var readTopic = bytes[index++] == 1; diff --git a/foreign/csharp/Iggy_SDK_Tests/Utils/Users/UsersFactory.cs b/foreign/csharp/Iggy_SDK_Tests/Utils/Users/UsersFactory.cs index aef0443bb5..d307e38851 100644 --- a/foreign/csharp/Iggy_SDK_Tests/Utils/Users/UsersFactory.cs +++ b/foreign/csharp/Iggy_SDK_Tests/Utils/Users/UsersFactory.cs @@ -33,10 +33,10 @@ internal static CreateUserRequest CreateUserRequest(string? username = null, str permissions ?? CreatePermissions()); } - internal static Dictionary CreateStreamPermissions(int streamId = 1, int topicId = 1) + internal static Dictionary CreateStreamPermissions(uint streamId = 1, uint topicId = 1) { - var streamsPermission = new Dictionary(); - var topicPermissions = new Dictionary(); + var streamsPermission = new Dictionary(); + var topicPermissions = new Dictionary(); topicPermissions.Add(topicId, new TopicPermissions { diff --git a/foreign/csharp/Iggy_SDK_Tests/VsrTests/ConsumerGroupClientStateTests.cs b/foreign/csharp/Iggy_SDK_Tests/VsrTests/ConsumerGroupClientStateTests.cs index c1c4f00ada..c1bb394fee 100644 --- a/foreign/csharp/Iggy_SDK_Tests/VsrTests/ConsumerGroupClientStateTests.cs +++ b/foreign/csharp/Iggy_SDK_Tests/VsrTests/ConsumerGroupClientStateTests.cs @@ -94,7 +94,8 @@ public void MemberHoldingNoPartitions_StaysRegistered() state.RegisterGroup(Key, Identifier.Numeric(1), Identifier.Numeric(2), Identifier.Numeric(3)); state.SetAssignment(Key, 1, []); - Assert.False(state.HasAssignment(Key)); + Assert.True(state.HasAssignment(Key)); + Assert.Null(state.NextGroupPartition(Key)); Assert.True(state.IsRegistered(Key)); state.DeregisterGroup(Key); @@ -102,6 +103,17 @@ public void MemberHoldingNoPartitions_StaysRegistered() Assert.False(state.IsRegistered(Key)); } + [Fact] + public void HasAssignment_ExpiresAfterTheRefreshInterval() + { + var state = new ConsumerGroupClientState(); + state.SetAssignment(Key, 1, []); + var now = Environment.TickCount64; + + Assert.True(state.HasAssignment(Key, now)); + Assert.False(state.HasAssignment(Key, now + ConsumerGroupClientState.AssignmentRefreshMs)); + } + [Fact] public void RegisteredGroups_ReturnsJoinedIdentifiers() { diff --git a/foreign/csharp/Iggy_SDK_Tests/VsrTests/EndpointFailoverTests.cs b/foreign/csharp/Iggy_SDK_Tests/VsrTests/EndpointFailoverTests.cs index 3f6ad316c6..399abb1c68 100644 --- a/foreign/csharp/Iggy_SDK_Tests/VsrTests/EndpointFailoverTests.cs +++ b/foreign/csharp/Iggy_SDK_Tests/VsrTests/EndpointFailoverTests.cs @@ -27,6 +27,7 @@ using Apache.Iggy.IggyClient.Implementations; using Apache.Iggy.Vsr; using Microsoft.Extensions.Logging.Abstractions; +using static Apache.Iggy.Tests.VsrTests.MockFrames; namespace Apache.Iggy.Tests.VsrTests; @@ -37,26 +38,6 @@ namespace Apache.Iggy.Tests.VsrTests; /// public sealed class EndpointFailoverTests { - private const int HeaderSize = 256; - private const int SizeOffset = 48; - private const int CommandOffset = 60; - private const int RequestIdOffset = 168; - private const int RequestOperationOffset = 176; - private const int RequestReservedOffset = 196; - private const int ReplyRequestIdOffset = 200; - private const int ReplyOperationOffset = 208; - private const int ReplyStatusOffset = 216; - - private const byte CommandReply = 8; - private const byte CommandEviction = 13; - private const int EvictionReasonOffset = 255; - private const byte EvictionStaleClient = 13; - private const byte OperationRegister = 1; - private const byte OperationNonReplicated = 2; - private const int GetClusterMetadataCode = 12; - private const int PingCode = 1; - private const uint TransientNotAccepted = 58; - [Fact] public async Task ResumesOnASurvivorAfterTheSignedInNodeDies() { @@ -65,11 +46,11 @@ public async Task ResumesOnASurvivorAfterTheSignedInNodeDies() // The primary leads, so the sign-in settles there and the roster is only remembered - not acted on - // until the node dies. - primary.Serve(request => request.Code == GetClusterMetadataCode - ? Reply(OperationNonReplicated, ClusterMetadata(primary.Port, survivor.Port, primary.Port)) + primary.Serve(request => request.Code == GET_CLUSTER_METADATA_CODE + ? Reply(OPERATION_NON_REPLICATED, ClusterMetadata(primary.Port, survivor.Port, primary.Port)) : Answer(request)); - survivor.Serve(request => request.Code == GetClusterMetadataCode - ? Reply(OperationNonReplicated, ClusterMetadata(primary.Port, survivor.Port, survivor.Port)) + survivor.Serve(request => request.Code == GET_CLUSTER_METADATA_CODE + ? Reply(OPERATION_NON_REPLICATED, ClusterMetadata(primary.Port, survivor.Port, survivor.Port)) : Answer(request)); var configuration = new IggyClientConfigurator @@ -114,13 +95,13 @@ public async Task WalksPastTwoRefusingReplicasToThePartitionPrimary() metadataLeader.Serve(request => request.Code switch { - GetClusterMetadataCode => Reply(OperationNonReplicated, + GET_CLUSTER_METADATA_CODE => Reply(OPERATION_NON_REPLICATED, ThreeNodeClusterMetadata(metadataLeader.Port, follower.Port, partitionPrimary.Port)), - (int)commandCode => Reply(request.Operation, [], TransientNotAccepted), + (int)commandCode => Reply(request.Operation, [], TRANSIENT_NOT_ACCEPTED), _ => Answer(request) }); follower.Serve(request => request.Code == (int)commandCode - ? Reply(request.Operation, [], TransientNotAccepted) + ? Reply(request.Operation, [], TRANSIENT_NOT_ACCEPTED) : Answer(request)); partitionPrimary.Serve(Answer); @@ -166,12 +147,12 @@ public async Task WalksTheWholeRosterBeyondTheMetadataRedirectCap() byte[] Refuse(MockRequest request) { return request.Code == (int)commandCode - ? Reply(request.Operation, [], TransientNotAccepted) + ? Reply(request.Operation, [], TRANSIENT_NOT_ACCEPTED) : Answer(request); } - metadataLeader.Serve(request => request.Code == GetClusterMetadataCode - ? Reply(OperationNonReplicated, RosterMetadata(metadataLeader.Port, roster)) + metadataLeader.Serve(request => request.Code == GET_CLUSTER_METADATA_CODE + ? Reply(OPERATION_NON_REPLICATED, RosterMetadata(metadataLeader.Port, roster)) : Refuse(request)); second.Serve(Refuse); third.Serve(Refuse); @@ -212,18 +193,18 @@ public async Task ServerEvictionReplaysTheRememberedSignIn() var evict = false; node.Serve(request => { - if (request.Operation == OperationRegister) + if (request.Operation == OPERATION_REGISTER) { - return Reply(OperationRegister, RegisterBody(session: 128)); + return Reply(OPERATION_REGISTER, RegisterBody(session: 128)); } if (evict) { evict = false; - return EvictionFrame(EvictionStaleClient); + return EvictionFrame(EVICTION_STALE_CLIENT); } - return Reply(OperationNonReplicated, request.Code == GetClusterMetadataCode + return Reply(OPERATION_NON_REPLICATED, request.Code == GET_CLUSTER_METADATA_CODE ? ClusterMetadata(node.Port, node.Port, node.Port) : []); }); @@ -268,18 +249,18 @@ public async Task ServerEvictionDuringAReplicatedWriteReplaysTheRememberedSignIn var evict = false; node.Serve(request => { - if (request.Operation == OperationRegister) + if (request.Operation == OPERATION_REGISTER) { - return Reply(OperationRegister, RegisterBody(session: 128)); + return Reply(OPERATION_REGISTER, RegisterBody(session: 128)); } if (evict) { evict = false; - return EvictionFrame(EvictionStaleClient); + return EvictionFrame(EVICTION_STALE_CLIENT); } - return Reply(OperationNonReplicated, request.Code == GetClusterMetadataCode + return Reply(OPERATION_NON_REPLICATED, request.Code == GET_CLUSTER_METADATA_CODE ? ClusterMetadata(node.Port, node.Port, node.Port) : []); }); @@ -331,8 +312,8 @@ public async Task ResumesOnASurvivorThatComesUpWhileTheClientIsRetrying() probe.Stop(); using var primary = new MockNode(); - primary.Serve(request => request.Code == GetClusterMetadataCode - ? Reply(OperationNonReplicated, ClusterMetadata(primary.Port, survivorPort, primary.Port)) + primary.Serve(request => request.Code == GET_CLUSTER_METADATA_CODE + ? Reply(OPERATION_NON_REPLICATED, ClusterMetadata(primary.Port, survivorPort, primary.Port)) : Answer(request)); var configuration = new IggyClientConfigurator @@ -364,8 +345,8 @@ public async Task ResumesOnASurvivorThatComesUpWhileTheClientIsRetrying() { await Task.Delay(300, TestContext.Current.CancellationToken); survivor = new MockNode(survivorPort); - survivor.Serve(request => request.Code == GetClusterMetadataCode - ? Reply(OperationNonReplicated, ClusterMetadata(primary.Port, survivorPort, survivorPort)) + survivor.Serve(request => request.Code == GET_CLUSTER_METADATA_CODE + ? Reply(OPERATION_NON_REPLICATED, ClusterMetadata(primary.Port, survivorPort, survivorPort)) : Answer(request)); }, TestContext.Current.CancellationToken); @@ -386,10 +367,10 @@ public async Task ResumesOnASurvivorThatComesUpWhileTheClientIsRetrying() private static byte[] EvictionFrame(byte reason) { - var frame = new byte[HeaderSize]; - BinaryPrimitives.WriteUInt32LittleEndian(frame.AsSpan(SizeOffset, 4), HeaderSize); - frame[CommandOffset] = CommandEviction; - frame[EvictionReasonOffset] = reason; + var frame = new byte[HEADER_SIZE]; + BinaryPrimitives.WriteUInt32LittleEndian(frame.AsSpan(SIZE_OFFSET, 4), HEADER_SIZE); + frame[COMMAND_OFFSET] = COMMAND_EVICTION; + frame[EVICTION_REASON_OFFSET] = reason; return frame; } @@ -397,8 +378,8 @@ private static byte[] EvictionFrame(byte reason) public async Task FailsFastWhenNothingEverSignedIn() { using var node = new MockNode(); - node.Serve(request => request.Code == GetClusterMetadataCode - ? Reply(OperationNonReplicated, ClusterMetadata(node.Port, node.Port, node.Port)) + node.Serve(request => request.Code == GET_CLUSTER_METADATA_CODE + ? Reply(OPERATION_NON_REPLICATED, ClusterMetadata(node.Port, node.Port, node.Port)) : Answer(request)); var configuration = new IggyClientConfigurator @@ -457,49 +438,6 @@ public async Task FailsFastWhenNothingEverSignedIn() return (false, lastError); } - /// A reply for anything the roster read does not claim: a register, or an empty read. - private static byte[] Answer(MockRequest request) - { - return request.Operation == OperationRegister - ? Reply(OperationRegister, RegisterBody(session: 128)) - : Reply(OperationNonReplicated, []); - } - - private static byte[] Reply(byte operation, byte[] body) - { - return Reply(operation, body, 0); - } - - private static byte[] Reply(byte operation, byte[] body, uint status) - { - var frame = new byte[HeaderSize + body.Length]; - BinaryPrimitives.WriteUInt32LittleEndian(frame.AsSpan(SizeOffset, 4), (uint)frame.Length); - frame[CommandOffset] = CommandReply; - frame[ReplyOperationOffset] = operation; - BinaryPrimitives.WriteUInt32LittleEndian(frame.AsSpan(ReplyStatusOffset, 4), status); - body.CopyTo(frame.AsSpan(HeaderSize)); - - return frame; - } - - /// - /// A register reply carries a committed result section, so its four leading zero bytes announce zero - /// entries and the typed payload starts right after them. A non-replicated read carries none. - /// - private static byte[] RegisterBody(ulong session) - { - var serverVersion = Encoding.UTF8.GetBytes("0.0.0"); - var body = new byte[4 + 17 + serverVersion.Length]; - var payload = body.AsSpan(4); - BinaryPrimitives.WriteUInt32LittleEndian(payload[..4], 7); - BinaryPrimitives.WriteUInt64LittleEndian(payload[4..12], session); - BinaryPrimitives.WriteUInt32LittleEndian(payload[12..16], 11 << 10); - payload[16] = (byte)serverVersion.Length; - serverVersion.CopyTo(payload[17..]); - - return body; - } - private static byte[] ClusterMetadata(ushort primaryPort, ushort survivorPort, ushort leaderPort) { var body = new List(); @@ -554,149 +492,4 @@ private static void WriteString(List body, string value) body.AddRange(BitConverter.GetBytes((uint)bytes.Length)); body.AddRange(bytes); } - - private readonly record struct MockRequest(byte Operation, int Code, ulong RequestId); - - /// - /// A loopback VSR node. Killing it drops the live sockets and stops accepting, so a redial is refused the - /// way a dead process refuses one. - /// - private sealed class MockNode : IDisposable - { - private readonly TcpListener _listener; - private readonly List _accepted = []; - private volatile bool _killed; - private int _connections; - private int _pings; - private int _registrations; - - /// - /// A port to bind, for a node that has to come up on an address the client already knows. Zero - /// takes whatever the OS hands out. - /// - public MockNode(ushort port = 0) - { - _listener = new TcpListener(IPAddress.Loopback, port); - _listener.Start(); - Port = (ushort)((IPEndPoint)_listener.LocalEndpoint).Port; - } - - public ushort Port { get; } - - public int Pings => Volatile.Read(ref _pings); - - public int Registrations => Volatile.Read(ref _registrations); - - public int Connections - { - get - { - lock (_accepted) - { - return _connections; - } - } - } - - public void Serve(Func handler) - { - _ = Task.Run(async () => - { - while (!_killed) - { - TcpClient connection; - try - { - connection = await _listener.AcceptTcpClientAsync(); - } - catch (Exception) - { - return; - } - - lock (_accepted) - { - _accepted.Add(connection); - _connections++; - } - - _ = Task.Run(() => Exchange(connection, handler)); - } - }); - } - - public void Kill() - { - _killed = true; - lock (_accepted) - { - foreach (var connection in _accepted) - { - connection.Close(); - } - - _accepted.Clear(); - } - - _listener.Stop(); - } - - public void Dispose() - { - Kill(); - } - - private async Task Exchange(TcpClient connection, Func handler) - { - try - { - await using var stream = connection.GetStream(); - var header = new byte[HeaderSize]; - while (!_killed) - { - await ReadExactly(stream, header); - var size = BinaryPrimitives.ReadUInt32LittleEndian(header.AsSpan(SizeOffset, 4)); - var body = new byte[size - HeaderSize]; - await ReadExactly(stream, body); - - var request = new MockRequest(header[RequestOperationOffset], - BinaryPrimitives.ReadInt32LittleEndian(header.AsSpan(RequestReservedOffset, 4)), - BinaryPrimitives.ReadUInt64LittleEndian(header.AsSpan(RequestIdOffset, 8))); - if (request.Operation == OperationRegister) - { - Interlocked.Increment(ref _registrations); - } - else if (request.Code == PingCode) - { - Interlocked.Increment(ref _pings); - } - - var reply = handler(request); - BinaryPrimitives.WriteUInt64LittleEndian(reply.AsSpan(ReplyRequestIdOffset, 8), - request.RequestId); - await stream.WriteAsync(reply); - await stream.FlushAsync(); - } - } - catch (Exception) - { - // A killed node and a client that went away look the same here. - } - } - - private static async Task ReadExactly(NetworkStream stream, byte[] buffer) - { - var read = 0; - while (read < buffer.Length) - { - var chunk = await stream.ReadAsync(buffer.AsMemory(read)); - if (chunk == 0) - { - throw new EndOfStreamException("Connection closed"); - } - - read += chunk; - } - } - } } diff --git a/foreign/csharp/Iggy_SDK_Tests/VsrTests/GroupPollingTests.cs b/foreign/csharp/Iggy_SDK_Tests/VsrTests/GroupPollingTests.cs new file mode 100644 index 0000000000..3eedf5e932 --- /dev/null +++ b/foreign/csharp/Iggy_SDK_Tests/VsrTests/GroupPollingTests.cs @@ -0,0 +1,150 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +using System.Buffers.Binary; +using Apache.Iggy.Configuration; +using Apache.Iggy.Contracts; +using Apache.Iggy.Enums; +using Apache.Iggy.IggyClient.Implementations; +using Apache.Iggy.Kinds; +using Apache.Iggy.Utils; +using Microsoft.Extensions.Logging.Abstractions; +using static Apache.Iggy.Tests.VsrTests.MockFrames; + +namespace Apache.Iggy.Tests.VsrTests; + +/// +/// Group polls resolve the partition client-side from the coordinator's assignment. Mirrors +/// foreign/go/client/tcp/tcp_group_polling_test.go. +/// +public sealed class GroupPollingTests +{ + private const int SyncGroupCode = CommandCodes.SYNC_CONSUMER_GROUP_CODE; + private const int PollMessagesCode = CommandCodes.POLL_MESSAGES_CODE; + + [Fact] + public async Task given_member_holding_no_partitions_when_polled_twice_should_report_no_assignment_and_sync_once() + { + using var node = new MockNode(); + node.Serve(request => request.Code == SyncGroupCode + ? Reply(OPERATION_NON_REPLICATED, AssignmentBody(9, [])) + : Answer(request)); + using var client = await ConnectAsync(node); + + var first = await PollOnceAsync(client); + var second = await PollOnceAsync(client); + + Assert.Equal(PolledMessages.NoAssignedPartition, first.PartitionId); + Assert.Empty(first.Messages); + Assert.Equal(PolledMessages.NoAssignedPartition, second.PartitionId); + Assert.Empty(second.Messages); + // The empty assignment is cached: a member that owns nothing must not re-sync on every poll. + Assert.Equal(1, node.Requests(SyncGroupCode)); + Assert.Equal(0, node.Requests(PollMessagesCode)); + } + + [Fact] + public async Task given_rebalance_outlasting_the_attempts_when_polled_should_report_no_assignment() + { + using var node = new MockNode(); + node.Serve(request => request.Code switch + { + SyncGroupCode => Reply(OPERATION_NON_REPLICATED, AssignmentBody(9, [0])), + // The server marks a stale assignment with the resync sentinel. + PollMessagesCode => Reply(request.Operation, EmptyBatchBody(uint.MaxValue)), + _ => Answer(request) + }); + using var client = await ConnectAsync(node); + + var polled = await PollOnceAsync(client); + + Assert.Equal(PolledMessages.NoAssignedPartition, polled.PartitionId); + Assert.Empty(polled.Messages); + Assert.Equal(2, node.Requests(PollMessagesCode)); + Assert.Equal(3, node.Requests(SyncGroupCode)); + } + + [Fact] + public async Task given_fenced_poll_when_resynced_should_poll_the_new_assignment() + { + using var node = new MockNode(); + var generation = 1ul; + node.Serve(request => + { + switch (request.Code) + { + case SyncGroupCode: + return Reply(OPERATION_NON_REPLICATED, AssignmentBody(generation, [(uint)generation])); + case PollMessagesCode when generation == 1: + // The member no longer owns the partition at this generation. + generation = 2; + return Reply(request.Operation, EmptyBatchBody(uint.MaxValue)); + case PollMessagesCode: + return Reply(request.Operation, EmptyBatchBody(2)); + default: + return Answer(request); + } + }); + using var client = await ConnectAsync(node); + + var polled = await PollOnceAsync(client); + + Assert.Equal(2u, polled.PartitionId); + Assert.Equal(2, node.Requests(SyncGroupCode)); + Assert.Equal(2, node.Requests(PollMessagesCode)); + } + + private static async Task ConnectAsync(MockNode node) + { + var configuration = new IggyClientConfigurator + { + BaseAddress = $"127.0.0.1:{node.Port}", + Protocol = Protocol.Tcp + }; + var client = new TcpMessageStream(configuration, NullLoggerFactory.Instance); + await client.ConnectAsync(TestContext.Current.CancellationToken); + + return client; + } + + private static Task PollOnceAsync(TcpMessageStream client) + { + return client.PollMessagesAsync(Identifier.Numeric(1), Identifier.Numeric(2), null, Consumer.Group(3), + PollingStrategy.Next(), 10, false, TestContext.Current.CancellationToken); + } + + private static byte[] AssignmentBody(ulong generation, uint[] partitions) + { + var body = new byte[12 + partitions.Length * 4]; + BinaryPrimitives.WriteUInt64LittleEndian(body.AsSpan(0, 8), generation); + BinaryPrimitives.WriteUInt32LittleEndian(body.AsSpan(8, 4), (uint)partitions.Length); + for (var index = 0; index < partitions.Length; index++) + { + BinaryPrimitives.WriteUInt32LittleEndian(body.AsSpan(12 + index * 4, 4), partitions[index]); + } + + return body; + } + + private static byte[] EmptyBatchBody(uint partitionId) + { + var body = new byte[16]; + BinaryPrimitives.WriteUInt32LittleEndian(body.AsSpan(0, 4), partitionId); + + return body; + } +} diff --git a/foreign/csharp/Iggy_SDK_Tests/VsrTests/MockNode.cs b/foreign/csharp/Iggy_SDK_Tests/VsrTests/MockNode.cs new file mode 100644 index 0000000000..55e247054a --- /dev/null +++ b/foreign/csharp/Iggy_SDK_Tests/VsrTests/MockNode.cs @@ -0,0 +1,251 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +using System.Buffers.Binary; +using System.Net; +using System.Net.Sockets; +using System.Text; +using static Apache.Iggy.Tests.VsrTests.MockFrames; + +namespace Apache.Iggy.Tests.VsrTests; + +/// VSR frame layout and reply builders shared by the loopback mock node tests. +internal static class MockFrames +{ + internal const int HEADER_SIZE = 256; + internal const int SIZE_OFFSET = 48; + internal const int COMMAND_OFFSET = 60; + internal const int REQUEST_ID_OFFSET = 168; + internal const int REQUEST_OPERATION_OFFSET = 176; + internal const int REQUEST_RESERVED_OFFSET = 196; + internal const int REPLY_REQUEST_ID_OFFSET = 200; + internal const int REPLY_OPERATION_OFFSET = 208; + internal const int REPLY_STATUS_OFFSET = 216; + + internal const byte COMMAND_REPLY = 8; + internal const byte COMMAND_EVICTION = 13; + internal const int EVICTION_REASON_OFFSET = 255; + internal const byte EVICTION_STALE_CLIENT = 13; + internal const byte OPERATION_REGISTER = 1; + internal const byte OPERATION_NON_REPLICATED = 2; + internal const int GET_CLUSTER_METADATA_CODE = 12; + internal const int PING_CODE = 1; + internal const uint TRANSIENT_NOT_ACCEPTED = 58; + + /// A reply for anything the roster read does not claim: a register, or an empty read. + internal static byte[] Answer(MockRequest request) + { + return request.Operation == OPERATION_REGISTER + ? Reply(OPERATION_REGISTER, RegisterBody(session: 128)) + : Reply(OPERATION_NON_REPLICATED, []); + } + + internal static byte[] Reply(byte operation, byte[] body) + { + return Reply(operation, body, 0); + } + + internal static byte[] Reply(byte operation, byte[] body, uint status) + { + var frame = new byte[HEADER_SIZE + body.Length]; + BinaryPrimitives.WriteUInt32LittleEndian(frame.AsSpan(SIZE_OFFSET, 4), (uint)frame.Length); + frame[COMMAND_OFFSET] = COMMAND_REPLY; + frame[REPLY_OPERATION_OFFSET] = operation; + BinaryPrimitives.WriteUInt32LittleEndian(frame.AsSpan(REPLY_STATUS_OFFSET, 4), status); + body.CopyTo(frame.AsSpan(HEADER_SIZE)); + + return frame; + } + + /// + /// A register reply carries a committed result section, so its four leading zero bytes announce zero + /// entries and the typed payload starts right after them. A non-replicated read carries none. + /// + internal static byte[] RegisterBody(ulong session) + { + var serverVersion = Encoding.UTF8.GetBytes("0.0.0"); + var body = new byte[4 + 17 + serverVersion.Length]; + var payload = body.AsSpan(4); + BinaryPrimitives.WriteUInt32LittleEndian(payload[..4], 7); + BinaryPrimitives.WriteUInt64LittleEndian(payload[4..12], session); + BinaryPrimitives.WriteUInt32LittleEndian(payload[12..16], 11 << 10); + payload[16] = (byte)serverVersion.Length; + serverVersion.CopyTo(payload[17..]); + + return body; + } +} + +internal readonly record struct MockRequest(byte Operation, int Code, ulong RequestId); + +/// +/// A loopback VSR node. Killing it drops the live sockets and stops accepting, so a redial is refused the +/// way a dead process refuses one. +/// +internal sealed class MockNode : IDisposable +{ + private readonly TcpListener _listener; + private readonly List _accepted = []; + private volatile bool _killed; + private int _connections; + private readonly List _recorded = []; + private int _pings; + private int _registrations; + + /// + /// A port to bind, for a node that has to come up on an address the client already knows. Zero + /// takes whatever the OS hands out. + /// + public MockNode(ushort port = 0) + { + _listener = new TcpListener(IPAddress.Loopback, port); + _listener.Start(); + Port = (ushort)((IPEndPoint)_listener.LocalEndpoint).Port; + } + + public ushort Port { get; } + + public int Pings => Volatile.Read(ref _pings); + + public int Registrations => Volatile.Read(ref _registrations); + + /// How many requests with the given command code the node has answered so far. + public int Requests(int code) + { + lock (_recorded) + { + return _recorded.Count(request => request.Code == code); + } + } + + public int Connections + { + get + { + lock (_accepted) + { + return _connections; + } + } + } + + public void Serve(Func handler) + { + _ = Task.Run(async () => + { + while (!_killed) + { + TcpClient connection; + try + { + connection = await _listener.AcceptTcpClientAsync(); + } + catch (Exception) + { + return; + } + + lock (_accepted) + { + _accepted.Add(connection); + _connections++; + } + + _ = Task.Run(() => Exchange(connection, handler)); + } + }); + } + + public void Kill() + { + _killed = true; + lock (_accepted) + { + foreach (var connection in _accepted) + { + connection.Close(); + } + + _accepted.Clear(); + } + + _listener.Stop(); + } + + public void Dispose() + { + Kill(); + } + + private async Task Exchange(TcpClient connection, Func handler) + { + try + { + await using var stream = connection.GetStream(); + var header = new byte[HEADER_SIZE]; + while (!_killed) + { + await ReadExactly(stream, header); + var size = BinaryPrimitives.ReadUInt32LittleEndian(header.AsSpan(SIZE_OFFSET, 4)); + var body = new byte[size - HEADER_SIZE]; + await ReadExactly(stream, body); + + var request = new MockRequest(header[REQUEST_OPERATION_OFFSET], + BinaryPrimitives.ReadInt32LittleEndian(header.AsSpan(REQUEST_RESERVED_OFFSET, 4)), + BinaryPrimitives.ReadUInt64LittleEndian(header.AsSpan(REQUEST_ID_OFFSET, 8))); + lock (_recorded) + { + _recorded.Add(request); + } + + if (request.Operation == OPERATION_REGISTER) + { + Interlocked.Increment(ref _registrations); + } + else if (request.Code == PING_CODE) + { + Interlocked.Increment(ref _pings); + } + + var reply = handler(request); + BinaryPrimitives.WriteUInt64LittleEndian(reply.AsSpan(REPLY_REQUEST_ID_OFFSET, 8), + request.RequestId); + await stream.WriteAsync(reply); + await stream.FlushAsync(); + } + } + catch (Exception) + { + // A killed node and a client that went away look the same here. + } + } + + private static async Task ReadExactly(NetworkStream stream, byte[] buffer) + { + var read = 0; + while (read < buffer.Length) + { + var chunk = await stream.ReadAsync(buffer.AsMemory(read)); + if (chunk == 0) + { + throw new EndOfStreamException("Connection closed"); + } + + read += chunk; + } + } +} From bad22c03c64ae0b1b843a0c3f8375f26c6d38228 Mon Sep 17 00:00:00 2001 From: Grzegorz Koszyk <112548209+numinnex@users.noreply.github.com> Date: Thu, 3 Sep 2026 10:35:58 +0200 Subject: [PATCH 049/182] feat(server): dedup partition writes with per-group client table slices (#3959) Implement deduplication for partition plane operations. --------- Co-authored-by: Piotr Gankiewicz Co-authored-by: Krishna Vishal Co-authored-by: Hubert Gruszecki --- core/configs/src/server_config/defaults.rs | 1 + core/configs/src/server_config/partition.rs | 52 ++ core/consensus/src/client_table.rs | 710 ++++++++++++++++- core/consensus/src/impls.rs | 65 +- core/consensus/src/lib.rs | 20 +- core/integration/tests/cluster/mod.rs | 1 + .../tests/cluster/partition_dedup.rs | 749 ++++++++++++++++++ core/partitions/src/iggy_partition.rs | 281 ++++++- core/partitions/src/iggy_partitions.rs | 63 +- core/partitions/src/state_transfer.rs | 300 ++++++- core/server/config.toml | 11 + core/server/src/boot/recovery.rs | 3 + core/server/src/dispatch/partition.rs | 121 ++- core/server/src/http/session.rs | 33 +- core/server/src/http/state.rs | 2 +- core/server/src/http/submit.rs | 15 +- core/server/src/partition_helpers.rs | 2 + core/shard/src/lib.rs | 351 +++++++- core/shard/src/metrics.rs | 10 +- core/shard/src/router.rs | 16 +- core/simulator/src/lib.rs | 264 +++--- scripts/ci/storage-compat.sh | 47 +- 22 files changed, 2824 insertions(+), 293 deletions(-) create mode 100644 core/integration/tests/cluster/partition_dedup.rs diff --git a/core/configs/src/server_config/defaults.rs b/core/configs/src/server_config/defaults.rs index 1569cfadad..743f25d91a 100644 --- a/core/configs/src/server_config/defaults.rs +++ b/core/configs/src/server_config/defaults.rs @@ -176,6 +176,7 @@ impl Default for PartitionConfig { let partition = &SERVER_CONFIG.partition; PartitionConfig { prepare_queue_depth: partition.prepare_queue_depth as usize, + dedup_clients_max: partition.dedup_clients_max as usize, evicted_ring_capacity: partition.evicted_ring_capacity as usize, evicted_ring_bytes_max: partition.evicted_ring_bytes_max.parse().unwrap(), transfer_served_cache_bytes_max: partition diff --git a/core/configs/src/server_config/partition.rs b/core/configs/src/server_config/partition.rs index d6e3273611..6d32c22320 100644 --- a/core/configs/src/server_config/partition.rs +++ b/core/configs/src/server_config/partition.rs @@ -113,6 +113,15 @@ pub const DEFAULT_EVICTED_RING_BYTES_MAX: u64 = 16 * 1024 * 1024; /// trips first evicts; this byte ceiling is the second typo guard. pub const MAX_EVICTED_RING_BYTES: u64 = 256 * 1024 * 1024; +/// Shipped default for [`PartitionConfig::dedup_clients_max`]; pinned against +/// the runtime constant by a bootstrap assert. +pub const PARTITION_DEDUP_CLIENTS_DEFAULT: usize = 4096; + +/// Ceiling for [`PartitionConfig::dedup_clients_max`]. A per-group budget, so +/// the ceiling bounds worst-case memory at roughly `partitions * this * 146 +/// bytes`: a 112-byte slot entry plus its index-map slot. +pub const PARTITION_DEDUP_CLIENTS_CEILING: usize = 1 << 16; + /// Capacity tunables for the per-partition consensus plane. #[derive(Debug, Deserialize, Serialize, Clone, ConfigEnv)] pub struct PartitionConfig { @@ -123,6 +132,18 @@ pub struct PartitionConfig { /// pinned request-buffer memory by the partition count. pub prepare_queue_depth: usize, + /// Distinct clients each partition group tracks request watermarks for, + /// deduplicating retried produces and consumer-offset writes. At capacity + /// the entry whose newest commit is oldest is evicted, which costs dedup + /// coverage for that client (its next replay re-executes, exactly as it + /// would have before dedup existed) and never correctness. Must be > 0 and + /// <= [`PARTITION_DEDUP_CLIENTS_CEILING`]. + /// + /// Unlike `[metadata] clients_table_max`, this budget is PER GROUP, so the + /// worst case scales with partition count: size it to the producers a + /// single partition actually sees, not the node's client total. + pub dedup_clients_max: usize, + /// Entries the evicted ring retains per multi-replica partition for /// journal repair after a peer rejoins. Larger widens the window a /// restarting peer can be served from the ring before falling back to @@ -177,6 +198,14 @@ impl Validatable for PartitionConfig { ); return Err(ConfigurationError::InvalidConfigurationValue); } + if self.dedup_clients_max == 0 || self.dedup_clients_max > PARTITION_DEDUP_CLIENTS_CEILING { + eprintln!( + "{COMPONENT} partition.dedup_clients_max ({}) must be > 0 and <= \ + {PARTITION_DEDUP_CLIENTS_CEILING}", + self.dedup_clients_max + ); + return Err(ConfigurationError::InvalidConfigurationValue); + } if self.evicted_ring_capacity == 0 { eprintln!("{COMPONENT} partition.evicted_ring_capacity must be > 0"); return Err(ConfigurationError::InvalidConfigurationValue); @@ -253,6 +282,29 @@ mod tests { ); } + #[test] + fn shipped_dedup_default_matches_the_runtime_constant() { + assert_eq!( + PartitionConfig::default().dedup_clients_max, + PARTITION_DEDUP_CLIENTS_DEFAULT, + "config.toml dedup_clients_max drifted from the runtime default" + ); + } + + #[test] + fn rejects_out_of_range_dedup_clients_max() { + for value in [0, PARTITION_DEDUP_CLIENTS_CEILING + 1] { + let config = PartitionConfig { + dedup_clients_max: value, + ..PartitionConfig::default() + }; + assert!( + config.validate().is_err(), + "dedup_clients_max {value} must be rejected" + ); + } + } + #[test] fn rejects_zero_prepare_queue_depth() { let config = PartitionConfig { diff --git a/core/consensus/src/client_table.rs b/core/consensus/src/client_table.rs index 0737e23bf3..e251219bc6 100644 --- a/core/consensus/src/client_table.rs +++ b/core/consensus/src/client_table.rs @@ -23,6 +23,7 @@ use server_common::{ MESSAGE_ALIGN, Message, iobuf::{Frozen, Owned}, }; +use std::cmp::Reverse; use std::collections::{HashMap, VecDeque}; use std::fmt; use std::mem::size_of; @@ -247,6 +248,12 @@ struct ClientEntry { /// first app op commits. Survives re-register: a resumed session keeps /// its dedup history. watermark: u64, + /// Partition-slice only: bit `i` set means request `watermark - i` has + /// committed, bit 0 being the watermark itself. A request below the + /// watermark with its bit clear is a reordered arrival still to execute, + /// not a duplicate; below the window everything reads as committed. Zero + /// on the metadata plane, whose reply ring plays this role. + committed_window: u128, /// `request_checksum` of the watermark request; catches a client reusing /// a request id for a different operation. Zero when unstamped (integrity /// fields are zeroed on the wire today), which disables the comparison. @@ -508,6 +515,89 @@ pub enum CommitReply { AdvancedFence, } +/// Which of the table's mechanisms an instance runs. +/// +/// The metadata plane needs all of them. A partition group's slice needs only +/// the watermark: it has no register to mint an epoch from, no result section +/// worth caching, and one table per group rather than per node, so the +/// preallocated slot array would reserve ~384 KiB per partition before a single +/// client connects. The predicates below are the only two combinations that +/// exist, so the mode is an enum rather than three independent flags. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum ClientTableMode { + /// Metadata plane: replies cached, epoch fenced, slots preallocated. + Metadata, + /// One partition consensus group's slice: watermark only. + PartitionSlice, +} + +impl ClientTableMode { + /// Keep committed replies so a duplicate replays the original bytes. Off: + /// duplicates answer [`RequestStatus::AlreadyApplied`] and the caller + /// synthesizes the reply. + #[must_use] + pub const fn cache_replies(self) -> bool { + matches!(self, Self::Metadata) + } + + /// Enforce the register-minted epoch fence. Off: entries carry no epoch, + /// `check_request` ignores the presented one, and a committed request may + /// create its own entry (there is no register to do it). + #[must_use] + pub const fn fence_epoch(self) -> bool { + matches!(self, Self::Metadata) + } + + /// Allocate every slot up front. Off: slots grow to the cap on demand. + /// Slot assignment is identical either way -- both hand out the lowest free + /// index -- so eviction order and the wire encoding are unchanged. + #[must_use] + pub const fn preallocate_slots(self) -> bool { + matches!(self, Self::Metadata) + } +} + +/// One partition slice entry in its wire and install form. +/// +/// Named fields rather than a tuple because `watermark` and `latest_commit` +/// are both `u64` and a positional swap would decode cleanly into the wrong +/// dedup decision. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct DedupWatermark { + pub client: u128, + /// Acting user the watermark belongs to. A different user committing under + /// the same client id resets the entry rather than inheriting it: the id is + /// client-supplied (or, for HTTP, re-minted after a logout), so it alone is + /// not an identity. + pub user_id: u32, + /// Highest committed request number. + pub watermark: u64, + /// Commit op of the newest request folded in; the eviction rank. + pub latest_commit: u64, + /// Bit `i` set: request `watermark - i` committed. See + /// [`COMMITTED_WINDOW_BITS`]. + pub committed_window: u128, +} + +/// Width of the per-entry committed-request window below the watermark. +/// +/// A client that pipelines writes can see one of them refused transiently and +/// replay it after later ids have committed, so "at or below the watermark" +/// alone would absorb that replay as a duplicate and lose the write. The window +/// records which ids under the watermark actually committed; an unmarked one +/// inside it executes, while one that has aged out below it reads as committed +/// and is absorbed with the operation's empty success. +/// +/// The width is in the CLIENT's request-id space, not in this group's writes: +/// `ConsensusSession` mints from one counter across every partition, stream and +/// metadata op, so a slice only ever sees the subset of those ids routed to it. +/// Coverage in a client's own writes to one group is this width divided by the +/// number of groups it interleaves, so 128 ids is around 16 writes per group +/// across 8 partitions, and a replay held back longer than that is absorbed and +/// lost. A wider bitmap divides by the same fanout: closing the gap needs +/// per-group request numbering, which waits on the clients-table follow-up. +pub const COMMITTED_WINDOW_BITS: u64 = 128; + /// VSR client table: per-session fence epoch + request-watermark dedup. /// /// Fixed-size slot array (source of truth) + `HashMap` index (O(1) lookup). @@ -530,11 +620,11 @@ pub enum CommitReply { /// /// ## Plane /// -/// Metadata-plane today. The design spans planes (one logical table, -/// group-resident slices); partition-plane integration arrives once -/// partition prepares carry real `(session_id, request)` instead of the -/// transport id (data-plane request numbering, IGGY-137). Until then the -/// partition plane stays at-least-once with no dedup. +/// This table is the metadata plane's. The partition plane runs the same +/// watermark rule in its own per-group slices ([`ClientTableMode::PartitionSlice`], +/// held by `partitions::IggyPartition::dedup`), which keep no reply ring and no +/// epoch: partition prepares carry the VSR client id and request number but no +/// session, so fencing a stale session waits on identity surviving reconnects. /// /// ## Tracking /// @@ -560,19 +650,32 @@ pub enum CommitReply { #[derive(Debug)] pub struct ClientTable { /// `None` = free slot. Deterministic iteration for eviction + serialization. + /// + /// Under [`ClientTableMode::preallocate_slots`] this is sized to + /// `clients_max` at construction; otherwise it grows to that cap on demand. + /// Every `Some` has exactly one `index` entry, so `index.len()` is the + /// occupied count. slots: Vec>, /// `client_id` -> slot index. Rebuilt on decode. index: HashMap, + /// Slot ceiling. Tracked explicitly because `slots.len()` is the allocated + /// length, which only equals the cap when slots are preallocated. + clients_max: usize, + mode: ClientTableMode, /// Fences of clients capacity eviction reclaimed, oldest at the front. /// /// Bounded by the slot count. A fence is the entry's header fields plus, at /// most, the watermark request's own reply, so it costs a fraction of the /// entry it replaces. Trimmed oldest-first. /// - /// Replica-local best-effort, NOT replicated state: the bound is - /// `slots.len()`, which `from_snapshot` and `decode` size per node, and a - /// state transfer replaces the table wholesale. Losing a fence degrades a - /// resume to the pre-fence behaviour; it never makes one more permissive. + /// Replica-local best-effort, NOT replicated state: the bound is the slot + /// ceiling, which `from_snapshot` and `decode` size per node, and a state + /// transfer replaces the table wholesale. Losing a fence degrades a resume + /// to the pre-fence behaviour; it never makes one more permissive. + /// + /// Only a plane that mints epochs fills this: a fence exists so a later + /// register revives the evicted session's watermark, and a plane with no + /// register has nothing to revive it with. evicted_fences: VecDeque, } @@ -590,11 +693,24 @@ impl ClientTable { /// `max_clients` caps slots; index pre-sized to avoid rehash storms. #[must_use] pub fn new(max_clients: usize) -> Self { - let mut slots = Vec::with_capacity(max_clients); - slots.resize_with(max_clients, || None); + Self::with_mode(max_clients, ClientTableMode::Metadata) + } + + /// `max_clients` caps slots; `mode` selects which mechanisms run. + #[must_use] + pub fn with_mode(max_clients: usize, mode: ClientTableMode) -> Self { + let (slots, index) = if mode.preallocate_slots() { + let mut slots = Vec::with_capacity(max_clients); + slots.resize_with(max_clients, || None); + (slots, HashMap::with_capacity(max_clients)) + } else { + (Vec::new(), HashMap::new()) + }; Self { slots, - index: HashMap::with_capacity(max_clients), + index, + clients_max: max_clients, + mode, evicted_fences: VecDeque::new(), } } @@ -611,7 +727,7 @@ impl ClientTable { self.index.is_empty(), "set_capacity must run before any client registers" ); - *self = Self::new(max_clients); + *self = Self::with_mode(max_clients, self.mode); } /// Snapshot the table for the metadata checkpoint: every occupied slot with its @@ -738,14 +854,18 @@ impl ClientTable { user_id: entry.user_id, watermark: entry.watermark, watermark_checksum: entry.watermark_checksum, + committed_window: 0, ring, client_id: entry.client_id, latest_commit, }); } + let clients_max = slots.len(); let mut table = Self { slots, index, + clients_max, + mode: ClientTableMode::Metadata, evicted_fences: VecDeque::with_capacity(snapshot.fences.len()), }; for (position, fence) in snapshot.fences.into_iter().enumerate() { @@ -793,7 +913,10 @@ impl ClientTable { ) -> RequestStatus { assert!(client_id != 0, "client_id 0 is reserved for internal use"); // Header validation guarantees both > 0 at wire layer. - debug_assert!(epoch > 0, "check_request: epoch must be > 0"); + debug_assert!( + epoch > 0 || !self.mode.fence_epoch(), + "check_request: epoch must be > 0 when fencing" + ); debug_assert!(request > 0, "check_request: request must be > 0"); // Epoch check before request: a fenced zombie must be rejected even @@ -803,17 +926,21 @@ impl ClientTable { }; let entry = self.slots[slot_idx].as_ref().expect("index/slot mismatch"); - if epoch < entry.epoch { - return RequestStatus::Fenced { - current: entry.epoch, - received: epoch, - }; - } - if epoch > entry.epoch { - return RequestStatus::EpochAhead { - current: entry.epoch, - received: epoch, - }; + // A plane with no register mints no epoch, so there is nothing to + // fence against and the presented value is ignored. + if self.mode.fence_epoch() { + if epoch < entry.epoch { + return RequestStatus::Fenced { + current: entry.epoch, + received: epoch, + }; + } + if epoch > entry.epoch { + return RequestStatus::EpochAhead { + current: entry.epoch, + received: epoch, + }; + } } if request > entry.watermark { @@ -917,7 +1044,7 @@ impl ClientTable { // must never hand one user another user's dedup history, nor its // cached reply bytes, merely because the key was reused. let fence = self.take_fence(client_id, user_id); - let freed = if self.index.len() >= self.slots.len() { + let freed = if self.index.len() >= self.clients_max { self.evict_oldest() } else { None @@ -946,6 +1073,7 @@ impl ClientTable { .as_ref() .map_or(REGISTER_REQUEST_ID, |fence| fence.watermark), watermark_checksum: fence.as_ref().map_or(0, |fence| fence.watermark_checksum), + committed_window: 0, ring, }); self.index.insert(client_id, slot_idx); @@ -1059,6 +1187,176 @@ impl ClientTable { CommitReply::Cached } + /// Watermark-plus-window dedup check for a plane that mints no epoch. + /// + /// `true` means `user_id` already committed this request under `client_id`, + /// so it must be answered rather than executed again: it is the watermark, + /// a marked id inside the [`COMMITTED_WINDOW_BITS`] window below it, or + /// anything older than the window. An unmarked id inside the window is a + /// reordered arrival (a transiently refused write replayed after its + /// successors committed) and reads as new. An entry another user left under + /// the same id is not evidence about this caller: the id alone is not an + /// identity (see [`DedupWatermark::user_id`]), so the request reads as new + /// and its commit resets the entry. + /// + /// # Panics + /// If called on a table that fences epochs -- that plane must go through + /// [`Self::check_request`], which enforces the fence. + #[must_use] + pub fn is_duplicate(&self, client_id: u128, user_id: u32, request: u64) -> bool { + debug_assert!( + !self.mode.fence_epoch(), + "is_duplicate: an epoch-fencing table must use check_request" + ); + let Some(&slot_idx) = self.index.get(&client_id) else { + return false; + }; + let entry = self.slots[slot_idx].as_ref().expect("index/slot mismatch"); + entry.user_id == user_id && request <= entry.watermark && entry.window_has(request) + } + + /// Record a committed request without a reply to cache. + /// + /// The entry point for a plane that runs + /// [`ClientTableMode::PartitionSlice`]: there is no register to create the + /// entry, so the first committed request creates it, and there is no result + /// section worth retaining, so a later duplicate answers + /// [`RequestStatus::AlreadyApplied`] and the caller synthesizes the reply. + /// + /// Idempotent and order-insensitive for one user: the watermark only rises + /// and the window only gains bits, so replaying an already-folded op is a + /// no-op and a state-transfer install followed by a re-walk of the same + /// commits converges. A commit above the watermark shifts the window up by + /// the gap (ids that age out read as committed from then on); one below it + /// sets its bit. A commit by a DIFFERENT user under the same client id + /// replaces the entry outright: nothing observes a logout here, so this is + /// what stops the next holder of a re-minted id from having its first + /// writes absorbed by the previous holder's watermark. + /// + /// # Panics + /// If called on a table whose mode caches replies -- that plane must go + /// through [`Self::commit_reply`] so the ring stays populated. + pub fn commit_request(&mut self, client_id: u128, user_id: u32, request: u64, commit_op: u64) { + debug_assert!( + !self.mode.cache_replies(), + "commit_request: a reply-caching table must use commit_reply" + ); + // Zero is the reserved client id, refused at every ingress (wire + // validation, the HTTP minter, the auto-commit guard at the call + // sites). Kept as a return rather than an assert so that an artifact + // slipping past the decoder degrades to no dedup for that entry instead + // of taking the replica down. + if client_id == 0 { + return; + } + + if let Some(&slot_idx) = self.index.get(&client_id) { + let entry = self.slots[slot_idx].as_mut().expect("index/slot mismatch"); + if entry.user_id != user_id { + entry.user_id = user_id; + entry.watermark = request; + entry.committed_window = 1; + entry.latest_commit = commit_op; + } else if request > entry.watermark { + let gap = request - entry.watermark; + entry.committed_window = if gap >= COMMITTED_WINDOW_BITS { + 1 + } else { + (entry.committed_window << gap) | 1 + }; + entry.watermark = request; + entry.latest_commit = commit_op; + } else { + let below = entry.watermark - request; + if below < COMMITTED_WINDOW_BITS && entry.committed_window & (1 << below) == 0 { + entry.committed_window |= 1 << below; + // Commits walk in op order, so a newly folded reordered id + // is the newest commit unless an install re-walk replays an + // older one. + entry.latest_commit = entry.latest_commit.max(commit_op); + } + } + return; + } + + let freed = if self.index.len() >= self.clients_max { + self.evict_oldest() + } else { + None + }; + let Some(slot_idx) = freed.or_else(|| self.first_free_slot()) else { + // Only reachable at a zero cap, which config validation rejects. + return; + }; + self.index.insert(client_id, slot_idx); + self.slots[slot_idx] = Some(ClientEntry::watermark_only( + client_id, user_id, request, 1, commit_op, + )); + } + + /// Replace every entry, as a state-transfer install does. An empty iterator + /// is the clear: there is no separate `clear`, and the one caller that + /// needs one (a failed install converging to empty) comes through here. + /// + /// The peer's cap may exceed this node's, so when the input is longer than + /// `clients_max` the entries with the newest commits survive, which is what + /// the oldest-commit eviction would have converged on had the surplus been + /// folded in one by one. Zero client ids are dropped, as + /// [`Self::commit_request`] drops them. + /// + /// # Panics + /// If called on a table whose mode caches replies (those install through + /// the snapshot / wire codecs, which carry the rings). + pub fn install_watermarks(&mut self, entries: impl IntoIterator) { + debug_assert!( + !self.mode.cache_replies(), + "install_watermarks: a reply-caching table installs via decode" + ); + self.slots.clear(); + self.index.clear(); + let mut entries: Vec = entries + .into_iter() + .filter(|entry| entry.client != 0) + .collect(); + entries.sort_unstable_by_key(|entry| Reverse(entry.latest_commit)); + entries.truncate(self.clients_max); + for entry in entries { + // The wire form is strictly ascending by client, so this only + // guards a caller-built iterator; the first (newest) copy wins. + if self.index.contains_key(&entry.client) { + continue; + } + self.index.insert(entry.client, self.slots.len()); + self.slots.push(Some(ClientEntry::watermark_only( + entry.client, + entry.user_id, + entry.watermark, + entry.committed_window, + entry.latest_commit, + ))); + } + } + + /// Every entry ascending by client: the deterministic form a wire encoding + /// needs. + #[must_use] + pub fn watermarks_sorted(&self) -> Vec { + let mut entries: Vec = self + .slots + .iter() + .flatten() + .map(|entry| DedupWatermark { + client: entry.client_id, + user_id: entry.user_id, + watermark: entry.watermark, + latest_commit: entry.latest_commit, + committed_window: entry.committed_window, + }) + .collect(); + entries.sort_unstable_by_key(|entry| entry.client); + entries + } + /// Remove a client session and cached replies. /// /// **LOCAL ONLY -- does NOT replicate.** Two correct call sites: @@ -1152,7 +1450,7 @@ impl ClientTable { /// latency shows up in the logs rather than as a silent re-execution. #[must_use] pub const fn fence_retention(&self) -> usize { - self.slots.len() + self.clients_max } /// Evict the client whose latest cached reply has the oldest commit. @@ -1212,6 +1510,12 @@ impl ClientTable { /// Record an evicted entry's dedup fence, trimming oldest-first. fn remember_fence(&mut self, entry: &ClientEntry) { + // Only a register revives a fence, and a plane that mints no epoch has + // none, so storing one there would cost a slot's worth of memory per + // group for something nothing can read back. + if !self.mode.fence_epoch() { + return; + } // Nothing committed under this session, so there is nothing to dedup. // Worth skipping rather than storing: `evict_oldest` ranks on the oldest // `latest_commit`, and a session idle since its register carries its own @@ -1310,8 +1614,25 @@ impl ClientTable { self.evicted_fences.remove(position) } - fn first_free_slot(&self) -> Option { - self.slots.iter().position(Option::is_none) + /// Lowest free slot, growing the array when slots are allocated lazily. + /// Assignment is identical to the preallocated case: both hand out the + /// lowest free index, so eviction order and the wire encoding do not + /// depend on the mode. + /// + /// The hole scan runs only when a hole exists (`index.len()` is the + /// occupied count): a lazily grown table below its cap has none, and + /// scanning it before every push would make the fill quadratic on the + /// commit path. + fn first_free_slot(&mut self) -> Option { + if self.index.len() < self.slots.len() + && let Some(index) = self.slots.iter().position(Option::is_none) + { + return Some(index); + } + (self.slots.len() < self.clients_max).then(|| { + self.slots.push(None); + self.slots.len() - 1 + }) } /// Latest cached reply for a client. @@ -1649,6 +1970,7 @@ impl ClientTable { user_id, watermark, watermark_checksum, + committed_window: 0, ring, client_id, latest_commit, @@ -1724,11 +2046,42 @@ impl ClientTable { /// can exceed `[metadata] clients_table_max`. #[must_use] pub const fn capacity(&self) -> usize { - self.slots.len() + self.clients_max } } impl ClientEntry { + /// Entry for a plane that mints no epoch and caches no reply: the fields a + /// [`ClientTableMode::PartitionSlice`] table never reads stay at their + /// zero values. Bit 0 of the window is forced on: the watermark itself is + /// committed by definition. + const fn watermark_only( + client_id: u128, + user_id: u32, + watermark: u64, + committed_window: u128, + commit_op: u64, + ) -> Self { + Self { + epoch: 0, + user_id, + watermark, + watermark_checksum: 0, + committed_window: committed_window | 1, + ring: VecDeque::new(), + client_id, + latest_commit: commit_op, + } + } + + /// Whether `request` (at or below the watermark) is inside the window and + /// marked committed, or below the window entirely. Callers check + /// `request <= watermark` first. + const fn window_has(&self, request: u64) -> bool { + let below = self.watermark - request; + below >= COMMITTED_WINDOW_BITS || self.committed_window & (1 << below) != 0 + } + /// Latest committed reply (register or app op). /// /// # Panics @@ -3002,6 +3355,303 @@ mod tests { // Capacity resize (boot-only) + // --- ClientTableMode::PartitionSlice: watermark-only dedup --- + // + // One consensus group's slice. No register mints entries here, no reply is + // cached, and slots grow on demand, so these pin the behaviour the + // partition plane actually relies on. + + const SLICE_USER: u32 = 3; + const OTHER_USER: u32 = 4; + + fn slice(clients_max: usize) -> ClientTable { + ClientTable::with_mode(clients_max, ClientTableMode::PartitionSlice) + } + + fn watermark(client: u128, watermark: u64, latest_commit: u64) -> DedupWatermark { + DedupWatermark { + client, + user_id: SLICE_USER, + watermark, + latest_commit, + committed_window: 1, + } + } + + fn clients_of(table: &ClientTable) -> Vec { + table + .watermarks_sorted() + .into_iter() + .map(|entry| entry.client) + .collect() + } + + #[test] + fn given_partition_slice_when_empty_should_admit_and_not_preallocate() { + // The reason this plane cannot use the metadata mode: one table per + // group, so preallocating the cap would reserve hundreds of KiB per + // partition before a single client connects. + let table = slice(4096); + assert_eq!(table.count(), 0); + assert_eq!(table.slots.len(), 0, "slots must grow on demand"); + assert!(!table.is_duplicate(7, SLICE_USER, 1)); + } + + #[test] + fn given_partition_slice_when_request_replayed_should_report_duplicate() { + // Only what committed is a duplicate: an id below the watermark that + // never committed is a reordered arrival and still executes. + let mut table = slice(4); + table.commit_request(7, SLICE_USER, 5, 100); + + assert!(table.is_duplicate(7, SLICE_USER, 5)); + assert!(!table.is_duplicate(7, SLICE_USER, 4)); + assert!(!table.is_duplicate(7, SLICE_USER, 6)); + } + + #[test] + fn given_partition_slice_when_request_id_gaps_should_accept_the_jump() { + // One client counter feeds several groups, so a slice legitimately sees + // only a subset of the ids that client mints; the skipped ids stay + // admissible in case they were routed here late rather than elsewhere. + let mut table = slice(4); + table.commit_request(7, SLICE_USER, 5, 100); + table.commit_request(7, SLICE_USER, 9, 101); + + assert!(table.is_duplicate(7, SLICE_USER, 5)); + assert!(table.is_duplicate(7, SLICE_USER, 9)); + assert!(!table.is_duplicate(7, SLICE_USER, 7)); + assert!(!table.is_duplicate(7, SLICE_USER, 10)); + } + + #[test] + fn given_partition_slice_when_commit_replayed_should_be_idempotent() { + let mut table = slice(4); + table.commit_request(7, SLICE_USER, 5, 100); + table.commit_request(7, SLICE_USER, 5, 100); + table.commit_request(7, SLICE_USER, 5, 100); + + assert_eq!(table.watermarks_sorted(), vec![watermark(7, 5, 100)]); + } + + #[test] + fn given_partition_slice_when_lower_id_commits_late_should_admit_then_absorb() { + // A pipelining client had request 2 refused transiently and replays it + // after 3 committed: the replay is a new write, not a duplicate, and + // only once it commits does it read as one. + let mut table = slice(4); + table.commit_request(7, SLICE_USER, 1, 100); + table.commit_request(7, SLICE_USER, 3, 101); + + assert!(!table.is_duplicate(7, SLICE_USER, 2)); + table.commit_request(7, SLICE_USER, 2, 102); + + assert!(table.is_duplicate(7, SLICE_USER, 2)); + assert!(table.is_duplicate(7, SLICE_USER, 1)); + assert!(table.is_duplicate(7, SLICE_USER, 3)); + assert!(!table.is_duplicate(7, SLICE_USER, 4)); + assert_eq!( + table.watermarks_sorted(), + vec![DedupWatermark { + client: 7, + user_id: SLICE_USER, + watermark: 3, + latest_commit: 102, + committed_window: 0b111, + }] + ); + } + + #[test] + fn given_partition_slice_when_id_ages_out_of_window_should_read_as_committed() { + // Below the window nothing is tracked, so the pre-window rule applies: + // absorbed. Inside it, an unmarked id stays admissible however the + // watermark moved. + let mut table = slice(4); + table.commit_request(7, SLICE_USER, 1, 100); + table.commit_request(7, SLICE_USER, 1 + COMMITTED_WINDOW_BITS + 10, 101); + + assert!(table.is_duplicate(7, SLICE_USER, 1)); + assert!(!table.is_duplicate(7, SLICE_USER, 1 + COMMITTED_WINDOW_BITS)); + assert!(!table.is_duplicate(7, SLICE_USER, 12)); + assert!(table.is_duplicate(7, SLICE_USER, 11)); + } + + #[test] + fn given_partition_slice_when_watermark_jumps_should_shift_the_window() { + let mut table = slice(4); + table.commit_request(7, SLICE_USER, 1, 100); + table.commit_request(7, SLICE_USER, 2, 101); + table.commit_request(7, SLICE_USER, 5, 102); + + // 5 (bit 0), 2 (bit 3), 1 (bit 4) committed; 3 and 4 did not. + assert_eq!(table.watermarks_sorted()[0].committed_window, 0b11001); + assert!(!table.is_duplicate(7, SLICE_USER, 3)); + assert!(!table.is_duplicate(7, SLICE_USER, 4)); + assert!(table.is_duplicate(7, SLICE_USER, 2)); + } + + #[test] + fn given_partition_slice_when_other_user_commits_under_same_id_should_reset() { + // The id is client-supplied (or re-minted after an HTTP logout), so the + // previous holder's watermark must not absorb the next holder's writes. + let mut table = slice(4); + table.commit_request(7, SLICE_USER, u64::MAX, 100); + + assert!(!table.is_duplicate(7, OTHER_USER, 1)); + table.commit_request(7, OTHER_USER, 1, 101); + + assert!(table.is_duplicate(7, OTHER_USER, 1)); + assert!(!table.is_duplicate(7, OTHER_USER, 2)); + assert!( + !table.is_duplicate(7, SLICE_USER, 5), + "the previous holder's history is gone with the reset" + ); + assert_eq!(table.count(), 1, "a reset reuses the slot"); + } + + #[test] + fn given_partition_slice_when_full_should_evict_the_oldest_commit() { + let mut table = slice(2); + table.commit_request(1, SLICE_USER, 1, 10); + table.commit_request(2, SLICE_USER, 1, 20); + table.commit_request(3, SLICE_USER, 1, 30); + + assert_eq!(table.count(), 2); + assert_eq!( + clients_of(&table), + vec![2, 3], + "oldest commit is the victim" + ); + } + + #[test] + fn given_partition_slice_when_entry_evicted_should_admit_its_replay_again() { + // Losing an entry costs dedup coverage, never correctness: the replay + // re-executes exactly as it would have before the slice existed. + let mut table = slice(1); + table.commit_request(1, SLICE_USER, 5, 10); + table.commit_request(2, SLICE_USER, 1, 20); + + assert!(!table.is_duplicate(1, SLICE_USER, 5)); + } + + #[test] + fn given_partition_slice_when_entry_touched_should_spare_it_from_eviction() { + let mut table = slice(2); + table.commit_request(1, SLICE_USER, 1, 10); + table.commit_request(2, SLICE_USER, 1, 20); + // Client 1 commits again, so client 2 now holds the oldest commit. + table.commit_request(1, SLICE_USER, 2, 30); + table.commit_request(3, SLICE_USER, 1, 40); + + assert_eq!(clients_of(&table), vec![1, 3]); + } + + #[test] + fn given_partition_slice_when_filled_to_cap_should_grow_without_holes() { + // The lazily grown array must hand out every index once and never + // rescan for a hole that cannot exist below the cap. + let mut table = slice(64); + for client in 1..=64u128 { + table.commit_request(client, SLICE_USER, 1, client as u64); + } + + assert_eq!(table.count(), 64); + assert_eq!(table.slots.len(), 64); + assert!(table.slots.iter().all(Option::is_some)); + } + + #[test] + fn given_partition_slice_when_watermarks_installed_should_replace_not_merge() { + let mut table = slice(4); + table.commit_request(9, SLICE_USER, 3, 1); + table.install_watermarks([watermark(1, 4, 50), watermark(2, 7, 60)]); + + assert_eq!(table.count(), 2); + assert!( + !table.is_duplicate(9, SLICE_USER, 3), + "install replaces rather than merges" + ); + assert!(table.is_duplicate(1, SLICE_USER, 4)); + assert!(!table.is_duplicate(2, SLICE_USER, 8)); + } + + #[test] + fn given_partition_slice_when_install_exceeds_cap_should_keep_newest_commits() { + // A peer with a larger cap ships more entries than fit; the survivors + // are the ones eviction would have converged on, not wire order. + let mut table = slice(2); + table.install_watermarks([ + watermark(1, 1, 300), + watermark(2, 1, 100), + watermark(3, 1, 200), + ]); + + assert_eq!(table.count(), 2); + assert_eq!(clients_of(&table), vec![1, 3]); + } + + #[test] + fn given_partition_slice_when_install_carries_user_should_keep_it() { + let mut table = slice(4); + table.install_watermarks([DedupWatermark { + client: 1, + user_id: OTHER_USER, + watermark: 4, + latest_commit: 50, + committed_window: 1, + }]); + + assert!(table.is_duplicate(1, OTHER_USER, 4)); + assert!(!table.is_duplicate(1, SLICE_USER, 4)); + } + + #[test] + fn given_partition_slice_when_exported_should_sort_ascending_by_client() { + let mut table = slice(8); + for (commit_op, client) in [30u128, 10, 20].into_iter().enumerate() { + table.commit_request(client, SLICE_USER, 1, commit_op as u64); + } + + assert_eq!(clients_of(&table), vec![10, 20, 30]); + } + + #[test] + fn given_partition_slice_when_cleared_should_admit_everything() { + let mut table = slice(4); + table.commit_request(7, SLICE_USER, 5, 100); + table.install_watermarks(std::iter::empty()); + + assert_eq!(table.count(), 0); + assert!(!table.is_duplicate(7, SLICE_USER, 5)); + } + + #[test] + fn given_partition_slice_when_client_is_reserved_zero_should_record_nothing() { + // Zero is reserved cluster-wide and refused at every ingress; a commit + // or install that still carries it degrades to no entry, not a panic. + let mut table = slice(4); + table.commit_request(0, SLICE_USER, 5, 100); + table.install_watermarks([watermark(0, 5, 100), watermark(1, 1, 101)]); + + assert_eq!(clients_of(&table), vec![1]); + } + + #[cfg(debug_assertions)] + #[test] + #[should_panic(expected = "an epoch-fencing table must use check_request")] + fn given_metadata_table_when_is_duplicate_called_should_panic() { + let _ = ClientTable::new(4).is_duplicate(7, SLICE_USER, 1); + } + + #[cfg(debug_assertions)] + #[test] + #[should_panic(expected = "a reply-caching table must use commit_reply")] + fn given_metadata_table_when_commit_request_called_should_panic() { + ClientTable::new(4).commit_request(7, SLICE_USER, 1, 1); + } + // Resizing an empty table swaps its slot count in: a smaller cap then // evicts once the new bound is reached. #[test] diff --git a/core/consensus/src/impls.rs b/core/consensus/src/impls.rs index 3f732bb48e..5ad273c35f 100644 --- a/core/consensus/src/impls.rs +++ b/core/consensus/src/impls.rs @@ -176,6 +176,14 @@ pub const PROBE_ATTEMPTS_MAX: u32 = 5; /// When exceeded, the client with the oldest committed request is evicted. pub const CLIENTS_TABLE_MAX: usize = 8192; +/// Default live dedup entries per PARTITION consensus group. +/// +/// Far below [`CLIENTS_TABLE_MAX`] because this budget is per group rather than +/// per node: the worst case scales with partition count, so it is sized to the +/// producers one partition sees. Pinned against the +/// `[partition] dedup_clients_max` default by a bootstrap assert. +pub const PARTITION_DEDUP_CLIENTS_MAX: usize = 4096; + #[derive(Debug)] pub struct PipelineEntry { pub header: PrepareHeader, @@ -286,13 +294,10 @@ pub struct RequestEntry { } impl RequestEntry { + /// Queued request on the network reply path: no in-process subscriber. #[must_use] pub const fn new(message: Message) -> Self { - Self { - message, - received_at: 0, - reply_sender: None, - } + Self::with_sender(message, None) } /// Queued request paired with a fresh receiver that resolves when the @@ -305,12 +310,22 @@ impl RequestEntry { message: Message, ) -> (Self, Receiver>) { let (sender, receiver) = oneshot::channel(); - let entry = Self { + (Self::with_sender(message, Some(sender)), receiver) + } + + /// Queued request carrying a sender the caller already owns, for a submit + /// that parked before reaching a prepare slot. `None` is the network reply + /// path; the other two constructors are this one with a fixed sender. + #[must_use] + pub const fn with_sender( + message: Message, + reply_sender: Option>>, + ) -> Self { + Self { message, received_at: 0, - reply_sender: Some(sender), - }; - (entry, receiver) + reply_sender, + } } /// Take the reply sender for hand-off to the promoted pipeline entry. @@ -656,6 +671,24 @@ impl LocalPipeline { .any(|r| r.message.header().client == client) } + /// True if either queue already holds this exact `(client, request)`. + /// + /// The partition-plane in-flight check. Narrower than + /// [`Self::has_message_from_client`] on purpose: the partition pipeline is + /// depth-`prepare_queue_depth` by design, so blocking every concurrent + /// request from one client would serialize it to one in-flight write per + /// group. Only an exact replay needs absorbing. + #[must_use] + pub fn has_message_from_client_request(&self, client: u128, request: u64) -> bool { + self.prepare_queue + .iter() + .any(|p| p.header.client == client && p.header.request == request) + || self.request_queue.iter().any(|r| { + let header = r.message.header(); + header.client == client && header.request == request + }) + } + /// Verify pipeline invariants. /// /// # Panics @@ -769,6 +802,10 @@ impl Pipeline for LocalPipeline { Self::has_message_from_client(self, client_id) } + fn has_message_from_client_request(&self, client_id: u128, request: u64) -> bool { + Self::has_message_from_client_request(self, client_id, request) + } + fn cancel_all_subscribers(&mut self) { Self::cancel_all_subscribers(self); } @@ -1763,6 +1800,16 @@ impl> VsrConsensus { self.pipeline.borrow().has_message_from_client(client_id) } + /// True iff this exact `(client, request)` is already in flight. The + /// partition plane's in-flight dedup: absorbs a replay without serializing + /// a client's pipeline depth. + #[must_use] + pub fn pipeline_has_message_from_client_request(&self, client_id: u128, request: u64) -> bool { + self.pipeline + .borrow() + .has_message_from_client_request(client_id, request) + } + /// Header of the oldest in-flight prepare. #[must_use] pub fn pipeline_head_header(&self) -> Option { diff --git a/core/consensus/src/lib.rs b/core/consensus/src/lib.rs index d5647b52f6..0abad9657b 100644 --- a/core/consensus/src/lib.rs +++ b/core/consensus/src/lib.rs @@ -70,13 +70,22 @@ pub trait Pipeline { fn verify(&self); /// True iff either queue carries `client_id`. Used by metadata-plane - /// preflight for in-flight dedup. Partition plane is at-least-once - /// and skips. Default `false`; falls through to slot dedup in + /// preflight for in-flight dedup; the partition plane uses the narrower + /// [`Self::has_message_from_client_request`] instead, to keep a client's + /// pipeline depth. Default `false`; falls through to slot dedup in /// `check_request`. fn has_message_from_client(&self, _client_id: u128) -> bool { false } + /// True iff either queue carries this exact `(client, request)`. The + /// partition-plane in-flight dedup check: narrow on purpose, so a client + /// keeps its pipeline depth and only an exact replay is absorbed. + /// Default `false`. + fn has_message_from_client_request(&self, _client_id: u128, _request: u64) -> bool { + false + } + /// Drop reply senders on every entry; receivers wake `Canceled`. /// View-change reset uses this to unblock awaiters while preserving /// pipeline for DVC reconciliation. @@ -161,8 +170,9 @@ where pub mod client_table; pub mod le_cursor; pub use client_table::{ - CachedReply, ClientEntrySnapshot, ClientTable, ClientTableDecodeError, ClientTableSnapshot, - ClientTableWireError, CommitReply, DISCONNECT_LOGOUT_REQUEST_ID, FenceSnapshot, SessionEnd, + CachedReply, ClientEntrySnapshot, ClientTable, ClientTableDecodeError, ClientTableMode, + ClientTableSnapshot, ClientTableWireError, CommitReply, DISCONNECT_LOGOUT_REQUEST_ID, + DedupWatermark, FenceSnapshot, SessionEnd, }; pub mod state_manifest; pub use state_manifest::{ @@ -176,7 +186,7 @@ pub use state_transfer::{ }; // One-shot per `PipelineEntry` for in-process commit awaiters. pub(crate) mod oneshot; -pub use oneshot::{Canceled, Receiver}; +pub use oneshot::{Canceled, Receiver, Sender, channel as oneshot_channel}; mod fatal; pub use fatal::{FatalReason, fatal}; diff --git a/core/integration/tests/cluster/mod.rs b/core/integration/tests/cluster/mod.rs index b794264a0a..57359759f3 100644 --- a/core/integration/tests/cluster/mod.rs +++ b/core/integration/tests/cluster/mod.rs @@ -26,6 +26,7 @@ mod metadata_checkpoint_restart; mod metadata_state_transfer; mod multi_shard_partition_convergence; mod parked_frame_redispatch; +mod partition_dedup; mod partition_primary_routing; mod partition_state_transfer; mod register_forwarding; diff --git a/core/integration/tests/cluster/partition_dedup.rs b/core/integration/tests/cluster/partition_dedup.rs new file mode 100644 index 0000000000..892211b741 --- /dev/null +++ b/core/integration/tests/cluster/partition_dedup.rs @@ -0,0 +1,749 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Spec tests for partition-plane request dedup (IGGY-274). +//! +//! Each partition consensus group keeps a slice of the VSR client table: +//! per-client request watermarks folded in at commit. A replay of an +//! already-committed `(client, request)` is answered with the empty success its +//! original earned instead of committing a second copy. +//! +//! The frames are hand-crafted on a raw TCP socket for the same reason +//! `client_table_restart` does it: the Rust SDK mints a fresh `client_id` and +//! request id per attempt, so it cannot express "the same request, twice" -- +//! which is precisely the input under test. The SDK is still used for setup and +//! for reading the log back, where it is the more honest observer. + +use bytes::{Bytes, BytesMut}; +use futures::future::join_all; +use iggy::prelude::*; +use iggy_binary_protocol::codec::WireEncode; +use iggy_binary_protocol::consensus::{ + Command, Operation, ReplyHeader, RequestHeader, read_size_field, +}; +use iggy_binary_protocol::requests::consumer_offsets::StoreConsumerOffsetRequest; +use iggy_binary_protocol::requests::messages::send_messages::{RawMessage, SendMessagesEncoder}; +use iggy_binary_protocol::requests::users::LoginRegisterRequest; +use iggy_binary_protocol::{ + AckLevel, ClientVersionInfo, HEADER_SIZE, IGGY_PROTOCOL_VERSION, WireConsumer, WireIdentifier, + WireName, WirePartitioning, +}; +use integration::harness::TestHarness; +use integration::iggy_harness; +use secrecy::SecretString; +use std::mem::offset_of; +use std::net::SocketAddr; +use std::time::Duration; +use tokio::io::{AsyncReadExt, AsyncWriteExt}; +use tokio::net::TcpStream; +use tokio::time::{Instant, sleep, timeout}; + +const STREAM_NAME: &str = "partition-dedup-stream"; +const TOPIC_NAME: &str = "partition-dedup-topic"; +const PARTITION_ID: u32 = 0; + +/// Fixed wire identity, so the replay frame is byte-identical to the original. +/// The SDK would randomize this. +const CLIENT_ID: u128 = 0x0DED_1234_5678; + +/// Second identity for liveness probes. The dedup watermark is a per-client +/// max, so a probe under [`CLIENT_ID`] would raise that client's watermark and +/// mask a missing transfer; the probe must not touch the identity under test. +const PROBE_CLIENT_ID: u128 = 0x0DED_9999_0001; + +const REPLY_WAIT: Duration = Duration::from_secs(10); +const COMMIT_BUDGET: Duration = Duration::from_secs(20); +const RETRY_PAUSE: Duration = Duration::from_millis(100); + +#[iggy_harness(server(system.sharding.cpu_allocation = "0..1"))] +async fn given_committed_send_when_replayed_should_absorb_without_a_second_copy( + harness: &mut TestHarness, +) { + let client = harness + .root_client_for_node(0) + .await + .expect("connect a root client"); + seed_topic(&client).await; + + let addr = harness.node(0).tcp_addr().expect("node tcp address"); + let (mut stream, session) = register(addr).await; + + let body = send_messages_body(b"only-once"); + let header = request_header(Operation::SendMessages, session, 1, body.len()); + + let original = exchange_until_committed(&mut stream, &header, &body).await; + assert_eq!(original, 0, "the original send must commit"); + + // Byte-identical replay: what a retry after a lost reply looks like. + let replayed = exchange_until_committed(&mut stream, &header, &body).await; + assert_eq!( + replayed, 0, + "an absorbed duplicate is a success, not an error" + ); + + let polled = poll_all(&client).await; + assert_eq!( + polled, 1, + "the replayed send must not append a second copy (got {polled} messages)" + ); +} + +#[iggy_harness(server(system.sharding.cpu_allocation = "0..1"))] +async fn given_committed_send_when_next_request_id_arrives_should_admit_it( + harness: &mut TestHarness, +) { + // The watermark must not wedge the client: the id above it still commits. + // Without this, "dedup works" and "the plane is broken" look identical. + let client = harness + .root_client_for_node(0) + .await + .expect("connect a root client"); + seed_topic(&client).await; + + let addr = harness.node(0).tcp_addr().expect("node tcp address"); + let (mut stream, session) = register(addr).await; + + for request in 1..=3u64 { + let body = send_messages_body(format!("message-{request}").as_bytes()); + let header = request_header(Operation::SendMessages, session, request, body.len()); + let status = exchange_until_committed(&mut stream, &header, &body).await; + assert_eq!(status, 0, "request {request} must commit"); + } + + let polled = poll_all(&client).await; + assert_eq!(polled, 3, "each distinct request id must append once"); +} + +#[iggy_harness(server(system.sharding.cpu_allocation = "0..1"))] +async fn given_gapped_request_id_when_sent_should_commit(harness: &mut TestHarness) { + // One client counter feeds every group it writes to, so a slice only ever + // sees a subset of the ids minted. Gaps must be legal, not a wedge. + let client = harness + .root_client_for_node(0) + .await + .expect("connect a root client"); + seed_topic(&client).await; + + let addr = harness.node(0).tcp_addr().expect("node tcp address"); + let (mut stream, session) = register(addr).await; + + for request in [1u64, 9, 40] { + let body = send_messages_body(format!("gap-{request}").as_bytes()); + let header = request_header(Operation::SendMessages, session, request, body.len()); + let status = exchange_until_committed(&mut stream, &header, &body).await; + assert_eq!(status, 0, "gapped request {request} must commit"); + } + + let polled = poll_all(&client).await; + assert_eq!(polled, 3, "a gapped id is new, not a duplicate"); +} + +#[iggy_harness(server(system.sharding.cpu_allocation = "0..1"))] +async fn given_committed_consumer_offset_when_replayed_should_absorb(harness: &mut TestHarness) { + // Dedup covers every replicated partition write, not just produces. A + // replayed offset store must answer success rather than committing twice. + let client = harness + .root_client_for_node(0) + .await + .expect("connect a root client"); + seed_topic(&client).await; + + let addr = harness.node(0).tcp_addr().expect("node tcp address"); + let (mut stream, session) = register(addr).await; + + // Seed a message so offset 0 is in range for the store. + let produce = send_messages_body(b"seed"); + let produce_header = request_header(Operation::SendMessages, session, 1, produce.len()); + assert_eq!( + exchange_until_committed(&mut stream, &produce_header, &produce).await, + 0, + "the seed produce must commit" + ); + + let body = store_offset_body(0); + let header = request_header(Operation::StoreConsumerOffset, session, 2, body.len()); + + let original = exchange_until_committed(&mut stream, &header, &body).await; + assert_eq!(original, 0, "the original offset store must commit"); + + let replayed = exchange_until_committed(&mut stream, &header, &body).await; + assert_eq!( + replayed, 0, + "a replayed offset store is absorbed as a success" + ); + + // The next id still gets through: the watermark must not wedge the client. + let next = store_offset_body(0); + let next_header = request_header(Operation::StoreConsumerOffset, session, 3, next.len()); + assert_eq!( + exchange_until_committed(&mut stream, &next_header, &next).await, + 0, + "the id above the watermark must still commit" + ); +} + +/// More connections than the prepare queue holds, each with one write in +/// flight at the same instant, so the surplus parks in the request queue and is +/// promoted into a prepare slot at a later commit. Every one of them must be +/// answered: a write promoted without the reply sender it parked with commits +/// but leaves its connection waiting out a timeout, and the count is the +/// second discriminator (each writer's single id must commit exactly once). +const CONCURRENT_WRITERS: u64 = 40; + +/// Distinct identity per writer, so each connection's watermark is its own and +/// the request ids can all be 1. +const WRITER_CLIENT_BASE: u128 = 0x0DED_C0DE_0000; + +#[iggy_harness( + cluster_nodes = 3, + server( + system.sharding.cpu_allocation = "0..1", + partition.prepare_queue_depth = "4" + ) +)] +async fn given_more_writers_than_prepare_slots_when_all_send_at_once_should_answer_every_one( + harness: &mut TestHarness, +) { + // Three nodes so a prepare needs a replication round trip to commit and the + // pipeline actually fills; a solo primary self-acks per frame and never + // exposes the request queue. The queue depth is pinned low so forty writers + // overflow it deterministically rather than by timing luck. + let client = harness + .root_client_for_node(0) + .await + .expect("connect a root client"); + seed_topic(&client).await; + + let addr = harness.node(0).tcp_addr().expect("node tcp address"); + // Register every connection first so the writes race each other, not the + // logins. + let mut connections = Vec::with_capacity(CONCURRENT_WRITERS as usize); + for writer in 0..CONCURRENT_WRITERS { + let client_id = WRITER_CLIENT_BASE + u128::from(writer); + let (stream, session) = register_client_with_budget(addr, client_id, COMMIT_BUDGET).await; + connections.push((client_id, stream, session)); + } + + let sends = connections + .iter_mut() + .map(|(client_id, stream, session)| async move { + let body = send_messages_body(format!("writer-{client_id:x}").as_bytes()); + let header = + request_header_for(*client_id, Operation::SendMessages, *session, 1, body.len()); + exchange_with_budget(stream, &header, &body, COMMIT_BUDGET).await + }); + let statuses = join_all(sends).await; + for (writer, status) in statuses.into_iter().enumerate() { + assert_eq!( + status, 0, + "writer {writer} must be answered with its commit (got status {status})" + ); + } + + let polled = poll_all(&client).await; + assert_eq!( + u64::from(polled), + CONCURRENT_WRITERS, + "every writer's single request must commit exactly once" + ); +} + +/// `StoreConsumerOffset` body for the raw connection's own consumer id. +fn store_offset_body(offset: u64) -> Bytes { + StoreConsumerOffsetRequest { + consumer: WireConsumer::consumer(WireIdentifier::numeric(1)), + stream_id: WireIdentifier::named(STREAM_NAME).expect("stream identifier"), + topic_id: WireIdentifier::named(TOPIC_NAME).expect("topic identifier"), + partition_id: Some(PARTITION_ID), + offset, + ack: AckLevel::Quorum, + } + .to_bytes() +} + +/// State-transfer choreography end to end: rejoin, view changes, and the +/// final commits ride slow CI runners. +const TRANSFER_BUDGET: Duration = Duration::from_secs(60); +const FINAL_COMMIT_BUDGET: Duration = Duration::from_secs(120); +const MARKER_POLL: Duration = Duration::from_millis(200); +const INSTALL_MARKER: &str = "partition state transfer installed"; + +/// Pre-stop produces fold into every replica's slice live; the rest commit +/// while node 2 is down. Total must push the evicted ring (capacity 64) past +/// the rejoiner's durable end, or repair closes the gap and no transfer runs. +const PRE_STOP_SENDS: u64 = 40; +/// The identity under test stops sending here; everything after comes from the +/// filler client. The rejoiner's tail repair re-applies the ring window (the +/// LAST ~64 commits) through the ordinary commit path, and the watermark is a +/// per-client max -- so if the tested client appeared anywhere in that window, +/// repair alone would cover every lower id and the artifact would be +/// redundant. The filler pushes the tested client's last send out of the ring, +/// leaving the transferred artifact as node 2's ONLY source for it. +const TESTED_CLIENT_SENDS: u64 = 140; +const FILLER_SENDS: u64 = 100; +const TOTAL_SENDS: u64 = TESTED_CLIENT_SENDS + FILLER_SENDS; +/// Replayed id: the tested client's watermark itself. Absorbing it requires an +/// entry for that client, which only the transferred artifact can supply. +const REPLAYED_REQUEST: u64 = TESTED_CLIENT_SENDS; + +/// Filler identity whose sends evict the tested client from the repair ring. +const FILLER_CLIENT_ID: u128 = 0x0DED_F111_E400; + +/// Sentinel status for an Eviction frame: the connection's session is gone and +/// the caller must reconnect and re-register before retrying. +const EVICTED: u32 = u32::MAX; + +/// Sentinel status for a socket the server closed mid-exchange (a node stopped +/// under the connection). Same contract as [`EVICTED`]: reconnect, re-register, +/// retry the identical frame. +const DISCONNECTED: u32 = u32::MAX - 1; + +#[iggy_harness( + cluster_nodes = 3, + server( + system.sharding.cpu_allocation = "0..1", + partition.evicted_ring_capacity = "64" + ) +)] +async fn given_transferred_dedup_slice_when_old_request_replays_should_absorb( + harness: &mut TestHarness, +) { + // Phase 1: node 0 is every group's view-0 primary. Produce the pre-stop + // window with node 2 live, then the rest with it stopped, so the second + // window exists on node 2 only via state transfer. + let client = harness + .root_client_for_node(0) + .await + .expect("connect a root client"); + seed_topic(&client).await; + let addr = harness.node(0).tcp_addr().expect("node 0 tcp address"); + let (mut stream, session) = register(addr).await; + raw_produce(&mut stream, session, 1..=PRE_STOP_SENDS, COMMIT_BUDGET).await; + sleep(Duration::from_secs(1)).await; + harness.stop_node(2).expect("stop node 2"); + raw_produce( + &mut stream, + session, + (PRE_STOP_SENDS + 1)..=TESTED_CLIENT_SENDS, + COMMIT_BUDGET, + ) + .await; + drop(stream); + let (mut filler, filler_session) = + register_client_with_budget(addr, FILLER_CLIENT_ID, COMMIT_BUDGET).await; + raw_produce_for( + &mut filler, + FILLER_CLIENT_ID, + filler_session, + 1..=FILLER_SENDS, + COMMIT_BUDGET, + ) + .await; + drop(filler); + drop(client); + + // Phase 2: the rejoin cannot repair past the survivors' evicted ring, so + // it converts to state transfer; the install carries the dedup section. + harness.restart_node(2).expect("restart node 2"); + await_marker(harness, 2, INSTALL_MARKER).await; + + // Phase 3: walk the primaries off node 0 and node 1 so the REPLAY is + // admitted by the transferred node. Stopping node 0 elects node 1 + // (view 1); after node 0 rejoins, stopping node 1 elects node 2 (view 2) + // with quorum {0, 2}. + harness.stop_node(0).expect("stop node 0"); + sleep(Duration::from_secs(2)).await; + harness.restart_node(0).expect("restart node 0"); + sleep(Duration::from_secs(2)).await; + harness.stop_node(1).expect("stop node 1"); + + // Phase 4, on node 2. The probe send goes FIRST and under a DIFFERENT + // client: its commit proves the view settled on node 2 and the rejoined + // node 0 is acking, and it pins the expected count -- while leaving + // CLIENT_ID's watermark exactly what the transfer installed (the watermark + // is a per-client max, so a same-client probe would mask a missing + // transfer). Sends reconnect + re-register on eviction; dedup keys on the + // client id and must hold across a re-register. + let addr = harness.node(2).tcp_addr().expect("node 2 tcp address"); + let fresh = send_reconnecting(addr, PROBE_CLIENT_ID, 1, FINAL_COMMIT_BUDGET).await; + assert_eq!( + fresh, 0, + "the probe client's send must commit on the new primary" + ); + + let replayed = send_reconnecting(addr, CLIENT_ID, REPLAYED_REQUEST, FINAL_COMMIT_BUDGET).await; + assert_eq!( + replayed, 0, + "a replay of a transferred watermark is absorbed as a success" + ); + + // The count is the discriminator: an absorbed replay leaves it at + // TOTAL_SENDS + 1; a re-execution (empty transferred slice) appends a + // second copy of the replayed payload. + let client = harness + .root_client_for_node(2) + .await + .expect("connect a root client to node 2"); + let polled = poll_up_to(&client, (TOTAL_SENDS + 16) as u32).await; + assert_eq!( + u64::from(polled), + TOTAL_SENDS + 1, + "the transferred slice must absorb the replay instead of re-executing it" + ); +} + +/// One send under `request`, surviving evictions: reconnect, re-register, and +/// retry the identical frame until it answers or the budget runs out. +async fn send_reconnecting(addr: SocketAddr, client: u128, request: u64, budget: Duration) -> u32 { + let deadline = Instant::now() + budget; + let body = send_messages_body(format!("send-{request}").as_bytes()); + loop { + let remaining = deadline.saturating_duration_since(Instant::now()); + assert!( + remaining > Duration::ZERO, + "request {request} did not resolve within {budget:?}" + ); + let (mut stream, session) = register_client_with_budget(addr, client, remaining).await; + let header = request_header_for( + client, + Operation::SendMessages, + session, + request, + body.len(), + ); + let status = exchange_with_budget(&mut stream, &header, &body, remaining).await; + if status != EVICTED && status != DISCONNECTED { + return status; + } + sleep(RETRY_PAUSE).await; + } +} + +/// Produce one single-message batch per request id over the lockstep raw +/// connection, waiting out each commit. +async fn raw_produce( + stream: &mut TcpStream, + session: u64, + requests: std::ops::RangeInclusive, + budget: Duration, +) { + raw_produce_for(stream, CLIENT_ID, session, requests, budget).await; +} + +async fn raw_produce_for( + stream: &mut TcpStream, + client: u128, + session: u64, + requests: std::ops::RangeInclusive, + budget: Duration, +) { + for request in requests { + let body = send_messages_body(format!("send-{request}").as_bytes()); + let header = request_header_for( + client, + Operation::SendMessages, + session, + request, + body.len(), + ); + let status = exchange_with_budget(stream, &header, &body, budget).await; + assert_ne!( + status, DISCONNECTED, + "server closed the lockstep connection under request {request}" + ); + assert_eq!(status, 0, "request {request} must commit"); + } +} + +async fn await_marker(harness: &TestHarness, node: usize, marker: &str) { + let deadline = Instant::now() + TRANSFER_BUDGET; + while !harness.node(node).stdout_contains(marker) { + assert!( + Instant::now() < deadline, + "node {node} never logged {marker:?} within {TRANSFER_BUDGET:?}" + ); + sleep(MARKER_POLL).await; + } +} + +async fn seed_topic(client: &IggyClient) { + client + .create_stream(STREAM_NAME) + .await + .expect("create stream"); + let stream_id = Identifier::named(STREAM_NAME).expect("stream identifier"); + client + .create_topic( + &stream_id, + TOPIC_NAME, + &TopicCreateOptions { + partitions_count: Some(1), + message_expiry: Some(IggyExpiry::NeverExpire), + // Every commit flushes and ring-evicts, which is what marches + // the repair floor past a rejoiner and forces the transfer the + // transferred-slice spec depends on. + messages_required_to_save: Some(1), + ..TopicCreateOptions::default() + }, + ) + .await + .expect("create topic"); +} + +async fn poll_all(client: &IggyClient) -> u32 { + poll_up_to(client, 100).await +} + +async fn poll_up_to(client: &IggyClient, max: u32) -> u32 { + let stream_id = Identifier::named(STREAM_NAME).expect("stream identifier"); + let topic_id = Identifier::named(TOPIC_NAME).expect("topic identifier"); + client + .poll_messages( + &stream_id, + &topic_id, + Some(PARTITION_ID), + &Consumer::new(Identifier::numeric(1).expect("consumer identifier")), + &PollingStrategy::offset(0), + max, + false, + ) + .await + .expect("poll messages") + .messages + .len() as u32 +} + +/// Full `SendMessages` body: metadata prefix, batch header, one message. +fn send_messages_body(payload: &[u8]) -> Bytes { + let stream_id = WireIdentifier::named(STREAM_NAME).expect("stream identifier"); + let topic_id = WireIdentifier::named(TOPIC_NAME).expect("topic identifier"); + let partitioning = WirePartitioning::PartitionId(PARTITION_ID); + let messages = [RawMessage { + // A fixed id keeps the replay byte-identical; a zero would be + // server-stamped and the two frames would diverge. + id: 0x5EED, + origin_timestamp: 0, + headers: None, + payload, + }]; + let size = SendMessagesEncoder::encoded_size(&stream_id, &topic_id, &partitioning, &messages); + let mut buf = BytesMut::with_capacity(size); + SendMessagesEncoder::encode(&mut buf, &stream_id, &topic_id, &partitioning, &messages) + .expect("encode send_messages body"); + buf.freeze() +} + +fn request_header( + operation: Operation, + session: u64, + request: u64, + body_len: usize, +) -> RequestHeader { + request_header_for(CLIENT_ID, operation, session, request, body_len) +} + +fn request_header_for( + client: u128, + operation: Operation, + session: u64, + request: u64, + body_len: usize, +) -> RequestHeader { + RequestHeader { + command: Command::Request, + operation, + size: u32::try_from(HEADER_SIZE + body_len).unwrap(), + client, + session, + request, + ..Default::default() + } +} + +/// Exchange until the server stops answering transiently, returning the reply +/// status. A transient means the request was never admitted, so replaying it +/// keeps the same id -- exactly what the SDK's own retry loop does. +async fn exchange_until_committed( + stream: &mut TcpStream, + header: &RequestHeader, + body: &Bytes, +) -> u32 { + let status = exchange_with_budget(stream, header, body, COMMIT_BUDGET).await; + assert_ne!( + status, DISCONNECTED, + "server closed the lockstep connection under request {}", + header.request + ); + status +} + +async fn exchange_with_budget( + stream: &mut TcpStream, + header: &RequestHeader, + body: &Bytes, + budget: Duration, +) -> u32 { + let deadline = Instant::now() + budget; + loop { + let status = exchange(stream, header, body).await; + if !is_transient(status) { + return status; + } + assert!( + Instant::now() < deadline, + "request {} stayed transient for {budget:?}", + header.request + ); + sleep(RETRY_PAUSE).await; + } +} + +/// Write one frame, read one frame, return the reply status. The connection is +/// lockstep, so the reply that comes back is this request's. A socket the +/// server closed (a node stopping under the connection) answers +/// [`DISCONNECTED`] rather than panicking, so the reconnecting callers can +/// treat it like an eviction; a reply that never comes is still a failure. +async fn exchange(stream: &mut TcpStream, header: &RequestHeader, body: &Bytes) -> u32 { + if stream.write_all(bytemuck::bytes_of(header)).await.is_err() { + return DISCONNECTED; + } + if !body.is_empty() && stream.write_all(body).await.is_err() { + return DISCONNECTED; + } + + let mut reply_header = [0u8; HEADER_SIZE]; + match timeout(REPLY_WAIT, stream.read_exact(&mut reply_header)).await { + Ok(Ok(_)) => {} + Ok(Err(_)) => return DISCONNECTED, + Err(_) => panic!("reply header timed out"), + } + + let command_offset = offset_of!(RequestHeader, command); + if reply_header[command_offset] == Command::Eviction as u8 { + // The session died (view change, epoch fence): the contract is + // reconnect + re-register, and dedup must still hold because it keys + // on the client id, not the session. + return EVICTED; + } + assert_eq!( + reply_header[command_offset], + Command::Reply as u8, + "expected a Reply frame" + ); + + let status_offset = offset_of!(ReplyHeader, status); + let status = u32::from_le_bytes( + reply_header[status_offset..status_offset + 4] + .try_into() + .unwrap(), + ); + let total_size = read_size_field(&reply_header).expect("reply size field") as usize; + if total_size > HEADER_SIZE { + let mut discard = vec![0u8; total_size - HEADER_SIZE]; + match timeout(REPLY_WAIT, stream.read_exact(&mut discard)).await { + Ok(Ok(_)) => {} + Ok(Err(_)) => return DISCONNECTED, + Err(_) => panic!("reply body timed out"), + } + } + status +} + +/// Register `CLIENT_ID` as root, returning the connection and its bound +/// session. The session binds to THIS socket server-side, so every frame in a +/// test must reuse the returned stream. +async fn register(addr: SocketAddr) -> (TcpStream, u64) { + register_with_budget(addr, COMMIT_BUDGET).await +} + +/// Fresh socket per attempt: a login refused mid-election may come back as an +/// eviction that poisons the connection. +async fn register_with_budget(addr: SocketAddr, budget: Duration) -> (TcpStream, u64) { + register_client_with_budget(addr, CLIENT_ID, budget).await +} + +async fn register_client_with_budget( + addr: SocketAddr, + client: u128, + budget: Duration, +) -> (TcpStream, u64) { + let deadline = Instant::now() + budget; + loop { + let mut stream = TcpStream::connect(addr).await.unwrap(); + if let Some(session) = login_on(&mut stream, client).await { + return (stream, session); + } + assert!( + Instant::now() < deadline, + "register did not commit within {budget:?}" + ); + sleep(RETRY_PAUSE).await; + } +} + +async fn login_on(stream: &mut TcpStream, client: u128) -> Option { + let body = LoginRegisterRequest { + version_info: ClientVersionInfo { + protocol_version: IGGY_PROTOCOL_VERSION, + sdk_name: WireName::new("iggy274-raw").unwrap(), + sdk_version: WireName::new("0.0.1").unwrap(), + }, + username: WireName::new(DEFAULT_ROOT_USERNAME).unwrap(), + password: SecretString::from(DEFAULT_ROOT_PASSWORD), + client_context: None, + } + .to_bytes(); + let header = request_header_for(client, Operation::Register, 0, 0, body.len()); + + stream.write_all(bytemuck::bytes_of(&header)).await.unwrap(); + stream.write_all(&body).await.unwrap(); + + let mut reply_header = [0u8; HEADER_SIZE]; + let Ok(Ok(_)) = timeout(REPLY_WAIT, stream.read_exact(&mut reply_header)).await else { + return None; + }; + let command_offset = offset_of!(RequestHeader, command); + if reply_header[command_offset] != Command::Reply as u8 { + return None; + } + + let status_offset = offset_of!(ReplyHeader, status); + let status = u32::from_le_bytes( + reply_header[status_offset..status_offset + 4] + .try_into() + .unwrap(), + ); + let total_size = read_size_field(&reply_header).expect("login reply size") as usize; + let mut reply_body = vec![0u8; total_size - HEADER_SIZE]; + let Ok(Ok(_)) = timeout(REPLY_WAIT, stream.read_exact(&mut reply_body)).await else { + return None; + }; + if status != 0 { + return None; + } + let session_offset = offset_of!(ReplyHeader, commit); + Some(u64::from_le_bytes( + reply_header[session_offset..session_offset + 8] + .try_into() + .unwrap(), + )) +} + +fn is_transient(code: u32) -> bool { + code == IggyError::TransientNotCommitted.as_code() + || code == IggyError::TransientNotAccepted.as_code() +} diff --git a/core/partitions/src/iggy_partition.rs b/core/partitions/src/iggy_partition.rs index 1af433af23..3cbdf3e177 100644 --- a/core/partitions/src/iggy_partition.rs +++ b/core/partitions/src/iggy_partition.rs @@ -36,9 +36,9 @@ use crate::{ PollingConsumer, }; use consensus::{ - CommitLogEvent, Consensus, PartitionDiagEvent, PipelineEntry, PlaneKind, Project, - ReplicaLogContext, RequestLogEvent, Sequencer, SimEventKind, VsrConsensus, ack_preflight, - ack_quorum_reached, build_deny_reply_from_request, build_reply_from_request, + ClientTable, ClientTableMode, CommitLogEvent, Consensus, PartitionDiagEvent, PipelineEntry, + PlaneKind, Project, ReplicaLogContext, RequestLogEvent, Sequencer, SimEventKind, VsrConsensus, + ack_preflight, ack_quorum_reached, build_deny_reply_from_request, build_reply_from_request, build_reply_message, drain_committable_prefix, emit_namespace_progress_event, emit_partition_diag, emit_sim_event, fence_old_prepare_by_commit, repaired_frontier_update, replicate_frozen_to_next_in_chain, replicate_preflight, restamp_prepare_view, @@ -53,7 +53,7 @@ use iggy_binary_protocol::responses::messages::{ use iggy_binary_protocol::{ AckLevel, GenericHeader, Operation, PrepareHeader, WireDecode, WireEncode, WireIdentifier, }; -use iggy_binary_protocol::{PrepareOkHeader, RoutedRequestHeader}; +use iggy_binary_protocol::{PrepareOkHeader, ReplyHeader, RoutedRequestHeader}; use iggy_common::{ ConsumerGroupId, ConsumerGroupOffsets, ConsumerKind, ConsumerOffset, ConsumerOffsets, IggyByteSize, IggyError, IggyExpiry, IggyTimestamp, PartitionStats, PollingKind, @@ -86,16 +86,24 @@ use tokio::sync::Mutex as TokioMutex; use tracing::{debug, error, warn}; // This struct aliases in terms of the code contained the `LocalPartition from `core/server/src/streaming/partitions/local_partition.rs`. -// -// Note: there is no per-client write dedup at the partition plane. -// `SendMessages` retries are at-least-once and may commit multiple times. -// Duplicate suppression is a consensus-layer concern: the VSR client table -// dedups by request id (at-most-once), so the data plane needs no message-id set. pub struct IggyPartition where B: MessageBus, { consensus: VsrConsensus, + /// This group's slice of the VSR client table, run in + /// [`ClientTableMode::PartitionSlice`]: per-client request watermarks + /// folded in at commit. Replica-local and memory-only: boot lifts the commit + /// frontier without re-applying the log, so a restarted replica comes back + /// with an empty slice while its peers keep theirs, and only commits folded + /// in after boot, or a state-transfer install, rebuild it. The mode turns + /// off what this plane cannot use -- no reply ring (`SendMessages` has no + /// result section, so a duplicate is answered by synthesizing the empty + /// success its original earned), no epoch fence (a partition group never + /// observes a `Register`), and no preallocated slot array (one table per + /// group, where preallocating the cap would reserve hundreds of KiB per + /// partition before a client connects). + dedup: ClientTable, pub log: SegmentedLog>, /// Highest durably persisted offset. pub offset: Arc, @@ -460,6 +468,10 @@ where let single_replica = consensus.replica_count() == 1; let partition = Self { consensus, + dedup: ClientTable::with_mode( + consensus::PARTITION_DEDUP_CLIENTS_MAX, + ClientTableMode::PartitionSlice, + ), log: SegmentedLog::default(), offset: Arc::new(AtomicU64::new(0)), dirty_offset: AtomicU64::new(0), @@ -555,6 +567,26 @@ where &self.consensus } + /// This group's dedup slice. Read at admission to classify a request, + /// written only from the commit path. + #[must_use] + pub(crate) const fn dedup(&self) -> &ClientTable { + &self.dedup + } + + /// Mutable slice, for the commit path and state-transfer install. + pub(crate) const fn dedup_mut(&mut self) -> &mut ClientTable { + &mut self.dedup + } + + /// Size the dedup slice to `[partition] dedup_clients_max`. Boot-only: + /// `set_capacity` replaces the table rather than evicting into the new + /// bound, and panics if the slice already holds an entry. Config + /// validation rejects a zero cap before it can reach here. + pub fn set_dedup_clients_max(&mut self, clients_max: usize) { + self.dedup.set_capacity(clients_max); + } + #[must_use] pub fn with_in_memory_storage( stats: Arc, @@ -1435,8 +1467,9 @@ where } /// `AckLevel::NoAck` fast path: persist, apply, send reply, no - /// replication. Single-replica durability. No reply cache: partition - /// plane is at-least-once; session lifecycle lives on metadata. + /// replication. Single-replica durability. Never recorded in the dedup + /// slice: it does not replicate, so folding it in would fork the slice + /// across replicas. Session lifecycle lives on metadata. #[allow(clippy::future_not_send)] async fn apply_consumer_offset_no_ack( &self, @@ -1444,6 +1477,7 @@ where kind: ConsumerKind, consumer_id: u32, offset: Option, + waiter: Option>>, ) { let pending = offset.map_or_else( || PendingConsumerOffsetCommit::delete(kind, consumer_id), @@ -1474,6 +1508,12 @@ where &request_header, committed_reply_body(request_header.operation), ); + // Same rule as the committed path: a submit's waiter takes the reply, + // because `header.client` is then the VSR consensus id. + if let Some(waiter) = waiter { + let _ = waiter.send(reply); + return; + } let reply_buffers = reply.into_generic().into_frozen(); if let Err(error) = self .consensus @@ -1951,16 +1991,28 @@ where /// Project a client request into a prepare. /// - /// At-least-once: no per-client dedup. `SendMessages` retry -> fresh - /// prepare, may re-commit at new offset. Consumers handle dedup - /// (message key / content / producer-id+seq). Session lifecycle + - /// eviction live on metadata plane. + /// A replay of a committed `(client, request)` is absorbed by this group's + /// dedup slice; anything above the watermark projects into a prepare. + /// Session lifecycle + eviction live on the metadata plane. + /// + /// `reply` is the in-process channel a `PartitionSubmit` carried in. When + /// present the committed reply fires on it instead of going to the bus: + /// the connection-owning shard writes it to the socket it holds, because + /// `header.client` is the VSR consensus id and carries no home-shard + /// routing. `None` keeps the bus path (auto-commit ops, tests). /// /// # Panics /// Panics if called when this partition's consensus instance is not the /// primary, is not in normal status, or is currently syncing. #[allow(clippy::future_not_send, clippy::too_many_lines)] - pub async fn on_request(&mut self, message: Message) { + pub async fn on_request( + &mut self, + message: Message, + reply: Option>>, + ) { + // Taken by whichever arm answers: the deny paths, the NoAck fast path, + // or the pipeline entry that fires it at commit. Exactly one runs. + let mut reply = reply; self.clear_pending_consumer_offset_commits_if_view_changed(); let namespace = IggyNamespace::from_raw(message.header().group); let client_id = message.header().client; @@ -2023,6 +2075,7 @@ where message.header(), IggyError::TransientNotAccepted.as_code(), "non-primary transient reply send failed", + reply.take(), ) .await; return; @@ -2054,6 +2107,63 @@ where _ => None, }; + // Dedup BEFORE the admission checks below: a replay of an + // already-committed delete must answer the success its original + // earned, not the typed 404 the existence check would raise now + // that the offset is gone. + // + // A replay racing its own in-flight original is absorbed here: the + // slice only knows committed ops, so it cannot yet see the copy + // still in the pipeline. Keyed on the exact `(client, request)` + // for the transports that keep several writes in flight per + // client (HTTP handlers on one session, the pipelining SDKs): + // matching any request from the client would serialize them to + // one in-flight write per group. A lockstep TCP connection never + // has a second request here to begin with. + // + // Those same transports can deliver a client's ids out of order: + // a write refused transiently here is replayed after its + // successors committed. The slice therefore keeps a committed-id + // window under the watermark (`consensus::COMMITTED_WINDOW_BITS`) + // and admits an unmarked id inside it instead of absorbing it. + if !is_auto_commit_client(client_id) { + if consensus.pipeline_has_message_from_client_request(client_id, request) { + Self::send_partition_deny_or_log( + consensus, + message.header(), + IggyError::TransientNotCommitted.as_code(), + "in-flight dedup transient reply send failed", + reply.take(), + ) + .await; + return; + } + // An absorbed duplicate answers the operation's empty success. + // For `SendMessages` that is LESS than the original reply + // carried: the offset confirmations are not retained (no reply + // ring in this mode), so a retried produce learns it committed + // but not where. + if self + .dedup + .is_duplicate(client_id, message.header().user_id, request) + { + let committed = build_reply_from_request( + &self.consensus, + message.header(), + committed_reply_body(message.header().operation), + ); + Self::deliver_reply_or_log( + &self.consensus, + message.header(), + committed, + reply.take(), + "duplicate reply send failed", + ) + .await; + return; + } + } + if matches!(message.header().operation, Operation::DeleteConsumerOffset) && let Some((kind, consumer_id, _, _)) = consumer_offset && let Err(error) = self.ensure_consumer_offset_exists(kind, consumer_id) @@ -2078,6 +2188,7 @@ where message.header(), error.as_code(), "delete_consumer_offset deny reply send failed", + reply.take(), ) .await; return; @@ -2112,6 +2223,7 @@ where message.header(), IggyError::InvalidOffset(requested_offset).as_code(), "store_consumer_offset deny reply send failed", + reply.take(), ) .await; return; @@ -2136,9 +2248,10 @@ where // request room -> buffer; both full -> drop+warn (client retries // via read-timeout). if consensus.pipeline_is_full() { - let push_result = - consensus.push_queued_request(consensus::RequestEntry::new(message)); - if push_result.is_err() { + let push_result = consensus.push_queued_request( + consensus::RequestEntry::with_sender(message, reply.take()), + ); + if let Err(mut refused) = push_result { emit_partition_diag( tracing::Level::WARN, &PartitionDiagEvent::new( @@ -2146,13 +2259,32 @@ where "on_request: prepare and request queues both full, dropping", ), ); + // The request provably never entered either queue, so a + // waiter can be told so instead of waiting out its + // timeout. + let waiter = refused.take_reply_sender(); + Self::send_partition_deny_or_log( + consensus, + refused.message.header(), + IggyError::TransientNotAccepted.as_code(), + "queues-full transient reply send failed", + waiter, + ) + .await; } return; } let prepare = message.project(consensus); consensus.verify_pipeline(); - consensus.pipeline_message(PlaneKind::Partitions, &prepare); + match reply.take() { + Some(sender) => consensus.pipeline_message_with_sender( + PlaneKind::Partitions, + &prepare, + sender, + ), + None => consensus.pipeline_message(PlaneKind::Partitions, &prepare), + } Disposition::Replicate(prepare) } }; @@ -2165,17 +2297,23 @@ where consumer_id, offset, } => { - self.apply_consumer_offset_no_ack(request_header, kind, consumer_id, offset) - .await; + self.apply_consumer_offset_no_ack( + request_header, + kind, + consumer_id, + offset, + reply.take(), + ) + .await; } } } /// Promote up to `slots_freed` buffered requests into prepares post-commit. /// - /// No preflight: partition plane is at-least-once with no `ClientTable` - /// dedup. Buffered `SendMessages` retry commits at fresh offset; consumers - /// dedup by message key / content / producer-id+seq. + /// Promotion runs no preflight: the request was classified at admission + /// and the slice cannot have gained a higher watermark for it since (only + /// a commit moves it, and this entry has not committed). /// /// Per-iteration `is_primary && is_normal && !is_transferring` asserts inlined /// (closure form's `&consensus` borrow conflicts with `&mut self`). Guards @@ -2192,7 +2330,7 @@ where pub async fn drain_request_queue_into_prepares(&mut self, slots_freed: usize) { for _ in 0..slots_freed { let req = self.consensus().pop_queued_request(); - let Some(req) = req else { break }; + let Some(mut req) = req else { break }; let prepare = { let consensus = self.consensus(); @@ -2208,9 +2346,19 @@ where !consensus.is_transferring(), "drain_request_queue_into_prepares: must not be transferring state" ); + // The waiter parked with the request; it must travel into the + // prepare slot or the commit has nobody to answer. + let reply_sender = req.take_reply_sender(); let prepare = req.message.project(consensus); consensus.verify_pipeline(); - consensus.pipeline_message(PlaneKind::Partitions, &prepare); + match reply_sender { + Some(sender) => consensus.pipeline_message_with_sender( + PlaneKind::Partitions, + &prepare, + sender, + ), + None => consensus.pipeline_message(PlaneKind::Partitions, &prepare), + } prepare }; self.on_replicate(prepare).await; @@ -3262,7 +3410,7 @@ where let committed_batch_stats = self.resolve_committed_visible_offsets(&drained); let mut messages_committed = false; - for (entry, batch_stats) in drained.into_iter().zip(committed_batch_stats) { + for (mut entry, batch_stats) in drained.into_iter().zip(committed_batch_stats) { let prepare_header = entry.header; if !self .commit_partition_entry( @@ -3312,6 +3460,20 @@ where self.consensus.advance_commit_min(prepare_header.op); + // Fold the committed request into this group's dedup slice. Runs on + // EVERY replica, not just the one that replies, so a promoted + // primary can absorb a replay of what its predecessor committed. + // Auto-commit ops carry the reserved sentinel client and no client + // ever replays them. + if !is_auto_commit_client(prepare_header.client) { + self.dedup.commit_request( + prepare_header.client, + prepare_header.user_id, + prepare_header.request, + prepare_header.op, + ); + } + let pipeline_depth = self.consensus.pipeline_len(); let event = CommitLogEvent { replica: ReplicaLogContext::from_consensus(&self.consensus, PlaneKind::Partitions), @@ -3329,9 +3491,11 @@ where pipeline_depth, ); - // No reply cache: at-least-once means retries re-commit at new - // offsets. Only primary delivers replies; backups just advance - // commit. Session lifecycle is metadata-only. + // No reply cache: an absorbed duplicate is answered by + // synthesizing the same empty success at admission, so no committed + // bytes need keeping. Only the primary delivers replies; backups + // just advance commit and fold the slice. Session lifecycle is + // metadata-only. // // A server-generated auto-commit op (a poll's `auto_commit`, // replicated for failover) carries the reserved @@ -3346,23 +3510,36 @@ where operation => committed_reply_body(operation), }; let reply = build_reply_message(&prepare_header, &body); - let reply_buffers = reply.into_generic().into_frozen(); emit_sim_event(SimEventKind::ClientReplyEmitted, &event); - if let Err(error) = self + // An in-process waiter takes the reply instead of the bus: it + // arrived as a `PartitionSubmit`, so `header.client` is the VSR + // consensus id and carries no home-shard routing. The awaiting + // shard owns the socket. A dropped receiver is ignored -- the + // client recovers on its own read-timeout. + // + // Without a waiter the bus is tried, and for a TCP client it + // cannot route the VSR id. That is the expected shape of every + // op re-committed after a view change (the rebuilt pipeline + // entries carry no sender), so it logs at debug: the original + // waiter was cancelled by the view change and the client is + // already on its read-timeout. + if let Some(sender) = entry.take_reply_sender() { + let _ = sender.send(reply); + } else if let Err(error) = self .consensus .message_bus() - .send_to_client(prepare_header.client, reply_buffers) + .send_to_client(prepare_header.client, reply.into_generic().into_frozen()) .await { - tracing::error!( + tracing::debug!( target: "iggy.partitions.diag", plane = "partitions", client = prepare_header.client, op = prepare_header.op, namespace_raw, %error, - "client reply forward failed, no retransmit path; client will time out", + "client reply not routable by the bus; client will time out", ); } } @@ -3570,14 +3747,34 @@ where /// Send `header`'s deny reply with `status` on `ReplyHeader.status` (empty /// body, op=0), logging a WARN under `send_fail_label` if the reply send /// fails. Callers deny on the primary, before the op enters the pipeline, - /// so nothing replicates. + /// so nothing replicates. `waiter` is the submit's in-process channel, + /// taken by the caller. async fn send_partition_deny_or_log( consensus: &VsrConsensus, header: &RoutedRequestHeader, status: u32, send_fail_label: &'static str, + waiter: Option>>, ) { let reply = build_deny_reply_from_request(consensus, header, status); + Self::deliver_reply_or_log(consensus, header, reply, waiter, send_fail_label).await; + } + + /// Deliver an admission-time reply (a deny, or an absorbed duplicate's + /// success). When `waiter` is present the reply goes there: `header.client` + /// is then the VSR consensus id, which the bus cannot route. Otherwise the + /// bus carries it, and a failed send logs a WARN under `send_fail_label`. + async fn deliver_reply_or_log( + consensus: &VsrConsensus, + header: &RoutedRequestHeader, + reply: Message, + waiter: Option>>, + send_fail_label: &'static str, + ) { + if let Some(waiter) = waiter { + let _ = waiter.send(reply); + return; + } if let Err(send_error) = consensus .message_bus() .send_to_client(header.client, reply.into_generic().into_frozen()) @@ -5664,7 +5861,7 @@ mod tests { let consumer_id: u32 = 5; partition - .on_request(delete_offset_request(client_id, 7, consumer_id)) + .on_request(delete_offset_request(client_id, 7, consumer_id), None) .await; { @@ -5702,7 +5899,7 @@ mod tests { ConsumerOffset::new(ConsumerKind::Consumer, consumer_id, 3, String::new()), ); partition - .on_request(delete_offset_request(client_id, 8, consumer_id)) + .on_request(delete_offset_request(client_id, 8, consumer_id), None) .await; assert_eq!( partition.consensus().pipeline_len(), @@ -5717,7 +5914,7 @@ mod tests { let client_id = 42; partition - .on_request(delete_offset_request(client_id, 7, 5)) + .on_request(delete_offset_request(client_id, 7, 5), None) .await; let sent = sent_to_clients.borrow(); @@ -7233,6 +7430,7 @@ mod tests { next_offset: 50, consumers: Vec::new(), groups: Vec::new(), + dedup: Vec::new(), }; let refused = partition .install_state_transfer(&repair_config(), 12, Vec::new(), &behind.encode(), 0) @@ -7260,6 +7458,7 @@ mod tests { next_offset: 0, consumers: Vec::new(), groups: Vec::new(), + dedup: Vec::new(), }; let accepted = partition .install_state_transfer(&repair_config(), 12, Vec::new(), &purged.encode(), 0) @@ -7299,6 +7498,7 @@ mod tests { next_offset: 50, consumers: Vec::new(), groups: Vec::new(), + dedup: Vec::new(), }; let refused = partition .install_state_transfer(&repair_config(), 12, Vec::new(), &offer.encode(), 1) @@ -7348,6 +7548,7 @@ mod tests { next_offset: 0, consumers: Vec::new(), groups: Vec::new(), + dedup: Vec::new(), }; let installed = partition .install_state_transfer(&repair_config(), 12, Vec::new(), &reset.encode(), 1) diff --git a/core/partitions/src/iggy_partitions.rs b/core/partitions/src/iggy_partitions.rs index 87887c9ee8..2a2ee6e1e5 100644 --- a/core/partitions/src/iggy_partitions.rs +++ b/core/partitions/src/iggy_partitions.rs @@ -21,12 +21,17 @@ use crate::poll_plan::PollPlan; use crate::types::PartitionsConfig; use crate::{IggyPartition, Partition, PollingArgs, PollingConsumer}; use ahash::AHashSet; -use consensus::{Consensus, Plane, PlaneIdentity, VsrConsensus}; +use consensus::{ + Consensus, Plane, PlaneIdentity, VsrConsensus, build_deny_reply_from_request_header, +}; use iggy_binary_protocol::{ - Command, ConsensusHeader, Operation, PrepareHeader, PrepareOkHeader, RoutedRequestHeader, + Command, ConsensusHeader, Operation, PrepareHeader, PrepareOkHeader, ReplyHeader, + RoutedRequestHeader, }; +use iggy_common::IggyError; use journal::superblock::{PingPongSuperblock, SuperblockStore}; use message_bus::MessageBus; +use server_common::Message; use server_common::send_messages::{ChecksumMode, convert_request_message, encrypt_batch_request}; use server_common::sharding::{IggyNamespace, LocalIdx, ShardId}; #[cfg(debug_assertions)] @@ -528,22 +533,35 @@ where } } -impl Plane> for IggyPartitions +impl IggyPartitions where B: MessageBus, SB: SuperblockStore, { - async fn on_request( + /// [`Plane::on_request`] carrying the in-process reply channel a + /// `PartitionSubmit` arrived with; `None` keeps the bus-reply path. + pub async fn on_request_with_reply( &self, message: as Consensus>::Message, + reply: Option>>, ) { let namespace = IggyNamespace::from_raw(message.header().group); + // Every exit below that drops the request answers its waiter first: a + // submit's reply cannot be routed by `header.client` (the VSR id), so + // an unanswered channel costs the client a full read-timeout for an + // outcome that was decided here and now. + let mut reply = reply; if self.is_tombstoned(&namespace) { warn!( target: "iggy.partitions.diag", namespace_raw = namespace.inner(), "dropping request: namespace tombstoned" ); + Self::answer_waiter( + reply.take(), + message.header(), + IggyError::TransientNotAccepted.as_code(), + ); return; } // At-rest encryption happens HERE, once, before the op enters @@ -559,6 +577,9 @@ where // `encrypt_batch_request`'s decode before re-encryption, and the // re-encrypted batch (checksum kept by `encrypt_batch_request`) then // re-enters `convert` as the canonical-vs-legacy discriminator. + // The header outlives the message consumed by the conversion, so a + // failure can still be answered with the frame's own identity. + let header = *message.header(); let canonical = convert_request_message(namespace, message, ChecksumMode::Compute) .and_then(|message| encrypt_batch_request(message, encryptor)); match canonical { @@ -570,6 +591,7 @@ where %error, "dropping send_messages: failed to encrypt batch at ingestion" ); + Self::answer_waiter(reply.take(), &header, error.as_code()); return; } } @@ -584,9 +606,40 @@ where operation = ?message.header().operation, "partition not initialized for namespace" ); + Self::answer_waiter( + reply.take(), + message.header(), + IggyError::TransientNotAccepted.as_code(), + ); return; }; - partition.on_request(message).await; + partition.on_request(message, reply).await; + } + + /// Deny a request that never reached its partition on the submit channel + /// it arrived with, if any. The bus path has no waiter to answer and keeps + /// its drop-and-warn behaviour. + fn answer_waiter( + waiter: Option>>, + header: &RoutedRequestHeader, + status: u32, + ) { + if let Some(waiter) = waiter { + let _ = waiter.send(build_deny_reply_from_request_header(header, status)); + } + } +} + +impl Plane> for IggyPartitions +where + B: MessageBus, + SB: SuperblockStore, +{ + async fn on_request( + &self, + message: as Consensus>::Message, + ) { + self.on_request_with_reply(message, None).await; } async fn on_replicate(&self, message: as Consensus>::Message) { diff --git a/core/partitions/src/state_transfer.rs b/core/partitions/src/state_transfer.rs index 5bd29807db..7444b9d76b 100644 --- a/core/partitions/src/state_transfer.rs +++ b/core/partitions/src/state_transfer.rs @@ -37,7 +37,9 @@ use crate::{IggyIndexWriter, IggyPartition}; use compio::io::{AsyncReadAtExt, AsyncWriteAtExt}; use consensus::le_cursor::{LeCursor, Truncated, split_verified_trailer}; use consensus::state_manifest::artifact_kind; -use consensus::{ArtifactProgress, Sequencer as _, StateArtifactHasher, state_artifact_checksum}; +use consensus::{ + ArtifactProgress, DedupWatermark, Sequencer as _, StateArtifactHasher, state_artifact_checksum, +}; use iggy_common::{ConsumerGroupId, ConsumerKind, ConsumerOffset, IggyByteSize}; use journal::superblock::SuperblockStore; use message_bus::MessageBus; @@ -50,15 +52,26 @@ use std::path::{Path, PathBuf}; use std::rc::Rc; use std::sync::atomic::Ordering; -/// Framing marker for the consumer-offsets wire artifact, "ICO1". -pub(crate) const CONSUMER_OFFSETS_MAGIC: [u8; 4] = *b"ICO1"; +/// Framing marker for the consumer-offsets wire artifact, "ICO2". Bumped with +/// the version when the dedup section was appended, so the magic alone tells +/// the two layouts apart. +pub(crate) const CONSUMER_OFFSETS_MAGIC: [u8; 4] = *b"ICO2"; /// Version byte following the magic. /// /// Any layout change bumps this, INCLUDING appended fields: the decoder /// deliberately fails closed on unknown versions and on trailing bytes, /// because a v2 field can change the meaning of fields v1 already read. -pub(crate) const CONSUMER_OFFSETS_VERSION: u8 = 1; +pub(crate) const CONSUMER_OFFSETS_VERSION: u8 = 2; + +/// The previous framing, "ICO1" at version 1: the same layout without the +/// dedup section. Still decoded so a rolling upgrade works in both orders -- +/// an upgraded replica rejoining behind the repair floor of an un-upgraded +/// primary installs its artifact with an empty slice (dedup for that window +/// degrades to at-least-once, exactly the pre-dedup behaviour) instead of +/// refusing it and re-pulling forever. +pub(crate) const CONSUMER_OFFSETS_MAGIC_V1: [u8; 4] = *b"ICO1"; +const CONSUMER_OFFSETS_VERSION_V1: u8 = 1; /// Per-section entry ceiling for the consumer-offsets artifact. /// @@ -67,6 +80,10 @@ pub(crate) const CONSUMER_OFFSETS_VERSION: u8 = 1; /// entry ceiling. pub(crate) const CONSUMER_OFFSETS_ENTRIES_MAX: u32 = 1 << 20; +/// Wire stride of one dedup entry: client u128 + watermark u64 + commit u64 + +/// user u32 + committed window u128. +const DEDUP_ENTRY_LEN: usize = 2 * size_of::() + 2 * size_of::() + size_of::(); + /// One in-flight partition state transfer on the receiving replica. /// /// Mirrors the metadata plane's session, plus `staged`: completed @@ -295,13 +312,19 @@ pub(crate) struct ConsumerOffsetsWire { pub consumers: Vec<(u32, u64)>, /// `(consumer group id, offset)`, ascending by id. pub groups: Vec<(u32, u64)>, + /// This group's dedup slice, ascending by client. Carried so a replica + /// rejoining behind the repair floor can absorb a replay of what the group + /// already committed instead of re-executing it. + pub dedup: Vec, } impl ConsumerOffsetsWire { /// Encode: `magic | version u8 | purge_generation u64 | next_offset u64 | - /// consumer_count u32 | group_count u32 | {id u32, offset u64}xN | - /// {id u32, offset u64}xM | XxHash3_64 trailer`. Little-endian - /// throughout. + /// consumer_count u32 | group_count u32 | dedup_count u32 | + /// {id u32, offset u64}xN | {id u32, offset u64}xM | + /// {client u128, watermark u64, latest_commit u64, user_id u32, + /// committed_window u128}xD | + /// XxHash3_64 trailer`. Little-endian throughout. #[must_use] pub fn encode(&self) -> Vec { // Size exactly rather than guess; the reservation assert keeps the @@ -309,8 +332,9 @@ impl ConsumerOffsetsWire { let reserved = CONSUMER_OFFSETS_MAGIC.len() + size_of::() + 2 * size_of::() - + 2 * size_of::() + + 3 * size_of::() + (self.consumers.len() + self.groups.len()) * (size_of::() + size_of::()) + + self.dedup.len() * DEDUP_ENTRY_LEN + size_of::(); let mut out = Vec::with_capacity(reserved); out.extend_from_slice(&CONSUMER_OFFSETS_MAGIC); @@ -321,10 +345,19 @@ impl ConsumerOffsetsWire { out.extend_from_slice(&(self.consumers.len() as u32).to_le_bytes()); #[allow(clippy::cast_possible_truncation)] out.extend_from_slice(&(self.groups.len() as u32).to_le_bytes()); + #[allow(clippy::cast_possible_truncation)] + out.extend_from_slice(&(self.dedup.len() as u32).to_le_bytes()); for (id, offset) in self.consumers.iter().chain(self.groups.iter()) { out.extend_from_slice(&id.to_le_bytes()); out.extend_from_slice(&offset.to_le_bytes()); } + for entry in &self.dedup { + out.extend_from_slice(&entry.client.to_le_bytes()); + out.extend_from_slice(&entry.watermark.to_le_bytes()); + out.extend_from_slice(&entry.latest_commit.to_le_bytes()); + out.extend_from_slice(&entry.user_id.to_le_bytes()); + out.extend_from_slice(&entry.committed_window.to_le_bytes()); + } debug_assert_eq!(out.len() + size_of::(), reserved, "encode reservation"); let trailer = state_artifact_checksum(&out); out.extend_from_slice(&trailer.to_le_bytes()); @@ -351,19 +384,32 @@ impl ConsumerOffsetsWire { })?; let mut cursor = LeCursor::new(content); let magic = cursor.take(CONSUMER_OFFSETS_MAGIC.len())?; - if magic != CONSUMER_OFFSETS_MAGIC { - return Err(ConsumerOffsetsWireError::BadMagic); - } let version = cursor.u8()?; - if version != CONSUMER_OFFSETS_VERSION { - return Err(ConsumerOffsetsWireError::UnsupportedVersion { version }); - } + let carries_dedup = if magic == CONSUMER_OFFSETS_MAGIC { + if version != CONSUMER_OFFSETS_VERSION { + return Err(ConsumerOffsetsWireError::UnsupportedVersion { version }); + } + true + } else if magic == CONSUMER_OFFSETS_MAGIC_V1 { + if version != CONSUMER_OFFSETS_VERSION_V1 { + return Err(ConsumerOffsetsWireError::UnsupportedVersion { version }); + } + false + } else { + return Err(ConsumerOffsetsWireError::BadMagic); + }; let purge_generation = cursor.u64()?; let next_offset = cursor.u64()?; let consumer_count = cursor.u32()?; let group_count = cursor.u32()?; + let dedup_count = if carries_dedup { cursor.u32()? } else { 0 }; let consumers = Self::decode_section(&mut cursor, "consumers", consumer_count)?; let groups = Self::decode_section(&mut cursor, "groups", group_count)?; + let dedup = if carries_dedup { + Self::decode_dedup_section(&mut cursor, dedup_count)? + } else { + Vec::new() + }; if !cursor.remaining().is_empty() { // Distinct from `Truncated`: extra bytes point at a NEWER // encoder, and telling the operator the artifact is short would @@ -377,9 +423,56 @@ impl ConsumerOffsetsWire { next_offset, consumers, groups, + dedup, }) } + /// Same guards as [`Self::decode_section`] at the dedup stride: peer count + /// against the ceiling, then against the bytes actually present, then + /// ascending-strict client order so the encoding stays canonical. Client + /// zero is the reserved id no ingress admits, so an artifact carrying it is + /// a peer bug and fails closed rather than being silently dropped at + /// install. + fn decode_dedup_section( + cursor: &mut LeCursor<'_>, + count: u32, + ) -> Result, ConsumerOffsetsWireError> { + if count > CONSUMER_OFFSETS_ENTRIES_MAX { + return Err(ConsumerOffsetsWireError::TooManyEntries { + section: "dedup", + count, + max: CONSUMER_OFFSETS_ENTRIES_MAX, + }); + } + if count as usize * DEDUP_ENTRY_LEN > cursor.remaining().len() { + return Err(ConsumerOffsetsWireError::Truncated); + } + let mut entries = Vec::with_capacity(count as usize); + let mut previous: Option = None; + for _ in 0..count { + let client = cursor.u128()?; + let watermark = cursor.u64()?; + let latest_commit = cursor.u64()?; + let user_id = cursor.u32()?; + let committed_window = cursor.u128()?; + if client == 0 { + return Err(ConsumerOffsetsWireError::ReservedClient); + } + if previous.is_some_and(|previous| client <= previous) { + return Err(ConsumerOffsetsWireError::NonAscendingClient { client }); + } + previous = Some(client); + entries.push(DedupWatermark { + client, + user_id, + watermark, + latest_commit, + committed_window, + }); + } + Ok(entries) + } + fn decode_section( cursor: &mut LeCursor<'_>, section: &'static str, @@ -397,9 +490,12 @@ impl ConsumerOffsetsWire { // The count is peer input and the reservation is 12 bytes per element // after alignment, so it is checked against the bytes actually present // before allocating: a ~30 byte artifact could otherwise ask for tens of - // megabytes across the two sections. 12 is the wire stride below -- the - // groups call sees exactly `12 * count` bytes remaining, so a wider - // guard would reject every non-empty artifact. + // megabytes across the two sections. 12 is the wire stride below, and + // the guard is a lower bound on purpose: what trails a section varies + // (groups trails consumers, the dedup section trails groups and is + // absent from a v1 artifact), so only "at least this many bytes + // present" holds for both calls. Demanding that `12 * count` be all + // that remains would reject valid input. if count as usize * (size_of::() + size_of::()) > cursor.remaining().len() { return Err(ConsumerOffsetsWireError::Truncated); } @@ -450,6 +546,14 @@ pub enum ConsumerOffsetsWireError { section: &'static str, id: u32, }, + /// Dedup clients are not strictly ascending. Same encoder bug as + /// [`Self::NonAscendingId`], on the u128-keyed section. + NonAscendingClient { + client: u128, + }, + /// A dedup entry carries client id zero, which is reserved and refused at + /// every ingress: a peer encoder bug. + ReservedClient, } impl From for ConsumerOffsetsWireError { @@ -490,6 +594,17 @@ impl fmt::Display for ConsumerOffsetsWireError { "consumer-offsets artifact {section} id {id} does not ascend \ (duplicate, or out of order)" ), + Self::NonAscendingClient { client } => write!( + f, + "consumer-offsets artifact dedup client {client} does not ascend \ + (duplicate, or out of order)" + ), + Self::ReservedClient => { + write!( + f, + "consumer-offsets artifact dedup entry carries reserved client 0" + ) + } } } } @@ -506,6 +621,20 @@ mod tests { next_offset: 43, consumers: vec![(1, 10), (7, 42)], groups: vec![(2, 5)], + dedup: vec![ + dedup_entry(11, 4, 90), + dedup_entry(usize::MAX as u128 + 5, 9, 91), + ], + } + } + + fn dedup_entry(client: u128, watermark: u64, latest_commit: u64) -> DedupWatermark { + DedupWatermark { + client, + user_id: 1, + watermark, + latest_commit, + committed_window: 0b1011, } } @@ -525,6 +654,7 @@ mod tests { next_offset: 0, consumers: Vec::new(), groups: Vec::new(), + dedup: Vec::new(), }; let encoded = empty.encode(); assert_eq!( @@ -584,6 +714,126 @@ mod tests { ); } + #[test] + fn given_unordered_dedup_clients_when_decoded_should_reject() { + let unordered = ConsumerOffsetsWire { + purge_generation: 0, + next_offset: 0, + consumers: Vec::new(), + groups: Vec::new(), + dedup: vec![dedup_entry(9, 1, 1), dedup_entry(4, 2, 2)], + }; + assert_eq!( + ConsumerOffsetsWire::decode(&unordered.encode()), + Err(ConsumerOffsetsWireError::NonAscendingClient { client: 4 }) + ); + } + + #[test] + fn given_reserved_client_in_dedup_when_decoded_should_reject() { + let reserved = ConsumerOffsetsWire { + purge_generation: 0, + next_offset: 0, + consumers: Vec::new(), + groups: Vec::new(), + dedup: vec![dedup_entry(0, 1, 1), dedup_entry(4, 2, 2)], + }; + assert_eq!( + ConsumerOffsetsWire::decode(&reserved.encode()), + Err(ConsumerOffsetsWireError::ReservedClient) + ); + } + + #[test] + fn given_v1_artifact_when_decoded_should_install_empty_dedup() { + // An un-upgraded primary still ships "ICO1": same fields minus the + // dedup count and section. It must decode, with nothing to absorb. + let mut bytes = Vec::new(); + bytes.extend_from_slice(&CONSUMER_OFFSETS_MAGIC_V1); + bytes.push(CONSUMER_OFFSETS_VERSION_V1); + bytes.extend_from_slice(&3u64.to_le_bytes()); + bytes.extend_from_slice(&43u64.to_le_bytes()); + bytes.extend_from_slice(&1u32.to_le_bytes()); + bytes.extend_from_slice(&1u32.to_le_bytes()); + for (id, offset) in [(7u32, 42u64), (2, 5)] { + bytes.extend_from_slice(&id.to_le_bytes()); + bytes.extend_from_slice(&offset.to_le_bytes()); + } + let trailer = state_artifact_checksum(&bytes); + bytes.extend_from_slice(&trailer.to_le_bytes()); + assert_eq!( + ConsumerOffsetsWire::decode(&bytes), + Ok(ConsumerOffsetsWire { + purge_generation: 3, + next_offset: 43, + consumers: vec![(7, 42)], + groups: vec![(2, 5)], + dedup: Vec::new(), + }) + ); + } + + #[test] + fn given_v1_magic_with_wrong_version_when_decoded_should_reject() { + let mut bytes = Vec::new(); + bytes.extend_from_slice(&CONSUMER_OFFSETS_MAGIC_V1); + bytes.push(CONSUMER_OFFSETS_VERSION); + bytes.extend_from_slice(&0u64.to_le_bytes()); + bytes.extend_from_slice(&0u64.to_le_bytes()); + bytes.extend_from_slice(&0u32.to_le_bytes()); + bytes.extend_from_slice(&0u32.to_le_bytes()); + let trailer = state_artifact_checksum(&bytes); + bytes.extend_from_slice(&trailer.to_le_bytes()); + assert_eq!( + ConsumerOffsetsWire::decode(&bytes), + Err(ConsumerOffsetsWireError::UnsupportedVersion { + version: CONSUMER_OFFSETS_VERSION + }) + ); + } + + #[test] + fn given_dedup_count_past_ceiling_when_decoded_should_reject_before_allocating() { + let mut bytes = Vec::new(); + bytes.extend_from_slice(&CONSUMER_OFFSETS_MAGIC); + bytes.push(CONSUMER_OFFSETS_VERSION); + bytes.extend_from_slice(&0u64.to_le_bytes()); + bytes.extend_from_slice(&0u64.to_le_bytes()); + bytes.extend_from_slice(&0u32.to_le_bytes()); + bytes.extend_from_slice(&0u32.to_le_bytes()); + bytes.extend_from_slice(&(CONSUMER_OFFSETS_ENTRIES_MAX + 1).to_le_bytes()); + let trailer = state_artifact_checksum(&bytes); + bytes.extend_from_slice(&trailer.to_le_bytes()); + assert_eq!( + ConsumerOffsetsWire::decode(&bytes), + Err(ConsumerOffsetsWireError::TooManyEntries { + section: "dedup", + count: CONSUMER_OFFSETS_ENTRIES_MAX + 1, + max: CONSUMER_OFFSETS_ENTRIES_MAX, + }) + ); + } + + #[test] + fn given_dedup_count_exceeding_bytes_when_decoded_should_reject_as_truncated() { + // Under the ceiling but past the bytes present: the stride guard is + // what stops a ~30 byte artifact reserving megabytes. + let mut bytes = Vec::new(); + bytes.extend_from_slice(&CONSUMER_OFFSETS_MAGIC); + bytes.push(CONSUMER_OFFSETS_VERSION); + bytes.extend_from_slice(&0u64.to_le_bytes()); + bytes.extend_from_slice(&0u64.to_le_bytes()); + bytes.extend_from_slice(&0u32.to_le_bytes()); + bytes.extend_from_slice(&0u32.to_le_bytes()); + bytes.extend_from_slice(&1_000u32.to_le_bytes()); + let trailer = state_artifact_checksum(&bytes); + bytes.extend_from_slice(&trailer.to_le_bytes()); + assert_eq!( + ConsumerOffsetsWire::decode(&bytes), + Err(ConsumerOffsetsWireError::Truncated) + ); + } + #[test] fn given_count_past_ceiling_when_decoded_should_reject_before_allocating() { let mut bytes = Vec::new(); @@ -593,6 +843,7 @@ mod tests { bytes.extend_from_slice(&0u64.to_le_bytes()); bytes.extend_from_slice(&(CONSUMER_OFFSETS_ENTRIES_MAX + 1).to_le_bytes()); bytes.extend_from_slice(&0u32.to_le_bytes()); + bytes.extend_from_slice(&0u32.to_le_bytes()); let trailer = state_artifact_checksum(&bytes); bytes.extend_from_slice(&trailer.to_le_bytes()); assert_eq!( @@ -612,6 +863,7 @@ mod tests { next_offset: 0, consumers: vec![(5, 1), (5, 2)], groups: Vec::new(), + dedup: Vec::new(), }; assert_eq!( ConsumerOffsetsWire::decode(&duplicate.encode()), @@ -625,6 +877,7 @@ mod tests { next_offset: 0, consumers: Vec::new(), groups: vec![(9, 1), (4, 2)], + dedup: Vec::new(), }; assert_eq!( ConsumerOffsetsWire::decode(&unordered.encode()), @@ -1704,11 +1957,13 @@ where // sealed segment while the counter stands at N, and the receiver // must resume minting at N either way. let next_offset = self.offset_frontier(); + let dedup = self.dedup().watermarks_sorted(); ConsumerOffsetsWire { purge_generation: self.applied_purge_generation, next_offset, consumers, groups, + dedup, } } @@ -2460,6 +2715,13 @@ where // file put a rejoin carrying thousands of consumers on the pump for // thousands of sequential open + write + optional fsync round trips; // the tick's superblock pre-pass sets the precedent for the width. + // The dedup slice is memory-only, so it installs here with the maps + // rather than being written anywhere. No frontier fence is needed: the + // install lifts `commit_min` to the offer's `commit_op`, so the commit + // walk that follows starts strictly above everything this artifact + // covers, and `record_commit` is idempotent besides. + self.dedup_mut() + .install_watermarks(offsets_wire.dedup.iter().copied()); let mut planned: Vec = Vec::with_capacity(offsets_wire.consumers.len() + offsets_wire.groups.len()); if let Some(dir) = self.consumer_offsets_path.clone() { @@ -2682,6 +2944,10 @@ where // promise rested on the caller clearing it first. self.segment_checksum_cache.borrow_mut().clear(); self.reuse_scan_memo.borrow_mut().take(); + // Degrade to at-least-once rather than keep watermarks that may now + // describe data this partition no longer holds: a stale entry would + // absorb a replay whose original was just unlinked. + self.dedup_mut().install_watermarks(std::iter::empty()); // Sweep EVERY segment file, not the in-memory count's worth: after // a late failure the renamed-in new chain is on disk while the diff --git a/core/server/config.toml b/core/server/config.toml index 79bdc46e88..5584b03788 100644 --- a/core/server/config.toml +++ b/core/server/config.toml @@ -985,6 +985,17 @@ clients_table_max = 8192 # u128 bitset, and this depth bounds that suffix. prepare_queue_depth = 32 +# Distinct clients each partition group tracks request watermarks for, so a +# retried produce or consumer-offset write is answered instead of committing a +# second time. At capacity the client whose newest commit is oldest is evicted; +# that client's next replay re-executes, exactly as it would have before dedup +# existed, so under-sizing degrades rather than breaks. Must be > 0 and <= 65536. +# +# Unlike [metadata] clients_table_max, this budget is PER GROUP, so worst-case +# memory scales with partition count. Size it to the producers one partition +# actually sees, not the node's client total. +dedup_clients_max = 4096 + # Entries the evicted ring retains per multi-replica partition for journal # repair after a peer rejoins. Larger widens the window a restarting peer can be # served from the ring before falling back to bulk sync, at the cost of pinned diff --git a/core/server/src/boot/recovery.rs b/core/server/src/boot/recovery.rs index 0115b5c609..e614e688f5 100644 --- a/core/server/src/boot/recovery.rs +++ b/core/server/src/boot/recovery.rs @@ -291,6 +291,9 @@ const _: () = assert!( configs::partition::DEFAULT_PARTITION_PREPARE_QUEUE_DEPTH == consensus::PIPELINE_PREPARE_QUEUE_MAX ); +const _: () = assert!( + configs::partition::PARTITION_DEDUP_CLIENTS_DEFAULT == consensus::PARTITION_DEDUP_CLIENTS_MAX +); const _: () = assert!(configs::metadata::DEFAULT_METADATA_CLIENTS_TABLE_MAX == consensus::CLIENTS_TABLE_MAX); const _: () = diff --git a/core/server/src/dispatch/partition.rs b/core/server/src/dispatch/partition.rs index 4da515f8e5..b2e0789bfa 100644 --- a/core/server/src/dispatch/partition.rs +++ b/core/server/src/dispatch/partition.rs @@ -319,9 +319,9 @@ fn build_auto_commit_request( operation: Operation::StoreConsumerOffset, size, client: AUTO_COMMIT_CLIENT_ID, - // The partition plane is sessionless (no `ClientTable` dedup); a - // nonzero session + request just satisfy the wire header - // validation. + // The reserved sentinel client is never deduped and never + // replied to; a nonzero session + request just satisfy the wire + // header validation. session: 1, request: 1, group: namespace.inner(), @@ -334,10 +334,18 @@ fn build_auto_commit_request( /// Route a partition data-plane op (`SendMessages` / consumer-offset writes) /// through the shard mesh by namespace: the op belongs to the partition's /// own consensus group, not the metadata group. The owning shard's -/// partitions plane runs at-least-once consensus and replies directly via -/// `send_to_client`. `header.client` therefore stays the TRANSPORT id -/// (home-shard routing bits), not the VSR session id -- partition ops are -/// sessionless ("session lifecycle is metadata-only"). +/// partitions plane dedups the request against its group's slice, so +/// `header.client` carries the VSR consensus id (the dedup key) rather than +/// the transport id. +/// +/// How the committed reply gets back depends on whether the bus can route that +/// id. HTTP registers each session under its own shard-0 transport id, so the +/// two are equal and the plane's `send_to_client` fires the session's +/// in-process reply slot directly; the request is dispatched and forgotten +/// here (`?ack=none` relies on exactly that: nothing listening, reply shed at +/// the bus). Every other transport registers under a client-chosen id the bus +/// cannot route, so the request is submitted with an in-process channel and +/// the reply is relayed to the socket this shard holds. /// /// Callers must have authenticated the transport already: `vsr_client_id` / /// `bound_session` come from its bound VSR session. Every failure before @@ -475,17 +483,98 @@ pub async fn dispatch_partition_request( let request = request.transmute_header(|header, new_header: &mut RoutedRequestHeader| { *new_header = header; new_header.group = namespace; - new_header.client = transport_client_id; - // Header validation requires `session > 0 && request > 0` for - // non-register ops. The partition plane itself is sessionless - // (at-least-once, no `ClientTable` dedup), so the bound VSR - // session merely satisfies validation. Current SDKs do number - // partition ops, but older and internal callers may still send - // zero, so a zero id is normalized to the compatibility value 1. + // The VSR consensus id, exactly as metadata ops are stamped: it is the + // dedup key every replica keys its slice by, and unlike the transport + // id it stays valid across nodes. Replies therefore cannot be routed + // by this field -- they ride the submit's channel back to this shard, + // which owns the socket. + new_header.client = vsr_client_id; + // Header validation requires `session > 0` for non-register ops. The + // slices mint no epoch of their own, so the bound VSR session only + // satisfies validation here. new_header.session = bound_session; - new_header.request = new_header.request.max(1); + // The session's owner as this shard resolved it, never the + // client-supplied value: the dedup slice keys on it to tell a re-minted + // id's next holder apart from its previous one, so it has to be + // trustworthy on every replica. Same rule as the metadata plane. The + // gate above failed closed on `None`, so `0` (unattributed) is only a + // type-level fallback here. + new_header.user_id = acting_user_id.unwrap_or(0); + }); + if vsr_client_id == transport_client_id { + shard.dispatch(request.into_generic()); + return; + } + relay_partition_reply( + shard, + IggyNamespace::from_raw(namespace), + request, + transport_client_id, + &header, + ) + .await; +} + +/// Submit a partition write whose reply the bus cannot route (the routed +/// header's `client` is the VSR consensus id, whose bits encode no home shard) +/// and relay the committed reply to the socket this shard holds. +/// +/// Only the submit runs on the caller's drain loop: it is what fixes the order +/// two writes from one connection reach the owning shard in. The wait for the +/// commit is spawned, so a connection keeps draining its queued polls and +/// metadata ops instead of holding them behind one replication round trip. +#[allow(clippy::future_not_send)] +async fn relay_partition_reply( + shard: &Rc>, + namespace: IggyNamespace, + request: Message, + transport_client_id: u128, + header: &RoutedRequestHeader, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let Ok(ticket) = shard.partition_submit(namespace, request) else { + // `PartitionSubmitRefused`: the frame never reached the owning shard, + // so this is a known outcome and the client can be told now rather + // than after its read-timeout. Same transient the plane itself answers + // for a request it could not admit. + send_deny_reply( + shard, + transport_client_id, + header, + IggyError::TransientNotAccepted.as_code(), + ) + .await; + return; + }; + let operation = header.operation; + let shard = Rc::clone(shard); + // Through the bus, not the runtime directly: the simulator supplies its + // own executor and virtual clock. + shard.bus.clone().spawn(async move { + let Some(reply) = shard.await_partition_submit(ticket).await else { + // Abandoned or expired. Deliberately silent: the outcome is + // unknown, and a synthesized failure could contradict a write that + // commits moments later. The client's read-timeout is the recovery. + return; + }; + if let Err(error) = shard + .bus + .send_to_client(transport_client_id, reply.into_frozen()) + .await + { + warn!( + transport_client_id, + operation = ?operation, + error = %error, + "failed to forward committed partition reply to its socket" + ); + } }); - shard.dispatch(request.into_generic()); } /// Serve `poll_messages`: resolve the partition namespace, run the read on diff --git a/core/server/src/http/session.rs b/core/server/src/http/session.rs index aa1db608c3..b03b94bf58 100644 --- a/core/server/src/http/session.rs +++ b/core/server/src/http/session.rs @@ -104,12 +104,18 @@ pub(in crate::http) struct HttpSession { /// a larger one would arrive at or below the watermark and be refused as a /// duplicate. pub(in crate::http) gate: Mutex, - /// Next data-plane request id. A separate, gate-free counter: partition ops - /// are at-least-once with no consensus dedup, so the id only correlates the - /// in-process reply slot and concurrent produces on one session are legal. - /// A plain `Cell` suffices on single-threaded shard 0; ids are minted - /// monotonically and never reused, which the slot-guard contract requires. - pub(in crate::http) data_request: Cell, + /// Serializes this session's data-plane writes the way `gate` does its + /// metadata writes: the guarded value is the NEXT request id, and the write + /// path holds the lock from the mint until the request has been handed to + /// the owning shard's inbox. The partition slice dedups on a per-client + /// watermark, so two handlers that minted in one order but reached the + /// shard in the other (one slept in the routable wait, say) would have the + /// lower id absorbed as a duplicate with a success status. Concurrent + /// awaits stay legal: the lock covers admission, not the commit round trip. + /// Shared with the `?ack=none` path so a shed reply's id never collides + /// with a live awaited slot on this session. Ids are minted monotonically + /// and never reused, which the slot-guard contract requires. + pub(in crate::http) data_gate: Mutex, /// Registry token of this session's lazily-installed in-process reply /// target (`None` until the first awaited partition write). Stored so /// session eviction can tear the registry entry down fenced by the same @@ -121,17 +127,6 @@ pub(in crate::http) struct HttpSession { pub(in crate::http) in_flight_writes: Cell, } -impl HttpSession { - /// Mint the next data-plane request id. Also consumed by the `?ack=none` - /// path, which installs no slot: sharing one counter keeps a shed reply's - /// id from ever colliding with a live awaited slot on this session. - pub(in crate::http) fn next_data_request_id(&self) -> u64 { - let id = self.data_request.get(); - self.data_request.set(id + 1); - id - } -} - /// Serializes first-use VSR registration per credential key so a herd of /// concurrent first-requests for one token runs exactly one `Register` instead /// of N that each mint a client id and orphan N-1 slots last-writer-wins. @@ -271,7 +266,7 @@ mod tests { user_id: DEFAULT_ROOT_USER_ID, expiry: u64::MAX, gate: Mutex::new(FIRST_REQUEST_ID), - data_request: Cell::new(FIRST_REQUEST_ID), + data_gate: Mutex::new(FIRST_REQUEST_ID), registry_token: Cell::new(None), in_flight_writes: Cell::new(0), }); @@ -303,7 +298,7 @@ mod tests { user_id: DEFAULT_ROOT_USER_ID, expiry, gate: Mutex::new(FIRST_REQUEST_ID), - data_request: Cell::new(FIRST_REQUEST_ID), + data_gate: Mutex::new(FIRST_REQUEST_ID), registry_token: Cell::new(None), in_flight_writes: Cell::new(0), }) diff --git a/core/server/src/http/state.rs b/core/server/src/http/state.rs index 73b2c6cb7b..5f1984cd41 100644 --- a/core/server/src/http/state.rs +++ b/core/server/src/http/state.rs @@ -367,7 +367,7 @@ impl HttpInner { user_id, expiry, gate: Mutex::new(FIRST_REQUEST_ID), - data_request: Cell::new(FIRST_REQUEST_ID), + data_gate: Mutex::new(FIRST_REQUEST_ID), registry_token: Cell::new(None), in_flight_writes: Cell::new(0), })) diff --git a/core/server/src/http/submit.rs b/core/server/src/http/submit.rs index d73b2005ee..0babb39528 100644 --- a/core/server/src/http/submit.rs +++ b/core/server/src/http/submit.rs @@ -377,7 +377,13 @@ pub(in crate::http) async fn partition_write_replicated( // timeout. Held across every exit below; released by `Drop`. let _in_flight = admit_partition_write(&session.in_flight_writes, &state.in_flight_writes)?; ensure_in_process_reply_target(state, session); - let request_id = session.next_data_request_id(); + // Held from the mint until `dispatch_partition_request` returns, which is + // past the owning shard's inbox: the ids of this session's writes must + // reach the partition in mint order or the watermark absorbs the overtaken + // one. Released before the commit wait so writes still overlap there. + let mut next_data_request_id = session.data_gate.lock().await; + let request_id = *next_data_request_id; + *next_data_request_id += 1; let message = build_request_message( operation, session.client_id, @@ -407,6 +413,7 @@ pub(in crate::http) async fn partition_write_replicated( Some(session.user_id), ) .await; + drop(next_data_request_id); let outcome = compio::time::timeout(PARTITION_WRITE_REPLY_TIMEOUT, receiver).await; // Removes the slot unless the reply already fired, so a late commit // reply after a timeout sheds at the bus instead of leaking a waiter. @@ -442,7 +449,10 @@ pub(in crate::http) async fn produce_unacked( body: &[u8], ) -> Result<(), PartitionWriteError> { let _in_flight = admit_partition_write(&session.in_flight_writes, &state.in_flight_writes)?; - let request_id = session.next_data_request_id(); + // Same gate as the acked path, for the same ordering reason. + let mut next_data_request_id = session.data_gate.lock().await; + let request_id = *next_data_request_id; + *next_data_request_id += 1; let message = build_request_message( Operation::SendMessages, session.client_id, @@ -459,6 +469,7 @@ pub(in crate::http) async fn produce_unacked( Some(session.user_id), ) .await; + drop(next_data_request_id); Ok(()) } diff --git a/core/server/src/partition_helpers.rs b/core/server/src/partition_helpers.rs index ad4471da98..ec39868e5c 100644 --- a/core/server/src/partition_helpers.rs +++ b/core/server/src/partition_helpers.rs @@ -797,6 +797,7 @@ async fn load_partition( config.partition.evicted_ring_capacity, config.partition.evicted_ring_bytes_max.as_bytes_u64(), ); + partition.set_dedup_clients_max(config.partition.dedup_clients_max); partition.set_partition_dir(partition_dir.clone()); // Before the hydrate: the durable record is keyed by incarnation, so a // `purge.gen` left behind by a previous life of this namespace reads 0. @@ -1210,6 +1211,7 @@ pub async fn build_partition_fresh( config.partition.evicted_ring_capacity, config.partition.evicted_ring_bytes_max.as_bytes_u64(), ); + partition.set_dedup_clients_max(config.partition.dedup_clients_max); partition.set_partition_dir(partition_dir); // Fresh dirs read generation 0; a dir surviving from a crashed process // (this "fresh" build races repair re-materialization) reads the last diff --git a/core/shard/src/lib.rs b/core/shard/src/lib.rs index 8b56a415a4..fe4deaea5a 100644 --- a/core/shard/src/lib.rs +++ b/core/shard/src/lib.rs @@ -402,6 +402,27 @@ pub type PartitionReadHandler = /// deadline. const PARTITION_READ_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(10); +/// Budget for a partition write's wait on its committed reply. Longer than a +/// read: the wait spans replication quorum plus any park-and-promote the +/// request rides through, and a view change mid-flight re-proposes under the +/// new primary. Expiry leaves the client to its own read-timeout. +const PARTITION_SUBMIT_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(30); + +/// A partition write admitted onto its owning shard's inbox, awaiting the +/// committed reply. Redeem with [`IggyShard::await_partition_submit`]. +pub struct PartitionSubmitTicket { + receiver: Receiver>>, + target: u16, +} + +/// The write never reached the owning shard. +/// +/// No sender existed for the target, or its inbox refused the frame. Either +/// way the outcome is known, unlike a reply that fails to arrive, so the caller +/// may deny the client outright. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct PartitionSubmitRefused; + /// Race `future` against a bus timer. /// /// `Some` if it finishes within `budget`, `None` if the timer fires first. @@ -703,6 +724,17 @@ pub enum LifecycleFrame { read: PartitionRead, reply: Sender, }, + /// Admit a partition write (`SendMessages` / consumer-offset write) on + /// the shard owning its namespace, carrying the channel its committed + /// reply travels back on. The partition plane cannot route a reply by + /// `header.client` -- that field is the VSR consensus id, whose bits + /// carry no home-shard routing -- so the reply returns to the + /// connection-owning shard, which writes it to the socket it holds. + /// See [`IggyShard::partition_submit`]. + PartitionSubmit { + request: Message, + reply: Sender>>, + }, /// Shard 0 broadcasts after a partition-shaped metadata commit; wakes /// the per-shard reconciler. No payload: reconciler re-reads target /// state. Drops covered by the periodic safety tick. @@ -1429,6 +1461,11 @@ where /// from. Cleared when the park map empties, so one episode warns once. shard_park_shedding: Cell, + /// Set once a partition submit has waited out its budget and warned; + /// cleared by the next reply that arrives. Gates the timeout warning to + /// one line per stall episode (see [`Self::await_partition_submit`]). + partition_submit_stalled: Cell, + /// Live ceiling on prepares served per `RequestPrepares` round. Defaults /// to [`REPAIR_CHUNK_MAX`]; the server overrides it from /// `[cluster] repair_chunk_max` at bootstrap. @@ -1607,6 +1644,7 @@ where parked_partition_bytes: Cell::new(0), redispatch_queue: RefCell::new(VecDeque::new()), shard_park_shedding: Cell::new(false), + partition_submit_stalled: Cell::new(false), metadata_repair: RefCell::new(None), metadata_transfer: RefCell::new(None), state_transfer_offers: RefCell::new(HashMap::new()), @@ -1872,6 +1910,121 @@ where } } + /// Admit a partition write on the shard owning `namespace`. Routes through + /// the shards table exactly like [`Self::partition_read`], self-sends + /// included, so a locally-owned partition takes the same path. + /// + /// Synchronous up to the inbox `try_send`, so two writes a caller admits + /// back to back reach the owning shard in that order; the committed reply + /// is awaited separately through [`Self::await_partition_submit`], which a + /// connection's drain loop spawns rather than blocks on. + /// + /// # Errors + /// [`PartitionSubmitRefused`] when the frame provably never reached the + /// owning shard (no sender for the target, or a full inbox), so the caller + /// can deny the client outright instead of leaving it to a read-timeout + /// for an outcome that is already known. + pub fn partition_submit( + &self, + namespace: IggyNamespace, + request: Message, + ) -> Result { + let target = self.shards_table.shard_for(namespace).unwrap_or_else(|| { + // Same fallback as `route_typed`: a miss means "not seeded yet", + // not "unroutable", and the owning shard parks what arrives early. + crate::shards_table::calculate_shard_from_consensus_ns( + namespace.inner(), + self.shard_count, + ) + }); + let (reply_tx, reply_rx) = channel::>>(1); + let frame = ShardFrame::lifecycle(LifecycleFrame::PartitionSubmit { + request, + reply: reply_tx, + }); + let Some(sender) = self.senders.get(target as usize) else { + self.metrics.record_frame_drop( + crate::metrics::frame_drop_variant::PARTITION, + crate::metrics::frame_drop_reason::UNROUTABLE, + ); + return Err(PartitionSubmitRefused); + }; + if let Err(error) = sender.try_send(frame) { + self.metrics.record_frame_drop( + crate::metrics::frame_drop_variant::PARTITION, + crate::coordinator::classify_try_send_err(&error), + ); + tracing::warn!( + shard = self.id, + target, + "partition_submit: inbox rejected PartitionSubmit frame: {error:?}" + ); + return Err(PartitionSubmitRefused); + } + Ok(PartitionSubmitTicket { + receiver: reply_rx, + target, + }) + } + + /// Wait out a submitted write's committed reply. + /// + /// `None` = reply channel dropped before a reply (view-change reset, park + /// eviction, shutdown) or budget expiry. The caller stays silent on `None`: + /// the outcome is unknown, so a synthesized failure could contradict a + /// write that commits moments later, and the client's own read-timeout is + /// the recovery. Both exits count under + /// `frame_drops_total{variant=partition}` with their own reasons. The + /// timeout warning fires once per stall episode, reset by the next reply + /// that does arrive: one wedged group would otherwise log a line per + /// request, and the counter carries the volume. + #[allow(clippy::future_not_send)] + pub async fn await_partition_submit( + &self, + ticket: PartitionSubmitTicket, + ) -> Option> { + let PartitionSubmitTicket { receiver, target } = ticket; + match bus_timeout(&self.bus, PARTITION_SUBMIT_TIMEOUT, receiver.recv()).await { + Some(Ok(Some(reply))) => { + self.partition_submit_stalled.set(false); + Some(reply) + } + Some(Ok(None) | Err(_)) => { + self.metrics.record_frame_drop( + crate::metrics::frame_drop_variant::PARTITION, + crate::metrics::frame_drop_reason::SUBMIT_ABANDONED, + ); + tracing::debug!( + shard = self.id, + target, + "partition_submit: reply channel dropped before commit" + ); + None + } + None => { + self.metrics.record_frame_drop( + crate::metrics::frame_drop_variant::PARTITION, + crate::metrics::frame_drop_reason::SUBMIT_TIMEOUT, + ); + if self.partition_submit_stalled.replace(true) { + tracing::debug!( + shard = self.id, + target, + "partition_submit: owning shard did not reply within budget" + ); + } else { + tracing::warn!( + shard = self.id, + target, + "partition_submit: owning shard did not reply within budget; \ + further expiries log at debug until a reply arrives" + ); + } + None + } + } + } + /// Return a clone of the shard-0 coordinator handle, if attached. /// Bootstrap uses this to wire the listener accept callbacks /// (replica + client) to coordinator-driven fd-delegation instead @@ -1937,6 +2090,7 @@ where parked_partition_bytes: Cell::new(0), redispatch_queue: RefCell::new(VecDeque::new()), shard_park_shedding: Cell::new(false), + partition_submit_stalled: Cell::new(false), metadata_repair: RefCell::new(None), metadata_transfer: RefCell::new(None), state_transfer_offers: RefCell::new(HashMap::new()), @@ -2413,6 +2567,12 @@ struct ParkedFrame { /// the outcome from a reply rather than a timeout. passes: u32, message: Message, + /// Channel the committed reply travels back on, for a frame that arrived + /// as a [`LifecycleFrame::PartitionSubmit`]. `None` for replicated + /// prepares and for writes admitted without a waiter. Dropping the frame + /// (expiry, teardown, shutdown) drops this, which wakes the awaiting + /// dispatch with a receive error it maps to silence. + reply: Option>>>, } impl ParkedFrame { @@ -2426,6 +2586,17 @@ impl ParkedFrame { } } +/// One staged frame classified for re-delivery. +/// +/// `reply` is the submit channel the frame parked with, if any: it decides which +/// admission path the pump re-enters, since a submit's committed reply cannot be +/// routed by `header.client` (that field is the VSR consensus id). +struct RedispatchedFrame { + message: MessageBag, + provenance: ParkProvenance, + reply: Option>>>, +} + /// What a frame keeps if it parks again after the pump re-delivers it. /// /// Production prevents that race by ranking redispatch above inbox work and by @@ -2573,12 +2744,13 @@ where /// arm. The queue borrow ends before dispatch awaits, so simulator /// materialisation can append off-pump without colliding with a suspended /// `RefCell` guard. - fn pop_redispatched_frame(&self) -> Option<(MessageBag, ParkProvenance)> { + fn pop_redispatched_frame(&self) -> Option { loop { let ParkedFrame { epoch, passes, message, + reply, } = self.redispatch_queue.borrow_mut().pop_front()?; let provenance = ParkProvenance { epoch, passes }; // Parked frames are stored generic (the buffer holds every variant @@ -2586,7 +2758,13 @@ where // the rare path - a post-`CreateTopic` convergence window, not the // per-message steady state the bag handoff exists for. match MessageBag::try_from(message) { - Ok(bag) => return Some((bag, provenance)), + Ok(message) => { + return Some(RedispatchedFrame { + message, + provenance, + reply, + }); + } Err(error) => { // The frame classified once already, on the way in, so this // is unreachable short of memory corruption. The consumed @@ -2607,6 +2785,42 @@ where } } + /// Deliver one staged frame, through the admission path it arrived on. + #[allow(clippy::future_not_send)] + pub(crate) async fn dispatch_redispatched_frame(&self, frame: RedispatchedFrame) + where + B: MessageBus + 'static, + MJ: JournalHandle, + ::Target: + Journal, Header = PrepareHeader>, + M: RestorableMetadataStm, + T: ShardsTable, + { + let RedispatchedFrame { + message, + provenance, + reply, + } = frame; + match (reply, message) { + (Some(reply), MessageBag::Request(request)) => { + self.dispatch_partition_submit(request, reply, Some(provenance)) + .await; + } + (Some(_), _) => { + // Only a client request parks with a waiter attached, so this is + // unreachable short of a classify that disagrees with the one + // the frame passed on the way in. Dropping the sender wakes the + // awaiting shard, which maps the receive error to silence. + tracing::error!( + shard = self.id, + "staged partition frame carries a reply channel but is not a client request; \ + dropping it" + ); + } + (None, message) => self.dispatch_message(message, Some(provenance)).await, + } + } + /// Test-only delivery of one staged frame. Production obtains frames through /// the router's ranked select arm, which also processes loopback after each /// one. This hook exists for the reconciler's defence-in-depth interleaving. @@ -2621,10 +2835,10 @@ where M: RestorableMetadataStm, T: ShardsTable, { - let Some((message, provenance)) = self.pop_redispatched_frame() else { + let Some(frame) = self.pop_redispatched_frame() else { return false; }; - self.dispatch_message(message, Some(provenance)).await; + self.dispatch_redispatched_frame(frame).await; true } @@ -2665,7 +2879,9 @@ where let header = request.header(); (header.operation, header.group) }; - match self.park_if_unmaterialised(request, routing.0, routing.1, provenance) { + match self + .park_if_unmaterialised(request, routing.0, routing.1, provenance, &mut None) + { // The incarnation fence runs only here, on client traffic. // A backup denying what the primary admitted would diverge // the replicas, so replicated frames are never fenced. @@ -2696,7 +2912,9 @@ where // A tombstoned prepare still flows to the plane: replicated // traffic has no client awaiting a reply on this node, and // the plane's own tombstone guard drops it. - match self.park_if_unmaterialised(prepare, routing.0, routing.1, provenance) { + match self + .park_if_unmaterialised(prepare, routing.0, routing.1, provenance, &mut None) + { ParkOutcome::Deliver(prepare) | ParkOutcome::Tombstoned(prepare) => { self.on_replicate(prepare).await; // A follower learns the cluster commit point from the @@ -3129,12 +3347,21 @@ where /// repair must fill. The `false` return is what makes callers bump /// `frame_drops_total{variant=partition,reason=park_dropped}`. fn deny_parked_client_request(&self, frame: ParkedFrame) -> bool { - if frame.message.header().command == Command::Request - && let Ok(request) = frame.message.try_into_typed::() - { - return self.stage_transient_deny(request.header()); + let ParkedFrame { message, reply, .. } = frame; + if message.header().command != Command::Request { + return false; } - false + let Ok(request) = message.try_into_typed::() else { + return false; + }; + // A submit cannot be answered through `stage_transient_deny`: it routes + // by `header.client`, which on a partition request is the VSR consensus + // id and addresses no connection. Its own channel reaches the shard + // holding the socket. + if reply.is_some() { + return Self::answer_partition_submit_transient(request.header(), reply); + } + self.stage_transient_deny(request.header()) } /// A parked frame addressed an incarnation this shard no longer holds. @@ -3204,6 +3431,7 @@ where operation: Operation, namespace_raw: u64, provenance: Option, + reply: &mut Option>>>, ) -> ParkOutcome where H: iggy_binary_protocol::ConsensusHeader, @@ -3331,6 +3559,7 @@ where epoch, passes, message: message.into_generic(), + reply: reply.take(), }); drop(pending); self.parked_partition_bytes @@ -3413,6 +3642,106 @@ where true } + /// Admit a `PartitionSubmit`: same gates as the [`MessageBag::Request`] + /// arm, but every refusal answers on `reply` instead of the bus, and the + /// admitted request carries an in-process reply channel down to the + /// pipeline entry so its committed reply comes back here rather than + /// being routed by `header.client`. + #[allow(clippy::future_not_send)] + pub async fn on_partition_submit( + &self, + request: Message, + reply: Sender>>, + ) where + B: MessageBus + 'static, + MJ: JournalHandle, + ::Target: + Journal, Header = PrepareHeader>, + M: RestorableMetadataStm, + T: ShardsTable, + { + self.dispatch_partition_submit(request, reply, None).await; + } + + /// [`Self::on_partition_submit`] carrying the park provenance of a submit + /// the pump is re-delivering, for the same reason + /// [`Self::dispatch_message`] carries it. + #[allow(clippy::future_not_send)] + async fn dispatch_partition_submit( + &self, + request: Message, + reply: Sender>>, + provenance: Option, + ) where + B: MessageBus + 'static, + MJ: JournalHandle, + ::Target: + Journal, Header = PrepareHeader>, + M: RestorableMetadataStm, + T: ShardsTable, + { + let routing = { + let header = request.header(); + (header.operation, header.group) + }; + // The frame takes a clone of the sender only when it parks; every other + // outcome answers on the original below, so no arm can lose the waiter + // to a `None` it would have to guard against. The clone is an `Rc` + // bump, and whichever half is not used drops with this scope. + match self.park_if_unmaterialised( + request, + routing.0, + routing.1, + provenance, + &mut Some(reply.clone()), + ) { + ParkOutcome::Deliver(request) + if !self.serves_committed_incarnation(routing.0, routing.1) => + { + Self::answer_partition_submit_transient(request.header(), Some(reply)); + } + ParkOutcome::Deliver(request) => { + let (sender, receiver) = consensus::oneshot_channel(); + self.plane + .partitions() + .on_request_with_reply(request, Some(sender)) + .await; + // Await OFF the pump: the commit that fires this receiver needs + // the pump to keep draining acks, so blocking here would + // deadlock the very reply being waited on. The task holds only + // owned channel halves, never a partitions borrow. + // + // Through the bus, not the runtime directly: the simulator + // supplies its own executor and virtual clock. + self.bus.spawn(async move { + let committed = receiver.await.ok().map(Message::into_generic); + let _ = reply.try_send(committed); + }); + } + ParkOutcome::Tombstoned(request) | ParkOutcome::Overflow(request) => { + Self::answer_partition_submit_transient(request.header(), Some(reply)); + } + // The clone travelled with the parked frame; it answers on drain + // or wakes the awaiter with a receive error when the frame expires. + ParkOutcome::Parked => {} + } + } + + /// Answer a refused `PartitionSubmit` with the same transient deny the bus + /// path sends, over the submit's own channel. `false` = nobody was + /// answered, so the frame still counts as dropped. + fn answer_partition_submit_transient( + request_header: &RoutedRequestHeader, + reply: Option>>>, + ) -> bool { + let Some(reply) = reply else { return false }; + let deny = build_deny_reply_from_request_header( + request_header, + IggyError::TransientNotAccepted.as_code(), + ); + reply.try_send(Some(deny.into_generic())).is_ok() + } + #[allow(clippy::future_not_send)] pub async fn on_request(&self, request: Message) where diff --git a/core/shard/src/metrics.rs b/core/shard/src/metrics.rs index 16007ffec1..4fbd8c7fcf 100644 --- a/core/shard/src/metrics.rs +++ b/core/shard/src/metrics.rs @@ -132,6 +132,12 @@ pub mod frame_drop_reason { pub const MISROUTED: &str = "misrouted"; pub const PARK_OVERFLOW: &str = "park_overflow"; pub const PARK_DROPPED: &str = "park_dropped"; + /// A partition write's reply channel was dropped before a reply arrived + /// (view-change pipeline reset, park teardown, shutdown): the outcome is + /// unknown and the client is left to its read-timeout. + pub const SUBMIT_ABANDONED: &str = "submit_abandoned"; + /// A partition write's reply did not arrive within the submit budget. + pub const SUBMIT_TIMEOUT: &str = "submit_timeout"; } // The tables only index the lazy fast-path cache below; a `{variant, reason}` @@ -139,7 +145,7 @@ pub mod frame_drop_reason { // site actually produces it, so the unreachable corners of the 7 x 9 cross // product never appear as permanent zero-valued series. const VARIANT_COUNT: usize = 7; -const REASON_COUNT: usize = 9; +const REASON_COUNT: usize = 11; const VARIANTS: [&str; VARIANT_COUNT] = [ frame_drop_variant::CONSENSUS, @@ -161,6 +167,8 @@ const REASONS: [&str; REASON_COUNT] = [ frame_drop_reason::MISROUTED, frame_drop_reason::PARK_OVERFLOW, frame_drop_reason::PARK_DROPPED, + frame_drop_reason::SUBMIT_ABANDONED, + frame_drop_reason::SUBMIT_TIMEOUT, ]; fn variant_index(s: &str) -> Option { diff --git a/core/shard/src/router.rs b/core/shard/src/router.rs index a866a9027c..1f4fa01170 100644 --- a/core/shard/src/router.rs +++ b/core/shard/src/router.rs @@ -351,13 +351,13 @@ where self.apply_reconcile_ops(); consensus_tick.set(rearm_tick()); } - (message, provenance) = poll_fn(|_| { + frame = poll_fn(|_| { self.pop_redispatched_frame().map_or(Poll::Pending, Poll::Ready) }).fuse() => { // One frame per select iteration. Ranking this arm above // the inbox preserves park order without making a full // queue stall ticks and commit broadcasts for other groups. - self.dispatch_message(message, Some(provenance)).await; + self.dispatch_redispatched_frame(frame).await; // A request handled by a solo primary self-acks here. If // loopback waited for another inbox frame, the request // would remain uncommitted indefinitely on a quiet shard. @@ -480,8 +480,8 @@ where M: RestorableMetadataStm, { loop { - while let Some((message, provenance)) = self.pop_redispatched_frame() { - self.dispatch_message(message, Some(provenance)).await; + while let Some(frame) = self.pop_redispatched_frame() { + self.dispatch_redispatched_frame(frame).await; self.process_loopback(loopback_buf, namespace_scratch).await; self.apply_reconcile_ops(); if let Some(fault) = self.first_partition_commit_fault() { @@ -718,6 +718,14 @@ where // times out. (self.on_partition_read)(namespace, read, reply); } + LifecycleFrame::PartitionSubmit { request, reply } => { + // Addressed to the shard owning the request's namespace (the + // sender resolved it via the shards table, same fallback as + // `route_typed`). Every refusal answers on `reply`, so the + // awaiting shard never waits out its budget on a decision + // already made. + self.on_partition_submit(request, reply).await; + } LifecycleFrame::MetadataCommitTick => { // Reconciler may not yet be wired (e.g. mid-bootstrap, or // single-shard tests that never enable the reconciler loop). diff --git a/core/simulator/src/lib.rs b/core/simulator/src/lib.rs index fda8e5bcfd..3d2fd3feaa 100644 --- a/core/simulator/src/lib.rs +++ b/core/simulator/src/lib.rs @@ -2002,122 +2002,6 @@ mod tests { ); } - /// At-least-once failover: a `SendMessages` retry on a new primary re-executes. - /// The retry reply carries a HIGHER `commit` op, proof of re-execution rather - /// than dedup, and the duplicate payload lives at two offsets. Consumers dedup - /// if they want at-most-once-per-payload. - #[test] - fn failover_retry_re_executes_under_at_least_once() { - server_common::MemoryPool::init_pool(&server_common::MemoryPoolSettings { - enabled: false, - size: iggy_common::IggyByteSize::from(0u64), - bucket_capacity: 1, - }); - - let replica_count: u8 = 5; - let client_id: u128 = 1; - let network_opts = packet::PacketSimulatorOptions { - node_count: replica_count, - client_count: 1, - ..packet::PacketSimulatorOptions::default() - }; - - let mut sim = Simulator::new( - replica_count as usize, - std::iter::once(client_id), - network_opts, - ); - let client = SimClient::new(client_id); - let ns = IggyNamespace::new(1, 1, 0); - sim.init_partition(ns); - sim.register_client_with_primary(&client); - - // Same `(client, session, request)` for replay; mirrors SDK's - // connection-loss retry. - let original_req = client.send_messages(ns, &[Bytes::from_static(b"failover-test")]); - let replay_req = original_req.deep_copy(); - let original_request_id = original_req.header().request; - - sim.submit_request(client_id, 0, original_req.into_generic()); - - let mut original_reply: Option> = None; - for _ in 0..200 { - let replies = sim.step(); - if !replies.is_empty() { - original_reply = Some(replies[0].deep_copy()); - break; - } - } - let original_reply = original_reply.expect("commit reply must arrive before primary crash"); - let original_commit_op = original_reply.header().commit; - assert_eq!( - original_reply.header().request, - original_request_id, - "sanity: original reply must echo the request id" - ); - - // Crash primary. Real-world: TCP buffer might have lost reply - // before ack; same retry path. - sim.replica_crash(0); - - // Steps for view change across 4 survivors. - for _ in 0..800 { - sim.step(); - } - - // Find new primary via any live replica. - let live = &sim.replicas[1].shards[0]; - let live_consensus = live - .plane - .partitions() - .get_by_ns(&ns) - .expect("partition must exist on a live replica") - .consensus(); - assert!( - live_consensus.view() > 0, - "view must have advanced past the crashed primary" - ); - let new_primary_idx = live_consensus.primary_index(live_consensus.view()); - assert_ne!( - new_primary_idx, 0, - "new primary must not be the crashed replica" - ); - - // Replay the SAME request to the new primary. No dedup, so re-execution. - sim.submit_request(client_id, new_primary_idx, replay_req.into_generic()); - - let mut retry_reply: Option> = None; - for _ in 0..200 { - let replies = sim.step(); - if !replies.is_empty() { - retry_reply = Some(replies[0].deep_copy()); - break; - } - } - let retry_reply = retry_reply.expect( - "reply must arrive after retry; new primary re-commits as \ - fresh prepare (at-least-once)", - ); - - // At-least-once: same request id (correlation), HIGHER commit op - // (re-execution). No dedup absorbs the retry. - assert_eq!( - retry_reply.header().request, - original_request_id, - "retry's reply must correlate to the request id" - ); - assert!( - retry_reply.header().commit > original_commit_op, - "retry must re-execute (commit op > original={original_commit_op}, got {})", - retry_reply.header().commit - ); - assert_eq!( - retry_reply.header().client, - client_id, - "retry must echo original client_id" - ); - } - /// Determinism: fresh simulator + workload from the same seed (network /// and workload) produces an identical reply-header sequence. #[test] @@ -2617,6 +2501,11 @@ mod tests { let client = SimClient::new(CLIENT_ID); sim.shell_login(&client); + // Both sends in flight at once. The simulated link delays each packet + // independently, so the lower request id can reach the primary after + // the higher one committed; the dedup slice's committed-id window is + // what keeps that reordered arrival a new write rather than an absorbed + // duplicate, and this loop is the check that both payloads commit. for payload in [ Bytes::from_static(b"parked-redispatch-0"), Bytes::from_static(b"parked-redispatch-1"), @@ -3228,6 +3117,149 @@ mod tests { ); } + /// Failover retry absorbed by the partition dedup slice: a `SendMessages` + /// replay of an already-committed `(client, request)` on a NEW primary is + /// answered without re-executing. The slice is folded in on every replica + /// at commit, so the promoted primary knows the watermark its predecessor + /// established -- that inheritance is what this test proves. + #[test] + #[allow(clippy::too_many_lines)] + fn failover_retry_absorbed_by_partition_dedup() { + server_common::MemoryPool::init_pool(&server_common::MemoryPoolSettings { + enabled: false, + size: iggy_common::IggyByteSize::from(0u64), + bucket_capacity: 1, + }); + + let replica_count: u8 = 5; + let client_id: u128 = 1; + let network_opts = packet::PacketSimulatorOptions { + node_count: replica_count, + client_count: 1, + ..packet::PacketSimulatorOptions::default() + }; + + let mut sim = Simulator::new( + replica_count as usize, + std::iter::once(client_id), + network_opts, + ); + let client = SimClient::new(client_id); + let ns = IggyNamespace::new(1, 1, 0); + sim.init_partition(ns); + sim.register_client_with_primary(&client); + + // Same `(client, session, request)` for replay; mirrors SDK's + // connection-loss retry. + let original_req = client.send_messages(ns, &[Bytes::from_static(b"failover-test")]); + let replay_req = original_req.deep_copy(); + let original_request_id = original_req.header().request; + + sim.submit_request(client_id, 0, original_req.into_generic()); + + let mut original_reply: Option> = None; + for _ in 0..200 { + let replies = sim.step(); + if !replies.is_empty() { + original_reply = Some(replies[0].deep_copy()); + break; + } + } + let original_reply = original_reply.expect("commit reply must arrive before primary crash"); + let original_commit_op = original_reply.header().commit; + // Offset after exactly one committed batch: the duplicate must not + // move it. + let offset_after_original = sim.replicas[1].shards[0] + .plane + .partitions() + .get_by_ns(&ns) + .expect("partition must exist on a live replica") + .stats + .current_offset(); + assert_eq!( + original_reply.header().request, + original_request_id, + "sanity: original reply must echo the request id" + ); + + // Crash primary. Real-world: TCP buffer might have lost reply + // before ack; same retry path. + sim.replica_crash(0); + + // Steps for view change across 4 survivors. + for _ in 0..800 { + sim.step(); + } + + // Find new primary via any live replica. + let live = &sim.replicas[1].shards[0]; + let live_consensus = live + .plane + .partitions() + .get_by_ns(&ns) + .expect("partition must exist on a live replica") + .consensus(); + assert!( + live_consensus.view() > 0, + "view must have advanced past the crashed primary" + ); + let new_primary_idx = live_consensus.primary_index(live_consensus.view()); + assert_ne!( + new_primary_idx, 0, + "new primary must not be the crashed replica" + ); + + // Replay the SAME request to the new primary: the dedup slice it + // inherited at commit must absorb it. + sim.submit_request(client_id, new_primary_idx, replay_req.into_generic()); + + let mut retry_reply: Option> = None; + for _ in 0..200 { + let replies = sim.step(); + if !replies.is_empty() { + retry_reply = Some(replies[0].deep_copy()); + break; + } + } + let retry_reply = retry_reply + .expect("reply must arrive after retry; the new primary absorbs it as a duplicate"); + + assert_eq!( + retry_reply.header().request, + original_request_id, + "retry's reply must correlate to the request id" + ); + assert_eq!( + retry_reply.header().client, + client_id, + "retry must echo original client_id" + ); + assert_eq!( + retry_reply.header().status, + 0, + "an absorbed duplicate is a success, not an error" + ); + // The absorbed answer is synthesized at admission, so it never earns a + // new op. Re-execution would have committed past the original. + assert!( + retry_reply.header().op <= original_commit_op, + "retry must NOT re-execute (original commit={original_commit_op}, reply op={})", + retry_reply.header().op + ); + // The payload committed exactly once. + let committed = sim.replicas[usize::from(new_primary_idx)].shards[0] + .plane + .partitions() + .get_by_ns(&ns) + .expect("partition must exist on the new primary") + .stats + .current_offset(); + assert_eq!( + committed, offset_after_original, + "duplicate must not append a second copy" + ); + } + /// With a positive crash probability /// the driver crashes followers (never the primary) but never below the /// survivor floor, while the per-tick invariants stay green and the diff --git a/scripts/ci/storage-compat.sh b/scripts/ci/storage-compat.sh index b4c3a72a23..952e69fb5e 100755 --- a/scripts/ci/storage-compat.sh +++ b/scripts/ci/storage-compat.sh @@ -86,7 +86,8 @@ REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" # its own CWD, and the baseline below builds from a worktree elsewhere on # disk, so a relative value would scatter the two builds and the lookup of # either binary across three directories. Exported so every cargo call here, -# nextest included, lands in the same place. +# nextest included, lands in the same place -- except the baseline build, which +# must have a target directory of its own (see the worktree build below). TARGET_DIR="${CARGO_TARGET_DIR:-${REPO_ROOT}/target}" mkdir -p "${TARGET_DIR}" TARGET_DIR="$(cd "${TARGET_DIR}" && pwd)" @@ -201,39 +202,51 @@ else WORKTREE_DIR="$(mktemp -d "${TMPDIR:-/tmp}/iggy-storage-compat.XXXXXX")" git worktree add --detach "${WORKTREE_DIR}" "${BASELINE_SHA}" - # Symmetric to the guard before the HEAD build below: a HEAD binary left by an - # earlier run makes cargo skip the uplift, and the cp further down would then - # capture that HEAD binary as the baseline, comparing HEAD against itself. - rm -f "${HEAD_SERVER}" - echo "Building baseline iggy-server from ${BASELINE_SHA}..." # Built from the worktree as CWD so the baseline's own rust-toolchain.toml - # applies, into the shared CARGO_TARGET_DIR so the registry graph compiles - # once. + # applies, and into a target directory of the worktree's own. + # + # The two trees MUST NOT share one. Cargo keys a workspace member's unit hash + # on its manifest path RELATIVE to the workspace root, and records that + # member's sources in the dep-info relative too, so `core/configs` in the + # worktree and `core/configs` here hash identically and both resolve against + # whichever root cargo is invoked from. Sharing a target directory therefore + # makes the second build read the first build's rlibs as fresh: HEAD's + # `server` would compile against MASTER's `configs`, `consensus`, + # `partitions` and `shard`. Any PR touching a crate below `server` fails to + # build here with errors that do not reproduce anywhere else. The duplicated + # dependency compile is the price of the two halves being what they claim. + # + # Inside the worktree so the cleanup trap reclaims it with the worktree; only + # the copied binary below outlives the run. # # Debug profile on both sides, and no --all-features: release would compile # debug_assert! out of the baseline while HEAD still panics on it, and # --all-features turns on the server's `disable-mimalloc`, so the two halves # would differ in ways the storage format never changed. + BASELINE_TARGET_DIR="${WORKTREE_DIR}/target" ( cd "${WORKTREE_DIR}" - cargo build --locked -p server --bin iggy-server + CARGO_TARGET_DIR="${BASELINE_TARGET_DIR}" cargo build --locked -p server --bin iggy-server ) - if [ ! -x "${HEAD_SERVER}" ]; then - echo "Baseline build did not produce ${HEAD_SERVER}" + BASELINE_BUILT="${BASELINE_TARGET_DIR}/debug/iggy-server" + if [ ! -x "${BASELINE_BUILT}" ]; then + echo "Baseline build did not produce ${BASELINE_BUILT}" exit 1 fi - # Copy before HEAD builds: both trees uplift to the same target/debug path. mkdir -p "$(dirname "${BASELINE_SERVER}")" - cp "${HEAD_SERVER}" "${BASELINE_SERVER}" + cp "${BASELINE_BUILT}" "${BASELINE_SERVER}" + # Now, not at cleanup: a second full debug dependency graph is several GB, and + # the HEAD build plus the integration test still have to fit on the runner. + rm -rf "${BASELINE_TARGET_DIR}" fi -# Delete the uplifted binary before building HEAD. Cargo skips re-uplifting when -# the destination already looks current, and the baseline build just refreshed -# it, so on a second run the BASELINE binary could survive at -# target/debug/iggy-server and the test would compare master against master. +# The baseline never writes here any more, but a binary left by an earlier run +# of this script (or by any other build in this tree) would satisfy the +# existence check below without cargo having produced it now. Delete it so that +# check means what it says. rm -f "${HEAD_SERVER}" echo "Building HEAD iggy-server from ${HEAD_SHA}..." From b0088408c7ffa4125b8df809f08270657cc53ccb Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Thu, 3 Sep 2026 11:03:14 +0200 Subject: [PATCH 050/182] chore(deps): Bump @humanfs/node from 0.16.7 to 0.16.8 in /examples/node (#4044) --- examples/node/package-lock.json | 36 +++++++++++++++++++++++---------- 1 file changed, 25 insertions(+), 11 deletions(-) diff --git a/examples/node/package-lock.json b/examples/node/package-lock.json index ab1a938567..980724cb06 100644 --- a/examples/node/package-lock.json +++ b/examples/node/package-lock.json @@ -26,18 +26,18 @@ "version": "0.10.0-edge.5", "license": "Apache-2.0", "dependencies": { - "@node-rs/xxhash": "1.7.6", + "@node-rs/xxhash": "1.7.7", "debug": "4.4.3", "generic-pool": "3.9.0", "uuidv7": "1.2.1" }, "devDependencies": { - "@commitlint/cli": "21.2.1", - "@commitlint/config-conventional": "21.2.0", + "@commitlint/cli": "21.2.2", + "@commitlint/config-conventional": "21.2.2", "@cucumber/cucumber": "13.2.1", "@swc-node/register": "1.12.1", "@types/debug": "4.1.13", - "@types/node": "26.1.2", + "@types/node": "26.2.0", "c8": "^12.0.0", "husky": "9.1.7", "typescript": "6.0.3", @@ -606,29 +606,43 @@ } }, "node_modules/@humanfs/core": { - "version": "0.19.1", - "resolved": "https://registry.npmjs.org/@humanfs/core/-/core-0.19.1.tgz", - "integrity": "sha512-5DyQ4+1JEUzejeK1JGICcideyfUbGixgS9jNgex5nqkW+cY7WZhxBigmieN5Qnw9ZosSNVC9KQKyb+GUaGyKUA==", + "version": "0.19.2", + "resolved": "https://registry.npmjs.org/@humanfs/core/-/core-0.19.2.tgz", + "integrity": "sha512-UhXNm+CFMWcbChXywFwkmhqjs3PRCmcSa/hfBgLIb7oQ5HNb1wS0icWsGtSAUNgefHeI+eBrA8I1fxmbHsGdvA==", "dev": true, "license": "Apache-2.0", + "dependencies": { + "@humanfs/types": "^0.15.0" + }, "engines": { "node": ">=18.18.0" } }, "node_modules/@humanfs/node": { - "version": "0.16.7", - "resolved": "https://registry.npmjs.org/@humanfs/node/-/node-0.16.7.tgz", - "integrity": "sha512-/zUx+yOsIrG4Y43Eh2peDeKCxlRt/gET6aHfaKpuq267qXdYDFViVHfMaLyygZOnl0kGWxFIgsBy8QFuTLUXEQ==", + "version": "0.16.8", + "resolved": "https://registry.npmjs.org/@humanfs/node/-/node-0.16.8.tgz", + "integrity": "sha512-gE1eQNZ3R++kTzFUpdGlpmy8kDZD/MLyHqDwqjkVQI0JMdI1D51sy1H958PNXYkM2rAac7e5/CnIKZrHtPh3BQ==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@humanfs/core": "^0.19.1", + "@humanfs/core": "^0.19.2", + "@humanfs/types": "^0.15.0", "@humanwhocodes/retry": "^0.4.0" }, "engines": { "node": ">=18.18.0" } }, + "node_modules/@humanfs/types": { + "version": "0.15.0", + "resolved": "https://registry.npmjs.org/@humanfs/types/-/types-0.15.0.tgz", + "integrity": "sha512-ZZ1w0aoQkwuUuC7Yf+7sdeaNfqQiiLcSRbfI08oAxqLtpXQr9AIVX7Ay7HLDuiLYAaFPu8oBYNq/QIi9URHJ3Q==", + "dev": true, + "license": "Apache-2.0", + "engines": { + "node": ">=18.18.0" + } + }, "node_modules/@humanwhocodes/module-importer": { "version": "1.0.1", "resolved": "https://registry.npmjs.org/@humanwhocodes/module-importer/-/module-importer-1.0.1.tgz", From 858279b58ed8d367f8aa9891d9e5fd1f23aa3b2a Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Thu, 3 Sep 2026 11:35:54 +0200 Subject: [PATCH 051/182] chore(deps): Bump @humanfs/node from 0.16.7 to 0.16.8 in /web (#4045) --- web/package-lock.json | 28 +++++++++++++++++++++------- 1 file changed, 21 insertions(+), 7 deletions(-) diff --git a/web/package-lock.json b/web/package-lock.json index 5d19d523be..4d8da6fcf1 100644 --- a/web/package-lock.json +++ b/web/package-lock.json @@ -738,29 +738,43 @@ } }, "node_modules/@humanfs/core": { - "version": "0.19.1", - "resolved": "https://registry.npmjs.org/@humanfs/core/-/core-0.19.1.tgz", - "integrity": "sha512-5DyQ4+1JEUzejeK1JGICcideyfUbGixgS9jNgex5nqkW+cY7WZhxBigmieN5Qnw9ZosSNVC9KQKyb+GUaGyKUA==", + "version": "0.19.2", + "resolved": "https://registry.npmjs.org/@humanfs/core/-/core-0.19.2.tgz", + "integrity": "sha512-UhXNm+CFMWcbChXywFwkmhqjs3PRCmcSa/hfBgLIb7oQ5HNb1wS0icWsGtSAUNgefHeI+eBrA8I1fxmbHsGdvA==", "dev": true, "license": "Apache-2.0", + "dependencies": { + "@humanfs/types": "^0.15.0" + }, "engines": { "node": ">=18.18.0" } }, "node_modules/@humanfs/node": { - "version": "0.16.7", - "resolved": "https://registry.npmjs.org/@humanfs/node/-/node-0.16.7.tgz", - "integrity": "sha512-/zUx+yOsIrG4Y43Eh2peDeKCxlRt/gET6aHfaKpuq267qXdYDFViVHfMaLyygZOnl0kGWxFIgsBy8QFuTLUXEQ==", + "version": "0.16.8", + "resolved": "https://registry.npmjs.org/@humanfs/node/-/node-0.16.8.tgz", + "integrity": "sha512-gE1eQNZ3R++kTzFUpdGlpmy8kDZD/MLyHqDwqjkVQI0JMdI1D51sy1H958PNXYkM2rAac7e5/CnIKZrHtPh3BQ==", "dev": true, "license": "Apache-2.0", "dependencies": { - "@humanfs/core": "^0.19.1", + "@humanfs/core": "^0.19.2", + "@humanfs/types": "^0.15.0", "@humanwhocodes/retry": "^0.4.0" }, "engines": { "node": ">=18.18.0" } }, + "node_modules/@humanfs/types": { + "version": "0.15.0", + "resolved": "https://registry.npmjs.org/@humanfs/types/-/types-0.15.0.tgz", + "integrity": "sha512-ZZ1w0aoQkwuUuC7Yf+7sdeaNfqQiiLcSRbfI08oAxqLtpXQr9AIVX7Ay7HLDuiLYAaFPu8oBYNq/QIi9URHJ3Q==", + "dev": true, + "license": "Apache-2.0", + "engines": { + "node": ">=18.18.0" + } + }, "node_modules/@humanwhocodes/module-importer": { "version": "1.0.1", "resolved": "https://registry.npmjs.org/@humanwhocodes/module-importer/-/module-importer-1.0.1.tgz", From a7fcf92829a3d465565dd6368d437cd70795f54e Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Thu, 3 Sep 2026 12:29:16 +0200 Subject: [PATCH 052/182] chore(deps): Bump the python group across 2 directories with 1 update (#4039) --- bdd/python/uv.lock | 38 +++++++++++++++++++------------------- examples/python/uv.lock | 38 +++++++++++++++++++------------------- 2 files changed, 38 insertions(+), 38 deletions(-) diff --git a/bdd/python/uv.lock b/bdd/python/uv.lock index b2510b94a0..44c64ab552 100644 --- a/bdd/python/uv.lock +++ b/bdd/python/uv.lock @@ -471,27 +471,27 @@ wheels = [ [[package]] name = "ruff" -version = "0.16.3" +version = "0.16.4" source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/61/b3/3213589383f8f1b3938781bd1278713f6d18621a14992b3e81fefb8a5ef9/ruff-0.16.3.tar.gz", hash = "sha256:e76d33a347661a84b5be6d043d0347fdc745dfdcf825a8f4fed64b5e26eebdf2", size = 4891904, upload-time = "2026-08-13T15:17:13.381Z" } +sdist = { url = "https://files.pythonhosted.org/packages/00/8f/d8074b1f25e003164087a8bfe79a0f1a3945135764dbb6aaab04103dcaf9/ruff-0.16.4.tar.gz", hash = "sha256:13171aa9d9af2240ee3504e639de73122c67e74036de5ba2e1d01422cd17e3dc", size = 4899731, upload-time = "2026-08-20T17:43:59.196Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/bf/96/493770daebd68c0a67f1549fdf519f53be51fc435186c0585bcc272fd76c/ruff-0.16.3-py3-none-linux_armv6l.whl", hash = "sha256:0c5710e247a58a4521e66e124ba9a74655b414f61ba3a2e9e3811e11098f48f7", size = 10902799, upload-time = "2026-08-13T15:16:27.382Z" }, - { url = "https://files.pythonhosted.org/packages/5e/e6/2becf3942fddc29a29b8df47691d456fb1085391a694f74d84513251418c/ruff-0.16.3-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:fe155130631a2471fd2e14a7a664a4dfbd7194b8229c3d7b2a40b21178639081", size = 11135539, upload-time = "2026-08-13T15:16:30.87Z" }, - { url = "https://files.pythonhosted.org/packages/3e/1e/4b8b72f0d006dbf19326aa99f9ca0ee2ff374187c4d301cf529a51aa06fe/ruff-0.16.3-py3-none-macosx_11_0_arm64.whl", hash = "sha256:e2ed719e14aa64d895c2ee922594a90a43c861a93f0575a95ff8c47cdbd13eb9", size = 10475095, upload-time = "2026-08-13T15:16:33.259Z" }, - { url = "https://files.pythonhosted.org/packages/92/32/2201fa49ba1f6c101ee321e83f051ac7a4b8d07b0ef6b4d3f2772b302275/ruff-0.16.3-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:9e0b1da805eb043654645d74d5de1e5ce2edc686e40790d2b86f56d71cc06a84", size = 10668771, upload-time = "2026-08-13T15:16:35.65Z" }, - { url = "https://files.pythonhosted.org/packages/c3/66/4afc5c8363bd04d45effce1b7c8713ca037d7a6740b7451a2403a6e3a972/ruff-0.16.3-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:a37bdea0bbe21780f590bf437d6412c8c4e1b6cd010f91a65c2c40c5e5f5f870", size = 10699568, upload-time = "2026-08-13T15:16:38.195Z" }, - { url = "https://files.pythonhosted.org/packages/53/fd/c67d246bf36bf1698551c56de39e95cd07f70e64433e0098e6267d77061b/ruff-0.16.3-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:09571e6d1288ed9be475207a3ac04ada404f1cd898104be0f6ab8d7df438575b", size = 11499365, upload-time = "2026-08-13T15:16:40.623Z" }, - { url = "https://files.pythonhosted.org/packages/67/0b/00ecbceb99a263af7b12f6f05ac3c92bc47b905e91adc3f207a836e3bc01/ruff-0.16.3-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:2c18c5a101eb540010638cc1ff3c84944d3adb3df62b8d98ca8f22ba484d3413", size = 12311728, upload-time = "2026-08-13T15:16:43.564Z" }, - { url = "https://files.pythonhosted.org/packages/54/b2/b7b3bb54f4d3f7db504e476ad4ab8de530dceebe2c061384b2757ee419e8/ruff-0.16.3-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:8457c44f15033c85ddbb77b15d451df9e24e4bd03b628396dd3610cedc3b8f82", size = 11699896, upload-time = "2026-08-13T15:16:46.209Z" }, - { url = "https://files.pythonhosted.org/packages/c7/30/4c468429ac195addc5ee1b717b6ab1b66632786737ca3b2ed3443fb0c26a/ruff-0.16.3-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:294b95c4ae0cda9388525c2047778aa758d6b8d4bb876fd4e9eaa3ebc92343eb", size = 11058736, upload-time = "2026-08-13T15:16:48.823Z" }, - { url = "https://files.pythonhosted.org/packages/43/67/7a113cdaddf24b64d7f75b1242a99d04c82fcef4f6921fdbb832beaffb5f/ruff-0.16.3-py3-none-manylinux_2_31_riscv64.whl", hash = "sha256:3d0c7c40c87c2a820509c31ba007968da6e1306468c067b2d82fbfdbcd0e8474", size = 11586911, upload-time = "2026-08-13T15:16:51.913Z" }, - { url = "https://files.pythonhosted.org/packages/f1/c1/2e66f24c0f3ead25a5e660111778685e505e5da353c82802bf49f0cbe7b9/ruff-0.16.3-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:9f738c0fdfa8eed0b2ce7fb27ee7258208a92a68d7949e62aa15164bc7b389da", size = 10954265, upload-time = "2026-08-13T15:16:54.763Z" }, - { url = "https://files.pythonhosted.org/packages/c2/ba/4cee23bf52cba9a058d3726de623624daf50ef9638868edd86f4126157f6/ruff-0.16.3-py3-none-musllinux_1_2_armv7l.whl", hash = "sha256:fb785f0be25abe69d320415cd4f833b59e17ba7613d9ba6a958023b6bceb0a50", size = 10709886, upload-time = "2026-08-13T15:16:57.339Z" }, - { url = "https://files.pythonhosted.org/packages/82/df/7da7194fa5d9dc0a285f7e6fa5a4722e7c63faac0b45b614ded9314363a1/ruff-0.16.3-py3-none-musllinux_1_2_i686.whl", hash = "sha256:c5536e3acfbf9563085aa2be7b13c629c3077e902afc5b941ac44024dbb9f506", size = 11210392, upload-time = "2026-08-13T15:17:00.171Z" }, - { url = "https://files.pythonhosted.org/packages/35/85/7795f6e817af050e7517bf3e7aa9b061cce70ef33d280aad902c956c1ecf/ruff-0.16.3-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:a2d85c02f9b8e165d85e6779184d38c4132de12603dab59c51c28e22584f9e4d", size = 11626910, upload-time = "2026-08-13T15:17:03.299Z" }, - { url = "https://files.pythonhosted.org/packages/78/9b/475b927cf27a5cbbda3c7bafb69ed6ff77e1d7923d5d85f17c2749d7ae32/ruff-0.16.3-py3-none-win32.whl", hash = "sha256:388cdf2166642bd9b13d52b5932d3170f34f8abed7e8d9a855f1d84b83645a0a", size = 10931415, upload-time = "2026-08-13T15:17:05.726Z" }, - { url = "https://files.pythonhosted.org/packages/b2/99/e2a2bfc4fbf0a1e8a916bc9ebe6fe6c58cc34c28e0ffc6ce281d572d1c2e/ruff-0.16.3-py3-none-win_amd64.whl", hash = "sha256:e80a7d69ca2a6d1c4d352ec91458cdca6e56c83cdbcabd93e4abe1e53591d948", size = 11445993, upload-time = "2026-08-13T15:17:08.353Z" }, - { url = "https://files.pythonhosted.org/packages/69/3e/4132e539aed78c148854d4997a2685b0ed4dc4e87110b59ce528564e184e/ruff-0.16.3-py3-none-win_arm64.whl", hash = "sha256:b8ca152da82c1acc1fa8d5874b15951935f0eef46f10e6954c83859011b6178a", size = 11399302, upload-time = "2026-08-13T15:17:10.908Z" }, + { url = "https://files.pythonhosted.org/packages/ff/80/779895ef584e089d22f2c6df0d0e99a65ec2df0805f1fffd439415b8c1f0/ruff-0.16.4-py3-none-linux_armv6l.whl", hash = "sha256:df4075f71ddac40b9934af60c3ec8a53047dd5a5fdc43224e6e4e8e9a27cb6f7", size = 10006909, upload-time = "2026-08-20T17:43:16.888Z" }, + { url = "https://files.pythonhosted.org/packages/a9/e6/f553199b5e8927a05cb5c422d921fd0656b29ab976e91c44802107c6b0da/ruff-0.16.4-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:0c95538517af68004306b0fb3214ff2f2af67a65092aee77cd9eb86db6656604", size = 10240201, upload-time = "2026-08-20T17:43:19.337Z" }, + { url = "https://files.pythonhosted.org/packages/1c/70/4a6dc4bb34da4dee35e30f09bbd1bfbdd26f33b62fb9b8df31f08a199cd2/ruff-0.16.4-py3-none-macosx_11_0_arm64.whl", hash = "sha256:963f83df8e69e575b64d67dd447ebbc917db41a14bf38d4593a4183e7aaa8255", size = 9835122, upload-time = "2026-08-20T17:43:21.708Z" }, + { url = "https://files.pythonhosted.org/packages/24/12/c6e22d686372c15bcb7af99831f1a1be96df696491babf4f24e4f942c527/ruff-0.16.4-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:32a5057c7ff3f6e6480a48fccfb3a412a690f48a3d03ac5cf08177d6c2da3ade", size = 9977162, upload-time = "2026-08-20T17:43:24.236Z" }, + { url = "https://files.pythonhosted.org/packages/46/49/72b10ec912f5ab5854992eaf7aa7cd36729b6937d9dc4e0fb41b3bf428ec/ruff-0.16.4-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:b3dce8d9b0c57c265b91885a66a567d8ea1372e8eb4e250fa8e5e3f579e99cff", size = 9829789, upload-time = "2026-08-20T17:43:26.966Z" }, + { url = "https://files.pythonhosted.org/packages/fa/80/0f30e32e7f6ee26edc39075502db9d368d788a44a79b55f763eb4ab03796/ruff-0.16.4-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:7dc651db49283c69f8e72c834eec4fe5573e4c646856aebece0ce385dceb2a80", size = 10527949, upload-time = "2026-08-20T17:43:29.384Z" }, + { url = "https://files.pythonhosted.org/packages/52/3d/86e8ad3542169e56cac3859a343afdb9df2ad54d35a59ce1e67baee83421/ruff-0.16.4-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:3817b87dbcabc92f13b05019257c5b89b5b4d51b5fb20f56fb5235ceb723cd07", size = 11333695, upload-time = "2026-08-20T17:43:31.872Z" }, + { url = "https://files.pythonhosted.org/packages/d0/16/481c29b380c20a0054a8261066665e1b3488e23636c49d0a43e75975b9bb/ruff-0.16.4-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:e9fce1499134b2c8c68e5166f95705a5812062bb93aacc5f9873bb1a27084bc7", size = 10727741, upload-time = "2026-08-20T17:43:34.596Z" }, + { url = "https://files.pythonhosted.org/packages/5e/b6/56bc0b8cf45b54b28b3a5e6381c8945d51b5b18adf659454c32295209a31/ruff-0.16.4-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:f2d812e482f5a7e02eee26cd73d2a37ebbdf47d795ea63ba1b89110ae93e9fb3", size = 10286522, upload-time = "2026-08-20T17:43:37.288Z" }, + { url = "https://files.pythonhosted.org/packages/e8/8b/b345b4fb110f2fbe2bd31eabd271e5e8b3b7e4ee6c0e02f2dc6be78db000/ruff-0.16.4-py3-none-manylinux_2_31_riscv64.whl", hash = "sha256:6baaf984aa7976edf93d3b627fe2d1d22ee94bbca05fa6f90fc76d73924e3454", size = 10584182, upload-time = "2026-08-20T17:43:39.984Z" }, + { url = "https://files.pythonhosted.org/packages/29/e5/827b34041c35f58774a9681a4213994c164fc987800f4dddabcf451da0bf/ruff-0.16.4-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:bdfcf0b28662eb890372d50f92c283bb94e67e7635ed93c7fd533970acff7b2b", size = 10134195, upload-time = "2026-08-20T17:43:42.351Z" }, + { url = "https://files.pythonhosted.org/packages/0f/10/d0bffcdd6729b87afc82ba0ef377173356a7dc8e972f5179968cf2fdf98c/ruff-0.16.4-py3-none-musllinux_1_2_armv7l.whl", hash = "sha256:b66b02cb9b04f537643cadf5768e5f98dc461890d530cb67113d71c8c76e605d", size = 9825821, upload-time = "2026-08-20T17:43:44.532Z" }, + { url = "https://files.pythonhosted.org/packages/f5/32/0db2a863b796ca62d83e92a07a3ccf00921b14db02059347576a2fda3d4b/ruff-0.16.4-py3-none-musllinux_1_2_i686.whl", hash = "sha256:8528bf9a4b291a60bf02ea453511e8ce6215bd2b982ee80405b66b008b6c30a0", size = 10267658, upload-time = "2026-08-20T17:43:46.989Z" }, + { url = "https://files.pythonhosted.org/packages/b2/a0/fbdeb59e48c6261f523e56c8f12e9c08fbe693786595cc7e3959207a9232/ruff-0.16.4-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:fbd85d2875fdd67e833213a651f613bbf25303abf6aa822a5121f4531195678d", size = 10697071, upload-time = "2026-08-20T17:43:49.891Z" }, + { url = "https://files.pythonhosted.org/packages/aa/28/0c6dd865859c6d17bc8ccc34cb72b0e02d6c7eb25e8a1e22b5bea681e2c0/ruff-0.16.4-py3-none-win32.whl", hash = "sha256:312769988007aaeb8e189b443ccdd03c0e6374489e053467be6d96518ebff76e", size = 10021687, upload-time = "2026-08-20T17:43:52.281Z" }, + { url = "https://files.pythonhosted.org/packages/a3/03/e724450f621698117f9aa6dd241c94d0274ae96781378dc86745ae29f0e7/ruff-0.16.4-py3-none-win_amd64.whl", hash = "sha256:05d9d27a18c4bcbefada602480ec9e01e0bc949d432e0ced5df77edac195919c", size = 10567657, upload-time = "2026-08-20T17:43:54.78Z" }, + { url = "https://files.pythonhosted.org/packages/0e/fe/da8b9e1347696bb22120b77280ec5ce25d500ca5cb39d5ad6e5c18de19c1/ruff-0.16.4-py3-none-win_arm64.whl", hash = "sha256:a3a61621c9b6f6a89573e938a080e648f1695baa3f58570a3a707bc51ff65a21", size = 10451579, upload-time = "2026-08-20T17:43:57.135Z" }, ] [[package]] diff --git a/examples/python/uv.lock b/examples/python/uv.lock index 118a278a92..bde592d6b1 100644 --- a/examples/python/uv.lock +++ b/examples/python/uv.lock @@ -107,27 +107,27 @@ requires-dist = [ [[package]] name = "ruff" -version = "0.16.3" +version = "0.16.4" source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/61/b3/3213589383f8f1b3938781bd1278713f6d18621a14992b3e81fefb8a5ef9/ruff-0.16.3.tar.gz", hash = "sha256:e76d33a347661a84b5be6d043d0347fdc745dfdcf825a8f4fed64b5e26eebdf2", size = 4891904, upload-time = "2026-08-13T15:17:13.381Z" } +sdist = { url = "https://files.pythonhosted.org/packages/00/8f/d8074b1f25e003164087a8bfe79a0f1a3945135764dbb6aaab04103dcaf9/ruff-0.16.4.tar.gz", hash = "sha256:13171aa9d9af2240ee3504e639de73122c67e74036de5ba2e1d01422cd17e3dc", size = 4899731, upload-time = "2026-08-20T17:43:59.196Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/bf/96/493770daebd68c0a67f1549fdf519f53be51fc435186c0585bcc272fd76c/ruff-0.16.3-py3-none-linux_armv6l.whl", hash = "sha256:0c5710e247a58a4521e66e124ba9a74655b414f61ba3a2e9e3811e11098f48f7", size = 10902799, upload-time = "2026-08-13T15:16:27.382Z" }, - { url = "https://files.pythonhosted.org/packages/5e/e6/2becf3942fddc29a29b8df47691d456fb1085391a694f74d84513251418c/ruff-0.16.3-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:fe155130631a2471fd2e14a7a664a4dfbd7194b8229c3d7b2a40b21178639081", size = 11135539, upload-time = "2026-08-13T15:16:30.87Z" }, - { url = "https://files.pythonhosted.org/packages/3e/1e/4b8b72f0d006dbf19326aa99f9ca0ee2ff374187c4d301cf529a51aa06fe/ruff-0.16.3-py3-none-macosx_11_0_arm64.whl", hash = "sha256:e2ed719e14aa64d895c2ee922594a90a43c861a93f0575a95ff8c47cdbd13eb9", size = 10475095, upload-time = "2026-08-13T15:16:33.259Z" }, - { url = "https://files.pythonhosted.org/packages/92/32/2201fa49ba1f6c101ee321e83f051ac7a4b8d07b0ef6b4d3f2772b302275/ruff-0.16.3-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:9e0b1da805eb043654645d74d5de1e5ce2edc686e40790d2b86f56d71cc06a84", size = 10668771, upload-time = "2026-08-13T15:16:35.65Z" }, - { url = "https://files.pythonhosted.org/packages/c3/66/4afc5c8363bd04d45effce1b7c8713ca037d7a6740b7451a2403a6e3a972/ruff-0.16.3-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:a37bdea0bbe21780f590bf437d6412c8c4e1b6cd010f91a65c2c40c5e5f5f870", size = 10699568, upload-time = "2026-08-13T15:16:38.195Z" }, - { url = "https://files.pythonhosted.org/packages/53/fd/c67d246bf36bf1698551c56de39e95cd07f70e64433e0098e6267d77061b/ruff-0.16.3-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:09571e6d1288ed9be475207a3ac04ada404f1cd898104be0f6ab8d7df438575b", size = 11499365, upload-time = "2026-08-13T15:16:40.623Z" }, - { url = "https://files.pythonhosted.org/packages/67/0b/00ecbceb99a263af7b12f6f05ac3c92bc47b905e91adc3f207a836e3bc01/ruff-0.16.3-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:2c18c5a101eb540010638cc1ff3c84944d3adb3df62b8d98ca8f22ba484d3413", size = 12311728, upload-time = "2026-08-13T15:16:43.564Z" }, - { url = "https://files.pythonhosted.org/packages/54/b2/b7b3bb54f4d3f7db504e476ad4ab8de530dceebe2c061384b2757ee419e8/ruff-0.16.3-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:8457c44f15033c85ddbb77b15d451df9e24e4bd03b628396dd3610cedc3b8f82", size = 11699896, upload-time = "2026-08-13T15:16:46.209Z" }, - { url = "https://files.pythonhosted.org/packages/c7/30/4c468429ac195addc5ee1b717b6ab1b66632786737ca3b2ed3443fb0c26a/ruff-0.16.3-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:294b95c4ae0cda9388525c2047778aa758d6b8d4bb876fd4e9eaa3ebc92343eb", size = 11058736, upload-time = "2026-08-13T15:16:48.823Z" }, - { url = "https://files.pythonhosted.org/packages/43/67/7a113cdaddf24b64d7f75b1242a99d04c82fcef4f6921fdbb832beaffb5f/ruff-0.16.3-py3-none-manylinux_2_31_riscv64.whl", hash = "sha256:3d0c7c40c87c2a820509c31ba007968da6e1306468c067b2d82fbfdbcd0e8474", size = 11586911, upload-time = "2026-08-13T15:16:51.913Z" }, - { url = "https://files.pythonhosted.org/packages/f1/c1/2e66f24c0f3ead25a5e660111778685e505e5da353c82802bf49f0cbe7b9/ruff-0.16.3-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:9f738c0fdfa8eed0b2ce7fb27ee7258208a92a68d7949e62aa15164bc7b389da", size = 10954265, upload-time = "2026-08-13T15:16:54.763Z" }, - { url = "https://files.pythonhosted.org/packages/c2/ba/4cee23bf52cba9a058d3726de623624daf50ef9638868edd86f4126157f6/ruff-0.16.3-py3-none-musllinux_1_2_armv7l.whl", hash = "sha256:fb785f0be25abe69d320415cd4f833b59e17ba7613d9ba6a958023b6bceb0a50", size = 10709886, upload-time = "2026-08-13T15:16:57.339Z" }, - { url = "https://files.pythonhosted.org/packages/82/df/7da7194fa5d9dc0a285f7e6fa5a4722e7c63faac0b45b614ded9314363a1/ruff-0.16.3-py3-none-musllinux_1_2_i686.whl", hash = "sha256:c5536e3acfbf9563085aa2be7b13c629c3077e902afc5b941ac44024dbb9f506", size = 11210392, upload-time = "2026-08-13T15:17:00.171Z" }, - { url = "https://files.pythonhosted.org/packages/35/85/7795f6e817af050e7517bf3e7aa9b061cce70ef33d280aad902c956c1ecf/ruff-0.16.3-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:a2d85c02f9b8e165d85e6779184d38c4132de12603dab59c51c28e22584f9e4d", size = 11626910, upload-time = "2026-08-13T15:17:03.299Z" }, - { url = "https://files.pythonhosted.org/packages/78/9b/475b927cf27a5cbbda3c7bafb69ed6ff77e1d7923d5d85f17c2749d7ae32/ruff-0.16.3-py3-none-win32.whl", hash = "sha256:388cdf2166642bd9b13d52b5932d3170f34f8abed7e8d9a855f1d84b83645a0a", size = 10931415, upload-time = "2026-08-13T15:17:05.726Z" }, - { url = "https://files.pythonhosted.org/packages/b2/99/e2a2bfc4fbf0a1e8a916bc9ebe6fe6c58cc34c28e0ffc6ce281d572d1c2e/ruff-0.16.3-py3-none-win_amd64.whl", hash = "sha256:e80a7d69ca2a6d1c4d352ec91458cdca6e56c83cdbcabd93e4abe1e53591d948", size = 11445993, upload-time = "2026-08-13T15:17:08.353Z" }, - { url = "https://files.pythonhosted.org/packages/69/3e/4132e539aed78c148854d4997a2685b0ed4dc4e87110b59ce528564e184e/ruff-0.16.3-py3-none-win_arm64.whl", hash = "sha256:b8ca152da82c1acc1fa8d5874b15951935f0eef46f10e6954c83859011b6178a", size = 11399302, upload-time = "2026-08-13T15:17:10.908Z" }, + { url = "https://files.pythonhosted.org/packages/ff/80/779895ef584e089d22f2c6df0d0e99a65ec2df0805f1fffd439415b8c1f0/ruff-0.16.4-py3-none-linux_armv6l.whl", hash = "sha256:df4075f71ddac40b9934af60c3ec8a53047dd5a5fdc43224e6e4e8e9a27cb6f7", size = 10006909, upload-time = "2026-08-20T17:43:16.888Z" }, + { url = "https://files.pythonhosted.org/packages/a9/e6/f553199b5e8927a05cb5c422d921fd0656b29ab976e91c44802107c6b0da/ruff-0.16.4-py3-none-macosx_10_12_x86_64.whl", hash = "sha256:0c95538517af68004306b0fb3214ff2f2af67a65092aee77cd9eb86db6656604", size = 10240201, upload-time = "2026-08-20T17:43:19.337Z" }, + { url = "https://files.pythonhosted.org/packages/1c/70/4a6dc4bb34da4dee35e30f09bbd1bfbdd26f33b62fb9b8df31f08a199cd2/ruff-0.16.4-py3-none-macosx_11_0_arm64.whl", hash = "sha256:963f83df8e69e575b64d67dd447ebbc917db41a14bf38d4593a4183e7aaa8255", size = 9835122, upload-time = "2026-08-20T17:43:21.708Z" }, + { url = "https://files.pythonhosted.org/packages/24/12/c6e22d686372c15bcb7af99831f1a1be96df696491babf4f24e4f942c527/ruff-0.16.4-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:32a5057c7ff3f6e6480a48fccfb3a412a690f48a3d03ac5cf08177d6c2da3ade", size = 9977162, upload-time = "2026-08-20T17:43:24.236Z" }, + { url = "https://files.pythonhosted.org/packages/46/49/72b10ec912f5ab5854992eaf7aa7cd36729b6937d9dc4e0fb41b3bf428ec/ruff-0.16.4-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:b3dce8d9b0c57c265b91885a66a567d8ea1372e8eb4e250fa8e5e3f579e99cff", size = 9829789, upload-time = "2026-08-20T17:43:26.966Z" }, + { url = "https://files.pythonhosted.org/packages/fa/80/0f30e32e7f6ee26edc39075502db9d368d788a44a79b55f763eb4ab03796/ruff-0.16.4-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:7dc651db49283c69f8e72c834eec4fe5573e4c646856aebece0ce385dceb2a80", size = 10527949, upload-time = "2026-08-20T17:43:29.384Z" }, + { url = "https://files.pythonhosted.org/packages/52/3d/86e8ad3542169e56cac3859a343afdb9df2ad54d35a59ce1e67baee83421/ruff-0.16.4-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:3817b87dbcabc92f13b05019257c5b89b5b4d51b5fb20f56fb5235ceb723cd07", size = 11333695, upload-time = "2026-08-20T17:43:31.872Z" }, + { url = "https://files.pythonhosted.org/packages/d0/16/481c29b380c20a0054a8261066665e1b3488e23636c49d0a43e75975b9bb/ruff-0.16.4-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:e9fce1499134b2c8c68e5166f95705a5812062bb93aacc5f9873bb1a27084bc7", size = 10727741, upload-time = "2026-08-20T17:43:34.596Z" }, + { url = "https://files.pythonhosted.org/packages/5e/b6/56bc0b8cf45b54b28b3a5e6381c8945d51b5b18adf659454c32295209a31/ruff-0.16.4-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:f2d812e482f5a7e02eee26cd73d2a37ebbdf47d795ea63ba1b89110ae93e9fb3", size = 10286522, upload-time = "2026-08-20T17:43:37.288Z" }, + { url = "https://files.pythonhosted.org/packages/e8/8b/b345b4fb110f2fbe2bd31eabd271e5e8b3b7e4ee6c0e02f2dc6be78db000/ruff-0.16.4-py3-none-manylinux_2_31_riscv64.whl", hash = "sha256:6baaf984aa7976edf93d3b627fe2d1d22ee94bbca05fa6f90fc76d73924e3454", size = 10584182, upload-time = "2026-08-20T17:43:39.984Z" }, + { url = "https://files.pythonhosted.org/packages/29/e5/827b34041c35f58774a9681a4213994c164fc987800f4dddabcf451da0bf/ruff-0.16.4-py3-none-musllinux_1_2_aarch64.whl", hash = "sha256:bdfcf0b28662eb890372d50f92c283bb94e67e7635ed93c7fd533970acff7b2b", size = 10134195, upload-time = "2026-08-20T17:43:42.351Z" }, + { url = "https://files.pythonhosted.org/packages/0f/10/d0bffcdd6729b87afc82ba0ef377173356a7dc8e972f5179968cf2fdf98c/ruff-0.16.4-py3-none-musllinux_1_2_armv7l.whl", hash = "sha256:b66b02cb9b04f537643cadf5768e5f98dc461890d530cb67113d71c8c76e605d", size = 9825821, upload-time = "2026-08-20T17:43:44.532Z" }, + { url = "https://files.pythonhosted.org/packages/f5/32/0db2a863b796ca62d83e92a07a3ccf00921b14db02059347576a2fda3d4b/ruff-0.16.4-py3-none-musllinux_1_2_i686.whl", hash = "sha256:8528bf9a4b291a60bf02ea453511e8ce6215bd2b982ee80405b66b008b6c30a0", size = 10267658, upload-time = "2026-08-20T17:43:46.989Z" }, + { url = "https://files.pythonhosted.org/packages/b2/a0/fbdeb59e48c6261f523e56c8f12e9c08fbe693786595cc7e3959207a9232/ruff-0.16.4-py3-none-musllinux_1_2_x86_64.whl", hash = "sha256:fbd85d2875fdd67e833213a651f613bbf25303abf6aa822a5121f4531195678d", size = 10697071, upload-time = "2026-08-20T17:43:49.891Z" }, + { url = "https://files.pythonhosted.org/packages/aa/28/0c6dd865859c6d17bc8ccc34cb72b0e02d6c7eb25e8a1e22b5bea681e2c0/ruff-0.16.4-py3-none-win32.whl", hash = "sha256:312769988007aaeb8e189b443ccdd03c0e6374489e053467be6d96518ebff76e", size = 10021687, upload-time = "2026-08-20T17:43:52.281Z" }, + { url = "https://files.pythonhosted.org/packages/a3/03/e724450f621698117f9aa6dd241c94d0274ae96781378dc86745ae29f0e7/ruff-0.16.4-py3-none-win_amd64.whl", hash = "sha256:05d9d27a18c4bcbefada602480ec9e01e0bc949d432e0ced5df77edac195919c", size = 10567657, upload-time = "2026-08-20T17:43:54.78Z" }, + { url = "https://files.pythonhosted.org/packages/0e/fe/da8b9e1347696bb22120b77280ec5ce25d500ca5cb39d5ad6e5c18de19c1/ruff-0.16.4-py3-none-win_arm64.whl", hash = "sha256:a3a61621c9b6f6a89573e938a080e648f1695baa3f58570a3a707bc51ff65a21", size = 10451579, upload-time = "2026-08-20T17:43:57.135Z" }, ] [[package]] From 873c6ce6ab1f3b2e21d356d0a5f02bdcb6a15f3c Mon Sep 17 00:00:00 2001 From: Justin Mclean Date: Thu, 3 Sep 2026 21:40:34 +1000 Subject: [PATCH 053/182] docs: mention the PHP SDK in the README (#4041) --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index 1700f19c95..3b00daf6c1 100644 --- a/README.md +++ b/README.md @@ -126,7 +126,7 @@ We do also publish edge/dev/nightly releases (e.g. `0.7.0-edge.1` or `apache/igg - [Node.js (TypeScript)](https://www.npmjs.com/package/apache-iggy) - [Go](https://pkg.go.dev/github.com/apache/iggy/foreign/go) -[C++](https://github.com/apache/iggy/tree/master/foreign/cpp) is work in progress. +[C++](https://github.com/apache/iggy/tree/master/foreign/cpp) and [PHP](https://github.com/apache/iggy/tree/master/foreign/php) are work in progress. --- From 05b3d74b90d56ab8d0ac7707c7ebd9b2870d1b5b Mon Sep 17 00:00:00 2001 From: Goutam Adwant <8672451+goutamadwant@users.noreply.github.com> Date: Thu, 3 Sep 2026 05:12:35 -0700 Subject: [PATCH 054/182] test(java): add Pinot connector E2E coverage (#3922) --- .../actions/java-gradle/pre-merge/action.yml | 6 + .../iggy-connector-pinot/build.gradle.kts | 25 +- .../iggy-connector-pinot/integration-test.sh | 234 --------- .../pinot/IggyPinotIntegrationTest.java | 465 ++++++++++++++++++ 4 files changed, 492 insertions(+), 238 deletions(-) delete mode 100755 foreign/java/external-processors/iggy-connector-pinot/integration-test.sh create mode 100644 foreign/java/external-processors/iggy-connector-pinot/src/test/java/org/apache/iggy/connector/pinot/IggyPinotIntegrationTest.java diff --git a/.github/actions/java-gradle/pre-merge/action.yml b/.github/actions/java-gradle/pre-merge/action.yml index 565e33c30f..d5568883b5 100644 --- a/.github/actions/java-gradle/pre-merge/action.yml +++ b/.github/actions/java-gradle/pre-merge/action.yml @@ -143,6 +143,11 @@ runs: mkdir -p reports/java-tests cp -r foreign/java/external-processors/iggy-connector-flink/iggy-connector-library/build/test-results reports/java-tests/flink fi + if [ -d "foreign/java/external-processors/iggy-connector-pinot/build/test-results" ]; then + echo "Found test reports in pinot" + mkdir -p reports/java-tests + cp -r foreign/java/external-processors/iggy-connector-pinot/build/test-results reports/java-tests/pinot + fi - name: Stop Iggy server if: always() && inputs.task == 'test' @@ -188,4 +193,5 @@ runs: paths: | foreign/java/java-sdk/build/test-results/**/TEST-*.xml foreign/java/external-processors/iggy-connector-flink/iggy-connector-library/build/test-results/**/TEST-*.xml + foreign/java/external-processors/iggy-connector-pinot/build/test-results/**/TEST-*.xml if: ${{ !cancelled() && inputs.task == 'test' }} diff --git a/foreign/java/external-processors/iggy-connector-pinot/build.gradle.kts b/foreign/java/external-processors/iggy-connector-pinot/build.gradle.kts index 769d1877b7..88d5e8bec4 100644 --- a/foreign/java/external-processors/iggy-connector-pinot/build.gradle.kts +++ b/foreign/java/external-processors/iggy-connector-pinot/build.gradle.kts @@ -19,6 +19,7 @@ plugins { id("iggy.java-library-conventions") + alias(libs.plugins.shadow) } dependencies { @@ -43,13 +44,19 @@ dependencies { testImplementation(platform(libs.junit.bom)) testImplementation(libs.bundles.testing) testImplementation(libs.pinot.spi) // Need Pinot SPI for tests + testImplementation(libs.testcontainers) testRuntimeOnly(libs.slf4j.simple) } -// Assemble connector plugin with all dependencies for Docker deployment -tasks.register("assemblePlugin") { - from(tasks.named("jar")) - from(configurations.runtimeClasspath) +tasks.shadowJar { + duplicatesStrategy = DuplicatesStrategy.EXCLUDE + relocate("io.netty", "org.apache.iggy.connector.pinot.shaded.io.netty") + mergeServiceFiles() +} + +// Assemble connector plugin with isolated dependencies for Docker deployment +tasks.register("assemblePlugin") { + from(tasks.named("shadowJar")) into(layout.buildDirectory.dir("plugin")) } @@ -57,6 +64,16 @@ tasks.named("jar") { finalizedBy("assemblePlugin") } +tasks.named("test") { + dependsOn("assemblePlugin") + inputs.dir(layout.projectDirectory.dir("deployment")) + inputs.property("useExternalServer", providers.environmentVariable("USE_EXTERNAL_SERVER").isPresent) + inputs.property("externalTcpPort", providers.environmentVariable("EXTERNAL_TCP_PORT").orElse("8090")) + systemProperty("iggy.pinot.image", "apachepinot/pinot:${libs.versions.pinot.get()}") + systemProperty("iggy.pinot.plugin.dir", layout.buildDirectory.dir("plugin").get().asFile.absolutePath) + systemProperty("iggy.pinot.deployment.dir", layout.projectDirectory.dir("deployment").asFile.absolutePath) +} + publishing { publications { named("maven") { diff --git a/foreign/java/external-processors/iggy-connector-pinot/integration-test.sh b/foreign/java/external-processors/iggy-connector-pinot/integration-test.sh deleted file mode 100755 index f83c5f8ec1..0000000000 --- a/foreign/java/external-processors/iggy-connector-pinot/integration-test.sh +++ /dev/null @@ -1,234 +0,0 @@ -#!/usr/bin/env bash -# Licensed to the Apache Software Foundation (ASF) under one -# or more contributor license agreements. See the NOTICE file -# distributed with this work for additional information -# regarding copyright ownership. The ASF licenses this file -# to you under the Apache License, Version 2.0 (the -# "License"); you may not use this file except in compliance -# with the License. You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, -# software distributed under the License is distributed on an -# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY -# KIND, either express or implied. See the License for the -# specific language governing permissions and limitations -# under the License. - -set -e - -# Colors for output -GREEN='\033[0;32m' -YELLOW='\033[1;33m' -RED='\033[0;31m' -NC='\033[0m' # No Color - -echo -e "${GREEN}=====================================${NC}" -echo -e "${GREEN}Iggy-Pinot Integration Test${NC}" -echo -e "${GREEN}=====================================${NC}" - -# Navigate to connector directory -cd "$(dirname "$0")" - -# Step 1: Build JARs -echo -e "\n${YELLOW}Step 1: Building JARs...${NC}" -cd ../../ -gradle :iggy-connector-pinot:jar :iggy:jar -cd external-processors/iggy-connector-pinot -echo -e "${GREEN}✓ JARs built successfully${NC}" - -# Step 2: Start Docker environment -echo -e "\n${YELLOW}Step 2: Starting Docker environment...${NC}" -docker-compose down -v -docker-compose up -d -echo -e "${GREEN}✓ Docker containers starting${NC}" - -# Step 3: Wait for services to be healthy -echo -e "\n${YELLOW}Step 3: Waiting for services to be healthy...${NC}" - -echo -n "Waiting for Iggy... " -for i in {1..30}; do - if curl --connect-timeout 3 --max-time 5 -s http://localhost:3000/ > /dev/null 2>&1; then - echo -e "${GREEN}✓${NC}" - break - fi - sleep 2 - echo -n "." -done - -echo -n "Waiting for Pinot Controller... " -for i in {1..60}; do - if curl --connect-timeout 3 --max-time 5 -s http://localhost:9000/health > /dev/null 2>&1; then - echo -e "${GREEN}✓${NC}" - break - fi - sleep 2 - echo -n "." -done - -echo -n "Waiting for Pinot Broker... " -for i in {1..60}; do - if curl --connect-timeout 3 --max-time 5 -s http://localhost:8099/health > /dev/null 2>&1; then - echo -e "${GREEN}✓${NC}" - break - fi - sleep 2 - echo -n "." -done - -echo -n "Waiting for Pinot Server... " -for i in {1..60}; do - if curl --connect-timeout 3 --max-time 5 -s http://localhost:8097/health > /dev/null 2>&1; then - echo -e "${GREEN}✓${NC}" - break - fi - sleep 2 - echo -n "." -done - -sleep 5 # Extra time for services to stabilize - -# Step 4: Login to Iggy and create stream/topic -echo -e "\n${YELLOW}Step 4: Logging in to Iggy and creating stream/topic...${NC}" - -# Login and get JWT token -TOKEN=$(curl -s -X POST "http://localhost:3000/users/login" \ - -H "Content-Type: application/json" \ - -d '{"username": "iggy", "password": "iggy"}' | jq -r '.access_token.token') - -if [ -z "$TOKEN" ] || [ "$TOKEN" = "null" ]; then - echo -e "${RED}✗ Failed to get authentication token${NC}" - exit 1 -fi - -echo -e "${GREEN}✓ Authenticated${NC}" - -# Create stream -curl -s -X POST "http://localhost:3000/streams" \ - -H "Authorization: Bearer $TOKEN" \ - -H "Content-Type: application/json" \ - -d '{"stream_id": 1, "name": "test-stream"}' \ - && echo -e "${GREEN}✓ Stream created${NC}" || echo -e "${RED}✗ Stream creation failed (may already exist)${NC}" - -# Create topic -TOPIC_RESPONSE=$(curl -s -X POST "http://localhost:3000/streams/test-stream/topics" \ - -H "Authorization: Bearer $TOKEN" \ - -H "Content-Type: application/json" \ - -d '{"topic_id": 1, "name": "test-events", "partitions_count": 2, "compression_algorithm": "none", "message_expiry": 0, "max_topic_size": 0}') - -if echo "$TOPIC_RESPONSE" | grep -q '"id"'; then - echo -e "${GREEN}✓ Topic created${NC}" -else - echo -e "${RED}✗ Topic creation failed: $TOPIC_RESPONSE${NC}" - exit 1 -fi - -# Create consumer group (topic-scoped, not stream-scoped) -curl -s -X POST "http://localhost:3000/streams/test-stream/topics/test-events/consumer-groups" \ - -H "Authorization: Bearer $TOKEN" \ - -H "Content-Type: application/json" \ - -d '{"name": "pinot-integration-test"}' \ - && echo -e "${GREEN}✓ Consumer group created${NC}" || echo -e "${YELLOW}Note: Consumer group may already exist${NC}" - -# Step 5: Create Pinot schema -echo -e "\n${YELLOW}Step 5: Creating Pinot schema...${NC}" -curl -X POST "http://localhost:9000/schemas" \ - -H "Content-Type: application/json" \ - -d @deployment/schema.json \ - && echo -e "${GREEN}✓ Schema created${NC}" || echo -e "${RED}✗ Schema creation failed${NC}" - -# Step 6: Create Pinot table -echo -e "\n${YELLOW}Step 6: Creating Pinot realtime table...${NC}" -TABLE_RESPONSE=$(curl -s -X POST "http://localhost:9000/tables" \ - -H "Content-Type: application/json" \ - -d @deployment/table.json) - -if echo "$TABLE_RESPONSE" | grep -q '"status":"Table test_events_REALTIME successfully added"'; then - echo -e "${GREEN}✓ Table created${NC}" -elif echo "$TABLE_RESPONSE" | grep -q '"code":500'; then - echo -e "${RED}✗ Table creation failed${NC}" - echo "$TABLE_RESPONSE" | jq '.' - exit 1 -else - echo -e "${GREEN}✓ Table created${NC}" -fi - -sleep 5 # Let table initialize - -# Step 7: Send test messages to Iggy -echo -e "\n${YELLOW}Step 7: Sending test messages to Iggy...${NC}" - -# Partition value for partition 0 (4-byte little-endian, base64 encoded) -PARTITION_VALUE=$(printf '\x00\x00\x00\x00' | base64) - -for i in {1..10}; do - TIMESTAMP=$(($(date +%s) * 1000)) - MESSAGE=$(cat < /dev/null 2>&1 - echo -e "${GREEN}✓ Message $i sent${NC}" - sleep 1 -done - -# Step 8: Wait for ingestion -echo -e "\n${YELLOW}Step 8: Waiting for Pinot to ingest messages...${NC}" -sleep 15 - -# Step 9: Query Pinot and verify data -echo -e "\n${YELLOW}Step 9: Querying Pinot for ingested data...${NC}" - -QUERY_RESULT=$(curl -s -X POST "http://localhost:8099/query/sql" \ - -H "Content-Type: application/json" \ - -d '{"sql": "SELECT COUNT(*) FROM test_events_REALTIME"}') - -echo "Query Result:" -echo "$QUERY_RESULT" | jq '.' - -# Extract count from result -COUNT=$(echo "$QUERY_RESULT" | jq -r '.resultTable.rows[0][0]' 2>/dev/null || echo "0") - -if [ "$COUNT" -gt "0" ]; then - echo -e "\n${GREEN}=====================================${NC}" - echo -e "${GREEN}✓ Integration Test PASSED!${NC}" - echo -e "${GREEN}Successfully ingested $COUNT messages${NC}" - echo -e "${GREEN}=====================================${NC}" - - # Show sample data - echo -e "\n${YELLOW}Sample data:${NC}" - curl -s -X POST "http://localhost:8099/query/sql" \ - -H "Content-Type: application/json" \ - -d '{"sql": "SELECT * FROM test_events_REALTIME LIMIT 5"}' | jq '.' - - EXIT_CODE=0 -else - echo -e "\n${RED}=====================================${NC}" - echo -e "${RED}✗ Integration Test FAILED!${NC}" - echo -e "${RED}No messages ingested${NC}" - echo -e "${RED}=====================================${NC}" - - # Show logs for debugging - echo -e "\n${YELLOW}Pinot Server logs:${NC}" - docker logs pinot-server --tail 50 - - EXIT_CODE=1 -fi - -# Cleanup option -echo -e "\n${YELLOW}To stop the environment: docker-compose down -v${NC}" -echo -e "${YELLOW}To view logs: docker-compose logs -f${NC}" - -exit $EXIT_CODE diff --git a/foreign/java/external-processors/iggy-connector-pinot/src/test/java/org/apache/iggy/connector/pinot/IggyPinotIntegrationTest.java b/foreign/java/external-processors/iggy-connector-pinot/src/test/java/org/apache/iggy/connector/pinot/IggyPinotIntegrationTest.java new file mode 100644 index 0000000000..5d78873da0 --- /dev/null +++ b/foreign/java/external-processors/iggy-connector-pinot/src/test/java/org/apache/iggy/connector/pinot/IggyPinotIntegrationTest.java @@ -0,0 +1,465 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.iggy.connector.pinot; + +import com.fasterxml.jackson.databind.JsonNode; +import com.fasterxml.jackson.databind.ObjectMapper; +import com.fasterxml.jackson.databind.node.ObjectNode; +import com.github.dockerjava.api.model.Capability; +import com.github.dockerjava.api.model.Ulimit; +import org.apache.iggy.client.blocking.tcp.IggyTcpClient; +import org.apache.iggy.identifier.StreamId; +import org.apache.iggy.identifier.TopicId; +import org.apache.iggy.message.Message; +import org.apache.iggy.message.Partitioning; +import org.apache.iggy.topic.CompressionAlgorithm; +import org.apache.pinot.spi.exception.QueryErrorCode; +import org.junit.jupiter.api.AfterAll; +import org.junit.jupiter.api.BeforeAll; +import org.junit.jupiter.api.Test; +import org.testcontainers.Testcontainers; +import org.testcontainers.containers.GenericContainer; +import org.testcontainers.containers.Network; +import org.testcontainers.containers.wait.strategy.Wait; +import org.testcontainers.images.PullPolicy; +import org.testcontainers.utility.DockerImageName; +import org.testcontainers.utility.MountableFile; + +import java.io.IOException; +import java.math.BigInteger; +import java.net.URI; +import java.net.http.HttpClient; +import java.net.http.HttpRequest; +import java.net.http.HttpResponse; +import java.nio.file.Files; +import java.nio.file.Path; +import java.time.Duration; +import java.time.Instant; +import java.util.ArrayList; +import java.util.List; +import java.util.Objects; +import java.util.UUID; +import java.util.function.Predicate; + +import static org.assertj.core.api.Assertions.assertThat; +import static org.assertj.core.api.Assertions.fail; + +class IggyPinotIntegrationTest { + + // The Java SDK speaks VSR, so use the same VSR-capable image as its integration tests. + private static final DockerImageName IGGY_IMAGE = DockerImageName.parse("apache/iggy:edge"); + private static final DockerImageName PINOT_IMAGE = DockerImageName.parse( + Objects.requireNonNull(System.getProperty("iggy.pinot.image"), "Missing iggy.pinot.image system property")); + private static final DockerImageName ZOOKEEPER_IMAGE = DockerImageName.parse("zookeeper:3.9"); + + private static final int IGGY_HTTP_PORT = 3000; + private static final int IGGY_TCP_PORT = 8090; + private static final int PINOT_CONTROLLER_PORT = 9000; + private static final int PINOT_BROKER_PORT = 8099; + private static final int PINOT_SERVER_ADMIN_PORT = 8097; + private static final String EXTERNAL_SERVER_HOST = "127.0.0.1"; + private static final String TESTCONTAINERS_HOST = "host.testcontainers.internal"; + private static final boolean USE_EXTERNAL_SERVER = System.getenv("USE_EXTERNAL_SERVER") != null; + + private static final String STREAM_NAME = "pinot-test-stream-" + UUID.randomUUID(); + private static final String TOPIC_NAME = "test-events"; + private static final String CONSUMER_GROUP_NAME = "pinot-integration-test"; + private static final String TABLE_NAME = "test_events_REALTIME"; + + private static final Duration HTTP_TIMEOUT = Duration.ofSeconds(30); + private static final Duration QUERY_TIMEOUT = Duration.ofSeconds(90); + private static final Duration STARTUP_TIMEOUT = Duration.ofMinutes(3); + + private static final ObjectMapper OBJECT_MAPPER = new ObjectMapper(); + private static final HttpClient HTTP_CLIENT = HttpClient.newBuilder() + .connectTimeout(HTTP_TIMEOUT) + .version(HttpClient.Version.HTTP_1_1) + .build(); + + private static Network network; + private static GenericContainer iggy; + private static GenericContainer zookeeper; + private static GenericContainer pinotController; + private static GenericContainer pinotBroker; + private static GenericContainer pinotServer; + private static IggyTcpClient iggyClient; + + @BeforeAll + static void startEnvironment() { + Path pluginDirectory = requiredDirectory("iggy.pinot.plugin.dir"); + Path deploymentDirectory = requiredDirectory("iggy.pinot.deployment.dir"); + + if (USE_EXTERNAL_SERVER) { + Testcontainers.exposeHostPorts(externalTcpPort()); + } + network = Network.newNetwork(); + try { + startZookeeper(); + if (!USE_EXTERNAL_SERVER) { + startIggy(); + } + startPinotController(pluginDirectory); + startPinotBroker(); + startPinotServer(pluginDirectory); + + iggyClient = IggyTcpClient.builder() + .host(iggyHost()) + .port(iggyPort()) + .credentials("iggy", "iggy") + .connectionTimeout(Duration.ofSeconds(10)) + .requestTimeout(Duration.ofSeconds(10)) + .buildAndLogin(); + createIggyResources(); + + postControllerResource("/schemas", Files.readString(deploymentDirectory.resolve("schema.json"))); + postControllerResource("/tables", tableConfiguration(deploymentDirectory.resolve("table.json"))); + awaitQuery("SELECT COUNT(*) FROM " + TABLE_NAME, IggyPinotIntegrationTest::hasResultRow); + } catch (IOException | RuntimeException e) { + throw new IllegalStateException("Failed to start the Iggy-Pinot test environment\n" + diagnostics(), e); + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + throw new IllegalStateException("Interrupted while starting the Iggy-Pinot test environment", e); + } + } + + @AfterAll + static void stopEnvironment() { + stop(pinotServer); + stop(pinotBroker); + stop(pinotController); + + if (iggyClient != null) { + try { + iggyClient.streams().deleteStream(StreamId.of(STREAM_NAME)); + } catch (RuntimeException ignored) { + // Startup may have failed before the stream was created. + } + try { + iggyClient.close(); + } catch (RuntimeException ignored) { + // Containers are still stopped below. + } + } + + stop(iggy); + stop(zookeeper); + if (network != null) { + network.close(); + } + } + + @Test + void shouldIngestAndMapJsonMessage() throws Exception { + String marker = "mapping-" + UUID.randomUUID(); + long timestamp = Instant.now().toEpochMilli(); + String payload = jsonMessage(marker, "account-updated", "mobile", 750L, timestamp); + + sendMessages(List.of(Message.of(payload))); + + JsonNode result = awaitQuery( + "SELECT * FROM " + TABLE_NAME + " WHERE userId = '" + marker + "' LIMIT 1", + IggyPinotIntegrationTest::hasResultRow); + + assertThat(value(result, "userId").asText()).isEqualTo(marker); + assertThat(value(result, "eventType").asText()).isEqualTo("account-updated"); + assertThat(value(result, "deviceType").asText()).isEqualTo("mobile"); + assertThat(value(result, "duration").asLong()).isEqualTo(750L); + assertThat(value(result, "timestamp").asLong()).isEqualTo(timestamp); + } + + @Test + void shouldIngestMessageBatch() throws Exception { + String marker = "batch-" + UUID.randomUUID(); + int batchSize = 10; + List messages = new ArrayList<>(batchSize); + for (int i = 0; i < batchSize; i++) { + messages.add(Message.of(jsonMessage( + marker + "-" + i, + marker, + i % 2 == 0 ? "desktop" : "mobile", + i * 100L, + Instant.now().toEpochMilli() + i))); + } + + sendMessages(messages); + + JsonNode result = awaitQuery( + "SELECT COUNT(*) FROM " + TABLE_NAME + " WHERE eventType = '" + marker + "'", + response -> firstValue(response).asInt() == batchSize); + + assertThat(firstValue(result).asInt()).isEqualTo(batchSize); + } + + private static void startZookeeper() { + zookeeper = new GenericContainer<>(ZOOKEEPER_IMAGE) + .withNetwork(network) + .withNetworkAliases("zookeeper") + .withExposedPorts(2181) + .withEnv("ZOOKEEPER_CLIENT_PORT", "2181") + .withEnv("ZOOKEEPER_TICK_TIME", "2000") + .waitingFor(Wait.forListeningPort().withStartupTimeout(STARTUP_TIMEOUT)); + zookeeper.start(); + } + + private static void startIggy() { + iggy = new GenericContainer<>(IGGY_IMAGE) + .withImagePullPolicy(PullPolicy.alwaysPull()) + .withNetwork(network) + .withNetworkAliases("iggy") + .withExposedPorts(IGGY_HTTP_PORT, IGGY_TCP_PORT) + .withEnv("IGGY_SYSTEM_LOGGING_LEVEL", "info") + .withEnv("IGGY_TCP_ADDRESS", "0.0.0.0:8090") + .withEnv("IGGY_HTTP_ENABLED", "true") + .withEnv("IGGY_HTTP_ADDRESS", "0.0.0.0:3000") + .withEnv("IGGY_ROOT_USERNAME", "iggy") + .withEnv("IGGY_ROOT_PASSWORD", "iggy") + .withEnv("IGGY_SYSTEM_SHARDING_CPU_ALLOCATION", "1") + .withCreateContainerCmdModifier(cmd -> cmd.getHostConfig() + .withCapAdd(Capability.SYS_NICE) + .withSecurityOpts(List.of("seccomp:unconfined")) + .withUlimits(List.of(new Ulimit("memlock", -1L, -1L)))) + .waitingFor(Wait.forHttp("/") + .forPort(IGGY_HTTP_PORT) + .forStatusCodeMatching(status -> status >= 200 && status < 500) + .withStartupTimeout(STARTUP_TIMEOUT)); + iggy.start(); + } + + private static String iggyHost() { + return USE_EXTERNAL_SERVER ? EXTERNAL_SERVER_HOST : iggy.getHost(); + } + + private static int iggyPort() { + return USE_EXTERNAL_SERVER ? externalTcpPort() : iggy.getMappedPort(IGGY_TCP_PORT); + } + + private static int externalTcpPort() { + String configured = System.getenv("EXTERNAL_TCP_PORT"); + return configured != null ? Integer.parseInt(configured) : IGGY_TCP_PORT; + } + + private static void startPinotController(Path pluginDirectory) { + pinotController = pinotContainer(pluginDirectory) + .withNetworkAliases("pinot-controller") + .withExposedPorts(PINOT_CONTROLLER_PORT) + .withCommand("StartController", "-zkAddress", "zookeeper:2181") + .withEnv("JAVA_OPTS", "-Xms512M -Xmx1G -XX:+UseG1GC -Dplugins.include=iggy-connector") + .waitingFor( + Wait.forHttp("/health").forPort(PINOT_CONTROLLER_PORT).withStartupTimeout(STARTUP_TIMEOUT)); + pinotController.start(); + } + + private static void startPinotBroker() { + pinotBroker = new GenericContainer<>(PINOT_IMAGE) + .withNetwork(network) + .withNetworkAliases("pinot-broker") + .withExposedPorts(PINOT_BROKER_PORT) + .withCommand("StartBroker", "-zkAddress", "zookeeper:2181") + .withEnv("JAVA_OPTS", "-Xms512M -Xmx1G -XX:+UseG1GC") + .waitingFor(Wait.forHttp("/health").forPort(PINOT_BROKER_PORT).withStartupTimeout(STARTUP_TIMEOUT)); + pinotBroker.start(); + } + + private static void startPinotServer(Path pluginDirectory) { + pinotServer = pinotContainer(pluginDirectory) + .withNetworkAliases("pinot-server") + .withExposedPorts(PINOT_SERVER_ADMIN_PORT) + .withCommand("StartServer", "-zkAddress", "zookeeper:2181") + .withEnv("JAVA_OPTS", "-Xms512M -Xmx1G -XX:+UseG1GC -Dplugins.include=iggy-connector") + .waitingFor( + Wait.forHttp("/health").forPort(PINOT_SERVER_ADMIN_PORT).withStartupTimeout(STARTUP_TIMEOUT)); + pinotServer.start(); + } + + private static GenericContainer pinotContainer(Path pluginDirectory) { + return new GenericContainer<>(PINOT_IMAGE) + .withNetwork(network) + .withCopyFileToContainer( + MountableFile.forHostPath(pluginDirectory), + "/opt/pinot/plugins/pinot-stream-ingestion/iggy-connector"); + } + + private static void createIggyResources() { + iggyClient.streams().createStream(STREAM_NAME); + StreamId streamId = StreamId.of(STREAM_NAME); + iggyClient + .topics() + .createTopic(streamId, 2L, CompressionAlgorithm.None, BigInteger.ZERO, BigInteger.ZERO, TOPIC_NAME); + iggyClient.consumerGroups().createConsumerGroup(streamId, TopicId.of(TOPIC_NAME), CONSUMER_GROUP_NAME); + } + + private static void sendMessages(List messages) { + iggyClient + .messages() + .sendMessages(StreamId.of(STREAM_NAME), TopicId.of(TOPIC_NAME), Partitioning.partitionId(0L), messages); + } + + private static String jsonMessage(String userId, String eventType, String deviceType, long duration, long timestamp) + throws IOException { + return OBJECT_MAPPER.writeValueAsString(OBJECT_MAPPER + .createObjectNode() + .put("userId", userId) + .put("eventType", eventType) + .put("deviceType", deviceType) + .put("duration", duration) + .put("timestamp", timestamp)); + } + + private static String tableConfiguration(Path tableConfigurationPath) throws IOException { + JsonNode tableConfiguration = OBJECT_MAPPER.readTree(Files.readString(tableConfigurationPath)); + ObjectNode streamConfigs = + (ObjectNode) tableConfiguration.required("tableIndexConfig").required("streamConfigs"); + streamConfigs.put("stream.iggy.stream.id", STREAM_NAME); + if (USE_EXTERNAL_SERVER) { + streamConfigs.put("stream.iggy.host", TESTCONTAINERS_HOST); + streamConfigs.put("stream.iggy.port", Integer.toString(externalTcpPort())); + } + return OBJECT_MAPPER.writeValueAsString(tableConfiguration); + } + + private static void postControllerResource(String path, String body) throws IOException, InterruptedException { + HttpResponse response = post(pinotController, PINOT_CONTROLLER_PORT, path, body); + if (response.statusCode() < 200 || response.statusCode() >= 300) { + throw new IllegalStateException("Pinot controller request to %s failed with status %d: %s" + .formatted(path, response.statusCode(), response.body())); + } + } + + private static JsonNode awaitQuery(String sql, Predicate success) { + long deadline = System.nanoTime() + QUERY_TIMEOUT.toNanos(); + String lastResponse = "No response received"; + + while (System.nanoTime() < deadline) { + try { + HttpResponse response = post( + pinotBroker, + PINOT_BROKER_PORT, + "/query/sql", + OBJECT_MAPPER.createObjectNode().put("sql", sql).toString()); + lastResponse = "HTTP " + response.statusCode() + ": " + response.body(); + if (response.statusCode() >= 200 && response.statusCode() < 300) { + JsonNode json = OBJECT_MAPPER.readTree(response.body()); + if (hasExceptionCode(json, QueryErrorCode.SQL_PARSING.getId())) { + return fail("Pinot rejected SQL query: %s%nResponse: %s".formatted(sql, response.body())); + } + if (json.path("exceptions").isEmpty() && success.test(json)) { + return json; + } + } + } catch (IOException e) { + lastResponse = e.toString(); + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + throw new IllegalStateException("Interrupted while waiting for Pinot query", e); + } + + try { + Thread.sleep(500); + } catch (InterruptedException e) { + Thread.currentThread().interrupt(); + throw new IllegalStateException("Interrupted while waiting for Pinot query", e); + } + } + + return fail("Pinot query did not reach the expected result within %s.%nSQL: %s%nLast response: %s%n%s" + .formatted(QUERY_TIMEOUT, sql, lastResponse, diagnostics())); + } + + private static HttpResponse post(GenericContainer container, int port, String path, String body) + throws IOException, InterruptedException { + URI uri = URI.create("http://" + container.getHost() + ":" + container.getMappedPort(port) + path); + HttpRequest request = HttpRequest.newBuilder(uri) + .timeout(HTTP_TIMEOUT) + .header("Content-Type", "application/json") + .POST(HttpRequest.BodyPublishers.ofString(body)) + .build(); + return HTTP_CLIENT.send(request, HttpResponse.BodyHandlers.ofString()); + } + + private static boolean hasResultRow(JsonNode response) { + return response.path("resultTable").path("rows").size() > 0; + } + + private static JsonNode firstValue(JsonNode response) { + return response.path("resultTable").path("rows").path(0).path(0); + } + + private static JsonNode value(JsonNode response, String columnName) { + JsonNode columnNames = response.path("resultTable").path("dataSchema").path("columnNames"); + for (int i = 0; i < columnNames.size(); i++) { + if (columnName.equals(columnNames.get(i).asText())) { + return response.path("resultTable").path("rows").path(0).path(i); + } + } + return fail("Pinot result did not contain column '%s': %s".formatted(columnName, response)); + } + + private static boolean hasExceptionCode(JsonNode response, int errorCode) { + for (JsonNode exception : response.path("exceptions")) { + if (exception.path("errorCode").asInt() == errorCode) { + return true; + } + } + return false; + } + + private static Path requiredDirectory(String property) { + String value = System.getProperty(property); + if (value == null) { + throw new IllegalStateException("Missing required system property: " + property); + } + Path directory = Path.of(value); + if (!Files.isDirectory(directory)) { + throw new IllegalStateException("Required directory does not exist: " + directory); + } + return directory; + } + + private static void stop(GenericContainer container) { + if (container != null) { + container.stop(); + } + } + + private static String diagnostics() { + return String.join( + "\n", + iggyDiagnostics(), + logs("Pinot controller", pinotController), + logs("Pinot broker", pinotBroker), + logs("Pinot server", pinotServer)); + } + + private static String iggyDiagnostics() { + if (USE_EXTERNAL_SERVER) { + return "Iggy uses external server at " + iggyHost() + ":" + iggyPort() + "."; + } + return logs("Iggy", iggy); + } + + private static String logs(String name, GenericContainer container) { + if (container == null || !container.isCreated()) { + return name + " was not created."; + } + String logs = container.getLogs(); + int start = Math.max(0, logs.length() - 4_000); + return "=== " + name + " logs ===\n" + logs.substring(start); + } +} From 412014a7b8f33e296d18501597ed996d7927ab24 Mon Sep 17 00:00:00 2001 From: Krishna Vishal Date: Thu, 3 Sep 2026 18:19:55 +0530 Subject: [PATCH 055/182] docs(simulator): add a README for the deterministic harness (#4047) --- core/simulator/README.md | 236 +++++++++++++++++++++++++++++++++++++++ 1 file changed, 236 insertions(+) create mode 100644 core/simulator/README.md diff --git a/core/simulator/README.md b/core/simulator/README.md new file mode 100644 index 0000000000..f1829b6c7c --- /dev/null +++ b/core/simulator/README.md @@ -0,0 +1,236 @@ +# Apache Iggy Simulator + +A deterministic harness that runs an Iggy cluster inside a single thread. The replicas are the real `shard::IggyShard`, running the real message pump, VSR consensus, metadata state machine and partitions. Only the environment under them is swapped: clock, journal storage, superblock, bus and network are in-memory doubles, and every task runs on a cooperative executor whose poll order is drawn from the run's seed. + +A run is a pure function of that seed. Failures print it, and `--seed ` replays the same task interleaving, the same dropped packets, the same crashes, down to the prepare timestamps recorded in the log. Three production consensus bugs have been found this way, each of them reproducible from one seed and a handful of flags. + +The crate is `publish = false`: a tool to run against the workspace, not a dependency. + +## Quick start + +```sh +# 10k ticks of partition-plane traffic, 3 replicas, perfect network +cargo run -p simulator --bin workload-fuzz + +# metadata plane against a hostile network with crashes and restarts +cargo run -p simulator --bin workload-fuzz -- \ + --seed 48 --ticks 4000 --plane metadata --faults heavy \ + --crash-prob 0.02 --restart-prob 0.08 --crash-primary + +# the properties below, pinned as regression tests +cargo test -p simulator + +# scripted demo: send, create and delete a stream, crash a follower, poll +cargo run -p simulator --bin simulator-ui +``` + +Nothing here needs Docker, a server binary, or a network stack. `core/integration` is where real processes are spawned. + +## Real code, simulated environment + +| Layer | What runs | +| --- | --- | +| Shard runtime, router, dispatch | `shard::IggyShard`, one `run_message_pump` task per shard | +| Consensus | `consensus::VsrConsensus`, one group for the metadata plane and one per partition namespace | +| Metadata, partitions | the `metadata` STM, single writer on shard 0 with reader mirrors on the peers; `partitions` on the hash-owning shard | +| Client requests | real wire messages built by `SimClient`; with `--shell`, the server's `on_client_request` path including login and session binding | +| Clock | `SimClock` over executor virtual time, epoch pinned at 2026-01-01 | +| Journal, storage | `SimJournal`, held by the harness so a WAL outlives the replica that wrote it | +| Superblock | `SimSuperblock`; RAM has no torn writes, so it carries none of `PingPongSuperblock`'s framing or checksums | +| Bus and network | `SimOutbox` per replica, drained into `PacketSimulator` | +| Scheduling | `DetExecutor`, uniform seeded pick among ready tasks | +| Snapshots | the real `SnapshotCoordinator` writing real files, only under `Simulator::with_checkpoints` | + +The `simulator` feature on `shard`, `metadata`, `partitions` and `server_common` exposes four seams that only this crate uses: `init_partition` (bypasses the reconciler), `seed_single_partition` (bypasses metadata consensus), `hold_borrow_across_await` (feeds the borrow detector a known-bad case), and a fixed-salt password hash so replicated user metadata replays. None has a production caller, and a `-p iggy-server` build excludes all four. + +## Determinism + +`SimSeeds::derive(seed)` splits the run's seed into one PRNG stream per consumer: + +| Stream | Drives | Why it is separate | +| --- | --- | --- | +| `network` | delays, drops, replays, partitions, clogs | a workload change must not move the packet trace | +| `workload` | which action, which entity, which argument | it shared a stream with `network` once, which is why children are drawn from a parent rather than salted off the seed | +| `executor` | task-poll and timer-fire order | shaking out order dependence must not perturb traffic | +| `entry_shard` | which shard receives each inbound packet | one draw per delivered packet, the highest-rate stream | +| `faults` | crash and restart scheduling | both probabilities at zero draws nothing, so a fault-free run replays bit-identically | +| `swarm` | `--faults swarm`'s parameter draw | correlating the loss probability with the loss events would collapse the explored space to a diagonal | + +Children are drawn from a parent PRNG in field order, so fields are appended to `SimSeeds`, never inserted: an insertion moves every child after it and re-locks every seeded baseline. + +Time never flows on its own. `DetExecutor::advance_time` is the only thing that moves it, sleeps anchor their deadline at creation (matching `compio::time::sleep`, which the pump's re-arm relies on), and timers fire in `(deadline, seq)` order so equal deadlines resolve FIFO. `run_pumps` gives `run_until_stalled` a 100k poll budget; exhausting it means a task is spin-waking, and the panic names the seed and the schedule hash. + +## One tick + +`Simulator::step()` returns the client replies delivered during it, in four phases: + +1. Fire the consensus tick timer (`CONSENSUS_TICK_INTERVAL`: view change, retransmits) and run every pump to quiescence. +2. Deliver ready packets into the shard routers. The receiving shard is a seeded draw and so usually not the owner, which makes the frame take the real router hop, as production does when the coordinator homes a peer connection on an arbitrary shard. Under `--shell` a client packet enters through `deliver_client_request` instead of raw `dispatch`. +3. Run the pumps again over the delivered frames and their loopback follow-ups, then drain each replica's outbox into the network. +4. Advance network time. + +Everything a step produces is on the wire before network time moves, so a reply cannot chain into another delivery inside the same tick. + +## The workload driver + +`workload::run_with_faults` is the loop the fuzzer and most tests share. Per tick it advances the workload clock, steps the fault injector, then resends timed-out requests and handshakes before sampling anything new, since a timed-out request still holds its client's slot. It then samples at most one request per idle client, steps the simulator, classifies the replies, recovers evicted clients, and asserts the per-tick invariants. + +Each workload action has a module under `workload/ops/`, 23 of them, one per `Action` variant, exposing the same seven items: `Input`, `Outcome`, `OUTCOMES`, `sample`, `build_message`, `classify_reply`, `predicted_effect`. Dispatch runs through the `op_dispatch!` macro, so an action without a module is a compile error rather than a silently unsampled op. Variant order in `Action` is part of the determinism contract: append, never insert or reorder. + +A resend reuses the original encoded message verbatim. Rebuilding it would draw a fresh request id, and the metadata client table dedups on that id: a renumbered retry commits a second time instead of returning the cached reply. Resends move to the next replica, so a client whose primary died finds the new one. Each client holds one request in flight (`CLIENT_REQUEST_QUEUE_MAX`). + +## What gets checked + +| Check | When | Catches | +| --- | --- | --- | +| `workload::invariants` | every tick | `commit_offset` or consensus `view` regressing on a live `(replica, namespace)`, in-flight requests over the per-client bound | +| `workload::state_checker` | every tick | two replicas holding different history at an op they both committed, both directions of `(commit_a == commit_b) == (checksum_a == checksum_b)`, and a break in the `parent == checksum` hash link | +| `Simulator::assert_inboxes_drained` | every executor quiescence | a frame sitting in an inbox nobody was woken for, which the next `advance_time` would otherwise drain silently | +| `workload::oracle` | after the drain | a live replica ahead of the leader, disagreement on any committed op two replicas share, a namespace in committed metadata with no host or more than one primary, and on a serial run a predicted `Shadow` that differs from the metadata committed on the leader | + +Agreement is asserted over the committed prefix, not over equal heads. A replica that missed the last commit broadcast, or rejoined a moment ago, is allowed to trail; what it may not do is hold different history at an op it did commit. + +The oracle reports what it actually compared (`ops_compared`, `replicas_compared`, `namespaces_checked`) because a check that ran against an empty chain passes silently. + +## Using it from a test + +```rust +let seed = 0xC0FF_EE00; +let client_id: u128 = 1; +let network = PacketSimulatorOptions { + node_count: 3, + client_count: 1, + seed, + ..PacketSimulatorOptions::default() +}; +let mut sim = Simulator::new(3, std::iter::once(client_id), network); +let client = SimClient::new(client_id); +let ns = IggyNamespace::new(1, 1, 0); +sim.init_partition(ns); +sim.register_client_with_primary(&client); + +let mut options = WorkloadOptions::new(seed, 3, vec![ns]); +options.weights = ActionWeights::new(&[ + (Action::CreateStream, 50), + (Action::DeleteStream, 25), + (Action::SendMessages, 25), +]); +let mut wl = Workload::new(options); + +let replies = workload::run(&mut sim, &mut wl, &[client], 2_000, u64::MAX); +assert!(replies > 0); +assert!(oracle::drive_to_quiesce(&mut sim, &mut wl, 5_000)); +oracle::assert_converged(&sim, &mut wl); +``` + +`MemoryPool::init_pool` has to run first with pooling disabled, since `PooledBuffer::from` panics on an uninitialised pool. +For the dispatch path use `Simulator::with_shards_shell`, where `shell_login` and `seed_stream_topic_partition` replace `register_client_with_primary`. To assert what was injected rather than trusting the probabilities to have fired, pass a caller-owned `FaultInjector` to `run_with_faults`; `replica_crash` and `replica_restart` drive faults directly. + +## workload-fuzz + +```text +workload-fuzz: seed=1 ticks=500 clients=1 replicas=3 plane=Partition faults=None shell=false crash_prob=0 quiesce=true +network: loss=0 replay=0 delay=1..3 partition=None/Symmetric p_partition=0 p_unpartition=0 stability=0/0 clog=0 clog_ticks=0 link_capacity=64 +ran 500 ticks; 50 replies; crashes=0 restarts=0 still down: 0 +coverage: replies_seen=50 replies_unknown=0 committed_rejections=0 samples_none=0 resends=0 denials=2 transients=0 evictions=0 + SendMessages: 28 commits, 0 denied (last status 0), 0 transient (last code 0) + StoreConsumerOffset: 16 commits, 0 denied (last status 0), 0 transient (last code 0) + DeleteConsumerOffset: 4 commits, 2 denied (last status 3021), 0 transient (last code 0) +quiesced and converged (leader-relative; entity oracle: held; evictions=0; ops_compared=1 replicas_compared=2 namespaces_checked=1) +coverage: replies_seen=51 replies_unknown=0 committed_rejections=0 samples_none=0 resends=0 denials=2 transients=0 evictions=0 +commands delivered: Request=52 Prepare=80 PrepareOk=78 Reply=52 Commit=56 RequestPrepares=3 RepairPrepare=4 RepairDone=3 +commands never delivered: Ping Pong PingClient PongClient StartViewChange DoViewChange StartView Eviction ... +workload-fuzz: OK (seed=1) +``` + +The two banner lines are the run's identity: every network parameter is printed, not the interesting subset, because under `--faults swarm` those values are the run. Coverage prints twice, before the quiesce assert and after the drain, so a failed drain still reports what the run managed to do. The command coverage lines answer "did this path get exercised?" by counting at delivery, so a command listed as delivered was seen on the wire rather than inferred from the source. + +An invalid network configuration exits 2 before the simulator starts. Every other failure is a panic, and the hook prints `reproduce with --seed N` ahead of it. + +### Flags + +| Group | Flags | +| --- | --- | +| Shape | `--replicas` (3), `--clients` (1), `--ticks` (10000), `--plane`, `--ack-quorum-ratio` (0.5), `--shell` | +| Faults | `--faults`, `--crash-prob`, `--restart-prob`, `--crash-primary`, `--min-survivors` (a commit quorum) | +| Network overrides | `--packet-loss-prob`, `--replay-prob`, `--one-way-delay-min`, `--one-way-delay-mean`, `--link-capacity`, `--partition-mode`, `--partition-symmetry`, `--partition-prob`, `--unpartition-prob`, `--partition-stability`, `--unpartition-stability`, `--clog-prob`, `--clog-duration-mean` | +| Checkpointing | `--journal-slots`, `--data-dir`, `--reuse-data-dir` | +| Quiesce | `--no-quiesce`, `--heal-before-quiesce` | +| Gates | `--min-commits` (1), `--min-ops-compared` (1), `--require-entity-oracle`, `--require-faults` | +| Durability study | `--restore-partition-frontier` | + +`--plane` picks the op mix: `partition` (writes and consumer offsets), `metadata` (replicated metadata mutations), `mixed` (stream creates over a write-heavy base), `uniform` (every action equally likely, the widest per-tick coverage). + +`--faults` picks a whole network profile and the individual network flags override single fields of it. +Severity costs progress, every lost frame paying a resend timeout: on one namespace with one client, a 3-replica cluster drains roughly 440 replies in 5000 ticks on a perfect network, 240 under `light`, 40 under `heavy`. +Budget ticks accordingly instead of reading a low reply count as a stall. +`--faults swarm` derives every network parameter from the seed, which is what a campaign wants: `none`, `light` and `heavy` are three fixed points, so a thousand seeds against `heavy` is the same network a thousand times. + +The gates exist because a run that commits nothing compares empty against empty and reports success. `--min-commits` and `--min-ops-compared` default to 1 for that reason; `--require-entity-oracle` additionally fails a run whose entity oracle was disarmed by an eviction and never re-armed. + +`--heal-before-quiesce` restarts every crashed replica and stops the network drawing faults before the drain. It is off by default: a drain that fails with a replica still down is how a wedge shows up, and healing first resolves the wedge instead of reporting it. + +### Checkpoints + +A checkpoint is forced by the metadata WAL running low on slots, so `--journal-slots N` is what makes one happen. Without it the journal is unbounded and WAL drain, `snapshot_op` movement, `RangeEvicted` and metadata state transfer are all unreachable however hard the cluster is driven. `N` must exceed `SnapshotCoordinator::CHECKPOINT_MARGIN`, or every commit checkpoints and the run measures that rather than the workload. + +Snapshots are written to a real directory, per process, whose path is printed. It refuses an existing `--data-dir` unless `--reuse-data-dir` says otherwise, because a restart now recovers from `snapshot.bin` and the directory would be input to the new run. + +## Reproducing and diagnosing + +```sh +RUST_BACKTRACE=1 RUST_LOG=shard=debug,consensus=debug \ + cargo run -p simulator --bin workload-fuzz -- --seed 13 --plane metadata \ + --crash-prob 0.01 --restart-prob 0.05 --ticks 30000 +``` + +Omitting `--seed` draws one and prints it on the banner, so an exploratory run stays replayable. The panic hook replaces the default one, so `RUST_BACKTRACE` is honoured explicitly and only when set; a campaign would otherwise be buried in backtraces. `RUST_LOG` selects the server's own tracing, which is the only record of a request dropped after logging, the shape that wedges a client's in-flight slot. + +A failed drain prints every outstanding request with its action, target and attempt count, then one line per replica carrying its commit range, repair barrier, transfer state, primary claim and armed repair session, which usually names the failure without a debugger: + +```text +replica 2: live | metadata ... commit=1407..1414 barrier=1359 transferring=false primary=false repair=..1407@0 +``` + +## Tests + +The suite lives at the end of `src/lib.rs`, with unit tests next to the modules they cover. Names follow `given_X_when_Y_should_Z` where there is a meaningful given; older behavioural names remain where there is not. Every test that touches replies initialises `MemoryPool` first, as above. + +What the suite pins, by example: `workload_replay_is_deterministic` and `multi_shard_replay_is_deterministic` (the seed contract), `committed_metadata_agrees_across_replicas`, `given_advanced_view_when_metadata_replica_restarts_should_recover_view_from_superblock`, `given_superblock_write_fails_when_primary_crashes_should_withhold_votes_and_not_elect`, `checkpointing_cluster_serves_a_chunked_state_transfer`, `shell_detects_partition_borrow_held_across_await`. + +Constructors: `Simulator::new` (one shard per replica), `with_shards` (metadata on shard 0, partitions hash-assigned across all shards), `with_shards_shell` (the same plus the real dispatch handlers), `with_checkpoints`. Multi-shard and shell-on-multi-shard are library-only; `workload-fuzz` always builds one shard per replica. + +## Not modelled + +- Storage faults, beyond two knobs on the superblock. `SimSuperblock::set_fail_writes` and `set_yield_writes` inject a persistent write fault and an fsync-wide suspension point; `MemStorage` under the journal never fails and never tears a write. Partition superblocks are storeless, which leaves partition view recovery untested. +- Segment files. Partition messages live in memory; they survive a restart only because the harness carries `RetainedPartitionState` across the rebuild. +- Partition-plane durability. Production's `load_partition` restores the view alone, so a restarted replica rejoins at op 0, invisible to quorum. `--restore-partition-frontier` looks past that at a system more durable than Iggy is. +- Packet corruption. Packets are delayed, dropped, duplicated, partitioned and clogged, never mangled. +- Multi-shard metadata. Shard 0 owns the only metadata consensus group. +- I/O of any kind except the checkpoint files: no `io_uring`, no sockets, no wall clock. + +## Layout + +```text +src/ +├── lib.rs Simulator: cluster construction, step(), crash/restart, partition seeding, tests +├── executor/ DetExecutor (seeded cooperative scheduler) + virtual clock +├── packet.rs PacketSimulator: delay, loss, replay, partitions, clogs, link capacity +├── network.rs Network: passthrough over PacketSimulator, submit/step/heal +├── bus.rs SimOutbox: per-replica staging that consensus sends into +├── replica.rs Replica type alias and new_shard() wiring, mirroring server bootstrap +├── deps.rs SimClock, MemStorage, SimJournal, SimSuperblock +├── client.rs SimClient: builds real wire requests +├── seeds.rs SimSeeds: one PRNG stream per consumer +├── ready_queue.rs min-heap with reservoir-sampled random ready removal +├── workload/ driver, op modules, shadow, auditor, invariants, state checker, oracle +└── bin/ + ├── workload-fuzz.rs the fuzzer + └── simulator-ui.rs scripted demo run +``` + +## CI + +A change under `core/simulator/**` marks the `rust-simulator` component in `.github/config/components.yml`, which scopes the Rust test tasks to it. The component is split out of `rust-cluster` so simulator-only changes do not trigger the foreign SDK suites. No fuzzing campaign runs in CI today; `workload-fuzz` is driven by hand, and a seed that fails should arrive as a test or as a finding write-up. + +Commits touching this crate take the `simulator` scope: `fix(simulator): ...`. From 7f0bc26d14c67e92d33868cdb805085bd59ffb1c Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Fri, 4 Sep 2026 00:52:31 +0200 Subject: [PATCH 056/182] chore(deps): Bump the node group across 2 directories with 3 updates (#4052) --- examples/node/package-lock.json | 8 +- examples/node/package.json | 2 +- foreign/node/package-lock.json | 142 ++++++++++++++++---------------- foreign/node/package.json | 2 +- 4 files changed, 77 insertions(+), 77 deletions(-) diff --git a/examples/node/package-lock.json b/examples/node/package-lock.json index 980724cb06..fdf8f7883a 100644 --- a/examples/node/package-lock.json +++ b/examples/node/package-lock.json @@ -15,7 +15,7 @@ "devDependencies": { "@types/debug": "^4.1.12", "@types/node": "^22.9.3", - "eslint": "^10.8.1", + "eslint": "^10.9.1", "jiti": "^2.7.0", "tsx": "^4.23.12", "typescript-eslint": "^8.47.0" @@ -1111,9 +1111,9 @@ } }, "node_modules/eslint": { - "version": "10.8.1", - "resolved": "https://registry.npmjs.org/eslint/-/eslint-10.8.1.tgz", - "integrity": "sha512-wqA7W2jbsC/BnV9Iv1UZpKVFkO1AdNoSmYW8NWG4HNOBbkAMvIqDZ27pI2f07dqn583NcIC44ckjAcOXDL1QbQ==", + "version": "10.9.1", + "resolved": "https://registry.npmjs.org/eslint/-/eslint-10.9.1.tgz", + "integrity": "sha512-9VaAkDURekixUQJy0oJYl2DcN6oKMfxay7XzaGYAWQwsb6qfKf+x76R2k1L8kb1boc+FyCAaTA9GmiKaaiaF+A==", "dev": true, "license": "MIT", "workspaces": [ diff --git a/examples/node/package.json b/examples/node/package.json index 013f2657b6..f02594a4ce 100644 --- a/examples/node/package.json +++ b/examples/node/package.json @@ -39,7 +39,7 @@ "devDependencies": { "@types/debug": "^4.1.12", "@types/node": "^22.9.3", - "eslint": "^10.8.1", + "eslint": "^10.9.1", "jiti": "^2.7.0", "tsx": "^4.23.12", "typescript-eslint": "^8.47.0" diff --git a/foreign/node/package-lock.json b/foreign/node/package-lock.json index 967611bc66..69b394e60f 100644 --- a/foreign/node/package-lock.json +++ b/foreign/node/package-lock.json @@ -20,7 +20,7 @@ "@cucumber/cucumber": "13.2.1", "@swc-node/register": "1.12.1", "@types/debug": "4.1.13", - "@types/node": "26.2.0", + "@types/node": "26.4.0", "c8": "^12.0.0", "husky": "9.1.7", "typescript": "6.0.3", @@ -2016,9 +2016,9 @@ "license": "MIT" }, "node_modules/@types/node": { - "version": "26.2.0", - "resolved": "https://registry.npmjs.org/@types/node/-/node-26.2.0.tgz", - "integrity": "sha512-5IviulTZeRNp2vAJ514cc/HUlY5nZ9fCbq9DMyC52BrhFZACo3nI0R7qBxhQmo/d27NFe96ur/b7Wwxklda+kg==", + "version": "26.4.0", + "resolved": "https://registry.npmjs.org/@types/node/-/node-26.4.0.tgz", + "integrity": "sha512-faiGnoIrLH/V8cibOMEAZ8pMw6oXqSukl29ra4mN8GdaB2ZewzeaLj+INpV5N+Z1eKWzY+IzaIZH2EIR6YZRNQ==", "dev": true, "license": "MIT", "dependencies": { @@ -2033,17 +2033,17 @@ "license": "MIT" }, "node_modules/@typescript-eslint/eslint-plugin": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/eslint-plugin/-/eslint-plugin-8.67.0.tgz", - "integrity": "sha512-Un7Heoyj65NREbKAyIrFxeM143NZpExWmy1Nep4DLeQOeLlTeumPjoNKnBrU5D5moWXbPJgRa5Uwcdu0faVNGQ==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/eslint-plugin/-/eslint-plugin-8.68.0.tgz", + "integrity": "sha512-WASHDpCm6qO5jj9g1a+8NiW5+GCkAyLReR56/4VruYmNgfUmqpxOfZ2Yfb8xGfJPWv5Qi6LSD8sXdces3vbp/Q==", "dev": true, "license": "MIT", "dependencies": { "@eslint-community/regexpp": "^4.12.2", - "@typescript-eslint/scope-manager": "8.67.0", - "@typescript-eslint/type-utils": "8.67.0", - "@typescript-eslint/utils": "8.67.0", - "@typescript-eslint/visitor-keys": "8.67.0", + "@typescript-eslint/scope-manager": "8.68.0", + "@typescript-eslint/type-utils": "8.68.0", + "@typescript-eslint/utils": "8.68.0", + "@typescript-eslint/visitor-keys": "8.68.0", "ignore": "^7.0.5", "natural-compare": "^1.4.0", "ts-api-utils": "^2.5.0" @@ -2056,7 +2056,7 @@ "url": "https://opencollective.com/typescript-eslint" }, "peerDependencies": { - "@typescript-eslint/parser": "^8.67.0", + "@typescript-eslint/parser": "^8.68.0", "eslint": "^8.57.0 || ^9.0.0 || ^10.0.0", "typescript": ">=4.8.4 <6.1.0" } @@ -2072,16 +2072,16 @@ } }, "node_modules/@typescript-eslint/parser": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/parser/-/parser-8.67.0.tgz", - "integrity": "sha512-fUBfTuuEulWqX6V8+O3PtScV01tzYYRUDTAirHFKoRAt7nOzoGiPt0M/bB47wWNy0coOOcgEwAMUtBpykMxl6w==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/parser/-/parser-8.68.0.tgz", + "integrity": "sha512-fHq2VC1kpyYfvEcbiMjOpySY4WS7voEp89yAThrHRX5sm9j2lzYppCb2umFMEed4fWcyeLjHxrz0mpjNBaBxMQ==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/scope-manager": "8.67.0", - "@typescript-eslint/types": "8.67.0", - "@typescript-eslint/typescript-estree": "8.67.0", - "@typescript-eslint/visitor-keys": "8.67.0", + "@typescript-eslint/scope-manager": "8.68.0", + "@typescript-eslint/types": "8.68.0", + "@typescript-eslint/typescript-estree": "8.68.0", + "@typescript-eslint/visitor-keys": "8.68.0", "debug": "^4.4.3" }, "engines": { @@ -2097,14 +2097,14 @@ } }, "node_modules/@typescript-eslint/project-service": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/project-service/-/project-service-8.67.0.tgz", - "integrity": "sha512-cvE8c7ulYeXN9fYuszhCeCsbzyVEXuhrRCybnBre7TUmqb5nRmBfQAwCj0O3WJFDeyAZt4VYv51vMCC9LHSdYw==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/project-service/-/project-service-8.68.0.tgz", + "integrity": "sha512-5GQtWZCXFcFYux955pvoS02WLc49pXNlvIxocKjS0clvwo3in1RdlzVKyiqQH9vE5AKWFLTaUgeQkOrTS+0Qxw==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/tsconfig-utils": "^8.67.0", - "@typescript-eslint/types": "^8.67.0", + "@typescript-eslint/tsconfig-utils": "^8.68.0", + "@typescript-eslint/types": "^8.68.0", "debug": "^4.4.3" }, "engines": { @@ -2119,14 +2119,14 @@ } }, "node_modules/@typescript-eslint/scope-manager": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/scope-manager/-/scope-manager-8.67.0.tgz", - "integrity": "sha512-EgvsleTwS4E+WzzSvem8fAUubLwatMNF1B5hHSLQxcvs7q2dtRhGyujHwLJSYlG41niJ7GP24Aha2+0mb1b2kg==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/scope-manager/-/scope-manager-8.68.0.tgz", + "integrity": "sha512-T5eXpcaJNg8bhjHJ8Rjp68Vq/QBteYtTKY8TZqVNPaUbuz0f6jI9t6aDkylwvalpAB9XTTFeFOjrjXAZ3YvmVA==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/types": "8.67.0", - "@typescript-eslint/visitor-keys": "8.67.0" + "@typescript-eslint/types": "8.68.0", + "@typescript-eslint/visitor-keys": "8.68.0" }, "engines": { "node": "^18.18.0 || ^20.9.0 || >=21.1.0" @@ -2137,9 +2137,9 @@ } }, "node_modules/@typescript-eslint/tsconfig-utils": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/tsconfig-utils/-/tsconfig-utils-8.67.0.tgz", - "integrity": "sha512-vV+LUSv5njUWsknE71fqKTlXUva+R76SaeORd6Zojcunk/6DvKFXONU3BrAs2H49mbygUXt6gbYunzwqNwlhdg==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/tsconfig-utils/-/tsconfig-utils-8.68.0.tgz", + "integrity": "sha512-F7zrGQfiJHojPwi8vhxZQC1tWtJzvL74cK/nqri2lk8YUXvYaYwl263xOJ69jDWPUk1hmcdoayFwk9lX09npVw==", "dev": true, "license": "MIT", "engines": { @@ -2154,15 +2154,15 @@ } }, "node_modules/@typescript-eslint/type-utils": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/type-utils/-/type-utils-8.67.0.tgz", - "integrity": "sha512-aVWDXbRmdXO9siTfX4ditQI1T9+zVcNazT48EJCD0v40/9RIFoUgZ05CmGEq9H2gixRpjUn/iplwvlcvutJW/Q==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/type-utils/-/type-utils-8.68.0.tgz", + "integrity": "sha512-X77zqoY1EjeWGs/0JNxeaMfp5C5lIz4Tw8y66F1Ne8Faq6g424sBNYM6xBAqElfGZPLpWS+CZAp0DXyKDzWiHg==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/types": "8.67.0", - "@typescript-eslint/typescript-estree": "8.67.0", - "@typescript-eslint/utils": "8.67.0", + "@typescript-eslint/types": "8.68.0", + "@typescript-eslint/typescript-estree": "8.68.0", + "@typescript-eslint/utils": "8.68.0", "debug": "^4.4.3", "ts-api-utils": "^2.5.0" }, @@ -2179,9 +2179,9 @@ } }, "node_modules/@typescript-eslint/types": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/types/-/types-8.67.0.tgz", - "integrity": "sha512-sBtgslww8nsMYUjhdPBiSyUqSzT8uR6g93A2QXnQC8+cGdjz0CyaOdqHDRJb1AtORbZCNUJBBeFA/tNR2uQmww==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/types/-/types-8.68.0.tgz", + "integrity": "sha512-9RnpsGJjrAllCMefGVVsImJM24YurhC0Q1h4UbvivtvOqXmR/vEJge2OoE++z9m6hyg8T1Q8t5SNT6tHSbrxcg==", "dev": true, "license": "MIT", "engines": { @@ -2193,16 +2193,16 @@ } }, "node_modules/@typescript-eslint/typescript-estree": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/typescript-estree/-/typescript-estree-8.67.0.tgz", - "integrity": "sha512-EKQBCE9yNlRJYm7jdTW5AhDacDUmSwQb0FAJAmK2EKYrNXIsa2vxcSZx6PvJ/dEdI6lS+Y9W+EXckLj0iPFGcw==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/typescript-estree/-/typescript-estree-8.68.0.tgz", + "integrity": "sha512-OKKsD0tYmoNiU5PW2zehO1yO56jYOm1ShYlxon/Z0SJNidAkdVg86eg9ruRuoXf8xfnuWZGbwDsStkoXbZtIIA==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/project-service": "8.67.0", - "@typescript-eslint/tsconfig-utils": "8.67.0", - "@typescript-eslint/types": "8.67.0", - "@typescript-eslint/visitor-keys": "8.67.0", + "@typescript-eslint/project-service": "8.68.0", + "@typescript-eslint/tsconfig-utils": "8.68.0", + "@typescript-eslint/types": "8.68.0", + "@typescript-eslint/visitor-keys": "8.68.0", "debug": "^4.4.3", "minimatch": "^10.2.2", "semver": "^7.7.3", @@ -2221,16 +2221,16 @@ } }, "node_modules/@typescript-eslint/utils": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/utils/-/utils-8.67.0.tgz", - "integrity": "sha512-U9D1FdwEWBwok3hxxSdhclMb0twvt9QnjIQ0VfQ1AiX2epnpSgv2ubVDsayOFyY8K6FX+AQ7E0FKWVG3iKsj1A==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/utils/-/utils-8.68.0.tgz", + "integrity": "sha512-PB5gJMMOg0Q5P1tsgWtEAqQacJXq0qEqRHDX/YJ4FaTMLfZPpHB3gjl2EJuiZyPABxmj4ZQYiY9m1bdAJ5y7tQ==", "dev": true, "license": "MIT", "dependencies": { "@eslint-community/eslint-utils": "^4.9.1", - "@typescript-eslint/scope-manager": "8.67.0", - "@typescript-eslint/types": "8.67.0", - "@typescript-eslint/typescript-estree": "8.67.0" + "@typescript-eslint/scope-manager": "8.68.0", + "@typescript-eslint/types": "8.68.0", + "@typescript-eslint/typescript-estree": "8.68.0" }, "engines": { "node": "^18.18.0 || ^20.9.0 || >=21.1.0" @@ -2245,13 +2245,13 @@ } }, "node_modules/@typescript-eslint/visitor-keys": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/visitor-keys/-/visitor-keys-8.67.0.tgz", - "integrity": "sha512-fkv8dHRDqfGtTHuJeebdrQ7cX6Ad4WAS00rgHh9UGvMycF1mjBfsxry1XsLIFhWZ6Judlh6UdzK+TYlbpCXgnA==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/visitor-keys/-/visitor-keys-8.68.0.tgz", + "integrity": "sha512-YR65gGdGvTUAWLldC3xLOvOzamdGzB4A5/N8rehEaHs3Zvoe39BhgY+u0SPch1OvrVTfLcc55wsSgK2NcnTS/A==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/types": "8.67.0", + "@typescript-eslint/types": "8.68.0", "eslint-visitor-keys": "^5.0.0" }, "engines": { @@ -2744,9 +2744,9 @@ } }, "node_modules/eslint": { - "version": "10.8.1", - "resolved": "https://registry.npmjs.org/eslint/-/eslint-10.8.1.tgz", - "integrity": "sha512-wqA7W2jbsC/BnV9Iv1UZpKVFkO1AdNoSmYW8NWG4HNOBbkAMvIqDZ27pI2f07dqn583NcIC44ckjAcOXDL1QbQ==", + "version": "10.9.1", + "resolved": "https://registry.npmjs.org/eslint/-/eslint-10.9.1.tgz", + "integrity": "sha512-9VaAkDURekixUQJy0oJYl2DcN6oKMfxay7XzaGYAWQwsb6qfKf+x76R2k1L8kb1boc+FyCAaTA9GmiKaaiaF+A==", "dev": true, "license": "MIT", "peer": true, @@ -3926,9 +3926,9 @@ "license": "ISC" }, "node_modules/picomatch": { - "version": "4.0.5", - "resolved": "https://registry.npmjs.org/picomatch/-/picomatch-4.0.5.tgz", - "integrity": "sha512-RvwwcruNjI1ncT5xRakeyS9Lf8lcItv34KD+aif+VH9kduAyfYBipGh12274xtenIPZ119/R9BdTBa8gAwSh0A==", + "version": "4.0.7", + "resolved": "https://registry.npmjs.org/picomatch/-/picomatch-4.0.7.tgz", + "integrity": "sha512-qcJu88Q2IWqJsDD529JKMdwGm/dvInW4HvQnRwiH9JtihJvzGOscDtHE3x1pBKeUOTysQ8kVmLnJ2kJu7yhcGA==", "dev": true, "license": "MIT", "engines": { @@ -4411,16 +4411,16 @@ } }, "node_modules/typescript-eslint": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/typescript-eslint/-/typescript-eslint-8.67.0.tgz", - "integrity": "sha512-S2udFs8tCKEKffuJ4TB1idGUZiXdCPGi3IPBGWXarbLQ5UPXORV8QEVzJ4gCRduURMb5EkpNCdjbk0eDIuI8Yg==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/typescript-eslint/-/typescript-eslint-8.68.0.tgz", + "integrity": "sha512-MHy0Y0ynqeEbx/S45+i/bBssdy3X6KNBfmJAP35GrgtNxu2TQ5K5xsFDhAnmsq1jvpdoZOPG1LGtJo0HWqYCrQ==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/eslint-plugin": "8.67.0", - "@typescript-eslint/parser": "8.67.0", - "@typescript-eslint/typescript-estree": "8.67.0", - "@typescript-eslint/utils": "8.67.0" + "@typescript-eslint/eslint-plugin": "8.68.0", + "@typescript-eslint/parser": "8.68.0", + "@typescript-eslint/typescript-estree": "8.68.0", + "@typescript-eslint/utils": "8.68.0" }, "engines": { "node": "^18.18.0 || ^20.9.0 || >=21.1.0" diff --git a/foreign/node/package.json b/foreign/node/package.json index 17cbfca34c..6722c71616 100644 --- a/foreign/node/package.json +++ b/foreign/node/package.json @@ -61,7 +61,7 @@ "@cucumber/cucumber": "13.2.1", "@swc-node/register": "1.12.1", "@types/debug": "4.1.13", - "@types/node": "26.2.0", + "@types/node": "26.4.0", "c8": "^12.0.0", "husky": "9.1.7", "typescript": "6.0.3", From 02f844f4c5cc368036ffbc4447a7c08ddab3bc7c Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Fri, 4 Sep 2026 01:16:43 +0200 Subject: [PATCH 057/182] chore(deps): Bump the web group across 1 directory with 3 updates (#4051) --- web/package-lock.json | 140 +++++++++++++++++++++--------------------- web/package.json | 6 +- 2 files changed, 73 insertions(+), 73 deletions(-) diff --git a/web/package-lock.json b/web/package-lock.json index 4d8da6fcf1..aa5e326ee9 100644 --- a/web/package-lock.json +++ b/web/package-lock.json @@ -29,10 +29,10 @@ "@sveltejs/vite-plugin-svelte": "^7.3.0", "@types/d3-interpolate": "^3.0.4", "@types/json-bigint": "^1.0.4", - "@types/node": "^26.2.0", + "@types/node": "^26.4.0", "autoprefixer": "^10.5.4", "esbuild": "^0.28.2", - "eslint": "^10.8.1", + "eslint": "^10.9.1", "eslint-config-prettier": "^10.1.8", "eslint-plugin-svelte": "^3.23.0", "globals": "^17.11.0", @@ -47,7 +47,7 @@ "ts-toolbelt": "^9.6.0", "tslib": "^2.8.1", "typescript": "^6.0.3", - "typescript-eslint": "^8.67.0", + "typescript-eslint": "^8.68.0", "vite": "^8.2.2", "zod": "^4.4.3" } @@ -2164,9 +2164,9 @@ "license": "MIT" }, "node_modules/@types/node": { - "version": "26.2.0", - "resolved": "https://registry.npmjs.org/@types/node/-/node-26.2.0.tgz", - "integrity": "sha512-5IviulTZeRNp2vAJ514cc/HUlY5nZ9fCbq9DMyC52BrhFZACo3nI0R7qBxhQmo/d27NFe96ur/b7Wwxklda+kg==", + "version": "26.4.0", + "resolved": "https://registry.npmjs.org/@types/node/-/node-26.4.0.tgz", + "integrity": "sha512-faiGnoIrLH/V8cibOMEAZ8pMw6oXqSukl29ra4mN8GdaB2ZewzeaLj+INpV5N+Z1eKWzY+IzaIZH2EIR6YZRNQ==", "dev": true, "license": "MIT", "dependencies": { @@ -2230,17 +2230,17 @@ } }, "node_modules/@typescript-eslint/eslint-plugin": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/eslint-plugin/-/eslint-plugin-8.67.0.tgz", - "integrity": "sha512-Un7Heoyj65NREbKAyIrFxeM143NZpExWmy1Nep4DLeQOeLlTeumPjoNKnBrU5D5moWXbPJgRa5Uwcdu0faVNGQ==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/eslint-plugin/-/eslint-plugin-8.68.0.tgz", + "integrity": "sha512-WASHDpCm6qO5jj9g1a+8NiW5+GCkAyLReR56/4VruYmNgfUmqpxOfZ2Yfb8xGfJPWv5Qi6LSD8sXdces3vbp/Q==", "dev": true, "license": "MIT", "dependencies": { "@eslint-community/regexpp": "^4.12.2", - "@typescript-eslint/scope-manager": "8.67.0", - "@typescript-eslint/type-utils": "8.67.0", - "@typescript-eslint/utils": "8.67.0", - "@typescript-eslint/visitor-keys": "8.67.0", + "@typescript-eslint/scope-manager": "8.68.0", + "@typescript-eslint/type-utils": "8.68.0", + "@typescript-eslint/utils": "8.68.0", + "@typescript-eslint/visitor-keys": "8.68.0", "ignore": "^7.0.5", "natural-compare": "^1.4.0", "ts-api-utils": "^2.5.0" @@ -2253,7 +2253,7 @@ "url": "https://opencollective.com/typescript-eslint" }, "peerDependencies": { - "@typescript-eslint/parser": "^8.67.0", + "@typescript-eslint/parser": "^8.68.0", "eslint": "^8.57.0 || ^9.0.0 || ^10.0.0", "typescript": ">=4.8.4 <6.1.0" } @@ -2269,16 +2269,16 @@ } }, "node_modules/@typescript-eslint/parser": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/parser/-/parser-8.67.0.tgz", - "integrity": "sha512-fUBfTuuEulWqX6V8+O3PtScV01tzYYRUDTAirHFKoRAt7nOzoGiPt0M/bB47wWNy0coOOcgEwAMUtBpykMxl6w==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/parser/-/parser-8.68.0.tgz", + "integrity": "sha512-fHq2VC1kpyYfvEcbiMjOpySY4WS7voEp89yAThrHRX5sm9j2lzYppCb2umFMEed4fWcyeLjHxrz0mpjNBaBxMQ==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/scope-manager": "8.67.0", - "@typescript-eslint/types": "8.67.0", - "@typescript-eslint/typescript-estree": "8.67.0", - "@typescript-eslint/visitor-keys": "8.67.0", + "@typescript-eslint/scope-manager": "8.68.0", + "@typescript-eslint/types": "8.68.0", + "@typescript-eslint/typescript-estree": "8.68.0", + "@typescript-eslint/visitor-keys": "8.68.0", "debug": "^4.4.3" }, "engines": { @@ -2294,14 +2294,14 @@ } }, "node_modules/@typescript-eslint/project-service": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/project-service/-/project-service-8.67.0.tgz", - "integrity": "sha512-cvE8c7ulYeXN9fYuszhCeCsbzyVEXuhrRCybnBre7TUmqb5nRmBfQAwCj0O3WJFDeyAZt4VYv51vMCC9LHSdYw==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/project-service/-/project-service-8.68.0.tgz", + "integrity": "sha512-5GQtWZCXFcFYux955pvoS02WLc49pXNlvIxocKjS0clvwo3in1RdlzVKyiqQH9vE5AKWFLTaUgeQkOrTS+0Qxw==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/tsconfig-utils": "^8.67.0", - "@typescript-eslint/types": "^8.67.0", + "@typescript-eslint/tsconfig-utils": "^8.68.0", + "@typescript-eslint/types": "^8.68.0", "debug": "^4.4.3" }, "engines": { @@ -2316,14 +2316,14 @@ } }, "node_modules/@typescript-eslint/scope-manager": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/scope-manager/-/scope-manager-8.67.0.tgz", - "integrity": "sha512-EgvsleTwS4E+WzzSvem8fAUubLwatMNF1B5hHSLQxcvs7q2dtRhGyujHwLJSYlG41niJ7GP24Aha2+0mb1b2kg==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/scope-manager/-/scope-manager-8.68.0.tgz", + "integrity": "sha512-T5eXpcaJNg8bhjHJ8Rjp68Vq/QBteYtTKY8TZqVNPaUbuz0f6jI9t6aDkylwvalpAB9XTTFeFOjrjXAZ3YvmVA==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/types": "8.67.0", - "@typescript-eslint/visitor-keys": "8.67.0" + "@typescript-eslint/types": "8.68.0", + "@typescript-eslint/visitor-keys": "8.68.0" }, "engines": { "node": "^18.18.0 || ^20.9.0 || >=21.1.0" @@ -2334,9 +2334,9 @@ } }, "node_modules/@typescript-eslint/tsconfig-utils": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/tsconfig-utils/-/tsconfig-utils-8.67.0.tgz", - "integrity": "sha512-vV+LUSv5njUWsknE71fqKTlXUva+R76SaeORd6Zojcunk/6DvKFXONU3BrAs2H49mbygUXt6gbYunzwqNwlhdg==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/tsconfig-utils/-/tsconfig-utils-8.68.0.tgz", + "integrity": "sha512-F7zrGQfiJHojPwi8vhxZQC1tWtJzvL74cK/nqri2lk8YUXvYaYwl263xOJ69jDWPUk1hmcdoayFwk9lX09npVw==", "dev": true, "license": "MIT", "engines": { @@ -2351,15 +2351,15 @@ } }, "node_modules/@typescript-eslint/type-utils": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/type-utils/-/type-utils-8.67.0.tgz", - "integrity": "sha512-aVWDXbRmdXO9siTfX4ditQI1T9+zVcNazT48EJCD0v40/9RIFoUgZ05CmGEq9H2gixRpjUn/iplwvlcvutJW/Q==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/type-utils/-/type-utils-8.68.0.tgz", + "integrity": "sha512-X77zqoY1EjeWGs/0JNxeaMfp5C5lIz4Tw8y66F1Ne8Faq6g424sBNYM6xBAqElfGZPLpWS+CZAp0DXyKDzWiHg==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/types": "8.67.0", - "@typescript-eslint/typescript-estree": "8.67.0", - "@typescript-eslint/utils": "8.67.0", + "@typescript-eslint/types": "8.68.0", + "@typescript-eslint/typescript-estree": "8.68.0", + "@typescript-eslint/utils": "8.68.0", "debug": "^4.4.3", "ts-api-utils": "^2.5.0" }, @@ -2376,9 +2376,9 @@ } }, "node_modules/@typescript-eslint/types": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/types/-/types-8.67.0.tgz", - "integrity": "sha512-sBtgslww8nsMYUjhdPBiSyUqSzT8uR6g93A2QXnQC8+cGdjz0CyaOdqHDRJb1AtORbZCNUJBBeFA/tNR2uQmww==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/types/-/types-8.68.0.tgz", + "integrity": "sha512-9RnpsGJjrAllCMefGVVsImJM24YurhC0Q1h4UbvivtvOqXmR/vEJge2OoE++z9m6hyg8T1Q8t5SNT6tHSbrxcg==", "devOptional": true, "license": "MIT", "engines": { @@ -2390,16 +2390,16 @@ } }, "node_modules/@typescript-eslint/typescript-estree": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/typescript-estree/-/typescript-estree-8.67.0.tgz", - "integrity": "sha512-EKQBCE9yNlRJYm7jdTW5AhDacDUmSwQb0FAJAmK2EKYrNXIsa2vxcSZx6PvJ/dEdI6lS+Y9W+EXckLj0iPFGcw==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/typescript-estree/-/typescript-estree-8.68.0.tgz", + "integrity": "sha512-OKKsD0tYmoNiU5PW2zehO1yO56jYOm1ShYlxon/Z0SJNidAkdVg86eg9ruRuoXf8xfnuWZGbwDsStkoXbZtIIA==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/project-service": "8.67.0", - "@typescript-eslint/tsconfig-utils": "8.67.0", - "@typescript-eslint/types": "8.67.0", - "@typescript-eslint/visitor-keys": "8.67.0", + "@typescript-eslint/project-service": "8.68.0", + "@typescript-eslint/tsconfig-utils": "8.68.0", + "@typescript-eslint/types": "8.68.0", + "@typescript-eslint/visitor-keys": "8.68.0", "debug": "^4.4.3", "minimatch": "^10.2.2", "semver": "^7.7.3", @@ -2418,16 +2418,16 @@ } }, "node_modules/@typescript-eslint/utils": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/utils/-/utils-8.67.0.tgz", - "integrity": "sha512-U9D1FdwEWBwok3hxxSdhclMb0twvt9QnjIQ0VfQ1AiX2epnpSgv2ubVDsayOFyY8K6FX+AQ7E0FKWVG3iKsj1A==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/utils/-/utils-8.68.0.tgz", + "integrity": "sha512-PB5gJMMOg0Q5P1tsgWtEAqQacJXq0qEqRHDX/YJ4FaTMLfZPpHB3gjl2EJuiZyPABxmj4ZQYiY9m1bdAJ5y7tQ==", "dev": true, "license": "MIT", "dependencies": { "@eslint-community/eslint-utils": "^4.9.1", - "@typescript-eslint/scope-manager": "8.67.0", - "@typescript-eslint/types": "8.67.0", - "@typescript-eslint/typescript-estree": "8.67.0" + "@typescript-eslint/scope-manager": "8.68.0", + "@typescript-eslint/types": "8.68.0", + "@typescript-eslint/typescript-estree": "8.68.0" }, "engines": { "node": "^18.18.0 || ^20.9.0 || >=21.1.0" @@ -2442,13 +2442,13 @@ } }, "node_modules/@typescript-eslint/visitor-keys": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/@typescript-eslint/visitor-keys/-/visitor-keys-8.67.0.tgz", - "integrity": "sha512-fkv8dHRDqfGtTHuJeebdrQ7cX6Ad4WAS00rgHh9UGvMycF1mjBfsxry1XsLIFhWZ6Judlh6UdzK+TYlbpCXgnA==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/@typescript-eslint/visitor-keys/-/visitor-keys-8.68.0.tgz", + "integrity": "sha512-YR65gGdGvTUAWLldC3xLOvOzamdGzB4A5/N8rehEaHs3Zvoe39BhgY+u0SPch1OvrVTfLcc55wsSgK2NcnTS/A==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/types": "8.67.0", + "@typescript-eslint/types": "8.68.0", "eslint-visitor-keys": "^5.0.0" }, "engines": { @@ -3012,9 +3012,9 @@ } }, "node_modules/eslint": { - "version": "10.8.1", - "resolved": "https://registry.npmjs.org/eslint/-/eslint-10.8.1.tgz", - "integrity": "sha512-wqA7W2jbsC/BnV9Iv1UZpKVFkO1AdNoSmYW8NWG4HNOBbkAMvIqDZ27pI2f07dqn583NcIC44ckjAcOXDL1QbQ==", + "version": "10.9.1", + "resolved": "https://registry.npmjs.org/eslint/-/eslint-10.9.1.tgz", + "integrity": "sha512-9VaAkDURekixUQJy0oJYl2DcN6oKMfxay7XzaGYAWQwsb6qfKf+x76R2k1L8kb1boc+FyCAaTA9GmiKaaiaF+A==", "dev": true, "license": "MIT", "workspaces": [ @@ -5042,16 +5042,16 @@ } }, "node_modules/typescript-eslint": { - "version": "8.67.0", - "resolved": "https://registry.npmjs.org/typescript-eslint/-/typescript-eslint-8.67.0.tgz", - "integrity": "sha512-S2udFs8tCKEKffuJ4TB1idGUZiXdCPGi3IPBGWXarbLQ5UPXORV8QEVzJ4gCRduURMb5EkpNCdjbk0eDIuI8Yg==", + "version": "8.68.0", + "resolved": "https://registry.npmjs.org/typescript-eslint/-/typescript-eslint-8.68.0.tgz", + "integrity": "sha512-MHy0Y0ynqeEbx/S45+i/bBssdy3X6KNBfmJAP35GrgtNxu2TQ5K5xsFDhAnmsq1jvpdoZOPG1LGtJo0HWqYCrQ==", "dev": true, "license": "MIT", "dependencies": { - "@typescript-eslint/eslint-plugin": "8.67.0", - "@typescript-eslint/parser": "8.67.0", - "@typescript-eslint/typescript-estree": "8.67.0", - "@typescript-eslint/utils": "8.67.0" + "@typescript-eslint/eslint-plugin": "8.68.0", + "@typescript-eslint/parser": "8.68.0", + "@typescript-eslint/typescript-estree": "8.68.0", + "@typescript-eslint/utils": "8.68.0" }, "engines": { "node": "^18.18.0 || ^20.9.0 || >=21.1.0" diff --git a/web/package.json b/web/package.json index 3217b3ec3d..fad235d1b2 100644 --- a/web/package.json +++ b/web/package.json @@ -22,10 +22,10 @@ "@sveltejs/vite-plugin-svelte": "^7.3.0", "@types/d3-interpolate": "^3.0.4", "@types/json-bigint": "^1.0.4", - "@types/node": "^26.2.0", + "@types/node": "^26.4.0", "autoprefixer": "^10.5.4", "esbuild": "^0.28.2", - "eslint": "^10.8.1", + "eslint": "^10.9.1", "eslint-config-prettier": "^10.1.8", "eslint-plugin-svelte": "^3.23.0", "globals": "^17.11.0", @@ -40,7 +40,7 @@ "ts-toolbelt": "^9.6.0", "tslib": "^2.8.1", "typescript": "^6.0.3", - "typescript-eslint": "^8.67.0", + "typescript-eslint": "^8.68.0", "vite": "^8.2.2", "zod": "^4.4.3" }, From 3cfc7df0d80bec2daf52fdbd3ffdbf16b587ca27 Mon Sep 17 00:00:00 2001 From: Maciej Modzelewski Date: Fri, 4 Sep 2026 10:14:46 +0200 Subject: [PATCH 058/182] test(java): boot the edge server consistently in both Java suites (#4054) The Pinot suite could not start the current apache/iggy:edge image: the server now refuses a wildcard bind without an advertised address (#3923) and the suite never set one. It also re-pulled the image on every run, which made it far slower than the SDK suite, and its readiness probe accepted any HTTP status below 500 from /. The SDK suite in turn still claimed the published image shipped the legacy server and only waited for the ports to open. Both suites now configure the container the same way: advertise an address clients can dial, let the server use every core, and gate on /ping before running tests. The Pinot suite reuses the locally cached image like the SDK suite does. --- .../pinot/IggyPinotIntegrationTest.java | 21 +++++------- .../iggy/client/BaseIntegrationTest.java | 32 +++++++++---------- 2 files changed, 23 insertions(+), 30 deletions(-) diff --git a/foreign/java/external-processors/iggy-connector-pinot/src/test/java/org/apache/iggy/connector/pinot/IggyPinotIntegrationTest.java b/foreign/java/external-processors/iggy-connector-pinot/src/test/java/org/apache/iggy/connector/pinot/IggyPinotIntegrationTest.java index 5d78873da0..af6744147d 100644 --- a/foreign/java/external-processors/iggy-connector-pinot/src/test/java/org/apache/iggy/connector/pinot/IggyPinotIntegrationTest.java +++ b/foreign/java/external-processors/iggy-connector-pinot/src/test/java/org/apache/iggy/connector/pinot/IggyPinotIntegrationTest.java @@ -38,7 +38,6 @@ import org.testcontainers.containers.GenericContainer; import org.testcontainers.containers.Network; import org.testcontainers.containers.wait.strategy.Wait; -import org.testcontainers.images.PullPolicy; import org.testcontainers.utility.DockerImageName; import org.testcontainers.utility.MountableFile; @@ -63,7 +62,7 @@ class IggyPinotIntegrationTest { - // The Java SDK speaks VSR, so use the same VSR-capable image as its integration tests. + // Mirrors the container in the SDK's BaseIntegrationTest so both suites exercise the same server. private static final DockerImageName IGGY_IMAGE = DockerImageName.parse("apache/iggy:edge"); private static final DockerImageName PINOT_IMAGE = DockerImageName.parse( Objects.requireNonNull(System.getProperty("iggy.pinot.image"), "Missing iggy.pinot.image system property")); @@ -74,6 +73,7 @@ class IggyPinotIntegrationTest { private static final int PINOT_CONTROLLER_PORT = 9000; private static final int PINOT_BROKER_PORT = 8099; private static final int PINOT_SERVER_ADMIN_PORT = 8097; + private static final String IGGY_NETWORK_ALIAS = "iggy"; private static final String EXTERNAL_SERVER_HOST = "127.0.0.1"; private static final String TESTCONTAINERS_HOST = "host.testcontainers.internal"; private static final boolean USE_EXTERNAL_SERVER = System.getenv("USE_EXTERNAL_SERVER") != null; @@ -220,25 +220,20 @@ private static void startZookeeper() { private static void startIggy() { iggy = new GenericContainer<>(IGGY_IMAGE) - .withImagePullPolicy(PullPolicy.alwaysPull()) .withNetwork(network) - .withNetworkAliases("iggy") + .withNetworkAliases(IGGY_NETWORK_ALIAS) .withExposedPorts(IGGY_HTTP_PORT, IGGY_TCP_PORT) - .withEnv("IGGY_SYSTEM_LOGGING_LEVEL", "info") - .withEnv("IGGY_TCP_ADDRESS", "0.0.0.0:8090") - .withEnv("IGGY_HTTP_ENABLED", "true") - .withEnv("IGGY_HTTP_ADDRESS", "0.0.0.0:3000") .withEnv("IGGY_ROOT_USERNAME", "iggy") .withEnv("IGGY_ROOT_PASSWORD", "iggy") - .withEnv("IGGY_SYSTEM_SHARDING_CPU_ALLOCATION", "1") + .withEnv("IGGY_TCP_ADDRESS", "0.0.0.0:" + IGGY_TCP_PORT) + .withEnv("IGGY_HTTP_ADDRESS", "0.0.0.0:" + IGGY_HTTP_PORT) + .withEnv("IGGY_NODE_ADVERTISED_ADDRESS", IGGY_NETWORK_ALIAS) + .withEnv("IGGY_SYSTEM_SHARDING_CPU_ALLOCATION", "all") .withCreateContainerCmdModifier(cmd -> cmd.getHostConfig() .withCapAdd(Capability.SYS_NICE) .withSecurityOpts(List.of("seccomp:unconfined")) .withUlimits(List.of(new Ulimit("memlock", -1L, -1L)))) - .waitingFor(Wait.forHttp("/") - .forPort(IGGY_HTTP_PORT) - .forStatusCodeMatching(status -> status >= 200 && status < 500) - .withStartupTimeout(STARTUP_TIMEOUT)); + .waitingFor(Wait.forHttp("/ping").forPort(IGGY_HTTP_PORT).withStartupTimeout(STARTUP_TIMEOUT)); iggy.start(); } diff --git a/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/BaseIntegrationTest.java b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/BaseIntegrationTest.java index 1e4fced76e..c25345ab19 100644 --- a/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/BaseIntegrationTest.java +++ b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/BaseIntegrationTest.java @@ -27,30 +27,30 @@ import org.slf4j.Logger; import org.slf4j.LoggerFactory; import org.testcontainers.containers.GenericContainer; +import org.testcontainers.containers.wait.strategy.Wait; import org.testcontainers.junit.jupiter.Testcontainers; import org.testcontainers.utility.DockerImageName; import java.util.List; /** - * Base for integration tests. The SDK speaks the VSR wire protocol, so the - * server under test must support it. + * Base for integration tests. By default the server under test is the + * {@code apache/iggy:edge} image started via testcontainers. The tag tracks + * master, so refresh the local copy with {@code docker pull apache/iggy:edge} + * whenever the SDK's wire protocol moves ahead of it. * - *

With {@code USE_EXTERNAL_SERVER} set, tests target an externally - * started VSR server on localhost, running standalone (single-node) mode. - * Start it from the repo root: + *

With {@code USE_EXTERNAL_SERVER} set, tests instead target an externally + * started server on localhost, running standalone (single-node) mode. Start + * it from the repo root: *

{@code
  * IGGY_ROOT_USERNAME=iggy IGGY_ROOT_PASSWORD=iggy cargo run --bin iggy-server
  * }
- * - *

Otherwise a server container is started via testcontainers. This works - * once the published image ships a VSR-capable server; until then the - * external-server mode is the only one that can pass. */ @Testcontainers public abstract class BaseIntegrationTest { protected static GenericContainer iggyServer; + private static final DockerImageName IGGY_IMAGE = DockerImageName.parse("apache/iggy:edge"); private static final String LOCALHOST_IP = "127.0.0.1"; private static final int HTTP_PORT = 3000; private static final int TCP_PORT = 8090; @@ -80,22 +80,20 @@ public static String serverHost() { static void setupContainer() { ResourceLeakDetector.setLevel(ResourceLeakDetector.Level.PARANOID); if (!USE_EXTERNAL_SERVER) { - // The published image still ships the legacy server, which does - // not speak the VSR wire protocol, so tests against this - // container fail until a VSR-capable server is released. Use - // USE_EXTERNAL_SERVER until then. log.info("Starting Iggy Server Container..."); - iggyServer = new GenericContainer<>(DockerImageName.parse("apache/iggy:edge")) + iggyServer = new GenericContainer<>(IGGY_IMAGE) .withExposedPorts(HTTP_PORT, TCP_PORT) .withEnv("IGGY_ROOT_USERNAME", "iggy") .withEnv("IGGY_ROOT_PASSWORD", "iggy") - .withEnv("IGGY_TCP_ADDRESS", "0.0.0.0:8090") - .withEnv("IGGY_HTTP_ADDRESS", "0.0.0.0:3000") - .withEnv("IGGY_NODE_ADVERTISED_ADDRESS", "127.0.0.1") + .withEnv("IGGY_TCP_ADDRESS", "0.0.0.0:" + TCP_PORT) + .withEnv("IGGY_HTTP_ADDRESS", "0.0.0.0:" + HTTP_PORT) + .withEnv("IGGY_NODE_ADVERTISED_ADDRESS", LOCALHOST_IP) + .withEnv("IGGY_SYSTEM_SHARDING_CPU_ALLOCATION", "all") .withCreateContainerCmdModifier(cmd -> cmd.getHostConfig() .withCapAdd(Capability.SYS_NICE) .withSecurityOpts(List.of("seccomp:unconfined")) .withUlimits(List.of(new Ulimit("memlock", -1L, -1L)))) + .waitingFor(Wait.forHttp("/ping").forPort(HTTP_PORT)) .withLogConsumer(frame -> System.out.print(frame.getUtf8String())); iggyServer.start(); } else { From b44724213546db9d68f808ef5061eb1ae51e4464 Mon Sep 17 00:00:00 2001 From: Grzegorz Koszyk <112548209+numinnex@users.noreply.github.com> Date: Fri, 4 Sep 2026 12:59:54 +0200 Subject: [PATCH 059/182] fix(cluster): serve metadata reads at or above the client's own writes (#4024) A client that commits a metadata write and then re-homes its session onto a backup can be served the pre-write state. auth.rs already documents the gap: the session epoch is the register's commit op, and on a backup that forwarded the proposal the local applied commit still lags it. A backup applies committed ops only as its commit walk advances, and nothing tied a read to the op the client's own write committed. Metadata reads now gate on the connection's committed watermark. One applied-frontier counter per process advances after every metadata apply and is shared with every shard, which for the first time gives a shard without consensus the applied position it had no way to observe. The watermark comes from the commit field replies already carry, seeded at bind from the session epoch. The fast path is a single atomic load with no awaits, so the shared-nothing read path is unchanged; a lagging node parks briefly, then fails the read retryable rather than answering stale. Over HTTP this closes the forwarded-register case, where a healthy backup forwards the register so the bound epoch can exceed the local frontier. It does not close the case where forwarding is active: the follower relays the write, its handler never runs, so the node that later serves the read holds no session and no watermark. Closing that needs the serving primary's commit op to travel back to the reading node, for instance a response header beside the view the forward middleware already relays. That is additive but touches every control-plane write response, so it is left out here and documented at the gate. The BDD delete-then-get steps now assert "not the stream we deleted" instead of "nothing at this id". The server hands a deleted stream's numeric id straight to the next create, so once scenarios share a server the old assertion cannot hold, and removing the polling loop without this would have left the spec flaky for an unrelated reason. --- .../apache/iggy/bdd/BasicMessagingSteps.java | 10 +- bdd/rust/tests/steps/streams.rs | 31 +- core/integration/tests/server/http_client.rs | 75 ++- .../tests/server/http_read_your_writes.rs | 219 +++++++++ .../tests/server/http_view_header.rs | 71 +-- core/integration/tests/server/mod.rs | 3 + core/metadata/src/applied_frontier.rs | 400 +++++++++++++++ core/metadata/src/impls/metadata.rs | 80 ++- core/metadata/src/lib.rs | 5 + core/server/src/boot/mod.rs | 17 +- core/server/src/boot/threads.rs | 3 + core/server/src/dispatch/mod.rs | 88 +++- core/server/src/dispatch/reads.rs | 419 +++++++++++++++- core/server/src/dispatch/submit.rs | 116 +++++ core/server/src/http.rs | 1 + core/server/src/http/error.rs | 26 +- core/server/src/http/extractor.rs | 4 +- core/server/src/http/forward.rs | 59 ++- core/server/src/http/handlers.rs | 131 +++-- core/server/src/http/reads.rs | 271 ++++++++++- core/server/src/http/reply.rs | 30 +- core/server/src/http/session.rs | 12 +- core/server/src/http/state.rs | 128 ++++- core/server/src/http/submit.rs | 27 +- core/server/src/lib.rs | 1 + core/server/src/responses.rs | 130 +++-- core/server/src/session_manager.rs | 144 +++++- core/shard/src/lib.rs | 6 +- core/shard/src/metrics.rs | 52 ++ core/simulator/src/client.rs | 88 ++-- core/simulator/src/lib.rs | 457 ++++++++++++++++++ core/simulator/src/replica.rs | 8 +- 32 files changed, 2793 insertions(+), 319 deletions(-) create mode 100644 core/integration/tests/server/http_read_your_writes.rs create mode 100644 core/metadata/src/applied_frontier.rs diff --git a/bdd/java/src/test/java/org/apache/iggy/bdd/BasicMessagingSteps.java b/bdd/java/src/test/java/org/apache/iggy/bdd/BasicMessagingSteps.java index 0fc05c28c0..f5b9a8c19f 100644 --- a/bdd/java/src/test/java/org/apache/iggy/bdd/BasicMessagingSteps.java +++ b/bdd/java/src/test/java/org/apache/iggy/bdd/BasicMessagingSteps.java @@ -142,8 +142,16 @@ public void deleteStreamByNumericId() { @Then("getting the stream by its numeric ID should return no stream") public void getStreamReturnsNoStream() { + // The assertion is "not the stream we deleted", not "nothing at this id": + // the server hands out the lowest free stream id, so once these scenarios + // run concurrently against one server a fresh create can legitimately + // occupy the deleted stream's id. Named, so a missing pre-value fails + // here instead of making the comparison below vacuously true. + assertNotNull(context.lastStreamName, "Stream should have been created"); Optional stream = getClient().streams().getStream(context.lastStreamId); - assertTrue(stream.isEmpty(), "Deleted stream should not be returned"); + assertTrue( + stream.isEmpty() || !stream.get().name().equals(context.lastStreamName), + "Deleted stream should not be returned"); } @When("I create a topic with name {string} in stream {int} with {int} partitions") diff --git a/bdd/rust/tests/steps/streams.rs b/bdd/rust/tests/steps/streams.rs index ba1248f3fa..f94ef4027c 100644 --- a/bdd/rust/tests/steps/streams.rs +++ b/bdd/rust/tests/steps/streams.rs @@ -18,11 +18,6 @@ use crate::common::global_context::GlobalContext; use cucumber::{given, then, when}; use iggy::prelude::{Identifier, StreamClient, StreamUpdateOptions}; -use std::time::Duration; -use tokio::time::{Instant, sleep}; - -const METADATA_CONVERGENCE_TIMEOUT: Duration = Duration::from_secs(2); -const METADATA_CONVERGENCE_POLL: Duration = Duration::from_millis(10); #[given("I have no streams in the system")] pub async fn given_no_streams(world: &mut GlobalContext) { @@ -128,18 +123,20 @@ pub async fn when_delete_stream_by_numeric_id(world: &mut GlobalContext) { #[then("getting the stream by its numeric ID should return no stream")] pub async fn then_get_stream_returns_no_stream(world: &mut GlobalContext) { - let deadline = Instant::now() + METADATA_CONVERGENCE_TIMEOUT; - loop { - get_stream_by_numeric_id(world).await; - if world.last_stream_name.is_none() { - return; - } - assert!( - Instant::now() < deadline, - "Deleted stream should not be returned after {METADATA_CONVERGENCE_TIMEOUT:?}" - ); - sleep(METADATA_CONVERGENCE_POLL).await; - } + // Read before the get overwrites it. The assertion is "not the stream we + // deleted", not "nothing at this id": `IdSlab::insert` hands out the lowest + // free key and these scenarios share one server, so a concurrent create can + // legitimately occupy the deleted stream's id. + let deleted = world + .last_stream_name + .clone() + .expect("Stream should have been created"); + get_stream_by_numeric_id(world).await; + assert_ne!( + world.last_stream_name.as_ref(), + Some(&deleted), + "Deleted stream should not be returned" + ); } async fn create_stream(world: &mut GlobalContext, stream_name: &str) { diff --git a/core/integration/tests/server/http_client.rs b/core/integration/tests/server/http_client.rs index 2ba24ac32b..8e51eb9425 100644 --- a/core/integration/tests/server/http_client.rs +++ b/core/integration/tests/server/http_client.rs @@ -16,16 +16,20 @@ // under the License. //! Shared HTTP transport plumbing for the server REST suites (`http_vsr`, -//! `http_rbac`): one authenticated `reqwest` session with the login-retry gate -//! and the generic verb helpers. Each suite keeps its own request shapes and -//! assertions as extension methods on [`HttpClient`], so the wire-contract and -//! listener-behavior separation between the suites stays intact. +//! `http_rbac`): one authenticated `reqwest` session with the login-retry gate, +//! the generic verb helpers, and the cluster-shaped helpers the multi-node +//! suites share (which node is the leader, which is a follower, and the retry +//! a follower needs before it can resolve the primary). Each suite keeps its +//! own request shapes and assertions as extension methods on [`HttpClient`], so +//! the wire-contract and listener-behavior separation between the suites stays +//! intact. +use std::future::Future; use std::time::{Duration, Instant}; use iggy::prelude::*; use integration::harness::TestHarness; -use reqwest::Response; +use reqwest::{Response, StatusCode}; use serde_json::{Value, json}; use tokio::time::sleep; @@ -196,6 +200,67 @@ impl HttpClient { } } +/// `http://host:port` of a harness node's HTTP listener. +pub fn node_url(harness: &TestHarness, node: usize) -> String { + let addr = harness.node(node).http_addr().expect("node http address"); + format!("http://{addr}") +} + +/// Harness indexes of the node the roster marks `Leader` and of one it marks +/// `Follower`. The harness emits the roster in node order, so a roster +/// position is a harness index. Every node reads `Follower` until shard 0 +/// publishes its first view, so the roster is polled within the shared +/// warmup budget until it marks a leader. +pub async fn leader_and_follower(harness: &TestHarness) -> (usize, usize) { + let client = harness + .root_client_for_node(0) + .await + .expect("connect to node 0"); + let deadline = Instant::now() + LOGIN_TIMEOUT; + loop { + let metadata = client + .get_cluster_metadata() + .await + .expect("get cluster metadata"); + let position = + |role: ClusterNodeRole| metadata.nodes.iter().position(|node| node.role == role); + if let (Some(leader), Some(follower)) = ( + position(ClusterNodeRole::Leader), + position(ClusterNodeRole::Follower), + ) { + return (leader, follower); + } + assert!( + Instant::now() < deadline, + "the roster did not mark a leader within {LOGIN_TIMEOUT:?}, got {metadata}" + ); + sleep(LOGIN_RETRY_INTERVAL).await; + } +} + +/// Repeat `request` while the follower answers 503, which it does until it +/// can resolve the primary from its own view; bounded by the shared warmup +/// budget, as cluster_metadata_vsr does. A 503 is the retry-safe class: the +/// request provably never entered a pipeline. +pub async fn until_primary_resolved(request: F) -> Response +where + F: Fn() -> Fut, + Fut: Future, +{ + let deadline = Instant::now() + LOGIN_TIMEOUT; + loop { + let response = request().await; + if response.status() != StatusCode::SERVICE_UNAVAILABLE { + return response; + } + assert!( + Instant::now() < deadline, + "follower did not resolve the primary within {LOGIN_TIMEOUT:?}" + ); + sleep(LOGIN_RETRY_INTERVAL).await; + } +} + /// Extract the JWT from a successful login response. pub async fn access_token(response: Response) -> String { let identity: IdentityInfo = response.json().await.expect("decode IdentityInfo"); diff --git a/core/integration/tests/server/http_read_your_writes.rs b/core/integration/tests/server/http_read_your_writes.rs new file mode 100644 index 0000000000..5fa54d65f8 --- /dev/null +++ b/core/integration/tests/server/http_read_your_writes.rs @@ -0,0 +1,219 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Read-your-writes over the REST listener, end to end: an unqualified read +//! must never answer below the metadata op the same caller was already told +//! committed. +//! +//! The window is a node that HANDED OUT a committed op it has not applied yet. +//! On the REST plane that node is a follower running a `Register`: the session +//! its first authenticated request mints is forwarded to the primary +//! (`dispatch::submit_register_local_or_forward`), so the follower answers with +//! an epoch its own commit walk can still be behind. Everything the metadata +//! group committed below that epoch is therefore state the caller has been +//! promised and this follower may not have applied. +//! +//! Each round seeds a stream through the primary over TCP, then authenticates +//! on the follower, which binds an epoch above that create. The follower's next +//! read is the assertion: it must not answer from before the stream existed. +//! Logout is the authenticated request that binds it, deliberately - it tears +//! the session entry down again, so the floor the read waits on has to outlive +//! the session that established it. +//! +//! The seeding runs over TCP rather than the primary's own REST listener so the +//! only HTTP sessions in play are the follower's: a second long-lived REST +//! session would be competing for VSR client ids with the fresh register each +//! round mints, which is a different subject. +//! +//! What each case can and cannot prove, stated plainly: +//! +//! - The wiring is proved deterministically. The forwarding case reads the +//! serving primary's applied op off the relayed write's `iggy-applied-op` +//! header and then off the follower's OWN answer to the following read: the +//! second can only be at or above the first if this node held that read until +//! its commit walk covered the op the primary handed the caller. That is the +//! invariant itself, in op numbers, not a proxy for it. +//! - Whether the gate actually PARKED is invisible from outside, and on a fast +//! local cluster the follower often applies within the same tick, so neither +//! case can force the park. The park, the wake and the expiry are pinned +//! deterministically next to the gate instead - `dispatch::reads`, +//! `metadata::applied_frontier`, and the simulator's +//! `metadata_read_frontier_tests`, which cuts replication to force the lag +//! the binary plane sees. +//! - A stale answer still fails loudly if the race does land: a 404, a list +//! missing the stream, or an applied op below the one the caller was handed. +//! A 503 fails too, on purpose - that is what the gate answers when the +//! follower never catches up inside its budget. + +use iggy::prelude::*; +use integration::iggy_harness; +use reqwest::{Response, StatusCode}; +use serde_json::{Value, json}; + +use crate::server::http_client::{ + HttpClient, leader_and_follower, node_url, until_primary_resolved, +}; + +/// Seed / read-back rounds. More than one because the lag is a race the test +/// cannot force: each round re-runs it with the follower's commit walk in a +/// different position relative to the epoch it just handed out. +const ROUNDS: u32 = 4; + +/// The serving node's applied metadata op, as stamped on every authenticated +/// success. A response without it is a contract violation, not an absence: the +/// relay records this value as the caller's read-your-writes floor, so a +/// missing header would silently reopen the stale read it exists to close. +fn applied_op(response: &Response) -> u64 { + response + .headers() + .get("iggy-applied-op") + .expect("every authenticated success carries iggy-applied-op") + .to_str() + .expect("iggy-applied-op must be ASCII") + .parse() + .expect("iggy-applied-op must be an op number") +} + +/// The `name` of every stream in a `GET /streams` list body. +fn stream_names(body: &Value) -> Vec { + body.as_array() + .expect("the stream list is a JSON array") + .iter() + .map(|stream| { + stream["name"] + .as_str() + .expect("every stream carries a name") + .to_owned() + }) + .collect() +} + +/// Three nodes, the smallest cluster with a quorum, one shard each so every +/// request is served by shard 0 where the metadata consensus lives. No +/// `http.jwt` secret and no `cluster.auth`: bearers are node-local and +/// follower-to-primary forwarding is off, so the follower answers its own +/// requests instead of relaying them (see `http_view_header`, which pins both +/// halves of that switch). +#[iggy_harness(cluster_nodes = 3, server(system.sharding.cpu_allocation = "0..1"))] +async fn given_a_follower_when_its_register_binds_a_committed_epoch_should_not_read_below_it( + harness: &TestHarness, +) { + let (leader, follower) = leader_and_follower(harness).await; + let seeder = harness + .root_client_for_node(leader) + .await + .expect("connect to the primary"); + + for round in 0..ROUNDS { + let stream = format!("read-your-writes-{round}"); + seeder + .create_stream(&stream) + .await + .expect("the primary must commit the stream this round reads back"); + + // Fresh bearer, then one authenticated request on the follower: it is + // what forwards the `Register` and binds an epoch above the create. + let http = HttpClient::login_root_no_redirect(node_url(harness, follower)).await; + let logout = until_primary_resolved(|| http.delete("/users/logout")).await; + assert_eq!( + logout.status(), + StatusCode::NO_CONTENT, + "the follower must bind and end a forwarded session" + ); + + // The list read resolves nothing, so it cannot 404 its way into looking + // correct: a stale answer here is a short list. + let read = http.get("/streams").await; + assert_eq!(read.status(), StatusCode::OK, "the stream list must serve"); + let names = stream_names(&read.json().await.expect("the stream list is JSON")); + assert!( + names.contains(&stream), + "the follower listed streams from before the epoch it had just handed out: {names:?}" + ); + + // The entity read, where the stale answer is a 404 instead. + let read = http.get(&format!("/streams/{stream}")).await; + assert_eq!( + read.status(), + StatusCode::OK, + "the follower answered a read below the epoch it had just handed out" + ); + let body: Value = read.json().await.expect("stream details are JSON"); + assert_eq!( + body["name"].as_str(), + Some(stream.as_str()), + "the read answered with another stream's state" + ); + } +} + +/// The same three nodes WITH cluster-wide bearer key material, which switches +/// follower-to-primary forwarding on. That is the other half of the window and +/// the one with a deterministic proof: the write is relayed, so this follower +/// never runs the local write path and learns what the caller was told +/// committed only from the primary's `iggy-applied-op`. Its own answer to the +/// read that follows carries its own applied op, which must be at or above it. +#[iggy_harness( + cluster_nodes = 3, + server( + system.sharding.cpu_allocation = "0..1", + http.jwt.encoding_secret = "0123456789abcdef0123456789abcdef", + http.jwt.decoding_secret = "0123456789abcdef0123456789abcdef" + ) +)] +async fn given_a_forwarding_follower_when_it_relays_a_write_should_not_read_below_the_primary_op( + harness: &TestHarness, +) { + let (_leader, follower) = leader_and_follower(harness).await; + let http = HttpClient::login_root_no_redirect(node_url(harness, follower)).await; + + for round in 0..ROUNDS { + let stream = format!("relayed-read-your-writes-{round}"); + let create = json!({ "name": stream }); + let created = until_primary_resolved(|| http.post_json("/streams", &create)).await; + assert_eq!( + created.status(), + StatusCode::OK, + "the follower must relay the write and answer the primary's reply" + ); + // The primary applies a metadata op before it replies, so this is at or + // above the op the caller now holds. + let committed_at = applied_op(&created); + assert!( + committed_at > 0, + "the relayed reply carried no applied op, so this node recorded no floor" + ); + + let read = http.get("/streams").await; + assert_eq!( + read.status(), + StatusCode::OK, + "the follower must serve the read rather than refuse it" + ); + assert!( + applied_op(&read) >= committed_at, + "the follower answered a read at op {} after handing the caller {committed_at}: \ + the relayed floor was not recorded or not waited for", + applied_op(&read) + ); + let names = stream_names(&read.json().await.expect("the stream list is JSON")); + assert!( + names.contains(&stream), + "the follower listed streams without the one it had just relayed: {names:?}" + ); + } +} diff --git a/core/integration/tests/server/http_view_header.rs b/core/integration/tests/server/http_view_header.rs index 8ed175a9b2..ca21921cdd 100644 --- a/core/integration/tests/server/http_view_header.rs +++ b/core/integration/tests/server/http_view_header.rs @@ -20,17 +20,13 @@ //! withheld wherever it could reach a caller that proved no credential. Raw //! `reqwest`, because the header itself is the contract. -use std::future::Future; -use std::time::Instant; - -use iggy::prelude::*; -use integration::harness::TestHarness; use integration::iggy_harness; use reqwest::{Response, StatusCode}; use serde_json::json; -use tokio::time::sleep; -use crate::server::http_client::{HttpClient, LOGIN_RETRY_INTERVAL, LOGIN_TIMEOUT}; +use crate::server::http_client::{ + HttpClient, leader_and_follower, node_url, until_primary_resolved, +}; const VIEW_HEADER: &str = "iggy-view"; @@ -91,67 +87,6 @@ async fn given_the_ping_route_when_it_succeeds_should_omit_the_iggy_view_header( ); } -/// `http://host:port` of a harness node's HTTP listener. -fn node_url(harness: &TestHarness, node: usize) -> String { - let addr = harness.node(node).http_addr().expect("node http address"); - format!("http://{addr}") -} - -/// Harness indexes of the node the roster marks `Leader` and of one it marks -/// `Follower`. The harness emits the roster in node order, so a roster -/// position is a harness index. Every node reads `Follower` until shard 0 -/// publishes its first view, so the roster is polled within the shared -/// warmup budget until it marks a leader. -async fn leader_and_follower(harness: &TestHarness) -> (usize, usize) { - let client = harness - .root_client_for_node(0) - .await - .expect("connect to node 0"); - let deadline = Instant::now() + LOGIN_TIMEOUT; - loop { - let metadata = client - .get_cluster_metadata() - .await - .expect("get cluster metadata"); - let position = - |role: ClusterNodeRole| metadata.nodes.iter().position(|node| node.role == role); - if let (Some(leader), Some(follower)) = ( - position(ClusterNodeRole::Leader), - position(ClusterNodeRole::Follower), - ) { - return (leader, follower); - } - assert!( - Instant::now() < deadline, - "the roster did not mark a leader within {LOGIN_TIMEOUT:?}, got {metadata}" - ); - sleep(LOGIN_RETRY_INTERVAL).await; - } -} - -/// Repeat `request` while the follower answers 503, which it does until it -/// can resolve the primary from its own view; bounded by the shared warmup -/// budget, as cluster_metadata_vsr does. A 503 is the retry-safe class: the -/// request provably never entered a pipeline. -async fn until_primary_resolved(request: F) -> Response -where - F: Fn() -> Fut, - Fut: Future, -{ - let deadline = Instant::now() + LOGIN_TIMEOUT; - loop { - let response = request().await; - if response.status() != StatusCode::SERVICE_UNAVAILABLE { - return response; - } - assert!( - Instant::now() < deadline, - "follower did not resolve the primary within {LOGIN_TIMEOUT:?}" - ); - sleep(LOGIN_RETRY_INTERVAL).await; - } -} - /// The view the primary stamps on its own successful response: the value a /// follower's redirect or relay must agree with. async fn primary_view(primary: &HttpClient) -> u64 { diff --git a/core/integration/tests/server/mod.rs b/core/integration/tests/server/mod.rs index a1da26b1e3..1c6411a8aa 100644 --- a/core/integration/tests/server/mod.rs +++ b/core/integration/tests/server/mod.rs @@ -51,6 +51,9 @@ mod http_tls; // The iggy-view response header: on authenticated success and redirect // responses only, never on errors or /ping, relayed from the primary. mod http_view_header; +// An unqualified REST read must not answer below what the same caller was told +// committed, on the node that accepted the write and has not applied it yet. +mod http_read_your_writes; // Binary GetClusterMetadata must serve the real roster from a VSR cluster. mod cluster_metadata_vsr; // A declared node.advertised_address outranks the bind address a diff --git a/core/metadata/src/applied_frontier.rs b/core/metadata/src/applied_frontier.rs new file mode 100644 index 0000000000..b759a3ff57 --- /dev/null +++ b/core/metadata/src/applied_frontier.rs @@ -0,0 +1,400 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! The node-wide applied metadata frontier and the wait a read parks on. + +use std::future::Future; +use std::pin::Pin; +use std::sync::Mutex; +use std::sync::PoisonError; +use std::sync::atomic::{AtomicU64, AtomicUsize, Ordering}; +use std::task::{Context, Poll, Waker}; +use std::time::Duration; + +/// Highest metadata op whose apply has been PUBLISHED on this node, shared by +/// every shard, plus the wakers of the reads waiting for it to reach them. +/// +/// `consensus.commit_min()` answers the same question but exists only on shard +/// 0, so a read served by a peer shard has no way to tell whether the node +/// caught up to an op its client already saw committed. One process-wide cell +/// does, for one `Acquire` load on the read fast path. +/// +/// The op is `Release`-written right after each apply's `publish()` and +/// `Acquire`-read; observing `>= op` therefore happens-after that publish, so a +/// following left-right `enter()` is guaranteed to see the op. `fetch_max` +/// rather than `store` because three writers move it -- the commit loop, the +/// recovery seed, and a state-transfer install -- and only monotonicity makes +/// their order irrelevant. +/// +/// A `std::sync::Mutex` guards the waiter list, not a `tokio` one: it is taken +/// and dropped inside [`Self::advance`] and inside one `poll`, never across an +/// `.await`, and it has to be `Sync` because the writer is shard 0's thread +/// while the sleepers are on every shard. `parked` keeps [`Self::advance`] off +/// that lock entirely in the normal case, where no read is waiting. +/// +/// Carries the read gates' budget too: it is config-derived (see the server's +/// `dispatch::reads`), and this cell is the one object minted before the shards +/// spawn that all of them can read, including peer shards with no consensus. +#[derive(Debug)] +pub struct AppliedFrontier { + op: AtomicU64, + /// Waits currently registered, so an advancing commit can skip the lock. + /// `Release` on the way in and `Acquire` on the way out, NOT relaxed: the + /// count is what tells [`Self::advance`] a waiter exists at all, so if it + /// reads zero it must be guaranteed that no registration it has to wake + /// happened before it. + parked: AtomicUsize, + waiters: Mutex, + read_budget: Duration, +} + +impl Default for AppliedFrontier { + fn default() -> Self { + Self::new(Self::DEFAULT_READ_BUDGET) + } +} + +/// Registered waits, keyed by an id so a re-poll can refresh its own waker and +/// a dropped wait (an HTTP client that disconnected mid-read) can remove it. +#[derive(Debug, Default)] +struct Waiters { + next_id: u64, + entries: Vec, +} + +#[derive(Debug)] +struct Waiter { + id: u64, + target: u64, + waker: Waker, +} + +impl AppliedFrontier { + /// The budget a held read gets when nothing sizes it from config: six + /// commit broadcasts at the built-in `COMMIT_MESSAGE_TICKS` interval. + /// + /// The server always overrides this from `[cluster] + /// commit_broadcast_interval`, and a test asserts the two agree at the + /// config default; this is what the simulator and unit fixtures get. + pub const DEFAULT_READ_BUDGET: Duration = Duration::from_millis(3_000); + + /// A frontier at zero whose held reads get `read_budget` before they fail + /// retryable. + #[must_use] + pub const fn new(read_budget: Duration) -> Self { + Self { + op: AtomicU64::new(0), + parked: AtomicUsize::new(0), + waiters: Mutex::new(Waiters { + next_id: 0, + entries: Vec::new(), + }), + read_budget, + } + } + + /// How long a read may be held before it must fail retryable. Read by both + /// planes' gates, which arm their own timers with it. + #[must_use] + pub const fn read_budget(&self) -> Duration { + self.read_budget + } + + /// Highest metadata op this NODE has applied and published. + #[must_use] + pub fn get(&self) -> u64 { + self.op.load(Ordering::Acquire) + } + + /// Publish `op` as applied and wake every read waiting at or below it. + /// Monotone, so a lower value is a no-op and wakes nobody. + /// + /// Must run AFTER the apply's `publish()` and, on the commit path, in the + /// same await-free region as `advance_commit_min`: a reader that sees the + /// frontier must be guaranteed to see the op's effects. + pub fn advance(&self, op: u64) { + if self.op.fetch_max(op, Ordering::Release) >= op { + return; + } + // The normal case is a commit with nobody waiting on it, and this runs + // on the commit path: skip the lock rather than contend it per op. + if self.parked.load(Ordering::Acquire) == 0 { + return; + } + let woken = { + let mut waiters = self.waiters.lock().unwrap_or_else(PoisonError::into_inner); + let mut woken = Vec::new(); + waiters.entries.retain(|waiter| { + if waiter.target > op { + return true; + } + woken.push(waiter.waker.clone()); + false + }); + self.parked.store(waiters.entries.len(), Ordering::Release); + woken + }; + // Woken OUTSIDE the guard: each waker's task deregisters through this + // same mutex, so waking under it would hand every reader a lock the + // commit path is still holding. + for waker in woken { + waker.wake(); + } + } + + /// A future that completes once the frontier covers `target`. + /// + /// Event-driven, not polled: the commit path wakes it, so a read resumes on + /// the commit it was waiting for rather than on the next tick. It has no + /// deadline of its own -- the caller composes one, because the two read + /// planes measure time differently (the shard bus timer, virtual under the + /// simulator, against `compio::time`). + pub const fn reached(&self, target: u64) -> Reached<'_> { + Reached { + frontier: self, + target, + id: None, + } + } + + /// Register a wait for `target` under `existing` (its id from an earlier + /// poll, if any), or report that the frontier already covers it. + /// + /// One lock acquisition, released with the return: the re-read under it is + /// what closes the race with [`Self::advance`], which bumps the op and only + /// then takes this lock, so an advance landing between a caller's load and + /// this call is one the wait would otherwise sleep through. + fn register(&self, existing: Option, target: u64, waker: &Waker) -> Registered { + let mut waiters = self.waiters.lock().unwrap_or_else(PoisonError::into_inner); + if self.get() >= target { + return Registered::Ready; + } + // Refresh rather than stack: a re-poll may arrive under a different + // task (a `select` re-driven elsewhere), and two entries for one wait + // would leak the first. + let id = if let Some(waiter) = + existing.and_then(|id| waiters.entries.iter_mut().find(|waiter| waiter.id == id)) + { + waiter.waker.clone_from(waker); + waiter.id + } else { + let id = waiters.next_id; + waiters.next_id += 1; + waiters.entries.push(Waiter { + id, + target, + waker: waker.clone(), + }); + id + }; + self.parked.store(waiters.entries.len(), Ordering::Release); + drop(waiters); + Registered::Waiting(id) + } + + /// Drop the registration `id`, if it is still listed. + fn deregister(&self, id: u64) { + let mut waiters = self.waiters.lock().unwrap_or_else(PoisonError::into_inner); + waiters.entries.retain(|waiter| waiter.id != id); + self.parked.store(waiters.entries.len(), Ordering::Release); + } + + /// Waits currently parked. For tests: a wait that outlives its future is a + /// leaked waker. + #[must_use] + pub fn waiting(&self) -> usize { + self.waiters + .lock() + .unwrap_or_else(PoisonError::into_inner) + .entries + .len() + } +} + +/// Outcome of registering a wait: nothing to wait for, or the id the wait is +/// listed under. +#[derive(Debug, Clone, Copy)] +enum Registered { + Ready, + Waiting(u64), +} + +/// The wait [`AppliedFrontier::reached`] hands out. Deregisters on drop, so a +/// cancelled read (a dropped axum handler future, a closed socket) leaves no +/// waker behind. +#[derive(Debug)] +pub struct Reached<'a> { + frontier: &'a AppliedFrontier, + target: u64, + id: Option, +} + +impl Future for Reached<'_> { + type Output = (); + + fn poll(self: Pin<&mut Self>, context: &mut Context<'_>) -> Poll<()> { + let this = self.get_mut(); + // Ahead of the registration, so the steady state (a frontier already + // past the target) costs one load and never touches the lock. + if this.frontier.get() >= this.target { + return Poll::Ready(()); + } + match this + .frontier + .register(this.id, this.target, context.waker()) + { + Registered::Ready => Poll::Ready(()), + Registered::Waiting(id) => { + this.id = Some(id); + Poll::Pending + } + } + } +} + +impl Drop for Reached<'_> { + fn drop(&mut self) { + if let Some(id) = self.id { + self.frontier.deregister(id); + } + } +} + +#[cfg(test)] +mod tests { + use super::AppliedFrontier; + use std::future::Future; + use std::pin::pin; + use std::sync::Arc; + use std::task::{Context, Poll}; + + /// A frontier already at or above the target is the steady state, and it + /// must cost neither a wake nor a registration: a gate that parked here + /// would put a commit's latency on every metadata read in the cluster. + #[test] + fn given_a_frontier_at_the_target_when_waiting_should_be_ready_without_registering() { + let frontier = AppliedFrontier::default(); + frontier.advance(7); + + let waker = futures::task::noop_waker(); + let mut context = Context::from_waker(&waker); + for target in [0, 7] { + let mut wait = pin!(frontier.reached(target)); + assert_eq!(wait.as_mut().poll(&mut context), Poll::Ready(())); + } + assert_eq!(frontier.waiting(), 0, "a ready wait registers nothing"); + } + + /// The wait is what replaces the poll loop, so the advance has to be what + /// wakes it: park below the target, advance past it, and the wait must be + /// woken and complete without any intervening timer. + #[test] + fn given_a_parked_wait_when_the_frontier_advances_should_wake_and_complete() { + let frontier = AppliedFrontier::default(); + let woken = Arc::new(std::sync::atomic::AtomicBool::new(false)); + let waker = futures::task::waker(Arc::new(FlagWaker { + woken: Arc::clone(&woken), + })); + let mut context = Context::from_waker(&waker); + + let mut wait = pin!(frontier.reached(9)); + assert_eq!(wait.as_mut().poll(&mut context), Poll::Pending); + assert_eq!(frontier.waiting(), 1); + + // Below the target: no wake, still parked. + frontier.advance(8); + assert!(!woken.load(std::sync::atomic::Ordering::Acquire)); + assert_eq!(frontier.waiting(), 1); + + frontier.advance(9); + assert!( + woken.load(std::sync::atomic::Ordering::Acquire), + "the advance past the target must wake the parked read" + ); + assert_eq!( + frontier.waiting(), + 0, + "a woken wait is off the list, so a later advance re-wakes nothing" + ); + assert_eq!(wait.as_mut().poll(&mut context), Poll::Ready(())); + } + + /// The commit path must not wake a reader while holding the lock that + /// reader needs to deregister, and it must not take that lock at all when + /// nothing is parked - a commit with no waiting read is the normal case. + #[test] + fn given_an_advance_when_waking_should_not_hold_the_waiter_lock() { + let frontier = Arc::new(AppliedFrontier::default()); + frontier.advance(4); + assert_eq!(frontier.waiting(), 0, "an advance with nobody parked"); + + // This waker re-enters the frontier's own lock, which is what a woken + // reader does when it deregisters. Waking under the guard therefore + // deadlocks this test outright rather than merely contending. + let waker = futures::task::waker(Arc::new(ReentrantWaker { + frontier: Arc::clone(&frontier), + })); + let mut context = Context::from_waker(&waker); + let mut wait = pin!(frontier.reached(9)); + assert_eq!(wait.as_mut().poll(&mut context), Poll::Pending); + + frontier.advance(9); + assert_eq!(frontier.waiting(), 0, "the woken wait is off the list"); + assert_eq!(wait.as_mut().poll(&mut context), Poll::Ready(())); + } + + /// A re-poll must not stack a second registration, and a dropped wait must + /// take its waker with it: an HTTP read is cancelled whenever its client + /// disconnects mid-wait, and a leaked waker would be a leak per disconnect. + #[test] + fn given_a_repolled_wait_when_dropped_should_leave_no_registration() { + let frontier = AppliedFrontier::default(); + let waker = futures::task::noop_waker(); + let mut context = Context::from_waker(&waker); + { + let mut wait = pin!(frontier.reached(9)); + assert_eq!(wait.as_mut().poll(&mut context), Poll::Pending); + assert_eq!(wait.as_mut().poll(&mut context), Poll::Pending); + assert_eq!(frontier.waiting(), 1, "a re-poll refreshes, never stacks"); + } + assert_eq!(frontier.waiting(), 0, "a dropped wait deregisters"); + } + + /// Stands in for a woken reader: waking takes the frontier's waiter lock, + /// exactly as the woken task's `deregister` does. + struct ReentrantWaker { + frontier: Arc, + } + + impl futures::task::ArcWake for ReentrantWaker { + fn wake_by_ref(arc_self: &Arc) { + let _parked = arc_self.frontier.waiting(); + } + } + + struct FlagWaker { + woken: Arc, + } + + impl futures::task::ArcWake for FlagWaker { + fn wake_by_ref(arc_self: &Arc) { + arc_self + .woken + .store(true, std::sync::atomic::Ordering::Release); + } + } +} diff --git a/core/metadata/src/impls/metadata.rs b/core/metadata/src/impls/metadata.rs index 0c19eb1563..dc2399ecf2 100644 --- a/core/metadata/src/impls/metadata.rs +++ b/core/metadata/src/impls/metadata.rs @@ -16,6 +16,7 @@ // under the License. use crate::MuxStateMachine; +use crate::applied_frontier::AppliedFrontier; use crate::stm::authz::gated_apply; use crate::stm::consumer_group::CompleteConsumerGroupRevocationRequest; use crate::stm::snapshot::{ @@ -66,6 +67,7 @@ use std::cell::{Cell, RefCell}; use std::mem::size_of; use std::path::Path; use std::rc::Rc; +use std::sync::Arc; use tracing::{debug, error, info, warn}; fn freeze_client_reply( @@ -769,6 +771,28 @@ pub struct IggyMetadata { /// whole snapshot on shard 0's pump, and hands each requester its own /// multi-MB copy. transfer_offer_cache: RefCell>>, + /// Highest metadata op whose apply has been PUBLISHED on this node, plus + /// the reads parked on it. Shared by every shard; see + /// [`AppliedFrontier`] for the ordering and the wake contract. + applied_frontier: Arc, +} + +impl IggyMetadata, J, S, M, SB> +where + B: MessageBus, +{ + /// Resume the applied frontier where recovery left the state machine. + /// + /// Recovery replays the committed WAL prefix before any listener binds, so + /// without this the frontier reads zero on a rebooted node and every read + /// whose caller holds a pre-restart commit parks until its deadline. A + /// no-op on a peer shard, which owns no consensus and shares shard 0's + /// cell. + pub fn seed_applied_frontier_from_consensus(&self) { + if let Some(consensus) = self.consensus.as_ref() { + self.advance_applied_frontier(consensus.commit_min()); + } + } } impl IggyMetadata @@ -808,11 +832,42 @@ where commit_notifier: RefCell::new(None), client_table_frontier: Cell::new(0), transfer_offer_cache: RefCell::new(None), + applied_frontier: Arc::default(), } } } impl IggyMetadata { + /// Share one process-wide applied frontier with every other shard. + /// + /// Consumed at construction rather than swapped in later: a shard that + /// served a read against its own private cell would gate on a number that + /// never moves. Shard 0 mints the cell in bootstrap, before any shard is + /// built, and hands each shard a clone. + #[must_use] + pub fn with_applied_frontier(mut self, applied_frontier: Arc) -> Self { + self.applied_frontier = applied_frontier; + self + } + + /// The node-wide applied frontier, readable on every shard. Reads gate on + /// it so a client cannot be served state older than a write it already saw + /// acked, and park on its wait when it is behind. + #[must_use] + pub const fn applied_frontier(&self) -> &Arc { + &self.applied_frontier + } + + /// Publish `op` as applied and wake the reads waiting at or below it. + /// Monotone, so a lower value is a no-op. + /// + /// Must run AFTER the apply's `publish()` and, on the commit path, in the + /// same await-free region as `advance_commit_min`: a reader that sees the + /// frontier must be guaranteed to see the op's effects. + pub fn advance_applied_frontier(&self, op: u64) { + self.applied_frontier.advance(op); + } + /// Slot capacity of the LIVE client table, i.e. the largest transferred /// table this replica can absorb. /// @@ -1474,10 +1529,15 @@ impl std::error::Error for StateTransferUnavailable { /// invites a caller to treat a completed install as a failure and redo it. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub struct InstallOutcome { - /// The receiver's new applied frontier, `max(snapshot_seq, + /// The receiver's applied position after the install, `max(snapshot_seq, /// local_applied)`. These differ whenever a serving peer offered a /// snapshot BEHIND this replica and the local state machine was kept. - pub applied_frontier: u64, + /// + /// Named apart from [`IggyMetadata::applied_frontier`] deliberately: that + /// one is the node-wide cell the read gate consults, which the install + /// raises to `snapshot_seq` alone, so the two carry different numbers + /// exactly when a behind-snapshot was kept. + pub installed_frontier: u64, /// Whether the transferred checkpoint's `(checkpoint_op, checksum)` /// pairing reached the durable superblock. /// @@ -1611,7 +1671,7 @@ where /// reconciler's periodic full diff against the committed STM, which /// reads the restored state on its next tick. /// - /// Returns an [`InstallOutcome`]: the new applied frontier, plus whether + /// Returns an [`InstallOutcome`]: the installed frontier, plus whether /// the transferred checkpoint's pairing reached the durable superblock. /// /// # Errors @@ -1839,6 +1899,7 @@ where if snapshot_seq > consensus.sequencer().current_sequence() { consensus.sequencer().set_sequence(snapshot_seq); } + self.advance_applied_frontier(snapshot_seq); } // Before the superblock write, so the durable record carries the frontier // this transfer just established rather than the pre-transfer one. @@ -1868,7 +1929,7 @@ where } Ok(InstallOutcome { - applied_frontier: snapshot_seq.max(local_applied), + installed_frontier: snapshot_seq.max(local_applied), pairing_durable, }) } @@ -2827,6 +2888,10 @@ where reply }; consensus.advance_commit_min(prepare_header.op); + // Paired with the counter bump, and before the reply leaves: a + // client that holds this reply may re-home onto any shard and read, + // and the read gate admits it only once the frontier covers the op. + self.advance_applied_frontier(prepare_header.op); emit_sim_event(SimEventKind::OperationCommitted, &event); // Fire subscriber BEFORE wire send. Slot already updated @@ -3574,6 +3639,7 @@ where prepare, ); consensus.advance_commit_min(op); + self.advance_applied_frontier(op); debug!("commit_journal: committed op={op}"); } } @@ -5674,6 +5740,12 @@ mod tests { journal_handle.header(1).is_some() && journal_handle.header(2).is_some(), "ops at or below the floor stay for the walk and tail repair" ); + assert_eq!( + md.applied_frontier().get(), + SNAPSHOT_SEQ, + "the snapshot IS ops up to its sequence applied, so the read gate has \ + to admit reads at the floor the install jumped to" + ); } #[compio::test] diff --git a/core/metadata/src/lib.rs b/core/metadata/src/lib.rs index 4d4d24003c..bc771641ce 100644 --- a/core/metadata/src/lib.rs +++ b/core/metadata/src/lib.rs @@ -17,10 +17,15 @@ //! Iggy metadata module +pub mod applied_frontier; pub mod impls; pub mod permissioner; pub mod stm; +// The node-wide read frontier the read gates park on; minted by the bootstrap +// before any shard exists, so it is named outside `impls::`. +pub use applied_frontier::{AppliedFrontier, Reached}; + // Re-export IggyMetadata for use in other modules pub use impls::metadata::{ BoundSession, CommitNotifier, IggyMetadata, MetadataSubmitError, StateTransferOffer, diff --git a/core/server/src/boot/mod.rs b/core/server/src/boot/mod.rs index 841b3e3ab0..5053c8f8dc 100644 --- a/core/server/src/boot/mod.rs +++ b/core/server/src/boot/mod.rs @@ -56,6 +56,7 @@ use crate::boot::threads::{ }; use crate::boot::topology::{RosterCells, resolve_tcp_topology}; use crate::dispatch::partition::make_partition_read_handler; +use crate::dispatch::reads::read_frontier_budget; use crate::dispatch::session_ops::warm_dummy_password_hash; use crate::dispatch::submit::make_metadata_submit_handler; use crate::dispatch::{ @@ -76,9 +77,9 @@ use journal::{Journal, JournalHandle}; use message_bus::replica::handshake::ReplicaHandshakeCtx; use message_bus::transports::tls::install_default_crypto_provider; use message_bus::{IggyMessageBus, ReplicaOwnerTable}; -use metadata::ReplicaIdentity; use metadata::impls::metadata::StreamsFrontend; use metadata::impls::recovery::recover; +use metadata::{AppliedFrontier, ReplicaIdentity}; use server_common::Message; use server_common::bootstrap::create_directories; use server_common::fs_utils::remove_dir_all; @@ -304,6 +305,13 @@ pub fn bootstrap( let mut shard_threads: Vec<(u16, thread::JoinHandle>)> = Vec::with_capacity(shards_count); let roster_cells = RosterCells::default(); + // Shared applied-metadata frontier: shard 0's commit path advances it and + // wakes the reads parked on it, every shard's read gate reads it. Minted + // here, before any shard exists, because a shard holding a private cell + // would gate reads on a number nothing moves - and it carries the held + // reads' budget, which is sized from the configured commit-broadcast + // cadence and which a peer shard has no other way to learn. + let metadata_applied_frontier = Arc::new(AppliedFrontier::new(read_frontier_budget(&config))); // Every shard's metric handles, minted before the threads spawn: each // shard bumps its own entry, and shard 0's HTTP scrape endpoint registers // the whole set (counters are Arc-backed, so cross-thread reads see the @@ -344,6 +352,7 @@ pub fn bootstrap( }; let roster_cells_for_shard = roster_cells.clone(); + let applied_frontier_for_shard = Arc::clone(&metadata_applied_frontier); let shard_metrics_for_shard = shard_metrics_all.clone(); let handle = match thread::Builder::new() .name(format!("shard-{shard_id}")) @@ -362,6 +371,7 @@ pub fn bootstrap( barrier_for_shard, owner_table_for_shard, roster_cells_for_shard, + applied_frontier_for_shard, shard_metrics_for_shard, ) }) { @@ -429,6 +439,7 @@ async fn shard_main( barrier: BootstrapBarrier, owner_table: Arc, roster_cells: RosterCells, + metadata_applied_frontier: Arc, shard_metrics_all: Vec, ) -> Result<(), ServerError> { let topology = resolve_tcp_topology(config, replica_id)?; @@ -580,7 +591,9 @@ async fn shard_main( superblock_for_metadata, mux_stm, Some(PathBuf::from(&config.system.path)), - ); + ) + .with_applied_frontier(metadata_applied_frontier); + metadata.seed_applied_frontier_from_consensus(); // Size the VSR client table before listeners bind and any client registers. // Must precede the recovered-table install below: the setter rebuilds the // table from scratch, so running it afterwards would drop every resumed diff --git a/core/server/src/boot/threads.rs b/core/server/src/boot/threads.rs index 2dc1efd729..2a68d89b96 100644 --- a/core/server/src/boot/threads.rs +++ b/core/server/src/boot/threads.rs @@ -28,6 +28,7 @@ use configs::sharding::{ INBOX_CAPACITY_MAX, SHUTDOWN_DRAIN_TIMEOUT_MAX, SHUTDOWN_POLL_INTERVAL_MAX, }; use message_bus::{IggyMessageBus, ReplicaOwnerTable}; +use metadata::AppliedFrontier; use partitions::FatalCommit; use server_common::executor::create_shard_executor; use shard::metrics::ShardMetrics; @@ -426,6 +427,7 @@ pub(in crate::boot) fn run_shard_thread( barrier: BootstrapBarrier, owner_table: Arc, roster_cells: RosterCells, + metadata_applied_frontier: Arc, shard_metrics_all: Vec, ) -> Result<(), ServerError> { // Armed for the whole thread body: a post-spawn error `?` or a panic @@ -469,6 +471,7 @@ pub(in crate::boot) fn run_shard_thread( barrier, owner_table, roster_cells, + metadata_applied_frontier, shard_metrics_all, )) .await diff --git a/core/server/src/dispatch/mod.rs b/core/server/src/dispatch/mod.rs index 8f69e7886e..68053050fa 100644 --- a/core/server/src/dispatch/mod.rs +++ b/core/server/src/dispatch/mod.rs @@ -32,7 +32,7 @@ mod authz; pub mod login_error; pub mod partition; -mod reads; +pub mod reads; pub mod session_ops; pub mod submit; #[cfg(test)] @@ -46,7 +46,7 @@ use crate::dispatch::session_ops::{ handle_login_register_request, handle_logout_request, send_login_eviction, send_unauthenticated_eviction, submit_disconnect_logout, }; -use crate::dispatch::submit::submit_client_request_on_owner; +use crate::dispatch::submit::{committed_reply_commit, submit_client_request_on_owner}; use crate::pat::maybe_rewrite_pat_request; use crate::responses::{ NonReplicatedResponse, build_deny_reply, build_raw_pat_reply, current_metadata_commit, @@ -94,6 +94,21 @@ use std::sync::Arc; use tracing::{debug, warn}; type ClientRequestQueues = Rc>>>>; + +/// Requests one client may have queued behind a request this shard has not +/// answered yet. +/// +/// The drain loop below serves one frame per client at a time, and a frame can +/// legitimately hold it for a while: a metadata write awaits consensus, and a +/// read can be HELD for the read-your-writes budget (see +/// `crate::dispatch::reads`). Without a cap, a client that keeps pipelining +/// through such a stall grows its queue - and this node's memory - unbounded. +/// +/// Overflow is ANSWERED, not dropped: `TransientNotAccepted` is the honest +/// code, since the frame provably never entered any pipeline, so the SDK may +/// re-issue it anywhere, including here once the queue drains. Sized far above +/// any SDK's in-flight window, so it only ever fires under a genuine stall. +const MAX_QUEUED_CLIENT_REQUESTS: usize = 1024; type ActiveClientRequests = Rc>>; pub fn make_client_request_handler( @@ -290,11 +305,18 @@ fn enqueue_client_request( S: 'static, SB: SuperblockStore + 'static, { - queues - .borrow_mut() - .entry(client_id) - .or_default() - .push_back(message); + { + let mut queues = queues.borrow_mut(); + let queue = queues.entry(client_id).or_default(); + if queue.len() >= MAX_QUEUED_CLIENT_REQUESTS { + // Borrow released before the deny, which spawns onto this same + // task and would otherwise re-enter the table. + drop(queues); + deny_overflowing_client_request(&shard, client_id, message); + return; + } + queue.push_back(message); + } if !active.borrow_mut().insert(client_id) { return; } @@ -314,6 +336,51 @@ fn enqueue_client_request( }); } +/// Answer a request that arrived with this client's queue already at +/// [`MAX_QUEUED_CLIENT_REQUESTS`] with the retryable transient denial. +/// +/// Spawned rather than awaited: the enqueue path is sync (it runs straight off +/// frame arrival) and the reply goes out on the bus. A frame whose header will +/// not even cast is dropped instead, exactly as the drain loop drops it. +fn deny_overflowing_client_request( + shard: &Rc>, + transport_client_id: u128, + message: Message, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + shard.metrics().record_client_request_denied_queue_full(); + let Ok(request) = message.try_into_typed::() else { + warn!( + transport_client_id, + "dropping over-queue client request with invalid header" + ); + return; + }; + let request = request.into_routed(); + debug!( + transport_client_id, + operation = ?request.header().operation, + queued = MAX_QUEUED_CLIENT_REQUESTS, + "denying client request retryable: this connection's request queue is full" + ); + let shard = Rc::clone(shard); + let bus = shard.bus.clone(); + bus.spawn(async move { + send_deny_reply( + &shard, + transport_client_id, + request.header(), + IggyError::TransientNotAccepted.as_code(), + ) + .await; + }); +} + #[allow(clippy::future_not_send)] async fn drain_client_requests( shard: Rc>, @@ -939,6 +1006,13 @@ async fn handle_client_request( // shard 0 can't route by the consensus client id (no home-shard bits). match submit_client_request_on_owner(shard, request).await { Some(reply) => { + // Recorded before the reply reaches the socket, so a read the client + // sends the instant it decodes this frame already sees the mark. + if let Some(commit) = committed_reply_commit(&reply) { + sessions + .borrow_mut() + .record_metadata_watermark(transport_client_id, commit); + } // The raw PAT token never enters consensus (it is non-deterministic // and secret), so the committed reply body is empty. Substitute the // raw-token response here, on the minting client's home shard, using diff --git a/core/server/src/dispatch/reads.rs b/core/server/src/dispatch/reads.rs index 2c8b43c0d8..807b342b00 100644 --- a/core/server/src/dispatch/reads.rs +++ b/core/server/src/dispatch/reads.rs @@ -38,14 +38,16 @@ use crate::shell::{ShellBus, ShellShard}; use crate::snapshot; use crate::wire::request_body; use bytes::Bytes; -use configs::server::ServerSystemConfig; +use configs::server::{ServerConfig, ServerSystemConfig}; use consensus::MetadataHandle; +use futures::future::{Either, select}; use iggy_binary_protocol::PrepareHeader; use iggy_binary_protocol::codes::{ - GET_CLIENT_CODE, GET_CLIENTS_CODE, GET_CONSUMER_OFFSET_CODE, GET_ME_CODE, - GET_PERSONAL_ACCESS_TOKENS_CODE, GET_SNAPSHOT_FILE_CODE, GET_STATS_CODE, PING_CODE, - POLL_MESSAGES_CODE, SYNC_CONSUMER_GROUP_CODE, + DESCRIBE_OPTIONS_CODE, GET_CLIENT_CODE, GET_CLIENTS_CODE, GET_CLUSTER_METADATA_CODE, + GET_CONSUMER_OFFSET_CODE, GET_ME_CODE, GET_PERSONAL_ACCESS_TOKENS_CODE, GET_SNAPSHOT_FILE_CODE, + GET_STATS_CODE, PING_CODE, POLL_MESSAGES_CODE, SYNC_CONSUMER_GROUP_CODE, }; +use iggy_binary_protocol::dispatch::lookup_command; use iggy_binary_protocol::requests::consumer_groups::SyncConsumerGroupRequest; use iggy_binary_protocol::requests::system::get_client::GetClientRequest; use iggy_binary_protocol::requests::system::get_snapshot::GetSnapshotRequest; @@ -59,13 +61,17 @@ use iggy_common::{IggyError, SnapshotCompression, SystemSnapshotType}; use journal::superblock::SuperblockStore; use journal::{Journal, JournalHandle}; use message_bus::framing::MAX_MESSAGE_SIZE; +use metadata::AppliedFrontier; use metadata::impls::metadata::StreamsFrontend; use metadata::permissioner::Permissioner; use server_common::Message; use std::cell::RefCell; +use std::future::Future; use std::net::IpAddr; +use std::pin::pin; use std::rc::Rc; use std::sync::Arc; +use std::time::Duration; use tracing::{debug, warn}; /// Per-user PATs, resolved from this shard's session (like `get_me`) and read @@ -122,6 +128,211 @@ async fn handle_get_me( .await; } +/// Commit broadcasts a held read waits out before it fails retryable. +/// +/// Sized for a node merely behind on its commit walk, NOT for a view change -- +/// detecting one costs `heartbeat_timeout` and escalating it another +/// `view_change_status_timeout`, and `recovery_barrier_deadline` budgets at +/// least 15s for the same event, so a read that waits out an election is a read +/// the caller should retry elsewhere. +const READ_FRONTIER_BROADCASTS: u32 = 6; + +/// How long a held metadata read may wait for this node's applied frontier, +/// sized from `[cluster] commit_broadcast_interval`: the thing the read is +/// short of is a commit broadcast, so the budget has to move with the +/// configured cadence rather than with the compile-time default that +/// `TimeoutManager::COMMIT_MESSAGE_TICKS` names (the runtime overrides it +/// through `set_commit_message_ticks`). +/// +/// Minted once per process into the shared [`AppliedFrontier`], because a peer +/// shard's read gate has neither consensus nor the cluster config. +#[must_use] +pub fn read_frontier_budget(config: &ServerConfig) -> Duration { + config + .cluster + .commit_broadcast_interval + .get_duration() + // No config ceiling on the interval, so plain `*` can overflow. + .saturating_mul(READ_FRONTIER_BROADCASTS) +} + +/// Whether `code`'s answer comes from the metadata state machine, and so must +/// not be served below the caller's watermark. +/// +/// The decision for both planes, consulted by every read path that CAN hold: +/// the binary spine's gated arms run it through [`authorize_and_hold_read`] and +/// the REST spine through `http::reads::gate_local_read`. The arms that never +/// consult it are exactly the codes named below, so this is the single list of +/// what is not gated: +/// +/// - `Ping` is the pre-auth liveness probe and reads nothing. +/// - `DescribeOptions` decodes a static catalog. +/// - `GetClusterMetadata` answers from the configured roster plus the +/// consensus view, and sits on the SDK's leader-discovery path, where the +/// wait would be real. +/// - `PollMessages` and `GetConsumerOffset` are partition-plane reads: their +/// answer comes from a partition group's own log, not the metadata STM, and +/// holding them would put metadata lag on the data path. +/// - `GetSnapshotFile` shells out to system tools off-thread; there is no +/// metadata answer to hold. +/// - A code this build does not know: its only outcome is `InvalidCommand`, +/// and parking a terminal error for the whole budget serves nobody. +/// +/// A deny-list otherwise, so a read code added later is gated by default: the +/// failure mode of forgetting to name one here is a wait, while forgetting to +/// add it to an allow-list is a silent stale read. +pub const fn read_needs_metadata_frontier(code: u32) -> bool { + !matches!( + code, + PING_CODE + | DESCRIBE_OPTIONS_CODE + | GET_CLUSTER_METADATA_CODE + | POLL_MESSAGES_CODE + | GET_CONSUMER_OFFSET_CODE + | GET_SNAPSHOT_FILE_CODE + ) && lookup_command(code).is_some() +} + +/// Whether a frontier wait actually parked. +/// +/// The caller's authorization resolved its scope off the pre-wait state +/// machine, and a wait that parked is one where that state machine moved, so +/// only the parked outcome forces the gate to run again. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum FrontierWait { + /// The frontier already covered the watermark: no await ran. + Ready, + /// The read parked, and the frontier caught up while it waited. + CaughtUp, +} + +/// The frontier never reached the watermark inside the budget. Each plane +/// renders it in its own error currency. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct FrontierUnreached; + +/// Hold a read until `frontier` covers `watermark`, or until `budget` expires. +/// +/// Event-driven: the wait is woken by the commit that advances the frontier +/// (see [`AppliedFrontier::advance`]), so a read resumes on the commit it was +/// short of rather than on a poll that happens to land after it. The caller +/// supplies the budget as a future because the two read planes measure time +/// differently -- the shard bus timer, which is virtual under the simulator, +/// against `compio::time` on the HTTP listener. +/// +/// Shared by both planes because one wait with two copies is one wait with a +/// drift vector, and split from its callers so the fast path, the park, the +/// wake and the expiry are all testable without a live shard. +/// +/// A caller with nothing to read back has `watermark == 0`, which the first +/// comparison satisfies: one `Acquire` load, no registration, no await. Expiry +/// is loud and carries both numbers, so a frontier that stopped moving is +/// visible instead of showing up as latency. +pub async fn hold_for_frontier( + frontier: &AppliedFrontier, + watermark: u64, + budget: impl Future, +) -> Result { + if frontier.get() >= watermark { + return Ok(FrontierWait::Ready); + } + let reached = pin!(frontier.reached(watermark)); + let budget = pin!(budget); + match select(reached, budget).await { + Either::Left(((), _)) => Ok(FrontierWait::CaughtUp), + // Reported by the caller, not here: a durably lagging node refuses + // every held read of every client for as long as it lags, so this is + // a counter plus a `debug!`, never a line per refusal. + Either::Right(((), _)) => Err(FrontierUnreached), + } +} + +/// Hold a local metadata read until this node has applied everything the +/// connection was told committed. +/// +/// A committed reply hands the client an op number; answering its next read +/// from a state machine below that op contradicts the frame the client is +/// holding. The lag is real on a node whose commit walk trails the client's +/// epoch -- a backup that forwarded the client's register binds a committed +/// session while its own `commit_journal` is still behind it (see +/// [`crate::dispatch::session_ops`]) -- so the gate is not about peer shards: +/// every shard of a node reads one shared frontier and gates identically. +/// +/// Fast path is a single `Acquire` load and no await, which is what keeps an +/// uncontended read shared-nothing. A park costs this connection more than the +/// read itself: the per-connection drain loop serves one frame at a time, so +/// the client's queued `SendMessages`, `PollMessages` and `PING` wait behind +/// the held read, and a client that keeps pipelining through the hold is +/// answered `TransientNotAccepted` once its queue hits +/// [`MAX_QUEUED_CLIENT_REQUESTS`](crate::dispatch::MAX_QUEUED_CLIENT_REQUESTS) +/// rather than growing it unbounded. No OTHER connection waits, and the budget +/// bounds the hold itself. Expiry fails loud and retryable rather than serving +/// state the client already saw replaced. +/// +/// The wait ends on the commit that closes the gap, not on a poll: the budget +/// timer is the only timer armed, so a read that resumes costs one wake. +#[allow(clippy::future_not_send)] +async fn await_metadata_read_frontier( + shard: &Rc>, + watermark: u64, +) -> Result +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let frontier = shard.plane.metadata().applied_frontier(); + let budget = frontier.read_budget(); + hold_for_frontier(frontier, watermark, shard.bus.sleep(budget)) + .await + // `TransientNotAccepted`, not `NotCommitted`: a read never entered a + // pipeline, so it is safe to re-issue anywhere, and it is the code that + // drives the SDK's roster walk rather than a replay against the same + // durably lagging replica. + .map_err(|FrontierUnreached| { + shard.metrics().record_metadata_read_frontier_refusal(); + debug!( + frontier = frontier.get(), + watermark, + ?budget, + "metadata read frontier unreached inside the budget; failing the read retryable" + ); + IggyError::TransientNotAccepted + }) +} + +/// Authorize a metadata read, then hold it for this node's applied frontier. +/// +/// Authorization first: a denial is terminal, and parking the connection for +/// the whole budget before answering one buys nothing. Then again on a wait +/// that parked -- the rule resolves its scope and the caller's grants off the +/// state machine, and a park is exactly the case where both moved under it. +#[allow(clippy::future_not_send)] +async fn authorize_and_hold_read( + shard: &Rc>, + code: u32, + watermark: u64, + authorize: impl Fn() -> Result<(), IggyError>, +) -> Result<(), IggyError> +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + authorize()?; + if !read_needs_metadata_frontier(code) { + return Ok(()); + } + if await_metadata_read_frontier(shard, watermark).await? == FrontierWait::CaughtUp { + authorize()?; + } + Ok(()) +} + #[allow(clippy::future_not_send, clippy::too_many_lines)] pub(in crate::dispatch) async fn handle_non_replicated_request( shard: &Rc>, @@ -138,10 +349,11 @@ pub(in crate::dispatch) async fn handle_non_replicated_request( { const CODE_RANGE: std::ops::Range = 0..4; let code = u32::from_le_bytes(request.header().reserved[CODE_RANGE].try_into().unwrap()); - // Acting user and peer address for the read gates below, resolved in one - // connection lookup. `user_id` is `None` only on the pre-auth path - // (PING), which serves ungated codes; the gated arms fail closed on it. - let (user_id, client_address) = sessions.borrow().read_context(transport_client_id); + // Acting user, peer address and read-your-writes floor for the gates + // below, resolved in one connection lookup. `user_id` is `None` only on the + // pre-auth path (PING), which serves ungated codes; the gated arms fail + // closed on it. + let (user_id, client_address, watermark) = sessions.borrow().read_context(transport_client_id); match code { PING_CODE => { // A ping is the client's liveness proof; reset its staleness clock @@ -167,13 +379,30 @@ pub(in crate::dispatch) async fn handle_non_replicated_request( } } GET_ME_CODE => { + // Self-scoped, so no permissioner rule -- but the consumer-group + // list it carries is read off the streams STM, so it is gated like + // any other metadata read. + if let Err(error) = authorize_and_hold_read(shard, code, watermark, || Ok(())).await { + send_non_replicated_deny(shard, &request, transport_client_id, error.as_code()) + .await; + return; + } handle_get_me(shard, sessions, transport_client_id, &request).await; } GET_PERSONAL_ACCESS_TOKENS_CODE => { + if let Err(error) = authorize_and_hold_read(shard, code, watermark, || Ok(())).await { + send_non_replicated_deny(shard, &request, transport_client_id, error.as_code()) + .await; + return; + } handle_get_personal_access_tokens(shard, sessions, transport_client_id, &request).await; } GET_CLIENTS_CODE => { - if let Err(error) = authorize_uid(shard, user_id, Permissioner::get_clients) { + if let Err(error) = authorize_and_hold_read(shard, code, watermark, || { + authorize_uid(shard, user_id, Permissioner::get_clients) + }) + .await + { send_non_replicated_deny(shard, &request, transport_client_id, error.as_code()) .await; return; @@ -197,7 +426,11 @@ pub(in crate::dispatch) async fn handle_non_replicated_request( .await; } GET_CLIENT_CODE => { - if let Err(error) = authorize_uid(shard, user_id, Permissioner::get_client) { + if let Err(error) = authorize_and_hold_read(shard, code, watermark, || { + authorize_uid(shard, user_id, Permissioner::get_client) + }) + .await + { send_non_replicated_deny(shard, &request, transport_client_id, error.as_code()) .await; return; @@ -250,7 +483,13 @@ pub(in crate::dispatch) async fn handle_non_replicated_request( } SYNC_CONSUMER_GROUP_CODE => { // Self-scoped: serves the caller's own assignment keyed by the - // header client id, so it carries no permissioner rule. + // header client id, so it carries no permissioner rule. The + // assignment itself is metadata-STM state, hence the gate. + if let Err(error) = authorize_and_hold_read(shard, code, watermark, || Ok(())).await { + send_non_replicated_deny(shard, &request, transport_client_id, error.as_code()) + .await; + return; + } handle_sync_consumer_group(shard, transport_client_id, &request).await; } _ => { @@ -269,6 +508,7 @@ pub(in crate::dispatch) async fn handle_non_replicated_request( code, &request, user_id, + watermark, &roster, client_ip, ) @@ -284,6 +524,7 @@ async fn handle_default_non_replicated( code: u32, request: &Message, user_id: Option, + watermark: u64, roster: &ClusterRoster, client_ip: Option, ) where @@ -295,8 +536,14 @@ async fn handle_default_non_replicated( { // Gate by command code before the shared builder runs. The builder stays // authz-free (it is byte-shared with the HTTP read path, which gates - // separately); a denial replies status!=0 with an empty body. - if let Err(error) = authorize_default_read(shard, code, request_body(request), user_id) { + // separately); a denial replies status!=0 with an empty body. The + // read-your-writes hold sits INSIDE the same call, behind that denial: an + // unauthorized read must fail now, not after the whole poll budget. + if let Err(error) = authorize_and_hold_read(shard, code, watermark, || { + authorize_default_read(shard, code, request_body(request), user_id) + }) + .await + { send_non_replicated_deny(shard, request, transport_client_id, error.as_code()).await; return; } @@ -489,3 +736,149 @@ async fn handle_sync_consumer_group( ) .await; } + +#[cfg(test)] +mod tests { + use super::{ + FrontierUnreached, FrontierWait, hold_for_frontier, read_frontier_budget, + read_needs_metadata_frontier, + }; + use configs::server::ServerConfig; + use iggy_binary_protocol::codes::{ + DESCRIBE_OPTIONS_CODE, GET_CLUSTER_METADATA_CODE, GET_CONSUMER_OFFSET_CODE, GET_ME_CODE, + GET_SNAPSHOT_FILE_CODE, GET_STREAM_CODE, PING_CODE, POLL_MESSAGES_CODE, + SYNC_CONSUMER_GROUP_CODE, + }; + use iggy_common::IggyDuration; + use metadata::AppliedFrontier; + use std::future::pending; + use std::sync::Arc; + use std::time::Duration; + + /// The budget has to move with the CONFIGURED commit cadence, not with the + /// compile-time default: `[cluster] commit_broadcast_interval` is what + /// sizes the timer the read is waiting on, and a cluster that widens it to + /// 2s would otherwise get a budget of one and a half broadcasts and refuse + /// reads on a backup that is merely a commit behind. + /// + /// The default arm also pins the fallback the simulator and the unit + /// fixtures run on, so the two cannot drift apart silently. + #[test] + fn given_a_configured_commit_cadence_when_sizing_the_budget_should_scale_with_it() { + let mut config = ServerConfig::default(); + assert_eq!( + read_frontier_budget(&config), + AppliedFrontier::DEFAULT_READ_BUDGET, + "the config default must agree with the frontier's built-in fallback" + ); + + config.cluster.commit_broadcast_interval = IggyDuration::from(Duration::from_secs(2)); + assert_eq!( + read_frontier_budget(&config), + Duration::from_secs(12), + "six broadcasts of the configured interval" + ); + } + + /// A caller with nothing to read back (`watermark == 0`) and one whose + /// watermark this node has already applied are the whole steady state, and + /// neither may cost a park: no registration, no await. The budget here is + /// a future that never completes, so a gate that parked would hang instead + /// of quietly costing a tick. + #[compio::test] + async fn given_a_frontier_at_the_watermark_when_gating_should_serve_without_parking() { + let frontier = AppliedFrontier::default(); + frontier.advance(9); + for watermark in [0, 7, 9] { + assert_eq!( + hold_for_frontier(&frontier, watermark, pending()).await, + Ok(FrontierWait::Ready), + "frontier 9 covers {watermark}, so the read must not park" + ); + } + assert_eq!(frontier.waiting(), 0, "a served read registers no wait"); + } + + /// The gate's whole point, and the reason the wait is event-driven: a read + /// whose caller was told op 9 committed is held while this node is at 4, + /// and the COMMIT that advances the frontier is what answers it - here a + /// detached task standing in for the commit path, with no timer in the + /// budget at all. The parked outcome is what tells the caller to re-run + /// the authorization it resolved off the pre-wait state machine. + #[compio::test] + async fn given_a_frontier_behind_the_watermark_when_it_advances_should_answer_the_held_read() { + let frontier = Arc::new(AppliedFrontier::default()); + frontier.advance(4); + let committer = Arc::clone(&frontier); + compio::runtime::spawn(async move { + // Yields first, so the read is provably parked before the advance: + // a gate that answered off the lagging frontier would already have + // returned by the time this runs. + compio::runtime::time::sleep(std::time::Duration::ZERO).await; + committer.advance(9); + }) + .detach(); + + assert_eq!( + hold_for_frontier(&frontier, 9, pending()).await, + Ok(FrontierWait::CaughtUp), + "the commit that closed the gap must answer the held read" + ); + assert_eq!(frontier.waiting(), 0, "the answered wait deregisters"); + } + + /// A node can legitimately never catch up (a durably lagging replica), so + /// the wait is bounded - and the exit is a refusal, never the stale answer. + /// Each plane renders it retryable: `TransientNotAccepted` on the binary + /// transports, the shared 503 over HTTP. + #[compio::test] + async fn given_a_frontier_that_never_catches_up_when_the_budget_expires_should_fail_retryable() + { + let frontier = AppliedFrontier::default(); + frontier.advance(4); + assert_eq!( + hold_for_frontier(&frontier, 9, std::future::ready(())).await, + Err(FrontierUnreached), + "an unreached frontier must refuse the read, not serve it" + ); + assert_eq!( + frontier.waiting(), + 0, + "the expired wait must not leave its waker behind" + ); + } + + /// The deny-list's whole point is that a read answered from the metadata + /// STM is gated even when nobody remembered to name it, so the arms that + /// are NOT gated are the ones worth pinning: the static catalog, the + /// roster read on the leader-discovery path, and a code this build cannot + /// serve at all (whose only outcome is `InvalidCommand`, which must not + /// wait out the budget first). + #[test] + fn given_a_read_code_when_classified_should_gate_all_but_the_named_exclusions() { + for code in [GET_STREAM_CODE, GET_ME_CODE, SYNC_CONSUMER_GROUP_CODE] { + assert!( + read_needs_metadata_frontier(code), + "code {code} answers from the metadata STM and must be gated" + ); + } + for code in [ + PING_CODE, + DESCRIBE_OPTIONS_CODE, + GET_CLUSTER_METADATA_CODE, + POLL_MESSAGES_CODE, + GET_CONSUMER_OFFSET_CODE, + GET_SNAPSHOT_FILE_CODE, + ] { + assert!( + !read_needs_metadata_frontier(code), + "code {code} has no metadata-STM answer to hold; the arms that skip \ + the gate are exactly these" + ); + } + assert!( + !read_needs_metadata_frontier(u32::MAX), + "an unknown code has no answer to hold, so it must not park" + ); + } +} diff --git a/core/server/src/dispatch/submit.rs b/core/server/src/dispatch/submit.rs index db3529b1f4..7176edf501 100644 --- a/core/server/src/dispatch/submit.rs +++ b/core/server/src/dispatch/submit.rs @@ -29,6 +29,7 @@ use crate::dispatch::session_ops::{ submit_register_local_or_forward, }; use crate::dispatch::upgrade_shard_handle; +use crate::responses::committed_reply_header; use crate::shell::{ShellBus, ShellShard, ShellShardHandle}; use consensus::MetadataHandle; use iggy_binary_protocol::{GenericHeader, PrepareHeader, RoutedRequestHeader}; @@ -192,3 +193,118 @@ where }); rx.recv().await.ok().flatten() } + +/// The commit position a SUCCESSFULLY COMMITTED metadata reply carries, or +/// `None` when the frame promises the caller nothing. +/// +/// Only a success promises. Every other frame on this path stamps `commit` with +/// the primary's `commit_max`, an op the caller was never told committed and, +/// on a backup-homed caller, one its own reads would then wait for: +/// +/// - an eviction is an `EvictionHeader` whose bytes would cast cleanly as a +/// reply, so the command is checked first (same guard as +/// `build_raw_pat_reply`); +/// - a request-level denial names itself in `ReplyHeader.status`, the channel +/// the SDK peeks before body decode (see `build_deny_reply`); +/// - a transient rejection did not commit and will be replayed; +/// - a TERMINAL pre-consensus rejection (`PreflightOutcome::Reject`, e.g. a +/// fenced session) is a result section carrying a non-transient code, which +/// is byte-identical to a COMMITTED business rejection (duplicate name, bad +/// expiry). Neither is separable here, and neither has to be: a rejection +/// mutated nothing, so the caller has nothing to read back from it, and +/// grading both as no-promise is the only reading that cannot make a read +/// wait for an op that never committed. +/// +/// The grading itself is [`committed_reply_header`], shared with the raw-PAT +/// splice, which must admit exactly the same frames. +/// +/// Shared with the HTTP write path, which grades the same frames off the same +/// submit entry point ([`submit_client_request_on_owner`]); one classifier is +/// what keeps the two planes' watermarks meaning the same thing. +pub fn committed_reply_commit(reply: &Message) -> Option { + committed_reply_header(reply).map(|header| header.commit) +} + +#[cfg(test)] +mod tests { + use super::committed_reply_commit; + use crate::dispatch::test_support::request_message; + use crate::responses::{build_deny_reply, build_reply_from_bytes}; + use bytes::Bytes; + use iggy_binary_protocol::Operation; + use iggy_common::IggyError; + + /// Commit position of the frames below. Above zero on purpose: `0` is the + /// "promised nothing" answer, so a fixture at zero could not tell a + /// classified success from a rejected frame. + const COMMIT: u64 = 9; + + /// A result-framed body: `[count][index][result]`, then the payload. + fn result_body(code: u32, payload: &[u8]) -> Bytes { + let mut body = Vec::new(); + let count = u32::from(code != 0); + body.extend_from_slice(&count.to_le_bytes()); + if count == 1 { + body.extend_from_slice(&0u32.to_le_bytes()); + body.extend_from_slice(&code.to_le_bytes()); + } + body.extend_from_slice(payload); + Bytes::from(body) + } + + /// The whole classification in one table: only a successful commit hands + /// the read gate a floor. Everything else stamps the primary's + /// `commit_max` into a frame that promised the caller nothing, and a floor + /// taken from one of those parks the caller's next read on a backup until + /// the budget expires. + #[test] + fn given_a_metadata_reply_when_classified_should_promise_only_a_committed_success() { + let request = request_message(Operation::CreateStream, 42, 7, 3, &[]); + + let committed = + build_reply_from_bytes(request.header(), 42, 7, COMMIT, &result_body(0, b"payload")) + .into_generic(); + assert_eq!( + committed_reply_commit(&committed), + Some(COMMIT), + "a committed success is the one frame that promises the caller its op" + ); + + for code in [ + IggyError::TransientNotCommitted.as_code(), + IggyError::TransientNotAccepted.as_code(), + IggyError::UserAlreadyExists.as_code(), + ] { + let rejected = + build_reply_from_bytes(request.header(), 42, 7, COMMIT, &result_body(code, &[])) + .into_generic(); + assert_eq!( + committed_reply_commit(&rejected), + None, + "result code {code} mutated nothing, so it promises no read floor" + ); + } + + let denied = build_deny_reply( + request.header(), + 42, + 7, + COMMIT, + IggyError::Unauthorized.as_code(), + ) + .into_generic(); + assert_eq!( + committed_reply_commit(&denied), + None, + "a request-level denial names itself in `status` and commits nothing" + ); + + // Any non-`Reply` command stands in for the eviction frame, whose bytes + // would otherwise cast cleanly as a `ReplyHeader`. + assert_eq!( + committed_reply_commit(&request.into_generic()), + None, + "only a `Reply` carries a commit position" + ); + } +} diff --git a/core/server/src/http.rs b/core/server/src/http.rs index cbbb515ad6..8748ea820e 100644 --- a/core/server/src/http.rs +++ b/core/server/src/http.rs @@ -232,6 +232,7 @@ pub fn start( in_flight_writes: Cell::new(0), forward, metrics: metrics::HttpMetrics::init(shard_metrics_all), + metadata_watermarks: Rc::default(), })); let app = router( state, diff --git a/core/server/src/http/error.rs b/core/server/src/http/error.rs index e369b5cfa3..8a0d04a318 100644 --- a/core/server/src/http/error.rs +++ b/core/server/src/http/error.rs @@ -471,6 +471,14 @@ pub(in crate::http) enum ReadError { /// [`service_unavailable`] body, retryable once the cluster re-commits the /// suffix. RecoveryIncomplete, + /// A metadata read waited out its budget with this node's applied frontier + /// still below the op the caller was told committed. Fail-closed on the + /// same retryable 503 as [`Self::RecoveryIncomplete`]: the two are the same + /// hazard (serving state the caller already saw replaced) reached from + /// different directions, and 503 is what the binary transports' equivalent + /// refusal (`TransientNotCommitted`) already renders as, so an SDK that + /// speaks both sees one answer. Never a 2xx with stale state. + MetadataFrontierUnreached, /// A partition read (poll / consumer-offset) got no reply from the owning /// shard within the mesh budget. 504 like a produce timeout: the outcome is /// unknown (the abandoned read may still be running), so the caller retries. @@ -486,7 +494,7 @@ impl IntoResponse for ReadError { Self::NotFound => CustomError::ResourceNotFound.into_response(), Self::NotPrimary => not_primary_response(), Self::RedirectToPrimary(location) => primary_redirect_response(&location), - Self::RecoveryIncomplete => service_unavailable(), + Self::RecoveryIncomplete | Self::MetadataFrontierUnreached => service_unavailable(), Self::Timeout => gateway_timeout_response( "partition_read_timeout", "the partition owner did not answer the read in time; retry", @@ -817,4 +825,20 @@ mod tests { not_primary.headers().get(RETRY_AFTER) ); } + + // An unreached read frontier must never degrade into a 2xx carrying stale + // state, and must not read as terminal either: it is the same retryable 503 + // the recovery barrier's expiry renders, so an SDK retries rather than + // surfacing the read as failed. + #[test] + fn metadata_frontier_unreached_renders_the_same_retryable_503_as_the_barrier() { + let frontier = ReadError::MetadataFrontierUnreached.into_response(); + let recovery = ReadError::RecoveryIncomplete.into_response(); + assert_eq!(frontier.status(), StatusCode::SERVICE_UNAVAILABLE); + assert_eq!(frontier.status(), recovery.status()); + assert_eq!( + frontier.headers().get(RETRY_AFTER), + Some(&HeaderValue::from(RETRY_AFTER_SECONDS)) + ); + } } diff --git a/core/server/src/http/extractor.rs b/core/server/src/http/extractor.rs index 92c3170ca3..075ab72965 100644 --- a/core/server/src/http/extractor.rs +++ b/core/server/src/http/extractor.rs @@ -117,7 +117,9 @@ impl FromRequestParts for Identity { let bearer = bearer_token(&parts.headers)?; // Verify only. The session key and expiry `resolve_credential` also - // returns feed the write path's session table; a read discards them. + // returns feed the write path's session table; a read discards them - + // its read-your-writes floor is keyed by user id, not by credential + // (see `MetadataWatermarks`). // The verify is `!Send` (a trusted-issuer JWT may await a JWKS fetch), // so bridge it with `SendWrapper` - sound only because compio pins this // future to shard 0's single thread, the only thread the JWKS client diff --git a/core/server/src/http/forward.rs b/core/server/src/http/forward.rs index fdfa2ea873..caf2aa8fa8 100644 --- a/core/server/src/http/forward.rs +++ b/core/server/src/http/forward.rs @@ -82,7 +82,7 @@ use crate::http::error::{ CustomError, error_response, gateway_timeout_response, primary_http_socket, with_retry_after, }; use crate::http::extractor::{bearer_token, resolve_credential}; -use crate::http::state::{ForwardState, HttpInner, VIEW_HEADER}; +use crate::http::state::{APPLIED_OP_HEADER, ForwardState, HttpInner, VIEW_HEADER}; use crate::server_error::ServerError; /// Marker stamped on every forwarded request. Loop guard only: a node that is @@ -130,10 +130,13 @@ const RESPONSE_CAPACITY_HINT: usize = 64 * 1024; /// Response headers copied from the primary's reply. Everything else is /// dropped, which subsumes the RFC 7230 hop-by-hop set: the relayed response /// is rebuilt, never streamed, so upstream `connection` / `transfer-encoding` -/// semantics cannot leak to the client. `iggy-view` is included so the -/// relayed response carries the serving primary's view, not this follower's -/// (the view layer only fills the header when absent). -const RELAYED_RESPONSE_HEADERS: [HeaderName; 3] = [CONTENT_TYPE, RETRY_AFTER, VIEW_HEADER]; +/// semantics cannot leak to the client. `iggy-view` and `iggy-applied-op` are +/// included so the relayed response carries the serving primary's view and +/// applied op, not this follower's (the response layer only fills either when +/// absent); the applied op is also what this node records as the caller's +/// read-your-writes floor, so dropping it here would reopen the stale read. +const RELAYED_RESPONSE_HEADERS: [HeaderName; 4] = + [CONTENT_TYPE, RETRY_AFTER, VIEW_HEADER, APPLIED_OP_HEADER]; /// Build the [`ForwardState`] at listener startup. /// @@ -268,9 +271,13 @@ async fn forward_or_pass(state: HttpState, request: Request, next: Next) -> Resp Ok(bearer) => bearer, Err(error) => return CustomError::from(error).into_response(), }; - if let Err(rejection) = resolve_credential(&state, bearer).await { - return rejection.into_response(); - } + // The user id is kept, not discarded: the relayed answer carries the + // primary's applied op, and this node has to record it as this caller's + // read-your-writes floor (see `record_relayed_floor`). + let user_id = match resolve_credential(&state, bearer).await { + Ok((_key, user_id, _expiry)) => user_id, + Err(rejection) => return rejection.into_response(), + }; let Some(_guard) = ForwardGuard::admit(&state.forward.in_flight) else { return with_retry_after(error_response( StatusCode::SERVICE_UNAVAILABLE, @@ -278,7 +285,41 @@ async fn forward_or_pass(state: HttpState, request: Request, next: Next) -> Resp "node is at its forward budget; retry with backoff", )); }; - forward(&state, request).await + let response = forward(&state, request).await; + record_relayed_floor(&state, user_id, &response); + response +} + +/// Record the serving primary's applied op as `user_id`'s read-your-writes +/// floor on THIS node. +/// +/// The relayed write ran on the primary, so the local write path never saw it +/// and left no floor behind, while the caller's next unqualified GET stays +/// local: without this, a `POST` followed by a `GET` through the same follower +/// can answer from before the write. Only a relayed SUCCESS counts - a 503 or a +/// 4xx promises the caller nothing - and the floor is monotone, so a slow relay +/// landing after a faster one cannot lower it. +/// +/// A missing or unparsable header is a no-op rather than a failure: it means +/// the peer is an older build, and a floor this node never learns is the +/// pre-existing behavior, not a new hazard. +fn record_relayed_floor(state: &HttpInner, user_id: u32, response: &Response) { + if !response.status().is_success() { + return; + } + let Some(applied) = response + .headers() + .get(APPLIED_OP_HEADER) + .and_then(|value| value.to_str().ok()) + .and_then(|value| value.parse::().ok()) + else { + debug!( + user_id, + "relayed response carried no applied op; the caller's floor stays where it was" + ); + return; + }; + state.metadata_watermarks.record(user_id, applied); } async fn forward_partition_or_pass(state: HttpState, request: Request, next: Next) -> Response { diff --git a/core/server/src/http/handlers.rs b/core/server/src/http/handlers.rs index 8483138096..2a082405b3 100644 --- a/core/server/src/http/handlers.rs +++ b/core/server/src/http/handlers.rs @@ -30,9 +30,10 @@ use axum::response::{IntoResponse, Response}; use chrono::Local; use consensus::{MetadataHandle, PartitionsHandle}; use iggy_binary_protocol::codes::{ - DESCRIBE_OPTIONS_CODE, GET_CONSUMER_GROUP_CODE, GET_CONSUMER_GROUPS_CODE, - GET_PERSONAL_ACCESS_TOKENS_CODE, GET_STATS_CODE, GET_STREAM_CODE, GET_STREAMS_CODE, - GET_TOPIC_CODE, GET_TOPICS_CODE, GET_USER_CODE, GET_USERS_CODE, + DESCRIBE_OPTIONS_CODE, GET_CLIENT_CODE, GET_CLIENTS_CODE, GET_CONSUMER_GROUP_CODE, + GET_CONSUMER_GROUPS_CODE, GET_CONSUMER_OFFSET_CODE, GET_PERSONAL_ACCESS_TOKENS_CODE, + GET_SNAPSHOT_FILE_CODE, GET_STATS_CODE, GET_STREAM_CODE, GET_STREAMS_CODE, GET_TOPIC_CODE, + GET_TOPICS_CODE, GET_USER_CODE, GET_USERS_CODE, POLL_MESSAGES_CODE, }; use iggy_binary_protocol::requests::consumer_groups::{ CreateConsumerGroupRequest, DeleteConsumerGroupRequest, GetConsumerGroupRequest, @@ -131,7 +132,7 @@ use crate::http::error::{ use crate::http::extractor::{Authenticated, Identity}; use crate::http::metrics::gauge_value; use crate::http::reads::{ - authorize_data_plane, authorize_read, read_local, resolve_gate_stream, resolve_gate_topic, + authorize_data_plane, gate_local_read, read_local, resolve_gate_stream, resolve_gate_topic, resolve_gate_topic_ids, resolve_gate_user, }; use crate::http::reply::{ @@ -334,9 +335,6 @@ pub(in crate::http) async fn get_stream( ) -> Result, ReadError> { let stream_id = Identifier::from_str_value(&stream_id).map_err(ReadError::Rejected)?; let wire_stream_id = identifier_to_wire(&stream_id).map_err(ReadError::Rejected)?; - // Resolve for the gate; a miss leaves it a pass-through so the read renders - // the existing 404 rather than a 403. - let scope = resolve_gate_stream(&state, &wire_stream_id); let request = GetStreamRequest { stream_id: wire_stream_id, }; @@ -347,8 +345,14 @@ pub(in crate::http) async fn get_stream( query.consistency, GET_STREAM_CODE, &body, + // Resolved when the rule RUNS, not here: `read_local` can park for the + // read-your-writes frontier, and an entity created during that wait + // would resolve to nothing on a pre-wait pass, where a miss is a + // pass-through. A miss still leaves the gate a pass-through so the read + // renders the existing 404 rather than a 403. |permissioner, uid| { - scope.map_or(Ok(()), |stream_id| permissioner.get_stream(uid, stream_id)) + resolve_gate_stream(&state, &request.stream_id) + .map_or(Ok(()), |stream_id| permissioner.get_stream(uid, stream_id)) }, )) .await?; @@ -370,7 +374,6 @@ pub(in crate::http) async fn get_topics( ) -> Result>, ReadError> { let stream_id = Identifier::from_str_value(&stream_id).map_err(ReadError::Rejected)?; let wire_stream_id = identifier_to_wire(&stream_id).map_err(ReadError::Rejected)?; - let scope = resolve_gate_stream(&state, &wire_stream_id); let request = GetTopicsRequest { stream_id: wire_stream_id, }; @@ -382,7 +385,8 @@ pub(in crate::http) async fn get_topics( GET_TOPICS_CODE, &body, |permissioner, uid| { - scope.map_or(Ok(()), |stream_id| permissioner.get_topics(uid, stream_id)) + resolve_gate_stream(&state, &request.stream_id) + .map_or(Ok(()), |stream_id| permissioner.get_topics(uid, stream_id)) }, )) .await?; @@ -406,7 +410,6 @@ pub(in crate::http) async fn get_topic( let topic_id = Identifier::from_str_value(&topic_id).map_err(ReadError::Rejected)?; let wire_stream_id = identifier_to_wire(&stream_id).map_err(ReadError::Rejected)?; let wire_topic_id = identifier_to_wire(&topic_id).map_err(ReadError::Rejected)?; - let scope = resolve_gate_topic(&state, &wire_stream_id, &wire_topic_id); let request = GetTopicRequest { stream_id: wire_stream_id, topic_id: wire_topic_id, @@ -419,9 +422,10 @@ pub(in crate::http) async fn get_topic( GET_TOPIC_CODE, &body, |permissioner, uid| { - scope.map_or(Ok(()), |(stream_id, topic_id)| { - permissioner.get_topic(uid, stream_id, topic_id) - }) + resolve_gate_topic(&state, &request.stream_id, &request.topic_id) + .map_or(Ok(()), |(stream_id, topic_id)| { + permissioner.get_topic(uid, stream_id, topic_id) + }) }, )) .await?; @@ -473,7 +477,6 @@ pub(in crate::http) async fn get_user( let user_id = Identifier::from_str_value(&user_id).map_err(ReadError::Rejected)?; identifier_to_wire(&user_id).map_err(ReadError::Rejected)? }; - let is_self = resolve_gate_user(&state, &wire_user_id) == Some(identity.user_id as usize); let request = GetUserRequest { user_id: wire_user_id, }; @@ -485,6 +488,9 @@ pub(in crate::http) async fn get_user( GET_USER_CODE, &body, |permissioner, uid| { + #[allow(clippy::cast_possible_truncation)] + let is_self = + resolve_gate_user(&state, &request.user_id) == Some(identity.user_id as usize); if is_self { Ok(()) } else { @@ -513,7 +519,6 @@ pub(in crate::http) async fn get_cgs( let topic_id = Identifier::from_str_value(&topic_id).map_err(ReadError::Rejected)?; let wire_stream_id = identifier_to_wire(&stream_id).map_err(ReadError::Rejected)?; let wire_topic_id = identifier_to_wire(&topic_id).map_err(ReadError::Rejected)?; - let scope = resolve_gate_topic(&state, &wire_stream_id, &wire_topic_id); let request = GetConsumerGroupsRequest { stream_id: wire_stream_id, topic_id: wire_topic_id, @@ -526,9 +531,10 @@ pub(in crate::http) async fn get_cgs( GET_CONSUMER_GROUPS_CODE, &body, |permissioner, uid| { - scope.map_or(Ok(()), |(stream_id, topic_id)| { - permissioner.get_consumer_groups(uid, stream_id, topic_id) - }) + resolve_gate_topic(&state, &request.stream_id, &request.topic_id) + .map_or(Ok(()), |(stream_id, topic_id)| { + permissioner.get_consumer_groups(uid, stream_id, topic_id) + }) }, )) .await?; @@ -552,7 +558,6 @@ pub(in crate::http) async fn get_cg( let group_id = Identifier::from_str_value(&group_id).map_err(ReadError::Rejected)?; let wire_stream_id = identifier_to_wire(&stream_id).map_err(ReadError::Rejected)?; let wire_topic_id = identifier_to_wire(&topic_id).map_err(ReadError::Rejected)?; - let scope = resolve_gate_topic(&state, &wire_stream_id, &wire_topic_id); let request = GetConsumerGroupRequest { stream_id: wire_stream_id, topic_id: wire_topic_id, @@ -566,9 +571,10 @@ pub(in crate::http) async fn get_cg( GET_CONSUMER_GROUP_CODE, &body, |permissioner, uid| { - scope.map_or(Ok(()), |(stream_id, topic_id)| { - permissioner.get_consumer_group(uid, stream_id, topic_id) - }) + resolve_gate_topic(&state, &request.stream_id, &request.topic_id) + .map_or(Ok(()), |(stream_id, topic_id)| { + permissioner.get_consumer_group(uid, stream_id, topic_id) + }) }, )) .await?; @@ -670,19 +676,26 @@ pub(in crate::http) async fn get_metrics( /// `POST /snapshot`: collect a diagnostic archive and return it as a ZIP /// download with the same headers the legacy server sets. /// -/// Gated on the snapshot rule (`read_servers || manage_servers`) via the -/// shared [`authorize_read`] gate. Collection shells out to system tools on a -/// dedicated OS thread (see `snapshot::collect`); this handler only awaits the -/// result handoff, which is `Send`, so no `SendWrapper` bridge is needed. +/// Gated on the snapshot rule (`read_servers || manage_servers`) through the +/// shared [`gate_local_read`], which also serves it behind the post-restart +/// barrier; the archive itself carries no metadata answer to hold. Collection +/// shells out to system tools on a dedicated OS thread (see +/// `snapshot::collect`); this handler only awaits the result handoff, which is +/// `Send`, so no `SendWrapper` bridge is needed. pub(in crate::http) async fn get_snapshot( State(state): State, identity: Identity, Query(query): Query, Json(command): Json, ) -> Result<(HeaderMap, Body), ReadError> { - authorize_read(&state, &identity, query.consistency, |permissioner, uid| { - permissioner.get_snapshot(uid) - })?; + SendWrapper::new(gate_local_read( + &state, + &identity, + query.consistency, + GET_SNAPSHOT_FILE_CODE, + Permissioner::get_snapshot, + )) + .await?; let archive = snapshot::collect( Arc::clone(&state.system_config), command.compression, @@ -727,18 +740,23 @@ pub(in crate::http) async fn get_cluster_metadata( /// Unlike the entity reads, connections live in each shard's session manager, /// not the metadata STM, so this scatter-gathers over the shard mesh /// (`list_all_clients`) instead of going through [`read_local`]. It still runs -/// the identical per-op + consistency gate via [`authorize_read`], so its -/// authorization matches every metadata read. The gather future is `!Send`, -/// bridged onto shard 0's thread by `SendWrapper` exactly as the write path -/// bridges its submit. +/// the identical gates via [`gate_local_read`] - the consumer-group counts it +/// reports come off the streams STM, so its binary twin holds it for the read +/// frontier too. The gather future is `!Send`, bridged onto shard 0's thread by +/// `SendWrapper` exactly as the write path bridges its submit. pub(in crate::http) async fn get_clients( State(state): State, identity: Identity, Query(query): Query, ) -> Result>, ReadError> { - authorize_read(&state, &identity, query.consistency, |permissioner, uid| { - permissioner.get_clients(uid) - })?; + SendWrapper::new(gate_local_read( + &state, + &identity, + query.consistency, + GET_CLIENTS_CODE, + Permissioner::get_clients, + )) + .await?; let infos = SendWrapper::new(state.shard.list_all_clients()).await; let response = GetClientsResponse { clients: infos @@ -763,9 +781,14 @@ pub(in crate::http) async fn get_client( Path(client_id): Path, Query(query): Query, ) -> Result, ReadError> { - authorize_read(&state, &identity, query.consistency, |permissioner, uid| { - permissioner.get_client(uid) - })?; + SendWrapper::new(gate_local_read( + &state, + &identity, + query.consistency, + GET_CLIENT_CODE, + Permissioner::get_client, + )) + .await?; let infos = SendWrapper::new(state.shard.list_all_clients()).await; // The wire client id is the u32 seq tail of the u128 transport id. #[allow(clippy::cast_possible_truncation)] @@ -1216,17 +1239,19 @@ pub(in crate::http) async fn poll_messages( ) -> Result, ReadError> { let stream_id = Identifier::from_str_value(&stream_id).map_err(ReadError::Rejected)?; let topic_id = Identifier::from_str_value(&topic_id).map_err(ReadError::Rejected)?; - let scope = resolve_gate_topic_ids(&state, &stream_id, &topic_id); - authorize_read( + SendWrapper::new(gate_local_read( &state, &identity, consistency.consistency, + POLL_MESSAGES_CODE, |permissioner, uid| { - scope.map_or(Ok(()), |(stream_id, topic_id)| { - permissioner.poll_messages(uid, stream_id, topic_id) - }) + resolve_gate_topic_ids(&state, &stream_id, &topic_id) + .map_or(Ok(()), |(stream_id, topic_id)| { + permissioner.poll_messages(uid, stream_id, topic_id) + }) }, - )?; + )) + .await?; let wire = poll_wire_request(&stream_id, &topic_id, &query).map_err(ReadError::Rejected)?; let (namespace, partition_id, consumer, args) = match resolve_poll_request(&state.shard, &wire, HTTP_READ_CLIENT_ID) { @@ -1295,17 +1320,19 @@ pub(in crate::http) async fn get_consumer_offset( ) -> Result, ReadError> { let stream_id = Identifier::from_str_value(&stream_id).map_err(ReadError::Rejected)?; let topic_id = Identifier::from_str_value(&topic_id).map_err(ReadError::Rejected)?; - let scope = resolve_gate_topic_ids(&state, &stream_id, &topic_id); - authorize_read( + SendWrapper::new(gate_local_read( &state, &identity, consistency.consistency, + GET_CONSUMER_OFFSET_CODE, |permissioner, uid| { - scope.map_or(Ok(()), |(stream_id, topic_id)| { - permissioner.get_consumer_offset(uid, stream_id, topic_id) - }) + resolve_gate_topic_ids(&state, &stream_id, &topic_id) + .map_or(Ok(()), |(stream_id, topic_id)| { + permissioner.get_consumer_offset(uid, stream_id, topic_id) + }) }, - )?; + )) + .await?; let wire = consumer_offset_wire_request(&stream_id, &topic_id, &query).map_err(ReadError::Rejected)?; let (namespace, partition_id, consumer) = diff --git a/core/server/src/http/reads.rs b/core/server/src/http/reads.rs index 3859b28ae3..dc3262ff37 100644 --- a/core/server/src/http/reads.rs +++ b/core/server/src/http/reads.rs @@ -15,10 +15,15 @@ // specific language governing permissions and limitations // under the License. -//! Read-path gates: the shared per-op RBAC + consistency check, the local -//! metadata-STM read entry, and the wire/domain identifier resolvers the read -//! and data-plane routes ground their scopes through. +//! Read-path gates: the shared per-op RBAC + consistency check, the two waits +//! a local read serves behind (the post-restart recovery barrier and the +//! per-user read-your-writes frontier), the local metadata-STM read +//! entry, and the wire/domain identifier resolvers the read and data-plane +//! routes ground their scopes through. +use crate::dispatch::reads::{ + FrontierUnreached, FrontierWait, hold_for_frontier, read_needs_metadata_frontier, +}; use crate::shell::ServerShard; use bytes::Bytes; use consensus::MetadataHandle; @@ -38,25 +43,23 @@ use crate::responses::{ NonReplicatedResponse, build_non_replicated_response, resolve_stream_id, resolve_topic_id, }; -/// The two cross-cutting gates every authenticated read enforces before it -/// touches state. Factored out of [`read_local`] so the cross-shard client -/// reads (`get_clients` / `get_client`) - which serve from the shard session -/// managers, not the local STM, and so cannot use [`read_local`] - still pass -/// the identical gate. Keeping it in one place is what guarantees no read route -/// can silently skip authz or answer a linearizable request on a follower. -/// -/// Per-op RBAC: run the route's `rule` against the caller's committed -/// permissions via the live permissioner. A denial (always `Unauthorized`) -/// renders 403 through the legacy `IggyError -> status` map; root holds every -/// grant, so its reads pass without a user-id short-circuit. A linearizable -/// read must come from the primary; on a follower it redirects (307) to the -/// primary's HTTP address when resolvable, else fails closed to a 503 (see +/// The per-op RBAC + consistency check itself, without the waits: run the +/// route's `rule` against the caller's committed permissions via the live +/// permissioner. A denial (always `Unauthorized`) renders 403 through the +/// legacy `IggyError -> status` map; root holds every grant, so its reads pass +/// without a user-id short-circuit. A linearizable read must come from the +/// primary; on a follower it redirects (307) to the primary's HTTP address when +/// resolvable, else fails closed to a 503 (see /// [`HttpInner::not_primary_read_error`]). -pub(in crate::http) fn authorize_read( +/// +/// Every read route reaches this through [`gate_local_read`], which is what +/// pairs it with the two waits a local read must serve behind. Callable on its +/// own only for a read that is NOT served from local state. +fn authorize_read( state: &HttpInner, identity: &Identity, consistency: Consistency, - rule: impl FnOnce(&Permissioner, u32) -> Result<(), IggyError>, + rule: impl Fn(&Permissioner, u32) -> Result<(), IggyError>, ) -> Result<(), ReadError> { state .shard @@ -91,10 +94,9 @@ pub(in crate::http) async fn read_local( consistency: Consistency, code: u32, body: &[u8], - rule: impl FnOnce(&Permissioner, u32) -> Result<(), IggyError>, + rule: impl Fn(&Permissioner, u32) -> Result<(), IggyError>, ) -> Result { - await_recovery_barrier(&state.shard).await?; - authorize_read(state, identity, consistency, rule)?; + gate_local_read(state, identity, consistency, code, rule).await?; let clients_count = if code == GET_STATS_CODE { u32::try_from(SendWrapper::new(state.shard.list_all_clients()).await.len()) .unwrap_or(u32::MAX) @@ -117,6 +119,112 @@ pub(in crate::http) async fn read_local( } } +/// Every gate a read served from THIS node's state has to pass, in the one +/// order that is safe. +/// +/// The chokepoint for the whole REST read surface: [`read_local`] runs it for +/// the metadata-STM entity reads, and the routes that cannot use `read_local` +/// call it directly - the cross-shard client reads, which serve from each +/// shard's session manager; the snapshot route, which shells out; and the +/// partition reads, which answer from a partition group's log. Skipping it is +/// how a route silently loses authorization, the post-restart barrier, or the +/// read-your-writes hold; which of the two waits actually applies is +/// [`read_needs_metadata_frontier`]'s decision, not the caller's. +/// +/// Order: +/// 1. the recovery barrier, so nothing is served off a WAL suffix that is +/// about to re-commit; +/// 2. authorization, because it is terminal: the linearizable follower +/// redirect must answer 307 immediately rather than after a park, and +/// holding a connection to then answer 403 buys nothing; +/// 3. the read-your-writes hold; +/// 4. authorization AGAIN if that hold actually parked. Every scoped route's +/// rule resolves its entity when the rule RUNS, and a park is precisely +/// the case where the state machine moved under it: an entity that did not +/// exist on the first pass resolved to nothing, where a scope miss is a +/// pass-through, and would be served with no permissioner call at all. Only +/// the parked outcome pays for the second pass. +pub(in crate::http) async fn gate_local_read( + state: &HttpInner, + identity: &Identity, + consistency: Consistency, + code: u32, + rule: impl Fn(&Permissioner, u32) -> Result<(), IggyError>, +) -> Result<(), ReadError> { + await_recovery_barrier(&state.shard).await?; + authorize_read(state, identity, consistency, &rule)?; + if read_needs_metadata_frontier(code) + && await_metadata_read_frontier(state, identity).await? == FrontierWait::CaughtUp + { + authorize_read(state, identity, consistency, &rule)?; + } + Ok(()) +} + +/// Hold a local metadata read until this node has applied everything the +/// calling user was told committed. +/// +/// A committed control-plane reply hands the caller an op number; answering its +/// next read from a state machine below that op contradicts the response it is +/// holding. The lag is real on a node that is not the metadata primary: a +/// healthy backup FORWARDS a `Register` to the primary +/// (`dispatch::submit_register_local_or_forward`) and binds the committed epoch +/// while its own commit walk is still behind it, so the caller holds an op that +/// node has not applied before it has issued a single write. A control-plane +/// write posted to a backup never commits there either: it is relayed to the +/// primary (forwarding on) or refused transient (forwarding off), so every op a +/// backup promises is one it learned from the primary rather than applied +/// itself. +/// +/// Adjacent to `?consistency=linearizable`, not in competition with it. That +/// asks for the freshest CLUSTER state and is answered by leaving this node +/// (307 to the primary), which [`authorize_read`] decides before this wait and +/// which this wait never sees. This gate makes an UNQUALIFIED read +/// read-your-writes for its own user, at no redirect and no consensus round +/// trip. Per user rather than per credential because a bearer is not stable: +/// a refreshed access token is a new credential for the same writer (see +/// [`crate::http::state::MetadataWatermarks`]). +/// +/// Scope is this node's own view, which a RELAYED write is part of: the relay +/// records the serving primary's applied op off the response header +/// (`http::forward::record_relayed_floor`), so a caller that posted through +/// this follower is held here too. A user who wrote through a different node +/// entirely still leaves no floor here, and neither does a relayed write +/// answered with the `TransientNotCommitted` 503, whose op may have committed +/// anyway (see `http::submit::submit_committed`). +/// +/// The wait itself is the binary plane's [`hold_for_frontier`]: woken by the +/// commit that advances the frontier, bounded by a `compio::time` timer like +/// the recovery barrier below, since this listener is pinned to shard 0's +/// compio thread and has no blocking pool to hand a wait to. Only this request +/// parks, and it parks on one wake rather than a timer per tick. +async fn await_metadata_read_frontier( + state: &HttpInner, + identity: &Identity, +) -> Result { + let frontier = state.shard.plane.metadata().applied_frontier(); + let budget = frontier.read_budget(); + hold_for_frontier( + frontier, + state.metadata_watermark(identity.user_id), + compio::time::sleep(budget), + ) + .await + .map_err(|FrontierUnreached| { + state + .shard + .metrics() + .record_metadata_read_frontier_refusal(); + tracing::debug!( + frontier = frontier.get(), + watermark = state.metadata_watermark(identity.user_id), + ?budget, + "metadata read frontier unreached inside the budget; failing read with retryable 503" + ); + ReadError::MetadataFrontierUnreached + }) +} + /// One recovery-barrier check's outcome, factored out of [`await_recovery_barrier`] /// so the expiry decision is unit-testable without a runtime: the loop reads the /// clock and injects whether the deadline has passed. @@ -161,7 +269,9 @@ const fn barrier_state(barrier: u64, commit_min: u64, expired: bool) -> BarrierW pub(in crate::http) async fn await_recovery_barrier( shard: &Rc, ) -> Result<(), ReadError> { - const POLL: std::time::Duration = std::time::Duration::from_millis(10); + // The consensus tick, like the read gate's cadence: what lifts this + // barrier is a commit walk, which advances on that clock. + const POLL: std::time::Duration = consensus::TICK_INTERVAL; let Some(consensus) = shard.plane.metadata().consensus.as_ref() else { return Ok(()); @@ -290,7 +400,122 @@ pub(in crate::http) fn authorize_data_plane( #[cfg(test)] mod tests { - use super::{BarrierWait, barrier_state}; + use super::{ + BarrierWait, FrontierWait, ReadError, barrier_state, hold_for_frontier, + read_needs_metadata_frontier, + }; + use crate::http::state::MetadataWatermarks; + use iggy_binary_protocol::codes::{ + DESCRIBE_OPTIONS_CODE, GET_CONSUMER_GROUPS_CODE, GET_PERSONAL_ACCESS_TOKENS_CODE, + GET_STATS_CODE, GET_STREAM_CODE, GET_STREAMS_CODE, GET_TOPIC_CODE, GET_TOPICS_CODE, + GET_USER_CODE, GET_USERS_CODE, + }; + use metadata::AppliedFrontier; + use std::future::pending; + use std::sync::Arc; + + /// Root's user id, the caller every fixture below writes and reads as. + const USER: u32 = 0; + + /// The gate exactly as [`await_metadata_read_frontier`] composes it: the + /// per-user floor a committed control-plane reply left behind, against the + /// node-wide applied frontier. Whether a read is HELD is decided by those + /// two numbers and nothing else, so this is the plane's own contract: + /// seeded floor, read held, commit that closes the gap serves it. + /// + /// The budget is a future that never completes, so a gate that failed to + /// park would answer instead of hanging, and a gate that failed to wake + /// would hang instead of answering. The end-to-end REST path (a follower + /// that binds a forwarded epoch, then reads through axum) is + /// `integration::server::http_read_your_writes`; the race it depends on + /// cannot be forced from outside the process, which is why the hold is + /// pinned here. + #[compio::test] + async fn given_a_recorded_floor_when_the_node_catches_up_should_serve_the_held_read() { + let watermarks = MetadataWatermarks::default(); + let frontier = Arc::new(AppliedFrontier::default()); + frontier.advance(4); + + // A caller this node never wrote for waits for nothing. + assert_eq!( + hold_for_frontier(&frontier, watermarks.get(USER), pending()).await, + Ok(FrontierWait::Ready), + "an unseeded caller has no write to read back" + ); + + // The committed reply's op becomes the floor, which is above what this + // node has applied: the read must not be answered yet. + watermarks.record(USER, 9); + let committer = Arc::clone(&frontier); + compio::runtime::spawn(async move { + compio::runtime::time::sleep(std::time::Duration::ZERO).await; + committer.advance(9); + }) + .detach(); + assert_eq!( + hold_for_frontier(&frontier, watermarks.get(USER), pending()).await, + Ok(FrontierWait::CaughtUp), + "the read must be held until the node applies the caller's own op" + ); + + // Caught up: back to the fast path, with the floor still in place. + assert_eq!( + hold_for_frontier(&frontier, watermarks.get(USER), pending()).await, + Ok(FrontierWait::Ready), + "an applied floor must not cost a park on every later read" + ); + } + + /// The refusal this plane renders when the frontier never arrives: the + /// shared retryable 503, never a 2xx carrying the pre-write state. + #[compio::test] + async fn given_a_floor_the_node_never_reaches_when_gating_should_render_the_retryable_503() { + let watermarks = MetadataWatermarks::default(); + watermarks.record(USER, 9); + let frontier = AppliedFrontier::default(); + frontier.advance(4); + + let outcome = hold_for_frontier(&frontier, watermarks.get(USER), std::future::ready(())) + .await + .map_err(|_| ReadError::MetadataFrontierUnreached); + assert!( + matches!(outcome, Err(ReadError::MetadataFrontierUnreached)), + "an unreached frontier must refuse the read, not serve it" + ); + } + + /// The HTTP read routes share the binary dispatch's exclusion list, so this + /// pins what that list means for the codes HTTP actually serves: every + /// entity read is gated, and the static option catalog is not. Forking the + /// list per plane is what this is here to catch. + /// + /// `GET_CLUSTER_METADATA` is deliberately absent: `/cluster/metadata` has + /// its own local handler and never reaches [`read_local`], so asserting it + /// here would pin a code this plane cannot produce. The shared predicate's + /// own arm for it is covered where it is used, in the dispatch spine. + #[test] + fn given_the_http_read_codes_when_classified_should_gate_all_but_the_static_catalog() { + for code in [ + GET_STREAMS_CODE, + GET_STREAM_CODE, + GET_TOPICS_CODE, + GET_TOPIC_CODE, + GET_USERS_CODE, + GET_USER_CODE, + GET_CONSUMER_GROUPS_CODE, + GET_PERSONAL_ACCESS_TOKENS_CODE, + GET_STATS_CODE, + ] { + assert!( + read_needs_metadata_frontier(code), + "code {code} answers from the metadata STM and must be gated" + ); + } + assert!( + !read_needs_metadata_frontier(DESCRIBE_OPTIONS_CODE), + "the option catalog is static; holding a read of it buys nothing" + ); + } #[test] fn barrier_state_ready_when_no_barrier_armed() { diff --git a/core/server/src/http/reply.rs b/core/server/src/http/reply.rs index 29babb6b3f..d44d18141b 100644 --- a/core/server/src/http/reply.rs +++ b/core/server/src/http/reply.rs @@ -38,6 +38,7 @@ use tracing::warn; use crate::dispatch::login_error::LoginRegisterError; use crate::http::error::{PartitionWriteError, WriteError}; +use crate::responses::reply_body; /// Discriminate a partition write reply. Partition replies carry no result /// section - a denial is empty-bodied and a committed body, where there is one, @@ -146,33 +147,6 @@ pub(in crate::http) fn committed_payload( } } -/// The transient variant of a reply-shaped pre-consensus rejection frame -/// (`[count=1][index=0][code]`, see `build_result_rejection_reply`), or `None` -/// for a committed outcome. Either transient means the op did not commit, so -/// the write path must replay the same request id rather than grade it as a -/// committed result or advance the session gate. The two codes are kept -/// distinct because they exhaust differently: `TransientNotAccepted` never -/// entered the pipeline and is safe to re-issue anywhere, while -/// `TransientNotCommitted` may still commit and only a same-session same-id -/// replay is safe. -pub(in crate::http) fn transient_code(reply: &Message) -> Option { - match result_code(reply_body(reply)) { - Some(code) if code == IggyError::TransientNotCommitted.as_code() => { - Some(IggyError::TransientNotCommitted) - } - Some(code) if code == IggyError::TransientNotAccepted.as_code() => { - Some(IggyError::TransientNotAccepted) - } - _ => None, - } -} - -/// The reply body past the generic header, bounded by the header's `size`. -fn reply_body(reply: &Message) -> &[u8] { - let size = reply.header().size as usize; - reply.as_slice().get(HEADER_SIZE..size).unwrap_or_default() -} - /// Decode the `GetStreamResponse` payload of a committed create-stream reply into /// `StreamDetails`. `payload` is the slice past the result section that /// [`submit_write`] already validated as a success. @@ -271,7 +245,7 @@ mod tests { use crate::responses::{ NonReplicatedResponse, build_deny_reply, build_empty_reply, build_reply_from_bytes, - build_reply_with_body, + build_reply_with_body, transient_code, }; use crate::http::wire::build_request_message; diff --git a/core/server/src/http/session.rs b/core/server/src/http/session.rs index b03b94bf58..b645ce5fd7 100644 --- a/core/server/src/http/session.rs +++ b/core/server/src/http/session.rs @@ -259,17 +259,7 @@ mod tests { /// construction plus the live cancellation smoke, not faked here. #[compio::test] async fn detached_task_advances_gate_and_ignores_dead_receiver() { - let session = Rc::new(HttpSession { - key: "jwt:test".to_owned(), - client_id: 7, - session: 1, - user_id: DEFAULT_ROOT_USER_ID, - expiry: u64::MAX, - gate: Mutex::new(FIRST_REQUEST_ID), - data_gate: Mutex::new(FIRST_REQUEST_ID), - registry_token: Cell::new(None), - in_flight_writes: Cell::new(0), - }); + let session = fake_session("jwt:test", 7, u64::MAX); let (result_slot, committed) = oneshot::channel::(); // The handler future dies (client disconnect) before the task runs. drop(committed); diff --git a/core/server/src/http/state.rs b/core/server/src/http/state.rs index 5f1984cd41..c55acfc661 100644 --- a/core/server/src/http/state.rs +++ b/core/server/src/http/state.rs @@ -57,6 +57,73 @@ use crate::shell::ServerShard; /// follower's possibly-stale one. pub(in crate::http) const VIEW_HEADER: HeaderName = HeaderName::from_static("iggy-view"); +/// Response header carrying the SERVING node's applied metadata op, stamped by +/// [`insert_view_header`] on the same responses as [`VIEW_HEADER`]. +/// +/// Load-bearing, not diagnostic: a follower that RELAYS a control-plane write +/// to the primary never runs the local write path, so nothing would record +/// what that caller was told committed, and its next unqualified GET - which +/// stays local - could answer from before its own write. The relay reads this +/// header off the primary's response and records it as the caller's floor (see +/// `http::forward`). Filled if absent, so a relayed response keeps the serving +/// node's number rather than the relaying follower's lower one. +/// +/// The op is a floor, not the caller's exact commit: the primary applies a +/// metadata op before it replies, so its applied frontier at reply time is at +/// or above the op the caller now holds. Above means waiting for a few of +/// someone else's committed ops too, which is stronger than read-your-writes +/// and never weaker. +pub(in crate::http) const APPLIED_OP_HEADER: HeaderName = + HeaderName::from_static("iggy-applied-op"); + +/// Per-user read-your-writes floors: the highest metadata op each user has +/// been told committed BY THIS NODE. +/// +/// Keyed by user id, and held outside the session table, both deliberately. A +/// session entry is dropped outright when its VSR slot dies +/// ([`HttpInner::forget_session`]) or when the expiry sweep runs, and +/// `POST /users/refresh-token` answers with a fresh `jti` that registers no +/// session at all: a floor living in the session entry, or keyed by the +/// credential, reads `0` again in all three cases while the caller's bearer +/// stays valid - the stale read this exists to prevent, for exactly the +/// callers still holding a committed reply. Keying by user also bounds the +/// table by the user count instead of by every token ever minted. +/// +/// One user's floor is shared by its credentials, which is stronger than +/// read-your-writes and never weaker: the extra ops a second credential waits +/// for are the same user's. +#[derive(Debug, Default)] +pub(in crate::http) struct MetadataWatermarks(RefCell>); + +impl MetadataWatermarks { + /// Highest metadata op `user_id` was told committed here, or `0` when it + /// was told none - no write of its ever ran on this node, so there is + /// nothing to read back. + /// + /// Deliberately not expiry-filtered: the number is a consistency floor, + /// not a capability, and the request that consults it has already + /// re-verified the bearer. + /// + /// Confines the `RefCell` borrow to this call, so it can never span the + /// read gate's `.await`. + pub(in crate::http) fn get(&self, user_id: u32) -> u64 { + self.0.borrow().get(&user_id).copied().unwrap_or(0) + } + + /// Raise `user_id`'s floor to `commit`. Monotone, so a reply that lands + /// out of order (concurrent requests on one credential are legal) cannot + /// lower it. + /// + /// Only COMMITTED metadata replies belong here; see + /// [`crate::dispatch::submit::committed_reply_commit`] for what that + /// excludes and why. + pub(in crate::http) fn record(&self, user_id: u32, commit: u64) { + let mut floors = self.0.borrow_mut(); + let floor = floors.entry(user_id).or_insert(0); + *floor = (*floor).max(commit); + } +} + /// Axum router state: shard-0's [`HttpInner`] behind an `Rc`, `!Send` yet /// bridged into axum's `Send + Sync` requirement by `SendWrapper`. Sound /// because the listener and every handler run on shard 0's compio thread - the @@ -122,6 +189,10 @@ pub(in crate::http) struct HttpInner { /// Legacy-parity metric registry served by the scrape route; the router's /// counting layer holds a clone of its request counter. pub(in crate::http) metrics: HttpMetrics, + /// Per-user read-your-writes floors the read gate holds reads against. + /// Behind `Rc` because the write path records from a detached task that + /// outlives its handler by design (see `submit_committed`). + pub(in crate::http) metadata_watermarks: Rc, } impl HttpInner { @@ -229,6 +300,13 @@ impl HttpInner { } } + /// Highest metadata op `user_id` was told committed here; see + /// [`MetadataWatermarks`] for why the floor is per user and lives outside + /// the session table. + pub(in crate::http) fn metadata_watermark(&self, user_id: u32) -> u64 { + self.metadata_watermarks.get(user_id) + } + /// Clone the live (non-expired) entry for `key`, if present. Confines the /// shared `RefCell` borrow to this call so it can never span an `.await`. fn live_session(&self, key: &str, now_secs: u64) -> Option> { @@ -360,6 +438,12 @@ impl HttpInner { ); return Err(AuthError::SessionIdTaken); } + // `bound.epoch` also floors the read gate: a HEALTHY BACKUP forwards the + // register to the primary (see `submit_register_local_or_forward`), so + // this node can hand back an epoch its own commit walk has not + // reached, and the caller's first read would otherwise be served from + // state older than the register it is holding. + self.metadata_watermarks.record(user_id, bound.epoch); Ok(Rc::new(HttpSession { key, client_id, @@ -454,15 +538,57 @@ pub(in crate::http) fn insert_view_header(state: &HttpInner, mut response: Respo .entry(VIEW_HEADER) .or_insert(HeaderValue::from(consensus.view())); } + // Same fill-if-absent rule, and for the same reason: the relay needs the + // op the SERVING node had applied, not this one's (see + // [`APPLIED_OP_HEADER`]). + response + .headers_mut() + .entry(APPLIED_OP_HEADER) + .or_insert(HeaderValue::from( + state.shard.plane.metadata().applied_frontier().get(), + )); response } #[cfg(test)] mod tests { - use super::register_submit_auth_error; + use super::{MetadataWatermarks, register_submit_auth_error}; use crate::http::error::AuthError; use metadata::MetadataSubmitError; + /// The floor is what the read gate waits for, so nothing may lower it: two + /// concurrent requests by one user can have their committed replies land + /// out of order, and the later-but-lower reply must not undo the + /// earlier-but-higher one. + #[test] + fn given_out_of_order_replies_when_recording_should_keep_the_floor_monotone() { + const USER: u32 = 3; + + let watermarks = MetadataWatermarks::default(); + assert_eq!( + watermarks.get(USER), + 0, + "a user this node never wrote for was promised nothing" + ); + + watermarks.record(USER, 50); + watermarks.record(USER, 7); + assert_eq!( + watermarks.get(USER), + 50, + "a lower commit must not lower the floor" + ); + } + + /// One user's floor is not another's: a busy writer must not park an + /// unrelated user's reads behind ops it never issued. + #[test] + fn given_two_users_when_one_writes_should_leave_the_other_floor_alone() { + let watermarks = MetadataWatermarks::default(); + watermarks.record(1, 50); + assert_eq!(watermarks.get(2), 0); + } + #[test] fn register_submit_errors_preserve_known_and_unknown_outcomes() { for error in [ diff --git a/core/server/src/http/submit.rs b/core/server/src/http/submit.rs index 0babb39528..a4864086e0 100644 --- a/core/server/src/http/submit.rs +++ b/core/server/src/http/submit.rs @@ -34,16 +34,15 @@ use tracing::warn; use crate::dispatch::partition::{dispatch_partition_request, resolve_delete_segments_truncate}; use crate::dispatch::session_ops::submit_logout_on_owner; -use crate::dispatch::submit::submit_client_request_on_owner; +use crate::dispatch::submit::{committed_reply_commit, submit_client_request_on_owner}; use crate::http::admission::admit_partition_write; use crate::http::error::{PartitionWriteError, WriteError}; -use crate::http::reply::{ - classify_partition_reply, committed_payload, eviction_error, transient_code, -}; +use crate::http::reply::{classify_partition_reply, committed_payload, eviction_error}; use crate::http::session::HttpSession; use crate::http::state::HttpInner; use crate::http::wire::build_request_message; use crate::pat::rewrite_pat_request_for_user; +use crate::responses::transient_code; use crate::shell::ServerShard; use crate::users::maybe_rewrite_user_password_request; use crate::wire::request_body; @@ -106,6 +105,7 @@ pub(in crate::http) async fn submit_committed( let (result_slot, committed) = oneshot::channel(); let shard = Rc::clone(&state.shard); let task_session = Rc::clone(session); + let watermarks = Rc::clone(&state.metadata_watermarks); let body = body.to_vec(); let max_tokens_per_user = state.max_tokens_per_user; // Detached so a client disconnect cannot abandon the gate mid-submit; @@ -113,6 +113,25 @@ pub(in crate::http) async fn submit_committed( compio::runtime::spawn(async move { let result = submit_gated(&shard, &task_session, operation, max_tokens_per_user, &body).await; + // Recorded here rather than after the await below, for the same reason + // the submit is detached: a caller that disconnected mid-write still + // committed the op, and its next request as this user must not be + // served state older than what committed. Ordered before the wake, so a + // read issued the instant the response lands already sees the mark. + // + // A follower with HTTP forwarding ON never runs this task: the + // middleware relays the write and records the floor from the serving + // primary's applied op instead (`http::forward::record_relayed_floor`). + // One gap survives that split - a relayed 503 carrying + // `TransientNotCommitted` is passed through untouched rather than + // retried, because its op may still commit, and only a 2xx records a + // floor. A write that did commit behind that code therefore leaves + // none, until this caller's next committed write raises it. + if let Ok((_, reply, _)) = &result + && let Some(commit) = committed_reply_commit(reply) + { + watermarks.record(task_session.user_id, commit); + } // A failed send means the handler died mid-await; the submit itself // already completed, which is the invariant that matters. let _ = result_slot.send(result); diff --git a/core/server/src/lib.rs b/core/server/src/lib.rs index 1b1036f0ee..b8131e558a 100644 --- a/core/server/src/lib.rs +++ b/core/server/src/lib.rs @@ -47,6 +47,7 @@ pub(crate) mod pat; pub(crate) mod responses; pub mod session_manager; pub mod shell; + pub(crate) mod users; pub(crate) mod wire; diff --git a/core/server/src/responses.rs b/core/server/src/responses.rs index 4c48c0514b..75a8d2511b 100644 --- a/core/server/src/responses.rs +++ b/core/server/src/responses.rs @@ -103,6 +103,7 @@ use std::rc::Rc; use std::sync::{Arc, OnceLock}; use sysinfo::System as SysinfoSystem; use system_stats::SystemProbe; +use tracing::warn; /// Build the `get_me` reply for the requesting connection. Identity /// (`user_id`, transport kind, peer address) comes from the per-shard @@ -1542,6 +1543,94 @@ pub fn build_reply_from_bytes( ) } +/// The reply body past the generic header, bounded by the header's `size` +/// rather than by the buffer length: `size` is the frame's authoritative +/// extent, so a short frame reads as "no result section" instead of into +/// allocation padding. +#[must_use] +pub fn reply_body(reply: &Message) -> &[u8] { + let size = reply.header().size as usize; + reply + .as_slice() + .get(std::mem::size_of::()..size) + .unwrap_or_default() +} + +/// The header of a SUCCESSFULLY COMMITTED metadata reply, or `None` when the +/// frame promises the caller nothing. +/// +/// Three checks, in this order, and both callers need all three: +/// +/// - an eviction is an `EvictionHeader` whose bytes would cast cleanly as a +/// `ReplyHeader`, so the command is checked FIRST: casting it would both +/// swallow the eviction and grade it as a commit; +/// - a request-level denial names itself in `ReplyHeader.status`, the channel +/// the SDK peeks before body decode (see [`build_deny_reply`]); +/// - a nonzero result section is a rejection, transient or committed. Every +/// reply here is result-framed (`Operation::is_result_framed` covers the +/// metadata ops; the partition plane grades through +/// `classify_partition_reply` instead), so a missing section is a malformed +/// frame, not a bare payload. +/// +/// The read-your-writes floor and the raw-PAT splice both hang off exactly +/// this predicate - the floor must not advance on a frame that committed +/// nothing, and the token must not be grafted onto a rejection body - so they +/// share one implementation rather than two that have to stay in step. +/// +/// A frame too short to hold a header, or one whose header will not cast, is +/// `None` with a warning: it is malformed, and the alternative is a panic on +/// the reply path. +#[must_use] +pub fn committed_reply_header(reply: &Message) -> Option<&ReplyHeader> { + if reply.header().command != Command::Reply { + return None; + } + let Some(bytes) = reply.as_slice().get(..std::mem::size_of::()) else { + warn!( + size = reply.header().size, + "metadata reply shorter than its own header" + ); + return None; + }; + let header = match bytemuck::checked::try_from_bytes::(bytes) { + Ok(header) => header, + Err(error) => { + warn!(?error, "metadata reply header failed to cast"); + return None; + } + }; + if header.status != 0 || result_code(reply_body(reply)) != Some(0) { + return None; + } + Some(header) +} + +/// The transient variant of a reply-shaped pre-consensus rejection frame +/// (`[count=1][index=0][code]`, see `build_result_rejection_reply`), or `None` +/// for a committed outcome. Either transient means the op did not commit, so +/// the write path must replay the same request id rather than grade it as a +/// committed result or advance the session gate. The two codes are kept +/// distinct because they exhaust differently: `TransientNotAccepted` never +/// entered the pipeline and is safe to re-issue anywhere, while +/// `TransientNotCommitted` may still commit and only a same-session same-id +/// replay is safe. +/// +/// Lives here rather than in the HTTP reply module both planes' write paths +/// grade through: the dispatch spine needs it too, and importing it from +/// `http` would close a module cycle. +#[must_use] +pub fn transient_code(reply: &Message) -> Option { + match result_code(reply_body(reply)) { + Some(code) if code == IggyError::TransientNotCommitted.as_code() => { + Some(IggyError::TransientNotCommitted) + } + Some(code) if code == IggyError::TransientNotAccepted.as_code() => { + Some(IggyError::TransientNotAccepted) + } + _ => None, + } +} + /// If a raw PAT token was minted (`CreatePersonalAccessToken`) and the commit /// succeeded, replace the committed reply -- whose body is empty because the /// raw token never entered consensus -- with a `RawPersonalAccessTokenResponse`, @@ -1556,40 +1645,15 @@ pub fn build_raw_pat_reply( let Some(raw) = raw_token else { return Ok(committed); }; - // `submit_request_in_process` hands back an `EvictionHeader`-backed message - // on the evict outcome (e.g. a `CreatePersonalAccessToken` whose session - // was evicted between bind and request). Its byte pattern is a valid - // `ReplyHeader`, so the checked cast below would silently pass and we would - // both swallow the eviction and ship a raw token whose hash never - // committed. Only rewrite a genuine committed `Reply`; pass anything else - // (the eviction) through untouched so the client learns its session died. - if committed.header().command != Command::Reply { + // Only a genuine committed success gets the secret spliced in. An eviction + // frame (a `CreatePersonalAccessToken` whose session died between bind and + // request), a request-level denial, and a rejection result section all pass + // through untouched, so the client decodes the typed outcome - or, for a + // transient, replays - instead of having a raw token grafted onto a + // rejection body whose hash never committed. + let Some(commit) = committed_reply_header(&committed).map(|header| header.commit) else { return Ok(committed); - } - let header_len = std::mem::size_of::(); - let committed_header = - bytemuck::checked::try_from_bytes::(&committed.as_slice()[..header_len]) - .map_err(|_| IggyError::InvalidFormat)?; - let commit = committed_header.commit; - let size = committed_header.size as usize; - // A `Reply` whose result section is nonzero is not a successful commit: - // a committed business rejection (duplicate name, invalid expiry) or a - // `TransientNotCommitted` retry frame, both with no payload and no token - // to ship. Splice the secret only into a genuine success; pass everything - // else through untouched so the client decodes the typed result (and, for - // a transient, replays) instead of having a raw token grafted onto a - // rejection body. Mirrors the HTTP handler's `committed_payload` gate. - // - // Bounded by the header's own `size` rather than running to the end of the - // buffer, so a short frame reads as "no result section" instead of into - // allocation padding. - let reply_body = committed - .as_slice() - .get(header_len..size) - .unwrap_or_default(); - if result_code(reply_body) != Some(0) { - return Ok(committed); - } + }; let token = WireName::new(raw.as_str()).map_err(|_| IggyError::InvalidFormat)?; let response = RawPersonalAccessTokenResponse { token }; let reply = build_result_framed_reply( diff --git a/core/server/src/session_manager.rs b/core/server/src/session_manager.rs index 84635a8b4f..d09bd4c079 100644 --- a/core/server/src/session_manager.rs +++ b/core/server/src/session_manager.rs @@ -79,6 +79,18 @@ pub struct Connection { pub last_heartbeat: Instant, /// Recorded at login; `None` until the connection authenticates. pub sdk: Option, + /// Highest metadata op this connection has been told committed. + /// + /// Seeded from the bound session (the register's own commit op, which + /// floors everything the client committed before it re-homed) and raised by + /// every committed reply relayed on this connection. The read gate holds a + /// local read until the node's applied frontier covers it, so a client + /// cannot be served state older than a write it already saw acked. + /// + /// Per-connection rather than per-client: the number only has to cover what + /// THIS socket was told, and a client that reconnects re-seeds from the + /// session it binds. + pub metadata_watermark: u64, } /// Bridges transport connections to consensus sessions. @@ -141,6 +153,7 @@ impl SessionManager { state: ConnectionState::Connected, last_heartbeat: Instant::now(), sdk: None, + metadata_watermark: 0, }); } @@ -215,6 +228,13 @@ impl SessionManager { match conn.state { ConnectionState::Connected => { conn.state = ConnectionState::Authenticated { user_id }; + // The floor belongs to whoever was told those ops committed, + // and this socket now serves someone else: a `Connected` + // connection is either fresh or one `bind_session` demoted, so + // carrying the old mark over would make the new login wait for + // a write it never issued. Never the other direction - the + // bind below re-seeds from the register epoch. + conn.metadata_watermark = 0; Ok(()) } _ => Err(SessionError::InvalidTransition { @@ -268,15 +288,47 @@ impl SessionManager { } // Now mutate the target connection. - self.connections.get_mut(&connection_id).unwrap().state = ConnectionState::Bound { + let bound = self + .connections + .get_mut(&connection_id) + .expect("bind_session: connection validated above, single-threaded"); + bound.state = ConnectionState::Bound { user_id, client_id, session, }; + // The session IS the register's commit op, so it floors every metadata + // op this client saw committed before it re-homed here. Without the + // seed a re-homed connection reads at zero and the gate admits the + // pre-write state its own last write already replaced. + bound.metadata_watermark = bound.metadata_watermark.max(session); self.client_to_connection.insert(client_id, connection_id); Ok(()) } + /// Raise this connection's metadata watermark to `commit`. Monotone, so a + /// late or out-of-order reply cannot lower it; no-op for an unknown + /// connection. + /// + /// Only committed replies belong here. A pre-consensus rejection stamps the + /// primary's `commit_max`, which is an op this connection was never + /// promised and, on a backup-homed connection, one it would then wait for. + pub fn record_metadata_watermark(&mut self, connection_id: u128, commit: u64) { + if let Some(conn) = self.connections.get_mut(&connection_id) { + conn.metadata_watermark = conn.metadata_watermark.max(commit); + } + } + + /// The highest metadata op this connection was told committed, or `0` when + /// it was told none (an unknown or still-unbound connection, which has no + /// write to read back). + #[must_use] + pub fn metadata_watermark(&self, connection_id: u128) -> u64 { + self.connections + .get(&connection_id) + .map_or(0, |conn| conn.metadata_watermark) + } + /// Look up the consensus session for a connection. /// /// Returns `(client_id, session)` if the connection is `Bound`, `None` otherwise. @@ -302,13 +354,17 @@ impl SessionManager { .map(|conn| conn.address) } - /// Acting user and transport peer address for a connection, in one map - /// lookup: the non-replicated dispatch path needs both, and the separate - /// accessors would walk the connection map twice per request. + /// Acting user, transport peer address and metadata watermark for a + /// connection, in one map lookup: the non-replicated dispatch path needs + /// all three per request, and the separate accessors would walk the + /// connection map (and take the shared borrow) once each. + /// + /// An unknown connection reads as a watermark of `0`: it was promised + /// nothing, so its reads wait for nothing. #[must_use] - pub fn read_context(&self, connection_id: u128) -> (Option, Option) { + pub fn read_context(&self, connection_id: u128) -> (Option, Option, u64) { let Some(conn) = self.connections.get(&connection_id) else { - return (None, None); + return (None, None, 0); }; let user_id = match conn.state { ConnectionState::Authenticated { user_id } | ConnectionState::Bound { user_id, .. } => { @@ -316,7 +372,7 @@ impl SessionManager { } ConnectionState::Connected => None, }; - (user_id, Some(conn.address)) + (user_id, Some(conn.address), conn.metadata_watermark) } /// Look up the authenticated user id for a connection. @@ -603,4 +659,78 @@ mod tests { "a second disconnect has nothing left to release" ); } + + /// The bind seed is what makes a re-homed connection safe: the register's + /// commit op floors every metadata op the client committed elsewhere, so + /// the read gate cannot admit the pre-write state on a node that has not + /// caught up. A recorder that could lower the mark would undo it. + #[test] + fn given_a_bound_connection_when_replies_arrive_should_keep_the_watermark_monotone() { + let mut mgr = SessionManager::new(); + let conn = 1; + mgr.ensure_connection(conn, addr(5200), ClientTransportKind::Tcp); + assert_eq!( + mgr.metadata_watermark(conn), + 0, + "an unbound connection was promised nothing" + ); + + mgr.login(conn, 3).unwrap(); + mgr.bind_session(conn, 100, 42).unwrap(); + assert_eq!( + mgr.metadata_watermark(conn), + 42, + "the bound session is the register's commit op and floors the mark" + ); + + mgr.record_metadata_watermark(conn, 50); + mgr.record_metadata_watermark(conn, 7); + assert_eq!( + mgr.metadata_watermark(conn), + 50, + "a lower commit must not lower the mark" + ); + } + + /// A socket that logs in again is serving a new caller, so it must not + /// inherit the floor of the one before it: the mark is what the PREVIOUS + /// login was told committed, and waiting for it would only ever delay the + /// new one. + #[test] + fn given_a_rebound_connection_when_it_logs_in_again_should_start_from_no_floor() { + let mut mgr = SessionManager::new(); + let conn = 1; + mgr.ensure_connection(conn, addr(5201), ClientTransportKind::Tcp); + mgr.login(conn, 3).unwrap(); + mgr.bind_session(conn, 100, 42).unwrap(); + mgr.record_metadata_watermark(conn, 50); + + // `bind_session` for the same client id on ANOTHER connection demotes + // this one to `Connected`, which is the state a re-login accepts. + mgr.ensure_connection(2, addr(5202), ClientTransportKind::Tcp); + mgr.login(2, 3).unwrap(); + mgr.bind_session(2, 100, 43).unwrap(); + assert_eq!( + mgr.metadata_watermark(conn), + 50, + "the demotion alone leaves the mark; the re-login is what clears it" + ); + + mgr.login(conn, 7).unwrap(); + assert_eq!( + mgr.metadata_watermark(conn), + 0, + "a different user on this socket was promised nothing" + ); + } + + /// An unknown connection is not an error: the disconnect callback can win + /// the race against a reply relay, and a gate reading `0` then serves the + /// read instead of parking a socket that is already gone. + #[test] + fn given_an_unknown_connection_when_recording_a_watermark_should_be_inert() { + let mut mgr = SessionManager::new(); + mgr.record_metadata_watermark(9, 5); + assert_eq!(mgr.metadata_watermark(9), 0); + } } diff --git a/core/shard/src/lib.rs b/core/shard/src/lib.rs index fe4deaea5a..a8adfe0667 100644 --- a/core/shard/src/lib.rs +++ b/core/shard/src/lib.rs @@ -6744,13 +6744,13 @@ where self.note_metadata_transfer_progress(); self.metadata_transfer_decode_failures.set(None); if outcome.pairing_durable { - // `applied_frontier`, not the transferred snapshot's op: the install + // `installed_frontier`, not the transferred snapshot's op: the install // returns `max(snapshot_seq, local_applied)`, which differs whenever a // serving peer offers a snapshot BEHIND this replica (checkpoints are // node-local) and the local state machine is kept instead. tracing::info!( shard = self.id, - applied_frontier = outcome.applied_frontier, + installed_frontier = outcome.installed_frontier, commit_op, table_frontier, "metadata state transfer installed; handing tail to journal repair" @@ -6761,7 +6761,7 @@ where // let every one of them pass on the degraded path. tracing::warn!( shard = self.id, - applied_frontier = outcome.applied_frontier, + installed_frontier = outcome.installed_frontier, commit_op, table_frontier, "metadata state transfer landed WITHOUT a durable checkpoint \ diff --git a/core/shard/src/metrics.rs b/core/shard/src/metrics.rs index 4fbd8c7fcf..fcf885a097 100644 --- a/core/shard/src/metrics.rs +++ b/core/shard/src/metrics.rs @@ -208,6 +208,8 @@ pub struct ShardMetrics { partition_frames_rejected_ahead_total: Counter, partition_requests_denied_transient_total: Counter, partition_repair_serves_deferred_purge_total: Counter, + metadata_read_frontier_refusals_total: Counter, + client_requests_denied_queue_full_total: Counter, } impl ShardMetrics { @@ -232,9 +234,49 @@ impl ShardMetrics { partition_frames_rejected_ahead_total: Counter::default(), partition_requests_denied_transient_total: Counter::default(), partition_repair_serves_deferred_purge_total: Counter::default(), + metadata_read_frontier_refusals_total: Counter::default(), + client_requests_denied_queue_full_total: Counter::default(), } } + /// Bumped every time a client request is answered with a retryable denial + /// because that client already has the maximum number of requests queued + /// behind one the shard has not answered yet. + /// + /// The queue only grows while a client pipelines faster than its own + /// frames are served, so a sustained rate means one connection is stalled + /// on something - a held metadata read, a slow commit - while it keeps + /// sending. + pub fn record_client_request_denied_queue_full(&self) { + self.client_requests_denied_queue_full_total.inc(); + } + + /// Current value of [`Self::record_client_request_denied_queue_full`], for + /// tests that assert the denial was counted. + #[must_use] + pub fn client_requests_denied_queue_full_value(&self) -> u64 { + self.client_requests_denied_queue_full_total.get() + } + + /// Bumped every time a metadata read is refused because this node's + /// applied frontier never reached what the caller was told committed. + /// + /// The counter is the signal, not a log line: a node that lags durably + /// refuses every held read of every client for as long as it lags, so the + /// refusal itself logs at `debug!` and this is what a dashboard alerts on. + /// Any sustained rate means reads on this node are failing retryable while + /// its commit walk stays behind. + pub fn record_metadata_read_frontier_refusal(&self) { + self.metadata_read_frontier_refusals_total.inc(); + } + + /// Current value of [`Self::record_metadata_read_frontier_refusal`], for + /// tests that assert a refusal was counted rather than scraping it. + #[must_use] + pub fn metadata_read_frontier_refusals_value(&self) -> u64 { + self.metadata_read_frontier_refusals_total.get() + } + /// Increment `frame_drops_total{variant, reason}` by 1. /// /// Callers should pass label constants from [`frame_drop_variant`] @@ -496,6 +538,16 @@ impl ShardMetrics { "partition repair serves or completions deferred until a committed purge applies", self.partition_repair_serves_deferred_purge_total.clone(), ); + registry.register( + "metadata_read_frontier_refusals", + "metadata reads refused because this node never applied the caller's committed op", + self.metadata_read_frontier_refusals_total.clone(), + ); + registry.register( + "client_requests_denied_queue_full", + "client requests denied retryable because that client's request queue was full", + self.client_requests_denied_queue_full_total.clone(), + ); } } diff --git a/core/simulator/src/client.rs b/core/simulator/src/client.rs index 1b1501aeed..99fe11a7f4 100644 --- a/core/simulator/src/client.rs +++ b/core/simulator/src/client.rs @@ -16,7 +16,7 @@ // under the License. use bytes::{Bytes, BytesMut}; -use iggy_binary_protocol::codes::POLL_MESSAGES_CODE; +use iggy_binary_protocol::codes::{GET_STREAM_CODE, POLL_MESSAGES_CODE}; use iggy_binary_protocol::primitives::consumer::WireConsumer; use iggy_binary_protocol::requests::consumer_groups::{ CreateConsumerGroupRequest, DeleteConsumerGroupRequest, @@ -36,7 +36,8 @@ use iggy_binary_protocol::requests::personal_access_tokens::{ }; use iggy_binary_protocol::requests::segments::DeleteSegmentsRequest; use iggy_binary_protocol::requests::streams::{ - CreateStreamRequest, DeleteStreamRequest, PurgeStreamRequest, UpdateStreamRequest, + CreateStreamRequest, DeleteStreamRequest, GetStreamRequest, PurgeStreamRequest, + UpdateStreamRequest, }; use iggy_binary_protocol::requests::topics::{ CreateTopicRequest, DeleteTopicRequest, PurgeTopicRequest, UpdateTopicRequest, @@ -600,35 +601,27 @@ impl SimClient { self.build_request_with_namespace(Operation::SendMessages, &buf, group) } - /// Build a `POLL_MESSAGES` request for an individual consumer, reading - /// `count` messages from offset 0 of `group`'s partition. + /// Build a `NonReplicated` request: `code` in the header's `reserved` + /// prefix, `group` as the routing namespace, `body` already encoded. /// - /// A `NonReplicated` read: the command code sits in the header's - /// `reserved` prefix, and the request id ECHOES the current counter - /// without advancing it (matching the SDK). The server ignores the id for - /// ops its `ClientTable` never sees, so burning one would buy nothing. - /// Requires a bound session (polls are auth-gated). + /// The request id ECHOES the current counter WITHOUT advancing it (matching + /// the SDK, and unlike [`Self::header`]): the server ignores the id for ops + /// its `ClientTable` never sees, so burning one would buy nothing - and + /// burning one here would desync every replicated request that follows. /// /// # Panics - /// Panics if the session is unbound or the request buffer is invalid. + /// Panics if the request buffer is invalid. #[allow(clippy::cast_possible_truncation)] - pub fn poll_messages(&self, group: IggyNamespace, count: u32) -> Message { - let (stream_id, topic_id, partition_id) = namespace_ids(group); - let body = PollMessagesRequest { - consumer: WireConsumer::consumer(WireIdentifier::Numeric(self.client_id as u32)), - stream_id, - topic_id, - partition_id, - strategy: WirePollingStrategy::first(), - count, - auto_commit: false, - } - .to_bytes(); - + fn non_replicated_request( + &self, + code: u32, + group: u64, + body: &[u8], + ) -> Message { let header_size = std::mem::size_of::(); let total_size = header_size + body.len(); let mut reserved = [0u8; 52]; - reserved[..4].copy_from_slice(&POLL_MESSAGES_CODE.to_le_bytes()); + reserved[..4].copy_from_slice(&code.to_le_bytes()); let header = RoutedRequestHeader { command: iggy_binary_protocol::Command::Request, operation: Operation::NonReplicated, @@ -637,15 +630,56 @@ impl SimClient { session: self.session_id(), request: self.request_counter.get(), reserved, - group: group.inner(), + group, ..Default::default() }; let mut buffer = Vec::with_capacity(total_size); buffer.extend_from_slice(bytemuck::bytes_of(&header)); - buffer.extend_from_slice(&body); + buffer.extend_from_slice(body); Message::try_from(Owned::<4096>::copy_from_slice(&buffer)) - .expect("poll request must be valid") + .expect("non-replicated request must be valid") + } + + /// Build a `POLL_MESSAGES` request for an individual consumer, reading + /// `count` messages from offset 0 of `group`'s partition. + /// + /// Requires a bound session (polls are auth-gated). + /// + /// # Panics + /// Panics if the session is unbound or the request buffer is invalid. + #[allow(clippy::cast_possible_truncation)] + pub fn poll_messages(&self, group: IggyNamespace, count: u32) -> Message { + let (stream_id, topic_id, partition_id) = namespace_ids(group); + let body = PollMessagesRequest { + consumer: WireConsumer::consumer(WireIdentifier::Numeric(self.client_id as u32)), + stream_id, + topic_id, + partition_id, + strategy: WirePollingStrategy::first(), + count, + auto_commit: false, + } + .to_bytes(); + self.non_replicated_request(POLL_MESSAGES_CODE, group.inner(), &body) + } + + /// Build a `GET_STREAM` read for `name`. + /// + /// A metadata read, so it is answered from whichever replica's state + /// machine the request lands on rather than routed to the primary; the + /// group is the metadata sentinel. Requires a bound session, since the read + /// is auth-gated. + /// + /// # Panics + /// Panics if `name` is not a valid wire name or the request buffer is + /// invalid. + pub fn get_stream(&self, name: &str) -> Message { + let body = GetStreamRequest { + stream_id: WireIdentifier::named(name).expect("stream name must be valid"), + } + .to_bytes(); + self.non_replicated_request(GET_STREAM_CODE, METADATA_GROUP, &body) } /// Store offset with explicit `AckLevel`. `NoAck` takes the primary's diff --git a/core/simulator/src/lib.rs b/core/simulator/src/lib.rs index 3d2fd3feaa..767f72196c 100644 --- a/core/simulator/src/lib.rs +++ b/core/simulator/src/lib.rs @@ -395,6 +395,10 @@ impl Simulator { // reader-mode mirror from it and reads committed metadata through the // shared handle. Built in index order, so shard 0's bundle exists first. let mut metadata_bundle: Option = None; + // One applied-metadata frontier per REPLICA, shared by its shards, + // as the server bootstrap mints one per process. Volatile: a restart + // below builds a fresh cell, matching a rebooted node. + let metadata_applied_frontier = Arc::::default(); for shard_idx in 0..shards_per_replica { let inbox = inboxes[usize::from(shard_idx)] .take() @@ -429,6 +433,7 @@ impl Simulator { (shard_idx == 0).then(|| replica_data_dir.clone()).flatten(), // Fresh boot: `init_partition` seeds later, before any workload. &[], + Arc::clone(&metadata_applied_frontier), ); if shard_idx == 0 { metadata_bundle = Some( @@ -1077,6 +1082,7 @@ impl Simulator { /// # Panics /// If the replica is not crashed, or its shard count does not fit `u16`; mesh /// construction caps it. + #[allow(clippy::too_many_lines)] pub fn replica_restart(&mut self, replica_index: u8) { assert!( self.crashed.contains(&replica_index), @@ -1131,6 +1137,7 @@ impl Simulator { let mut stop_txs = Vec::with_capacity(usize::from(shards_per_replica)); let mut pump_tasks = Vec::with_capacity(usize::from(shards_per_replica)); let mut metadata_bundle: Option = None; + let metadata_applied_frontier = Arc::::default(); for shard_idx in 0..shards_per_replica { let inbox = inboxes[usize::from(shard_idx)] .take() @@ -1162,6 +1169,7 @@ impl Simulator { metadata_incarnation, (shard_idx == 0).then(|| replica_data_dir.clone()).flatten(), &seed_namespaces, + Arc::clone(&metadata_applied_frontier), ); if shard_idx == 0 { metadata_bundle = @@ -5097,3 +5105,452 @@ mod repair_frontier_tests { ); } } + +#[cfg(test)] +mod metadata_read_frontier_tests { + //! A client that committed a metadata write and then re-homed onto a + //! lagging backup must never be served the pre-write state. + //! + //! The window is not peer shards on one node: a committed reply is only + //! produced after `gated_apply` published, and every shard reads the same + //! left-right buffers. It is a node whose commit walk trails the epoch the + //! client already holds. Register forwarding is the supported way to get + //! there -- a backup verifies the credentials itself, forwards only the + //! consensus proposal, and binds the committed session while its own + //! `commit_journal` is still behind that op (see + //! `server::dispatch::session_ops`). + //! + //! Blocking replication INTO one backup while the quorum commits without + //! it produces the lag deterministically, and the login still completes + //! because `ForwardRegister` / `ForwardRegisterResult` are left flowing. + //! + //! One shard per replica for the end-to-end read, deliberately. The harness + //! homes each inbound client packet on a seeded-random shard while every + //! shard owns its own `SessionManager`, so on a multi-shard replica a bound + //! session cannot reliably receive its own follow-up requests -- and the lag + //! under test is the node's, not a shard's. + //! + //! The second test is the other half: peer shards are not the WINDOW, but + //! they are how a peer-homed read learns the node's position at all, and + //! sharing one frontier cell across a replica's shards is what makes that + //! work. It runs multi-shard and drives the cell directly, so it needs no + //! session and dodges the homing problem entirely. + + use super::*; + use crate::client::SimClient; + use iggy_binary_protocol::responses::streams::get_stream::GetStreamResponse; + use iggy_binary_protocol::{Command, RoutedRequestHeader, WireDecode}; + + /// Replica 0 leads the metadata plane at view 0, so this one is a backup + /// for the whole run and is the node the client re-homes onto. + const LAGGING: u8 = 1; + + /// The gate's own budget, in `sim.step()`s: one step advances the virtual + /// clock by one consensus tick, so the tick count of the budget IS the step + /// count a held read survives. + /// + /// The simulator's replicas carry the frontier's built-in default, since + /// they are built without a `[cluster]` config to size it from. + #[allow(clippy::cast_possible_truncation)] + const BUDGET_STEPS: u32 = (metadata::AppliedFrontier::DEFAULT_READ_BUDGET.as_millis() + / shard::CONSENSUS_TICK_INTERVAL.as_millis()) as u32; + + /// Steps the read is given while the backup is still cut off. A server that + /// answers a metadata read from an unconverged state answers within a + /// couple of these; the gate must hold the read past all of them. + /// + /// A fifth of the budget, so expiry cannot masquerade as a held read and + /// the convergence phase below still has most of the budget left. + const STALE_WINDOW_STEPS: u32 = BUDGET_STEPS / 5; + + /// Steps the convergence phase spends waiting for the held read. + /// + /// Deliberately PAST the budget rather than exactly up to it: repair that + /// lands one tick late would otherwise flip this test onto the expiry path + /// and fail on the status assertion, which reads as "the gate is broken" + /// when it means "convergence was slow". Overshooting instead lets the + /// expired read be reported as what it is. Expiry has its own case, which + /// never restores replication at all. + const CONVERGE_STEPS: u32 = BUDGET_STEPS - STALE_WINDOW_STEPS + BUDGET_STEPS / 2; + + /// The frames that would let the backup learn the committed writes. Journal + /// repair and `StartView` adoption are cut with the same knife as live + /// replication: any one of them left open closes the lag this exercises. + const REPLICATION_FRAMES: [Command; 5] = [ + Command::Prepare, + Command::Commit, + Command::RepairPrepare, + Command::RepairDone, + Command::StartView, + ]; + + /// Both cases below need the same shape: a stream created everywhere, then + /// deleted on a quorum that excludes `LAGGING` while the client re-homes + /// onto it, so the client holds a committed epoch above a delete that + /// backup has not applied. Returns the sim, the re-homed client, and the + /// op the delete committed at. + /// + /// Replication into `LAGGING` is left CUT: each case decides whether to + /// restore it. + fn backup_behind_a_deleted_stream( + seed: u64, + stream_name: &str, + client_id: u128, + ) -> (Simulator, SimClient, u64) { + server_common::MemoryPool::init_pool(&server_common::MemoryPoolSettings { + enabled: false, + size: iggy_common::IggyByteSize::from(0u64), + bucket_capacity: 1, + }); + + let replica_count: u8 = 3; + let network_opts = packet::PacketSimulatorOptions { + node_count: replica_count, + client_count: 1, + seed, + ..packet::PacketSimulatorOptions::default() + }; + let mut sim = Simulator::with_shards_shell( + usize::from(replica_count), + 1, + std::iter::once(client_id), + network_opts, + ); + + let client = SimClient::new(client_id); + sim.shell_login(&client); + + // The create lands on every replica: the backup has to HOLD the stream + // for the read to be able to serve a stale one. + let created = commit_write(&mut sim, client_id, 0, client.create_stream(stream_name)); + step_until_applied(&mut sim, LAGGING, created); + + // Cut replication into the backup, so the delete commits on the quorum + // formed by the primary and the remaining replica and never reaches it. + // Journal repair and `StartView` adoption go with it: any one of them + // left open closes the lag this exercises. + set_replication(&mut sim, LAGGING, false); + + let deleted = commit_write(&mut sim, client_id, 0, client.delete_stream(stream_name)); + assert!( + deleted > created, + "the delete must commit above the create, else the read cannot \ + distinguish the two states" + ); + + // Re-home the same client onto the backup. The login is forwarded, so + // the session it binds IS a committed op above the delete while the + // backup's own applied frontier is still below it. + sim.shell_login_via(&client, LAGGING); + assert!( + sim.network.delivered_any(Command::ForwardRegister), + "no ForwardRegister crossed the wire: the backup answered the login \ + itself, so the client never re-homed" + ); + let lagging_commit = metadata_commit(&sim, usize::from(LAGGING)); + assert!( + (created..deleted).contains(&lagging_commit), + "the backup applied up to op {lagging_commit}, outside the window \ + [{created}, {deleted}) these tests need: it must hold the create and \ + miss the delete" + ); + assert_eq!( + read_stream_name_on(&sim, LAGGING, stream_name), + Some(stream_name.to_string()), + "the backup no longer holds the deleted stream, so a read cannot \ + serve a stale one and the assertions prove nothing" + ); + (sim, client, deleted) + } + + /// A stream deleted before the client re-homed must not come back on the + /// backup that has not applied the delete yet. + #[test] + fn given_backup_behind_the_client_epoch_when_reading_a_deleted_stream_should_not_serve_it() { + let stream_name = "read-your-writes"; + let client_id: u128 = 1; + let (mut sim, client, deleted) = + backup_behind_a_deleted_stream(0x1A7E_0F31, stream_name, client_id); + + let read = client.get_stream(stream_name); + let request_id = read.header().request; + sim.submit_request(client_id, LAGGING, read.into_generic()); + + // Phase 1: still cut off. Any answer here is served from state that + // predates the delete the client already saw committed. + let mut early = None; + for _ in 0..STALE_WINDOW_STEPS { + if let Some(reply) = sim + .step() + .into_iter() + .find(|reply| reply.header().request == request_id) + { + early = Some(reply); + break; + } + } + if let Some(reply) = early { + panic!( + "the backup answered a metadata read while its applied frontier \ + ({}) was below the client's committed epoch ({deleted}): status={}, \ + stream={:?}", + metadata_commit(&sim, usize::from(LAGGING)), + reply.header().status, + read_stream_name(&reply), + ); + } + + // Phase 2: restore replication. The held read must answer from the + // converged state, which no longer holds the stream. + set_replication(&mut sim, LAGGING, true); + for step in 0..CONVERGE_STEPS { + if let Some(reply) = sim + .step() + .into_iter() + .find(|reply| reply.header().request == request_id) + { + assert_eq!( + reply.header().status, + 0, + "the read was refused rather than answered {step} steps into a \ + restored link: the gate expired at its {BUDGET_STEPS}-step budget, \ + of which phase 1 spent {STALE_WINDOW_STEPS}, so repair was slower \ + than the budget rather than the gate being wrong" + ); + assert_eq!( + read_stream_name(&reply), + None, + "the converged backup still serves the deleted stream" + ); + return; + } + } + panic!( + "no answer to the held metadata read within {CONVERGE_STEPS} steps of \ + restored replication; backup applied frontier {}, client epoch {deleted}", + metadata_commit(&sim, usize::from(LAGGING)), + ); + } + + /// The other half of the bound: a backup that NEVER catches up must refuse + /// the held read rather than serve the state the client already saw + /// replaced, and it must do so inside the budget rather than hanging. + /// + /// Same setup as the test above with replication left cut, so the only + /// possible outcomes are the refusal this asserts or a stale answer. + #[test] + fn given_a_backup_that_never_converges_when_reading_should_refuse_inside_the_budget() { + let stream_name = "read-your-writes-expiry"; + let client_id: u128 = 1; + let (mut sim, client, deleted) = + backup_behind_a_deleted_stream(0x1A7E_0F32, stream_name, client_id); + + let read = client.get_stream(stream_name); + let request_id = read.header().request; + sim.submit_request(client_id, LAGGING, read.into_generic()); + + // Overshoot the budget: the refusal must land inside it, and a read + // still unanswered after it is a hang, which the panic below names. + for _ in 0..(BUDGET_STEPS + BUDGET_STEPS / 2) { + if let Some(reply) = sim + .step() + .into_iter() + .find(|reply| reply.header().request == request_id) + { + assert_ne!( + reply.header().status, + 0, + "the cut-off backup answered a read below the client's committed \ + epoch ({deleted}) instead of refusing it: stream={:?}", + read_stream_name(&reply), + ); + assert_eq!(read_stream_name(&reply), None, "a refusal carries no body"); + return; + } + } + panic!( + "the held read neither answered nor expired within {} steps; backup applied \ + frontier {}, client epoch {deleted}", + BUDGET_STEPS + BUDGET_STEPS / 2, + metadata_commit(&sim, usize::from(LAGGING)), + ); + } + + /// Shards per replica for the sharing test below. Two is the whole + /// population that matters: shard 0 and one peer. + const SHARED_FRONTIER_SHARDS: u16 = 2; + + /// Advance applied to shard 0, chosen above whatever recovery seeded so a + /// peer reading the pre-advance value cannot pass by accident. + const SHARED_FRONTIER_ADVANCE: u64 = 7; + + /// A peer shard owns no metadata consensus, so `commit_min` -- the number + /// the pre-existing read barrier gates on -- does not exist there at all. + /// The applied frontier is the only thing its read gate can consult, and it + /// arrives by being ONE cell per process rather than one per shard. + /// + /// Nothing else in the system observes that. A private cell per shard still + /// compiles, still serves every read, and its only symptom is that each + /// metadata read homed on a peer shard parks for the whole deadline and + /// then fails retryable -- a latency cliff behind a warning, not an error. + /// So the invariant is asserted directly rather than through a request: a + /// peer must see an advance it did not make. + #[test] + fn given_a_peer_shard_when_shard_zero_advances_the_frontier_should_observe_it() { + server_common::MemoryPool::init_pool(&server_common::MemoryPoolSettings { + enabled: false, + size: iggy_common::IggyByteSize::from(0u64), + bucket_capacity: 1, + }); + + let replica_count: u8 = 3; + let network_opts = packet::PacketSimulatorOptions { + node_count: replica_count, + client_count: 1, + seed: 0x5EED_5A11, + ..packet::PacketSimulatorOptions::default() + }; + let sim = Simulator::with_shards( + usize::from(replica_count), + SHARED_FRONTIER_SHARDS, + std::iter::once(1u128), + network_opts, + ); + + let shards = &sim.replicas[0].shards; + assert_eq!( + shards.len(), + usize::from(SHARED_FRONTIER_SHARDS), + "the replica did not build the peer shard this test needs" + ); + for (shard_idx, shard) in shards.iter().enumerate().skip(1) { + assert!( + shard.plane.metadata().consensus.is_none(), + "shard {shard_idx} owns consensus, so it is not the peer whose \ + only source for the frontier is shard 0's cell" + ); + } + + let advanced = + shards[0].plane.metadata().applied_frontier().get() + SHARED_FRONTIER_ADVANCE; + shards[0] + .plane + .metadata() + .advance_applied_frontier(advanced); + + for (shard_idx, shard) in shards.iter().enumerate() { + assert_eq!( + shard.plane.metadata().applied_frontier().get(), + advanced, + "shard {shard_idx} did not observe shard 0's advance: the \ + applied-frontier cell is per shard, not per process, so every \ + metadata read homed here parks until its deadline" + ); + } + } + + /// Open or close every replication route from the primary into `replica`. + fn set_replication(sim: &mut Simulator, replica: u8, open: bool) { + for frame in REPLICATION_FRAMES { + let filter = sim + .network + .link_filter_mut(ProcessId::Replica(0), ProcessId::Replica(replica)); + if open { + filter.insert(frame); + } else { + filter.remove(frame); + } + } + } + + /// Step until `replica` has applied `op`, so the state the read sees is the + /// one this test set up rather than whatever the last reply happened to + /// leave behind. + fn step_until_applied(sim: &mut Simulator, replica: u8, op: u64) { + for _ in 0..SETUP_TOTAL_STEPS { + if metadata_commit(sim, usize::from(replica)) >= op { + return; + } + sim.step(); + } + panic!( + "replica {replica} never applied op {op} (stuck at {})", + metadata_commit(sim, usize::from(replica)), + ); + } + + /// Read `name` straight out of a replica's committed metadata, bypassing + /// the read path under test. + fn read_stream_name_on(sim: &Simulator, replica: u8, name: &str) -> Option { + sim.replicas[usize::from(replica)].shards[0] + .plane + .metadata() + .mux_stm + .streams() + .read(|inner| { + inner + .items + .iter() + .find(|(_, stream)| &*stream.name == name) + .map(|(_, stream)| stream.name.to_string()) + }) + } + + /// Submit one replicated metadata write to `target`, step until its reply + /// lands, and return the op it committed at. + fn commit_write( + sim: &mut Simulator, + client_id: u128, + target: u8, + request: Message, + ) -> u64 { + let request_id = request.header().request; + sim.submit_request(client_id, target, request.into_generic()); + for _ in 0..SETUP_TOTAL_STEPS { + if let Some(reply) = sim + .step() + .into_iter() + .find(|reply| reply.header().request == request_id) + { + assert_eq!( + reply.header().status, + 0, + "metadata write {request_id} was refused" + ); + assert!( + !setup_reply_is_transient(&reply), + "metadata write {request_id} was rejected in transit, so it never \ + committed and cannot anchor the read below" + ); + return reply.header().commit; + } + } + panic!("metadata write {request_id} never committed"); + } + + /// The stream name a `GetStream` reply carries, or `None` for the + /// empty-body not-found answer. + fn read_stream_name(reply: &Message) -> Option { + let body = reply + .as_slice() + .get(size_of::()..reply.header().size as usize) + .unwrap_or_default(); + if body.is_empty() { + return None; + } + let (response, _) = GetStreamResponse::decode(body) + .expect("a non-empty GetStream reply must decode as GetStreamResponse"); + Some(response.stream.name.as_str().to_string()) + } + + /// Committed metadata op on a replica's shard 0. + fn metadata_commit(sim: &Simulator, replica_idx: usize) -> u64 { + sim.replicas[replica_idx].shards[0] + .plane + .metadata() + .consensus + .as_ref() + .expect("shard 0 owns metadata consensus") + .commit_min() + } +} diff --git a/core/simulator/src/replica.rs b/core/simulator/src/replica.rs index 869dc61857..5132085fd9 100644 --- a/core/simulator/src/replica.rs +++ b/core/simulator/src/replica.rs @@ -30,7 +30,7 @@ use metadata::stm::mux::WithFactory; use metadata::stm::snapshot::RestoreSnapshot; use metadata::stm::stream::{Streams, StreamsInner}; use metadata::stm::user::{Users, UsersInner}; -use metadata::{IggyMetadata, apply_committed_prepare}; +use metadata::{AppliedFrontier, IggyMetadata, apply_committed_prepare}; use partitions::{IggyPartitions, PartitionPathLayout, PartitionsConfig}; use server::boot::wire_shell_handlers; use server::shell::{ShellHandlers, ShellShardHandle}; @@ -157,6 +157,7 @@ pub fn new_shard( incarnation: u128, data_dir: Option, seed_namespaces: &[(server_common::sharding::IggyNamespace, u32)], + applied_frontier: Arc, ) -> (Rc, Option) { // Metadata is single-writer, mirroring the server bootstrap. Shard 0 owns // the only writable STM; every peer shard rebuilds a reader-mode mirror from @@ -304,7 +305,8 @@ pub fn new_shard( superblock, mux, data_dir, - ); + ) + .with_applied_frontier(applied_frontier); // Both halves are load-bearing: the pairing keeps a later view-change superblock // write from regressing to `(0, 0)`, and the folded table is the floor the replayed @@ -368,6 +370,8 @@ pub fn new_shard( ); } } + // Same seed the server bootstrap runs after its own replay. + metadata.seed_applied_frontier_from_consensus(); // Mint the peers' read-side bundle AFTER reconstruction so it reflects the // recovered state. Shard 0 only; peers pass it back in as `reader_bundle`. let metadata_bundle = (shard_idx == 0).then(|| metadata.mux_stm.factory_bundle()); From 39c1bd0a50af7321ad1472df2c2d50ded2042329 Mon Sep 17 00:00:00 2001 From: Krishna Vishal Date: Fri, 4 Sep 2026 16:55:26 +0530 Subject: [PATCH 060/182] fix(partitions): reserve offsets before confirming them (#3975) --- core/configs/src/server_config/cluster.rs | 9 +- core/configs/src/server_config/defaults.rs | 3 + core/configs/src/server_config/displays.rs | 7 +- core/configs/src/server_config/partition.rs | 109 + core/consensus/src/impls.rs | 4 +- core/consensus/src/vsr_state.rs | 160 +- .../tests/cluster/crash_offset_reuse.rs | 446 ++- core/integration/tests/server/http_vsr.rs | 98 + core/journal/src/superblock.rs | 4 +- core/metadata/src/impls/recovery.rs | 1 + core/partitions/src/iggy_partition.rs | 2391 +++++++++++++++-- core/partitions/src/lib.rs | 15 + core/partitions/src/log.rs | 27 +- core/partitions/src/segment_anchor.rs | 329 +++ core/partitions/src/state_transfer.rs | 37 +- core/server/config.toml | 39 +- core/server/src/boot/recovery.rs | 131 +- core/server/src/partition_helpers.rs | 424 ++- core/server/src/segment_recovery.rs | 390 ++- core/server/src/server_error.rs | 15 + core/shard/src/lib.rs | 136 +- 21 files changed, 4320 insertions(+), 455 deletions(-) create mode 100644 core/partitions/src/segment_anchor.rs diff --git a/core/configs/src/server_config/cluster.rs b/core/configs/src/server_config/cluster.rs index 88d9b195fa..09b9be3d44 100644 --- a/core/configs/src/server_config/cluster.rs +++ b/core/configs/src/server_config/cluster.rs @@ -276,10 +276,11 @@ pub struct ClusterConfig { /// be > 0 and <= `MAX_REPAIR_CHUNK_MAX`. #[serde(default = "default_repair_chunk_max")] pub repair_chunk_max: usize, - /// How long the metadata superblock may stay unwritable before the replica - /// fail-stops. While wedged the replica is already fenced quorum-invisible - /// and peers elect around it; this converts the log-only limp into a - /// distinct exit status a supervisor can act on. Zero (and the `0` / + /// How long a superblock may stay unwritable before the replica fail-stops. + /// Applies per plane: the metadata superblock, and each PARTITION's own. While + /// wedged the group is already fenced quorum-invisible and peers elect around + /// it; this converts the log-only limp into a distinct exit status a + /// supervisor can act on. Zero (and the `0` / /// `disabled` / `unlimited` sentinels, which all parse to zero) disables /// the fail-stop; nonzero values below /// `MIN_SUPERBLOCK_WEDGED_FATAL_TIMEOUT` are rejected at boot. diff --git a/core/configs/src/server_config/defaults.rs b/core/configs/src/server_config/defaults.rs index 743f25d91a..303ccd0a36 100644 --- a/core/configs/src/server_config/defaults.rs +++ b/core/configs/src/server_config/defaults.rs @@ -41,6 +41,7 @@ use crate::common::server::{ ConsumerGroupConfig, DataMaintenanceConfig, HeartbeatConfig, PersonalAccessTokenConfig, TelemetryConfig, }; +use std::num::NonZeroU32; use std::sync::Arc; // Same embedded TOML the shared sections read; re-exported so sibling @@ -177,6 +178,8 @@ impl Default for PartitionConfig { PartitionConfig { prepare_queue_depth: partition.prepare_queue_depth as usize, dedup_clients_max: partition.dedup_clients_max as usize, + offset_reservation_lease: NonZeroU32::new(partition.offset_reservation_lease as u32) + .expect("the embedded config.toml carries a nonzero offset_reservation_lease"), evicted_ring_capacity: partition.evicted_ring_capacity as usize, evicted_ring_bytes_max: partition.evicted_ring_bytes_max.parse().unwrap(), transfer_served_cache_bytes_max: partition diff --git a/core/configs/src/server_config/displays.rs b/core/configs/src/server_config/displays.rs index aa4f79f980..5c46c7b82c 100644 --- a/core/configs/src/server_config/displays.rs +++ b/core/configs/src/server_config/displays.rs @@ -56,10 +56,11 @@ impl Display for PartitionConfig { fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { write!( f, - "{{ prepare_queue_depth: {}, evicted_ring_capacity: {}, \ - evicted_ring_bytes_max: {}, transfer_served_cache_bytes_max: {}, \ - transfer_artifact_bytes_max: {} }}", + "{{ prepare_queue_depth: {}, offset_reservation_lease: {}, \ + evicted_ring_capacity: {}, evicted_ring_bytes_max: {}, \ + transfer_served_cache_bytes_max: {}, transfer_artifact_bytes_max: {} }}", self.prepare_queue_depth, + self.offset_reservation_lease, self.evicted_ring_capacity, self.evicted_ring_bytes_max, self.transfer_served_cache_bytes_max, diff --git a/core/configs/src/server_config/partition.rs b/core/configs/src/server_config/partition.rs index 6d32c22320..86ee3a7c4c 100644 --- a/core/configs/src/server_config/partition.rs +++ b/core/configs/src/server_config/partition.rs @@ -44,6 +44,7 @@ use crate::common::validators::SEGMENT_MAX_SIZE_BYTES; use configs::ConfigEnv; use iggy_common::{IggyByteSize, Validatable}; use serde::{Deserialize, Serialize}; +use std::num::NonZeroU32; /// Mirrors `consensus::PIPELINE_PREPARE_QUEUE_MAX`. pub const DEFAULT_PARTITION_PREPARE_QUEUE_DEPTH: usize = 32; @@ -97,6 +98,22 @@ pub const DEFAULT_TRANSFER_SERVED_CACHE_BYTES_MAX: u64 = /// count. pub const MAX_TRANSFER_BYTES: u64 = 64 * 1024 * 1024 * 1024; +/// Upper bound on `offset_reservation_lease`: a typo guard, so a slipped digit +/// cannot reach the arithmetic in the append fence. +pub const MAX_OFFSET_RESERVATION_LEASE: u32 = 16 * 1024 * 1024; + +/// Mirrors `partitions::DEFAULT_OFFSET_RESERVATION_LEASE`; pinned against drift +/// by `default_offset_reservation_lease_matches_partitions_constant` in the +/// server crate, which can see both. +pub const DEFAULT_OFFSET_RESERVATION_LEASE: u32 = 64 * 1024; + +/// Serde fallback for a `[partition]` table that omits +/// `offset_reservation_lease`. +fn default_offset_reservation_lease() -> NonZeroU32 { + NonZeroU32::new(DEFAULT_OFFSET_RESERVATION_LEASE) + .expect("DEFAULT_OFFSET_RESERVATION_LEASE is a nonzero literal") +} + /// Mirrors `partitions::EVICTED_RING_CAPACITY`. pub const DEFAULT_EVICTED_RING_CAPACITY: usize = 4096; @@ -144,6 +161,27 @@ pub struct PartitionConfig { /// single partition actually sees, not the node's client total. pub dedup_clients_max: usize, + /// Offsets claimed in the superblock ahead of the mint counter before an + /// append, so a crash-restarted replica resumes above what it confirmed. + /// One superblock write per block: lowering it raises the fsync rate, + /// raising it wastes at most one block per crash. Must be <= + /// [`MAX_OFFSET_RESERVATION_LEASE`]. + /// + /// SINGLE-REPLICA groups only: a replicated group acks once a quorum has + /// journaled the batch, claims nothing, and ignores this. + /// + /// `NonZeroU32` rather than a `u32` with a floor check: a zero lease claims + /// nothing and would write the superblock before every append, and the type + /// is what stops the partition-side setter from having to silently coerce it + /// to one, a coercion that hid wiring errors the validator could not see. + /// + /// The serde fallback is for the providers that do NOT merge the embedded + /// defaults: the file provider does, so a partial `[partition]` table only + /// fails through direct deserialization or an alternate provider. + #[serde(default = "default_offset_reservation_lease")] + #[config_env(leaf)] + pub offset_reservation_lease: NonZeroU32, + /// Entries the evicted ring retains per multi-replica partition for /// journal repair after a peer rejoins. Larger widens the window a /// restarting peer can be served from the ring before falling back to @@ -206,6 +244,14 @@ impl Validatable for PartitionConfig { ); return Err(ConfigurationError::InvalidConfigurationValue); } + if self.offset_reservation_lease.get() > MAX_OFFSET_RESERVATION_LEASE { + eprintln!( + "{COMPONENT} partition.offset_reservation_lease ({}) exceeds the maximum \ + ({MAX_OFFSET_RESERVATION_LEASE})", + self.offset_reservation_lease + ); + return Err(ConfigurationError::InvalidConfigurationValue); + } if self.evicted_ring_capacity == 0 { eprintln!("{COMPONENT} partition.evicted_ring_capacity must be > 0"); return Err(ConfigurationError::InvalidConfigurationValue); @@ -332,6 +378,69 @@ mod tests { assert!(config.validate().is_ok()); } + fn lease(value: u32) -> NonZeroU32 { + NonZeroU32::new(value).expect("a nonzero test lease") + } + + /// A `[partition]` table as an alternate provider hands it over: every other + /// field present, the lease optional. + fn partial_table(lease: Option) -> String { + let entry = lease.map_or_else(String::new, |lease| { + format!(r#""offset_reservation_lease": {lease},"#) + }); + format!( + r#"{{"prepare_queue_depth": 32, {entry} + "dedup_clients_max": 4096, + "evicted_ring_capacity": 4096, + "evicted_ring_bytes_max": "16 MiB", + "transfer_served_cache_bytes_max": "64 MiB", + "transfer_artifact_bytes_max": "64 MiB"}}"# + ) + } + + /// The floor is the type's, so a zero cannot be constructed to validate -- + /// it is refused at deserialization instead. + #[test] + fn rejects_zero_offset_reservation_lease_at_deserialization() { + let error = serde_json::from_str::(&partial_table(Some(0))) + .expect_err("a zero lease reserves nothing and must not deserialize"); + assert!( + error.to_string().contains("nonzero"), + "the refusal must name the constraint, got {error}" + ); + } + + /// The providers that do not merge the embedded defaults hand over a partial + /// table, which must still deserialize. + #[test] + fn given_a_table_without_the_lease_when_deserialized_should_fall_back_to_the_default() { + let config = serde_json::from_str::(&partial_table(None)) + .expect("a table omitting the lease must deserialize"); + assert_eq!( + config.offset_reservation_lease.get(), + DEFAULT_OFFSET_RESERVATION_LEASE + ); + assert!(config.validate().is_ok()); + } + + #[test] + fn rejects_offset_reservation_lease_above_ceiling() { + let config = PartitionConfig { + offset_reservation_lease: lease(MAX_OFFSET_RESERVATION_LEASE + 1), + ..PartitionConfig::default() + }; + assert!(config.validate().is_err()); + } + + #[test] + fn accepts_offset_reservation_lease_at_ceiling() { + let config = PartitionConfig { + offset_reservation_lease: lease(MAX_OFFSET_RESERVATION_LEASE), + ..PartitionConfig::default() + }; + assert!(config.validate().is_ok()); + } + #[test] fn rejects_zero_evicted_ring_capacity() { let config = PartitionConfig { diff --git a/core/consensus/src/impls.rs b/core/consensus/src/impls.rs index 5ad273c35f..159e0ca5e6 100644 --- a/core/consensus/src/impls.rs +++ b/core/consensus/src/impls.rs @@ -2139,8 +2139,10 @@ impl> VsrConsensus { checkpoint_op, checkpoint_checksum, // Consensus mints no message offsets: the PARTITION plane stamps - // this in before it writes (`IggyPartition::write_superblock`). + // both of these in before it writes + // (`IggyPartition::write_superblock`). offset_frontier: 0, + offset_reserved: 0, } } diff --git a/core/consensus/src/vsr_state.rs b/core/consensus/src/vsr_state.rs index 9040906adc..3023d3df13 100644 --- a/core/consensus/src/vsr_state.rs +++ b/core/consensus/src/vsr_state.rs @@ -31,20 +31,38 @@ use std::fmt; /// Number of bytes [`VsrState::to_bytes`] produces: `cluster`(16) + /// `replica_id`(1) + `replica_count`(1) + `view`(4) + `log_view`(4) + /// `commit_max`(8) + `checkpoint_op`(8) + `checkpoint_checksum`(16) + -/// `offset_frontier`(8). -pub const ENCODED_LEN: usize = 66; +/// `offset_frontier`(8) + `offset_reserved`(8). +/// +/// Growing this is one-way: a record of this length is [`VsrStateError::WrongLength`] +/// to every build that predates the field, so a ROLLBACK needs the data directory +/// wiped even though the upgrade does not. Stated for operators beside +/// `partition.offset_reservation_lease` in `config.toml`. +pub const ENCODED_LEN: usize = 74; -/// The layout before `offset_frontier` was appended. +/// The layout before `offset_reserved` was appended. +/// +/// [`VsrState::try_from`] accepts records of this length. Without it every +/// superblock already on disk decodes as [`VsrStateError::WrongLength`], which +/// the metadata plane treats as a durability violation and refuses the whole +/// node's boot on. +/// +/// "Already on disk" is not a clustering concern. `PingPongSuperblock::open` +/// runs unconditionally in metadata recovery, so EVERY server writes one of +/// these into `metadata/superblock.a`, clustered or not, on every view change +/// and checkpoint. Every release from `server-0.9.0-edge.2` (the first that +/// carries this module at all) through `edge.6` wrote exactly 66 bytes, +/// single-node deployments that never enabled clustering included. +/// +/// The 58-byte layout that preceded it is deliberately NOT accepted: it left +/// trunk before any release carried it -- `server-0.8.2-edge.1` has no +/// `vsr_state.rs`, and `edge.2` already wrote 66 -- so a record that short is +/// corruption, not history. /// -/// [`VsrState::try_from`] still accepts records of this length and zero-fills -/// the new field. Without it every superblock already on disk -- the metadata -/// plane writes one on every view change and checkpoint, single-node included -- -/// would decode as [`VsrStateError::WrongLength`] and refuse boot as a -/// durability violation. A version bump instead of this would not help on its -/// own: `classify` compares the version for exact equality, so a v2 build turns -/// every v1 record into `Unreadable`, which is the same refusal wearing a -/// different name. -pub const ENCODED_LEN_WITHOUT_FRONTIER: usize = 58; +/// A version bump instead of this would not help on its own: `classify` compares +/// the version for exact equality, so a v2 build turns every v1 record into +/// `Unreadable`, which is the same refusal wearing a different name -- and it +/// would make this tolerance unreachable, refusing even records that decode. +pub const ENCODED_LEN_WITHOUT_RESERVATION: usize = 66; /// The durable consensus state of one replica for one consensus group. /// @@ -97,6 +115,20 @@ pub struct VsrState { /// /// Always `0` on the metadata plane, which mints no message offsets. pub offset_frontier: u64, + /// PARTITION plane: a monotone CEILING on the offsets this replica may + /// already have minted, claimed ahead of the counter in blocks so an append + /// pays one superblock write per block instead of one per batch. + /// + /// Never folded into [`Self::offset_frontier`]. The frontier is a claim + /// about DATA, which state transfer's rewind guard refuses to destroy; a + /// reservation names no bytes, only "an offset up to here may have reached + /// a client". Comparing an offer against it refuses every legitimate offer + /// below the lease headroom, and the replica cycles transfer -> refusal -> + /// backoff forever. Boot seeds the mint counter from both; the rewind guard + /// reads the frontier alone. + /// + /// Always `0` on the metadata plane, which mints no message offsets. + pub offset_reserved: u64, } impl VsrState { @@ -113,6 +145,7 @@ impl VsrState { out[34..42].copy_from_slice(&self.checkpoint_op.to_le_bytes()); out[42..58].copy_from_slice(&self.checkpoint_checksum.to_le_bytes()); out[58..66].copy_from_slice(&self.offset_frontier.to_le_bytes()); + out[66..74].copy_from_slice(&self.offset_reserved.to_le_bytes()); out } } @@ -121,16 +154,27 @@ impl TryFrom<&[u8]> for VsrState { type Error = VsrStateError; fn try_from(bytes: &[u8]) -> Result { - // Length-tolerant: a pre-`offset_frontier` record is padded out and the - // new field reads as 0, which is exactly "no recorded frontier" (the - // read sites filter it). One length check up front then puts every - // field slice below in bounds by construction, so the `try_into`s - // cannot fail. + // Length-tolerant for ONE legacy layout, the pre-`offset_reserved` record + // every tagged release wrote. The one length check then puts every field + // slice below in bounds by construction, so the `try_into`s cannot fail. + // + // `offset_reserved` is filled from `offset_frontier`, not zeroed. The two + // agree on a record written before the reservation existed: the frontier + // is what that build's data proved, and the write side clamps the + // reservation up to it anyway, so this is the same value a first write + // under this build would record. A 0 would instead claim "nothing + // reserved" for offsets the frontier says exist. + // + // It cannot recover what the old build never wrote down -- offsets acked + // out of RAM above the frontier are gone with the process either way -- + // but refusing the record recovers nothing and costs the node its boot. let mut padded = [0u8; ENCODED_LEN]; match bytes.len() { ENCODED_LEN => padded.copy_from_slice(bytes), - ENCODED_LEN_WITHOUT_FRONTIER => { - padded[..ENCODED_LEN_WITHOUT_FRONTIER].copy_from_slice(bytes); + ENCODED_LEN_WITHOUT_RESERVATION => { + padded[..ENCODED_LEN_WITHOUT_RESERVATION].copy_from_slice(bytes); + padded[ENCODED_LEN_WITHOUT_RESERVATION..ENCODED_LEN] + .copy_from_slice(&bytes[58..66]); } actual => { return Err(VsrStateError::WrongLength { @@ -150,6 +194,7 @@ impl TryFrom<&[u8]> for VsrState { checkpoint_op: u64::from_le_bytes(field(bytes, 34)), checkpoint_checksum: u128::from_le_bytes(field(bytes, 42)), offset_frontier: u64::from_le_bytes(field(bytes, 58)), + offset_reserved: u64::from_le_bytes(field(bytes, 66)), }; // A record violating `log_view <= view` decodes into a replica that looks // healthy locally while `DoViewChangeHeader::validate` makes every peer drop @@ -190,8 +235,8 @@ impl fmt::Display for VsrStateError { Self::WrongLength { expected, actual } => { write!( f, - "VsrState needs {expected} bytes (or {ENCODED_LEN_WITHOUT_FRONTIER}, \ - the layout before the offset frontier), got {actual}" + "VsrState needs {expected} bytes (or {ENCODED_LEN_WITHOUT_RESERVATION}, \ + the layout before the offset reservation), got {actual}" ) } Self::LogViewAheadOfView { view, log_view } => write!( @@ -224,7 +269,8 @@ mod tests { commit_max: 6, checkpoint_op: 7, checkpoint_checksum: 8, - offset_frontier: 0, + offset_frontier: 9, + offset_reserved: 10, }; let bytes = state.to_bytes(); assert_eq!(bytes.len(), ENCODED_LEN); @@ -236,17 +282,20 @@ mod tests { assert_eq!(bytes[26], 6, "commit_max low byte"); assert_eq!(bytes[34], 7, "checkpoint_op low byte"); assert_eq!(bytes[42], 8, "checkpoint_checksum low byte"); + assert_eq!(bytes[58], 9, "offset_frontier low byte"); + assert_eq!(bytes[66], 10, "offset_reserved low byte"); assert_eq!(VsrState::try_from(&bytes[..]).unwrap(), state); assert!(VsrState::try_from(&bytes[..ENCODED_LEN - 1]).is_err()); } - /// A superblock written before `offset_frontier` existed must still decode: - /// the metadata plane writes one on every view change, so an exact-length - /// decode turns an in-place upgrade into a boot refusal on every deployment - /// that ever ran. + /// A superblock written before `offset_reserved` existed must still decode: + /// every release from `server-0.9.0-edge.2` to `edge.6` wrote that layout on + /// both planes -- the metadata one unconditionally, so single-node + /// deployments that never enabled clustering have one -- and the metadata + /// plane refuses the whole node's boot on a record it cannot decode. #[test] - fn given_pre_frontier_record_when_decoded_should_accept_and_zero_fill() { + fn given_pre_reservation_record_when_decoded_should_fill_from_the_frontier() { let full = VsrState { cluster: 3, replica_id: 1, @@ -257,23 +306,62 @@ mod tests { checkpoint_op: 7, checkpoint_checksum: 5, offset_frontier: 77, + offset_reserved: 88, } .to_bytes(); - let legacy = &full[..ENCODED_LEN_WITHOUT_FRONTIER]; - let decoded = VsrState::try_from(legacy).expect("a pre-frontier record must decode"); - assert_eq!(decoded.offset_frontier, 0, "the new field zero-fills"); + let legacy = &full[..ENCODED_LEN_WITHOUT_RESERVATION]; + let decoded = VsrState::try_from(legacy).expect("a pre-reservation record must decode"); + assert_eq!(decoded.offset_frontier, 77); + assert_eq!( + decoded.offset_reserved, 77, + "the reservation fills from the frontier, not from zero: a 0 would claim \ + nothing was reserved for offsets the frontier says exist" + ); assert_eq!(decoded.view, 9); assert_eq!(decoded.log_view, 8); assert_eq!(decoded.commit_max, 41); assert_eq!(decoded.checkpoint_op, 7); assert_eq!(decoded.checkpoint_checksum, 5); - // Anything that is neither layout is still refused. - assert!(matches!( - VsrState::try_from(&full[..40]), - Err(VsrStateError::WrongLength { .. }) - )); + // Anything that is neither layout is still refused, the 58-byte + // pre-frontier layout included: `server-0.8.2-edge.1` has no + // `vsr_state.rs` and `edge.2` already wrote 66, so no release ever put a + // 58-byte record on a disk and one that short is corruption, not + // history. + for short in [40, 58] { + assert!( + matches!( + VsrState::try_from(&full[..short]), + Err(VsrStateError::WrongLength { .. }) + ), + "a {short}-byte record must not decode" + ); + } + } + + /// A zero frontier is the shape a record written before either offset field + /// existed decodes into, and the fill must not turn that into a claim. + #[test] + fn given_pre_reservation_record_with_no_frontier_when_decoded_should_reserve_nothing() { + let full = VsrState { + cluster: 3, + replica_id: 1, + replica_count: 3, + view: 2, + log_view: 2, + commit_max: 0, + checkpoint_op: 0, + checkpoint_checksum: 0, + offset_frontier: 0, + offset_reserved: 0, + } + .to_bytes(); + + let decoded = VsrState::try_from(&full[..ENCODED_LEN_WITHOUT_RESERVATION]) + .expect("a pre-reservation record must decode"); + assert_eq!(decoded.offset_frontier, 0); + assert_eq!(decoded.offset_reserved, 0); } #[test] @@ -294,9 +382,11 @@ mod tests { // Distinct and nonzero: with 0 here a transposed write over the // trailing field would still satisfy every assertion below. offset_frontier: 9, + offset_reserved: 11, } .to_bytes(); assert_eq!(bytes[58], 9, "offset_frontier must occupy bytes 58..66"); + assert_eq!(bytes[66], 11, "offset_reserved must occupy bytes 66..74"); bytes[22] = 5; // log_view = 5, view stays 4 assert_eq!( diff --git a/core/integration/tests/cluster/crash_offset_reuse.rs b/core/integration/tests/cluster/crash_offset_reuse.rs index 447a6cc892..92e4ae89f9 100644 --- a/core/integration/tests/cluster/crash_offset_reuse.rs +++ b/core/integration/tests/cluster/crash_offset_reuse.rs @@ -15,17 +15,17 @@ // specific language governing permissions and limitations // under the License. -//! RED SPEC, expected to FAIL: offset identity across a crash. +//! Offset identity across a crash. //! //! `SendMessagesResponse::confirmations` hands clients concrete base offsets, -//! which makes offset reuse client-visible: a client that recorded offset N -//! for its message must never see the server confirm a DIFFERENT message at -//! N later. A solo node acks below the flush thresholds from RAM only and -//! persists no offset watermark, so after a SIGKILL it restarts the partition -//! log at the last flushed position and re-mints offsets it already -//! confirmed. Passes only once a durable watermark (or durable journal tail) -//! keeps post-restart offsets above everything ever acked. +//! which makes offset reuse client-visible: a client that recorded offset N for +//! its message must never see the server confirm a DIFFERENT message at N +//! later. A solo node acks below the flush thresholds from RAM only, so nothing +//! in the segments says those offsets were ever handed out. What keeps them +//! from being re-minted is the offset RESERVATION in the partition superblock, +//! claimed by the append fence before any of them exist and read back by boot. +use std::path::Path; use std::time::Duration; use iggy::prelude::*; @@ -97,13 +97,8 @@ async fn wait_until_serving(harness: &TestHarness, budget: Duration) -> IggyClie } } -// TODO(hubcio): fix this test -#[ignore = "confirmed offsets re-minted after a crash; no durable offset watermark"] -#[iggy_harness(cluster_nodes = 1)] -async fn given_confirmed_sends_below_flush_threshold_when_a_solo_node_is_killed_should_not_remint_offsets( - harness: &mut TestHarness, -) { - let client = harness.tcp_root_client().await.unwrap(); +/// Create the stream and its single-partition topic. +async fn create_topic(client: &IggyClient, messages_required_to_save: Option) { client .create_stream(STREAM_NAME) .await @@ -115,20 +110,83 @@ async fn given_confirmed_sends_below_flush_threshold_when_a_solo_node_is_killed_ &TopicCreateOptions { partitions_count: Some(1), message_expiry: Some(IggyExpiry::NeverExpire), + messages_required_to_save, ..TopicCreateOptions::default() }, ) .await .expect("create topic"); +} - let acked = produce_acked(&client, "pre-crash", PRE_CRASH_SENDS).await; - let highest_confirmed = *acked.last().expect("confirmed sends"); - drop(client); +/// Base offsets of every segment file under `root`, from the file names, which +/// are the on-disk claim about where each range begins. +/// +/// Reading them is the only way to assert the shape the re-anchor produces. A +/// black-box offset assertion passes either way on the boot that WRITES the +/// wrong shape; the cost only lands on the boot that reads it back, where the +/// walk refuses and the solo arm tombstones the partition. +fn segment_base_offsets(root: &Path) -> Vec { + let mut offsets = Vec::new(); + let mut stack = vec![root.to_path_buf()]; + while let Some(dir) = stack.pop() { + let Ok(entries) = std::fs::read_dir(&dir) else { + continue; + }; + for entry in entries.flatten() { + let path = entry.path(); + if path.is_dir() { + stack.push(path); + } else if path.extension().is_some_and(|extension| extension == "log") + && let Some(stem) = path.file_stem().and_then(|stem| stem.to_str()) + && let Ok(offset) = stem.parse::() + { + offsets.push(offset); + } + } + } + offsets.sort_unstable(); + offsets +} + +/// Poll until the only node has the replicated consumer offset on disk. Panics +/// at the deadline. +async fn wait_for_stored_offset_on_disk(harness: &TestHarness, expected: u64, budget: Duration) { + let data_path = harness.node(0).data_path(); + let deadline = tokio::time::Instant::now() + budget; + loop { + if integration::harness::disk::read_replicated_consumer_offset(&data_path) == Some(expected) + { + return; + } + assert!( + tokio::time::Instant::now() < deadline, + "the node did not persist consumer offset {expected} within {budget:?} \ + (found {:?})", + integration::harness::disk::read_replicated_consumer_offset(&data_path), + ); + sleep(POLL_INTERVAL).await; + } +} +/// Kill the node, bring it back, and return a client onto the restarted one. +async fn crash_and_recover(harness: &mut TestHarness) -> IggyClient { harness.kill_node(0).expect("SIGKILL the only node"); harness.restart_node(0).expect("restart it"); + wait_until_serving(harness, SERVE_TIMEOUT).await +} - let client = wait_until_serving(harness, SERVE_TIMEOUT).await; +#[iggy_harness(cluster_nodes = 1)] +async fn given_confirmed_sends_below_flush_threshold_when_a_solo_node_is_killed_should_not_remint_offsets( + harness: &mut TestHarness, +) { + let client = harness.tcp_root_client().await.unwrap(); + create_topic(&client, None).await; + + let acked = produce_acked(&client, "pre-crash", PRE_CRASH_SENDS).await; + let highest_confirmed = *acked.last().expect("confirmed sends"); + drop(client); + + let client = crash_and_recover(harness).await; let post_crash_offset = produce_acked(&client, "post-crash", 1).await[0]; assert!( @@ -140,4 +198,354 @@ async fn given_confirmed_sends_below_flush_threshold_when_a_solo_node_is_killed_ so two different messages now share an offset and consumers reading by offset get \ silently different data" ); + + // The shape this boot WROTE is only paid for by the boot that reads it back: + // nothing reached a segment before the crash, so the re-anchor emptied the + // chain and had to plant at the append point rather than leave the segment + // named 0 with the first mint a lease block inside it. + let bases = segment_base_offsets(harness.test_dir()); + assert!( + bases.iter().any(|&base| base > highest_confirmed), + "the re-anchor left no segment above the pre-crash offsets, so the post-crash \ + mint landed inside a segment named below it: segment bases {bases:?}, last \ + offset confirmed before the crash {highest_confirmed}" + ); + drop(client); + + // And the boot that reads it: a hole inside a segment refuses the walk and + // tombstones the partition, which shows up here as a node that never serves + // the stream again. + let client = crash_and_recover(harness).await; + let third_life = produce_acked(&client, "third-life", 1).await[0]; + assert!( + third_life > post_crash_offset, + "the second restart re-minted: {post_crash_offset} was confirmed after the \ + first crash, yet the node came back and handed out {third_life}" + ); +} + +/// The graceful stop is the runbook answer to an incident, so it must not be the +/// action that undoes the fix. Between the two boots here the node takes no +/// traffic at all: nothing reaches a segment, the committed frontier stays at +/// what the crash left, and the reservation is the only record that offsets were +/// handed out. +/// +/// End to end, not a probe of one mechanism: three separate things carry the +/// append point across this sequence (the boot seed, the segment the re-anchor +/// plants, and the collapse writing the append point rather than the committed +/// frontier), so any one of them alone keeps this green. The collapse is pinned +/// on its own by `given_an_unspent_reservation_when_collapsing_should_leave_it_standing` +/// in `core/partitions`. +#[iggy_harness(cluster_nodes = 1)] +async fn given_a_crash_restarted_node_when_stopped_cleanly_should_still_not_remint_offsets( + harness: &mut TestHarness, +) { + let client = harness.tcp_root_client().await.unwrap(); + create_topic(&client, None).await; + + let acked = produce_acked(&client, "pre-crash", PRE_CRASH_SENDS).await; + let highest_confirmed = *acked.last().expect("confirmed sends"); + drop(client); + + // Boot one: reads the reservation back and never spends it. + let client = crash_and_recover(harness).await; + drop(client); + + // The clean stop, then boot two. `restart_node` stops the running node with + // SIGTERM and waits for it, so the shutdown flush and its collapse both run. + harness + .restart_node(0) + .expect("cleanly restart the only node"); + let client = wait_until_serving(harness, SERVE_TIMEOUT).await; + let post_stop_offset = produce_acked(&client, "post-clean-stop", 1).await[0]; + + assert!( + post_stop_offset > highest_confirmed, + "a clean stop between the crash and the next send re-minted: offset \ + {highest_confirmed} was handed to a client before the SIGKILL, and after a \ + graceful restart the node confirmed {post_stop_offset}. The shutdown collapse \ + dropped a reservation the boot had not spent yet" + ); +} + +/// The combination neither clean-stop nor flushed case covers on its own: +/// `given_a_crash_restarted_node_when_stopped_cleanly_should_still_not_remint_offsets` +/// stops cleanly but takes no traffic between boots, so its collapse only ever +/// sees an UNSPENT reservation, while every flushed case re-enters through +/// SIGKILL and never runs the collapse at all. +/// +/// Here the second life spends the reservation, appends past the flush threshold, +/// and then stops gracefully, so the collapse writes an append point over a chain +/// the re-anchor already planted a gap into. The boot that follows has to accept +/// that chain and resume above it; a refusal tombstones the partition and shows +/// up as a node that never serves the stream again. +#[iggy_harness(cluster_nodes = 1)] +async fn given_a_flushed_crash_restarted_node_when_stopped_cleanly_should_still_not_remint_offsets( + harness: &mut TestHarness, +) { + const FLUSH_THRESHOLD: u32 = 4; + /// Past the threshold, so the life leaves a chain and a flushed tail. + const SENDS_PER_LIFE: u32 = 6; + + let client = harness.tcp_root_client().await.unwrap(); + create_topic(&client, Some(FLUSH_THRESHOLD)).await; + + let acked = produce_acked(&client, "first-life", SENDS_PER_LIFE).await; + let first_life_max = *acked.last().expect("confirmed sends"); + drop(client); + + let client = crash_and_recover(harness).await; + let second_life = produce_acked(&client, "second-life", SENDS_PER_LIFE).await; + let second_life_max = *second_life.last().expect("confirmed sends"); + assert!( + second_life[0] > first_life_max, + "the crash restart re-minted: confirmed {first_life_max} before the SIGKILL, \ + then {} after it", + second_life[0] + ); + drop(client); + + // SIGTERM and wait, so the shutdown flush and its collapse both run over the + // re-anchored chain. + harness + .restart_node(0) + .expect("cleanly restart the only node"); + let client = wait_until_serving(harness, SERVE_TIMEOUT).await; + let post_stop = produce_acked(&client, "post-clean-stop", 1).await[0]; + + assert!( + post_stop > second_life_max, + "a clean stop after a spent reservation re-minted: {second_life_max} was \ + confirmed to a client, and the graceful restart handed out {post_stop}. The \ + collapse recorded the committed frontier rather than the append point, so the \ + next boot resumed inside offsets the previous life had already confirmed" + ); + + let bases = segment_base_offsets(harness.test_dir()); + assert!( + bases.iter().any(|&base| base > first_life_max), + "the chain lost the re-anchor's plant across the clean stop: bases {bases:?}, \ + last offset confirmed before the crash {first_life_max}" + ); +} + +/// The behaviour the reservation is SOLD on, which every other case here leaves +/// implicit: a consumer positioned inside the pre-crash range must keep reading +/// forward across the hole the reservation creates, and must land on the real +/// post-crash message rather than on a phantom offset inside the gap. +/// +/// The other cases assert only that newly confirmed offsets rise. That is the +/// producer's half. A consumer that stored a position, survived the crash and +/// then polled `Next` is what actually walks the boundary the re-anchor planted: +/// `disk_poll_start` has to carry the walk on into the segment above the gap, +/// and the stored offset has to still mean the same place. +#[iggy_harness(cluster_nodes = 1)] +async fn given_a_stored_consumer_position_when_polling_across_a_reservation_hole_should_not_miss( + harness: &mut TestHarness, +) { + const CONSUMER_ID: u32 = 7; + const POST_CRASH_PAYLOAD: &str = "across-the-hole-000"; + + let stream = Identifier::named(STREAM_NAME).unwrap(); + let topic = Identifier::named(TOPIC_NAME).unwrap(); + let client = harness.tcp_root_client().await.unwrap(); + create_topic(&client, None).await; + + let acked = produce_acked(&client, "pre-crash", PRE_CRASH_SENDS).await; + let highest_confirmed = *acked.last().expect("confirmed sends"); + // Positioned one BELOW the last confirmed offset, so the pre-crash tail is + // still ahead of the consumer when the node dies. A position at the tail + // would make the first post-crash poll indistinguishable from a fresh read. + let stored = highest_confirmed - 1; + let consumer = Consumer::new(Identifier::numeric(CONSUMER_ID).unwrap()); + client + .store_consumer_offset(&consumer, &stream, &topic, Some(PARTITION_ID), stored) + .await + .expect("store the pre-crash consumer position"); + // Gated on the position reaching DISK before the SIGKILL. The ack is granted + // at commit and the consumer-offset write is threshold-gated like any other, + // so without this the crash can legitimately take the position with it and + // the assertion below races. + wait_for_stored_offset_on_disk(harness, stored, SERVE_TIMEOUT).await; + drop(client); + + let client = crash_and_recover(harness).await; + assert_eq!( + client + .get_consumer_offset(&consumer, &stream, &topic, Some(PARTITION_ID)) + .await + .expect("read the consumer offset back") + .expect("the stored position survived the crash") + .stored_offset, + stored, + "the position a consumer committed before the crash must mean the same \ + offset after it, or every consumer silently re-reads or skips" + ); + + // One message above the hole, with a payload no earlier send used, so the + // poll cannot pass by matching something the pre-crash range already held. + let post_crash = produce_acked(&client, "across-the-hole", 1).await[0]; + assert!( + post_crash > highest_confirmed, + "the premise: the restart must mint above the confirmed range" + ); + + let polled = client + .poll_messages( + &stream, + &topic, + Some(PARTITION_ID), + &consumer, + &PollingStrategy::next(), + 1, + false, + ) + .await + .expect("poll forward from the stored position"); + + let message = polled + .messages + .first() + .unwrap_or_else(|| panic!("`Next` from offset {stored} served nothing at all")); + assert_eq!( + String::from_utf8_lossy(&message.payload), + POST_CRASH_PAYLOAD, + "the consumer must land on the post-crash message, not on stale bytes or a \ + phantom offset inside the reservation's hole" + ); + assert_eq!( + message.header.offset, post_crash, + "and it must be served at the offset the producer was confirmed at: the hole \ + between {highest_confirmed} and {post_crash} is unwritten offset space, not \ + messages a consumer may be handed" + ); +} + +/// The replicated fence path, which no other case here reaches. A three-node +/// group acks a send once a quorum has journaled it, so it claims no reservation +/// and re-anchors nothing -- and one node crashing must still not disturb the +/// offsets the group hands out. +#[iggy_harness(cluster_nodes = 3)] +async fn given_a_replicated_group_when_a_node_is_killed_should_not_remint_offsets( + harness: &mut TestHarness, +) { + let client = harness.tcp_root_client().await.unwrap(); + create_topic(&client, None).await; + + let acked = produce_acked(&client, "pre-crash", PRE_CRASH_SENDS).await; + let highest_confirmed = *acked.last().expect("confirmed sends"); + assert_eq!( + acked, + (0..u64::from(PRE_CRASH_SENDS)).collect::>(), + "the pre-crash run mints a contiguous range from zero" + ); + drop(client); + + let client = crash_and_recover(harness).await; + let post_crash_offset = produce_acked(&client, "post-crash", 1).await[0]; + + assert!( + post_crash_offset > highest_confirmed, + "a replicated group re-minted after one node restarted: {highest_confirmed} was \ + confirmed before the SIGKILL, then {post_crash_offset} after it" + ); + + // No reservation was claimed, so no gap was planted: a replicated group's + // segment boundaries have to stay a function of its batches alone, or the + // reconciler's offset-keyed segment GC never converges. + let bases = segment_base_offsets(harness.test_dir()); + assert!( + bases.iter().all(|&base| base <= post_crash_offset), + "a replicated group planted a segment above every offset it minted, so it \ + re-anchored around a reservation it should never have claimed: bases {bases:?}" + ); +} + +/// The fix has to survive its own side effect: the hole the reservation leaves +/// between the recovered segments and the new append point makes the next boot +/// REFUSE the chain if it lands INSIDE a segment, which tombstones the partition +/// on a solo node. +/// +/// So the SECOND crash is the one that matters, and only if the run between the +/// two reaches disk, which is what the flush threshold is for. +#[iggy_harness(cluster_nodes = 1)] +async fn given_a_crash_restarted_node_when_it_flushes_and_crashes_again_should_still_not_remint_offsets( + harness: &mut TestHarness, +) { + const FLUSH_THRESHOLD: u32 = 4; + /// Past the threshold, so every life leaves a chain for the next boot. + const SENDS_PER_LIFE: u32 = 6; + + let client = harness.tcp_root_client().await.unwrap(); + create_topic(&client, Some(FLUSH_THRESHOLD)).await; + + let acked = produce_acked(&client, "first-life", SENDS_PER_LIFE).await; + let first_life_max = *acked.last().expect("confirmed sends"); + drop(client); + + // The boot that consumes a reservation and re-anchors, then flushes the + // hole's far side to disk. + let client = crash_and_recover(harness).await; + let second_life = produce_acked(&client, "second-life", SENDS_PER_LIFE).await; + let second_life_min = second_life[0]; + let second_life_max = *second_life.last().expect("confirmed sends"); + assert!( + second_life_min > first_life_max, + "the first restart re-minted: confirmed {first_life_max} before the crash, \ + then {second_life_min} after it" + ); + drop(client); + + // ON a segment boundary, not inside one: a segment still named 0 while + // holding the second life's offsets claims a range it does not have. + let bases = segment_base_offsets(harness.test_dir()); + assert!( + bases.iter().any(|&base| base > first_life_max), + "no segment is anchored above the pre-crash offsets, so the second life \ + appended into a segment named for the first: segment bases {bases:?}, last \ + offset confirmed before the crash {first_life_max}" + ); + + // The first boot that has to read a chain the re-anchor wrote. + let client = crash_and_recover(harness).await; + let third_life = produce_acked(&client, "third-life", 1).await[0]; + assert!( + third_life > second_life_max, + "the SECOND restart re-minted: {second_life_max} was confirmed between the \ + two crashes, yet the node came back and handed out {third_life}. A hole \ + left INSIDE a segment costs the tail that proved the frontier" + ); +} + +/// The partially-flushed shape a real workload crashes in: the run below the +/// threshold is confirmed out of the journal while everything before it is on +/// disk. The recovered chain then ends BELOW the reservation with bytes in it, +/// so the re-anchor has to seal it rather than append into the gap. +#[iggy_harness(cluster_nodes = 1)] +async fn given_sends_straddling_the_flush_threshold_when_the_node_is_killed_should_not_remint_offsets( + harness: &mut TestHarness, +) { + const FLUSH_THRESHOLD: u32 = 4; + const STRADDLING_SENDS: u32 = 6; + + let client = harness.tcp_root_client().await.unwrap(); + create_topic(&client, Some(FLUSH_THRESHOLD)).await; + + let acked = produce_acked(&client, "straddle", STRADDLING_SENDS).await; + let highest_confirmed = *acked.last().expect("confirmed sends"); + assert_eq!( + acked, + (0..u64::from(STRADDLING_SENDS)).collect::>(), + "the pre-crash run mints a contiguous range from zero" + ); + drop(client); + + let client = crash_and_recover(harness).await; + let post_crash = produce_acked(&client, "post-straddle", 1).await[0]; + assert!( + post_crash > highest_confirmed, + "offsets confirmed out of the journal above the last flushed one were \ + re-minted: {highest_confirmed} went to a client before the SIGKILL, and the \ + first send after it was confirmed at {post_crash}" + ); } diff --git a/core/integration/tests/server/http_vsr.rs b/core/integration/tests/server/http_vsr.rs index 2ae84bac2f..0438937154 100644 --- a/core/integration/tests/server/http_vsr.rs +++ b/core/integration/tests/server/http_vsr.rs @@ -735,6 +735,104 @@ async fn given_ack_none_when_producing_should_return_202_and_commit(harness: &Te } } +/// The FIRST send to a partition that has never minted an offset, on both +/// produce routes, against the only cluster shape where the offset reservation +/// runs at all. +/// +/// `cluster_nodes = 1` is load-bearing, not a speed-up: `request_mint_ceiling` +/// returns `None` above one replica, so this suite's three-node default leaves +/// the whole reservation path as dead code and proves nothing here. +/// +/// What it guards is the BOUNCE REMOVAL. A first send used to be answered +/// `TransientNotAccepted` so the shard tick would claim the block off the +/// request pump, and neither HTTP route can carry a retryable refusal back to +/// the caller: the acked route has no transient replay loop, and `?ack=none` +/// never reads a reply at all, so that bounce answered 202 and dropped the +/// message. +/// +/// It is NOT sensitive to where the claim is taken. With the bounce gone the +/// inline fence at the mint writes the same block on the first send, so both +/// routes still commit with the create-time claim deleted; +/// `given_a_fresh_solo_partition_when_building_should_record_its_first_claim` +/// in `core/server/src/partition_helpers.rs` reads the durable record and is +/// the test that fails without it. +/// +/// Both partitions are produced to exactly once, so a per-partition regression +/// cannot hide behind a second send. +#[iggy_harness(cluster_nodes = 1)] +async fn given_a_solo_topic_when_producing_its_first_http_messages_should_commit_them( + harness: &TestHarness, +) { + const ACKED_PARTITION: u32 = 0; + const UNACKED_PARTITION: u32 = 1; + + let http = HttpClient::login_root(harness).await; + http.create_stream_and_topic("http-first-send", "first", 2) + .await; + + let response = http + .produce( + "http-first-send", + "first", + ACKED_PARTITION, + vec![text_message(1, "first-acked".to_string())], + ) + .await; + assert_eq!( + response.status(), + StatusCode::CREATED, + "the first acked send to a never-minted partition must commit, not be refused" + ); + let polled = http + .poll("http-first-send", "first", ACKED_PARTITION, 0, 10) + .await; + assert_eq!(polled.messages.len(), 1, "the first acked send is durable"); + assert_eq!( + polled.messages[0].payload, + bytes::Bytes::from("first-acked") + ); + + let response = http + .produce_with_query( + "http-first-send", + "first", + UNACKED_PARTITION, + vec![text_message(2, "first-unacked".to_string())], + "?ack=none", + ) + .await; + assert_eq!( + response.status(), + StatusCode::ACCEPTED, + "ack=none must answer before the commit" + ); + + // 202 says nothing about the commit, which is the whole hazard: a refusal + // on this route is answered the same way and leaves no trace. Only the poll + // proves the message survived. + let deadline = Instant::now() + ASYNC_COMMIT_TIMEOUT; + loop { + if let Some(polled) = http + .try_poll("http-first-send", "first", UNACKED_PARTITION, 0, 10) + .await + && !polled.messages.is_empty() + { + assert_eq!(polled.messages.len(), 1, "exactly one message was produced"); + assert_eq!( + polled.messages[0].payload, + bytes::Bytes::from("first-unacked"), + "the first ack=none send to a never-minted partition must not be dropped" + ); + break; + } + assert!( + Instant::now() < deadline, + "the first ack=none send never became pollable within {ASYNC_COMMIT_TIMEOUT:?}" + ); + sleep(ASYNC_COMMIT_RETRY_INTERVAL).await; + } +} + /// End-to-end RBAC proof: an ungranted user is 403 on a metadata read and on a /// data-plane produce, root stays 200/201, and the auth-only cluster-metadata /// route is never gated. Exercises the HTTP per-op gates (read + partition diff --git a/core/journal/src/superblock.rs b/core/journal/src/superblock.rs index fd8b761244..77f49ffe09 100644 --- a/core/journal/src/superblock.rs +++ b/core/journal/src/superblock.rs @@ -70,8 +70,8 @@ const MIN_RECORD_LEN: usize = HEADER_LEN + CHECKSUM_LEN; /// Ceiling on a record's payload, bounding every allocation this module makes from /// a length it read off disk (`PrepareJournal::MAX_ENTRY_SIZE` bounds the WAL for the -/// same reason). The only payload today is a [`consensus::VsrState`], 66 bytes now -/// that it carries the offset frontier (58 before it, a length its decode still +/// same reason). The only payload today is a [`consensus::VsrState`], 74 bytes now +/// that it carries the offset reservation (66 before it, a length its decode still /// accepts); the headroom is for a payload that grows fields, not for bulk data. /// `read_slot` treats a longer file as corrupt WITHOUT reading it, and /// `build_record` refuses to write one, so a length this store could have diff --git a/core/metadata/src/impls/recovery.rs b/core/metadata/src/impls/recovery.rs index 39d5b9847d..60ffcf58f1 100644 --- a/core/metadata/src/impls/recovery.rs +++ b/core/metadata/src/impls/recovery.rs @@ -843,6 +843,7 @@ mod tests { checkpoint_op, checkpoint_checksum, offset_frontier: 0, + offset_reserved: 0, } } diff --git a/core/partitions/src/iggy_partition.rs b/core/partitions/src/iggy_partition.rs index 3cbdf3e177..19db69e4c0 100644 --- a/core/partitions/src/iggy_partition.rs +++ b/core/partitions/src/iggy_partition.rs @@ -79,12 +79,26 @@ use std::cell::{Cell, RefCell}; use std::collections::{HashMap, HashSet}; use std::fmt; use std::hash::Hash; +use std::num::NonZeroU32; use std::rc::Rc; use std::sync::Arc; use std::sync::atomic::{AtomicU64, Ordering}; use tokio::sync::Mutex as TokioMutex; use tracing::{debug, error, warn}; +/// Which of a partition's offset counters are live. +/// +/// Two bits, not one: a reservation-seeded boot makes the append counter live +/// with nothing committed behind it, and folding them together reports +/// `offset_frontier() == 1` for a partition holding nothing. +#[derive(Debug, Default, Clone, Copy)] +pub struct OffsetSpace { + /// The APPEND counter is live, so the next mint continues it. + pub append_live: bool, + /// The COMMITTED counter names data. + pub committed_seeded: bool, +} + // This struct aliases in terms of the code contained the `LocalPartition from `core/server/src/streaming/partitions/local_partition.rs`. pub struct IggyPartition where @@ -120,7 +134,7 @@ where pub stats: Arc, pub created_at: IggyTimestamp, pub revision_id: u64, - pub should_increment_offset: bool, + pub(crate) offset_space: OffsetSpace, pub write_lock: Arc>, pub(crate) consumer_offsets_path: Option, pub(crate) consumer_group_offsets_path: Option, @@ -239,6 +253,17 @@ where /// segments that were the only other witness, after which the rebuild /// re-mints offsets the group already handed out. durable_offset_frontier: Cell, + /// The `offset_reserved` ceiling the last successful superblock write + /// recorded, seeded at boot from the record that write left behind. + /// + /// A `Cell` rather than a re-read of the record because every append reads + /// it, and the steady state ("the block still covers this batch") has to + /// cost nothing. Kept apart from [`Self::durable_offset_frontier`]: see + /// `consensus::VsrState::offset_reserved`. + durable_offset_reserved: Cell, + /// Offsets the append fence claims per superblock write; installed by boot + /// from `PartitionsConfig`. + offset_reservation_lease: u64, /// In-flight state transfer for this group (rejoin whose repair floor was /// refused); tail repair takes over at install. See /// [`PartitionTransferSession`]. @@ -300,7 +325,7 @@ where .field("namespace", &self.consensus.group()) .field("offset", &self.offset) .field("dirty_offset", &self.dirty_offset) - .field("should_increment_offset", &self.should_increment_offset) + .field("offset_space", &self.offset_space) .field("partition_dir", &self.partition_dir) .field("repair", &self.repair) .field("recovered_durable_offset", &self.recovered_durable_offset) @@ -481,7 +506,7 @@ where stats, created_at: IggyTimestamp::now(), revision_id: 0, - should_increment_offset: false, + offset_space: OffsetSpace::default(), write_lock: Arc::new(TokioMutex::new(())), consumer_offsets_path: None, consumer_group_offsets_path: None, @@ -504,6 +529,8 @@ where superblock_retry_after_micros: Cell::new(0), purge_deferred: false, durable_offset_frontier: Cell::new(0), + durable_offset_reserved: Cell::new(0), + offset_reservation_lease: u64::from(crate::DEFAULT_OFFSET_RESERVATION_LEASE), transfer: None, transfer_attempts: 0, transfer_failures: 0, @@ -606,7 +633,7 @@ where partition .dirty_offset .store(start_offset, Ordering::Relaxed); - partition.should_increment_offset = false; + partition.set_offset_space_used(false); partition.stats.increment_segments_count(1); partition } @@ -631,6 +658,8 @@ where self.superblock = Some(superblock); self.durable_offset_frontier .set(recovered.map_or(0, |state| state.offset_frontier)); + self.durable_offset_reserved + .set(recovered.map_or(0, |state| state.offset_reserved)); } /// Persist this group's VSR state to its superblock when the view changed @@ -707,24 +736,87 @@ where /// view it is in must not act in it, so it withholds every view-scoped /// send for this group, goes quiet, and its peers elect around it. Only /// THIS partition's group is fenced; the rest of the node keeps serving. + /// A RUN of failures is terminal for the process, not the group: the shard + /// tick fail-stops on `superblock_wedged`. #[allow(clippy::future_not_send)] async fn write_superblock(&self, superblock: &SB, offset_frontier: u64) -> bool { - // ADVANCE direction: never below what this replica has already minted, - // and never below what the record ALREADY holds. Both bounds are - // needed and neither implies the other -- a failed install leaves the - // counter behind the record it wrote before the swap, so maxing against - // the counter alone lets the fence that follows lower the durable - // frontier. The reset direction goes through `write_superblock_inner`. - let advanced = offset_frontier - .max(self.offset_frontier()) - .max(self.durable_offset_frontier.get()); - self.write_superblock_inner(superblock, advanced).await + self.write_superblock_advancing(superblock, offset_frontier, 0) + .await + } + + /// [`Self::write_superblock`] for a caller that also has a reservation to + /// claim. Both fields advance, neither can regress. + #[allow(clippy::future_not_send)] + async fn write_superblock_advancing( + &self, + superblock: &SB, + offset_frontier: u64, + offset_reserved: u64, + ) -> bool { + // ADVANCE direction; the reset direction goes through + // `write_superblock_inner`. Both bounds inside `advanced_frontier` are + // needed: a failed install leaves the chain behind the record it wrote + // before the swap, so maxing against the data alone would let the fence + // that follows lower the durable frontier. + let advanced = self.advanced_frontier(offset_frontier); + // Nothing but the record witnesses a reservation, so a caller with no + // claim of its own (every view-change write) passes 0 and carries the + // recorded one forward; dropping it would let the next boot seed the + // counter below what an earlier append already fenced. + let reserved = offset_reserved.max(self.durable_offset_reserved.get()); + self.write_superblock_inner(superblock, advanced, reserved) + .await + } + + /// The advance rule for the frontier, shared by every writer that claims + /// one: never below what this replica holds, never below what the record + /// already says. Held messages, NOT [`Self::mint_frontier`], which stands a + /// lease block above them after a reservation-seeded boot and names none of + /// them. + fn advanced_frontier(&self, claim: u64) -> u64 { + claim + .max(self.held_offset_frontier()) + .max(self.durable_offset_frontier.get()) + } + + /// Record an incoming state-transfer frontier, advancing the frontier and + /// SETTING the reservation to the frontier this write records. + /// + /// Not to the offer: `write_superblock_inner` clamps the reservation up to + /// the frontier it writes, which is `advanced_frontier(frontier)` and sits + /// above the offer whenever an earlier over-claiming write left the durable + /// frontier higher. + /// + /// The one place the otherwise-monotone reservation may come down, and the + /// one place it must: the offer describes the group's committed log, so a + /// local reservation above it covers offsets this replica never confirmed. + /// Carried forward, it re-seeds the counter a lease block above the group + /// after the next restart, where every replicated prepare fails the + /// `base_offset == dirty_offset + 1` check. + #[allow(clippy::future_not_send)] + #[must_use = "the bool is the durability verdict; dropping it silently ignores a failed write"] + pub async fn install_offset_frontier_at(&self, frontier: u64) -> bool { + let Some(superblock) = self.superblock.as_ref().map(Rc::clone) else { + return true; + }; + if self.superblock_write_is_backed_off() { + return false; + } + let _superblock_guard = self.superblock_lock.acquire().await; + let advanced = self.advanced_frontier(frontier); + self.write_superblock_inner(superblock.as_ref(), advanced, frontier) + .await } /// The write itself; the advance and reset directions differ only in the - /// frontier they hand in. + /// values they hand in. #[allow(clippy::future_not_send)] - async fn write_superblock_inner(&self, superblock: &SB, offset_frontier: u64) -> bool { + async fn write_superblock_inner( + &self, + superblock: &SB, + offset_frontier: u64, + offset_reserved: u64, + ) -> bool { // The pairing fields stay `(0, 0)` and `commit_max` is a dead write // on this plane: nothing reads either back (`restore_partition_view` // restores view/log_view only), because recovery re-derives the @@ -737,18 +829,36 @@ where // write stamps the current counter, so whichever write lands last (a // view change, or the explicit persist an install issues) leaves a // lower bound boot can re-seed from. + // Sampled BEFORE the write: the writers that bypass the gate (the append + // fence, the quarantine record, the shutdown collapse) attempt inside an + // open window, and counting those makes the wedge threshold a function of + // producer retry rate rather than of elapsed time. + let inside_backoff_window = self.superblock_write_is_backed_off(); let mut state = self.consensus.vsr_state(0, 0); state.offset_frontier = offset_frontier; + // A frontier of N says offsets below N exist, so a reservation under it + // is not a reservation. Clamped here rather than per caller so the + // reset direction gets it too, where a stale higher reservation would + // seed the next boot into the offset space the reset just erased. + state.offset_reserved = offset_reserved.max(offset_frontier); match superblock.write(&state.to_bytes()).await { Ok(()) => { self.consensus .mark_superblock_durable(state.view, state.log_view); self.durable_offset_frontier.set(state.offset_frontier); + self.durable_offset_reserved.set(state.offset_reserved); self.superblock_write_failures.set(0); self.superblock_retry_after_micros.set(0); true } Err(error) => { + // One failure per backoff step. `superblock_wedged` reads the + // count as elapsed time (the window is capped at 1 s, the + // default threshold is 2 m), which a per-attempt count would + // turn into seconds under a handful of retrying producers. + if inside_backoff_window { + return false; + } let failures = self.superblock_write_failures.get() + 1; self.superblock_write_failures.set(failures); let backoff = SUPERBLOCK_RETRY_BACKOFF_BASE_MICROS @@ -784,37 +894,131 @@ where /// proved. /// /// The record is a lower bound, never a completeness claim: it exists - /// because three paths leave a replica whose counter would otherwise - /// restart at 0 while the group is at N (a transfer install of an all-GC'd - /// origin, a crash inside the install's swap window, and the - /// fence-and-rebuild path, which needs no crash at all). Restarting the + /// because four paths leave a replica whose counter would otherwise + /// restart below where the group already is (a transfer install of an + /// all-GC'd origin, a crash inside the install's swap window, the + /// fence-and-rebuild path, which needs no crash at all, and a crash while + /// acked messages were still resident in the journal). Restarting the /// counter is not a lag -- replicas re-stamp `base_offset` from it and /// recompute `batch_checksum` over the result, so the next replicated /// prepare would persist different bytes here than on every peer, silently. /// + /// The two counters are seeded separately, because the record carries two + /// bounds. `offset_frontier` is what messages reached, so it seeds the + /// COMMITTED counter; `offset_reserved` is what may have been handed to a + /// client, so it seeds the APPEND counter and nothing else. Folding the + /// reservation into the committed one would publish a `current_offset` over + /// a lease-block hole, and `store_consumer_offset` would then admit offsets + /// inside it. + /// + /// The reservation is SOLO ONLY. A backup mints nothing: it re-stamps what + /// the primary sends and rejects anything that does not continue its own + /// counter, so an append point a lease block above its group would have every + /// peer refuse the batch. A replicated group is also less exposed, since an + /// ack there means a quorum journaled the batch and the hole needs a + /// FULL-cluster crash. + /// /// Lives HERE rather than in the server crate so the boot paths and the /// simulator share one implementation. A copy in the harness was a copy of /// the max rule that had lost the max, in the one place built to catch /// violations of it. pub fn restore_offset_frontier(&mut self, recovered: Option<&consensus::VsrState>) { - let Some(frontier) = recovered - .map(|state| state.offset_frontier) - .filter(|&f| f > 0) - else { + let Some(state) = recovered else { return; }; - let recovered_end = frontier - 1; - if self.should_increment_offset && self.offset.load(Ordering::Acquire) >= recovered_end { + let frontier = state.offset_frontier; + let reserved = if self.consensus.replica_count() == 1 { + state.offset_reserved + } else { + 0 + }; + // NOT `frontier > 0`: the shape a crash before the first flush leaves is + // a zero frontier and a nonzero reservation, because the append fence + // runs before the journal append, so the first record a partition ever + // writes names no data at all. + let append_point = frontier.max(reserved); + if append_point == 0 { + return; + } + let seeded = self.offset_space.append_live; + // Each counter takes its own max, since the record's two bounds move + // independently: a graceful stop collapses the reservation onto the + // append point while the frontier stays where the data ended. + let committed_restored = frontier.checked_sub(1).is_some_and(|committed_end| { + let raise = !seeded || self.offset.load(Ordering::Acquire) < committed_end; + if raise { + self.offset.store(committed_end, Ordering::Release); + } + raise + }); + let append_end = append_point - 1; + let append_restored = !seeded || self.dirty_offset.load(Ordering::Relaxed) < append_end; + if append_restored { + self.dirty_offset.store(append_end, Ordering::Relaxed); + } + if !committed_restored && !append_restored { return; } tracing::info!( namespace_raw = self.consensus().group(), offset_frontier = frontier, - "restored partition offset frontier from its superblock" + offset_reserved = reserved, + append_point, + "restored partition offset counters from its superblock" ); - self.offset.store(recovered_end, Ordering::Release); - self.dirty_offset.store(recovered_end, Ordering::Relaxed); - self.should_increment_offset = true; + self.offset_space.append_live = true; + // Only the frontier names data. A record carrying a reservation alone is + // the pre-first-flush shape, where the committed counter still seeds + // nothing. + self.offset_space.committed_seeded |= frontier > 0; + } + + /// Path of the anchor whose lifecycle is this segment's. + /// + /// Anchors are unlinked with the segment they sit beside. Left behind, a + /// purge resets the offset space to 0 and the stale record still `covers` + /// the bounds a later plant reuses. + pub(crate) fn anchor_cleanup_path(&self, start_offset: u64) -> Option { + self.partition_dir + .as_deref() + .map(|dir| crate::segment_anchor::anchor_path(dir, start_offset)) + } + + /// Seed or clear BOTH offset-space bits. + /// + /// For the callers that genuinely move both: a fresh or purged partition + /// (neither counter names anything) and a boot off segments (both do, + /// because a segment on disk holds committed messages only). Everything on + /// the live path moves ONE bit -- see [`Self::note_append_live`] and + /// [`Self::note_committed_seeded`]. + pub const fn set_offset_space_used(&mut self, used: bool) { + self.offset_space = OffsetSpace { + append_live: used, + committed_seeded: used, + }; + } + + /// The append counter is live: an offset has been journaled, so the next + /// mint continues from it. + /// + /// APPEND only. A journaled offset is not a committed one: `build_poll_plan` + /// gates on `OffsetSpace::committed_seeded`, and seeding that here would + /// serve a resident offset 0 to a consumer before the first commit and let + /// the frontier persist name data no quorum agreed on. A view change may + /// still truncate this offset away. + pub const fn note_append_live(&mut self) { + self.offset_space.append_live = true; + } + + /// The committed counter names data: an offset has passed commit, so it is + /// pollable and the frontier may record it. + /// + /// Implies the append counter is live too -- nothing commits that was not + /// journaled first -- but the reverse does not hold, which is the whole + /// reason the two bits are separate. + pub const fn note_committed_seeded(&mut self) { + self.offset_space.append_live = true; + self.offset_space.committed_seeded = true; } /// Whether this partition ever stamped an offset, i.e. whether its offset @@ -823,7 +1027,7 @@ where /// that never took a write: both report `(0, 0)`. #[cfg(any(test, feature = "simulator"))] pub const fn offset_space_used(&self) -> bool { - self.should_increment_offset + self.offset_space.append_live } /// Adopt a log carried over from a previous incarnation of this partition, @@ -854,7 +1058,7 @@ where // prepare mint from a base no peer agrees on. // // Keyed on the RETIRED incarnation's flag, never on `(0, 0)` or on this - // partition's own `should_increment_offset`. One message at offset 0 reports + // partition's own `append_live`. One message at offset 0 reports // the same two zeroes as an empty partition, and this instance is freshly // built so its own flag is always false. The arithmetic test would therefore // adopt the log, skip the counters, and let the next write stamp @@ -869,13 +1073,74 @@ where .max(self.dirty_offset.load(Ordering::Relaxed)); self.offset.store(durable, Ordering::Release); self.dirty_offset.store(dirty, Ordering::Relaxed); - self.should_increment_offset = true; + self.set_offset_space_used(true); // Everything carried over is already persisted as far as this replica is // concerned, so the flush and commit paths must not re-persist or re-count // it, the same contract boot gives a partition recovered from segments. self.recovered_durable_offset = Some(durable); } + /// The in-memory half of [`Self::reanchor_to_offset_frontier`], for a + /// simulator partition rebuilt over a restored offset counter. + /// + /// The production re-anchor cannot be reused here: it creates and unlinks + /// real segment files, and an in-memory partition carries no + /// `partition_dir`, so it would write a chain to whatever path the config + /// resolves. What it does to the chain's SHAPE is the part the simulator + /// needs, and this makes exactly the same two decisions on the same two + /// conditions. + /// + /// Without it a restored replica keeps a single segment named at 0 while the + /// counter resumes a lease block above it, and the next mint lands INSIDE + /// that segment. Production can never reach that shape -- boot plants at the + /// append point -- so the harness both diverges from what it is modelling and + /// cannot expose the chain refusal a real node would hit on the boot after. + #[cfg(any(test, feature = "simulator"))] + pub fn reanchor_in_memory_to_mint_frontier(&mut self, segment_size: IggyByteSize) { + let frontier = self.mint_frontier(); + if frontier == 0 { + return; + } + // Empty tails named below the append point claim a range they do not + // hold. No files to unlink, so the retire is the whole job. + while let Some(segment) = self.log.segments().last() { + if segment.size.as_bytes_u64() > 0 || segment.start_offset >= frontier { + break; + } + if self.log.retire_back().is_none() { + break; + } + self.stats.decrement_segments_count(1); + } + let tail = self + .log + .segments() + .last() + .map(|segment| (segment.end_offset, segment.size.as_bytes_u64())); + let plant = match tail { + // An emptied chain: plant at the append point, the same as boot's + // `None` arm. Nothing precedes it, so there is no gap to record. + None => true, + // A SIZED tail below the append point gets sealed and planted past. + // An empty one either just went or is already named at the frontier + // and can take the appends as it stands. + Some((sealed_end, size)) if size > 0 && sealed_end.saturating_add(1) < frontier => { + self.log.active_segment_mut().sealed = true; + true + } + Some(_) => false, + }; + if plant { + self.log.add_persisted_segment( + crate::Segment::new(frontier, segment_size), + server_common::SegmentStorage::default(), + None, + None, + ); + self.stats.increment_segments_count(1); + } + } + /// Copy this incarnation's offset counter into the shared /// [`PartitionStats`], making it the value readers (offset validation, /// `get_topic`, `get_stats`) see. @@ -894,28 +1159,83 @@ where .set_current_offset(self.offset.load(Ordering::Acquire)); } - /// The next message offset this replica will mint, `0` while the offset - /// space is still empty. The value stamped into the durable record. + /// One past the highest COMMITTED offset, `0` while the offset space is + /// still empty. The value stamped into the durable record, and what a + /// transfer offer advertises. + /// + /// Not [`Self::mint_frontier`]: after a reservation-seeded boot the append + /// point stands a lease block above this, and neither the record nor an + /// offer may claim offsets no message reached. #[must_use] pub fn offset_frontier(&self) -> u64 { - if self.should_increment_offset { + if self.offset_space.committed_seeded { self.offset.load(Ordering::Acquire).saturating_add(1) } else { 0 } } + /// The offset the next mint will take, which is where the segment chain has + /// to be anchored for the appends that follow to land contiguously. + /// + /// Above [`Self::offset_frontier`] by exactly the offsets this replica has + /// journaled but not committed, plus -- on the first boot after a crash that + /// took acked-but-unflushed messages with it -- the lease block the durable + /// reservation claimed. That gap is the whole point: the reservation is the + /// only surviving witness that those offsets were handed to a client, so the + /// counter resumes above them instead of re-minting them. + #[must_use] + pub fn mint_frontier(&self) -> u64 { + if self.offset_space.append_live { + self.dirty_offset.load(Ordering::Relaxed).saturating_add(1) + } else { + 0 + } + } + + /// One past the highest offset this replica holds and may not lose: named by + /// a sized segment, or committed and still resident in the journal. `0` when + /// it holds nothing. + /// + /// The committed arm is not redundant with the disk arm, since the + /// threshold-gated flush routinely leaves committed messages unnamed by any + /// segment. It reads the COMMITTED counter and not `journal.info`, whose + /// `current_offset` is the DIRTY tail: a view change truncates that tail + /// (`truncate_uncommitted_from`) while the durable frontier only advances, so + /// recording it would leave every later boot seeding the counter above the + /// group, where each replicated prepare fails the + /// `base_offset == dirty_offset + 1` check until a state transfer -- again on + /// the boot after that one. + #[must_use] + pub fn held_offset_frontier(&self) -> u64 { + // Reverse search, not a scan-and-max: the chain is ordered and the + // contiguity guard keeps end offsets ascending, so the LAST sized segment + // is the highest one. Only the trailing empties are walked. + let on_disk = self + .log + .segments() + .iter() + .rev() + .find(|segment| segment.size.as_bytes_u64() > 0) + .map_or(0, |segment| segment.end_offset.saturating_add(1)); + on_disk.max(self.offset_frontier()) + } + /// Force the durable record to catch up with the current offset frontier, /// outside the view-change gate. /// /// [`Self::persist_superblock_if_needed`] fires on `(view, log_view)` /// changes only, which is the right trigger for the split-brain fence and /// the wrong one for the frontier: an install can move the counter by - /// millions without touching the view. Called where the frontier changes - /// with nothing else durable naming it -- after a state-transfer install - /// and after the convergence that follows a failed one. Returns whether the - /// record now holds it; a failure is logged by the writer and left to the - /// ordinary retry, since the install itself already succeeded. + /// millions without touching the view. + /// + /// One production caller, at the END of a state-transfer install, which is + /// also the converge path that follows a failed one. It pairs with the + /// [`Self::persist_offset_frontier_at`] the install writes BEFORE its + /// destructive swap: that one is a lower bound across the swap window, this + /// one records what the installed chain actually holds. Returns whether the + /// record now holds it; a failure is logged there and left to the ordinary + /// retry, since the install itself already succeeded. #[allow(clippy::future_not_send)] #[must_use = "the bool is the durability verdict; dropping it silently ignores a failed write"] pub async fn persist_offset_frontier(&self) -> bool { @@ -959,7 +1279,9 @@ where return false; } let _superblock_guard = self.superblock_lock.acquire().await; - self.write_superblock_inner(superblock.as_ref(), frontier) + // Reservation reset with it: left above, it would seed the next boot + // back into the offset space this reset just left behind. + self.write_superblock_inner(superblock.as_ref(), frontier, frontier) .await } @@ -977,6 +1299,11 @@ where /// `intended` is the frontier the caller knows the group is at, written /// verbatim; `None` means the live counter is authoritative and the /// advancing form applies. + /// + /// The reservation keeps its max either way: `vsr_state` calls it a monotone + /// ceiling, and only a purge or an install may bring it down. Inert while + /// the one `Some` caller is the replicated `ConvergeFailed` arm, where the + /// fence never ran and it already equals the frontier. #[allow(clippy::future_not_send)] #[must_use = "the bool is the durability verdict; dropping it silently ignores a failed write"] pub async fn record_frontier_before_quarantine(&self, intended: Option) -> bool { @@ -986,7 +1313,8 @@ where let _superblock_guard = self.superblock_lock.acquire().await; match intended { Some(frontier) => { - self.write_superblock_inner(superblock.as_ref(), frontier) + let reserved = frontier.max(self.durable_offset_reserved.get()); + self.write_superblock_inner(superblock.as_ref(), frontier, reserved) .await } None => { @@ -1007,6 +1335,48 @@ where self.consensus.clock_realtime_micros() < self.superblock_retry_after_micros.get() } + /// Drop the reservation back onto the frontier, once a graceful flush has + /// made the segments account for every offset this replica confirmed. + /// + /// The reservation is there for the crash case, where they do not. Left + /// standing it would make every ordinary restart resume a lease block + /// higher and hole the offset space for nothing. + /// + /// Collapses onto [`Self::mint_frontier`], not [`Self::offset_frontier`]: the + /// append point is what the next boot has to resume at, and on a boot that + /// consumed a reservation without appending it is the reservation itself, so + /// reading the committed frontier here would write a record BELOW what an + /// earlier life already confirmed to a client. A clean stop is the runbook + /// answer to an incident, which would make it the one action that undoes the + /// protection. + /// + /// The frontier field still records only what is held: a graceful stop + /// flushes the committed prefix, but the journal can hold an uncommitted tail + /// that the next view legitimately truncates. + /// + /// Callers must have flushed FIRST, and must not call this when the flush + /// failed: the claim it makes is precisely that the flush succeeded. + /// + /// BYPASSES the retry backoff, like `record_frontier_before_quarantine`: the + /// stop is the last chance, not deferred work. Skipped, the reservation + /// stands and the next boot seeds the append counter a lease block above the + /// data, holing the offset space for nothing. + #[allow(clippy::future_not_send)] + #[must_use = "the bool is the durability verdict; a failed collapse leaves a gap"] + pub async fn collapse_offset_reservation(&self) -> bool { + let append_point = self.mint_frontier(); + if self.durable_offset_reserved.get() <= append_point { + return true; + } + let Some(superblock) = self.superblock.as_ref().map(Rc::clone) else { + return true; + }; + let _superblock_guard = self.superblock_lock.acquire().await; + let held = self.advanced_frontier(0); + self.write_superblock_inner(superblock.as_ref(), held, append_point) + .await + } + /// [`Self::persist_offset_frontier`] for a frontier this replica has not /// reached yet. /// @@ -1030,6 +1400,316 @@ where self.write_superblock(superblock.as_ref(), frontier).await } + /// Whether the offset reservation is close enough to being consumed that it + /// should be extended NOW, off the append path. + /// + /// The append fence is correct but badly placed: it writes the superblock + /// inline in the shard's frame pump, where the consensus tick is a sibling + /// arm, so its two fsyncs delay heartbeats for every group on the core. The + /// fix is to make the fence's fast path + /// (`durable_offset_reserved > end_offset`) the only path it ever takes under + /// load, by extending from the tick instead. + /// + /// HALF a block of headroom, which is a wide margin on purpose: over-claiming + /// costs nothing but offset space, while arriving late puts the write back on + /// the append path. Floored at 1, since validation admits a lease of 1 and + /// `1 / 2` would never trigger, leaving every append to pay the inline claim. + /// A partition that has never minted is skipped: extending every idle + /// partition at boot would write a superblock per partition for nothing, and + /// the first block is already claimed where the partition is created, off + /// the append path. + #[must_use] + pub fn needs_offset_reservation_extension(&self) -> bool { + if self.consensus.replica_count() > 1 || self.superblock.is_none() { + return false; + } + if !self.offset_space.append_live { + return false; + } + let headroom = self + .durable_offset_reserved + .get() + .saturating_sub(self.mint_frontier()); + headroom < (self.offset_reservation_lease / 2).max(1) + } + + /// Extend the reservation a full block past the CEILING already on disk. + /// + /// Pairs with [`Self::needs_offset_reservation_extension`]; the caller is the + /// shard tick, so this write is off the append path. A failure needs no + /// handling beyond the writer's own logging and backoff: the fence at the + /// mint is still there, and it is what refuses the append if the ceiling + /// never caught up. + /// + /// From the ceiling, NOT [`Self::mint_frontier`]. The trigger fires while the + /// append point still sits under the ceiling -- that being the point of + /// extending early -- so claiming a lease past the append point buys back only + /// the headroom the trigger had left, about half a lease, and doubles the + /// write rate the default lease is sized for. Maxed against the append point + /// so a ceiling that somehow fell behind still comes forward. + #[allow(clippy::future_not_send)] + pub async fn extend_offset_reservation(&self) -> bool { + let Some(superblock) = self.superblock.as_ref().map(Rc::clone) else { + return true; + }; + if self.superblock_write_is_backed_off() { + return false; + } + let _superblock_guard = self.superblock_lock.acquire().await; + let ceiling = self.durable_offset_reserved.get().max(self.mint_frontier()); + self.write_claim_from(superblock.as_ref(), ceiling).await + } + + /// Upper bound on the offsets a pending `SendMessages` request will mint, for + /// fencing it BEFORE it enters the pipeline. + /// + /// `project` assigns an op, not a base offset, so the exact range is unknown + /// until the mint runs under `write_lock`. This is deliberately loose: a + /// request pipelined behind others can land above it, and the fence at the + /// mint stays as the exact check. It does not need to be tight -- the claim + /// runs a whole lease block past whatever it is handed, so one of these + /// covers every batch in flight unless a run of them crosses a block + /// boundary. + /// + /// `None` above one replica, where nothing is reserved, and when the body is + /// not one canonical batch, which `convert_request_message` has already + /// rejected by the time this runs. + /// + /// Header decode, NOT `decode_batch_slice`: the verifying decode fails on + /// every ordinary send, because `convert_request_message` runs at + /// [`ChecksumMode::Skip`] and leaves `batch_checksum` zeroed, which would + /// silently drop the fence back to the mint. It also re-hashes every body + /// `admit_wire_request` already hashed. + fn request_mint_ceiling(&self, message: &Message) -> Option { + if self.consensus.replica_count() > 1 { + return None; + } + let body = message + .as_slice() + .get(std::mem::size_of::()..message.header().size as usize)?; + let count = BatchHeader::decode(body).ok()?.message_count; + // The batch's LAST offset, not one past it: the claim adds the exclusive + // successor and the lease itself, so a ceiling one too high wastes an + // offset on every claim and overstates what a crash can lose. + // + // Saturating rather than `None` on overflow. `None` means "no fence + // applies here" and would send an exhausted offset space on to the mint, + // where the refusal fences the partition and takes the node down; a + // saturated ceiling reaches the fence instead and comes back as a + // retryable transient. + let last = u64::from(count).saturating_sub(1); + Some(self.mint_frontier().saturating_add(last)) + } + + /// The append fence: make sure the durable record already permits every + /// offset up to and including `end_offset` before the caller lets them + /// exist. + /// + /// `SendMessagesResponse` hands clients concrete base offsets and the poll + /// path serves committed messages out of the resident journal, so an offset + /// is client-visible long before the threshold-gated flush names it in a + /// segment, and a crash in between hands a second message an offset a client + /// already holds. Fencing here rather than at commit puts it upstream of + /// every way a NEWLY minted offset escapes -- the reply, the poll tier, the + /// peers a prepare reaches -- on the one path both a primary's mint and a + /// backup's re-stamp take. Journal repair + /// (`append_repaired_send_messages`) is the exception and needs none: it + /// re-journals offsets a peer already minted and fenced, so there is nothing + /// new to claim. Claiming through `end_offset + 1 + lease` rather than from + /// the live counter needs no special case for an oversized batch. + /// + /// SOLO ONLY, like everything the reservation feeds: `restore_offset_frontier` + /// seeds no counter from it above one replica, and the boot re-anchor that + /// shapes the chain around it never runs there either. A replicated group + /// paying a superblock write per block would buy nothing -- and an ack there + /// already means a quorum journaled the batch, so re-minting needs a + /// FULL-cluster crash. + /// + /// `false` when the write was attempted and failed. Fail-closed: the send is + /// rejected with nothing externalised, exactly as a failed view persist + /// withholds its sends. + /// + /// BYPASSES the retry backoff, like `record_frontier_before_quarantine` and + /// for the same reason: this is the fence at the MINT, where a refusal is + /// terminal. The failure cell is shared with every other superblock writer on + /// the partition, so a purge's frontier reset or the tick's own extension + /// failing once would otherwise open a 20 ms window (up to 1 s after repeats) + /// in which this takes the node down over a fault on another path entirely. + /// The refusal is also not deferred work: `superblock_wedged` is the gate that + /// decides a run of failures is terminal. + /// + /// The ADMITTED path does honour the backoff, through + /// `reserve_offsets_through_retryable`, because a refusal there costs + /// one client retry and re-running a full atomic replace per retry starves + /// the shard pump on a failing disk. + /// + /// WHERE it is called decides how much a refusal costs. Ahead of the pipeline + /// (`on_request`) the client gets a retryable transient and the group keeps + /// serving. At the mint the op already has its number and its ack is already + /// skipped, so `commit_max` can never pass it and nothing later can commit + /// either: `on_replicate` fences the partition there and takes the node down. + /// At CREATE (`build_partition_fresh`) nothing has been externalised at all, + /// so a refusal fails the build and leaves the namespace unmaterialised for + /// the reconciler to retry, boot included. Going live without the + /// block instead would let the first send land inside the backoff the failed + /// write just armed, where the admitted path refuses it with a transient. + #[allow(clippy::future_not_send)] + #[must_use = "the bool is the fence verdict; dropping it lets the append escape unreserved"] + pub async fn reserve_offsets_through(&self, end_offset: u64) -> bool { + if self.consensus.replica_count() > 1 { + return true; + } + // A frontier: offsets strictly below it are permitted, so covering + // `end_offset` needs a record strictly above it. + if self.durable_offset_reserved.get() > end_offset { + return true; + } + let Some(superblock) = self.superblock.as_ref().map(Rc::clone) else { + return true; + }; + let _superblock_guard = self.superblock_lock.acquire().await; + // A batch queued behind another append's write finds the block already + // extended. + if self.durable_offset_reserved.get() > end_offset { + return true; + } + self.write_offset_claim(superblock.as_ref(), end_offset) + .await + } + + /// [`Self::reserve_offsets_through`] for the ADMITTED path, where a refusal + /// costs the client one retry rather than the process its life. + /// + /// Identical except that it honours the superblock retry backoff, AFTER the + /// coverage fast path: a batch the record already permits owes the disk + /// nothing and must not be refused by a window some other writer opened. + /// + /// The fence at the mint deliberately does not honour it. By the time that + /// one runs the op has its number and its ack is already skipped, so a + /// refusal fences the partition and takes the node down; attempting the write + /// against a disk that just refused one is strictly better than declining to + /// try. Here the client simply retries, so re-running a full create, write, + /// file fsync, rename and directory fsync per retry inside an open window + /// buys nothing and starves the shard pump for as long as the fault lasts. + #[allow(clippy::future_not_send)] + #[must_use = "the bool is the fence verdict; dropping it lets the append escape unreserved"] + async fn reserve_offsets_through_retryable(&self, end_offset: u64) -> bool { + if self.consensus.replica_count() > 1 || self.durable_offset_reserved.get() > end_offset { + return true; + } + if self.superblock_write_is_backed_off() { + return false; + } + self.reserve_offsets_through(end_offset).await + } + + /// The reservation preflight every admitted send passes, whether it arrives + /// at [`Self::on_request`] or is promoted out of the request queue. + /// + /// `true` when the send may be projected. `false` when it was ANSWERED here + /// and must go no further: the client holds a `TransientNotAccepted`, which + /// admitted nothing, so it may re-issue anywhere without double-apply risk. + /// + /// Two ways to come back `false`, neither reaching the mint: an open + /// superblock backoff window, and a claim that was attempted and failed. + /// + /// `waiter` is the submit's in-process reply channel, taken only on a + /// refusal: the deny goes there because `header.client` is then the VSR + /// consensus id, which the bus cannot route. + #[allow(clippy::future_not_send)] + async fn admit_reserved_send( + &self, + message: &Message, + waiter: &mut Option>>, + ) -> bool { + if message.header().operation != Operation::SendMessages { + return true; + } + let Some(ceiling) = self.request_mint_ceiling(message) else { + return true; + }; + if !self.reserve_offsets_through_retryable(ceiling).await { + self.deny_unreserved_send(message.header(), waiter.take()) + .await; + return false; + } + true + } + + /// Answer a send the reservation would not cover with a retryable transient, + /// and say why. + /// + /// `TransientNotAccepted`, per its contract: nothing was admitted, so the + /// client may re-issue anywhere without double-apply risk. It does make the + /// SDK recheck the leader and walk the roster, which finds no better node + /// when the fault is this one's disk -- wasteful, but the weaker code would + /// claim an unknown outcome for a request that provably has none. + #[allow(clippy::future_not_send)] + async fn deny_unreserved_send( + &self, + header: &RoutedRequestHeader, + waiter: Option>>, + ) { + let consensus = self.consensus(); + emit_partition_diag( + tracing::Level::WARN, + &PartitionDiagEvent::new( + ReplicaLogContext::from_consensus(consensus, PlaneKind::Partitions), + "refusing a send: the offset reservation could not be extended", + ) + .with_operation(Operation::SendMessages), + ); + Self::send_partition_deny_or_log( + consensus, + header, + IggyError::TransientNotAccepted.as_code(), + "unreserved send transient reply send failed", + waiter, + ) + .await; + } + + /// Claim a block past `end_offset` unconditionally. + /// + /// Split from [`Self::reserve_offsets_through`] because the tick's extension + /// has to write while the ceiling still covers the append point -- that is the + /// point of extending early -- and the fence's coverage fast path would + /// short-circuit exactly that call. The advancing write is what keeps this + /// monotone: a claim below the record cannot lower it. + /// + /// `false` for an `end_offset` of `u64::MAX`, which EXHAUSTS the offset space: + /// the record is an exclusive frontier, so covering an offset needs a value + /// strictly above it and `u64::MAX + 1` does not exist. Saturating instead + /// would record `u64::MAX`, report success for an offset it does not cover, + /// and let the next boot seed the append counter at `u64::MAX - 1` and re-mint + /// an offset a client already holds -- the one defect this whole path exists + /// to prevent, at the one offset where it would be silent. + #[allow(clippy::future_not_send)] + async fn write_offset_claim(&self, superblock: &SB, end_offset: u64) -> bool { + let Some(exclusive) = end_offset.checked_add(1) else { + tracing::error!( + namespace_raw = self.consensus.group(), + end_offset, + "refusing an append that would exhaust the partition's offset space" + ); + return false; + }; + self.write_claim_from(superblock, exclusive).await + } + + /// Claim a lease of offsets past `exclusive`, the lowest offset the record + /// does not yet permit. + /// + /// Saturating on the lease is safe where [`Self::write_offset_claim`]'s + /// successor is not: a claim clamped to `u64::MAX` still sits strictly above + /// every `end_offset` below it, so the coverage test still holds. Only the + /// successor itself can push the frontier off the end of the space. + #[allow(clippy::future_not_send)] + async fn write_claim_from(&self, superblock: &SB, exclusive: u64) -> bool { + let claim = exclusive.saturating_add(self.offset_reservation_lease); + self.write_superblock_advancing(superblock, 0, claim).await + } + /// Burn one transfer stall round; `true` once the budget is exhausted. /// Lives on the partition, not the session, so a re-minted session /// cannot reset it (see [`Self::transfer_attempts`]). @@ -1135,6 +1815,18 @@ where .unwrap_or(config.messages_required_to_save) } + /// Install the offset-reservation block size resolved from this node's + /// `PartitionsConfig`. + /// + /// Carried on the partition because the fence runs inside `on_request` / + /// `on_replicate`, which take no config. `NonZeroU32` because a zero block + /// reserves nothing and would write the superblock before every append: + /// coercing it here instead would contradict the configuration validator and + /// hide a wiring error that handed this a zero. + pub const fn set_offset_reservation_lease(&mut self, lease: NonZeroU32) { + self.offset_reservation_lease = lease.get() as u64; + } + /// Whether this partition's segments reserve their bytes on open. #[must_use] pub fn effective_preallocate_segments(&self, config: &PartitionsConfig) -> bool { @@ -1604,7 +2296,7 @@ where // commit). Also used below as the poll's high-water bound: this function // is fully synchronous, so the single load cannot drift mid-plan. let commit_offset = self.offsets().commit_offset; - if !self.should_increment_offset || args.count == 0 { + if !self.offset_space.committed_seeded || args.count == 0 { return PollPlan { commit_offset, auto_commit: None, @@ -1852,7 +2544,10 @@ where return Err(IggyError::CannotAppendMessage); } - let dirty_offset = if self.should_increment_offset { + // Only here: this is the only path that mints. A backup re-stamps what + // the primary sends (`append_received_send_messages_to_journal`) and + // must follow it exactly, so raising ITS counter would fork the group. + let dirty_offset = if self.offset_space.append_live { self.dirty_offset .load(Ordering::Relaxed) .checked_add(1) @@ -1885,6 +2580,15 @@ where self.fatal.as_ref() } + /// Consecutive superblock write failures for this group, for the shard's + /// wedge fail-stop. A partition that cannot record its state withholds every + /// view-scoped send and refuses every append, so past some window it is + /// serving nothing and a supervisor should be handling it instead. + #[must_use] + pub const fn superblock_write_failures(&self) -> u64 { + self.superblock_write_failures.get() + } + /// Fence this partition after the shutdown flush failed to persist its /// committed journal prefix: that data is cluster-committed and now lives /// only in this process's memory, so the shard must not report a clean @@ -2244,6 +2948,18 @@ where offset, } } else { + // Fence AHEAD of the pipeline for a mint, not only at the mint. + // A refusal at the mint arrives after the sequencer took the op, + // where the only honest answer left is to fence the partition and + // take the node down (`on_replicate`). Here the request has + // entered nothing, so a transient disk fault costs the client one + // retry instead of costing the process its life. + // + // The preflight answers the client itself on a refusal; see + // [`Self::admit_reserved_send`]. + if !self.admit_reserved_send(&message, &mut reply).await { + return; + } // Two-queue: prepare slot -> project+replicate; prepare full + // request room -> buffer; both full -> drop+warn (client retries // via read-timeout). @@ -2311,9 +3027,18 @@ where /// Promote up to `slots_freed` buffered requests into prepares post-commit. /// - /// Promotion runs no preflight: the request was classified at admission - /// and the slice cannot have gained a higher watermark for it since (only - /// a commit moves it, and this entry has not committed). + /// Promotion runs no DEDUP preflight: the request was classified at + /// admission and the slice cannot have gained a higher watermark for it + /// since (only a commit moves it, and this entry has not committed). + /// + /// The RESERVATION preflight is repeated per promotion, and re-derives the + /// ceiling from the live mint frontier rather than trusting the one the + /// request was admitted under. Several queued batches can cumulatively cross + /// the lease while they wait, and without this the first one past it reaches + /// the exact fence at the mint, where a refusal fences the partition and + /// takes the node down instead of returning `TransientNotAccepted`. A refused + /// promotion ends the drain: the ones behind it want the same claim and would + /// each be answered with the same transient. /// /// Per-iteration `is_primary && is_normal && !is_transferring` asserts inlined /// (closure form's `&consensus` borrow conflicts with `&mut self`). Guards @@ -2332,6 +3057,16 @@ where let req = self.consensus().pop_queued_request(); let Some(mut req) = req else { break }; + // Taken before the preflight so a refusal answers the parked waiter + // instead of waking it with `Canceled`. + let mut reply_sender = req.take_reply_sender(); + if !self + .admit_reserved_send(&req.message, &mut reply_sender) + .await + { + break; + } + let prepare = { let consensus = self.consensus(); assert!( @@ -2348,7 +3083,6 @@ where ); // The waiter parked with the request; it must travel into the // prepare slot or the commit has nobody to answer. - let reply_sender = req.take_reply_sender(); let prepare = req.message.project(consensus); consensus.verify_pipeline(); match reply_sender { @@ -2593,16 +3327,57 @@ where let frozen_for_forward = match replicated_result { Ok(frozen) => frozen, Err(error) => { + // A BACKUP refusing here is the design, not a fault: it rejects + // any prepare whose `base_offset` does not continue its own + // counter, which is exactly what a dropped or out-of-order + // prepare leaves, and withholding `PrepareOk` is the fail-closed + // answer. The primary retransmits, journal repair fills the gap, + // and the group elects around the replica if it cannot catch up. + // Nothing here is owed an ack this replica already skipped. + // + // On the PRIMARY the same return is unrecoverable. The op is + // already in the pipeline with the sequencer advanced past it + // (`push_prepare_entry`), and this sits ahead of + // `send_prepare_ok`, so it never gets its ack and `commit_max` + // can never pass it: every later op journals fine and queues + // behind it forever. Nothing lifts that -- the prepare timeout + // only backs off, a solo group's retransmit target is itself, + // this plane has no `repair_primary_self_acks`, and a solo group + // never starts a view change. Clients get no reply at all, since + // replies are generated on commit, so they wait out their read + // timeout, and once the queues fill so does every send after. + // + // So fence there, the way a failed local commit of a + // cluster-committed op does: the shard picks `fatal` up on its + // next tick and takes the node down. A one-second superblock + // backoff must not cost a partition the rest of the process's + // life in the dark. emit_partition_diag( - tracing::Level::WARN, + if is_backup { + tracing::Level::WARN + } else { + tracing::Level::ERROR + }, &PartitionDiagEvent::new( self.diag_ctx(), - "failed to apply replicated partition operation", + if is_backup { + "failed to apply replicated partition operation" + } else { + "failed to apply an operation this replica sequenced; \ + fencing the partition and shutting down" + }, ) .with_operation(header.operation) .with_op(header.op) .with_error(error.to_string()), ); + if !is_backup && self.fatal.is_none() { + self.fatal = Some(FatalCommit { + namespace_raw: self.namespace().inner(), + op: header.op, + operation: header.operation, + }); + } return; } }; @@ -2882,7 +3657,7 @@ where if validated.message_count == 0 { return Err(IggyError::InvalidCommand); } - let expected_offset = if self.should_increment_offset { + let expected_offset = if self.offset_space.append_live { self.dirty_offset .load(Ordering::Relaxed) .checked_add(1) @@ -2916,6 +3691,19 @@ where .checked_add(u64::from(batch_messages_count) - 1) .ok_or(IggyError::CannotAppendMessage)?; + // Past this line the offsets are in the journal, hence committable, + // pollable, confirmable and forwardable, and no later gate can take + // them back. See [`Self::reserve_offsets_through`]. + // + // LOCK ORDER: the caller holds `write_lock` and this takes + // `superblock_lock` under it. The install path takes them in the + // reverse order, safely only because `reset_offset_frontier_at` drops + // `superblock_lock` before `try_install` takes `write_lock`. Never hold + // `superblock_lock` across a `write_lock` acquire. + if !self.reserve_offsets_through(last_dirty_offset).await { + return Err(IggyError::CannotAppendMessage); + } + let segment_index = self.log.segments().len() - 1; let current_position = self.log.segments()[segment_index].current_position; let next_position = current_position @@ -2949,7 +3737,7 @@ where .await .map_err(|_| IggyError::CannotAppendMessage)?; - self.should_increment_offset = true; + self.note_append_live(); self.dirty_offset .store(last_dirty_offset, Ordering::Relaxed); self.log.segments_mut()[segment_index].current_position = next_position; @@ -3040,7 +3828,15 @@ where let next_offset = next_offset.max(minimum_next_offset); self.dirty_offset .store(next_offset.saturating_sub(1), Ordering::Relaxed); - self.should_increment_offset = next_offset > 0; + // The APPEND bit follows the rewound counter. The committed bit does + // not: this drops an uncommitted suffix, which by definition names + // nothing that ever committed, so raising it here would make a + // truncation the event that publishes offsets no quorum agreed on. + // A rewind to zero is the exception -- nothing is left at all. + self.offset_space.append_live = next_offset > 0; + if next_offset == 0 { + self.offset_space.committed_seeded = false; + } } self.consensus.invalidate_local_dvc_suffix(); Ok(removed) @@ -3340,6 +4136,7 @@ where // gated, so counting here would leave the stats lagging the visible // offset until a flush and would double-count once it fires. if let Some(durable_offset) = durable_offset { + self.note_committed_seeded(); self.offset.store(durable_offset, Ordering::Release); self.stats.set_current_offset(durable_offset); } @@ -3644,6 +4441,12 @@ where if let Some(batch_stats) = batch_stats { let end_offset = batch_stats.end_offset(); + // The committed counter now names data, which is what makes + // it pollable and persistable. Outside the recovered-offset + // guard below: that guard only skips re-counting stats a + // previous life already persisted, and those offsets are + // committed either way. + self.note_committed_seeded(); // A repaired batch at or below the boot-time recovered // durable offset was already counted (and persisted) // before the restart; skip it. Live traffic always sits @@ -4043,98 +4846,72 @@ where } async fn rotate_segment(&mut self, config: &PartitionsConfig) -> Result<(), IggyError> { - let namespace = self.namespace(); - let old_segment_index = self.log.segments().len() - 1; - let active_segment = self.log.active_segment_mut(); - active_segment.sealed = true; - let start_offset = active_segment.end_offset + 1; + let start_offset = self.log.active_segment().end_offset + 1; + self.rotate_segment_at(config, start_offset).await + } - let segment_size = self.effective_segment_size(config); - let enforce_fsync = self.effective_enforce_fsync(config); - let preallocate_segments = self.effective_preallocate_segments(config); - let segment = Segment::new(start_offset, segment_size); - // Prefer the active writer's location: a per-topic path override or a - // config change after the initial segment was created must not scatter - // one partition's segments across two directories. The config layout - // only decides for a partition with no writer yet. - let (messages_path, index_path) = self.partition_dir().map_or_else( - || { - ( - config.get_messages_path( - namespace.stream_id(), - namespace.topic_id(), - namespace.partition_id(), - start_offset, - ), - config.get_index_path( - namespace.stream_id(), - namespace.topic_id(), - namespace.partition_id(), + /// Seal the active segment and plant a fresh empty one at `start_offset`. + /// + /// Shared by the size-driven roll, which plants at `end_offset + 1`, and the + /// boot re-anchor, which plants at the append point the reservation moved the + /// counter to. One seal path, and one order: the new segment's files are + /// created BEFORE the sealed segment's writers are torn down, so a failed + /// create leaves the chain serviceable. + async fn rotate_segment_at( + &mut self, + config: &PartitionsConfig, + start_offset: u64, + ) -> Result<(), IggyError> { + let namespace = self.namespace(); + let sealed_index = self.log.segments().len() - 1; + let sealed_end = self.log.active_segment().end_offset; + debug_assert!( + start_offset > sealed_end, + "a plant at {start_offset} overlaps the sealed tail ending at {sealed_end}" + ); + // A wider gap than the roll's own is legitimate only with the anchor + // already durable, which is the caller's obligation + // (`record_reanchor_gap`) and is otherwise readable nowhere in here. + #[cfg(debug_assertions)] + if start_offset > sealed_end.saturating_add(1) + && let Some(partition_dir) = self.partition_dir() + { + debug_assert!( + matches!( + crate::segment_anchor::read_anchor(&partition_dir, start_offset).await, + Ok(Some(anchor)) if anchor.covers( start_offset, - ), - ) - }, - |dir| { - ( - format!("{dir}/{start_offset:0>20}.log"), - format!("{dir}/{start_offset:0>20}.index"), - ) - }, - ); - - let storage = SegmentStorage::new(&messages_path, &index_path, 0, 0, false) - .await - .map_err(|_| IggyError::CannotCreateSegmentLogFile(messages_path.clone()))?; - let messages_size_bytes = storage - .messages_writer - .as_ref() - .ok_or_else(|| IggyError::CannotCreateSegmentLogFile(messages_path.clone()))? - .size_counter(); - let messages_writer = Rc::new( - MessagesWriter::new( - &messages_path, - messages_size_bytes, - enforce_fsync, - false, - preallocate_segments.then_some(segment_size), - ) - .await - .map_err(|_| IggyError::CannotCreateSegmentLogFile(messages_path.clone()))?, - ); - let index_size_bytes = storage - .index_writer - .as_ref() - .ok_or_else(|| IggyError::CannotCreateSegmentIndexFile(index_path.clone()))? - .size_counter(); - let index_writer = Rc::new( - IggyIndexWriter::new(&index_path, index_size_bytes, enforce_fsync, false) - .await - .map_err(|_| IggyError::CannotCreateSegmentIndexFile(index_path.clone()))?, - ); + self.log.segments()[sealed_index].start_offset, + sealed_end, + ) + ), + "a plant at {start_offset} leaves a gap past {sealed_end} with no anchor" + ); + } + self.log.active_segment_mut().sealed = true; + self.install_empty_segment(config, start_offset).await?; + self.stats.increment_segments_count(1); - let old_storage = &mut self.log.storages_mut()[old_segment_index]; - let _ = old_storage.shutdown(); - self.log.messages_writers_mut()[old_segment_index] = None; - self.log.index_writers_mut()[old_segment_index] = None; + let sealed_storage = &mut self.log.storages_mut()[sealed_index]; + let _ = sealed_storage.shutdown(); + self.log.messages_writers_mut()[sealed_index] = None; + self.log.index_writers_mut()[sealed_index] = None; // Drop the sealed segment's in-memory index cache: only the ACTIVE // segment's cache is ever read (the `commit_messages` flush staging), // so a sealed cache is dead weight. - self.log.indexes_mut()[old_segment_index] = None; + self.log.indexes_mut()[sealed_index] = None; // The read fd cached while this segment was active is not counted by // the sealed LRU budget, so it must not survive the seal; the next // sealed poll re-fills the fresh slot under the LRU's rules. - self.log.reset_read_state(old_segment_index); - - self.log - .add_persisted_segment(segment, storage, Some(messages_writer), Some(index_writer)); - self.stats.increment_segments_count(1); + self.log.reset_read_state(sealed_index); debug!( target: "iggy.partitions.diag", plane = "partitions", namespace_raw = namespace.inner(), + sealed_end, start_offset, - "rotated to new segment" + "sealed the active segment and planted a fresh one" ); Ok(()) } @@ -4258,7 +5035,11 @@ where let _ = storage.shutdown(); drop(storage); - for path in messages_path.into_iter().chain(index_path) { + for path in messages_path + .into_iter() + .chain(index_path) + .chain(self.anchor_cleanup_path(segment.start_offset)) + { match compio::fs::remove_file(&path).await { Ok(()) => {} Err(error) if error.kind() == std::io::ErrorKind::NotFound => {} @@ -4386,6 +5167,218 @@ where Ok(()) } + /// Re-anchor the append point after boot re-seeded the offset counter above + /// what the recovered segment chain holds. + /// + /// A hole INSIDE a segment is not survivable: `recover_segment_bounds` walks + /// a segment from its FILENAME with a running `expected_offset` and REFUSES + /// at the first offset that does not continue it (`OffsetDiscontinuity`), + /// which on a solo group tombstones the partition. A surviving index does not + /// help: a first entry that is not the file-name offset makes recovery + /// discard the index and walk from byte 0, reaching the same refusal. On a + /// segment BOUNDARY every reader copes -- absolute offsets in the index, + /// `disk_poll_start` walking on into later segments, and a chain guard that + /// admits a forward gap the reservation covers. + /// + /// So an empty tail is unlinked (its name claims a range it does not hold), a + /// sized tail (the only copy of its messages) is sealed with a fresh segment + /// planted at the append point, and a chain the unlinks emptied is planted + /// directly -- `ensure_initial_segment` names its segment for the COMMITTED + /// frontier and would put the first mint inside it. + /// + /// # Errors + /// [`IggyError`] when the fresh segment cannot be created, leaving the + /// partition without a serviceable chain. + #[allow(clippy::future_not_send, clippy::too_many_lines)] + pub async fn reanchor_to_offset_frontier( + &mut self, + config: &PartitionsConfig, + ) -> Result<(), IggyError> { + // Where the next append will land: the counter, or an armed mint floor + // above it. The floor is the whole reason a hole can appear, so + // anchoring to the counter alone would leave the chain as unprepared. + let frontier = self.mint_frontier(); + if frontier == 0 { + return Ok(()); + } + let namespace = self.namespace(); + let mut retired = 0usize; + while let Some(segment) = self.log.segments().last() { + if segment.size.as_bytes_u64() > 0 || segment.start_offset >= frontier { + break; + } + let Some((segment, mut storage)) = self.log.retire_back() else { + break; + }; + let (messages_path, index_path) = storage.segment_and_index_paths(); + let _ = storage.shutdown(); + drop(storage); + for path in messages_path + .into_iter() + .chain(index_path) + .chain(self.anchor_cleanup_path(segment.start_offset)) + { + match compio::fs::remove_file(&path).await { + Ok(()) => {} + Err(error) if error.kind() == std::io::ErrorKind::NotFound => {} + Err(error) => { + // Refused, not logged. The segment is already out of the + // in-memory chain, so a file left behind becomes a + // non-tail empty segment as soon as the plant lands -- + // `[sized][stale empty][planted]` -- which the next boot + // refuses outright as `EmptyNonTailSegment`. Failing boot + // here says so while the directory is still readable. + error!( + target: "iggy.partitions.diag", + plane = "partitions", + namespace_raw = namespace.inner(), + path = %path, + %error, + "failed to unlink a stale empty segment during the boot \ + re-anchor; refusing to plant beside it" + ); + return Err(IggyError::CannotDeleteFile); + } + } + } + tracing::info!( + target: "iggy.partitions.diag", + plane = "partitions", + namespace_raw = namespace.inner(), + start_offset = segment.start_offset, + offset_frontier = frontier, + "unlinked an empty segment named below the restored offset frontier" + ); + // Boot DOES count the recovered chain -- `load_persisted_segments` + // increments per segment before it looks at the size, so empty tails + // are in the total -- and retention pairs its own retire with a + // decrement. Without this the count stays one high on the wire for + // the life of the process. + self.stats.decrement_segments_count(1); + retired += 1; + } + // Durable before anything is planted beside them: a crash in between + // would boot the stale name back into the chain. Refused, not logged: + // the emptied-chain arm plants through `install_empty_segment`, which + // fsyncs no directory of its own, so a swallowed error here is the whole + // promise gone. + if retired > 0 + && let Some(partition_dir) = self.partition_dir.clone() + && let Err(error) = crate::state_transfer::fsync_dir(&partition_dir).await + { + error!( + target: "iggy.partitions.diag", + plane = "partitions", + namespace_raw = namespace.inner(), + partition_dir, + %error, + "boot re-anchor could not fsync the partition dir after unlinking; \ + refusing to plant beside a name that may come back" + ); + return Err(IggyError::CannotSyncFile); + } + // Bounds copied out: the plant below takes `&mut self`, so the borrow on + // the chain cannot still be live. + let tail = self + .log + .segments() + .last() + .map(|segment| (segment.start_offset, segment.end_offset, segment.size)); + match tail { + // An EMPTIED chain still needs the plant, and it cannot be left to + // the caller's `ensure_initial_segment`, which names the segment for + // the COMMITTED frontier and knows nothing of the append point. On + // the shape a crash before the first flush leaves -- committed + // frontier 0, append point a lease block up -- that plants + // `0.log` and then mints inside it, the hole this function exists to + // prevent. The index does not save it either: a first entry that is + // not the file-name offset makes recovery discard the index and walk + // from byte 0, where the discontinuity tombstones the partition. + // + // No anchor: with nothing before it the plant leaves no gap, so the + // chain guard has no pair to judge. + None => { + self.install_empty_segment(config, frontier).await?; + self.stats.increment_segments_count(1); + tracing::info!( + target: "iggy.partitions.diag", + plane = "partitions", + namespace_raw = namespace.inner(), + offset_frontier = frontier, + "planted a fresh segment at the restored offset frontier over an \ + empty recovered chain" + ); + } + // Only a SIZED tail: an empty one either just went, or is already + // named at the frontier and can take the appends as it is. + Some((sealed_start, sealed_end, size)) + if size.as_bytes_u64() > 0 && sealed_end.saturating_add(1) < frontier => + { + self.record_reanchor_gap(frontier, sealed_start, sealed_end) + .await?; + self.rotate_segment_at(config, frontier).await?; + tracing::info!( + target: "iggy.partitions.diag", + plane = "partitions", + namespace_raw = namespace.inner(), + sealed_end, + offset_frontier = frontier, + "sealed the recovered tail and planted a fresh segment at the \ + restored offset frontier" + ); + } + Some(_) => {} + } + Ok(()) + } + + /// Write the anchor that makes the gap a plant at `frontier` leaves + /// legitimate, and make it durable before the segment exists. + /// + /// The chain guard admits a forward gap only when the far side carries an + /// anchor naming exactly the near side, so this record is what separates the + /// re-anchor's own gap from a lost segment. Ordering is load-bearing in one + /// direction only: an anchor with no segment is swept at the next boot, + /// while a segment with no anchor reads as damage. + /// + /// # Errors + /// [`IggyError::CannotCreateSegmentLogFile`] naming the anchor path. The + /// caller must not plant. + #[allow(clippy::future_not_send)] + async fn record_reanchor_gap( + &self, + frontier: u64, + sealed_start: u64, + sealed_end: u64, + ) -> Result<(), IggyError> { + // No directory means an in-memory partition, whose chain no boot reads. + let Some(partition_dir) = self.partition_dir() else { + return Ok(()); + }; + let anchor = crate::segment_anchor::SegmentAnchor { + planted_start: frontier, + sealed_start, + sealed_end, + }; + if let Err(error) = crate::segment_anchor::write_anchor(&partition_dir, anchor).await { + error!( + target: "iggy.partitions.diag", + plane = "partitions", + namespace_raw = self.namespace().inner(), + offset_frontier = frontier, + sealed_start, + sealed_end, + %error, + "could not record the boot re-anchor's gap; refusing to plant a segment \ + the next boot would read as a lost one" + ); + return Err(IggyError::CannotCreateSegmentLogFile( + crate::segment_anchor::anchor_path(&partition_dir, frontier), + )); + } + Ok(()) + } + /// Record the purge's frontier reset BEFORE the purge touches anything. /// /// The unlinks are made durable by their own directory fsync, so a crash @@ -4477,7 +5470,7 @@ where // Drain every segment (including the active one) and unlink its files. let segment_count = self.log.segments().len(); for _ in 0..segment_count { - let Some((_, mut storage)) = self.log.retire_front() else { + let Some((segment, mut storage)) = self.log.retire_front() else { break; }; @@ -4485,7 +5478,11 @@ where let _ = storage.shutdown(); drop(storage); - for path in messages_path.into_iter().chain(index_path) { + for path in messages_path + .into_iter() + .chain(index_path) + .chain(self.anchor_cleanup_path(segment.start_offset)) + { match compio::fs::remove_file(&path).await { Ok(()) => {} Err(error) if error.kind() == std::io::ErrorKind::NotFound => {} @@ -4532,7 +5529,7 @@ where // whole body. self.offset.store(start_offset, Ordering::Release); self.dirty_offset.store(start_offset, Ordering::Relaxed); - self.should_increment_offset = false; + self.set_offset_space_used(false); // Recreate a fresh empty segment at offset 0 with real writers. Every // segment is drained by now, so a failure here is the fence case. @@ -5060,7 +6057,7 @@ where .await .map_err(|_| IggyError::CannotAppendMessage)?; - self.should_increment_offset = true; + self.note_append_live(); self.dirty_offset .store(dirty.max(last_offset), Ordering::Relaxed); self.log.segments_mut()[segment_index].current_position = next_position; @@ -5351,170 +6348,920 @@ fn leading_oversized_end(segments: &[Segment], max_bytes: u64) -> Option { up_to } -/// `end_offset` of the `count`-th oldest sealed (non-active) segment of -/// `segments`, or `None` when there is no deletable sealed segment. Clamps to -/// the last sealed segment when fewer than `count` exist. -fn nth_oldest_sealed_end(segments: &[Segment], count: u32) -> Option { - if count == 0 { - return None; +/// `end_offset` of the `count`-th oldest sealed (non-active) segment of +/// `segments`, or `None` when there is no deletable sealed segment. Clamps to +/// the last sealed segment when fewer than `count` exist. +fn nth_oldest_sealed_end(segments: &[Segment], count: u32) -> Option { + if count == 0 { + return None; + } + // Exclude the active (last) segment, take the leading sealed run, then the + // `count`-th of those (or the last available when fewer exist). + let last_idx = segments.len().saturating_sub(1); + segments + .iter() + .take(last_idx) + .take_while(|segment| segment.sealed) + .take(count as usize) + .map(|segment| segment.end_offset) + .last() +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::iggy_index::{IGGY_INDEX_SIZE, IggyIndex, IggyIndexCache}; + use crate::iggy_index_reader::IggyIndexReader; + use crate::poll_plan::{DiskReadOutcome, SealedSegmentHandle}; + use bytes::Bytes; + use compio::io::AsyncWriteAtExt; + use consensus::LocalPipeline; + use iggy_binary_protocol::{Command, ReplyHeader, WireConsumer, WireEncode}; + use message_bus::{BusMessage, SendError}; + use server_common::MESSAGE_ALIGN; + use server_common::send_messages::{ + COMMAND_HEADER_SIZE, IggyMessage, IggyMessageHeader, IggyMessages, SendMessagesOwned, + decode_batch_slice, + }; + use std::cell::RefCell; + use std::rc::Rc; + + const TEST_CLUSTER: u128 = 1; + + pub(super) fn test_partition() -> IggyPartition { + let namespace = IggyNamespace::new(1, 1, 0); + let consensus = VsrConsensus::new( + TEST_CLUSTER, + 0, + 1, + namespace.inner(), + IggyMessageBus::new(0), + LocalPipeline::new(), + ); + consensus.init(); + IggyPartition::with_in_memory_storage( + Arc::new(PartitionStats::default()), + consensus, + IggyByteSize::from(1024 * 1024), + false, + ) + } + + /// A SOLO partition, the shape the offset reservation is scoped to. + fn solo_recording_partition() -> IggyPartition { + let namespace = IggyNamespace::new(1, 1, 0); + let consensus = VsrConsensus::new( + TEST_CLUSTER, + 0, + 1, + namespace.inner(), + IggyMessageBus::new(0), + LocalPipeline::new(), + ); + consensus.init(); + IggyPartition::with_in_memory_storage( + Arc::new(PartitionStats::default()), + consensus, + IggyByteSize::from(1024 * 1024), + false, + ) + } + + /// Partition whose consensus already advanced to `(view, log_view)` with + /// nothing marked durable, as after a view change and before the persist + /// gate runs. + fn partition_at_view( + view: u32, + log_view: u32, + ) -> IggyPartition { + let namespace = IggyNamespace::new(1, 1, 0); + let mut consensus = VsrConsensus::new( + TEST_CLUSTER, + 0, + 3, + namespace.inner(), + IggyMessageBus::new(0), + LocalPipeline::new(), + ); + consensus.set_view(view); + consensus.set_log_view(log_view); + consensus.init_as_backup(); + IggyPartition::with_in_memory_storage( + Arc::new(PartitionStats::default()), + consensus, + IggyByteSize::from(1024 * 1024), + false, + ) + } + + /// In-memory superblock double: records every payload, counts attempts, + /// and injects write failures. + #[derive(Default)] + struct RecordingSuperblock { + writes: RefCell>>, + attempts: Cell, + fail_writes: Cell, + } + + impl journal::superblock::SuperblockStore for RecordingSuperblock { + async fn write(&self, payload: &[u8]) -> std::io::Result<()> { + self.attempts.set(self.attempts.get() + 1); + if self.fail_writes.get() { + return Err(std::io::Error::other("injected superblock write failure")); + } + self.writes.borrow_mut().push(payload.to_vec()); + Ok(()) + } + + async fn read_latest(&self) -> std::io::Result { + Ok(self + .writes + .borrow() + .last() + .map_or(journal::superblock::SuperblockContents::Empty, |bytes| { + journal::superblock::SuperblockContents::Present(bytes.clone()) + })) + } + } + + #[compio::test] + async fn given_storeless_partition_when_persist_gate_runs_should_mark_current_view_durable() { + let partition = partition_at_view(2, 1); + assert!(partition.consensus().needs_superblock_persist()); + + assert!(partition.persist_superblock_if_needed().await); + + assert!( + !partition.consensus().needs_superblock_persist(), + "a storeless partition must record durable = current, or the dispatch \ + tripwire would fire on its first view-scoped send" + ); + } + + #[compio::test] + async fn given_advanced_view_when_persist_gate_runs_should_write_vsr_state_once() { + let mut partition = partition_at_view(3, 2); + let store = Rc::new(RecordingSuperblock::default()); + partition.set_superblock(store.clone(), None); + + assert!(partition.persist_superblock_if_needed().await); + + let state = consensus::VsrState::try_from(store.writes.borrow()[0].as_slice()) + .expect("recorded payload decodes as a VsrState"); + assert_eq!(state.cluster, TEST_CLUSTER); + assert_eq!(state.view, 3); + assert_eq!(state.log_view, 2); + assert_eq!( + (state.checkpoint_op, state.checkpoint_checksum), + (0, 0), + "no partition checkpoint exists yet, so the pairing fields stay zero" + ); + assert!(!partition.consensus().needs_superblock_persist()); + + assert!(partition.persist_superblock_if_needed().await); + assert_eq!( + store.attempts.get(), + 1, + "an unchanged view must take the lock-free fast path, not rewrite" + ); + } + + /// The `offset_frontier` of the most recent recorded write. + fn last_recorded_frontier(store: &RecordingSuperblock) -> u64 { + let writes = store.writes.borrow(); + let bytes = writes.last().expect("a superblock write landed"); + consensus::VsrState::try_from(bytes.as_slice()) + .expect("recorded payload decodes as a VsrState") + .offset_frontier + } + + fn last_recorded_reservation(store: &RecordingSuperblock) -> u64 { + let writes = store.writes.borrow(); + let bytes = writes.last().expect("a superblock write landed"); + consensus::VsrState::try_from(bytes.as_slice()) + .expect("recorded payload decodes as a VsrState") + .offset_reserved + } + + fn recorded_state(offset_frontier: u64, offset_reserved: u64) -> consensus::VsrState { + consensus::VsrState { + cluster: TEST_CLUSTER, + replica_id: 0, + replica_count: 1, + view: 1, + log_view: 1, + commit_max: 0, + checkpoint_op: 0, + checkpoint_checksum: 0, + offset_frontier, + offset_reserved, + } + } + + fn test_lease(value: u32) -> NonZeroU32 { + NonZeroU32::new(value).expect("a nonzero test lease") + } + + /// One superblock write per block, not per batch: a fence writing per append + /// would put two fsyncs in front of every produce. + #[compio::test] + async fn given_appends_inside_the_block_when_fencing_should_write_the_superblock_once() { + let store = Rc::new(RecordingSuperblock::default()); + let mut partition = solo_recording_partition(); + partition.set_superblock(store.clone(), None); + partition.set_offset_reservation_lease(test_lease(16)); + + assert!(partition.reserve_offsets_through(0).await); + assert_eq!(store.attempts.get(), 1, "the first offset claims a block"); + assert_eq!( + last_recorded_reservation(&store), + 17, + "the claim runs one past the offset plus the lease" + ); + + for offset in 1..=16 { + assert!(partition.reserve_offsets_through(offset).await); + } + assert_eq!( + store.attempts.get(), + 1, + "every offset inside the block is covered by the claim already on disk" + ); + + assert!(partition.reserve_offsets_through(17).await); + assert_eq!( + store.attempts.get(), + 2, + "the first offset past the block extends it" + ); + assert_eq!(last_recorded_reservation(&store), 34); + } + + /// A restored counter above the chain must move the chain, not just the + /// counter. Left as one segment named at 0, the next mint lands INSIDE it -- + /// a shape production's boot never produces, so the harness would model + /// something the server cannot reach and could not expose the chain refusal + /// the boot after would hit. + #[test] + fn given_a_restored_counter_above_an_empty_chain_when_reanchoring_should_plant_at_the_mint() { + let mut partition = solo_recording_partition(); + let recovered = recorded_state(0, 65_537); + partition.set_superblock(Rc::new(RecordingSuperblock::default()), Some(&recovered)); + partition.restore_offset_frontier(Some(&recovered)); + assert_eq!(partition.mint_frontier(), 65_537); + assert_eq!( + partition.log.segments().len(), + 1, + "the premise: one empty segment named at 0, as the rebuild leaves it" + ); + + partition.reanchor_in_memory_to_mint_frontier(IggyByteSize::from(1024 * 1024)); + + let starts: Vec = partition + .log + .segments() + .iter() + .map(|segment| segment.start_offset) + .collect(); + assert_eq!( + starts, + vec![65_537], + "the empty segment claiming 0.. is retired and one planted at the \ + append point, exactly as boot's emptied-chain arm does" + ); + } + + /// A SIZED tail is the only copy of its messages, so it is sealed and the + /// plant goes past it, leaving the gap the chain guard admits by anchor. + #[test] + fn given_a_restored_counter_above_a_sized_tail_when_reanchoring_should_seal_and_plant_past_it() + { + let mut partition = solo_recording_partition(); + { + let tail = partition.log.active_segment_mut(); + tail.size = IggyByteSize::from(4_096); + tail.end_offset = 9; + } + partition.note_committed_seeded(); + partition.offset.store(9, Ordering::Release); + partition.dirty_offset.store(9, Ordering::Relaxed); + let recovered = recorded_state(10, 65_547); + partition.restore_offset_frontier(Some(&recovered)); + assert_eq!(partition.mint_frontier(), 65_547); + + partition.reanchor_in_memory_to_mint_frontier(IggyByteSize::from(1024 * 1024)); + + let segments = partition.log.segments(); + assert_eq!(segments.len(), 2, "the sized tail is kept, not retired"); + assert!(segments[0].sealed, "and sealed before the plant lands"); + assert_eq!(segments[0].end_offset, 9); + assert_eq!( + segments[1].start_offset, 65_547, + "the plant names the append point, so the next mint starts a segment \ + rather than landing inside one" + ); + } + + /// A tail already named AT the append point takes the appends as it stands. + /// Planting beside it would leave two segments claiming the same start + /// offset, which no chain guard admits. + #[test] + fn given_a_chain_already_anchored_at_the_mint_when_reanchoring_should_leave_it_alone() { + let segment_size = IggyByteSize::from(1024 * 1024); + let mut partition = solo_recording_partition(); + { + let tail = partition.log.active_segment_mut(); + tail.size = IggyByteSize::from(4_096); + tail.end_offset = 9; + } + // The shape a clean boot leaves: the flushed tail, then an empty segment + // already named for the append point. + partition.log.add_persisted_segment( + crate::Segment::new(10, segment_size), + server_common::SegmentStorage::default(), + None, + None, + ); + partition.note_committed_seeded(); + partition.offset.store(9, Ordering::Release); + partition.dirty_offset.store(9, Ordering::Relaxed); + assert_eq!(partition.mint_frontier(), 10); + + partition.reanchor_in_memory_to_mint_frontier(segment_size); + + let starts: Vec = partition + .log + .segments() + .iter() + .map(|segment| segment.start_offset) + .collect(); + assert_eq!( + starts, + vec![0, 10], + "the empty tail is named at the append point, so it is neither retired \ + nor planted beside" + ); + assert!( + !partition.log.segments()[0].sealed, + "and nothing was sealed, since no plant needed a gap" + ); + } + + /// The record is an EXCLUSIVE frontier, so `u64::MAX` can never be covered: + /// covering it would need `u64::MAX + 1`. Saturating and reporting success + /// there confirms an offset to a client that the next boot re-mints, which is + /// the exact defect this path exists to prevent, at the one offset where it + /// would be silent. + #[compio::test] + async fn given_an_exhausted_offset_space_when_fencing_should_refuse_rather_than_confirm() { + let store = Rc::new(RecordingSuperblock::default()); + let mut partition = solo_recording_partition(); + partition.set_superblock(store.clone(), None); + partition.set_offset_reservation_lease(test_lease(16)); + + assert!( + partition.reserve_offsets_through(u64::MAX - 1).await, + "the last representable offset is still reservable" + ); + assert_eq!( + last_recorded_reservation(&store), + u64::MAX, + "a claim clamped to the top of the space still sits strictly above \ + the offset it covers" + ); + + let attempts = store.attempts.get(); + assert!( + !partition.reserve_offsets_through(u64::MAX).await, + "the terminal offset must be refused, not confirmed" + ); + assert_eq!( + store.attempts.get(), + attempts, + "and refused without attempting a write it could not make correct" + ); + } + + /// The boot after that refusal: the counter resumes AT the terminal offset + /// and the fence keeps refusing it, so nothing a client holds is reissued. + #[compio::test] + async fn given_a_saturated_reservation_when_restored_should_keep_refusing_the_terminal_offset() + { + let store = Rc::new(RecordingSuperblock::default()); + let mut partition = solo_recording_partition(); + let recovered = recorded_state(0, u64::MAX); + partition.set_superblock(store.clone(), Some(&recovered)); + partition.set_offset_reservation_lease(test_lease(16)); + partition.restore_offset_frontier(Some(&recovered)); + + assert_eq!( + partition.mint_frontier(), + u64::MAX, + "the append point resumes above every offset the reservation covered" + ); + assert!( + !partition.reserve_offsets_through(u64::MAX).await, + "the one offset the record never covered must stay unmintable" + ); + } + + /// The tick claims a lease past the CEILING. Extending past the append point + /// instead buys back only the headroom the trigger had left -- about half a + /// lease -- and doubles the write rate the default lease is sized for. + #[compio::test] + async fn given_a_tick_extension_when_it_writes_should_advance_the_ceiling_a_full_lease() { + let store = Rc::new(RecordingSuperblock::default()); + let mut partition = solo_recording_partition(); + partition.set_superblock(store.clone(), None); + partition.set_offset_reservation_lease(test_lease(16)); + + assert!(partition.reserve_offsets_through(0).await); + assert_eq!(last_recorded_reservation(&store), 17); + partition.note_append_live(); + + // Half the block consumed, which is where the trigger fires. + partition.dirty_offset.store(11, Ordering::Relaxed); + assert!(partition.needs_offset_reservation_extension()); + assert!(partition.extend_offset_reservation().await); + assert_eq!( + last_recorded_reservation(&store), + 33, + "a full lease past the ceiling of 17, not past the append point of 12" + ); + + // And the trigger is genuinely satisfied for a full block of appends, + // rather than re-firing after another half. + for offset in 12..=24 { + partition.dirty_offset.store(offset, Ordering::Relaxed); + assert!( + !partition.needs_offset_reservation_extension(), + "offset {offset} still sits a full half-lease under the new ceiling" + ); + } + } + + /// A storeless partition (in-memory, simulated) reserves nothing at all, so + /// the tick must never reach a write for one. + #[test] + fn given_a_storeless_partition_when_ticking_should_not_extend() { + let mut partition = solo_recording_partition(); + partition.set_offset_reservation_lease(test_lease(16)); + // Without this the assert would pass on `!append_live` alone, leaving + // the store check it is named for untested. + partition.note_append_live(); + assert!(partition.superblock.is_none(), "the premise: no store"); + assert!(!partition.needs_offset_reservation_extension()); + } + + /// Inside an open backoff window the ADMITTED path refuses without touching + /// the disk the last writer just found broken. Every producer retry otherwise + /// re-runs a full atomic replace, which starves the shard pump for as long as + /// the fault lasts. + #[compio::test] + async fn given_an_open_backoff_window_when_preflighting_a_send_should_refuse_without_writing() { + let store = Rc::new(RecordingSuperblock::default()); + let mut partition = solo_recording_partition(); + partition.set_superblock(store.clone(), None); + partition.set_offset_reservation_lease(test_lease(16)); + partition + .superblock_retry_after_micros + .set(partition.consensus().clock_realtime_micros() + 1_000_000); + + assert!( + !partition.reserve_offsets_through_retryable(0).await, + "a claim that would need a write is refused inside the window" + ); + assert_eq!( + store.attempts.get(), + 0, + "and refused without any I/O at all" + ); + + // The fence at the MINT keeps its bypass: a refusal there fences the + // partition and takes the node down, so trying the write is strictly + // better than declining to. + assert!(partition.reserve_offsets_through(0).await); + assert_eq!(store.attempts.get(), 1); + + // A batch the record already covers owes the disk nothing, so the open + // window must not refuse it either. + assert!( + partition.reserve_offsets_through_retryable(0).await, + "the coverage fast path wins over the backoff window" + ); + assert_eq!(store.attempts.get(), 1); + } + + /// A journaled offset is not a committed one. Seeding the committed bit at + /// the append would serve a resident offset 0 to a consumer before the first + /// commit and let the frontier persist name data no quorum agreed on. + #[test] + fn given_a_journaled_offset_when_uncommitted_should_not_seed_the_committed_counter() { + let mut partition = solo_recording_partition(); + + partition.note_append_live(); + partition.dirty_offset.store(0, Ordering::Relaxed); + assert!(partition.offset_space.append_live); + assert!( + !partition.offset_space.committed_seeded, + "the append moves the append counter alone" + ); + assert_eq!(partition.mint_frontier(), 1, "the next mint continues it"); + assert_eq!( + partition.offset_frontier(), + 0, + "and the frontier a persist would record still names no data" + ); + + partition.note_committed_seeded(); + assert_eq!( + partition.offset_frontier(), + 1, + "commit is what publishes the offset" + ); + } + + /// Fail-closed: offsets the record does not cover would be confirmed to a + /// client with nothing durable saying they were handed out. + /// The fence ahead of the pipeline must bound a mint the ordinary send path + /// actually takes. `convert_request_message` runs at `ChecksumMode::Skip`, so + /// a verifying decode of its output fails and the ceiling would come back + /// `None`, dropping every solo send back to the fence at the mint, where a + /// refusal fences the partition and exits the node. + #[test] + fn given_a_checksumless_send_when_bounding_the_mint_should_read_the_batch_header() { + let partition = solo_recording_partition(); + let namespace = IggyNamespace::from_raw(partition.consensus().group()); + let message = checksumless_send_request(namespace, 3); + let body = &message.as_slice() + [std::mem::size_of::()..message.header().size as usize]; + + assert!( + decode_batch_slice(body).is_err(), + "the premise: a Skip-converted body does not survive a verifying decode" + ); + assert_eq!( + partition.request_mint_ceiling(&message), + Some(2), + "the ceiling is the batch's LAST offset: three messages from a frontier \ + of 0 mint 0, 1 and 2, and the claim adds the exclusive successor itself" + ); + } + + /// Nothing is reserved above one replica, so the ceiling is not computed + /// there either. + #[test] + fn given_a_replicated_group_when_bounding_the_mint_should_not_compute_a_ceiling() { + let partition = partition_at_view(1, 1); + assert!(partition.consensus().replica_count() > 1); + let namespace = IggyNamespace::from_raw(partition.consensus().group()); + assert_eq!( + partition.request_mint_ceiling(&checksumless_send_request(namespace, 3)), + None + ); + } + + #[compio::test] + async fn given_failing_superblock_when_fencing_should_refuse() { + let store = Rc::new(RecordingSuperblock::default()); + let mut partition = solo_recording_partition(); + partition.set_superblock(store.clone(), None); + partition.set_offset_reservation_lease(test_lease(4)); + store.fail_writes.set(true); + + assert!( + !partition.reserve_offsets_through(0).await, + "an unrecordable claim must refuse the append" + ); + } + + /// The point of extending from the tick: after it runs, the append path finds + /// the ceiling already covering it and writes nothing. Without this the two + /// fsyncs land in front of a produce, inside the frame pump the consensus + /// tick shares. + #[compio::test] + async fn given_a_consumed_block_when_the_tick_extends_should_leave_the_append_path_writeless() { + let store = Rc::new(RecordingSuperblock::default()); + let mut partition = solo_recording_partition(); + partition.set_superblock(store.clone(), None); + partition.set_offset_reservation_lease(test_lease(16)); + + // First append pays for the first claim, as it must: nothing durable yet. + assert!(partition.reserve_offsets_through(0).await); + assert_eq!(store.attempts.get(), 1); + partition.set_offset_space_used(true); + + // Inside the block there is nothing to do, from either caller. + partition.dirty_offset.store(4, Ordering::Relaxed); + assert!( + !partition.needs_offset_reservation_extension(), + "a full block of headroom needs no extension" + ); + + // Past half the block the tick takes the write. + partition.dirty_offset.store(11, Ordering::Relaxed); + assert!( + partition.needs_offset_reservation_extension(), + "under half a block of headroom the tick must extend" + ); + assert!(partition.extend_offset_reservation().await); + assert_eq!( + store.attempts.get(), + 2, + "the tick wrote, not the append path" + ); + + // And now the append path is writeless across the rest of the old block. + let before = store.attempts.get(); + for offset in 12..=16 { + assert!(partition.reserve_offsets_through(offset).await); + } + assert_eq!( + store.attempts.get(), + before, + "every append after the extension must take the fence's fast path" + ); + } + + /// The extension must not fire for a partition that has never minted, or boot + /// would write a superblock per idle partition for nothing. + #[test] + fn given_an_untouched_partition_when_ticking_should_not_extend_the_reservation() { + let mut partition = solo_recording_partition(); + partition.set_superblock(Rc::new(RecordingSuperblock::default()), None); + assert!(!partition.offset_space.append_live); + assert!(!partition.needs_offset_reservation_extension()); + } + + /// A boot that consumed a reservation has ZERO headroom -- the append point + /// sits exactly at the ceiling -- so the extension has to fire before the + /// first produce rather than after it. + #[test] + fn given_a_reservation_seeded_boot_when_ticking_should_extend_before_the_first_produce() { + let mut partition = solo_recording_partition(); + let recovered = recorded_state(0, 65_537); + partition.set_superblock(Rc::new(RecordingSuperblock::default()), Some(&recovered)); + partition.restore_offset_frontier(Some(&recovered)); + assert_eq!(partition.mint_frontier(), 65_537); + assert!( + partition.needs_offset_reservation_extension(), + "the seeded append point is at the ceiling, so the next append would \ + otherwise pay for the write" + ); + } + + /// A replicated group pays nothing for a protection it cannot use: nothing + /// seeds its counter from the reservation, its chain never gets re-anchored, + /// and an ack there means a quorum journaled the batch. + #[compio::test] + async fn given_a_replicated_group_when_fencing_should_not_write_at_all() { + let store = Rc::new(RecordingSuperblock::default()); + let mut partition = partition_at_view(1, 1); + assert!(partition.consensus().replica_count() > 1); + partition.set_superblock(store.clone(), None); + partition.set_offset_reservation_lease(test_lease(16)); + store.fail_writes.set(true); + + assert!( + partition.reserve_offsets_through(1_000).await, + "the fence is inert above one replica, so not even a failing store can \ + refuse the append" + ); + assert_eq!(store.attempts.get(), 0, "and it attempts no write"); } - // Exclude the active (last) segment, take the leading sealed run, then the - // `count`-th of those (or the last available when fewer exist). - let last_idx = segments.len().saturating_sub(1); - segments - .iter() - .take(last_idx) - .take_while(|segment| segment.sealed) - .take(count as usize) - .map(|segment| segment.end_offset) - .last() -} -#[cfg(test)] -mod tests { - use super::*; - use crate::iggy_index::{IGGY_INDEX_SIZE, IggyIndex, IggyIndexCache}; - use crate::iggy_index_reader::IggyIndexReader; - use crate::poll_plan::{DiskReadOutcome, SealedSegmentHandle}; - use bytes::Bytes; - use compio::io::AsyncWriteAtExt; - use consensus::LocalPipeline; - use iggy_binary_protocol::{Command, ReplyHeader, WireConsumer, WireEncode}; - use message_bus::{BusMessage, SendError}; - use server_common::MESSAGE_ALIGN; - use server_common::send_messages::{ - COMMAND_HEADER_SIZE, IggyMessage, IggyMessageHeader, IggyMessages, SendMessagesOwned, - }; - use std::cell::RefCell; - use std::rc::Rc; + /// The reservation seeds the APPEND counter and only on a solo group. The + /// whole fix: after a crash below the flush thresholds it is the only witness + /// that those offsets were confirmed. + #[test] + fn given_a_recorded_reservation_when_solo_should_seed_the_append_counter() { + let mut partition = solo_recording_partition(); + assert_eq!( + partition.consensus().replica_count(), + 1, + "the reservation seed is scoped to solo groups" + ); + let store = Rc::new(RecordingSuperblock::default()); - const TEST_CLUSTER: u128 = 1; + partition.set_superblock(store.clone(), None); + partition.restore_offset_frontier(None); + assert_eq!( + partition.mint_frontier(), + 0, + "nothing recorded, nothing seeded" + ); - pub(super) fn test_partition() -> IggyPartition { - let namespace = IggyNamespace::new(1, 1, 0); - let consensus = VsrConsensus::new( - TEST_CLUSTER, + // The shape a crash before the first flush leaves: the fence recorded a + // block, and no message ever reached a segment, so the frontier is 0. + let recovered = recorded_state(0, 65_537); + partition.set_superblock(store, Some(&recovered)); + partition.restore_offset_frontier(Some(&recovered)); + assert_eq!( + partition.mint_frontier(), + 65_537, + "the first mint must land above every offset the reservation covered" + ); + assert_eq!( + partition.offset_frontier(), 0, - 1, - namespace.inner(), - IggyMessageBus::new(0), - LocalPipeline::new(), + "the committed frontier must not inherit the reservation, nor the \ + append counter's own liveness: nothing was flushed, so the partition \ + holds no offset at all" ); - consensus.init(); - IggyPartition::with_in_memory_storage( - Arc::new(PartitionStats::default()), - consensus, - IggyByteSize::from(1024 * 1024), - false, - ) } - /// Partition whose consensus already advanced to `(view, log_view)` with - /// nothing marked durable, as after a view change and before the persist - /// gate runs. - fn partition_at_view( - view: u32, - log_view: u32, - ) -> IggyPartition { - let namespace = IggyNamespace::new(1, 1, 0); - let mut consensus = VsrConsensus::new( - TEST_CLUSTER, + /// A replicated group must not take the jump: a backup rejects any prepare + /// whose `base_offset` does not continue its own counter, so an append point a + /// lease block above the group has every peer refuse the batch. + #[test] + fn given_a_recorded_reservation_when_replicated_should_not_seed_the_append_counter() { + let mut partition = partition_at_view(1, 1); + assert!(partition.consensus().replica_count() > 1); + let recovered = recorded_state(0, 70_000); + partition.set_superblock(Rc::new(RecordingSuperblock::default()), Some(&recovered)); + partition.restore_offset_frontier(Some(&recovered)); + assert_eq!( + partition.mint_frontier(), 0, - 3, - namespace.inner(), - IggyMessageBus::new(0), - LocalPipeline::new(), + "a replicated group's offsets are the group's to decide" ); - consensus.set_view(view); - consensus.set_log_view(log_view); - consensus.init_as_backup(); - IggyPartition::with_in_memory_storage( - Arc::new(PartitionStats::default()), - consensus, - IggyByteSize::from(1024 * 1024), - false, - ) } - /// In-memory superblock double: records every payload, counts attempts, - /// and injects write failures. - #[derive(Default)] - struct RecordingSuperblock { - writes: RefCell>>, - attempts: Cell, - fail_writes: Cell, + /// The committed frontier still seeds both counters: it is a claim about data + /// every replica shares, so it is not gated on the replica count. + #[test] + fn given_a_recorded_frontier_when_replicated_should_seed_both_counters() { + let mut partition = partition_at_view(1, 1); + assert!(partition.consensus().replica_count() > 1); + let recovered = recorded_state(40, 70_000); + partition.set_superblock(Rc::new(RecordingSuperblock::default()), Some(&recovered)); + partition.restore_offset_frontier(Some(&recovered)); + assert_eq!(partition.offset_frontier(), 40); + assert_eq!( + partition.mint_frontier(), + 40, + "the reservation is ignored here, so the append point is the frontier" + ); } - impl journal::superblock::SuperblockStore for RecordingSuperblock { - async fn write(&self, payload: &[u8]) -> std::io::Result<()> { - self.attempts.set(self.attempts.get() + 1); - if self.fail_writes.get() { - return Err(std::io::Error::other("injected superblock write failure")); - } - self.writes.borrow_mut().push(payload.to_vec()); - Ok(()) - } + /// The rewind fence asks whether an offer would destroy something this replica + /// COMMITTED. An append point standing a lease block above that -- what a + /// reservation-seeded boot leaves until the first append -- claims no + /// messages, so it must not turn a legitimate offer inside the block into a + /// refusal the replica would then cycle on forever. + #[compio::test] + async fn given_an_append_point_above_every_message_when_an_offer_arrives_should_not_call_it_a_rewind() + { + let partition_dir = transfer_fence_dir("append-point-is-not-data").await; + let mut partition = test_partition(); + partition.set_partition_dir(partition_dir.clone()); + partition.set_offset_space_used(true); + partition.dirty_offset.store(69_999, Ordering::Relaxed); + assert_eq!(partition.mint_frontier(), 70_000); + assert_eq!( + partition.offset_frontier(), + 1, + "nothing committed: the append point speaks for no messages" + ); - async fn read_latest(&self) -> std::io::Result { - Ok(self - .writes - .borrow() - .last() - .map_or(journal::superblock::SuperblockContents::Empty, |bytes| { - journal::superblock::SuperblockContents::Present(bytes.clone()) - })) - } + let offer = crate::state_transfer::ConsumerOffsetsWire { + purge_generation: 0, + next_offset: 1_030, + consumers: Vec::new(), + groups: Vec::new(), + dedup: Vec::new(), + }; + let outcome = partition + .install_state_transfer(&repair_config(), 12, Vec::new(), &offer.encode(), 0) + .await; + assert!( + !matches!( + outcome, + Err(crate::state_transfer::PartitionInstallError::OfferRewindsDurableData { .. }) + ), + "the offer destroys nothing this replica holds, so the fence must let it \ + through: got {outcome:?}" + ); + + let _ = std::fs::remove_dir_all(&partition_dir); } + /// After a clean shutdown the segments account for every confirmed offset, + /// so collapsing the reservation keeps the offset space dense across an + /// ordinary restart instead of jumping a lease block every time. #[compio::test] - async fn given_storeless_partition_when_persist_gate_runs_should_mark_current_view_durable() { - let partition = partition_at_view(2, 1); - assert!(partition.consensus().needs_superblock_persist()); - - assert!(partition.persist_superblock_if_needed().await); + async fn given_a_flushed_partition_when_collapsing_should_drop_the_reservation_to_the_frontier() + { + let store = Rc::new(RecordingSuperblock::default()); + let mut partition = solo_recording_partition(); + partition.set_superblock(store.clone(), Some(&recorded_state(0, 65_537))); + // A flushed partition: the append point and the committed head agree. + partition.set_offset_space_used(true); + partition.offset.store(24, Ordering::Release); + partition.dirty_offset.store(24, Ordering::Relaxed); + + assert!(partition.collapse_offset_reservation().await); + assert_eq!(last_recorded_frontier(&store), 25); + assert_eq!( + last_recorded_reservation(&store), + 25, + "a clean stop leaves no claim above what the segments prove" + ); - assert!( - !partition.consensus().needs_superblock_persist(), - "a storeless partition must record durable = current, or the dispatch \ - tripwire would fire on its first view-scoped send" + // And the restart that follows mints where it left off, not a block up. + let recovered = recorded_state(25, 25); + let mut restarted = solo_recording_partition(); + restarted.set_superblock(store.clone(), Some(&recovered)); + restarted.restore_offset_frontier(Some(&recovered)); + assert_eq!( + restarted.mint_frontier(), + 25, + "the collapsed record puts the append point back at the frontier" ); } + /// The graceful stop must not undo the crash protection. A boot that read a + /// reservation back and then took no traffic has a committed frontier of 0 + /// while the reservation is the only record that offsets were confirmed, so a + /// collapse reading the committed frontier would write it away -- and a clean + /// stop is the runbook answer to an incident, which would make it the one + /// action that re-opens the defect. #[compio::test] - async fn given_advanced_view_when_persist_gate_runs_should_write_vsr_state_once() { - let mut partition = partition_at_view(3, 2); + async fn given_an_unspent_reservation_when_collapsing_should_leave_it_standing() { let store = Rc::new(RecordingSuperblock::default()); - partition.set_superblock(store.clone(), None); + let recovered = recorded_state(0, 65_537); + let mut partition = solo_recording_partition(); + partition.set_superblock(store.clone(), Some(&recovered)); + partition.restore_offset_frontier(Some(&recovered)); - assert!(partition.persist_superblock_if_needed().await); + assert!(partition.collapse_offset_reservation().await); + assert_eq!( + store.attempts.get(), + 0, + "the reservation already covers the append point, so there is nothing to \ + collapse and nothing to write" + ); - let state = consensus::VsrState::try_from(store.writes.borrow()[0].as_slice()) - .expect("recorded payload decodes as a VsrState"); - assert_eq!(state.cluster, TEST_CLUSTER); - assert_eq!(state.view, 3); - assert_eq!(state.log_view, 2); + // The restart after the clean stop still resumes above every offset the + // crashed incarnation confirmed. + let mut restarted = solo_recording_partition(); + restarted.set_superblock(store, Some(&recovered)); + restarted.restore_offset_frontier(Some(&recovered)); + assert_eq!(restarted.mint_frontier(), 65_537); + } + + /// An install is the one place the reservation may come down: left high, it + /// re-seeds the counter above the group and every replicated prepare fails + /// the `base_offset == dirty_offset + 1` check. + #[compio::test] + async fn given_install_frontier_when_recorded_should_set_the_reservation_down_to_it() { + let store = Rc::new(RecordingSuperblock::default()); + let mut partition = partition_at_view(1, 1); + partition.set_superblock(store.clone(), Some(&recorded_state(0, 70_000))); + + assert!(partition.install_offset_frontier_at(1_030).await); assert_eq!( - (state.checkpoint_op, state.checkpoint_checksum), - (0, 0), - "no partition checkpoint exists yet, so the pairing fields stay zero" + last_recorded_reservation(&store), + 1_030, + "the install's frontier replaces the stale reservation" ); - assert!(!partition.consensus().needs_superblock_persist()); + assert_eq!(last_recorded_frontier(&store), 1_030); + } - assert!(partition.persist_superblock_if_needed().await); + /// A purge resets the offset space to zero and the reservation goes with it: + /// a survivor would re-seed the counter into the space just erased. + #[compio::test] + async fn given_purge_reset_when_recorded_should_clear_the_reservation() { + let store = Rc::new(RecordingSuperblock::default()); + let mut partition = partition_at_view(1, 1); + partition.set_superblock(store.clone(), Some(&recorded_state(500, 70_000))); + + assert!(partition.reset_offset_frontier_at(0).await); + assert_eq!(last_recorded_frontier(&store), 0); assert_eq!( - store.attempts.get(), - 1, - "an unchanged view must take the lock-free fast path, not rewrite" + last_recorded_reservation(&store), + 0, + "a reset that left the reservation behind would resurrect the old space" ); } - /// The `offset_frontier` of the most recent recorded write. - fn last_recorded_frontier(store: &RecordingSuperblock) -> u64 { - let writes = store.writes.borrow(); - let bytes = writes.last().expect("a superblock write landed"); - consensus::VsrState::try_from(bytes.as_slice()) - .expect("recorded payload decodes as a VsrState") - .offset_frontier + /// The reservation can never sit below the frontier: "offsets under N exist" + /// is stronger than "offsets under N may have been handed out". + #[compio::test] + async fn given_reservation_below_the_frontier_when_written_should_clamp_it_up() { + let store = Rc::new(RecordingSuperblock::default()); + let mut partition = partition_at_view(1, 1); + partition.set_superblock(store.clone(), None); + partition.set_offset_space_used(true); + partition.offset.store(99, Ordering::Release); + + assert!(partition.persist_offset_frontier_at(100).await); + assert_eq!(last_recorded_frontier(&store), 100); + assert_eq!( + last_recorded_reservation(&store), + 100, + "the reservation is clamped up to the frontier it accompanies" + ); } /// The fence path persists the frontier while the live counter still sits @@ -5563,6 +7310,7 @@ mod tests { checkpoint_op: 0, checkpoint_checksum: 0, offset_frontier: 4_200, + offset_reserved: 0, }; partition.set_superblock(store.clone(), Some(&recovered)); assert_eq!(partition.offset_frontier(), 0, "nothing minted locally"); @@ -5585,7 +7333,7 @@ mod tests { let store = Rc::new(RecordingSuperblock::default()); partition.set_superblock(store.clone(), None); partition.offset.store(9_000, Ordering::Release); - partition.should_increment_offset = true; + partition.set_offset_space_used(true); assert!(partition.persist_offset_frontier().await); assert_eq!(last_recorded_frontier(&store), 9_001); @@ -5609,7 +7357,7 @@ mod tests { let store = Rc::new(RecordingSuperblock::default()); partition.set_superblock(store.clone(), None); partition.offset.store(9_000, Ordering::Release); - partition.should_increment_offset = true; + partition.set_offset_space_used(true); store.fail_writes.set(true); assert!( @@ -5820,6 +7568,44 @@ mod tests { (partition, sent_to_clients) } + /// A `SendMessages` request in the shape `convert_request_message` leaves at + /// [`ChecksumMode::Skip`]: canonical batch, `batch_checksum` zeroed. + fn checksumless_send_request( + namespace: IggyNamespace, + message_count: u32, + ) -> Message { + let mut batch = IggyMessages::with_capacity(message_count as usize); + for _ in 0..message_count { + batch.push(IggyMessage { + header: IggyMessageHeader { + payload_length: 8, + ..Default::default() + }, + payload: Bytes::from_static(b"abcdefgh"), + user_headers: None, + }); + } + let mut owned = + SendMessagesOwned::from_messages(namespace, &batch).expect("build send_messages batch"); + owned.header.batch_checksum = 0; + + let header_size = std::mem::size_of::(); + let total = header_size + COMMAND_HEADER_SIZE + owned.blob.len(); + let mut message = Message::::new(total); + let body = &mut message.as_mut_slice()[header_size..]; + owned.header.encode_into(&mut body[..COMMAND_HEADER_SIZE]); + body[COMMAND_HEADER_SIZE..].copy_from_slice(&owned.blob); + message.transmute_header(|_, header: &mut RoutedRequestHeader| { + header.command = Command::Request; + header.operation = Operation::SendMessages; + header.client = 1; + header.session = 1; + header.request = 1; + header.group = namespace.inner(); + header.size = u32::try_from(total).expect("request size fits u32"); + }) + } + fn delete_offset_request( client_id: u128, request_id: u64, @@ -7422,8 +9208,15 @@ mod tests { let partition_dir = transfer_fence_dir("rewind-refused").await; let mut partition = test_partition(); partition.set_partition_dir(partition_dir.clone()); - partition.should_increment_offset = true; + partition.set_offset_space_used(true); partition.offset.store(99, Ordering::Release); + // Committed and resident: the threshold-gated flush leaves exactly this + // shape, and the fence has to count it as data. + { + let info = &mut partition.log.journal_mut().info; + info.messages_count = 100; + info.current_offset = 99; + } let behind = crate::state_transfer::ConsumerOffsetsWire { purge_generation: 0, @@ -7474,6 +9267,53 @@ mod tests { let _ = std::fs::remove_dir_all(&partition_dir); } + /// A chain installed EMPTY at frontier N holds no sized segment and no + /// journal entry, so a fence reading only held bytes reads 0 and skips + /// itself, letting a stale offer rewind the counter under offsets this + /// replica already claimed. The committed frontier is what carries N. + #[compio::test] + async fn given_an_empty_chain_installed_at_a_frontier_when_a_stale_offer_arrives_should_refuse() + { + let partition_dir = transfer_fence_dir("empty-install-rewind").await; + let mut partition = test_partition(); + partition.set_partition_dir(partition_dir.clone()); + // What an install of an all-GC'd origin leaves: the counter at the group + // frontier, nothing on disk, nothing resident. + partition.set_offset_space_used(true); + partition.offset.store(4_095, Ordering::Release); + partition.dirty_offset.store(4_095, Ordering::Relaxed); + assert_eq!( + partition.held_offset_frontier(), + 4_096, + "the committed arm has to carry a frontier no byte on disk names" + ); + + let stale = crate::state_transfer::ConsumerOffsetsWire { + purge_generation: 0, + next_offset: 1_000, + consumers: Vec::new(), + groups: Vec::new(), + dedup: Vec::new(), + }; + let refused = partition + .install_state_transfer(&repair_config(), 12, Vec::new(), &stale.encode(), 0) + .await; + assert!( + matches!( + refused, + Err( + crate::state_transfer::PartitionInstallError::OfferRewindsDurableData { + offer_next_offset: 1_000, + local_next_offset: 4_096, + } + ) + ), + "expected a rewind refusal, got {refused:?}" + ); + + let _ = std::fs::remove_dir_all(&partition_dir); + } + /// The canonical post-restart rejoin: this replica applied a purge before /// the restart but was killed before the purge's `purge.gen` record step, /// so the metadata plane's COMMITTED generation is 1 while its own @@ -7485,8 +9325,15 @@ mod tests { let partition_dir = transfer_fence_dir("restart-purge-rewind").await; let mut partition = test_partition(); partition.set_partition_dir(partition_dir.clone()); - partition.should_increment_offset = true; + partition.set_offset_space_used(true); partition.offset.store(99, Ordering::Release); + // Committed and resident: the threshold-gated flush leaves exactly this + // shape, and the fence has to count it as data. + { + let info = &mut partition.log.journal_mut().info; + info.messages_count = 100; + info.current_offset = 99; + } assert_eq!( partition.applied_purge_generation(), 0, @@ -7535,7 +9382,7 @@ mod tests { let partition_dir = transfer_fence_dir("missed-purge-reset").await; let mut partition = test_partition(); partition.set_partition_dir(partition_dir.clone()); - partition.should_increment_offset = true; + partition.set_offset_space_used(true); partition.offset.store(99, Ordering::Release); assert_eq!( partition.applied_purge_generation(), diff --git a/core/partitions/src/lib.rs b/core/partitions/src/lib.rs index 818872029d..013bc6de6a 100644 --- a/core/partitions/src/lib.rs +++ b/core/partitions/src/lib.rs @@ -28,6 +28,7 @@ mod messages_writer; pub mod offset_storage; mod poll_plan; mod segment; +pub mod segment_anchor; pub mod state_transfer; mod types; @@ -39,6 +40,20 @@ pub use iggy_index_writer::IggyIndexWriter; pub use iggy_partition::{IggyPartition, PurgeError, SegmentRemoval}; pub use iggy_partitions::IggyPartitions; pub use journal::{EVICTED_RING_BYTES_MAX, EVICTED_RING_CAPACITY}; + +/// Offsets a partition claims in its superblock ahead of the mint counter +/// before it will append, so a crash-restarted replica resumes above every +/// offset it confirmed instead of re-minting it. +/// +/// One superblock write (two fsyncs) per block: at 100k messages/s a 1Ki block +/// costs ~200 fsyncs/s, 64Ki costs ~3/s. The waste is at most one block of a +/// `u64` space per crash, visible only as a segment boundary at boot. +/// +/// Lives HERE and not in `iggy_common`: it is a server-side write-path default +/// that no client ever reads, and the shared crate is the client-facing API. +/// Both consumers -- the fallback in [`IggyPartition`] and the `[partition]` +/// config default boot installs -- already depend on this crate. +pub const DEFAULT_OFFSET_RESERVATION_LEASE: u32 = 64 * 1024; pub use messages_writer::MessagesWriter; pub use offset_storage::delete_persisted_offset; pub use poll_plan::{AutoCommitApplied, PollPlan}; diff --git a/core/partitions/src/log.rs b/core/partitions/src/log.rs index 92d9c1768d..64ae2f5209 100644 --- a/core/partitions/src/log.rs +++ b/core/partitions/src/log.rs @@ -192,8 +192,9 @@ where } /// Mutable segment views. Length mutation lives in - /// [`Self::add_persisted_segment`] / [`Self::retire_front`] only, so the - /// parallel vecs cannot desync from the outside. + /// [`Self::add_persisted_segment`] / [`Self::retire_front`] / + /// [`Self::retire_back`] only, so the parallel vecs cannot desync from the + /// outside. pub fn segments_mut(&mut self) -> &mut [Segment] { &mut self.segments } @@ -314,6 +315,28 @@ where Some((segment, storage)) } + /// Retire the NEWEST segment, the mirror of [`Self::retire_front`]. + /// + /// Boot re-anchor only, and only for an EMPTY tail named below the re-seeded + /// counter, which would otherwise claim a range it does not hold. Nothing + /// else may take from the back: a sized tail is the only copy of its + /// messages. + pub fn retire_back(&mut self) -> Option<(Segment, SegmentStorage)> { + if self.segments.is_empty() { + return None; + } + self.debug_assert_lockstep(); + let segment = self.segments.pop()?; + let storage = self.storage.pop()?; + self.indexes.pop(); + self.messages_writers.pop(); + self.index_writers.pop(); + self.sealed_read_state.pop(); + self.sealed_lru + .retain(|&offset| offset != segment.start_offset); + Some((segment, storage)) + } + fn debug_assert_lockstep(&self) { debug_assert!( self.segments.len() == self.storage.len() diff --git a/core/partitions/src/segment_anchor.rs b/core/partitions/src/segment_anchor.rs new file mode 100644 index 0000000000..1891ce5fe7 --- /dev/null +++ b/core/partitions/src/segment_anchor.rs @@ -0,0 +1,329 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! The record that makes a gap in the segment chain legitimate. +//! +//! Recovery derives each segment's bounds from its own bytes, so a gap has no +//! author: the boot re-anchor's planted gap and a lost segment look identical. +//! The re-anchor writes its intent down instead, and the chain guard admits a +//! forward gap only when the far side carries an anchor naming exactly the near +//! side. +//! +//! Written and directory-fsynced BEFORE that segment is created. A crash in the +//! window then leaves an anchor with no segment, which the boot sweep collects; +//! the other order leaves a planted segment with no anchor, which the guard +//! reads as damage on an intact chain. + +use crate::state_transfer::STAGING_SUFFIX; +use compio::io::AsyncWriteAtExt; +use consensus::state_artifact_checksum; +use std::io; + +/// File extension for an anchor record, `{start_offset:020}.anchor` beside the +/// `{start_offset:020}.log` it belongs to. +pub const ANCHOR_EXTENSION: &str = "anchor"; + +/// [`ANCHOR_EXTENSION`] as a filename suffix, for the directory sweeps that +/// match on one. +pub const ANCHOR_SUFFIX: &str = ".anchor"; + +/// Leading bytes of an anchor record, so a file that is not one (a truncated +/// write, an operator's copy) is refused rather than decoded. +const ANCHOR_MAGIC: u64 = u64::from_le_bytes(*b"IGGYANCH"); + +/// `magic`(8) + `planted_start`(8) + `sealed_start`(8) + `sealed_end`(8) + +/// `checksum`(8). +pub const ANCHOR_ENCODED_LEN: usize = 40; + +/// The gap one planted segment is allowed to leave behind it. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct SegmentAnchor { + /// Start offset of the segment this record sits beside, i.e. the FAR side of + /// the gap. + /// + /// Redundant with the file name and deliberately so: the name is not + /// checksummed, so without this field the payload authenticates the + /// predecessor bounds while saying nothing about which plant it authorises. + /// Valid anchor bytes copied beside a later segment would then cover a wider + /// gap after the same predecessor -- exactly the operator's copy this module + /// claims to refuse. + pub planted_start: u64, + /// Start offset of the segment that was sealed, i.e. the near side of the + /// gap. Names WHICH segment, so an anchor cannot be satisfied by a + /// different file that happens to end where this one expects. + pub sealed_start: u64, + /// End offset the sealed segment held when it was sealed. The gap runs from + /// here to the planted segment's own start offset. + pub sealed_end: u64, +} + +impl SegmentAnchor { + /// Encode to the fixed little-endian on-disk layout. + #[must_use] + pub fn to_bytes(&self) -> [u8; ANCHOR_ENCODED_LEN] { + let mut out = [0u8; ANCHOR_ENCODED_LEN]; + out[0..8].copy_from_slice(&ANCHOR_MAGIC.to_le_bytes()); + out[8..16].copy_from_slice(&self.planted_start.to_le_bytes()); + out[16..24].copy_from_slice(&self.sealed_start.to_le_bytes()); + out[24..32].copy_from_slice(&self.sealed_end.to_le_bytes()); + let checksum = state_artifact_checksum(&out[0..32]); + out[32..40].copy_from_slice(&checksum.to_le_bytes()); + out + } + + /// Decode a record, returning `None` for anything this build did not write: + /// a wrong length, a wrong magic, or a checksum that does not match. + /// + /// A `None` is never treated as "no gap was intended". It means the record + /// proves nothing, so the gap it would have covered stays damage. + #[must_use] + pub fn from_bytes(bytes: &[u8]) -> Option { + if bytes.len() != ANCHOR_ENCODED_LEN { + return None; + } + let field = |at: usize| -> u64 { + let mut raw = [0u8; 8]; + raw.copy_from_slice(&bytes[at..at + 8]); + u64::from_le_bytes(raw) + }; + if field(0) != ANCHOR_MAGIC || field(32) != state_artifact_checksum(&bytes[0..32]) { + return None; + } + Some(Self { + planted_start: field(8), + sealed_start: field(16), + sealed_end: field(24), + }) + } + + /// Whether this anchor legitimises the gap between the segment starting at + /// `sealed_start` / ending at `sealed_end` and the segment planted at + /// `planted_start`. + /// + /// All THREE bounds must match. The predecessor pair alone would let an + /// anchor left by an earlier incarnation of the chain cover a gap it never + /// saw; the plant alone would let any predecessor satisfy it. Binding the + /// plant is what stops valid bytes from being copied beside a later segment + /// to authorise a wider gap after the same predecessor. + #[must_use] + pub const fn covers(&self, planted_start: u64, sealed_start: u64, sealed_end: u64) -> bool { + self.planted_start == planted_start + && self.sealed_start == sealed_start + && self.sealed_end == sealed_end + } +} + +/// Path of the anchor record beside the segment starting at `start_offset`. +#[must_use] +pub fn anchor_path(partition_dir: &str, start_offset: u64) -> String { + format!("{partition_dir}/{start_offset:0>20}.{ANCHOR_EXTENSION}") +} + +/// Write the anchor for a segment about to be planted at `start_offset`, then +/// fsync the directory so the record cannot arrive after the segment it +/// describes. +/// +/// # Errors +/// +/// Any I/O failure. The caller must NOT plant the segment: a planted segment +/// whose anchor is missing reads as damage on the next boot. +pub async fn write_anchor(partition_dir: &str, anchor: SegmentAnchor) -> io::Result<()> { + // Temp, fsync, rename, fsync the dir, like the superblock in this same + // directory. There is no second slot to fall back on, so a truncating + // in-place write would leave a torn record where the guard needs either the + // old one or the new one. + // + // The path comes from the record's own `planted_start` rather than a second + // parameter: the guard matches the two for equality, so a caller that could + // pass them separately could write a record that never satisfies anything. + let path = anchor_path(partition_dir, anchor.planted_start); + // `STAGING_SUFFIX` rather than a suffix of its own: every sweep already + // unlinks it unconditionally, boot included, so a torn write leaves nothing + // a later guard can read. + let tmp_path = format!("{path}{STAGING_SUFFIX}"); + let mut file = compio::fs::File::create(&tmp_path).await?; + let (result, _buf) = file + .write_all_at(anchor.to_bytes().to_vec(), 0) + .await + .into(); + result?; + file.sync_all().await?; + compio::fs::rename(&tmp_path, &path).await?; + crate::state_transfer::fsync_dir(partition_dir).await +} + +/// Read the anchor beside the segment starting at `start_offset`. +/// +/// `Ok(None)` when the file is absent, is not exactly [`ANCHOR_ENCODED_LEN`] +/// bytes, or does not decode; all of them mean the same thing to the guard, so +/// the caller needs no distinction. +/// +/// The length is checked from the metadata BEFORE any read, so a corrupt or +/// foreign file left at this path cannot size an allocation on the boot path. +/// `from_bytes` would refuse it either way, but only after reading all of it. +/// +/// # Errors +/// +/// Any other stat or read failure. NOT folded into `Ok(None)`: an `EACCES` or +/// `EIO` over a healthy planted chain would read as no gap intended, refusing +/// it as damage for as long as the fault lasts. +pub async fn read_anchor( + partition_dir: &str, + start_offset: u64, +) -> io::Result> { + let path = anchor_path(partition_dir, start_offset); + let length = match compio::fs::metadata(&path).await { + Ok(metadata) => metadata.len(), + Err(error) if error.kind() == io::ErrorKind::NotFound => return Ok(None), + Err(error) => return Err(error), + }; + if length != ANCHOR_ENCODED_LEN as u64 { + return Ok(None); + } + match compio::fs::read(&path).await { + Ok(bytes) => Ok(SegmentAnchor::from_bytes(&bytes)), + Err(error) if error.kind() == io::ErrorKind::NotFound => Ok(None), + Err(error) => Err(error), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn anchor() -> SegmentAnchor { + SegmentAnchor { + planted_start: 8_192, + sealed_start: 7, + sealed_end: 4_095, + } + } + + #[test] + fn given_an_anchor_when_round_tripped_should_decode_identically() { + let bytes = anchor().to_bytes(); + assert_eq!(bytes.len(), ANCHOR_ENCODED_LEN); + assert_eq!(SegmentAnchor::from_bytes(&bytes), Some(anchor())); + } + + #[test] + fn given_a_flipped_bit_when_decoded_should_refuse() { + // Every byte the checksum covers, so a corrupted record can never read as + // a legitimate gap. + for index in 0..32 { + let mut bytes = anchor().to_bytes(); + bytes[index] ^= 1; + assert_eq!( + SegmentAnchor::from_bytes(&bytes), + None, + "a record corrupted at byte {index} must not decode" + ); + } + } + + #[test] + fn given_a_wrong_length_when_decoded_should_refuse() { + let bytes = anchor().to_bytes(); + assert_eq!(SegmentAnchor::from_bytes(&bytes[..39]), None); + assert_eq!(SegmentAnchor::from_bytes(&[]), None); + } + + #[test] + fn given_an_anchor_when_matching_a_different_predecessor_should_not_cover_it() { + let anchor = SegmentAnchor { + planted_start: 30, + sealed_start: 10, + sealed_end: 20, + }; + assert!(anchor.covers(30, 10, 20)); + assert!(!anchor.covers(30, 10, 21), "a different end must not match"); + assert!( + !anchor.covers(30, 0, 20), + "the same end under a different segment must not match: an anchor left \ + by an earlier chain would otherwise cover a gap it never saw" + ); + } + + /// The copy this module claims to refuse: valid, checksum-clean bytes moved + /// beside a LATER segment. Without the plant in the payload the record still + /// authenticates, still names the same predecessor, and authorises a gap that + /// is now arbitrarily wide. + #[test] + fn given_valid_anchor_bytes_copied_beside_a_later_segment_should_not_cover_the_wider_gap() { + let planted = SegmentAnchor { + planted_start: 100, + sealed_start: 0, + sealed_end: 99, + }; + let decoded = SegmentAnchor::from_bytes(&planted.to_bytes()) + .expect("the copied bytes are checksum-clean, which is the premise"); + + assert!( + decoded.covers(100, 0, 99), + "beside its own segment it holds" + ); + assert!( + !decoded.covers(5_000, 0, 99), + "the same bytes beside a segment planted at 5000 must not authorise \ + the gap 100..5000 after the same predecessor" + ); + } + + /// A foreign or corrupt file at the anchor path must be refused from its + /// metadata, not by reading however many bytes it happens to hold. + #[compio::test] + async fn given_an_oversized_file_at_the_anchor_path_when_read_should_refuse_it() { + let dir = tempfile::tempdir().expect("tempdir"); + let partition_dir = dir.path().to_str().expect("utf-8 tempdir"); + let path = anchor_path(partition_dir, 42); + + std::fs::write(&path, vec![0u8; 1 << 20]).expect("plant an oversized file"); + assert_eq!( + read_anchor(partition_dir, 42) + .await + .expect("an oversized file is refused, not an error"), + None + ); + + std::fs::write(&path, anchor().to_bytes()).expect("plant a real record"); + assert_eq!( + read_anchor(partition_dir, 42) + .await + .expect("a well-formed record reads back"), + Some(anchor()) + ); + } + + /// The path is derived from the record, so a write always lands where the + /// guard will look for it. + #[compio::test] + async fn given_an_anchor_when_written_should_land_at_its_own_planted_start() { + let dir = tempfile::tempdir().expect("tempdir"); + let partition_dir = dir.path().to_str().expect("utf-8 tempdir"); + + write_anchor(partition_dir, anchor()) + .await + .expect("write the anchor"); + + assert_eq!( + read_anchor(partition_dir, anchor().planted_start) + .await + .expect("read it back"), + Some(anchor()) + ); + } +} diff --git a/core/partitions/src/state_transfer.rs b/core/partitions/src/state_transfer.rs index 7444b9d76b..1872d52165 100644 --- a/core/partitions/src/state_transfer.rs +++ b/core/partitions/src/state_transfer.rs @@ -32,6 +32,7 @@ use crate::offset_storage::{ PURGE_GENERATION_FILE, delete_persisted_offset, persist_offset, persist_purge_generation, }; use crate::segment::Segment; +use crate::segment_anchor::ANCHOR_SUFFIX; use crate::types::PartitionsConfig; use crate::{IggyIndexWriter, IggyPartition}; use compio::io::{AsyncReadAtExt, AsyncWriteAtExt}; @@ -1487,7 +1488,7 @@ pub async fn quarantine_segment_files(partition_dir: &str) -> std::io::Result std::io::Result committed_purge_generation || (self.applied_purge_generation < committed_purge_generation && offsets_wire.next_offset == 0); + // The COMMITTED frontier, which is what an offer is comparable against: + // `held_offset_frontier` reads 0 for a chain installed empty at frontier + // N (its disk arm filters empty segments and the install clears the + // journal), and a 0 skips the guard below entirely, letting a stale offer + // rewind the counter under data this replica already claimed. The append + // point is not usable either -- it can stand a lease block high -- but + // only on a solo group, which never receives an offer. let local_next_offset = self.offset_frontier(); if !purge_advances && local_next_offset > 0 && offsets_wire.next_offset < local_next_offset { @@ -2295,11 +2303,16 @@ where // zero `.log` files and re-seed the counter from the pre-purge frontier, // above a group that restarted at the offer's, and the next prepare // would stamp a `base_offset` and `batch_checksum` no peer shares. + // + // Identical to the advancing form on every group that can receive an + // offer today, since the reservation is solo-only and there equals the + // frontier. Spelled out because the shape is what makes it a reset, not + // the arithmetic that currently coincides. let frontier_durable = if purge_advances { self.reset_offset_frontier_at(offsets_wire.next_offset) .await } else { - self.persist_offset_frontier_at(offsets_wire.next_offset) + self.install_offset_frontier_at(offsets_wire.next_offset) .await }; if !frontier_durable { @@ -2411,11 +2424,19 @@ where // the NEWEST suffix, which is contiguous) and drop the in-memory // vectors in lockstep, exactly as `purge` does. let namespace_raw = self.consensus().group(); - while let Some((_, mut storage)) = self.log.retire_front() { + while let Some((segment, mut storage)) = self.log.retire_front() { let (messages_path, index_path) = storage.segment_and_index_paths(); let _ = storage.shutdown(); drop(storage); - for path in messages_path.into_iter().chain(index_path) { + // Anchors go with the chain they describe, as everywhere else. + // Unreachable for an install today, since anchors are planted only + // by the solo boot re-anchor, but a record outliving its segment is + // the one way a later gap gets admitted for free. + for path in messages_path + .into_iter() + .chain(index_path) + .chain(self.anchor_cleanup_path(segment.start_offset)) + { match compio::fs::remove_file(&path).await { Ok(()) => {} Err(error) if error.kind() == std::io::ErrorKind::NotFound => {} @@ -2805,7 +2826,7 @@ where let end = next_offset.saturating_sub(1); self.offset.store(end, Ordering::Release); self.dirty_offset.store(end, Ordering::Relaxed); - self.should_increment_offset = next_offset > 0; + self.set_offset_space_used(next_offset > 0); self.recovered_durable_offset = installed_end; // Where the group's offset space starts on this replica: everything // below is represented by this install, so the repair floor check @@ -2959,7 +2980,7 @@ where .into_iter() .filter(|path| { path.to_str().is_some_and(|path| { - [".log", ".index", STAGING_SUFFIX] + [".log", ".index", STAGING_SUFFIX, ANCHOR_SUFFIX] .iter() .any(|extension| path.ends_with(extension)) }) @@ -3005,7 +3026,7 @@ where let end = minted_next_offset.saturating_sub(1); self.offset.store(end, Ordering::Release); self.dirty_offset.store(end, Ordering::Relaxed); - self.should_increment_offset = minted_next_offset > 0; + self.set_offset_space_used(minted_next_offset > 0); self.recovered_durable_offset = None; // The frontier claims "everything below me is represented here", and the // repair floor check accepts any floor at or below it. Nothing was diff --git a/core/server/config.toml b/core/server/config.toml index 5584b03788..3362651867 100644 --- a/core/server/config.toml +++ b/core/server/config.toml @@ -664,15 +664,25 @@ repair_retry_interval = "1s" # the queue and drops frames. Must be > 0 and <= 1024. repair_chunk_max = 128 -# How long the metadata superblock may stay unwritable before the replica -# fail-stops (duration). A replica that cannot persist its view is already -# fenced quorum-invisible and retries with capped backoff; past this window the -# process exits with a distinct status so a supervisor restarts or replaces it -# instead of an operator finding the wedge in logs. "0" disables the fail-stop -# and leaves the replica fenced indefinitely. Nonzero values must be at least -# 30s so a transient disk hiccup cannot kill the process. +# How long a superblock may stay unwritable before the replica fail-stops +# (duration). Applies to the metadata superblock and to each partition's own. A +# group that cannot persist its view is already fenced quorum-invisible and +# retries with capped backoff; past this window the process exits with a distinct +# status so a supervisor restarts or replaces it instead of an operator finding +# the wedge in logs. "0" disables the fail-stop and leaves the group fenced +# indefinitely. Nonzero values must be at least 30s so a transient disk hiccup +# cannot kill the process. superblock_wedged_fatal_timeout = "2m" +# DOWNGRADE, every deployment: this release grew the superblock record from 66 +# to 74 bytes to carry partition.offset_reservation_lease's claim. The metadata +# superblock is written by EVERY server, clustered or not, so upgrading rewrites +# it at the new length on a plain single node too. A build that predates the +# field refuses a record of that length and treats the refusal as a durability +# violation, which fails the whole node's boot -- so rolling BACK to +# server-0.9.0-edge.6 or earlier needs the data directory wiped on every node. +# Upgrading needs nothing: the 66-byte record still decodes. + # Replica-to-replica authentication (PSK + BLAKE3 keyed-MAC handshake). [cluster.auth] # When true, every replica peer must complete the authenticated handshake or be @@ -996,6 +1006,21 @@ prepare_queue_depth = 32 # actually sees, not the node's client total. dedup_clients_max = 4096 +# How many offsets a partition claims in its superblock ahead of the mint +# counter before it will append, so a crash-restarted replica resumes above +# every offset it confirmed to a client instead of re-minting it for a different +# message. One superblock write (two fsyncs) per block: lowering it raises the +# fsync rate on the write path, raising it wastes at most one block of the u64 +# offset space per crash. Must be > 0 and <= 16777216. +# +# SINGLE-REPLICA groups only. A replicated group acks a send once a quorum has +# journaled it, so re-minting there needs a full-cluster crash; it claims +# nothing, pays no write, and ignores this value. +# +# DOWNGRADE: see the note on superblock_wedged_fatal_timeout in [cluster]. +# Upgrading needs nothing. +offset_reservation_lease = 65536 + # Entries the evicted ring retains per multi-replica partition for journal # repair after a peer rejoins. Larger widens the window a restarting peer can be # served from the ring before falling back to bulk sync, at the cost of pinned diff --git a/core/server/src/boot/recovery.rs b/core/server/src/boot/recovery.rs index e614e688f5..eb99734d69 100644 --- a/core/server/src/boot/recovery.rs +++ b/core/server/src/boot/recovery.rs @@ -50,7 +50,7 @@ use std::cell::RefCell; use std::rc::Rc; use std::sync::Arc; use std::time::Duration; -use tracing::{info, warn}; +use tracing::{error, info, warn}; #[allow(clippy::too_many_arguments, clippy::too_many_lines)] pub(in crate::boot) async fn build_shard_for_thread( @@ -172,7 +172,7 @@ pub(in crate::boot) async fn build_shard_for_thread( // `Arc` atomics race only against other atomic adds. for (stream_id, topic_id, partition_stats, partition_metadata, topic_runtime) in owned { let namespace = IggyNamespace::new(stream_id, topic_id, partition_metadata.id); - let Some(partition) = load_partition_or_fence( + let loaded = load_partition_or_fence( config, namespace, partition_stats, @@ -184,9 +184,26 @@ pub(in crate::boot) async fn build_shard_for_thread( Rc::clone(&bus), &partitions, ) - .await? - else { - continue; + .await; + let partition = match loaded { + Ok(Some(partition)) => partition, + Ok(None) => continue, + // A refused claim is a failed superblock write, not damage: the + // namespace stays materialisable, so skipping it here costs one + // partition its start instead of the whole shard, and the + // reconciler's addition pass retries it within a tick. + Err(error @ ServerError::PartitionOffsetReservationClaim { .. }) => { + error!( + stream_id, + topic_id, + partition_id = partition_metadata.id, + %error, + "skipping this partition at boot; the reconciler retries its first \ + offset reservation claim" + ); + continue; + } + Err(error) => return Err(error), }; partitions.insert(namespace, partition); shards_table.insert( @@ -333,11 +350,9 @@ const _: () = assert!(consensus::DVC_HEADERS_MAX == iggy_binary_protocol::consensus::DVC_HEADERS_MAX); const _: () = assert!(consensus::DVC_HEADERS_MAX == u128::BITS as usize); -/// `[cluster] superblock_wedged_fatal_timeout` as a consecutive-failure count. -/// Retries pin at the backoff cap after warmup, so the window divided by -/// [`journal::superblock::SUPERBLOCK_RETRY_BACKOFF_MAX_MICROS`] bounds how -/// long a wedged replica may limp before it fail-stops. Zero stays zero -/// (fail-stop disabled). +/// `[cluster] superblock_wedged_fatal_timeout` as a consecutive-failure count, +/// which is the only shape the shard's `superblock_wedged` can compare. Zero +/// stays zero (fail-stop disabled). fn superblock_wedged_fatal_failures(config: &ServerConfig) -> u64 { superblock_window_to_failures( config @@ -347,12 +362,44 @@ fn superblock_wedged_fatal_failures(config: &ServerConfig) -> u64 { ) } +/// The failure count whose arrival time is the first at or past `window`. +/// +/// Walks the real retry schedule rather than dividing by the backoff cap. Only +/// the retries past warmup pin at the cap: the first six wait 20, 40, 80, 160, +/// 320 and 640 ms, so they spend 1.26 s of the window where a flat division +/// charges them six. The default 2 m window came out as 120 failures, which +/// arrive after about 114.26 s -- the fail-stop firing almost six seconds before +/// the window the operator configured. fn superblock_window_to_failures(window: Duration) -> u64 { if window.is_zero() { return 0; } - let cap_micros = u128::from(journal::superblock::SUPERBLOCK_RETRY_BACKOFF_MAX_MICROS); - u64::try_from((window.as_micros() / cap_micros).max(1)).unwrap_or(u64::MAX) + // `write_superblock_inner` records failure N and only then arms the wait + // that follows it, so failure N ARRIVES at the sum of the N-1 waits before + // it -- the first arrives at zero. The loop sums forward until the window is + // covered, and the count that satisfies it is one past the last wait summed. + let window_micros = window.as_micros(); + let mut elapsed = 0u128; + let mut waits = 0u64; + while elapsed < window_micros { + waits += 1; + elapsed += u128::from(superblock_retry_backoff_micros(waits)); + } + // A window shorter than the very first retry lands here with one wait + // summed, giving two: a threshold of one would fail-stop on the first + // failure, before any of the window had elapsed at all. + waits.saturating_add(1) +} + +/// The wait `IggyPartition`'s superblock writer arms after its `failures`-th +/// consecutive failure. +/// +/// Mirrors that arithmetic exactly. A divergence here does not fail a test, it +/// moves the fail-stop to a time no operator asked for. +fn superblock_retry_backoff_micros(failures: u64) -> u64 { + journal::superblock::SUPERBLOCK_RETRY_BACKOFF_BASE_MICROS + .saturating_mul(1 << failures.min(journal::superblock::SUPERBLOCK_RETRY_BACKOFF_MAX_SHIFT)) + .min(journal::superblock::SUPERBLOCK_RETRY_BACKOFF_MAX_MICROS) } /// Floor for the post-restart read-recovery deadline (see @@ -592,16 +639,53 @@ mod tests { ); assert_eq!( superblock_window_to_failures(Duration::from_mins(2)), - 120, - "past warmup one retry rides each 1s backoff cap" + 126, + "the six warmup retries spend 1.26s, not 6s: 120 would fire at ~114.26s" + ); + assert_eq!( + superblock_window_to_failures(Duration::from_secs(30)), + 36, + "the configured floor for a nonzero window" ); assert_eq!( superblock_window_to_failures(Duration::from_micros(500)), - 1, - "a sub-cap window still needs one failure to fire" + 2, + "a window shorter than the first retry must not fail-stop on the \ + very first failure, before any of it elapsed" ); } + /// The threshold is a floor on elapsed time, never a ceiling: the failure it + /// names must arrive at or after the configured window, and its predecessor + /// must arrive before it. Walked against the writer's own schedule. + #[test] + fn given_a_fatal_window_when_converted_should_never_fire_before_it_elapses() { + for window in [ + Duration::from_secs(30), + Duration::from_secs(45), + Duration::from_mins(2), + Duration::from_mins(10), + ] { + let threshold = superblock_window_to_failures(window); + // Failure N arrives at the sum of the N-1 waits before it. + let arrival = |count: u64| -> u128 { + (1..count) + .map(|wait| u128::from(superblock_retry_backoff_micros(wait))) + .sum() + }; + assert!( + arrival(threshold) >= window.as_micros(), + "{window:?}: failure {threshold} arrives at {}us, inside the window", + arrival(threshold) + ); + assert!( + arrival(threshold - 1) < window.as_micros(), + "{window:?}: failure {} already covers the window, so {threshold} is late", + threshold - 1 + ); + } + } + #[test] fn default_cluster_heartbeat_timeout_matches_consensus_constant() { // The config default lives in core/server/config.toml (a string, @@ -849,6 +933,21 @@ mod tests { ); } + #[test] + fn default_offset_reservation_lease_matches_partitions_constant() { + // `IggyPartition::new` falls back to the partitions constant (simulator, + // unit tests) while boot installs this one, so drift would have the + // fence write at a different rate in the simulator than in production. + let config_default = + configs::partition::PartitionConfig::default().offset_reservation_lease; + assert_eq!( + config_default.get(), + partitions::DEFAULT_OFFSET_RESERVATION_LEASE, + "[partition] offset_reservation_lease default drifted from \ + partitions::DEFAULT_OFFSET_RESERVATION_LEASE" + ); + } + #[test] fn default_evicted_ring_capacity_matches_partitions_constant() { // Belt and suspenders with the static assert above; this pins the diff --git a/core/server/src/partition_helpers.rs b/core/server/src/partition_helpers.rs index ec39868e5c..15a519a223 100644 --- a/core/server/src/partition_helpers.rs +++ b/core/server/src/partition_helpers.rs @@ -46,7 +46,9 @@ use journal::superblock::{PingPongSuperblock, SuperblockContents}; use message_bus::IggyMessageBus; use metadata::stm::stream::Partition; use metadata::{IdentityField, ReplicaIdentity}; -use partitions::{IggyIndexWriter, IggyPartition, IggyPartitions, MessagesWriter, Segment}; +use partitions::{ + IggyIndexWriter, IggyPartition, IggyPartitions, MessagesWriter, PartitionsConfig, Segment, +}; use server_common::SegmentStorage; use server_common::fs_utils::remove_dir_all; use server_common::sharding::IggyNamespace; @@ -143,15 +145,16 @@ pub async fn create_partition_file_hierarchy( /// Populate `partition` with consumer-offset / consumer-group-offset storage. /// /// Hydrates from on-disk state if files exist (recovery path) or -/// configures empty maps (fresh partition path). `current_offset` bounds -/// recovered offsets so a partition that lost its tail does not surface -/// consumer offsets ahead of its current log head. +/// configures empty maps (fresh partition path). Recovered offsets are bounded +/// so a partition that lost its tail does not surface consumer offsets ahead of +/// an offset it never handed out, and `current_offset` is where a bounded one +/// lands. /// /// # Errors /// /// Returns [`ServerError::ConsumerOffsetsLoad`] when the on-disk files -/// exist but fail to decode. A stored offset ahead of `current_offset` is -/// clamped (with a warning), not an error. +/// exist but fail to decode. A stored offset past the offset space is clamped +/// to `current_offset` (with a warning), not an error. pub fn configure_consumer_offsets( partition: &mut IggyPartition>, config: &ServerConfig, @@ -169,6 +172,17 @@ pub fn configure_consumer_offsets( config .system .get_consumer_group_offsets_path(stream_id, topic_id, partition_id); + // The bound is the offset space this replica could have MINTED, not the data + // it can still serve. A boot re-anchor leaves the append point a lease block + // above the recovered chain, so on the restart after a crash that took + // acked-but-unflushed messages, a position stored before that crash names a + // real offset sitting under an empty chain -- confirmed to a client, and not + // "past the log" the way a torn offset file is. Bounding it by the data head + // instead walks a committed consumer position BACKWARD across the restart, + // which is the silent re-read the reservation exists to prevent. + // `mint_frontier` is one past the next mint, and reads 0 on the fresh-build + // path, where the max leaves `current_offset` in charge as before. + let offset_space_ceiling = current_offset.max(partition.mint_frontier().saturating_sub(1)); let loaded_consumer_offsets = load_partition_consumer_offsets( &consumer_offsets_path, @@ -182,7 +196,7 @@ pub fn configure_consumer_offsets( let guard = consumer_offsets.pin(); for offset in loaded_consumer_offsets { let recovered_offset = offset.offset.load(Ordering::Relaxed); - if recovered_offset > current_offset { + if recovered_offset > offset_space_ceiling { // A crash can persist an offset ahead of the flushed data // (offsets are stored eagerly, messages flush later). Clamp to // the recovered head so the consumer resumes instead of being @@ -191,6 +205,7 @@ pub fn configure_consumer_offsets( consumer_id = offset.consumer_id, recovered_offset, current_offset, + offset_space_ceiling, stream_id, topic_id, partition_id, @@ -213,11 +228,12 @@ pub fn configure_consumer_offsets( let guard = consumer_group_offsets.pin(); for (group_id, offset) in loaded_group_offsets { let recovered_offset = offset.offset.load(Ordering::Relaxed); - if recovered_offset > current_offset { + if recovered_offset > offset_space_ceiling { warn!( consumer_group_id = group_id.0, recovered_offset, current_offset, + offset_space_ceiling, stream_id, topic_id, partition_id, @@ -328,7 +344,7 @@ pub async fn ensure_initial_segment( // `rposition(|s| s.start_offset <= offset)` routes every poll for `0..N-1` // into it, the next boot makes that shape durable, and this replica starts // offering peers a segment that claims `[0..N]`. - let start_offset = partition.offset_frontier(); + let start_offset = partition.mint_frontier(); let messages_path = config .system @@ -535,6 +551,7 @@ pub async fn load_partition_or_fence( // outgrow clippy's `large_futures` cap, and this runs once per partition. match Box::pin(load_partition( config, + partitions.config(), namespace, Arc::clone(&partition_stats), partition_metadata, @@ -662,7 +679,7 @@ pub async fn load_partition_or_fence( Box::pin(build_partition_fresh( config, namespace, - partition_stats, + Arc::clone(&partition_stats), partition_metadata.created_revision, topic_runtime, cluster_id, @@ -708,6 +725,7 @@ pub async fn load_partition_or_fence( #[allow(clippy::too_many_arguments)] async fn load_partition( config: &ServerConfig, + partitions_config: &PartitionsConfig, namespace: IggyNamespace, stats: Arc, partition_metadata: &Partition, @@ -798,6 +816,7 @@ async fn load_partition( config.partition.evicted_ring_bytes_max.as_bytes_u64(), ); partition.set_dedup_clients_max(config.partition.dedup_clients_max); + partition.set_offset_reservation_lease(config.partition.offset_reservation_lease); partition.set_partition_dir(partition_dir.clone()); // Before the hydrate: the durable record is keyed by incarnation, so a // `purge.gen` left behind by a previous life of this namespace reads 0. @@ -813,6 +832,28 @@ async fn load_partition( ) .await?; + partition.created_at = partition_metadata.created_at; + restore_partition_offsets(&mut partition, partitions_config, recovered_state.as_ref()).await?; + let current_offset = partition.offset.load(Ordering::Acquire); + + configure_consumer_offsets(&mut partition, config, namespace, current_offset)?; + ensure_initial_segment(&mut partition, config, stream_id, topic_id, partition_id).await?; + + Ok(partition) +} + +/// Restore the offset counter of a recovered partition from what boot could +/// prove about its offset space, then put the next append point where the +/// recovery walk can read it back. +/// +/// Three carriers, weakest last: the sized segments' end offset, an empty +/// chain's file name (a state-transfer install at the group frontier), and the +/// superblock's durable frontier as a lower bound over both. +async fn restore_partition_offsets( + partition: &mut IggyPartition>, + partitions_config: &PartitionsConfig, + recovered_state: Option<&VsrState>, +) -> Result<(), ServerError> { let sized_end = partition .log .segments() @@ -825,15 +866,23 @@ async fn load_partition( // frontier after the origin GC'd everything: the file name carries the // frontier, and re-minting offsets from 0 here would fork this // replica's batch stamps from the rest of the group after a restart. + // + // Bounded by the durable frontier: an install writes it at the group + // frontier, so the name is corroborated, while the boot re-anchor and + // `ensure_initial_segment` plant at `mint_frontier()`, a RESERVATION that + // names no data and leaves the frontier far below. Without the bound two + // crashes under the flush threshold promote 65537 to committed on a + // partition holding nothing, and `store_consumer_offset` admits the hole. + let durable_frontier = recovered_state.map_or(0, |state| state.offset_frontier); let empty_frontier = partition .log .segments() .iter() .map(|segment| segment.start_offset) .max() + .map(|start| start.min(durable_frontier)) .filter(|&start| sized_end.is_none() && start > 0); let current_offset = sized_end.or_else(|| empty_frontier.map(|start| start - 1)); - partition.created_at = partition_metadata.created_at; partition.recovered_durable_offset = sized_end; // The OFFSET COUNTER is restored from that file name (above), but the // `installed_frontier` CLAIM deliberately is not: the claim says "everything @@ -851,18 +900,27 @@ async fn load_partition( let counter = current_offset.unwrap_or(0); partition.offset.store(counter, Ordering::Release); partition.dirty_offset.store(counter, Ordering::Relaxed); - partition.should_increment_offset = current_offset.is_some(); + partition.set_offset_space_used(current_offset.is_some()); // The durable frontier is a LOWER BOUND on top of what the segments proved: // it is the only carrier left when the segments that named the frontier are // gone (an all-GC'd origin's install, a crash inside the swap window), and // taking the max means real recovered data always wins. - partition.restore_offset_frontier(recovered_state.as_ref()); - let current_offset = partition.offset.load(Ordering::Acquire); - - configure_consumer_offsets(&mut partition, config, namespace, current_offset)?; - ensure_initial_segment(&mut partition, config, stream_id, topic_id, partition_id).await?; - - Ok(partition) + partition.restore_offset_frontier(recovered_state); + // Minting from the reservation leaves a hole between the recovered chain + // and the new append point, and the recovery walk REFUSES a hole inside a + // segment (tombstoning the partition on the solo arm), so put it on a + // segment boundary instead. + // + // Solo only, in step with the reservation itself: a replicated group's + // segment boundaries must be a function of the batches alone or the + // reconciler's offset-keyed segment GC never converges. + if partition.consensus().replica_count() == 1 { + partition + .reanchor_to_offset_frontier(partitions_config) + .await + .map_err(|error| ServerError::Iggy(Box::new(error)))?; + } + Ok(()) } /// Recover this partition's persisted segment chain, stamping each segment @@ -887,26 +945,18 @@ async fn recover_partition_segments( let enforce_fsync = runtime_options .enforce_fsync .unwrap_or(iggy_common::DEFAULT_ENFORCE_FSYNC); - load_persisted_segments( - config, - stream_id, - topic_id, - partition_id, - segment_size, - enforce_fsync, - stats, - ) - .await - .map_err(|source| { - error!( - stream_id, - topic_id, - partition_id, - error = %source, - "failed to load partition log during server bootstrap" - ); - source - }) + load_persisted_segments(config, namespace, segment_size, enforce_fsync, stats) + .await + .map_err(|source| { + error!( + stream_id, + topic_id, + partition_id, + error = %source, + "failed to load partition log during server bootstrap" + ); + source + }) } /// Reopen writers over a recovered segment chain. @@ -1063,11 +1113,13 @@ fn hydrate_reopen_error( /// already on disk is routed through the loader instead, so a prior /// life's segments are hydrated rather than built over. /// -/// Steps performed (all idempotent on retry after a partial failure): +/// Steps performed. 1 to 4 are idempotent on retry after a partial failure; the +/// claim is last precisely because it is not (see its own comment): /// 1. Create directory hierarchy on disk. /// 2. Build per-partition VSR consensus group, resuming any superblock-recorded view. /// 3. Configure empty consumer-offset storage with the on-disk paths set. /// 4. Provision the initial segment + writers (offset 0). +/// 5. Claim the group's first offset-reservation block (solo groups with a store). /// /// The namespace arrives packed, so its components are in range by /// construction. Metadata admission is what bounds them. @@ -1078,14 +1130,14 @@ fn hydrate_reopen_error( /// `seed_view` comment below for why a group left at view 0 is unreachable. A /// restart materialization ignores it and probes for the live view instead. /// -/// The returned partition's `offset` / `dirty_offset` are `0` and -/// `should_increment_offset` is `false`, mirroring a clean append starting -/// at the empty segment. +/// The returned partition's `offset` / `dirty_offset` are `0` and its +/// `OffsetSpace` is unused, mirroring a clean append starting at the empty +/// segment. /// /// # Errors /// -/// Returns [`ServerError`] when directory creation, superblock recovery, or -/// segment provisioning fails. +/// Returns [`ServerError`] when directory creation, superblock recovery, +/// segment provisioning, or the first offset-reservation claim fails. #[allow(clippy::too_many_arguments)] pub async fn build_partition_fresh( config: &ServerConfig, @@ -1212,6 +1264,7 @@ pub async fn build_partition_fresh( config.partition.evicted_ring_bytes_max.as_bytes_u64(), ); partition.set_dedup_clients_max(config.partition.dedup_clients_max); + partition.set_offset_reservation_lease(config.partition.offset_reservation_lease); partition.set_partition_dir(partition_dir); // Fresh dirs read generation 0; a dir surviving from a crashed process // (this "fresh" build races repair re-materialization) reads the last @@ -1224,7 +1277,7 @@ pub async fn build_partition_fresh( partition.created_at = IggyTimestamp::now(); partition.offset.store(0, Ordering::Release); partition.dirty_offset.store(0, Ordering::Relaxed); - partition.should_increment_offset = false; + partition.set_offset_space_used(false); debug_assert!( !partition.log.has_segments(), "fresh partition must not carry recovered segments" @@ -1245,11 +1298,59 @@ pub async fn build_partition_fresh( // frontier before quarantining, and the boot-path chain refusal to carry // the refused chain's max `end_offset` on its error. partition.restore_offset_frontier(recovered_state.as_ref()); + let current_offset = partition.offset.load(Ordering::Acquire); configure_consumer_offsets(&mut partition, config, namespace, current_offset)?; ensure_initial_segment(&mut partition, config, stream_id, topic_id, partition_id).await?; + // Claim the first offset-reservation block HERE so no send ever pays the + // create, write, file fsync, rename and directory fsync of a first claim + // inline in the shard's request pump, where the consensus tick is a sibling + // arm. It is a NEW write on a path that otherwise only READS the superblock: + // one atomic replace per created partition, serialised with its siblings in + // the reconciler's addition loop, so it lengthens the window a produce + // arriving with the create spends parked. + // + // LAST of the steps, because a claim written before a step that then fails + // outlives the create. The reconciler routes any namespace whose directory + // exists to `load_partition_or_fence`, and step 1 made that directory, so + // the retry comes back through the loader: `restore_offset_frontier` there + // resumes the append point at the recorded reservation and holes every + // offset below it on a partition that never took a write. + // + // `0`, not `mint_frontier()`: a rebuild recovers its append point exactly ON + // the reservation it recorded, so asking to cover the frontier would fail + // the callee's strict `>` and rewrite the record on every rebuild. Asking + // only for offset 0 leaves that same check to skip every partition already + // carrying a reservation, which pays one inline fence on its first send + // instead, the cost a graceful stop and boot already carries. No-op above + // one replica and with no store attached, where nothing is reserved. + // + // The shard tick takes over from the first mint onward + // (`needs_offset_reservation_extension`), which stays gated on a partition + // that has minted so boot cannot write a superblock per idle partition. + if !partition.reserve_offsets_through(0).await { + // Not degraded-but-live: the failed write armed the group's superblock + // retry backoff, and `reserve_offsets_through_retryable` refuses every + // send arriving inside it with a transient the HTTP plane does not + // replay. Neither caller escalates: the reconciler backs the namespace + // off and its retry materialises through the loader, whose partition + // carries a clear backoff cell, and boot skips the partition, leaving it + // to that same retry. + // + // `ensure_initial_segment` folded its segment into the parent topic + // before the claim ran, and the retry counts the same file again while + // recovering it. + partition.stats.zero_out_all(); + return Err(ServerError::PartitionOffsetReservationClaim { + stream_id, + topic_id, + partition_id, + namespace_raw: namespace.inner(), + }); + } + Ok(partition) } @@ -1312,7 +1413,10 @@ pub async fn delete_partitions_from_disk( #[cfg(test)] mod tests { use super::*; + use configs::server::ServerSystemConfig; use journal::superblock::SuperblockStore; + use partitions::PartitionPathLayout; + use server_common::sharding::ShardId; const CLUSTER: u128 = 7; const REPLICA: u8 = 1; @@ -1329,6 +1433,19 @@ mod tests { checkpoint_op: 0, checkpoint_checksum: 0, offset_frontier: 0, + offset_reserved: 0, + } + } + + /// The solo shape the reservation is scoped to, with a claim already + /// recorded and nothing flushed behind it. + fn reserved_solo_state(reserved: u64) -> VsrState { + VsrState { + replica_id: 0, + replica_count: 1, + commit_max: 0, + offset_reserved: reserved, + ..recorded_state(0, 0) } } @@ -1344,6 +1461,221 @@ mod tests { } } + /// The offset reservation is solo-only, so every test that touches it builds + /// under this identity. + const fn solo_identity() -> ReplicaIdentity { + ReplicaIdentity { + cluster: CLUSTER, + replica_id: 0, + replica_count: 1, + } + } + + fn solo_config(root: &tempfile::TempDir) -> ServerConfig { + ServerConfig { + system: Arc::new(ServerSystemConfig { + path: root.path().to_string_lossy().into_owned(), + ..ServerSystemConfig::default() + }), + ..ServerConfig::default() + } + } + + async fn build_solo_partition( + config: &ServerConfig, + ) -> Result>, ServerError> { + build_partition_fresh( + config, + IggyNamespace::new(1, 1, 0), + Arc::new(PartitionStats::default()), + 0, + TopicRuntimeOptions::default(), + CLUSTER, + 0, + 1, + 0, + Rc::new(IggyMessageBus::new(0)), + ) + .await + } + + /// The container the loader tombstones into. Its config is never read on the + /// paths under test, which stop before `restore_partition_offsets`. + fn solo_partitions() -> IggyPartitions> { + IggyPartitions::new( + ShardId::new(0), + PartitionsConfig { + messages_required_to_save: 1, + size_of_messages_required_to_save: IggyByteSize::from(1024_u64), + enforce_fsync: false, + validate_checksum: true, + segment_size: IggyByteSize::from(1_048_576_u64), + preallocate_segments: false, + encryptor: None, + path_layout: PartitionPathLayout::default(), + }, + ) + } + + /// The reservation the partition left on disk, which is the only copy a + /// restart or a first send can read. + async fn recorded_reservation(dir: &str) -> u64 { + let (_store, recorded) = open_partition_superblock(dir, solo_identity()) + .await + .expect("reopen the partition superblock"); + recorded + .expect("a partition that recorded a reservation") + .offset_reserved + } + + /// The create claims the first lease block, so the DURABLE record covers the + /// first send before it arrives. Asserting on the returned partition alone + /// would pass with no claim at all: the inline fence at the mint writes the + /// same block on the first send, which is exactly what this moves off the + /// append path. + #[compio::test] + async fn given_a_fresh_solo_partition_when_building_should_record_its_first_claim() { + let root = tempfile::tempdir().expect("tempdir"); + let config = solo_config(&root); + let dir = config.system.get_partition_path(1, 1, 0); + + let partition = build_solo_partition(&config) + .await + .expect("build a fresh partition"); + drop(partition); + + assert_eq!( + recorded_reservation(&dir).await, + 1 + u64::from(config.partition.offset_reservation_lease.get()), + "the create must leave a full lease block covering offset 0 on disk" + ); + } + + /// A rebuild that reads a reservation back must NAME its planted segment for + /// the append point. Named 0, the segment takes the first append's + /// `base_offset` of N instead, `rposition(|s| s.start_offset <= offset)` + /// routes every poll for `0..N-1` into it, and the next boot makes that + /// durable. + #[compio::test] + async fn given_a_recorded_reservation_when_building_fresh_should_plant_at_the_append_point() { + const RESERVED: u64 = 65_537; + let root = tempfile::tempdir().expect("tempdir"); + let config = solo_config(&root); + let dir = config.system.get_partition_path(1, 1, 0); + + let (store, recovered) = open_partition_superblock(&dir, solo_identity()) + .await + .expect("open a fresh partition superblock"); + assert!(recovered.is_none()); + store + .write(&reserved_solo_state(RESERVED).to_bytes()) + .await + .expect("record the reservation"); + drop(store); + + let partition = build_solo_partition(&config) + .await + .expect("rebuild the partition over its recorded reservation"); + + assert_eq!( + partition.mint_frontier(), + RESERVED, + "the append point must resume above every offset the reservation covered" + ); + assert_eq!( + partition.offset_frontier(), + 0, + "nothing was flushed, so the committed frontier names no data" + ); + + let planted: Vec = std::fs::read_dir(&dir) + .expect("list the partition dir") + .flatten() + .filter_map(|entry| { + let path = entry.path(); + (path.extension()? == "log") + .then(|| path.file_name()?.to_str().map(str::to_owned)) + .flatten() + }) + .collect(); + assert_eq!( + planted, + vec![format!("{RESERVED:0>20}.log")], + "the initial segment must be named for the append point, not offset 0" + ); + + drop(partition); + assert_eq!( + recorded_reservation(&dir).await, + RESERVED, + "a rebuild resumes ON its recorded reservation, so re-claiming here would \ + burn a lease block and two fsyncs per rebuild" + ); + } + + /// The loader's fence-and-rebuild arm is the claim's SECOND caller. A + /// refused claim is a failed superblock write, not damage, so it has to come + /// back as an error every caller can retry: absorbing it into a tombstone + /// here would darken the namespace for the life of the process over a fault + /// the next attempt may not even hit. + #[compio::test] + async fn given_a_failing_claim_when_rebuilding_a_fenced_chain_should_refuse_not_tombstone() { + let root = tempfile::tempdir().expect("tempdir"); + let config = solo_config(&root); + let namespace = IggyNamespace::new(1, 1, 0); + let dir = config.system.get_partition_path(1, 1, 0); + std::fs::create_dir_all(&dir).expect("partition dir"); + // Two empty segments make the first a NON-tail empty, the refusal a solo + // group rebuilds through (zero recoverable bytes) instead of tombstoning + // where it stands. + for start_offset in [0, 1] { + std::fs::File::create(config.system.get_messages_file_path(1, 1, 0, start_offset)) + .expect("empty segment log"); + } + // The rebuild's claim is this group's first superblock write, so it + // targets slot A. A directory where its temp file goes fails the atomic + // replace and nothing else: the slot reads still find the store empty, + // and the quarantine moves segment files only. + std::fs::create_dir(Path::new(&dir).join("superblock.a.tmp")).expect("block slot A"); + + let stats = Arc::new(PartitionStats::default()); + let partitions = solo_partitions(); + let loaded = load_partition_or_fence( + &config, + namespace, + Arc::clone(&stats), + &Partition::new(0, namespace.inner(), IggyTimestamp::now(), 0, 0), + TopicRuntimeOptions::default(), + CLUSTER, + 0, + 1, + Rc::new(IggyMessageBus::new(0)), + &partitions, + ) + .await; + + match loaded { + Err(ServerError::PartitionOffsetReservationClaim { .. }) => {} + Err(other) => panic!("expected the claim's own refusal, got {other}"), + Ok(Some(_)) => panic!("the planted directory must fail the rebuild's claim"), + Ok(None) => panic!("a refused claim must reach the caller, not be absorbed here"), + } + assert!( + std::fs::metadata(format!("{dir}.fenced.0")).is_ok(), + "the quarantine must have run, or this asserts on the wrong arm" + ); + assert!( + !partitions.is_tombstoned(&namespace), + "a transient write failure must leave the namespace materialisable" + ); + assert_eq!( + stats.segments_count_inconsistent(), + 0, + "the rebuild counted its initial segment before the claim refused, and the \ + retry counts the same file again" + ); + } + #[compio::test] async fn given_fresh_partition_dir_when_superblock_opened_should_yield_no_state() { let root = tempfile::tempdir().expect("tempdir"); diff --git a/core/server/src/segment_recovery.rs b/core/server/src/segment_recovery.rs index 2c889fa92b..0e192c0794 100644 --- a/core/server/src/segment_recovery.rs +++ b/core/server/src/segment_recovery.rs @@ -30,15 +30,17 @@ use crate::server_error::{PartitionRecoveryRefusal, ServerError}; use configs::server::ServerConfig; use iggy_common::{IggyByteSize, IggyError, MAX_MESSAGE_SIZE_UPPER_BYTES, PartitionStats}; +use partitions::segment_anchor::ANCHOR_EXTENSION; use partitions::state_transfer::STAGING_SUFFIX; use partitions::{IggyIndex, IggyIndexReader, Segment}; use server_common::send_messages::{BatchHeader, COMMAND_HEADER_SIZE, decode_batch_slice}; +use server_common::sharding::IggyNamespace; use server_common::{SegmentStorage, yield_to_reactor}; use std::fs; use std::io; use std::os::unix::fs::FileExt; use std::path::{Path, PathBuf}; -use tracing::{error, warn}; +use tracing::{error, info, warn}; const LOG_EXTENSION: &str = "log"; const INDEX_EXTENSION: &str = "index"; @@ -173,16 +175,21 @@ pub struct RecoveredSegment { /// makes a durable index entry evidence about the log (see /// [`PartitionRecoveryRefusal::FsyncedLogLoss`]), so passing it wrong either /// refuses healthy chains or hides previously durable data loss. +/// +/// Takes no offset ceiling. A legitimate gap is proved by the anchor the boot +/// re-anchor writes beside the segment it plants, not inferred from how far the +/// superblock's reservation happens to reach. #[allow(clippy::too_many_lines)] pub async fn load_persisted_segments( config: &ServerConfig, - stream_id: usize, - topic_id: usize, - partition_id: usize, + namespace: IggyNamespace, segment_size: IggyByteSize, enforce_fsync: bool, stats: &PartitionStats, ) -> Result, ServerError> { + let stream_id = namespace.stream_id(); + let topic_id = namespace.topic_id(); + let partition_id = namespace.partition_id(); let partition_path = config .system .get_partition_path(stream_id, topic_id, partition_id); @@ -298,9 +305,9 @@ pub async fn load_persisted_segments( last.segment.sealed = false; } - // Pass B: the chain guard reads only the planned bounds, so it can refuse - // BEFORE anything is truncated. - ensure_contiguous_chain(identity, &planned)?; + // Pass B: the chain guard reads the planned bounds and, for a gap, the + // anchor beside it, so it can refuse BEFORE anything is truncated. + ensure_contiguous_chain(identity, &planned).await?; // Pass C: the chain is accepted; make disk match the bounds and open // storage over them. @@ -529,13 +536,35 @@ struct ScanScratch { /// chain and push `current_offset` past data this replica does not hold. /// Refuse loudly instead of serving a holed log. /// +/// A FORWARD gap is admitted only when the far side carries a +/// [`SegmentAnchor`] naming exactly the near side, which is the record the boot +/// re-anchor writes before it plants. Nothing else legitimises a gap: an +/// overlap, a backwards pair, or a gap with no anchor is damage. +/// +/// The anchors are what a monotone offset ceiling could not be. A ceiling says +/// only "some boot claimed up to N", and every plant base sits below the current +/// N -- but so does the successor of a segment that was deleted, so a lost middle +/// segment read as a plant. The anchor is written by the one component that +/// creates legitimate gaps, names which segment it sealed, and is swept as soon +/// as its segment is gone. +/// +/// Retention needs no allowance: it removes a contiguous FRONT prefix, so the +/// remaining chain stays contiguous and no interior gap appears. +/// +/// # Known residual +/// +/// An anchor whose planted segment survived while the sealed segment it names was +/// itself lost still reads as legitimate, because the pair the anchor describes +/// is then simply absent from the chain and the guard never examines it. Catching +/// that needs a durable count of the chain, which this record does not carry. +/// /// Runs on the planned bounds alone, BEFORE any truncation, so the segment /// files a refusal quarantines are exactly the bytes boot found. The refusal /// names the partition and its directory so the caller can fence THAT group /// rather than abort the node's boot: the shapes it rejects are exactly what /// a failed quarantine leaves behind, and one damaged local chain must not /// take the whole node down. -fn ensure_contiguous_chain( +async fn ensure_contiguous_chain( identity: PartitionIdentity<'_>, planned: &[PlannedSegment], ) -> Result<(), ServerError> { @@ -567,7 +596,41 @@ fn ensure_contiguous_chain( } // `checked_add`, not `+`: an end offset at u64::MAX must read as a // hole (no start offset can follow it), not overflow. - if previous.end_offset.checked_add(1) != Some(next.start_offset) { + if previous.end_offset.checked_add(1) == Some(next.start_offset) { + continue; + } + // FORWARD only, and only with the plant's own record beside it. Start + // offsets come off the file names so they ascend, but each end offset is + // walked from that file's own bytes with nothing clamping it against the + // next start, so a half-installed transfer or an operator copy can leave + // a pair that overlaps -- and no re-anchor ever plants a segment whose + // range a predecessor already covers. + // Read HERE rather than collected up front: only a gap needs an anchor, + // so a contiguous chain -- every chain that never crashed mid-block -- + // opens no file at all, and the ones that do are already walking this + // pair. An unreadable anchor is unknown, not absent, and treating it as + // absent would refuse a healthy chain for as long as the fault lasts. + let read = + partitions::segment_anchor::read_anchor(identity.partition_path, next.start_offset) + .await + .map_err(|error| { + error!( + partition_path = identity.partition_path, + start_offset = next.start_offset, + %error, + "failed to read a segment anchor during recovery" + ); + ServerError::from(IggyError::CannotReadFile) + })?; + let anchored = previous.end_offset < next.start_offset + && read.is_some_and(|anchor| { + anchor.covers( + next.start_offset, + previous.start_offset, + previous.end_offset, + ) + }); + if !anchored { return Err(identity.refusal(PartitionRecoveryRefusal::Hole { previous_start: previous.start_offset, previous_end: previous.end_offset, @@ -575,6 +638,14 @@ fn ensure_contiguous_chain( recoverable_bytes, })); } + info!( + partition_path = identity.partition_path, + previous_start = previous.start_offset, + previous_end = previous.end_offset, + next_start = next.start_offset, + "admitted a gap in the recovered segment chain: the planted segment \ + carries the boot re-anchor's own record of it" + ); } Ok(()) } @@ -636,7 +707,11 @@ fn sweep_scratch_files_and_collect_offsets(partition_path: &str) -> Result orphan_candidates.push(path), + // An anchor outlives nothing: it describes the gap in front of ONE + // segment, so once that segment is gone (retention, a failed plant + // that never landed) the record can only mislead a later guard into + // admitting a gap it never saw. + Some(INDEX_EXTENSION | ANCHOR_EXTENSION) => orphan_candidates.push(path), _ => {} } } @@ -2407,6 +2482,7 @@ mod tests { use super::*; use bytes::Bytes; use configs::server::ServerSystemConfig; + use partitions::segment_anchor::SegmentAnchor; use server_common::send_messages::{ IggyMessage, IggyMessageHeader, IggyMessages, SendMessagesOwned, calculate_batch_checksum, }; @@ -2565,6 +2641,44 @@ mod tests { (messages_path, index_path) } + /// Path of the anchor beside the segment planted at `start_offset`. + fn anchor_fixture_path(config: &ServerConfig, start_offset: u64) -> String { + partitions::segment_anchor::anchor_path( + &config + .system + .get_partition_path(STREAM_ID, TOPIC_ID, PARTITION_ID), + start_offset, + ) + } + + /// Write the anchor a plant at `planted_start` leaves behind, naming the tail + /// it sealed. + fn write_anchor_fixture( + config: &ServerConfig, + planted_start: u64, + sealed_start: u64, + sealed_end: u64, + ) { + let anchor = SegmentAnchor { + planted_start, + sealed_start, + sealed_end, + }; + fs::write( + anchor_fixture_path(config, planted_start), + anchor.to_bytes(), + ) + .expect("write anchor fixture"); + } + + /// Recover expecting a refusal, with `context` naming what should have failed. + async fn refusal(config: &ServerConfig, context: &str) -> ServerError { + match recover(config).await { + Ok(recovered) => panic!("{context}, got {} segments", recovered.len()), + Err(error) => error, + } + } + fn len_of(path: &str) -> u64 { fs::metadata(path).expect("stat fixture file").len() } @@ -2584,18 +2698,23 @@ mod tests { } async fn recover(config: &ServerConfig) -> Result, ServerError> { - recover_under_fsync(config, false).await + recover_with(config, false).await } async fn recover_under_fsync( config: &ServerConfig, enforce_fsync: bool, + ) -> Result, ServerError> { + recover_with(config, enforce_fsync).await + } + + async fn recover_with( + config: &ServerConfig, + enforce_fsync: bool, ) -> Result, ServerError> { load_persisted_segments( config, - STREAM_ID, - TOPIC_ID, - PARTITION_ID, + IggyNamespace::new(STREAM_ID, TOPIC_ID, PARTITION_ID), IggyByteSize::from(SEGMENT_MAX_SIZE), enforce_fsync, &PartitionStats::default(), @@ -3018,6 +3137,249 @@ mod tests { assert_eq!(bytes_of(&next_index_path), next_index); } + /// Valid, checksum-clean anchor bytes COPIED beside a later segment must not + /// authorise the wider gap they now sit in front of. The record names its own + /// plant, and the guard matches that against the file it was found beside, so + /// an operator's `cp` buys nothing. + #[compio::test] + async fn given_an_anchor_copied_beside_a_later_segment_when_recovering_should_refuse() { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + write_segment(&config, 0, &encoded_batch(0, 3), &index_entry(0, 0)); + write_segment(&config, 500, &encoded_batch(500, 1), &index_entry(500, 0)); + // The anchor a legitimate plant at 10 would have left, moved beside the + // segment at 500 without touching a byte of it. + let stolen = fs::read({ + write_anchor_fixture(&config, 10, 0, 2); + anchor_fixture_path(&config, 10) + }) + .expect("read the legitimate anchor"); + fs::remove_file(anchor_fixture_path(&config, 10)).expect("unlink the original"); + fs::write(anchor_fixture_path(&config, 500), &stolen).expect("copy it beside 500"); + + let error = refusal(&config, "a copied anchor must not cover a wider gap").await; + assert!( + matches!( + &error, + ServerError::PartitionRecoveryRefused { + reason: PartitionRecoveryRefusal::Hole { .. }, + .. + } + ), + "expected a hole refusal, got {error:?}" + ); + } + + /// The shape the boot re-anchor leaves: a sealed tail, then the next segment + /// planted above it, with the anchor beside the plant naming the tail. + #[compio::test] + async fn given_an_anchored_gap_when_recovering_should_accept_the_chain() { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + write_segment(&config, 0, &encoded_batch(0, 3), &index_entry(0, 0)); + write_segment(&config, 10, &encoded_batch(10, 1), &index_entry(10, 0)); + write_anchor_fixture(&config, 10, 0, 2); + + let recovered = recover(&config) + .await + .expect("a gap the plant recorded is the re-anchor's, not damage"); + + assert_eq!(recovered.len(), 2); + assert_eq!(recovered[0].segment.start_offset, 0); + assert_eq!(recovered[1].segment.start_offset, 10); + } + + /// The regression this record exists to close: with no anchor the gap is a + /// segment that went missing, and admitting it serves a holed log silently. + #[compio::test] + async fn given_an_unanchored_gap_when_recovering_should_refuse() { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + write_segment(&config, 0, &encoded_batch(0, 3), &index_entry(0, 0)); + write_segment(&config, 10, &encoded_batch(10, 1), &index_entry(10, 0)); + + let error = refusal(&config, "a gap no plant recorded must refuse recovery").await; + + assert!( + matches!( + &error, + ServerError::PartitionRecoveryRefused { + reason: PartitionRecoveryRefusal::Hole { .. }, + .. + } + ), + "expected a hole refusal, got {error:?}" + ); + } + + /// An anchor names WHICH segment it sealed, so one left behind by an earlier + /// chain cannot legitimise a gap it never saw. + #[compio::test] + async fn given_an_anchor_naming_another_segment_when_recovering_should_refuse() { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + write_segment(&config, 0, &encoded_batch(0, 3), &index_entry(0, 0)); + write_segment(&config, 10, &encoded_batch(10, 1), &index_entry(10, 0)); + // Right shape, wrong predecessor: this anchor describes a tail ending at + // 5, and the chain's tail ends at 2. + write_anchor_fixture(&config, 10, 0, 5); + + let error = refusal(&config, "an anchor for a different tail must not cover it").await; + assert!( + matches!( + &error, + ServerError::PartitionRecoveryRefused { + reason: PartitionRecoveryRefusal::Hole { .. }, + .. + } + ), + "expected a hole refusal, got {error:?}" + ); + } + + /// A corrupt anchor proves nothing, so the gap it would have covered stays + /// damage rather than becoming legitimate by default. + #[compio::test] + async fn given_a_corrupt_anchor_when_recovering_should_refuse() { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + write_segment(&config, 0, &encoded_batch(0, 3), &index_entry(0, 0)); + write_segment(&config, 10, &encoded_batch(10, 1), &index_entry(10, 0)); + let path = anchor_fixture_path(&config, 10); + let mut bytes = SegmentAnchor { + planted_start: 10, + sealed_start: 0, + sealed_end: 2, + } + .to_bytes(); + bytes[8] ^= 1; + fs::write(&path, bytes).expect("write corrupt anchor"); + + let error = refusal(&config, "a corrupt anchor must not cover a gap").await; + assert!( + matches!( + &error, + ServerError::PartitionRecoveryRefused { + reason: PartitionRecoveryRefusal::Hole { .. }, + .. + } + ), + "expected a hole refusal, got {error:?}" + ); + } + + /// Every crash cycle leaves one more re-anchor gap, and the guard runs on the + /// NEXT boot -- before the re-anchor -- so by the second cycle the earlier gap + /// is no longer the last pair. A rule keyed on the last pair alone would + /// refuse this, and the solo arm tombstones a chain with bytes in it, taking a + /// healthy partition dark from the second crash onward. + #[compio::test] + async fn given_gaps_from_several_crash_cycles_when_recovering_should_accept_the_chain() { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + write_segment(&config, 0, &encoded_batch(0, 3), &index_entry(0, 0)); + write_segment(&config, 10, &encoded_batch(10, 2), &index_entry(10, 0)); + write_segment(&config, 20, &encoded_batch(20, 1), &index_entry(20, 0)); + write_anchor_fixture(&config, 10, 0, 2); + write_anchor_fixture(&config, 20, 10, 11); + + let recovered = recover(&config) + .await + .expect("anchored gaps accumulate one per crash, not one total"); + + assert_eq!(recovered.len(), 3); + assert_eq!(recovered[0].segment.start_offset, 0); + assert_eq!(recovered[1].segment.start_offset, 10); + assert_eq!(recovered[2].segment.start_offset, 20); + } + + /// One lost segment in the middle of a chain whose OTHER gaps are all + /// anchored. The anchors say nothing about this pair, so it must still refuse + /// -- the shape a single monotone ceiling could not separate from a plant. + #[compio::test] + async fn given_a_lost_segment_among_anchored_gaps_when_recovering_should_refuse() { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + write_segment(&config, 0, &encoded_batch(0, 3), &index_entry(0, 0)); + write_segment(&config, 10, &encoded_batch(10, 2), &index_entry(10, 0)); + // 20 was an ordinary rotation off 12, then went missing; 30 is a plant. + write_segment(&config, 30, &encoded_batch(30, 1), &index_entry(30, 0)); + write_anchor_fixture(&config, 10, 0, 2); + + let error = refusal(&config, "a lost middle segment must refuse recovery").await; + assert!( + matches!( + &error, + ServerError::PartitionRecoveryRefused { + reason: PartitionRecoveryRefusal::Hole { + previous_start: 10, + previous_end: 11, + next_start: 30, + .. + }, + .. + } + ), + "expected a hole refusal naming the lost pair, got {error:?}" + ); + } + + /// No re-anchor plants a segment whose range a predecessor already covers, so + /// an anchor must not launder an overlap either. Reachable because each end + /// offset is walked from its own file with nothing clamping it against the + /// next start. + #[compio::test] + async fn given_overlapping_segments_when_recovering_should_refuse_even_when_anchored() { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + // `0.log` walks to 0..=15 while `10.log` claims 10 onward: the pair runs + // backwards, which no legitimate chain does. + write_segment(&config, 0, &encoded_batch(0, 16), &index_entry(0, 0)); + write_segment(&config, 10, &encoded_batch(10, 3), &index_entry(10, 0)); + write_anchor_fixture(&config, 10, 0, 15); + + let error = refusal(&config, "an overlap must refuse however it is recorded").await; + assert!( + matches!( + &error, + ServerError::PartitionRecoveryRefused { + reason: PartitionRecoveryRefusal::Hole { .. }, + .. + } + ), + "expected a hole refusal, got {error:?}" + ); + } + + /// An anchor whose segment is gone can only mislead a later guard, so the boot + /// sweep collects it the way it collects an orphaned index. + #[compio::test] + async fn given_an_anchor_with_no_segment_when_recovering_should_sweep_it() { + let tmp = tempdir().expect("tempdir"); + let config = test_config(&tmp); + prepare_partition_dir(&config); + write_segment(&config, 0, &encoded_batch(0, 3), &index_entry(0, 0)); + // The window a crash between the anchor write and the plant leaves. + write_anchor_fixture(&config, 10, 0, 2); + let orphan = anchor_fixture_path(&config, 10); + + let recovered = recover(&config).await.expect("recover the intact chain"); + + assert_eq!(recovered.len(), 1); + assert!( + !Path::new(&orphan).exists(), + "an anchor naming a segment that does not exist must be swept" + ); + } + #[compio::test] async fn given_non_monotone_index_entries_when_recovering_should_rebuild_the_index_from_the_log() { diff --git a/core/server/src/server_error.rs b/core/server/src/server_error.rs index 7f0f86c5e3..b9e744755c 100644 --- a/core/server/src/server_error.rs +++ b/core/server/src/server_error.rs @@ -201,6 +201,21 @@ pub enum ServerError { partition_id: usize, reason: PartitionRecoveryRefusal, }, + /// Fails the create rather than letting the partition go live without its + /// first reservation: the failed write arms the group's superblock retry + /// backoff, and a send arriving inside that window is refused with a + /// transient the HTTP plane does not replay. `namespace_raw` joins this to + /// the write's own `iggy.partitions.diag` line, which carries the cause. + #[error( + "partition {stream_id}/{topic_id}/{partition_id} (namespace {namespace_raw}) could not \ + claim its first offset reservation" + )] + PartitionOffsetReservationClaim { + stream_id: usize, + topic_id: usize, + partition_id: usize, + namespace_raw: u64, + }, #[error( "shard {shard_id} aborted while waiting for shard-0 to broadcast the metadata \ factory bundle; shard 0 dropped its sender (most likely it failed to recover)" diff --git a/core/shard/src/lib.rs b/core/shard/src/lib.rs index a8adfe0667..f44c40f3e0 100644 --- a/core/shard/src/lib.rs +++ b/core/shard/src/lib.rs @@ -936,6 +936,16 @@ impl RestorableMetadataStm for M where /// so the bounded per-peer bus queue can never drop a burst tail. Clamped /// against the live bus ceiling by /// [`IggyShard::state_chunk_len_max`] rather than assumed to fit. +/// Superblock writes issued at once when a whole shard's groups need one in the +/// same pass: a node-wide view change, or a graceful stop collapsing every +/// partition's offset reservation. +/// +/// Each write is a create + write + 2 fsyncs. Serial, a few hundred groups on +/// ordinary storage overrun the view-change escalation window (and, on the stop +/// path, a supervisor's kill timeout); unbounded, they dump the whole burst of +/// fds and fsyncs onto the reactor in one pass. +const SUPERBLOCK_FAN_OUT: usize = 16; + const STATE_CHUNK_LEN: u32 = 256 * 1024; /// Bus frame ceiling assumed before bootstrap overrides it. Matches the @@ -3990,11 +4000,10 @@ where } // Retained log before the frontier restore, so the restore maxes against // the offsets the log proved rather than the zeroes of an empty one. - // `restore_offset_frontier` STORES `recovered_end` once past its guard, so - // it can lower `dirty_offset`; harmless only because `write_superblock` - // maxes the recorded frontier against `offset_frontier()`. The order also - // keeps that restore's precondition (`should_increment_offset` already set - // by a recovered offset space) meaningful. + // `restore_offset_frontier` takes each counter's own max against what is + // already loaded, so neither can be lowered here -- but only if the log is + // adopted first, or those maxes are taken against the zeroes of a + // partition that has not got its offsets back yet. if let Some(state) = retained { partition.adopt_retained_log(state); // OPT-IN, off by default: it models durability Iggy does not have. @@ -4027,6 +4036,12 @@ where // the restore at all, a simulator replica rebuilt against a retained // store resumes minting at 0 while its group is at N. partition.restore_offset_frontier(recovered_state.as_ref()); + // And the chain transition that restore obliges, which production's boot + // does through `reanchor_to_offset_frontier`. A restored counter can sit + // a lease block above the chain, and leaving the tail named below it puts + // the next mint inside a segment -- a shape boot never produces, so the + // harness would be modelling something the server cannot reach. + partition.reanchor_in_memory_to_mint_frontier(partitions.config().segment_size); partitions.insert(namespace, partition); if self.redispatch_parked_frames(namespace, epoch) { // This mutation occurs outside the pump, unlike production's @@ -6817,40 +6832,49 @@ where // partitions-plane borrow is held across the tick `.await`. namespace_scratch.extend(partitions.namespaces().copied()); - // Pre-pass: issue every group's pending superblock persist - // CONCURRENTLY. A cluster-wide view change makes every group on - // this shard need one in the same tick, and each `atomic_replace` - // is a create + write + 2 fsyncs; run serially, a few hundred - // groups on ordinary storage exceed the 5s view-change escalation - // and loop elections. The persists are independent (each group owns - // its store, lock, and failure bookkeeping, all behind `&self`), - // and the per-group loop below re-checks the gate on its lock-free - // fast path, so gating semantics are unchanged. + // Pre-pass: issue every group's pending superblock write CONCURRENTLY. + // A cluster-wide view change makes every group on this shard need one in + // the same tick, and each `atomic_replace` is a create + write + 2 + // fsyncs; run serially, a few hundred groups on ordinary storage exceed + // the 5s view-change escalation and loop elections. The writes are + // independent (each group owns its store, lock, and failure bookkeeping, + // all behind `&self`), and the per-group loop below re-checks the persist + // gate on its lock-free fast path, so gating semantics are unchanged. + // + // The offset-reservation extension rides the same pre-pass, which is the + // whole point of it being here: the append fence writes the superblock + // INLINE in this pump, where those two fsyncs delay the tick above for + // every group on the core. Extending at half a block of headroom keeps + // the fence on its lock-free fast path under load, so the write happens + // here instead of in front of a produce. Ordered BEFORE the persist + // because any write marks the view durable, so one write can satisfy + // both and the persist gate below then finds nothing to do. let pending_persists: Vec<_> = namespace_scratch .iter() .copied() .filter(|namespace| { - partitions - .get_by_ns(namespace) - .is_some_and(|partition| partition.consensus().needs_superblock_persist()) + partitions.get_by_ns(namespace).is_some_and(|partition| { + partition.consensus().needs_superblock_persist() + || partition.needs_offset_reservation_extension() + }) }) .map(|namespace| async move { if let Some(partition) = partitions.get_by_ns(&namespace) { - // The only dropped durability verdict in the tree: this pre-pass - // exists to coalesce the writes, and the per-group loop below re-runs - // the same gate on its lock-free fast path and withholds every - // view-scoped send when it fails, so the verdict here is redundant - // rather than ignored. + // Verdicts dropped on purpose. The reservation is backstopped + // by the fence at the mint, which refuses the append if the + // ceiling never caught up; and the persist gate is re-run by + // the per-group loop below on its lock-free fast path, which + // withholds every view-scoped send when it fails. + if partition.needs_offset_reservation_extension() { + let _ = partition.extend_offset_reservation().await; + } let _ = partition.persist_superblock_if_needed().await; } }) .collect(); - // Capped fan-out: each persist is a create + write + 2 fsyncs, and a - // node-wide view change over many partitions must not dump an - // unbounded fd/fsync burst onto the reactor in one tick. let mut pending_persists = pending_persists.into_iter(); loop { - let chunk: Vec<_> = pending_persists.by_ref().take(16).collect(); + let chunk: Vec<_> = pending_persists.by_ref().take(SUPERBLOCK_FAN_OUT).collect(); if chunk.is_empty() { break; } @@ -6880,6 +6904,27 @@ where } continue; } + // Same bound the metadata plane fail-stops on, applied per group, + // and it exits the NODE rather than fencing the group: a partition + // whose superblock keeps refusing withholds every view-scoped send, + // and on a solo group refuses every append too, so it serves nothing + // while the process still reports healthy. + let superblock_failures = partition.superblock_write_failures(); + if superblock_wedged( + superblock_failures, + self.superblock_wedged_fatal_failures.get(), + ) { + consensus::fatal( + FatalReason::SuperblockWedged, + &format!( + "partition superblock persist failed {superblock_failures} consecutive \ + times for namespace {}, past the [cluster] \ + superblock_wedged_fatal_timeout window; exiting so a supervisor handles \ + the wedge instead of the replica limping fenced", + namespace.inner() + ), + ); + } let consensus = partition.consensus(); // Only while a view change is live. A `Normal` tick has no consumer: @@ -7103,6 +7148,7 @@ where partitions = namespaces.len(), "shutdown flush: draining committed journals to segment storage" ); + let mut collapse_pending = Vec::new(); for namespace in namespaces { let Some(partition) = partitions.get_mut_by_ns(&namespace) else { continue; @@ -7121,7 +7167,45 @@ where // after this flush). A partition already fenced by the commit // path keeps its original fault. partition.fence_flush_failure(); + // The collapse claims the segments account for every confirmed + // offset, which a failed flush is exactly the case against, so + // leave the reservation standing. + continue; + } + collapse_pending.push(namespace); + } + + // Collapsed CONCURRENTLY, for the same reason the tick coalesces its view + // persists: see [`SUPERBLOCK_FAN_OUT`]. Each group owns its store, lock + // and failure bookkeeping, all behind `&self`. + // + // The flushes above stay serial: they take `&mut`, and the writers they + // drive are the shard's, not the partition's. + let mut pending = collapse_pending + .into_iter() + .map(|namespace| async move { + // The segments now prove where the offset space ends, so the + // reservation has nothing left to witness. Without the collapse + // every clean stop would leave a lease-block-wide hole. + let Some(partition) = partitions.get_by_ns(&namespace) else { + return; + }; + if !partition.collapse_offset_reservation().await { + tracing::warn!( + namespace_raw = namespace.inner(), + "could not collapse the offset reservation on shutdown; the restart \ + will resume above it and leave a gap in the offset space" + ); + } + }) + .collect::>() + .into_iter(); + loop { + let chunk: Vec<_> = pending.by_ref().take(SUPERBLOCK_FAN_OUT).collect(); + if chunk.is_empty() { + break; } + futures::future::join_all(chunk).await; } } From be2265c9ea132887a0b3414d43e431ed66d51cfd Mon Sep 17 00:00:00 2001 From: Hubert Gruszecki Date: Fri, 4 Sep 2026 18:08:43 +0200 Subject: [PATCH 061/182] refactor(server): make dispatch routing and failure paths explicit (#4036) --- .../tcp_test/offset_feature_deserialize.go | 9 +- .../tests/server/legacy_login_vsr.rs | 53 +- core/integration/tests/server/mod.rs | 7 + .../tests/server/poll_semantics_vsr.rs | 57 +- core/integration/tests/server/raw_tcp.rs | 207 +++ .../tests/server/unknown_code_vsr.rs | 92 + core/metadata/src/impls/metadata.rs | 18 +- core/server/src/boot/mod.rs | 126 +- core/server/src/boot/recovery.rs | 27 +- core/server/src/boot/threads.rs | 601 +++++- core/server/src/dispatch/authz.rs | 309 ++-- core/server/src/dispatch/failure.rs | 629 +++++++ core/server/src/dispatch/mod.rs | 1638 +++++++++-------- core/server/src/dispatch/partition.rs | 569 +++--- core/server/src/dispatch/reads.rs | 101 +- core/server/src/dispatch/session_ops.rs | 282 +-- core/server/src/dispatch/test_support.rs | 13 +- core/server/src/http/handlers.rs | 13 +- core/server/src/http/submit.rs | 23 +- core/server/src/lib.rs | 1 + core/server/src/responses.rs | 11 +- core/server/src/rewrite.rs | 570 ++++++ core/server/src/server_error.rs | 21 + core/server/src/session_manager.rs | 106 +- core/simulator/src/replica.rs | 2 +- 25 files changed, 3923 insertions(+), 1562 deletions(-) create mode 100644 core/integration/tests/server/raw_tcp.rs create mode 100644 core/integration/tests/server/unknown_code_vsr.rs create mode 100644 core/server/src/dispatch/failure.rs create mode 100644 core/server/src/rewrite.rs diff --git a/bdd/go/tests/tcp_test/offset_feature_deserialize.go b/bdd/go/tests/tcp_test/offset_feature_deserialize.go index 28bb0b89c8..229d200ddb 100644 --- a/bdd/go/tests/tcp_test/offset_feature_deserialize.go +++ b/bdd/go/tests/tcp_test/offset_feature_deserialize.go @@ -21,6 +21,7 @@ import ( "context" iggcon "github.com/apache/iggy/foreign/go/contracts" + ierror "github.com/apache/iggy/foreign/go/errors" "github.com/onsi/ginkgo/v2" "github.com/onsi/gomega" ) @@ -175,7 +176,10 @@ var _ = ginkgo.Describe("GET CONSUMER OFFSET:", func() { consumer := iggcon.NewGroupConsumer(randomU32Identifier()) partitionId := uint32(1) - offset, err := client.GetConsumerOffset( + // An unresolved stream is an addressing error, not a fresh + // consumer: a nil offset would let a typo'd or deleted target read + // back as "no offset stored" and silently reprocess. + _, err := client.GetConsumerOffset( context.Background(), consumer, randomU32Identifier(), @@ -183,8 +187,7 @@ var _ = ginkgo.Describe("GET CONSUMER OFFSET:", func() { &partitionId, ) - itShouldNotReturnError(err) - itShouldReturnNilOffsetForNewConsumerGroup(offset) + itShouldReturnSpecificError(err, ierror.ErrStreamIdNotFound) }) }) diff --git a/core/integration/tests/server/legacy_login_vsr.rs b/core/integration/tests/server/legacy_login_vsr.rs index 98b685e1b3..f0a88c1342 100644 --- a/core/integration/tests/server/legacy_login_vsr.rs +++ b/core/integration/tests/server/legacy_login_vsr.rs @@ -27,19 +27,15 @@ //! TCP socket: a header-only non-replicated frame carrying the code in the //! reserved command slot. -use iggy_binary_protocol::HEADER_SIZE; +use iggy_binary_protocol::EvictionReason; use iggy_binary_protocol::codes::{LOGIN_USER_CODE, LOGIN_WITH_PERSONAL_ACCESS_TOKEN_CODE}; -use iggy_binary_protocol::consensus::{Command, Operation, RequestHeader}; +use iggy_binary_protocol::consensus::Command; use integration::harness::TestHarness; use integration::iggy_harness; -use std::mem::offset_of; -use std::time::Duration; -use tokio::io::{AsyncReadExt, AsyncWriteExt}; -use tokio::net::TcpStream; -use tokio::time::timeout; -// Wire byte pinned to `EvictionReason::MalformedLogin` in consensus::header. -const EVICTION_REASON_MALFORMED_LOGIN: u8 = 15; +use crate::server::raw_tcp::{ + connect, eviction_reason, frame_command, non_replicated_header, read_frame_header, write_frame, +}; #[iggy_harness] async fn given_legacy_login_user_code_when_sent_raw_should_evict_malformed_login( @@ -60,45 +56,26 @@ async fn given_legacy_pat_login_code_when_sent_raw_should_evict_malformed_login( /// eviction. The reject runs before the session gate, so this unbound socket /// exercises the same path a bound connection would. async fn assert_legacy_login_code_evicted(harness: &TestHarness, code: u32) { - let mut header = RequestHeader { - command: Command::Request, - operation: Operation::NonReplicated, - size: u32::try_from(HEADER_SIZE).unwrap(), - // NonReplicated leaves session / request unchecked, but the header - // validator still requires a nonzero client id. - client: 0xC0FFEE, - session: 0, - request: 0, - ..Default::default() - }; - // A non-replicated command code travels in the first 4 reserved bytes. - header.reserved[..4].copy_from_slice(&code.to_le_bytes()); + // NonReplicated leaves session / request unchecked, but the header + // validator still requires a nonzero client id. + let header = non_replicated_header(0xC0FFEE, 0, 0, code); - let addr = harness - .server() - .tcp_addr() - .expect("server must expose a TCP address"); - let mut stream = TcpStream::connect(addr).await.unwrap(); - stream.write_all(bytemuck::bytes_of(&header)).await.unwrap(); + let mut stream = connect(harness).await; + write_frame(&mut stream, &header, &[]).await; - // Eviction is header-only: exactly 256 bytes. The timeout makes the + // Eviction is header-only: exactly 256 bytes. The bounded read makes the // fail-fast contract explicit -- a regression that silently drops the // frame trips this instead of hanging until the test wall clock. - let mut reply = [0u8; HEADER_SIZE]; - timeout(Duration::from_secs(5), stream.read_exact(&mut reply)) - .await - .expect("server must answer a legacy login code within 5s, not stall") - .expect("reading the eviction frame must succeed"); + let reply = read_frame_header(&mut stream).await; - let command_offset = offset_of!(RequestHeader, command); assert_eq!( - reply[command_offset], + frame_command(&reply), Command::Eviction as u8, "expected an Eviction frame for legacy login code {code}, not a Reply" ); assert_eq!( - reply[HEADER_SIZE - 1], - EVICTION_REASON_MALFORMED_LOGIN, + eviction_reason(&reply), + EvictionReason::MalformedLogin as u8, "legacy login code {code} must evict with MalformedLogin" ); } diff --git a/core/integration/tests/server/mod.rs b/core/integration/tests/server/mod.rs index 1c6411a8aa..e32f2c5ab4 100644 --- a/core/integration/tests/server/mod.rs +++ b/core/integration/tests/server/mod.rs @@ -21,9 +21,16 @@ mod a2a_jwt; mod cg; // Flush (FLUSH_UNSAVED_BUFFER) has no the server primitive; it must deny typed. mod flush_vsr; +// Raw TCP framing (connect, hand-crafted frames, root register) for the +// server suites that send what the SDK cannot. +pub(crate) mod raw_tcp; // Legacy login codes (LOGIN_USER / LOGIN_WITH_PAT) have no the server handler; // they must evict typed (MalformedLogin), not stall or reply empty-ok. mod legacy_login_vsr; +// A non-replicated code no read serves (unknown, or table-listed without an +// arm) must deny typed (InvalidCommand) at the read gate, not stall or reply +// empty-ok. +mod unknown_code_vsr; // A failed credential login must report the credential failure, not the // payload shape it fell through to. mod login_credentials_vsr; diff --git a/core/integration/tests/server/poll_semantics_vsr.rs b/core/integration/tests/server/poll_semantics_vsr.rs index a3a90f06d0..2430397424 100644 --- a/core/integration/tests/server/poll_semantics_vsr.rs +++ b/core/integration/tests/server/poll_semantics_vsr.rs @@ -19,7 +19,7 @@ //! topic does not have must surface a typed `PartitionNotFound`, not an empty //! poll a consumer would read as end-of-partition; a poll whose stream or //! topic does not resolve must surface the legacy `StreamIdNotFound` / -//! `TopicIdNotFound` the same way; the partition addressing error on +//! `TopicIdNotFound` the same way; the same three addressing errors on //! `get_consumer_offset` must not decode as "no offset stored"; and a //! timestamp poll must be at-or-after, including the message stamped exactly at //! the queried timestamp (the timestamp replies report per message). @@ -203,6 +203,61 @@ async fn given_missing_partition_when_getting_consumer_offset_should_reject_part ); } +/// The same swallow one level up: an unresolved STREAM or TOPIC also answered +/// the empty body, so a typo'd or deleted target read back as a fresh +/// consumer. The consumer then resumes from its configured default and +/// silently reprocesses, with no error anywhere. The poll path denies both +/// codes, and the identical read over REST 404s. +#[iggy_harness( + test_client_transport = [Tcp] +)] +async fn given_missing_stream_when_getting_consumer_offset_should_reject_stream_not_found( + harness: &TestHarness, +) { + let client = harness.tcp_root_client().await.expect("tcp root client"); + let stream_id = Identifier::from_str_value("no-such-stream").expect("stream identifier"); + let topic_id = Identifier::from_str_value("no-such-topic").expect("topic identifier"); + + let result = client + .get_consumer_offset(&Consumer::default(), &stream_id, &topic_id, Some(0)) + .await; + + let expected = IggyError::StreamIdNotFound(Identifier::default()).as_code(); + assert!( + matches!(&result, Err(error) if error.as_code() == expected), + "get_consumer_offset on a missing stream must surface Err(StreamIdNotFound), \ + got {result:?}" + ); +} + +#[iggy_harness( + test_client_transport = [Tcp] +)] +async fn given_missing_topic_when_getting_consumer_offset_should_reject_topic_not_found( + harness: &TestHarness, +) { + let client = harness.tcp_root_client().await.expect("tcp root client"); + client + .create_stream("offset-topicless-stream") + .await + .expect("create stream"); + let stream_id = + Identifier::from_str_value("offset-topicless-stream").expect("stream identifier"); + let topic_id = Identifier::from_str_value("no-such-topic").expect("topic identifier"); + + let result = client + .get_consumer_offset(&Consumer::default(), &stream_id, &topic_id, Some(0)) + .await; + + let expected = + IggyError::TopicIdNotFound(Identifier::default(), Identifier::default()).as_code(); + assert!( + matches!(&result, Err(error) if error.as_code() == expected), + "get_consumer_offset on a missing topic of an existing stream must surface \ + Err(TopicIdNotFound), got {result:?}" + ); +} + #[iggy_harness( test_client_transport = [Tcp] )] diff --git a/core/integration/tests/server/raw_tcp.rs b/core/integration/tests/server/raw_tcp.rs new file mode 100644 index 0000000000..cfe5d8c698 --- /dev/null +++ b/core/integration/tests/server/raw_tcp.rs @@ -0,0 +1,207 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Raw TCP framing for the server suites that hand-craft client frames the +//! SDK cannot emit: connect to the harness server, write one request frame, +//! read the header the server answers with, and register root so a frame +//! can ride a bound session. + +use std::mem::offset_of; +use std::time::Duration; + +use iggy::prelude::*; +use iggy_binary_protocol::codec::{WireDecode, WireEncode}; +use iggy_binary_protocol::consensus::{ + Command, Operation, ReplyHeader, RequestHeader, read_size_field, result_code, + result_section_len, +}; +use iggy_binary_protocol::requests::users::LoginRegisterRequest; +use iggy_binary_protocol::responses::users::LoginRegisterResponse; +use iggy_binary_protocol::{ + ClientVersionInfo, EvictionHeader, HEADER_SIZE, IGGY_PROTOCOL_VERSION, WireName, +}; +use integration::harness::TestHarness; +use secrecy::SecretString; +use tokio::io::{AsyncReadExt, AsyncWriteExt}; +use tokio::net::TcpStream; +use tokio::time::{Instant, sleep, timeout}; + +/// Per-frame reply wait. A server that drops the frame answers nothing at +/// all, so an unanswered read is a verdict, not a reason to wait longer. +const REPLY_WAIT: Duration = Duration::from_secs(5); + +/// Budget for the register to commit: right after boot the single node may +/// still be electing itself and answers transient rejections meanwhile. +const COMMIT_BUDGET: Duration = Duration::from_secs(15); + +const RETRY_PAUSE: Duration = Duration::from_millis(100); + +pub(crate) async fn connect(harness: &TestHarness) -> TcpStream { + let addr = harness + .server() + .tcp_addr() + .expect("server must expose a TCP address"); + TcpStream::connect(addr).await.unwrap() +} + +pub(crate) fn request_header( + operation: Operation, + client: u128, + session: u64, + request: u64, + body_len: usize, +) -> RequestHeader { + RequestHeader { + command: Command::Request, + operation, + size: u32::try_from(HEADER_SIZE + body_len).unwrap(), + client, + session, + request, + ..Default::default() + } +} + +/// A header-only `NonReplicated` frame; the command code travels in the +/// first 4 reserved bytes. +pub(crate) fn non_replicated_header( + client: u128, + session: u64, + request: u64, + code: u32, +) -> RequestHeader { + let mut header = request_header(Operation::NonReplicated, client, session, request, 0); + header.reserved[..4].copy_from_slice(&code.to_le_bytes()); + header +} + +pub(crate) async fn write_frame(stream: &mut TcpStream, header: &RequestHeader, body: &[u8]) { + stream.write_all(bytemuck::bytes_of(header)).await.unwrap(); + if !body.is_empty() { + stream.write_all(body).await.unwrap(); + } +} + +/// Read the header of the next server frame, within [`REPLY_WAIT`]. +pub(crate) async fn read_frame_header(stream: &mut TcpStream) -> [u8; HEADER_SIZE] { + let mut header = [0u8; HEADER_SIZE]; + timeout(REPLY_WAIT, stream.read_exact(&mut header)) + .await + .expect("server must answer within the reply wait, not stall") + .expect("reply header read failed"); + header +} + +/// Write one frame and read one Reply off the lockstep connection, the body +/// sized by the reply's size field. +pub(crate) async fn exchange( + stream: &mut TcpStream, + header: &RequestHeader, + body: &[u8], +) -> ([u8; HEADER_SIZE], Vec) { + write_frame(stream, header, body).await; + let reply_header = read_frame_header(stream).await; + let command = frame_command(&reply_header); + assert_eq!( + command, + Command::Reply as u8, + "expected a Reply frame, got command byte {command} (an Eviction carries reason {})", + eviction_reason(&reply_header) + ); + + let total_size = read_size_field(&reply_header).expect("reply size field") as usize; + // `read_size_field` is a bare 4-byte LE read with no floor (the SDK adds + // its own guard, which this helper cannot inherit), so an under-sized + // field would either panic on subtract overflow or wrap into a ~1.8e19 + // allocation and abort - either way hiding the regression under test. + assert!( + total_size >= HEADER_SIZE, + "reply size field {total_size} is below the {HEADER_SIZE}-byte header" + ); + let mut reply_body = vec![0u8; total_size - HEADER_SIZE]; + timeout(REPLY_WAIT, stream.read_exact(&mut reply_body)) + .await + .expect("reply body timed out") + .expect("reply body read failed"); + (reply_header, reply_body) +} + +pub(crate) fn frame_command(header: &[u8; HEADER_SIZE]) -> u8 { + header[offset_of!(RequestHeader, command)] +} + +/// The typed reason byte of an `Eviction` frame, read by field offset rather +/// than by a hardcoded index: `EvictionReason` is renumberable, so a literal +/// would keep passing while checking a different reason. +pub(crate) fn eviction_reason(header: &[u8; HEADER_SIZE]) -> u8 { + header[offset_of!(EvictionHeader, reason)] +} + +pub(crate) fn reply_status(reply_header: &[u8; HEADER_SIZE]) -> u32 { + let offset = offset_of!(ReplyHeader, status); + u32::from_le_bytes(reply_header[offset..offset + 4].try_into().unwrap()) +} + +/// Root login/register on the socket as `client`, replayed on transient +/// rejections until it commits; returns the bound session id. The session +/// binds to THIS transport connection server-side, so a frame that must ride +/// it has to reuse the stream. +pub(crate) async fn register_root(stream: &mut TcpStream, client: u128) -> u64 { + let body = LoginRegisterRequest { + version_info: ClientVersionInfo { + protocol_version: IGGY_PROTOCOL_VERSION, + sdk_name: WireName::new("raw-tcp").unwrap(), + sdk_version: WireName::new("0.0.1").unwrap(), + }, + username: WireName::new(DEFAULT_ROOT_USERNAME).unwrap(), + password: SecretString::from(DEFAULT_ROOT_PASSWORD), + client_context: None, + } + .to_bytes(); + let header = request_header(Operation::Register, client, 0, 0, body.len()); + let deadline = Instant::now() + COMMIT_BUDGET; + loop { + let (reply_header, reply_body) = exchange(stream, &header, &body).await; + // A pre-commit deny rides the status word; a committed verdict rides + // the result section that leads a Register reply body. + let code = match reply_status(&reply_header) { + 0 => result_code(&reply_body).expect("register reply must carry a result section"), + status => status, + }; + if code == 0 { + let payload_start = result_section_len(&reply_body).unwrap(); + let response = LoginRegisterResponse::decode_from(&reply_body[payload_start..]) + .expect("register payload must decode"); + assert_ne!(response.session, 0, "server must bind a nonzero session"); + return response.session; + } + assert!( + is_transient(code), + "register rejected with a terminal code {code}" + ); + assert!( + Instant::now() < deadline, + "register did not commit within {COMMIT_BUDGET:?}" + ); + sleep(RETRY_PAUSE).await; + } +} + +fn is_transient(code: u32) -> bool { + code == IggyError::TransientNotCommitted.as_code() + || code == IggyError::TransientNotAccepted.as_code() +} diff --git a/core/integration/tests/server/unknown_code_vsr.rs b/core/integration/tests/server/unknown_code_vsr.rs new file mode 100644 index 0000000000..622e5bac10 --- /dev/null +++ b/core/integration/tests/server/unknown_code_vsr.rs @@ -0,0 +1,92 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Armless non-replicated command codes against the server (vsr). The read +//! gate is total over the protocol command table, so a bound session sending +//! a `NonReplicated` header whose reserved command slot carries a code the +//! reads have no arm for must get a typed `InvalidCommand` deny Reply. Two +//! codes pin the two halves. One no table entry claims (the SDK forwards +//! unknown codes untouched, `COMMAND_TABLE` being a registry rather than a +//! capability list): the shared response builder's catch-all already denied +//! it `InvalidCommand`, so that test pins the pre-existing deny now that the +//! gate owns it. One the table lists but no read serves (`LOGOUT_USER`, which +//! the SDK only ever sends as `Operation::Logout`): the old gate fell open +//! and let the builder acknowledge it empty-ok, as if a logout had happened, +//! so that test pins the closed fail-open. The frames are hand-crafted on a +//! raw TCP socket to pin the status word the SDK maps through +//! `IggyError::from_code`. + +use iggy::prelude::*; +use iggy_binary_protocol::codes::LOGOUT_USER_CODE; +use iggy_binary_protocol::lookup_command; +use integration::harness::TestHarness; +use integration::iggy_harness; + +use crate::server::raw_tcp::{ + connect, exchange, non_replicated_header, register_root, reply_status, +}; + +/// A code no `COMMAND_TABLE` entry claims. +const UNKNOWN_CODE: u32 = 9999; + +/// The header validator requires a nonzero client id; the value is otherwise +/// free since nothing here reconnects. +const CLIENT_ID: u128 = 0xBAD_C0DE; + +#[iggy_harness] +async fn given_bound_session_when_unknown_non_replicated_code_sent_should_deny_invalid_command( + harness: &TestHarness, +) { + assert!( + lookup_command(UNKNOWN_CODE).is_none(), + "test needs a code absent from COMMAND_TABLE" + ); + assert_non_replicated_code_denied_invalid_command(harness, UNKNOWN_CODE).await; +} + +#[iggy_harness] +async fn given_bound_session_when_table_listed_code_without_read_arm_sent_should_deny_invalid_command( + harness: &TestHarness, +) { + assert!( + lookup_command(LOGOUT_USER_CODE).is_some_and(|meta| !meta.is_replicated()), + "test needs a non-replicated COMMAND_TABLE entry" + ); + assert_non_replicated_code_denied_invalid_command(harness, LOGOUT_USER_CODE).await; +} + +/// Register root on a raw socket, send a header-only `NonReplicated` frame +/// carrying `code` in the reserved command slot on that bound connection, and +/// assert the server answers with an `InvalidCommand` deny Reply. +async fn assert_non_replicated_code_denied_invalid_command(harness: &TestHarness, code: u32) { + let mut stream = connect(harness).await; + let session = register_root(&mut stream, CLIENT_ID).await; + + let header = non_replicated_header(CLIENT_ID, session, 1, code); + let (reply_header, reply_body) = exchange(&mut stream, &header, &[]).await; + + assert_eq!( + reply_status(&reply_header), + IggyError::InvalidCommand.as_code(), + "a bound session sending non-replicated code {code} must be denied InvalidCommand" + ); + assert!( + reply_body.is_empty(), + "a deny Reply carries an empty body, got {} bytes", + reply_body.len() + ); +} diff --git a/core/metadata/src/impls/metadata.rs b/core/metadata/src/impls/metadata.rs index dc2399ecf2..726a013c13 100644 --- a/core/metadata/src/impls/metadata.rs +++ b/core/metadata/src/impls/metadata.rs @@ -736,8 +736,13 @@ pub struct IggyMetadata { /// policy. superblock_write_failures: Cell, superblock_retry_after_micros: Cell, - /// State machine - lives on all shards - pub mux_stm: M, + /// State machine - lives on all shards. + /// + /// Shared so shard 0's bootstrap can keep a clone alive past every + /// fallible step that owns this struct: the peer shards read through + /// handles minted off this writer, and dropping it makes their + /// `LeftRight::read` panic. See `server/src/boot::shard_main`. + pub mux_stm: Rc, pub allocator: ConsensusGroupAllocator, /// Snapshot coordinator - present when persistent checkpointing is configured. pub coordinator: Option>, @@ -809,9 +814,10 @@ where journal: Option, snapshot: Option, superblock: Option>, - mux_stm: M, + mux_stm: impl Into>, data_dir: Option, ) -> Self { + let mux_stm = mux_stm.into(); let allocator = ConsensusGroupAllocator::new(mux_stm.streams().highest_partition_consensus_group_id()); let coordinator = data_dir.map(|dir| SnapshotCoordinator::new(dir, IggySnapshot::create)); @@ -2860,7 +2866,7 @@ where // Normal op: apply SM, commit_reply. `Err` is decode/corruption // only; a business rejection commits as a deterministic no-op // whose `code` rides the reply body, replayed on retry. - let apply = gated_apply(&self.mux_stm, prepare).unwrap_or_else(|err| { + let apply = gated_apply(&*self.mux_stm, prepare).unwrap_or_else(|err| { panic!( "on_ack: committed metadata op={} failed to apply: {err}", prepare_header.op @@ -3271,7 +3277,7 @@ where // (see the phantom-op comment at the call site). let client_table = self.client_table.borrow().to_snapshot(); let checksum = match coordinator.persist_snapshot( - &self.mux_stm, + &*self.mux_stm, snap_op, created_at, Some(client_table), @@ -3632,7 +3638,7 @@ where // table, while their state-machine effects still have to replay // (the snapshot sits at a lower op). apply_committed_prepare( - &self.mux_stm, + &*self.mux_stm, &self.client_table, self.client_table_mutation_allowed(header.op), |operation| self.fire_commit_notifier(operation), diff --git a/core/server/src/boot/mod.rs b/core/server/src/boot/mod.rs index 5053c8f8dc..fc2de43979 100644 --- a/core/server/src/boot/mod.rs +++ b/core/server/src/boot/mod.rs @@ -47,12 +47,12 @@ use crate::boot::listeners::{ make_replica_delegation_fns, make_shard_zero_client_accept_fns, start_tcp_runtime, }; use crate::boot::recovery::{ - RecoveredOwnerState, build_shard_for_thread, restore_metadata_consensus, + RecoveredOwnerState, ShardBuild, build_shard_for_thread, restore_metadata_consensus, }; use crate::boot::threads::{ - StopSignals, await_pump_drain, install_panic_hook, join_partial_shard_survivors, - resolve_shard_assignments, run_shard_thread, spawn_shutdown_watchdog, - validate_sharding_runtime_knobs, + PeerExitCountdown, PeerExitWait, ShutdownDeadline, StopSignals, await_pump_drain, + install_panic_hook, join_partial_shard_survivors, resolve_shard_assignments, run_shard_thread, + spawn_shutdown_watchdog, validate_sharding_runtime_knobs, }; use crate::boot::topology::{RosterCells, resolve_tcp_topology}; use crate::dispatch::partition::make_partition_read_handler; @@ -60,8 +60,8 @@ use crate::dispatch::reads::read_frontier_budget; use crate::dispatch::session_ops::warm_dummy_password_hash; use crate::dispatch::submit::make_metadata_submit_handler; use crate::dispatch::{ - make_client_request_handler, make_deferred_client_request_handler, - make_deferred_replica_message_handler, make_list_clients_handler, + make_deferred_client_request_handler, make_deferred_replica_message_handler, + make_list_clients_handler, }; use crate::server_error::ServerError; use crate::session_manager::SessionManager; @@ -278,6 +278,23 @@ pub fn bootstrap( // thread body or in a task compio's `spawn` would swallow, escapes it. let first_panic = install_panic_hook(Arc::clone(&shutdown_flag)); let config = Arc::new(config); + // One post-shutdown budget for the whole process: the main thread's + // shard joins and shard 0's peer wait both count down this instant + // instead of each arming a full `shutdown_join_timeout`. The floor keeps a + // spent join budget from releasing the metadata writer while a peer still + // reads through it, so it covers a peer's whole exit: one poll interval to + // observe the shutdown flag, then one drain budget to drain. Flooring at + // the drain alone is short by a poll interval, and + // `shutdown_join_timeout == shutdown_drain_timeout` is a legal config. + let shutdown_deadline = Arc::new(ShutdownDeadline::new( + config.system.sharding.shutdown_join_timeout.get_duration(), + config + .system + .sharding + .shutdown_drain_timeout + .get_duration() + .saturating_add(config.system.sharding.shutdown_poll_interval.get_duration()), + )); // One owner table per server process, Arc-cloned into every shard's bus so // any shard's bus reads the same atomic slots that the owning // shard's installer / disconnect path writes. @@ -302,6 +319,11 @@ pub fn bootstrap( // count so a sender never blocks (each peer sends exactly once). let (ready_tx, ready_rx) = crossfire::mpmc::bounded_async::(metadata_peers); + // Shard 0 blocks on this at exit until every peer thread is done (see + // `PeerExitCountdown`). Sized to the real peer count, not the clamped + // channel capacity above: a single-shard server must not wait at all. + let peer_exit = Arc::new(PeerExitCountdown::new(shards_count.saturating_sub(1))); + let mut shard_threads: Vec<(u16, thread::JoinHandle>)> = Vec::with_capacity(shards_count); let roster_cells = RosterCells::default(); @@ -354,6 +376,8 @@ pub fn bootstrap( let roster_cells_for_shard = roster_cells.clone(); let applied_frontier_for_shard = Arc::clone(&metadata_applied_frontier); let shard_metrics_for_shard = shard_metrics_all.clone(); + let peer_exit_for_shard = Arc::clone(&peer_exit); + let shutdown_deadline_for_shard = Arc::clone(&shutdown_deadline); let handle = match thread::Builder::new() .name(format!("shard-{shard_id}")) .spawn(move || -> Result<(), ServerError> { @@ -373,6 +397,8 @@ pub fn bootstrap( roster_cells_for_shard, applied_frontier_for_shard, shard_metrics_for_shard, + peer_exit_for_shard, + shutdown_deadline_for_shard, ) }) { Ok(handle) => handle, @@ -389,10 +415,14 @@ pub fn bootstrap( drop(metadata_bundle_rx); drop(ready_tx); drop(ready_rx); - join_partial_shard_survivors( - shard_threads, - config.system.sharding.shutdown_join_timeout.get_duration(), + // Shard 0 spawns first, so the vec holds it plus every peer + // that made it; the rest are peers shard 0's exit wait would + // otherwise sit out its whole budget for. + let spawned_peers = shard_threads.len().saturating_sub(1); + peer_exit.peers_never_spawned( + shards_count.saturating_sub(1).saturating_sub(spawned_peers), ); + join_partial_shard_survivors(shard_threads, &shutdown_deadline); return Err(ServerError::ShardSpawnFailed { shard_id, source }); } }; @@ -416,7 +446,7 @@ pub fn bootstrap( Ok(ShardHandles { shutdown_flag, shard_threads, - join_timeout: config.system.sharding.shutdown_join_timeout.get_duration(), + deadline: shutdown_deadline, first_panic, }) } @@ -441,6 +471,8 @@ async fn shard_main( roster_cells: RosterCells, metadata_applied_frontier: Arc, shard_metrics_all: Vec, + peer_exit: Arc, + shutdown_deadline: Arc, ) -> Result<(), ServerError> { let topology = resolve_tcp_topology(config, replica_id)?; let bus = Rc::new(IggyMessageBus::with_config_and_owner_table( @@ -478,7 +510,12 @@ async fn shard_main( // metadata VSR; per-commit `publish()` (in `WriteCell::apply`) // bounds reader staleness to one op. let data_dir = Path::new(&config.system.path); - let (mux_stm, owner_state) = match metadata_handoff { + // The bundle broadcast is deliberately NOT inside the owner arm: it is + // the first moment a peer can hold a read handle over shard 0's writer, + // so the writer must first be parked in a binding that outlives the peer + // wait armed below. A `recover()` failure inside the arm is safe for the + // same reason -- no peer holds a handle yet. + let (mux_stm, pending_bundle_tx, owner_state) = match metadata_handoff { MetadataHandoff::Owner { bundle_tx } => { // Root is created locally at boot (never journaled), so replay // must start from the same baseline or every WAL-created user @@ -520,17 +557,9 @@ async fn shard_main( } } }); - broadcast_metadata_bundle( - shard_id, - &bundle_tx, - recovered.mux_stm.factory_bundle(), - total_shards.saturating_sub(1), - &shutdown_flag_for_handoff, - poll_interval, - ) - .await?; ( - recovered.mux_stm, + Rc::new(recovered.mux_stm), + Some(bundle_tx), Some(RecoveredOwnerState { journal: recovered.journal, snapshot: recovered.snapshot, @@ -551,10 +580,40 @@ async fn shard_main( poll_interval, ) .await?; - (ServerMuxStateMachine::from_factory_bundle(bundle), None) + ( + Rc::new(ServerMuxStateMachine::from_factory_bundle(bundle)), + None, + None, + ) } }; + // Shard 0 owns the metadata state machine's only write handle, and the + // peers read through it until their runtimes are gone. Declared after + // `mux_stm` so it drops first: every exit from here on, clean or `?`, + // waits for the peers before the write side goes. The `Rc` above is what + // makes that hold -- the handle no longer travels by value into the + // fallible shard build, which would drop it inside the callee. + let _peer_exit_wait = (shard_id == 0).then(|| { + PeerExitWait::new( + peer_exit, + Arc::clone(&shutdown_flag_for_handoff), + shutdown_deadline, + ) + }); + + if let Some(bundle_tx) = pending_bundle_tx { + broadcast_metadata_bundle( + shard_id, + &bundle_tx, + mux_stm.factory_bundle(), + total_shards.saturating_sub(1), + &shutdown_flag_for_handoff, + poll_interval, + ) + .await?; + } + // Metadata consensus + journal + snapshot live only on shard 0. // `IggyShard::tick_metadata` short-circuits when `consensus.is_none()`, // so peer shards have no caller that reads `journal` or `snapshot`. @@ -589,7 +648,7 @@ async fn shard_main( journal_for_metadata, snapshot_for_metadata, superblock_for_metadata, - mux_stm, + Rc::clone(&mux_stm), Some(PathBuf::from(&config.system.path)), ) .with_applied_frontier(metadata_applied_frontier); @@ -625,7 +684,12 @@ async fn shard_main( // Heap-pin like `run_shard_thread` pins `shard_main`: the builder future // carries the whole shard construction state machine and outgrew clippy's // `large_futures` cap; one allocation per shard startup. - let (shard, sessions) = Box::pin(build_shard_for_thread( + let ShardBuild { + shard, + sessions, + on_client_request, + shard_handle, + } = Box::pin(build_shard_for_thread( shard_id, total_shards, config, @@ -892,11 +956,15 @@ async fn shard_main( boot_view, shard.plane.metadata().client_table.borrow().client_ids(), ); - let on_client_request = make_client_request_handler( - &shard, - &sessions, - Arc::clone(&config.system), - config.personal_access_token.max_tokens_per_user, + // The request handler strands every frame until the weak + // self-reference is backfilled, so the build must have done that + // before the first listener binds. + debug_assert!( + shard_handle + .borrow() + .as_ref() + .is_some_and(|weak| weak.upgrade().is_some()), + "shard self-reference must be backfilled before listeners bind" ); let (accepted_replica, dialed_replica) = make_replica_delegation_fns(Rc::clone(&coord), &bus); diff --git a/core/server/src/boot/recovery.rs b/core/server/src/boot/recovery.rs index eb99734d69..5d21e1f7c5 100644 --- a/core/server/src/boot/recovery.rs +++ b/core/server/src/boot/recovery.rs @@ -23,7 +23,8 @@ use crate::partition_helpers::load_partition_or_fence; use crate::server_error::ServerError; use crate::session_manager::SessionManager; use crate::shell::{ - ServerMetadata, ServerShard, ShellHandlers, consensus_timers, repair_retry_ticks, + ServerMetadata, ServerShard, ShellHandlers, ShellShardHandle, consensus_timers, + repair_retry_ticks, }; use configs::server::ServerConfig; use consensus::{ @@ -35,6 +36,7 @@ use journal::Journal; use journal::prepare_journal::PrepareJournal; use journal::superblock::PingPongSuperblock; use message_bus::IggyMessageBus; +use message_bus::client_listener::RequestHandler; use metadata::impls::metadata::{IggySnapshot, StreamsFrontend}; use metadata::stm::snapshot::Snapshot; use partitions::{IggyPartitions, PartitionsConfig}; @@ -52,6 +54,18 @@ use std::sync::Arc; use std::time::Duration; use tracing::{error, info, warn}; +/// A shard built for its thread, with what `shard_main` wires after the +/// build: the session manager its request plane shares, the shard's one +/// client-request handler (shard 0 hands the same instance to its local +/// transports), and the weak self-reference the deferred handlers +/// upgrade per frame, already backfilled. +pub(in crate::boot) struct ShardBuild { + pub shard: Rc, + pub sessions: Rc>, + pub on_client_request: RequestHandler, + pub shard_handle: ShellShardHandle, PrepareJournal, IggySnapshot>, +} + #[allow(clippy::too_many_arguments, clippy::too_many_lines)] pub(in crate::boot) async fn build_shard_for_thread( shard_id: u16, @@ -65,7 +79,7 @@ pub(in crate::boot) async fn build_shard_for_thread( reply_inbox: ShardReceiver, metrics: ShardMetrics, roster_cells: &RosterCells, -) -> Result<(Rc, Rc>), ServerError> { +) -> Result { let shard_local_id = ShardId::new(shard_id); let total_partitions = metadata.mux_stm.streams().read(|inner| { inner @@ -243,7 +257,7 @@ pub(in crate::boot) async fn build_shard_for_thread( ShardIdentity::new(shard_id, shard_name), Rc::clone(&bus), on_replica_message, - on_client_request, + Rc::clone(&on_client_request), on_metadata_submit, on_list_clients, on_partition_read, @@ -289,7 +303,12 @@ pub(in crate::boot) async fn build_shard_for_thread( usize::try_from(config.message_bus.max_message_size.as_bytes_u64()).unwrap_or(usize::MAX), ); *shard_handle.borrow_mut() = Some(Rc::downgrade(&shard)); - Ok((shard, sessions)) + Ok(ShardBuild { + shard, + sessions, + on_client_request, + shard_handle, + }) } // Pin the configs-crate default literals (duplicated there to avoid a diff --git a/core/server/src/boot/threads.rs b/core/server/src/boot/threads.rs index 2a68d89b96..83c4e408c4 100644 --- a/core/server/src/boot/threads.rs +++ b/core/server/src/boot/threads.rs @@ -25,7 +25,8 @@ use crate::shard_allocator::{ShardAllocator, ShardInfo}; use compio::runtime::ResumeUnwind; use configs::server::ServerConfig; use configs::sharding::{ - INBOX_CAPACITY_MAX, SHUTDOWN_DRAIN_TIMEOUT_MAX, SHUTDOWN_POLL_INTERVAL_MAX, + INBOX_CAPACITY_MAX, RECONCILE_PERIODIC_INTERVAL_MAX, SHUTDOWN_DRAIN_TIMEOUT_MAX, + SHUTDOWN_JOIN_TIMEOUT_MAX, SHUTDOWN_POLL_INTERVAL_MAX, }; use message_bus::{IggyMessageBus, ReplicaOwnerTable}; use metadata::AppliedFrontier; @@ -36,7 +37,7 @@ use shard::{Receiver as ShardReceiver, Sender, ShardFrame, TaggedSender}; use std::backtrace::Backtrace; use std::rc::Rc; use std::sync::atomic::{AtomicBool, Ordering}; -use std::sync::{Arc, OnceLock}; +use std::sync::{Arc, Condvar, Mutex, OnceLock, PoisonError}; use std::time::{Duration, Instant}; use std::{panic, thread}; use tracing::{error, info, warn}; @@ -46,12 +47,12 @@ use tracing::{error, info, warn}; /// Carries the cross-thread shutdown flag, one OS-thread `JoinHandle` /// per shard, and the first panic `install_panic_hook` recorded. The /// caller flips the flag via [`Self::install_ctrlc_handler`] and then -/// drains every shard via [`Self::join_all`], bounded by `join_timeout` -/// (`system.sharding.shutdown_join_timeout`). +/// drains every shard via [`Self::join_all`], bounded by the shared +/// `ShutdownDeadline` (`system.sharding.shutdown_join_timeout`). pub struct ShardHandles { pub(in crate::boot) shutdown_flag: Arc, pub(in crate::boot) shard_threads: Vec<(u16, thread::JoinHandle>)>, - pub(in crate::boot) join_timeout: Duration, + pub(in crate::boot) deadline: Arc, pub(in crate::boot) first_panic: Arc>, } @@ -94,7 +95,9 @@ impl ShardHandles { /// deadline passes is abandoned (its `JoinHandle` dropped, the OS /// thread left to die with the process) and reported as /// [`ShardJoinFailureKind::Wedged`]: a wedged pump or listener must - /// not block process exit forever. + /// not block process exit forever. Shard 0 gets the peer-wait floor + /// as grace on top, because its `PeerExitWait` is allowed to hold the + /// metadata writer that long past a spent budget. /// /// # Errors /// @@ -107,31 +110,32 @@ impl ShardHandles { /// compio's `spawn` caught, which no thread result can carry. pub fn join_all(self) -> Result<(), ServerError> { let mut failures: Vec = Vec::new(); - // Armed on the first poll that observes the shutdown flag, shared - // across all shards: one budget covers the whole drain, not one - // budget per shard. - let mut deadline: Option = None; // Shards run thread-per-core with compio's blocking fallback pool // disabled, so an io_uring opcode the kernel lacks aborts every shard // with the same panic. Surface the actionable diagnostic once. let mut io_uring_diagnostic_shown = false; for (shard_id, handle) in self.shard_threads { - let Some(joined) = join_until_shutdown_deadline( - handle, - &self.shutdown_flag, - self.join_timeout, - &mut deadline, - ) else { + // Shard 0's `PeerExitWait` is allowed to overrun the budget by + // its floor, so its join has to tolerate the same overrun or a + // shutdown that correctly held the writer for a slow peer reports + // the shard that honoured the fence as wedged. + let grace = if shard_id == 0 { + self.deadline.peer_wait_floor + } else { + Duration::ZERO + }; + let waited = self.deadline.budget.saturating_add(grace); + let Some(joined) = + join_until_shutdown_deadline(handle, &self.shutdown_flag, &self.deadline, grace) + else { error!( shard_id, - waited = ?self.join_timeout, + ?waited, "shard thread still running at the shutdown join deadline; abandoning it" ); failures.push(ShardJoinFailure { shard_id, - kind: ShardJoinFailureKind::Wedged { - waited: self.join_timeout, - }, + kind: ShardJoinFailureKind::Wedged { waited }, }); continue; }; @@ -218,28 +222,24 @@ pub(in crate::boot) fn install_panic_hook(shutdown_flag: Arc) -> Arc const JOIN_POLL_INTERVAL: Duration = Duration::from_millis(25); /// Join `handle`, waiting indefinitely while the server runs. The -/// `join_timeout` clock starts only when `shutdown_flag` is observed set -/// (arming the caller-shared `deadline` once, so all shards drain under -/// ONE budget); a running server parked here for hours must never be -/// mistaken for a wedged shard. `None` means the thread was still -/// running at the post-shutdown deadline and the handle was dropped -/// (the OS thread keeps running detached; process exit reaps it). -/// `JoinHandle` has no timed join, so this polls `is_finished` at -/// [`JOIN_POLL_INTERVAL`]; the closing `join()` on a finished thread -/// returns immediately. +/// budget clock starts only when `shutdown_flag` is observed set (arming +/// the shared [`ShutdownDeadline`], so every shard join and shard 0's +/// peer wait drain under ONE budget); a running server parked here for +/// hours must never be mistaken for a wedged shard. `grace` extends this +/// caller's share of the budget by an overrun the joined thread is allowed +/// (shard 0's peer-wait floor). `None` means the thread was still running +/// at the post-shutdown deadline and the handle was dropped (the OS thread +/// keeps running detached; process exit reaps it). `JoinHandle` has no +/// timed join, so this polls `is_finished` at [`JOIN_POLL_INTERVAL`]; the +/// closing `join()` on a finished thread returns immediately. fn join_until_shutdown_deadline( handle: thread::JoinHandle>, shutdown_flag: &AtomicBool, - join_timeout: Duration, - deadline: &mut Option, + deadline: &ShutdownDeadline, + grace: Duration, ) -> Option>> { while !handle.is_finished() { - if deadline.is_none() && shutdown_flag.load(Ordering::Relaxed) { - *deadline = Some(Instant::now() + join_timeout); - } - if let Some(deadline) = deadline - && Instant::now() >= *deadline - { + if shutdown_flag.load(Ordering::Relaxed) && deadline.remaining_with_grace(grace).is_zero() { return None; } thread::sleep(JOIN_POLL_INTERVAL); @@ -274,9 +274,8 @@ fn panic_payload_to_string(payload: &(dyn std::any::Any + Send)) -> String { /// spawn error instead of hanging on a wedged shard. pub(in crate::boot) fn join_partial_shard_survivors( shard_threads: Vec<(u16, thread::JoinHandle>)>, - join_timeout: Duration, + deadline: &ShutdownDeadline, ) { - let deadline = Instant::now() + join_timeout; let mut remaining = shard_threads; loop { let mut still_running = Vec::with_capacity(remaining.len()); @@ -289,7 +288,7 @@ pub(in crate::boot) fn join_partial_shard_survivors( } } remaining = still_running; - if remaining.is_empty() || Instant::now() >= deadline { + if remaining.is_empty() || deadline.remaining().is_zero() { break; } thread::sleep(JOIN_POLL_INTERVAL); @@ -297,7 +296,7 @@ pub(in crate::boot) fn join_partial_shard_survivors( for (shard_id, _survivor) in remaining { error!( shard_id, - waited = ?join_timeout, + waited = ?deadline.budget, "survivor shard thread still running at the shutdown join deadline; abandoning it" ); } @@ -334,6 +333,199 @@ impl Drop for ShutdownOnDrop { } } +/// The single post-shutdown budget, shared by the main thread's shard +/// joins and shard 0's peer wait. +/// +/// Both waits are bounded by `system.sharding.shutdown_join_timeout` and +/// they NEST: shard 0 cannot start waiting for its peers until its own +/// drain returned, which is already inside the join budget. Arming one +/// instant on first use, whichever wait gets there first, keeps the two +/// inside one deadline instead of stacking two full budgets, so a +/// correct shutdown cannot report shard 0 as wedged. +/// +/// The two waits enforce different things, so the budget is shared but not +/// fungible: the join bounds LIVENESS (process exit must not hang on a +/// wedged shard) while the peer wait enforces SAFETY (the metadata writer +/// must outlive every reader). `peer_wait_floor` is what keeps an exhausted +/// join budget from cancelling the safety fence, and the joins in turn +/// tolerate that floor as grace (see [`ShardHandles::join_all`]). +pub(in crate::boot) struct ShutdownDeadline { + armed: OnceLock, + budget: Duration, + peer_wait_floor: Duration, +} + +impl ShutdownDeadline { + pub(in crate::boot) const fn new(budget: Duration, peer_wait_floor: Duration) -> Self { + Self { + armed: OnceLock::new(), + budget, + peer_wait_floor, + } + } + + /// Time left in the shared budget plus `grace`, arming it on the first + /// call. Callers must only reach this once the shutdown flag is set: a + /// running server would otherwise start the clock. + fn remaining_with_grace(&self, grace: Duration) -> Duration { + self.armed + .get_or_init(|| Instant::now() + self.budget) + .checked_add(grace) + .map_or(Duration::MAX, |limit| { + limit.saturating_duration_since(Instant::now()) + }) + } + + fn remaining(&self) -> Duration { + self.remaining_with_grace(Duration::ZERO) + } + + /// How long shard 0 may block on its peers: what is left of the shared + /// budget, floored at a peer's whole exit (one poll interval to observe the + /// shutdown flag plus one drain budget) so a join that already spent the + /// budget cannot release the metadata writer while a peer still reads + /// through it. Only [`PeerExitWait::drop`]'s panic arm skips the fence. + fn remaining_for_peer_wait(&self) -> Duration { + self.remaining().max(self.peer_wait_floor) + } +} + +/// Peer shards still running: each [`PeerExitGuard`] counts one out, +/// [`PeerExitWait`] blocks shard 0 until the count is zero. +/// +/// Shard 0 owns the metadata state machine's only write handle and every +/// peer reads through handles that stop working the moment it drops, so +/// the shard that owns the writer must outlive every reader on every exit +/// but one: [`PeerExitWait::drop`] skips the wait while shard 0 is +/// panicking, because parking an unwinding thread on a `Condvar` inside +/// `runtime.block_on` would stall the `io_uring` driver, and a panic in +/// shard 0's own pump can be exactly what the peers are blocked on. A +/// peer mid-read then panics too; the first panic is already recorded and +/// the process is going down either way. +pub(in crate::boot) struct PeerExitCountdown { + running: Mutex, + all_exited: Condvar, +} + +impl PeerExitCountdown { + pub(in crate::boot) const fn new(peers: usize) -> Self { + Self { + running: Mutex::new(peers), + all_exited: Condvar::new(), + } + } + + /// Block until every peer has counted itself out or `timeout` elapses. + /// `Err` carries the number of peers still running at the deadline. + fn wait(&self, timeout: Duration) -> Result<(), usize> { + let (guard, _) = self + .all_exited + .wait_timeout_while( + self.running.lock().unwrap_or_else(PoisonError::into_inner), + timeout, + |running| *running > 0, + ) + .unwrap_or_else(PoisonError::into_inner); + let running = *guard; + drop(guard); + if running == 0 { Ok(()) } else { Err(running) } + } + + fn peer_exited(&self) { + let mut running = self.running.lock().unwrap_or_else(PoisonError::into_inner); + *running = running.saturating_sub(1); + if *running == 0 { + self.all_exited.notify_all(); + } + } + + /// Count out peers that never spawned. The countdown is sized before the + /// spawn loop, so a failed `thread::Builder::spawn` leaves peers whose + /// [`PeerExitGuard`] will never exist: without this shard 0 waits out its + /// whole join budget for threads that were never there. + pub(in crate::boot) fn peers_never_spawned(&self, count: usize) { + for _ in 0..count { + self.peer_exited(); + } + } +} + +/// Counts one peer shard out of the [`PeerExitCountdown`] on drop. +/// +/// Held by `run_shard_thread` from before the runtime exists, so it drops +/// after the runtime and its tasks are gone: past that point nothing on +/// the peer's thread can still read shard 0's metadata. Drop runs on every +/// exit path, the error `?` returns and panic unwinds included. +struct PeerExitGuard { + countdown: Arc, +} + +impl PeerExitGuard { + const fn new(countdown: Arc) -> Self { + Self { countdown } + } +} + +impl Drop for PeerExitGuard { + fn drop(&mut self) { + self.countdown.peer_exited(); + } +} + +/// Shard 0's side of the [`PeerExitCountdown`]: blocks on drop until every +/// peer has exited, so whatever is declared before it outlives every +/// peer's reads. +/// +/// Flips the shutdown flag before waiting: a peer parked on its bus token +/// only starts its drain once the flag is set, and the thread-level +/// `ShutdownOnDrop` flips it only after `block_on` returns, which is after +/// this wait. Bounded by [`ShutdownDeadline::remaining_for_peer_wait`] -- +/// what is left of the budget [`ShardHandles::join_all`] shares, but never +/// less than one drain -- with the same abandon-and-log policy, so a wedged +/// peer cannot hold shard 0 indefinitely while a spent join budget still +/// cannot cut the wait to nothing. +pub(in crate::boot) struct PeerExitWait { + countdown: Arc, + shutdown_flag: Arc, + deadline: Arc, +} + +impl PeerExitWait { + pub(in crate::boot) const fn new( + countdown: Arc, + shutdown_flag: Arc, + deadline: Arc, + ) -> Self { + Self { + countdown, + shutdown_flag, + deadline, + } + } +} + +impl Drop for PeerExitWait { + fn drop(&mut self) { + // The panic is the fault to report and `ShutdownOnDrop` still + // flips the flag for the peers; blocking an unwinding thread here + // would only delay it. + if thread::panicking() { + return; + } + self.shutdown_flag.store(true, Ordering::Relaxed); + let remaining = self.deadline.remaining_for_peer_wait(); + if let Err(peers_running) = self.countdown.wait(remaining) { + warn!( + peers_running, + waited = ?remaining, + budget = ?self.deadline.budget, + "peer shards still running at the shutdown join deadline; \ + releasing shard 0 anyway" + ); + } + } +} + /// Resolve the operator's `cpu_allocation` into concrete shard /// assignments plus the checked `u16` shard count. /// @@ -364,10 +556,13 @@ pub(in crate::boot) fn resolve_shard_assignments( } /// Re-validate the runtime sharding knobs that the per-shard runtime -/// consumes directly. Mirrors `ShardingConfig::validate` so a caller -/// that built the config without running it (e.g. tests, embedded -/// usage) cannot OOM at boot or wedge process exit with an out-of-range -/// value. +/// consumes directly: the two inbox capacities, the three shutdown +/// durations and their ordering, and the reconcile tick. Mirrors +/// `ShardingConfig::validate` for exactly those, so a caller that built the +/// config without running it (e.g. tests, embedded usage) cannot OOM at +/// boot, starve a core, or wedge process exit with an out-of-range value. +/// `cpu_allocation` is not re-checked here - [`resolve_shard_assignments`] +/// resolves it through the allocator and rejects it there. pub(in crate::boot) fn validate_sharding_runtime_knobs( sharding: &configs::sharding::ShardingConfig, ) -> Result<(), ServerError> { @@ -407,6 +602,32 @@ pub(in crate::boot) fn validate_sharding_runtime_knobs( drain: drain_timeout, }); } + let join_timeout = sharding.shutdown_join_timeout.get_duration(); + if join_timeout > SHUTDOWN_JOIN_TIMEOUT_MAX { + return Err(ServerError::InvalidShutdownJoinTimeout { + value: join_timeout, + max: SHUTDOWN_JOIN_TIMEOUT_MAX, + }); + } + // A join budget under the drain budget abandons shards mid-drain, + // interrupting the WAL fsync / replica drain, and now also cuts shard + // 0's peer wait short of the writer's last reader. + if join_timeout < drain_timeout { + return Err(ServerError::ShutdownJoinBelowDrain { + join: join_timeout, + drain: drain_timeout, + }); + } + // Zero feeds `run_reconciler`'s sleep inside an unconditional loop, so the + // tick arm is ready every iteration and `reconcile_once` runs back to back, + // starving the pump on that core. + let reconcile_interval = sharding.reconcile_periodic_interval.get_duration(); + if reconcile_interval.is_zero() || reconcile_interval > RECONCILE_PERIODIC_INTERVAL_MAX { + return Err(ServerError::InvalidReconcilePeriodicInterval { + value: reconcile_interval, + max: RECONCILE_PERIODIC_INTERVAL_MAX, + }); + } Ok(()) } @@ -429,11 +650,17 @@ pub(in crate::boot) fn run_shard_thread( roster_cells: RosterCells, metadata_applied_frontier: Arc, shard_metrics_all: Vec, + peer_exit: Arc, + shutdown_deadline: Arc, ) -> Result<(), ServerError> { // Armed for the whole thread body: a post-spawn error `?` or a panic // unwind here must flip `shutdown_flag` so sibling watchdogs drive // their bus shutdown instead of parking forever on `bus.token().wait()`. let mut shutdown_guard = ShutdownOnDrop::new(Arc::clone(&shutdown_flag)); + // Declared before the runtime so it drops after it: a peer counts + // itself out only once no task of its runtime can read shard 0's + // metadata any more. + let _peer_exit_guard = (shard_id != 0).then(|| PeerExitGuard::new(Arc::clone(&peer_exit))); assignment .bind_cpu() @@ -473,6 +700,8 @@ pub(in crate::boot) fn run_shard_thread( roster_cells, metadata_applied_frontier, shard_metrics_all, + peer_exit, + shutdown_deadline, )) .await }); @@ -652,6 +881,248 @@ mod tests { ); } + #[test] + fn peer_exit_countdown_releases_the_waiter_once_every_peer_is_out() { + let countdown = Arc::new(PeerExitCountdown::new(2)); + let guards: [PeerExitGuard; 2] = + std::array::from_fn(|_| PeerExitGuard::new(Arc::clone(&countdown))); + assert_eq!( + countdown.wait(Duration::from_millis(10)), + Err(2), + "two live peers must hold the waiter past a short deadline" + ); + let peers: Vec<_> = guards + .into_iter() + .map(|guard| { + thread::spawn(move || { + thread::sleep(Duration::from_millis(20)); + drop(guard); + }) + }) + .collect(); + assert_eq!( + countdown.wait(Duration::from_secs(30)), + Ok(()), + "the last guard drop must release the waiter" + ); + for peer in peers { + peer.join() + .expect("peer thread dropped its guard without panicking"); + } + } + + #[test] + fn peer_exit_guard_counts_out_during_unwind() { + let countdown = Arc::new(PeerExitCountdown::new(1)); + let guard = PeerExitGuard::new(Arc::clone(&countdown)); + // `resume_unwind` skips the panic hook, so the unwind is silent. + let unwound = panic::catch_unwind(panic::AssertUnwindSafe(|| { + let _guard = guard; + panic::resume_unwind(Box::new("peer shard body panicked")); + })); + assert!(unwound.is_err()); + assert_eq!( + countdown.wait(Duration::ZERO), + Ok(()), + "a guard dropped by a panic unwind must still count its peer out" + ); + } + + /// Stops a parked stand-in thread once the test's binding goes out of + /// scope. The joiner under test drops the `JoinHandle`, so the thread + /// cannot be joined back; flagging it keeps it from parking for the life + /// of the test binary. + struct ThreadStopper(Arc); + + impl Drop for ThreadStopper { + fn drop(&mut self) { + self.0.store(true, Ordering::Relaxed); + } + } + + /// A shard thread that never returns on its own: the stand-in for a + /// wedged pump that the deadline tests need. + fn spawn_wedged_shard() -> (thread::JoinHandle>, ThreadStopper) { + let stop = Arc::new(AtomicBool::new(false)); + let handle = thread::spawn({ + let stop = Arc::clone(&stop); + move || -> Result<(), ServerError> { + while !stop.load(Ordering::Relaxed) { + thread::sleep(JOIN_POLL_INTERVAL); + } + Ok(()) + } + }); + (handle, ThreadStopper(stop)) + } + + #[test] + fn peer_exit_wait_flips_the_flag_before_waiting() { + // Stands in for a peer parked on its bus token: it exits only once + // the shutdown flag is set, so a waiter that set the flag after + // waiting would sit out the whole budget. + let countdown = Arc::new(PeerExitCountdown::new(1)); + let shutdown_flag = Arc::new(AtomicBool::new(false)); + let peer = thread::spawn({ + let guard = PeerExitGuard::new(Arc::clone(&countdown)); + let shutdown_flag = Arc::clone(&shutdown_flag); + move || { + while !shutdown_flag.load(Ordering::Relaxed) { + thread::sleep(Duration::from_millis(1)); + } + drop(guard); + } + }); + let started = Instant::now(); + drop(PeerExitWait::new( + countdown, + Arc::clone(&shutdown_flag), + Arc::new(ShutdownDeadline::new( + Duration::from_secs(30), + Duration::from_secs(10), + )), + )); + assert!( + started.elapsed() < Duration::from_secs(5), + "the waiter must not sit out its budget on a peer that waits for the flag" + ); + assert!(shutdown_flag.load(Ordering::Relaxed)); + peer.join() + .expect("peer thread dropped its guard without panicking"); + } + + #[test] + fn peer_exit_wait_gives_up_at_the_deadline() { + let countdown = Arc::new(PeerExitCountdown::new(1)); + let _wedged_peer = PeerExitGuard::new(Arc::clone(&countdown)); + // Degenerate budget == floor: the floor cannot extend the wait, so + // the abandon is bounded by the budget alone. + let timeout = Duration::from_millis(50); + let started = Instant::now(); + drop(PeerExitWait::new( + countdown, + Arc::new(AtomicBool::new(false)), + Arc::new(ShutdownDeadline::new(timeout, timeout)), + )); + let waited = started.elapsed(); + assert!( + waited >= timeout && waited < Duration::from_secs(5), + "a wedged peer must be abandoned at the deadline, waited {waited:?}" + ); + } + + /// Regression: the two waits nest (shard 0 cannot start waiting for its + /// peers until its own drain returned, already inside the join budget). + /// With a budget each, a clean shutdown reported shard 0 as wedged and + /// exited non-zero. What is left of the shared budget is what the peer + /// wait gets, whenever that clears the floor. + #[test] + fn peer_wait_inherits_the_shared_budget_above_its_floor() { + const BUDGET: Duration = Duration::from_secs(30); + const FLOOR: Duration = Duration::from_secs(10); + let deadline = ShutdownDeadline::new(BUDGET, FLOOR); + let inherited = deadline.remaining_for_peer_wait(); + assert!( + inherited > FLOOR && inherited <= BUDGET, + "an unspent budget must be inherited, not replaced by the floor: {inherited:?}" + ); + } + + /// The other half of the same nesting: the join budget bounds LIVENESS + /// (process exit must not hang on a wedged shard) while this wait enforces + /// SAFETY (shard 0 owns the metadata write handle every peer reads + /// through). Funding the fence out of the join budget let an exit-latency + /// timeout cancel it: the drain spent the budget, `wait(ZERO)` returned + /// without blocking, and shard 0 dropped the writer under a live reader. + #[test] + fn peer_wait_honours_a_live_peer_past_a_spent_join_budget() { + const BUDGET: Duration = Duration::from_millis(20); + const FLOOR: Duration = Duration::from_millis(200); + let deadline = Arc::new(ShutdownDeadline::new(BUDGET, FLOOR)); + let shutdown_flag = AtomicBool::new(true); + // Stands in for the slow drain that arms and then spends the shared + // budget before shard 0 can reach its peer wait. + let (wedged_shard, _stopper) = spawn_wedged_shard(); + assert!( + join_until_shutdown_deadline(wedged_shard, &shutdown_flag, &deadline, Duration::ZERO) + .is_none(), + "the join must spend the budget it armed" + ); + assert!(deadline.remaining().is_zero(), "the budget must be spent"); + + let countdown = Arc::new(PeerExitCountdown::new(1)); + let _live_peer = PeerExitGuard::new(Arc::clone(&countdown)); + let started = Instant::now(); + drop(PeerExitWait::new( + countdown, + Arc::new(AtomicBool::new(false)), + Arc::clone(&deadline), + )); + let waited = started.elapsed(); + assert!( + waited >= FLOOR, + "a spent join budget must not release the metadata writer while a peer still reads \ + through it, waited {waited:?}" + ); + } + + #[test] + fn peer_exit_wait_ignores_peers_that_never_spawned() { + // A failed spawn leaves peers with no guard to count them out; the + // one that did spawn must still be the only thing the waiter waits on. + let countdown = Arc::new(PeerExitCountdown::new(3)); + countdown.peers_never_spawned(2); + let spawned = PeerExitGuard::new(Arc::clone(&countdown)); + assert_eq!( + countdown.wait(Duration::from_millis(10)), + Err(1), + "the peer that did spawn must still hold the waiter" + ); + drop(spawned); + assert_eq!(countdown.wait(Duration::ZERO), Ok(())); + } + + #[test] + fn peer_exit_wait_with_no_peers_returns_at_once() { + let started = Instant::now(); + drop(PeerExitWait::new( + Arc::new(PeerExitCountdown::new(0)), + Arc::new(AtomicBool::new(false)), + Arc::new(ShutdownDeadline::new( + Duration::from_secs(30), + Duration::from_secs(10), + )), + )); + assert!( + started.elapsed() < Duration::from_secs(1), + "a single-shard server has nobody to wait for" + ); + } + + #[test] + fn peer_exit_wait_never_blocks_an_unwinding_thread() { + let countdown = Arc::new(PeerExitCountdown::new(1)); + let _still_running = PeerExitGuard::new(Arc::clone(&countdown)); + let wait = PeerExitWait::new( + countdown, + Arc::new(AtomicBool::new(false)), + Arc::new(ShutdownDeadline::new( + Duration::from_secs(30), + Duration::from_secs(10), + )), + ); + let started = Instant::now(); + let unwound = panic::catch_unwind(panic::AssertUnwindSafe(|| { + let _wait = wait; + panic::resume_unwind(Box::new("shard 0 body panicked")); + })); + assert!(unwound.is_err()); + assert!( + started.elapsed() < Duration::from_secs(1), + "a waiter dropped during unwind must return without waiting" + ); + } + #[compio::test] async fn pump_drain_timeout_is_not_reported_as_clean() { let mut config = ServerConfig::default(); @@ -714,19 +1185,15 @@ mod tests { thread::sleep(Duration::from_millis(300)); Ok(()) }); - let mut deadline = None; - let joined = join_until_shutdown_deadline( - handle, - &shutdown_flag, - Duration::from_millis(20), - &mut deadline, - ); + let deadline = ShutdownDeadline::new(Duration::from_millis(20), Duration::from_millis(20)); + let joined = + join_until_shutdown_deadline(handle, &shutdown_flag, &deadline, Duration::ZERO); assert!( matches!(joined, Some(Ok(Ok(())))), "a running server must be awaited indefinitely, not abandoned as wedged" ); assert!( - deadline.is_none(), + deadline.armed.get().is_none(), "the join deadline must not arm before the shutdown flag flips" ); } @@ -744,7 +1211,10 @@ mod tests { let handles = ShardHandles { shutdown_flag: Arc::new(AtomicBool::new(true)), shard_threads: vec![(0, handle)], - join_timeout: Duration::from_secs(1), + deadline: Arc::new(ShutdownDeadline::new( + Duration::from_secs(1), + Duration::from_millis(500), + )), first_panic, }; let error = handles @@ -781,24 +1251,17 @@ mod tests { #[test] fn join_abandons_a_wedged_shard_after_the_shutdown_deadline() { let shutdown_flag = AtomicBool::new(true); - // Never finishes: stands in for a wedged pump. The thread leaks - // into the test process, which exits right after. - let handle = thread::spawn(|| -> Result<(), ServerError> { - loop { - thread::sleep(Duration::from_secs(1)); - } - }); - let mut deadline = None; - let joined = join_until_shutdown_deadline( - handle, - &shutdown_flag, - Duration::from_millis(100), - &mut deadline, - ); + let (handle, _stopper) = spawn_wedged_shard(); + let deadline = ShutdownDeadline::new(Duration::from_millis(100), Duration::from_millis(50)); + let joined = + join_until_shutdown_deadline(handle, &shutdown_flag, &deadline, Duration::ZERO); assert!( joined.is_none(), "a shard still running past the post-shutdown budget must be abandoned" ); - assert!(deadline.is_some(), "the deadline arms once the flag is set"); + assert!( + deadline.armed.get().is_some(), + "the deadline arms once the flag is set" + ); } } diff --git a/core/server/src/dispatch/authz.rs b/core/server/src/dispatch/authz.rs index 436aa2222e..c6fb2c6e49 100644 --- a/core/server/src/dispatch/authz.rs +++ b/core/server/src/dispatch/authz.rs @@ -29,9 +29,10 @@ use std::rc::Rc; use consensus::MetadataHandle; use iggy_binary_protocol::codes::{ - DESCRIBE_OPTIONS_CODE, GET_CLUSTER_METADATA_CODE, GET_CONSUMER_GROUP_CODE, - GET_CONSUMER_GROUPS_CODE, GET_PERSONAL_ACCESS_TOKENS_CODE, GET_STATS_CODE, GET_STREAM_CODE, - GET_STREAMS_CODE, GET_TOPIC_CODE, GET_TOPICS_CODE, GET_USER_CODE, GET_USERS_CODE, + DESCRIBE_OPTIONS_CODE, FLUSH_UNSAVED_BUFFER_CODE, GET_CLUSTER_METADATA_CODE, + GET_CONSUMER_GROUP_CODE, GET_CONSUMER_GROUPS_CODE, GET_PERSONAL_ACCESS_TOKENS_CODE, + GET_STATS_CODE, GET_STREAM_CODE, GET_STREAMS_CODE, GET_TOPIC_CODE, GET_TOPICS_CODE, + GET_USER_CODE, GET_USERS_CODE, }; use iggy_binary_protocol::requests::consumer_groups::{ GetConsumerGroupRequest, GetConsumerGroupsRequest, @@ -39,20 +40,15 @@ use iggy_binary_protocol::requests::consumer_groups::{ use iggy_binary_protocol::requests::streams::GetStreamRequest; use iggy_binary_protocol::requests::topics::{GetTopicRequest, GetTopicsRequest}; use iggy_binary_protocol::requests::users::GetUserRequest; -use iggy_binary_protocol::{ - Operation, PrepareHeader, RoutedRequestHeader, WireDecode, WireIdentifier, -}; +use iggy_binary_protocol::{Operation, PrepareHeader, WireDecode, WireIdentifier, lookup_command}; use iggy_common::IggyError; use journal::superblock::SuperblockStore; use journal::{Journal, JournalHandle}; use metadata::impls::metadata::StreamsFrontend; use metadata::permissioner::Permissioner; use server_common::Message; -use tracing::warn; -use crate::responses::{ - build_deny_reply, current_metadata_commit, resolve_stream_id, resolve_topic_id, -}; +use crate::responses::{resolve_stream_id, resolve_topic_id}; use crate::shell::{ShellBus, ShellShard}; /// Authorize a partition-plane op on its resolved (stream, topic) for the @@ -132,74 +128,6 @@ where decision.err().map(|error| error.as_code()) } -/// Reply to a request rejected before it reached its plane with the request's -/// own frame: empty body + nonzero `status`. The nonzero status is the whole -/// point: the SDK peeks it and surfaces the typed error, whereas a status-0 -/// frame reads as a committed ack for work that never happened. Silence is no -/// better, the connection decodes replies in lockstep and would wedge on every -/// later request. -#[allow(clippy::future_not_send)] -pub(in crate::dispatch) async fn send_deny_reply( - shard: &Rc>, - transport_client_id: u128, - request_header: &RoutedRequestHeader, - status: u32, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let commit = current_metadata_commit(shard); - let reply = build_deny_reply(request_header, transport_client_id, 0, commit, status); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - status, - error = %error, - operation = ?request_header.operation, - "failed to surface request denial" - ); - } -} - -/// Deny a request from an unbound transport without disclosing the metadata -/// commit frontier. The status is the only field a pre-authenticated caller -/// needs, while the live commit would expose cluster write activity. -#[allow(clippy::future_not_send)] -pub(in crate::dispatch) async fn send_unbound_deny_reply( - shard: &Rc>, - transport_client_id: u128, - request_header: &RoutedRequestHeader, - status: u32, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let reply = build_deny_reply(request_header, transport_client_id, 0, 0, status); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - status, - error = %error, - operation = ?request_header.operation, - "failed to surface unbound request denial" - ); - } -} - /// Run an unscoped non-replicated-read rule for the acting user. A `None` user /// id (only the pre-auth path, which serves ungated codes) fails closed. pub(in crate::dispatch) fn authorize_uid( @@ -263,6 +191,11 @@ where /// topic]) against committed state first. The PAT list is self-scoped, so /// authentication is its whole rule, and `GET_CLUSTER_METADATA` -- which /// describes the private replica network -- is gated the same way. +/// +/// The gate is total over the protocol command table: every code the builder +/// serves has a named arm, and the tail refuses everything else (a replicated +/// code inside a `NonReplicated` header, a table-listed code with no arm, an +/// unknown code) instead of deferring it to the builder's catch-all. pub(in crate::dispatch) fn authorize_default_read( shard: &Rc>, code: u32, @@ -276,8 +209,8 @@ where S: 'static, SB: SuperblockStore + 'static, { - // A `u32` match cannot be exhaustive: every gated code is named explicitly, - // and the final arm is the ungated set the builder serves without a rule. + // A `u32` match cannot be exhaustive, so totality is by construction: + // every code with a decision is named, and the tail refuses the rest. match code { GET_STATS_CODE => authorize_uid(shard, user_id, Permissioner::get_stats), GET_USERS_CODE => authorize_uid(shard, user_id, Permissioner::get_users), @@ -328,46 +261,17 @@ where |request| (&request.stream_id, &request.topic_id), Permissioner::get_consumer_groups, ), - _ => Ok(()), - } -} - -/// Reply to a denied non-replicated read with the request's reply frame: empty -/// body + nonzero `status`. The SDK peeks the status before body decode and -/// surfaces the typed error, so a poll denial never reaches the empty-poll -/// "0 messages" body path. -#[allow(clippy::future_not_send)] -pub(in crate::dispatch) async fn send_non_replicated_deny( - shard: &Rc>, - request: &Message, - transport_client_id: u128, - status: u32, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let commit = current_metadata_commit(shard); - let reply = build_deny_reply( - request.header(), - request.header().client, - request.header().session, - commit, - status, - ); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - status, - error = %error, - "failed to surface non-replicated authz denial" - ); + // No on-demand flush primitive exists, and flush has no HTTP route, so + // this arm is the only thing answering `FeatureUnavailable` for it. + FLUSH_UNSAVED_BUFFER_CODE => Err(IggyError::FeatureUnavailable), + // A replicated code smuggled inside a `NonReplicated` header keeps the + // builder's `FeatureUnavailable`; a table-listed code with no arm above + // and an unknown code are both refused as `InvalidCommand`. The builder + // refuses the same set, so a new caller cannot land on a fail-open. + _ => match lookup_command(code) { + Some(meta) if meta.is_replicated() => Err(IggyError::FeatureUnavailable), + _ => Err(IggyError::InvalidCommand), + }, } } @@ -508,3 +412,168 @@ where Some((stream_id, topic_id)) }) } + +#[cfg(test)] +mod tests { + use super::*; + use crate::dispatch::test_support::{FIRST_BOOT, SpyBus, TestShard, test_shard}; + use iggy_binary_protocol::COMMAND_TABLE; + use iggy_binary_protocol::codes::{ + CREATE_STREAM_CODE, GET_CLIENT_CODE, GET_CLIENTS_CODE, GET_CONSUMER_OFFSET_CODE, + GET_ME_CODE, GET_SNAPSHOT_FILE_CODE, LOGIN_REGISTER_CODE, LOGIN_REGISTER_WITH_PAT_CODE, + LOGIN_USER_CODE, LOGIN_WITH_PERSONAL_ACCESS_TOKEN_CODE, LOGOUT_USER_CODE, PING_CODE, + POLL_MESSAGES_CODE, SYNC_CONSUMER_GROUP_CODE, + }; + use iggy_common::defaults::DEFAULT_ROOT_USER_ID; + + /// A code no `COMMAND_TABLE` entry claims. + const UNKNOWN_CODE: u32 = 9999; + + /// The gate's answer as the wire sees it: status 0 allows, anything else + /// is the deny code stamped into `ReplyHeader.status`. + #[derive(Debug, Clone, Copy, PartialEq, Eq)] + enum Verdict { + Allow, + Deny(u32), + } + + /// Shard with root seeded, so the root column below exercises the + /// permissioner rules rather than a missing-user deny. + fn gate_shard() -> Rc { + let bus = SpyBus::default(); + let shard = Rc::new(test_shard(&bus, 0, 1, FIRST_BOOT)); + shard + .plane + .metadata() + .mux_stm + .users() + .ensure_root_user("iggy", "hash"); + shard + } + + /// Gate a header-only frame (`body = &[]`) for `user_id`. + fn gate(shard: &Rc, code: u32, user_id: Option) -> Verdict { + match authorize_default_read(shard, code, &[], user_id) { + Ok(()) => Verdict::Allow, + Err(error) => Verdict::Deny(error.as_code()), + } + } + + /// Expected verdicts per non-replicated code for a header-only frame, as + /// `(code, unbound caller, root)`. Header-only means the identifier-scoped + /// arms never decode a body and defer to the builder's own error for both + /// callers. Codes the reads router serves on its own arms (`PING`, `GET_ME`, + /// `GET_CLIENTS`, ...) and the codes the classifier settles (the legacy + /// logins) never reach the gate, which refuses them like any other armless + /// code. + fn expected_header_only_verdicts() -> Vec<(u32, Verdict, Verdict)> { + let allow = Verdict::Allow; + let unauthenticated = Verdict::Deny(IggyError::Unauthenticated.as_code()); + let invalid_command = Verdict::Deny(IggyError::InvalidCommand.as_code()); + let feature_unavailable = Verdict::Deny(IggyError::FeatureUnavailable.as_code()); + vec![ + (PING_CODE, invalid_command, invalid_command), + (GET_STATS_CODE, unauthenticated, allow), + (GET_SNAPSHOT_FILE_CODE, invalid_command, invalid_command), + (GET_CLUSTER_METADATA_CODE, unauthenticated, allow), + (GET_ME_CODE, invalid_command, invalid_command), + (GET_CLIENT_CODE, invalid_command, invalid_command), + (GET_CLIENTS_CODE, invalid_command, invalid_command), + (GET_USER_CODE, allow, allow), + (GET_USERS_CODE, unauthenticated, allow), + (LOGIN_USER_CODE, invalid_command, invalid_command), + (LOGOUT_USER_CODE, invalid_command, invalid_command), + (LOGIN_REGISTER_CODE, invalid_command, invalid_command), + (GET_PERSONAL_ACCESS_TOKENS_CODE, unauthenticated, allow), + ( + LOGIN_WITH_PERSONAL_ACCESS_TOKEN_CODE, + invalid_command, + invalid_command, + ), + (POLL_MESSAGES_CODE, invalid_command, invalid_command), + ( + FLUSH_UNSAVED_BUFFER_CODE, + feature_unavailable, + feature_unavailable, + ), + (GET_CONSUMER_OFFSET_CODE, invalid_command, invalid_command), + (GET_STREAM_CODE, allow, allow), + (GET_STREAMS_CODE, unauthenticated, allow), + (GET_TOPIC_CODE, allow, allow), + (GET_TOPICS_CODE, allow, allow), + (GET_CONSUMER_GROUP_CODE, allow, allow), + (GET_CONSUMER_GROUPS_CODE, allow, allow), + (SYNC_CONSUMER_GROUP_CODE, invalid_command, invalid_command), + ( + LOGIN_REGISTER_WITH_PAT_CODE, + invalid_command, + invalid_command, + ), + (DESCRIBE_OPTIONS_CODE, unauthenticated, allow), + ] + } + + /// Every non-replicated table entry has a named decision above. A new + /// entry without a row fails by name, so it cannot slip past the gate + /// unnoticed; a row without a table entry is stale and fails too. + #[test] + fn gate_is_total_over_the_command_table() { + let shard = gate_shard(); + let expected = expected_header_only_verdicts(); + for meta in COMMAND_TABLE.iter().filter(|meta| !meta.is_replicated()) { + let Some((_, unbound, root)) = expected.iter().find(|row| row.0 == meta.code) else { + panic!( + "non-replicated command {} ({}) has no ratchet row: decide it in \ + authorize_default_read and add the row", + meta.name, meta.code + ); + }; + assert_eq!( + gate(&shard, meta.code, None), + *unbound, + "{} ({}) from an unbound caller", + meta.name, + meta.code + ); + assert_eq!( + gate(&shard, meta.code, Some(DEFAULT_ROOT_USER_ID)), + *root, + "{} ({}) from root", + meta.name, + meta.code + ); + } + for (code, _, _) in &expected { + assert!( + lookup_command(*code).is_some_and(|meta| !meta.is_replicated()), + "ratchet row {code} names no non-replicated table entry" + ); + } + } + + #[test] + fn unknown_code_is_refused_as_invalid_command() { + assert!( + lookup_command(UNKNOWN_CODE).is_none(), + "test needs a code absent from COMMAND_TABLE" + ); + let shard = gate_shard(); + let invalid_command = Verdict::Deny(IggyError::InvalidCommand.as_code()); + assert_eq!(gate(&shard, UNKNOWN_CODE, None), invalid_command); + assert_eq!( + gate(&shard, UNKNOWN_CODE, Some(DEFAULT_ROOT_USER_ID)), + invalid_command + ); + } + + #[test] + fn replicated_code_in_a_non_replicated_frame_is_refused_as_feature_unavailable() { + let shard = gate_shard(); + let feature_unavailable = Verdict::Deny(IggyError::FeatureUnavailable.as_code()); + assert_eq!(gate(&shard, CREATE_STREAM_CODE, None), feature_unavailable); + assert_eq!( + gate(&shard, CREATE_STREAM_CODE, Some(DEFAULT_ROOT_USER_ID)), + feature_unavailable + ); + } +} diff --git a/core/server/src/dispatch/failure.rs b/core/server/src/dispatch/failure.rs new file mode 100644 index 0000000000..37bb5a5c61 --- /dev/null +++ b/core/server/src/dispatch/failure.rs @@ -0,0 +1,629 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! The wire failure channels and the one send exit for host-built frames. +//! +//! Every frame the dispatch host builds, success or rejection, leaves through +//! [`send_host_frame`], so the send-failure log has one shape. Which channel a +//! failure rides is a wire contract with the SDK: +//! +//! | channel | carrier | when | +//! |---|---|---| +//! | [`FrameChannel::TypedDeny`] | Reply, nonzero status + empty body, or a result-framed rejection body | rejections that must unblock the SDK's lockstep request slot: checksum, authz, pre-consensus rewrite, unknown or unsupported non-replicated code, unbound non-PING read, transient replay hints | +//! | [`FrameChannel::Eviction`] | session-terminal Eviction frame with a typed reason | the client must register again: `NoSession`, `MalformedLogin`, heartbeat and login evictions. The reason rides the channel label, since one `context` covers four of them | +//! | [`FrameChannel::ResyncSentinel`] | status-0 poll reply, body carries `RESYNC_REQUIRED_PARTITION_SENTINEL` | a fenced consumer-group poll: the consumer must re-sync its assignment; HTTP mirrors it as `resync_required_polled_messages` in `crate::http::wire` | +//! | [`FrameChannel::EmptyFrame`] | status-0 fail-fast body, the 16-byte empty poll | the partition cannot answer yet; the SDK fails fast (empty poll) and retries. A permanent client error never rides this channel: an undecodable body, an unresolved target, and a request the resolve rejects all deny typed, because there is nothing to retry | +//! | [`FrameChannel::Reply`] | status-0 success frame | host-built success replies: login/register, ping, logout, non-replicated read bodies, committed metadata replies | +//! | silent drop | no frame | one deliberate case, a transient consensus submit failure: the SDK read-timeout replays the same request id, and a synthesized failure could contradict a write that commits moments later. A header `RequestHeader::validate` rejected also drops, but that one is a GAP, not a contract - the fields decode, so a deny could be echoed under the transport id, and the client instead waits out its read timeout | +//! | HTTP status | HTTP status code | the HTTP spine maps the same rejections in `crate::http::error`; it never rides these frames | +//! +//! The last two send nothing, so [`FrameChannel`] has no variant for them. +//! +//! Scope: the table covers `crate::dispatch` only. Two neighbours answer on +//! their own paths by design - the partitions engine builds and sends +//! produce/poll replies, and the shard crate builds client-shaped denies of +//! its own (`IggyShard::deny_partition_request_transient` and +//! `stage_transient_deny`, both `TypedDeny`-shaped, the latter shedding the +//! frame outright when its lifecycle queue is full). + +use crate::responses::{NonReplicatedResponse, build_deny_reply, current_metadata_commit}; +use crate::rewrite::RewriteStage; +use crate::shell::{ShellBus, ShellShard}; +use bytes::Bytes; +use consensus::{ + EvictionContext, MetadataHandle, build_eviction_message, + build_incompatible_protocol_eviction_message, build_result_rejection_reply, +}; +use iggy_binary_protocol::{EvictionReason, PrepareHeader, RoutedRequestHeader}; +use iggy_common::IggyError; +use journal::superblock::SuperblockStore; +use journal::{Journal, JournalHandle}; +use message_bus::BusMessage; +use server_common::Message; +use std::rc::Rc; +use tracing::warn; + +/// Labels the channel a host-built frame rides, for the send-failure log. +/// The taxonomy, including the two channels that never construct a frame, +/// is on the module doc. +#[derive(Clone, Copy, Debug)] +pub(in crate::dispatch) enum FrameChannel { + TypedDeny, + /// The reason travels with the channel: five call sites share the + /// `"login_rejection"` context across four distinct reasons, so + /// `context` alone cannot tell a `MalformedLogin` send failure from an + /// `InvalidCredentials` one. + Eviction(EvictionReason), + ResyncSentinel, + EmptyFrame, + Reply, +} + +impl std::fmt::Display for FrameChannel { + fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::TypedDeny => formatter.write_str("typed_deny"), + Self::Eviction(reason) => write!(formatter, "eviction({reason:?})"), + Self::ResyncSentinel => formatter.write_str("resync_sentinel"), + Self::EmptyFrame => formatter.write_str("empty_frame"), + Self::Reply => formatter.write_str("reply"), + } + } +} + +/// The one send exit for host-built client frames. Best-effort: a failed +/// send means the connection is gone (or its queue is full), and there is +/// nothing left to reply on, so the error is logged and dropped. `frame` is +/// any bus message: a contiguous frozen frame or the vectored poll reply. +#[allow(clippy::future_not_send)] +pub(in crate::dispatch) async fn send_host_frame( + bus: &B, + transport_client_id: u128, + frame: impl Into, + channel: FrameChannel, + context: &'static str, +) { + if let Err(send_error) = bus.send_to_client(transport_client_id, frame).await { + warn!( + transport_client_id, + error = %send_error, + channel = %channel, + context, + "failed to send host frame to client" + ); + } +} + +/// Reply to a request rejected before it reached its plane with the request's +/// own frame: empty body + nonzero `status`. The nonzero status is the whole +/// point: the SDK peeks it and surfaces the typed error, whereas a status-0 +/// frame reads as a committed ack for work that never happened. Silence is no +/// better, the connection decodes replies in lockstep and would wedge on every +/// later request. +#[allow(clippy::future_not_send)] +pub(in crate::dispatch) async fn send_deny_reply( + shard: &Rc>, + transport_client_id: u128, + request_header: &RoutedRequestHeader, + status: u32, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let commit = current_metadata_commit(shard); + let reply = build_deny_reply(request_header, transport_client_id, 0, commit, status); + send_host_frame( + &shard.bus, + transport_client_id, + reply.into_generic().into_frozen(), + FrameChannel::TypedDeny, + "request_denial", + ) + .await; +} + +/// Deny a request from an unbound transport without disclosing the metadata +/// commit frontier. The status is the only field a pre-authenticated caller +/// needs, while the live commit would expose cluster write activity. +#[allow(clippy::future_not_send)] +pub(in crate::dispatch) async fn send_unbound_deny_reply( + shard: &Rc>, + transport_client_id: u128, + request_header: &RoutedRequestHeader, + status: u32, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let reply = build_deny_reply(request_header, transport_client_id, 0, 0, status); + send_host_frame( + &shard.bus, + transport_client_id, + reply.into_generic().into_frozen(), + FrameChannel::TypedDeny, + "unbound_request_denial", + ) + .await; +} + +/// Reply to a denied non-replicated read with the request's reply frame: empty +/// body + nonzero `status`. The SDK peeks the status before body decode and +/// surfaces the typed error, so a poll denial never reaches the empty-poll +/// "0 messages" body path. +#[allow(clippy::future_not_send)] +pub(in crate::dispatch) async fn send_non_replicated_deny( + shard: &Rc>, + request: &Message, + transport_client_id: u128, + status: u32, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let commit = current_metadata_commit(shard); + let reply = build_deny_reply( + request.header(), + request.header().client, + request.header().session, + commit, + status, + ); + send_host_frame( + &shard.bus, + transport_client_id, + reply.into_generic().into_frozen(), + FrameChannel::TypedDeny, + "non_replicated_denial", + ) + .await; +} + +/// Reject a request before it reaches consensus: warn, then send the typed +/// deny reply. A silent drop would wedge every later request on the +/// connection until the socket read timeout. `stage` names the chain step +/// for both log lines, so the set stays enumerable. +#[allow(clippy::future_not_send)] +pub(in crate::dispatch) async fn send_pre_consensus_deny( + shard: &Rc>, + transport_client_id: u128, + request_header: &RoutedRequestHeader, + error: &IggyError, + stage: RewriteStage, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let context = stage.as_str(); + warn!( + transport_client_id, + error = %error, + operation = ?request_header.operation, + context, + "denying request pre-consensus" + ); + let commit = current_metadata_commit(shard); + let reply = build_deny_reply( + request_header, + transport_client_id, + 0, + commit, + error.as_code(), + ); + send_host_frame( + &shard.bus, + transport_client_id, + reply.into_generic().into_frozen(), + FrameChannel::TypedDeny, + context, + ) + .await; +} + +/// Result-framed rejection Reply: status 0, body `[count=1][index][code]`. +/// The SDK decodes the nonzero result code, so a transient code makes it +/// replay the same request at once instead of waiting out its read timeout. +/// Replying empty instead would surface as a hard `InvalidFormat` decode +/// failure and break the replay. +#[allow(clippy::future_not_send)] +pub(in crate::dispatch) async fn send_result_rejection( + shard: &Rc>, + transport_client_id: u128, + request_header: &RoutedRequestHeader, + error: &IggyError, + context: &'static str, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let commit = current_metadata_commit(shard); + let reply = build_result_rejection_reply(request_header, commit, error.as_code()); + send_host_frame( + &shard.bus, + transport_client_id, + reply.into_generic().into_frozen(), + FrameChannel::TypedDeny, + context, + ) + .await; +} + +/// Best-effort session-terminal `Eviction` frame: the client's session is +/// gone (or was never granted), so it must register again. Every frame +/// transport decodes `Command::Eviction` and maps the typed reason +/// (`NoSession` -> `Unauthenticated`, ...), so clients fail fast with the +/// real cause instead of a body-decode failure or a timeout. Consensus +/// context (cluster/view/replica) is stamped on the metadata shard and +/// zeroed elsewhere; the SDK only reads the reason, plus the protocol +/// window on `IncompatibleProtocol`. +#[allow(clippy::future_not_send)] +pub(in crate::dispatch) async fn send_eviction( + shard: &Rc>, + transport_client_id: u128, + vsr_client_id: u128, + reason: EvictionReason, + context: &'static str, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let ctx = shard.plane.metadata().consensus.as_ref().map_or( + EvictionContext { + cluster: 0, + view: 0, + replica: 0, + }, + EvictionContext::from_consensus, + ); + let eviction = match reason { + EvictionReason::IncompatibleProtocol => { + build_incompatible_protocol_eviction_message(ctx, vsr_client_id) + } + _ => build_eviction_message(ctx, vsr_client_id, reason), + }; + send_host_frame( + &shard.bus, + transport_client_id, + eviction.into_generic().into_frozen(), + FrameChannel::Eviction(reason), + context, + ) + .await; +} + +/// Send a non-replicated reply body to a client, stamping the current +/// metadata commit. Shared by the non-replicated read arms; `channel` +/// labels the body shape (a real answer, a fail-fast empty poll, or the +/// re-sync sentinel) for the send-failure log. +#[allow(clippy::future_not_send)] +pub(in crate::dispatch) async fn send_non_replicated_bytes( + shard: &Rc>, + request: &Message, + transport_client_id: u128, + bytes: Bytes, + channel: FrameChannel, + context: &'static str, +) where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let commit = current_metadata_commit(shard); + let reply = NonReplicatedResponse::Bytes(bytes).into_reply( + request.header(), + request.header().client, + request.header().session, + commit, + ); + send_host_frame( + &shard.bus, + transport_client_id, + reply.into_generic().into_frozen(), + channel, + context, + ) + .await; +} + +// Byte snapshots pinning each channel's frame to the pre-refactor inline +// construction. What they hold is that routing a rejection through this +// module changed no byte a client sees: same command, same status, same +// header echo, same body length per channel. They are NOT the wire +// contract - a deliberate protocol change updates them. +#[cfg(test)] +mod tests { + use super::*; + use crate::dispatch::handle_client_request; + use crate::dispatch::test_support::{ + FIRST_BOOT, SpyBus, TestShard, request_message, test_shard, + }; + use crate::responses::build_empty_reply; + use crate::session_manager::SessionManager; + use configs::server::ServerSystemConfig; + use iggy_binary_protocol::Operation; + use iggy_binary_protocol::codes::PING_CODE; + use iggy_common::RESYNC_REQUIRED_PARTITION_SENTINEL; + use std::cell::RefCell; + use std::sync::Arc; + + const TRANSPORT: u128 = 42; + const VSR_CLIENT: u128 = 7; + const SESSION: u64 = 3; + const REQUEST: u64 = 5; + + fn snapshot_shard() -> (SpyBus, Rc) { + let bus = SpyBus::default(); + let shard = Rc::new(test_shard(&bus, 0, 1, FIRST_BOOT)); + (bus, shard) + } + + fn sole_client_frame(bus: &SpyBus) -> (u128, Vec) { + let replies = bus.client_replies.borrow(); + assert_eq!(replies.len(), 1, "expected exactly one client-bound frame"); + replies[0].clone() + } + + fn poll_request() -> Message { + request_message(Operation::NonReplicated, VSR_CLIENT, SESSION, REQUEST, &[]) + } + + /// The 16-byte empty `PolledMessages` body exactly as the pre-refactor + /// `partition::empty_polled_messages_body` built it. + fn old_empty_polled_messages_body(partition_id: u32) -> Bytes { + let mut body = Vec::with_capacity(16); + body.extend_from_slice(&partition_id.to_le_bytes()); + body.extend_from_slice(&0u64.to_le_bytes()); + body.extend_from_slice(&0u32.to_le_bytes()); + Bytes::from(body) + } + + fn old_eviction_context(shard: &Rc) -> EvictionContext { + shard.plane.metadata().consensus.as_ref().map_or( + EvictionContext { + cluster: 0, + view: 0, + replica: 0, + }, + EvictionContext::from_consensus, + ) + } + + #[compio::test] + async fn snapshot_typed_deny_commit_stamped_frame_unchanged() { + let (bus, shard) = snapshot_shard(); + let request = poll_request(); + let status = IggyError::Unauthenticated.as_code(); + let old = build_deny_reply( + request.header(), + TRANSPORT, + 0, + current_metadata_commit(&shard), + status, + ) + .into_generic(); + + send_deny_reply(&shard, TRANSPORT, request.header(), status).await; + + let (target, frame) = sole_client_frame(&bus); + assert_eq!(target, TRANSPORT); + assert_eq!(frame, old.as_slice().to_vec()); + } + + #[compio::test] + async fn snapshot_typed_deny_unbound_frame_unchanged() { + let (bus, shard) = snapshot_shard(); + let request = poll_request(); + let status = IggyError::Unauthenticated.as_code(); + let old = build_deny_reply(request.header(), TRANSPORT, 0, 0, status).into_generic(); + + send_unbound_deny_reply(&shard, TRANSPORT, request.header(), status).await; + + let (target, frame) = sole_client_frame(&bus); + assert_eq!(target, TRANSPORT); + assert_eq!(frame, old.as_slice().to_vec()); + } + + #[compio::test] + async fn snapshot_typed_deny_non_replicated_frame_unchanged() { + let (bus, shard) = snapshot_shard(); + let request = poll_request(); + let status = IggyError::Unauthenticated.as_code(); + let old = build_deny_reply( + request.header(), + request.header().client, + request.header().session, + current_metadata_commit(&shard), + status, + ) + .into_generic(); + + send_non_replicated_deny(&shard, &request, TRANSPORT, status).await; + + let (target, frame) = sole_client_frame(&bus); + assert_eq!(target, TRANSPORT); + assert_eq!(frame, old.as_slice().to_vec()); + } + + #[compio::test] + async fn snapshot_typed_deny_result_framed_frame_unchanged() { + let (bus, shard) = snapshot_shard(); + let request = poll_request(); + let code = IggyError::TransientNotAccepted; + let old = build_result_rejection_reply( + request.header(), + current_metadata_commit(&shard), + code.as_code(), + ) + .into_generic(); + + send_result_rejection(&shard, TRANSPORT, request.header(), &code, "snapshot").await; + + let (target, frame) = sole_client_frame(&bus); + assert_eq!(target, TRANSPORT); + assert_eq!(frame, old.as_slice().to_vec()); + } + + #[compio::test] + async fn snapshot_eviction_frame_unchanged() { + let (bus, shard) = snapshot_shard(); + let ctx = old_eviction_context(&shard); + + // Old `send_unauthenticated_eviction`: reason NoSession, client id = + // the transport id. + let old = build_eviction_message(ctx, TRANSPORT, EvictionReason::NoSession).into_generic(); + send_eviction( + &shard, + TRANSPORT, + TRANSPORT, + EvictionReason::NoSession, + "snapshot", + ) + .await; + let (target, frame) = sole_client_frame(&bus); + assert_eq!(target, TRANSPORT); + assert_eq!(frame, old.as_slice().to_vec()); + bus.client_replies.borrow_mut().clear(); + + // Old `send_login_eviction`: reason MalformedLogin, client id = the + // request's VSR client id. + let old = + build_eviction_message(ctx, VSR_CLIENT, EvictionReason::MalformedLogin).into_generic(); + send_eviction( + &shard, + TRANSPORT, + VSR_CLIENT, + EvictionReason::MalformedLogin, + "snapshot", + ) + .await; + let (target, frame) = sole_client_frame(&bus); + assert_eq!(target, TRANSPORT); + assert_eq!(frame, old.as_slice().to_vec()); + bus.client_replies.borrow_mut().clear(); + + // Old `send_login_eviction` on `IncompatibleProtocol`: the protocol + // window rides the frame, client id = the request's VSR client id. + let old = build_incompatible_protocol_eviction_message(ctx, VSR_CLIENT).into_generic(); + send_eviction( + &shard, + TRANSPORT, + VSR_CLIENT, + EvictionReason::IncompatibleProtocol, + "snapshot", + ) + .await; + let (target, frame) = sole_client_frame(&bus); + assert_eq!(target, TRANSPORT); + assert_eq!(frame, old.as_slice().to_vec()); + } + + /// The HTTP mirror (`resync_required_polled_messages` in + /// `crate::http::wire`) is a JSON DTO, not a wire frame, so the shared + /// contract asserted here is the sentinel constant itself; the DTO's own + /// test pins its `partition_id` to the same constant. + #[compio::test] + async fn snapshot_resync_sentinel_frame_unchanged() { + let (bus, shard) = snapshot_shard(); + let request = poll_request(); + let body = old_empty_polled_messages_body(RESYNC_REQUIRED_PARTITION_SENTINEL); + assert_eq!( + body[..4], + RESYNC_REQUIRED_PARTITION_SENTINEL.to_le_bytes(), + "sentinel poll body must lead with the re-sync sentinel partition id" + ); + let old = NonReplicatedResponse::Bytes(body.clone()) + .into_reply( + request.header(), + request.header().client, + request.header().session, + current_metadata_commit(&shard), + ) + .into_generic(); + + send_non_replicated_bytes( + &shard, + &request, + TRANSPORT, + body, + FrameChannel::ResyncSentinel, + "poll_messages", + ) + .await; + + let (target, frame) = sole_client_frame(&bus); + assert_eq!(target, TRANSPORT); + assert_eq!(frame, old.as_slice().to_vec()); + } + + /// The PING reply is the one host-built success Reply the funnel serves + /// without a consensus round; the old side is the frame the reads + /// router's PING arm built inline before the exit existed. + #[compio::test] + async fn snapshot_reply_frame_unchanged() { + let (bus, shard) = snapshot_shard(); + let sessions = Rc::new(RefCell::new(SessionManager::new())); + let system_config = Arc::new(ServerSystemConfig::default()); + let request = request_message(Operation::NonReplicated, VSR_CLIENT, SESSION, REQUEST, &[]) + .transmute_header(|header, ping: &mut RoutedRequestHeader| { + *ping = header; + ping.reserved[..4].copy_from_slice(&PING_CODE.to_le_bytes()); + // The funnel promotes the client wire header with `group` + // unset, so the old side must build from the same bytes. + ping.group = 0; + }); + let old = build_empty_reply( + request.header(), + request.header().client, + request.header().session, + current_metadata_commit(&shard), + ) + .into_generic(); + + handle_client_request( + &shard, + &sessions, + &system_config, + 1, + TRANSPORT, + request.into_generic(), + ) + .await; + + let (target, frame) = sole_client_frame(&bus); + assert_eq!(target, TRANSPORT); + assert_eq!(frame, old.as_slice().to_vec()); + } +} diff --git a/core/server/src/dispatch/mod.rs b/core/server/src/dispatch/mod.rs index 68053050fa..50b40dcd25 100644 --- a/core/server/src/dispatch/mod.rs +++ b/core/server/src/dispatch/mod.rs @@ -20,7 +20,9 @@ //! The tree: [`session_ops`] (login/register/logout and their replica //! forwards), [`partition`] (the partition data plane, both mesh ends), //! [`reads`] (the non-replicated read router), [`submit`] (the shard-0 -//! metadata-submit RPC), `authz` (the wire-path authorization gates). +//! metadata-submit RPC), `authz` (the wire-path authorization gates), +//! `failure` (the wire failure channels and the one send exit for +//! host-built frames). //! //! Deliberate asymmetry (the two authz gates): replicated metadata ops are //! authorized in-apply by the STM, in committed order on every replica; @@ -30,6 +32,7 @@ //! error contract (404-before-403) is pinned client-visible behavior. mod authz; +mod failure; pub mod login_error; pub mod partition; pub mod reads; @@ -39,61 +42,44 @@ pub mod submit; mod test_support; use crate::consumer_group::maybe_rewrite_consumer_group_request; -use crate::dispatch::authz::{send_deny_reply, send_unbound_deny_reply}; +use crate::dispatch::failure::{ + FrameChannel, send_deny_reply, send_eviction, send_host_frame, send_pre_consensus_deny, + send_unbound_deny_reply, +}; use crate::dispatch::partition::{dispatch_partition_request, handle_delete_segments_request}; use crate::dispatch::reads::handle_non_replicated_request; use crate::dispatch::session_ops::{ - handle_login_register_request, handle_logout_request, send_login_eviction, - send_unauthenticated_eviction, submit_disconnect_logout, + handle_login_register_request, handle_logout_request, submit_disconnect_logout, }; use crate::dispatch::submit::{committed_reply_commit, submit_client_request_on_owner}; -use crate::pat::maybe_rewrite_pat_request; -use crate::responses::{ - NonReplicatedResponse, build_deny_reply, build_raw_pat_reply, current_metadata_commit, -}; -use crate::segment_cleaner::UNENFORCEABLE_TOPIC_SIZE_WARN; -use crate::session_manager::SessionManager; +use crate::responses::build_raw_pat_reply; +use crate::rewrite::{RewriteDeny, RewriteStage, tcp_chain}; +use crate::session_manager::{ConnectionContext, SessionManager}; use crate::shell::{ShellBus, ShellShard, ShellShardHandle}; -use crate::users::maybe_rewrite_user_password_request; -use crate::wire::{request_body, verify_request_checksum}; -use bytes::Bytes; +use crate::wire::verify_request_checksum; +use ahash::{AHashMap, AHashSet}; use configs::server::ServerSystemConfig; -use consensus::MetadataHandle; use iggy_binary_protocol::PrepareHeader; use iggy_binary_protocol::codes::{ - GET_CLUSTER_METADATA_CODE, LOGIN_USER_CODE, LOGIN_WITH_PERSONAL_ACCESS_TOKEN_CODE, PING_CODE, + LOGIN_USER_CODE, LOGIN_WITH_PERSONAL_ACCESS_TOKEN_CODE, PING_CODE, }; -use iggy_binary_protocol::requests::partitions::{ - CreatePartitionsRequest, DeletePartitionsRequest, -}; -use iggy_binary_protocol::requests::streams::{CreateStreamRequest, UpdateStreamRequest}; -use iggy_binary_protocol::requests::topics::{CreateTopicRequest, UpdateTopicRequest}; -use iggy_binary_protocol::requests::users::{CreateUserRequest, UpdateUserRequest}; use iggy_binary_protocol::{ - EvictionReason, GenericHeader, MAX_PARTITIONS_PER_REQUEST, Operation, RequestHeader, - RoutedRequestHeader, WireDecode, WireIdentifier, WireOptions, -}; -use iggy_common::{ - IggyByteSize, IggyError, MaxTopicSize, TopicCreateOptions, UPDATABLE_STREAM_OPTION_KEYS, - UPDATABLE_TOPIC_OPTION_KEYS, UPDATABLE_USER_OPTION_KEYS, validate_preallocated_topic_bytes, - validate_topic_segment_size, + EvictionReason, GenericHeader, Operation, RequestHeader, RoutedRequestHeader, }; +use iggy_common::IggyError; use journal::superblock::SuperblockStore; use journal::{Journal, JournalHandle}; -use message_bus::BusMessage; use message_bus::client_listener::RequestHandler; use message_bus::replica::listener::MessageHandler; -use metadata::impls::metadata::StreamsFrontend; -use metadata::stm::stream::Streams; use server_common::Message; use shard::{ConnectedClientInfo, ListClientsHandler}; use std::cell::RefCell; -use std::collections::{HashMap, HashSet, VecDeque}; +use std::collections::VecDeque; use std::rc::Rc; use std::sync::Arc; -use tracing::{debug, warn}; +use tracing::{debug, error, warn}; -type ClientRequestQueues = Rc>>>>; +type ClientRequestQueues = Rc>>>>; /// Requests one client may have queued behind a request this shard has not /// answered yet. @@ -109,50 +95,7 @@ type ClientRequestQueues = Rc>>; - -pub fn make_client_request_handler( - shard: &Rc>, - sessions: &Rc>, - system_config: Arc, - max_tokens_per_user: u32, -) -> RequestHandler -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let shard = Rc::clone(shard); - let sessions = Rc::clone(sessions); - let queues: ClientRequestQueues = Rc::new(RefCell::new(HashMap::new())); - let active: ActiveClientRequests = Rc::new(RefCell::new(HashSet::new())); - let sessions_for_disconnect = Rc::clone(&sessions); - let shard_for_disconnect = Rc::clone(&shard); - shard - .bus - .set_client_connection_lost_fn(Rc::new(move |client_id| { - if let Some((vsr_client_id, session)) = sessions_for_disconnect - .borrow_mut() - .remove_connection(client_id) - { - submit_disconnect_logout(Rc::clone(&shard_for_disconnect), vsr_client_id, session); - } - })); - Rc::new(move |client_id, message| { - enqueue_client_request( - Rc::clone(&shard), - Rc::clone(&sessions), - Arc::clone(&system_config), - max_tokens_per_user, - Rc::clone(&queues), - Rc::clone(&active), - client_id, - message, - ); - }) -} +type ActiveClientRequests = Rc>>; /// Build the per-shard [`ListClientsHandler`]: on a `ListClients` /// broadcast, serialize this shard's locally-homed connected clients from @@ -187,6 +130,12 @@ where }) } +/// Build the shard's one client-request handler: per-client FIFO queues +/// drained one task per client, and the bus connection-lost hook that +/// logs a dropped connection out. Every transport on the shard must +/// share the instance (shard 0 hands it to its local QUIC, TCP-TLS and +/// WSS listeners as well), or a client's ordering guarantee and the +/// disconnect hook split by transport. pub fn make_deferred_client_request_handler( bus: &B, shard_handle: &ShellShardHandle, @@ -203,50 +152,55 @@ where { let shard_handle = Rc::clone(shard_handle); let sessions = Rc::clone(sessions); - let queues: ClientRequestQueues = Rc::new(RefCell::new(HashMap::new())); - let active: ActiveClientRequests = Rc::new(RefCell::new(HashSet::new())); + let queues: ClientRequestQueues = Rc::new(RefCell::new(AHashMap::new())); + let active: ActiveClientRequests = Rc::new(RefCell::new(AHashSet::new())); + let queues_for_disconnect = Rc::clone(&queues); let sessions_for_disconnect = Rc::clone(&sessions); let shard_handle_for_disconnect = Rc::clone(&shard_handle); let bus_for_spawn = (*bus).clone(); bus.set_client_connection_lost_fn(Rc::new(move |client_id| { + // The socket is gone, so nothing will drain what a live drain task + // left queued. The active slot is NOT released here: the transport + // task runs this hook while a drain may be suspended at an `.await`, + // and clearing the slot would let a frame the dispatch task still has + // buffered spawn a second drain over the same queue. The drain task's + // own guard covers every exit, the panic compio catches included. + queues_for_disconnect.borrow_mut().remove(&client_id); + // Upgrade FIRST: `remove_connection` strips the `SessionManager` + // entry, so running it ahead of a failed upgrade would drop the + // binding without ever submitting the replicated `Logout`, leaking + // the `ClientTable` entry and its consumer-group memberships. The + // window is pre-build / post-runtime-drop only. + let Some(shard) = upgrade_shard_handle(&shard_handle_for_disconnect) else { + // Nothing reaps what stays behind: the heartbeat verifier is + // optional and only collects `Bound` / `Authenticated` sessions, + // so a `Connected` row survives to process exit. + error!( + client_id, + "client connection lost with no live shard; session and client-table entries \ + leak until process exit" + ); + return; + }; if let Some((vsr_client_id, session)) = sessions_for_disconnect .borrow_mut() .remove_connection(client_id) - && let Some(shard) = upgrade_shard_handle(&shard_handle_for_disconnect) { submit_disconnect_logout(shard, vsr_client_id, session); } })); Rc::new(move |client_id, message| { - let shard_handle = Rc::clone(&shard_handle); - let sessions = Rc::clone(&sessions); - let system_config = Arc::clone(&system_config); - let queues = Rc::clone(&queues); - let active = Rc::clone(&active); - queues - .borrow_mut() - .entry(client_id) - .or_default() - .push_back(message); - if !active.borrow_mut().insert(client_id) { - return; - } - bus_for_spawn.spawn(async move { - let Some(shard) = upgrade_shard_handle(&shard_handle) else { - active.borrow_mut().remove(&client_id); - return; - }; - drain_client_requests( - shard, - sessions, - system_config, - max_tokens_per_user, - queues, - active, - client_id, - ) - .await; - }); + enqueue_client_request( + &bus_for_spawn, + &shard_handle, + &sessions, + &system_config, + max_tokens_per_user, + &queues, + &active, + client_id, + message, + ); }) } @@ -290,12 +244,13 @@ where #[allow(clippy::too_many_arguments)] fn enqueue_client_request( - shard: Rc>, - sessions: Rc>, - system_config: Arc, + bus: &B, + shard_handle: &ShellShardHandle, + sessions: &Rc>, + system_config: &Arc, max_tokens_per_user: u32, - queues: ClientRequestQueues, - active: ActiveClientRequests, + queues: &ClientRequestQueues, + active: &ActiveClientRequests, client_id: u128, message: Message, ) where @@ -312,7 +267,7 @@ fn enqueue_client_request( // Borrow released before the deny, which spawns onto this same // task and would otherwise re-enter the table. drop(queues); - deny_overflowing_client_request(&shard, client_id, message); + deny_overflowing_client_request(shard_handle, client_id, message); return; } queue.push_back(message); @@ -321,15 +276,25 @@ fn enqueue_client_request( return; } - let bus = shard.bus.clone(); + let shard_handle = Rc::clone(shard_handle); + let sessions = Rc::clone(sessions); + let system_config = Arc::clone(system_config); + let queues = Rc::clone(queues); + let active = Rc::clone(active); bus.spawn(async move { + let _slot = ActiveDrainSlot { active, client_id }; + // The handle is set once the shard is built. A frame that beats it + // stays queued and the client's next frame drains both, which needs + // the slot released on this path too - the guard above does it. + let Some(shard) = upgrade_shard_handle(&shard_handle) else { + return; + }; drain_client_requests( shard, sessions, system_config, max_tokens_per_user, queues, - active, client_id, ) .await; @@ -340,10 +305,12 @@ fn enqueue_client_request( /// [`MAX_QUEUED_CLIENT_REQUESTS`] with the retryable transient denial. /// /// Spawned rather than awaited: the enqueue path is sync (it runs straight off -/// frame arrival) and the reply goes out on the bus. A frame whose header will -/// not even cast is dropped instead, exactly as the drain loop drops it. +/// frame arrival) and the reply goes out on the bus. Two shapes are dropped +/// rather than answered: a frame whose header will not even cast, exactly as +/// the drain loop drops it, and a frame that arrives before the shard is +/// built, which leaves nothing to render a reply from. fn deny_overflowing_client_request( - shard: &Rc>, + shard_handle: &ShellShardHandle, transport_client_id: u128, message: Message, ) where @@ -353,6 +320,17 @@ fn deny_overflowing_client_request( S: 'static, SB: SuperblockStore + 'static, { + // Unreachable in practice: a listener binds only after the build backfills + // the weak self-reference, which `crate::boot` pins with a `debug_assert`. + // Dropped rather than admitted past the cap -- the cap exists to refuse + // this frame, and the deny itself needs the shard for its metric and reply. + let Some(shard) = upgrade_shard_handle(shard_handle) else { + warn!( + transport_client_id, + "dropping over-queue client request received before the shard was built" + ); + return; + }; shard.metrics().record_client_request_denied_queue_full(); let Ok(request) = message.try_into_typed::() else { warn!( @@ -368,7 +346,6 @@ fn deny_overflowing_client_request( queued = MAX_QUEUED_CLIENT_REQUESTS, "denying client request retryable: this connection's request queue is full" ); - let shard = Rc::clone(shard); let bus = shard.bus.clone(); bus.spawn(async move { send_deny_reply( @@ -381,6 +358,24 @@ fn deny_overflowing_client_request( }); } +/// Holds a client's one drain slot for as long as its drain task lives. +/// +/// Releasing from the task's own `Drop` covers every exit, the panic compio +/// catches included: a slot left taken with no live drain queues every later +/// frame for that client forever. It also keeps the release out of the +/// connection-lost hook, which fires from the transport task and could +/// otherwise clear the slot under a drain suspended at an `.await`. +struct ActiveDrainSlot { + active: ActiveClientRequests, + client_id: u128, +} + +impl Drop for ActiveDrainSlot { + fn drop(&mut self) { + self.active.borrow_mut().remove(&self.client_id); + } +} + #[allow(clippy::future_not_send)] async fn drain_client_requests( shard: Rc>, @@ -388,7 +383,6 @@ async fn drain_client_requests( system_config: Arc, max_tokens_per_user: u32, queues: ClientRequestQueues, - active: ActiveClientRequests, client_id: u128, ) where B: ShellBus, @@ -398,7 +392,7 @@ async fn drain_client_requests( SB: SuperblockStore + 'static, { loop { - let Some(message) = pop_next_client_request(&queues, &active, client_id) else { + let Some(message) = pop_next_client_request(&queues, client_id) else { return; }; handle_client_request( @@ -415,211 +409,110 @@ async fn drain_client_requests( fn pop_next_client_request( queues: &ClientRequestQueues, - active: &ActiveClientRequests, client_id: u128, ) -> Option> { + // Freed as soon as it drains empty. Keeping it would cost the connection + // one `VecDeque` (at its peak depth, since `pop_front` never gives + // capacity back), and the connection-lost hook cannot be the sole owner: + // the dispatch task can still deliver a buffered frame after the hook + // ran, and `client_id` is never reused, so that entry would live until + // process exit. let mut queues = queues.borrow_mut(); - let Some(queue) = queues.get_mut(&client_id) else { - active.borrow_mut().remove(&client_id); - return None; - }; + let queue = queues.get_mut(&client_id)?; let message = queue.pop_front(); if queue.is_empty() { queues.remove(&client_id); } - if message.is_none() { - active.borrow_mut().remove(&client_id); - } message } -/// Per-request partitions-count cap, shared by create-topic, create-partitions -/// and delete-partitions admission. Runs pre-consensus like -/// [`validate_topic_bounds`]: an oversized count must not burn a replicated -/// log entry (create-partitions admission would also allocate that many -/// consensus-group ids before replicating). -/// -/// Zero passes here because a zero-partition TOPIC is legal (legacy -/// `create_topic` admits `0..=MAX`); the add/remove requests reject it in -/// [`validate_partitions_change_count`]. -const fn validate_partitions_count(partitions_count: u32) -> Result<(), IggyError> { - if partitions_count > MAX_PARTITIONS_PER_REQUEST { - return Err(IggyError::TooManyPartitions); - } - Ok(()) +/// Where the funnel routes a client request. Derived by [`classify`], which +/// IS the routing: [`handle_client_request`] matches on its result. The +/// variant ORDER mirrors the order of the checks inside [`classify`], and +/// that order is semantics (documented there). +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(in crate::dispatch) enum RequestClass { + /// Legacy pre-register login code: rejected with a typed + /// `MalformedLogin` eviction before the session gate. + LegacyLogin, + /// Non-replicated code other than PING on an unbound transport: + /// denied `Unauthenticated` (plain deny reply, never an eviction). + UnauthenticatedRead, + /// Non-replicated read for the reads router. + NonReplicatedRead, + /// The register handshake (`session == 0 && request == 0`). + LoginRegister, + Logout, + /// Replicated operation on an unbound transport: `Eviction(NoSession)`. + UnboundReplicated, + /// Neither a partition nor a metadata consensus op: resolved to a + /// replicated `TruncatePartition` by the owning shard. + DeleteSegments, + /// Partition-plane operation. + Partition, + /// Everything else: replicated metadata consensus. + ReplicatedMetadata, } -/// [`validate_partitions_count`] plus the zero rejection that create-partitions -/// and delete-partitions carry: adding or removing zero partitions is a no-op -/// that would still burn a replicated log entry, bump `Streams::revision` and -/// force every shard through a rebalance pass. Legacy rejects it with -/// `TooManyPartitions` in both handlers (`1..=MAX` on create, `== 0` on -/// delete), so the code matches rather than inventing a new one. -const fn validate_partitions_change_count(partitions_count: u32) -> Result<(), IggyError> { - if partitions_count == 0 { - return Err(IggyError::TooManyPartitions); +/// Route a client request: [`handle_client_request`] matches on the result, +/// so this function IS the routing and the order of the checks below is the +/// semantics. Pure: `bound` stands for +/// `sessions.get_session(transport_client_id).is_some()`, the only session +/// fact the checks consult. +/// +/// The pins, in check order: +/// - the legacy-login rejection precedes the session gate: a legacy code +/// must get the typed `MalformedLogin` eviction, not the generic +/// unauthenticated deny; +/// - the pre-auth allowlist is PING only; `GET_CLUSTER_METADATA` is +/// deliberately NOT pre-auth (see the funnel's auth-bypass guard); +/// - poll-messages and consumer-offset reads are non-replicated CODES, not +/// partition operations: they classify +/// [`RequestClass::NonReplicatedRead`] and route inside the reads router; +/// - a `Register` with `session != 0` or `request != 0` falls through to +/// the default. Those rows are classify-only: `RequestHeader::validate` +/// rejects such a header before the funnel sees it, so no mid-session +/// register ever reaches consensus; +/// - `DeleteSegments` is neither a partition nor a metadata op, so its +/// check sits before `is_partition`; +/// - the checksum and heartbeat pre-gates run BEFORE classification in the +/// funnel. +pub(in crate::dispatch) fn classify(header: &RoutedRequestHeader, bound: bool) -> RequestClass { + if header.operation == Operation::NonReplicated { + let nr_code = non_replicated_code(header); + if matches!( + nr_code, + LOGIN_USER_CODE | LOGIN_WITH_PERSONAL_ACCESS_TOKEN_CODE + ) { + return RequestClass::LegacyLogin; + } + if nr_code != PING_CODE && !bound { + return RequestClass::UnauthenticatedRead; + } + return RequestClass::NonReplicatedRead; } - validate_partitions_count(partitions_count) -} - -/// Static create-topic bounds shared by the TCP and HTTP ingresses. Runs -/// pre-consensus: a rejected request must not burn a replicated log entry, -/// and `prepare_request` errors evict the session instead of denying typed. -/// `ServerDefault` is exempt from the size floor (it resolves against server -/// config at admission, matching legacy); `Unlimited` passes numerically. -/// `segment_size_bytes` is the topic's RESOLVED segment size (explicit -/// option, else this node's default), so a per-topic segment above the -/// global default still floors the topic cap. -pub fn validate_topic_bounds( - partitions_count: u32, - max_topic_size: MaxTopicSize, - segment_size_bytes: u64, -) -> Result<(), IggyError> { - validate_partitions_count(partitions_count)?; - validate_topic_size_floor(max_topic_size, segment_size_bytes) -} - -/// A topic cap below one segment can never be enforced: the first segment -/// already exceeds it. Split out of [`validate_topic_bounds`] because update -/// admission checks the cap without a partitions count to check. -pub fn validate_topic_size_floor( - max_topic_size: MaxTopicSize, - segment_size_bytes: u64, -) -> Result<(), IggyError> { - if !matches!(max_topic_size, MaxTopicSize::ServerDefault) - && max_topic_size.as_bytes_u64() < segment_size_bytes - { - return Err(IggyError::InvalidTopicSize( - max_topic_size, - IggyByteSize::from(segment_size_bytes), - )); + if header.operation == Operation::Register && header.session == 0 && header.request == 0 { + return RequestClass::LoginRegister; } - Ok(()) -} - -/// Announce an accepted `max_topic_size` the server cannot enforce as written. -/// -/// [`validate_topic_size_floor`] admits any cap of one segment or more, but -/// retention runs PER PARTITION and floors each partition's share at one SEALED -/// segment, which reaches up to one maximum bus frame past `segment_size`. A cap -/// between the two is stored and echoed back verbatim while the server actually -/// keeps `(segment_size + max_message_size) * partitions_count`, so the only -/// moment an operator can be told is the one where they set it. -/// -/// Warns rather than rejects: which caps are accepted is client-visible wire -/// behavior, and tightening it would break topics that already exist. -pub fn warn_unenforceable_topic_size( - max_topic_size: MaxTopicSize, - segment_size_bytes: u64, - max_message_size_bytes: usize, - partitions_count: u32, -) { - let MaxTopicSize::Custom(configured) = max_topic_size else { - return; - }; - let max_message_size_bytes = u64::try_from(max_message_size_bytes).unwrap_or(u64::MAX); - let per_partition_floor = segment_size_bytes.saturating_add(max_message_size_bytes); - let topic_floor = per_partition_floor.saturating_mul(u64::from(partitions_count)); - if configured.as_bytes_u64() >= topic_floor { - return; + if header.operation == Operation::Logout { + return RequestClass::Logout; } - warn!( - max_topic_size = configured.as_bytes_u64(), - partitions_count, - segment_size = segment_size_bytes, - enforced_per_partition = per_partition_floor, - "{UNENFORCEABLE_TOPIC_SIZE_WARN}" - ); -} - -/// Announce the same unenforceable cap when partitions are ADDED to a topic. -/// -/// The cap is topic-wide but enforcement is per partition, so every added -/// partition shrinks the share: a cap that cleared the floor when the topic was -/// created can stop clearing it here. The request carries only the delta, so -/// the stored cap, segment size and current partition count come from metadata. -pub fn warn_unenforceable_topic_size_on_partition_add( - streams: &Streams, - stream_id: &WireIdentifier, - topic_id: &WireIdentifier, - max_message_size_bytes: usize, - added_partitions_count: u32, -) { - let Some(((stream_slab, topic_slab), _)) = streams.partition_count_context(stream_id, topic_id) - else { - return; - }; - let Some((_, max_topic_size, partitions_count, segment_size)) = - streams.topic_retention_config(stream_slab, topic_slab) - else { - return; - }; - warn_unenforceable_topic_size( - max_topic_size, - segment_size.map_or(iggy_common::DEFAULT_SEGMENT_SIZE, |segment_size| { - segment_size.as_bytes_u64() - }), - max_message_size_bytes, - u32::try_from(partitions_count) - .unwrap_or(u32::MAX) - .saturating_add(added_partitions_count), - ); -} - -/// Reject option keys outside the resource's catalog, pre-consensus. Unknown -/// keys are rejected rather than skipped: a silently ignored knob would hand -/// the client server defaults without it ever learning. Streams and users -/// have no catalog keys yet, so `known` is empty for both until one lands. -pub fn validate_option_keys(options: &WireOptions, known: &[&str]) -> Result<(), IggyError> { - for entry in options { - // Wire validation already enforced UTF-8 string keys. - let key = String::from_utf8_lossy(entry.key); - if !known.contains(&key.as_ref()) { - return Err(IggyError::UnsupportedOptionKey(key.into_owned())); - } + if !bound { + return RequestClass::UnboundReplicated; + } + if header.operation == Operation::DeleteSegments { + return RequestClass::DeleteSegments; } - Ok(()) + if header.operation.is_partition() { + return RequestClass::Partition; + } + RequestClass::ReplicatedMetadata } -/// Reject a request before it reaches consensus: warn, then send the typed -/// deny reply. A silent drop would wedge every later request on the -/// connection until the socket read timeout. `context` labels the rejection -/// site in both log lines. -#[allow(clippy::future_not_send)] -async fn send_pre_consensus_deny( - shard: &Rc>, - header: &RoutedRequestHeader, - transport_client_id: u128, - error: &IggyError, - context: &'static str, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - warn!( - transport_client_id, - error = %error, - operation = ?header.operation, - context, - "denying request pre-consensus" - ); - let commit = current_metadata_commit(shard); - let reply = build_deny_reply(header, transport_client_id, 0, commit, error.as_code()); - if let Err(send_error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %send_error, - context, - "failed to send pre-consensus deny reply" - ); - } +/// The command code a `NonReplicated` header carries in its first four +/// reserved bytes. +fn non_replicated_code(header: &RoutedRequestHeader) -> u32 { + u32::from_le_bytes(header.reserved[..4].try_into().unwrap()) } #[allow(clippy::future_not_send, clippy::too_many_lines)] @@ -673,69 +566,66 @@ async fn handle_client_request( return; } - ensure_transport_connection(shard, sessions, transport_client_id); - - // Any request is liveness proof, not just PING: an idle-but-active client - // (e.g. an admin issuing reads between long sleeps) must not be evicted by - // the heartbeat verifier. A genuinely dead connection sends nothing, so the - // intended stale-client eviction still fires. No-ops for an unbound client. - sessions.borrow_mut().record_heartbeat(transport_client_id); + // ONE `connections` walk for the whole prologue. Any request is liveness + // proof, not just PING: an idle-but-active client (e.g. an admin issuing + // reads between long sleeps) must not be evicted by the heartbeat + // verifier. A genuinely dead connection sends nothing, so the intended + // stale-client eviction still fires. + // Bound, not matched: a `match` on `borrow_mut()` would hold the guard + // across the arm that borrows again. + let mut touched = sessions.borrow_mut().touch_connection(transport_client_id); + if touched.is_none() { + // A transport's first frame: the peer address and transport kind live + // on the bus, so only this path pays that lookup. + ensure_transport_connection(shard, sessions, transport_client_id); + touched = sessions.borrow_mut().touch_connection(transport_client_id); + } + let ConnectionContext { + bound, + user_id, + address: client_address, + metadata_watermark, + } = touched.unwrap_or_default(); - let header = *request.header(); - if header.operation == Operation::NonReplicated { - // Auth bypass guard: `PING`, the liveness probe, is the only pre-auth - // code, on every roster shape. `GET_CLUSTER_METADATA` describes the - // private replica network and is not something an unauthenticated - // caller gets to read; a client that dialed a backup no longer needs - // it to find the leader, because the backup authenticates the login - // locally and forwards only the consensus proposal - // (`submit_register_local_or_forward`). Every other non-replicated - // code MUST go through Register first, which binds the acting user - // the per-op authz gates resolve. - let nr_code = u32::from_le_bytes(request.header().reserved[..4].try_into().unwrap()); - // Legacy (pre-register) login codes. The server authenticates only via - // the Register handshake (LOGIN_REGISTER / LOGIN_REGISTER_WITH_PAT, - // Operation::Register); the vsr SDK funnels both logins there and never - // emits these. Reject them uniformly with a typed MalformedLogin (the - // SDK maps it to InvalidFormat) before the session gate, so a legacy or - // foreign client fails fast instead of getting the generic - // Unauthenticated deny the pre-auth guard would send unbound, or the - // silent empty-ok Reply the bound non-replicated path would send. - if matches!( - nr_code, - LOGIN_USER_CODE | LOGIN_WITH_PERSONAL_ACCESS_TOKEN_CODE - ) { + // Borrowed, not copied: the 256-byte header is only worth a by-value + // snapshot where an arm rewrites it (`ReplicatedMetadata`) and still has + // to echo the client's original fields on a deny. + let header = request.header(); + match classify(header, bound.is_some()) { + RequestClass::LegacyLogin => { + // Legacy (pre-register) login codes. The server authenticates only via + // the Register handshake (LOGIN_REGISTER / LOGIN_REGISTER_WITH_PAT, + // Operation::Register); the vsr SDK funnels both logins there and never + // emits these. Reject them uniformly with a typed MalformedLogin (the + // SDK maps it to InvalidFormat) before the session gate, so a legacy or + // foreign client fails fast instead of getting the generic + // Unauthenticated deny the pre-auth guard would send unbound, or the + // silent empty-ok Reply the bound non-replicated path would send. + let nr_code = non_replicated_code(header); warn!( transport_client_id, code = nr_code, "rejecting legacy login code; server requires the register handshake" ); - send_login_eviction( + send_eviction( shard, transport_client_id, header.client, EvictionReason::MalformedLogin, + "legacy_login_rejection", ) .await; - return; } - let allowed_pre_auth = nr_code == PING_CODE; - if !allowed_pre_auth && sessions.borrow().get_session(transport_client_id).is_none() { - // Foreign SDKs still probe `GET_CLUSTER_METADATA` before login - // until they are fixed, so that rejection is routine traffic and - // logs at debug rather than warn. - if nr_code == GET_CLUSTER_METADATA_CODE { - debug!( - transport_client_id, - "denying pre-auth cluster-metadata read with Unauthenticated" - ); - } else { - warn!( - transport_client_id, - code = nr_code, - "denying pre-auth non-replicated read with Unauthenticated" - ); - } + RequestClass::UnauthenticatedRead => { + let nr_code = non_replicated_code(header); + // No per-code exemption: every in-tree SDK reads the roster only + // after login, so an unauthenticated roster read is a real event + // and not something to hide at debug. + warn!( + transport_client_id, + code = nr_code, + "denying pre-auth non-replicated read with Unauthenticated" + ); // A plain deny Reply, not an Eviction: there is no session to // evict, and an Eviction is session-terminal by wire contract, // so SDKs would tear down the very connection their login is @@ -748,360 +638,203 @@ async fn handle_client_request( IggyError::Unauthenticated.as_code(), ) .await; - return; - } - handle_non_replicated_request(shard, sessions, system_config, transport_client_id, request) - .await; - return; - } - - if header.operation == Operation::Register && header.session == 0 && header.request == 0 { - handle_login_register_request(shard, sessions, transport_client_id, request).await; - return; - } - - if header.operation == Operation::Logout { - handle_logout_request(shard, sessions, transport_client_id, request).await; - return; - } - - let bound = sessions.borrow().get_session(transport_client_id); - if bound.is_none() { - // Replicated request on an unbound transport. Without this short- - // circuit, the rewrite below overwrites `header.client` with - // `transport_client_id` and dispatches; the request_preflight then - // rejects with `NoSession`/`Fenced` and the failure disappears - // silently, wedging the SDK until the socket timeout. A typed - // `Eviction(NoSession)` is right here, unlike the pre-auth read - // guard above: a replicated request implies the client believes it - // has a session, and that session is gone, so it must register - // again. An empty status-0 Reply is not safe here, because - // SendMessages is the one replicated operation without a result - // section, and its decoder would read the empty body as a - // successful send. - warn!( - transport_client_id, - operation = ?header.operation, - "rejecting replicated request from unbound transport with Eviction(NoSession)" - ); - send_unauthenticated_eviction(shard, transport_client_id).await; - return; - } - - // DeleteSegments is neither a partition nor a metadata consensus op: the - // owning shard resolves the requested count to a concrete offset, then a - // `TruncatePartition` is replicated through metadata (Option A). Each - // replica's reconciler trims to the committed watermark. Handle it here, - // ahead of the partition/metadata routing below. - if header.operation == Operation::DeleteSegments { - handle_delete_segments_request(shard, transport_client_id, bound, &request).await; - return; - } - - if header.operation.is_partition() { - // `bound` is Some here (unbound transports returned above). - let (vsr_client_id, bound_session) = bound.unwrap_or((0, 0)); - // `get_session` discards the acting user id the partition gate needs; - // resolve it from the same bound connection. A bound transport always - // has one, but the gate fails closed on `None` rather than trust that. - let acting_user_id = sessions.borrow().get_user_id(transport_client_id); - dispatch_partition_request( - shard, - request, - vsr_client_id, - bound_session, - transport_client_id, - acting_user_id, - ) - .await; - return; - } - - let request = request.transmute_header(|header, new_header: &mut RoutedRequestHeader| { - *new_header = header; - // Metadata-plane ops route by operation: stamp the sentinel group. - new_header.group = server_common::sharding::METADATA_GROUP; - // `bound` is always Some here (unbound transports early-return above); - // this sets the consensus client id + session for the replicated op. - if let Some((bound_client_id, bound_session)) = bound { - new_header.client = bound_client_id; - new_header.session = bound_session; } - }); - let (request, raw_pat_token) = match maybe_rewrite_pat_request( - sessions, - transport_client_id, - max_tokens_per_user, - |user_id| { - shard - .plane - .metadata() - .mux_stm - .users() - .read(|users| users.pat_count_of(user_id)) - }, - request, - ) { - Ok(rewritten) => rewritten, - Err(error) => { - // Token cap reached, malformed body, or a lost session binding. - send_pre_consensus_deny( + RequestClass::NonReplicatedRead => { + // The auth-bypass guard is `classify`'s `UnauthenticatedRead` class: + // `PING`, the liveness probe, is the only pre-auth code, on every + // roster shape. `GET_CLUSTER_METADATA` describes the private replica + // network and is not something an unauthenticated caller gets to + // read; a client that dialed a backup no longer needs it to find the + // leader, because the backup authenticates the login locally and + // forwards only the consensus proposal + // (`submit_register_local_or_forward`). Every other non-replicated + // code MUST go through Register first, which binds the acting user + // the per-op authz gates resolve. + handle_non_replicated_request( shard, - &header, + sessions, + system_config, transport_client_id, - &error, - "personal-access-token", + request, + (user_id, client_address, metadata_watermark), ) .await; - return; } - }; - // Hash raw passwords and, for ChangePassword, verify the current password - // on the primary before replication; see `crate::users`. Replicas store the - // hash directly. A wrong current password is not denied here: it rides - // consensus and applies as a committed InvalidCredentials no-op, so the only - // Err returned is a malformed body. - let request = match maybe_rewrite_user_password_request(shard, request) { - Ok(rewritten) => rewritten, - Err(error) => { - // Malformed body: deny fast with InvalidCommand. - send_pre_consensus_deny(shard, &header, transport_client_id, &error, "user-password") - .await; - return; + RequestClass::LoginRegister => { + handle_login_register_request(shard, sessions, transport_client_id, request).await; } - }; - // Static bounds run pre-consensus so a rejected request burns no - // replicated log entry; HTTP covers the same bounds via - // `command.validate()`. A body that fails to decode denies typed too - // (`InvalidCommand`), instead of riding consensus just to fail there. - let bounds = match header.operation { - Operation::CreateTopic => CreateTopicRequest::decode_from(request_body(&request)) - .map_err(|_| IggyError::InvalidCommand) - .and_then(|create_topic| { - // `parse` doubles as the catalog gate: an unknown key or a - // malformed value denies typed here, pre-consensus. - let options = TopicCreateOptions::parse(&create_topic.options)?; - if let Some(segment_size) = options.segment_size { - validate_topic_segment_size( - segment_size.as_bytes_u64(), - iggy_common::MAX_TOPIC_SEGMENT_SIZE, - )?; - } - let segment_size = options.segment_size.map_or_else( - || iggy_common::DEFAULT_SEGMENT_SIZE, - |segment_size| segment_size.as_bytes_u64(), - ); - if options - .preallocate_segments - .unwrap_or(iggy_common::DEFAULT_PREALLOCATE_SEGMENTS) - { - validate_preallocated_topic_bytes(segment_size, create_topic.partitions_count)?; - } - let max_topic_size = options - .max_topic_size - .unwrap_or(MaxTopicSize::ServerDefault); - validate_topic_bounds(create_topic.partitions_count, max_topic_size, segment_size)?; - warn_unenforceable_topic_size( - max_topic_size, - segment_size, - shard.bus_max_message_size(), - create_topic.partitions_count, - ); - Ok(()) - }), - Operation::CreatePartitions => CreatePartitionsRequest::decode_from(request_body(&request)) - .map_err(|_| IggyError::InvalidCommand) - .and_then(|create_partitions| { - validate_partitions_change_count(create_partitions.partitions_count)?; - let metadata = shard.plane.metadata(); - warn_unenforceable_topic_size_on_partition_add( - metadata.mux_stm.streams(), - &create_partitions.stream_id, - &create_partitions.topic_id, - shard.bus_max_message_size(), - create_partitions.partitions_count, - ); - Ok(()) - }), - Operation::DeletePartitions => DeletePartitionsRequest::decode_from(request_body(&request)) - .map_err(|_| IggyError::InvalidCommand) - .and_then(|delete_partitions| { - validate_partitions_change_count(delete_partitions.partitions_count) - }), - // Only the updatable subset: the create-time knobs are pushed to - // partitions when the topic is built and nothing re-pushes them, so - // accepting one here would store a value no partition ever sees. - Operation::UpdateTopic => UpdateTopicRequest::decode_from(request_body(&request)) - .map_err(|_| IggyError::InvalidCommand) - .and_then(|update_topic| { - validate_option_keys(&update_topic.options, UPDATABLE_TOPIC_OPTION_KEYS)?; - let options = TopicCreateOptions::parse(&update_topic.options)?; - let Some(max_topic_size) = options.max_topic_size else { - return Ok(()); - }; - // An update can lower the cap below one segment just as a - // create can, and the stored map would then report a size the - // topic can never enforce. The floor is this topic's own - // segment size, since that key is create-only. - let metadata = shard.plane.metadata(); - let streams = metadata.mux_stm.streams(); - let segment_size = streams - .topic_segment_size(&update_topic.stream_id, &update_topic.topic_id) - .map_or_else( - || iggy_common::DEFAULT_SEGMENT_SIZE, - |segment_size| segment_size.as_bytes_u64(), - ); - validate_topic_size_floor(max_topic_size, segment_size)?; - let partitions_count = streams - .topic_partitions_count(&update_topic.stream_id, &update_topic.topic_id) - .unwrap_or(0); - warn_unenforceable_topic_size( - max_topic_size, - segment_size, - shard.bus_max_message_size(), - u32::try_from(partitions_count).unwrap_or(u32::MAX), - ); - Ok(()) - }), - Operation::UpdateStream => UpdateStreamRequest::decode_from(request_body(&request)) - .map_err(|_| IggyError::InvalidCommand) - .and_then(|update_stream| { - validate_option_keys(&update_stream.options, UPDATABLE_STREAM_OPTION_KEYS) - }), - Operation::UpdateUser => UpdateUserRequest::decode_from(request_body(&request)) - .map_err(|_| IggyError::InvalidCommand) - .and_then(|update_user| { - validate_option_keys(&update_user.options, UPDATABLE_USER_OPTION_KEYS) - }), - Operation::CreateStream => CreateStreamRequest::decode_from(request_body(&request)) - .map_err(|_| IggyError::InvalidCommand) - .and_then(|create_stream| validate_option_keys(&create_stream.options, &[])), - Operation::CreateUser => CreateUserRequest::decode_from(request_body(&request)) - .map_err(|_| IggyError::InvalidCommand) - .and_then(|create_user| validate_option_keys(&create_user.options, &[])), - _ => Ok(()), - }; - if let Err(error) = bounds { - send_pre_consensus_deny(shard, &header, transport_client_id, &error, "static-bounds").await; - return; - } - // Enrich consumer-group Join/Leave with the client's VSR id (+ topic - // partition count for Join) before replication; see `crate::consumer_group`. - let request = match maybe_rewrite_consumer_group_request(shard, request).await { - Ok(rewritten) => rewritten, - Err(error) => { + RequestClass::Logout => { + handle_logout_request(shard, sessions, transport_client_id, request).await; + } + RequestClass::UnboundReplicated => { + // Replicated request on an unbound transport. Without this short- + // circuit, the rewrite below overwrites `header.client` with + // `transport_client_id` and dispatches; the request_preflight then + // rejects with `NoSession`/`Fenced` and the failure disappears + // silently, wedging the SDK until the socket timeout. A typed + // `Eviction(NoSession)` is right here, unlike the plain deny of + // `UnauthenticatedRead`: a replicated request implies the client + // believes it has a session, and that session is gone, so it must + // register again. An empty status-0 Reply is not safe here, because + // SendMessages is the one replicated operation without a result + // section, and its decoder would read the empty body as a + // successful send. warn!( transport_client_id, - error = %error, operation = ?header.operation, - "dropping consumer-group request with invalid payload" + "rejecting replicated request from unbound transport with Eviction(NoSession)" ); - return; + // The eviction context is best-effort off the metadata consensus + // (peer shards have none; zeroes are cosmetic -- the SDK only + // reads the reason), and the evicted id is the transport id: no + // VSR session exists to name. + send_eviction( + shard, + transport_client_id, + transport_client_id, + EvictionReason::NoSession, + "unbound_replicated_request", + ) + .await; } - }; - let request_header = *request.header(); - // Replicated request: run consensus on the metadata owner (shard 0) and - // bring the committed reply back here. This shard owns the connection, - // so it writes the reply to the socket via the transport client id -- - // shard 0 can't route by the consensus client id (no home-shard bits). - match submit_client_request_on_owner(shard, request).await { - Some(reply) => { - // Recorded before the reply reaches the socket, so a read the client - // sends the instant it decodes this frame already sees the mark. - if let Some(commit) = committed_reply_commit(&reply) { - sessions - .borrow_mut() - .record_metadata_watermark(transport_client_id, commit); - } - // The raw PAT token never enters consensus (it is non-deterministic - // and secret), so the committed reply body is empty. Substitute the - // raw-token response here, on the minting client's home shard, using - // the confirmed commit position from the committed reply. - let reply = match build_raw_pat_reply(&request_header, reply, raw_pat_token) { - Ok(reply) => reply, + RequestClass::DeleteSegments => { + // DeleteSegments is neither a partition nor a metadata consensus op: the + // owning shard resolves the requested count to a concrete offset, then a + // `TruncatePartition` is replicated through metadata (Option A). Each + // replica's reconciler trims to the committed watermark. `classify` + // names it ahead of the partition and metadata classes. + handle_delete_segments_request(shard, transport_client_id, bound, &request).await; + } + RequestClass::Partition => { + // `bound` is Some here: `classify` sends unbound transports to + // `UnboundReplicated`. + let (vsr_client_id, bound_session) = bound.unwrap_or((0, 0)); + // The acting user comes from the prologue's lookup. A bound + // transport always has one, but the gate below fails closed on + // `None` rather than trust that. + dispatch_partition_request( + shard, + request, + vsr_client_id, + bound_session, + transport_client_id, + user_id, + ) + .await; + } + RequestClass::ReplicatedMetadata => { + // The one arm that needs the by-value copy: the rewrite below + // stamps the consensus client / session / group over the header, + // and a pre-consensus deny still has to echo what the client sent. + let header = *header; + let request = + request.transmute_header(|header, new_header: &mut RoutedRequestHeader| { + *new_header = header; + // Metadata-plane ops route by operation: stamp the sentinel group. + new_header.group = server_common::sharding::METADATA_GROUP; + // `bound` is always Some here (`classify` sends unbound transports to + // `UnboundReplicated`); this sets the consensus client id + session + // for the replicated op. + if let Some((bound_client_id, bound_session)) = bound { + new_header.client = bound_client_id; + new_header.session = bound_session; + } + }); + let (request, raw_pat_token) = match tcp_chain( + shard, + sessions, + transport_client_id, + max_tokens_per_user, + request, + ) { + Ok(rewritten) => rewritten, + Err(RewriteDeny { stage, error }) => { + send_pre_consensus_deny(shard, transport_client_id, &header, &error, stage) + .await; + return; + } + }; + // Enrich consumer-group Join/Leave with the client's VSR id (+ topic + // partition count for Join) before replication; see `crate::consumer_group`. + let request = match maybe_rewrite_consumer_group_request(shard, request).await { + Ok(rewritten) => rewritten, Err(error) => { - warn!( + // Both of the rewrite's own failures are `InvalidCommand` + // decode errors, so a replay cannot help: deny typed + // instead of leaving the lockstep connection to its read + // timeout. (Its third error path needs a body past + // `u32::MAX` against a 64 MiB message cap, so no client + // frame reaches it; the deny is correct there too.) + send_pre_consensus_deny( + shard, transport_client_id, - error = %error, - "failed to build raw PAT reply" - ); + &header, + &error, + RewriteStage::ConsumerGroup, + ) + .await; return; } }; - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - operation = ?header.operation, - "failed to deliver committed reply to client" - ); + let request_header = *request.header(); + // Replicated request: run consensus on the metadata owner (shard 0) and + // bring the committed reply back here. This shard owns the connection, + // so it writes the reply to the socket via the transport client id -- + // shard 0 can't route by the consensus client id (no home-shard bits). + match submit_client_request_on_owner(shard, request).await { + Some(reply) => { + // Recorded before the reply reaches the socket, so a read the client + // sends the instant it decodes this frame already sees the mark. + if let Some(commit) = committed_reply_commit(&reply) { + sessions + .borrow_mut() + .record_metadata_watermark(transport_client_id, commit); + } + // The raw PAT token never enters consensus (it is non-deterministic + // and secret), so the committed reply body is empty. Substitute the + // raw-token response here, on the minting client's home shard, using + // the confirmed commit position from the committed reply. + let reply = match build_raw_pat_reply(&request_header, reply, raw_pat_token) { + Ok(reply) => reply, + Err(error) => { + warn!( + transport_client_id, + error = %error, + "failed to build raw PAT reply" + ); + // The op COMMITTED; only the reply could not be + // rendered. A typed deny is still the right frame: + // silence wedges the lockstep connection on a + // request that succeeded server-side. The raw + // token is unrecoverable either way -- it lives in + // that one reply and a retry dedups to the empty + // committed body -- so the caller must delete the + // token and mint a new one. + send_deny_reply(shard, transport_client_id, &header, error.as_code()) + .await; + return; + } + }; + send_host_frame( + &shard.bus, + transport_client_id, + reply.into_frozen(), + FrameChannel::Reply, + "committed_reply", + ) + .await; + } + None => { + // Transient submit failure (not primary / not caught up / dedup + // absorbed). Stay silent; the SDK read-timeout replays. + warn!( + transport_client_id, + operation = ?header.operation, + "replicated request not committed (transient); client will replay" + ); + } } } - None => { - // Transient submit failure (not primary / not caught up / dedup - // absorbed). Stay silent; the SDK read-timeout replays. - warn!( - transport_client_id, - operation = ?header.operation, - "replicated request not committed (transient); client will replay" - ); - } - } -} - -/// Send a non-replicated reply body to a client, stamping the current -/// metadata commit. Shared by the `get_me` / `get_clients` / `get_client` -/// arms. -#[allow(clippy::future_not_send)] -async fn send_non_replicated_bytes( - shard: &Rc>, - request: &Message, - transport_client_id: u128, - bytes: Bytes, - label: &'static str, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let commit = current_metadata_commit(shard); - let reply = NonReplicatedResponse::Bytes(bytes).into_reply( - request.header(), - request.header().client, - request.header().session, - commit, - ); - send_reply_frame( - shard, - transport_client_id, - reply.into_generic().into_frozen(), - label, - ) - .await; -} - -/// Hand a built reply frame to the bus for `transport_client_id`. -#[allow(clippy::future_not_send)] -async fn send_reply_frame( - shard: &Rc>, - transport_client_id: u128, - frame: impl Into, - label: &'static str, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - if let Err(error) = shard.bus.send_to_client(transport_client_id, frame).await { - warn!(transport_client_id, label, error = %error, "failed to send non-replicated reply"); } } @@ -1146,7 +879,13 @@ mod tests { use crate::cluster_meta::ClusterRoster; use crate::dispatch::test_support::{FIRST_BOOT, SpyBus, TestMux, TestShard, test_shard}; use iggy_binary_protocol::Command; + use iggy_binary_protocol::codes::{ + GET_CLUSTER_METADATA_CODE, GET_CONSUMER_OFFSET_CODE, POLL_MESSAGES_CODE, + }; + use iggy_binary_protocol::{EvictionHeader, ReplyHeader}; + use journal::prepare_journal::PrepareJournal; use metadata::IggyMetadata; + use metadata::impls::metadata::IggySnapshot; use partitions::{IggyPartitions, PartitionPathLayout, PartitionsConfig}; use server_common::MESSAGE_ALIGN; use server_common::sharding::ShardId; @@ -1299,7 +1038,6 @@ mod tests { #[compio::test] async fn pre_auth_cluster_metadata_denied_on_every_roster() { use configs::cluster::{ClusterNodeConfig, TransportPorts}; - use iggy_binary_protocol::codes::GET_CLUSTER_METADATA_CODE; use iggy_binary_protocol::{GenericHeader, ReplyHeader}; const TRANSPORT: u128 = 91; @@ -1400,89 +1138,467 @@ mod tests { } } + /// Client-wire frame for the funnel tests, shaped like the SDK sends it + /// (the funnel promotes it to the routed shape itself). + fn wire_request( + operation: Operation, + client: u128, + session: u64, + request: u64, + body: &[u8], + ) -> Message { + let header_size = size_of::(); + let total = header_size + body.len(); + let mut message = Message::::new(total); + { + let slice = message.as_mut_slice(); + slice[header_size..total].copy_from_slice(body); + let header = + bytemuck::checked::from_bytes_mut::(&mut slice[..header_size]); + *header = RequestHeader { + command: Command::Request, + operation, + size: u32::try_from(total).expect("test request fits u32"), + client, + session, + request, + ..Default::default() + }; + } + message + } + + fn non_replicated_request(client: u128, nr_code: u32) -> Message { + let mut message = wire_request(Operation::NonReplicated, client, 0, 0, &[]); + { + let header_size = size_of::(); + let header = bytemuck::checked::from_bytes_mut::( + &mut message.as_mut_slice()[..header_size], + ); + header.reserved[..4].copy_from_slice(&nr_code.to_le_bytes()); + } + message.into_generic() + } + + const fn frame_command(frame: &[u8]) -> u8 { + frame[std::mem::offset_of!(GenericHeader, command)] + } + + const fn eviction_reason_byte(frame: &[u8]) -> u8 { + frame[std::mem::offset_of!(EvictionHeader, reason)] + } + + fn reply_status(frame: &[u8]) -> u32 { + const STATUS_OFFSET: usize = std::mem::offset_of!(ReplyHeader, status); + u32::from_le_bytes(frame[STATUS_OFFSET..STATUS_OFFSET + 4].try_into().unwrap()) + } + + /// The classifier is the funnel's probe chain as data: every row pins + /// one routing decision, and the labeled rows pin the probe ORDERINGS + /// the funnel relies on. + #[allow(clippy::too_many_lines)] #[test] - fn create_topic_bounds_deny_pre_consensus() { - let segment_size = iggy_common::DEFAULT_SEGMENT_SIZE; - assert!(segment_size > 0, "default segment size must be nonzero"); + fn classify_pins_probe_order() { + fn routed(operation: Operation, session: u64, request: u64) -> RoutedRequestHeader { + RoutedRequestHeader { + command: Command::Request, + operation, + client: 7, + session, + request, + ..Default::default() + } + } + fn nr(nr_code: u32) -> RoutedRequestHeader { + let mut header = routed(Operation::NonReplicated, 0, 0); + header.reserved[..4].copy_from_slice(&nr_code.to_le_bytes()); + header + } + let table = [ + ( + "legacy login beats the session gate (unbound)", + nr(LOGIN_USER_CODE), + false, + RequestClass::LegacyLogin, + ), + ( + "legacy login beats the bound reads route", + nr(LOGIN_USER_CODE), + true, + RequestClass::LegacyLogin, + ), + ( + "legacy PAT login beats the session gate (unbound)", + nr(LOGIN_WITH_PERSONAL_ACCESS_TOKEN_CODE), + false, + RequestClass::LegacyLogin, + ), + ( + "legacy PAT login beats the bound reads route", + nr(LOGIN_WITH_PERSONAL_ACCESS_TOKEN_CODE), + true, + RequestClass::LegacyLogin, + ), + ( + "ping is the only pre-auth code", + nr(PING_CODE), + false, + RequestClass::NonReplicatedRead, + ), + ( + "cluster metadata is not pre-auth", + nr(GET_CLUSTER_METADATA_CODE), + false, + RequestClass::UnauthenticatedRead, + ), + ( + "poll-messages is a non-replicated code, not a partition op", + nr(POLL_MESSAGES_CODE), + true, + RequestClass::NonReplicatedRead, + ), + ( + "consumer-offset read is a non-replicated code, not a partition op", + nr(GET_CONSUMER_OFFSET_CODE), + true, + RequestClass::NonReplicatedRead, + ), + ( + "the register handshake", + routed(Operation::Register, 0, 0), + false, + RequestClass::LoginRegister, + ), + ( + "register with a session falls to the default", + routed(Operation::Register, 1, 0), + true, + RequestClass::ReplicatedMetadata, + ), + ( + "register with a request number falls to the default", + routed(Operation::Register, 0, 1), + true, + RequestClass::ReplicatedMetadata, + ), + ( + "logout, bound", + routed(Operation::Logout, 1, 1), + true, + RequestClass::Logout, + ), + ( + "logout beats the session gate", + routed(Operation::Logout, 1, 1), + false, + RequestClass::Logout, + ), + ( + "unbound metadata op", + routed(Operation::CreateStream, 1, 1), + false, + RequestClass::UnboundReplicated, + ), + ( + "the session gate beats the delete-segments probe", + routed(Operation::DeleteSegments, 1, 1), + false, + RequestClass::UnboundReplicated, + ), + ( + "the session gate beats the partition probe", + routed(Operation::SendMessages, 1, 1), + false, + RequestClass::UnboundReplicated, + ), + ( + "delete-segments is neither partition nor metadata", + routed(Operation::DeleteSegments, 1, 1), + true, + RequestClass::DeleteSegments, + ), + ( + "send-messages is partition-plane", + routed(Operation::SendMessages, 1, 1), + true, + RequestClass::Partition, + ), + ( + "store-consumer-offset is partition-plane", + routed(Operation::StoreConsumerOffset, 1, 1), + true, + RequestClass::Partition, + ), + ( + "create-topic is replicated metadata", + routed(Operation::CreateTopic, 1, 1), + true, + RequestClass::ReplicatedMetadata, + ), + ]; + for (label, header, bound, expected) in table { + assert_eq!(classify(&header, bound), expected, "{label}"); + } + } - assert!( - validate_topic_bounds( - MAX_PARTITIONS_PER_REQUEST, - MaxTopicSize::ServerDefault, - segment_size + /// A legacy login code must get the typed `MalformedLogin` eviction + /// BEFORE the session gate runs: an unbound sender must not fall into + /// the generic Unauthenticated deny the pre-auth guard sends for other + /// codes. + #[compio::test] + async fn legacy_login_codes_evicted_before_session_gate() { + const TRANSPORT: u128 = 94; + let bus = SpyBus::default(); + let shard = Rc::new(test_shard(&bus, 0, 1, FIRST_BOOT)); + let sessions = Rc::new(RefCell::new(SessionManager::new())); + let system_config = Arc::new(ServerSystemConfig::default()); + + for code in [LOGIN_USER_CODE, LOGIN_WITH_PERSONAL_ACCESS_TOKEN_CODE] { + handle_client_request( + &shard, + &sessions, + &system_config, + 1, + TRANSPORT, + non_replicated_request(TRANSPORT, code), ) - .is_ok(), - "the partition cap itself is admissible" + .await; + let replies = bus.client_replies.borrow(); + assert_eq!(replies.len(), 1, "code {code} must produce one frame"); + let (client, frame) = &replies[0]; + assert_eq!(*client, TRANSPORT); + assert_eq!( + frame_command(frame), + Command::Eviction as u8, + "legacy code {code} must be evicted, not denied" + ); + assert_eq!( + eviction_reason_byte(frame), + EvictionReason::MalformedLogin as u8, + "legacy code {code} must carry MalformedLogin" + ); + drop(replies); + bus.client_replies.borrow_mut().clear(); + } + } + + /// A replicated op from an unbound transport must get the typed + /// `Eviction(NoSession)`: the client believes it has a session and that + /// session is gone, so a silent drop or an empty Reply would wedge or + /// mislead it. + #[compio::test] + async fn unbound_replicated_request_gets_no_session_eviction() { + const TRANSPORT: u128 = 95; + let bus = SpyBus::default(); + let shard = Rc::new(test_shard(&bus, 0, 1, FIRST_BOOT)); + let sessions = Rc::new(RefCell::new(SessionManager::new())); + let system_config = Arc::new(ServerSystemConfig::default()); + + handle_client_request( + &shard, + &sessions, + &system_config, + 1, + TRANSPORT, + wire_request(Operation::CreateStream, TRANSPORT, 1, 1, &[]).into_generic(), + ) + .await; + let replies = bus.client_replies.borrow(); + assert_eq!(replies.len(), 1, "unbound replicated op must be answered"); + let (client, frame) = &replies[0]; + assert_eq!(*client, TRANSPORT); + assert_eq!( + frame_command(frame), + Command::Eviction as u8, + "unbound replicated op must be evicted, not denied or dropped" ); - assert!( - matches!( - validate_topic_bounds( - MAX_PARTITIONS_PER_REQUEST + 1, - MaxTopicSize::ServerDefault, - segment_size - ), - Err(IggyError::TooManyPartitions) - ), - "one past the partition cap must deny" + assert_eq!( + eviction_reason_byte(frame), + EvictionReason::NoSession as u8, + "the eviction must carry NoSession" ); - // ServerDefault is numerically 0 yet exempt from the segment-size - // floor: it resolves against server config, matching legacy. - assert!(validate_topic_bounds(1, MaxTopicSize::ServerDefault, segment_size).is_ok()); - assert!(validate_topic_bounds(1, MaxTopicSize::Unlimited, segment_size).is_ok()); - let below_floor = MaxTopicSize::Custom((segment_size - 1).into()); - assert!( - matches!( - validate_topic_bounds(1, below_floor, segment_size), - Err(IggyError::InvalidTopicSize(size, floor)) - if size == below_floor && floor == IggyByteSize::from(segment_size) - ), - "custom size below the segment size must deny with the bounds" + } + + /// PING is the one pre-auth code: an unbound transport's ping must get a + /// normal status-0 Reply, not an eviction and not a deny. + #[compio::test] + async fn pre_auth_ping_allowed() { + const TRANSPORT: u128 = 96; + let bus = SpyBus::default(); + let shard = Rc::new(test_shard(&bus, 0, 1, FIRST_BOOT)); + let sessions = Rc::new(RefCell::new(SessionManager::new())); + let system_config = Arc::new(ServerSystemConfig::default()); + + handle_client_request( + &shard, + &sessions, + &system_config, + 1, + TRANSPORT, + non_replicated_request(TRANSPORT, PING_CODE), + ) + .await; + let replies = bus.client_replies.borrow(); + assert_eq!(replies.len(), 1, "pre-auth ping must be answered"); + let (client, frame) = &replies[0]; + assert_eq!(*client, TRANSPORT); + assert_eq!( + frame_command(frame), + Command::Reply as u8, + "pre-auth ping must get a Reply, not an eviction" ); - let at_floor = MaxTopicSize::Custom(IggyByteSize::from(segment_size)); - assert!( - validate_topic_bounds(1, at_floor, segment_size).is_ok(), - "a topic exactly one segment large is admissible" + assert_eq!(reply_status(frame), 0, "pre-auth ping must succeed"); + } + + /// The request-checksum gate runs before every probe: a replicated op + /// from an UNBOUND transport with a bad stamp must get the checksum deny + /// Reply, not the `Eviction(NoSession)` the session gate would send. + #[compio::test] + async fn checksum_mismatch_denies_before_everything() { + const TRANSPORT: u128 = 97; + const BODY: &[u8] = b"stream-body"; + let bus = SpyBus::default(); + let shard = Rc::new(test_shard(&bus, 0, 1, FIRST_BOOT)); + let sessions = Rc::new(RefCell::new(SessionManager::new())); + let system_config = Arc::new(ServerSystemConfig::default()); + + let mut message = wire_request(Operation::CreateStream, TRANSPORT, 1, 1, BODY); + { + let header_size = size_of::(); + let header = bytemuck::checked::from_bytes_mut::( + &mut message.as_mut_slice()[..header_size], + ); + // Nonzero (zero means unstamped and skips the check) and never + // the body's real checksum. + header.request_checksum = u128::from(iggy_common::calculate_checksum(BODY)) + 1; + } + handle_client_request( + &shard, + &sessions, + &system_config, + 1, + TRANSPORT, + message.into_generic(), + ) + .await; + let replies = bus.client_replies.borrow(); + assert_eq!(replies.len(), 1, "bad stamp must be answered"); + let (client, frame) = &replies[0]; + assert_eq!(*client, TRANSPORT); + assert_eq!( + frame_command(frame), + Command::Reply as u8, + "a bad stamp must get the checksum deny, not the NoSession eviction" ); + assert_eq!( + reply_status(frame), + IggyError::InvalidFormat.as_code(), + "the deny status must carry the checksum error" + ); + } + + /// The late-bound self-reference as the shard build creates it: unset + /// until the shard exists. + fn unset_shard_handle() -> ShellShardHandle { + Rc::new(RefCell::new(None)) } + /// Yield so the drain task the enqueue spawned on the bus can run. + async fn run_spawned_tasks() { + compio::time::sleep(std::time::Duration::from_millis(1)).await; + } + + /// One handler per shard means one connection-lost hook on its bus: a + /// second install would overwrite the first and orphan the sessions + /// the first one bound. #[test] - fn partitions_count_cap_denies_pre_consensus() { - assert!( - validate_partitions_count(MAX_PARTITIONS_PER_REQUEST).is_ok(), - "the cap itself is admissible" + fn deferred_handler_installs_one_connection_lost_hook() { + let bus = SpyBus::default(); + let _handler = make_deferred_client_request_handler( + &bus, + &unset_shard_handle(), + &Rc::new(RefCell::new(SessionManager::new())), + Arc::new(ServerSystemConfig::default()), + 1, + ); + assert_eq!( + bus.connection_lost_hooks.get(), + 1, + "one factory call must install exactly one connection-lost hook" + ); + } + + /// The handler is built before the shard it serves. A frame that + /// arrives while the self-reference is still unset stays queued, and + /// the enqueue must release the client's active slot: otherwise every + /// later frame for that client finds the slot taken and nothing is + /// ever drained, the stranded frame included. + #[compio::test] + async fn deferred_handler_drains_after_the_shard_handle_is_set() { + const TRANSPORT: u128 = 98; + let bus = SpyBus::default(); + let shard_handle = unset_shard_handle(); + let handler = make_deferred_client_request_handler( + &bus, + &shard_handle, + &Rc::new(RefCell::new(SessionManager::new())), + Arc::new(ServerSystemConfig::default()), + 1, ); + + handler(TRANSPORT, non_replicated_request(TRANSPORT, PING_CODE)); + run_spawned_tasks().await; assert!( - matches!( - validate_partitions_count(MAX_PARTITIONS_PER_REQUEST + 1), - Err(IggyError::TooManyPartitions) - ), - "one past the cap must deny" + bus.client_replies.borrow().is_empty(), + "nothing can be served before the shard exists" + ); + + let shard = Rc::new(test_shard(&bus, 0, 1, FIRST_BOOT)); + *shard_handle.borrow_mut() = Some(Rc::downgrade(&shard)); + handler(TRANSPORT, non_replicated_request(TRANSPORT, PING_CODE)); + for _ in 0..500 { + if bus.client_replies.borrow().len() == 2 { + break; + } + run_spawned_tasks().await; + } + + let replies = bus.client_replies.borrow(); + assert_eq!( + replies.len(), + 2, + "the stranded ping and the new one must both be served once the shard exists" ); - // Zero passes the shared cap because a zero-partition TOPIC is legal - // (legacy `create_topic` admits `0..=MAX`). - assert!(validate_partitions_count(0).is_ok()); + for (client, frame) in replies.iter() { + assert_eq!(*client, TRANSPORT); + assert_eq!(frame_command(frame), Command::Reply as u8); + assert_eq!(reply_status(frame), 0, "a served ping succeeds"); + } } + /// The per-client queue entry goes with its last frame. The + /// connection-lost hook cannot be its only owner: the dispatch task still + /// delivers frames it had buffered when the socket closed, which re-create + /// the entry after the hook ran, and `client_id` is monotonic - so an + /// entry nothing frees again lives until process exit. #[test] - fn zero_partitions_change_denies_pre_consensus() { - // Adding or removing zero partitions is a no-op that would still burn - // a replicated log entry and force a rebalance. Legacy rejects it with - // `TooManyPartitions` in both handlers, so the code matches. + fn draining_a_client_queue_frees_its_entry() { + const CLIENT: u128 = 7; + let queues: ClientRequestQueues = Rc::new(RefCell::new(AHashMap::new())); + queues + .borrow_mut() + .entry(CLIENT) + .or_default() + .push_back(non_replicated_request(CLIENT, PING_CODE)); + + assert!(pop_next_client_request(&queues, CLIENT).is_some()); assert!( - matches!( - validate_partitions_change_count(0), - Err(IggyError::TooManyPartitions) - ), - "adding or removing zero partitions must deny" + queues.borrow().is_empty(), + "the entry must be freed as soon as it drains empty" ); - assert!(validate_partitions_change_count(1).is_ok()); - assert!(validate_partitions_change_count(MAX_PARTITIONS_PER_REQUEST).is_ok()); assert!( - matches!( - validate_partitions_change_count(MAX_PARTITIONS_PER_REQUEST + 1), - Err(IggyError::TooManyPartitions) - ), - "the cap still applies" + pop_next_client_request(&queues, CLIENT).is_none(), + "a drained client has nothing left to pop" ); } } diff --git a/core/server/src/dispatch/partition.rs b/core/server/src/dispatch/partition.rs index b2e0789bfa..ac20090abc 100644 --- a/core/server/src/dispatch/partition.rs +++ b/core/server/src/dispatch/partition.rs @@ -28,23 +28,25 @@ //! replies to committed writes itself, straight from the owning shard, while //! everything the host builds here -- read bodies, denies, empty-poll shapes //! -- goes out on the connection's home shard. The reply path is therefore -//! split by plane, not unified, and the deny helpers in `authz` are the +//! split by plane, not unified, and the deny helpers in `failure` are the //! third leg (typed status replies for requests that never reach a plane). use crate::consumer_group::maybe_rewrite_consumer_offset_request; -use crate::dispatch::authz::{ - authorize_partition_op, authorize_partition_read, send_deny_reply, send_non_replicated_deny, +use crate::dispatch::authz::{authorize_partition_op, authorize_partition_read}; +use crate::dispatch::failure::{ + FrameChannel, send_deny_reply, send_host_frame, send_non_replicated_bytes, + send_non_replicated_deny, send_result_rejection, }; use crate::dispatch::submit::submit_client_request_on_owner; -use crate::dispatch::{send_non_replicated_bytes, send_reply_frame, upgrade_shard_handle}; +use crate::dispatch::upgrade_shard_handle; use crate::responses::{ - build_consumer_offset_body, build_empty_reply, build_polled_messages_reply, - current_metadata_commit, resolve_partition_namespace, resolve_partition_request_namespace, + build_consumer_offset_body, build_polled_messages_reply, current_metadata_commit, + resolve_partition_namespace, resolve_partition_request_namespace, }; use crate::shell::{ShellBus, ShellShard, ShellShardHandle}; use crate::wire::{request_body, usize_to_u32}; use bytes::Bytes; -use consensus::{Consensus, MetadataHandle, PartitionsHandle, build_result_rejection_reply}; +use consensus::{Consensus, MetadataHandle, PartitionsHandle}; use iggy_binary_protocol::PrepareHeader; use iggy_binary_protocol::primitives::consumer::WireConsumer; use iggy_binary_protocol::primitives::polling_strategy::WirePollingStrategy; @@ -57,10 +59,10 @@ use iggy_binary_protocol::{ AckLevel, Command, KIND_CONSUMER_GROUP, Operation, RoutedRequestHeader, WireDecode, WireEncode, WireIdentifier, }; -use iggy_common::{IggyError, PollingStrategy}; +use iggy_common::{IggyError, PollingStrategy, RESYNC_REQUIRED_PARTITION_SENTINEL}; use journal::superblock::SuperblockStore; use journal::{Journal, JournalHandle}; -use message_bus::AUTO_COMMIT_CLIENT_ID; +use message_bus::{AUTO_COMMIT_CLIENT_ID, BusMessage}; use metadata::impls::metadata::{ StreamsFrontend, build_truncate_partition_client_message, build_truncate_partition_client_message_with_identifiers, @@ -469,14 +471,19 @@ pub async fn dispatch_partition_request( // metadata access to resolve it. let request = match maybe_rewrite_consumer_offset_request(shard, request) { Ok(rewritten) => rewritten, + // Not reachable through the wire path: the same body already decoded in + // `resolve_partition_request_namespace` above, and re-encoding it can + // only fail past `u32::MAX` bytes against a 64 MiB request cap. Denying + // typed keeps a future re-encode failure from acking work the partition + // plane never saw. Err(error) => { warn!( transport_client_id, error = %error, operation = ?header.operation, - "failed to rewrite consumer-offset request; replying empty" + "failed to rewrite consumer-offset request; replying denied" ); - send_empty_partition_reply(shard, transport_client_id, &header).await; + send_deny_reply(shard, transport_client_id, &header, error.as_code()).await; return; } }; @@ -581,8 +588,12 @@ async fn relay_partition_reply( /// the owning shard ([`shard::IggyShard::partition_read`]), and re-encode /// the stored batches into the legacy wire `PolledMessages` body. /// -/// Failures reply with an empty body so the SDK fails fast on decode -/// instead of hanging until its read timeout. +/// A partition that cannot answer yet replies with the 16-byte empty poll, +/// which the SDK reads as 0 messages and retries; a consumer-group poll the +/// coordinator fenced replies with the same shape carrying the re-sync +/// sentinel. Every permanent client error (undecodable body, authz, and every +/// rejection the resolve raises) denies with a nonzero status instead, so none +/// of them can be mistaken for an empty partition. #[allow(clippy::future_not_send)] pub(in crate::dispatch) async fn handle_poll_messages( shard: &Rc>, @@ -597,21 +608,23 @@ pub(in crate::dispatch) async fn handle_poll_messages( SB: SuperblockStore + 'static, { let Ok(wire) = PollMessagesRequest::decode_from(request_body(request)) else { - // Undecodable poll: keep the fail-fast empty-poll shape. - send_non_replicated_bytes( + // A permanent client error, so it must not borrow the empty-poll + // shape: that body decodes as a successful 0-message poll, and a + // consumer looping on it never learns why. The server already sends + // nonzero-status poll replies (authz, unresolved target). + send_non_replicated_deny( shard, request, transport_client_id, - empty_polled_messages_body(0), - "poll_messages", + IggyError::InvalidCommand.as_code(), ) .await; return; }; // Gate on (stream, topic) before touching the partition plane. A resolution - // miss falls through to the resolve path below (empty-poll / not-found); a - // denial replies status!=0 with an empty body, distinct from the empty-poll - // "0 messages" shape. + // miss denies typed on the resolve path below; a denial here replies + // status!=0 with an empty body, distinct from the empty-poll "0 messages" + // shape. if let Some(status) = authorize_partition_read( shard, &wire.stream_id, @@ -624,89 +637,132 @@ pub(in crate::dispatch) async fn handle_poll_messages( send_non_replicated_deny(shard, request, transport_client_id, status).await; return; } - let body = match resolve_poll_request(shard, &wire, request.header().client) { - Ok((namespace, partition_id, consumer, args)) => { - match shard - .partition_read(namespace, PartitionRead::Poll { consumer, args }) - .await - { - Some(PartitionReadReply::Poll { - fragments, - current_offset, - }) => match build_polled_messages_reply( - request.header(), - current_metadata_commit(shard), - partition_id, - current_offset, - fragments, - shard.plane.partitions().config().encryptor.as_deref(), - ) { - Ok(reply) => { - send_reply_frame(shard, transport_client_id, reply, "poll_messages").await; - return; - } - Err(error) => { - warn!( - transport_client_id, - error = %error, - "failed to re-encode polled batches; replying empty poll" - ); - empty_polled_messages_body(partition_id) - } - }, - other => { - warn!( + let (body, channel) = match resolve_poll_request(shard, &wire, request.header().client) { + Ok(resolved) => { + match read_polled_messages(shard, transport_client_id, request, resolved).await { + Ok(reply) => { + send_host_frame( + &shard.bus, transport_client_id, - namespace = namespace.inner(), - reply_was_none = other.is_none(), - "partition read failed; replying empty poll" - ); - empty_polled_messages_body(partition_id) + reply, + FrameChannel::Reply, + "poll_messages", + ) + .await; + return; } + Err(fallback) => fallback, } } + // A generation fence: the client's cached assignment went stale after a + // rebalance. The empty poll carries the re-sync sentinel so the SDK + // re-syncs and retries instead of reading end-of-partition. + Err(error @ IggyError::ConsumerGroupPartitionNotOwned(..)) => { + warn!( + transport_client_id, + error = %error, + "poll_messages fenced; replying re-sync sentinel" + ); + empty_poll_fallback(RESYNC_REQUIRED_PARTITION_SENTINEL) + } + // Everything else the resolve rejects is a permanent client error: an + // unresolved stream, topic, or partition, a polling strategy or + // consumer kind outside the wire enum, or a group poll that omitted + // its partition id. All must surface typed -- the empty poll is a + // successful 0-message read, and a consumer looping on it never learns + // why. Err(error) => { - // A stream, topic, or partition id that does not resolve is a - // client addressing error and must surface as a typed rejection, - // not an empty poll a consumer would read as end-of-partition. - if matches!( - error, - IggyError::PartitionNotFound(..) - | IggyError::StreamIdNotFound(_) - | IggyError::TopicIdNotFound(..) - ) { - warn!( - transport_client_id, - error = %error, - "poll_messages rejected: target not found" - ); - send_non_replicated_deny(shard, request, transport_client_id, error.as_code()) - .await; - return; - } - // A zero-byte body would panic the SDK's `PolledMessages` - // decoder; reply the 16-byte empty-poll shape instead. A generation - // fence (the client's cached assignment is stale after a rebalance) - // carries the re-sync sentinel so the SDK re-syncs and retries - // rather than treating the empty poll as end-of-partition. warn!( transport_client_id, error = %error, - "poll_messages request rejected; replying empty poll" + "poll_messages rejected; replying denied" ); - let partition_id = if matches!(error, IggyError::ConsumerGroupPartitionNotOwned(..)) { - iggy_common::RESYNC_REQUIRED_PARTITION_SENTINEL - } else { - 0 - }; - empty_polled_messages_body(partition_id) + send_non_replicated_deny(shard, request, transport_client_id, error.as_code()).await; + return; } }; - send_non_replicated_bytes(shard, request, transport_client_id, body, "poll_messages").await; + send_non_replicated_bytes( + shard, + request, + transport_client_id, + body, + channel, + "poll_messages", + ) + .await; } -/// Serve `get_consumer_offset`. An empty body decodes as `None` on the SDK -/// side (no offset stored / partition unknown). +/// Run the resolved poll on the owning shard and re-encode the stored +/// batches into the wire `PolledMessages` reply. A failed read or re-encode +/// hands back the fail-fast empty poll for the partition instead. +#[allow(clippy::future_not_send)] +async fn read_polled_messages( + shard: &Rc>, + transport_client_id: u128, + request: &Message, + (namespace, partition_id, consumer, args): DecodedPollRequest, +) -> Result +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + match shard + .partition_read(namespace, PartitionRead::Poll { consumer, args }) + .await + { + Some(PartitionReadReply::Poll { + fragments, + current_offset, + }) => build_polled_messages_reply( + request.header(), + current_metadata_commit(shard), + partition_id, + current_offset, + fragments, + shard.plane.partitions().config().encryptor.as_deref(), + ) + .map_err(|error| { + warn!( + transport_client_id, + error = %error, + "failed to re-encode polled batches; replying empty poll" + ); + empty_poll_fallback(partition_id) + }), + other => { + warn!( + transport_client_id, + namespace = namespace.inner(), + reply_was_none = other.is_none(), + "partition read failed; replying empty poll" + ); + Err(empty_poll_fallback(partition_id)) + } + } +} + +/// The fail-fast poll reply for a partition that could not answer: the +/// 16-byte empty poll for `partition_id`, riding the re-sync sentinel +/// channel when the id is the sentinel and the empty-frame channel +/// otherwise. +fn empty_poll_fallback(partition_id: u32) -> (Bytes, FrameChannel) { + let channel = if partition_id == RESYNC_REQUIRED_PARTITION_SENTINEL { + FrameChannel::ResyncSentinel + } else { + FrameChannel::EmptyFrame + }; + (empty_polled_messages_body(partition_id), channel) +} + +/// Serve `get_consumer_offset`. An empty status-0 body decodes as `None` on +/// the SDK side, so it is reserved for "this consumer has no stored offset" -- +/// an unresolved consumer group included, since a deleted group has no offset +/// to report. A malformed request or an unresolved stream, topic, or partition +/// denies with a nonzero status instead, so an addressing typo cannot read +/// back as a fresh consumer. // TODO(hubcio): plain local partition_read with no primary gate, so a // follower answers from its own (possibly lagging) offset state. Needs the // same is-caught-up-primary gate the auto-commit path has, or an explicit @@ -725,13 +781,15 @@ pub(in crate::dispatch) async fn handle_get_consumer_offset( SB: SuperblockStore + 'static, { let Ok(wire) = GetConsumerOffsetRequest::decode_from(request_body(request)) else { - // Undecodable: an empty body decodes as None (no offset) on the SDK. - send_non_replicated_bytes( + // Same rule as the poll above: an empty body is the legitimate + // "no offset stored" answer, so a malformed request must not send + // one. Byte-identical frames on the same channel with the same + // context would also be indistinguishable in the send-failure log. + send_non_replicated_deny( shard, request, transport_client_id, - Bytes::new(), - "get_consumer_offset", + IggyError::InvalidCommand.as_code(), ) .await; return; @@ -749,7 +807,7 @@ pub(in crate::dispatch) async fn handle_get_consumer_offset( return; } let body = match resolve_consumer_offset_request(shard, &wire) { - Ok((namespace, partition_id, consumer)) => { + Ok(Some((namespace, partition_id, consumer))) => { match shard .partition_read(namespace, PartitionRead::ConsumerOffset { consumer }) .await @@ -761,71 +819,37 @@ pub(in crate::dispatch) async fn handle_get_consumer_offset( _ => Bytes::new(), } } - // A partition id that does not exist in a resolvable topic is a client - // addressing error, the same one the poll path denies typed. An empty - // body decodes as `None` -- indistinguishable from "this consumer has - // no stored offset yet" -- so the caller cannot tell a typo from a - // fresh consumer. - Err(error @ IggyError::PartitionNotFound(..)) => { + // An unresolved group has no offset to report, the one thing the + // empty body may mean. + Ok(None) => Bytes::new(), + // Everything the resolve rejects is a client error: an unresolved + // stream, topic, or partition, or `resolve_partition_namespace`'s + // generic rejection. All must surface typed -- an empty body decodes as + // `None`, indistinguishable from "no stored offset yet", so a typo'd or + // deleted target would read back as a fresh consumer and resume from + // its configured default with no error anywhere. The same read over + // REST 404s. + Err(error) => { warn!( transport_client_id, error = %error, - "get_consumer_offset rejected: partition not found" + "get_consumer_offset rejected; replying denied" ); send_non_replicated_deny(shard, request, transport_client_id, error.as_code()).await; return; } - Err(error) => { - warn!( - transport_client_id, - error = %error, - "get_consumer_offset request rejected; replying empty" - ); - Bytes::new() - } }; send_non_replicated_bytes( shard, request, transport_client_id, body, + FrameChannel::Reply, "get_consumer_offset", ) .await; } -/// Ack a consumer-offset op whose body could not be rewritten for the -/// partition plane with an empty Reply. The SDK connection processes replies -/// in lockstep, so a silent drop wedges every subsequent request on that -/// connection. -#[allow(clippy::future_not_send)] -async fn send_empty_partition_reply( - shard: &Rc>, - transport_client_id: u128, - request_header: &RoutedRequestHeader, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let commit = current_metadata_commit(shard); - let reply = build_empty_reply(request_header, transport_client_id, 0, commit); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - operation = ?request_header.operation, - "failed to surface empty partition reply" - ); - } -} - /// Wait (bounded) until this shard holds a routing row for `namespace`. Fast /// path: row already present -> no wait. /// @@ -961,10 +985,15 @@ where /// namespace, partition, and polling consumer. Shared by the TCP dispatch and /// the HTTP route; needs no client id because offset reads are not fenced /// (any client may read a group's offset, member or not). +/// +/// `Ok(None)` is the one outcome that means "no offset exists to report" (an +/// unresolved group). It is a separate outcome rather than an error code +/// because [`resolve_partition_namespace`] also rejects an addressing miss +/// with `InvalidIdentifier`, and the caller must deny that one typed. pub fn resolve_consumer_offset_request( shard: &Rc>, wire: &GetConsumerOffsetRequest, -) -> Result<(IggyNamespace, u32, PollingConsumer), IggyError> +) -> Result, IggyError> where B: ShellBus, MJ: JournalHandle + 'static, @@ -981,19 +1010,21 @@ where // it, member or not), the same key the write path is rewritten to. An // unresolved group (e.g. deleted) has no offset, so the read reports None. let consumer = if wire.consumer.kind == KIND_CONSUMER_GROUP { - let group_id = shard + let Some(group_id) = shard .plane .metadata() .mux_stm .streams() .resolve_consumer_group_id(&wire.stream_id, &wire.topic_id, &wire.consumer.id) - .ok_or(IggyError::InvalidIdentifier)?; + else { + return Ok(None); + }; #[allow(clippy::cast_possible_truncation)] PollingConsumer::ConsumerGroup(group_id as usize, partition_id as usize) } else { polling_consumer_from_wire(&wire.consumer, partition_id)? }; - Ok((namespace, partition_id, consumer)) + Ok(Some((namespace, partition_id, consumer))) } fn polling_consumer_from_wire( @@ -1045,8 +1076,9 @@ fn polling_strategy_from_wire( /// The consensus reply is forwarded verbatim: nothing-to-delete commits a /// no-op `TruncatePartition(0)` and acks, while a not-primary rejection /// reaches the client as `TransientNotCommitted` so the SDK replays instead -/// of mistaking a dropped delete for success. Only a malformed / unresolvable -/// request is acked empty without a commit. +/// of mistaking a dropped delete for success. Nothing is acked empty: a +/// malformed body denies typed, and an unresolvable namespace commits a typed +/// rejection against the client's raw identifiers so its retry dedups. #[allow(clippy::future_not_send)] pub(in crate::dispatch) async fn handle_delete_segments_request( shard: &Rc>, @@ -1087,7 +1119,7 @@ pub(in crate::dispatch) async fn handle_delete_segments_request( ) .await { - Ok(truncate) => Some(truncate), + Ok(truncate) => truncate, // The owning partition has not converged on the committed log yet, so // the delete cannot be resolved to a watermark. Reply with the // result-framed transient rejection (under the TruncatePartition @@ -1104,67 +1136,50 @@ pub(in crate::dispatch) async fn handle_delete_segments_request( 0, 0, ); - let reply = build_result_rejection_reply( + send_result_rejection( + shard, + transport_client_id, template.header(), - current_metadata_commit(shard), - IggyError::TransientNotAccepted.as_code(), - ); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - "delete_segments: failed to send transient rejection" - ); - } + &IggyError::TransientNotAccepted, + "delete_segments_transient_rejection", + ) + .await; return; } - Err(_) => None, - }; - - let reply = if let Some(truncate) = truncate { - // Forward the consensus reply verbatim, exactly like the generic - // metadata path: a committed success acks the delete, and a - // result-framed `TransientNotCommitted` rejection makes the SDK - // replay the request. Acking unconditionally here would swallow a - // not-primary rejection and drop the delete on the floor while the - // client believes it succeeded. - let Some(reply) = submit_client_request_on_owner(shard, truncate).await else { - // Transient submit failure (not primary / view change). Stay - // silent; the SDK read-timeout replays the same request id, - // which re-resolves and commits. Acking here would advance the - // client past an unrecorded request and gap the next metadata - // op. - warn!( - transport_client_id, - "delete_segments: transient submit; client will replay" - ); + // Undecodable body (never produced by the SDK): deny typed rather + // than ack empty. Both keep the lockstep stream framed, but a + // status-0 ack reads as a completed trim. Unresolvable-but-well-formed + // targets commit a typed rejection instead (see the resolve). + Err(error) => { + send_deny_reply(shard, transport_client_id, &header, error.as_code()).await; return; - }; - reply - } else { - // Undecodable body (never produced by the SDK): ack empty so the - // lockstep stream stays framed; the typed decoder surfaces the - // failure client-side. Unresolvable-but-well-formed targets commit a - // typed rejection instead (see the resolve), so only a wire-corrupt - // request can gap the sequence here. - let commit = current_metadata_commit(shard); - build_empty_reply(&header, transport_client_id, session, commit).into_generic() + } }; - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_frozen()) - .await - { + + // Forward the consensus reply verbatim, exactly like the generic metadata + // path: a committed success acks the delete, and a result-framed + // `TransientNotCommitted` rejection makes the SDK replay the request. + // Acking unconditionally here would swallow a not-primary rejection and + // drop the delete on the floor while the client believes it succeeded. + let Some(reply) = submit_client_request_on_owner(shard, truncate).await else { + // Transient submit failure (not primary / view change). Stay silent; + // the SDK read-timeout replays the same request id, which re-resolves + // and commits. Acking here would advance the client past an + // unrecorded request and gap the next metadata op. warn!( transport_client_id, - error = %error, - "delete_segments: failed to send reply" + "delete_segments: transient submit; client will replay" ); - } + return; + }; + send_host_frame( + &shard.bus, + transport_client_id, + reply.into_frozen(), + FrameChannel::Reply, + "delete_segments_reply", + ) + .await; } /// Resolve a client `DeleteSegments` to the `TruncatePartition` that commits the @@ -1175,9 +1190,11 @@ pub(in crate::dispatch) async fn handle_delete_segments_request( /// `request` number; `client_id` / `session` are the bound VSR identity the /// truncate commits under. A resolvable namespace with nothing sealed to delete /// still yields a `TruncatePartition(up_to_offset = 0)` so the metadata request -/// sequence stays contiguous. `Err` on a malformed body or an unresolved -/// namespace: the TCP caller drops it to a silent replay, the HTTP caller renders -/// the error. +/// sequence stays contiguous, and an UNRESOLVABLE one yields the truncate +/// against the client's raw identifiers (the apply rejects it as a committed +/// result). `Err` is only `InvalidCommand` for a malformed body and +/// `TransientNotAccepted` for a partition behind the commit frontier: the TCP +/// caller denies typed, the HTTP caller renders the error. #[allow(clippy::future_not_send)] #[allow(clippy::cast_possible_truncation)] pub async fn resolve_delete_segments_truncate( @@ -1279,7 +1296,7 @@ where mod tests { use super::*; use crate::dispatch::test_support::{ - SpyBus, TestMux, TestShard, prepare_message, request_message, + SpyBus, TestMux, TestShard, prepare_message, request_message, test_shard, }; use iggy_binary_protocol::ReplyHeader; use iggy_binary_protocol::primitives::partition_assignment::CreatedPartitionAssignment; @@ -1289,6 +1306,7 @@ mod tests { CreateTopicRequest, CreateTopicWithAssignmentsRequest, }; use iggy_binary_protocol::{WireName, WireOptions, WirePartitioning}; + use iggy_common::Identifier; use iggy_common::defaults::DEFAULT_ROOT_USER_ID; use metadata::IggyMetadata; use metadata::stm::StateMachine as _; @@ -1302,6 +1320,145 @@ mod tests { ShardIdentity, shard_channel, }; + /// An undecodable request body is a PERMANENT client error, so every read + /// on this path must answer a nonzero status. The fail-fast shapes these + /// used to borrow all decode as success: the 16-byte empty poll reads as a + /// 0-message poll, an empty consumer-offset body as "no offset stored", + /// and a status-0 `delete_segments` ack as a completed trim. A client on a + /// skewed protocol would loop on those forever with no error to surface. + #[compio::test] + async fn undecodable_read_bodies_must_deny_typed_not_fabricate_success() { + const TRANSPORT: u128 = 91; + const VSR_CLIENT: u128 = 1; + const SESSION: u64 = 1; + const STATUS_OFFSET: usize = std::mem::offset_of!(ReplyHeader, status); + // Shorter than any of the three request encodings, so each decoder + // fails on length before it can interpret a field. + const TRUNCATED_BODY: &[u8] = &[0x01]; + + let bus = SpyBus::default(); + let shard = Rc::new(test_shard(&bus, 0, 1, 1)); + + let poll = request_message( + Operation::NonReplicated, + VSR_CLIENT, + SESSION, + 1, + TRUNCATED_BODY, + ); + handle_poll_messages(&shard, TRANSPORT, &poll, Some(DEFAULT_ROOT_USER_ID)).await; + + let offset = request_message( + Operation::NonReplicated, + VSR_CLIENT, + SESSION, + 2, + TRUNCATED_BODY, + ); + handle_get_consumer_offset(&shard, TRANSPORT, &offset, Some(DEFAULT_ROOT_USER_ID)).await; + + let delete = request_message( + Operation::DeleteSegments, + VSR_CLIENT, + SESSION, + 3, + TRUNCATED_BODY, + ); + handle_delete_segments_request(&shard, TRANSPORT, Some((VSR_CLIENT, SESSION)), &delete) + .await; + + let replies = bus.client_replies.borrow(); + assert_eq!( + replies.len(), + 3, + "one deny frame per undecodable request, none of them silent" + ); + for (label, (client, frame)) in ["poll_messages", "get_consumer_offset", "delete_segments"] + .into_iter() + .zip(replies.iter()) + { + assert_eq!(*client, TRANSPORT, "{label} deny must target the transport"); + let status = + u32::from_le_bytes(frame[STATUS_OFFSET..STATUS_OFFSET + 4].try_into().unwrap()); + assert_eq!( + status, + IggyError::InvalidCommand.as_code(), + "{label} with an undecodable body must carry a nonzero status" + ); + } + } + + /// A WELL-FORMED read against a stream that does not exist is a permanent + /// addressing error, and it reaches the server through a public SDK method + /// (a typo'd or deleted stream). Both fail-fast shapes decode as success: + /// the 16-byte empty poll as a 0-message read, and an empty + /// consumer-offset body as `None`, which is indistinguishable from a fresh + /// consumer -- so the caller resumes from its configured default and + /// silently reprocesses. The same lookup 404s over REST. + #[compio::test] + async fn unresolved_read_targets_must_deny_typed_not_read_as_empty() { + const TRANSPORT: u128 = 91; + const VSR_CLIENT: u128 = 1; + const SESSION: u64 = 1; + const STATUS_OFFSET: usize = std::mem::offset_of!(ReplyHeader, status); + const MISSING_STREAM: u32 = 404; + + let bus = SpyBus::default(); + let shard = Rc::new(test_shard(&bus, 0, 1, 1)); + + let poll_body = PollMessagesRequest { + consumer: WireConsumer::consumer(WireIdentifier::Numeric(1)), + stream_id: WireIdentifier::Numeric(MISSING_STREAM), + topic_id: WireIdentifier::Numeric(1), + partition_id: Some(1), + strategy: WirePollingStrategy::offset(0), + count: 10, + auto_commit: false, + } + .to_bytes(); + let poll = request_message(Operation::NonReplicated, VSR_CLIENT, SESSION, 1, &poll_body); + handle_poll_messages(&shard, TRANSPORT, &poll, Some(DEFAULT_ROOT_USER_ID)).await; + + let offset_body = GetConsumerOffsetRequest { + consumer: WireConsumer::consumer(WireIdentifier::Numeric(1)), + stream_id: WireIdentifier::Numeric(MISSING_STREAM), + topic_id: WireIdentifier::Numeric(1), + partition_id: Some(1), + } + .to_bytes(); + let offset = request_message( + Operation::NonReplicated, + VSR_CLIENT, + SESSION, + 2, + &offset_body, + ); + handle_get_consumer_offset(&shard, TRANSPORT, &offset, Some(DEFAULT_ROOT_USER_ID)).await; + + let replies = bus.client_replies.borrow(); + assert_eq!( + replies.len(), + 2, + "one deny frame per unresolved read, none of them silent" + ); + for (label, (client, frame)) in ["poll_messages", "get_consumer_offset"] + .into_iter() + .zip(replies.iter()) + { + assert_eq!(*client, TRANSPORT, "{label} deny must target the transport"); + let status = + u32::from_le_bytes(frame[STATUS_OFFSET..STATUS_OFFSET + 4].try_into().unwrap()); + assert_eq!( + status, + IggyError::StreamIdNotFound( + Identifier::numeric(MISSING_STREAM).expect("a nonzero numeric identifier") + ) + .as_code(), + "{label} against a missing stream must carry the not-found status" + ); + } + } + /// A partition write whose routable wait exhausts (namespace committed, /// but no reconciler ever seeds this shard's routing row -- the state a /// teardown/rematerialise churn leaves behind) must answer a nonzero diff --git a/core/server/src/dispatch/reads.rs b/core/server/src/dispatch/reads.rs index 807b342b00..bcbb7de69f 100644 --- a/core/server/src/dispatch/reads.rs +++ b/core/server/src/dispatch/reads.rs @@ -26,9 +26,11 @@ //! the HTTP layer), never in the builder. use crate::cluster_meta::ClusterRoster; -use crate::dispatch::authz::{authorize_default_read, authorize_uid, send_non_replicated_deny}; +use crate::dispatch::authz::{authorize_default_read, authorize_uid}; +use crate::dispatch::failure::{ + FrameChannel, send_host_frame, send_non_replicated_bytes, send_non_replicated_deny, +}; use crate::dispatch::partition::{handle_get_consumer_offset, handle_poll_messages}; -use crate::dispatch::send_non_replicated_bytes; use crate::responses::{ build_empty_reply, build_get_me_response, build_get_personal_access_tokens_response, build_non_replicated_response, connected_client_to_response, current_metadata_commit, @@ -67,7 +69,7 @@ use metadata::permissioner::Permissioner; use server_common::Message; use std::cell::RefCell; use std::future::Future; -use std::net::IpAddr; +use std::net::{IpAddr, SocketAddr}; use std::pin::pin; use std::rc::Rc; use std::sync::Arc; @@ -96,6 +98,7 @@ async fn handle_get_personal_access_tokens( request, transport_client_id, response.to_bytes(), + FrameChannel::Reply, "get_personal_access_tokens", ) .await; @@ -123,6 +126,7 @@ async fn handle_get_me( request, transport_client_id, response.to_bytes(), + FrameChannel::Reply, "get_me", ) .await; @@ -340,6 +344,13 @@ pub(in crate::dispatch) async fn handle_non_replicated_request( system_config: &Arc, transport_client_id: u128, request: Message, + // Acting user, peer address and read-your-writes floor for the read gates + // below, resolved by the funnel in the same connection lookup as the + // heartbeat. `user_id` is `None` only on the pre-auth path (PING), which + // serves ungated codes; the gated arms fail closed on it. An unknown + // connection arrives with a floor of `0`: it was promised nothing, so its + // reads wait for nothing. + (user_id, client_address, watermark): (Option, Option, u64), ) where B: ShellBus, MJ: JournalHandle + 'static, @@ -349,16 +360,10 @@ pub(in crate::dispatch) async fn handle_non_replicated_request( { const CODE_RANGE: std::ops::Range = 0..4; let code = u32::from_le_bytes(request.header().reserved[CODE_RANGE].try_into().unwrap()); - // Acting user, peer address and read-your-writes floor for the gates - // below, resolved in one connection lookup. `user_id` is `None` only on the - // pre-auth path (PING), which serves ungated codes; the gated arms fail - // closed on it. - let (user_id, client_address, watermark) = sessions.borrow().read_context(transport_client_id); match code { PING_CODE => { - // A ping is the client's liveness proof; reset its staleness clock - // so the heartbeat verifier doesn't evict an active connection. - sessions.borrow_mut().record_heartbeat(transport_client_id); + // No `record_heartbeat` here: the funnel records one for EVERY + // frame before classification, so a ping is already covered. let commit = current_metadata_commit(shard); let reply = build_empty_reply( request.header(), @@ -366,17 +371,14 @@ pub(in crate::dispatch) async fn handle_non_replicated_request( request.header().session, commit, ); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - "failed to send non-replicated ping reply" - ); - } + send_host_frame( + &shard.bus, + transport_client_id, + reply.into_generic().into_frozen(), + FrameChannel::Reply, + "ping_reply", + ) + .await; } GET_ME_CODE => { // Self-scoped, so no permissioner rule -- but the consumer-group @@ -421,6 +423,7 @@ pub(in crate::dispatch) async fn handle_non_replicated_request( &request, transport_client_id, response.to_bytes(), + FrameChannel::Reply, "get_clients", ) .await; @@ -469,8 +472,15 @@ pub(in crate::dispatch) async fn handle_non_replicated_request( } .to_bytes() }); - send_non_replicated_bytes(shard, &request, transport_client_id, bytes, "get_client") - .await; + send_non_replicated_bytes( + shard, + &request, + transport_client_id, + bytes, + FrameChannel::Reply, + "get_client", + ) + .await; } GET_SNAPSHOT_FILE_CODE => { handle_get_snapshot(shard, system_config, transport_client_id, &request, user_id).await; @@ -544,6 +554,27 @@ async fn handle_default_non_replicated( }) .await { + // Same line as the builder-`Err` branch below: `send_non_replicated_deny` + // logs only on send FAILURE, so a refusal that reaches the client would + // otherwise leave nothing server-side - including the refusal this gate + // now issues for every armless or unknown non-replicated code. + // The frontier wait reaches here through the same `Err` and has + // already counted and logged itself, so it stays at `debug!`: a durably + // lagging node refuses every held read of every client for as long as + // it lags, and warning here would be a line per refusal. + if matches!(error, IggyError::TransientNotAccepted) { + debug!( + transport_client_id, + code, "denying non-replicated VSR request; read frontier unreached" + ); + } else { + warn!( + transport_client_id, + code, + error = %error, + "denying non-replicated VSR request" + ); + } send_non_replicated_deny(shard, request, transport_client_id, error.as_code()).await; return; } @@ -571,18 +602,14 @@ async fn handle_default_non_replicated( request.header().session, commit, ); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - code, - error = %error, - "failed to send non-replicated VSR reply" - ); - } + send_host_frame( + &shard.bus, + transport_client_id, + reply.into_generic().into_frozen(), + FrameChannel::Reply, + "non_replicated_reply", + ) + .await; } Err(error) => { // Surface the builder's typed error (unsupported op, undecodable @@ -659,6 +686,7 @@ async fn handle_get_snapshot( request, transport_client_id, GetSnapshotResponse { data: archive }.to_bytes(), + FrameChannel::Reply, "get_snapshot", ) .await; @@ -732,6 +760,7 @@ async fn handle_sync_consumer_group( request, transport_client_id, body, + FrameChannel::Reply, "sync_consumer_group", ) .await; diff --git a/core/server/src/dispatch/session_ops.rs b/core/server/src/dispatch/session_ops.rs index d43bcc17b6..a8d2e62db3 100644 --- a/core/server/src/dispatch/session_ops.rs +++ b/core/server/src/dispatch/session_ops.rs @@ -32,6 +32,9 @@ //! two together, so logout and eviction must release BOTH -- every teardown //! path below pairs `remove_connection` with a replicated `Logout`. +use crate::dispatch::failure::{ + FrameChannel, send_eviction, send_host_frame, send_result_rejection, +}; use crate::dispatch::login_error::LoginRegisterError; use crate::responses::{ build_deny_reply, build_empty_reply, build_login_register_reply, current_metadata_commit, @@ -39,11 +42,7 @@ use crate::responses::{ use crate::session_manager::{ClientSdkInfo, SessionManager}; use crate::shell::{ShellBus, ShellShard}; use crate::wire::request_body; -use consensus::{ - Consensus, DISCONNECT_LOGOUT_REQUEST_ID, EvictionContext, MetadataHandle, - build_eviction_message, build_incompatible_protocol_eviction_message, - build_result_rejection_reply, -}; +use consensus::{Consensus, DISCONNECT_LOGOUT_REQUEST_ID, MetadataHandle}; use iggy_binary_protocol::PrepareHeader; use iggy_binary_protocol::requests::users::{LoginRegisterRequest, LoginRegisterWithPatRequest}; use iggy_binary_protocol::{ @@ -241,10 +240,14 @@ where let commit = current_metadata_commit(shard).max(session); let reply = build_login_register_reply(request_header, vsr_client_id, session, commit, user_id); - let _ = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await; + send_host_frame( + &shard.bus, + transport_client_id, + reply.into_generic().into_frozen(), + FrameChannel::Reply, + "login_replay_reply", + ) + .await; return Ok(()); } @@ -288,17 +291,14 @@ where // lower number would make one frame contradict itself. let commit = current_metadata_commit(shard).max(session); let reply = build_login_register_reply(request_header, vsr_client_id, session, commit, user_id); - let send_result = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await; - if let Err(error) = send_result { - warn!( - transport_client_id, - error = %error, - "failed to send login/register reply" - ); - } + send_host_frame( + &shard.bus, + transport_client_id, + reply.into_generic().into_frozen(), + FrameChannel::Reply, + "login_register_reply", + ) + .await; Ok(()) } @@ -331,58 +331,28 @@ async fn surface_login_failure( SB: SuperblockStore + 'static, { if error.is_terminal() { - send_login_eviction( + send_eviction( shard, transport_client_id, request_header.client, eviction_reason_for(error), + "login_rejection", ) .await; } else { // Which code the hint carries is what tells the client whether the // replay may move to another node: see `transient_login_code`. - send_login_transient_reply( + send_result_rejection( shard, transport_client_id, request_header, - transient_login_code(error), + &transient_login_code(error), + "login_transient_replay_hint", ) .await; } } -/// Result-framed transient Reply on a non-terminal failed Register. The SDK -/// decodes the nonzero result code and replays the same login on the same -/// connection. Only call for transient errors -- see -/// [`surface_login_failure`]. -#[allow(clippy::future_not_send)] -async fn send_login_transient_reply( - shard: &Rc>, - transport_client_id: u128, - request_header: &RoutedRequestHeader, - code: IggyError, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let commit = current_metadata_commit(shard); - let reply = build_result_rejection_reply(request_header, commit, code.as_code()); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - "failed to send login transient reply" - ); - } -} - /// Wire code for a transient (non-terminal) login/register failure. /// /// `TransientNotAccepted` asserts nothing was committed: the register never @@ -420,53 +390,6 @@ const fn eviction_reason_for(error: &LoginRegisterError) -> EvictionReason { } } -/// Reject a replicated request from an unbound transport with a typed -/// `Eviction(NoSession)` frame: the session the client believes it has is -/// gone, so it must register again. Pre-auth non-replicated reads get a -/// deny Reply instead (no session exists, so nothing is evicted). -/// -/// The SDK's reply decoder maps eviction reasons to typed errors -/// (`NoSession` -> `Unauthenticated`), so clients fail fast with the same -/// error the legacy server returns instead of a body-decode failure. The -/// eviction context is best-effort off the metadata consensus (peer shards -/// have none; zeroes are cosmetic -- the SDK only reads the reason). -#[allow(clippy::future_not_send)] -pub(in crate::dispatch) async fn send_unauthenticated_eviction( - shard: &Rc>, - transport_client_id: u128, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let ctx = shard.plane.metadata().consensus.as_ref().map_or( - consensus::EvictionContext { - cluster: 0, - view: 0, - replica: 0, - }, - consensus::EvictionContext::from_consensus, - ); - let eviction = consensus::build_eviction_message( - ctx, - transport_client_id, - iggy_binary_protocol::EvictionReason::NoSession, - ); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, eviction.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - "failed to send unauthenticated eviction" - ); - } -} - /// Per-shard heartbeat verifier: evict connections that have not pinged within /// `1.2 x interval`. Mirrors the legacy `verify_heartbeats` periodic task. /// Eviction reuses the disconnect path (drops the client from its consumer @@ -552,35 +475,21 @@ async fn evict_stale_client( if let Some((vsr_client_id, session)) = bound { submit_disconnect_logout(Rc::clone(shard), vsr_client_id, session); } - let ctx = shard.plane.metadata().consensus.as_ref().map_or( - consensus::EvictionContext { - cluster: 0, - view: 0, - replica: 0, - }, - consensus::EvictionContext::from_consensus, - ); - let eviction = consensus::build_eviction_message( - ctx, + // The eviction itself is done and observed at this point (session + // dropped, `Logout` submitted). The client notice below is best-effort + // and logs its own send failure, so this line must not claim it. + warn!( transport_client_id, - iggy_binary_protocol::EvictionReason::StaleClient, + "evicted stale client (missed heartbeat); sending the eviction notice" ); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, eviction.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - "failed to send stale-client eviction" - ); - } else { - warn!( - transport_client_id, - "evicted stale client (missed heartbeat)" - ); - } + send_eviction( + shard, + transport_client_id, + transport_client_id, + EvictionReason::StaleClient, + "stale_client_eviction", + ) + .await; } /// Answer a backup's forwarded `Register` from the node it named primary. @@ -1180,17 +1089,14 @@ pub(in crate::dispatch) async fn handle_logout_request( ); let commit = current_metadata_commit(shard); let reply = build_empty_reply(request.header(), transport_client_id, 0, commit); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - "failed to send unbound logout reply" - ); - } + send_host_frame( + &shard.bus, + transport_client_id, + reply.into_generic().into_frozen(), + FrameChannel::Reply, + "unbound_logout_reply", + ) + .await; return; }; @@ -1210,17 +1116,14 @@ pub(in crate::dispatch) async fn handle_logout_request( commit, transient_logout_code(&error).as_code(), ); - if let Err(send_error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %send_error, - "failed to send logout deny reply" - ); - } + send_host_frame( + &shard.bus, + transport_client_id, + reply.into_generic().into_frozen(), + FrameChannel::TypedDeny, + "logout_transient_deny", + ) + .await; return; } }; @@ -1228,17 +1131,14 @@ pub(in crate::dispatch) async fn handle_logout_request( sessions.borrow_mut().remove_connection(transport_client_id); let reply = build_empty_reply(request.header(), vsr_client_id, session, commit); - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, reply.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - "failed to send logout reply" - ); - } + send_host_frame( + &shard.bus, + transport_client_id, + reply.into_generic().into_frozen(), + FrameChannel::Reply, + "logout_reply", + ) + .await; } /// Preserve the client identity when a Logout may already have entered the @@ -1281,11 +1181,12 @@ pub(in crate::dispatch) async fn handle_login_register_request( transport_client_id, "rejecting login: body has no decodable version prefix" ); - send_login_eviction( + send_eviction( shard, transport_client_id, vsr_client_id, EvictionReason::MalformedLogin, + "login_rejection", ) .await; return; @@ -1298,11 +1199,12 @@ pub(in crate::dispatch) async fn handle_login_register_request( sdk_version = %version_info.sdk_version, "rejecting login: incompatible protocol version" ); - send_login_eviction( + send_eviction( shard, transport_client_id, vsr_client_id, EvictionReason::IncompatibleProtocol, + "login_rejection", ) .await; return; @@ -1395,11 +1297,12 @@ pub(in crate::dispatch) async fn handle_login_register_request( transport_client_id, "rejecting register request: invalid credentials" ); - send_login_eviction( + send_eviction( shard, transport_client_id, request.header().client, EvictionReason::InvalidCredentials, + "login_rejection", ) .await; return; @@ -1409,61 +1312,16 @@ pub(in crate::dispatch) async fn handle_login_register_request( transport_client_id, "rejecting register request with unsupported payload shape" ); - send_login_eviction( + send_eviction( shard, transport_client_id, request.header().client, EvictionReason::MalformedLogin, + "login_rejection", ) .await; } -/// Best-effort login-rejection eviction. Terminal one-way frame; a gone -/// connection has nothing to recover, so the send error is logged and -/// dropped. Consensus context (cluster/view/replica) is stamped on the -/// metadata shard and zeroed elsewhere -- the SDK only reads the reason, -/// plus the protocol window on `IncompatibleProtocol`. -#[allow(clippy::future_not_send)] -pub(in crate::dispatch) async fn send_login_eviction( - shard: &Rc>, - transport_client_id: u128, - vsr_client_id: u128, - reason: EvictionReason, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let ctx = shard.plane.metadata().consensus.as_ref().map_or( - EvictionContext { - cluster: 0, - view: 0, - replica: 0, - }, - EvictionContext::from_consensus, - ); - let eviction = match reason { - EvictionReason::IncompatibleProtocol => { - build_incompatible_protocol_eviction_message(ctx, vsr_client_id) - } - _ => build_eviction_message(ctx, vsr_client_id, reason), - }; - if let Err(error) = shard - .bus - .send_to_client(transport_client_id, eviction.into_generic().into_frozen()) - .await - { - warn!( - transport_client_id, - error = %error, - reason = ?reason, - "failed to send login eviction" - ); - } -} - #[cfg(test)] mod tests { use super::*; diff --git a/core/server/src/dispatch/test_support.rs b/core/server/src/dispatch/test_support.rs index e52578c64d..c0a0057345 100644 --- a/core/server/src/dispatch/test_support.rs +++ b/core/server/src/dispatch/test_support.rs @@ -54,12 +54,16 @@ pub type RecordedReplies = Rc)>>>; pub type RecordedReplicaSends = Rc)>>>; /// Records every client-bound reply and replica-bound frame (target + -/// bytes) instead of writing to a socket; everything else is a no-op. The -/// two `ShellBus` halves are stubbed. +/// bytes) instead of writing to a socket, and counts connection-lost hook +/// installs; everything else is a no-op. The two `ShellBus` halves are +/// stubbed. #[derive(Debug, Clone, Default)] pub struct SpyBus { pub client_replies: RecordedReplies, pub replica_sends: RecordedReplicaSends, + /// Installs of the client connection-lost hook, one per handler built + /// on this bus. + pub connection_lost_hooks: Rc>, /// Resolve [`MessageBus::sleep`] immediately instead of arming a real /// timer. The register forward is the only path here that races a /// timer, and its budget is five seconds -- too long to wait for in a @@ -143,7 +147,10 @@ impl ConnectionInstaller for SpyBus { fn client_meta(&self, _client_id: u128) -> Option> { None } - fn set_client_connection_lost_fn(&self, _f: ClientConnectionLostFn) {} + fn set_client_connection_lost_fn(&self, _f: ClientConnectionLostFn) { + self.connection_lost_hooks + .set(self.connection_lost_hooks.get() + 1); + } } /// Consensus incarnations standing for two successive boots of one node, as diff --git a/core/server/src/http/handlers.rs b/core/server/src/http/handlers.rs index 2a082405b3..fc877054bb 100644 --- a/core/server/src/http/handlers.rs +++ b/core/server/src/http/handlers.rs @@ -121,10 +121,6 @@ use shard::{PartitionRead, PartitionReadReply}; use crate::dispatch::partition::{resolve_consumer_offset_request, resolve_poll_request}; use crate::dispatch::session_ops::{verify_login_credentials, verify_pat_credentials}; -use crate::dispatch::{ - validate_option_keys, validate_topic_bounds, validate_topic_size_floor, - warn_unenforceable_topic_size, warn_unenforceable_topic_size_on_partition_add, -}; use crate::http::error::{ Consistency, ConsistencyQuery, CustomError, PartitionWriteError, ProduceAck, ProduceQuery, ReadError, WriteError, @@ -150,6 +146,10 @@ use crate::http::wire::{ use crate::responses::{ build_polled_messages_body, build_raw_pat_reply, connected_client_to_response, }; +use crate::rewrite::{ + validate_option_keys, validate_topic_bounds, validate_topic_size_floor, + warn_unenforceable_topic_size, warn_unenforceable_topic_size_on_partition_add, +}; use crate::snapshot; /// `GET /ping` response body, matching the legacy HTTP server's health probe. @@ -1335,8 +1335,9 @@ pub(in crate::http) async fn get_consumer_offset( .await?; let wire = consumer_offset_wire_request(&stream_id, &topic_id, &query).map_err(ReadError::Rejected)?; - let (namespace, partition_id, consumer) = - resolve_consumer_offset_request(&state.shard, &wire).map_err(|_| ReadError::NotFound)?; + let (namespace, partition_id, consumer) = resolve_consumer_offset_request(&state.shard, &wire) + .map_err(|_| ReadError::NotFound)? + .ok_or(ReadError::NotFound)?; let reply = SendWrapper::new( state .shard diff --git a/core/server/src/http/submit.rs b/core/server/src/http/submit.rs index a4864086e0..08c23aa63c 100644 --- a/core/server/src/http/submit.rs +++ b/core/server/src/http/submit.rs @@ -23,12 +23,10 @@ use std::rc::Rc; use std::time::{Duration, Instant}; use bytes::Bytes; -use consensus::MetadataHandle; use futures::channel::oneshot; use iggy_binary_protocol::consensus::Command; use iggy_binary_protocol::{GenericHeader, Operation, ReplyHeader, RoutedRequestHeader}; use iggy_common::IggyError; -use metadata::impls::metadata::StreamsFrontend; use server_common::{MESSAGE_ALIGN, Message, iobuf::Frozen}; use tracing::warn; @@ -41,10 +39,9 @@ use crate::http::reply::{classify_partition_reply, committed_payload, eviction_e use crate::http::session::HttpSession; use crate::http::state::HttpInner; use crate::http::wire::build_request_message; -use crate::pat::rewrite_pat_request_for_user; use crate::responses::transient_code; +use crate::rewrite::http_chain; use crate::shell::ServerShard; -use crate::users::maybe_rewrite_user_password_request; use crate::wire::request_body; /// Bound on a partition write's (produce / consumer-offset write) wait for its @@ -217,22 +214,8 @@ async fn submit_gated( request_id, body, ); - let (message, raw_token) = rewrite_pat_request_for_user( - session.user_id, - max_tokens_per_user, - |user_id| { - shard - .plane - .metadata() - .mux_stm - .users() - .read(|users| users.pat_count_of(user_id)) - }, - message, - ) - .map_err(WriteError::Rejected)?; - let message = - maybe_rewrite_user_password_request(shard, message).map_err(WriteError::Rejected)?; + let (message, raw_token) = http_chain(shard, session.user_id, max_tokens_per_user, message) + .map_err(WriteError::Rejected)?; // `DeleteSegments` is not itself a consensus op: resolve it to the metadata // `TruncatePartition` that commits the trim before it reaches consensus, // mirroring the TCP dispatch. The truncate rides this session's burned diff --git a/core/server/src/lib.rs b/core/server/src/lib.rs index b8131e558a..7eb70057f8 100644 --- a/core/server/src/lib.rs +++ b/core/server/src/lib.rs @@ -45,6 +45,7 @@ pub(crate) mod consumer_group; pub(crate) mod dispatch; pub(crate) mod pat; pub(crate) mod responses; +pub(crate) mod rewrite; pub mod session_manager; pub mod shell; diff --git a/core/server/src/responses.rs b/core/server/src/responses.rs index 75a8d2511b..6cea586424 100644 --- a/core/server/src/responses.rs +++ b/core/server/src/responses.rs @@ -595,10 +595,15 @@ where // than let the catch-all's empty-ok attest an artifact that was never // produced. GET_SNAPSHOT_FILE_CODE => Err(IggyError::InvalidCommand), + // Sequenced AFTER the named arms above, so flush keeps answering + // `FeatureUnavailable`. A table-listed non-replicated code with no arm + // is a routing bug and an unknown code is a client bug; the empty-ok + // that used to cover both attested a read that never ran. Only the + // named arms return `Empty`, and there it means "resolved to nothing" + // (the 404 the HTTP path maps). _ => match iggy_binary_protocol::dispatch::lookup_command(code) { - Some(meta) if !meta.is_replicated() => Ok(NonReplicatedResponse::Empty), - Some(_) => Err(IggyError::FeatureUnavailable), - None => Err(IggyError::InvalidCommand), + Some(meta) if meta.is_replicated() => Err(IggyError::FeatureUnavailable), + _ => Err(IggyError::InvalidCommand), }, } } diff --git a/core/server/src/rewrite.rs b/core/server/src/rewrite.rs new file mode 100644 index 0000000000..5f00c588cc --- /dev/null +++ b/core/server/src/rewrite.rs @@ -0,0 +1,570 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! The two pre-consensus request-rewrite chains, side by side. +//! +//! [`tcp_chain`] serves the TCP funnel, [`http_chain`] the HTTP submit. Each +//! step rewrites or validates a request BEFORE consensus, so a rejected +//! request burns no replicated log entry and no plaintext secret enters +//! consensus. The chains enter the PAT rewrite through different functions +//! on purpose: TCP resolves the acting user from the transport +//! `SessionManager` ([`maybe_rewrite_pat_request`]), HTTP authenticates +//! against its own session table and passes the resolved `user_id` +//! ([`rewrite_pat_request_for_user`]). +//! +//! Three more links complete the chains but live in their spines, because +//! they fire partition-read mesh RPCs or are plane-specific. Two are async: +//! the consumer-group Join/Leave enrichment ([`crate::consumer_group`], the +//! TCP funnel calls it after [`tcp_chain`]) and the `DeleteSegments` -> +//! `TruncatePartition` resolution +//! (`dispatch::partition::resolve_delete_segments_truncate`, called by both +//! spines). The third, the consumer-offset rewrite on the partition path +//! (`consumer_group::maybe_rewrite_consumer_offset_request`, called by +//! `dispatch::partition`), is synchronous. + +use crate::pat::{maybe_rewrite_pat_request, rewrite_pat_request_for_user}; +use crate::segment_cleaner::UNENFORCEABLE_TOPIC_SIZE_WARN; +use crate::session_manager::SessionManager; +use crate::shell::{ShellBus, ShellShard}; +use crate::users::maybe_rewrite_user_password_request; +use crate::wire::request_body; +use consensus::MetadataHandle; +use iggy_binary_protocol::requests::partitions::{ + CreatePartitionsRequest, DeletePartitionsRequest, +}; +use iggy_binary_protocol::requests::streams::{CreateStreamRequest, UpdateStreamRequest}; +use iggy_binary_protocol::requests::topics::{CreateTopicRequest, UpdateTopicRequest}; +use iggy_binary_protocol::requests::users::{CreateUserRequest, UpdateUserRequest}; +use iggy_binary_protocol::{ + MAX_PARTITIONS_PER_REQUEST, Operation, PrepareHeader, RoutedRequestHeader, WireDecode, + WireIdentifier, WireOptions, +}; +use iggy_common::{ + IggyByteSize, IggyError, MaxTopicSize, TopicCreateOptions, UPDATABLE_STREAM_OPTION_KEYS, + UPDATABLE_TOPIC_OPTION_KEYS, UPDATABLE_USER_OPTION_KEYS, validate_preallocated_topic_bytes, + validate_topic_segment_size, +}; +use journal::superblock::SuperblockStore; +use journal::{Journal, JournalHandle}; +use metadata::impls::metadata::StreamsFrontend; +use metadata::stm::stream::Streams; +use server_common::Message; +use std::cell::RefCell; +use std::rc::Rc; +use tracing::warn; + +/// The pre-consensus rewrite stages, in chain order. +/// +/// One owner for the set: the funnel's consumer-group rewrite runs outside +/// [`tcp_chain`] but denies through the same path, so it names a variant +/// here instead of passing a bare literal. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum RewriteStage { + PersonalAccessToken, + UserPassword, + StaticBounds, + ConsumerGroup, +} + +impl RewriteStage { + /// The `context` log label, in the same `snake_case` shape as every other + /// frame context. + #[must_use] + pub const fn as_str(self) -> &'static str { + match self { + Self::PersonalAccessToken => "personal_access_token", + Self::UserPassword => "user_password", + Self::StaticBounds => "static_bounds", + Self::ConsumerGroup => "consumer_group", + } + } +} + +/// A staged pre-consensus rejection: `stage` labels the chain step for the +/// deny log line, `error` is the typed code the deny reply carries. +pub struct RewriteDeny { + pub stage: RewriteStage, + pub error: IggyError, +} + +/// The TCP funnel's pre-consensus rewrite chain, in order: the PAT rewrite +/// (resolving the acting user from the transport `sessions` binding), the +/// password rewrite, then the static bounds gate. Mirrors [`http_chain`] +/// plus that bounds step, which the binary wire needs because it has no +/// `command.validate()` layer. Returns the rewritten request and the raw +/// PAT token the funnel substitutes into the committed reply; a rejection +/// names the failing stage for the deny log line. +pub fn tcp_chain( + shard: &Rc>, + sessions: &Rc>, + transport_client_id: u128, + max_tokens_per_user: u32, + request: Message, +) -> Result<(Message, Option), RewriteDeny> +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let (request, raw_pat_token) = maybe_rewrite_pat_request( + sessions, + transport_client_id, + max_tokens_per_user, + |user_id| { + shard + .plane + .metadata() + .mux_stm + .users() + .read(|users| users.pat_count_of(user_id)) + }, + request, + ) + // Token cap reached, malformed body, or a lost session binding. + .map_err(|error| RewriteDeny { + stage: RewriteStage::PersonalAccessToken, + error, + })?; + // Hash raw passwords and, for ChangePassword, verify the current password + // on the primary before replication; see `crate::users`. Replicas store the + // hash directly. A wrong current password is not denied here: it rides + // consensus and applies as a committed InvalidCredentials no-op, so the only + // Err returned is a malformed body. + let request = + maybe_rewrite_user_password_request(shard, request).map_err(|error| RewriteDeny { + stage: RewriteStage::UserPassword, + error, + })?; + static_bounds(shard, &request).map_err(|error| RewriteDeny { + stage: RewriteStage::StaticBounds, + error, + })?; + Ok((request, raw_pat_token)) +} + +/// The HTTP submit's pre-consensus rewrite chain: the PAT rewrite for the +/// already-authenticated `user_id`, then the password rewrite. Mirrors +/// [`tcp_chain`] minus the session lookup (the HTTP listener authenticates +/// against its own session table and resolves the acting user itself) and +/// minus the static bounds step: HTTP enforces the same bounds in its +/// handlers via `command.validate()` plus the validators below. Returns the +/// rewritten request and the raw PAT token the caller substitutes into the +/// committed reply. +pub fn http_chain( + shard: &Rc>, + user_id: u32, + max_tokens_per_user: u32, + request: Message, +) -> Result<(Message, Option), IggyError> +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + let (request, raw_token) = rewrite_pat_request_for_user( + user_id, + max_tokens_per_user, + |user_id| { + shard + .plane + .metadata() + .mux_stm + .users() + .read(|users| users.pat_count_of(user_id)) + }, + request, + )?; + let request = maybe_rewrite_user_password_request(shard, request)?; + Ok((request, raw_token)) +} + +/// Per-request partitions-count cap, shared by create-topic, create-partitions +/// and delete-partitions admission. Runs pre-consensus like +/// [`validate_topic_bounds`]: an oversized count must not burn a replicated +/// log entry (create-partitions admission would also allocate that many +/// consensus-group ids before replicating). +/// +/// Zero passes here because a zero-partition TOPIC is legal (legacy +/// `create_topic` admits `0..=MAX`); the add/remove requests reject it in +/// [`validate_partitions_change_count`]. +const fn validate_partitions_count(partitions_count: u32) -> Result<(), IggyError> { + // The two transports carry the cap under different names: this one on the + // binary path, `MAX_PARTITIONS_COUNT` in the HTTP DTO validators. The + // parity is a documented contract ([`http_chain`]), so it fails the build + // rather than the next cross-transport test. + const _: () = assert!( + MAX_PARTITIONS_PER_REQUEST == iggy_common::MAX_PARTITIONS_COUNT, + "the binary and HTTP partitions-count caps must stay equal" + ); + + if partitions_count > MAX_PARTITIONS_PER_REQUEST { + return Err(IggyError::TooManyPartitions); + } + Ok(()) +} + +/// [`validate_partitions_count`] plus the zero rejection that create-partitions +/// and delete-partitions carry: adding or removing zero partitions is a no-op +/// that would still burn a replicated log entry, bump `Streams::revision` and +/// force every shard through a rebalance pass. Legacy rejects it with +/// `TooManyPartitions` in both handlers (`1..=MAX` on create, `== 0` on +/// delete), so the code matches rather than inventing a new one. +const fn validate_partitions_change_count(partitions_count: u32) -> Result<(), IggyError> { + if partitions_count == 0 { + return Err(IggyError::TooManyPartitions); + } + validate_partitions_count(partitions_count) +} + +/// Static create-topic bounds shared by the TCP and HTTP ingresses. Runs +/// pre-consensus: a rejected request must not burn a replicated log entry, +/// and `prepare_request` errors evict the session instead of denying typed. +/// `ServerDefault` is exempt from the size floor (it resolves against server +/// config at admission, matching legacy); `Unlimited` passes numerically. +/// `segment_size_bytes` is the topic's RESOLVED segment size (explicit +/// option, else this node's default), so a per-topic segment above the +/// global default still floors the topic cap. +pub fn validate_topic_bounds( + partitions_count: u32, + max_topic_size: MaxTopicSize, + segment_size_bytes: u64, +) -> Result<(), IggyError> { + validate_partitions_count(partitions_count)?; + validate_topic_size_floor(max_topic_size, segment_size_bytes) +} + +/// A topic cap below one segment can never be enforced: the first segment +/// already exceeds it. Split out of [`validate_topic_bounds`] because update +/// admission checks the cap without a partitions count to check. +pub fn validate_topic_size_floor( + max_topic_size: MaxTopicSize, + segment_size_bytes: u64, +) -> Result<(), IggyError> { + if !matches!(max_topic_size, MaxTopicSize::ServerDefault) + && max_topic_size.as_bytes_u64() < segment_size_bytes + { + return Err(IggyError::InvalidTopicSize( + max_topic_size, + IggyByteSize::from(segment_size_bytes), + )); + } + Ok(()) +} + +/// Announce an accepted `max_topic_size` the server cannot enforce as written. +/// +/// [`validate_topic_size_floor`] admits any cap of one segment or more, but +/// retention runs PER PARTITION and floors each partition's share at one SEALED +/// segment, which reaches up to one maximum bus frame past `segment_size`. A cap +/// between the two is stored and echoed back verbatim while the server actually +/// keeps `(segment_size + max_message_size) * partitions_count`, so the only +/// moment an operator can be told is the one where they set it. +/// +/// Warns rather than rejects: which caps are accepted is client-visible wire +/// behavior, and tightening it would break topics that already exist. +pub fn warn_unenforceable_topic_size( + max_topic_size: MaxTopicSize, + segment_size_bytes: u64, + max_message_size_bytes: usize, + partitions_count: u32, +) { + let MaxTopicSize::Custom(configured) = max_topic_size else { + return; + }; + let max_message_size_bytes = u64::try_from(max_message_size_bytes).unwrap_or(u64::MAX); + let per_partition_floor = segment_size_bytes.saturating_add(max_message_size_bytes); + let topic_floor = per_partition_floor.saturating_mul(u64::from(partitions_count)); + if configured.as_bytes_u64() >= topic_floor { + return; + } + warn!( + max_topic_size = configured.as_bytes_u64(), + partitions_count, + segment_size = segment_size_bytes, + enforced_per_partition = per_partition_floor, + "{UNENFORCEABLE_TOPIC_SIZE_WARN}" + ); +} + +/// Announce the same unenforceable cap when partitions are ADDED to a topic. +/// +/// The cap is topic-wide but enforcement is per partition, so every added +/// partition shrinks the share: a cap that cleared the floor when the topic was +/// created can stop clearing it here. The request carries only the delta, so +/// the stored cap, segment size and current partition count come from metadata. +pub fn warn_unenforceable_topic_size_on_partition_add( + streams: &Streams, + stream_id: &WireIdentifier, + topic_id: &WireIdentifier, + max_message_size_bytes: usize, + added_partitions_count: u32, +) { + let Some(((stream_slab, topic_slab), _)) = streams.partition_count_context(stream_id, topic_id) + else { + return; + }; + let Some((_, max_topic_size, partitions_count, segment_size)) = + streams.topic_retention_config(stream_slab, topic_slab) + else { + return; + }; + warn_unenforceable_topic_size( + max_topic_size, + segment_size.map_or(iggy_common::DEFAULT_SEGMENT_SIZE, |segment_size| { + segment_size.as_bytes_u64() + }), + max_message_size_bytes, + u32::try_from(partitions_count) + .unwrap_or(u32::MAX) + .saturating_add(added_partitions_count), + ); +} + +/// Reject option keys outside the resource's catalog, pre-consensus. Unknown +/// keys are rejected rather than skipped: a silently ignored knob would hand +/// the client server defaults without it ever learning. Streams and users +/// have no catalog keys yet, so `known` is empty for both until one lands. +pub fn validate_option_keys(options: &WireOptions, known: &[&str]) -> Result<(), IggyError> { + for entry in options { + // Wire validation already enforced UTF-8 string keys. + let key = String::from_utf8_lossy(entry.key); + if !known.contains(&key.as_ref()) { + return Err(IggyError::UnsupportedOptionKey(key.into_owned())); + } + } + Ok(()) +} + +/// Static bounds run pre-consensus so a rejected request burns no +/// replicated log entry; HTTP covers the same bounds via +/// `command.validate()`. A body that fails to decode denies typed too +/// (`InvalidCommand`), instead of riding consensus just to fail there. +#[allow(clippy::too_many_lines)] +fn static_bounds( + shard: &Rc>, + request: &Message, +) -> Result<(), IggyError> +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + match request.header().operation { + Operation::CreateTopic => CreateTopicRequest::decode_from(request_body(request)) + .map_err(|_| IggyError::InvalidCommand) + .and_then(|create_topic| { + // `parse` doubles as the catalog gate: an unknown key or a + // malformed value denies typed here, pre-consensus. + let options = TopicCreateOptions::parse(&create_topic.options)?; + if let Some(segment_size) = options.segment_size { + validate_topic_segment_size( + segment_size.as_bytes_u64(), + iggy_common::MAX_TOPIC_SEGMENT_SIZE, + )?; + } + let segment_size = options.segment_size.map_or_else( + || iggy_common::DEFAULT_SEGMENT_SIZE, + |segment_size| segment_size.as_bytes_u64(), + ); + if options + .preallocate_segments + .unwrap_or(iggy_common::DEFAULT_PREALLOCATE_SEGMENTS) + { + validate_preallocated_topic_bytes(segment_size, create_topic.partitions_count)?; + } + let max_topic_size = options + .max_topic_size + .unwrap_or(MaxTopicSize::ServerDefault); + validate_topic_bounds(create_topic.partitions_count, max_topic_size, segment_size)?; + warn_unenforceable_topic_size( + max_topic_size, + segment_size, + shard.bus_max_message_size(), + create_topic.partitions_count, + ); + Ok(()) + }), + Operation::CreatePartitions => CreatePartitionsRequest::decode_from(request_body(request)) + .map_err(|_| IggyError::InvalidCommand) + .and_then(|create_partitions| { + validate_partitions_change_count(create_partitions.partitions_count)?; + let metadata = shard.plane.metadata(); + warn_unenforceable_topic_size_on_partition_add( + metadata.mux_stm.streams(), + &create_partitions.stream_id, + &create_partitions.topic_id, + shard.bus_max_message_size(), + create_partitions.partitions_count, + ); + Ok(()) + }), + Operation::DeletePartitions => DeletePartitionsRequest::decode_from(request_body(request)) + .map_err(|_| IggyError::InvalidCommand) + .and_then(|delete_partitions| { + validate_partitions_change_count(delete_partitions.partitions_count) + }), + // Only the updatable subset: the create-time knobs are pushed to + // partitions when the topic is built and nothing re-pushes them, so + // accepting one here would store a value no partition ever sees. + Operation::UpdateTopic => UpdateTopicRequest::decode_from(request_body(request)) + .map_err(|_| IggyError::InvalidCommand) + .and_then(|update_topic| { + validate_option_keys(&update_topic.options, UPDATABLE_TOPIC_OPTION_KEYS)?; + let options = TopicCreateOptions::parse(&update_topic.options)?; + let Some(max_topic_size) = options.max_topic_size else { + return Ok(()); + }; + // An update can lower the cap below one segment just as a + // create can, and the stored map would then report a size the + // topic can never enforce. The floor is this topic's own + // segment size, since that key is create-only. + let metadata = shard.plane.metadata(); + let streams = metadata.mux_stm.streams(); + let segment_size = streams + .topic_segment_size(&update_topic.stream_id, &update_topic.topic_id) + .map_or_else( + || iggy_common::DEFAULT_SEGMENT_SIZE, + |segment_size| segment_size.as_bytes_u64(), + ); + validate_topic_size_floor(max_topic_size, segment_size)?; + let partitions_count = streams + .topic_partitions_count(&update_topic.stream_id, &update_topic.topic_id) + .unwrap_or(0); + warn_unenforceable_topic_size( + max_topic_size, + segment_size, + shard.bus_max_message_size(), + u32::try_from(partitions_count).unwrap_or(u32::MAX), + ); + Ok(()) + }), + Operation::UpdateStream => UpdateStreamRequest::decode_from(request_body(request)) + .map_err(|_| IggyError::InvalidCommand) + .and_then(|update_stream| { + validate_option_keys(&update_stream.options, UPDATABLE_STREAM_OPTION_KEYS) + }), + Operation::UpdateUser => UpdateUserRequest::decode_from(request_body(request)) + .map_err(|_| IggyError::InvalidCommand) + .and_then(|update_user| { + validate_option_keys(&update_user.options, UPDATABLE_USER_OPTION_KEYS) + }), + Operation::CreateStream => CreateStreamRequest::decode_from(request_body(request)) + .map_err(|_| IggyError::InvalidCommand) + .and_then(|create_stream| validate_option_keys(&create_stream.options, &[])), + Operation::CreateUser => CreateUserRequest::decode_from(request_body(request)) + .map_err(|_| IggyError::InvalidCommand) + .and_then(|create_user| validate_option_keys(&create_user.options, &[])), + _ => Ok(()), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn create_topic_bounds_deny_pre_consensus() { + let segment_size = iggy_common::DEFAULT_SEGMENT_SIZE; + assert!(segment_size > 0, "default segment size must be nonzero"); + + assert!( + validate_topic_bounds( + MAX_PARTITIONS_PER_REQUEST, + MaxTopicSize::ServerDefault, + segment_size + ) + .is_ok(), + "the partition cap itself is admissible" + ); + assert!( + matches!( + validate_topic_bounds( + MAX_PARTITIONS_PER_REQUEST + 1, + MaxTopicSize::ServerDefault, + segment_size + ), + Err(IggyError::TooManyPartitions) + ), + "one past the partition cap must deny" + ); + // ServerDefault is numerically 0 yet exempt from the segment-size + // floor: it resolves against server config, matching legacy. + assert!(validate_topic_bounds(1, MaxTopicSize::ServerDefault, segment_size).is_ok()); + assert!(validate_topic_bounds(1, MaxTopicSize::Unlimited, segment_size).is_ok()); + let below_floor = MaxTopicSize::Custom((segment_size - 1).into()); + assert!( + matches!( + validate_topic_bounds(1, below_floor, segment_size), + Err(IggyError::InvalidTopicSize(size, floor)) + if size == below_floor && floor == IggyByteSize::from(segment_size) + ), + "custom size below the segment size must deny with the bounds" + ); + let at_floor = MaxTopicSize::Custom(IggyByteSize::from(segment_size)); + assert!( + validate_topic_bounds(1, at_floor, segment_size).is_ok(), + "a topic exactly one segment large is admissible" + ); + } + + #[test] + fn partitions_count_cap_denies_pre_consensus() { + assert!( + validate_partitions_count(MAX_PARTITIONS_PER_REQUEST).is_ok(), + "the cap itself is admissible" + ); + assert!( + matches!( + validate_partitions_count(MAX_PARTITIONS_PER_REQUEST + 1), + Err(IggyError::TooManyPartitions) + ), + "one past the cap must deny" + ); + // Zero passes the shared cap because a zero-partition TOPIC is legal + // (legacy `create_topic` admits `0..=MAX`). + assert!(validate_partitions_count(0).is_ok()); + } + + #[test] + fn zero_partitions_change_denies_pre_consensus() { + // Adding or removing zero partitions is a no-op that would still burn + // a replicated log entry and force a rebalance. Legacy rejects it with + // `TooManyPartitions` in both handlers, so the code matches. + assert!( + matches!( + validate_partitions_change_count(0), + Err(IggyError::TooManyPartitions) + ), + "adding or removing zero partitions must deny" + ); + assert!(validate_partitions_change_count(1).is_ok()); + assert!(validate_partitions_change_count(MAX_PARTITIONS_PER_REQUEST).is_ok()); + assert!( + matches!( + validate_partitions_change_count(MAX_PARTITIONS_PER_REQUEST + 1), + Err(IggyError::TooManyPartitions) + ), + "the cap still applies" + ); + } +} diff --git a/core/server/src/server_error.rs b/core/server/src/server_error.rs index b9e744755c..517d3be635 100644 --- a/core/server/src/server_error.rs +++ b/core/server/src/server_error.rs @@ -122,6 +122,27 @@ pub enum ServerError { poll: std::time::Duration, drain: std::time::Duration, }, + #[error("system.sharding.shutdown_join_timeout must be <= {max:?}; got {value:?}")] + InvalidShutdownJoinTimeout { + value: std::time::Duration, + max: std::time::Duration, + }, + #[error( + "system.sharding.shutdown_join_timeout ({join:?}) must be >= \ + shutdown_drain_timeout ({drain:?})" + )] + ShutdownJoinBelowDrain { + join: std::time::Duration, + drain: std::time::Duration, + }, + #[error( + "system.sharding.reconcile_periodic_interval must be in (0, {max:?}]; got {value:?}. \ + Note that \"0\", \"none\", \"unlimited\", and \"disabled\" all parse to zero" + )] + InvalidReconcilePeriodicInterval { + value: std::time::Duration, + max: std::time::Duration, + }, #[error("failed to serialize current server config")] CurrentConfigSerialize(#[source] toml::ser::Error), #[error("failed to write current server config at {path}")] diff --git a/core/server/src/session_manager.rs b/core/server/src/session_manager.rs index d09bd4c079..de7bfead33 100644 --- a/core/server/src/session_manager.rs +++ b/core/server/src/session_manager.rs @@ -26,13 +26,36 @@ //! transport connection and the consensus-level `(client_id, session)` pair. use crate::cluster_meta::ClusterRoster; +use ahash::AHashMap; use message_bus::installer::conn_info::ClientTransportKind; use shard::ConnectedClientInfo; -use std::collections::HashMap; use std::net::SocketAddr; use std::rc::Rc; use std::time::{Duration, Instant}; +/// What the request funnel resolves from one `connections` lookup per frame. +/// +/// The bound consensus session, the acting user, the transport peer address +/// and the read-your-writes floor. `Default` (everything absent, no address, +/// floor `0`) stands for a connection neither this map nor the bus knows. +#[derive(Debug, Clone, Copy, Default)] +pub(crate) struct ConnectionContext { + /// `(client_id, session)` once register committed, `None` before. + pub bound: Option<(u128, u64)>, + /// Acting user from `login`, `None` while still `Connected`. + pub user_id: Option, + /// Peer address recorded by [`SessionManager::ensure_connection`]; the + /// non-replicated reads pick the advertised address from it, and + /// `None` degrades to the catch-all address. + pub address: Option, + /// Read-your-writes floor: the metadata commit this connection's own + /// writes have reached, which its reads must not be served below. + /// + /// A connection this map does not know reads as `0` -- it was promised + /// nothing, so its reads wait for nothing. + pub metadata_watermark: u64, +} + /// Connection lifecycle states. /// /// ```text @@ -107,10 +130,14 @@ pub struct Connection { /// per consensus session). If a client reconnects with the same `client_id`, /// the old connection must be evicted first. pub struct SessionManager { - connections: HashMap, + /// `ahash` over `std`: connection and client ids are server-minted, so + /// there is no `HashDoS` surface, and both maps sit on the per-frame path. + /// Neither is order-sensitive (`iter_clients` and `collect_stale` both + /// consume the whole map). + connections: AHashMap, /// Reverse index: `client_id` → `connection_id` for fast lookup when /// a consensus reply arrives and needs routing to the right connection. - client_to_connection: HashMap, + client_to_connection: AHashMap, /// This shard's copy of the configured cluster roster, served by the /// `GetClusterMetadata` read. Lives here because it is the /// per-shard context already threaded to the non-replicated read path; @@ -122,8 +149,8 @@ impl SessionManager { #[must_use] pub fn new() -> Self { Self { - connections: HashMap::new(), - client_to_connection: HashMap::new(), + connections: AHashMap::new(), + client_to_connection: AHashMap::new(), cluster_roster: Rc::new(ClusterRoster::disabled()), } } @@ -157,12 +184,31 @@ impl SessionManager { }); } - /// Record a liveness heartbeat (`ping`) for a connection, resetting its - /// staleness clock. No-op for an unknown connection. - pub fn record_heartbeat(&mut self, connection_id: u128) { - if let Some(conn) = self.connections.get_mut(&connection_id) { - conn.last_heartbeat = Instant::now(); - } + /// The request funnel's per-frame view of a connection: stamp the + /// liveness clock and read back everything the dispatch arms resolve + /// from it, in ONE map lookup. + /// + /// `None` means the connection is not registered yet, which happens + /// only on a transport's first frame; the caller installs it from the + /// bus metadata ([`Self::ensure_connection`]) and asks again. + pub(crate) fn touch_connection(&mut self, connection_id: u128) -> Option { + let conn = self.connections.get_mut(&connection_id)?; + conn.last_heartbeat = Instant::now(); + let (bound, user_id) = match conn.state { + ConnectionState::Bound { + user_id, + client_id, + session, + } => (Some((client_id, session)), Some(user_id)), + ConnectionState::Authenticated { user_id } => (None, Some(user_id)), + ConnectionState::Connected => (None, None), + }; + Some(ConnectionContext { + bound, + user_id, + address: Some(conn.address), + metadata_watermark: conn.metadata_watermark, + }) } /// Connection ids whose last heartbeat is older than `max_age` -- the @@ -331,7 +377,11 @@ impl SessionManager { /// Look up the consensus session for a connection. /// - /// Returns `(client_id, session)` if the connection is `Bound`, `None` otherwise. + /// Returns `(client_id, session)` if the connection is `Bound`, `None` + /// otherwise. The request funnel reads it through the crate-private + /// `touch_connection` instead, which resolves it in the same lookup as + /// the heartbeat; this is for the session-op paths that only need the + /// binding. #[must_use] pub fn get_session(&self, connection_id: u128) -> Option<(u128, u64)> { let conn = self.connections.get(&connection_id)?; @@ -343,38 +393,6 @@ impl SessionManager { } } - /// The transport-level peer address a connection arrived from, recorded by - /// [`Self::ensure_connection`] for every transport. The non-replicated - /// read path uses it to pick the advertised address a client is told - /// about; `None` (unknown connection) degrades to the catch-all address. - #[must_use] - pub fn connection_address(&self, connection_id: u128) -> Option { - self.connections - .get(&connection_id) - .map(|conn| conn.address) - } - - /// Acting user, transport peer address and metadata watermark for a - /// connection, in one map lookup: the non-replicated dispatch path needs - /// all three per request, and the separate accessors would walk the - /// connection map (and take the shared borrow) once each. - /// - /// An unknown connection reads as a watermark of `0`: it was promised - /// nothing, so its reads wait for nothing. - #[must_use] - pub fn read_context(&self, connection_id: u128) -> (Option, Option, u64) { - let Some(conn) = self.connections.get(&connection_id) else { - return (None, None, 0); - }; - let user_id = match conn.state { - ConnectionState::Authenticated { user_id } | ConnectionState::Bound { user_id, .. } => { - Some(user_id) - } - ConnectionState::Connected => None, - }; - (user_id, Some(conn.address), conn.metadata_watermark) - } - /// Look up the authenticated user id for a connection. #[must_use] pub fn get_user_id(&self, connection_id: u128) -> Option { diff --git a/core/simulator/src/replica.rs b/core/simulator/src/replica.rs index 5132085fd9..28816aa465 100644 --- a/core/simulator/src/replica.rs +++ b/core/simulator/src/replica.rs @@ -362,7 +362,7 @@ pub fn new_shard( // every op above the floor must mutate the table. A frontier only fences a // LIVE table a state transfer just replaced. apply_committed_prepare( - &metadata.mux_stm, + &*metadata.mux_stm, &metadata.client_table, true, |_| {}, From bc9ef0c3f938b82ac0fe3c3a89f64d0242660edd Mon Sep 17 00:00:00 2001 From: Maxim Levkov Date: Fri, 4 Sep 2026 13:02:41 -0700 Subject: [PATCH 062/182] fix(connectors): warn when the runtime API is exposed without a key (#3804) --- core/connectors/runtime/README.md | 86 ++- core/connectors/runtime/config.toml | 9 +- .../runtime/example_config/config.toml | 9 +- core/connectors/runtime/src/api/config.rs | 111 +++- core/connectors/runtime/src/api/mod.rs | 573 +++++++++++++++++- 5 files changed, 780 insertions(+), 8 deletions(-) diff --git a/core/connectors/runtime/README.md b/core/connectors/runtime/README.md index 590d0a8f87..23803704d0 100644 --- a/core/connectors/runtime/README.md +++ b/core/connectors/runtime/README.md @@ -197,10 +197,17 @@ Connector runtime has an optional HTTP API that can be enabled by setting the `e ```toml [http] # Optional HTTP API configuration enabled = true +# Loopback on purpose: the configuration endpoints return plugin credentials in +# plaintext and also accept writes. Set api_key in the same edit if you move +# this off loopback, and http.tls unless cleartext is acceptable. address = "127.0.0.1:8081" -api_key = "" # Optional API key for authentication to be passed as `api-key` header +api_key = "" # Optional API key for authentication to be passed as `api-key` header; empty disables authentication [http.cors] # Optional CORS configuration for HTTP API +# Enabling this with the shipped allowed_origins = ["*"] lets any page the +# operator visits read these endpoints cross-origin, whatever address is bound. +# Pin the origins you own, or set api_key, before turning it on. +# "*" is honored only as the first entry; anywhere else it panics at startup. enabled = false allowed_methods = ["GET", "POST", "PUT", "DELETE"] allowed_origins = ["*"] @@ -219,29 +226,104 @@ cert_file = "core/certs/iggy_cert.pem" key_file = "core/certs/iggy_key.pem" ``` +> [!IMPORTANT] +> **Treat this API as privileged. It reads and it writes.** +> +> The configuration endpoints return plugin configuration exactly as stored, +> credentials included - a database connection string, an S3 secret key, a +> webhook signing secret. Nothing redacts them on the way out. The runtime +> masks the `api-key` in its own logs and redacts the state-store headers when +> it serializes them, but no such path exists for plugin configuration. +> +> The exposure is not limited to disclosure. Publishing a configuration with +> `POST /{sinks,sources}/{key}/configs` and then calling `POST .../restart` is +> enough to repoint a connector at a destination of the caller's choosing, +> because `restart` re-reads the stored configuration and starts the connector +> from it - on the local provider that is whatever version was published last, +> with no activation step in between. The runtime then forwards the operator's +> topic data using its own Iggy credentials. `PUT .../configs/active` and +> `DELETE .../configs` sit behind the same key. +> +> A rewritten plugin `path` is not loaded by the restart. `start_connector` +> reuses the container `dlopen`ed at boot and only re-runs the plugin's init +> with the new configuration, so a hostile path sits in the stored config until +> the next time the runtime process starts, which makes it deferred code +> execution rather than immediate. +> +> `api_key` is empty by default, which means authentication is **off** by +> default. Only `/` and `/health` are exempt once it is set, so everything above +> sits behind that one empty string, and the loopback default `address` is what +> confines it to local processes. +> +> Three edits take that containment away: +> +> - **Moving `address` off loopback.** Set `api_key` in the same edit. The +> runtime warns at startup when the address resolves beyond loopback with no +> key configured, but nothing prevents it. +> - **Enabling `[http.cors]` with the shipped `allowed_origins = ["*"]`.** That +> becomes `AllowOrigin::any()`, and the CORS layer wraps *outside* +> authentication, so tower-http answers the preflight before `resolve_api_key` +> runs. A browser is a local process, so with a wildcard origin and no key any +> page the operator visits gets a green preflight for a configuration `POST` +> and can then read *and rewrite* configuration cross-origin - the whole +> publish-then-restart repoint above, not just disclosure. The rewrite half +> needs the method on `allowed_methods` and `content-type` on +> `allowed_headers`, both of which the shipped block grants. +> +> Setting `api_key` closes it, since an attacker's page cannot supply the +> header. Pinning `allowed_origins` to origins the operator owns closes the +> cross-origin *read*, and no startup warning fires for that case. One fires +> for `null`, which is narrower than `*` but not owned by anyone: a browser +> sends it from a sandboxed iframe, a `data:` URL and a `file://` page, all of +> which an attacker can produce, so pinning it buys nothing. No origin setting +> closes the CORS-simple route below; `api_key` does. +> - **Leaving `http.tls.enabled = false`.** It ships disabled, so the `api-key` +> header and the credential-bearing responses both travel in cleartext. Enable +> TLS alongside `api_key` whenever this API leaves loopback. The runtime warns +> at startup whenever the address resolves beyond loopback with TLS off, and +> keeps warning after `api_key` is set, because the key crosses in the clear +> too. Terminating TLS at an ingress or a service mesh is a valid answer to +> that warning; the runtime cannot see it, so the line stays. +> +> And one that needs no edit at all. `POST .../restart` carries no body and no +> content type, which makes it a CORS-simple request: a page can issue it with +> `mode: 'no-cors'` and the browser sends it whatever `[http.cors]` says, +> because CORS gates reading a response rather than issuing a request. So on the +> shipped keyless default, any page the operator visits can restart any +> connector, and no startup warning covers it because it is true of the defaults +> rather than of an edit. Chrome's private network access blocks the +> public-origin case; Firefox and Safari do not, a page served from a local +> origin bypasses it everywhere, and setting `allow_private_network = true` +> hands back the case Chrome would otherwise block. `api_key` is what closes +> this one. + Currently, it does expose the following endpoints: - `GET /`: welcome message. - `GET /health`: health status of the runtime. - `GET /stats`: runtime statistics including process info, memory/CPU usage, and connector status. -- `GET /metrics`: Prometheus-formatted metrics (when `http.metrics.enabled` is `true`). +- `GET {http.metrics.endpoint}` (default `/metrics`): Prometheus-formatted metrics, when `http.metrics.enabled` is `true`. - `GET /sinks`: list of sinks. - `GET /sinks/{key}`: sink details. - `GET /sinks/{key}/configs`: list of configuration versions for the sink. - `POST /sinks/{key}/configs`: add a new configuration version for the sink. +- `DELETE /sinks/{key}/configs`: delete one configuration version for the sink - the `version` query parameter, or on the local provider the active version when it is omitted. - `GET /sinks/{key}/configs/{version}`: configuration details for a specific version. - `GET /sinks/{key}/configs/active`: active configuration details. - `PUT /sinks/{key}/configs/active`: activate a specific configuration version for the sink. - `GET /sinks/{key}/configs/plugin`: sink plugin config, including the optional `format` query parameter to specify the config format. +- `POST /sinks/{key}/restart`: stop the sink and start it again from its highest stored configuration version, which on the local provider is not necessarily the active one ([#3848](https://github.com/apache/iggy/issues/3848)). - `GET /sinks/{key}/transforms`: sink transforms to be applied to the fields. - `GET /sources`: list of sources. - `GET /sources/{key}`: source details. - `GET /sources/{key}/configs`: list of configuration versions for the source. - `POST /sources/{key}/configs`: add a new configuration version for the source. +- `DELETE /sources/{key}/configs`: delete one configuration version for the source - the `version` query parameter, or on the local provider the active version when it is omitted. - `GET /sources/{key}/configs/{version}`: configuration details for a specific version. - `GET /sources/{key}/configs/active`: active configuration details. - `PUT /sources/{key}/configs/active`: activate a specific configuration version for the source. - `GET /sources/{key}/configs/plugin`: source plugin config, including the optional `format` query parameter to specify the config format. +- `POST /sources/{key}/restart`: stop the source and start it again from its highest stored configuration version, which on the local provider is not necessarily the active one ([#3848](https://github.com/apache/iggy/issues/3848)). - `GET /sources/{key}/transforms`: source transforms to be applied to the fields. ## Telemetry diff --git a/core/connectors/runtime/config.toml b/core/connectors/runtime/config.toml index d52909122b..9a3e3befbc 100644 --- a/core/connectors/runtime/config.toml +++ b/core/connectors/runtime/config.toml @@ -17,10 +17,17 @@ [http] # Optional HTTP API configuration enabled = true +# Loopback on purpose: the configuration endpoints return plugin credentials in +# plaintext and also accept writes. Set api_key in the same edit if you move +# this off loopback, and http.tls unless cleartext is acceptable. address = "127.0.0.1:8081" -api_key = "" # Optional API key for authentication to be passed as `api-key` header +api_key = "" # Optional API key for authentication to be passed as `api-key` header; empty disables authentication [http.cors] # Optional CORS configuration for HTTP API +# Enabling this with the shipped allowed_origins = ["*"] lets any page the +# operator visits read these endpoints cross-origin, whatever address is bound. +# Pin the origins you own, or set api_key, before turning it on. +# "*" is honored only as the first entry; anywhere else it panics at startup. enabled = false allowed_methods = ["GET", "POST", "PUT", "DELETE"] allowed_origins = ["*"] diff --git a/core/connectors/runtime/example_config/config.toml b/core/connectors/runtime/example_config/config.toml index 29afad4ce6..a6690b801d 100644 --- a/core/connectors/runtime/example_config/config.toml +++ b/core/connectors/runtime/example_config/config.toml @@ -17,10 +17,17 @@ [http] # Optional HTTP API configuration enabled = true +# Loopback on purpose: the configuration endpoints return plugin credentials in +# plaintext and also accept writes. Set api_key in the same edit if you move +# this off loopback, and http.tls unless cleartext is acceptable. address = "127.0.0.1:8081" -api_key = "" # Optional API key for authentication to be passed as `api-key` header +api_key = "" # Optional API key for authentication to be passed as `api-key` header; empty disables authentication [http.cors] # Optional CORS configuration for HTTP API +# Enabling this with the shipped allowed_origins = ["*"] lets any page the +# operator visits read these endpoints cross-origin, whatever address is bound. +# Pin the origins you own, or set api_key, before turning it on. +# "*" is honored only as the first entry; anywhere else it panics at startup. enabled = false allowed_methods = ["GET", "POST", "PUT", "DELETE"] allowed_origins = ["*"] diff --git a/core/connectors/runtime/src/api/config.rs b/core/connectors/runtime/src/api/config.rs index 1221bd3e8f..78f5125a36 100644 --- a/core/connectors/runtime/src/api/config.rs +++ b/core/connectors/runtime/src/api/config.rs @@ -73,6 +73,44 @@ pub struct HttpCorsConfig { pub allow_private_network: bool, } +impl HttpCorsConfig { + /// Whether `configure_cors` will turn this into `AllowOrigin::any()`. + /// + /// Only the first entry decides, because that is what the mapping below + /// reads. A `"*"` in any later position never reaches a served request: + /// `AllowOrigin::list` panics on a wildcard, so such a config takes the + /// process down at startup rather than allowing anything. + /// + /// Compared untrimmed, because `configure_cors` compares untrimmed: `" *"` + /// becomes a list entry no `Origin` matches, so it allows nothing and there + /// is nothing to warn about. `core/server/src/http.rs` trims before the same + /// comparison; porting that here means trimming in both places at once, + /// since trimming only this one would warn about a closed config and + /// trimming only the mapping would open one silently. + pub fn allows_any_origin(&self) -> bool { + self.allowed_origins + .first() + .is_some_and(|origin| origin == "*") + } + + /// Whether the resulting policy admits an origin nobody owns, assuming one + /// gets built: a wildcard past the first position panics `configure_cors`, + /// so that shape warns and then dies. + /// + /// `*` is one. `null` is the other: a browser sends it from a sandboxed + /// iframe, a `data:` URL and a `file://` page, so listing it hands the + /// cross-origin read to whoever gets the operator to open a page. + /// `AllowOrigin::list` echoes any listed value back on a match, so a + /// `null` anywhere counts, not only first. Untrimmed for the reason above. + /// + /// Separate from `allows_any_origin` because `configure_cors` has to keep + /// mapping `["null"]` to a one-entry list rather than widening it, and + /// defined in terms of it so the two cannot disagree about `*`. + pub fn allows_unowned_origin(&self) -> bool { + self.allows_any_origin() || self.allowed_origins.iter().any(|origin| origin == "null") + } +} + #[derive(Debug, Default, Deserialize, Serialize, Clone, ConfigEnv)] pub struct HttpTlsConfig { pub enabled: bool, @@ -111,7 +149,7 @@ pub fn map_connector_config( pub fn configure_cors(config: &HttpCorsConfig) -> CorsLayer { let allowed_origins = match &config.allowed_origins { origins if origins.is_empty() => AllowOrigin::default(), - origins if origins.first().unwrap() == "*" => AllowOrigin::any(), + _ if config.allows_any_origin() => AllowOrigin::any(), origins => AllowOrigin::list(origins.iter().map(|s| s.parse().unwrap())), }; @@ -223,3 +261,74 @@ impl std::fmt::Display for HttpCorsConfig { ) } } + +#[cfg(test)] +mod tests { + use super::*; + + fn cors(allowed_origins: &[&str]) -> HttpCorsConfig { + HttpCorsConfig { + allowed_origins: allowed_origins.iter().map(|s| (*s).to_owned()).collect(), + ..HttpCorsConfig::default() + } + } + + #[test] + fn given_a_leading_wildcard_when_classified_should_allow_any_origin() { + assert!(cors(&["*"]).allows_any_origin()); + assert!(cors(&["*"]).allows_unowned_origin()); + } + + #[test] + fn given_pinned_or_absent_origins_when_classified_should_allow_none() { + assert!( + !cors(&[]).allows_any_origin(), + "an empty list emits no header" + ); + assert!(!cors(&[]).allows_unowned_origin()); + assert!(!cors(&["https://console.example"]).allows_any_origin()); + assert!(!cors(&["https://console.example"]).allows_unowned_origin()); + assert!( + !cors(&["https://console.example", "*"]).allows_any_origin(), + "only the first entry reaches `AllowOrigin::any`; a later `*` makes \ + `AllowOrigin::list` panic, so that config never serves a request" + ); + } + + #[test] + fn given_the_classified_origin_shapes_when_built_should_produce_a_layer() { + // The predicates above describe what `configure_cors` does with each + // shape, so a shape that cannot build would make the warning the last + // line an operator sees before the process dies. + for origins in [ + &["*"][..], + &[][..], + &["https://console.example"][..], + &["null"][..], + ] { + let _layer = configure_cors(&cors(origins)); + } + } + + #[test] + #[should_panic(expected = "Wildcard origin")] + fn given_a_trailing_wildcard_when_built_should_panic_rather_than_allow_nothing() { + let _layer = configure_cors(&cors(&["https://console.example", "*"])); + } + + #[test] + fn given_a_null_origin_when_classified_should_report_an_unowned_one() { + assert!( + cors(&["null"]).allows_unowned_origin(), + "a sandboxed iframe, a data: URL and a file:// page all send Origin: null" + ); + assert!( + cors(&["https://console.example", "null"]).allows_unowned_origin(), + "`AllowOrigin::list` echoes any listed value, so position does not matter" + ); + assert!( + !cors(&["null"]).allows_any_origin(), + "`configure_cors` still has to build a one-entry list, not widen to any()" + ); + } +} diff --git a/core/connectors/runtime/src/api/mod.rs b/core/connectors/runtime/src/api/mod.rs index 38cd8f6ed1..62950fa6e6 100644 --- a/core/connectors/runtime/src/api/mod.rs +++ b/core/connectors/runtime/src/api/mod.rs @@ -22,9 +22,14 @@ use axum::{Json, Router, extract::State, middleware, routing::get}; use axum_server::tls_rustls::RustlsConfig; use config::{HttpConfig, configure_cors}; use iggy_connector_sdk::api::ConnectorRuntimeStats; -use std::{net::SocketAddr, path::PathBuf, sync::Arc}; -use tokio::spawn; -use tracing::{error, info}; +use secrecy::ExposeSecret; +use std::{ + net::{IpAddr, SocketAddr}, + path::PathBuf, + sync::Arc, +}; +use tokio::{net::lookup_host, spawn}; +use tracing::{error, info, warn}; mod auth; pub mod config; @@ -34,6 +39,12 @@ mod sink; mod source; const NAME: &str = env!("CARGO_PKG_NAME"); +/// Where the full exposure surface is written down, so the startup warnings +/// can name one thing each instead of restating it. A URL rather than a repo +/// path: the published image carries the binary and its licences, nothing else, +/// so the reader most likely to see these lines has no source tree. +const EXPOSURE_DOC: &str = + "https://github.com/apache/iggy/blob/master/core/connectors/runtime/README.md"; pub async fn init(config: &HttpConfig, context: Arc) { if !config.enabled { @@ -41,6 +52,8 @@ pub async fn init(config: &HttpConfig, context: Arc) { return; } + warn_on_weak_containment(config).await; + let mut system_router = Router::new().route("/stats", get(get_stats)); if config.metrics.enabled { @@ -121,6 +134,83 @@ pub async fn init(config: &HttpConfig, context: Arc) { }); } +/// Warns once for each configuration change that leaves this API without a +/// compensating control: no key beyond loopback, no key with an unowned CORS +/// origin, no TLS beyond loopback. +/// +/// Separate warnings rather than one, because the three compose independently +/// and an operator who closes one has not necessarily closed the others. Each +/// carries a pointer instead of its own remediation: the README holds the full +/// exposure surface, including the parts no startup warning can help with +/// because they are already true of the shipped defaults. +async fn warn_on_weak_containment(config: &HttpConfig) { + let unauthenticated = config.api_key.expose_secret().is_empty(); + let beyond_loopback = resolves_beyond_loopback(&config.address).await; + + if unauthenticated && beyond_loopback { + warn!( + "{NAME} HTTP API on {} has no api_key: anyone who reaches it can read and rewrite every connector configuration. {EXPOSURE_DOC}", + config.address + ); + } + + // Loopback does not contain this one. A browser is a local process and the + // CORS layer wraps outside authentication, so the shipped + // `allowed_origins = ["*"]` lets any page the operator visits read these + // endpoints cross-origin. Pinning origins the operator owns closes that + // read, and an empty list emits no header at all, so neither is warned + // about. `null` is not such an origin, so a list holding it still is. No + // origin setting closes `POST .../restart`, which is CORS-simple and + // reaches the handler regardless; that one is true of the defaults, so the + // README carries it rather than a warning. + if unauthenticated && config.cors.enabled && config.cors.allows_unowned_origin() { + warn!( + "{NAME} HTTP API has http.cors open to * or null with no api_key: any page the operator visits can read and rewrite its config. {EXPOSURE_DOC}" + ); + } + + if beyond_loopback && !config.tls.enabled { + warn!( + "{NAME} HTTP API on {} has http.tls disabled: the api-key header and the credentials in its responses cross in cleartext. {EXPOSURE_DOC}", + config.address + ); + } +} + +/// Whether `address` resolves to anything outside loopback. +/// +/// Resolves rather than parses because `address` is a free-form `String` that +/// takes a hostname, as `[iggy] address` does in the same file. Not because the +/// default needs it: the embedded `config.toml` is the first figment layer, so +/// the effective default is `127.0.0.1:8081` and would parse. An address that +/// cannot resolve counts as exposed, since it is about to fail the bind anyway +/// and staying quiet about one we could not classify is the wrong direction to +/// be wrong in. +/// +/// Classification only. Do not bind what this resolves: `TcpListener::bind` +/// walks every resolved address and takes the first that works, so collapsing +/// to one would drop the `localhost` -> `[::1, 127.0.0.1]` fallback on hosts +/// with IPv6 disabled. The cost is resolving twice at startup, which is the +/// trade for keeping that fallback. +async fn resolves_beyond_loopback(address: &str) -> bool { + let Ok(resolved) = lookup_host(address).await else { + return true; + }; + let addresses: Vec = resolved.collect(); + // Empty is reported as exposed rather than confined: `all` over nothing is + // vacuously true, which would quietly invert the policy above. + addresses.is_empty() + || !addresses.iter().all(|address| match address.ip() { + // `Ipv6Addr::is_loopback` matches only `::1`, so an IPv4-mapped + // `::ffff:127.0.0.1` binds to 127.0.0.1 and would still read as + // exposed. + IpAddr::V6(v6) => v6 + .to_ipv4_mapped() + .map_or(v6.is_loopback(), |v4| v4.is_loopback()), + ip => ip.is_loopback(), + }) +} + async fn get_metrics(State(context): State>) -> String { context.metrics.get_formatted_output() } @@ -128,3 +218,480 @@ async fn get_metrics(State(context): State>) -> String { async fn get_stats(State(context): State>) -> Json { Json(stats::get_runtime_stats(&context).await) } + +#[cfg(test)] +mod tests { + use super::*; + use crate::configs::connectors::create_connectors_config_provider; + use crate::configs::runtime::{ConnectorsConfig, LocalConnectorsConfig}; + use crate::manager::sink::SinkManager; + use crate::manager::source::SourceManager; + use crate::metrics::Metrics; + use crate::state::FileStateFactory; + use crate::stream::IggyClients; + use iggy::prelude::IggyClient; + use iggy_common::IggyTimestamp; + use secrecy::SecretString; + use std::fmt::Write as _; + use std::sync::Mutex; + use tempfile::TempDir; + use tracing::Level; + use tracing::field::{Field, Visit}; + use tracing::subscriber::DefaultGuard; + use tracing_subscriber::Layer as _; + use tracing_subscriber::filter::LevelFilter; + use tracing_subscriber::layer::{Context as LayerContext, SubscriberExt}; + + /// Reserved for documentation by RFC 5737, so the bind fails and the tests + /// that use it reach the warning without listening anywhere. Hosts running + /// `net.ipv4.ip_nonlocal_bind=1`, which keepalived and haproxy boxes set, + /// bind it regardless; `routable_address_binds` excuses those. + const UNASSIGNABLE_ROUTABLE_ADDRESS: &str = "192.0.2.1:8081"; + + /// The pointer has to be fetchable by whoever reads the warning, and the + /// published image ships no source tree. Asserting the const against itself + /// elsewhere cannot catch a value that stops being a URL. + #[test] + fn given_the_exposure_pointer_when_read_should_be_a_fetchable_url() { + assert!( + EXPOSURE_DOC.starts_with("https://"), + "a repo-relative path does not exist for an operator running the image" + ); + } + const EPHEMERAL_LOOPBACK_ADDRESS: &str = "127.0.0.1:0"; + + type Captured = Arc>>; + + fn config(address: &str, api_key: &str) -> HttpConfig { + HttpConfig { + address: address.to_owned(), + api_key: SecretString::from(api_key.to_owned()), + ..HttpConfig::default() + } + } + + /// Captures events for the current thread only, for as long as the guard + /// lives. Not a global subscriber: that slot is process-wide, and a shared + /// buffer would leave negative assertions hostage to the rest of the binary. + /// + /// `#[tokio::test]` builds a current-thread runtime, so a task spawned by + /// the test body sees this subscriber. Under a multi-thread flavour the + /// capture would come back empty and these tests would fail, not pass. + /// + /// Filtered rather than checking the level in `on_event`: a layer with no + /// filter reports no `max_level_hint`, which pushes the global max level to + /// TRACE and stops every callsite in the binary short-circuiting. + fn capture_events() -> (DefaultGuard, Captured) { + let captured: Captured = Arc::new(Mutex::new(Vec::new())); + let layer = CaptureEvents { + captured: Arc::clone(&captured), + } + .with_filter(LevelFilter::INFO); + let guard = tracing::subscriber::set_default(tracing_subscriber::registry().with(layer)); + (guard, captured) + } + + fn warnings(captured: &Captured) -> Vec { + captured + .lock() + .expect("the capture mutex is only held to push a line") + .iter() + .filter(|(level, _)| *level == Level::WARN) + .map(|(_, message)| message.clone()) + .collect() + } + + /// Whether `init` got as far as serving. The positive control for tests + /// whose real assertion is that something was not warned about. + fn started_serving(captured: &Captured) -> bool { + logged_info(captured, "Started") + } + + fn logged_info(captured: &Captured, needle: &str) -> bool { + captured + .lock() + .expect("the capture mutex is only held to push a line") + .iter() + .any(|(level, message)| *level == Level::INFO && message.contains(needle)) + } + + /// Whether this host binds an address RFC 5737 reserves for documentation. + /// + /// `net.ipv4.ip_nonlocal_bind=1`, which keepalived and haproxy boxes set, + /// makes it succeed. The two tests that read a failed bind as their + /// ordering control stand those two assertions down there rather than going + /// red over a sysctl. They do not skip: a skipped test reports `ok` with no + /// signal, and the warning itself is required on every host. + fn routable_address_binds() -> bool { + std::net::TcpListener::bind(UNASSIGNABLE_ROUTABLE_ADDRESS).is_ok() + } + + struct CaptureEvents { + captured: Captured, + } + + impl tracing_subscriber::Layer for CaptureEvents { + fn on_event(&self, event: &tracing::Event<'_>, _context: LayerContext<'_, S>) { + let mut recorded = Recorded(String::new()); + event.record(&mut recorded); + self.captured + .lock() + .expect("the capture mutex is only held to push a line") + .push((*event.metadata().level(), recorded.0)); + } + } + + /// Unconditional on purpose: singling out the `message` field would add a + /// branch whose other side nothing here takes. + struct Recorded(String); + + impl Visit for Recorded { + fn record_debug(&mut self, _field: &Field, value: &dyn std::fmt::Debug) { + let _ = write!(self.0, "{value:?} "); + } + } + + /// The cheapest context `init` will accept. Nothing here reaches Iggy. + /// + /// `api_key` is a parameter because the guard reads `config.api_key` while + /// the middleware enforces `context.api_key`; a test that set only one + /// would exercise a state the runtime cannot reach. + async fn context(api_key: &str) -> (Arc, TempDir) { + let directory = tempfile::tempdir().expect("a temp dir must be available"); + let config_provider = + create_connectors_config_provider(&ConnectorsConfig::Local(LocalConnectorsConfig { + config_dir: directory.path().display().to_string(), + })) + .await + .expect("an empty config dir must initialize with no connectors"); + + let context = RuntimeContext { + sinks: SinkManager::new(vec![]), + sources: SourceManager::new(vec![]), + api_key: SecretString::from(api_key.to_owned()), + config_provider: Arc::from(config_provider), + metrics: Arc::new(Metrics::init()), + start_time: IggyTimestamp::now(), + iggy_clients: Arc::new(IggyClients { + producer: IggyClient::default(), + consumer: IggyClient::default(), + }), + state_factory: Arc::new(FileStateFactory::new( + directory.path().display().to_string(), + )), + }; + (Arc::new(context), directory) + } + + #[tokio::test] + async fn given_loopback_addresses_when_classified_should_report_contained() { + assert!(!resolves_beyond_loopback("127.0.0.1:8081").await); + assert!(!resolves_beyond_loopback("[::1]:8081").await); + assert!( + !resolves_beyond_loopback("localhost:8081").await, + "`address` accepts a hostname, and parsing alone would misjudge one" + ); + assert!( + !resolves_beyond_loopback("[::ffff:127.0.0.1]:8081").await, + "an IPv4-mapped address binds to 127.0.0.1, and `Ipv6Addr::is_loopback` \ + alone matches only `::1`" + ); + } + + #[tokio::test] + async fn given_routable_or_unresolvable_addresses_when_classified_should_report_exposed() { + assert!( + resolves_beyond_loopback("0.0.0.0:8081").await, + "binding every interface to reach the API from outside a container \ + is the case this exists to catch" + ); + assert!(resolves_beyond_loopback("192.0.2.10:8081").await); + // About to fail the bind regardless, so staying quiet about an address + // we cannot classify is the wrong direction to be wrong in. + assert!(resolves_beyond_loopback("not a valid address").await); + } + + #[tokio::test] + async fn given_no_key_and_a_routable_address_when_initialized_should_warn_before_binding() { + let bind_is_refused = !routable_address_binds(); + // Fixture first: `create_connectors_config_provider` logs, and anything + // it warns about later would otherwise land in this buffer. + let (context, _directory) = context("").await; + let (_capture, captured) = capture_events(); + let config = config(UNASSIGNABLE_ROUTABLE_ADDRESS, ""); + + // `init` panics when the bind fails, which is what makes this the + // ordering test: the warning has to already be out by then, or an + // operator whose bind fails never learns the API was unauthenticated. + let bind_failed = tokio::spawn(async move { init(&config, context).await }) + .await + .is_err(); + + if bind_is_refused { + assert!( + bind_failed, + "a documentation-range address must not bind here" + ); + assert!( + !started_serving(&captured), + "the bind must not have completed, or this proves nothing about ordering" + ); + } + // Keyless and untrusting on a routable address, so the address warning + // and the cleartext one both fire. Asserted with `all` and a count + // rather than `any`: either message could stop naming the address and + // the other would still satisfy an `any`. + let warnings = warnings(&captured); + assert_eq!( + warnings.len(), + 2, + "init must consult the guard for both the missing key and the missing \ + TLS: {warnings:?}" + ); + assert!( + warnings + .iter() + .all(|warning| warning.contains(UNASSIGNABLE_ROUTABLE_ADDRESS)), + "each names the address it is exposing: {warnings:?}" + ); + assert!( + warnings + .iter() + .all(|warning| warning.contains(EXPOSURE_DOC)), + "each carries the pointer instead of its own remediation: {warnings:?}" + ); + assert!( + warnings + .iter() + .any(|warning| warning.contains("read and rewrite")), + "the exposure is not read-only, and saying so is the point: {warnings:?}" + ); + } + + #[tokio::test] + async fn given_loopback_address_when_initialized_should_serve_without_warning() { + let (context, _directory) = context("").await; + let (_capture, captured) = capture_events(); + + init(&config(EPHEMERAL_LOOPBACK_ADDRESS, ""), context).await; + + // Positive control first: without it the assertion below passes for any + // reason `init` returns early, including `enabled` ever defaulting to + // false, and the only in-`init` loopback coverage disappears silently. + assert!( + started_serving(&captured), + "init must reach the listener, or the assertion below proves nothing" + ); + // Any warning at all, not one matching this address: matching on the + // address goes vacuous the moment the message is reworded. + assert!( + warnings(&captured).is_empty(), + "the shipped posture is loopback with no key; warning about it would \ + teach operators to ignore the ones that matter: {:?}", + warnings(&captured) + ); + } + + #[tokio::test] + async fn given_wildcard_cors_and_no_key_when_initialized_should_warn_despite_loopback() { + let (context, _directory) = context("").await; + let (_capture, captured) = capture_events(); + let mut config = config(EPHEMERAL_LOOPBACK_ADDRESS, ""); + config.cors.enabled = true; + config.cors.allowed_origins = vec!["*".to_owned()]; + + init(&config, context).await; + + assert!(started_serving(&captured)); + assert!( + warnings(&captured) + .iter() + .any(|warning| warning.contains("http.cors")), + "loopback does not contain CORS: a browser is a local process and the \ + layer wraps outside authentication" + ); + assert!( + warnings(&captured) + .iter() + .all(|warning| warning.contains(EXPOSURE_DOC)), + "each warning carries the pointer instead of its own remediation: {:?}", + warnings(&captured) + ); + } + + #[tokio::test] + async fn given_owned_cors_origins_when_initialized_should_not_warn_about_cors() { + let (context, _directory) = context("").await; + let (_capture, captured) = capture_events(); + let mut config = config(EPHEMERAL_LOOPBACK_ADDRESS, ""); + config.cors.enabled = true; + config.cors.allowed_origins = vec!["https://console.example".to_owned()]; + + init(&config, context).await; + + assert!( + started_serving(&captured), + "init must reach the listener, or the assertion below proves nothing" + ); + assert!( + warnings(&captured).is_empty(), + "an operator who pinned origins they own closed the cross-origin read, \ + and warning at them is what teaches everyone to ignore the rest: {:?}", + warnings(&captured) + ); + } + + #[tokio::test] + async fn given_a_null_cors_origin_when_initialized_should_warn_despite_the_pin() { + let (context, _directory) = context("").await; + let (_capture, captured) = capture_events(); + let mut config = config(EPHEMERAL_LOOPBACK_ADDRESS, ""); + config.cors.enabled = true; + config.cors.allowed_origins = vec!["null".to_owned()]; + + init(&config, context).await; + + assert!(started_serving(&captured)); + assert!( + warnings(&captured) + .iter() + .any(|warning| warning.contains("http.cors")), + "`null` is a pinned origin that nobody owns, so pinning it buys nothing \ + a wildcard would not have given away" + ); + assert!( + warnings(&captured) + .iter() + .all(|warning| warning.contains(EXPOSURE_DOC)), + "each warning carries the pointer instead of its own remediation: {:?}", + warnings(&captured) + ); + } + + #[tokio::test] + async fn given_a_key_but_no_tls_beyond_loopback_when_initialized_should_still_warn() { + let bind_is_refused = !routable_address_binds(); + let (context, _directory) = context("configured").await; + let (_capture, captured) = capture_events(); + let config = config(UNASSIGNABLE_ROUTABLE_ADDRESS, "configured"); + + let bind_failed = tokio::spawn(async move { init(&config, context).await }) + .await + .is_err(); + + if bind_is_refused { + assert!( + bind_failed, + "a documentation-range address must not bind here" + ); + } + let warnings = warnings(&captured); + assert!( + warnings.iter().any(|warning| warning.contains("http.tls")), + "setting a key does not stop the key and the credential-bearing \ + responses crossing the network in cleartext" + ); + assert_eq!( + warnings.len(), + 1, + "the key that was set must not still be reported as missing, and a \ + count cannot go vacuous the way matching on the old wording would: \ + {warnings:?}" + ); + assert!( + warnings + .iter() + .all(|warning| warning.contains(EXPOSURE_DOC)), + "each warning carries the pointer instead of its own remediation: {warnings:?}" + ); + } + + #[tokio::test] + async fn given_wildcard_cors_and_a_key_when_initialized_should_not_warn_about_cors() { + let (context, _directory) = context("configured").await; + let (_capture, captured) = capture_events(); + let mut config = config(EPHEMERAL_LOOPBACK_ADDRESS, "configured"); + config.cors.enabled = true; + config.cors.allowed_origins = vec!["*".to_owned()]; + + init(&config, context).await; + + assert!(started_serving(&captured)); + assert!( + warnings(&captured).is_empty(), + "an attacker's page cannot supply the api-key header, so a key closes \ + the cross-origin path a wildcard opens: {:?}", + warnings(&captured) + ); + } + + #[tokio::test] + async fn given_cors_switched_off_when_initialized_should_not_warn_about_its_origins() { + let (context, _directory) = context("").await; + let (_capture, captured) = capture_events(); + // The shipped block verbatim: a wildcard list that is not switched on. + // Warning here would fire on every stock deployment. + let mut config = config(EPHEMERAL_LOOPBACK_ADDRESS, ""); + config.cors.allowed_origins = vec!["*".to_owned()]; + + init(&config, context).await; + + assert!(started_serving(&captured)); + assert!( + warnings(&captured).is_empty(), + "no CORS layer is installed unless http.cors is enabled, so the origin \ + list allows nothing on its own: {:?}", + warnings(&captured) + ); + } + + #[tokio::test] + async fn given_tls_enabled_beyond_loopback_when_initialized_should_not_warn_about_cleartext() { + let (context, _directory) = context("configured").await; + let (_capture, captured) = capture_events(); + let mut config = config(UNASSIGNABLE_ROUTABLE_ADDRESS, "configured"); + config.tls.enabled = true; + + // The TLS branch loads a certificate this config does not name, so + // `init` panics there. That is after the warning block, which is what + // makes the panic the positive control. + let reached_the_certificate = tokio::spawn(async move { init(&config, context).await }) + .await + .is_err(); + + assert!( + reached_the_certificate, + "init must get past the warning block, or the assertion below proves nothing" + ); + assert!( + warnings(&captured).is_empty(), + "TLS on the listener is exactly what the cleartext warning asks for: {:?}", + warnings(&captured) + ); + } + + #[tokio::test] + async fn given_a_disabled_api_when_initialized_should_warn_about_nothing() { + let (context, _directory) = context("").await; + let (_capture, captured) = capture_events(); + // Routable, keyless and untrusting in every direction, but switched off. + let mut config = config(UNASSIGNABLE_ROUTABLE_ADDRESS, ""); + config.enabled = false; + config.cors.enabled = true; + + init(&config, context).await; + + // Positive control: without it this passes for any reason `init` returns + // early, including one that never reaches the guard at all. + assert!( + logged_info(&captured, "HTTP API is disabled"), + "init must take the disabled path, or the assertion below proves nothing" + ); + assert!( + warnings(&captured).is_empty(), + "an API that is not listening exposes nothing, and warning about one \ + is the false positive that teaches operators to ignore the rest: {:?}", + warnings(&captured) + ); + } +} From 9914f5a812f22a56dabd1dc9b725d956a40e2185 Mon Sep 17 00:00:00 2001 From: Grzegorz Koszyk <112548209+numinnex@users.noreply.github.com> Date: Fri, 4 Sep 2026 22:54:15 +0200 Subject: [PATCH 063/182] fix(shard): drive partition repair and the commit walk from the tick (#4006) --- core/configs/src/server_config/cluster.rs | 45 + core/configs/src/server_config/defaults.rs | 5 + core/consensus/src/observability.rs | 22 + .../tests/cluster/parked_frame_redispatch.rs | 19 +- .../tests/data_integrity/storage_compat.rs | 146 +- core/partitions/Cargo.toml | 8 + core/partitions/src/iggy_partition.rs | 130 +- core/partitions/src/journal.rs | 105 +- core/partitions/src/lib.rs | 7 +- core/partitions/src/types.rs | 28 + core/server/config.toml | 20 +- core/server/src/boot/recovery.rs | 42 +- core/server/src/partition_reconciler.rs | 38 +- core/server/src/shell.rs | 13 + core/shard/src/lib.rs | 1247 ++++++++++++++++- core/shard/src/metrics.rs | 56 +- core/shard/src/router.rs | 2 + core/simulator/Cargo.toml | 4 + core/simulator/src/lib.rs | 858 ++++++++++++ scripts/ci/storage-compat.sh | 17 + 20 files changed, 2702 insertions(+), 110 deletions(-) diff --git a/core/configs/src/server_config/cluster.rs b/core/configs/src/server_config/cluster.rs index 09b9be3d44..929200f087 100644 --- a/core/configs/src/server_config/cluster.rs +++ b/core/configs/src/server_config/cluster.rs @@ -170,6 +170,14 @@ fn default_repair_retry_interval() -> IggyDuration { SERVER_CONFIG.cluster.repair_retry_interval.parse().unwrap() } +fn default_repair_gap_debounce_interval() -> IggyDuration { + SERVER_CONFIG + .cluster + .repair_gap_debounce_interval + .parse() + .unwrap() +} + /// serde fallback for configs written before the field existed; the value /// itself lives in `core/server/config.toml` like every other default. fn default_repair_chunk_max() -> usize { @@ -264,10 +272,28 @@ pub struct ClusterConfig { /// partition repair loops. Sizes the retry threshold in consensus ticks. /// Zero (and the `0` / `disabled` / `unlimited` sentinels, which all parse /// to zero) is rejected at boot. + /// + /// Paces streams that are already open only. How long a hole waits before a + /// stream is opened for it is `repair_gap_debounce_interval`. #[serde(default = "default_repair_retry_interval")] #[serde_as(as = "DisplayFromStr")] #[config_env(leaf)] pub repair_retry_interval: IggyDuration, + /// How long a partition backup must hold committed ops it cannot walk to + /// before the shard sweep OPENS a repair session for it. + /// + /// Separate from `repair_retry_interval`, which paces an already-open + /// stream: this one decides how long a replication hole stays open, so + /// raising the retry interval to quiet repair chatter must not widen it. + /// Floored at `PARTITION_GAP_DEBOUNCE_TICKS_MIN` consensus ticks, since one + /// tick of lag is ordinary pipelining and repair against it would fire on + /// healthy traffic; the shard crate owns the floor and `config.toml` states + /// its value. Zero (and the `0` / `disabled` / `unlimited` sentinels, which + /// all parse to zero) is rejected at boot. + #[serde(default = "default_repair_gap_debounce_interval")] + #[serde_as(as = "DisplayFromStr")] + #[config_env(leaf)] + pub repair_gap_debounce_interval: IggyDuration, /// Prepares a peer serves per repair round before the requester walks to /// the next chunk. Each frame rides the per-peer message-bus queue, so this /// must stay below `message_bus.peer_queue_capacity` or a full round @@ -1002,6 +1028,13 @@ impl Validatable for ClusterConfig { // The repair retry interval sizes a tick threshold that has to advance; // `0` / `disabled` / `unlimited` all collapse to zero and would wedge // every stalled repair stream - reject them. + if self.repair_gap_debounce_interval.get_duration().is_zero() { + eprintln!( + "Invalid cluster configuration: cluster.repair_gap_debounce_interval must be \ + nonzero (it debounces the partition sweep's repair arm)" + ); + return Err(ConfigurationError::InvalidConfigurationValue); + } if self.repair_retry_interval.get_duration().is_zero() { eprintln!( "Invalid cluster configuration: cluster.repair_retry_interval must be nonzero \ @@ -1471,6 +1504,7 @@ mod tests { ), view_probe_attempts_max: default_view_probe_attempts_max(), repair_retry_interval: default_repair_retry_interval(), + repair_gap_debounce_interval: default_repair_gap_debounce_interval(), repair_chunk_max: default_repair_chunk_max(), superblock_wedged_fatal_timeout: default_superblock_wedged_fatal_timeout(), nodes: Vec::new(), @@ -1878,6 +1912,7 @@ mod cluster_validate_tests { ), view_probe_attempts_max: default_view_probe_attempts_max(), repair_retry_interval: default_repair_retry_interval(), + repair_gap_debounce_interval: default_repair_gap_debounce_interval(), repair_chunk_max: default_repair_chunk_max(), superblock_wedged_fatal_timeout: default_superblock_wedged_fatal_timeout(), nodes, @@ -2005,6 +2040,16 @@ mod cluster_validate_tests { assert!(c.validate().is_err()); } + #[test] + fn validate_rejects_zero_repair_gap_debounce_interval() { + // Same sentinel family as the retry interval above: `0`, `disabled` and + // `unlimited` all parse to zero, and a zero debounce would arm repair + // against a single reordered prepare on every partition. + let mut c = cfg(vec![node("n1", 0), node("n2", 1)]); + c.repair_gap_debounce_interval = IggyDuration::new(Duration::ZERO); + assert!(c.validate().is_err()); + } + #[test] fn validate_rejects_zero_repair_chunk_max() { let mut c = cfg(vec![node("n1", 0), node("n2", 1)]); diff --git a/core/configs/src/server_config/defaults.rs b/core/configs/src/server_config/defaults.rs index 303ccd0a36..330f39877c 100644 --- a/core/configs/src/server_config/defaults.rs +++ b/core/configs/src/server_config/defaults.rs @@ -112,6 +112,11 @@ impl Default for ClusterConfig { .repair_retry_interval .parse() .unwrap(), + repair_gap_debounce_interval: SERVER_CONFIG + .cluster + .repair_gap_debounce_interval + .parse() + .unwrap(), repair_chunk_max: SERVER_CONFIG.cluster.repair_chunk_max as usize, nodes: SERVER_CONFIG .cluster diff --git a/core/consensus/src/observability.rs b/core/consensus/src/observability.rs index 4519693d60..87473c3ad3 100644 --- a/core/consensus/src/observability.rs +++ b/core/consensus/src/observability.rs @@ -402,6 +402,11 @@ pub struct PartitionDiagEvent<'a> { pub message: &'static str, pub operation: Option, pub op: Option, + /// This replica's sequencer position when the event fired. Set where `op` + /// alone cannot say which side of the frontier the frame landed on: a + /// prepare above it is a forward gap, one at or below it is a replay of an + /// op already sequenced. + pub sequence: Option, pub prepare_checksum: Option, pub reason: Option<&'a str>, pub error: Option>, @@ -415,6 +420,7 @@ impl<'a> PartitionDiagEvent<'a> { message, operation: None, op: None, + sequence: None, prepare_checksum: None, reason: None, error: None, @@ -433,6 +439,12 @@ impl<'a> PartitionDiagEvent<'a> { self } + #[must_use] + pub const fn with_sequence(mut self, sequence: u64) -> Self { + self.sequence = Some(sequence); + self + } + #[must_use] pub const fn with_prepare_checksum(mut self, prepare_checksum: u128) -> Self { self.prepare_checksum = Some(prepare_checksum); @@ -469,6 +481,7 @@ fn emit_partition_diag_error(event: &PartitionDiagEvent<'_>) { let ctx = event.replica; let operation = event.operation.map_or("", operation_as_str); let op = event.op.unwrap_or_default(); + let sequence = event.sequence.unwrap_or_default(); let prepare_checksum = event.prepare_checksum.unwrap_or_default(); let reason = event.reason.unwrap_or(""); let error = event.error.as_deref().unwrap_or(""); @@ -490,6 +503,7 @@ fn emit_partition_diag_error(event: &PartitionDiagEvent<'_>) { role = ctx.role.as_str(), operation, op, + sequence, prepare_checksum, reason, error, @@ -501,6 +515,7 @@ fn emit_partition_diag_warn(event: &PartitionDiagEvent<'_>) { let ctx = event.replica; let operation = event.operation.map_or("", operation_as_str); let op = event.op.unwrap_or_default(); + let sequence = event.sequence.unwrap_or_default(); let prepare_checksum = event.prepare_checksum.unwrap_or_default(); let reason = event.reason.unwrap_or(""); let error = event.error.as_deref().unwrap_or(""); @@ -522,6 +537,7 @@ fn emit_partition_diag_warn(event: &PartitionDiagEvent<'_>) { role = ctx.role.as_str(), operation, op, + sequence, prepare_checksum, reason, error, @@ -533,6 +549,7 @@ fn emit_partition_diag_info(event: &PartitionDiagEvent<'_>) { let ctx = event.replica; let operation = event.operation.map_or("", operation_as_str); let op = event.op.unwrap_or_default(); + let sequence = event.sequence.unwrap_or_default(); let prepare_checksum = event.prepare_checksum.unwrap_or_default(); let reason = event.reason.unwrap_or(""); let error = event.error.as_deref().unwrap_or(""); @@ -554,6 +571,7 @@ fn emit_partition_diag_info(event: &PartitionDiagEvent<'_>) { role = ctx.role.as_str(), operation, op, + sequence, prepare_checksum, reason, error, @@ -565,6 +583,7 @@ fn emit_partition_diag_debug(event: &PartitionDiagEvent<'_>) { let ctx = event.replica; let operation = event.operation.map_or("", operation_as_str); let op = event.op.unwrap_or_default(); + let sequence = event.sequence.unwrap_or_default(); let prepare_checksum = event.prepare_checksum.unwrap_or_default(); let reason = event.reason.unwrap_or(""); let error = event.error.as_deref().unwrap_or(""); @@ -586,6 +605,7 @@ fn emit_partition_diag_debug(event: &PartitionDiagEvent<'_>) { role = ctx.role.as_str(), operation, op, + sequence, prepare_checksum, reason, error, @@ -597,6 +617,7 @@ fn emit_partition_diag_trace(event: &PartitionDiagEvent<'_>) { let ctx = event.replica; let operation = event.operation.map_or("", operation_as_str); let op = event.op.unwrap_or_default(); + let sequence = event.sequence.unwrap_or_default(); let prepare_checksum = event.prepare_checksum.unwrap_or_default(); let reason = event.reason.unwrap_or(""); let error = event.error.as_deref().unwrap_or(""); @@ -618,6 +639,7 @@ fn emit_partition_diag_trace(event: &PartitionDiagEvent<'_>) { role = ctx.role.as_str(), operation, op, + sequence, prepare_checksum, reason, error, diff --git a/core/integration/tests/cluster/parked_frame_redispatch.rs b/core/integration/tests/cluster/parked_frame_redispatch.rs index 3b1a525ed4..23bc22be73 100644 --- a/core/integration/tests/cluster/parked_frame_redispatch.rs +++ b/core/integration/tests/cluster/parked_frame_redispatch.rs @@ -23,8 +23,9 @@ //! that parked. This test pins the park path positively, and on the replica //! where getting it wrong creates a replica gap: a client request that never //! reaches the plane is answered with a retriable status and the SDK replays it, -//! while a replicated PREPARE has no client behind it and must wait for a later -//! commit heartbeat to arm repair if this path drops it. +//! while a replicated PREPARE has no client behind it: nothing re-sends it once +//! its op has quorum, so the backup gap-stops and waits out `tick_partitions`' +//! repair debounce before anything refetches it. //! //! What makes the window wide on a backup is the commit broadcast. A backup //! learns a metadata commit from the `commit` field of the next prepare on that @@ -50,7 +51,7 @@ //! - Every acked message is readable in dense offset order, each producer's own //! sends stay in the order it made them, and all three replicas hold //! byte-identical segments. A prepare lost to the gap check leaves a backup -//! permanently short, since the gap never closes on its own. +//! short until the repair driver's next pass closes the gap. //! //! The harness removes an ambient `RUST_LOG` when this test supplies its explicit //! logging level, and the log oracle falls back from captured stdout to the @@ -98,8 +99,13 @@ const DEGRADED_MARKERS: [&str; 3] = [ ]; /// `IggyPartition::on_replicate`'s backup gap check. A re-dispatch that appends -/// behind an op already queued on the inbox surfaces here, and the dropped op -/// forces repair that correct redispatch ordering should never need. +/// behind an op already queued on the inbox surfaces here. The dropped op is +/// refetched by `tick_partitions`' level-triggered repair driver, but only after +/// its debounce interval, so a re-dispatch that trips this has already stalled +/// the replica for ~1s and the marker still means the ordering broke. +/// +/// The metadata plane logs the same string, so the counting below pairs it with +/// `PARTITION_PLANE_FIELD`. const GAP_MARKER: &str = "dropping out-of-order prepare (gap)"; const PARTITION_PLANE_FIELD: &str = "plane=\"partitions\""; @@ -347,7 +353,8 @@ fn assert_no_degraded_park_paths(harness: &TestHarness) { assert_eq!( partition_gaps, 0, "node {node} logged {GAP_MARKER:?}: a re-dispatched prepare lost its arrival \ - position and forced avoidable partition repair" + position, and the op it displaced is recoverable only by waiting out the \ + repair driver's debounce" ); } } diff --git a/core/integration/tests/data_integrity/storage_compat.rs b/core/integration/tests/data_integrity/storage_compat.rs index 5c7ce66471..ea09f47e07 100644 --- a/core/integration/tests/data_integrity/storage_compat.rs +++ b/core/integration/tests/data_integrity/storage_compat.rs @@ -57,17 +57,30 @@ //! unseals the last segment, so a chain that ends on a rotation boundary //! would only ever hand that path an empty file. //! -//! # Structural false positive to avoid +//! # The configuration is held constant across the swap //! -//! The harness forwards every parent `IGGY_*` variable to the server child, -//! and the server treats an unknown `IGGY_*` name as a `debug_assert`. A pull -//! request that adds a config leaf AND a test that sets it therefore makes -//! the BASELINE binary die on startup, which looks like a compatibility -//! break. Nothing here detects that: `resolve_config_paths` validates the -//! name against the catalog of the build under test, which is the wrong side -//! of the swap, so it only proves the name exists on HEAD. Keeping every -//! override to a name that already exists on the merge base stays a rule the -//! author has to follow by hand. +//! Both boots read the BASELINE's `core/server/config.toml`, extracted next +//! to the baseline binary by `scripts/ci/storage-compat.sh`. Neither side may +//! fall back to the server's relative default path: figment resolves that by +//! walking up from the test process's directory, so the BASELINE parses THIS +//! branch's file, and a key added under a `deny_unknown_fields` table (every +//! `[cluster]` and `[node]` one) fails its extraction with +//! `Config(CannotLoadConfiguration)`. That reads as a compatibility break and +//! is not one. +//! +//! One file across the swap leaves the binary as the only variable, and it +//! pins the two rules the config plane already carries: a field this branch +//! adds needs its `#[serde(default)]`, because the build under test boots on +//! a file written before the field existed, and a key this branch takes away +//! needs its `RelocatedKey` entry. Deleting a key from a +//! `deny_unknown_fields` table, or relocating one, makes the SECOND boot +//! refuse the file; strip that key from the extracted copy when it happens. +//! +//! The overrides below reach both binaries as `IGGY_*` variables, and +//! `resolve_config_paths` checks them against the catalog of the build under +//! test only. [`assert_overrides_known_to_the_baseline`] covers the other +//! side, where an unknown `IGGY_*` name trips a `debug_assert` and takes the +//! debug binary down at boot. //! //! `IGGY_TEST_VERBOSE` makes the harness inherit the server's stdout instead //! of capturing it, which would make the tombstone check vacuous. The @@ -88,13 +101,21 @@ use std::time::{Duration, Instant}; use tokio::time::sleep; /// Absolute path to the baseline `iggy-server` binary. See -/// [`baseline_server_binary`] for why a relative value is refused. +/// [`baseline_path_from_env`] for why a relative value is refused. /// /// NOT `IGGY_`-prefixed on purpose: the harness forwards every parent /// `IGGY_*` variable to the server child, where an unknown name trips a /// `debug_assert` and kills the debug build this test runs against. const BASELINE_SERVER_ENV: &str = "COMPAT_BASELINE_SERVER"; +/// Absolute path to the baseline's `core/server/config.toml`, which both +/// boots read. Same naming rule as [`BASELINE_SERVER_ENV`]. +const BASELINE_CONFIG_ENV: &str = "COMPAT_BASELINE_CONFIG"; + +/// Selects the server's config file, ahead of the relative default path the +/// module documentation warns about. +const CONFIG_PATH_ENV: &str = "IGGY_CONFIG_PATH"; + /// `metadata.journal_slots` floor for `prepare_queue_depth = 32` /// (`4 * max(64, 32)`). With the 64-op checkpoint margin a checkpoint fires /// once the journal holds 192 committed ops. @@ -263,11 +284,17 @@ const GRACEFUL_SHUTDOWN_MARKER: &str = "server shutdown complete"; #[ignore = "needs a baseline iggy-server built from master; run via scripts/ci/storage-compat.sh"] async fn should_read_back_a_data_directory_written_by_the_baseline_server() { let baseline = baseline_server_binary(); - let envs = resolve_config_paths(&HashMap::from([( + let baseline_config = baseline_server_config(); + let overrides = HashMap::from([( "metadata.journal_slots".to_string(), JOURNAL_SLOTS.to_string(), - )])) - .expect("metadata.journal_slots resolves against the live config catalog"); + )]); + assert_overrides_known_to_the_baseline(&overrides, &baseline_config); + let mut envs = resolve_config_paths(&overrides) + .expect("metadata.journal_slots resolves against the live config catalog"); + // `extra_envs` is re-applied on every `start()`, so the swapped-in binary + // reads this same file. + envs.insert(CONFIG_PATH_ENV.to_string(), baseline_config.clone()); let mut harness = TestHarness::builder() .cluster_nodes(1) @@ -279,7 +306,13 @@ async fn should_read_back_a_data_directory_written_by_the_baseline_server() { ) .build() .unwrap(); - harness.start().await.unwrap(); + harness.start().await.unwrap_or_else(|error| { + panic!( + "the baseline server did not start. It boots on the merge base's own config file, \ + so a Config(...) error below means the harness fed it something the merge base \ + does not carry: {error}" + ) + }); let data_path = harness.server().data_path(); let client = harness.tcp_root_client().await.unwrap(); @@ -677,7 +710,14 @@ async fn should_read_back_a_data_directory_written_by_the_baseline_server() { // The swap: `None` selects the cargo-built binary of the crate under test. harness.server_mut().set_executable_path(None); - harness.restart_server().await.unwrap(); + harness.restart_server().await.unwrap_or_else(|error| { + panic!( + "the server under test did not restart on the data directory the baseline wrote. \ + A Config(...) error below is a configuration break rather than a storage one: \ + this branch dropped or renamed a key the merge base's config file still sets, or \ + added a field without a `#[serde(default)]`: {error}" + ) + }); // `tcp_root_client` hands out a client the harness does not own, so the // reconnect loop inside `restart_server` never touched it. Take a fresh one. @@ -877,46 +917,88 @@ async fn should_read_back_a_data_directory_written_by_the_baseline_server() { ); } -/// Resolve the baseline binary to an absolute, existing path. +/// Resolve one of the baseline artefacts to an absolute, existing path. /// -/// The absolute requirement is not cosmetic. `ServerHandle::start` only spawns -/// the configured path directly when it has more than one component or already -/// exists; a bare name that does not exist falls through to +/// The absolute requirement is not cosmetic for the binary. `ServerHandle::start` +/// only spawns the configured path directly when it has more than one component +/// or already exists; a bare name that does not exist falls through to /// `Command::cargo_bin`, which resolves the binary of the crate under test. So /// a single-component value would boot the HEAD build while the test reported /// it as the baseline, comparing master against master and passing forever. /// The same silent-green family as running a stale binary. -fn baseline_server_binary() -> String { - let raw = std::env::var(BASELINE_SERVER_ENV).unwrap_or_else(|_| { +fn baseline_path_from_env(variable: &str, artefact: &str) -> String { + let raw = std::env::var(variable).unwrap_or_else(|_| { panic!( - "{BASELINE_SERVER_ENV} is unset. It must be an ABSOLUTE path to the iggy-server \ - binary built from master. Without it this test would boot the build under test \ - twice and prove nothing, so it refuses to run. Use scripts/ci/storage-compat.sh, \ - which builds the baseline and exports the variable." + "{variable} is unset. It must be an ABSOLUTE path to the baseline {artefact}. \ + Without it this test would run entirely on the build under test and prove nothing, \ + so it refuses to run. Use scripts/ci/storage-compat.sh, which builds the baseline \ + and exports both variables." ) }); let path = PathBuf::from(&raw); assert!( path.is_absolute(), - "{BASELINE_SERVER_ENV}={raw:?} is relative. It must be an ABSOLUTE path: a bare binary \ - name that does not exist on disk is silently resolved as the build under test, which \ - would compare master against master and report green forever." + "{variable}={raw:?} is relative. It must be an ABSOLUTE path: a bare name that does not \ + exist on disk is silently resolved as the build under test, which would compare master \ + against master and report green forever." ); let resolved = fs::canonicalize(&path).unwrap_or_else(|error| { panic!( - "{BASELINE_SERVER_ENV}={raw:?} does not resolve: {error}. It must point at an \ - iggy-server binary built from master." + "{variable}={raw:?} does not resolve: {error}. It must point at the baseline \ + {artefact}." ) }); assert!( resolved.is_file(), - "{BASELINE_SERVER_ENV}={raw:?} resolves to {}, which is not a file.", + "{variable}={raw:?} resolves to {}, which is not a file.", resolved.display() ); resolved.display().to_string() } +fn baseline_server_binary() -> String { + baseline_path_from_env(BASELINE_SERVER_ENV, "iggy-server binary") +} + +fn baseline_server_config() -> String { + baseline_path_from_env(BASELINE_CONFIG_ENV, "core/server/config.toml") +} + +/// Refuse an override the baseline binary cannot read. +/// +/// Every override travels to both binaries as an `IGGY_*` variable, and an +/// unknown name trips a `debug_assert` in the server's env provider, which +/// takes the debug baseline down at boot with no mention of the variable. +/// `resolve_config_paths` cannot catch that: it validates against the catalog +/// of the build under test, which is the wrong side of the swap. The baseline +/// config file spells out every knob that build has, so it stands in for the +/// catalog here. +fn assert_overrides_known_to_the_baseline(overrides: &HashMap, config: &str) { + let raw = fs::read_to_string(config) + .unwrap_or_else(|error| panic!("reading the baseline config {config}: {error}")); + let doc: toml::Value = toml::from_str(&raw) + .unwrap_or_else(|error| panic!("parsing the baseline config {config}: {error}")); + + for path in overrides.keys() { + // The `system.` fallback mirrors `resolve_config_paths`. + let prefixed = format!("system.{path}"); + let known = [path.as_str(), prefixed.as_str()] + .into_iter() + .any(|candidate| config_key(&doc, candidate).is_some()); + assert!( + known, + "the override '{path}' names no key in the baseline config {config}, so the \ + baseline server has no such config leaf and dies on the variable at boot. Drive \ + the behaviour through a key that already exists on the merge base." + ); + } +} + +fn config_key<'a>(doc: &'a toml::Value, path: &str) -> Option<&'a toml::Value> { + path.split('.').try_fold(doc, |node, key| node.get(key)) +} + /// Options of the topic that carries the segment chain. Every key but /// `compression_algorithm` is sent, so that one must come back derived while /// the rest come back explicit. diff --git a/core/partitions/Cargo.toml b/core/partitions/Cargo.toml index 1f509cd2e6..3a1b48cecb 100644 --- a/core/partitions/Cargo.toml +++ b/core/partitions/Cargo.toml @@ -37,6 +37,14 @@ publish = false # requests it. No production caller. simulator = [] +# Fault injection for the harness (`IggyPartition::inject_commit_failure`): +# makes a local commit fail the way a full disk does, so the shard sweep's +# fence path can be driven over an in-memory partition. Requested ONLY by +# `core/simulator`'s dev-dependencies, which the v3 resolver keeps out of any +# non-test build, so the branch it gates cannot reach a shipped binary the way +# a `[dependencies]` feature would through unification. +fault-injection = [] + [dependencies] ahash = { workspace = true } bytemuck = { workspace = true } diff --git a/core/partitions/src/iggy_partition.rs b/core/partitions/src/iggy_partition.rs index 19db69e4c0..2128ae517b 100644 --- a/core/partitions/src/iggy_partition.rs +++ b/core/partitions/src/iggy_partition.rs @@ -30,7 +30,7 @@ use crate::poll_plan::{ }; use crate::segment::Segment; use crate::state_transfer::{PartitionTransferSession, PendingTransferRearm}; -use crate::types::{FatalCommit, RepairConclusion, RepairSession}; +use crate::types::{COMMIT_WALK_OPS_MAX, FatalCommit, RepairConclusion, RepairSession}; use crate::{ AppendResult, Partition, PartitionOffsets, PartitionsConfig, PollQueryResult, PollingArgs, PollingConsumer, @@ -154,6 +154,36 @@ where /// set when the recovery handshake finds this replica behind the group's /// commit frontier, cleared when `RepairDone` completes the walk. pub repair: Option, + /// Consecutive shard-sweep ticks this partition has been seen gap-stopped + /// (committed ops it cannot walk to, because the op at its commit frontier + /// plus one is missing). Debounces the sweep's level-triggered repair arm, + /// and is spent by whichever site opens the repair session. + /// + /// `Cell` for the same reason as [`Self::prepare_gap_drops`]: the sweep + /// drives it from the shared borrow it probes the partition through, so the + /// in-flight scan the arm budget needs can run without a `&mut` outstanding. + pub gap_ticks: Cell, + /// Prepares the backup gap check destroyed since the shard last drained + /// the count, folded into `partition_prepare_gap_drops_total` there. + /// Replicated traffic has no client to answer and retransmit skips ops that + /// already reached quorum, so nothing else records that the frame existed. + /// It counts what the ordering check destroyed, which is neither the holes + /// nor only them: a gap opened by the last prepare of a burst leaves it at + /// zero, and a duplicate delivery of an op this replica already sequenced + /// bumps it without any hole existing. Zero proves nothing, and nonzero is + /// a reason to look at the `sequence` field on the drop log, which is what + /// separates the two shapes. + /// + /// `Cell`: the shard drains it once per sweep, off the same shared borrow + /// the rest of the tick reads the partition through, so a `&mut` here would + /// buy a second lookup of the same group per tick. + prepare_gap_drops: Cell, + /// Fault injection for the harness, armed by + /// [`Self::inject_commit_failure`]. Behind a feature only the simulator's + /// DEV-dependencies turn on, so no build that ships this crate compiles the + /// branch it gates. + #[cfg(any(test, feature = "fault-injection"))] + injected_commit_failure: bool, /// Highest message offset recovered from segments at boot (`None` when /// the partition booted empty). Repaired batches at or below this line /// are already persisted and counted; the flush and commit paths skip @@ -276,6 +306,10 @@ where /// [`Self::note_transfer_rearm_scheduled`]; livelock across attempts is /// bounded by [`Self::transfer_failures`] and its exponential backoff. transfer_attempts: u32, + /// Consecutive stalled re-requests on the live repair session, against + /// [`crate::types::REPAIR_MAX_STALL_RETRIES`]. Survives the session, so rotating + /// the peer cannot reset it; cleared by real progress. + repair_attempts: u32, /// CONSECUTIVE transfer failures of any class (decode, spill, install, /// peer-unavailable, stall exhaustion). Deliberately NOT keyed on the /// offered generation: a committing primary advances its generation @@ -514,6 +548,10 @@ where consumer_offset_enforce_fsync: false, runtime_options: TopicRuntimeOptions::default(), repair: None, + gap_ticks: Cell::new(0), + prepare_gap_drops: Cell::new(0), + #[cfg(any(test, feature = "fault-injection"))] + injected_commit_failure: false, recovered_durable_offset: None, installed_frontier: None, fatal: None, @@ -533,6 +571,7 @@ where offset_reservation_lease: u64::from(crate::DEFAULT_OFFSET_RESERVATION_LEASE), transfer: None, transfer_attempts: 0, + repair_attempts: 0, transfer_failures: 0, transfer_refusals: 0, transfer_rearm: None, @@ -1710,6 +1749,51 @@ where self.write_superblock_advancing(superblock, 0, claim).await } + /// Take and clear the gap-drop count ([`Self::prepare_gap_drops`]). + #[must_use = "dropping the count loses the only record those prepares existed"] + pub const fn take_prepare_gap_drops(&self) -> u64 { + self.prepare_gap_drops.replace(0) + } + + /// Read the gap-drop count without clearing it, so a test can prove the + /// count was still buffered on the partition at the moment it was removed. + #[cfg(any(test, feature = "fault-injection"))] + #[must_use] + pub const fn prepare_gap_drops(&self) -> u64 { + self.prepare_gap_drops.get() + } + + /// Fail the next local commit of a committed `SendMessages` op the way a + /// full disk fails it, fencing the partition. + /// + /// The commit path's own fence is covered by a `/dev/full` writer in this + /// crate's tests, but the shard sweep that must OBSERVE the fence runs over + /// in-memory partitions in the simulator, where no device can be made to + /// fail. + #[cfg(any(test, feature = "fault-injection"))] + pub const fn inject_commit_failure(&mut self) { + self.injected_commit_failure = true; + } + + /// Burn one repair stall round; `true` once the budget is exhausted and the + /// session should be re-armed against a different peer. + /// + /// On the PARTITION, not the session, for the same reason the transfer + /// budget is: the rotation mints a new session, which would otherwise reset + /// the count and re-target forever without ever giving up on the ring. + #[must_use = "the bool is the rotate verdict; dropping it disables the stall budget"] + pub const fn burn_repair_attempt(&mut self) -> bool { + self.repair_attempts += 1; + self.repair_attempts > crate::types::REPAIR_MAX_STALL_RETRIES + } + + /// Real repair progress (any in-window frame from the serving peer): reset + /// the stall budget, so it bounds CONSECUTIVE stalls rather than the ones a + /// long healthy stream accumulates. + pub const fn note_repair_progress(&mut self) { + self.repair_attempts = 0; + } + /// Burn one transfer stall round; `true` once the budget is exhausted. /// Lives on the partition, not the session, so a re-minted session /// cannot reset it (see [`Self::transfer_attempts`]). @@ -3159,7 +3243,7 @@ where return; } - let journal_holds_op = self.log.journal().inner.header_by_op(header.op).is_some(); + let journal_holds_op = self.log.journal().inner.holds_op(header.op); if journal_holds_op { // Retransmit after downstream flap: durable here but commit // hasn't caught up. Re-forward + re-ACK so primary's view of @@ -3251,6 +3335,10 @@ where let is_backup = self.consensus().is_follower(); if is_backup { if header.op != current_op + 1 { + // `sequence` is what separates the two shapes this line covers: + // a forward gap (op above the sequencer, the hole the repair + // driver closes) and a retransmit of an op this replica already + // sequenced. Without it they read identically. emit_partition_diag( tracing::Level::WARN, &PartitionDiagEvent::new( @@ -3258,8 +3346,11 @@ where "dropping out-of-order prepare (gap)", ) .with_operation(header.operation) - .with_op(header.op), + .with_op(header.op) + .with_sequence(current_op), ); + self.prepare_gap_drops + .set(self.prepare_gap_drops.get().saturating_add(1)); return; } } else { @@ -3486,6 +3577,24 @@ where } } + /// Apply the committed prefix this replica can reach, from the pipeline if + /// the primary populated one and from the journal otherwise. + /// + /// The journal half is bounded by [`COMMIT_WALK_OPS_MAX`], for EVERY caller + /// and not just the tick sweep: `on_commit`, `StartView` adoption, the + /// post-transfer tail and the post-repair walk all reach this with the same + /// resident backlog behind them, and all four run on the shard pump. The + /// pipeline half is left alone, being the primary's own in-flight window, + /// which the pipeline depth already caps. + /// + /// Nothing is lost by stopping early: the run is re-derived from + /// `commit_min` on the next call, and the sweep's walk predicate stays true + /// until the group drains, so the next tick resumes exactly here. The one + /// caller with no next tick is the shutdown drain, and it is covered twice + /// over: [`Self::flush_committed_messages`] persists the committed prefix + /// by bytes rather than by walk, and what stays un-applied sits above the + /// `commit_min` the superblock records, which is what makes the restarted + /// replica ask its peers for it. #[allow(clippy::future_not_send)] pub async fn commit_journal(&mut self, config: &PartitionsConfig) { if self.fatal.is_some() { @@ -3504,7 +3613,7 @@ where // double-count against `advance_commit_min`. let mut drained = drain_committable_prefix(self.consensus()); if drained.is_empty() { - drained = self.collect_committable_from_journal(); + drained = self.collect_committable_from_journal(COMMIT_WALK_OPS_MAX); } if drained.is_empty() { return; @@ -3529,13 +3638,13 @@ where /// the journal keeps its committed entries until they are flushed /// (`commit_messages` drains only the committed prefix), so this read finds /// every committed op while the uncommitted tail stays resident. - fn collect_committable_from_journal(&self) -> Vec { + fn collect_committable_from_journal(&self, max_ops: usize) -> Vec { let from_op = self.consensus.commit_min() + 1; let commit_max = self.consensus.commit_max(); self.log .journal() .inner - .committed_headers_from(from_op, commit_max) + .committed_headers_from(from_op, commit_max, max_ops) .into_iter() .map(PipelineEntry::new) .collect() @@ -3843,6 +3952,10 @@ where } async fn commit_messages(&mut self, config: &PartitionsConfig) -> Result<(), IggyError> { + #[cfg(any(test, feature = "fault-injection"))] + if std::mem::take(&mut self.injected_commit_failure) { + return Err(IggyError::CannotSaveMessagesToSegment); + } self.commit_messages_inner(config, false).await } @@ -5773,11 +5886,12 @@ where return; } // Any in-window frame proves the stream is alive; only silence - // should age the stall counter. + // should age the stall counter or spend the rotation budget. if let Some(session) = self.repair.as_mut() { session.idle_ticks = 0; + self.repair_attempts = 0; } - if self.log.journal().inner.header_by_op(header.op).is_some() { + if self.log.journal().inner.holds_op(header.op) { return; } let applied = if header.operation == Operation::SendMessages { diff --git a/core/partitions/src/journal.rs b/core/partitions/src/journal.rs index 45f3b6d1d8..1bffcd16f4 100644 --- a/core/partitions/src/journal.rs +++ b/core/partitions/src/journal.rs @@ -762,6 +762,30 @@ where headers.iter().find(|header| header.op == op).copied() } + /// Whether `op` is resident, in O(log n) instead of `header_by_op`'s scan. + /// Callers that only need presence must use this: a miss is the common case + /// on the residency checks, and a miss is exactly when the scan walks the + /// whole vec. + /// + /// It answers off `op_to_storage_offset`, so it is `header_by_op(op).is_some()` + /// everywhere except INSIDE [`Self::append_with_meta`], which pushes the + /// header before the storage write and inserts the offset after it: a task + /// that interleaves at that await sees the header without the offset and is + /// answered `false`. Both are cleared together at every clear site + /// ([`Self::commit`], [`Self::evict_prefix`], the restore path), so that + /// window is the only divergence. + /// + /// The window is unreachable for this journal: `PartitionJournalMemStorage` + /// writes to memory and its `write_at` never yields, so no task can observe + /// the half-inserted state. That is what lets `apply_repaired_prepare` lean + /// on this for idempotence, where a false negative would re-journal an op + /// the log already holds. A future yielding `Storage` has to insert the + /// offset before the write, or move that check back to the header vec. + pub fn holds_op(&self, op: u64) -> bool { + let op_to_storage_offset = unsafe { &*self.op_to_storage_offset.get() }; + op_to_storage_offset.contains_key(&op) + } + /// Presence and message-carrying shape of the repair window `(floor, to_op]` /// in ONE pass over the header vec. /// @@ -842,27 +866,64 @@ where } /// Headers for the contiguous op run `from_op ..= commit_max`, in op order, - /// stopping at the first missing op. A replication gap must not be skipped: - /// the caller advances `commit_min` strictly by one, so a hole would break - /// that contract. Headers are append-ordered, which is op-ascending on a - /// backup, so this is a single linear scan: drop ops below `from_op`, take - /// while contiguous, stop at the first gap or past `commit_max`. - pub fn committed_headers_from(&self, from_op: u64, commit_max: u64) -> Vec { + /// stopping at the first missing op and at `limit` headers. A replication + /// gap must not be skipped: the caller advances `commit_min` strictly by + /// one, so a hole would break that contract. + /// + /// `limit` bounds what one caller commits in a single pass. The run is + /// re-derived from the caller's own `commit_min` every time, so a truncated + /// answer is resumed, not lost. + pub fn committed_headers_from( + &self, + from_op: u64, + commit_max: u64, + limit: usize, + ) -> Vec { // Walk by OP, not by append position: after a rejoin the journal // interleaves live tail ops (which arrive while repair is still // streaming) with repaired window ops, so append order is no longer // op-ascending and a positional sequential scan would break at the // first interleave boundary forever. - let mut result = Vec::new(); - let mut op = from_op; - while op <= commit_max { - let Some(header) = self.header_by_op(op) else { - break; - }; - result.push(header); - op += 1; + // + // ONE pass over the headers, like `repaired_window_shape`, not a + // `header_by_op` probe per op: that probe is itself a linear scan, so + // probing walked the window against the whole vec, and the walk this + // feeds runs per group per tick over a rejoin's entire backlog. + if from_op > commit_max { + return Vec::new(); } - result + let headers = unsafe { &*self.headers.get() }; + // Sized off the RUN the caller asked for, capped by the headers that + // could possibly cover it. Sizing off `headers.len()` alone would make + // a one-op run allocate a slot per resident header, and `commit_max` is + // a cluster frontier this replica may be arbitrarily far below, so + // neither bound can be dropped. `saturating_add` because `commit_max` + // is `u64::MAX` on a saturated frontier, where `+ 1` would panic in + // debug and wrap to an empty span (a walk that never resumes) in + // release. + let span = (commit_max - from_op) + .saturating_add(1) + .min(limit as u64) + .min(headers.len() as u64); + // `span` is a `min` of two `usize`-derived values on a 64-bit target, + // so the narrowing is lossless; `slot` is then compared against it as + // a `u64` BEFORE any cast, so nothing depends on the cast to bound it. + #[allow(clippy::cast_possible_truncation)] + let span = span as usize; + let mut slots: Vec> = vec![None; span]; + for header in headers { + if header.op < from_op || header.op - from_op >= span as u64 { + continue; + } + #[allow(clippy::cast_possible_truncation)] + let slot = (header.op - from_op) as usize; + // First writer wins, matching the `header_by_op` probe this + // replaces (`find` returns the earliest match). + slots[slot].get_or_insert(*header); + } + // Stops at the first hole: a replication gap must not be skipped, or + // `advance_commit_min`'s sequential contract breaks. + slots.into_iter().map_while(|header| header).collect() } /// Oldest message offset still resident in the in-memory journal, if @@ -1555,7 +1616,7 @@ mod tests { // Contiguous run from op 1 stops before the missing op 3 even though // op 4 is resident and within commit_max. - let run = journal.committed_headers_from(1, 4); + let run = journal.committed_headers_from(1, 4, usize::MAX); let ops: Vec = run.iter().map(|header| header.op).collect(); assert_eq!( ops, @@ -1564,9 +1625,19 @@ mod tests { ); assert!( - journal.committed_headers_from(5, 4).is_empty(), + journal.committed_headers_from(5, 4, usize::MAX).is_empty(), "from_op past commit_max yields nothing" ); + + // The limit truncates the run rather than skipping ahead in it, so the + // caller resumes at the op it stopped on. + let bounded = journal.committed_headers_from(1, 4, 1); + let ops: Vec = bounded.iter().map(|header| header.op).collect(); + assert_eq!(ops, vec![1], "the limit must cut the run at its front"); + assert!( + journal.committed_headers_from(1, 4, 0).is_empty(), + "a zero limit commits nothing" + ); } /// Three-message batch with the broker append time (`base_timestamp`) diff --git a/core/partitions/src/lib.rs b/core/partitions/src/lib.rs index 013bc6de6a..059c795a0f 100644 --- a/core/partitions/src/lib.rs +++ b/core/partitions/src/lib.rs @@ -61,9 +61,10 @@ pub use segment::Segment; use server_common::Message; pub use server_common::send_messages::{IggyMessage, IggyMessageHeader, IggyMessages}; pub use types::{ - AppendResult, FatalCommit, Fragment, PartitionOffsets, PartitionPathLayout, PartitionsConfig, - PollFragments, PollQueryResult, PollingArgs, PollingConsumer, REPAIR_RETRY_TICKS, - RepairConclusion, RepairSession, SendMessagesResult, + AppendResult, COMMIT_WALK_OPS_MAX, FatalCommit, Fragment, PartitionOffsets, + PartitionPathLayout, PartitionsConfig, PollFragments, PollQueryResult, PollingArgs, + PollingConsumer, REPAIR_MAX_STALL_RETRIES, REPAIR_RETRY_TICKS, RepairConclusion, RepairSession, + SendMessagesResult, }; /// A partition's message log, named so a caller can carry one across a rebuild. diff --git a/core/partitions/src/types.rs b/core/partitions/src/types.rs index 4762b772a7..f2a3bd7918 100644 --- a/core/partitions/src/types.rs +++ b/core/partitions/src/types.rs @@ -251,6 +251,34 @@ impl Default for PartitionOffsets { /// from the serving peer. pub const REPAIR_RETRY_TICKS: u32 = 100; +/// Consecutive stalled re-requests tolerated before a repair session is +/// abandoned and re-armed against a different peer. +/// +/// A session pins its peer, and `repair.is_some()` fences the sweep's detector +/// and every edge-triggered arming site while it stands, so a peer that went +/// away (or never was reachable, which the gap-stopped-primary rotation can +/// pick) would wedge the group harder than having no session at all. Three +/// rounds is ~3s at the default retry interval: long enough that an ordinary +/// dropped frame is re-requested rather than re-targeted, short enough that a +/// dead peer costs one debounce interval, not a view change. +pub const REPAIR_MAX_STALL_RETRIES: u32 = 3; + +/// Committed ops one journal-driven commit walk applies before returning to the +/// pump. +/// +/// `commit_journal`'s journal half returns the whole resident +/// `(commit_min, commit_max]` run, and applying it reaches a segment flush per +/// batch with no await the pump can interleave: after a rejoin that is the +/// entire backlog in one call, with the consensus tick stopped behind it. Every +/// caller is re-driven (the shard sweep's walk backstop is level-triggered and +/// stays true until the group drains), so a truncated walk resumes on the next +/// tick instead of losing anything. +/// +/// Sized as a batch big enough that an ordinary group drains in one pass and +/// small enough that a full one cannot hold the pump for the view-change +/// escalation window. +pub const COMMIT_WALK_OPS_MAX: usize = 64; + /// One in-flight journal-repair stream for a partition group. #[derive(Debug, Clone, Copy)] pub struct RepairSession { diff --git a/core/server/config.toml b/core/server/config.toml index 3362651867..3bae72a0cc 100644 --- a/core/server/config.toml +++ b/core/server/config.toml @@ -655,9 +655,27 @@ view_probe_attempts_max = 5 # remaining window from the serving peer (duration). Repair frames are # fire-and-forget over the lossy bus, so a session with no retry wedges forever # on a single dropped frame. Paces both the metadata and partition repair loops; -# must be nonzero. +# must be nonzero. Paces streams that are already OPEN only: how long a hole +# waits before one is opened for it is repair_gap_debounce_interval below. repair_retry_interval = "1s" +# How long a partition backup must hold committed ops it cannot walk to before +# the shard's tick sweep opens a repair session for it (duration). The +# level-triggered floor under every edge-triggered arming site, which a produce +# stream can starve; must be nonzero. +# +# Floored at 50 consensus ticks (500ms at the 10ms tick), so values under that +# arm no sooner: one tick of lag is ordinary pipelining, and repair against it +# would fire on healthy traffic. +# +# Recovery latency for one hole is max(this, 500ms) plus up to one tick of sweep +# granularity. Under a correlated fault the sweep opens at most 3 sessions per +# tick and holds at most 8 at once, so the Nth group waiting on this shard adds +# ceil(N / 3) ticks on top, and more while sessions are already in flight. That +# cap is what keeps a node-wide rejoin from putting every group's repair stream +# on one serving peer at once. +repair_gap_debounce_interval = "1s" + # Prepares a peer serves per repair round before the requester walks to the next # chunk (integer). Each frame rides the per-peer message-bus queue, so this must # stay strictly below message_bus.peer_queue_capacity or a full round overruns diff --git a/core/server/src/boot/recovery.rs b/core/server/src/boot/recovery.rs index 5d21e1f7c5..27d7a45b7c 100644 --- a/core/server/src/boot/recovery.rs +++ b/core/server/src/boot/recovery.rs @@ -24,7 +24,7 @@ use crate::server_error::ServerError; use crate::session_manager::SessionManager; use crate::shell::{ ServerMetadata, ServerShard, ShellHandlers, ShellShardHandle, consensus_timers, - repair_retry_ticks, + repair_gap_debounce_ticks, repair_retry_ticks, }; use configs::server::ServerConfig; use consensus::{ @@ -285,6 +285,7 @@ pub(in crate::boot) async fn build_shard_for_thread( // Repair pacing is shared by both planes' repair loops, so it is a // per-shard tunable set once here rather than per consensus group. shard.set_repair_retry_ticks(repair_retry_ticks(config)); + shard.set_partition_gap_debounce_ticks(repair_gap_debounce_ticks(config)); shard.set_superblock_wedged_fatal_failures(superblock_wedged_fatal_failures(config)); shard.set_served_segment_cache_bytes_max( config @@ -939,6 +940,45 @@ mod tests { ); } + /// The floor is prose in `config.toml` ("50 consensus ticks (500ms at the + /// 10ms tick)"), which is the number an operator sizes + /// `repair_gap_debounce_interval` against. Nothing else would notice it + /// drifting. + #[test] + fn documented_gap_debounce_floor_matches_the_shard_constant() { + assert_eq!( + shard::PARTITION_GAP_DEBOUNCE_TICKS_MIN, + 50, + "the gap debounce floor moved; core/server/config.toml states it in \ + ticks and milliseconds under [cluster] repair_gap_debounce_interval" + ); + assert_eq!( + u128::from(shard::PARTITION_GAP_DEBOUNCE_TICKS_MIN) + * shard::CONSENSUS_TICK_INTERVAL.as_millis(), + 500, + "the floor is no longer 500ms; core/server/config.toml states that \ + figure under [cluster] repair_gap_debounce_interval" + ); + } + + #[test] + fn default_repair_gap_debounce_interval_matches_partitions_constant() { + // Same lockstep the retry interval keeps: the shipped config.toml value + // is what an un-configured replica and the simulator run on, and the + // shard's own default is the compile-time constant. + let config_default = configs::cluster::ClusterConfig::default() + .repair_gap_debounce_interval + .get_duration() + .as_millis(); + let built_in = + u128::from(partitions::REPAIR_RETRY_TICKS) * shard::CONSENSUS_TICK_INTERVAL.as_millis(); + assert_eq!( + config_default, built_in, + "[cluster] repair_gap_debounce_interval default drifted from the shard's \ + compile-time debounce" + ); + } + #[test] fn default_repair_chunk_max_matches_shard_constant() { // Belt and suspenders with the static assert above: that pins the diff --git a/core/server/src/partition_reconciler.rs b/core/server/src/partition_reconciler.rs index 4db62a2b7a..e35f623784 100644 --- a/core/server/src/partition_reconciler.rs +++ b/core/server/src/partition_reconciler.rs @@ -109,17 +109,33 @@ //! `park_dropped` when it parked and then lost its namespace. A prepare has //! nobody to answer, so the counter is the only record it existed. //! +//! A shed or discarded *prepare* is not recovered by retransmit once its op has +//! reached quorum (`consensus::retransmit_targets` skips entries with +//! `ok_quorum_received`), so the backup gap-stops. `tick_partitions` opens a +//! repair session for it: its level-triggered detector arms once a partition has +//! been gap-stopped for `[cluster] repair_gap_debounce_interval`, independently of the +//! edge-triggered arming sites (`StartView` adoption, the commit heartbeat, the +//! post-transfer tail), whose edges a produce stream can starve. A repair range +//! the primary has already evicted escalates to partition state transfer. The park policy +//! above still shrinks the exposure to a genuinely exhausted byte budget and a +//! namespace this shard cannot serve; the driver bounds how long either costs, +//! and `partition_prepare_gap_drops_total` counts what reached the gap check. +//! //! # Known gaps //! -//! Recorded here because both were previously carried as a TODO on the +//! Recorded here because they were previously carried as a TODO on the //! materialization barrier this module used to promise, and the barrier is gone //! (see above) while these are not: //! -//! A shed prepare is not retransmitted once its op reached quorum, but it is not -//! stranded until a view change. A later `CommitMessage` that advances the -//! backup's frontier runs `maybe_request_partition_repair`; an evicted repair -//! range escalates to partition state transfer. The park policy still avoids -//! manufacturing that recovery work unless a byte budget is already spent. +//! A shed prepare is not retransmitted once its op reached quorum, and the +//! commit heartbeat is no answer to it: a follower advances `commit_max` from +//! every prepare header before the gap check drops the frame, so under produce +//! the heartbeat lands as `Accepted` and the backstop inside that branch never +//! runs. What closes it is the level-triggered sweep above, which means the +//! residual exposure is its debounce (`[cluster] repair_gap_debounce_interval`, +//! floored at 500ms) plus the arm cap under a correlated fault, during which the +//! backup serves a short prefix. The park policy still avoids manufacturing that +//! recovery work unless a byte budget is already spent. //! //! TODO(krishna): `serves_committed_incarnation` and the park stamp both call //! `Streams::created_revision_for_namespace`, now on the per-request fence path. @@ -3532,9 +3548,11 @@ mod tests { drop(inbox); } - /// The replicated-prepare shape, which no other test covers and where both - /// park critical are worst: a prepare has no client, so discarding it forces - /// the backup to wait for a later commit heartbeat and journal repair. + /// The replicated-prepare shape, which no other test covers and where the + /// park path's stakes are highest: a prepare has no client, so + /// `deny_parked_client_request` no-ops on it and anything that discards it + /// loses committed data silently, recoverable only once `tick_partitions`' + /// repair driver notices the gap it left. /// /// A backup receives the prepare before its own metadata commits (so the frame /// parks unstamped), then applies the commit and materialises. The prepare must @@ -3574,7 +3592,7 @@ mod tests { shard.redispatched_frame_count(), 1, "the parked prepare must be staged for re-dispatch; discarding it is \ - an avoidable gap, since a prepare has no client to retry it" + silent committed-data loss, since a prepare has no client to answer" ); let (served, answered) = drain_inbox(&inbox); assert_eq!( diff --git a/core/server/src/shell.rs b/core/server/src/shell.rs index 661c34b9ce..18145cb689 100644 --- a/core/server/src/shell.rs +++ b/core/server/src/shell.rs @@ -200,3 +200,16 @@ pub(crate) fn repair_retry_ticks(config: &ServerConfig) -> u32 { )) .unwrap_or(u32::MAX) } + +/// `[cluster] repair_gap_debounce_interval` in consensus ticks: how long a +/// partition backup holds a hole before the sweep opens a repair session for +/// it. Deliberately NOT the retry interval above: that one paces an open +/// stream, and pairing them means quieting retry chatter also widens how long a +/// replication hole stays open. The shard applies +/// [`shard::PARTITION_GAP_DEBOUNCE_TICKS_MIN`] as a floor on top. +pub(crate) fn repair_gap_debounce_ticks(config: &ServerConfig) -> u32 { + u32::try_from(duration_to_ticks( + config.cluster.repair_gap_debounce_interval.get_duration(), + )) + .unwrap_or(u32::MAX) +} diff --git a/core/shard/src/lib.rs b/core/shard/src/lib.rs index f44c40f3e0..8181831fd8 100644 --- a/core/shard/src/lib.rs +++ b/core/shard/src/lib.rs @@ -1486,6 +1486,27 @@ where /// `[cluster] repair_retry_interval` at bootstrap. repair_retry_ticks: Cell, + /// Live repair sessions on this shard, republished by every partition sweep + /// and incremented as sessions open, for + /// [`PARTITION_REPAIRS_INFLIGHT_MAX`]. A tally rather than a scan because + /// the arming funnel holds a `&mut` to one partition, which a scan over the + /// plane would alias; one sweep stale at worst. + partition_repairs_inflight: Cell, + + /// Live gap debounce in consensus ticks: how long a partition holds a hole + /// before the sweep opens a repair session for it. Defaults to + /// [`partitions::REPAIR_RETRY_TICKS`]; the server overrides it from + /// `[cluster] repair_gap_debounce_interval` at bootstrap. + partition_gap_debounce_ticks: Cell, + + /// Namespace the next partition sweep starts from: the first group the + /// per-tick WALK budget turned away last pass, `None` to start at the front. + /// + /// The sweep visits namespaces in `BTreeMap` order, so without a carried + /// cursor the leading groups would spend the whole budget on every pass and + /// the tail would never be reached. See [`rotate_sweep_to_cursor`]. + partition_walk_cursor: Cell>, + /// Consecutive metadata superblock write failures tolerated before the /// process fail-stops. Defaults to 0 (disabled) so the simulator and tests /// keep a wedged-but-fenced replica alive; the server arms it from @@ -1667,6 +1688,9 @@ where partition_artifact_len_max: Cell::new(PARTITION_ARTIFACT_LEN_DEFAULT), repair_chunk_max: Cell::new(REPAIR_CHUNK_MAX), repair_retry_ticks: Cell::new(partitions::REPAIR_RETRY_TICKS), + partition_gap_debounce_ticks: Cell::new(partitions::REPAIR_RETRY_TICKS), + partition_repairs_inflight: Cell::new(0), + partition_walk_cursor: Cell::new(None), superblock_wedged_fatal_failures: Cell::new(0), bus_max_message_size: Cell::new(DEFAULT_BUS_MAX_MESSAGE_SIZE), metadata_transfer_attempts: Cell::new(0), @@ -1681,6 +1705,14 @@ where self.repair_retry_ticks.set(ticks); } + /// Override the partition sweep's gap debounce (consensus ticks) from + /// configuration. Called once per shard at bootstrap; the simulator and + /// tests keep the compile-time [`partitions::REPAIR_RETRY_TICKS`] default. + /// [`PARTITION_GAP_DEBOUNCE_TICKS_MIN`] still floors whatever is set. + pub fn set_partition_gap_debounce_ticks(&self, ticks: u32) { + self.partition_gap_debounce_ticks.set(ticks); + } + /// Arm the superblock fail-stop bound (consecutive write failures). /// Called once per shard at bootstrap; the simulator and tests keep the /// disabled default (0) so a wedged-but-fenced replica stays observable @@ -2113,6 +2145,9 @@ where partition_artifact_len_max: Cell::new(PARTITION_ARTIFACT_LEN_DEFAULT), repair_chunk_max: Cell::new(REPAIR_CHUNK_MAX), repair_retry_ticks: Cell::new(partitions::REPAIR_RETRY_TICKS), + partition_gap_debounce_ticks: Cell::new(partitions::REPAIR_RETRY_TICKS), + partition_repairs_inflight: Cell::new(0), + partition_walk_cursor: Cell::new(None), superblock_wedged_fatal_failures: Cell::new(0), bus_max_message_size: Cell::new(DEFAULT_BUS_MAX_MESSAGE_SIZE), metadata_transfer_attempts: Cell::new(0), @@ -2413,7 +2448,15 @@ where self.discard_parked_partition_frames(namespace); self.metrics.record_partition_removed(); confirmed_remove = true; - if removed.is_none() { + if let Some(partition) = removed { + // Tail of the gap-drop count. The tick sweep drains it + // per pass, but `get_by_ns` stops answering the moment + // the reconciler tombstones the namespace, so whatever + // the last pass before the tombstone left would go to + // the floor with the partition value. + self.metrics + .record_partition_prepare_gap_drops(partition.take_prepare_gap_drops()); + } else { tracing::trace!( shard = self_shard_id, namespace_raw = namespace.inner(), @@ -3221,8 +3264,9 @@ where /// buffer exists to absorb -- the partition primary materialises and /// replicates as soon as its own metadata commits, well before a lagging /// backup applies the same commit. Treating that as "prior incarnation" - /// destroys live traffic and forces the backup to recover a gap that the - /// park path could have delivered directly. + /// destroys live traffic: a replicated prepare has no client to answer, so + /// it would be dropped and the backup left gap-stopped until + /// `tick_partitions`' level-triggered driver notices and repairs it. /// The residual is unchanged from before the stamp existed -- a frame parked /// while the namespace was absent, then recreated under a new incarnation, /// is served against the replacement -- and closing it needs a wire-level @@ -3485,12 +3529,13 @@ where let existing = pending.get_mut(&namespace); let parked_len = existing.as_ref().map_or(0, |entry| entry.frames.len()); let namespace_bytes = existing.as_ref().map_or(0, |entry| entry.bytes); - // A prepare is never shed on a byte budget before the budget is spent. - // It has no client to retry it, and recovery requires a later commit - // heartbeat to expose the gap and arm same-view repair. A request costs - // only a retry, so it is refused the moment admitting it would cross a - // budget. This caps prepare residency at one frame of overshoot per - // budget (worst case + // A prepare is never shed on a byte budget. No client to answer, and + // recovery is slow: `consensus::retransmit_targets` skips an op that + // already reached quorum, so shedding one gap-stops the backup until + // `tick_partitions`' driver repairs it, where shedding a request costs + // one retry. A request is refused the moment admitting it + // would cross a budget; a prepare only once one is already spent. Caps + // prepare residency at one frame of overshoot per budget (worst case // `MAX_PARKED_BYTES` + `max_message_size`, 80 MiB per shard) instead of // at the budget, and is what makes an oversize frame parkable at all. let namespace_budget_spent = parked_len > 0 @@ -6825,12 +6870,16 @@ where ); let partitions = self.plane.partitions(); let repair_retry_ticks = self.repair_retry_ticks.get(); + let gap_debounce_ticks = self.partition_gap_debounce_ticks.get(); // Fan out over every group (each partition's heartbeat/retransmit timer // must advance), so the keyed single-namespace lookup the control-frame // handlers use does not apply here. The namespaces are snapshotted into // the pump's owned scratch (as `process_loopback` does) so no // partitions-plane borrow is held across the tick `.await`. namespace_scratch.extend(partitions.namespaces().copied()); + // Resume where the last sweep ran out of walk budget, so the cap below + // spreads over every group instead of replaying the same prefix. + rotate_sweep_to_cursor(namespace_scratch, self.partition_walk_cursor.get()); // Pre-pass: issue every group's pending superblock write CONCURRENTLY. // A cluster-wide view change makes every group on this shard need one in @@ -6890,12 +6939,45 @@ where // mid-sweep is seen on the next tick, the same latency a capped arm // already accepts. let mut transfers_inflight: Option = None; + // Live repair sessions seen this pass, published at the end for the arm + // fn's concurrency cap. + let mut repairs_live = 0usize; + // Repair sessions this sweep has opened, against + // `PARTITION_REPAIR_ARMS_PER_TICK_MAX`. + let mut repair_arms = 0usize; + // Commit walks this sweep has run, against + // `PARTITION_WALKS_PER_TICK_MAX`. + let mut walks = 0usize; + // First group the WALK budget turned away, which becomes the next + // sweep's starting point. Recorded per SWEEP, not per group: the cursor + // only has to name where the budget ran out, and every group after it + // is reached on the next pass by the rotation above. + // + // The walk cap alone, because only its eligible set regenerates: a + // walked group is walk-stalled again on the next produce, so a fixed + // start would re-spend the budget on the same prefix forever. An ARMED + // group leaves the gap-stopped set for the life of its session, so the + // arm cap drains its own queue in namespace order with no cursor, and + // letting an arm deferral move this one would pull the walk's resume + // point backwards and break the `ceil(groups / cap)` bound below. + // + // It always advances: the walk budget is fresh at the group the sweep + // starts on, so the first group can never be the deferred one, and a + // cursor that stood still would re-skip the same tail forever. + let mut walk_cursor: Option = None; let mut fatal: Option = None; for namespace in namespace_scratch.drain(..) { let Some(partition) = partitions.get_by_ns(&namespace) else { continue; }; + // Ahead of the fence check and every `continue` below: the count is + // the only record those prepares existed, and a partition that + // fences here never ticks again. + let gap_drops = partition.take_prepare_gap_drops(); + if gap_drops > 0 { + self.metrics.record_partition_prepare_gap_drops(gap_drops); + } // A fenced partition must not tick: its consensus would emit // view-scoped sends for a log the cluster has already passed. if let Some(fault) = partition.fatal() { @@ -6999,7 +7081,7 @@ where ); continue; } - partition.repair.as_mut().and_then(|session| { + let due = partition.repair.as_mut().and_then(|session| { if !consensus_normal { return None; } @@ -7016,11 +7098,51 @@ where cluster, self_id, )) - }) + }); + // A session pins its peer and fences every arming site while it + // stands, so a peer that cannot answer wedges the group harder + // than having no session at all -- and the gap-stopped-primary + // rotation above can pick a peer that is simply down. Past the + // budget the session is dropped and re-armed one step around + // the ring; an ordinary lost frame is re-requested long before + // that. + due.map(|stalled| (stalled, partition.burn_repair_attempt())) }; - if let Some((peer, nonce, from_op, to_op, cluster, self_id)) = stalled + if let Some(((peer, nonce, from_op, to_op, cluster, self_id), rotate)) = stalled && from_op <= to_op { + if rotate { + let Some(partition) = partitions.get_mut_by_ns(&namespace) else { + continue; + }; + let consensus = partition.consensus(); + let primary = consensus.primary_index(consensus.view()); + let next_peer = + next_transfer_peer(self_id, peer, consensus.replica_count(), primary); + tracing::warn!( + shard = self.id, + namespace_raw = namespace.inner(), + peer, + next_peer, + from_op, + to_op, + "partition repair stalled past its retry budget; re-arming from \ + another replica" + ); + partition.repair = None; + partition.note_repair_progress(); + if next_peer == peer { + // The ring had nobody else to offer (a solo group, or a + // two-replica group whose only peer is the one that + // went quiet). Dropping the session is still the right + // move: it unfences the detector, which re-arms after + // its debounce and logs the state each interval. + continue; + } + self.maybe_request_partition_repair(partition, next_peer) + .await; + continue; + } tracing::info!( shard = self.id, namespace_raw = namespace.inner(), @@ -7041,6 +7163,160 @@ where .await; } + // Level-triggered gap detector. Every other partition arming site + // is edge-triggered and the edges are starvable: the commit-heartbeat + // backstop needs `CommitOutcome::Advanced`, and a follower has + // already advanced `commit_max` from each prepare header in + // `replicate_preflight` before the gap check dropped the prepare, so + // under produce load the heartbeat lands as `Accepted` and the gap + // wedges until an unrelated view change. Those edges stay the fast + // path; this is the ~1s floor under them. + // + // Runs entirely on the shared borrow: the in-flight scan below reads + // every partition on the shard, so it must not run under a `&mut`, + // and `gap_ticks` is a `Cell` for exactly that reason. + let (walk_stalled, arm_peer) = { + let Some(partition) = partitions.get_by_ns(&namespace) else { + continue; + }; + // Live sessions, tallied on the borrow this sweep already takes + // rather than by a scan: `maybe_request_partition_repair` reads + // the tally to refuse over the concurrency cap, and it is called + // from four edge sites that hold a `&mut` and so could not scan + // at all. Counted here, after the stall block above has cleared + // whatever finished, so the tally the NEXT sweep and every edge + // site in between read is one full pass old at worst. + if partition.repair.is_some() { + repairs_live += 1; + } + let probe = partition_gap_probe(partition); + let walk_stalled = partition_is_walk_stalled(&probe); + // The RATE cap only. The concurrency cap lives in the arm fn, + // which is the funnel every arming site goes through; resolved + // before the debounce either way, so a refusal keeps the group + // due rather than spending its arm. + let may_arm = partition_is_gap_stopped(&probe) + && repair_arms < PARTITION_REPAIR_ARMS_PER_TICK_MAX; + let mut gap_ticks = partition.gap_ticks.get(); + let verdict = drive_partition_gap_debounce( + &probe, + &mut gap_ticks, + gap_debounce_ticks, + may_arm, + ); + partition.gap_ticks.set(gap_ticks); + let arm_peer = match verdict { + GapArm::NotDue | GapArm::Deferred => None, + GapArm::Arm => { + let consensus = partition.consensus(); + let self_id = consensus.replica(); + let primary = consensus.primary_index(consensus.view()); + // A gap-stopped PRIMARY cannot ask itself, and leaving + // it to warn wedged the group: no edge-triggered site + // re-drives a primary's own hole, and the next op to + // commit walks `advance_commit_min` into its sequential + // assert. Any replica in `Normal` or `ViewChange` serves + // `RequestPrepares`, and a primary's window is its + // COMMITTED prefix (the suffix widening needs a pending + // view log, which a settled primary has none of), so a + // peer holding those ops holds them identically. + // + // The pick is positional, not liveness-aware: a dead + // choice leaves the session re-requesting on the stall + // timer, which is where a repair abandon budget (what + // `burn_transfer_attempt` gives transfers) would rotate + // it. Still strictly better than the warn this replaced, + // which recovered nothing at all. + let peer = if primary == self_id { + next_transfer_peer(self_id, self_id, consensus.replica_count(), primary) + } else { + primary + }; + if peer == self_id { + // Solo group: the rotation had nobody to return. + // Restart the debounce so this repeats at its + // interval rather than every tick. + partition.gap_ticks.set(0); + tracing::warn!( + shard = self.id, + namespace_raw = namespace.inner(), + commit_min = probe.commit_min, + commit_max = probe.commit_max, + "partition is gap-stopped below its own commit frontier with no \ + peer to repair from" + ); + None + } else { + Some(peer) + } + } + }; + (walk_stalled, arm_peer) + }; + if let Some(peer) = arm_peer { + let Some(partition) = partitions.get_mut_by_ns(&namespace) else { + continue; + }; + // Logged by `maybe_request_partition_repair` at info, with the + // same fields plus the window it settled on. A refusal there + // (the concurrency cap, or a guard the probe cannot see) spends + // no rate budget and leaves the debounce satisfied, so the group + // is due again next pass. + if self.maybe_request_partition_repair(partition, peer).await { + repair_arms += 1; + } + } + + // Capped like the repair arm, and for the same reason: a node-wide + // rejoin leaves every group on the shard walk-stalled in the same + // tick, and each walk reaches a segment flush. Undebounced, though + // -- the predicate guarantees the walk finds at least the next op, + // so it cannot spin: `partition_is_walk_stalled` reads residency off + // `op_to_storage_offset` while the walk reads `headers`, and those + // two are written and cleared together (see `Journal::holds_op`), so + // a group the predicate admits has an op for the walk to take. + if walk_stalled { + if walks >= PARTITION_WALKS_PER_TICK_MAX { + // Deferred, not dropped: this group becomes the next + // sweep's starting point, so a shard with more owed walks + // than budget drains them round-robin. Without the cursor + // the leading groups would take the whole budget every + // pass and the tail would keep its committed ops resident + // indefinitely. + walk_cursor.get_or_insert(namespace); + } else { + let config = partitions.config(); + let Some(partition) = partitions.get_mut_by_ns(&namespace) else { + continue; + }; + let consensus = partition.consensus(); + // Debug, not info: an in-flight repair journals bodies + // without walking them, so this is the steady state for the + // whole duration of a rejoin and would be one line per group + // per tick. + tracing::debug!( + shard = self.id, + namespace_raw = namespace.inner(), + commit_min = consensus.commit_min(), + commit_max = consensus.commit_max(), + "partition commit walk parked over resident committed ops; resuming" + ); + partition.commit_journal(config).await; + walks += 1; + // Re-read, because the aggregate above was sampled BEFORE + // this walk: a local commit failure fences the partition + // here, and reporting the stale verdict would let the pump + // keep serving a divergent replica until the next tick + // noticed. + if let Some(fault) = partition.fatal() { + if fatal.is_none() { + fatal = Some(fault.clone()); + } + continue; + } + } + } + // Transfer stall retry: descriptor and chunk frames are // fire-and-forget, so a lost one must not wedge the session (and // the rejoin behind it) forever. Budget-bounded: a peer that died @@ -7129,6 +7405,17 @@ where } } + // The first group the walk cap turned away is where the next sweep + // enters the group set; a sweep that turned nobody away resets to the + // front, since a stale cursor would keep re-entering at a point no cap + // chose. + self.partition_walk_cursor.set(walk_cursor); + // Republished from this pass, arms included: the arm fn has been + // incrementing it as sessions opened, and this is the recount that + // retires whatever completed. + self.partition_repairs_inflight + .set(repairs_live + repair_arms); + fatal } @@ -7965,13 +8252,48 @@ where /// `maybe_request_metadata_repair`: no-op unless Normal, not /// transferring, behind the frontier, and no session live. #[allow(clippy::future_not_send)] - async fn maybe_request_partition_repair(&self, partition: &mut IggyPartition, peer: u8) + /// Open a journal-repair session against `peer`, if this partition needs + /// one and the shard has room for it. `true` when a session was recorded. + /// + /// THE funnel: the tick sweep and the four edge-triggered sites + /// (`StartView` adoption, the commit heartbeat, the post-transfer tail, the + /// post-repair walk) all arrive here, so the concurrency ceiling and the + /// debounce reset live here rather than in any one caller. A node-wide view + /// change drives `on_start_view` for every group at once, which is exactly + /// the burst the sweep's own rate cap would not see. + async fn maybe_request_partition_repair( + &self, + partition: &mut IggyPartition, + peer: u8, + ) -> bool where B: MessageBus, { let consensus = partition.consensus(); if !consensus.is_normal() || consensus.is_transferring() || partition.repair.is_some() { - return; + return false; + } + // Read, never scanned: the tally is republished by each sweep (see + // `tick_partitions`), so it is at worst one tick stale, which is all a + // concurrency ceiling needs. Callers here hold a `&mut` to one + // partition, so a scan over the plane would alias it. + if self.partition_repairs_inflight.get() >= PARTITION_REPAIRS_INFLIGHT_MAX { + tracing::debug!( + shard = self.id, + namespace_raw = consensus.group(), + peer, + "partition repair not armed: shard is at its live-session ceiling" + ); + return false; + } + // Never against self. The session is recorded below BEFORE the send, + // and a self-addressed `RequestPrepares` cannot be delivered (the + // replica registry holds no entry for this node), so the session would + // stand forever: `repair_finished` needs a `commit_min` only the reply + // can advance, the stall retry re-sends to the same peer, and + // `repair.is_some()` fences every other arming site meanwhile. + if peer == consensus.replica() { + return false; } // The window ends at the group head when suffix bodies are missing, // not at the commit point. A backup that adopted a StartView holds @@ -7986,27 +8308,29 @@ where let commit_lag = consensus.commit_min() < commit_to_op; let head = consensus.sequencer().current_sequence(); if !commit_lag && head <= commit_to_op { - return; + return false; } - let canonical_suffix = consensus - .with_pending_view_log(|pending| pending_covers_suffix(pending, commit_to_op, head)) - .unwrap_or(false); - let missing_suffix = canonical_suffix - && !partition - .log - .journal() - .inner - .repaired_window_shape(commit_to_op, head) - .complete; + let missing_suffix = partition_missing_suffix(partition); if !commit_lag && !missing_suffix { - return; + return false; } let nonce = iggy_common::random_id::get_uuid(); let from_op = consensus.commit_min() + 1; - let fetch_to_op = if missing_suffix { head } else { commit_to_op }; + // The widening is for the replica that is LEVEL with the commit + // frontier and short of bodies above it. Widening while a commit lag + // stands would ask for `(commit_min, head]` -- the whole committed + // prefix this replica already holds, refetched -- and the suffix is + // reached anyway once the lag closes, on the arm after it. + let fetch_to_op = if commit_lag { commit_to_op } else { head }; let cluster = consensus.cluster(); let self_id = consensus.replica(); let namespace = consensus.group(); + // Spent here for the same reason the ceiling is: the four edge-triggered + // sites never touch it, so a short edge-armed repair would leave the + // count saturated and hand the next real gap an arm on its first tick. + partition.gap_ticks.set(0); + self.partition_repairs_inflight + .set(self.partition_repairs_inflight.get() + 1); partition.repair = Some(partitions::RepairSession { nonce, view: consensus.view(), @@ -8023,6 +8347,7 @@ where from_op, commit_to_op, fetch_to_op, + peer, "partition behind the group frontier; requesting repair" ); self.send_request_prepares( @@ -8035,6 +8360,7 @@ where namespace, ) .await; + true } /// Receiver side of a partition descriptor: accept the manifest, adopt @@ -9286,6 +9612,302 @@ fn repair_serve_ceiling(requested_to_op: u64, commit_max: u64, head: u64) -> u64 requested_to_op.min(commit_max.max(head)) } +/// Repair sessions the partition tick sweep will OPEN per pass. +/// +/// The RATE half of the pair: it spreads the cost of OPENING sessions, while +/// [`PARTITION_REPAIRS_INFLIGHT_MAX`] bounds how many stand at once. One arm is +/// a `RequestPrepares` plus a repair stream the serving peer walks +/// synchronously, and a node-wide gap (a rejoin, a lossy link) makes every group +/// on this shard due in the same tick. +/// +/// Over-cap groups stay due with their debounce satisfied and arm on a later +/// pass. No cursor: an armed group leaves the gap-stopped set for the life of +/// its session, so the queue drains in namespace order on its own, and letting +/// a deferred arm move the walk cursor would pull the walk's resume point +/// backwards. +const PARTITION_REPAIR_ARMS_PER_TICK_MAX: usize = 3; + +/// Live repair sessions this shard will hold at once. +/// +/// The concurrency ceiling the rate cap above is not: without it a node-wide +/// rejoin puts every group's stream in flight within `groups / arms` ticks, and +/// each one is a window the SERVING peer walks on its own pump, so the cost +/// lands on a node that has nothing wrong with it. Sized at twice +/// [`IggyShard::PARTITION_TRANSFERS_INFLIGHT_MAX`]: a repair streams journal +/// entries the peer already holds resident, where a transfer reads and hashes +/// whole segments, so more of them fit in the same serving budget. +/// +/// Applied inside `maybe_request_partition_repair`, not at any one caller: the +/// four edge-triggered sites arm from frame handlers, and a node-wide view +/// change drives `on_start_view` for every group on the shard at once, which no +/// per-sweep budget can see. Over-cap groups stay gap-stopped with their +/// debounce satisfied, so they arm as sessions complete. +const PARTITION_REPAIRS_INFLIGHT_MAX: usize = 8; + +/// Commit walks the partition tick sweep will RUN per pass. +/// +/// Same correlated-fan-out argument as the repair arm, and the walk is the +/// costlier half: `commit_journal` reaches `commit_messages`, which flushes a +/// segment and fsyncs under `enforce_fsync`. +/// +/// The two caps together are what bound the tick: this one bounds how many +/// groups a sweep walks, [`partitions::COMMIT_WALK_OPS_MAX`] bounds how far +/// each walk goes (for every caller of `commit_journal`, not just this one), +/// and the product is the sweep's worst case. Deliberately NOT the +/// superblock pre-pass's number: that one runs its fan-out CONCURRENTLY under +/// `join_all` and drains every group in the same body, while these walks are +/// serial and what is over budget waits for the next tick. +/// +/// Capping cannot starve a partition: the walk carries no debounce counter and +/// clears its own predicate (a walk either advances `commit_min` or fences the +/// partition), and [`rotate_sweep_to_cursor`] resumes the next sweep at the +/// first group this one turned away, so the eligible set drains in +/// `ceil(groups / cap)` ticks however many groups are owed at once. +const PARTITION_WALKS_PER_TICK_MAX: usize = 16; + +/// Floor under the gap detector's debounce, in ticks. +/// +/// The debounce reads `[cluster] repair_gap_debounce_interval`, and +/// `duration_to_ticks` floors that at one tick. One tick of lag is ordinary +/// pipelining, so without a floor of its own a shortened interval would arm +/// repair against a single reordered prepare. +/// +/// Public because it bounds what that operator knob can do: gap recovery starts +/// after `max(repair_gap_debounce_interval, this)`, which the `[cluster]` +/// documentation states. +pub const PARTITION_GAP_DEBOUNCE_TICKS_MIN: u32 = 50; + +/// What the tick sweep reads off one partition to decide whether it is +/// gap-stopped. Split out so the guards, the debounce and the per-tick cap are +/// testable without a shard, a bus, or a journal. +/// +/// The flags are independent readings of one instant, not states of one +/// machine, and the exhaustive predicate test below enumerates them as such, so +/// the lint's two-variant enums would only rename `true` and `false`. +#[allow(clippy::struct_excessive_bools)] +#[derive(Debug, Clone, Copy)] +struct GapProbe { + normal: bool, + transferring: bool, + /// Whether a repair session, a transfer, or a scheduled transfer re-arm + /// already owns this partition's recovery. Arming a second one would race + /// it, or defeat the re-arm's backoff as `arm_partition_transfer` documents. + recovery_owned: bool, + commit_min: u64, + commit_max: u64, + /// Whether `commit_min + 1` is resident in the local journal. + /// + /// Read only when the guards above already hold, and `false` otherwise: + /// both predicates test the lag first, so a probe that fails it is + /// answered without touching the journal at all. See + /// [`partition_gap_probe`]. + next_op_resident: bool, + /// Whether this replica adopted suffix headers above `commit_max` whose + /// bodies never arrived. Its own recovery shape, disjoint from the lag + /// below the frontier: the group cannot gather quorum for that suffix until + /// the bodies land, and the only other site that notices is the single + /// `on_start_view` edge that adopted them. See [`partition_missing_suffix`]. + missing_suffix: bool, +} + +/// Whether this replica holds committed ops it cannot walk to, because the op +/// one past its commit frontier is missing from its journal. +/// +/// The journal-hole half is not redundant: a follower advances `commit_max` +/// from every prepare header in `replicate_preflight`, so `commit_min < +/// commit_max` is transiently true on every healthy pipelined tick and a bare +/// lag test would arm repair against ordinary produce. +const fn partition_is_gap_stopped(probe: &GapProbe) -> bool { + if !probe.normal || probe.transferring || probe.recovery_owned { + return false; + } + // The lag decides first, and a walkable lag wins outright. A replica that + // is BOTH short of a suffix and behind its own frontier would otherwise arm + // over `(commit_min, head]` -- refetching a committed prefix it already + // holds resident -- and would claim this predicate and the walk at once. + // The walk closes the lag within a tick or two (the suffix cannot commit + // meanwhile, so `commit_max` stands still), and the suffix arms on the pass + // after that. + if probe.commit_min < probe.commit_max { + return !probe.next_op_resident; + } + probe.missing_suffix +} + +/// The gap predicate's disjoint sibling, not its complement: everything the +/// walk needs is resident, it just never ran (a heartbeat carrying a known +/// commit is `Accepted`, and an idle group offers no other edge). +/// +/// The two split on `next_op_resident` while a lag stands, and +/// [`partition_is_gap_stopped`] defers to that split even for a missing suffix, +/// so they cannot both hold. Both are false whenever a shared guard fails. Not +/// gated on `recovery_owned`: repair fetches bodies without walking them, so +/// gating parks the walk all session. +const fn partition_is_walk_stalled(probe: &GapProbe) -> bool { + probe.normal + && !probe.transferring + && probe.commit_min < probe.commit_max + && probe.next_op_resident +} + +/// What the debounce says about one partition on one sweep. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum GapArm { + /// Not gap-stopped, or gap-stopped for less than the debounce. + NotDue, + /// Due, but this sweep's arm budget is spent. The debounce stays satisfied, + /// so the group is due again on the next pass rather than serving a fresh + /// interval. It moves no cursor: the sweep resumes where the WALK budget + /// ran out, and arms drain their own queue as sessions open. + Deferred, + /// Open a repair session now. + Arm, +} + +/// Count one sweep tick against `gap_ticks` and answer whether this partition +/// may arm repair now. +/// +/// Level-triggered, because every edge-triggered arming site is starvable: the +/// commit-heartbeat backstop fires only on `CommitOutcome::Advanced`, and under +/// sustained produce the prepares consume the advance in preflight before the +/// gap check drops them, so the heartbeat lands as `Accepted` and the gap wedges +/// until an unrelated view change. +/// +/// `budget_available` is the sweep's per-tick arm rate; the live-session +/// ceiling is applied by `maybe_request_partition_repair`, which every arming +/// site funnels through. A refused arm keeps its debounce satisfied rather than +/// starting over, so the group arms on the next pass with a slot free. Spending it is `maybe_request_partition_repair`'s job, which resets +/// `gap_ticks` for EVERY arming site, not just this one: an edge-armed repair +/// that completes before the next sweep would otherwise leave the count +/// saturated and hand the next gap an arm on its first tick. +const fn drive_partition_gap_debounce( + probe: &GapProbe, + gap_ticks: &mut u32, + debounce_ticks: u32, + budget_available: bool, +) -> GapArm { + if !partition_is_gap_stopped(probe) { + *gap_ticks = 0; + return GapArm::NotDue; + } + let debounce_ticks = if debounce_ticks < PARTITION_GAP_DEBOUNCE_TICKS_MIN { + PARTITION_GAP_DEBOUNCE_TICKS_MIN + } else { + debounce_ticks + }; + *gap_ticks = gap_ticks.saturating_add(1); + if *gap_ticks < debounce_ticks { + return GapArm::NotDue; + } + if budget_available { + GapArm::Arm + } else { + GapArm::Deferred + } +} + +/// Rotate a sweep's namespace snapshot so it resumes at `cursor`. +/// +/// The per-tick caps are what make this necessary: the snapshot is in ascending +/// namespace order, so a shard whose leading groups stay eligible would spend +/// the whole budget on them every pass and never reach the tail. `cursor` names +/// the first group a cap turned away last pass, so every eligible group is +/// served within `ceil(groups / cap)` sweeps. +/// +/// A cursor whose namespace was removed meanwhile resumes at its successor, and +/// one past the last namespace wraps to the front. `None` means the previous +/// sweep turned nobody away. +fn rotate_sweep_to_cursor(namespaces: &mut [IggyNamespace], cursor: Option) { + let Some(cursor) = cursor else { + return; + }; + debug_assert!( + namespaces.is_sorted(), + "the sweep snapshot must be in namespace order for the cursor to resume in it", + ); + // `partition_point` answers in `0..=len`, and `rotate_left(len)` is the + // no-op that wraps a cursor past the last namespace back to the front. + namespaces.rotate_left(namespaces.partition_point(|namespace| *namespace < cursor)); +} + +/// Whether this replica holds adopted suffix HEADERS above `commit_max` whose +/// bodies never arrived. +/// +/// The shape `maybe_request_partition_repair` widens its window for, read here +/// so the sweep's detector and the arm agree by construction. A backup that +/// adopted a `StartView` withholds its ack for those ops until the body is +/// journaled, and the primary's retransmit is dropped by the backup gap check +/// because adoption already advanced the sequencer to the head: nothing else +/// delivers them, and the group wedges one op below its head. +/// +/// Ordered cheapest-first, because it runs per group per tick: no suffix at all +/// is one comparison, and a suffix nobody adopted is one `Option` check. Only a +/// group that has both pays the header-vec walk. +fn partition_missing_suffix(partition: &IggyPartition) -> bool +where + B: MessageBus, + SB: SuperblockStore, +{ + let consensus = partition.consensus(); + let commit_max = consensus.commit_max(); + let head = consensus.sequencer().current_sequence(); + if head <= commit_max { + return false; + } + let canonical_suffix = consensus + .with_pending_view_log(|pending| pending_covers_suffix(pending, commit_max, head)) + .unwrap_or(false); + canonical_suffix + && !partition + .log + .journal() + .inner + .repaired_window_shape(commit_max, head) + .complete +} + +/// Read the gap probe off a live partition. +fn partition_gap_probe(partition: &IggyPartition) -> GapProbe +where + B: MessageBus, + SB: SuperblockStore, +{ + let consensus = partition.consensus(); + let commit_min = consensus.commit_min(); + let commit_max = consensus.commit_max(); + let recovery_owned = partition.transfer.is_some() + || partition.transfer_rearm.is_some() + || partition.repair.is_some(); + let normal = consensus.is_normal(); + let transferring = consensus.is_transferring(); + // Residency last, and only once the guards both predicates share already + // hold. This runs for every group on the shard on every tick, and the + // caught-up steady state (`commit_min == commit_max`) would otherwise pay + // a journal lookup whose answer both predicates discard. + let next_op_resident = normal + && !transferring + && commit_min < commit_max + && partition + .log + .journal() + .inner + .holds_op(commit_min.saturating_add(1)); + // Same discipline, one guard deeper: the suffix test walks the header vec, + // so it runs only for a group that HAS an unfinished suffix and already + // owes nothing else. + let missing_suffix = + normal && !transferring && !recovery_owned && partition_missing_suffix(partition); + GapProbe { + normal, + transferring, + recovery_owned, + commit_min, + commit_max, + next_op_resident, + missing_suffix, + } +} + /// Whether the parked `StartView` log names every op in the uncommitted /// suffix `(commit_max, head]`, in descending order. Only this canonical list /// makes fetching bodies above the commit point safe. @@ -10519,3 +11141,574 @@ mod superblock_fail_stop_tests { assert!(superblock_wedged(121, 120)); } } + +#[cfg(test)] +mod sweep_scheduler_tests { + //! Fairness of the partition sweep's per-tick caps. + //! + //! The caps exist so a node-wide rejoin cannot put every group's walk (each + //! reaching a segment flush) into one tick body. They are only ACCEPTABLE + //! because the sweep resumes where the WALK budget ran out: the snapshot is + //! in ascending namespace order, so a fixed start would spend every pass on + //! the same leading groups and leave the tail holding committed ops it can + //! never walk to. + //! + //! Both budgets are modelled, because they share the sweep: the arm cap + //! runs first and can turn groups away ahead of the walks, and the walk + //! fairness bound has to survive that. What keeps them independent is that + //! the arm cap moves no cursor, which is the property these runs pin. + + use super::{ + IggyNamespace, PARTITION_REPAIR_ARMS_PER_TICK_MAX, PARTITION_REPAIRS_INFLIGHT_MAX, + PARTITION_WALKS_PER_TICK_MAX, rotate_sweep_to_cursor, + }; + + /// Comfortably past `PARTITION_WALKS_PER_TICK_MAX`, and deliberately not a + /// multiple of it, so the wrap lands mid-snapshot on most passes. + const GROUPS: usize = 40; + + /// Every third group is gap-stopped rather than walk-stalled. The two + /// predicates are disjoint below the commit frontier, so a group is in one + /// set or the other, and this spreads the arm-capped ones through the + /// snapshot instead of parking them in one block. + const fn is_gap_stopped(partition: usize) -> bool { + partition.is_multiple_of(3) + } + + fn namespaces() -> Vec { + (0..GROUPS) + .map(|partition| IggyNamespace::new(1, 1, partition)) + .collect() + } + + /// Per-group tallies, indexed by partition id. No `Default`: empty vecs + /// next to a `new` that sizes them by `GROUPS` would panic on first index. + struct Served { + walks: Vec, + arms: Vec, + } + + impl Served { + fn new() -> Self { + Self { + walks: vec![0; GROUPS], + arms: vec![0; GROUPS], + } + } + } + + /// Sweeps a session stays open for before it completes. Long enough that + /// the live-session ceiling actually binds (it is reached on the third + /// sweep at three arms a pass), so the model spends passes waiting on + /// capacity the way a real rejoin does. + const SESSION_SWEEPS: u32 = 6; + + /// Repair state per group, standing in for `partition.repair` (which fences + /// a group out of the gap-stopped set while it stands) and for the hole + /// itself (which a completed session closes, so the group stops being + /// eligible rather than arming again). + struct Repairs { + live: Vec>, + done: Vec, + } + + impl Repairs { + fn new() -> Self { + Self { + live: vec![None; GROUPS], + done: vec![false; GROUPS], + } + } + + /// Age every open session by one sweep, closing the gaps that finish. + fn retire(&mut self) { + for (partition, session) in self.live.iter_mut().enumerate() { + let Some(remaining) = session else { + continue; + }; + *remaining -= 1; + if *remaining == 0 { + *session = None; + self.done[partition] = true; + } + } + } + + fn live_count(&self) -> usize { + self.live.iter().filter(|session| session.is_some()).count() + } + + fn is_due(&self, partition: usize) -> bool { + is_gap_stopped(partition) && !self.done[partition] && self.live[partition].is_none() + } + } + + /// One sweep of `tick_partitions`' scheduling: rotate to the carried walk + /// cursor, retire whatever finished, then spend the rate cap, the live + /// ceiling and the walk budget in snapshot order. Answers with the cursor + /// this sweep leaves behind. + fn sweep( + cursor: Option, + served: &mut Served, + repairs: &mut Repairs, + ) -> Option { + repairs.retire(); + let mut snapshot = namespaces(); + rotate_sweep_to_cursor(&mut snapshot, cursor); + let mut walks = 0; + let mut arms = 0; + let mut walk_deferred = None; + for namespace in snapshot { + let partition = namespace.partition_id(); + if is_gap_stopped(partition) { + // Both ceilings, in the order the sweep applies them: the rate + // cap it counts itself, then the live-session count the arm fn + // refuses on. + if !repairs.is_due(partition) + || arms >= PARTITION_REPAIR_ARMS_PER_TICK_MAX + || repairs.live_count() >= PARTITION_REPAIRS_INFLIGHT_MAX + { + continue; + } + served.arms[partition] += 1; + repairs.live[partition] = Some(SESSION_SWEEPS); + arms += 1; + continue; + } + if walks < PARTITION_WALKS_PER_TICK_MAX { + served.walks[partition] += 1; + walks += 1; + } else { + walk_deferred.get_or_insert(namespace); + } + } + walk_deferred + } + + #[test] + fn given_more_eligible_groups_than_the_walk_budget_when_swept_should_reach_every_one() { + let mut served = Served::new(); + let mut repairs = Repairs::new(); + let mut cursor = None; + let walk_eligible = (0..GROUPS).filter(|p| !is_gap_stopped(*p)).count(); + let passes = walk_eligible.div_ceil(PARTITION_WALKS_PER_TICK_MAX); + for _ in 0..passes { + cursor = sweep(cursor, &mut served, &mut repairs); + } + let unserved: Vec<_> = (0..GROUPS) + .filter(|partition| !is_gap_stopped(*partition) && served.walks[*partition] == 0) + .collect(); + assert!( + unserved.is_empty(), + "{} of {walk_eligible} walk-eligible groups never had their walk run in \ + {passes} sweeps (partitions {unserved:?}); the budget is being spent on \ + the same leading groups every pass", + unserved.len() + ); + } + + #[test] + fn given_a_sustained_backlog_when_swept_should_keep_every_group_within_one_walk() { + // Sustained, because the starvation this guards against only shows over + // many passes: one sweep serves the head no matter how the cursor moves. + // Walk-eligible groups stay eligible throughout (a walked group is + // walk-stalled again on the next produce), so a fair scheduler owes them + // walks in round-robin and none may drift a full round behind. + let mut served = Served::new(); + let mut repairs = Repairs::new(); + let mut cursor = None; + for _ in 0..10 * GROUPS { + cursor = sweep(cursor, &mut served, &mut repairs); + } + let walks: Vec = (0..GROUPS) + .filter(|partition| !is_gap_stopped(*partition)) + .map(|partition| served.walks[partition]) + .collect(); + let most = walks.iter().max().copied().unwrap_or_default(); + let fewest = walks.iter().min().copied().unwrap_or_default(); + assert!( + most - fewest <= 1, + "walks are not spread evenly: one group got {most}, another {fewest}" + ); + } + + #[test] + fn given_a_capped_arm_backlog_when_swept_should_arm_every_gap_stopped_group() { + // The arm caps carry no cursor because arming REMOVES a group from the + // eligible set: the front of the queue drains, so the tail is reached + // without one. If that ever stops holding, this run wedges. + // + // Both ceilings are modelled, so the run also pins that the live-session + // cap only DELAYS: a group refused for capacity keeps its debounce and + // arms once a session retires. Bounded by the slower of the two, plus a + // session's life for the last batch to have somewhere to go. + let mut served = Served::new(); + let mut repairs = Repairs::new(); + let mut cursor = None; + let gap_stopped = (0..GROUPS).filter(|p| is_gap_stopped(*p)).count(); + let by_rate = gap_stopped.div_ceil(PARTITION_REPAIR_ARMS_PER_TICK_MAX); + let by_capacity = + gap_stopped.div_ceil(PARTITION_REPAIRS_INFLIGHT_MAX) * SESSION_SWEEPS as usize; + for _ in 0..by_rate.max(by_capacity) + SESSION_SWEEPS as usize { + cursor = sweep(cursor, &mut served, &mut repairs); + } + let unarmed: Vec<_> = (0..GROUPS) + .filter(|partition| is_gap_stopped(*partition) && served.arms[*partition] == 0) + .collect(); + assert!( + unarmed.is_empty(), + "gap-stopped groups {unarmed:?} never armed; the arm caps are queueing \ + behind the same prefix and need a cursor after all" + ); + assert!( + served.arms.iter().all(|arms| *arms <= 1), + "a group armed twice while its first session was still open" + ); + } + + #[test] + fn given_the_live_session_ceiling_when_swept_should_never_exceed_it() { + // The ceiling exists because each session is a window the SERVING peer + // walks on its own pump; the rate cap alone would let a rejoin put every + // group's stream in flight within `groups / 3` passes. + let mut served = Served::new(); + let mut repairs = Repairs::new(); + let mut cursor = None; + for _ in 0..10 * GROUPS { + cursor = sweep(cursor, &mut served, &mut repairs); + let live = repairs.live_count(); + assert!( + live <= PARTITION_REPAIRS_INFLIGHT_MAX, + "{live} sessions live at once, past the ceiling of \ + {PARTITION_REPAIRS_INFLIGHT_MAX}" + ); + } + } + + #[test] + fn given_a_cursor_naming_a_removed_namespace_when_rotating_should_resume_at_its_successor() { + // The group the cap turned away can be deleted before the next sweep; + // the resume point is then the next namespace above it, not the front. + let mut snapshot = namespaces(); + let removed = snapshot.remove(20); + rotate_sweep_to_cursor(&mut snapshot, Some(removed)); + assert_eq!( + snapshot.first().copied(), + Some(IggyNamespace::new(1, 1, 21)) + ); + } + + #[test] + fn given_a_cursor_past_every_namespace_when_rotating_should_wrap_to_the_front() { + let mut snapshot = namespaces(); + rotate_sweep_to_cursor(&mut snapshot, Some(IggyNamespace::new(1, 1, GROUPS))); + assert_eq!(snapshot, namespaces()); + } + + #[test] + fn given_no_deferral_last_pass_when_rotating_should_start_at_the_front() { + let mut snapshot = namespaces(); + rotate_sweep_to_cursor(&mut snapshot, None); + assert_eq!(snapshot, namespaces()); + } +} + +#[cfg(test)] +mod gap_detector_tests { + //! The level-triggered repair arm the partition tick sweep runs. + //! + //! Its whole reason to exist is that the edge-triggered arming sites are + //! starvable, so the guards it shares with them and the debounce that keeps + //! it off healthy traffic are the parts worth pinning. + + use super::{ + GapArm, GapProbe, PARTITION_GAP_DEBOUNCE_TICKS_MIN, drive_partition_gap_debounce, + partition_is_gap_stopped, partition_is_walk_stalled, + }; + + const DEBOUNCE: u32 = 100; + + /// A gap-stopped follower: committed through op 10, walkable only to 5, + /// because op 6 is not in its journal. + const fn gap_stopped() -> GapProbe { + GapProbe { + normal: true, + transferring: false, + recovery_owned: false, + commit_min: 5, + commit_max: 10, + next_op_resident: false, + missing_suffix: false, + } + } + + /// A walk-stalled follower: the same lag, but op 6 IS in its journal, so + /// nothing needs fetching and the walk just has to run. + fn walk_stalled() -> GapProbe { + GapProbe { + next_op_resident: true, + ..gap_stopped() + } + } + + /// A follower level with its commit frontier that holds adopted suffix + /// headers whose bodies never arrived. + fn missing_suffix() -> GapProbe { + GapProbe { + commit_min: 10, + missing_suffix: true, + ..gap_stopped() + } + } + + #[test] + fn given_a_lagging_follower_with_the_next_op_resident_when_probed_should_not_be_gap_stopped() { + // The half that keeps the predicate honest. A follower advances + // commit_max from every prepare header in preflight, so commit_min < + // commit_max is transiently true on any pipelined tick; without the + // journal-hole test the driver would request repair against ordinary + // produce, on every partition, forever. + assert!(!partition_is_gap_stopped(&walk_stalled())); + assert!(partition_is_gap_stopped(&gap_stopped())); + } + + #[test] + fn given_a_caught_up_follower_when_probed_should_not_be_gap_stopped() { + let caught_up = GapProbe { + commit_min: 10, + ..gap_stopped() + }; + assert!(!partition_is_gap_stopped(&caught_up)); + } + + #[test] + fn given_adopted_suffix_headers_without_bodies_when_probed_should_be_gap_stopped() { + // The shape a commit-frontier test alone misses: both marks sit below + // the head, so there is no lag to see, and the only other site that + // notices is the single `on_start_view` edge that adopted the headers. + // Left out, that class hangs until an unrelated view change. + assert!(partition_is_gap_stopped(&missing_suffix())); + assert!( + !partition_is_gap_stopped(&GapProbe { + missing_suffix: false, + ..missing_suffix() + }), + "a caught-up replica with a complete suffix has nothing to repair" + ); + } + + #[test] + fn given_a_replica_outside_normal_status_when_probed_should_not_be_gap_stopped() { + // A view change owns the log while it runs, and `maybe_request_partition_repair` + // refuses outside Normal anyway; arming here would only burn a nonce. + for probe in [gap_stopped(), missing_suffix()] { + assert!(!partition_is_gap_stopped(&GapProbe { + normal: false, + ..probe + })); + assert!(!partition_is_gap_stopped(&GapProbe { + transferring: true, + ..probe + })); + } + } + + #[test] + fn given_recovery_already_owned_when_probed_should_not_be_gap_stopped() { + // A session, a transfer, or a scheduled transfer re-arm all own the + // recovery; a second one would race it or defeat the re-arm's backoff. + for probe in [gap_stopped(), missing_suffix()] { + assert!(!partition_is_gap_stopped(&GapProbe { + recovery_owned: true, + ..probe + })); + } + } + + #[test] + fn given_a_gap_stopped_follower_when_debouncing_should_arm_only_at_the_threshold() { + let probe = gap_stopped(); + let mut gap_ticks = 0; + for tick in 1..DEBOUNCE { + assert_eq!( + drive_partition_gap_debounce(&probe, &mut gap_ticks, DEBOUNCE, true), + GapArm::NotDue, + "armed at tick {tick}, before the debounce elapsed" + ); + } + assert_eq!( + drive_partition_gap_debounce(&probe, &mut gap_ticks, DEBOUNCE, true), + GapArm::Arm + ); + } + + #[test] + fn given_a_debounce_in_progress_when_the_gap_closes_should_reset_the_counter() { + let stopped = gap_stopped(); + let walkable = GapProbe { + next_op_resident: true, + ..stopped + }; + let mut gap_ticks = 0; + for _ in 0..DEBOUNCE - 1 { + drive_partition_gap_debounce(&stopped, &mut gap_ticks, DEBOUNCE, true); + } + assert_eq!(gap_ticks, DEBOUNCE - 1); + + assert_eq!( + drive_partition_gap_debounce(&walkable, &mut gap_ticks, DEBOUNCE, true), + GapArm::NotDue + ); + assert_eq!(gap_ticks, 0, "progress must restart the debounce"); + assert_eq!( + drive_partition_gap_debounce(&stopped, &mut gap_ticks, DEBOUNCE, true), + GapArm::NotDue, + "a fresh gap must serve its own debounce, not inherit the old count" + ); + } + + #[test] + fn given_a_follower_with_resident_committed_ops_when_probed_should_be_walk_stalled() { + assert!(partition_is_walk_stalled(&walk_stalled())); + assert!( + !partition_is_walk_stalled(&gap_stopped()), + "a missing next op is repair's job; a walk over it would stop dead" + ); + } + + #[test] + fn given_a_caught_up_follower_when_probed_should_not_be_walk_stalled() { + let caught_up = GapProbe { + commit_min: 10, + ..walk_stalled() + }; + assert!(!partition_is_walk_stalled(&caught_up)); + } + + #[test] + fn given_a_replica_outside_normal_status_when_probed_should_not_be_walk_stalled() { + let electing = GapProbe { + normal: false, + ..walk_stalled() + }; + assert!(!partition_is_walk_stalled(&electing)); + + // Same gate as the on-commit arm: a walk during a transfer can advance + // commit_min past the incoming frontier. + let installing = GapProbe { + transferring: true, + ..walk_stalled() + }; + assert!(!partition_is_walk_stalled(&installing)); + } + + #[test] + fn given_recovery_already_owned_when_the_next_op_is_resident_should_still_be_walk_stalled() { + // Deliberate: `apply_repaired_prepare` journals without walking, so a + // gated walk would sit parked for the whole session while the resident + // prefix is already applicable. + assert!(partition_is_walk_stalled(&GapProbe { + recovery_owned: true, + ..walk_stalled() + })); + } + + #[test] + fn given_any_probe_when_evaluated_should_never_be_both_gap_stopped_and_walk_stalled() { + // If they ever overlap, one tick both arms repair and walks the window + // it is fetching, and the arm refetches a resident committed prefix. + for normal in [false, true] { + for transferring in [false, true] { + for recovery_owned in [false, true] { + for (commit_min, commit_max) in [(5, 10), (10, 10)] { + for next_op_resident in [false, true] { + for missing_suffix in [false, true] { + let probe = GapProbe { + normal, + transferring, + recovery_owned, + commit_min, + commit_max, + next_op_resident, + missing_suffix, + }; + assert!( + !(partition_is_gap_stopped(&probe) + && partition_is_walk_stalled(&probe)), + "both predicates claim {probe:?}" + ); + } + } + } + } + } + } + } + + #[test] + fn given_a_missing_suffix_over_a_walkable_lag_when_evaluated_should_prefer_the_walk() { + // Arming here would request `(commit_min, head]`: the committed prefix + // this replica already holds resident, refetched, plus the suffix. The + // walk closes the lag first -- `commit_max` cannot move while the suffix + // is short of quorum -- and the suffix arms on the pass after. + let probe = GapProbe { + missing_suffix: true, + ..walk_stalled() + }; + assert!(partition_is_walk_stalled(&probe)); + assert!( + !partition_is_gap_stopped(&probe), + "a walkable lag must win the tick; the suffix arm waits for it to close" + ); + assert!( + partition_is_gap_stopped(&GapProbe { + commit_min: probe.commit_max, + ..probe + }), + "and the same replica arms once the lag is gone" + ); + } + + #[test] + fn given_the_arm_budget_spent_when_debouncing_should_defer_without_losing_the_debounce() { + let probe = gap_stopped(); + let mut gap_ticks = DEBOUNCE; + assert_eq!( + drive_partition_gap_debounce(&probe, &mut gap_ticks, DEBOUNCE, false), + GapArm::Deferred, + "a spent budget must refuse the arm" + ); + assert!( + gap_ticks > DEBOUNCE, + "a refused group stays due; restarting its debounce would push the \ + arm a whole interval out per contended tick" + ); + assert_eq!( + drive_partition_gap_debounce(&probe, &mut gap_ticks, DEBOUNCE, true), + GapArm::Arm, + "the same group arms on the next pass with a slot free" + ); + } + + #[test] + fn given_a_debounce_shorter_than_the_floor_when_driven_should_hold_until_the_floor() { + // `repair_retry_interval` is an operator knob whose primary meaning is + // the stalled-stream retry, and `duration_to_ticks` floors it at one + // tick. One tick of lag is ordinary pipelining, so without a floor of + // its own a shortened retry interval would arm repair against a single + // reordered prepare. + let probe = gap_stopped(); + let mut gap_ticks = 0; + for tick in 1..PARTITION_GAP_DEBOUNCE_TICKS_MIN { + assert_eq!( + drive_partition_gap_debounce(&probe, &mut gap_ticks, 1, true), + GapArm::NotDue, + "a 1-tick debounce armed at tick {tick}, under the floor" + ); + } + assert_eq!( + drive_partition_gap_debounce(&probe, &mut gap_ticks, 1, true), + GapArm::Arm + ); + } +} diff --git a/core/shard/src/metrics.rs b/core/shard/src/metrics.rs index fcf885a097..3e4b1c14fa 100644 --- a/core/shard/src/metrics.rs +++ b/core/shard/src/metrics.rs @@ -80,11 +80,14 @@ pub struct FrameDropLabel { /// (`reason=park_overflow`), a parked frame retired with no client to answer /// (`reason=park_dropped`), an incarnation rejection, or a routing send the /// target inbox refused. A shed client request is answered with a retriable -/// status. A shed prepare may no longer be covered by retransmit once its op -/// reached quorum, but a later `CommitMessage` that advances the backup's -/// frontier arms same-view journal repair. If the primary evicted the range, -/// repair escalates to partition state transfer. The counter therefore signals -/// a data-plane gap or recovery burden, not a requirement for a view change. +/// status, so the client recovers. A shed *prepare* has nobody to answer and is +/// not covered by retransmit once its op has reached quorum +/// (`consensus::retransmit_targets` skips `ok_quorum_received`), so the backup +/// gap-stops; `tick_partitions`' sweep is what repairs it, escalating to +/// partition state transfer when the primary has evicted the range, and +/// `partition_prepare_gap_drops_total` is what counts the prepares that reached +/// the gap check. The counter therefore signals a data-plane gap or recovery +/// burden, not a requirement for a view change. pub mod frame_drop_variant { pub const CONSENSUS: &str = "consensus"; pub const FD_TRANSFER: &str = "fd_transfer"; @@ -208,6 +211,7 @@ pub struct ShardMetrics { partition_frames_rejected_ahead_total: Counter, partition_requests_denied_transient_total: Counter, partition_repair_serves_deferred_purge_total: Counter, + partition_prepare_gap_drops_total: Counter, metadata_read_frontier_refusals_total: Counter, client_requests_denied_queue_full_total: Counter, } @@ -234,6 +238,7 @@ impl ShardMetrics { partition_frames_rejected_ahead_total: Counter::default(), partition_requests_denied_transient_total: Counter::default(), partition_repair_serves_deferred_purge_total: Counter::default(), + partition_prepare_gap_drops_total: Counter::default(), metadata_read_frontier_refusals_total: Counter::default(), client_requests_denied_queue_full_total: Counter::default(), } @@ -447,6 +452,42 @@ impl ShardMetrics { self.partition_repair_serves_deferred_purge_total.get() } + /// Add the prepares a partition's backup gap check destroyed since the last + /// sweep. Drained per tick from `IggyPartition::take_prepare_gap_drops`, + /// and once more when `ConfirmRemove` drops the partition: a tombstoned + /// namespace is invisible to the sweep, so the tail it left would otherwise + /// go to the floor with the value. + /// + /// Counts the prepares that ARRIVED after a hole, not the holes: a gap + /// opened by the last prepare of a burst leaves this at zero. A nonzero + /// value proves the repair driver has work; a zero one proves nothing. + /// + /// Deliberately NOT a `frame_drops_total{variant=partition}` reason: that + /// family means the bus or the router shed a frame, and the simulator + /// asserts it stays at zero on runs with no injected loss. A gap drop is a + /// protocol-ordering drop that any real loss produces, and the sweep repairs + /// it, so folding the two would turn a routing-fault alert into noise. + /// + /// Shard-scoped, with no namespace label: a server runs hundreds of groups + /// per shard, so labelling by namespace is unbounded cardinality, and every + /// other partition counter in this file is shard-scoped for the same + /// reason. The per-group detail is in the arm's log line. + /// + /// The metadata plane's own gap drop is NOT counted here, and has no + /// counter of its own: its repair is armed by the same edge-triggered sites + /// this plane's sweep exists to backstop, so that plane is starvable in the + /// same way and is simply not instrumented for it yet. + pub fn record_partition_prepare_gap_drops(&self, drops: u64) { + self.partition_prepare_gap_drops_total.inc_by(drops); + } + + /// Snapshot of `partition_prepare_gap_drops_total`. Test/simulator accessor. + #[cfg(any(test, feature = "simulator"))] + #[must_use] + pub fn partition_prepare_gap_drops_value(&self) -> u64 { + self.partition_prepare_gap_drops_total.get() + } + /// Snapshot of `partition_frames_rejected_stale_total`. Test/simulator /// accessor, readable from any crate under those cfgs so the crates that /// drive the reconciler can assert a reject did not happen. @@ -538,6 +579,11 @@ impl ShardMetrics { "partition repair serves or completions deferred until a committed purge applies", self.partition_repair_serves_deferred_purge_total.clone(), ); + registry.register( + "partition_prepare_gap_drops", + "replicated prepares dropped out of order by a backup's gap check", + self.partition_prepare_gap_drops_total.clone(), + ); registry.register( "metadata_read_frontier_refusals", "metadata reads refused because this node never applied the caller's committed op", diff --git a/core/shard/src/router.rs b/core/shard/src/router.rs index 1f4fa01170..4a3a0d2149 100644 --- a/core/shard/src/router.rs +++ b/core/shard/src/router.rs @@ -379,6 +379,8 @@ where self.process_frame(frame).await; self.process_loopback(&mut loopback_buf, &mut namespace_scratch).await; // Tail drain catches reconcile ops whose marker was dropped. + // Anything it stages is served by the arm above + // on the next pass, before this arm can run again. self.apply_reconcile_ops(); } // Guaranteed reply-lane service: `select_biased!` diff --git a/core/simulator/Cargo.toml b/core/simulator/Cargo.toml index 0e46f73fc6..0ac51b6af5 100644 --- a/core/simulator/Cargo.toml +++ b/core/simulator/Cargo.toml @@ -51,6 +51,10 @@ tracing = { workspace = true } tracing-subscriber = { workspace = true } [dev-dependencies] +# Test-only feature on a crate this package already depends on: the fault +# hooks the partition-driver tests arm must not compile into anything that +# ships. See `partitions/Cargo.toml`. +partitions = { workspace = true, features = ["fault-injection"] } tempfile = { workspace = true } [lints.clippy] diff --git a/core/simulator/src/lib.rs b/core/simulator/src/lib.rs index 767f72196c..8a9d7c6317 100644 --- a/core/simulator/src/lib.rs +++ b/core/simulator/src/lib.rs @@ -5106,6 +5106,864 @@ mod repair_frontier_tests { } } +#[cfg(test)] +mod partition_repair_driver_tests { + //! A backup that missed a committed partition prepare recovers in Normal + //! status, without waiting for a view change. + //! + //! Every edge-triggered arming site is starvable. A follower advances + //! `commit_max` from each prepare header in `replicate_preflight`, before + //! the gap check drops the prepare, so under produce the primary's commit + //! heartbeat lands as `CommitOutcome::Accepted` and the backstop that would + //! arm repair on `Advanced` never runs. These tests starve that edge + //! outright (no commit heartbeat for the group reaches the lagging replica) + //! so nothing but the level-triggered detector in `tick_partitions` can + //! close the gap. + + use super::*; + use bytes::Bytes; + use consensus::Status; + use iggy_binary_protocol::{ + CommitHeader, PrepareHeader, RepairRangeReplyHeader, RequestPreparesHeader, + }; + use packet::Packet; + use std::sync::atomic::{AtomicU64, Ordering}; + + /// Chain replication runs 0 -> 1 -> 2 and stops before the primary, so + /// replica 2 forwards to nobody: it is the only replica whose losses do not + /// also starve its successor, and therefore the group, of quorum. + const LAGGING: u8 = 2; + + const CLIENT_ID: u128 = 1; + + /// Ops that replicate cleanly before the fault, so the gap opens above a + /// committed prefix rather than at the group's first op. + const WARMUP_SENDS: usize = 3; + + /// Ticks stepped after each produce. Keeps one send's round trip inside its + /// own window so the workload is legible tick by tick. + const STEPS_PER_SEND: usize = 12; + + /// Produces issued with the fault standing. `partitions::REPAIR_RETRY_TICKS` + /// is 100, so this must carry the run past the debounce while prepares keep + /// consuming the `commit_max` advance the heartbeat backstop needs. + const GAP_SENDS: usize = 12; + + /// Quiet ticks after the produce stops, for the repair stream to land. + /// + /// Budgeted against `NORMAL_HEARTBEAT_TICKS` (500): with this group's commit + /// heartbeats withheld, the lagging replica elects once that timer fires, and + /// an election would heal the gap through `on_start_view` instead. Every + /// phase after the fault is installed has to fit inside it. + const QUIET_STEPS: usize = 160; + + /// Budget for the group to settle once the fault is lifted. Generous rather + /// than tuned: the drain loop breaks on convergence, and the tail above the + /// repaired window waits on whichever interval-driven site picks it up. + const DRAIN_STEPS: usize = 600; + + /// Ticks of healthy load in the no-false-positive test, several debounce + /// intervals' worth so the sweep gets many chances to arm. + const LOAD_TICKS: usize = 4 * partitions::REPAIR_RETRY_TICKS as usize; + + /// Paced at a round trip rather than one submit per tick: the pipeline caps + /// at `PIPELINE_PREPARE_QUEUE_MAX`, so submitting faster than the group + /// commits just collects transient rejections and the group goes quiet. + const TICKS_PER_SEND: usize = 4; + + /// Ops the healthy run must have committed for its verdict to mean anything. + /// A produce's round trip is four one-way hops plus tick granularity, so this + /// network commits on the order of one op per fifteen ticks however hard the + /// client pushes; the floor only has to prove the group was live across + /// several debounce intervals, not that it was saturated. + const COMMITTED_MIN: u64 = 20; + + /// Produces issued with the fault standing in the walk-starvation test: few + /// enough that the gap debounce fires only after produce stops, so the + /// repair window closes at the last carried commit and lands fully resident. + const STRAND_SENDS: usize = 3; + + /// Quiet budget for the walk-starvation test: debounce, repair stream, then + /// the drain, kept under `NORMAL_HEARTBEAT_TICKS` so an election cannot be + /// the healer. + const STRAND_QUIET_STEPS: usize = 300; + + /// Defines this test's `withhold_one_prepare` chain hook over the statics it + /// names: swallow the FIRST partition prepare for `$namespace`, once, and + /// record its op in `$withheld_op`. + /// + /// A macro because link hooks are bare `fn` pointers, so the body cannot + /// capture, and the statics must stay per-test: the sibling tests in this + /// binary run in parallel and would otherwise share one fault. The statics + /// are still declared in each test, where its fault setup is read. + macro_rules! withhold_one_prepare { + ($namespace:ident, $withheld_op:ident) => { + fn withhold_one_prepare(packet: &Packet) -> bool { + let Some(header) = prepare_for(packet, $namespace.load(Ordering::Relaxed)) else { + return false; + }; + $withheld_op + .compare_exchange(0, header.op, Ordering::Relaxed, Ordering::Relaxed) + .is_ok() + } + }; + } + + /// The partition-plane prepare a packet carries, if it carries one for + /// `group`. + fn prepare_for(packet: &Packet, group: u64) -> Option { + if packet.message.header().command != Command::Prepare { + return None; + } + let header: &PrepareHeader = + bytemuck::checked::from_bytes(&packet.message.as_slice()[..size_of::()]); + (header.group == group).then_some(*header) + } + + /// Whether a packet is a commit heartbeat for `group`. + fn is_commit_for(packet: &Packet, group: u64) -> bool { + if packet.message.header().command != Command::Commit { + return false; + } + let header: &CommitHeader = + bytemuck::checked::from_bytes(&packet.message.as_slice()[..size_of::()]); + header.group == group + } + + /// Whether a packet is a repair request for `group`. + fn is_request_prepares_for(packet: &Packet, group: u64) -> bool { + if packet.message.header().command != Command::RequestPrepares { + return false; + } + let header: &RequestPreparesHeader = bytemuck::checked::from_bytes( + &packet.message.as_slice()[..size_of::()], + ); + header.group == group + } + + /// Whether a packet is a repair stream terminator for `group`. + fn is_repair_done_for(packet: &Packet, group: u64) -> bool { + if packet.message.header().command != Command::RepairDone { + return false; + } + let header: &RepairRangeReplyHeader = bytemuck::checked::from_bytes( + &packet.message.as_slice()[..size_of::()], + ); + header.group == group + } + + fn cluster(seed: u64) -> (Simulator, SimClient) { + server_common::MemoryPool::init_pool(&server_common::MemoryPoolSettings { + enabled: false, + size: iggy_common::IggyByteSize::from(0u64), + bucket_capacity: 1, + }); + let replica_count: u8 = 3; + let network_opts = packet::PacketSimulatorOptions { + node_count: replica_count, + client_count: 1, + seed, + ..packet::PacketSimulatorOptions::default() + }; + let sim = Simulator::new( + replica_count as usize, + std::iter::once(CLIENT_ID), + network_opts, + ); + (sim, SimClient::new(CLIENT_ID)) + } + + /// Submit `sends` produces against `namespace`, stepping between each. + fn produce( + sim: &mut Simulator, + client: &SimClient, + namespace: IggyNamespace, + sends: usize, + tag: &str, + ) { + for index in 0..sends { + let msg = client.send_messages(namespace, &[Bytes::from(format!("{tag}-{index}"))]); + sim.submit_request(client.client_id(), 0, msg.into_generic()); + for _ in 0..STEPS_PER_SEND { + sim.step(); + } + } + } + + /// `(status, view, commit_min, commit_max)` of one replica's partition group. + fn group_state( + sim: &Simulator, + replica: u8, + namespace: IggyNamespace, + ) -> (Status, u32, u64, u64) { + let shard = sim.replicas[replica as usize].partition_shard(namespace); + let partition = shard + .plane + .partitions() + .get_by_ns(&namespace) + .expect("the replica hosts the group"); + let consensus = partition.consensus(); + ( + consensus.status(), + consensus.view(), + consensus.commit_min(), + consensus.commit_max(), + ) + } + + fn journal_holds(sim: &Simulator, replica: u8, namespace: IggyNamespace, op: u64) -> bool { + let shard = sim.replicas[replica as usize].partition_shard(namespace); + shard + .plane + .partitions() + .get_by_ns(&namespace) + .is_some_and(|partition| partition.log.journal().inner.holds_op(op)) + } + + fn gap_drops(sim: &Simulator, replica: u8, namespace: IggyNamespace) -> u64 { + sim.replicas[replica as usize] + .partition_shard(namespace) + .metrics() + .partition_prepare_gap_drops_value() + } + + /// Gap drops still buffered on the partition, i.e. recorded since the sweep + /// last drained them into the shard counter. + fn buffered_gap_drops(sim: &Simulator, replica: u8, namespace: IggyNamespace) -> u64 { + sim.replicas[replica as usize] + .partition_shard(namespace) + .plane + .partitions() + .get_by_ns(&namespace) + .map_or(0, partitions::IggyPartition::prepare_gap_drops) + } + + fn transfer_armed(sim: &Simulator, replica: u8, namespace: IggyNamespace) -> bool { + let shard = sim.replicas[replica as usize].partition_shard(namespace); + shard + .plane + .partitions() + .get_by_ns(&namespace) + .is_some_and(|partition| { + partition.transfer.is_some() + || partition.consensus().state_transfer_stage() + != consensus::StateTransferStage::Idle + }) + } + + #[test] + fn given_a_backup_that_dropped_a_committed_prepare_when_heartbeat_advances_are_starved_should_repair_in_normal_status() + { + // Statics, not captures: the link hooks are bare `fn` pointers. Declared + // inside the test because the sibling tests in this binary run in + // parallel and would otherwise share them. + static GAP_NS: AtomicU64 = AtomicU64::new(0); + static WITHHELD_OP: AtomicU64 = AtomicU64::new(0); + + withhold_one_prepare!(GAP_NS, WITHHELD_OP); + + /// Primary -> 2: withhold this group's commit heartbeats, so the + /// `Advanced` backstop can never run, and withhold retransmits of the + /// dropped op. The retransmit half stands in for production behaviour + /// rather than adding a fault: `consensus::retransmit_targets` skips an + /// op that already reached quorum, and this op reaches quorum on 0 and 1 + /// alone. Without it the retry timer could heal the gap and the test + /// would pass with no repair driver at all. + fn starve_commit_edge(packet: &Packet) -> bool { + let group = GAP_NS.load(Ordering::Relaxed); + if let Some(header) = prepare_for(packet, group) { + return header.op == WITHHELD_OP.load(Ordering::Relaxed); + } + is_commit_for(packet, group) + } + + let (mut sim, client) = cluster(0x5EED_0232); + let namespace = IggyNamespace::new(1, 1, 0); + sim.init_partition(namespace); + sim.register_client_with_primary(&client); + GAP_NS.store(namespace.inner(), Ordering::Relaxed); + WITHHELD_OP.store(0, Ordering::Relaxed); + + produce(&mut sim, &client, namespace, WARMUP_SENDS, "warmup"); + let (_, _, warm_commit_min, _) = group_state(&sim, LAGGING, namespace); + assert!( + warm_commit_min > 0, + "the lagging replica committed nothing before the fault, so the gap \ + below would open at the group's first op" + ); + + *sim.network + .link_drop_packet_fn(ProcessId::Replica(1), ProcessId::Replica(LAGGING)) = + Some(withhold_one_prepare); + *sim.network + .link_drop_packet_fn(ProcessId::Replica(0), ProcessId::Replica(LAGGING)) = + Some(starve_commit_edge); + + produce(&mut sim, &client, namespace, GAP_SENDS, "gap"); + + let withheld = WITHHELD_OP.load(Ordering::Relaxed); + assert_ne!( + withheld, 0, + "no partition prepare crossed the chain link, so the fault never armed" + ); + assert!( + gap_drops(&sim, LAGGING, namespace) > 0, + "the lagging replica never reached its backup gap check, so the \ + prepares after the withheld op were not dropped as a gap" + ); + + for _ in 0..QUIET_STEPS { + sim.step(); + } + + // Judged with the blockade still standing, so the tick driver is the only + // thing that can have armed the repair: no commit heartbeat for this + // group has reached the replica since the gap opened, and `on_commit` is + // where the `Advanced` backstop lives. + let (status, view, commit_min, _) = group_state(&sim, LAGGING, namespace); + assert_eq!( + view, 0, + "a view change healed the gap instead of the repair driver; the test \ + proves nothing about normal status" + ); + assert_eq!(status, Status::Normal, "the replica left Normal status"); + assert!( + journal_holds(&sim, LAGGING, namespace, withheld), + "op {withheld} was never repaired back into the lagging replica's journal" + ); + assert!( + commit_min >= withheld, + "the commit walk never crossed the repaired hole: stopped at \ + {commit_min}, the withheld op is {withheld}" + ); + + // Lift the blockade and let the group settle. The commit walk is driven + // by arriving frames, so with this group's heartbeats withheld the tail + // above the repaired window has nothing to advance it; that is the + // injected fault, not the gap under test. + *sim.network + .link_drop_packet_fn(ProcessId::Replica(0), ProcessId::Replica(LAGGING)) = None; + for _ in 0..DRAIN_STEPS { + sim.step(); + let (_, _, commit_min, commit_max) = group_state(&sim, LAGGING, namespace); + if commit_min == commit_max { + break; + } + } + let (status, view, commit_min, commit_max) = group_state(&sim, LAGGING, namespace); + assert_eq!((status, view), (Status::Normal, 0)); + assert_eq!( + commit_min, commit_max, + "the lagging replica is still gap-stopped: committed through \ + {commit_max} but walkable only to {commit_min}" + ); + } + + #[test] + fn given_a_repair_armed_by_the_tick_driver_when_the_range_is_evicted_should_convert_to_state_transfer() + { + static GAP_NS: AtomicU64 = AtomicU64::new(0); + static WITHHELD_OP: AtomicU64 = AtomicU64::new(0); + + withhold_one_prepare!(GAP_NS, WITHHELD_OP); + + fn starve_commit_edge(packet: &Packet) -> bool { + let group = GAP_NS.load(Ordering::Relaxed); + if let Some(header) = prepare_for(packet, group) { + return header.op == WITHHELD_OP.load(Ordering::Relaxed); + } + is_commit_for(packet, group) + } + + let (mut sim, client) = cluster(0x5EED_0233); + let namespace = IggyNamespace::new(1, 1, 0); + sim.init_partition(namespace); + sim.register_client_with_primary(&client); + GAP_NS.store(namespace.inner(), Ordering::Relaxed); + WITHHELD_OP.store(0, Ordering::Relaxed); + + produce(&mut sim, &client, namespace, WARMUP_SENDS, "warmup"); + + *sim.network + .link_drop_packet_fn(ProcessId::Replica(1), ProcessId::Replica(LAGGING)) = + Some(withhold_one_prepare); + *sim.network + .link_drop_packet_fn(ProcessId::Replica(0), ProcessId::Replica(LAGGING)) = + Some(starve_commit_edge); + + produce(&mut sim, &client, namespace, GAP_SENDS, "gap"); + assert_ne!( + WITHHELD_OP.load(Ordering::Relaxed), + 0, + "no partition prepare crossed the chain link, so the fault never armed" + ); + + // Compact the serving side past the gap. This plane's journal is + // memory-only, so wiping it IS its retention floor moving: the serve + // path reads `repair_retained_from` as `None` and reports eviction from + // its own commit frontier, which is exactly what a peer that + // checkpointed past the requested window answers. + { + let primary = sim.replicas[0].partition_shard(namespace); + let partition = primary + .plane + .partitions() + .get_by_ns(&namespace) + .expect("the primary hosts the group"); + partition.log.journal().inner.clear_all(); + } + + for _ in 0..QUIET_STEPS { + sim.step(); + if transfer_armed(&sim, LAGGING, namespace) { + break; + } + } + + let (status, view, ..) = group_state(&sim, LAGGING, namespace); + assert_eq!( + view, 0, + "a view change armed the recovery instead of the tick-armed repair session" + ); + assert!( + transfer_armed(&sim, LAGGING, namespace), + "the tick-armed repair session hit an evicted range but never converted \ + to a state transfer (status {status:?})" + ); + } + + #[test] + fn given_healthy_pipelined_traffic_when_no_gap_exists_should_not_arm_repair() { + static GAP_NS: AtomicU64 = AtomicU64::new(0); + static REPAIR_REQUESTS: AtomicU64 = AtomicU64::new(0); + + /// Observer, not a fault: counts this group's repair requests and passes + /// every packet through. + fn count_repair_requests(packet: &Packet) -> bool { + if is_request_prepares_for(packet, GAP_NS.load(Ordering::Relaxed)) { + REPAIR_REQUESTS.fetch_add(1, Ordering::Relaxed); + } + false + } + + // TWO replicas, so quorum spans both: no op can commit without the + // backup's ack, every reordering-induced gap therefore blocks quorum, and + // `consensus::retransmit_targets` refills it. At three, the network's + // per-tick delivery shuffle lets an op commit on the primary and its + // first chain hop while the last replica loses it for good, which is the + // very fault the sibling tests inject -- it would then be repaired here, + // correctly, and this assertion would fire on a healthy driver. + server_common::MemoryPool::init_pool(&server_common::MemoryPoolSettings { + enabled: false, + size: iggy_common::IggyByteSize::from(0u64), + bucket_capacity: 1, + }); + let replica_count: u8 = 2; + let mut sim = Simulator::new( + replica_count as usize, + std::iter::once(CLIENT_ID), + packet::PacketSimulatorOptions { + node_count: replica_count, + client_count: 1, + seed: 0x5EED_0234, + ..packet::PacketSimulatorOptions::default() + }, + ); + let client = SimClient::new(CLIENT_ID); + let namespace = IggyNamespace::new(1, 1, 0); + sim.init_partition(namespace); + sim.register_client_with_primary(&client); + GAP_NS.store(namespace.inner(), Ordering::Relaxed); + REPAIR_REQUESTS.store(0, Ordering::Relaxed); + + for (from, to) in [(0u8, 1u8), (1, 0)] { + *sim.network + .link_drop_packet_fn(ProcessId::Replica(from), ProcessId::Replica(to)) = + Some(count_repair_requests); + } + + // Sustained, not bursty: a produce every tick, for several debounce + // intervals, so the sweep gets many chances to arm against ordinary + // pipelining. + for tick in 0..LOAD_TICKS { + if tick % TICKS_PER_SEND == 0 { + let msg = + client.send_messages(namespace, &[Bytes::from(format!("healthy-{tick}"))]); + sim.submit_request(client.client_id(), 0, msg.into_generic()); + } + sim.step(); + } + for _ in 0..QUIET_STEPS { + sim.step(); + } + + let committed = group_state(&sim, 1, namespace).2; + let sends = LOAD_TICKS / TICKS_PER_SEND; + assert!( + committed >= COMMITTED_MIN, + "the backup committed only {committed} ops across {sends} sends, so the \ + sweep was never driven over a loaded group" + ); + // No per-tick lag sampling: `sim.step()` runs the pumps to quiescence, so + // every sample lands AFTER the tick whose walk backstop drained whatever + // the step's produce left, and a run that asserted something about those + // samples would be asserting over an empty set. The commit lag a naive + // detector misreads lives inside a step, and it is pinned where it can + // be held still: exhaustively in `gap_detector_tests`, and end to end by + // the walk-starvation run in this module, which strands a backup in + // exactly that state and proves nothing arms repair over it. + // + // What this run adds is the part only a live group can show: a loaded, + // lossless cluster produces no repair traffic at all. + for replica in 0..replica_count { + let (status, view, commit_min, commit_max) = group_state(&sim, replica, namespace); + assert_eq!( + (status, view), + (Status::Normal, 0), + "replica {replica} left view 0 / Normal, so a view change could \ + account for repair traffic" + ); + // Residual lag is transient at worst: a heartbeat carrying a commit + // the replica already knows is `CommitOutcome::Accepted`, so a backup + // can sit with committed-but-unwalked ops until the tick sweep's + // walk-stalled backstop resumes the walk. Every one of them is + // RESIDENT, which is what separates it from a gap. + if commit_min < commit_max { + assert!( + journal_holds(&sim, replica, namespace, commit_min + 1), + "replica {replica} lags at {commit_min} of {commit_max} with op \ + {} missing, so a no-loss run produced a real hole", + commit_min + 1 + ); + } + } + assert_eq!( + REPAIR_REQUESTS.load(Ordering::Relaxed), + 0, + "the tick driver requested repair on a healthy group across {sends} sends; \ + its gap predicate is reading ordinary commit lag as a journal hole" + ); + } + + #[test] + fn given_a_backup_holding_resident_committed_ops_when_every_walk_edge_is_starved_should_drain_in_normal_status() + { + static GAP_NS: AtomicU64 = AtomicU64::new(0); + static WITHHELD_OP: AtomicU64 = AtomicU64::new(0); + static WITHHELD_DONES: AtomicU64 = AtomicU64::new(0); + + withhold_one_prepare!(GAP_NS, WITHHELD_OP); + + /// Primary -> 2: withhold every direct prepare (live ones ride the + /// chain, so this starves only retransmit heals), the group's commit + /// heartbeats, and its repair terminators. The repaired ops themselves + /// pass, so the window lands resident while `complete_repair`, the walk + /// the terminator would run, never fires. + fn starve_walk_edges(packet: &Packet) -> bool { + let group = GAP_NS.load(Ordering::Relaxed); + if prepare_for(packet, group).is_some() { + return true; + } + if is_repair_done_for(packet, group) { + WITHHELD_DONES.fetch_add(1, Ordering::Relaxed); + return true; + } + is_commit_for(packet, group) + } + + let (mut sim, client) = cluster(0x5EED_0235); + let namespace = IggyNamespace::new(1, 1, 0); + sim.init_partition(namespace); + sim.register_client_with_primary(&client); + GAP_NS.store(namespace.inner(), Ordering::Relaxed); + WITHHELD_OP.store(0, Ordering::Relaxed); + WITHHELD_DONES.store(0, Ordering::Relaxed); + + produce(&mut sim, &client, namespace, WARMUP_SENDS, "warmup"); + let (_, _, warm_commit_min, _) = group_state(&sim, LAGGING, namespace); + assert!( + warm_commit_min > 0, + "the lagging replica committed nothing before the fault, so the gap \ + below would open at the group's first op" + ); + + *sim.network + .link_drop_packet_fn(ProcessId::Replica(1), ProcessId::Replica(LAGGING)) = + Some(withhold_one_prepare); + *sim.network + .link_drop_packet_fn(ProcessId::Replica(0), ProcessId::Replica(LAGGING)) = + Some(starve_walk_edges); + + produce(&mut sim, &client, namespace, STRAND_SENDS, "strand"); + + let withheld = WITHHELD_OP.load(Ordering::Relaxed); + assert_ne!( + withheld, 0, + "no partition prepare crossed the chain link, so the fault never armed" + ); + + for _ in 0..STRAND_QUIET_STEPS { + sim.step(); + let (_, view, commit_min, commit_max) = group_state(&sim, LAGGING, namespace); + if view != 0 || (commit_min >= withheld && commit_min == commit_max) { + break; + } + } + + assert!( + WITHHELD_DONES.load(Ordering::Relaxed) > 0, + "no repair terminator was withheld, so the walk was never starved and \ + a green run would not prove the tick backstop" + ); + assert!( + journal_holds(&sim, LAGGING, namespace, withheld), + "op {withheld} was never repaired back into the lagging replica's journal" + ); + let (status, view, commit_min, commit_max) = group_state(&sim, LAGGING, namespace); + assert_eq!( + view, 0, + "a view change drained the walk instead of the tick backstop; the test \ + proves nothing about normal status" + ); + assert_eq!(status, Status::Normal, "the replica left Normal status"); + for op in commit_min + 1..=commit_max { + assert!( + journal_holds(&sim, LAGGING, namespace, op), + "op {op} is not resident, so this run stranded on a repair gap, \ + not a parked walk" + ); + } + assert_eq!( + commit_min, commit_max, + "the walk never resumed over resident committed ops: walkable to \ + {commit_min}, committed through {commit_max}, every op between resident" + ); + } + + #[test] + fn given_a_gap_stopped_backup_when_repair_is_armed_should_spend_its_debounce() { + // The reset lives in `maybe_request_partition_repair`, the funnel every + // arming site goes through, precisely because the four edge-triggered + // sites never touch `gap_ticks` themselves. Driven here rather than + // modelled: a saturated count hands the NEXT gap an arm on its first + // tick, and the shape that reaches it is a repair short enough that no + // sweep ever observes the partition with its recovery owned. + static GAP_NS: AtomicU64 = AtomicU64::new(0); + static WITHHELD_OP: AtomicU64 = AtomicU64::new(0); + + withhold_one_prepare!(GAP_NS, WITHHELD_OP); + + /// Primary -> 2: withhold this group's commit heartbeats, so the + /// edge-triggered backstop cannot be what arms the repair below. + fn starve_commit_edge(packet: &Packet) -> bool { + let group = GAP_NS.load(Ordering::Relaxed); + if let Some(header) = prepare_for(packet, group) { + return header.op == WITHHELD_OP.load(Ordering::Relaxed); + } + is_commit_for(packet, group) + } + + let (mut sim, client) = cluster(0x5EED_0238); + let namespace = IggyNamespace::new(1, 1, 0); + sim.init_partition(namespace); + sim.register_client_with_primary(&client); + GAP_NS.store(namespace.inner(), Ordering::Relaxed); + WITHHELD_OP.store(0, Ordering::Relaxed); + + produce(&mut sim, &client, namespace, WARMUP_SENDS, "warmup"); + *sim.network + .link_drop_packet_fn(ProcessId::Replica(1), ProcessId::Replica(LAGGING)) = + Some(withhold_one_prepare); + *sim.network + .link_drop_packet_fn(ProcessId::Replica(0), ProcessId::Replica(LAGGING)) = + Some(starve_commit_edge); + // Short, deliberately: long enough to open the gap, well under the + // debounce, so the driver has not armed on its own and the sweep below + // is the one that does it. + produce(&mut sim, &client, namespace, STRAND_SENDS, "gap"); + assert_ne!( + WITHHELD_OP.load(Ordering::Relaxed), + 0, + "no partition prepare crossed the chain link, so the fault never armed" + ); + let (_, _, commit_min, commit_max) = group_state(&sim, LAGGING, namespace); + assert!( + commit_min < commit_max && !journal_holds(&sim, LAGGING, namespace, commit_min + 1), + "the lagging replica is not gap-stopped (walkable to {commit_min} of \ + {commit_max}), so the sweep has nothing to arm" + ); + + // Saturate the debounce by hand and take one sweep: the arm has to be + // what spends it, whatever the tick counted on the way in. + let shard = sim.replicas[LAGGING as usize].partition_shard(namespace); + { + let partitions = shard.plane.partitions(); + let partition = partitions + .get_by_ns(&namespace) + .expect("the replica hosts the group"); + assert!( + partition.repair.is_none(), + "a session was already open, so this sweep would arm nothing" + ); + partition.gap_ticks.set(u32::MAX); + } + let mut namespace_scratch = Vec::new(); + futures::executor::block_on(shard.tick_partitions(&mut namespace_scratch)); + + let partitions = shard.plane.partitions(); + let partition = partitions + .get_by_ns(&namespace) + .expect("the replica hosts the group"); + assert!( + partition.repair.is_some(), + "the sweep never opened a session, so the reset below proves nothing" + ); + assert_eq!( + partition.gap_ticks.get(), + 0, + "the arm left the debounce saturated; a repair that completes before the \ + next sweep would then hand the following gap an arm on its first tick" + ); + } + + #[test] + fn given_a_local_commit_failure_in_the_tick_walk_when_the_partition_fences_should_return_the_fault() + { + static GAP_NS: AtomicU64 = AtomicU64::new(0); + + /// Primary -> 2: withhold this group's commit heartbeats, so the last + /// produced op stays journaled-but-uncommitted here and the walk the + /// test drives below is the first one that can reach it. + fn starve_commit_edge(packet: &Packet) -> bool { + is_commit_for(packet, GAP_NS.load(Ordering::Relaxed)) + } + + let (mut sim, client) = cluster(0x5EED_0236); + let namespace = IggyNamespace::new(1, 1, 0); + sim.init_partition(namespace); + sim.register_client_with_primary(&client); + GAP_NS.store(namespace.inner(), Ordering::Relaxed); + + produce(&mut sim, &client, namespace, WARMUP_SENDS, "warmup"); + *sim.network + .link_drop_packet_fn(ProcessId::Replica(0), ProcessId::Replica(LAGGING)) = + Some(starve_commit_edge); + produce(&mut sim, &client, namespace, 1, "fence"); + + let (_, _, commit_min, _) = group_state(&sim, LAGGING, namespace); + let stranded = commit_min + 1; + assert!( + journal_holds(&sim, LAGGING, namespace, stranded), + "op {stranded} is not resident, so the walk below would find nothing \ + to commit and nothing to fail on" + ); + + // The sweep is invoked DIRECTLY rather than through `sim.step()`: what + // is under test is the verdict of the tick whose own walk raised the + // fault, and a pump-driven tick would fence on one pass and be asked + // for its verdict on the next, where the pre-walk read already answers. + let shard = sim.replicas[LAGGING as usize].partition_shard(namespace); + { + let partitions = shard.plane.partitions(); + let partition = partitions + .get_mut_by_ns(&namespace) + .expect("the replica hosts the group"); + // Committed as far as this replica knows, with the body resident: + // walk-stalled, which is the state the tick's backstop walks. + partition.consensus().advance_commit_max(stranded); + partition.inject_commit_failure(); + } + + let mut namespace_scratch = Vec::new(); + let fault = futures::executor::block_on(shard.tick_partitions(&mut namespace_scratch)) + .expect( + "the tick returned clean over a partition its own walk fenced; the pump \ + would keep serving a divergent replica on that verdict", + ); + assert_eq!(fault.namespace_raw, namespace.inner()); + assert_eq!( + fault.op, stranded, + "the fault must name the op whose local commit failed" + ); + assert!( + sim.replicas[LAGGING as usize] + .partition_shard(namespace) + .plane + .partitions() + .get_by_ns(&namespace) + .is_some_and(|partition| partition.fatal().is_some()), + "the partition must stay fenced after the tick reported the fault" + ); + } + + #[test] + fn given_buffered_gap_drops_when_the_namespace_is_removed_should_still_count_them() { + static GAP_NS: AtomicU64 = AtomicU64::new(0); + static WITHHELD_OP: AtomicU64 = AtomicU64::new(0); + + // Every prepare after the withheld one reaches the backup's gap check + // and is destroyed there, which is what the counter records. + withhold_one_prepare!(GAP_NS, WITHHELD_OP); + + let (mut sim, client) = cluster(0x5EED_0237); + let namespace = IggyNamespace::new(1, 1, 0); + sim.init_partition(namespace); + sim.register_client_with_primary(&client); + GAP_NS.store(namespace.inner(), Ordering::Relaxed); + WITHHELD_OP.store(0, Ordering::Relaxed); + + produce(&mut sim, &client, namespace, WARMUP_SENDS, "warmup"); + *sim.network + .link_drop_packet_fn(ProcessId::Replica(1), ProcessId::Replica(LAGGING)) = + Some(withhold_one_prepare); + + // Stepped one at a time and stopped on the step that buffered a drop: + // the sweep drains the count on the following tick, and the whole point + // of this run is to remove the namespace while the count is still on + // the partition. + let mut buffered = 0; + 'produce: for index in 0..GAP_SENDS { + let message = client.send_messages(namespace, &[Bytes::from(format!("gap-{index}"))]); + sim.submit_request(client.client_id(), 0, message.into_generic()); + for _ in 0..STEPS_PER_SEND { + sim.step(); + buffered = buffered_gap_drops(&sim, LAGGING, namespace); + if buffered > 0 { + break 'produce; + } + } + } + assert!( + buffered > 0, + "no prepare reached the lagging replica's gap check, so the removal \ + below would have nothing to lose" + ); + let counted = gap_drops(&sim, LAGGING, namespace); + + // The reconciler's teardown order: the tombstone lands first and the + // disk delete runs before `ConfirmRemove`, so from here `get_mut_by_ns` + // answers `None` and the sweep can never drain the count again. + { + let shard = sim.replicas[LAGGING as usize].partition_shard(namespace); + shard.plane.partitions().tombstone(namespace); + shard.enqueue_reconcile_op(shard::ReconcileOp::ConfirmRemove { namespace }); + } + sim.step(); + + let shard = sim.replicas[LAGGING as usize].partition_shard(namespace); + assert!( + shard.plane.partitions().get_by_ns(&namespace).is_none(), + "ConfirmRemove must have dropped the partition, or this run proves nothing" + ); + assert_eq!( + shard.metrics().partition_prepare_gap_drops_value(), + counted + buffered, + "the {buffered} prepare(s) buffered on the partition went to the floor \ + with it; the drops are the only record those frames existed" + ); + } +} + #[cfg(test)] mod metadata_read_frontier_tests { //! A client that committed a metadata write and then re-homed onto a diff --git a/scripts/ci/storage-compat.sh b/scripts/ci/storage-compat.sh index 952e69fb5e..edfabe02c9 100755 --- a/scripts/ci/storage-compat.sh +++ b/scripts/ci/storage-compat.sh @@ -31,6 +31,9 @@ source "$(dirname "${BASH_SOURCE[0]}")/lib/init.sh" # The test boots the baseline, seeds a data directory, swaps the binary to HEAD # and restarts against that same directory. # +# The baseline's own core/server/config.toml is extracted next to its binary and +# both boots read it, so the swap changes the binary and nothing else. +# # Both binaries MUST be built here. `core/integration` has no dependency on the # `server` package; the harness only LOCATES a binary, through # `assert_cmd::Command::cargo_bin`, which falls back to whatever file happens to @@ -186,6 +189,19 @@ fi # every run. BASELINE_SERVER="${TARGET_DIR}/storage-compat/${BASELINE_SHA}/iggy-server" +# The baseline's own config file, handed to BOTH boots. Neither binary may fall +# back to the server's relative default path: figment resolves that by walking +# up from the test process's directory, which hands the BASELINE this branch's +# file, and a key added under a `deny_unknown_fields` table (every [cluster] and +# [node] one) then fails its extraction. Written on every run, cached binary or +# not, so the pair can never drift apart. +BASELINE_CONFIG="${TARGET_DIR}/storage-compat/${BASELINE_SHA}/config.toml" +mkdir -p "$(dirname "${BASELINE_CONFIG}")" +if ! git show "${BASELINE_SHA}:core/server/config.toml" >"${BASELINE_CONFIG}"; then + echo "Baseline ${BASELINE_SHA} carries no core/server/config.toml" + exit 1 +fi + if [ "${REBUILD_BASELINE}" -eq 1 ]; then rm -f "${BASELINE_SERVER}" fi @@ -260,6 +276,7 @@ fi # Absolute path: ServerHandle only treats this value as a literal path when it # has more than one component, otherwise it falls back to a cargo_bin lookup. export COMPAT_BASELINE_SERVER="${BASELINE_SERVER}" +export COMPAT_BASELINE_CONFIG="${BASELINE_CONFIG}" echo "Running storage compatibility test..." # --run-ignored only: the test is #[ignore]d, so the normal lanes skip it. From 40be16846cd044d7450b078f7d207183f386e0ae Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Fri, 4 Sep 2026 23:03:23 +0200 Subject: [PATCH 064/182] chore(deps): Bump github.com/onsi/gomega from 1.42.1 to 1.43.0 in /bdd/go in the go group across 1 directory (#4057) --- bdd/go/go.mod | 2 +- bdd/go/go.sum | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/bdd/go/go.mod b/bdd/go/go.mod index d25ad1c130..ba8a624d9e 100644 --- a/bdd/go/go.mod +++ b/bdd/go/go.mod @@ -9,7 +9,7 @@ require ( github.com/cucumber/godog v0.16.0 github.com/google/uuid v1.6.0 github.com/onsi/ginkgo/v2 v2.32.1 - github.com/onsi/gomega v1.42.1 + github.com/onsi/gomega v1.43.0 ) require ( diff --git a/bdd/go/go.sum b/bdd/go/go.sum index 3ed6d2cc2c..75d27b37c7 100644 --- a/bdd/go/go.sum +++ b/bdd/go/go.sum @@ -53,8 +53,8 @@ github.com/mfridman/tparse v0.18.0 h1:wh6dzOKaIwkUGyKgOntDW4liXSo37qg5AXbIhkMV3v github.com/mfridman/tparse v0.18.0/go.mod h1:gEvqZTuCgEhPbYk/2lS3Kcxg1GmTxxU7kTC8DvP0i/A= github.com/onsi/ginkgo/v2 v2.32.1 h1:6tlvcDm/3sE8lGJbZ4+d4mO3RLy24/tQWOFzVSQNIfw= github.com/onsi/ginkgo/v2 v2.32.1/go.mod h1:+aXOY+vzZ5mu2iI2HpTZUPmM//oQfsNFX6gU9kNcA44= -github.com/onsi/gomega v1.42.1 h1:iN1rCUX+44NZ1Dc97MPoeFYbFR0vh8zxoxMFwKdyZ6I= -github.com/onsi/gomega v1.42.1/go.mod h1:REff/hsDsodHoKlWsP2mAPhu1+5/6hVYNf9rIEBpeSg= +github.com/onsi/gomega v1.43.0 h1:VlG/1FxqNxhSO+lq/OHBNaaqwiBK/mO8JbVkX9Y+FeU= +github.com/onsi/gomega v1.43.0/go.mod h1:REff/hsDsodHoKlWsP2mAPhu1+5/6hVYNf9rIEBpeSg= github.com/rogpeppe/go-internal v1.14.1 h1:UQB4HGPB6osV0SQTLymcB4TgvyWu6ZyliaW0tI/otEQ= github.com/rogpeppe/go-internal v1.14.1/go.mod h1:MaRKkUm5W0goXpeCfT7UZI6fk/L7L7so1lCWt35ZSgc= github.com/spf13/pflag v1.0.10 h1:4EBh2KAYBwaONj6b2Ye1GiHfwjqyROoF4RwYO+vPwFk= From 63d3273b714c168102dde52b88fe3fc50edd73b9 Mon Sep 17 00:00:00 2001 From: Maxim Levkov Date: Fri, 4 Sep 2026 16:29:34 -0700 Subject: [PATCH 065/182] feat(ci): add /pin and /unpin triage commands (#4060) --- .github/workflows/pr-triage-apply.yml | 106 +++++++++++++++++++++++++- CONTRIBUTING.md | 2 + 2 files changed, 106 insertions(+), 2 deletions(-) diff --git a/.github/workflows/pr-triage-apply.yml b/.github/workflows/pr-triage-apply.yml index 3c5e7e6b15..d5840f4f30 100644 --- a/.github/workflows/pr-triage-apply.yml +++ b/.github/workflows/pr-triage-apply.yml @@ -189,6 +189,10 @@ jobs: script: | const LABEL_REVIEW = 'S-waiting-on-review'; const LABEL_AUTHOR = 'S-waiting-on-author'; + // Orthogonal to the S-* pair rather than a third state: `pinned` + // is the only label stale-prs.yml exempts, so it is added and + // removed on its own and never goes through replaceStateLabel. + const LABEL_PINNED = 'pinned'; const COMMITTER_ASSOCS = new Set(['MEMBER', 'COLLABORATOR', 'OWNER']); // Wider gate for the implicit changes_requested -> author // flip: a formal "Request changes" review is a deliberate, @@ -228,6 +232,7 @@ jobs: '- `/ready` - back to `S-waiting-on-review` after addressing feedback', '- `/author` - flip to `S-waiting-on-author` while you finish changes', '- `/request-review @user-or-team` - request a reviewer', + '- `/pin` - exempt the PR from the stale bot, `/unpin` to undo', '', 'See [CONTRIBUTING.md](https://github.com/apache/iggy/blob/master/CONTRIBUTING.md#pr-triage-commands) for details.', ].join('\n'); @@ -300,6 +305,72 @@ jobs: }), `setLabels ${add ?? '(clear S-*)'}`); }; + // Create the label when absent rather than assume it exists. + // stale-prs.yml names `pinned` as its only exemption, but no + // repo config guarantees the label itself was ever created, and + // an exemption naming a label nothing can carry is inert. + // Whether the labels API materializes an unknown name is + // undocumented, so do it explicitly. Needs `issues: write`. + const ensurePinnedLabel = async () => { + try { + await withRetry(() => github.rest.issues.getLabel({ + owner: context.repo.owner, + repo: context.repo.repo, + name: LABEL_PINNED, + }), `getLabel ${LABEL_PINNED}`); + return true; + } catch (e) { + if (e.status !== 404) { + core.warning(`${LABEL_PINNED}: lookup failed: ${e.message}`); + return false; + } + } + try { + await withRetry(() => github.rest.issues.createLabel({ + owner: context.repo.owner, + repo: context.repo.repo, + name: LABEL_PINNED, + color: 'd4c5f9', + description: 'Exempt from the stale bot', + }), `createLabel ${LABEL_PINNED}`); + core.info(`created label ${LABEL_PINNED}`); + return true; + } catch (e) { + // 422 means a concurrent run created it between our get and create. + if (e.status === 422) return true; + core.warning(`${LABEL_PINNED}: create failed: ${e.message}`); + return false; + } + }; + + // Add-only / remove-only, so unlike replaceStateLabel there is no + // list-then-PUT window an unrelated label change can be lost in. + const setPinned = async (on) => { + if (on) { + if (!await ensurePinnedLabel()) return false; + await withRetry(() => github.rest.issues.addLabels({ + owner: context.repo.owner, + repo: context.repo.repo, + issue_number: prNumber, + labels: [LABEL_PINNED], + }), `addLabels ${LABEL_PINNED}`); + return true; + } + try { + await withRetry(() => github.rest.issues.removeLabel({ + owner: context.repo.owner, + repo: context.repo.repo, + issue_number: prNumber, + name: LABEL_PINNED, + }), `removeLabel ${LABEL_PINNED}`); + } catch (e) { + // 404 is the PR not carrying the label, or the label not existing + // at all. Both mean it is already in the state /unpin asks for. + if (e.status !== 404) throw e; + } + return true; + }; + // Best-effort commenter feedback. A failed reaction or reply // must never fail the run, so each swallows its own errors. // Reactions exist only for issue comments — review-triggered @@ -504,7 +575,7 @@ jobs: // The trailing `(?:\s|$)` rejects suffixed prose like // `/ready-to-merge` — `\b` fires at hyphen/slash and would // silently flip state. - const COMMAND_RE = /^[ \t]*\/(request-review|ready|author)(?:\s|$)/m; + const COMMAND_RE = /^[ \t]*\/(request-review|ready|author|unpin|pin)(?:\s|$)/m; if (!COMMAND_RE.test(body) && !reviewWantsAuthor) { core.info('no command in body and no actionable review state, skipping'); return; @@ -569,6 +640,8 @@ jobs: let sawReassign = false; let sawReady = false; let sawAuthor = false; + let sawPin = false; + let sawUnpin = false; // Outcome tracking for commenter feedback (see react() above). // applied: at least one command took effect. denied: at least // one recognized command was rejected for lack of permission. @@ -650,6 +723,34 @@ jobs: } continue; } + // Same gate as /ready and /request-review: the PR author or a + // committer. Exempting a PR from the stale bot is reversible and + // visible as a label, so it does not need the wider /author gate. + if (!sawPin && /^\/pin(?:\s|$)/.test(line)) { + sawPin = true; + if (!(isCommitter || isPrAuthor)) { + core.info(`/pin: ignored, ${commentAuthor} lacks permission`); + denied = true; + } else if (await setPinned(true)) { + core.info(`/pin: added ${LABEL_PINNED}`); + applied = true; + } else { + await reply('`/pin`: could not apply the `pinned` label.'); + } + continue; + } + if (!sawUnpin && /^\/unpin(?:\s|$)/.test(line)) { + sawUnpin = true; + if (!(isCommitter || isPrAuthor)) { + core.info(`/unpin: ignored, ${commentAuthor} lacks permission`); + denied = true; + } else { + await setPinned(false); + core.info(`/unpin: removed ${LABEL_PINNED}`); + applied = true; + } + continue; + } } if (reviewers.size > 0 || teamReviewers.size > 0) { @@ -699,7 +800,8 @@ jobs: } } - if (!sawReassign && !sawReady && !sawAuthor && !reviewWantsAuthor) { + if (!sawReassign && !sawReady && !sawAuthor && !sawPin && !sawUnpin + && !reviewWantsAuthor) { core.info('no command matched'); } diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index f4540816f6..70eaea20c7 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -146,6 +146,8 @@ line in a regular PR comment (not an inline review reply): | `/ready` | author or maintainer | mark `S-waiting-on-review` | | `/author` | maintainer or returning contributor | mark `S-waiting-on-author` | | `/request-review @user-or-team ...` | author or maintainer | request review from the listed `@user` / `@org/team` handles | +| `/pin` | author or maintainer | add `pinned`, exempting the PR from the stale bot | +| `/unpin` | author or maintainer | remove `pinned` | Some labels move on their own: opening or marking a non-draft PR ready sets `S-waiting-on-review`; a "Request changes" review sets `S-waiting-on-author`; From 6c47193fe44ca581ecdecdaa6fa9ce58370c650c Mon Sep 17 00:00:00 2001 From: Maxim Levkov Date: Fri, 4 Sep 2026 16:43:34 -0700 Subject: [PATCH 066/182] docs: say where a closed PR's findings go, and fix the stale timing (#4059) --- CONTRIBUTING.md | 14 +++++++++++++- 1 file changed, 13 insertions(+), 1 deletion(-) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 70eaea20c7..3d8744e6d6 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -166,7 +166,19 @@ PRs may be closed if: - Code not ran and tested locally - Mixed purposes or purposes not clear - Can't answer questions about the change -- Inactivity for longer than 7 days +- Inactivity, see [Stale PRs](#stale-prs) below + +Whoever closes leaves a comment saying why. If the thread holds a finding +that outlives the change, open an issue for it and link it from that +comment. The closed thread is the first place someone looks to find out +whether anything fell through. + +### Stale PRs + +A bot labels a PR `S-stale` after 7 days without activity and closes it 7 +days after that. A push, comment, review, reopen, or ready-for-review +clears the label. Drafts and PRs labeled `pinned` are exempt. Issues are +never labeled or closed by it. A closed PR can be reopened. ## Questions? From 59a9508cd58ec63e8a7dc78899f3f54e48fdacd3 Mon Sep 17 00:00:00 2001 From: Hubert Gruszecki Date: Sat, 5 Sep 2026 12:46:10 +0200 Subject: [PATCH 067/182] ci(python): restore Windows wheels for the PyPI release (#4049) --- .github/workflows/_build_python_wheels.yml | 78 +++++++++++++++-- scripts/ci/sync-python-interpreter-version.sh | 87 ++++++++++++++++++- 2 files changed, 159 insertions(+), 6 deletions(-) diff --git a/.github/workflows/_build_python_wheels.yml b/.github/workflows/_build_python_wheels.yml index ae066bb246..2a0cafc276 100644 --- a/.github/workflows/_build_python_wheels.yml +++ b/.github/workflows/_build_python_wheels.yml @@ -184,8 +184,17 @@ jobs: retention-days: 7 windows: - if: false # TODO(hubcio): temporarily disabled runs-on: windows-latest + strategy: + fail-fast: false + # Windows exposes no `python3.X` on PATH and maturin's py-launcher + # fallback does not see setup-python tool-cache installs, so one + # job cannot target several interpreters. One job per version + # instead, each falling through to maturin's default of the + # interpreter on PATH. 3.10 is Windows-only excluded, see + # WHEEL_MATRIX_SKIP in scripts/ci/sync-python-interpreter-version.sh. + matrix: + python-version: ["3.11", "3.12", "3.13"] steps: - name: Download latest copy script from master if: inputs.use_latest_ci @@ -209,22 +218,51 @@ jobs: - name: Setup Python uses: actions/setup-python@v7.0.0 with: - python-version: "3.10" + python-version: ${{ matrix.python-version }} architecture: x64 + - name: Setup Rust with cache + uses: ./.github/actions/utils/setup-rust-with-cache + with: + shared-key: python-wheels-windows + + # aws-lc-sys, reached through quinn's default rustls provider, + # assembles its x86_64 crypto with NASM. The runner image ships + # CMake but not NASM, and Chocolatey leaves it off PATH. + - name: Install NASM + shell: pwsh + run: | + $nasmDir = "C:\Program Files\NASM" + choco install nasm --no-progress -y + Add-Content -Path $env:GITHUB_PATH -Value $nasmDir + # GITHUB_PATH only reaches later steps, so prove the install + # landed where expected here rather than in a linker error. + $env:PATH = "$nasmDir;$env:PATH" + nasm -v + + - name: Generate third-party license manifest + shell: bash + run: | + TARGET="x86_64-pc-windows-msvc" + # HOME is Windows-style here, which MSYS tar mishandles as -C. + CARGO_BIN="$(cygpath -u "$HOME")/.cargo/bin" + curl -sSfL "https://github.com/EmbarkStudios/cargo-about/releases/download/0.9.0/cargo-about-0.9.0-${TARGET}.tar.gz" \ + | tar -xz -C "$CARGO_BIN" --strip-components=1 "cargo-about-0.9.0-${TARGET}/cargo-about.exe" + ./scripts/ci/third-party-licenses.sh --generate --manifest foreign/python/Cargo.toml --output foreign/python/LICENSE-binary + - name: Build wheels uses: PyO3/maturin-action@v1 with: target: x86_64 working-directory: foreign/python - args: --release --out dist --interpreter python3.10 python3.11 python3.12 python3.13 + args: --release --out dist sccache: "true" - name: Upload wheels if: inputs.upload_artifacts uses: actions/upload-artifact@v7 with: - name: wheels-windows-x64 + name: wheels-windows-x64-py${{ matrix.python-version }} path: foreign/python/dist retention-days: 7 @@ -265,7 +303,7 @@ jobs: collect: name: Collect all wheels - needs: [linux, macos, sdist] # TODO(hubcio): add windows back when re-enabled + needs: [linux, macos, windows, sdist] if: ${{ !cancelled() }} runs-on: ubuntu-latest outputs: @@ -312,3 +350,33 @@ jobs: - id: output run: echo "artifact_name=python-wheels-all" >> $GITHUB_OUTPUT + + # Runs last so the artifact and the summary survive for inspection + # when a leg is missing. A failed build job uploads nothing, so a + # short count means one did not contribute. These counts are not + # derived from the matrices above - update them together. + - name: Verify wheel coverage + run: | + EXPECTED_LINUX=16 # 4 build variants x 4 interpreters + EXPECTED_MACOS=8 # 2 targets x 4 interpreters + EXPECTED_WINDOWS=3 # 1 interpreter per job, 3.10 excluded + EXPECTED_SDIST=1 + + missing=0 + check() { + local label="$1" pattern="$2" expected="$3" actual + actual=$(find dist -maxdepth 1 -name "$pattern" | wc -l) + if [ "$actual" -ne "$expected" ]; then + echo "::error::${label}: expected ${expected}, found ${actual}" + missing=1 + else + echo "${label}: ${actual}" + fi + } + + check Linux '*linux*.whl' "$EXPECTED_LINUX" + check macOS '*macosx*.whl' "$EXPECTED_MACOS" + check Windows '*win*.whl' "$EXPECTED_WINDOWS" + check sdist '*.tar.gz' "$EXPECTED_SDIST" + + exit "$missing" diff --git a/scripts/ci/sync-python-interpreter-version.sh b/scripts/ci/sync-python-interpreter-version.sh index cc3ced1a36..8c074be73f 100755 --- a/scripts/ci/sync-python-interpreter-version.sh +++ b/scripts/ci/sync-python-interpreter-version.sh @@ -24,8 +24,15 @@ source "$(dirname "${BASH_SOURCE[0]}")/lib/init.sh" # Colors for output RED='\033[0;31m' GREEN='\033[0;32m' +YELLOW='\033[1;33m' NC='\033[0m' # No Color +# Minor versions deliberately not built as Windows wheels. Every other +# version in the pyproject classifiers must appear in the wheel job +# matrix, so adding a new supported version still fails this check +# until the matrix is updated. +WHEEL_MATRIX_SKIP=("3.10") + # Default mode MODE="" @@ -266,6 +273,12 @@ ensure_lock_python_requirement() { fi } +# Every supported minor version, oldest first, as declared by the +# pyproject classifiers. +read_classifier_versions() { + sed -nE 's/^ "Programming Language :: Python :: ([0-9]+\.[0-9]+)",$/\1/p' "$SOURCE_FILE" +} + ensure_wheel_interpreters() { local file="$1" local classifier_versions=() @@ -284,7 +297,7 @@ ensure_wheel_interpreters() { fi classifier_versions=() - while IFS= read -r _py_cls_tmp; do classifier_versions+=("$_py_cls_tmp"); done < <(sed -nE 's/^ "Programming Language :: Python :: ([0-9]+\.[0-9]+)",$/\1/p' "$SOURCE_FILE") + while IFS= read -r _py_cls_tmp; do classifier_versions+=("$_py_cls_tmp"); done < <(read_classifier_versions) if [ "${#classifier_versions[@]}" -eq 0 ]; then echo -e "${RED}✗${NC} $SOURCE_FILE: could not find Python version classifiers" @@ -327,6 +340,77 @@ ensure_wheel_interpreters() { fi } +# The Windows wheel job builds one interpreter per matrix entry rather +# than passing --interpreter, so its version list needs its own check. +ensure_wheel_matrix() { + local file="$1" + local expected="" + local current + local version + local skipped + local seen=0 + + TOTAL_CHECKS=$((TOTAL_CHECKS + 1)) + + if [ ! -f "$file" ]; then + echo -e "${RED}✗${NC} $file: file does not exist" + FAILED=1 + return + fi + + while IFS= read -r version; do + seen=$((seen + 1)) + for skipped in "${WHEEL_MATRIX_SKIP[@]}"; do + [ "$version" = "$skipped" ] && continue 2 + done + expected+=", \"${version}\"" + done < <(read_classifier_versions) + + if [ "$seen" -eq 0 ]; then + echo -e "${RED}✗${NC} $SOURCE_FILE: could not find Python version classifiers" + FAILED=1 + return + fi + + if [ -z "$expected" ]; then + echo -e "${RED}✗${NC} $SOURCE_FILE: WHEEL_MATRIX_SKIP excludes every classifier version" + FAILED=1 + return + fi + expected="[${expected#, }]" + + current=$(sed -nE 's/^[[:space:]]*python-version: (\[.*\])$/\1/p' "$file") + + if [ -z "$current" ]; then + echo -e "${RED}✗${NC} $file: could not find wheel matrix python-version list" + FAILED=1 + return + fi + + # --fix rewrites every match, so refuse to touch a file that grew a + # second list rather than collapsing both onto the same versions. + if [ "$(printf '%s\n' "$current" | wc -l)" -gt 1 ]; then + echo -e "${RED}✗${NC} $file: multiple python-version lists, cannot pick one" + FAILED=1 + return + fi + + if [ "$current" = "$expected" ]; then + echo -e "${GREEN}✓${NC} $file: wheel matrix Python versions" + return + fi + + if [ "$MODE" = "fix" ]; then + sed -i.bak -E "s|^([[:space:]]*python-version: )\[.*\]$|\\1${expected}|" "$file" + rm -f "$file.bak" + FIXED_CHECKS=$((FIXED_CHECKS + 1)) + echo -e "${GREEN}Fixed${NC} $file: wheel matrix Python versions" + else + echo -e "${RED}✗${NC} $file: wheel matrix Python versions are not $expected" + FAILED=1 + fi +} + ensure_classifiers "$SOURCE_FILE" PYTHON_VERSION_FILES=( @@ -401,6 +485,7 @@ ensure_line \ "wheel workflow setup-python versions" ensure_wheel_interpreters ".github/workflows/_build_python_wheels.yml" +ensure_wheel_matrix ".github/workflows/_build_python_wheels.yml" PYLOCK_FILES=( "foreign/python/pylock.toml" From 7b78014c938bf52331f1d34ec5552047b4b1e7ee Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=C5=81ukasz=20Zborek?= Date: Sun, 6 Sep 2026 23:31:57 +0200 Subject: [PATCH 068/182] fix(csharp): update string length validation to use UTF-8 byte count (#4072) Identifier, Partitioning and header types checked string.Length, so a non-ASCII name under 255 characters could exceed 255 bytes and be truncated into the one byte length prefix, sending a frame the server parses as a shorter name followed by garbage. Most TcpContracts skipped the check entirely. Identifier also compared and hashed Value by reference. Every length-prefixed name, credential, token and header field now goes through one WireName rule mirroring the server's 1 to 255 byte bound. Identifier and Partitioning derive Length from Value and reject oversize arrays in the initializer. Identifier equality uses byte contents. User contracts validate credentials against CredentialBounds before serializing. Contract tests cover empty, at-limit, over-limit and non-ASCII inputs for each entry point. #4056 --- .../Iggy_SDK/Contracts/Tcp/TcpContracts.cs | 60 +++-- .../csharp/Iggy_SDK/Extensions/Extensions.cs | 8 +- foreign/csharp/Iggy_SDK/Headers/HeaderKey.cs | 17 +- .../csharp/Iggy_SDK/Headers/HeaderValue.cs | 12 +- foreign/csharp/Iggy_SDK/Identifier.cs | 56 +++-- .../csharp/Iggy_SDK/IggyClient/IIggyStream.cs | 2 +- .../csharp/Iggy_SDK/IggyClient/IIggyTopic.cs | 4 +- .../Implementations/HttpMessageStream.cs | 2 +- foreign/csharp/Iggy_SDK/Iggy_SDK.csproj | 2 +- foreign/csharp/Iggy_SDK/Kinds/Partitioning.cs | 46 ++-- .../Iggy_SDK/Utils/TcpMessageStreamHelpers.cs | 5 +- foreign/csharp/Iggy_SDK/Utils/WireName.cs | 53 +++++ .../WireNameLengthContractsTests.cs | 217 ++++++++++++++++++ .../UtilityTests/HeaderValueTests.cs | 7 + .../IdentifiersByteSerializationTests.cs | 131 +++++++++++ 15 files changed, 526 insertions(+), 96 deletions(-) create mode 100644 foreign/csharp/Iggy_SDK/Utils/WireName.cs create mode 100644 foreign/csharp/Iggy_SDK_Tests/ContractsTests/WireNameLengthContractsTests.cs diff --git a/foreign/csharp/Iggy_SDK/Contracts/Tcp/TcpContracts.cs b/foreign/csharp/Iggy_SDK/Contracts/Tcp/TcpContracts.cs index 74b3821d7a..4a825f6d42 100644 --- a/foreign/csharp/Iggy_SDK/Contracts/Tcp/TcpContracts.cs +++ b/foreign/csharp/Iggy_SDK/Contracts/Tcp/TcpContracts.cs @@ -27,13 +27,14 @@ using Apache.Iggy.Headers; using Apache.Iggy.Kinds; using Apache.Iggy.Messages; +using Apache.Iggy.Utils; +using Apache.Iggy.Vsr; using Partitioning = Apache.Iggy.Kinds.Partitioning; namespace Apache.Iggy.Contracts.Tcp; internal static class TcpContracts { - private const int MaxWireNameLength = 255; ///

Frames wider than this are built on the heap instead of the stack. private const int MaxStackAllocBytes = 1024; @@ -43,7 +44,7 @@ internal static class TcpContracts internal static byte[] LoginWithPersonalAccessToken(string token) { - var tokenLength = Encoding.UTF8.GetByteCount(token); + var tokenLength = WireName.ByteCount(token, nameof(token)); Span bytes = stackalloc byte[5 + tokenLength]; bytes[0] = (byte)tokenLength; Encoding.UTF8.GetBytes(token, bytes[1..(1 + tokenLength)]); @@ -52,7 +53,7 @@ internal static byte[] LoginWithPersonalAccessToken(string token) internal static byte[] DeletePersonalRequestToken(string name) { - var nameLength = Encoding.UTF8.GetByteCount(name); + var nameLength = WireName.ByteCount(name, nameof(name)); Span bytes = stackalloc byte[5 + nameLength]; bytes[0] = (byte)nameLength; Encoding.UTF8.GetBytes(name, bytes[1..(1 + nameLength)]); @@ -61,7 +62,7 @@ internal static byte[] DeletePersonalRequestToken(string name) internal static byte[] CreatePersonalAccessToken(string name, ulong? expiry) { - var nameLength = Encoding.UTF8.GetByteCount(name); + var nameLength = WireName.ByteCount(name, nameof(name)); Span bytes = stackalloc byte[1 + nameLength + 8]; bytes[0] = (byte)nameLength; Encoding.UTF8.GetBytes(name, bytes[1..(1 + nameLength)]); @@ -94,13 +95,11 @@ internal static byte[] LoginUser(string userName, string password, string? versi { var bytes = new List(); - var usernameBytes = Encoding.UTF8.GetBytes(userName); - bytes.Add((byte)usernameBytes.Length); - bytes.AddRange(usernameBytes); + bytes.Add((byte)WireName.ByteCount(userName, nameof(userName))); + bytes.AddRange(Encoding.UTF8.GetBytes(userName)); - var passwordBytes = Encoding.UTF8.GetBytes(password); - bytes.Add((byte)passwordBytes.Length); - bytes.AddRange(passwordBytes); + bytes.Add((byte)WireName.ByteCount(password, nameof(password))); + bytes.AddRange(Encoding.UTF8.GetBytes(password)); if (!string.IsNullOrEmpty(version)) { @@ -129,6 +128,8 @@ internal static byte[] LoginUser(string userName, string password, string? versi internal static byte[] ChangePassword(Identifier userId, string currentPassword, string newPassword) { + CredentialBounds.ValidatePassword(currentPassword); + CredentialBounds.ValidatePassword(newPassword); var currentPasswordLength = Encoding.UTF8.GetByteCount(currentPassword); var newPasswordLength = Encoding.UTF8.GetByteCount(newPassword); var length = userId.Length + 2 + currentPasswordLength + newPasswordLength + 2; @@ -157,9 +158,15 @@ internal static byte[] UpdatePermissions(Identifier userId, Permissions? permiss internal static byte[] UpdateUser(Identifier userId, string? userName, UserStatus? status) { + if (userName is not null) + { + CredentialBounds.ValidateUsername(userName); + } + var userNameLength = userName is null ? 0 : Encoding.UTF8.GetByteCount(userName); - var length = userId.Length + 2 + userNameLength - + (status is not null ? 2 : 1) + 1 + 1; + var length = userId.Length + 2 + + (userName is null ? 1 : 2 + userNameLength) + + (status is not null ? 2 : 1); Span bytes = stackalloc byte[length]; bytes.WriteBytesFromIdentifier(userId); @@ -196,6 +203,8 @@ internal static byte[] UpdateUser(Identifier userId, string? userName, UserStatu internal static byte[] CreateUser(string userName, string password, UserStatus status, Permissions? permissions = null) { + CredentialBounds.ValidateUsername(userName); + CredentialBounds.ValidatePassword(password); var userNameLength = Encoding.UTF8.GetByteCount(userName); var passwordLength = Encoding.UTF8.GetByteCount(password); var permissionsBytes = permissions is not null ? GetBytesFromPermissions(permissions) : []; @@ -558,7 +567,7 @@ internal static void WriteHeadersTo(Span destination, Dictionary bytes = stackalloc byte[nameLength + 1]; bytes[0] = (byte)nameLength; Encoding.UTF8.GetBytes(name, bytes[1..]); @@ -567,7 +576,7 @@ internal static byte[] CreateStream(string name) internal static byte[] UpdateStream(Identifier streamId, string name) { - var nameLength = Encoding.UTF8.GetByteCount(name); + var nameLength = WireName.ByteCount(name, nameof(name)); Span bytes = stackalloc byte[streamId.Length + nameLength + 3]; bytes.WriteBytesFromIdentifier(streamId); var position = 2 + streamId.Length; @@ -578,7 +587,7 @@ internal static byte[] UpdateStream(Identifier streamId, string name) internal static byte[] CreateGroup(Identifier streamId, Identifier topicId, string name) { - var nameLength = Encoding.UTF8.GetByteCount(name); + var nameLength = WireName.ByteCount(name, nameof(name)); Span bytes = stackalloc byte[2 + streamId.Length + 2 + topicId.Length + 1 + nameLength]; bytes.WriteBytesFromStreamAndTopicIdentifiers(streamId, topicId); var position = 2 + streamId.Length + 2 + topicId.Length; @@ -664,7 +673,7 @@ internal static byte[] UpdateTopic(Identifier streamId, Identifier topicId, stri } var optionsLength = HeadersByteLength(options); - var nameLength = WireNameLength(name, nameof(name)); + var nameLength = WireName.ByteCount(name, nameof(name)); var length = 4 + streamId.Length + topicId.Length + 1 + nameLength + optionsLength; var rented = length > MaxStackAllocBytes ? ArrayPool.Shared.Rent(length) : null; try @@ -718,7 +727,7 @@ internal static byte[] CreateTopic(Identifier streamId, string name, uint partit } var optionsLength = HeadersByteLength(options); - var nameLength = WireNameLength(name, nameof(name)); + var nameLength = WireName.ByteCount(name, nameof(name)); var length = 2 + streamId.Length + 4 + 1 + nameLength + optionsLength; var rented = length > MaxStackAllocBytes ? ArrayPool.Shared.Rent(length) : null; try @@ -742,23 +751,6 @@ internal static byte[] CreateTopic(Identifier streamId, string name, uint partit } } - /// - /// UTF-8 byte count of a length-prefixed wire name, bounded by what its one-byte prefix can carry. - /// - private static int WireNameLength(string name, string parameterName) - { - var length = Encoding.UTF8.GetByteCount(name); - if (length > MaxWireNameLength) - { - // Truncating into the prefix would ship a frame the server parses as a shorter - // name followed by garbage, instead of a request it can reject. - throw new ArgumentException( - $"{parameterName} must be at most {MaxWireNameLength} UTF-8 bytes, got {length}.", parameterName); - } - - return length; - } - internal static byte[] GetTopicById(Identifier streamId, Identifier topicId) { Span bytes = stackalloc byte[2 + streamId.Length + 2 + topicId.Length]; diff --git a/foreign/csharp/Iggy_SDK/Extensions/Extensions.cs b/foreign/csharp/Iggy_SDK/Extensions/Extensions.cs index 3e2063a3c9..137279ba9b 100644 --- a/foreign/csharp/Iggy_SDK/Extensions/Extensions.cs +++ b/foreign/csharp/Iggy_SDK/Extensions/Extensions.cs @@ -75,12 +75,12 @@ internal static void WriteBytesFromStreamAndTopicIdentifiers(this Span byt { bytes[startPos] = streamId.Kind.GetByte(); bytes[startPos + 1] = (byte)streamId.Length; - streamId.Value.CopyTo(bytes[(startPos + 2)..(startPos + 2 + streamId.Length)]); + streamId.Bytes.CopyTo(bytes[(startPos + 2)..(startPos + 2 + streamId.Length)]); var position = startPos + 2 + streamId.Length; bytes[position] = topicId.Kind.GetByte(); bytes[position + 1] = (byte)topicId.Length; - topicId.Value.CopyTo(bytes[(position + 2)..(position + 2 + topicId.Length)]); + topicId.Bytes.CopyTo(bytes[(position + 2)..(position + 2 + topicId.Length)]); } [MethodImpl(MethodImplOptions.AggressiveInlining)] @@ -88,7 +88,7 @@ internal static void WriteBytesFromIdentifier(this Span bytes, Identifier { bytes[startPos + 0] = identifier.Kind.GetByte(); bytes[startPos + 1] = (byte)identifier.Length; - identifier.Value.CopyTo(bytes[(startPos + 2)..]); + identifier.Bytes.CopyTo(bytes[(startPos + 2)..]); } [MethodImpl(MethodImplOptions.AggressiveInlining)] @@ -96,7 +96,7 @@ internal static void WriteBytesFromPartitioning(this Span bytes, Partition { bytes[startPos + 0] = identifier.Kind.GetByte(); bytes[startPos + 1] = (byte)identifier.Length; - identifier.Value.CopyTo(bytes[(startPos + 2)..]); + identifier.Bytes.CopyTo(bytes[(startPos + 2)..]); } } diff --git a/foreign/csharp/Iggy_SDK/Headers/HeaderKey.cs b/foreign/csharp/Iggy_SDK/Headers/HeaderKey.cs index b6a19ad213..b7fffbb248 100644 --- a/foreign/csharp/Iggy_SDK/Headers/HeaderKey.cs +++ b/foreign/csharp/Iggy_SDK/Headers/HeaderKey.cs @@ -16,6 +16,7 @@ // under the License. using System.Text; +using Apache.Iggy.Utils; namespace Apache.Iggy.Headers; @@ -37,20 +38,18 @@ namespace Apache.Iggy.Headers; /// /// Creates a HeaderKey from a string value. /// - /// The string value (must be 1-255 characters). + /// The string value (must be 1-255 UTF-8 bytes). /// A new HeaderKey with String kind. /// Thrown when value length is invalid. public static HeaderKey FromString(string val) { - if (val.Length is 0 or > 255) - { - throw new ArgumentException("Value has incorrect size, must be between 1 and 255", nameof(val)); - } + var bytes = Encoding.UTF8.GetBytes(val); + WireName.Validate(bytes.Length, nameof(val)); return new HeaderKey { Kind = HeaderKind.String, - Value = Encoding.UTF8.GetBytes(val) + Value = bytes }; } @@ -92,11 +91,7 @@ public override int GetHashCode() { var hash = new HashCode(); hash.Add(Kind); - foreach (var b in Value) - { - hash.Add(b); - } - + hash.AddBytes(Value); return hash.ToHashCode(); } diff --git a/foreign/csharp/Iggy_SDK/Headers/HeaderValue.cs b/foreign/csharp/Iggy_SDK/Headers/HeaderValue.cs index d551965e52..8ce5d9f8a6 100644 --- a/foreign/csharp/Iggy_SDK/Headers/HeaderValue.cs +++ b/foreign/csharp/Iggy_SDK/Headers/HeaderValue.cs @@ -19,6 +19,7 @@ using System.Globalization; using System.Text; using Apache.Iggy.Extensions; +using Apache.Iggy.Utils; namespace Apache.Iggy.Headers; @@ -42,8 +43,11 @@ public readonly struct HeaderValue ///
/// Raw bytes /// + /// Thrown when the value is empty or longer than 255 bytes. public static HeaderValue FromBytes(byte[] value) { + WireName.Validate(value.Length, nameof(value)); + return new HeaderValue { Kind = HeaderKind.Raw, @@ -59,15 +63,13 @@ public static HeaderValue FromBytes(byte[] value) /// public static HeaderValue FromString(string value) { - if (value.Length is 0 or > 255) - { - throw new ArgumentException("Value has incorrect size, must be between 1 and 255", nameof(value)); - } + var bytes = Encoding.UTF8.GetBytes(value); + WireName.Validate(bytes.Length, nameof(value)); return new HeaderValue { Kind = HeaderKind.String, - Value = Encoding.UTF8.GetBytes(value) + Value = bytes }; } diff --git a/foreign/csharp/Iggy_SDK/Identifier.cs b/foreign/csharp/Iggy_SDK/Identifier.cs index 8800fc542a..69e3692890 100644 --- a/foreign/csharp/Iggy_SDK/Identifier.cs +++ b/foreign/csharp/Iggy_SDK/Identifier.cs @@ -18,6 +18,7 @@ using System.Buffers.Binary; using System.Text; using Apache.Iggy.Enums; +using Apache.Iggy.Utils; namespace Apache.Iggy; @@ -32,14 +33,36 @@ namespace Apache.Iggy; public required IdKind Kind { get; init; } /// - /// Identifier length in bytes. + /// Identifier length in bytes, always derived from . + /// The initializer is kept for compatibility and its value is ignored. /// - public required int Length { get; init; } + public int Length + { + get => _value.Length; + init { } + } /// - /// Identifier value as bytes. + /// Copy of the identifier value as bytes, at most 255 of them. /// - public required byte[] Value { get; init; } + /// Thrown when the value is longer than 255 bytes. + public required byte[] Value + { + get => _value.ToArray(); + init + { + ArgumentNullException.ThrowIfNull(value); + ArgumentOutOfRangeException.ThrowIfGreaterThan(value.Length, WireName.MAX_LENGTH, nameof(Value)); + _value = value.ToArray(); + } + } + + /// + /// Read-only view of the value bytes for serialization, without the defensive copy of . + /// + internal ReadOnlySpan Bytes => _value; + + private readonly byte[] _value; /// /// Creates a numeric identifier from a value. @@ -65,7 +88,6 @@ public static Identifier Numeric(uint value) return new Identifier { Kind = IdKind.Numeric, - Length = 4, Value = bytes }; } @@ -78,16 +100,13 @@ public static Identifier Numeric(uint value) /// Thrown when the value is too long or too short. public static Identifier String(string value) { - if (value.Length is 0 or > 255) - { - throw new ArgumentException("Value has incorrect size, must be between 1 and 255", nameof(value)); - } + var bytes = Encoding.UTF8.GetBytes(value); + WireName.Validate(bytes.Length, nameof(value)); return new Identifier { Kind = IdKind.String, - Length = value.Length, - Value = Encoding.UTF8.GetBytes(value) + Value = bytes }; } @@ -96,8 +115,8 @@ public override string ToString() { return Kind switch { - IdKind.Numeric => BitConverter.ToInt32(Value).ToString(), - IdKind.String => Encoding.UTF8.GetString(Value), + IdKind.Numeric => BitConverter.ToInt32(_value).ToString(), + IdKind.String => Encoding.UTF8.GetString(_value), _ => throw new ArgumentOutOfRangeException() }; } @@ -114,7 +133,7 @@ public uint GetUInt32() throw new InvalidOperationException("Identifier is not numeric"); } - return BinaryPrimitives.ReadUInt32LittleEndian(Value); + return BinaryPrimitives.ReadUInt32LittleEndian(_value); } /// @@ -129,7 +148,7 @@ public string GetString() throw new InvalidOperationException("Identifier is not string"); } - return Encoding.UTF8.GetString(Value); + return Encoding.UTF8.GetString(_value); } /// @@ -139,7 +158,7 @@ public string GetString() /// True if the current identifier is equal to the other identifier; otherwise, false. public bool Equals(Identifier other) { - return Kind == other.Kind && Value.Equals(other.Value); + return Kind == other.Kind && Bytes.SequenceEqual(other.Bytes); } /// @@ -151,6 +170,9 @@ public override bool Equals(object? obj) /// public override int GetHashCode() { - return HashCode.Combine((int)Kind, Value); + var hash = new HashCode(); + hash.Add(Kind); + hash.AddBytes(_value); + return hash.ToHashCode(); } } diff --git a/foreign/csharp/Iggy_SDK/IggyClient/IIggyStream.cs b/foreign/csharp/Iggy_SDK/IggyClient/IIggyStream.cs index 233139f0ba..c861b99382 100644 --- a/foreign/csharp/Iggy_SDK/IggyClient/IIggyStream.cs +++ b/foreign/csharp/Iggy_SDK/IggyClient/IIggyStream.cs @@ -29,7 +29,7 @@ public interface IIggyStream /// Creates a new stream with the specified name. /// /// - /// The stream name must be unique within the Iggy instance and has a maximum length of 255 characters. + /// The stream name must be unique within the Iggy instance and has a maximum length of 255 UTF-8 bytes. /// /// The unique name of the stream to create. /// The cancellation token to cancel the operation. diff --git a/foreign/csharp/Iggy_SDK/IggyClient/IIggyTopic.cs b/foreign/csharp/Iggy_SDK/IggyClient/IIggyTopic.cs index 391fdff6e2..e44461e9af 100644 --- a/foreign/csharp/Iggy_SDK/IggyClient/IIggyTopic.cs +++ b/foreign/csharp/Iggy_SDK/IggyClient/IIggyTopic.cs @@ -55,7 +55,7 @@ public interface IIggyTopic /// Additional parameters control message expiry, compression, replication, and maximum size. /// /// The identifier of the stream where the topic will be created (numeric ID or name). - /// The unique name of the topic (max 255 characters). + /// The unique name of the topic (max 255 UTF-8 bytes). /// The number of partitions for the topic (max 1000). /// The compression algorithm to use for messages (default: None). /// The message expiry period (0 for server default, MaxValue for never expire). @@ -86,7 +86,7 @@ public interface IIggyTopic /// /// The identifier of the stream containing the topic (numeric ID or name). /// The identifier of the topic to update (numeric ID or name). - /// The new name for the topic (max 255 characters). + /// The new name for the topic (max 255 UTF-8 bytes). /// The new compression algorithm to use (default: None). /// The new maximum size of the topic in bytes (0 = unlimited). /// The new message expiry period (0 for server default, MaxValue for never expire). diff --git a/foreign/csharp/Iggy_SDK/IggyClient/Implementations/HttpMessageStream.cs b/foreign/csharp/Iggy_SDK/IggyClient/Implementations/HttpMessageStream.cs index 8310f26c73..61fe4ce5f5 100644 --- a/foreign/csharp/Iggy_SDK/IggyClient/Implementations/HttpMessageStream.cs +++ b/foreign/csharp/Iggy_SDK/IggyClient/Implementations/HttpMessageStream.cs @@ -999,7 +999,7 @@ private async ValueTask ResolvePartitioningAsync(Identifier stream var partition = partitioning.Kind switch { Enums.Partitioning.Balanced => _groupState.NextBalancedPartition(key, partitionCount.Value), - Enums.Partitioning.MessageKey => XxHash32.HashToUInt32(partitioning.Value) % partitionCount.Value, + Enums.Partitioning.MessageKey => XxHash32.HashToUInt32(partitioning.Bytes) % partitionCount.Value, _ => throw new FeatureUnavailableException() }; diff --git a/foreign/csharp/Iggy_SDK/Iggy_SDK.csproj b/foreign/csharp/Iggy_SDK/Iggy_SDK.csproj index 3cacac12e1..29c45d0ed4 100644 --- a/foreign/csharp/Iggy_SDK/Iggy_SDK.csproj +++ b/foreign/csharp/Iggy_SDK/Iggy_SDK.csproj @@ -26,7 +26,7 @@ under the License. net8.0;net10.0 Apache.Iggy Apache.Iggy - 0.9.0-edge.7 + 0.9.0-edge.8 true diff --git a/foreign/csharp/Iggy_SDK/Kinds/Partitioning.cs b/foreign/csharp/Iggy_SDK/Kinds/Partitioning.cs index 9efbdd46ef..662e50f179 100644 --- a/foreign/csharp/Iggy_SDK/Kinds/Partitioning.cs +++ b/foreign/csharp/Iggy_SDK/Kinds/Partitioning.cs @@ -17,6 +17,7 @@ using System.Buffers.Binary; using System.Text; +using Apache.Iggy.Utils; namespace Apache.Iggy.Kinds; @@ -31,14 +32,36 @@ public readonly struct Partitioning public required Enums.Partitioning Kind { get; init; } /// - /// Length of the partitioning value. + /// Length of the partitioning value in bytes, always derived from . + /// The initializer is kept for compatibility and its value is ignored. /// - public required int Length { get; init; } + public int Length + { + get => _value.Length; + init { } + } /// - /// Partitioning value as bytes. + /// Copy of the partitioning value as bytes, at most 255 of them. /// - public required byte[] Value { get; init; } + /// Thrown when the value is longer than 255 bytes. + public required byte[] Value + { + get => _value.ToArray(); + init + { + ArgumentNullException.ThrowIfNull(value); + ArgumentOutOfRangeException.ThrowIfGreaterThan(value.Length, WireName.MAX_LENGTH, nameof(Value)); + _value = value.ToArray(); + } + } + + /// + /// Read-only view of the value bytes for serialization, without the defensive copy of . + /// + internal ReadOnlySpan Bytes => _value; + + private readonly byte[] _value; /// /// Creates a partitioning strategy that use default partitioning (balanced). @@ -49,7 +72,6 @@ public static Partitioning None() return new Partitioning { Kind = Enums.Partitioning.Balanced, - Length = 0, Value = [] }; } @@ -78,7 +100,6 @@ public static Partitioning PartitionId(uint value) return new Partitioning { Kind = Enums.Partitioning.PartitionId, - Length = 4, Value = bytes }; } @@ -91,16 +112,13 @@ public static Partitioning PartitionId(uint value) /// Thrown when the value size is incorrect public static Partitioning EntityIdString(string value) { - if (value.Length is 0 or > 255) - { - throw new ArgumentException("Value has incorrect size, must be between 1 and 255", nameof(value)); - } + var bytes = Encoding.UTF8.GetBytes(value); + WireName.Validate(bytes.Length, nameof(value)); return new Partitioning { Kind = Enums.Partitioning.MessageKey, - Length = value.Length, - Value = Encoding.UTF8.GetBytes(value) + Value = bytes }; } @@ -120,7 +138,6 @@ public static Partitioning EntityIdBytes(byte[] value) return new Partitioning { Kind = Enums.Partitioning.MessageKey, - Length = value.Length, Value = value }; } @@ -137,7 +154,6 @@ public static Partitioning EntityIdInt(int value) return new Partitioning { Kind = Enums.Partitioning.MessageKey, - Length = 4, Value = bytes.ToArray() }; } @@ -154,7 +170,6 @@ public static Partitioning EntityIdUlong(ulong value) return new Partitioning { Kind = Enums.Partitioning.MessageKey, - Length = 8, Value = bytes.ToArray() }; } @@ -170,7 +185,6 @@ public static Partitioning EntityIdGuid(Guid value) return new Partitioning { Kind = Enums.Partitioning.MessageKey, - Length = 16, Value = bytes }; } diff --git a/foreign/csharp/Iggy_SDK/Utils/TcpMessageStreamHelpers.cs b/foreign/csharp/Iggy_SDK/Utils/TcpMessageStreamHelpers.cs index 3aaab7c188..030d79ab2f 100644 --- a/foreign/csharp/Iggy_SDK/Utils/TcpMessageStreamHelpers.cs +++ b/foreign/csharp/Iggy_SDK/Utils/TcpMessageStreamHelpers.cs @@ -61,10 +61,7 @@ internal static byte[] GetBytesFromIdentifier(Identifier identifier) _ => throw new ArgumentOutOfRangeException() }; bytes[1] = (byte)identifier.Length; - for (var i = 0; i < identifier.Length; i++) - { - bytes[i + 2] = identifier.Value[i]; - } + identifier.Bytes.CopyTo(bytes[2..]); return bytes.ToArray(); } diff --git a/foreign/csharp/Iggy_SDK/Utils/WireName.cs b/foreign/csharp/Iggy_SDK/Utils/WireName.cs new file mode 100644 index 0000000000..206fcf1c48 --- /dev/null +++ b/foreign/csharp/Iggy_SDK/Utils/WireName.cs @@ -0,0 +1,53 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +using System.Text; + +namespace Apache.Iggy.Utils; + +/// +/// Length rule for every length-prefixed name, password, token and header field on the wire, +/// mirroring the server's WireName and header field limits: 1 to 255 bytes. +/// +internal static class WireName +{ + internal const int MAX_LENGTH = 255; + + /// + /// Validates the UTF-8 byte count of and returns it. + /// + internal static int ByteCount(string value, string parameterName) + { + return Validate(Encoding.UTF8.GetByteCount(value), parameterName); + } + + /// + /// Validates an already computed UTF-8 byte count and returns it. + /// + internal static int Validate(int byteCount, string parameterName) + { + if (byteCount is 0 or > MAX_LENGTH) + { + // Truncating into the prefix would ship a frame the server parses as a shorter + // name followed by garbage, instead of a request it can reject. + throw new ArgumentException( + $"{parameterName} must be 1 to {MAX_LENGTH} UTF-8 bytes, got {byteCount}.", parameterName); + } + + return byteCount; + } +} diff --git a/foreign/csharp/Iggy_SDK_Tests/ContractsTests/WireNameLengthContractsTests.cs b/foreign/csharp/Iggy_SDK_Tests/ContractsTests/WireNameLengthContractsTests.cs new file mode 100644 index 0000000000..0b62ec86c2 --- /dev/null +++ b/foreign/csharp/Iggy_SDK_Tests/ContractsTests/WireNameLengthContractsTests.cs @@ -0,0 +1,217 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +using Apache.Iggy.Contracts.Tcp; +using Apache.Iggy.Enums; +using Apache.Iggy.Exceptions; +using Partitioning = Apache.Iggy.Kinds.Partitioning; +using Apache.Iggy.Vsr; + +namespace Apache.Iggy.Tests.ContractsTests; + +public sealed class WireNameLengthContractsTests +{ + // 129 characters, but 258 UTF-8 bytes: a character count would let this through. + private static readonly string OverLimit = new('ż', 129); + + // 128 characters, exactly 255 UTF-8 bytes: the widest name the one-byte prefix can carry. + private static readonly string AtLimit = new string('ż', 127) + "a"; + + private static readonly Identifier Id = Identifier.Numeric(1); + + public static TheoryData> EmptyCases => new() + { + { nameof(TcpContracts.CreateStream), "name", () => TcpContracts.CreateStream("") }, + { nameof(TcpContracts.UpdateStream), "name", () => TcpContracts.UpdateStream(Id, "") }, + { nameof(TcpContracts.CreateGroup), "name", () => TcpContracts.CreateGroup(Id, Id, "") }, + { nameof(TcpContracts.CreatePersonalAccessToken), "name", () => TcpContracts.CreatePersonalAccessToken("", null) }, + { nameof(TcpContracts.DeletePersonalRequestToken), "name", () => TcpContracts.DeletePersonalRequestToken("") }, + { nameof(TcpContracts.LoginWithPersonalAccessToken), "token", () => TcpContracts.LoginWithPersonalAccessToken("") }, + { nameof(TcpContracts.LoginUser) + "/userName", "userName", () => TcpContracts.LoginUser("", "pass", null, null) }, + { nameof(TcpContracts.LoginUser) + "/password", "password", () => TcpContracts.LoginUser("user", "", null, null) } + }; + + public static TheoryData> OverLimitCases => new() + { + { nameof(TcpContracts.CreateStream), "name", () => TcpContracts.CreateStream(OverLimit) }, + { nameof(TcpContracts.UpdateStream), "name", () => TcpContracts.UpdateStream(Id, OverLimit) }, + { nameof(TcpContracts.CreateGroup), "name", () => TcpContracts.CreateGroup(Id, Id, OverLimit) }, + { nameof(TcpContracts.CreatePersonalAccessToken), "name", () => TcpContracts.CreatePersonalAccessToken(OverLimit, null) }, + { nameof(TcpContracts.DeletePersonalRequestToken), "name", () => TcpContracts.DeletePersonalRequestToken(OverLimit) }, + { nameof(TcpContracts.LoginWithPersonalAccessToken), "token", () => TcpContracts.LoginWithPersonalAccessToken(OverLimit) }, + { nameof(TcpContracts.LoginUser) + "/userName", "userName", () => TcpContracts.LoginUser(OverLimit, "pass", null, null) }, + { nameof(TcpContracts.LoginUser) + "/password", "password", () => TcpContracts.LoginUser("user", OverLimit, null, null) } + }; + + // Each frame carries the name right after a fixed-size prefix; the offset says where its length byte sits. + public static TheoryData> AtLimitCases => new() + { + { nameof(TcpContracts.CreateStream), 0, () => TcpContracts.CreateStream(AtLimit) }, + { nameof(TcpContracts.UpdateStream), 2 + Id.Length, () => TcpContracts.UpdateStream(Id, AtLimit) }, + { nameof(TcpContracts.CreateGroup), 2 * (2 + Id.Length), () => TcpContracts.CreateGroup(Id, Id, AtLimit) }, + { nameof(TcpContracts.CreatePersonalAccessToken), 0, () => TcpContracts.CreatePersonalAccessToken(AtLimit, null) }, + { nameof(TcpContracts.DeletePersonalRequestToken), 0, () => TcpContracts.DeletePersonalRequestToken(AtLimit) }, + { nameof(TcpContracts.LoginWithPersonalAccessToken), 0, () => TcpContracts.LoginWithPersonalAccessToken(AtLimit) }, + { nameof(TcpContracts.LoginUser) + "/userName", 0, () => TcpContracts.LoginUser(AtLimit, "pass", null, null) }, + { nameof(TcpContracts.LoginUser) + "/password", 1 + 4, () => TcpContracts.LoginUser("user", AtLimit, null, null) } + }; + + // User management shares the server's credential bounds with the login path, not the 1-255 wire rule. + public static TheoryData> CredentialCases => new() + { + { nameof(TcpContracts.CreateUser) + "/empty userName", VsrError.INVALID_USERNAME, () => TcpContracts.CreateUser("", "pass", UserStatus.Active) }, + { nameof(TcpContracts.CreateUser) + "/short userName", VsrError.INVALID_USERNAME, () => TcpContracts.CreateUser("ab", "pass", UserStatus.Active) }, + { nameof(TcpContracts.CreateUser) + "/long userName", VsrError.INVALID_USERNAME, () => TcpContracts.CreateUser(new string('a', 51), "pass", UserStatus.Active) }, + { nameof(TcpContracts.CreateUser) + "/empty password", VsrError.INVALID_PASSWORD, () => TcpContracts.CreateUser("user", "", UserStatus.Active) }, + { nameof(TcpContracts.CreateUser) + "/long password", VsrError.INVALID_PASSWORD, () => TcpContracts.CreateUser("user", new string('a', 101), UserStatus.Active) }, + { nameof(TcpContracts.UpdateUser) + "/empty userName", VsrError.INVALID_USERNAME, () => TcpContracts.UpdateUser(Id, "", null) }, + { nameof(TcpContracts.UpdateUser) + "/long userName", VsrError.INVALID_USERNAME, () => TcpContracts.UpdateUser(Id, new string('a', 51), null) }, + { nameof(TcpContracts.ChangePassword) + "/empty current", VsrError.INVALID_PASSWORD, () => TcpContracts.ChangePassword(Id, "", "new") }, + { nameof(TcpContracts.ChangePassword) + "/long current", VsrError.INVALID_PASSWORD, () => TcpContracts.ChangePassword(Id, new string('a', 101), "new") }, + { nameof(TcpContracts.ChangePassword) + "/empty new", VsrError.INVALID_PASSWORD, () => TcpContracts.ChangePassword(Id, "old", "") }, + { nameof(TcpContracts.ChangePassword) + "/long new", VsrError.INVALID_PASSWORD, () => TcpContracts.ChangePassword(Id, "old", new string('a', 101)) } + }; + + [Theory] + [MemberData(nameof(OverLimitCases))] + public void Contract_WithAStringOverTheWireLimitInBytes_Throws(string contract, string parameterName, + Func serialize) + { + Assert.NotEmpty(contract); + var exception = Assert.Throws(serialize); + Assert.Equal(parameterName, exception.ParamName); + } + + [Theory] + [MemberData(nameof(EmptyCases))] + public void Contract_WithAnEmptyString_Throws(string contract, string parameterName, Func serialize) + { + Assert.NotEmpty(contract); + var exception = Assert.Throws(serialize); + Assert.Equal(parameterName, exception.ParamName); + } + + [Theory] + [MemberData(nameof(AtLimitCases))] + public void Contract_WithAStringOfExactly255Bytes_PrefixesTheFullLength(string contract, int prefixOffset, + Func serialize) + { + Assert.NotEmpty(contract); + var bytes = serialize(); + + Assert.Equal(255, bytes[prefixOffset]); + Assert.Equal((byte)'a', bytes[prefixOffset + 255]); + } + + [Theory] + [MemberData(nameof(CredentialCases))] + public void UserContract_WithACredentialOutsideTheServerBounds_ThrowsTheTypedStatus(string contract, + int statusCode, Func serialize) + { + Assert.NotEmpty(contract); + var exception = Assert.Throws(serialize); + Assert.Equal(statusCode, exception.StatusCode); + Assert.False(exception.FromServer); + } + + [Fact] + public void CreateUser_WithCredentialsAtTheServerBounds_Serializes() + { + var bytes = TcpContracts.CreateUser(new string('u', 50), new string('p', 100), UserStatus.Active); + + Assert.Equal(50, bytes[0]); + Assert.Equal(100, bytes[1 + 50]); + } + + [Fact] + public void LoginUser_WithEmptyVersionAndContext_Serializes() + { + var bytes = TcpContracts.LoginUser("user", "pass", "", ""); + + Assert.Equal(1 + 4 + 1 + 4 + 4 + 4, bytes.Length); + } + + [Fact] + public void UpdateUser_WithStatusOnly_SerializesExactlyOneNameFlagAndStatusPair() + { + var bytes = TcpContracts.UpdateUser(Id, null, UserStatus.Inactive); + + Assert.Equal(new byte[] { 1, 4, 1, 0, 0, 0, 0, 1, (byte)UserStatus.Inactive }, bytes); + } + + [Fact] + public void UpdateUser_WithNameOnly_SerializesExactlyOneNameAndStatusFlag() + { + var bytes = TcpContracts.UpdateUser(Id, "abc", null); + + Assert.Equal(new byte[] { 1, 4, 1, 0, 0, 0, 1, 3, (byte)'a', (byte)'b', (byte)'c', 0 }, bytes); + } + + [Fact] + public void CreateStream_WithANonAsciiName_PrefixesTheUtf8ByteCount() + { + var bytes = TcpContracts.CreateStream("café"); + + Assert.Equal(5, bytes[0]); + Assert.Equal(6, bytes.Length); + } + + [Fact] + public void GetUser_WithANonAsciiStringIdentifier_SerializesTheUtf8Bytes() + { + var bytes = TcpContracts.GetUser(Identifier.String("café")); + + Assert.Equal(new byte[] { 2, 5, (byte)'c', (byte)'a', (byte)'f', 0xC3, 0xA9 }, bytes); + } + + [Fact] + public void UpdateStream_WithANonAsciiStringIdentifier_PlacesTheNameAfterTheUtf8Bytes() + { + var bytes = TcpContracts.UpdateStream(Identifier.String("café"), "topic"); + + Assert.Equal( + new byte[] + { + 2, 5, (byte)'c', (byte)'a', (byte)'f', 0xC3, 0xA9, + 5, (byte)'t', (byte)'o', (byte)'p', (byte)'i', (byte)'c' + }, bytes); + } + + [Fact] + public void Identifier_BuiltWithAnObjectInitializerOver255Bytes_Throws() + { + var exception = Assert.Throws(() => + new Identifier { Kind = IdKind.String, Value = new byte[300] }); + Assert.Equal("Value", exception.ParamName); + } + + [Fact] + public void Partitioning_BuiltWithAnObjectInitializerOver255Bytes_Throws() + { + var exception = Assert.Throws(() => + new Partitioning { Kind = Enums.Partitioning.MessageKey, Value = new byte[300] }); + Assert.Equal("Value", exception.ParamName); + } + + [Fact] + public void Identifier_BuiltWithAnObjectInitializerOf255Bytes_KeepsTheFullLength() + { + var identifier = new Identifier { Kind = IdKind.String, Value = new byte[255] }; + + Assert.Equal(255, identifier.Length); + } +} diff --git a/foreign/csharp/Iggy_SDK_Tests/UtilityTests/HeaderValueTests.cs b/foreign/csharp/Iggy_SDK_Tests/UtilityTests/HeaderValueTests.cs index 961c6fda9c..a89f7fe587 100644 --- a/foreign/csharp/Iggy_SDK_Tests/UtilityTests/HeaderValueTests.cs +++ b/foreign/csharp/Iggy_SDK_Tests/UtilityTests/HeaderValueTests.cs @@ -34,6 +34,13 @@ public void Raw_ReturnsCorrectValue() Assert.Equal(data, header.Value); } + [Fact] + public void Raw_ThrowsArgumentExceptionForInvalidValue() + { + Assert.Throws(() => HeaderValue.FromBytes([])); + Assert.Throws(() => HeaderValue.FromBytes(new byte[256])); + } + [Fact] public void String_ThrowsArgumentExceptionForInvalidValue() { diff --git a/foreign/csharp/Iggy_SDK_Tests/UtilityTests/IdentifiersByteSerializationTests.cs b/foreign/csharp/Iggy_SDK_Tests/UtilityTests/IdentifiersByteSerializationTests.cs index 60aa552eb8..6e1235b387 100644 --- a/foreign/csharp/Iggy_SDK_Tests/UtilityTests/IdentifiersByteSerializationTests.cs +++ b/foreign/csharp/Iggy_SDK_Tests/UtilityTests/IdentifiersByteSerializationTests.cs @@ -15,7 +15,10 @@ // specific language governing permissions and limitations // under the License. +using Apache.Iggy.Enums; using Apache.Iggy.Kinds; +using Partitioning = Apache.Iggy.Kinds.Partitioning; +using Apache.Iggy.Headers; namespace Apache.Iggy.Tests.UtilityTests; @@ -30,6 +33,27 @@ public void StringIdentifier_WithInvalidLength_ShouldThrowArgumentException() Assert.Throws(() => Identifier.String(val)); } + [Theory] + [InlineData("café", 5)] + [InlineData("naïve-café", 12)] + [InlineData("日本語", 9)] + public void StringIdentifier_WithNonAscii_ShouldUseUtf8ByteLength(string value, int expectedLength) + { + var identifier = Identifier.String(value); + + Assert.Equal(expectedLength, identifier.Length); + Assert.Equal(expectedLength, identifier.Value.Length); + Assert.Equal(value, identifier.GetString()); + } + + [Fact] + public void StringIdentifier_WithNonAsciiExceeding255Bytes_ShouldThrowArgumentException() + { + var val = new string('あ', 200); + + Assert.Throws(() => Identifier.String(val)); + } + [Fact] public void KeyEntityId_WithInvalidLength_ShouldThrowArgumentException() { @@ -39,6 +63,35 @@ public void KeyEntityId_WithInvalidLength_ShouldThrowArgumentException() Assert.Throws(() => Partitioning.EntityIdString(val)); } + [Theory] + [InlineData("café", 5)] + [InlineData("日本語", 9)] + public void KeyEntityId_WithNonAscii_ShouldUseUtf8ByteLength(string value, int expectedLength) + { + var partitioning = Partitioning.EntityIdString(value); + + Assert.Equal(expectedLength, partitioning.Length); + Assert.Equal(expectedLength, partitioning.Value.Length); + } + + [Fact] + public void KeyEntityId_WithNonAsciiExceeding255Bytes_ShouldThrowArgumentException() + { + Assert.Throws(() => Partitioning.EntityIdString(new string('あ', 200))); + } + + [Fact] + public void HeaderKey_WithNonAsciiExceeding255Bytes_ShouldThrowArgumentException() + { + Assert.Throws(() => HeaderKey.FromString(new string('あ', 200))); + } + + [Fact] + public void HeaderValue_WithNonAsciiExceeding255Bytes_ShouldThrowArgumentException() + { + Assert.Throws(() => HeaderValue.FromString(new string('あ', 200))); + } + [Fact] public void KeyBytes_WithInvalidLength_ShouldThrowArgumentException() { @@ -64,10 +117,88 @@ public void PartitionId_WithNegativeValue_ShouldThrow() Assert.Throws(() => Partitioning.PartitionId(-1)); } + [Fact] + public void Identifier_WithSameKindAndValue_ShouldBeEqual() + { + Assert.Equal(Identifier.Numeric(1), Identifier.Numeric(1)); + Assert.Equal(Identifier.String("name"), Identifier.String("name")); + Assert.Equal(Identifier.Numeric(1).GetHashCode(), Identifier.Numeric(1).GetHashCode()); + Assert.NotEqual(Identifier.Numeric(1), Identifier.Numeric(2)); + Assert.NotEqual(Identifier.Numeric(1), Identifier.String("1")); + } + [Fact] public void Consumer_WithNegativeId_ShouldThrow() { Assert.Throws(() => Consumer.New(-1)); Assert.Throws(() => Consumer.Group(-1)); } + + [Fact] + public void Identifier_BuiltWithALegacyLengthInitializer_DerivesLengthFromValue() + { + var identifier = new Identifier { Kind = IdKind.String, Length = 1, Value = "café"u8.ToArray() }; + + Assert.Equal(5, identifier.Length); + Assert.Equal(Identifier.String("café"), identifier); + } + + [Fact] + public void Partitioning_BuiltWithALegacyLengthInitializer_DerivesLengthFromValue() + { + var partitioning = new Partitioning + { + Kind = Enums.Partitioning.MessageKey, + Length = 1, + Value = "café"u8.ToArray() + }; + + Assert.Equal(5, partitioning.Length); + } + + [Fact] + public void Identifier_WhenTheInitializerArrayIsMutated_KeepsTheOriginalValue() + { + var bytes = "abc"u8.ToArray(); + var identifier = new Identifier { Kind = IdKind.String, Value = bytes }; + var lookup = new HashSet { identifier }; + + bytes[0] = (byte)'z'; + + Assert.Equal("abc", identifier.GetString()); + Assert.Contains(Identifier.String("abc"), lookup); + } + + [Fact] + public void Identifier_WhenTheValueCopyIsMutated_KeepsTheOriginalValue() + { + var identifier = Identifier.String("abc"); + var lookup = new HashSet { identifier }; + + identifier.Value[0] = (byte)'z'; + + Assert.Equal("abc", identifier.GetString()); + Assert.Contains(Identifier.String("abc"), lookup); + } + + [Fact] + public void Partitioning_WhenTheInitializerArrayIsMutated_KeepsTheOriginalValue() + { + var bytes = "abc"u8.ToArray(); + var partitioning = Partitioning.EntityIdBytes(bytes); + + bytes[0] = (byte)'z'; + + Assert.Equal("abc"u8.ToArray(), partitioning.Value); + } + + [Fact] + public void Partitioning_WhenTheValueCopyIsMutated_KeepsTheOriginalValue() + { + var partitioning = Partitioning.EntityIdString("abc"); + + partitioning.Value[0] = (byte)'z'; + + Assert.Equal("abc"u8.ToArray(), partitioning.Value); + } } From 25e9d44dde95372d0f3e48011eb05aa3251410ed Mon Sep 17 00:00:00 2001 From: sipher <110331483+sipher-01@users.noreply.github.com> Date: Mon, 7 Sep 2026 03:16:07 +0530 Subject: [PATCH 069/182] fix(go): update MaxPayloadSize from 10 MB to 64 MB to match size update in rust message construction (#4080) Closes #4078 --- foreign/go/contracts/messages.go | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/foreign/go/contracts/messages.go b/foreign/go/contracts/messages.go index 4dc3d14a5a..faa64508ec 100644 --- a/foreign/go/contracts/messages.go +++ b/foreign/go/contracts/messages.go @@ -30,8 +30,8 @@ const ( // // Constraints // - Minimum payload size: 1 byte (empty payloads are not allowed) - // - Maximum payload size: 10 MB - MaxPayloadSize = 10 * 1000 * 1000 + // - Maximum payload size: 64 MB + MaxPayloadSize = 64 * 1000 * 1000 // MaxUserHeadersSize is maximum allowed size in bytes for user-defined headers. // From cb57ce2bd569d415bdd02b4638ba76c134472138 Mon Sep 17 00:00:00 2001 From: Piotr Gankiewicz Date: Mon, 7 Sep 2026 07:56:10 +0200 Subject: [PATCH 070/182] fix(connectors): validate protobuf field lengths consistently (#4069) Schema-based protobuf decoding did not reliably reject invalid field lengths. Use shared checked conversion, addition, and bounds validation for known, preserved unknown, and skipped fields. Add regression coverage for invalid lengths and valid decoding with an in-memory schema. --- core/connectors/sdk/src/decoders/proto.rs | 135 +++++++++++++++++++--- 1 file changed, 120 insertions(+), 15 deletions(-) diff --git a/core/connectors/sdk/src/decoders/proto.rs b/core/connectors/sdk/src/decoders/proto.rs index 1b44bbd840..9e97ba52bc 100644 --- a/core/connectors/sdk/src/decoders/proto.rs +++ b/core/connectors/sdk/src/decoders/proto.rs @@ -398,11 +398,7 @@ impl ProtoStreamDecoder { } 2 => { let (length, mut new_cursor) = self.parse_simple_varint(data, cursor)?; - let end_cursor = new_cursor + length as usize; - - if end_cursor > data.len() { - return Err(Error::InvalidProtobufPayload); - } + let end_cursor = Self::length_delimited_end(new_cursor, length, data.len())?; let field_data = &data[new_cursor..end_cursor]; new_cursor = end_cursor; @@ -451,11 +447,7 @@ impl ProtoStreamDecoder { } 2 => { let (length, mut new_cursor) = self.parse_simple_varint(data, cursor)?; - let end_cursor = new_cursor + length as usize; - - if end_cursor > data.len() { - return Err(Error::InvalidProtobufPayload); - } + let end_cursor = Self::length_delimited_end(new_cursor, length, data.len())?; let field_data = &data[new_cursor..end_cursor]; new_cursor = end_cursor; @@ -475,11 +467,7 @@ impl ProtoStreamDecoder { } 2 => { let (length, new_cursor) = self.parse_simple_varint(data, cursor)?; - let end_cursor = new_cursor + length as usize; - - if end_cursor > data.len() { - return Err(Error::InvalidProtobufPayload); - } + let end_cursor = Self::length_delimited_end(new_cursor, length, data.len())?; Ok(end_cursor) } @@ -487,6 +475,14 @@ impl ProtoStreamDecoder { } } + fn length_delimited_end(cursor: usize, length: u64, data_len: usize) -> Result { + usize::try_from(length) + .ok() + .and_then(|length| cursor.checked_add(length)) + .filter(|end_cursor| *end_cursor <= data_len) + .ok_or(Error::InvalidProtobufPayload) + } + fn apply_field_transformations(&self, payload: Payload) -> Result { if let Some(mappings) = &self.config.field_mappings { match payload { @@ -828,4 +824,113 @@ mod tests { "Should fallback gracefully when schema loading fails" ); } + + fn one_field_schema_decoder(preserve_unknown_fields: bool) -> ProtoStreamDecoder { + let file_descriptor_set = prost_types::FileDescriptorSet { + file: vec![prost_types::FileDescriptorProto { + name: Some("one_field.proto".to_string()), + package: Some("test".to_string()), + message_type: vec![prost_types::DescriptorProto { + name: Some("OneField".to_string()), + field: vec![prost_types::FieldDescriptorProto { + name: Some("name".to_string()), + number: Some(1), + r#type: Some(Type::String as i32), + ..Default::default() + }], + ..Default::default() + }], + ..Default::default() + }], + }; + let decoder = ProtoStreamDecoder::new(ProtoConfig { + descriptor_set: Some(file_descriptor_set.encode_to_vec()), + message_type: Some("test.OneField".to_string()), + use_any_wrapper: false, + preserve_unknown_fields, + ..ProtoConfig::default() + }); + assert!( + decoder.message_descriptor.is_some(), + "schema must be loaded for these tests to exercise the field parser" + ); + decoder + } + + fn decode_with_loaded_schema( + decoder: &ProtoStreamDecoder, + payload: &[u8], + ) -> Result { + decoder.decode_with_message_descriptor( + payload, + decoder.message_descriptor.as_ref().unwrap(), + decoder.file_descriptor_set.as_ref().unwrap(), + ) + } + + #[test] + fn decode_should_succeed_given_valid_message_with_loaded_schema() { + let decoder = one_field_schema_decoder(false); + let payload = vec![0x0a, 0x03, b'a', b'b', b'c']; + + let result = decoder.decode(payload); + + let Ok(Payload::Json(simd_json::OwnedValue::Object(map))) = result else { + panic!("Expected JSON object"); + }; + assert_eq!( + map.get("name"), + Some(&simd_json::OwnedValue::String("abc".to_string())) + ); + } + + #[test] + fn decode_should_fail_given_known_field_length_that_overflows_usize() { + let decoder = one_field_schema_decoder(false); + let payload = vec![ + 0x0a, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x01, + ]; + + let result = decode_with_loaded_schema(&decoder, &payload); + + assert!(matches!(result, Err(Error::InvalidProtobufPayload))); + assert!(decoder.decode(payload).is_err()); + } + + #[test] + fn decode_should_fail_given_unknown_preserved_field_length_that_overflows_usize() { + let decoder = one_field_schema_decoder(true); + let payload = vec![ + 0x9a, 0x06, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x01, + ]; + + let result = decode_with_loaded_schema(&decoder, &payload); + + assert!(matches!(result, Err(Error::InvalidProtobufPayload))); + assert!(decoder.decode(payload).is_err()); + } + + #[test] + fn decode_should_fail_given_skipped_field_length_that_wraps_cursor() { + let decoder = one_field_schema_decoder(false); + let payload = vec![ + 0x9a, 0x06, 0xf4, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0xff, 0x01, + ]; + + let result = decode_with_loaded_schema(&decoder, &payload); + + assert!(matches!(result, Err(Error::InvalidProtobufPayload))); + assert!(decoder.decode(payload).is_err()); + } + + #[test] + fn decode_should_fail_given_field_length_past_end_of_payload() { + let decoder = one_field_schema_decoder(false); + let payload = vec![0x0a, 0x10, b'a']; + + let result = decode_with_loaded_schema(&decoder, &payload); + + assert!(matches!(result, Err(Error::InvalidProtobufPayload))); + assert!(decoder.decode(payload).is_err()); + } } From 3462fed5c4e1d4ffdca0549b7a90e6bcc089d88c Mon Sep 17 00:00:00 2001 From: Matthew Patton Date: Mon, 7 Sep 2026 02:24:24 -0400 Subject: [PATCH 071/182] fix(shard): widen partition repair when a view-adopted suffix is missing bodies (#4074) --- core/shard/src/lib.rs | 18 ++++++++++++------ 1 file changed, 12 insertions(+), 6 deletions(-) diff --git a/core/shard/src/lib.rs b/core/shard/src/lib.rs index 8181831fd8..73cf6a9f0d 100644 --- a/core/shard/src/lib.rs +++ b/core/shard/src/lib.rs @@ -8316,12 +8316,18 @@ where } let nonce = iggy_common::random_id::get_uuid(); let from_op = consensus.commit_min() + 1; - // The widening is for the replica that is LEVEL with the commit - // frontier and short of bodies above it. Widening while a commit lag - // stands would ask for `(commit_min, head]` -- the whole committed - // prefix this replica already holds, refetched -- and the suffix is - // reached anyway once the lag closes, on the arm after it. - let fetch_to_op = if commit_lag { commit_to_op } else { head }; + // Capping at `commit_to_op` while a commit lag stands avoids + // re-asking for `(commit_min, commit_to_op]`, the committed prefix + // this replica already holds. But a restarted node has a commit lag + // by construction, and if it also adopted a StartView suffix, the + // missing bodies sit above `commit_to_op`, not within it -- so the + // cap only holds when no suffix is missing; otherwise it must widen + // to `head` to ever reach those bodies. + let fetch_to_op = if commit_lag && !missing_suffix { + commit_to_op + } else { + head + }; let cluster = consensus.cluster(); let self_id = consensus.replica(); let namespace = consensus.group(); From 64b0bae7175d42df120bb9e93c66ceb8a2e2e07f Mon Sep 17 00:00:00 2001 From: BingEdward <1042653432@qq.com> Date: Mon, 7 Sep 2026 15:27:13 +0800 Subject: [PATCH 072/182] feat(python): expose get_stats on IggyClient (#4018) --- .../python-maturin/pre-merge/action.yml | 9 + core/bench/report/src/types/server_stats.rs | 2 +- core/common/src/types/stats/mod.rs | 2 +- .../Iggy_SDK/Contracts/StatsResponse.cs | 2 +- foreign/python/apache_iggy.pyi | 245 ++++++++++ foreign/python/src/client.rs | 25 + foreign/python/src/lib.rs | 5 + foreign/python/src/stats.rs | 449 ++++++++++++++++++ foreign/python/tests/test_stats.py | 250 ++++++++++ 9 files changed, 986 insertions(+), 3 deletions(-) create mode 100644 foreign/python/src/stats.rs create mode 100644 foreign/python/tests/test_stats.py diff --git a/.github/actions/python-maturin/pre-merge/action.yml b/.github/actions/python-maturin/pre-merge/action.yml index 6a28d0e5be..234c36efed 100644 --- a/.github/actions/python-maturin/pre-merge/action.yml +++ b/.github/actions/python-maturin/pre-merge/action.yml @@ -90,6 +90,15 @@ runs: echo "pyrefly version: $(uv run pyrefly --version)" shell: bash + # The crate is outside the root workspace, so the Rust test jobs never + # reach it; without this its unit tests would only ever be compiled. + - name: Rust unit tests for Python extension + if: inputs.task == 'lint' + run: | + cd foreign/python + cargo test --manifest-path Cargo.toml + shell: bash + - name: Build Python wheel if: inputs.task == 'test' run: | diff --git a/core/bench/report/src/types/server_stats.rs b/core/bench/report/src/types/server_stats.rs index 0eca6a9391..e2b24f37b7 100644 --- a/core/bench/report/src/types/server_stats.rs +++ b/core/bench/report/src/types/server_stats.rs @@ -72,7 +72,7 @@ pub struct BenchmarkServerStats { pub kernel_version: String, /// The version of the Iggy server. pub iggy_server_version: String, - /// The semantic version of the Iggy server in the numeric format e.g. 1.2.3 -> 100200300 (major * 1000000 + minor * 1000 + patch). + /// The semantic version of the Iggy server in the numeric format e.g. 1.2.3 -> 1002003 (major * 1000000 + minor * 1000 + patch). pub iggy_server_semver: Option, /// Cache metrics per partition #[serde(with = "cache_metrics_serializer")] diff --git a/core/common/src/types/stats/mod.rs b/core/common/src/types/stats/mod.rs index 5de7f03b43..fcec5cf4c6 100644 --- a/core/common/src/types/stats/mod.rs +++ b/core/common/src/types/stats/mod.rs @@ -70,7 +70,7 @@ pub struct Stats { pub kernel_version: String, /// The version of the Iggy server. pub iggy_server_version: String, - /// The semantic version of the Iggy server in the numeric format e.g. 1.2.3 -> 100200300 (major * 1000000 + minor * 1000 + patch). + /// The semantic version of the Iggy server in the numeric format e.g. 1.2.3 -> 1002003 (major * 1000000 + minor * 1000 + patch). pub iggy_server_semver: Option, /// Cache metrics per partition #[serde(with = "cache_metrics_serializer")] diff --git a/foreign/csharp/Iggy_SDK/Contracts/StatsResponse.cs b/foreign/csharp/Iggy_SDK/Contracts/StatsResponse.cs index 772fbaa574..4b6efa627d 100644 --- a/foreign/csharp/Iggy_SDK/Contracts/StatsResponse.cs +++ b/foreign/csharp/Iggy_SDK/Contracts/StatsResponse.cs @@ -148,7 +148,7 @@ public sealed class StatsResponse public required string IggyServerVersion { get; init; } /// - /// Semantic version of the Iggy server in the numeric format e.g. 1.2.3 -> 100200300 (major * 1000000 + minor * 1000 + + /// Semantic version of the Iggy server in the numeric format e.g. 1.2.3 -> 1002003 (major * 1000000 + minor * 1000 + /// patch). /// public uint IggyServerSemver { get; init; } diff --git a/foreign/python/apache_iggy.pyi b/foreign/python/apache_iggy.pyi index c0f603e58d..ed15ed2f54 100644 --- a/foreign/python/apache_iggy.pyi +++ b/foreign/python/apache_iggy.pyi @@ -30,6 +30,8 @@ __all__ = [ "AutoCommitAfter", "AutoCommitWhen", "AutoLogin", + "CacheMetrics", + "CacheMetricsKey", "Consumer", "ConsumerGroup", "ConsumerGroupDetails", @@ -49,6 +51,7 @@ __all__ = [ "SendMessage", "SendMessagesConfirmation", "SendMessagesResponse", + "Stats", "StreamDetails", "StreamPermissions", "TcpConfig", @@ -290,6 +293,57 @@ class AutoLogin: """ def __repr__(self) -> builtins.str: ... +@typing.final +class CacheMetrics: + r""" + Cache metrics for a specific partition. + """ + @property + def hits(self) -> builtins.int: + r""" + Number of cache hits. + """ + @property + def misses(self) -> builtins.int: + r""" + Number of cache misses. + """ + @property + def hit_ratio(self) -> builtins.float: + r""" + Hit ratio (hits / (hits + misses)). + """ + def __repr__(self) -> builtins.str: ... + +@typing.final +class CacheMetricsKey: + r""" + Key identifying the partition a `CacheMetrics` entry belongs to. + + Hashable and comparable, so it can key the `Stats.cache_metrics` dict. + """ + @property + def stream_id(self) -> builtins.int: + r""" + The unique identifier (numeric) of the stream. + """ + @property + def topic_id(self) -> builtins.int: + r""" + The unique identifier (numeric) of the topic within the stream. + """ + @property + def partition_id(self) -> builtins.int: + r""" + The unique identifier (numeric) of the partition within the topic. + """ + def __eq__(self, other: builtins.object, /) -> builtins.bool: ... + def __hash__(self) -> builtins.int: ... + def __new__( + cls, stream_id: builtins.int, topic_id: builtins.int, partition_id: builtins.int + ) -> CacheMetricsKey: ... + def __repr__(self) -> builtins.str: ... + class Consumer: r""" The consumer polling the messages. It selects both the consumer kind and the @@ -872,6 +926,21 @@ class IggyClient: Sends a ping request to the server to check connectivity. Raises `RuntimeError` if the connection fails. """ + def get_stats(self) -> collections.abc.Awaitable[Stats]: + r""" + Get the statistics and details of the server and its running process. + + Requires an authenticated session whose user holds the `read_servers` + or `manage_servers` global permission. + + Returns: + An awaitable that resolves to `Stats`. + + Raises: + RuntimeError: If the client is not connected, the session is not + authenticated, the user lacks the permission, or the request + fails. + """ def describe_options( self, scope: builtins.str ) -> collections.abc.Awaitable[list[OptionSpec]]: @@ -1836,6 +1905,182 @@ class SendMessagesResponse: with an offset a client has already recorded. """ +@typing.final +class Stats: + r""" + The statistics and details of the server and its running process. + + The fields are gathered from several sources while the request is served + (metadata counters, a process probe, a disk probe), so they are not an + atomic snapshot of one instant. + """ + @property + def process_id(self) -> builtins.int: + r""" + The unique identifier of the server process. + """ + @property + def cpu_usage(self) -> builtins.float: + r""" + The CPU usage of the server process, in percent summed over the cores + it ran on, so it exceeds 100 whenever the process uses more than one + core. + + Measured as a delta since the previous `get_stats` served by the same + server shard, so the first sample a shard serves is 0. + """ + @property + def total_cpu_usage(self) -> builtins.float: + r""" + The total CPU usage of the system, in percent averaged over the cores + the server may run on when confined by an affinity/cpuset mask (over + every host core otherwise), so it stays within 0-100. + + Same per-shard delta sampling as `cpu_usage`: the first sample a shard + serves is 0. + """ + @property + def memory_usage(self) -> builtins.int: + r""" + The memory usage of the server process, in bytes. + """ + @property + def total_memory(self) -> builtins.int: + r""" + The total memory of the system, in bytes, or the effective cgroup memory + limit when the server runs inside a memory-capped cgroup (container, + systemd slice). + """ + @property + def available_memory(self) -> builtins.int: + r""" + The available memory of the system, in bytes, scoped to the cgroup + limit when one applies. + """ + @property + def run_time(self) -> datetime.timedelta: + r""" + The run time of the server process, with whole-second precision. + """ + @property + def start_time(self) -> builtins.int: + r""" + The start time of the server process, in microseconds since the Unix + epoch, with whole-second precision. + """ + @property + def read_bytes(self) -> builtins.int: + r""" + The total number of bytes read. + """ + @property + def written_bytes(self) -> builtins.int: + r""" + The total number of bytes written. + """ + @property + def messages_size_bytes(self) -> builtins.int: + r""" + The total size of the messages, in bytes. + """ + @property + def streams_count(self) -> builtins.int: + r""" + The total number of streams. + """ + @property + def topics_count(self) -> builtins.int: + r""" + The total number of topics. + """ + @property + def partitions_count(self) -> builtins.int: + r""" + The total number of partitions. + """ + @property + def segments_count(self) -> builtins.int: + r""" + The total number of segments. + """ + @property + def messages_count(self) -> builtins.int: + r""" + The total number of messages. + """ + @property + def clients_count(self) -> builtins.int: + r""" + The total number of connected clients. + """ + @property + def consumer_groups_count(self) -> builtins.int: + r""" + The total number of consumer groups. + """ + @property + def hostname(self) -> builtins.str: + r""" + The name of the host the server runs on. + """ + @property + def os_name(self) -> builtins.str: + r""" + The name of the operating system. + """ + @property + def os_version(self) -> builtins.str: + r""" + The version of the operating system. + """ + @property + def kernel_version(self) -> builtins.str: + r""" + The version of the kernel. + """ + @property + def iggy_server_version(self) -> builtins.str: + r""" + The version of the Iggy server. + """ + @property + def iggy_server_semver(self) -> builtins.int | None: + r""" + The numeric semantic version of the Iggy server, or `None` when unknown. + E.g. 1.2.3 -> 1002003 (major * 1000000 + minor * 1000 + patch). + """ + @property + def cache_metrics(self) -> builtins.dict[CacheMetricsKey, CacheMetrics]: + r""" + Cache metrics per partition. + + Current servers do not populate this and reply with an empty map. Each + access builds a fresh dict, so mutating the returned dict does not + change the stats. + """ + @property + def threads_count(self) -> builtins.int: + r""" + The number of threads in the server process. + """ + @property + def free_disk_space(self) -> builtins.int: + r""" + The available (free) disk space for the data directory, in bytes. + + 0 when the server does not know its data directory or the disk probe + fails. + """ + @property + def total_disk_space(self) -> builtins.int: + r""" + The total disk space for the data directory, in bytes. + + 0 when the server does not know its data directory or the disk probe + fails. + """ + def __repr__(self) -> builtins.str: ... + @typing.final class StreamDetails: @property diff --git a/foreign/python/src/client.rs b/foreign/python/src/client.rs index 10c5c776b0..8252156a36 100644 --- a/foreign/python/src/client.rs +++ b/foreign/python/src/client.rs @@ -42,6 +42,7 @@ use crate::options::OptionSpec as PyOptionSpec; use crate::permissions::Permissions as PyPermissions; use crate::receive_message::{PollingStrategy, ReceiveMessage}; use crate::send_message::{SendMessage, SendMessagesResponse as PySendMessagesResponse}; +use crate::stats::Stats as PyStats; use crate::stream::StreamDetails; use crate::topic::{IggyExpiry, MaxTopicSize, Topic, TopicDetails}; use crate::user::{ @@ -157,6 +158,30 @@ impl IggyClient { }) } + /// Get the statistics and details of the server and its running process. + /// + /// Requires an authenticated session whose user holds the `read_servers` + /// or `manage_servers` global permission. + /// + /// Returns: + /// An awaitable that resolves to `Stats`. + /// + /// Raises: + /// RuntimeError: If the client is not connected, the session is not + /// authenticated, the user lacks the permission, or the request + /// fails. + #[gen_stub(override_return_type(type_repr="collections.abc.Awaitable[Stats]", imports=("collections.abc")))] + fn get_stats<'a>(&self, py: Python<'a>) -> PyResult> { + let inner = self.inner.clone(); + future_into_py(py, async move { + let stats = inner + .get_stats() + .await + .map_err(|e| PyErr::new::(e.to_string()))?; + Ok(PyStats::from(stats)) + }) + } + /// Describe the option catalog for a resource scope. /// /// This is the discovery surface for the `options` argument on diff --git a/foreign/python/src/lib.rs b/foreign/python/src/lib.rs index 5f2e128264..d4397d5ab0 100644 --- a/foreign/python/src/lib.rs +++ b/foreign/python/src/lib.rs @@ -24,6 +24,7 @@ mod options; mod permissions; mod receive_message; mod send_message; +mod stats; mod stream; mod topic; mod user; @@ -40,6 +41,7 @@ use permissions::{GlobalPermissions, Permissions, StreamPermissions, TopicPermis use pyo3::prelude::*; use receive_message::{PollingStrategy, ReceiveMessage}; use send_message::{SendMessage, SendMessagesConfirmation, SendMessagesResponse}; +use stats::{CacheMetrics, CacheMetricsKey, Stats}; use stream::StreamDetails; use topic::{IggyExpiry, MaxTopicSize, Partition, Topic, TopicDetails}; use user::{UserInfo, UserInfoDetails, UserStatus}; @@ -57,6 +59,9 @@ fn apache_iggy(_py: Python, m: &Bound<'_, PyModule>) -> PyResult<()> { m.add_class::()?; m.add_class::()?; m.add_class::()?; + m.add_class::()?; + m.add_class::()?; + m.add_class::()?; m.add_class::()?; m.add_class::()?; m.add_class::()?; diff --git a/foreign/python/src/stats.rs b/foreign/python/src/stats.rs new file mode 100644 index 0000000000..af48425a0d --- /dev/null +++ b/foreign/python/src/stats.rs @@ -0,0 +1,449 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use crate::duration::iggy_duration_to_py_delta; +use iggy::prelude::{ + CacheMetrics as RustCacheMetrics, CacheMetricsKey as RustCacheMetricsKey, Stats as RustStats, +}; +use pyo3::prelude::*; +use pyo3::types::{PyDelta, PyDict}; +use pyo3_stub_gen::derive::{gen_stub_pyclass, gen_stub_pymethods}; + +/// Key identifying the partition a `CacheMetrics` entry belongs to. +/// +/// Hashable and comparable, so it can key the `Stats.cache_metrics` dict. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +#[gen_stub_pyclass] +#[pyclass(eq, frozen, hash, skip_from_py_object)] +pub struct CacheMetricsKey { + /// The unique identifier (numeric) of the stream. + #[pyo3(get)] + pub stream_id: u32, + /// The unique identifier (numeric) of the topic within the stream. + #[pyo3(get)] + pub topic_id: u32, + /// The unique identifier (numeric) of the partition within the topic. + #[pyo3(get)] + pub partition_id: u32, +} + +impl From<&RustCacheMetricsKey> for CacheMetricsKey { + fn from(key: &RustCacheMetricsKey) -> Self { + Self { + stream_id: key.stream_id, + topic_id: key.topic_id, + partition_id: key.partition_id, + } + } +} + +#[gen_stub_pymethods] +#[pymethods] +impl CacheMetricsKey { + #[new] + fn new(stream_id: u32, topic_id: u32, partition_id: u32) -> Self { + Self { + stream_id, + topic_id, + partition_id, + } + } + + fn __repr__(&self) -> String { + format!( + "CacheMetricsKey(stream_id={}, topic_id={}, partition_id={})", + self.stream_id, self.topic_id, self.partition_id + ) + } +} + +/// Cache metrics for a specific partition. +#[gen_stub_pyclass] +#[pyclass] +pub struct CacheMetrics { + /// Number of cache hits. + #[pyo3(get)] + pub hits: u64, + /// Number of cache misses. + #[pyo3(get)] + pub misses: u64, + /// Hit ratio (hits / (hits + misses)). + #[pyo3(get)] + pub hit_ratio: f32, +} + +impl From<&RustCacheMetrics> for CacheMetrics { + fn from(metrics: &RustCacheMetrics) -> Self { + Self { + hits: metrics.hits, + misses: metrics.misses, + hit_ratio: metrics.hit_ratio, + } + } +} + +#[gen_stub_pymethods] +#[pymethods] +impl CacheMetrics { + fn __repr__(&self) -> String { + format!( + "CacheMetrics(hits={}, misses={}, hit_ratio={})", + self.hits, self.misses, self.hit_ratio + ) + } +} + +/// The statistics and details of the server and its running process. +/// +/// The fields are gathered from several sources while the request is served +/// (metadata counters, a process probe, a disk probe), so they are not an +/// atomic snapshot of one instant. +#[gen_stub_pyclass] +#[pyclass] +pub struct Stats { + pub(crate) inner: RustStats, +} + +impl From for Stats { + fn from(stats: RustStats) -> Self { + Self { inner: stats } + } +} + +#[gen_stub_pymethods] +#[pymethods] +impl Stats { + /// The unique identifier of the server process. + #[getter] + pub fn process_id(&self) -> u32 { + self.inner.process_id + } + + /// The CPU usage of the server process, in percent summed over the cores + /// it ran on, so it exceeds 100 whenever the process uses more than one + /// core. + /// + /// Measured as a delta since the previous `get_stats` served by the same + /// server shard, so the first sample a shard serves is 0. + #[getter] + pub fn cpu_usage(&self) -> f32 { + self.inner.cpu_usage + } + + /// The total CPU usage of the system, in percent averaged over the cores + /// the server may run on when confined by an affinity/cpuset mask (over + /// every host core otherwise), so it stays within 0-100. + /// + /// Same per-shard delta sampling as `cpu_usage`: the first sample a shard + /// serves is 0. + #[getter] + pub fn total_cpu_usage(&self) -> f32 { + self.inner.total_cpu_usage + } + + /// The memory usage of the server process, in bytes. + #[getter] + pub fn memory_usage(&self) -> u64 { + self.inner.memory_usage.as_bytes_u64() + } + + /// The total memory of the system, in bytes, or the effective cgroup memory + /// limit when the server runs inside a memory-capped cgroup (container, + /// systemd slice). + #[getter] + pub fn total_memory(&self) -> u64 { + self.inner.total_memory.as_bytes_u64() + } + + /// The available memory of the system, in bytes, scoped to the cgroup + /// limit when one applies. + #[getter] + pub fn available_memory(&self) -> u64 { + self.inner.available_memory.as_bytes_u64() + } + + /// The run time of the server process, with whole-second precision. + #[getter] + #[gen_stub(override_return_type(type_repr = "datetime.timedelta", imports=("datetime")))] + pub fn run_time<'a>(&self, py: Python<'a>) -> PyResult> { + iggy_duration_to_py_delta(py, self.inner.run_time) + } + + /// The start time of the server process, in microseconds since the Unix + /// epoch, with whole-second precision. + #[getter] + pub fn start_time(&self) -> u64 { + self.inner.start_time.as_micros() + } + + /// The total number of bytes read. + #[getter] + pub fn read_bytes(&self) -> u64 { + self.inner.read_bytes.as_bytes_u64() + } + + /// The total number of bytes written. + #[getter] + pub fn written_bytes(&self) -> u64 { + self.inner.written_bytes.as_bytes_u64() + } + + /// The total size of the messages, in bytes. + #[getter] + pub fn messages_size_bytes(&self) -> u64 { + self.inner.messages_size_bytes.as_bytes_u64() + } + + /// The total number of streams. + #[getter] + pub fn streams_count(&self) -> u32 { + self.inner.streams_count + } + + /// The total number of topics. + #[getter] + pub fn topics_count(&self) -> u32 { + self.inner.topics_count + } + + /// The total number of partitions. + #[getter] + pub fn partitions_count(&self) -> u32 { + self.inner.partitions_count + } + + /// The total number of segments. + #[getter] + pub fn segments_count(&self) -> u32 { + self.inner.segments_count + } + + /// The total number of messages. + #[getter] + pub fn messages_count(&self) -> u64 { + self.inner.messages_count + } + + /// The total number of connected clients. + #[getter] + pub fn clients_count(&self) -> u32 { + self.inner.clients_count + } + + /// The total number of consumer groups. + #[getter] + pub fn consumer_groups_count(&self) -> u32 { + self.inner.consumer_groups_count + } + + /// The name of the host the server runs on. + #[getter] + pub fn hostname(&self) -> String { + self.inner.hostname.clone() + } + + /// The name of the operating system. + #[getter] + pub fn os_name(&self) -> String { + self.inner.os_name.clone() + } + + /// The version of the operating system. + #[getter] + pub fn os_version(&self) -> String { + self.inner.os_version.clone() + } + + /// The version of the kernel. + #[getter] + pub fn kernel_version(&self) -> String { + self.inner.kernel_version.clone() + } + + /// The version of the Iggy server. + #[getter] + pub fn iggy_server_version(&self) -> String { + self.inner.iggy_server_version.clone() + } + + /// The numeric semantic version of the Iggy server, or `None` when unknown. + /// E.g. 1.2.3 -> 1002003 (major * 1000000 + minor * 1000 + patch). + #[getter] + #[gen_stub(override_return_type(type_repr = "builtins.int | None"))] + pub fn iggy_server_semver(&self) -> Option { + self.inner.iggy_server_semver + } + + /// Cache metrics per partition. + /// + /// Current servers do not populate this and reply with an empty map. Each + /// access builds a fresh dict, so mutating the returned dict does not + /// change the stats. + #[getter] + #[gen_stub(override_return_type(type_repr = "builtins.dict[CacheMetricsKey, CacheMetrics]"))] + pub fn cache_metrics<'a>(&self, py: Python<'a>) -> PyResult> { + let dict = PyDict::new(py); + for (key, metrics) in &self.inner.cache_metrics { + dict.set_item(CacheMetricsKey::from(key), CacheMetrics::from(metrics))?; + } + Ok(dict) + } + + /// The number of threads in the server process. + #[getter] + pub fn threads_count(&self) -> u32 { + self.inner.threads_count + } + + /// The available (free) disk space for the data directory, in bytes. + /// + /// 0 when the server does not know its data directory or the disk probe + /// fails. + #[getter] + pub fn free_disk_space(&self) -> u64 { + self.inner.free_disk_space.as_bytes_u64() + } + + /// The total disk space for the data directory, in bytes. + /// + /// 0 when the server does not know its data directory or the disk probe + /// fails. + #[getter] + pub fn total_disk_space(&self) -> u64 { + self.inner.total_disk_space.as_bytes_u64() + } + + fn __repr__(&self) -> String { + format!( + "Stats(hostname='{}', iggy_server_version='{}', streams_count={}, \ + topics_count={}, partitions_count={}, messages_count={}, clients_count={})", + self.inner.hostname, + self.inner.iggy_server_version, + self.inner.streams_count, + self.inner.topics_count, + self.inner.partitions_count, + self.inner.messages_count, + self.inner.clients_count + ) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use iggy::prelude::{IggyByteSize, IggyDuration, IggyTimestamp}; + use std::collections::HashMap; + + /// Two entries sharing a stream and topic, so a key collision in the + /// `HashMap` -> `PyDict` conversion drops one of them. + fn cache_metrics_entries() -> HashMap { + HashMap::from([ + ( + RustCacheMetricsKey { + stream_id: 1, + topic_id: 2, + partition_id: 3, + }, + RustCacheMetrics { + hits: 7, + misses: 3, + hit_ratio: 0.7, + }, + ), + ( + RustCacheMetricsKey { + stream_id: 1, + topic_id: 2, + partition_id: 4, + }, + RustCacheMetrics { + hits: 0, + misses: 5, + hit_ratio: 0.0, + }, + ), + ]) + } + + fn rust_stats(iggy_server_semver: Option) -> RustStats { + RustStats { + process_id: 1, + cpu_usage: 0.0, + total_cpu_usage: 0.0, + memory_usage: IggyByteSize::default(), + total_memory: IggyByteSize::default(), + available_memory: IggyByteSize::default(), + run_time: IggyDuration::default(), + start_time: IggyTimestamp::default(), + read_bytes: IggyByteSize::default(), + written_bytes: IggyByteSize::default(), + messages_size_bytes: IggyByteSize::default(), + streams_count: 0, + topics_count: 0, + partitions_count: 0, + segments_count: 0, + messages_count: 0, + clients_count: 0, + consumer_groups_count: 0, + hostname: String::new(), + os_name: String::new(), + os_version: String::new(), + kernel_version: String::new(), + iggy_server_version: String::new(), + iggy_server_semver, + cache_metrics: cache_metrics_entries(), + threads_count: 0, + free_disk_space: IggyByteSize::default(), + total_disk_space: IggyByteSize::default(), + } + } + + #[test] + fn given_stats_when_converting_should_preserve_semver_option() { + Python::initialize(); + + assert_eq!(Stats::from(rust_stats(None)).iggy_server_semver(), None); + assert_eq!( + Stats::from(rust_stats(Some(1_002_003))).iggy_server_semver(), + Some(1_002_003) + ); + } + + #[test] + fn given_populated_cache_metrics_when_reading_should_key_every_entry() { + Python::initialize(); + + let stats = Stats::from(rust_stats(None)); + + Python::attach(|py| { + let dict = stats.cache_metrics(py).expect("build cache metrics dict"); + assert_eq!(dict.len(), cache_metrics_entries().len()); + + let entry = dict + .get_item(CacheMetricsKey::new(1, 2, 4)) + .expect("look up cache metrics entry") + .expect("entry for an independently built key"); + assert_eq!( + entry + .getattr("misses") + .and_then(|misses| misses.extract::()) + .expect("read misses"), + 5 + ); + }); + } +} diff --git a/foreign/python/tests/test_stats.py b/foreign/python/tests/test_stats.py new file mode 100644 index 0000000000..2b4423cb8a --- /dev/null +++ b/foreign/python/tests/test_stats.py @@ -0,0 +1,250 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +import datetime + +import pytest + +from apache_iggy import CacheMetricsKey, GlobalPermissions, IggyClient, Permissions +from apache_iggy import SendMessage as Message + +from .utils import ( + get_server_config, + login_fresh_client, + unique_credentials, + wait_for_server, +) + +# Fields that describe the server process itself and must not change between +# two calls within one server run. +PROCESS_IDENTITY_FIELDS = ( + "process_id", + "start_time", + "hostname", + "os_name", + "os_version", + "kernel_version", + "iggy_server_version", + "iggy_server_semver", +) + + +class TestStats: + """Test server statistics retrieval.""" + + @pytest.mark.asyncio + async def test_get_stats(self, iggy_client: IggyClient, unique_name): + """Sending messages moves the server counts reported by get_stats.""" + stats_before = await iggy_client.get_stats() + + stream_name = unique_name() + topic_name = unique_name() + await iggy_client.create_stream(stream_name) + await iggy_client.create_topic( + stream=stream_name, name=topic_name, partitions_count=1 + ) + await iggy_client.send_messages( + stream=stream_name, + topic=topic_name, + partitioning=0, + messages=[Message(f"stats message {i}") for i in range(3)], + ) + + stats = await iggy_client.get_stats() + + # `>=` rather than exact equality: the counters are server-global, so + # concurrently running tests (e.g. under pytest-xdist) may bump them too. + assert stats.streams_count >= stats_before.streams_count + 1 + assert stats.topics_count >= stats_before.topics_count + 1 + assert stats.partitions_count >= stats_before.partitions_count + 1 + assert stats.messages_count >= stats_before.messages_count + 3 + assert stats.messages_size_bytes > stats_before.messages_size_bytes + assert stats.clients_count >= 1 + + assert stats.iggy_server_version + assert stats.hostname + assert stats.os_name + assert stats.os_version + assert stats.kernel_version + assert stats.process_id > 0 + # sysinfo cannot enumerate a process's threads on macOS, so a server + # running there reports 0. + if stats.os_name != "Darwin": + assert stats.threads_count > 0 + assert stats.start_time > 0 + assert stats.total_memory > 0 + assert stats.available_memory <= stats.total_memory + assert stats.total_disk_space > 0 + assert stats.free_disk_space <= stats.total_disk_space + + assert isinstance(stats.run_time, datetime.timedelta) + assert stats.run_time >= stats_before.run_time + for field in PROCESS_IDENTITY_FIELDS: + assert getattr(stats, field) == getattr(stats_before, field) + + assert f"streams_count={stats.streams_count}" in repr(stats) + assert stats.hostname in repr(stats) + + @pytest.mark.asyncio + async def test_get_stats_reflects_topology( + self, iggy_client: IggyClient, unique_name + ): + """Streams, topics, partitions, consumer groups and a second client show + up in the counters, and deleting the streams brings them back down.""" + stats_before = await iggy_client.get_stats() + + streams = [unique_name() for _ in range(2)] + topics_per_stream = 2 + partitions_per_topic = 3 + topics = [] + for stream_name in streams: + await iggy_client.create_stream(stream_name) + for _ in range(topics_per_stream): + topic_name = unique_name() + await iggy_client.create_topic( + stream=stream_name, + name=topic_name, + partitions_count=partitions_per_topic, + ) + await iggy_client.create_consumer_group( + stream_name, topic_name, unique_name() + ) + topics.append((stream_name, topic_name)) + # Bound only to keep a second connection open until the test ends. + _second_client = await login_fresh_client("iggy", "iggy") + + topics_created = len(topics) + partitions_created = topics_created * partitions_per_topic + + stats = await iggy_client.get_stats() + + assert stats.streams_count >= stats_before.streams_count + len(streams) + assert stats.topics_count >= stats_before.topics_count + topics_created + assert ( + stats.partitions_count >= stats_before.partitions_count + partitions_created + ) + # Every new partition opens with one segment. + assert stats.segments_count >= stats_before.segments_count + partitions_created + assert ( + stats.consumer_groups_count + >= stats_before.consumer_groups_count + topics_created + ) + # Both `iggy_client` and `_second_client` are connected at this point, so + # the cross-shard total covers them regardless of what else the server + # reaped in between. + assert stats.clients_count >= 2 + + # The SDK has no delete_stream, so the streams stay behind and only the + # topic-level counters are expected to drop. The baseline is re-read + # right before the deletes to keep the window in which a concurrent + # test can bump the server-global counters as small as possible. + stats_before_delete = await iggy_client.get_stats() + for stream_name, topic_name in topics: + await iggy_client.delete_topic(stream_name, topic_name) + + stats_after = await iggy_client.get_stats() + + assert ( + stats_after.topics_count + <= stats_before_delete.topics_count - topics_created + ) + assert ( + stats_after.partitions_count + <= stats_before_delete.partitions_count - partitions_created + ) + assert ( + stats_after.segments_count + <= stats_before_delete.segments_count - partitions_created + ) + assert ( + stats_after.consumer_groups_count + <= stats_before_delete.consumer_groups_count - topics_created + ) + + @pytest.mark.asyncio + async def test_get_stats_cache_metrics_dict(self, iggy_client: IggyClient): + """cache_metrics is a dict the server currently leaves empty.""" + stats = await iggy_client.get_stats() + + assert stats.cache_metrics == {} + + @pytest.mark.unit + def test_cache_metrics_key_is_constructible_and_hashable(self): + """A key built in Python can address a cache_metrics dict entry.""" + key = CacheMetricsKey(stream_id=1, topic_id=2, partition_id=3) + + assert key.stream_id == 1 + assert key.topic_id == 2 + assert key.partition_id == 3 + assert repr(key) == "CacheMetricsKey(stream_id=1, topic_id=2, partition_id=3)" + + equal_key = CacheMetricsKey(stream_id=1, topic_id=2, partition_id=3) + other_key = CacheMetricsKey(stream_id=1, topic_id=2, partition_id=4) + assert key == equal_key + assert key != other_key + assert hash(key) == hash(equal_key) + + # An equal key constructed independently hits the same dict slot. + metrics_by_key = {key: "metrics"} + assert metrics_by_key[equal_key] == "metrics" + assert other_key not in metrics_by_key + + @pytest.mark.asyncio + async def test_get_stats_requires_connection_and_auth(self): + """get_stats fails before connecting, before login, and after logout.""" + host, port = get_server_config() + wait_for_server(host, port) + + client = IggyClient(f"{host}:{port}") + with pytest.raises(RuntimeError, match="Not connected"): + await client.get_stats() + + await client.connect() + with pytest.raises(RuntimeError, match="Unauthenticated"): + await client.get_stats() + + await client.login_user("iggy", "iggy") + await client.get_stats() + + await client.logout_user() + with pytest.raises(RuntimeError, match="Unauthenticated"): + await client.get_stats() + + @pytest.mark.asyncio + @pytest.mark.parametrize("flag", ["read_servers", "manage_servers"]) + async def test_get_stats_requires_server_permission( + self, iggy_client: IggyClient, unique_name, flag + ): + """A user without read_servers or manage_servers is denied; either grants.""" + username, password = unique_credentials(unique_name) + created = await iggy_client.create_user(username, password) + + try: + denied_client = await login_fresh_client(username, password) + with pytest.raises(RuntimeError, match="Unauthorized"): + await denied_client.get_stats() + + await iggy_client.update_permissions( + created.id, + Permissions(global_permissions=GlobalPermissions(**{flag: True})), + ) + + granted_client = await login_fresh_client(username, password) + stats = await granted_client.get_stats() + assert stats.process_id > 0 + finally: + await iggy_client.delete_user(created.id) From 08ff476f679a2b71c5d9a1e2ee70db58a82a75bd Mon Sep 17 00:00:00 2001 From: Justin Mclean Date: Mon, 7 Sep 2026 17:54:02 +1000 Subject: [PATCH 073/182] docs: add AI assistance expectations to CONTRIBUTING (#4081) --- CONTRIBUTING.md | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 3d8744e6d6..ddbd31c5f3 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -28,6 +28,23 @@ These require design discussion in the issue before coding: Authors of PRs must run the code locally. "Relying on CI" is not acceptable. +### AI Assistance + +You are responsible for the code you submit, even if a tool wrote it. + +Using an AI assistant to help you code is fine. Submitting code you don't understand is +not. Before you open a PR you must be able to explain what every part of the change does +and why, answer review questions about it yourself, and defend the design without going +back to the tool for an answer. If you can't, the PR isn't ready. + +While you're new to the project, please keep to **one open PR at a time**. Review takes +longer than writing, so a queue of changes from one contributor holds up everyone else's. + +Maintainers may close a PR at first review if it reads as a relay between the reviewer and +a model, rather than a change the author understands and takes responsibility for. That is +a judgment about the submission, not about you, and it does not bar you from contributing +again if you come back with a change you can take responsibility for. + ### Green CI Maintainers will not start reviewing a PR while its CI is failing. Get the From 9a7ead986651eeb12c1cf53b6bcf08dfb3a0a3a2 Mon Sep 17 00:00:00 2001 From: Richard Cocks <50965970+richardcocks@users.noreply.github.com> Date: Mon, 7 Sep 2026 09:14:09 +0100 Subject: [PATCH 074/182] perf(node, csharp, java, go): generate message ids from random bytes instead of via UUIDs (#4076) Closes #4066 --- .../Iggy_SDK/Contracts/Tcp/TcpContracts.cs | 5 +- foreign/go/internal/command/message.go | 9 +- .../apache/iggy/serde/BytesSerializer.java | 4 +- .../apache/iggy/serde/MessageIdGenerator.java | 74 ++++++++++ .../iggy/serde/MessageIdGeneratorTest.java | 80 +++++++++++ .../src/wire/message/message.utils.test.ts | 129 ++++++++++++++++++ .../node/src/wire/message/message.utils.ts | 44 +++++- foreign/node/src/wire/number.utils.test.ts | 67 +++++++++ foreign/node/src/wire/number.utils.ts | 25 ++-- 9 files changed, 412 insertions(+), 25 deletions(-) create mode 100644 foreign/java/java-sdk/src/main/java/org/apache/iggy/serde/MessageIdGenerator.java create mode 100644 foreign/java/java-sdk/src/test/java/org/apache/iggy/serde/MessageIdGeneratorTest.java create mode 100644 foreign/node/src/wire/message/message.utils.test.ts create mode 100644 foreign/node/src/wire/number.utils.test.ts diff --git a/foreign/csharp/Iggy_SDK/Contracts/Tcp/TcpContracts.cs b/foreign/csharp/Iggy_SDK/Contracts/Tcp/TcpContracts.cs index 4a825f6d42..4167dea90e 100644 --- a/foreign/csharp/Iggy_SDK/Contracts/Tcp/TcpContracts.cs +++ b/foreign/csharp/Iggy_SDK/Contracts/Tcp/TcpContracts.cs @@ -354,13 +354,14 @@ internal static int CreateMessage(Span bytes, Identifier streamId, Identif // The producer owns message ids: a zero id is minted before the frame checksum covers it. var originTimestamp = ulong.MaxValue; + Span idBytes = stackalloc byte[16]; foreach (var message in messages) { if (message.Header.Id == 0) { - message.Header = message.Header with { Id = Guid.NewGuid().ToUInt128() }; + Random.Shared.NextBytes(idBytes); + message.Header = message.Header with { Id = BinaryPrimitives.ReadUInt128LittleEndian(idBytes) }; } - originTimestamp = Math.Min(originTimestamp, message.Header.OriginTimestamp); } diff --git a/foreign/go/internal/command/message.go b/foreign/go/internal/command/message.go index 19d277a001..4b995d8e87 100644 --- a/foreign/go/internal/command/message.go +++ b/foreign/go/internal/command/message.go @@ -22,10 +22,10 @@ import ( "errors" "fmt" "math" + "math/rand/v2" "github.com/apache/iggy/foreign/go/contracts" "github.com/apache/iggy/foreign/go/internal/batch" - "github.com/google/uuid" "github.com/klauspost/compress/s2" "github.com/zeebo/xxh3" ) @@ -96,11 +96,8 @@ func (s *SendMessages) AppendBinary(b []byte) ([]byte, error) { // The id sits under the frame checksum, so it must exist before the // frame is hashed; the server never mints ids. if message.Header.Id == (iggcon.MessageID{}) { - id, err := uuid.NewRandom() - if err != nil { - return b, err - } - message.Header.Id = iggcon.MessageID(id) + binary.LittleEndian.PutUint64(message.Header.Id[0:8], rand.Uint64()) + binary.LittleEndian.PutUint64(message.Header.Id[8:16], rand.Uint64()) } // The header lengths and the appended slices must agree, or every // message boundary after a mismatch mis-frames; deriving both from diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/serde/BytesSerializer.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/serde/BytesSerializer.java index 6a67b138f7..7a20e27fa1 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/serde/BytesSerializer.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/serde/BytesSerializer.java @@ -33,7 +33,6 @@ import org.apache.iggy.message.MessageId; import org.apache.iggy.message.Partitioning; import org.apache.iggy.message.PollingStrategy; -import org.apache.iggy.message.UuidMessageId; import org.apache.iggy.user.GlobalPermissions; import org.apache.iggy.user.Permissions; import org.apache.iggy.user.StreamPermissions; @@ -45,7 +44,6 @@ import java.util.List; import java.util.Map; import java.util.Optional; -import java.util.UUID; /** * Unified serializer for both blocking and async clients. @@ -340,7 +338,7 @@ private static long batchChecksum( */ private static byte[] encodedMessageId(MessageId id) { if (id.toBigInteger().signum() == 0) { - return readAllBytes(new UuidMessageId(UUID.randomUUID()).toBytes()); + return MessageIdGenerator.mint(); } return readAllBytes(id.toBytes()); } diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/serde/MessageIdGenerator.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/serde/MessageIdGenerator.java new file mode 100644 index 0000000000..a66cbf9ab8 --- /dev/null +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/serde/MessageIdGenerator.java @@ -0,0 +1,74 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.iggy.serde; + +import javax.crypto.Cipher; +import javax.crypto.spec.IvParameterSpec; +import javax.crypto.spec.SecretKeySpec; +import java.security.GeneralSecurityException; +import java.security.SecureRandom; + +/** + * Mints the opaque 16-byte id stamped on a message whose caller passed a zero id. + * + *

Each thread runs its own AES-CTR keystream: a random 128-bit key and IV are drawn once from + * {@link SecureRandom} and never reused or decrypted, so the counter blocks are cryptographically + * strong random bytes. The id therefore has 128-bit collision resistance backed by 256 bits of + * independent per-thread seed, and each mint is a single AES-NI block. This is a randomness source, + * not a security primitive: the id is opaque, is not keyed on, and need not stay secret, so the + * keystream is never reseeded. + */ +final class MessageIdGenerator { + + /** Constant plaintext encrypted to read a block of keystream; never mutated. */ + private static final byte[] PLAINTEXT = new byte[16]; + + /** Draws each thread's key and IV once, from a cryptographic source. */ + private static final SecureRandom SEEDER = new SecureRandom(); + + private static final ThreadLocal KEYSTREAM = ThreadLocal.withInitial(MessageIdGenerator::newKeystream); + + private MessageIdGenerator() {} + + /** Returns a fresh 16-byte id from the calling thread's keystream. */ + static byte[] mint() { + byte[] minted = new byte[16]; + try { + KEYSTREAM.get().update(PLAINTEXT, 0, 16, minted); + } catch (GeneralSecurityException e) { + throw new IllegalStateException("failed to mint a message id", e); + } + return minted; + } + + private static Cipher newKeystream() { + byte[] key = new byte[16]; + byte[] iv = new byte[16]; + SEEDER.nextBytes(key); + SEEDER.nextBytes(iv); + try { + Cipher keystream = Cipher.getInstance("AES/CTR/NoPadding"); + keystream.init(Cipher.ENCRYPT_MODE, new SecretKeySpec(key, "AES"), new IvParameterSpec(iv)); + return keystream; + } catch (GeneralSecurityException e) { + throw new IllegalStateException("failed to initialise the message-id keystream", e); + } + } +} diff --git a/foreign/java/java-sdk/src/test/java/org/apache/iggy/serde/MessageIdGeneratorTest.java b/foreign/java/java-sdk/src/test/java/org/apache/iggy/serde/MessageIdGeneratorTest.java new file mode 100644 index 0000000000..31f372bff8 --- /dev/null +++ b/foreign/java/java-sdk/src/test/java/org/apache/iggy/serde/MessageIdGeneratorTest.java @@ -0,0 +1,80 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.iggy.serde; + +import org.junit.jupiter.api.Test; + +import java.math.BigInteger; +import java.util.ArrayList; +import java.util.HashSet; +import java.util.List; +import java.util.Set; + +import static org.assertj.core.api.Assertions.assertThat; + +/** + * The zero-id mint: ids fill the full 16 bytes, never repeat within a thread (the AES-CTR counter + * always advances), and never collide across threads (each thread seeds an independent keystream). + */ +class MessageIdGeneratorTest { + + @Test + void shouldMintSixteenBytes() { + assertThat(MessageIdGenerator.mint()).hasSize(16); + } + + @Test + void shouldMintDistinctIdsWithinAThread() { + var count = 100_000; + Set ids = new HashSet<>(count * 2); + for (var i = 0; i < count; i++) { + ids.add(new BigInteger(1, MessageIdGenerator.mint())); + } + assertThat(ids).hasSize(count); + } + + @Test + void shouldMintDistinctIdsAcrossThreads() throws InterruptedException { + var threads = 4; + var perThread = 25_000; + List> perThreadIds = new ArrayList<>(); + List workers = new ArrayList<>(); + for (var t = 0; t < threads; t++) { + List mine = new ArrayList<>(perThread); + perThreadIds.add(mine); + var worker = new Thread(() -> { + for (var i = 0; i < perThread; i++) { + mine.add(new BigInteger(1, MessageIdGenerator.mint())); + } + }); + workers.add(worker); + worker.start(); + } + for (var worker : workers) { + worker.join(); + } + + Set all = new HashSet<>(threads * perThread * 2); + for (var ids : perThreadIds) { + all.addAll(ids); + } + assertThat(all).hasSize(threads * perThread); + } +} diff --git a/foreign/node/src/wire/message/message.utils.test.ts b/foreign/node/src/wire/message/message.utils.test.ts new file mode 100644 index 0000000000..c58a9882eb --- /dev/null +++ b/foreign/node/src/wire/message/message.utils.test.ts @@ -0,0 +1,129 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { u128ToBuf } from "../number.utils.js"; +import { + isValidMessageId, + serializeMessageId, + resolveMessageId, + mintMessageId, +} from "./message.utils.js"; + +const MESSAGE_ID_SIZE = 16; +const MAX_U128 = (1n << 128n) - 1n; +const NIL_UUID = "00000000-0000-0000-0000-000000000000"; +const isZero = (b: Buffer) => b.every((byte) => byte === 0); + +describe("isValidMessageId", () => { + it("accepts undefined, string, number, and bigint", () => { + assert.ok(isValidMessageId(undefined)); + assert.ok(isValidMessageId("id")); + assert.ok(isValidMessageId(7)); + assert.ok(isValidMessageId(7n)); + }); + + it("rejects other types", () => { + assert.ok(!isValidMessageId(null)); + assert.ok(!isValidMessageId({})); + }); +}); + +describe("serializeMessageId", () => { + it("serializes undefined to a zero u128", () => { + assert.deepEqual(serializeMessageId(), Buffer.alloc(MESSAGE_ID_SIZE, 0)); + }); + + it("serializes a number as little-endian u128", () => { + assert.deepEqual(serializeMessageId(7), u128ToBuf(7n)); + }); + + it("serializes a bigint as little-endian u128", () => { + assert.deepEqual(serializeMessageId(8n), u128ToBuf(8n)); + }); + + it("serializes a UUID string to the same bytes as its numeric value", () => { + const uuid = "00000000-0000-0000-0000-000000000007"; + assert.deepEqual(serializeMessageId(uuid), u128ToBuf(7n)); + }); + + it("accepts the largest u128", () => { + assert.deepEqual(serializeMessageId(MAX_U128), u128ToBuf(MAX_U128)); + }); + + it("rejects a numeric id at or above 2^128", () => { + assert.throws(() => serializeMessageId(1n << 128n), /2\^128/); + }); + + it("rejects a negative numeric id", () => { + assert.throws(() => serializeMessageId(-1n), />= 0/); + }); + + it("rejects an unparsable string", () => { + assert.throws(() => serializeMessageId("not-a-uuid"), /invalid message id/); + }); + + it("rejects an invalid type", () => { + assert.throws(() => serializeMessageId({}), /invalid message id/); + }); +}); + +describe("resolveMessageId", () => { + it("mints a non-zero id for undefined, 0, and 0n", () => { + for (const id of [undefined, 0, 0n]) { + const b = resolveMessageId(id); + assert.equal(b.length, MESSAGE_ID_SIZE); + assert.ok(!isZero(b)); + } + }); + + it("mints for the all-zero nil UUID string", () => { + assert.ok(!isZero(resolveMessageId(NIL_UUID))); + }); + + it("passes a provided non-zero id through unchanged", () => { + assert.deepEqual(resolveMessageId(7n), serializeMessageId(7n)); + }); +}); + +describe("mintMessageId", () => { + it("returns a non-zero 16-byte buffer", () => { + const b = mintMessageId(); + assert.equal(b.length, MESSAGE_ID_SIZE); + assert.ok(!isZero(b)); + }); + + it("produces unique, non-zero ids across pool refills", () => { + const seen = new Set(); + for (let i = 0; i < 10_000; i++) { + const hex = mintMessageId().toString("hex"); + assert.notEqual(hex, "0".repeat(MESSAGE_ID_SIZE * 2)); + seen.add(hex); + } + assert.equal(seen.size, 10_000); + }); + + it("returns owned bytes that survive a later refill", () => { + const first = mintMessageId().toString("hex"); + const held = mintMessageId(); + const snapshot = held.toString("hex"); + for (let i = 0; i < 10_000; i++) mintMessageId(); + assert.equal(held.toString("hex"), snapshot); + assert.notEqual(snapshot, first); + }); +}); diff --git a/foreign/node/src/wire/message/message.utils.ts b/foreign/node/src/wire/message/message.utils.ts index 8ed7c0b289..576dd01d4b 100644 --- a/foreign/node/src/wire/message/message.utils.ts +++ b/foreign/node/src/wire/message/message.utils.ts @@ -15,7 +15,7 @@ // specific language governing permissions and limitations // under the License. -import { uuidv4 } from 'uuidv7'; +import { randomFillSync } from 'node:crypto'; import { uint32ToBuf, u128ToBuf, uint8ToBuf } from '../number.utils.js'; import { serializeHeaders, type Headers } from './header.utils.js'; import { serializeIdentifier, type Id } from '../identifier.utils.js'; @@ -32,6 +32,9 @@ import { /** Size of the message ID in bytes (u128) */ const MESSAGE_ID_SIZE = 16; +/** Exclusive upper bound for a numeric message ID: it must be < 2^128 */ +const MESSAGE_ID_UPPER_BOUND = 1n << BigInt((MESSAGE_ID_SIZE * 8)); + /** Largest representable frame timestamp delta (u32, microseconds) */ const MAX_TIMESTAMP_DELTA = 0xFFFF_FFFFn; @@ -99,6 +102,8 @@ export const serializeMessageId = (id?: unknown) => { throw new Error(`invalid message id: '${id}' (numeric id must be >= 0)`) const idValue = 'number' === typeof id ? BigInt(id) : id; + if (idValue >= MESSAGE_ID_UPPER_BOUND) + throw new Error(`invalid message id: '${id}' (numeric id must be < 2^${MESSAGE_ID_SIZE * 8})`) return u128ToBuf(idValue); } @@ -114,17 +119,44 @@ export const serializeMessageId = (id?: unknown) => { } +/** Number of ids drawn from the pool per CSPRNG refill */ +const ID_POOL_COUNT = 4096; + +/** Pooled random bytes and a cursor into them, filled lazily on first mint */ +const idPool = Buffer.allocUnsafe(ID_POOL_COUNT * MESSAGE_ID_SIZE); +let idPoolCursor = idPool.length; // past the end -> refill on first use + +/** + * Mints a random 16-byte message ID from the pool, refilling when drained. + * + * @returns 16-byte buffer of random bytes owned by the caller + */ +export const mintMessageId = (): Buffer => { + if (idPoolCursor + MESSAGE_ID_SIZE > idPool.length) { + randomFillSync(idPool); + idPoolCursor = 0; + } + const id = Buffer.allocUnsafe(MESSAGE_ID_SIZE); + idPool.copy(id, 0, idPoolCursor, idPoolCursor + MESSAGE_ID_SIZE); + idPoolCursor += MESSAGE_ID_SIZE; + return id; +}; + /** - * Serializes a message ID, minting a random UUID when the ID is - * absent or zero. + * Resolves a message ID to a 16-byte little-endian buffer, minting a random + * one when the ID is absent or zero. * * @param id - Optional message ID * @returns 16-byte little-endian buffer containing a non-zero ID */ -const resolveMessageId = (id?: MessageIdKind): Buffer => { +export const resolveMessageId = (id?: MessageIdKind): Buffer => { + // An absent or zero id mints a random one. + if (id === undefined || id === 0 || id === 0n) + return mintMessageId(); const bId = serializeMessageId(id); - return bId.every((byte) => byte === 0) - ? u128ToBuf(BigInt(`0x${uuidv4().replaceAll('-', '')}`)) + // A string id can still be the all-zero nil UUID; mint in that case too. + return 'string' === typeof id && bId.every((byte) => byte === 0) + ? mintMessageId() : bId; }; diff --git a/foreign/node/src/wire/number.utils.test.ts b/foreign/node/src/wire/number.utils.test.ts new file mode 100644 index 0000000000..1a53f552ae --- /dev/null +++ b/foreign/node/src/wire/number.utils.test.ts @@ -0,0 +1,67 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { u128ToBuf } from "./number.utils.js"; + +const MAX_U128 = (1n << 128n) - 1n; +const hex = (v: bigint) => u128ToBuf(v).toString("hex"); + +describe("u128ToBuf", () => { + it("encodes zero as 16 zero bytes", () => { + assert.equal(hex(0n), "0".repeat(32)); + }); + + it("encodes a small value little-endian", () => { + assert.equal(hex(7n), "07" + "0".repeat(30)); + }); + + it("orders all 16 bytes little-endian", () => { + // Distinct bytes so a byte-order slip is visible. + assert.equal( + hex(0x0102030405060708090a0b0c0d0e0f10n), + "100f0e0d0c0b0a090807060504030201", + ); + }); + + it("spans the 64-bit half boundary", () => { + assert.equal(hex((1n << 64n) - 1n), "ff".repeat(8) + "00".repeat(8)); + assert.equal(hex(1n << 64n), "00".repeat(8) + "01" + "00".repeat(7)); + }); + + it("encodes the largest u128 as all ones", () => { + assert.equal(hex(MAX_U128), "ff".repeat(16)); + }); + + it("round-trips through readBigUInt64LE halves", () => { + for (const v of [0n, 1n, 42n, 1n << 64n, MAX_U128]) { + const b = u128ToBuf(v); + const low = b.readBigUInt64LE(0); + const high = b.readBigUInt64LE(8); + assert.equal((high << 64n) | low, v); + } + }); + + it("throws for a value at or above 2^128", () => { + assert.throws(() => u128ToBuf(1n << 128n)); + }); + + it("throws for a negative value", () => { + assert.throws(() => u128ToBuf(-1n)); + }); +}); diff --git a/foreign/node/src/wire/number.utils.ts b/foreign/node/src/wire/number.utils.ts index 3ea8f691fb..aa2ce6a589 100644 --- a/foreign/node/src/wire/number.utils.ts +++ b/foreign/node/src/wire/number.utils.ts @@ -147,17 +147,26 @@ export const doubleToBuf = (v: number) => { return b; } +/** Mask selecting the low 64 bits of a u128, for little-endian half-writes. */ +const U64_MASK = 0xFFFF_FFFF_FFFF_FFFFn; + /** - * Converts a BigInt to a 128-bit unsigned integer Buffer in little-endian format. + * Converts a u128 value to a 16-byte unsigned integer Buffer in little-endian + * format. * - * @param num - BigInt value to convert - * @param width - Width in bytes (default: 16) - * @returns Buffer containing the value in little-endian byte order + * Writes the two 64-bit halves directly rather than round-tripping through a hex + * string. `value` must be a non-negative integer below 2^128; anything outside + * that range throws (a `RangeError` from the underlying write), which is + * preferable to silently truncating a message id. + * + * @param value - u128 value (0 <= value < 2^128) + * @returns 16-byte buffer containing the value in little-endian byte order */ -export function u128ToBuf(num: bigint, width = 16): Buffer { - const hex = num.toString(16); - const b = Buffer.from(hex.padStart(width * 2, '0').slice(0, width * 2), 'hex'); - return b.reverse(); +export function u128ToBuf(value: bigint): Buffer { + const b = Buffer.allocUnsafe(16); + b.writeBigUInt64LE(value & U64_MASK, 0); + b.writeBigUInt64LE(value >> 64n, 8); + return b; } /** From 227786432dc2f105c8d1139ae30a6aa8df8e2e56 Mon Sep 17 00:00:00 2001 From: Grzegorz Koszyk <112548209+numinnex@users.noreply.github.com> Date: Mon, 7 Sep 2026 10:45:09 +0200 Subject: [PATCH 075/182] fix(shard): drive metadata repair and the commit walk from the tick (#4008) A metadata backup that fell behind had no way back on its own, the same starvation the partition plane just had. A follower advances commit_max from every prepare header in replicate_preflight, before the gap check drops the prepare, so the commit heartbeat lands as Accepted rather than Advanced and the arming site inside that branch never fires. Under sustained metadata traffic the gap wedged until an unrelated view change. The same starvation left a backup holding resident committed ops that nothing re-drove, since the follower walk runs at the tail of an accepted prepare and the gap check returns before it. The metadata tick now evaluates the same two level-triggered checks the partition driver uses, off one probe. A hole below commit_max arms the existing repair against the primary, debounced on the repair retry interval. Resident committed ops re-drive the walk, follower only, because a backup's walk ships no wire replies while a stranded primary is resume_stranded_commits' job. A gap below the serving peer's retention floor needs no special handling: the serve path answers RangeEvicted and the existing conversion arms a state transfer. --- core/configs/src/server_config/cluster.rs | 8 +- core/metadata/src/impls/metadata.rs | 153 +++- core/partitions/src/iggy_partition.rs | 4 +- core/server/config.toml | 19 +- core/server/src/boot/recovery.rs | 6 +- core/server/src/shell.rs | 6 +- core/shard/src/lib.rs | 864 ++++++++++++++++++---- core/shard/src/metrics.rs | 39 +- core/simulator/src/deps.rs | 14 + core/simulator/src/lib.rs | 726 +++++++++++++++++- 10 files changed, 1664 insertions(+), 175 deletions(-) diff --git a/core/configs/src/server_config/cluster.rs b/core/configs/src/server_config/cluster.rs index 929200f087..0667c131cc 100644 --- a/core/configs/src/server_config/cluster.rs +++ b/core/configs/src/server_config/cluster.rs @@ -279,13 +279,15 @@ pub struct ClusterConfig { #[serde_as(as = "DisplayFromStr")] #[config_env(leaf)] pub repair_retry_interval: IggyDuration, - /// How long a partition backup must hold committed ops it cannot walk to - /// before the shard sweep OPENS a repair session for it. + /// How long a backup must hold committed ops it cannot walk to before the + /// shard tick OPENS a repair session for it. Paces both planes' detectors: + /// the partition sweep in `tick_partitions` and the metadata one in + /// `tick_metadata`. /// /// Separate from `repair_retry_interval`, which paces an already-open /// stream: this one decides how long a replication hole stays open, so /// raising the retry interval to quiet repair chatter must not widen it. - /// Floored at `PARTITION_GAP_DEBOUNCE_TICKS_MIN` consensus ticks, since one + /// Floored at `REPAIR_GAP_DEBOUNCE_TICKS_MIN` consensus ticks, since one /// tick of lag is ordinary pipelining and repair against it would fire on /// healthy traffic; the shard crate owns the floor and `config.toml` states /// its value. Zero (and the `0` / `disabled` / `unlimited` sentinels, which diff --git a/core/metadata/src/impls/metadata.rs b/core/metadata/src/impls/metadata.rs index 726a013c13..9fbd167cfd 100644 --- a/core/metadata/src/impls/metadata.rs +++ b/core/metadata/src/impls/metadata.rs @@ -218,6 +218,20 @@ impl IggySnapshot { /// accepted (unverified, loudly) while a PRESENT but mismatching one refuses boot. A /// bare checksum could not tell those apart, and guessing wrong in either direction is /// unacceptable: silently accepting corruption, or bricking a healthy node. +/// Committed ops one [`IggyMetadata::commit_journal`] call applies before +/// returning to the pump. +/// +/// The twin of `partitions::COMMIT_WALK_OPS_MAX`, and needed for the same +/// reason: the walk reads a WAL body and applies it per op with no await the +/// pump can interleave, and the resident `(commit_min, commit_max]` run is the +/// whole backlog after a repair or a rejoin, not the pipeline depth. +/// +/// Every caller is re-driven, so a truncated walk resumes rather than losing +/// anything: `tick_metadata`'s walk backstop covers a follower and +/// `resume_stranded_commits` covers the primary, both level-triggered on +/// `commit_min < commit_max` every tick. +const COMMIT_WALK_OPS_MAX: usize = 64; + const SNAPSHOT_TRAILER_MAGIC: u32 = 0x4953_4E50; /// `magic` + the payload's [`checkpoint_checksum`]. @@ -776,6 +790,16 @@ pub struct IggyMetadata { /// whole snapshot on shard 0's pump, and hands each requester its own /// multi-MB copy. transfer_offer_cache: RefCell>>, + /// Prepares the backup gap check destroyed since `tick_metadata` last + /// drained the count into `metadata_prepare_gap_drops_total`. What it does + /// and does not prove is `IggyPartition::prepare_gap_drops`, verbatim; what + /// differs is the frontier the check runs against, the journal head rather + /// than the sequencer, so this also counts the ops that fall outside what + /// metadata repair can refill (an interior hole below the head, a forward + /// gap above `commit_max`). + /// + /// `Cell` because every method on this type takes `&self`. + prepare_gap_drops: Cell, /// Highest metadata op whose apply has been PUBLISHED on this node, plus /// the reads parked on it. Shared by every shard; see /// [`AppliedFrontier`] for the ordering and the wake contract. @@ -838,12 +862,19 @@ where commit_notifier: RefCell::new(None), client_table_frontier: Cell::new(0), transfer_offer_cache: RefCell::new(None), + prepare_gap_drops: Cell::new(0), applied_frontier: Arc::default(), } } } impl IggyMetadata { + /// Take and clear the gap-drop count (`prepare_gap_drops`). + #[must_use = "dropping the count loses the only record those prepares existed"] + pub const fn take_prepare_gap_drops(&self) -> u64 { + self.prepare_gap_drops.replace(0) + } + /// Share one process-wide applied frontier with every other shard. /// /// Consumed at construction rather than swapped in later: a shard that @@ -1280,6 +1311,8 @@ where sequencer_op = current_op, "on_replicate: dropping out-of-order prepare (gap)" ); + self.prepare_gap_drops + .set(self.prepare_gap_drops.get().saturating_add(1)); return; } } else { @@ -3608,15 +3641,26 @@ where let consensus = self.consensus.as_ref().unwrap(); let journal = self.journal.as_ref().unwrap(); + let mut applied = 0usize; while consensus.commit_min() < consensus.commit_max() { + if applied == COMMIT_WALK_OPS_MAX { + debug!( + "commit_journal: stopping at op={} after {applied} ops; resuming next tick", + consensus.commit_min() + ); + break; + } + applied += 1; let op = consensus.commit_min() + 1; let Some(header) = journal.handle().header(op as usize) else { // Gap-stop: the walk halts at the first missing prepare and - // resumes once it is refilled. Live drops refill via the - // primary's prepare retransmit; a replica behind at recovery - // or after StartView adoption arms a `MetadataRepairSession` - // (shard) that re-requests the missing window. + // resumes once it is refilled -- by the primary's retransmit + // while the op still lacks quorum, otherwise by a + // `MetadataRepairSession` (shard), armed at recovery, at + // StartView adoption, or by `tick_metadata`'s gap detector, + // which is the only one of those a live drop under sustained + // traffic reaches. break; }; let header = *header; @@ -4253,6 +4297,20 @@ mod tests { IggyMetadata::new(None, None, None, None, TestMux::default(), None) } + #[test] + fn take_prepare_gap_drops_drains_the_count() { + let md = peer_metadata(); + assert_eq!(md.take_prepare_gap_drops(), 0); + + md.prepare_gap_drops.set(2); + assert_eq!(md.take_prepare_gap_drops(), 2); + assert_eq!( + md.take_prepare_gap_drops(), + 0, + "a second drain must not re-report drops the metrics already counted" + ); + } + #[test] fn commit_notifier_fires_with_received_operation() { let md = peer_metadata(); @@ -5035,6 +5093,93 @@ mod tests { ); } + /// The walk cap bounds ONE `commit_journal` call, not the backlog. + /// + /// The resident `(commit_min, commit_max]` run after a repair window or a + /// rejoin is the whole backlog, and the walk applies each op with no await + /// the pump can interleave, so an uncapped call holds the shard for all of + /// it. Every caller is re-driven every tick, so stopping short loses + /// nothing; a cap that did NOT resume would pin `commit_min` until the next + /// op to commit trips `advance_commit_min`'s sequential assert. + #[compio::test] + async fn commit_journal_stops_at_the_walk_cap_and_resumes_on_the_next_call() { + const CLIENT: u128 = 1; + const SESSION: u64 = 1; + const ACTING_USER: u32 = 7; + /// The cap, as an op count. + const CAP: u64 = COMMIT_WALK_OPS_MAX as u64; + /// One op past the cap, so the first call must stop short and the + /// second must have something left to finish. + const OPS: u64 = CAP + 1; + + let dir = tempfile::tempdir().unwrap(); + std::fs::create_dir_all(dir.path().join(crate::impls::METADATA_DIR)).unwrap(); + let journal = + journal::prepare_journal::PrepareJournal::open(&dir.path().join("journal.wal"), 0) + .await + .unwrap(); + // Replica 1 of 3 at view 0: a backup, so `on_replicate` journals each + // prepare without the primary's pipeline commit path running under it. + let consensus = VsrConsensus::new( + 1, + 1, + 3, + server_common::sharding::METADATA_GROUP, + NoopBus, + LocalPipeline::new(), + ); + consensus.init(); + let md: IggyMetadata<_, journal::prepare_journal::PrepareJournal, (), TestMux> = + IggyMetadata::new( + Some(consensus), + Some(journal), + None, + None, + TestMux::default(), + Some(dir.path().to_path_buf()), + ); + let consensus = md.consensus.as_ref().unwrap(); + md.client_table.borrow_mut().commit_register( + CLIENT, + ACTING_USER, + register_reply(CLIENT, SESSION), + ); + + for request in 1..=OPS { + let prepare = md + .prepare_request(create_stream_request( + CLIENT, + request, + &format!("s{request}"), + )) + .expect("CreateStream is client-allowed"); + md.on_replicate(prepare).await; + } + let journal = md.journal.as_ref().unwrap(); + assert_eq!(journal.last_op(), Some(OPS), "every op must be resident"); + assert_eq!(consensus.commit_min(), 0, "no commit heartbeat has landed"); + + // What a repair window or a rejoin leaves behind: the whole run + // committed by the group and resident here, none of it walked. + consensus.advance_commit_max(OPS); + + md.commit_journal().await; + assert_eq!( + consensus.commit_min(), + CAP, + "one call walked the whole backlog; the pump is blocked for as long \ + as the resident run is, however long that is" + ); + + md.commit_journal().await; + assert_eq!( + consensus.commit_min(), + OPS, + "the walk did not resume where it stopped, so `commit_min` is pinned \ + below `commit_max` with no other re-driver" + ); + } + /// A state-transfer receiver admits the first live prepare above the floor it /// installed, instead of waiting for an op the snapshot already contains. /// diff --git a/core/partitions/src/iggy_partition.rs b/core/partitions/src/iggy_partition.rs index 2128ae517b..356a10a7ef 100644 --- a/core/partitions/src/iggy_partition.rs +++ b/core/partitions/src/iggy_partition.rs @@ -159,7 +159,7 @@ where /// plus one is missing). Debounces the sweep's level-triggered repair arm, /// and is spent by whichever site opens the repair session. /// - /// `Cell` for the same reason as [`Self::prepare_gap_drops`]: the sweep + /// `Cell` for the same reason as `prepare_gap_drops`: the sweep /// drives it from the shared borrow it probes the partition through, so the /// in-flight scan the arm budget needs can run without a `&mut` outstanding. pub gap_ticks: Cell, @@ -1749,7 +1749,7 @@ where self.write_superblock_advancing(superblock, 0, claim).await } - /// Take and clear the gap-drop count ([`Self::prepare_gap_drops`]). + /// Take and clear the gap-drop count (`prepare_gap_drops`). #[must_use = "dropping the count loses the only record those prepares existed"] pub const fn take_prepare_gap_drops(&self) -> u64 { self.prepare_gap_drops.replace(0) diff --git a/core/server/config.toml b/core/server/config.toml index 3bae72a0cc..e369125132 100644 --- a/core/server/config.toml +++ b/core/server/config.toml @@ -659,21 +659,22 @@ view_probe_attempts_max = 5 # waits before one is opened for it is repair_gap_debounce_interval below. repair_retry_interval = "1s" -# How long a partition backup must hold committed ops it cannot walk to before -# the shard's tick sweep opens a repair session for it (duration). The -# level-triggered floor under every edge-triggered arming site, which a produce -# stream can starve; must be nonzero. +# How long a backup must hold committed ops it cannot walk to before the shard's +# tick opens a repair session for it (duration). The level-triggered floor under +# every edge-triggered arming site, which a produce stream can starve; must be +# nonzero. Paces the partition sweep and the metadata detector alike. # # Floored at 50 consensus ticks (500ms at the 10ms tick), so values under that # arm no sooner: one tick of lag is ordinary pipelining, and repair against it # would fire on healthy traffic. # # Recovery latency for one hole is max(this, 500ms) plus up to one tick of sweep -# granularity. Under a correlated fault the sweep opens at most 3 sessions per -# tick and holds at most 8 at once, so the Nth group waiting on this shard adds -# ceil(N / 3) ticks on top, and more while sessions are already in flight. That -# cap is what keeps a node-wide rejoin from putting every group's repair stream -# on one serving peer at once. +# granularity. Under a correlated fault the partition sweep opens at most 3 +# sessions per tick and holds at most 8 at once, so the Nth group waiting on +# this shard adds ceil(N / 3) ticks on top, and more while sessions are already +# in flight. That cap is what keeps a node-wide rejoin from putting every +# group's repair stream on one serving peer at once. There is one metadata group +# per node, so its detector is not rate-capped. repair_gap_debounce_interval = "1s" # Prepares a peer serves per repair round before the requester walks to the next diff --git a/core/server/src/boot/recovery.rs b/core/server/src/boot/recovery.rs index 27d7a45b7c..46558ffc7d 100644 --- a/core/server/src/boot/recovery.rs +++ b/core/server/src/boot/recovery.rs @@ -285,7 +285,7 @@ pub(in crate::boot) async fn build_shard_for_thread( // Repair pacing is shared by both planes' repair loops, so it is a // per-shard tunable set once here rather than per consensus group. shard.set_repair_retry_ticks(repair_retry_ticks(config)); - shard.set_partition_gap_debounce_ticks(repair_gap_debounce_ticks(config)); + shard.set_repair_gap_debounce_ticks(repair_gap_debounce_ticks(config)); shard.set_superblock_wedged_fatal_failures(superblock_wedged_fatal_failures(config)); shard.set_served_segment_cache_bytes_max( config @@ -947,13 +947,13 @@ mod tests { #[test] fn documented_gap_debounce_floor_matches_the_shard_constant() { assert_eq!( - shard::PARTITION_GAP_DEBOUNCE_TICKS_MIN, + shard::REPAIR_GAP_DEBOUNCE_TICKS_MIN, 50, "the gap debounce floor moved; core/server/config.toml states it in \ ticks and milliseconds under [cluster] repair_gap_debounce_interval" ); assert_eq!( - u128::from(shard::PARTITION_GAP_DEBOUNCE_TICKS_MIN) + u128::from(shard::REPAIR_GAP_DEBOUNCE_TICKS_MIN) * shard::CONSENSUS_TICK_INTERVAL.as_millis(), 500, "the floor is no longer 500ms; core/server/config.toml states that \ diff --git a/core/server/src/shell.rs b/core/server/src/shell.rs index 18145cb689..10da7aecf7 100644 --- a/core/server/src/shell.rs +++ b/core/server/src/shell.rs @@ -202,11 +202,11 @@ pub(crate) fn repair_retry_ticks(config: &ServerConfig) -> u32 { } /// `[cluster] repair_gap_debounce_interval` in consensus ticks: how long a -/// partition backup holds a hole before the sweep opens a repair session for -/// it. Deliberately NOT the retry interval above: that one paces an open +/// backup holds a hole before the tick opens a repair session for it, on either +/// plane. Deliberately NOT the retry interval above: that one paces an open /// stream, and pairing them means quieting retry chatter also widens how long a /// replication hole stays open. The shard applies -/// [`shard::PARTITION_GAP_DEBOUNCE_TICKS_MIN`] as a floor on top. +/// [`shard::REPAIR_GAP_DEBOUNCE_TICKS_MIN`] as a floor on top. pub(crate) fn repair_gap_debounce_ticks(config: &ServerConfig) -> u32 { u32::try_from(duration_to_ticks( config.cluster.repair_gap_debounce_interval.get_duration(), diff --git a/core/shard/src/lib.rs b/core/shard/src/lib.rs index 73cf6a9f0d..008db6fbc8 100644 --- a/core/shard/src/lib.rs +++ b/core/shard/src/lib.rs @@ -884,6 +884,10 @@ pub const REPAIR_CHUNK_MAX: u64 = 128; struct MetadataRepairSession { nonce: u128, to_op: u64, + /// Consensus view this session was armed in. A later view decides the log + /// again, so the window this names may no longer be the one to fetch; + /// `partitions::RepairSession::view` fences the partition twin the same way. + view: u32, /// Re-request target on stall. peer: u8, /// Ticks since the stream last made progress; at @@ -1319,6 +1323,21 @@ where /// repair takes over at install. See [`MetadataTransferSession`]. metadata_transfer: RefCell>, + /// Consecutive ticks the metadata group has been seen gap-stopped + /// (committed ops it cannot walk to, because the op one past its commit + /// frontier is missing from the WAL). Debounces `tick_metadata`'s + /// level-triggered repair arm; the partition twin is + /// `IggyPartition::gap_ticks`. `Cell` because the tick drives it through + /// `&self`, and shard-level rather than plane-level because there is one + /// metadata group per node (precedent: [`Self::metadata_transfer_attempts`]). + metadata_gap_ticks: Cell, + + /// Op the tick's commit walk last stopped on without moving, or `0`. The + /// journal names it but cannot produce its body, so the gap probe counts it + /// as absent and lets repair fetch it. Cleared implicitly: any advance of + /// `commit_min` makes it stop matching `commit_min + 1`. + metadata_walk_stuck_op: Cell, + /// Serving-side cache of state-transfer offers, both planes, keyed by /// `(namespace, requester replica id)`. Bounded by the replica count times /// the groups this shard serves; replaced per fresh nonce. @@ -1493,11 +1512,11 @@ where /// plane would alias; one sweep stale at worst. partition_repairs_inflight: Cell, - /// Live gap debounce in consensus ticks: how long a partition holds a hole - /// before the sweep opens a repair session for it. Defaults to - /// [`partitions::REPAIR_RETRY_TICKS`]; the server overrides it from - /// `[cluster] repair_gap_debounce_interval` at bootstrap. - partition_gap_debounce_ticks: Cell, + /// Live gap debounce in consensus ticks: how long a group holds a hole + /// before the tick opens a repair session for it. Shared by both planes. + /// Defaults to [`partitions::REPAIR_RETRY_TICKS`]; the server overrides it + /// from `[cluster] repair_gap_debounce_interval` at bootstrap. + repair_gap_debounce_ticks: Cell, /// Namespace the next partition sweep starts from: the first group the /// per-tick WALK budget turned away last pass, `None` to start at the front. @@ -1543,6 +1562,12 @@ where /// [`Self::metadata_transfer_decode_failures`]. metadata_transfer_attempts: Cell, + /// Consecutive stalled re-requests on the live metadata repair session, + /// against [`partitions::REPAIR_MAX_STALL_RETRIES`]. Survives the session, + /// so rotating the peer cannot reset it; cleared by an accepted repaired + /// prepare. + metadata_repair_attempts: Cell, + /// Decode failures charged against one snapshot generation, as /// `(snapshot_seq, failures)`. `None` until a pulled artifact set first /// fails to decode; cleared by a successful install. Past @@ -1678,6 +1703,8 @@ where partition_submit_stalled: Cell::new(false), metadata_repair: RefCell::new(None), metadata_transfer: RefCell::new(None), + metadata_gap_ticks: Cell::new(0), + metadata_walk_stuck_op: Cell::new(0), state_transfer_offers: RefCell::new(HashMap::new()), partition_offer_builds: RefCell::new(HashMap::new()), served_segment_cache: RefCell::new(ServedSegmentCache::default()), @@ -1688,12 +1715,13 @@ where partition_artifact_len_max: Cell::new(PARTITION_ARTIFACT_LEN_DEFAULT), repair_chunk_max: Cell::new(REPAIR_CHUNK_MAX), repair_retry_ticks: Cell::new(partitions::REPAIR_RETRY_TICKS), - partition_gap_debounce_ticks: Cell::new(partitions::REPAIR_RETRY_TICKS), + repair_gap_debounce_ticks: Cell::new(partitions::REPAIR_RETRY_TICKS), partition_repairs_inflight: Cell::new(0), partition_walk_cursor: Cell::new(None), superblock_wedged_fatal_failures: Cell::new(0), bus_max_message_size: Cell::new(DEFAULT_BUS_MAX_MESSAGE_SIZE), metadata_transfer_attempts: Cell::new(0), + metadata_repair_attempts: Cell::new(0), metadata_transfer_decode_failures: Cell::new(None), }) } @@ -1705,12 +1733,13 @@ where self.repair_retry_ticks.set(ticks); } - /// Override the partition sweep's gap debounce (consensus ticks) from - /// configuration. Called once per shard at bootstrap; the simulator and - /// tests keep the compile-time [`partitions::REPAIR_RETRY_TICKS`] default. - /// [`PARTITION_GAP_DEBOUNCE_TICKS_MIN`] still floors whatever is set. - pub fn set_partition_gap_debounce_ticks(&self, ticks: u32) { - self.partition_gap_debounce_ticks.set(ticks); + /// Override the tick gap debounce (consensus ticks) from configuration, + /// for both planes' detectors. Called once per shard at bootstrap; the + /// simulator and tests keep the compile-time + /// [`partitions::REPAIR_RETRY_TICKS`] default. + /// [`REPAIR_GAP_DEBOUNCE_TICKS_MIN`] still floors whatever is set. + pub fn set_repair_gap_debounce_ticks(&self, ticks: u32) { + self.repair_gap_debounce_ticks.set(ticks); } /// Arm the superblock fail-stop bound (consecutive write failures). @@ -2135,6 +2164,8 @@ where partition_submit_stalled: Cell::new(false), metadata_repair: RefCell::new(None), metadata_transfer: RefCell::new(None), + metadata_gap_ticks: Cell::new(0), + metadata_walk_stuck_op: Cell::new(0), state_transfer_offers: RefCell::new(HashMap::new()), partition_offer_builds: RefCell::new(HashMap::new()), served_segment_cache: RefCell::new(ServedSegmentCache::default()), @@ -2145,12 +2176,13 @@ where partition_artifact_len_max: Cell::new(PARTITION_ARTIFACT_LEN_DEFAULT), repair_chunk_max: Cell::new(REPAIR_CHUNK_MAX), repair_retry_ticks: Cell::new(partitions::REPAIR_RETRY_TICKS), - partition_gap_debounce_ticks: Cell::new(partitions::REPAIR_RETRY_TICKS), + repair_gap_debounce_ticks: Cell::new(partitions::REPAIR_RETRY_TICKS), partition_repairs_inflight: Cell::new(0), partition_walk_cursor: Cell::new(None), superblock_wedged_fatal_failures: Cell::new(0), bus_max_message_size: Cell::new(DEFAULT_BUS_MAX_MESSAGE_SIZE), metadata_transfer_attempts: Cell::new(0), + metadata_repair_attempts: Cell::new(0), metadata_transfer_decode_failures: Cell::new(None), } } @@ -4511,17 +4543,16 @@ where // its own WAL (a late joiner missed the ops below the // primary's active window; the primary only retransmits // uncommitted ops, never the committed prefix). Without - // this, such a replica learns it is behind and does - // nothing about it -- metadata repair is otherwise only - // rooted at StartView adoption, which a same-view - // late joiner never sees. Request repair from the - // primary; if it has checkpointed past the gap the - // repair floor evicts and the handler above converts to - // state transfer. Idempotent: `maybe_request_metadata_repair` - // no-ops when caught up, already transferring, or a - // session is live, so a caught-up replica and a - // cold-start node (commit_max == commit_min == 0) both - // skip it. + // this, such a replica waits out `tick_metadata`'s + // debounced gap detector; this edge is the fast path, + // for the runs where a heartbeat does land as + // `Advanced`. Request repair from the primary; if it + // has checkpointed past the gap the repair floor evicts + // and the handler above converts to state transfer. + // Idempotent: `maybe_request_metadata_repair` no-ops + // when caught up, already transferring, or a session is + // live, so a caught-up replica and a cold-start node + // (commit_max == commit_min == 0) both skip it. self.maybe_request_metadata_repair(consensus, header.replica) .await; } @@ -4976,6 +5007,15 @@ where let Some(journal) = planes.0.journal.as_ref() else { return; }; + // Above the two returns below, not after them: only SILENCE should + // age the stream, and an in-scope frame proves the peer is serving + // whether or not this replica still needs the op it carries. The + // ops a re-request re-serves are exactly the ones already held, so + // counting accepted frames alone rotates away from a live peer. + if let Some(session) = self.metadata_repair.borrow_mut().as_mut() { + session.idle_ticks = 0; + } + self.note_metadata_repair_progress(); let journal = journal.handle(); #[allow(clippy::cast_possible_truncation)] if journal.header(header.op as usize).is_some() { @@ -5075,10 +5115,10 @@ where // peer's served-through claim: repair frames ride a // lossy best-effort bus, so a fully-served stream can // still arrive with holes. Anything short keeps the - // session armed; while the walk is making progress the - // next chunk is pulled immediately (the window is served - // in `REPAIR_CHUNK_MAX` slices), and a stalled one is - // left to the retry timer. + // session armed; the next chunk is pulled as soon as this + // one is walked (the window is served in + // `REPAIR_CHUNK_MAX` slices), and a window still holed + // below `served_through` is left to the retry timer. let commit_min = consensus.commit_min(); let done = commit_min >= session.to_op; tracing::info!( @@ -5090,7 +5130,7 @@ where ); if done { *self.metadata_repair.borrow_mut() = None; - } else if commit_min > before { + } else if repair_chunk_walked(before, commit_min, header.op) { self.send_request_prepares( consensus.cluster(), consensus.replica(), @@ -5510,19 +5550,43 @@ where B: MessageBus, P: Pipeline, { + // `ViewChange` too: a parked view change repairs toward its merged log + // and cannot start until the window fills. Gating on `Normal` alone + // defers a dropped frame to the 500-tick escalation, and closes the one + // session that legitimately runs outside `Normal`. + let repairing_view = + consensus.view_log_is_pending() && consensus.is_primary_for_view(consensus.view()); + + // Closed at the TOP of the tick, not after an idle window: a standing + // session fences every arming site and holds the gap debounce at zero + // (`recovery_owned`), so waiting a full retry interval to notice costs + // that interval on every arm behind it. + let superseded = self.metadata_repair.borrow().is_some_and(|session| { + metadata_repair_superseded( + &session, + consensus.commit_min(), + consensus.view(), + consensus.is_normal(), + repairing_view, + ) + }); + if superseded { + tracing::debug!( + shard = self.id, + commit_min = consensus.commit_min(), + view = consensus.view(), + "metadata repair session walked or superseded; closing it" + ); + *self.metadata_repair.borrow_mut() = None; + self.note_metadata_repair_progress(); + return; + } + // Stall retry (mirrors `tick_partitions`): a lost frame must not wedge it. let repair_retry_ticks = self.repair_retry_ticks.get(); let stalled = { - // `ViewChange` too: a parked view change repairs toward its merged log - // and cannot start until the window fills. Gating on `Normal` alone - // defers a dropped frame to the 500-tick escalation. - let repairing_view = - consensus.view_log_is_pending() && consensus.is_primary_for_view(consensus.view()); let mut session = self.metadata_repair.borrow_mut(); session.as_mut().and_then(|session| { - if !consensus.is_normal() && !repairing_view { - return None; - } session.idle_ticks += 1; if session.idle_ticks < repair_retry_ticks { return None; @@ -5532,6 +5596,39 @@ where }) }; if let Some((peer, nonce, to_op)) = stalled { + // A session pins its peer and fences every arming site while it + // stands, so a peer that cannot answer wedges the plane harder than + // having no session at all -- and the gap-stopped-primary rotation + // can pick a peer that is simply down. Past the budget the session + // is dropped and re-armed one step around the ring; an ordinary lost + // frame is re-requested long before that. + if self.burn_metadata_repair_attempt() { + let next_peer = next_transfer_peer( + consensus.replica(), + peer, + consensus.replica_count(), + consensus.primary_index(consensus.view()), + ); + tracing::warn!( + shard = self.id, + peer, + next_peer, + to_op, + "metadata repair stalled past its retry budget; re-arming from another \ + replica" + ); + *self.metadata_repair.borrow_mut() = None; + self.note_metadata_repair_progress(); + if next_peer != peer { + self.maybe_request_metadata_repair(consensus, next_peer) + .await; + } + // Nobody else to name (a solo group, or a two-replica group + // whose only peer went quiet): dropping the session is still + // right, since it unfences the detector, which re-arms after its + // debounce and logs the state each interval. + return; + } // Primary-elect only. Its window starts at the merged log's commit // point, which can sit below local `commit_min` (the headers inherited // from senders behind the canonical log_view live there), so @@ -5561,6 +5658,20 @@ where consensus.group(), ) .await; + } else { + // `from_op` past `to_op` without `commit_min` reaching it: the + // primary-elect window above starts at the merged log's commit + // point, which can sit above what this replica has walked. The + // top-of-tick check closes the ordinary case; this closes the + // one it cannot see. + tracing::info!( + shard = self.id, + to_op, + peer, + "metadata repair window fully requested; closing the stalled session" + ); + *self.metadata_repair.borrow_mut() = None; + self.note_metadata_repair_progress(); } } } @@ -5811,6 +5922,7 @@ where *self.metadata_repair.borrow_mut() = Some(MetadataRepairSession { nonce, to_op: pending.op_head, + view: consensus.view(), peer, idle_ticks: 0, }); @@ -5834,14 +5946,31 @@ where } /// Start metadata tail journal-repair from `peer` when the commit walk - /// gap-stopped below the known frontier. Shared by `StartView` adoption - /// and the post-install step of a state transfer. + /// gap-stopped below the known frontier. + /// + /// Every TAIL arming site funnels through here -- `StartView` adoption, the + /// commit-heartbeat backstop, the state-transfer fallbacks, and + /// `tick_metadata`'s gap detector -- so the guards below are what make the + /// level-triggered one idempotent. The one session this does not mint is + /// the view-change repair `advance_pending_metadata_view` builds inline: it + /// repairs toward a merged log rather than the commit frontier, from a peer + /// that offered the body rather than from the primary, so none of the + /// guards below describe it. #[allow(clippy::future_not_send)] async fn maybe_request_metadata_repair

(&self, consensus: &VsrConsensus, peer: u8) where B: MessageBus, P: Pipeline, { + // Never against self. A self-addressed `RequestPrepares` cannot be + // delivered (the replica registry holds no entry for this node), and the + // send fails AFTER the session is recorded, so the session would stand + // forever: nothing advances `commit_min` to close it, the stall retry + // re-sends to the same place, and `metadata_repair.is_some()` fences + // every other arming site meanwhile. + if peer == consensus.replica() { + return; + } if consensus.is_normal() && !consensus.is_transferring() && consensus.commit_min() < consensus.commit_max() @@ -5850,9 +5979,15 @@ where let nonce = iggy_common::random_id::get_uuid(); let to_op = consensus.commit_max(); let from_op = consensus.commit_min() + 1; + // Spent here rather than at the detector, so the edge-triggered + // sites spend it too: an edge-armed repair that completes before + // the next tick would otherwise leave the count saturated and hand + // the next real gap an arm on its first tick. + self.metadata_gap_ticks.set(0); *self.metadata_repair.borrow_mut() = Some(MetadataRepairSession { nonce, to_op, + view: consensus.view(), peer, idle_ticks: 0, }); @@ -5860,6 +5995,7 @@ where shard = self.id, from_op, to_op, + peer, "metadata behind the group frontier; requesting repair" ); self.send_request_prepares( @@ -6395,6 +6531,30 @@ where budget.clamp(1, STATE_CHUNK_LEN as usize) } + /// Burn one stalled repair round; `true` once the budget is exhausted and + /// the session should be re-armed against a different peer. + /// + /// The partition twin is `IggyPartition::burn_repair_attempt`, and it lives + /// on the shard here for the same reason `metadata_transfer_attempts` does: + /// one metadata group per node. It has to outlive the SESSION either way, + /// or the rotation that mints a new one would reset the count and re-target + /// forever without ever giving up on a peer. + fn burn_metadata_repair_attempt(&self) -> bool { + let attempts = self.metadata_repair_attempts.get() + 1; + self.metadata_repair_attempts.set(attempts); + attempts > partitions::REPAIR_MAX_STALL_RETRIES + } + + /// The serving peer answered: reset the budget, so it bounds CONSECUTIVE + /// silence rather than the stalls a long healthy stream accumulates. + /// + /// Any in-scope repair frame, not only an accepted one. A re-request + /// re-serves ops this replica already holds, so charging those as silence + /// rotates away from a peer that is answering. + fn note_metadata_repair_progress(&self) { + self.metadata_repair_attempts.set(0); + } + /// Burn one retry round; `true` once the budget is exhausted. fn burn_metadata_transfer_attempt(&self) -> bool { let attempts = self.metadata_transfer_attempts.get() + 1; @@ -6870,7 +7030,7 @@ where ); let partitions = self.plane.partitions(); let repair_retry_ticks = self.repair_retry_ticks.get(); - let gap_debounce_ticks = self.partition_gap_debounce_ticks.get(); + let gap_debounce_ticks = self.repair_gap_debounce_ticks.get(); // Fan out over every group (each partition's heartbeat/retransmit timer // must advance), so the keyed single-namespace lookup the control-frame // handlers use does not apply here. The namespaces are snapshotted into @@ -7190,50 +7350,27 @@ where repairs_live += 1; } let probe = partition_gap_probe(partition); - let walk_stalled = partition_is_walk_stalled(&probe); + let walk_stalled = group_is_walk_stalled(&probe); // The RATE cap only. The concurrency cap lives in the arm fn, // which is the funnel every arming site goes through; resolved // before the debounce either way, so a refusal keeps the group // due rather than spending its arm. - let may_arm = partition_is_gap_stopped(&probe) + let may_arm = group_is_gap_stopped(&probe) && repair_arms < PARTITION_REPAIR_ARMS_PER_TICK_MAX; let mut gap_ticks = partition.gap_ticks.get(); - let verdict = drive_partition_gap_debounce( - &probe, - &mut gap_ticks, - gap_debounce_ticks, - may_arm, - ); + let verdict = + drive_group_gap_debounce(&probe, &mut gap_ticks, gap_debounce_ticks, may_arm); partition.gap_ticks.set(gap_ticks); let arm_peer = match verdict { GapArm::NotDue | GapArm::Deferred => None, GapArm::Arm => { let consensus = partition.consensus(); - let self_id = consensus.replica(); - let primary = consensus.primary_index(consensus.view()); - // A gap-stopped PRIMARY cannot ask itself, and leaving - // it to warn wedged the group: no edge-triggered site - // re-drives a primary's own hole, and the next op to - // commit walks `advance_commit_min` into its sequential - // assert. Any replica in `Normal` or `ViewChange` serves - // `RequestPrepares`, and a primary's window is its - // COMMITTED prefix (the suffix widening needs a pending - // view log, which a settled primary has none of), so a - // peer holding those ops holds them identically. - // - // The pick is positional, not liveness-aware: a dead - // choice leaves the session re-requesting on the stall - // timer, which is where a repair abandon budget (what - // `burn_transfer_attempt` gives transfers) would rotate - // it. Still strictly better than the warn this replaced, - // which recovered nothing at all. - let peer = if primary == self_id { - next_transfer_peer(self_id, self_id, consensus.replica_count(), primary) - } else { - primary - }; - if peer == self_id { - // Solo group: the rotation had nobody to return. + let peer = gap_repair_peer( + consensus.replica(), + consensus.replica_count(), + consensus.primary_index(consensus.view()), + ); + if peer.is_none() { // Restart the debounce so this repeats at its // interval rather than every tick. partition.gap_ticks.set(0); @@ -7245,10 +7382,8 @@ where "partition is gap-stopped below its own commit frontier with no \ peer to repair from" ); - None - } else { - Some(peer) } + peer } }; (walk_stalled, arm_peer) @@ -7271,7 +7406,7 @@ where // rejoin leaves every group on the shard walk-stalled in the same // tick, and each walk reaches a segment flush. Undebounced, though // -- the predicate guarantees the walk finds at least the next op, - // so it cannot spin: `partition_is_walk_stalled` reads residency off + // so it cannot spin: `group_is_walk_stalled` reads residency off // `op_to_storage_offset` while the walk reads `headers`, and those // two are written and cleared together (see `Journal::holds_op`), so // a group the predicate admits has an op for the walk to take. @@ -9218,7 +9353,114 @@ where } } + /// Drop the WAL entry at `stuck_op` and the suffix above it, so repair can + /// refill a header whose body the commit walk cannot read. + /// + /// Nothing else clears it: `on_repair_prepare` returns early for an op + /// whose header is resident, and the append under it is refused anyway. + /// `stuck_op` is at `commit_min + 1` under `commit_max`, so a quorum holds + /// it and repair can serve it back. + /// + /// SERIALIZATION: same argument as `reconcile_metadata_view_divergence`, + /// which is the other shard-side `truncate_from` caller. This runs on the + /// pump between frames, so no append is in flight for these ops. #[allow(clippy::future_not_send)] + async fn drop_unwalkable_metadata_entry

( + &self, + consensus: &VsrConsensus, + journal: &MJ, + stuck_op: u64, + ) where + B: MessageBus, + P: Pipeline, + MJ: JournalHandle, + ::Target: + Journal, Header = PrepareHeader>, + { + match journal.handle().truncate_from(stuck_op).await { + Ok(removed) => { + // The snapshot's `(op, commit)` tag does not move when entries + // are removed under it, so the next `DoViewChange` would + // otherwise advertise headers this replica can no longer serve. + consensus.invalidate_local_dvc_suffix(); + tracing::warn!( + shard = self.id, + stuck_op, + removed, + "metadata commit walk found a resident header with no body at op {stuck_op}; \ + dropped {removed} entries from it so repair can refill the range" + ); + } + Err(error) => { + tracing::error!( + shard = self.id, + stuck_op, + %error, + "could not drop the unwalkable entry at op {stuck_op}; journal repair skips \ + ops it already holds a header for, so this replica will not walk past \ + it until it is restarted" + ); + } + } + } + + /// Read the gap probe off the metadata plane; [`partition_gap_probe`]'s + /// twin. A shard method because the recovery slots live here, on the shard, + /// not on the plane. + fn metadata_gap_probe

(&self, consensus: &VsrConsensus, journal: &MJ) -> GapProbe + where + B: MessageBus, + P: Pipeline, + MJ: JournalHandle, + ::Target: + Journal, Header = PrepareHeader>, + { + let commit_min = consensus.commit_min(); + let commit_max = consensus.commit_max(); + let normal = consensus.is_normal(); + let transferring = consensus.is_transferring(); + let recovery_owned = + self.metadata_transfer.borrow().is_some() || self.metadata_repair.borrow().is_some(); + // Residency last, and only once the guards both predicates share hold, + // as in `partition_gap_probe`: a caught-up plane would otherwise pay a + // journal lookup whose answer both predicates discard. + // + // Safe against the snapshot floor: a checkpoint drains only to + // `commit_min`, so `commit_min + 1` never sits below it and a `None` is + // a real hole. + // + // The header ring is only half of what the walk needs. `commit_journal` + // reads the BODY through `entry()`, which answers `None` for an op the + // ring names but the WAL cannot produce, and then breaks without moving + // `commit_min`. Reading the body here instead is not an option (it is an + // async WAL read, per tick, on the walk's fast path), so the walk + // reports the op it stopped on and this treats that op as absent -- + // which it is, for every purpose this probe serves. Without it the two + // disagree forever: the walk cannot move, the probe keeps calling the + // group walk-stalled, the debounce keeps resetting, and repair never + // arms. + // + // Self-clearing: any path that advances `commit_min` past the stuck op + // leaves `stuck_op != commit_min + 1`, so nothing has to retract it. + let next_op = commit_min.saturating_add(1); + #[allow(clippy::cast_possible_truncation)] + let next_op_resident = normal + && !transferring + && commit_min < commit_max + && self.metadata_walk_stuck_op.get() != next_op + && journal.handle().header(next_op as usize).is_some(); + GapProbe { + normal, + transferring, + recovery_owned, + commit_min, + commit_max, + next_op_resident, + missing_suffix: false, + } + } + + #[allow(clippy::future_not_send, clippy::too_many_lines)] pub async fn tick_metadata(&self) where B: MessageBus, @@ -9283,6 +9525,106 @@ where self.advance_pending_metadata_view().await; self.expire_idle_state_transfer_offers(); + // Level-triggered gap detector, the metadata twin of the one in + // `tick_partitions`, and starvable in exactly the same way: + // `replicate_preflight` advances `commit_max` before the gap check + // drops the prepare, so under sustained traffic the heartbeat lands as + // `Accepted` and the `Advanced`-gated arm in `on_commit` never fires. + // + // Placed before the transfer-stall block below: that block's exhausted + // branch returns early, so a detector after it would be skipped on the + // tick that abandons a transfer. + if let Some(journal) = metadata.journal.as_ref() { + let gap_drops = metadata.take_prepare_gap_drops(); + if gap_drops > 0 { + self.metrics.record_metadata_prepare_gap_drops(gap_drops); + } + let probe = self.metadata_gap_probe(consensus, journal); + let mut gap_ticks = self.metadata_gap_ticks.get(); + // Always budgeted: one metadata group per node, so there is no + // correlated fan-out for a per-tick rate cap to spread. + let verdict = drive_group_gap_debounce( + &probe, + &mut gap_ticks, + self.repair_gap_debounce_ticks.get(), + true, + ); + self.metadata_gap_ticks.set(gap_ticks); + if verdict == GapArm::Arm { + match gap_repair_peer( + consensus.replica(), + consensus.replica_count(), + consensus.primary_index(consensus.view()), + ) { + None => { + // Restart the debounce so this repeats at its interval, + // not every tick. + self.metadata_gap_ticks.set(0); + tracing::warn!( + shard = self.id, + commit_min = probe.commit_min, + commit_max = probe.commit_max, + "metadata is gap-stopped below its own commit frontier with no peer \ + to repair from" + ); + } + // Always repair, never classify the gap up front: a window + // below the peer's retention floor is answered + // `RangeEvicted`, and `on_repair_range_reply` converts that + // to a state transfer. The floor is only ever learned + // through that refusal. The arm logs the window it settled + // on, so nothing is logged here. + Some(peer) => self.maybe_request_metadata_repair(consensus, peer).await, + } + } + // Undebounced, like the partition walk arm, and unrated: there is + // one group to walk here rather than a shard-wide fan-out, so + // nothing needs spreading across ticks. How FAR one walk goes is + // still capped, inside `commit_journal` itself. + // + // Both roles, like the partition arm. `resume_stranded_commits` + // above re-drives a primary's PIPELINE, and `(commit_min, + // commit_max]` is journal-only once it has run, so an inherited + // prefix or the tail of a capped walk has no other re-driver here + // and pins `commit_min` until the next op to commit trips + // `advance_commit_min`'s sequential assert. + // + // Not gated on `recovery_owned` (repaired prepares are journaled + // without being walked, so gating parks the walk for the whole + // session); `group_is_walk_stalled` itself refuses mid-transfer, + // where a walk past the incoming `snapshot_seq` would break the + // install. + if group_is_walk_stalled(&probe) { + // Debug, not info: a repair stream journals its prepares without + // walking them, so this is the steady state for the whole + // duration of a rejoin and would be one line per tick. + tracing::debug!( + shard = self.id, + commit_min = probe.commit_min, + commit_max = probe.commit_max, + "metadata commit walk parked over resident committed ops; resuming" + ); + metadata.commit_journal().await; + // A walk that moved nothing found the header and not the body. + // Recording the op stops the detector calling this a parked + // walk, but arming repair alone cannot refill it: the ingest + // skips an op whose header is resident and `append` refuses the + // slot under it, so the header has to go first. + let walked = consensus.commit_min(); + if walked == probe.commit_min { + let stuck_op = walked.saturating_add(1); + // Once per op: a failed truncation leaves the header where + // it is, and retrying every tick only repeats the error. + if self.metadata_walk_stuck_op.replace(stuck_op) != stuck_op { + self.drop_unwalkable_metadata_entry(consensus, journal, stuck_op) + .await; + } + } else { + self.metadata_walk_stuck_op.set(0); + } + } + } + // Stall retry for an in-flight state transfer: descriptor or chunk // frames are fire-and-forget, so a lost one must not wedge the // session (and the boot flow behind it) forever. @@ -9318,9 +9660,17 @@ where consensus.set_state_transfer_stage(consensus::StateTransferStage::Idle); } metadata.commit_journal().await; - let current_primary = consensus.primary_index(consensus.view()); - self.maybe_request_metadata_repair(consensus, current_primary) - .await; + // Rotated, not `primary_index` raw: this replica can BE the + // primary here (a leading replica that transferred to catch up + // on a checkpoint it lacked), and the arm refuses self. + if let Some(next_peer) = gap_repair_peer( + consensus.replica(), + consensus.replica_count(), + consensus.primary_index(consensus.view()), + ) { + self.maybe_request_metadata_repair(consensus, next_peer) + .await; + } return; } tracing::info!( @@ -9681,12 +10031,15 @@ const PARTITION_WALKS_PER_TICK_MAX: usize = 16; /// Public because it bounds what that operator knob can do: gap recovery starts /// after `max(repair_gap_debounce_interval, this)`, which the `[cluster]` /// documentation states. -pub const PARTITION_GAP_DEBOUNCE_TICKS_MIN: u32 = 50; +pub const REPAIR_GAP_DEBOUNCE_TICKS_MIN: u32 = 50; -/// What the tick sweep reads off one partition to decide whether it is +/// What a tick driver reads off one consensus group to decide whether it is /// gap-stopped. Split out so the guards, the debounce and the per-tick cap are /// testable without a shard, a bus, or a journal. /// +/// Both planes fill it: `partition_gap_probe` off a live partition, and +/// `IggyShard::metadata_gap_probe` off the metadata plane's consensus and WAL. +/// /// The flags are independent readings of one instant, not states of one /// machine, and the exhaustive predicate test below enumerates them as such, so /// the lint's two-variant enums would only rename `true` and `false`. @@ -9696,8 +10049,10 @@ struct GapProbe { normal: bool, transferring: bool, /// Whether a repair session, a transfer, or a scheduled transfer re-arm - /// already owns this partition's recovery. Arming a second one would race - /// it, or defeat the re-arm's backoff as `arm_partition_transfer` documents. + /// already owns this group's recovery. Arming a second one would race it, + /// or defeat the re-arm's backoff as `arm_partition_transfer` documents. + /// The re-arm shape is the partition plane's alone; metadata has no + /// re-arm state, so its probe reads the other two. recovery_owned: bool, commit_min: u64, commit_max: u64, @@ -9713,6 +10068,12 @@ struct GapProbe { /// below the frontier: the group cannot gather quorum for that suffix until /// the bodies land, and the only other site that notices is the single /// `on_start_view` edge that adopted them. See [`partition_missing_suffix`]. + /// + /// Always `false` on a metadata probe: the shape it names is read off the + /// partition's own journal window, and the metadata plane's equivalent is + /// still only noticed at the `advance_pending_metadata_view` edge. So the + /// metadata detector covers the hole BELOW the frontier and nothing above + /// it. missing_suffix: bool, } @@ -9722,8 +10083,8 @@ struct GapProbe { /// The journal-hole half is not redundant: a follower advances `commit_max` /// from every prepare header in `replicate_preflight`, so `commit_min < /// commit_max` is transiently true on every healthy pipelined tick and a bare -/// lag test would arm repair against ordinary produce. -const fn partition_is_gap_stopped(probe: &GapProbe) -> bool { +/// lag test would arm repair against ordinary traffic. +const fn group_is_gap_stopped(probe: &GapProbe) -> bool { if !probe.normal || probe.transferring || probe.recovery_owned { return false; } @@ -9745,18 +10106,18 @@ const fn partition_is_gap_stopped(probe: &GapProbe) -> bool { /// commit is `Accepted`, and an idle group offers no other edge). /// /// The two split on `next_op_resident` while a lag stands, and -/// [`partition_is_gap_stopped`] defers to that split even for a missing suffix, +/// [`group_is_gap_stopped`] defers to that split even for a missing suffix, /// so they cannot both hold. Both are false whenever a shared guard fails. Not /// gated on `recovery_owned`: repair fetches bodies without walking them, so /// gating parks the walk all session. -const fn partition_is_walk_stalled(probe: &GapProbe) -> bool { +const fn group_is_walk_stalled(probe: &GapProbe) -> bool { probe.normal && !probe.transferring && probe.commit_min < probe.commit_max && probe.next_op_resident } -/// What the debounce says about one partition on one sweep. +/// What the debounce says about one group on one tick. #[derive(Debug, Clone, Copy, PartialEq, Eq)] enum GapArm { /// Not gap-stopped, or gap-stopped for less than the debounce. @@ -9765,39 +10126,46 @@ enum GapArm { /// so the group is due again on the next pass rather than serving a fresh /// interval. It moves no cursor: the sweep resumes where the WALK budget /// ran out, and arms drain their own queue as sessions open. + /// + /// Partition-plane only. The metadata driver holds one group per node, so + /// it always passes a budget and never sees this. Deferred, /// Open a repair session now. Arm, } -/// Count one sweep tick against `gap_ticks` and answer whether this partition -/// may arm repair now. +/// Count one tick against `gap_ticks` and answer whether this group may arm +/// repair now. /// -/// Level-triggered, because every edge-triggered arming site is starvable: the -/// commit-heartbeat backstop fires only on `CommitOutcome::Advanced`, and under -/// sustained produce the prepares consume the advance in preflight before the -/// gap check drops them, so the heartbeat lands as `Accepted` and the gap wedges -/// until an unrelated view change. +/// Level-triggered, because every edge-triggered arming site is starvable, on +/// both planes: the commit-heartbeat backstop fires only on +/// `CommitOutcome::Advanced`, and under sustained traffic the prepares consume +/// the advance in preflight before the gap check drops them, so the heartbeat +/// lands as `Accepted` and the gap wedges until an unrelated view change. /// -/// `budget_available` is the sweep's per-tick arm rate; the live-session -/// ceiling is applied by `maybe_request_partition_repair`, which every arming -/// site funnels through. A refused arm keeps its debounce satisfied rather than -/// starting over, so the group arms on the next pass with a slot free. Spending it is `maybe_request_partition_repair`'s job, which resets -/// `gap_ticks` for EVERY arming site, not just this one: an edge-armed repair -/// that completes before the next sweep would otherwise leave the count -/// saturated and hand the next gap an arm on its first tick. -const fn drive_partition_gap_debounce( +/// `budget_available` is the partition sweep's per-tick arm rate; the +/// live-session ceiling is applied by `maybe_request_partition_repair`, which +/// every partition arming site funnels through. A refused arm keeps its +/// debounce satisfied rather than starting over, so the group arms on the next +/// pass with a slot free. +/// +/// Spending the count is the arm function's job, not this one's, and it resets +/// `gap_ticks` for EVERY arming site rather than only the tick: an edge-armed +/// repair that completes before the next tick would otherwise leave the count +/// saturated and hand the next gap an arm on its first tick. Metadata's twin of +/// that reset lives in `maybe_request_metadata_repair`. +const fn drive_group_gap_debounce( probe: &GapProbe, gap_ticks: &mut u32, debounce_ticks: u32, budget_available: bool, ) -> GapArm { - if !partition_is_gap_stopped(probe) { + if !group_is_gap_stopped(probe) { *gap_ticks = 0; return GapArm::NotDue; } - let debounce_ticks = if debounce_ticks < PARTITION_GAP_DEBOUNCE_TICKS_MIN { - PARTITION_GAP_DEBOUNCE_TICKS_MIN + let debounce_ticks = if debounce_ticks < REPAIR_GAP_DEBOUNCE_TICKS_MIN { + REPAIR_GAP_DEBOUNCE_TICKS_MIN } else { debounce_ticks }; @@ -9812,6 +10180,61 @@ const fn drive_partition_gap_debounce( } } +/// The peer a gap-stopped replica asks for repair, or `None` when there is +/// nobody to ask. +/// +/// The primary, except when this replica IS the primary: no site re-drives a +/// settled primary's own hole, so leaving it to warn wedges the group, and the +/// next op to commit walks `advance_commit_min` into its sequential assert. Any +/// replica in `Normal` or `ViewChange` serves `RequestPrepares`, and a +/// gap-stopped replica's window is its COMMITTED prefix, which every peer that +/// holds those ops holds identically. +/// +/// Positional, not liveness-aware. A dead pick is corrected by the stall +/// budget on either plane, which drops the session and rotates one step further +/// around the ring rather than re-requesting from it forever. +/// +/// Shared by both planes so the rule cannot drift: the partition sweep and +/// `tick_metadata` arm off the same predicate and owe the same answer. +const fn gap_repair_peer(self_id: u8, replica_count: u8, primary: u8) -> Option { + let peer = if primary == self_id { + next_transfer_peer(self_id, self_id, replica_count, primary) + } else { + primary + }; + // A solo group (or a ring with nobody else live to name) rotates back to + // self, which no session can be opened against. + if peer == self_id { None } else { Some(peer) } +} + +/// Whether a standing metadata repair session should be closed at the top of +/// the tick: its window is walked, the view that decided that window has +/// moved, or this replica has left the status the session belongs to. +/// +/// `repairing_view` is the primary-elect repairing toward its merged log, the +/// one session that runs outside `Normal`. Pinned by +/// `metadata_repair_session_tests`. +const fn metadata_repair_superseded( + session: &MetadataRepairSession, + commit_min: u64, + view: u32, + normal: bool, + repairing_view: bool, +) -> bool { + commit_min >= session.to_op || session.view != view || !(normal || repairing_view) +} + +/// Whether a walked `RepairDone` should pull the next chunk of the window. +/// +/// `served_through` is the terminator's own op. Chunk progress, not this +/// walk's: `tick_metadata` walks the same journal, so it can consume a chunk +/// between the chunk's last prepare and its terminator, and requiring +/// `commit_min` to move HERE idles the session a full retry interval on every +/// such landing. +const fn repair_chunk_walked(before: u64, commit_min: u64, served_through: u64) -> bool { + commit_min > before || commit_min >= served_through +} + /// Rotate a sweep's namespace snapshot so it resumes at `cursor`. /// /// The per-tick caps are what make this necessary: the snapshot is in ascending @@ -11421,15 +11844,19 @@ mod sweep_scheduler_tests { #[cfg(test)] mod gap_detector_tests { - //! The level-triggered repair arm the partition tick sweep runs. + //! The level-triggered repair arm the partition and metadata tick drivers + //! share. //! //! Its whole reason to exist is that the edge-triggered arming sites are //! starvable, so the guards it shares with them and the debounce that keeps - //! it off healthy traffic are the parts worth pinning. + //! it off healthy traffic are the parts worth pinning. Probes are built + //! here by hand: what the two planes read off their own state is + //! `partition_gap_probe`'s and `metadata_gap_probe`'s business, and the + //! simulator's driver suites cover those end to end. use super::{ - GapArm, GapProbe, PARTITION_GAP_DEBOUNCE_TICKS_MIN, drive_partition_gap_debounce, - partition_is_gap_stopped, partition_is_walk_stalled, + GapArm, GapProbe, REPAIR_GAP_DEBOUNCE_TICKS_MIN, drive_group_gap_debounce, + group_is_gap_stopped, group_is_walk_stalled, }; const DEBOUNCE: u32 = 100; @@ -11474,8 +11901,8 @@ mod gap_detector_tests { // commit_max is transiently true on any pipelined tick; without the // journal-hole test the driver would request repair against ordinary // produce, on every partition, forever. - assert!(!partition_is_gap_stopped(&walk_stalled())); - assert!(partition_is_gap_stopped(&gap_stopped())); + assert!(!group_is_gap_stopped(&walk_stalled())); + assert!(group_is_gap_stopped(&gap_stopped())); } #[test] @@ -11484,7 +11911,7 @@ mod gap_detector_tests { commit_min: 10, ..gap_stopped() }; - assert!(!partition_is_gap_stopped(&caught_up)); + assert!(!group_is_gap_stopped(&caught_up)); } #[test] @@ -11493,9 +11920,9 @@ mod gap_detector_tests { // the head, so there is no lag to see, and the only other site that // notices is the single `on_start_view` edge that adopted the headers. // Left out, that class hangs until an unrelated view change. - assert!(partition_is_gap_stopped(&missing_suffix())); + assert!(group_is_gap_stopped(&missing_suffix())); assert!( - !partition_is_gap_stopped(&GapProbe { + !group_is_gap_stopped(&GapProbe { missing_suffix: false, ..missing_suffix() }), @@ -11508,11 +11935,11 @@ mod gap_detector_tests { // A view change owns the log while it runs, and `maybe_request_partition_repair` // refuses outside Normal anyway; arming here would only burn a nonce. for probe in [gap_stopped(), missing_suffix()] { - assert!(!partition_is_gap_stopped(&GapProbe { + assert!(!group_is_gap_stopped(&GapProbe { normal: false, ..probe })); - assert!(!partition_is_gap_stopped(&GapProbe { + assert!(!group_is_gap_stopped(&GapProbe { transferring: true, ..probe })); @@ -11524,7 +11951,7 @@ mod gap_detector_tests { // A session, a transfer, or a scheduled transfer re-arm all own the // recovery; a second one would race it or defeat the re-arm's backoff. for probe in [gap_stopped(), missing_suffix()] { - assert!(!partition_is_gap_stopped(&GapProbe { + assert!(!group_is_gap_stopped(&GapProbe { recovery_owned: true, ..probe })); @@ -11537,13 +11964,13 @@ mod gap_detector_tests { let mut gap_ticks = 0; for tick in 1..DEBOUNCE { assert_eq!( - drive_partition_gap_debounce(&probe, &mut gap_ticks, DEBOUNCE, true), + drive_group_gap_debounce(&probe, &mut gap_ticks, DEBOUNCE, true), GapArm::NotDue, "armed at tick {tick}, before the debounce elapsed" ); } assert_eq!( - drive_partition_gap_debounce(&probe, &mut gap_ticks, DEBOUNCE, true), + drive_group_gap_debounce(&probe, &mut gap_ticks, DEBOUNCE, true), GapArm::Arm ); } @@ -11557,17 +11984,17 @@ mod gap_detector_tests { }; let mut gap_ticks = 0; for _ in 0..DEBOUNCE - 1 { - drive_partition_gap_debounce(&stopped, &mut gap_ticks, DEBOUNCE, true); + drive_group_gap_debounce(&stopped, &mut gap_ticks, DEBOUNCE, true); } assert_eq!(gap_ticks, DEBOUNCE - 1); assert_eq!( - drive_partition_gap_debounce(&walkable, &mut gap_ticks, DEBOUNCE, true), + drive_group_gap_debounce(&walkable, &mut gap_ticks, DEBOUNCE, true), GapArm::NotDue ); assert_eq!(gap_ticks, 0, "progress must restart the debounce"); assert_eq!( - drive_partition_gap_debounce(&stopped, &mut gap_ticks, DEBOUNCE, true), + drive_group_gap_debounce(&stopped, &mut gap_ticks, DEBOUNCE, true), GapArm::NotDue, "a fresh gap must serve its own debounce, not inherit the old count" ); @@ -11575,9 +12002,9 @@ mod gap_detector_tests { #[test] fn given_a_follower_with_resident_committed_ops_when_probed_should_be_walk_stalled() { - assert!(partition_is_walk_stalled(&walk_stalled())); + assert!(group_is_walk_stalled(&walk_stalled())); assert!( - !partition_is_walk_stalled(&gap_stopped()), + !group_is_walk_stalled(&gap_stopped()), "a missing next op is repair's job; a walk over it would stop dead" ); } @@ -11588,7 +12015,7 @@ mod gap_detector_tests { commit_min: 10, ..walk_stalled() }; - assert!(!partition_is_walk_stalled(&caught_up)); + assert!(!group_is_walk_stalled(&caught_up)); } #[test] @@ -11597,7 +12024,7 @@ mod gap_detector_tests { normal: false, ..walk_stalled() }; - assert!(!partition_is_walk_stalled(&electing)); + assert!(!group_is_walk_stalled(&electing)); // Same gate as the on-commit arm: a walk during a transfer can advance // commit_min past the incoming frontier. @@ -11605,7 +12032,7 @@ mod gap_detector_tests { transferring: true, ..walk_stalled() }; - assert!(!partition_is_walk_stalled(&installing)); + assert!(!group_is_walk_stalled(&installing)); } #[test] @@ -11613,7 +12040,7 @@ mod gap_detector_tests { // Deliberate: `apply_repaired_prepare` journals without walking, so a // gated walk would sit parked for the whole session while the resident // prefix is already applicable. - assert!(partition_is_walk_stalled(&GapProbe { + assert!(group_is_walk_stalled(&GapProbe { recovery_owned: true, ..walk_stalled() })); @@ -11639,8 +12066,8 @@ mod gap_detector_tests { missing_suffix, }; assert!( - !(partition_is_gap_stopped(&probe) - && partition_is_walk_stalled(&probe)), + !(group_is_gap_stopped(&probe) + && group_is_walk_stalled(&probe)), "both predicates claim {probe:?}" ); } @@ -11661,13 +12088,13 @@ mod gap_detector_tests { missing_suffix: true, ..walk_stalled() }; - assert!(partition_is_walk_stalled(&probe)); + assert!(group_is_walk_stalled(&probe)); assert!( - !partition_is_gap_stopped(&probe), + !group_is_gap_stopped(&probe), "a walkable lag must win the tick; the suffix arm waits for it to close" ); assert!( - partition_is_gap_stopped(&GapProbe { + group_is_gap_stopped(&GapProbe { commit_min: probe.commit_max, ..probe }), @@ -11680,7 +12107,7 @@ mod gap_detector_tests { let probe = gap_stopped(); let mut gap_ticks = DEBOUNCE; assert_eq!( - drive_partition_gap_debounce(&probe, &mut gap_ticks, DEBOUNCE, false), + drive_group_gap_debounce(&probe, &mut gap_ticks, DEBOUNCE, false), GapArm::Deferred, "a spent budget must refuse the arm" ); @@ -11690,7 +12117,7 @@ mod gap_detector_tests { arm a whole interval out per contended tick" ); assert_eq!( - drive_partition_gap_debounce(&probe, &mut gap_ticks, DEBOUNCE, true), + drive_group_gap_debounce(&probe, &mut gap_ticks, DEBOUNCE, true), GapArm::Arm, "the same group arms on the next pass with a slot free" ); @@ -11705,16 +12132,173 @@ mod gap_detector_tests { // reordered prepare. let probe = gap_stopped(); let mut gap_ticks = 0; - for tick in 1..PARTITION_GAP_DEBOUNCE_TICKS_MIN { + for tick in 1..REPAIR_GAP_DEBOUNCE_TICKS_MIN { assert_eq!( - drive_partition_gap_debounce(&probe, &mut gap_ticks, 1, true), + drive_group_gap_debounce(&probe, &mut gap_ticks, 1, true), GapArm::NotDue, "a 1-tick debounce armed at tick {tick}, under the floor" ); } assert_eq!( - drive_partition_gap_debounce(&probe, &mut gap_ticks, 1, true), + drive_group_gap_debounce(&probe, &mut gap_ticks, 1, true), GapArm::Arm ); } } + +#[cfg(test)] +mod metadata_repair_session_tests { + //! The three rules a standing metadata repair session lives by: who it is + //! opened against, when it is closed, and when a walked terminator pulls + //! the next chunk of its window. + //! + //! All three wedge the plane rather than failing loudly. A session fences + //! every other arming site and holds the gap debounce at zero while it + //! stands, so one opened against nobody, or kept past the view that decided + //! its window, or that stops pulling chunks, pins the commit frontier with + //! nothing else able to arm. + + use super::{ + MetadataRepairSession, gap_repair_peer, metadata_repair_superseded, next_transfer_peer, + repair_chunk_walked, + }; + + /// Armed at view 3, against a window ending at op 20. + const fn session() -> MetadataRepairSession { + MetadataRepairSession { + nonce: 7, + to_op: 20, + view: 3, + peer: 0, + idle_ticks: 0, + } + } + + #[test] + fn given_a_gap_stopped_backup_when_picking_a_peer_should_ask_the_primary() { + assert_eq!(gap_repair_peer(2, 3, 0), Some(0)); + assert_eq!(gap_repair_peer(1, 5, 3), Some(3)); + } + + #[test] + fn given_a_gap_stopped_primary_when_picking_a_peer_should_never_ask_itself() { + // The case `maybe_request_metadata_repair`'s self-guard exists for: no + // other site re-drives a settled primary's own hole, and a + // self-addressed request fails to send AFTER the session is recorded. + for replica_count in 2..=7u8 { + for primary in 0..replica_count { + let peer = gap_repair_peer(primary, replica_count, primary); + assert_ne!(peer, Some(primary), "count {replica_count}"); + assert!(peer.is_some(), "count {replica_count}"); + } + } + } + + #[test] + fn given_a_solo_group_when_picking_a_peer_should_answer_nobody() { + assert_eq!(gap_repair_peer(0, 1, 0), None); + } + + #[test] + fn given_a_silent_peer_when_the_stall_budget_is_spent_should_rotate_off_it() { + // Re-arming against the peer that just went quiet spends another whole + // budget on it, and the ring is the only thing that names anyone else. + for replica_count in 3..=7u8 { + for primary in 0..replica_count { + let self_id = (primary + 1) % replica_count; + let failed = gap_repair_peer(self_id, replica_count, primary).expect("a peer"); + let next = next_transfer_peer(self_id, failed, replica_count, primary); + assert_ne!(next, failed, "count {replica_count}, primary {primary}"); + assert_ne!(next, self_id, "count {replica_count}, primary {primary}"); + } + } + } + + #[test] + fn given_two_replicas_when_the_stall_budget_is_spent_should_name_the_same_peer_back() { + // Which is how the caller reads "nobody else to ask" and drops the + // session instead of re-arming it. + assert_eq!(next_transfer_peer(1, 0, 2, 0), 0); + } + + #[test] + fn given_a_session_whose_window_is_walked_when_checked_should_be_superseded() { + let session = session(); + assert!(metadata_repair_superseded( + &session, + session.to_op, + session.view, + true, + false + )); + assert!(!metadata_repair_superseded( + &session, + session.to_op - 1, + session.view, + true, + false + )); + } + + #[test] + fn given_a_session_armed_in_an_earlier_view_when_checked_should_be_superseded() { + let session = session(); + assert!(metadata_repair_superseded( + &session, + 0, + session.view + 1, + true, + false + )); + } + + #[test] + fn given_a_replica_that_left_normal_when_checked_should_be_superseded() { + let session = session(); + assert!(metadata_repair_superseded( + &session, + 0, + session.view, + false, + false + )); + } + + #[test] + fn given_a_primary_elect_repairing_its_merged_log_when_checked_should_stand() { + // The one session that runs outside `Normal`; + // `advance_pending_metadata_view` cannot start the view until its + // window fills. + let session = session(); + assert!(!metadata_repair_superseded( + &session, + 0, + session.view, + false, + true + )); + assert!( + metadata_repair_superseded(&session, 0, session.view + 1, false, true), + "not even the primary-elect's session survives the next view" + ); + } + + #[test] + fn given_a_chunk_this_walk_moved_when_checked_should_pull_the_next_chunk() { + assert!(repair_chunk_walked(5, 8, 12)); + } + + #[test] + fn given_a_chunk_the_tick_already_walked_when_checked_should_pull_the_next_chunk() { + // `tick_metadata` walks the same journal, so the terminator can arrive + // with nothing left for its own walk to move. + assert!(repair_chunk_walked(12, 12, 12)); + } + + #[test] + fn given_a_window_still_holed_below_the_terminator_when_checked_should_wait_for_the_retry() { + // A frame was lost inside the served chunk: re-requesting now would + // race the retry timer for the same window. + assert!(!repair_chunk_walked(5, 5, 12)); + } +} diff --git a/core/shard/src/metrics.rs b/core/shard/src/metrics.rs index 3e4b1c14fa..2d9535c25e 100644 --- a/core/shard/src/metrics.rs +++ b/core/shard/src/metrics.rs @@ -212,6 +212,7 @@ pub struct ShardMetrics { partition_requests_denied_transient_total: Counter, partition_repair_serves_deferred_purge_total: Counter, partition_prepare_gap_drops_total: Counter, + metadata_prepare_gap_drops_total: Counter, metadata_read_frontier_refusals_total: Counter, client_requests_denied_queue_full_total: Counter, } @@ -239,6 +240,7 @@ impl ShardMetrics { partition_requests_denied_transient_total: Counter::default(), partition_repair_serves_deferred_purge_total: Counter::default(), partition_prepare_gap_drops_total: Counter::default(), + metadata_prepare_gap_drops_total: Counter::default(), metadata_read_frontier_refusals_total: Counter::default(), client_requests_denied_queue_full_total: Counter::default(), } @@ -473,10 +475,9 @@ impl ShardMetrics { /// other partition counter in this file is shard-scoped for the same /// reason. The per-group detail is in the arm's log line. /// - /// The metadata plane's own gap drop is NOT counted here, and has no - /// counter of its own: its repair is armed by the same edge-triggered sites - /// this plane's sweep exists to backstop, so that plane is starvable in the - /// same way and is simply not instrumented for it yet. + /// The metadata plane's own gap drop is NOT counted here: it has its own + /// counter, and its own level-triggered driver in `tick_metadata`. See + /// [`Self::record_metadata_prepare_gap_drops`]. pub fn record_partition_prepare_gap_drops(&self, drops: u64) { self.partition_prepare_gap_drops_total.inc_by(drops); } @@ -488,6 +489,31 @@ impl ShardMetrics { self.partition_prepare_gap_drops_total.get() } + /// Add the prepares the metadata backup gap check destroyed since the last + /// tick, drained from `IggyMetadata::take_prepare_gap_drops`. + /// + /// The sibling of [`Self::record_partition_prepare_gap_drops`], and it + /// carries every caveat that one does: it counts the prepares that ARRIVED + /// after a hole rather than the holes, so a nonzero value proves the + /// metadata repair driver has work and a zero one proves nothing. There is + /// one metadata group per node, so unlike the partition counter it needs no + /// argument about namespace cardinality. + /// + /// Deliberately NOT a `frame_drops_total` reason, for the same reason: that + /// family means the bus or the router shed a frame and the simulator + /// asserts it stays at zero on runs with no injected loss, while a gap drop + /// is a protocol-ordering drop the tick repairs. + pub fn record_metadata_prepare_gap_drops(&self, drops: u64) { + self.metadata_prepare_gap_drops_total.inc_by(drops); + } + + /// Snapshot of `metadata_prepare_gap_drops_total`. Test/simulator accessor. + #[cfg(any(test, feature = "simulator"))] + #[must_use] + pub fn metadata_prepare_gap_drops_value(&self) -> u64 { + self.metadata_prepare_gap_drops_total.get() + } + /// Snapshot of `partition_frames_rejected_stale_total`. Test/simulator /// accessor, readable from any crate under those cfgs so the crates that /// drive the reconciler can assert a reject did not happen. @@ -584,6 +610,11 @@ impl ShardMetrics { "replicated prepares dropped out of order by a backup's gap check", self.partition_prepare_gap_drops_total.clone(), ); + registry.register( + "metadata_prepare_gap_drops", + "replicated metadata prepares dropped out of order by a backup's gap check", + self.metadata_prepare_gap_drops_total.clone(), + ); registry.register( "metadata_read_frontier_refusals", "metadata reads refused because this node never applied the caller's committed op", diff --git a/core/simulator/src/deps.rs b/core/simulator/src/deps.rs index 39b12c8454..ace5328595 100644 --- a/core/simulator/src/deps.rs +++ b/core/simulator/src/deps.rs @@ -435,6 +435,20 @@ impl SimJournal { headers.remove(&op).is_some() } + /// Forget one op's BODY, leaving its header resident: what a commit walk + /// reading `entry()` cannot get past on its own, since the header the + /// gap check and the repair ingest both read is still there. + /// + /// Tests only, and the sibling of [`Self::forget_op`]: no drop pattern on + /// a link produces this, because the header and the body land in the same + /// append. + pub fn forget_body(&self, op: u64) -> bool { + #[cfg(debug_assertions)] + let _guard = JournalAccessGuard::new(&self.accessing); + let offsets = unsafe { &mut *self.offsets.get() }; + offsets.remove(&op).is_some() + } + /// The committed watermark to restore after a restart, mirroring /// `metadata::recover`. On a solo cluster every appended op commits the instant /// it is durable, so the head IS the commit point; otherwise the highest diff --git a/core/simulator/src/lib.rs b/core/simulator/src/lib.rs index 8a9d7c6317..373e9fd23c 100644 --- a/core/simulator/src/lib.rs +++ b/core/simulator/src/lib.rs @@ -5209,9 +5209,14 @@ mod partition_repair_driver_tests { }; } - /// The partition-plane prepare a packet carries, if it carries one for - /// `group`. - fn prepare_for(packet: &Packet, group: u64) -> Option { + /// The prepare a packet carries, if it carries one for `group`. Keyed on + /// the group, so `metadata_repair_driver_tests` reads the metadata plane's + /// prepares with the same helper. + /// + /// `pub`, not `pub(super)`, in this and the helpers below: the module is + /// private and `#[cfg(test)]`, so both spell the same reach, and + /// `clippy::redundant_pub_crate` refuses the narrower one. + pub fn prepare_for(packet: &Packet, group: u64) -> Option { if packet.message.header().command != Command::Prepare { return None; } @@ -5221,7 +5226,7 @@ mod partition_repair_driver_tests { } /// Whether a packet is a commit heartbeat for `group`. - fn is_commit_for(packet: &Packet, group: u64) -> bool { + pub fn is_commit_for(packet: &Packet, group: u64) -> bool { if packet.message.header().command != Command::Commit { return false; } @@ -5231,7 +5236,7 @@ mod partition_repair_driver_tests { } /// Whether a packet is a repair request for `group`. - fn is_request_prepares_for(packet: &Packet, group: u64) -> bool { + pub fn is_request_prepares_for(packet: &Packet, group: u64) -> bool { if packet.message.header().command != Command::RequestPrepares { return false; } @@ -5242,7 +5247,7 @@ mod partition_repair_driver_tests { } /// Whether a packet is a repair stream terminator for `group`. - fn is_repair_done_for(packet: &Packet, group: u64) -> bool { + pub fn is_repair_done_for(packet: &Packet, group: u64) -> bool { if packet.message.header().command != Command::RepairDone { return false; } @@ -5252,7 +5257,7 @@ mod partition_repair_driver_tests { header.group == group } - fn cluster(seed: u64) -> (Simulator, SimClient) { + pub fn cluster(seed: u64) -> (Simulator, SimClient) { server_common::MemoryPool::init_pool(&server_common::MemoryPoolSettings { enabled: false, size: iggy_common::IggyByteSize::from(0u64), @@ -5964,6 +5969,713 @@ mod partition_repair_driver_tests { } } +#[cfg(test)] +mod metadata_repair_driver_tests { + //! A backup that missed a committed metadata prepare recovers in Normal + //! status, without waiting for a view change. + //! + //! The metadata twin of `partition_repair_driver_tests`, driven by the + //! detector in `tick_metadata`: the same preflight `commit_max` advance + //! starves the `Advanced`-gated arm in `on_commit`, and + //! `retry_stalled_metadata_repair` re-drives only a session that already + //! exists. Every fault here is keyed on `METADATA_GROUP`, since both + //! planes share `Prepare` and `Commit` on the same links. + + use super::partition_repair_driver_tests::{ + cluster, is_commit_for, is_repair_done_for, is_request_prepares_for, prepare_for, + }; + use super::*; + use consensus::Status; + use journal::Journal; + use packet::Packet; + use server_common::sharding::METADATA_GROUP; + use std::sync::atomic::{AtomicU64, Ordering}; + + /// Chain replication runs 0 -> 1 -> 2 and stops before the primary, so + /// replica 2 is the only one whose losses cannot starve the group of + /// quorum (see `partition_repair_driver_tests::LAGGING`). + const LAGGING: u8 = 2; + + const CLIENT_ID: u128 = 1; + + /// Ops committed cleanly before the fault, so the gap opens above a + /// committed prefix rather than at the group's first op. + const WARMUP_SENDS: usize = 3; + + /// Ticks stepped after each stream creation, one round trip's worth. + const STEPS_PER_SEND: usize = 12; + + /// Creations issued with the fault standing in the gap test. Long enough + /// that prepares keep consuming the `commit_max` advance the heartbeat + /// backstop needs; the debounce may elapse mid-produce, which the verdict + /// tolerates (the withheld heartbeats mean only the tick driver can arm). + const GAP_SENDS: usize = 12; + + /// Creations issued with the fault standing in the eviction and + /// walk-starvation tests: few enough (under the debounce) that it fires + /// only after the traffic stops, so the floor stamp or the starved walk + /// edge is in place before the arm runs. + const SHORT_GAP_SENDS: usize = 3; + + /// Quiet ticks for the repair stream to land, kept under + /// `NORMAL_HEARTBEAT_TICKS` (500) so no election can be the healer. + const QUIET_STEPS: usize = 160; + + /// Budget for the group to settle once the fault is lifted; the drain + /// loop breaks on convergence. + const DRAIN_STEPS: usize = 600; + + /// Quiet budget for the walk-starvation test: debounce, repair stream, + /// then the drain, still under `NORMAL_HEARTBEAT_TICKS`. + const STRAND_QUIET_STEPS: usize = 300; + + /// Stall interval for the rotation run, shortened from + /// `partitions::REPAIR_RETRY_TICKS` so a whole spent budget plus the gap + /// debounce (floored at `shard::REPAIR_GAP_DEBOUNCE_TICKS_MIN`) fits under + /// `NORMAL_HEARTBEAT_TICKS`. At the production interval the run would only + /// prove that an election heals the gap. + const ROTATE_RETRY_TICKS: u32 = 10; + + /// Quiet budget for the rotation run: debounce, the spent budget, then the + /// repair stream from the replica it rotates onto. + const ROTATE_QUIET_STEPS: usize = 250; + + /// Ticks of healthy load in the no-false-positive test, several debounce + /// intervals' worth so the driver gets many chances to arm. + const LOAD_TICKS: usize = 4 * partitions::REPAIR_RETRY_TICKS as usize; + + /// Paced at a fraction of a round trip; faster submission only collects + /// transient rejections once the prepare pipeline fills. + const TICKS_PER_SEND: usize = 4; + + /// Ops the healthy run must have committed for its verdict to mean + /// anything: enough to prove the group was live across several debounce + /// intervals, not that it was saturated. + const COMMITTED_MIN: u64 = 20; + + /// Defines this test's `withhold_one_prepare` chain hook over the static it + /// names: swallow the FIRST metadata prepare, once, and record its op in + /// `$withheld_op`. + /// + /// A macro for the same reason as the partition twin: link hooks are bare + /// `fn` pointers, so the body cannot capture, and the statics must stay + /// per-test or the siblings in this binary would share one fault. + macro_rules! withhold_one_metadata_prepare { + ($withheld_op:ident) => { + fn withhold_one_prepare(packet: &Packet) -> bool { + let Some(header) = prepare_for(packet, METADATA_GROUP) else { + return false; + }; + $withheld_op + .compare_exchange(0, header.op, Ordering::Relaxed, Ordering::Relaxed) + .is_ok() + } + }; + } + + /// Defines this test's primary -> backup hook: withhold this group's commit + /// heartbeats, so the `Advanced` backstop can never run, and withhold + /// retransmits of the op `$withheld_op` names. + /// + /// The retransmit half stands in for production behaviour rather than + /// adding a fault: `consensus::retransmit_targets` skips an op that already + /// reached quorum, and this op reaches quorum on 0 and 1 alone. + macro_rules! starve_commit_edge { + ($withheld_op:ident) => { + fn starve_commit_edge(packet: &Packet) -> bool { + if let Some(header) = prepare_for(packet, METADATA_GROUP) { + return header.op == $withheld_op.load(Ordering::Relaxed); + } + is_commit_for(packet, METADATA_GROUP) + } + }; + } + + /// `(status, view, commit_min, commit_max)` of one replica's metadata group. + fn metadata_state(sim: &Simulator, replica: u8) -> (Status, u32, u64, u64) { + let metadata = sim.replicas[replica as usize].shards[0].plane.metadata(); + let consensus = metadata + .consensus + .as_ref() + .expect("shard 0 owns metadata consensus"); + ( + consensus.status(), + consensus.view(), + consensus.commit_min(), + consensus.commit_max(), + ) + } + + #[allow(clippy::cast_possible_truncation)] + fn journal_holds(sim: &Simulator, replica: u8, op: u64) -> bool { + sim.replicas[replica as usize] + .metadata_journal + .header(op as usize) + .is_some() + } + + fn gap_drops(sim: &Simulator, replica: u8) -> u64 { + sim.replicas[replica as usize].shards[0] + .metrics() + .metadata_prepare_gap_drops_value() + } + + fn transfer_armed(sim: &Simulator, replica: u8) -> bool { + let metadata = sim.replicas[replica as usize].shards[0].plane.metadata(); + metadata.consensus.as_ref().is_some_and(|consensus| { + consensus.state_transfer_stage() != consensus::StateTransferStage::Idle + }) + } + + /// Submit `sends` stream creations to the primary, stepping between each. + fn create_streams(sim: &mut Simulator, client: &SimClient, sends: usize, tag: &str) { + for index in 0..sends { + let msg = client.create_stream(&format!("{tag}-{index}")); + sim.submit_request(client.client_id(), 0, msg.into_generic()); + for _ in 0..STEPS_PER_SEND { + sim.step(); + } + } + } + + #[test] + fn given_a_backup_that_dropped_a_committed_metadata_prepare_when_heartbeat_advances_are_starved_should_repair_in_normal_status() + { + // Statics, not captures: the link hooks are bare `fn` pointers, + // declared inside the test so parallel siblings cannot share them. + static WITHHELD_OP: AtomicU64 = AtomicU64::new(0); + withhold_one_metadata_prepare!(WITHHELD_OP); + starve_commit_edge!(WITHHELD_OP); + + let (mut sim, client) = cluster(0x5EED_0240); + sim.register_client_with_primary(&client); + WITHHELD_OP.store(0, Ordering::Relaxed); + + create_streams(&mut sim, &client, WARMUP_SENDS, "md-warm"); + let (_, _, warm_commit_min, _) = metadata_state(&sim, LAGGING); + assert!( + warm_commit_min > 0, + "the lagging replica committed nothing before the fault, so the gap \ + below would open at the group's first op" + ); + + *sim.network + .link_drop_packet_fn(ProcessId::Replica(1), ProcessId::Replica(LAGGING)) = + Some(withhold_one_prepare); + *sim.network + .link_drop_packet_fn(ProcessId::Replica(0), ProcessId::Replica(LAGGING)) = + Some(starve_commit_edge); + + create_streams(&mut sim, &client, GAP_SENDS, "md-gap"); + + let withheld = WITHHELD_OP.load(Ordering::Relaxed); + assert_ne!( + withheld, 0, + "no metadata prepare crossed the chain link, so the fault never armed" + ); + assert!( + gap_drops(&sim, LAGGING) > 0, + "the lagging replica never reached its backup gap check, so the \ + prepares after the withheld op were not dropped as a gap" + ); + + for _ in 0..QUIET_STEPS { + sim.step(); + } + + // Judged with the blockade still standing: no commit heartbeat for + // this group has reached the replica since the gap opened, so only + // the tick driver can have armed the repair. + let (status, view, commit_min, _) = metadata_state(&sim, LAGGING); + assert_eq!( + view, 0, + "a view change healed the gap instead of the repair driver; the test \ + proves nothing about normal status" + ); + assert_eq!(status, Status::Normal, "the replica left Normal status"); + assert!( + journal_holds(&sim, LAGGING, withheld), + "op {withheld} was never repaired back into the lagging replica's WAL" + ); + assert!( + commit_min >= withheld, + "the commit walk never crossed the repaired hole: stopped at \ + {commit_min}, the withheld op is {withheld}" + ); + + // Lift the blockade and let the group settle; the tail above the + // repaired window waits on the heartbeats the fault withheld. + *sim.network + .link_drop_packet_fn(ProcessId::Replica(0), ProcessId::Replica(LAGGING)) = None; + for _ in 0..DRAIN_STEPS { + sim.step(); + let (_, _, commit_min, commit_max) = metadata_state(&sim, LAGGING); + if commit_min == commit_max { + break; + } + } + let (status, view, commit_min, commit_max) = metadata_state(&sim, LAGGING); + assert_eq!((status, view), (Status::Normal, 0)); + assert_eq!( + commit_min, commit_max, + "the lagging replica is still gap-stopped: committed through \ + {commit_max} but walkable only to {commit_min}" + ); + } + + #[test] + fn given_a_metadata_repair_armed_by_the_tick_driver_when_the_floor_is_evicted_should_convert_to_state_transfer() + { + static WITHHELD_OP: AtomicU64 = AtomicU64::new(0); + withhold_one_metadata_prepare!(WITHHELD_OP); + starve_commit_edge!(WITHHELD_OP); + + let (mut sim, client) = cluster(0x5EED_0241); + sim.register_client_with_primary(&client); + WITHHELD_OP.store(0, Ordering::Relaxed); + + create_streams(&mut sim, &client, WARMUP_SENDS, "md-warm"); + + *sim.network + .link_drop_packet_fn(ProcessId::Replica(1), ProcessId::Replica(LAGGING)) = + Some(withhold_one_prepare); + *sim.network + .link_drop_packet_fn(ProcessId::Replica(0), ProcessId::Replica(LAGGING)) = + Some(starve_commit_edge); + + create_streams(&mut sim, &client, SHORT_GAP_SENDS, "md-gap"); + assert_ne!( + WITHHELD_OP.load(Ordering::Relaxed), + 0, + "no metadata prepare crossed the chain link, so the fault never armed" + ); + + // Move the primary's retention floor past the whole gap window before + // the debounce can arm (`SHORT_GAP_SENDS`): the serve path reads only + // the snapshot watermark, so the request is answered `RangeEvicted` + // (see `stamp_metadata_snapshot`). + let primary_commit_min = metadata_state(&sim, 0).2; + sim.stamp_metadata_snapshot(0, primary_commit_min); + + for _ in 0..QUIET_STEPS { + sim.step(); + if transfer_armed(&sim, LAGGING) { + break; + } + } + + let (status, view, ..) = metadata_state(&sim, LAGGING); + assert_eq!( + view, 0, + "a view change armed the recovery instead of the tick-armed repair session" + ); + assert!( + transfer_armed(&sim, LAGGING), + "the tick-armed repair session hit an evicted floor but never converted \ + to a state transfer (status {status:?})" + ); + } + + #[test] + fn given_a_backup_holding_resident_committed_metadata_ops_when_every_walk_edge_is_starved_should_drain_in_normal_status() + { + static WITHHELD_OP: AtomicU64 = AtomicU64::new(0); + static WITHHELD_DONES: AtomicU64 = AtomicU64::new(0); + withhold_one_metadata_prepare!(WITHHELD_OP); + + /// Primary -> 2: withhold every direct prepare (live ones ride the + /// chain, so this starves only retransmit heals), the group's commit + /// heartbeats, and its repair terminators. The repaired ops themselves + /// pass, so the window lands resident while the `RepairDone` that would + /// run the walk never fires. + fn starve_walk_edges(packet: &Packet) -> bool { + if prepare_for(packet, METADATA_GROUP).is_some() { + return true; + } + if is_repair_done_for(packet, METADATA_GROUP) { + WITHHELD_DONES.fetch_add(1, Ordering::Relaxed); + return true; + } + is_commit_for(packet, METADATA_GROUP) + } + + let (mut sim, client) = cluster(0x5EED_0242); + sim.register_client_with_primary(&client); + WITHHELD_OP.store(0, Ordering::Relaxed); + WITHHELD_DONES.store(0, Ordering::Relaxed); + + create_streams(&mut sim, &client, WARMUP_SENDS, "md-warm"); + let (_, _, warm_commit_min, _) = metadata_state(&sim, LAGGING); + assert!( + warm_commit_min > 0, + "the lagging replica committed nothing before the fault, so the gap \ + below would open at the group's first op" + ); + + *sim.network + .link_drop_packet_fn(ProcessId::Replica(1), ProcessId::Replica(LAGGING)) = + Some(withhold_one_prepare); + *sim.network + .link_drop_packet_fn(ProcessId::Replica(0), ProcessId::Replica(LAGGING)) = + Some(starve_walk_edges); + + create_streams(&mut sim, &client, SHORT_GAP_SENDS, "md-strand"); + + let withheld = WITHHELD_OP.load(Ordering::Relaxed); + assert_ne!( + withheld, 0, + "no metadata prepare crossed the chain link, so the fault never armed" + ); + + for _ in 0..STRAND_QUIET_STEPS { + sim.step(); + let (_, view, commit_min, commit_max) = metadata_state(&sim, LAGGING); + if view != 0 || (commit_min >= withheld && commit_min == commit_max) { + break; + } + } + + assert!( + WITHHELD_DONES.load(Ordering::Relaxed) > 0, + "no repair terminator was withheld, so the walk was never starved and \ + a green run would not prove the tick backstop" + ); + assert!( + journal_holds(&sim, LAGGING, withheld), + "op {withheld} was never repaired back into the lagging replica's WAL" + ); + let (status, view, commit_min, commit_max) = metadata_state(&sim, LAGGING); + assert_eq!( + view, 0, + "a view change drained the walk instead of the tick backstop; the test \ + proves nothing about normal status" + ); + assert_eq!(status, Status::Normal, "the replica left Normal status"); + for op in commit_min + 1..=commit_max { + assert!( + journal_holds(&sim, LAGGING, op), + "op {op} is not resident, so this run stranded on a repair gap, \ + not a parked walk" + ); + } + assert_eq!( + commit_min, commit_max, + "the walk never resumed over resident committed ops: walkable to \ + {commit_min}, committed through {commit_max}, every op between resident" + ); + } + + #[test] + fn given_a_resident_header_with_no_body_when_the_walk_stops_on_it_should_drop_it_for_repair() { + static WITHHELD_OP: AtomicU64 = AtomicU64::new(0); + withhold_one_metadata_prepare!(WITHHELD_OP); + starve_commit_edge!(WITHHELD_OP); + + let (mut sim, client) = cluster(0x5EED_0245); + sim.register_client_with_primary(&client); + WITHHELD_OP.store(0, Ordering::Relaxed); + sim.replicas[LAGGING as usize].shards[0].set_repair_retry_ticks(ROTATE_RETRY_TICKS); + + create_streams(&mut sim, &client, WARMUP_SENDS, "md-warm"); + + *sim.network + .link_drop_packet_fn(ProcessId::Replica(1), ProcessId::Replica(LAGGING)) = + Some(withhold_one_prepare); + *sim.network + .link_drop_packet_fn(ProcessId::Replica(0), ProcessId::Replica(LAGGING)) = + Some(starve_commit_edge); + + create_streams(&mut sim, &client, SHORT_GAP_SENDS, "md-gap"); + assert_ne!( + WITHHELD_OP.load(Ordering::Relaxed), + 0, + "no metadata prepare crossed the chain link, so the fault never armed" + ); + + // Corrupt the op the walk is about to read, once repair has put a + // window back but before the walk has consumed it. Injected here + // rather than by dropping packets: the header and the body land in the + // same append, so no link fault produces this state. + let mut doomed = 0; + for _ in 0..ROTATE_QUIET_STEPS { + sim.step(); + let (_, view, commit_min, commit_max) = metadata_state(&sim, LAGGING); + if view != 0 { + break; + } + let next_op = commit_min + 1; + if commit_min < commit_max && journal_holds(&sim, LAGGING, next_op) { + assert!( + sim.replicas[LAGGING as usize] + .metadata_journal + .forget_body(next_op), + "op {next_op} had no body to forget" + ); + doomed = next_op; + break; + } + } + assert_ne!( + doomed, 0, + "the repair window never landed resident above the commit point, so \ + there was no walkable op to corrupt" + ); + assert!( + journal_holds(&sim, LAGGING, doomed), + "op {doomed}'s header must stay resident, or the walk would read this \ + as an ordinary hole and repair would refill it unaided" + ); + + for _ in 0..ROTATE_QUIET_STEPS { + sim.step(); + let (_, view, commit_min, _) = metadata_state(&sim, LAGGING); + if view != 0 || commit_min >= doomed { + break; + } + } + + let (status, view, commit_min, commit_max) = metadata_state(&sim, LAGGING); + assert_eq!( + view, 0, + "a view change healed the log instead of the walk; the test proves \ + nothing about the unwalkable entry" + ); + assert_eq!(status, Status::Normal, "the replica left Normal status"); + assert!( + commit_min >= doomed, + "the walk never crossed op {doomed}: the repair ingest skips an op \ + whose header is resident and `append` refuses the slot under it, so \ + the header has to be dropped first. Stopped at {commit_min} of \ + {commit_max}" + ); + } + + #[test] + fn given_a_repair_peer_that_never_answers_when_the_budget_is_spent_should_re_arm_from_another_replica() + { + static WITHHELD_OP: AtomicU64 = AtomicU64::new(0); + static ASKED_PRIMARY: AtomicU64 = AtomicU64::new(0); + static ASKED_SUCCESSOR: AtomicU64 = AtomicU64::new(0); + withhold_one_metadata_prepare!(WITHHELD_OP); + + /// The peer `gap_repair_peer` names for a gap-stopped backup is the + /// primary, so the session can only heal by rotating off it. + fn blackhole(_packet: &Packet) -> bool { + true + } + + /// Observers, not faults: every packet passes. + fn count_requests_to_primary(packet: &Packet) -> bool { + if is_request_prepares_for(packet, METADATA_GROUP) { + ASKED_PRIMARY.fetch_add(1, Ordering::Relaxed); + } + false + } + fn count_requests_to_successor(packet: &Packet) -> bool { + if is_request_prepares_for(packet, METADATA_GROUP) { + ASKED_SUCCESSOR.fetch_add(1, Ordering::Relaxed); + } + false + } + + let (mut sim, client) = cluster(0x5EED_0244); + sim.register_client_with_primary(&client); + WITHHELD_OP.store(0, Ordering::Relaxed); + ASKED_PRIMARY.store(0, Ordering::Relaxed); + ASKED_SUCCESSOR.store(0, Ordering::Relaxed); + sim.replicas[LAGGING as usize].shards[0].set_repair_retry_ticks(ROTATE_RETRY_TICKS); + + create_streams(&mut sim, &client, WARMUP_SENDS, "md-warm"); + let (_, _, warm_commit_min, _) = metadata_state(&sim, LAGGING); + assert!( + warm_commit_min > 0, + "the lagging replica committed nothing before the fault, so the gap \ + below would open at the group's first op" + ); + + *sim.network + .link_drop_packet_fn(ProcessId::Replica(1), ProcessId::Replica(LAGGING)) = + Some(withhold_one_prepare); + // Everything, not just this group's commit edge: the primary must be + // unable to answer the repair it is about to be asked for, while the + // chain (0 -> 1 -> 2) keeps carrying the prepares that advance + // `commit_max` past the hole. + *sim.network + .link_drop_packet_fn(ProcessId::Replica(0), ProcessId::Replica(LAGGING)) = + Some(blackhole); + *sim.network + .link_drop_packet_fn(ProcessId::Replica(LAGGING), ProcessId::Replica(0)) = + Some(count_requests_to_primary); + *sim.network + .link_drop_packet_fn(ProcessId::Replica(LAGGING), ProcessId::Replica(1)) = + Some(count_requests_to_successor); + + create_streams(&mut sim, &client, SHORT_GAP_SENDS, "md-gap"); + let withheld = WITHHELD_OP.load(Ordering::Relaxed); + assert_ne!( + withheld, 0, + "no metadata prepare crossed the chain link, so the fault never armed" + ); + + for _ in 0..ROTATE_QUIET_STEPS { + sim.step(); + let (_, view, commit_min, _) = metadata_state(&sim, LAGGING); + if view != 0 || commit_min >= withheld { + break; + } + } + + let (status, view, commit_min, commit_max) = metadata_state(&sim, LAGGING); + assert_eq!( + view, 0, + "a view change healed the gap instead of the rotation; the test proves \ + nothing about the stall budget" + ); + assert_eq!(status, Status::Normal, "the replica left Normal status"); + assert!( + ASKED_PRIMARY.load(Ordering::Relaxed) > 1, + "the session was never re-requested from the silent primary, so the \ + stall budget (`partitions::REPAIR_MAX_STALL_RETRIES`) was not spent \ + and any rotation below came from somewhere else" + ); + assert!( + ASKED_SUCCESSOR.load(Ordering::Relaxed) > 0, + "the budget ran out against a peer that cannot answer and the session \ + was never re-armed anywhere else; nothing re-drives it, so the plane \ + stays gap-stopped at {commit_min} of {commit_max}" + ); + assert!( + journal_holds(&sim, LAGGING, withheld), + "op {withheld} was never repaired back into the lagging replica's WAL" + ); + assert!( + commit_min >= withheld, + "the commit walk never crossed the repaired hole: stopped at \ + {commit_min}, the withheld op is {withheld}" + ); + } + + /// Regression canary, expected green even without the tick driver: a + /// healthy metadata backup walks at every accepted prepare's tail + /// (`on_replicate`), so per-tick lag never survives to quiescence and the + /// journal-hole half of the predicate is pinned by `gap_detector_tests` + /// instead. What this run pins is that the driver stays silent under + /// sustained pipelined load. + #[test] + fn given_healthy_metadata_traffic_when_no_gap_exists_should_not_arm_repair() { + static REPAIR_REQUESTS: AtomicU64 = AtomicU64::new(0); + + /// Observer, not a fault: counts this group's repair requests and + /// passes every packet through. + fn count_repair_requests(packet: &Packet) -> bool { + if is_request_prepares_for(packet, METADATA_GROUP) { + REPAIR_REQUESTS.fetch_add(1, Ordering::Relaxed); + } + false + } + + // TWO replicas, as in the partition twin: quorum spans both, so no op + // can commit while the backup misses it, and every reordering-induced + // gap blocks quorum until retransmit refills it. + server_common::MemoryPool::init_pool(&server_common::MemoryPoolSettings { + enabled: false, + size: iggy_common::IggyByteSize::from(0u64), + bucket_capacity: 1, + }); + let replica_count: u8 = 2; + let mut sim = Simulator::new( + replica_count as usize, + std::iter::once(CLIENT_ID), + packet::PacketSimulatorOptions { + node_count: replica_count, + client_count: 1, + seed: 0x5EED_0243, + ..packet::PacketSimulatorOptions::default() + }, + ); + let client = SimClient::new(CLIENT_ID); + sim.register_client_with_primary(&client); + REPAIR_REQUESTS.store(0, Ordering::Relaxed); + + for (from, to) in [(0u8, 1u8), (1, 0)] { + *sim.network + .link_drop_packet_fn(ProcessId::Replica(from), ProcessId::Replica(to)) = + Some(count_repair_requests); + } + + // Sustained, not bursty, and the run asserts the load went somewhere, + // so a green result cannot come from a workload that never loaded the + // group. + let mut lag_run = 0u32; + let mut longest_lag_run = 0u32; + let sample = |sim: &Simulator, lag_run: &mut u32, longest: &mut u32| { + let (_, _, commit_min, commit_max) = metadata_state(sim, 1); + if commit_min < commit_max { + *lag_run += 1; + *longest = (*longest).max(*lag_run); + } else { + *lag_run = 0; + } + }; + for tick in 0..LOAD_TICKS { + if tick % TICKS_PER_SEND == 0 { + let msg = client.create_stream(&format!("healthy-{tick}")); + sim.submit_request(client.client_id(), 0, msg.into_generic()); + } + sim.step(); + sample(&sim, &mut lag_run, &mut longest_lag_run); + } + for _ in 0..QUIET_STEPS { + sim.step(); + sample(&sim, &mut lag_run, &mut longest_lag_run); + } + + let committed = metadata_state(&sim, 1).2; + let sends = LOAD_TICKS / TICKS_PER_SEND; + assert!( + committed >= COMMITTED_MIN, + "the backup committed only {committed} ops across {sends} sends, so the \ + driver was never ticked over a loaded group" + ); + // Asserted, not merely recorded: on this plane the tail walk in + // `on_replicate` leaves a healthy backup caught up at every quiescence + // point, so any lag at all is a change in that behaviour rather than + // ordinary pipelining. The predicate itself is pinned by + // `gap_detector_tests`; what this holds is the premise the repair check + // below rests on. + assert_eq!( + longest_lag_run, 0, + "healthy two-replica metadata traffic left the backup lagging for \ + {longest_lag_run} consecutive ticks; the run below then proves nothing \ + about false positives, and this test should sample residency the way \ + the partition twin's walk-starvation run does" + ); + for replica in 0..replica_count { + let (status, view, commit_min, commit_max) = metadata_state(&sim, replica); + assert_eq!( + (status, view), + (Status::Normal, 0), + "replica {replica} left view 0 / Normal, so a view change could \ + account for repair traffic" + ); + if commit_min < commit_max { + assert!( + journal_holds(&sim, replica, commit_min + 1), + "replica {replica} lags at {commit_min} of {commit_max} with op \ + {} missing, so a no-loss run produced a real hole", + commit_min + 1 + ); + } + } + assert_eq!( + REPAIR_REQUESTS.load(Ordering::Relaxed), + 0, + "the tick driver requested repair on a healthy group; its gap predicate \ + is reading ordinary commit lag as a journal hole" + ); + } +} + #[cfg(test)] mod metadata_read_frontier_tests { //! A client that committed a metadata write and then re-homed onto a From 0bdd8c6ca83e4b80f46e2c3793491c0ad4aa9e56 Mon Sep 17 00:00:00 2001 From: Piotr Gankiewicz Date: Mon, 7 Sep 2026 11:26:35 +0200 Subject: [PATCH 076/182] fix(partitions): bound consumer offset state per partition (#4063) Every distinct consumer id stored one offset file and one map entry per partition with no upper bound, so a partition could accumulate durable state until the filesystem or memory gave out and the replica fenced itself. Offset deletes acknowledged locally could also free a primary slot while backups kept every older generation, and follower-served auto-commit cursors could be promoted into committed state across leadership changes. Add a per-partition, per-kind admission limit enforced on the primary before a new key enters consensus or touches disk. Admission counts durable keys plus in-flight reservations, so existing consumers keep updating at the limit while new keys get a typed TooManyConsumerOffsets reply. Replicated partitions now commit every offset store and delete through VSR regardless of the acknowledgement byte, and consumer-group cleanup uses replicated deletes, which keeps file counts identical on all replicas. State transfer exports committed values only, and recovery preserves state above a lowered limit with a warning. Auto-commit hands its reservation to the partition pump through an internal frame instead of a per-poll waiter, phantom cursors are reclaimed on promotion, and the reconciler no longer runs a full pass on every group offset advance. Missing consumer groups return the existing not-found codes. The limit ships as [partition] consumer_offsets_max with a default of 4096, and the new error code is mirrored in the Go, Java, and Node SDKs. --- Cargo.lock | 22 +- Cargo.toml | 10 +- .../tests/tcp_test/offset_feature_delete.go | 21 +- bdd/python/uv.lock | 2 +- core/ai/mcp/Cargo.toml | 2 +- core/bench/Cargo.toml | 2 +- core/bench/dashboard/frontend/Cargo.toml | 2 +- core/bench/dashboard/server/Cargo.toml | 2 +- core/bench/report/Cargo.toml | 2 +- core/bench/src/benchmarks/common.rs | 10 +- core/binary_protocol/Cargo.toml | 2 +- .../src/primitives/ack_level.rs | 9 +- .../delete_consumer_offset.rs | 4 +- .../consumer_offsets/store_consumer_offset.rs | 4 +- core/cli/Cargo.toml | 2 +- core/common/Cargo.toml | 2 +- core/common/src/error/iggy_error.rs | 10 + .../src/traits/consumer_offset_client.rs | 7 + core/common/src/traits/message_client.rs | 10 + core/configs/src/server_config/defaults.rs | 2 + core/configs/src/server_config/displays.rs | 6 +- core/configs/src/server_config/partition.rs | 72 + core/connectors/sdk/Cargo.toml | 2 +- core/integration/src/harness/disk.rs | 33 +- .../tests/cluster/consumer_offset_quota.rs | 357 +++ core/integration/tests/cluster/mod.rs | 1 + .../tests/server/consumer_offset_quota_vsr.rs | 514 +++ core/integration/tests/server/mod.rs | 1 + core/integration/tests/server/raw_tcp.rs | 5 + core/message_bus/src/lib.rs | 2 +- core/metadata/src/impls/metadata.rs | 6 +- core/metadata/src/stm/user.rs | 4 +- .../src/consumer_offset_capacity.rs | 717 +++++ core/partitions/src/iggy_partition.rs | 2771 +++++++++++++++-- core/partitions/src/iggy_partitions.rs | 51 +- core/partitions/src/lib.rs | 19 +- core/partitions/src/offset_storage.rs | 197 +- core/partitions/src/poll_plan.rs | 352 ++- core/partitions/src/state_transfer.rs | 727 +++-- core/partitions/src/types.rs | 5 + core/sdk/Cargo.toml | 2 +- core/sdk/src/clients/consumer.rs | 32 +- core/sdk/src/leader_aware.rs | 25 + core/sdk/src/quic/quic_client.rs | 14 +- core/sdk/src/tcp/tcp_client.rs | 27 +- core/sdk/src/websocket/websocket_client.rs | 14 +- core/server/Cargo.toml | 2 +- core/server/config.toml | 30 + core/server/src/boot/recovery.rs | 11 + core/server/src/consumer_group.rs | 46 +- core/server/src/dispatch/mod.rs | 1 + core/server/src/dispatch/partition.rs | 399 ++- core/server/src/dispatch/session_ops.rs | 1 + core/server/src/dispatch/test_support.rs | 1 + core/server/src/http/handlers.rs | 18 +- core/server/src/offset_recovery.rs | 354 ++- core/server/src/partition_helpers.rs | 119 +- core/server/src/partition_reconciler.rs | 380 ++- core/server/src/responses.rs | 51 +- core/shard/src/lib.rs | 194 +- core/shard/src/metrics.rs | 124 +- core/shard/src/router.rs | 9 + core/simulator/src/client.rs | 4 +- core/simulator/src/lib.rs | 239 +- core/simulator/src/replica.rs | 1 + core/simulator/src/workload/effect.rs | 2 +- core/simulator/src/workload/shadow.rs | 2 +- examples/node/package-lock.json | 2 +- examples/python/uv.lock | 2 +- foreign/go/contracts/version.go | 2 +- foreign/go/errors/errors.yaml | 4 + foreign/go/errors/errors_gen.go | 17 + foreign/go/errors/errors_test.go | 19 + .../apache/iggy/exception/IggyErrorCode.java | 3 + .../iggy/exception/IggyErrorCodeTest.java | 1 + foreign/node/package-lock.json | 4 +- foreign/node/package.json | 2 +- foreign/node/src/wire/error.code.test.ts | 4 + foreign/node/src/wire/error.code.ts | 1 + foreign/python/Cargo.toml | 4 +- foreign/python/pyproject.toml | 2 +- foreign/python/uv.lock | 2 +- 82 files changed, 7052 insertions(+), 1088 deletions(-) create mode 100644 core/integration/tests/cluster/consumer_offset_quota.rs create mode 100644 core/integration/tests/server/consumer_offset_quota_vsr.rs create mode 100644 core/partitions/src/consumer_offset_capacity.rs diff --git a/Cargo.lock b/Cargo.lock index 56ecb71dd9..7856dd4133 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1862,7 +1862,7 @@ checksum = "3a8241f3ebb85c056b509d4327ad0358fbbba6ffb340bf388f26350aeda225b1" [[package]] name = "bench-dashboard-frontend" -version = "0.8.0-edge.3" +version = "0.8.0-edge.4" dependencies = [ "bench-dashboard-shared", "bench-report", @@ -1892,7 +1892,7 @@ dependencies = [ [[package]] name = "bench-report" -version = "0.4.0-edge.3" +version = "0.4.0-edge.4" dependencies = [ "charming", "colored", @@ -6594,7 +6594,7 @@ checksum = "cd62e6b5e86ea8eeeb8db1de02880a6abc01a397b2ebb64b5d74ac255318f5cb" [[package]] name = "iggy" -version = "0.11.0-edge.6" +version = "0.11.0-edge.7" dependencies = [ "async-broadcast", "async-dropper", @@ -6628,7 +6628,7 @@ dependencies = [ [[package]] name = "iggy-bench" -version = "0.6.0-edge.6" +version = "0.6.0-edge.7" dependencies = [ "async-trait", "bench-report", @@ -6657,7 +6657,7 @@ dependencies = [ [[package]] name = "iggy-bench-dashboard-server" -version = "0.8.0-edge.3" +version = "0.8.0-edge.4" dependencies = [ "actix-cors", "actix-files", @@ -6685,7 +6685,7 @@ dependencies = [ [[package]] name = "iggy-cli" -version = "0.14.0-edge.6" +version = "0.14.0-edge.7" dependencies = [ "anyhow", "apple-native-keyring-store", @@ -6791,7 +6791,7 @@ dependencies = [ [[package]] name = "iggy-mcp" -version = "0.5.0-edge.5" +version = "0.5.0-edge.6" dependencies = [ "axum", "axum-server", @@ -6825,7 +6825,7 @@ dependencies = [ [[package]] name = "iggy_binary_protocol" -version = "0.11.0-edge.6" +version = "0.11.0-edge.7" dependencies = [ "aligned-vec", "bytemuck", @@ -6838,7 +6838,7 @@ dependencies = [ [[package]] name = "iggy_common" -version = "0.11.0-edge.6" +version = "0.11.0-edge.7" dependencies = [ "aes-gcm 0.11.1", "async-broadcast", @@ -7203,7 +7203,7 @@ dependencies = [ [[package]] name = "iggy_connector_sdk" -version = "0.4.0-edge.3" +version = "0.4.0-edge.4" dependencies = [ "anyhow", "apache-avro 0.22.0", @@ -12160,7 +12160,7 @@ dependencies = [ [[package]] name = "server" -version = "0.9.0-edge.6" +version = "0.9.0-edge.7" dependencies = [ "ahash 0.8.12", "argon2", diff --git a/Cargo.toml b/Cargo.toml index af7a697ca9..beb92e80f4 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -198,13 +198,13 @@ hyper-util = { version = "0.1.20", features = ["server-auto", "service"] } iceberg = "0.10.1" iceberg-catalog-rest = "0.10.1" iceberg-storage-opendal = "0.10.1" -iggy = { path = "core/sdk", version = "0.11.0-edge.6" } -iggy-cli = { path = "core/cli", version = "0.14.0-edge.6" } +iggy = { path = "core/sdk", version = "0.11.0-edge.7" } +iggy-cli = { path = "core/cli", version = "0.14.0-edge.7" } iggy-gateway-kafka = { path = "gateways/kafka" } -iggy_binary_protocol = { path = "core/binary_protocol", version = "0.11.0-edge.6" } -iggy_common = { path = "core/common", version = "0.11.0-edge.6" } +iggy_binary_protocol = { path = "core/binary_protocol", version = "0.11.0-edge.7" } +iggy_common = { path = "core/common", version = "0.11.0-edge.7" } iggy_connector_doris_sink = { path = "core/connectors/sinks/doris_sink" } -iggy_connector_sdk = { path = "core/connectors/sdk", version = "0.4.0-edge.3" } +iggy_connector_sdk = { path = "core/connectors/sdk", version = "0.4.0-edge.4" } indexmap = "2.14.1" integration = { path = "core/integration" } ipnet = "2.12.1" diff --git a/bdd/go/tests/tcp_test/offset_feature_delete.go b/bdd/go/tests/tcp_test/offset_feature_delete.go index 0d8d93d925..f515375609 100644 --- a/bdd/go/tests/tcp_test/offset_feature_delete.go +++ b/bdd/go/tests/tcp_test/offset_feature_delete.go @@ -129,12 +129,25 @@ var _ = ginkgo.Describe("DELETE CONSUMER OFFSET:", func() { &partitionId, ) - // A consumer-offset request is routed by its packed namespace, so - // the shard that answers reports a missing resource rather than - // naming the group. - itShouldReturnSpecificError(err, ierror.ErrResourceNotFound) + // The stream and topic resolve, so the server names the group. + itShouldReturnSpecificError(err, ierror.ErrConsumerGroupIdNotFound) }) + ginkgo.Context("and attempts to delete an offset from a non-existing named consumer group", func() { + client := createAuthorizedConnection() + streamId, _ := successfullyCreateStream(prefix, client) + defer deleteStreamAfterTests(streamId, client) + topicId, _ := successfullyCreateTopic(streamId, client) + streamIdentifier, _ := iggcon.NewIdentifier(streamId) + topicIdentifier, _ := iggcon.NewIdentifier(topicId) + groupIdentifier, _ := iggcon.NewIdentifier("missing-offset-group") + partitionId := uint32(1) + err := client.DeleteConsumerOffset( + context.Background(), iggcon.NewGroupConsumer(groupIdentifier), + streamIdentifier, topicIdentifier, &partitionId, + ) + itShouldReturnSpecificError(err, ierror.ErrConsumerGroupNameNotFound) + }) ginkgo.Context("and attempts to delete an offset from a non-existing stream", func() { client := createAuthorizedConnection() consumer := iggcon.NewGroupConsumer(randomU32Identifier()) diff --git a/bdd/python/uv.lock b/bdd/python/uv.lock index 44c64ab552..b10e81b848 100644 --- a/bdd/python/uv.lock +++ b/bdd/python/uv.lock @@ -8,7 +8,7 @@ exclude-newer-span = "P7D" [[package]] name = "apache-iggy" -version = "0.9.0.dev6" +version = "0.9.0.dev7" source = { directory = "../../foreign/python" } [package.metadata] diff --git a/core/ai/mcp/Cargo.toml b/core/ai/mcp/Cargo.toml index 5386c5c4a0..1d1953f9f2 100644 --- a/core/ai/mcp/Cargo.toml +++ b/core/ai/mcp/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "iggy-mcp" -version = "0.5.0-edge.5" +version = "0.5.0-edge.6" description = "MCP Server for Iggy message streaming platform" edition = "2024" license = "Apache-2.0" diff --git a/core/bench/Cargo.toml b/core/bench/Cargo.toml index e2a3c8d8d0..7a3f1ce421 100644 --- a/core/bench/Cargo.toml +++ b/core/bench/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "iggy-bench" -version = "0.6.0-edge.6" +version = "0.6.0-edge.7" edition = "2024" license = "Apache-2.0" repository = "https://github.com/apache/iggy" diff --git a/core/bench/dashboard/frontend/Cargo.toml b/core/bench/dashboard/frontend/Cargo.toml index 595cec1d09..6b023d6314 100644 --- a/core/bench/dashboard/frontend/Cargo.toml +++ b/core/bench/dashboard/frontend/Cargo.toml @@ -18,7 +18,7 @@ [package] name = "bench-dashboard-frontend" license = "Apache-2.0" -version = "0.8.0-edge.3" +version = "0.8.0-edge.4" edition = "2024" publish = false diff --git a/core/bench/dashboard/server/Cargo.toml b/core/bench/dashboard/server/Cargo.toml index 726df33be8..68570b1123 100644 --- a/core/bench/dashboard/server/Cargo.toml +++ b/core/bench/dashboard/server/Cargo.toml @@ -18,7 +18,7 @@ [package] name = "iggy-bench-dashboard-server" license = "Apache-2.0" -version = "0.8.0-edge.3" +version = "0.8.0-edge.4" edition = "2024" publish = false diff --git a/core/bench/report/Cargo.toml b/core/bench/report/Cargo.toml index a418cdd256..bc5649a576 100644 --- a/core/bench/report/Cargo.toml +++ b/core/bench/report/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "bench-report" -version = "0.4.0-edge.3" +version = "0.4.0-edge.4" edition = "2024" description = "Benchmark report and chart generation library for iggy-bench binary and iggy-benchmarks-dashboard web app" license = "Apache-2.0" diff --git a/core/bench/src/benchmarks/common.rs b/core/bench/src/benchmarks/common.rs index bd5a38247c..ed166bd10c 100644 --- a/core/bench/src/benchmarks/common.rs +++ b/core/bench/src/benchmarks/common.rs @@ -44,7 +44,11 @@ pub async fn create_consumer( "Consumer #{} → joining consumer group #{}", consumer_id, consumer_group_id ); - let cg_identifier = Identifier::try_from(*consumer_group_id).unwrap(); + // By name: the group was created by name and the server assigns its + // own numeric id, which need not equal the bench's counter. + let cg_identifier = + Identifier::named(&format!("{CONSUMER_GROUP_NAME_PREFIX}-{consumer_group_id}")) + .unwrap(); client .join_consumer_group(stream_id, topic_id, &cg_identifier) .await @@ -209,8 +213,10 @@ pub fn build_consumer_futures( consumer_id }; let stream_id = format!("bench-stream-{stream_idx}"); + // Groups are created as `BASE + 1..=cg_count`, one per stream, so the + // consumer must land on the group of the stream it polls. let consumer_group_id = if cg_count > 0 { - Some(CONSUMER_GROUP_BASE_ID + (consumer_id % cg_count)) + Some(CONSUMER_GROUP_BASE_ID + stream_idx) } else { None }; diff --git a/core/binary_protocol/Cargo.toml b/core/binary_protocol/Cargo.toml index 0a83a4c010..be97141e3f 100644 --- a/core/binary_protocol/Cargo.toml +++ b/core/binary_protocol/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "iggy_binary_protocol" -version = "0.11.0-edge.6" +version = "0.11.0-edge.7" description = "Wire protocol types and codec for the Iggy binary protocol. Shared between server and SDK." edition = "2024" rust-version.workspace = true diff --git a/core/binary_protocol/src/primitives/ack_level.rs b/core/binary_protocol/src/primitives/ack_level.rs index a2ddb2389c..d1fc8064b8 100644 --- a/core/binary_protocol/src/primitives/ack_level.rs +++ b/core/binary_protocol/src/primitives/ack_level.rs @@ -20,9 +20,12 @@ use crate::WireError; /// Acknowledgement policy for consumer-offset write commands. /// /// Wire format: single `u8` discriminant. -/// - `NoAck(0)`: leader-local write only; respond as soon as the in-memory -/// and on-disk state have been updated. Matches the fast path used by -/// `PollMessages` auto-commit. +/// - `NoAck(0)`: local fast path for a single-replica partition. Replicated +/// partitions commit offset writes through VSR before replying, including +/// when this acknowledgement value is selected. +/// On a single replica, a directory-sync failure can be reported after the +/// mutation became visible. Its crash durability is then unknown. Retrying +/// a deletion that already took effect can return `ConsumerOffsetNotFound`. /// - `Quorum(1)`: submit through the partition VSR consensus pipeline and /// respond only after the write has been committed by a quorum of replicas. /// This is the default for explicit client writes. diff --git a/core/binary_protocol/src/requests/consumer_offsets/delete_consumer_offset.rs b/core/binary_protocol/src/requests/consumer_offsets/delete_consumer_offset.rs index 0140817940..e5ca29d1ad 100644 --- a/core/binary_protocol/src/requests/consumer_offsets/delete_consumer_offset.rs +++ b/core/binary_protocol/src/requests/consumer_offsets/delete_consumer_offset.rs @@ -24,8 +24,8 @@ use bytes::{BufMut, BytesMut}; /// `DeleteConsumerOffset` request. /// -/// Adds an `ack` byte: `NoAck` = leader-local fast path, `Quorum` = VSR -/// pipeline. +/// The `ack` byte selects the local fast path only for `NoAck` on a +/// single-replica partition. Replicated partitions use VSR for both values. /// /// Wire format: /// ```text diff --git a/core/binary_protocol/src/requests/consumer_offsets/store_consumer_offset.rs b/core/binary_protocol/src/requests/consumer_offsets/store_consumer_offset.rs index 5261d5c531..2283607987 100644 --- a/core/binary_protocol/src/requests/consumer_offsets/store_consumer_offset.rs +++ b/core/binary_protocol/src/requests/consumer_offsets/store_consumer_offset.rs @@ -24,8 +24,8 @@ use bytes::{BufMut, BytesMut}; /// `StoreConsumerOffset` request. /// -/// Adds an `ack` byte: `NoAck` = leader-local fast path, `Quorum` = VSR -/// pipeline. +/// The `ack` byte selects the local fast path only for `NoAck` on a +/// single-replica partition. Replicated partitions use VSR for both values. /// /// Wire format: /// ```text diff --git a/core/cli/Cargo.toml b/core/cli/Cargo.toml index a2f8756bb7..f8a2b89f16 100644 --- a/core/cli/Cargo.toml +++ b/core/cli/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "iggy-cli" -version = "0.14.0-edge.6" +version = "0.14.0-edge.7" edition = "2024" rust-version.workspace = true authors = ["bartosz.ciesla@gmail.com"] diff --git a/core/common/Cargo.toml b/core/common/Cargo.toml index 16aaa621c5..2e059fc3eb 100644 --- a/core/common/Cargo.toml +++ b/core/common/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "iggy_common" -version = "0.11.0-edge.6" +version = "0.11.0-edge.7" description = "Iggy is the persistent message streaming platform written in Rust, supporting QUIC, TCP and HTTP transport protocols, capable of processing millions of messages per second." edition = "2024" rust-version.workspace = true diff --git a/core/common/src/error/iggy_error.rs b/core/common/src/error/iggy_error.rs index e79d96b897..6a31bcb487 100644 --- a/core/common/src/error/iggy_error.rs +++ b/core/common/src/error/iggy_error.rs @@ -333,6 +333,8 @@ pub enum IggyError { NotResolvedConsumer(Identifier) = 3022, #[error("Cannot open consumer offsets file for path: {0}")] CannotOpenConsumerOffsetsFile(String) = 3023, + #[error("Consumer offset limit reached for partition, raise [partition] consumer_offsets_max")] + TooManyConsumerOffsets = 3024, #[error("Segment not found")] SegmentNotFound = 4000, #[error("Segment with start offset: {0} and partition with ID: {1} is closed")] @@ -627,4 +629,12 @@ mod tests { IggyError::from_code_as_string(GROUP_NAME_ERROR_CODE) ) } + + #[test] + fn too_many_consumer_offsets_round_trips_by_code() { + let error = IggyError::TooManyConsumerOffsets; + assert_eq!(error.as_code(), 3024); + assert_eq!(IggyError::from_code(3024), error); + assert_eq!(IggyError::from_code_as_string(3024), error.as_string()); + } } diff --git a/core/common/src/traits/consumer_offset_client.rs b/core/common/src/traits/consumer_offset_client.rs index ffdddb7e66..6409ff81d0 100644 --- a/core/common/src/traits/consumer_offset_client.rs +++ b/core/common/src/traits/consumer_offset_client.rs @@ -24,6 +24,7 @@ pub trait ConsumerOffsetClient { /// Store the consumer offset for a specific consumer or consumer group for the given stream and topic by unique IDs or names. /// /// Authentication is required, and the permission to poll the messages. + /// A new key at the per-partition limit returns [`IggyError::TooManyConsumerOffsets`] (3024). async fn store_consumer_offset( &self, consumer: &Consumer, @@ -35,6 +36,8 @@ pub trait ConsumerOffsetClient { /// Get the consumer offset for a specific consumer or consumer group for the given stream and topic by unique IDs or names. /// /// Authentication is required, and the permission to poll the messages. + /// An absent offset key returns `Ok(None)` when its stream, topic, and consumer group resolve. + /// A local auto-commit cursor can be visible before its durable store commits. async fn get_consumer_offset( &self, consumer: &Consumer, @@ -45,6 +48,10 @@ pub trait ConsumerOffsetClient { /// Delete the consumer offset for a specific consumer or consumer group for the given stream and topic by unique IDs or names. /// /// Authentication is required, and the permission to poll the messages. + /// A missing durable key returns [`IggyError::ConsumerOffsetNotFound`] (3021). + /// Missing numeric and named groups return codes 5000 and 5003 respectively. + /// A replica unable to admit the request returns [`IggyError::TransientNotAccepted`]. + /// Deletion does not allocate a new key and is allowed at the offset limit. async fn delete_consumer_offset( &self, consumer: &Consumer, diff --git a/core/common/src/traits/message_client.rs b/core/common/src/traits/message_client.rs index 18c6d2294a..b710176475 100644 --- a/core/common/src/traits/message_client.rs +++ b/core/common/src/traits/message_client.rs @@ -30,6 +30,16 @@ pub trait MessageClient { /// /// Polling a consumer group the client is not (or no longer) a member of fails with `ConsumerGroupMemberNotFound` rather than returning an empty batch, so the caller can rejoin. /// A member that holds no partitions gets an empty batch whose `partition_id` is [`NO_ASSIGNED_PARTITION`](crate::NO_ASSIGNED_PARTITION). + /// + /// With server-side auto-commit enabled, a new consumer offset key can be + /// rejected with `TooManyConsumerOffsets` at the partition's configured + /// limit. That poll returns no messages. Existing keys remain writable, + /// and polling without auto-commit does not allocate a stored offset. + /// A refused auto-commit submission returns `TransientNotAccepted` with no + /// messages and may be retried. A capacity error requires capacity to be freed. + /// Local cursors without a committed offset can be evicted at the limit. + /// Their next `Next` poll resumes from the earliest retained messages, + /// which can redeliver messages from earlier polls. #[allow(clippy::too_many_arguments)] async fn poll_messages( &self, diff --git a/core/configs/src/server_config/defaults.rs b/core/configs/src/server_config/defaults.rs index 330f39877c..a08913968e 100644 --- a/core/configs/src/server_config/defaults.rs +++ b/core/configs/src/server_config/defaults.rs @@ -183,6 +183,8 @@ impl Default for PartitionConfig { PartitionConfig { prepare_queue_depth: partition.prepare_queue_depth as usize, dedup_clients_max: partition.dedup_clients_max as usize, + consumer_offsets_max: partition.consumer_offsets_max as usize, + consumer_offset_enforce_fsync: partition.consumer_offset_enforce_fsync, offset_reservation_lease: NonZeroU32::new(partition.offset_reservation_lease as u32) .expect("the embedded config.toml carries a nonzero offset_reservation_lease"), evicted_ring_capacity: partition.evicted_ring_capacity as usize, diff --git a/core/configs/src/server_config/displays.rs b/core/configs/src/server_config/displays.rs index 5c46c7b82c..d0284e5889 100644 --- a/core/configs/src/server_config/displays.rs +++ b/core/configs/src/server_config/displays.rs @@ -56,10 +56,14 @@ impl Display for PartitionConfig { fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { write!( f, - "{{ prepare_queue_depth: {}, offset_reservation_lease: {}, \ + "{{ prepare_queue_depth: {}, dedup_clients_max: {}, consumer_offsets_max: {}, \ + consumer_offset_enforce_fsync: {}, offset_reservation_lease: {}, \ evicted_ring_capacity: {}, evicted_ring_bytes_max: {}, \ transfer_served_cache_bytes_max: {}, transfer_artifact_bytes_max: {} }}", self.prepare_queue_depth, + self.dedup_clients_max, + self.consumer_offsets_max, + self.consumer_offset_enforce_fsync, self.offset_reservation_lease, self.evicted_ring_capacity, self.evicted_ring_bytes_max, diff --git a/core/configs/src/server_config/partition.rs b/core/configs/src/server_config/partition.rs index 86ee3a7c4c..56257936c6 100644 --- a/core/configs/src/server_config/partition.rs +++ b/core/configs/src/server_config/partition.rs @@ -139,6 +139,23 @@ pub const PARTITION_DEDUP_CLIENTS_DEFAULT: usize = 4096; /// bytes`: a 112-byte slot entry plus its index-map slot. pub const PARTITION_DEDUP_CLIENTS_CEILING: usize = 1 << 16; +/// Shipped per-kind consumer-offset cardinality limit for one partition. +/// +/// Mirrors `partitions::DEFAULT_CONSUMER_OFFSETS_MAX`. The server crate pins +/// the duplicated literals because configs must not depend on partitions. +pub const PARTITION_CONSUMER_OFFSETS_DEFAULT: usize = 4096; + +/// Configuration typo guard for the per-kind consumer-offset limit. +/// +/// The runtime state-transfer decoder accepts `1 << 20` entries per section. +/// Keeping the configurable ceiling at one quarter leaves room for recovered +/// over-limit state while the shipped default remains much smaller. +pub const PARTITION_CONSUMER_OFFSETS_CEILING: usize = 1 << 18; + +fn default_consumer_offsets_max() -> usize { + PARTITION_CONSUMER_OFFSETS_DEFAULT +} + /// Capacity tunables for the per-partition consensus plane. #[derive(Debug, Deserialize, Serialize, Clone, ConfigEnv)] pub struct PartitionConfig { @@ -161,6 +178,19 @@ pub struct PartitionConfig { /// single partition actually sees, not the node's client total. pub dedup_clients_max: usize, + /// Distinct durable consumer-offset keys admitted per kind by one + /// partition primary. Existing keys remain writable at the limit. + #[serde(default = "default_consumer_offsets_max")] + pub consumer_offsets_max: usize, + + /// Whether consumer-offset files are written crash-safe: data-synced, then + /// renamed over the prior cursor, with the directory synced once per commit + /// walk. Independent of the topic's `enforce_fsync`, which governs message + /// and index files. Off, an offset file is rewritten in place with no sync. + /// A lost or torn cursor can cause replay from the earliest retained data. + #[serde(default)] + pub consumer_offset_enforce_fsync: bool, + /// Offsets claimed in the superblock ahead of the mint counter before an /// append, so a crash-restarted replica resumes above what it confirmed. /// One superblock write per block: lowering it raises the fsync rate, @@ -244,6 +274,16 @@ impl Validatable for PartitionConfig { ); return Err(ConfigurationError::InvalidConfigurationValue); } + if self.consumer_offsets_max == 0 + || self.consumer_offsets_max > PARTITION_CONSUMER_OFFSETS_CEILING + { + eprintln!( + "{COMPONENT} partition.consumer_offsets_max ({}) must be > 0 and <= \ + {PARTITION_CONSUMER_OFFSETS_CEILING}", + self.consumer_offsets_max + ); + return Err(ConfigurationError::InvalidConfigurationValue); + } if self.offset_reservation_lease.get() > MAX_OFFSET_RESERVATION_LEASE { eprintln!( "{COMPONENT} partition.offset_reservation_lease ({}) exceeds the maximum \ @@ -337,6 +377,34 @@ mod tests { ); } + #[test] + fn shipped_consumer_offsets_default_matches_constant() { + assert_eq!( + PartitionConfig::default().consumer_offsets_max, + PARTITION_CONSUMER_OFFSETS_DEFAULT + ); + } + + #[test] + fn rejects_out_of_range_consumer_offsets_max() { + for value in [0, PARTITION_CONSUMER_OFFSETS_CEILING + 1] { + let config = PartitionConfig { + consumer_offsets_max: value, + ..PartitionConfig::default() + }; + assert!(config.validate().is_err()); + } + } + + #[test] + fn accepts_consumer_offsets_max_at_ceiling() { + let config = PartitionConfig { + consumer_offsets_max: PARTITION_CONSUMER_OFFSETS_CEILING, + ..PartitionConfig::default() + }; + assert!(config.validate().is_ok()); + } + #[test] fn rejects_out_of_range_dedup_clients_max() { for value in [0, PARTITION_DEDUP_CLIENTS_CEILING + 1] { @@ -420,6 +488,10 @@ mod tests { config.offset_reservation_lease.get(), DEFAULT_OFFSET_RESERVATION_LEASE ); + assert_eq!( + config.consumer_offsets_max, + PARTITION_CONSUMER_OFFSETS_DEFAULT + ); assert!(config.validate().is_ok()); } diff --git a/core/connectors/sdk/Cargo.toml b/core/connectors/sdk/Cargo.toml index fae1b4a0ad..5d6b98daa9 100644 --- a/core/connectors/sdk/Cargo.toml +++ b/core/connectors/sdk/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "iggy_connector_sdk" -version = "0.4.0-edge.3" +version = "0.4.0-edge.4" description = "Iggy is the persistent message streaming platform written in Rust, supporting QUIC, TCP and HTTP transport protocols, capable of processing millions of messages per second." edition = "2024" license = "Apache-2.0" diff --git a/core/integration/src/harness/disk.rs b/core/integration/src/harness/disk.rs index 97d42b16c3..4b72890f70 100644 --- a/core/integration/src/harness/disk.rs +++ b/core/integration/src/harness/disk.rs @@ -26,7 +26,7 @@ use std::path::{Path, PathBuf}; use std::time::Duration; use consensus::VsrState; -use iggy::prelude::{ClusterClient, ClusterNodeRole}; +use iggy::prelude::{ClusterClient, ClusterNodeRole, ConsumerKind}; use journal::superblock::{SLOT_FILE_NAMES, SuperblockContents, decode_slots}; use tokio::time::sleep; @@ -39,6 +39,37 @@ const CONVERGENCE_POLL_INTERVAL: Duration = Duration::from_millis(200); const CONVERGENCE_DEADLINE: Duration = Duration::from_secs(20); const CONVERGENCE_STABLE_POLLS: u32 = 3; +/// Numeric regular offset files in one partition's selected kind directory. +pub fn consumer_offset_file_ids( + data_path: &Path, + stream_id: u32, + topic_id: u32, + partition_id: u32, + kind: ConsumerKind, +) -> std::io::Result> { + let kind_dir = match kind { + ConsumerKind::Consumer => "consumers", + ConsumerKind::ConsumerGroup => "groups", + }; + let dir = data_path.join(format!( + "streams/{stream_id}/topics/{topic_id}/partitions/{partition_id}/offsets/{kind_dir}" + )); + let entries = fs::read_dir(dir)?; + let mut ids = BTreeSet::new(); + for entry in entries { + let entry = entry?; + if entry.file_type()?.is_file() + && let Some(id) = entry + .file_name() + .to_str() + .and_then(|name| name.parse().ok()) + { + ids.insert(id); + } + } + Ok(ids) +} + /// A partition segment `.log`, named for its 20-digit zero-padded base offset /// (see `partitions::state_transfer`'s path builders). /// diff --git a/core/integration/tests/cluster/consumer_offset_quota.rs b/core/integration/tests/cluster/consumer_offset_quota.rs new file mode 100644 index 0000000000..3aa9a7d9d3 --- /dev/null +++ b/core/integration/tests/cluster/consumer_offset_quota.rs @@ -0,0 +1,357 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use crate::server::raw_tcp; +use iggy::prelude::*; +use iggy_binary_protocol::codec::WireEncode; +use iggy_binary_protocol::consensus::Operation; +use iggy_binary_protocol::requests::consumer_offsets::{ + DeleteConsumerOffsetRequest, StoreConsumerOffsetRequest, +}; +use iggy_binary_protocol::{AckLevel, WireConsumer, WireIdentifier}; +use integration::harness::{TestHarness, disk}; +use integration::iggy_harness; +use std::path::PathBuf; +use std::time::Duration; +use tokio::time::{Instant, sleep}; + +const STREAM_NAME: &str = "cluster-offset-quota-stream"; +const TOPIC_NAME: &str = "cluster-offset-quota-topic"; +const PARTITION_ID: u32 = 0; +const RAW_CLIENT_ID: u128 = 0xC0FF_EE02; +const WAIT: Duration = Duration::from_secs(20); + +#[iggy_harness( + cluster_nodes = 3, + server(partition.consumer_offsets_max = "4") +)] +async fn given_replicated_partition_when_no_ack_offsets_mutate_should_converge_and_stay_bounded( + harness: &mut TestHarness, +) { + let leader = disk::leader_node_index(harness).await; + let client = harness + .root_client_for_node(leader) + .await + .expect("root client"); + let stream = Identifier::named(STREAM_NAME).expect("stream identifier"); + let topic = Identifier::named(TOPIC_NAME).expect("topic identifier"); + let stream_details = client + .create_stream(STREAM_NAME) + .await + .expect("create stream"); + let topic_details = client + .create_topic( + &stream, + TOPIC_NAME, + &TopicCreateOptions { + partitions_count: Some(1), + message_expiry: Some(IggyExpiry::NeverExpire), + ..TopicCreateOptions::default() + }, + ) + .await + .expect("create topic"); + let mut messages = vec![ + IggyMessage::builder() + .payload("cluster-offset-quota".into()) + .build() + .expect("build message"), + ]; + client + .send_messages( + &stream, + &topic, + &Partitioning::partition_id(PARTITION_ID), + &mut messages, + ) + .await + .expect("seed non-empty partition"); + + let address = harness.node(leader).tcp_addr().expect("leader TCP address"); + let mut raw = raw_tcp::connect_to(address).await; + let session = raw_tcp::register_root(&mut raw, RAW_CLIENT_ID).await; + let wire_stream = WireIdentifier::Numeric(stream_details.id); + let wire_topic = WireIdentifier::Numeric(topic_details.id); + let raw_consumer_id = 41; + + let store = StoreConsumerOffsetRequest { + consumer: WireConsumer::consumer(WireIdentifier::Numeric(raw_consumer_id)), + stream_id: wire_stream.clone(), + topic_id: wire_topic.clone(), + partition_id: Some(PARTITION_ID), + offset: 0, + ack: AckLevel::NoAck, + } + .to_bytes(); + let store_header = raw_tcp::request_header( + Operation::StoreConsumerOffset, + RAW_CLIENT_ID, + session, + 1, + store.len(), + ); + let (reply, _) = raw_tcp::exchange(&mut raw, &store_header, &store).await; + assert_eq!(raw_tcp::reply_status(&reply), 0); + wait_for_file_state( + harness, + stream_details.id, + topic_details.id, + raw_consumer_id, + true, + ) + .await; + + let delete = DeleteConsumerOffsetRequest { + consumer: WireConsumer::consumer(WireIdentifier::Numeric(raw_consumer_id)), + stream_id: wire_stream, + topic_id: wire_topic, + partition_id: Some(PARTITION_ID), + ack: AckLevel::NoAck, + } + .to_bytes(); + let delete_header = raw_tcp::request_header( + Operation::DeleteConsumerOffset, + RAW_CLIENT_ID, + session, + 2, + delete.len(), + ); + let (reply, _) = raw_tcp::exchange(&mut raw, &delete_header, &delete).await; + assert_eq!(raw_tcp::reply_status(&reply), 0); + wait_for_file_state( + harness, + stream_details.id, + topic_details.id, + raw_consumer_id, + false, + ) + .await; + + let auto_commit_consumer = Consumer::new(Identifier::numeric(1).expect("consumer identifier")); + let polled = client + .poll_messages( + &stream, + &topic, + Some(PARTITION_ID), + &auto_commit_consumer, + &PollingStrategy::first(), + 1, + true, + ) + .await + .expect("auto-commit poll within limit"); + assert_eq!(polled.messages.len(), 1); + wait_for_file_state(harness, stream_details.id, topic_details.id, 1, true).await; + + for consumer_id in 1..=4 { + client + .store_consumer_offset( + &Consumer::new(Identifier::numeric(consumer_id).expect("consumer identifier")), + &stream, + &topic, + Some(PARTITION_ID), + 0, + ) + .await + .expect("store offset within limit"); + } + let rejected = client + .store_consumer_offset( + &Consumer::new(Identifier::numeric(5).expect("consumer identifier")), + &stream, + &topic, + Some(PARTITION_ID), + 0, + ) + .await; + assert!(matches!(rejected, Err(IggyError::TooManyConsumerOffsets))); + wait_for_max_file_count(harness, stream_details.id, topic_details.id, 4).await; + + let backup = (0..harness.cluster_size()) + .find(|node| *node != leader) + .expect("backup replica"); + harness + .kill_node(backup) + .expect("pause one replica during group churn"); + for generation in 0..6 { + let name = format!("quota-group-{generation}"); + client + .create_consumer_group(&stream, &topic, &name) + .await + .expect("create group"); + let group = Identifier::named(&name).expect("group identifier"); + client + .join_consumer_group(&stream, &topic, &group) + .await + .expect("join group"); + client + .store_consumer_offset( + &Consumer::group(group.clone()), + &stream, + &topic, + Some(PARTITION_ID), + 0, + ) + .await + .expect("store valid group offset"); + client + .delete_consumer_group(&stream, &topic, &group) + .await + .expect("delete group metadata"); + let deadline = Instant::now() + WAIT; + loop { + let ids = disk::consumer_offset_file_ids( + &harness.node(leader).data_path(), + stream_details.id, + topic_details.id, + PARTITION_ID, + ConsumerKind::ConsumerGroup, + ) + .expect("group offset directory"); + if ids.is_empty() { + break; + } + assert!( + Instant::now() < deadline, + "replicated cleanup did not delete {ids:?}" + ); + sleep(Duration::from_millis(50)).await; + } + } + harness + .restart_node(backup) + .expect("restart lagging replica"); + client + .send_messages( + &stream, + &topic, + &Partitioning::partition_id(PARTITION_ID), + &mut messages, + ) + .await + .expect("produce after rejoin"); + client + .store_consumer_offset( + &Consumer::new(Identifier::numeric(1).unwrap()), + &stream, + &topic, + Some(PARTITION_ID), + 1, + ) + .await + .expect("store convergence marker"); + let deadline = Instant::now() + WAIT; + loop { + let caught_up = (0..harness.cluster_size()).all(|node| { + std::fs::read(offset_file( + harness, + node, + stream_details.id, + topic_details.id, + 1, + )) + .ok() + .and_then(|bytes| bytes.first_chunk::<8>().copied()) + .map(u64::from_le_bytes) + == Some(1) + }); + if caught_up { + break; + } + assert!( + Instant::now() < deadline, + "replica never applied the post-cleanup marker" + ); + sleep(Duration::from_millis(50)).await; + } + for node in 0..harness.cluster_size() { + assert!( + disk::consumer_offset_file_ids( + &harness.node(node).data_path(), + stream_details.id, + topic_details.id, + PARTITION_ID, + ConsumerKind::ConsumerGroup + ) + .expect("rejoined group directory") + .is_empty(), + "node {node} retained deleted group generations" + ); + } +} + +async fn wait_for_file_state( + harness: &TestHarness, + stream_id: u32, + topic_id: u32, + consumer_id: u32, + expected: bool, +) { + let deadline = Instant::now() + WAIT; + loop { + let states: Vec = (0..harness.cluster_size()) + .map(|node| offset_file(harness, node, stream_id, topic_id, consumer_id).exists()) + .collect(); + if states.iter().all(|state| *state == expected) { + return; + } + assert!( + Instant::now() < deadline, + "consumer offset file state did not converge to {expected}: {states:?}" + ); + sleep(Duration::from_millis(100)).await; + } +} + +async fn wait_for_max_file_count(harness: &TestHarness, stream_id: u32, topic_id: u32, max: usize) { + let deadline = Instant::now() + WAIT; + loop { + let counts: Vec = (0..harness.cluster_size()) + .map(|node| { + disk::consumer_offset_file_ids( + &harness.node(node).data_path(), + stream_id, + topic_id, + PARTITION_ID, + ConsumerKind::Consumer, + ) + .expect("consumer offset directory") + .len() + }) + .collect(); + if counts.iter().all(|count| *count == max) { + return; + } + assert!( + Instant::now() < deadline, + "consumer offset files did not converge at {max}: {counts:?}" + ); + sleep(Duration::from_millis(100)).await; + } +} + +fn offset_file( + harness: &TestHarness, + node: usize, + stream_id: u32, + topic_id: u32, + consumer_id: u32, +) -> PathBuf { + harness.node(node).data_path().join(format!( + "streams/{stream_id}/topics/{topic_id}/partitions/{PARTITION_ID}/offsets/consumers/{consumer_id}" + )) +} diff --git a/core/integration/tests/cluster/mod.rs b/core/integration/tests/cluster/mod.rs index 57359759f3..e9f81fb936 100644 --- a/core/integration/tests/cluster/mod.rs +++ b/core/integration/tests/cluster/mod.rs @@ -17,6 +17,7 @@ mod client_table_adversarial; mod client_table_restart; +mod consumer_offset_quota; mod crash_durability; mod crash_offset_reuse; mod crash_recovery_corruption; diff --git a/core/integration/tests/server/consumer_offset_quota_vsr.rs b/core/integration/tests/server/consumer_offset_quota_vsr.rs new file mode 100644 index 0000000000..ef454b81e9 --- /dev/null +++ b/core/integration/tests/server/consumer_offset_quota_vsr.rs @@ -0,0 +1,514 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use iggy::prelude::*; +use iggy_binary_protocol::codec::WireEncode; +use iggy_binary_protocol::consensus::Operation; +use iggy_binary_protocol::requests::consumer_offsets::StoreConsumerOffsetRequest; +use iggy_binary_protocol::{AckLevel, WireConsumer, WireIdentifier}; +use iggy_common::store_consumer_offset::StoreConsumerOffset; +use integration::harness::TestBinary; +use integration::iggy_harness; +use reqwest::StatusCode; +use std::collections::BTreeMap; +use std::fs; + +use super::http_client::HttpClient; +use super::raw_tcp; + +const STREAM_NAME: &str = "consumer-offset-quota-stream"; +const TOPIC_NAME: &str = "consumer-offset-quota-topic"; +const PARTITION_ID: u32 = 0; +const LIMIT: u32 = 4; + +#[iggy_harness( + cluster_nodes = 1, + server(partition.consumer_offsets_max = "4") +)] +async fn given_full_consumer_offset_table_when_creating_another_should_reject_without_new_file( + harness: &TestHarness, +) { + let client = harness.tcp_root_client().await.expect("TCP root client"); + let stream = Identifier::named(STREAM_NAME).expect("stream identifier"); + let topic = Identifier::named(TOPIC_NAME).expect("topic identifier"); + let stream_details = client + .create_stream(STREAM_NAME) + .await + .expect("create stream"); + let topic_details = client + .create_topic( + &stream, + TOPIC_NAME, + &TopicCreateOptions { + partitions_count: Some(1), + message_expiry: Some(IggyExpiry::NeverExpire), + ..TopicCreateOptions::default() + }, + ) + .await + .expect("create topic"); + let mut messages = vec![ + IggyMessage::builder() + .payload("offset-quota".into()) + .build() + .expect("build message"), + ]; + client + .send_messages( + &stream, + &topic, + &Partitioning::partition_id(PARTITION_ID), + &mut messages, + ) + .await + .expect("seed non-empty partition"); + + client + .create_user( + "offset-poll-only", + "password123", + UserStatus::Active, + Some(Permissions { + global: GlobalPermissions::default(), + streams: Some(BTreeMap::from([( + stream_details.id as usize, + StreamPermissions { + topics: Some(BTreeMap::from([( + topic_details.id as usize, + TopicPermissions { + poll_messages: true, + ..Default::default() + }, + )])), + ..Default::default() + }, + )])), + }), + ) + .await + .expect("create a topic-scoped consumer"); + let client = harness.tcp_new_client().await.expect("consumer TCP client"); + client + .login_user("offset-poll-only", "password123") + .await + .expect("consumer login"); + + let first_consumer = Consumer::new(Identifier::numeric(1).unwrap()); + let polled = client + .poll_messages( + &stream, + &topic, + Some(PARTITION_ID), + &first_consumer, + &PollingStrategy::first(), + 1, + true, + ) + .await + .expect("new auto-commit consumer fits"); + assert_eq!(polled.messages.len(), 1); + let first_file = harness.server().data_path().join(format!( + "streams/{}/topics/{}/partitions/{PARTITION_ID}/offsets/consumers/1", + stream_details.id, topic_details.id + )); + let deadline = tokio::time::Instant::now() + std::time::Duration::from_secs(10); + while !first_file.is_file() { + assert!( + tokio::time::Instant::now() < deadline, + "auto-commit never reached its file" + ); + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + assert!( + client + .poll_messages( + &stream, + &topic, + Some(PARTITION_ID), + &first_consumer, + &PollingStrategy::next(), + 1, + true + ) + .await + .expect("next poll") + .messages + .is_empty() + ); + + for consumer_id in 1..=LIMIT { + client + .store_consumer_offset( + &Consumer::new(Identifier::numeric(consumer_id).expect("consumer identifier")), + &stream, + &topic, + Some(PARTITION_ID), + 0, + ) + .await + .expect("store offset within limit"); + } + + let rejected_consumer = + Consumer::new(Identifier::numeric(LIMIT + 1).expect("consumer identifier")); + let rejected = client + .store_consumer_offset(&rejected_consumer, &stream, &topic, Some(PARTITION_ID), 0) + .await; + assert!( + matches!(rejected, Err(IggyError::TooManyConsumerOffsets)), + "the first key above the limit must receive the typed capacity error" + ); + + client + .store_consumer_offset( + &Consumer::new(Identifier::numeric(1).expect("consumer identifier")), + &stream, + &topic, + Some(PARTITION_ID), + 0, + ) + .await + .expect("existing key remains writable at the limit"); + + let poll_rejected = client + .poll_messages( + &stream, + &topic, + Some(PARTITION_ID), + &rejected_consumer, + &PollingStrategy::first(), + 1, + true, + ) + .await; + assert!( + matches!(poll_rejected, Err(IggyError::TooManyConsumerOffsets)), + "auto-commit must not return data when its new key cannot be admitted" + ); + client + .poll_messages( + &stream, + &topic, + Some(PARTITION_ID), + &rejected_consumer, + &PollingStrategy::first(), + 1, + false, + ) + .await + .expect("the same poll succeeds when auto-commit is disabled"); + + client + .delete_consumer_offset( + &Consumer::new(Identifier::numeric(1).expect("consumer identifier")), + &stream, + &topic, + Some(PARTITION_ID), + ) + .await + .expect("delete one accepted offset"); + client + .store_consumer_offset(&rejected_consumer, &stream, &topic, Some(PARTITION_ID), 0) + .await + .expect("delete releases one durable slot"); + + let mut raw = raw_tcp::connect(harness).await; + let raw_client_id = 0xC0FF_EE03; + let session = raw_tcp::register_root(&mut raw, raw_client_id).await; + let unresolved_group = StoreConsumerOffsetRequest { + consumer: WireConsumer::consumer_group(WireIdentifier::Numeric(999)), + stream_id: WireIdentifier::Numeric(stream_details.id), + topic_id: WireIdentifier::Numeric(topic_details.id), + partition_id: Some(PARTITION_ID), + offset: 0, + ack: AckLevel::Quorum, + } + .to_bytes(); + let header = raw_tcp::request_header( + Operation::StoreConsumerOffset, + raw_client_id, + session, + 1, + unresolved_group.len(), + ); + let (reply, _) = raw_tcp::exchange(&mut raw, &header, &unresolved_group).await; + assert_eq!( + raw_tcp::reply_status(&reply), + IggyError::ConsumerGroupIdNotFound(Identifier::numeric(999).unwrap(), topic.clone()) + .as_code() + ); + + let file_count = integration::harness::disk::consumer_offset_file_ids( + &harness.server().data_path(), + stream_details.id, + topic_details.id, + PARTITION_ID, + ConsumerKind::Consumer, + ) + .expect("consumer offsets directory") + .len(); + assert_eq!(file_count, LIMIT as usize); + let group_file_count = integration::harness::disk::consumer_offset_file_ids( + &harness.server().data_path(), + stream_details.id, + topic_details.id, + PARTITION_ID, + ConsumerKind::ConsumerGroup, + ) + .map_or(0, |ids| ids.len()); + assert_eq!(group_file_count, 0); + + let named_group = StoreConsumerOffsetRequest { + consumer: WireConsumer::consumer_group(WireIdentifier::named("unknown-group").unwrap()), + stream_id: WireIdentifier::Numeric(stream_details.id), + topic_id: WireIdentifier::Numeric(topic_details.id), + partition_id: Some(PARTITION_ID), + offset: 0, + ack: AckLevel::Quorum, + } + .to_bytes(); + let header = raw_tcp::request_header( + Operation::StoreConsumerOffset, + raw_client_id, + session, + 2, + named_group.len(), + ); + let (reply, _) = raw_tcp::exchange(&mut raw, &header, &named_group).await; + assert_eq!( + raw_tcp::reply_status(&reply), + IggyError::ConsumerGroupNameNotFound("unknown-group".to_owned(), topic.clone()).as_code() + ); + + let unknown_stream = StoreConsumerOffsetRequest { + consumer: WireConsumer::consumer_group(WireIdentifier::Numeric(999)), + stream_id: WireIdentifier::Numeric(999_999), + topic_id: WireIdentifier::Numeric(topic_details.id), + partition_id: Some(PARTITION_ID), + offset: 0, + ack: AckLevel::Quorum, + } + .to_bytes(); + let header = raw_tcp::request_header( + Operation::StoreConsumerOffset, + raw_client_id, + session, + 3, + unknown_stream.len(), + ); + let (reply, _) = raw_tcp::exchange(&mut raw, &header, &unknown_stream).await; + assert_eq!( + raw_tcp::reply_status(&reply), + IggyError::ResourceNotFound(String::new()).as_code(), + "a missing stream is not reported as a missing group" + ); + + let http = HttpClient::login_root(harness).await; + let response = http + .client + .put(http.url(&format!( + "/streams/{STREAM_NAME}/topics/{TOPIC_NAME}/consumer-offsets" + ))) + .bearer_auth(&http.token) + .json(&StoreConsumerOffset { + consumer: Consumer::new(Identifier::numeric(6).expect("consumer identifier")), + partition_id: Some(PARTITION_ID), + offset: 0, + }) + .send() + .await + .expect("HTTP capacity request"); + assert_eq!(response.status(), StatusCode::BAD_REQUEST); + let body: serde_json::Value = response.json().await.expect("HTTP error body"); + assert_eq!(body["id"], 3024); + assert_eq!(body["code"], "too_many_consumer_offsets"); + + let metrics = http + .client + .get(http.url("/metrics")) + .bearer_auth(&http.token) + .send() + .await + .expect("metrics response") + .text() + .await + .expect("metrics text"); + let denied_for = |kind: &str| -> u64 { + let kind_label = format!("kind=\"{kind}\""); + let mut samples = 0; + let total = metrics + .lines() + .filter_map(|line| { + line.strip_prefix("partition_consumer_offsets_denied_total{")? + .split_once('}') + }) + .filter(|(labels, _)| labels.split(',').any(|label| label == kind_label)) + .map(|(_, value)| { + samples += 1; + value + .split_whitespace() + .last() + .expect("counter value") + .parse::() + .expect("numeric counter") + }) + .sum(); + assert!( + samples > 0, + "missing {kind} denial metric in response: {metrics}" + ); + total + }; + assert_eq!( + denied_for("consumer"), + 3, + "one explicit TCP, one poll, and one HTTP denial, all on the consumer kind" + ); + assert_eq!( + denied_for("consumer_group"), + 0, + "no consumer group offset was denied in this test" + ); +} + +#[iggy_harness( + cluster_nodes = 1, + server(partition.consumer_offsets_max = "2") +)] +async fn given_full_consumer_offset_table_when_server_restarts_should_preserve_admission_state( + harness: &mut TestHarness, +) { + let client = harness.tcp_root_client().await.expect("TCP root client"); + let stream = Identifier::named("consumer-offset-restart-stream").expect("stream identifier"); + let topic = Identifier::named("consumer-offset-restart-topic").expect("topic identifier"); + let stream_details = client + .create_stream("consumer-offset-restart-stream") + .await + .expect("create stream"); + let topic_details = client + .create_topic( + &stream, + "consumer-offset-restart-topic", + &TopicCreateOptions { + partitions_count: Some(1), + message_expiry: Some(IggyExpiry::NeverExpire), + ..TopicCreateOptions::default() + }, + ) + .await + .expect("create topic"); + let mut messages = vec![ + IggyMessage::builder() + .payload("offset-restart".into()) + .build() + .expect("build message"), + ]; + client + .send_messages( + &stream, + &topic, + &Partitioning::partition_id(PARTITION_ID), + &mut messages, + ) + .await + .expect("seed non-empty partition"); + for consumer_id in 1..=2 { + client + .store_consumer_offset( + &Consumer::new(Identifier::numeric(consumer_id).expect("consumer identifier")), + &stream, + &topic, + Some(PARTITION_ID), + 0, + ) + .await + .expect("store offset before restart"); + } + drop(client); + harness.server_mut().stop().expect("stop server"); + let offsets_dir = harness.server().data_path().join(format!( + "streams/{}/topics/{}/partitions/{PARTITION_ID}/offsets/consumers", + stream_details.id, topic_details.id + )); + fs::copy(offsets_dir.join("1"), offsets_dir.join("3")) + .expect("create first historical over-limit offset"); + fs::copy(offsets_dir.join("1"), offsets_dir.join("4")) + .expect("create second historical over-limit offset"); + harness.server_mut().start().expect("restart server"); + let client = harness + .root_client() + .await + .expect("post-restart root client"); + + let rejected = client + .store_consumer_offset( + &Consumer::new(Identifier::numeric(5).expect("consumer identifier")), + &stream, + &topic, + Some(PARTITION_ID), + 0, + ) + .await; + assert!(matches!(rejected, Err(IggyError::TooManyConsumerOffsets))); + client + .store_consumer_offset( + &Consumer::new(Identifier::numeric(4).expect("consumer identifier")), + &stream, + &topic, + Some(PARTITION_ID), + 0, + ) + .await + .expect("recovered existing key remains writable"); + client + .delete_consumer_offset( + &Consumer::new(Identifier::numeric(4).expect("consumer identifier")), + &stream, + &topic, + Some(PARTITION_ID), + ) + .await + .expect("delete recovered key"); + client + .delete_consumer_offset( + &Consumer::new(Identifier::numeric(3).expect("consumer identifier")), + &stream, + &topic, + Some(PARTITION_ID), + ) + .await + .expect("delete second historical key"); + client + .delete_consumer_offset( + &Consumer::new(Identifier::numeric(2).expect("consumer identifier")), + &stream, + &topic, + Some(PARTITION_ID), + ) + .await + .expect("delete below configured limit"); + client + .store_consumer_offset( + &Consumer::new(Identifier::numeric(5).expect("consumer identifier")), + &stream, + &topic, + Some(PARTITION_ID), + 0, + ) + .await + .expect("deleting below the limit releases a slot"); +} diff --git a/core/integration/tests/server/mod.rs b/core/integration/tests/server/mod.rs index e32f2c5ab4..85a690696f 100644 --- a/core/integration/tests/server/mod.rs +++ b/core/integration/tests/server/mod.rs @@ -78,6 +78,7 @@ mod partition_view_durability_vsr; // 80-case race matrix with hardcoded HTTP variants (test_matrix bypasses // the harness transport filter). mod concurrent_addition; +mod consumer_offset_quota_vsr; mod general; // The per-shard segment cleaner deletes expired / oversize segments from disk. mod message_cleanup; diff --git a/core/integration/tests/server/raw_tcp.rs b/core/integration/tests/server/raw_tcp.rs index cfe5d8c698..8711a60595 100644 --- a/core/integration/tests/server/raw_tcp.rs +++ b/core/integration/tests/server/raw_tcp.rs @@ -21,6 +21,7 @@ //! can ride a bound session. use std::mem::offset_of; +use std::net::SocketAddr; use std::time::Duration; use iggy::prelude::*; @@ -55,6 +56,10 @@ pub(crate) async fn connect(harness: &TestHarness) -> TcpStream { .server() .tcp_addr() .expect("server must expose a TCP address"); + connect_to(addr).await +} + +pub(crate) async fn connect_to(addr: SocketAddr) -> TcpStream { TcpStream::connect(addr).await.unwrap() } diff --git a/core/message_bus/src/lib.rs b/core/message_bus/src/lib.rs index 74836bb8e8..d8038b6126 100644 --- a/core/message_bus/src/lib.rs +++ b/core/message_bus/src/lib.rs @@ -797,7 +797,7 @@ impl IggyMessageBus { } /// Construct a bus with explicit runtime tunables and a pre-allocated - /// owner table. Server-ng bootstrap uses this so every shard's bus + /// owner table. Server bootstrap uses this so every shard's bus /// shares the same atomic slots; tests use [`Self::with_tunables`] /// which allocates a fresh table per bus. /// diff --git a/core/metadata/src/impls/metadata.rs b/core/metadata/src/impls/metadata.rs index 9fbd167cfd..3c4d19f47d 100644 --- a/core/metadata/src/impls/metadata.rs +++ b/core/metadata/src/impls/metadata.rs @@ -693,7 +693,7 @@ pub fn apply_committed_prepare( pub type CommitNotifier = std::rc::Rc; pub struct IggyMetadata { - /// `Some` on shard 0, `None` on other shards. Server-ng bootstrap + /// `Some` on shard 0, `None` on other shards. Server bootstrap /// holds the invariant: only shard 0 owns the metadata consensus /// replica; every other shard reconstructs `mux_stm` from the /// `MetadataHandoff::Waiter` factory bundle broadcast by shard 0 @@ -928,7 +928,7 @@ impl IggyMetadata { } /// Install (or replace) the post-commit notifier. Passing `None` - /// removes any previous one. Server-ng bootstrap calls this on shard 0 + /// removes any previous one. Server bootstrap calls this on shard 0 /// only; peer shards never commit metadata locally. pub fn set_commit_notifier(&self, notifier: Option) { *self.commit_notifier.borrow_mut() = notifier; @@ -937,7 +937,7 @@ impl IggyMetadata { /// Seed the coordinator's last-checkpoint pairing at boot from the recovered /// snapshot, so the first post-boot view-change superblock write records the real /// `(checkpoint_op, checksum)` instead of `(0, 0)`. No-op without a coordinator - /// (peer shards, the simulator). Server-ng bootstrap calls this on shard 0 after + /// (peer shards, the simulator). Server bootstrap calls this on shard 0 after /// cross-checking the pairing. pub fn seed_checkpoint_ref(&self, checkpoint_op: u64, checkpoint_checksum: u128) { if let Some(coordinator) = &self.coordinator { diff --git a/core/metadata/src/stm/user.rs b/core/metadata/src/stm/user.rs index 1e79d78532..c78ea7507b 100644 --- a/core/metadata/src/stm/user.rs +++ b/core/metadata/src/stm/user.rs @@ -642,14 +642,14 @@ impl StateHandler for UpdatePermissionsRequest { /// The success reply here is deliberately empty: the raw token the caller needs /// is the one thing this apply must never see. /// -/// The primary mints the raw token and its hash at ingress (server-ng +/// The primary mints the raw token and its hash at ingress (server /// `pat::rewrite_pat_request_for_user`) and replicates only the hash. Minting /// inside this apply would call `ring::rand` on every replica and diverge the /// token index, and replicating the raw token would persist a live credential in /// every WAL and snapshot. So the raw token leaves the primary by a side channel /// (`maybe_rewrite_pat_request` returns it alongside the rewritten request) and /// the home shard splices it into this op's reply as a typed -/// `RawPersonalAccessTokenResponse` (server-ng `responses::build_raw_pat_reply`). +/// `RawPersonalAccessTokenResponse` (server `responses::build_raw_pat_reply`). /// /// One consequence rides on that: the secret exists only on the wire of the /// original reply, so a replayed request cannot be served from the client-table diff --git a/core/partitions/src/consumer_offset_capacity.rs b/core/partitions/src/consumer_offset_capacity.rs new file mode 100644 index 0000000000..aaaa932087 --- /dev/null +++ b/core/partitions/src/consumer_offset_capacity.rs @@ -0,0 +1,717 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use iggy_common::ConsumerKind; +use std::cell::{Cell, RefCell}; +use std::collections::hash_map::Entry; +use std::collections::{HashMap, HashSet}; +use std::rc::Rc; +use std::sync::Arc; +use std::sync::atomic::{AtomicU64, AtomicUsize, Ordering}; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct DurableOffsetState { + pub(crate) committed_offset: u64, + pub(crate) persisted_high_water: u64, +} + +#[derive(Debug, Default)] +pub struct DurableConsumerOffsets { + consumers: RefCell>, + groups: RefCell>, + membership_epoch: Cell, +} + +impl DurableConsumerOffsets { + pub(crate) fn get(&self, kind: ConsumerKind, id: u32) -> Option { + self.entries(kind).borrow().get(&id).copied() + } + + pub(crate) fn contains(&self, kind: ConsumerKind, id: u32) -> bool { + self.entries(kind).borrow().contains_key(&id) + } + + pub(crate) fn count(&self, kind: ConsumerKind) -> usize { + self.entries(kind).borrow().len() + } + + pub(crate) fn covers(&self, kind: ConsumerKind, id: u32, offset: u64) -> bool { + self.get(kind, id).is_some_and(|state| { + state.committed_offset >= offset && state.persisted_high_water >= offset + }) + } + + pub(crate) fn record_explicit( + &self, + kind: ConsumerKind, + id: u32, + committed_offset: u64, + persisted_high_water: u64, + ) -> bool { + let created = self + .entries(kind) + .borrow_mut() + .insert( + id, + DurableOffsetState { + committed_offset, + persisted_high_water, + }, + ) + .is_none(); + if created { + self.bump_membership_epoch(); + } + created + } + + pub(crate) fn record_auto_commit( + &self, + kind: ConsumerKind, + id: u32, + committed_offset: u64, + persisted_high_water: u64, + ) { + let mut entries = self.entries(kind).borrow_mut(); + match entries.entry(id) { + Entry::Occupied(mut entry) => { + let state = entry.get_mut(); + state.committed_offset = state.committed_offset.max(committed_offset); + state.persisted_high_water = state.persisted_high_water.max(persisted_high_water); + } + Entry::Vacant(entry) => { + entry.insert(DurableOffsetState { + committed_offset, + persisted_high_water, + }); + drop(entries); + self.bump_membership_epoch(); + } + } + } + + pub(crate) fn remove(&self, kind: ConsumerKind, id: u32) -> bool { + let removed = self.entries(kind).borrow_mut().remove(&id).is_some(); + if removed { + self.bump_membership_epoch(); + } + removed + } + + pub(crate) fn clear(&self) { + self.consumers.borrow_mut().clear(); + self.groups.borrow_mut().clear(); + self.bump_membership_epoch(); + } + + #[cfg(any(test, feature = "simulator"))] + pub(crate) fn committed_entries(&self, kind: ConsumerKind) -> Vec<(u32, u64)> { + self.entries(kind) + .borrow() + .iter() + .map(|(id, state)| (*id, state.committed_offset)) + .collect() + } + + pub(crate) fn with_entries( + &self, + kind: ConsumerKind, + read: impl FnOnce(&HashMap) -> T, + ) -> T { + read(&self.entries(kind).borrow()) + } + + const fn entries(&self, kind: ConsumerKind) -> &RefCell> { + match kind { + ConsumerKind::Consumer => &self.consumers, + ConsumerKind::ConsumerGroup => &self.groups, + } + } + + fn bump_membership_epoch(&self) { + self.membership_epoch + .set(self.membership_epoch.get().wrapping_add(1)); + } +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct ConsumerOffsetCapacityError { + pub kind: ConsumerKind, + pub occupied: usize, + pub limit: usize, + pub first_in_episode: bool, + pub uncertain: bool, +} + +impl From for iggy_common::IggyError { + fn from(error: ConsumerOffsetCapacityError) -> Self { + if error.uncertain { + Self::TransientNotAccepted + } else { + Self::TooManyConsumerOffsets + } + } +} + +#[derive(Debug)] +pub struct ConsumerOffsetCapacity { + kind: ConsumerKind, + limit: Cell, + pending: RefCell>, + provisional: RefCell>>, + active_provisional_keys: Arc, + stranded: RefCell>, + uncertain: Cell, + durable_warned: Cell, + map_warned: Cell, + reclaim_epoch: Arc, + last_reclaim: Cell>, +} + +impl ConsumerOffsetCapacity { + pub(crate) fn new(kind: ConsumerKind, limit: usize) -> Self { + Self { + kind, + limit: Cell::new(limit), + pending: RefCell::new(HashMap::new()), + provisional: RefCell::new(HashMap::new()), + active_provisional_keys: Arc::new(AtomicUsize::new(0)), + stranded: RefCell::new(HashSet::new()), + uncertain: Cell::new(false), + durable_warned: Cell::new(false), + map_warned: Cell::new(false), + reclaim_epoch: Arc::new(AtomicU64::new(0)), + last_reclaim: Cell::new(None), + } + } + + pub(crate) fn set_limit(&self, limit: usize) { + self.limit.set(limit); + } + + pub(crate) const fn limit(&self) -> usize { + self.limit.get() + } + + pub(crate) fn try_reserve( + &self, + id: u32, + durable: &DurableConsumerOffsets, + ) -> Result<(), ConsumerOffsetCapacityError> { + self.check(id, durable)?; + *self.pending.borrow_mut().entry(id).or_default() += 1; + Ok(()) + } + + pub(crate) fn check( + &self, + id: u32, + durable: &DurableConsumerOffsets, + ) -> Result<(), ConsumerOffsetCapacityError> { + if self.holds(id, durable) || self.stranded.borrow().contains(&id) { + return Ok(()); + } + let limit = self.limit.get(); + let durable_count = durable.count(self.kind); + let fixed = durable_count + .saturating_add(self.pending.borrow().len()) + .saturating_add(self.stranded.borrow().len()); + let provisional_len = self.active_provisional_keys.load(Ordering::Relaxed); + let upper_bound = fixed.saturating_add(provisional_len); + if !self.uncertain.get() && upper_bound < limit { + self.durable_warned.set(false); + return Ok(()); + } + // A full durable table cannot gain room by pruning provisional keys. + let occupied = if durable_count >= limit { + durable_count + } else { + self.occupied(durable) + }; + if self.uncertain.get() || occupied >= limit { + return Err(ConsumerOffsetCapacityError { + kind: self.kind, + occupied, + limit, + first_in_episode: !self.durable_warned.replace(true), + uncertain: self.uncertain.get(), + }); + } + self.durable_warned.set(false); + Ok(()) + } + + pub(crate) fn reserve_provisional( + self: &Rc, + id: u32, + durable: &Rc, + ) -> Result { + self.check(id, durable)?; + let mut provisional = self.provisional.borrow_mut(); + if provisional.len() >= self.limit.get() && !provisional.contains_key(&id) { + provisional.retain(|_, token| token.active.load(Ordering::Relaxed) > 0); + } + let token = Arc::clone(provisional.entry(id).or_insert_with(|| { + Arc::new(ProvisionalToken { + reclaim_epoch: Arc::clone(&self.reclaim_epoch), + active_keys: Arc::clone(&self.active_provisional_keys), + active: AtomicUsize::new(0), + }) + })); + if token.active.fetch_add(1, Ordering::Relaxed) == 0 { + token.active_keys.fetch_add(1, Ordering::Relaxed); + } + Ok(AutoCommitReservation { + token, + kind: self.kind, + consumer_id: id, + }) + } + + pub(crate) fn owns(&self, reservation: &AutoCommitReservation) -> bool { + reservation.kind == self.kind + && self + .provisional + .borrow() + .get(&reservation.consumer_id) + .is_some_and(|token| Arc::ptr_eq(token, &reservation.token)) + } + + pub(crate) fn holds(&self, id: u32, durable: &DurableConsumerOffsets) -> bool { + durable.contains(self.kind, id) + || self.pending.borrow().contains_key(&id) + || self + .provisional + .borrow() + .get(&id) + .is_some_and(|token| token.active.load(Ordering::Relaxed) > 0) + } + + /// Assigns the pending count outright while [`Self::release_reservation`] + /// decrements it. Both take `&self` and neither locks: they are serialized + /// by their call sites, which all run under the partition's `&mut self` on + /// its own shard thread. + pub(crate) fn set_pending_count(&self, id: u32, count: usize) { + if count == 0 { + if self.pending.borrow_mut().remove(&id).is_some() { + self.note_local_key_change(); + } + } else { + self.pending.borrow_mut().insert(id, count); + } + } + + /// See [`Self::set_pending_count`] for the serialization contract. + pub(crate) fn release_reservation(&self, id: u32) { + let mut pending = self.pending.borrow_mut(); + let Some(count) = pending.get_mut(&id) else { + return; + }; + if *count == 1 { + pending.remove(&id); + self.note_local_key_change(); + } else { + *count -= 1; + } + } + + pub(crate) const fn is_uncertain(&self) -> bool { + self.uncertain.get() + } + + pub(crate) fn rebuild( + &self, + durable: &DurableConsumerOffsets, + pending_ids: impl IntoIterator, + ) { + let mut pending = self.pending.borrow_mut(); + pending.clear(); + for id in pending_ids { + *pending.entry(id).or_default() += 1; + } + drop(pending); + self.note_local_key_change(); + self.uncertain.set(false); + self.rearm_if_below_limit(durable); + } + + pub(crate) fn mark_uncertain(&self) { + self.pending.borrow_mut().clear(); + self.uncertain.set(true); + self.note_local_key_change(); + } + + pub(crate) fn record_stranded(&self, id: u32) { + self.stranded.borrow_mut().insert(id); + } + + pub(crate) fn clear_stranded(&self, id: u32) { + self.stranded.borrow_mut().remove(&id); + } + + pub(crate) fn is_stranded(&self, id: u32) -> bool { + self.stranded.borrow().contains(&id) + } + + /// Keys whose file could not be loaded or unlinked. Cleared only by a + /// later store or delete of the same key, never by `rebuild` or + /// `mark_uncertain`, so a permanently unwritable file keeps this above + /// zero. Exported as a gauge so that refusal has a signal. + pub(crate) fn stranded_count(&self) -> usize { + self.stranded.borrow().len() + } + + pub(crate) fn rearm_if_below_limit(&self, durable: &DurableConsumerOffsets) { + if !self.durable_warned.get() + || self.uncertain.get() + || durable.count(self.kind) >= self.limit.get() + { + return; + } + if self.occupied(durable) < self.limit.get() { + self.durable_warned.set(false); + } + } + + pub(crate) fn admit_local_map_key( + &self, + map_len: usize, + durable_full: bool, + ) -> Result<(), ConsumerOffsetCapacityError> { + let limit = self.limit.get(); + if map_len < limit { + self.map_warned.set(false); + return Ok(()); + } + Err(ConsumerOffsetCapacityError { + kind: self.kind, + occupied: map_len, + limit, + first_in_episode: !self.map_warned.replace(true), + uncertain: !durable_full, + }) + } + + pub(crate) fn note_local_key_change(&self) { + self.reclaim_epoch.fetch_add(1, Ordering::Relaxed); + } + + pub(crate) fn forget_inactive_provisional(&self, id: u32) { + let mut provisional = self.provisional.borrow_mut(); + if provisional + .get(&id) + .is_some_and(|token| token.active.load(Ordering::Relaxed) == 0) + { + provisional.remove(&id); + } + } + + pub(crate) fn should_reclaim(&self, durable: &DurableConsumerOffsets) -> bool { + if self.uncertain.get() { + return false; + } + let epoch = ( + self.reclaim_epoch.load(Ordering::Relaxed), + durable.membership_epoch.get(), + ); + self.last_reclaim.replace(Some(epoch)) != Some(epoch) + } + + /// Keys this kind holds: the durable set plus every pending, active + /// provisional or stranded key the durable set does not already count. + pub(crate) fn occupied(&self, durable: &DurableConsumerOffsets) -> usize { + let pending = self.pending.borrow(); + let provisional = self.provisional.borrow(); + let stranded = self.stranded.borrow(); + let mut local: HashSet = pending.keys().copied().collect(); + local.extend( + provisional + .iter() + .filter(|(_, token)| token.active.load(Ordering::Relaxed) > 0) + .map(|(id, _)| *id), + ); + local.extend(stranded.iter().copied()); + durable.count(self.kind) + + local + .into_iter() + .filter(|id| !durable.contains(self.kind, *id)) + .count() + } +} + +/// Keeps a provisional key occupied until the pump admits or drops its request. +#[derive(Debug)] +pub struct AutoCommitReservation { + token: Arc, + pub(crate) kind: ConsumerKind, + pub(crate) consumer_id: u32, +} + +#[derive(Debug)] +struct ProvisionalToken { + reclaim_epoch: Arc, + active_keys: Arc, + active: AtomicUsize, +} + +impl Drop for AutoCommitReservation { + fn drop(&mut self) { + if self.token.active.fetch_sub(1, Ordering::Relaxed) == 1 { + self.token.active_keys.fetch_sub(1, Ordering::Relaxed); + self.token.reclaim_epoch.fetch_add(1, Ordering::Relaxed); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn given_cached_inactive_tokens_when_admitting_should_count_only_active_keys() { + let durable = Rc::new(DurableConsumerOffsets::default()); + let capacity = Rc::new(ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 2)); + let first = capacity.reserve_provisional(1, &durable).unwrap(); + let repeated = capacity.reserve_provisional(1, &durable).unwrap(); + assert_eq!(capacity.active_provisional_keys.load(Ordering::Relaxed), 1); + drop(first); + assert_eq!(capacity.active_provisional_keys.load(Ordering::Relaxed), 1); + drop(repeated); + assert_eq!(capacity.active_provisional_keys.load(Ordering::Relaxed), 0); + assert_eq!(capacity.provisional.borrow().len(), 1); + capacity.check(2, &durable).unwrap(); + let second = capacity.reserve_provisional(2, &durable).unwrap(); + assert_eq!(capacity.active_provisional_keys.load(Ordering::Relaxed), 1); + drop(second); + capacity.check(3, &durable).unwrap(); + assert_eq!(capacity.active_provisional_keys.load(Ordering::Relaxed), 0); + } + + #[test] + fn given_low_occupancy_when_accounting_is_uncertain_should_reject_new_keys() { + let durable = DurableConsumerOffsets::default(); + let capacity = ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 100); + capacity.try_reserve(1, &durable).unwrap(); + capacity.mark_uncertain(); + assert!(capacity.check(2, &durable).unwrap_err().uncertain); + capacity.rebuild(&durable, [1]); + capacity.check(2, &durable).unwrap(); + } + + #[test] + fn given_overlapping_accounting_sets_when_sum_exceeds_limit_should_use_exact_occupancy() { + let durable = DurableConsumerOffsets::default(); + durable.record_explicit(ConsumerKind::Consumer, 1, 0, 0); + let capacity = ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 2); + capacity.try_reserve(1, &durable).unwrap(); + capacity.record_stranded(1); + capacity.check(2, &durable).unwrap(); + } + + #[test] + fn given_unchanged_protection_when_reclaim_repeats_should_skip_until_guard_drops() { + let durable = Rc::new(DurableConsumerOffsets::default()); + let capacity = Rc::new(ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 2)); + let held = capacity.reserve_provisional(7, &durable).unwrap(); + assert!(capacity.should_reclaim(&durable)); + for _ in 0..100 { + assert!(!capacity.should_reclaim(&durable)); + } + drop(held); + assert!(capacity.should_reclaim(&durable)); + assert!(!capacity.should_reclaim(&durable)); + capacity.note_local_key_change(); + assert!(capacity.should_reclaim(&durable)); + } + + #[test] + fn given_repeated_reservation_for_same_key_should_reuse_token_allocation() { + let durable = Rc::new(DurableConsumerOffsets::default()); + let capacity = Rc::new(ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 2)); + let first = capacity.reserve_provisional(7, &durable).unwrap(); + let token = Arc::clone(&first.token); + drop(first); + drop(capacity.reserve_provisional(8, &durable).unwrap()); + assert_eq!(capacity.provisional.borrow().len(), 2); + let second = capacity.reserve_provisional(7, &durable).unwrap(); + assert!(Arc::ptr_eq(&token, &second.token)); + } + + #[test] + fn given_inactive_token_cache_at_limit_when_new_key_arrives_should_prune_it() { + let durable = Rc::new(DurableConsumerOffsets::default()); + let capacity = Rc::new(ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 2)); + drop(capacity.reserve_provisional(7, &durable).unwrap()); + drop(capacity.reserve_provisional(8, &durable).unwrap()); + assert_eq!(capacity.provisional.borrow().len(), 2); + let _third = capacity.reserve_provisional(9, &durable).unwrap(); + assert_eq!(capacity.provisional.borrow().len(), 1); + assert!(capacity.provisional.borrow().contains_key(&9)); + } + + #[test] + fn given_local_map_pressure_when_durable_has_room_should_return_transient_error() { + let capacity = ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 1); + assert!(matches!( + iggy_common::IggyError::from(capacity.admit_local_map_key(1, false).unwrap_err()), + iggy_common::IggyError::TransientNotAccepted + )); + assert!(matches!( + iggy_common::IggyError::from(capacity.admit_local_map_key(1, true).unwrap_err()), + iggy_common::IggyError::TooManyConsumerOffsets + )); + } + + #[test] + fn given_provisional_and_journal_reservations_when_rebuilt_and_canceled_should_preserve_journal_slot() + { + let durable = Rc::new(DurableConsumerOffsets::default()); + let capacity = Rc::new(ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 1)); + let provisional = capacity + .reserve_provisional(7, &durable) + .expect("reserve poll"); + capacity.rebuild(&durable, [7]); + drop(provisional); + assert!(capacity.check(8, &durable).is_err()); + capacity.set_pending_count(7, 0); + assert!(capacity.check(8, &durable).is_ok()); + } + + #[test] + fn given_dropped_submit_when_guard_leaves_scope_should_release_only_its_key() { + let durable = Rc::new(DurableConsumerOffsets::default()); + let capacity = Rc::new(ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 2)); + let first = capacity + .reserve_provisional(7, &durable) + .expect("reserve first poll"); + let second = capacity + .reserve_provisional(8, &durable) + .expect("reserve second poll"); + assert!(capacity.check(9, &durable).is_err()); + drop(first); + assert!(capacity.check(9, &durable).is_ok()); + assert_eq!(capacity.occupied(&durable), 1); + drop(second); + assert_eq!(capacity.occupied(&durable), 0); + } + + #[test] + fn given_full_durable_set_when_reserving_new_key_should_reject() { + let durable = DurableConsumerOffsets::default(); + durable.record_explicit(ConsumerKind::Consumer, 1, 0, 0); + let capacity = ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 1); + let error = capacity + .try_reserve(2, &durable) + .expect_err("new key must be rejected"); + assert_eq!(error.occupied, 1); + assert_eq!(error.limit, 1); + assert!(error.first_in_episode); + } + + #[test] + fn given_same_pending_key_when_reserved_twice_should_consume_one_slot() { + let durable = DurableConsumerOffsets::default(); + let capacity = ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 1); + assert_eq!(capacity.try_reserve(7, &durable), Ok(())); + assert_eq!(capacity.try_reserve(7, &durable), Ok(())); + assert!(capacity.try_reserve(8, &durable).is_err()); + capacity.release_reservation(7); + assert!( + capacity.try_reserve(8, &durable).is_err(), + "one of two reservations still owns the slot" + ); + capacity.release_reservation(7); + assert!( + capacity.try_reserve(8, &durable).is_ok(), + "the slot is released after the last reservation" + ); + } + + #[test] + fn given_stranded_file_when_reserving_same_and_different_ids_should_only_reuse_exact_path() { + let durable = DurableConsumerOffsets::default(); + let capacity = ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 1); + capacity.record_stranded(7); + assert!(capacity.try_reserve(8, &durable).is_err()); + assert!( + capacity.try_reserve(7, &durable).is_ok(), + "rewriting the same path does not allocate another file" + ); + } + + #[test] + fn given_uncertain_rebuild_when_reserving_new_key_should_fail_closed() { + let durable = DurableConsumerOffsets::default(); + let capacity = ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 4); + capacity.mark_uncertain(); + let error = capacity + .try_reserve(7, &durable) + .expect_err("unknown pending state must block new keys"); + assert_eq!(error.occupied, 0); + assert!(error.uncertain); + assert!(error.first_in_episode); + assert!( + !capacity + .try_reserve(8, &durable) + .expect_err("the same uncertain episode stays closed") + .first_in_episode + ); + } + + #[test] + fn given_capacity_episode_when_occupancy_drops_should_rearm_first_warning() { + let durable = DurableConsumerOffsets::default(); + durable.record_explicit(ConsumerKind::Consumer, 1, 0, 0); + let capacity = ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 1); + assert!( + capacity + .try_reserve(2, &durable) + .expect_err("full table") + .first_in_episode + ); + assert!( + !capacity + .try_reserve(2, &durable) + .expect_err("same full table") + .first_in_episode + ); + durable.remove(ConsumerKind::Consumer, 1); + capacity.rearm_if_below_limit(&durable); + durable.record_explicit(ConsumerKind::Consumer, 3, 0, 0); + assert!( + capacity + .try_reserve(4, &durable) + .expect_err("new full episode") + .first_in_episode + ); + } + + #[test] + fn given_persisted_state_when_checking_coverage_should_preserve_membership() { + let durable = DurableConsumerOffsets::default(); + durable.record_explicit(ConsumerKind::Consumer, 3, 11, 11); + assert!(durable.contains(ConsumerKind::Consumer, 3)); + assert_eq!(durable.count(ConsumerKind::Consumer), 1); + assert_eq!( + durable.get(ConsumerKind::Consumer, 3), + Some(DurableOffsetState { + committed_offset: 11, + persisted_high_water: 11, + }) + ); + } +} diff --git a/core/partitions/src/iggy_partition.rs b/core/partitions/src/iggy_partition.rs index 356a10a7ef..f25616ad31 100644 --- a/core/partitions/src/iggy_partition.rs +++ b/core/partitions/src/iggy_partition.rs @@ -15,6 +15,9 @@ // specific language governing permissions and limitations // under the License. +use crate::consumer_offset_capacity::{ + ConsumerOffsetCapacity, ConsumerOffsetCapacityError, DurableConsumerOffsets, +}; use crate::iggy_index_writer::IggyIndexWriter; use crate::journal::{MessageLookup, PartitionJournal, PartitionJournalMemStorage}; use crate::log::JournalInfo; @@ -51,7 +54,7 @@ use iggy_binary_protocol::responses::messages::{ SendMessagesConfirmationResponse, SendMessagesResponse, }; use iggy_binary_protocol::{ - AckLevel, GenericHeader, Operation, PrepareHeader, WireDecode, WireEncode, WireIdentifier, + AckLevel, Operation, PrepareHeader, WireDecode, WireEncode, WireIdentifier, }; use iggy_binary_protocol::{PrepareOkHeader, ReplyHeader, RoutedRequestHeader}; use iggy_common::{ @@ -67,8 +70,8 @@ use journal::superblock::{ }; use message_bus::{IggyMessageBus, MessageBus, is_auto_commit_client}; use server_common::{ - MESSAGE_ALIGN, Message, SegmentStorage, - iobuf::{Frozen, Owned}, + Message, SegmentStorage, + iobuf::Frozen, send_messages::{ BatchHeader, ChecksumMode, convert_request_message, decode_prepare_slice, decode_prepare_slice_trusted, stamp_prepare_for_persistence, @@ -205,20 +208,27 @@ where /// server down without the partition moving again in the meantime. fatal: Option, pub(crate) pending_consumer_offset_commits: HashMap, - /// Committed-only mirror of each consumer's persisted offset file: the - /// last value this replica durably wrote per (kind, consumer id). Fed - /// exclusively by the file-writing paths (replicated commit-apply, the - /// primary-local `NoAck` store, purge/delete/reclaim) and never by the - /// eager poll-path in-memory apply, so both readers see committed state - /// only: the auto-commit persist gate (skip or blind-write, no per-commit - /// file read) and the submit-side coalesce gate - /// ([`Self::is_auto_commit_offset_covered`]). A cold key (first touch - /// after boot) folds against the file once via `persist_offset_max`, so - /// the tracker rebuilds from disk lazily and deterministically. - /// `RefCell`: mutated from `&self` paths on the single shard thread; - /// borrows never cross an await. - pub(crate) persisted_offsets: RefCell>, + pub(crate) queued_auto_commit_reservations: + RefCell>>, + /// Committed consumer-offset membership and values. This is deliberately + /// separate from the eager poll maps because follower-local and uncommitted + /// auto-commit progress must never consume a durable slot or enter a state + /// transfer artifact. + pub(crate) durable_consumer_offsets: Rc, + pub(crate) consumer_offset_capacity: Rc, + pub(crate) consumer_group_offset_capacity: Rc, pub(crate) observed_view: u32, + offset_reservations_need_resync: Cell, + offset_reservations_scan_state: Option<(u64, u64, u64, Option)>, + consumer_group_offsets_reconcile_epoch: Rc>, + consumer_offset_dirs_dirty: [Cell; 2], + /// Latest operation of each kind in the current commit walk. Even a covered + /// store depends on an earlier unsynced directory entry of its kind. + consumer_offset_dirs_touched: [Cell>; 2], + #[cfg(test)] + consumer_offset_dir_sync_fault: Cell>, + #[cfg(test)] + offset_dir_sync_count: Cell, /// Highest `PurgeTopic` generation this replica has locally applied (reset /// the partition to empty). The reconciler compares the committed metadata /// generation against this and resets only when it advances, so a redundant @@ -556,8 +566,26 @@ where installed_frontier: None, fatal: None, pending_consumer_offset_commits: HashMap::new(), - persisted_offsets: RefCell::new(HashMap::new()), + queued_auto_commit_reservations: RefCell::new(HashMap::new()), + durable_consumer_offsets: Rc::new(DurableConsumerOffsets::default()), + consumer_offset_capacity: Rc::new(ConsumerOffsetCapacity::new( + ConsumerKind::Consumer, + crate::DEFAULT_CONSUMER_OFFSETS_MAX, + )), + consumer_group_offset_capacity: Rc::new(ConsumerOffsetCapacity::new( + ConsumerKind::ConsumerGroup, + crate::DEFAULT_CONSUMER_OFFSETS_MAX, + )), observed_view, + offset_reservations_need_resync: Cell::new(false), + offset_reservations_scan_state: None, + consumer_group_offsets_reconcile_epoch: Rc::new(Cell::new(0)), + consumer_offset_dirs_dirty: [Cell::new(false), Cell::new(false)], + consumer_offset_dirs_touched: [Cell::new(None), Cell::new(None)], + #[cfg(test)] + consumer_offset_dir_sync_fault: Cell::new(None), + #[cfg(test)] + offset_dir_sync_count: Cell::new(0), applied_purge_generation: 0, created_revision: 0, purge_floor_op: 0, @@ -653,6 +681,12 @@ where self.dedup.set_capacity(clients_max); } + /// Set the per-kind durable consumer-offset limit for this partition. + pub fn set_consumer_offsets_max(&mut self, offsets_max: usize) { + self.consumer_offset_capacity.set_limit(offsets_max); + self.consumer_group_offset_capacity.set_limit(offsets_max); + } + #[must_use] pub fn with_in_memory_storage( stats: Arc, @@ -1090,8 +1124,34 @@ where durable_offset, write_offset, offset_space_used, + consumer_offsets, + consumer_group_offsets, } = state; self.log = log; + for (consumer_id, offset) in consumer_offsets { + self.consumer_offsets.pin().insert( + consumer_id as usize, + ConsumerOffset::new(ConsumerKind::Consumer, consumer_id, offset, String::new()), + ); + self.durable_consumer_offsets.record_explicit( + ConsumerKind::Consumer, + consumer_id, + offset, + offset, + ); + } + for (group_id, offset) in consumer_group_offsets { + self.consumer_group_offsets.pin().insert( + ConsumerGroupId(group_id as usize), + ConsumerOffset::new(ConsumerKind::ConsumerGroup, group_id, offset, String::new()), + ); + self.durable_consumer_offsets.record_explicit( + ConsumerKind::ConsumerGroup, + group_id, + offset, + offset, + ); + } // Empty carry-over: the previous incarnation never took a write, so there // is no offset space to restore and claiming one would make the next // prepare mint from a base no peer agrees on. @@ -1119,6 +1179,12 @@ where self.recovered_durable_offset = Some(durable); } + #[cfg(any(test, feature = "simulator"))] + #[must_use] + pub fn retained_consumer_offsets(&self, kind: ConsumerKind) -> Vec<(u32, u64)> { + self.durable_consumer_offsets.committed_entries(kind) + } + /// The in-memory half of [`Self::reanchor_to_offset_frontier`], for a /// simulator partition rebuilt over a restored offset counter. /// @@ -1943,6 +2009,52 @@ where self.consumer_offset_enforce_fsync = consumer_offset_enforce_fsync; } + /// Seed committed membership from one recovered offset file. The logical + /// value may be clamped to the recovered message frontier while the file + /// high-water retains the value read from disk. + pub fn seed_recovered_consumer_offset( + &self, + kind: ConsumerKind, + consumer_id: u32, + committed_offset: u64, + persisted_high_water: u64, + ) { + self.durable_consumer_offsets.record_explicit( + kind, + consumer_id, + committed_offset, + persisted_high_water, + ); + } + + pub fn seed_stranded_consumer_offset(&self, kind: ConsumerKind, consumer_id: u32) -> bool { + if !self.durable_consumer_offsets.contains(kind, consumer_id) { + self.consumer_offset_capacity_for(kind) + .record_stranded(consumer_id); + return true; + } + false + } + + #[must_use] + pub fn durable_consumer_offset_count(&self, kind: ConsumerKind) -> usize { + self.durable_consumer_offsets.count(kind) + } + + #[must_use] + pub fn occupied_consumer_offset_count(&self, kind: ConsumerKind) -> usize { + self.consumer_offset_capacity_for(kind) + .occupied(&self.durable_consumer_offsets) + } + + #[must_use] + pub fn consumer_offset_map_count(&self, kind: ConsumerKind) -> usize { + match kind { + ConsumerKind::Consumer => self.consumer_offsets.len(), + ConsumerKind::ConsumerGroup => self.consumer_group_offsets.len(), + } + } + /// Stage a consumer offset upsert for the replicated op. The prepare /// must already have been appended to `self.log.journal` by the caller /// so `VsrAction::RetransmitPrepares` can recover it during a view @@ -1963,7 +2075,11 @@ where } else { PendingConsumerOffsetCommit::upsert(kind, consumer_id, offset) }; - self.pending_consumer_offset_commits.insert(op, pending); + let replaced = self.pending_consumer_offset_commits.insert(op, pending); + if let Some(replaced) = replaced { + self.refresh_consumer_offset_reservation(replaced.kind, replaced.consumer_id); + } + self.refresh_consumer_offset_reservation(kind, consumer_id); } /// Stage a consumer offset delete for the replicated op. See @@ -1982,7 +2098,9 @@ where consumer_id: u32, ) { let pending = PendingConsumerOffsetCommit::delete(kind, consumer_id); - self.pending_consumer_offset_commits.insert(op, pending); + if let Some(replaced) = self.pending_consumer_offset_commits.insert(op, pending) { + self.refresh_consumer_offset_reservation(replaced.kind, replaced.consumer_id); + } } pub(crate) async fn apply_staged_consumer_offset_commit( @@ -2008,8 +2126,16 @@ where // durably stored; the in-memory update is idempotent on replay // because we look up by (kind, id). self.persist_consumer_offset_commit(pending).await?; - self.apply_consumer_offset_commit(pending)?; + let operation = match pending.mutation { + PendingConsumerOffsetMutation::Upsert(_) => Operation::StoreConsumerOffset, + PendingConsumerOffsetMutation::Delete => Operation::DeleteConsumerOffset, + }; + // A covered store also depends on an earlier dirty directory entry. + self.consumer_offset_dirs_touched[crate::state_transfer::consumer_kind_index(pending.kind)] + .set(Some((op, operation))); + self.apply_consumer_offset_commit(pending); self.pending_consumer_offset_commits.remove(&op); + self.refresh_consumer_offset_reservation(pending.kind, pending.consumer_id); Ok(()) } @@ -2017,15 +2143,13 @@ where &self, pending: PendingConsumerOffsetCommit, ) -> Result<(), IggyError> { - let Some(path) = self.persisted_offset_path(pending.kind, pending.consumer_id) else { - return Ok(()); - }; - let key = (pending.kind, pending.consumer_id); + let path = self.persisted_offset_path(pending.kind, pending.consumer_id); + let capacity = self.consumer_offset_capacity_for(pending.kind); match pending.mutation { // A server auto-commit persists monotonically: its op offset can // trail the durably-recorded value (disk-tier polls replicate in // IO-completion order), so a plain overwrite would rewind the file - // and re-deliver on restart. The `persisted_offsets` tracker keeps + // and re-deliver on restart. The durable offset tracker keeps // the fold off the file: a covered offset skips the write, an // advancing one blind-writes, and only a cold key (first commit // after boot) reads the file once. Explicit client stores @@ -2033,58 +2157,123 @@ where // in-memory `upsert_offset_max` vs `upsert_offset` split in the // commit-apply. PendingConsumerOffsetMutation::Upsert(offset) if pending.auto_commit => { - let tracked = self.persisted_offsets.borrow().get(&key).copied(); - let persisted = match tracked { - Some(high_water) if offset <= high_water => return Ok(()), - Some(_) => { - persist_offset(&path, offset, self.consumer_offset_enforce_fsync).await?; - offset + let tracked = self + .durable_consumer_offsets + .get(pending.kind, pending.consumer_id); + let (persisted_high_water, written) = match (path.as_deref(), tracked) { + (None, _) => (offset, false), + (Some(_), Some(state)) if offset <= state.persisted_high_water => { + (state.persisted_high_water, false) + } + (Some(path), Some(state)) => { + let value = state.committed_offset.max(offset); + persist_offset(path, value, self.consumer_offset_enforce_fsync).await?; + (value, true) } - None => { - persist_offset_max(&path, offset, self.consumer_offset_enforce_fsync) - .await? + (Some(path), None) => { + let persisted = + persist_offset_max(path, offset, self.consumer_offset_enforce_fsync) + .await?; + (persisted.offset, persisted.written) } }; - self.persisted_offsets.borrow_mut().insert(key, persisted); + self.durable_consumer_offsets.record_auto_commit( + pending.kind, + pending.consumer_id, + if tracked.is_none() { + persisted_high_water + } else { + offset + }, + persisted_high_water, + ); + if written && self.consumer_offset_enforce_fsync { + self.mark_consumer_offset_dir_dirty(pending.kind); + } + capacity.clear_stranded(pending.consumer_id); + if pending.kind == ConsumerKind::ConsumerGroup && tracked.is_none() { + self.mark_consumer_group_offsets_need_reconcile(); + } Ok(()) } PendingConsumerOffsetMutation::Upsert(offset) => { - persist_offset(&path, offset, self.consumer_offset_enforce_fsync).await?; - self.persisted_offsets.borrow_mut().insert(key, offset); + if let Some(path) = path.as_deref() { + persist_offset(path, offset, self.consumer_offset_enforce_fsync).await?; + } + let created = self.durable_consumer_offsets.record_explicit( + pending.kind, + pending.consumer_id, + offset, + offset, + ); + if path.is_some() && self.consumer_offset_enforce_fsync { + self.mark_consumer_offset_dir_dirty(pending.kind); + } + capacity.clear_stranded(pending.consumer_id); + if pending.kind == ConsumerKind::ConsumerGroup && created { + self.mark_consumer_group_offsets_need_reconcile(); + } Ok(()) } PendingConsumerOffsetMutation::Delete => { - delete_persisted_offset(&path).await?; - self.persisted_offsets.borrow_mut().remove(&key); + if let Some(path) = path.as_deref() { + // Keep the logical state until the file is removed. The + // partition journal is memory-only, so acknowledging an + // unsuccessful unlink would let boot resurrect the key. + match delete_persisted_offset(path).await { + Ok(removed) => { + if removed && self.consumer_offset_enforce_fsync { + self.mark_consumer_offset_dir_dirty(pending.kind); + } + capacity.clear_stranded(pending.consumer_id); + } + Err(error) => { + capacity.record_stranded(pending.consumer_id); + warn!( + target: "iggy.partitions.diag", + plane = "partitions", + replica_id = self.consensus.replica(), + namespace_raw = self.namespace().inner(), + kind = ?pending.kind, + consumer_id = pending.consumer_id, + path, + %error, + "committed consumer offset delete could not remove its file" + ); + return Err(error); + } + } + } + self.durable_consumer_offsets + .remove(pending.kind, pending.consumer_id); + capacity.forget_inactive_provisional(pending.consumer_id); + capacity.rearm_if_below_limit(&self.durable_consumer_offsets); Ok(()) } } } - /// Whether the committed high-water for this consumer already covers - /// `offset`, so a poll's auto-commit submit cannot advance it and may be - /// skipped instead of burning a consensus op. Reads committed state only - /// (the tracker is fed at commit-apply, never by the eager poll-path - /// apply): an offset covered in memory but not yet committed keeps - /// resubmitting until the covering op actually lands, so a dropped - /// in-flight op self-heals on the next poll. + /// Reject completion from a removed partition or a view whose retained + /// prepares have not yet been accounted for by the pump. #[must_use] - pub fn is_auto_commit_offset_covered( - &self, - kind: ConsumerKind, - consumer_id: u32, - offset: u64, - ) -> bool { - self.persisted_offsets - .borrow() - .get(&(kind, consumer_id)) - .is_some_and(|&high_water| offset <= high_water) - } - - fn apply_consumer_offset_commit( - &self, - pending: PendingConsumerOffsetCommit, - ) -> Result<(), IggyError> { + pub fn auto_commit_admission_ready(&self, applied: &crate::AutoCommitApplied) -> bool { + applied.belongs_to(&self.durable_consumer_offsets) + && self.observed_view == self.consensus.view() + && !self.offset_reservations_need_resync.get() + && (!self + .consumer_offset_capacity_for(applied.kind) + .is_uncertain() + || self + .durable_consumer_offsets + .contains(applied.kind, applied.consumer_id)) + } + + fn apply_consumer_offset_commit(&self, pending: PendingConsumerOffsetCommit) { + if pending.kind == ConsumerKind::ConsumerGroup + && matches!(pending.mutation, PendingConsumerOffsetMutation::Delete) + { + self.mark_consumer_group_offsets_need_reconcile(); + } match pending.mutation { PendingConsumerOffsetMutation::Upsert(offset) if pending.kind == ConsumerKind::Consumer => @@ -2104,7 +2293,6 @@ where pending.auto_commit, create, ); - Ok(()) } PendingConsumerOffsetMutation::Upsert(offset) if pending.kind == ConsumerKind::ConsumerGroup => @@ -2133,28 +2321,15 @@ where pending.auto_commit, create, ); - Ok(()) } - // Commit-time apply keeps its invariant check on the PRIMARY: - // admission verified the offset exists there, so a miss on the - // primary is real divergence (log corruption / out-of-order apply) - // and must surface rather than silently mask a split state. A - // FOLLOWER may legitimately miss the offset: `AckLevel::NoAck` - // stores apply on the primary only and are never replicated, - // so a later quorum delete finds nothing on the backups -- erroring - // there would fail the committed apply, panic the replica as - // divergent, and crash-loop on every journal replay. The - // prepare-time race is handled by not re-checking existence at - // staging (see `stage_consumer_offset_delete`). + // Two independently admitted deletes can commit after both saw + // the key present. Deletion is idempotent on every replica, while + // admission still rejects a request for an already absent key. PendingConsumerOffsetMutation::Delete if pending.kind == ConsumerKind::Consumer => { let id = pending.consumer_id; let guard = self.consumer_offsets.pin(); let key = usize::try_from(id).expect("u32 consumer id must fit usize"); - let removed = guard.remove(&key).is_some(); - if !removed && !self.consensus.is_follower() { - return Err(IggyError::ConsumerOffsetNotFound(key)); - } - Ok(()) + guard.remove(&key); } PendingConsumerOffsetMutation::Delete if pending.kind == ConsumerKind::ConsumerGroup => @@ -2164,13 +2339,9 @@ where let key = ConsumerGroupId( usize::try_from(group_id).expect("u32 group id must fit usize"), ); - let removed = guard.remove(&key).is_some(); - if !removed && !self.consensus.is_follower() { - return Err(IggyError::ConsumerOffsetNotFound(key.0)); - } - Ok(()) + guard.remove(&key); } - _ => Ok(()), + _ => (), } } @@ -2186,38 +2357,52 @@ where .collect() } - /// Reclaim every stored consumer-group offset whose group id is no longer - /// `is_live`, returning the owned persisted-file paths the caller must unlink. - /// - /// Fully synchronous (no `.await`): the in-memory papaya remove happens here, - /// the disk unlink is deferred to the caller on owned `String` data so no - /// borrow of `self` survives across the await. This is the only safe shape - /// for the reconciler, which runs on a sibling task to the pump that may - /// realloc the partitions vec during that await. The remove-then-unlink - /// ordering matches the crash-safe GC invariant (monotonic, never-reused - /// group ids mean a recreated group never reads a dead group's offset). + /// Snapshot dead group keys for deletion through the partition's VSR log. + /// A local unlink could free the primary's quota while backups retained + /// every older generation, so reclamation uses the same ordered delete as + /// an explicit consumer-offset request. #[must_use] - #[allow(clippy::cast_possible_truncation)] - pub fn reclaim_dead_group_offsets(&self, is_live: impl Fn(u64) -> bool) -> Vec { - let pinned = self.consumer_group_offsets.pin(); - let dead: Vec = pinned - .keys() - .map(|key| key.0 as u64) - .filter(|group_id| !is_live(*group_id)) - .collect(); - let mut paths = Vec::with_capacity(dead.len()); - for group_id in dead { - pinned.remove(&ConsumerGroupId(group_id as usize)); - self.persisted_offsets - .borrow_mut() - .remove(&(ConsumerKind::ConsumerGroup, group_id as u32)); - if let Some(path) = - self.persisted_offset_path(ConsumerKind::ConsumerGroup, group_id as u32) + pub fn dead_consumer_group_offset_ids(&self, is_live: impl Fn(u64) -> bool) -> Vec { + if !self.consensus.is_primary() || !self.consensus.is_normal() { + return Vec::new(); + } + // A stranded file already failed normal loading or unlink. Reissuing + // replicated deletes every reconciliation pass cannot make its + // filesystem writable and would create a permanent commit loop. + // A single-replica explicit deletion can retry after repair. On a + // replicated partition the file must be repaired or removed locally, + // because older peers do not recognize a delete for a map-missing key. + let capacity = self.consumer_offset_capacity_for(ConsumerKind::ConsumerGroup); + let mut dead = Vec::new(); + for key in self.consumer_group_offsets.pin().keys() { + let Ok(id) = u32::try_from(key.0) else { + continue; + }; + if !is_live(u64::from(id)) + && !capacity.is_stranded(id) + && self + .durable_consumer_offsets + .contains(ConsumerKind::ConsumerGroup, id) { - paths.push(path); + dead.push(id); } } - paths + dead.sort_unstable(); + dead.dedup(); + dead + } + + pub(crate) fn set_consumer_group_offsets_reconcile_epoch(&mut self, epoch: Rc>) { + self.consumer_group_offsets_reconcile_epoch = epoch; + self.mark_consumer_group_offsets_need_reconcile(); + } + + fn mark_consumer_group_offsets_need_reconcile(&self) { + self.consumer_group_offsets_reconcile_epoch.set( + self.consumer_group_offsets_reconcile_epoch + .get() + .wrapping_add(1), + ); } /// Cooperative-rebalance classification: a group's `(last_polled, committed)` @@ -2261,21 +2446,59 @@ where ); if let Err(error) = self.persist_consumer_offset_commit(pending).await { + if offset.is_some() { + self.release_consumer_offset_reservation(kind, consumer_id); + } emit_partition_diag( tracing::Level::WARN, &PartitionDiagEvent::new(self.diag_ctx(), "no_ack offset persist failed") .with_operation(request_header.operation) .with_error(error.to_string()), ); + Self::send_partition_deny_or_log( + &self.consensus, + &request_header, + error.as_code(), + "no_ack offset failure reply send failed", + waiter, + ) + .await; return; } - if let Err(error) = self.apply_consumer_offset_commit(pending) { + self.apply_consumer_offset_commit(pending); + // Only this request's kind: dirt on the other directory belongs to + // whichever path left it, and its failure is not this client's answer. + // Visibility does not prove crash durability. A failed required barrier + // must remain an error even after the mutation becomes visible. + let mut kinds = [false; 2]; + kinds[crate::state_transfer::consumer_kind_index(kind)] = true; + let failed = self.flush_consumer_offset_directories_for(kinds).await; + if failed.iter().any(|failed| *failed) { + if offset.is_some() { + self.release_consumer_offset_reservation(kind, consumer_id); + } else { + // Retain retry admission after the visible deletion so a new + // delete can retry its directory barrier instead of returning + // ConsumerOffsetNotFound. + self.consumer_offset_capacity_for(kind) + .record_stranded(consumer_id); + } emit_partition_diag( tracing::Level::WARN, - &PartitionDiagEvent::new(self.diag_ctx(), "no_ack offset apply failed") - .with_operation(request_header.operation) - .with_error(error.to_string()), + &PartitionDiagEvent::new( + self.diag_ctx(), + "no_ack offset directory sync failed after a visible local mutation", + ) + .with_operation(request_header.operation), ); + Self::send_partition_deny_or_log( + &self.consensus, + &request_header, + IggyError::CannotSyncFile.as_code(), + "no_ack offset directory sync failure reply send failed", + waiter, + ) + .await; return; } @@ -2284,6 +2507,9 @@ where &request_header, committed_reply_body(request_header.operation), ); + if offset.is_some() { + self.release_consumer_offset_reservation(kind, consumer_id); + } // Same rule as the committed path: a submit's waiter takes the reply, // because `header.client` is then the VSR consensus id. if let Some(waiter) = waiter { @@ -2323,11 +2549,105 @@ where } } + pub(crate) fn consumer_offset_capacity_for( + &self, + kind: ConsumerKind, + ) -> &ConsumerOffsetCapacity { + match kind { + ConsumerKind::Consumer => &self.consumer_offset_capacity, + ConsumerKind::ConsumerGroup => &self.consumer_group_offset_capacity, + } + } + + fn reserve_consumer_offset( + &self, + kind: ConsumerKind, + consumer_id: u32, + ) -> Result<(), ConsumerOffsetCapacityError> { + self.consumer_offset_capacity_for(kind) + .try_reserve(consumer_id, &self.durable_consumer_offsets) + } + + fn release_consumer_offset_reservation(&self, kind: ConsumerKind, consumer_id: u32) { + self.consumer_offset_capacity_for(kind) + .release_reservation(consumer_id); + } + + fn refresh_consumer_offset_reservation(&self, kind: ConsumerKind, consumer_id: u32) { + let count = self + .pending_consumer_offset_commits + .values() + .filter(|pending| { + pending.kind == kind + && pending.consumer_id == consumer_id + && matches!(pending.mutation, PendingConsumerOffsetMutation::Upsert(_)) + }) + .count(); + let capacity = self.consumer_offset_capacity_for(kind); + capacity.set_pending_count(consumer_id, count); + capacity.rearm_if_below_limit(&self.durable_consumer_offsets); + } + + async fn admit_consumer_offset_key( + &self, + header: &RoutedRequestHeader, + kind: ConsumerKind, + consumer_id: u32, + waiter: &mut Option>>, + ) -> bool { + if self.consumer_offset_capacity_for(kind).is_uncertain() + && !self.durable_consumer_offsets.contains(kind, consumer_id) + { + Self::send_partition_deny_or_log( + &self.consensus, + header, + IggyError::TransientNotAccepted.as_code(), + "consumer offset accounting unavailable reply send failed", + waiter.take(), + ) + .await; + return false; + } + let Err(capacity_error) = self.reserve_consumer_offset(kind, consumer_id) else { + return true; + }; + if capacity_error.first_in_episode { + warn!( + target: "iggy.partitions.diag", + plane = "partitions", + replica_id = self.consensus.replica(), + namespace_raw = self.namespace().inner(), + ?kind, + occupied = capacity_error.occupied, + limit = capacity_error.limit, + config = "[partition] consumer_offsets_max", + "consumer offset admission limit reached" + ); + } + Self::send_partition_deny_or_log( + &self.consensus, + header, + IggyError::TooManyConsumerOffsets.as_code(), + "consumer offset capacity deny reply send failed", + waiter.take(), + ) + .await; + false + } + fn ensure_consumer_offset_exists( &self, kind: ConsumerKind, consumer_id: u32, ) -> Result<(), IggyError> { + if self.durable_consumer_offsets.contains(kind, consumer_id) { + return Ok(()); + } + // A local-only cursor is not a replicated key. Older replicas reject + // missing-key deletes when replaying as primary during a rolling upgrade. + if self.consensus.replica_count() > 1 { + return Err(IggyError::ConsumerOffsetNotFound(consumer_id as usize)); + } let found = match kind { ConsumerKind::Consumer => { let key = usize::try_from(consumer_id).expect("u32 consumer id must fit usize"); @@ -2341,7 +2661,10 @@ where } }; - if found { + let local_stranded = self + .consumer_offset_capacity_for(kind) + .is_stranded(consumer_id); + if found || local_stranded { Ok(()) } else { Err(IggyError::ConsumerOffsetNotFound( @@ -2355,14 +2678,201 @@ where ReplicaLogContext::from_consensus(self.consensus(), PlaneKind::Partitions) } - fn clear_pending_consumer_offset_commits_if_view_changed(&mut self) { + fn store_offset_range_error(&self, offset: u64) -> Option { + let current = self.stats.current_offset(); + (offset > current || (current == 0 && self.stats.messages_count_inconsistent() == 0)) + .then_some(IggyError::InvalidOffset(offset)) + } + + fn resynchronize_consumer_offset_reservations(&mut self) { + self.resynchronize_consumer_offset_reservations_inner(false); + } + + /// Retry incomplete accounting at most once per shard tick, after progress. + pub fn retry_consumer_offset_reservations(&mut self) { + if self.consumer_offset_capacity.is_uncertain() + || self.consumer_group_offset_capacity.is_uncertain() + { + self.resynchronize_consumer_offset_reservations_inner(true); + } + } + + #[allow(clippy::too_many_lines)] + fn resynchronize_consumer_offset_reservations_inner(&mut self, from_tick: bool) { let current_view = self.consensus.view(); + let scan_state = ( + self.consensus.commit_min(), + self.consensus.commit_max(), + self.consensus.sequencer().current_sequence(), + self.log.journal().inner.last_op(), + ); + let uncertain = self.consumer_offset_capacity.is_uncertain() + || self.consumer_group_offset_capacity.is_uncertain(); if current_view == self.observed_view { - return; + if uncertain && !from_tick { + return; + } + let retry_requested = self.offset_reservations_need_resync.get() + || (uncertain && self.offset_reservations_scan_state != Some(scan_state)); + if !retry_requested { + return; + } } - self.pending_consumer_offset_commits.clear(); + if current_view != self.observed_view { + self.queued_auto_commit_reservations.borrow_mut().clear(); + self.mark_consumer_group_offsets_need_reconcile(); + } + + let from_op = self + .consensus + .commit_min() + .max(self.purge_floor_op) + .saturating_add(1); + let commit_max = self.consensus.commit_max(); + let to_op = self + .consensus + .sequencer() + .current_sequence() + .min(self.log.journal().inner.last_op().unwrap_or(commit_max)); + // Committed offset prepares still need local apply, even if a message + // flush already evicted their journal bytes. Never drop their staging. + // Within one view an op is assigned once, so uncommitted staging is + // kept too; only a view change can replace what sits at those ops, and + // only then is the tail decoded again. + let same_view = current_view == self.observed_view; + let mut rebuilt: HashMap<_, _> = self + .pending_consumer_offset_commits + .iter() + .filter(|(op, _)| { + **op >= from_op && (**op <= commit_max || (same_view && **op <= to_op)) + }) + .map(|(op, pending)| (*op, *pending)) + .collect(); + let headers = self.log.journal().inner.repair_headers_in(from_op..=to_op); + let uncommitted_from = from_op.max(commit_max.saturating_add(1)); + let expected = to_op + .checked_sub(uncommitted_from) + .map_or(0, |span| span.saturating_add(1)); + let mut decode_failed = + headers.keys().filter(|op| **op >= uncommitted_from).count() as u64 != expected; + for (op, header) in headers { + if !matches!( + header.operation, + Operation::StoreConsumerOffset | Operation::DeleteConsumerOffset + ) { + continue; + } + if rebuilt.contains_key(&op) { + continue; + } + match self.restage_consumer_offset_from_journal(op) { + Ok(pending) => { + rebuilt.insert(op, pending); + } + Err(error) => { + error!( + target: "iggy.partitions.diag", + plane = "partitions", + replica_id = self.consensus.replica(), + namespace_raw = self.namespace().inner(), + op, + %error, + "failed to rebuild consumer offset reservations after view change" + ); + decode_failed = true; + break; + } + } + } + self.pending_consumer_offset_commits = rebuilt; + if decode_failed { + self.consumer_offset_capacity.mark_uncertain(); + self.consumer_group_offset_capacity.mark_uncertain(); + } else { + let consumer_ids = self + .pending_consumer_offset_commits + .values() + .filter(|pending| { + pending.kind == ConsumerKind::Consumer + && matches!(pending.mutation, PendingConsumerOffsetMutation::Upsert(_)) + }) + .map(|pending| pending.consumer_id); + self.consumer_offset_capacity + .rebuild(&self.durable_consumer_offsets, consumer_ids); + let group_ids = self + .pending_consumer_offset_commits + .values() + .filter(|pending| { + pending.kind == ConsumerKind::ConsumerGroup + && matches!(pending.mutation, PendingConsumerOffsetMutation::Upsert(_)) + }) + .map(|pending| pending.consumer_id); + self.consumer_group_offset_capacity + .rebuild(&self.durable_consumer_offsets, group_ids); + } self.observed_view = current_view; + self.offset_reservations_scan_state = Some(scan_state); + // The shard tick retries uncertainty after journal or frontier progress. + self.offset_reservations_need_resync.set(false); + } + + fn reclaim_phantom_offsets(&self, kind: ConsumerKind, map_count: usize) { + let capacity = self.consumer_offset_capacity_for(kind); + if self.durable_consumer_offsets.count(kind) >= capacity.limit() + || !capacity.should_reclaim(&self.durable_consumer_offsets) + { + return; + } + let needed = map_count.saturating_sub(capacity.limit()).saturating_add(1); + match kind { + ConsumerKind::Consumer => { + self.reclaim_phantom_offset_keys(&self.consumer_offsets, kind, needed, |key| { + u32::try_from(*key).ok() + }); + } + ConsumerKind::ConsumerGroup => self.reclaim_phantom_offset_keys( + &self.consumer_group_offsets, + kind, + needed, + |key| u32::try_from(key.0).ok(), + ), + } + } + + fn reclaim_phantom_offset_keys( + &self, + offsets: &papaya::HashMap, + kind: ConsumerKind, + mut remaining: usize, + consumer_id: impl Fn(&K) -> Option, + ) { + let capacity = self.consumer_offset_capacity_for(kind); + let map = offsets.pin(); + for (key, _) in &map { + if let Some(id) = consumer_id(key) + && !capacity.holds(id, &self.durable_consumer_offsets) + && map.remove(key).is_some() + { + capacity.forget_inactive_provisional(id); + // This cursor was never durable, so the consumer's next `Next` + // poll restarts from offset 0 and redelivers. At-least-once + // permits it; an operator should still see it happen. + warn!( + target: "iggy.partitions.diag", + plane = "partitions", + namespace_raw = self.namespace().inner(), + ?kind, + consumer_id = id, + "reclaimed a local consumer offset cursor with no durable backing; its next \ + poll restarts from offset 0" + ); + remaining -= 1; + if remaining == 0 { + break; + } + } + } } /// Build an owned [`PollPlan`] synchronously (no `.await`), so the caller @@ -2421,6 +2931,30 @@ where } }; + if args.auto_commit + && let Ok(pending) = PendingConsumerOffsetCommit::try_from_polling_consumer(consumer, 0) + { + let capacity = self.consumer_offset_capacity_for(pending.kind); + if !capacity.is_uncertain() { + let exists = match pending.kind { + ConsumerKind::Consumer => self + .consumer_offsets + .pin() + .contains_key(&(pending.consumer_id as usize)), + ConsumerKind::ConsumerGroup => self + .consumer_group_offsets + .pin() + .contains_key(&ConsumerGroupId(pending.consumer_id as usize)), + }; + if !exists { + let map_count = self.consumer_offset_map_count(pending.kind); + if map_count >= capacity.limit() { + self.reclaim_phantom_offsets(pending.kind, map_count); + } + } + } + } + // Past the empty-return guards: only now build the auto-commit context, // whose offset-path `format!()` is wasted on the early returns above. let auto_commit = self.auto_commit_ctx(consumer, args.auto_commit); @@ -2533,7 +3067,15 @@ where create_path: self.consumer_group_offsets_path.clone(), }, }; - Some(AutoCommitCtx { target }) + let capacity = match pending.kind { + ConsumerKind::Consumer => Rc::clone(&self.consumer_offset_capacity), + ConsumerKind::ConsumerGroup => Rc::clone(&self.consumer_group_offset_capacity), + }; + Some(AutoCommitCtx { + target, + capacity, + durable: Rc::clone(&self.durable_consumer_offsets), + }) } /// Synchronous in-memory journal poll, for the resident tier. Never awaits @@ -2580,17 +3122,6 @@ where .map(|journaled| journaled.result) } - #[allow(clippy::cast_possible_truncation)] - fn store_consumer_offset( - &self, - consumer: PollingConsumer, - offset: u64, - ) -> Result<(), IggyError> { - let pending = PendingConsumerOffsetCommit::try_from_polling_consumer(consumer, offset)?; - self.apply_consumer_offset_commit(pending)?; - Ok(()) - } - fn get_consumer_offset(&self, consumer: PollingConsumer) -> Option { match consumer { PollingConsumer::Consumer(id, _) => self @@ -2798,13 +3329,36 @@ where message: Message, reply: Option>>, ) { - // Taken by whichever arm answers: the deny paths, the NoAck fast path, - // or the pipeline entry that fires it at commit. Exactly one runs. - let mut reply = reply; - self.clear_pending_consumer_offset_commits_if_view_changed(); - let namespace = IggyNamespace::from_raw(message.header().group); - let client_id = message.header().client; - let request = message.header().request; + self.on_request_with_reservation(message, reply, None).await; + } + + #[allow(clippy::too_many_lines)] + pub(crate) async fn on_request_with_reservation( + &mut self, + message: Message, + reply: Option>>, + mut reservation: Option, + ) { + if reservation.as_ref().is_some_and(|reservation| { + !self + .consumer_offset_capacity_for(reservation.kind) + .owns(reservation) + }) { + debug!( + namespace_raw = self.namespace().inner(), + kind = ?reservation.as_ref().map(|reservation| reservation.kind), + consumer_id = ?reservation.as_ref().map(|reservation| reservation.consumer_id), + "dropping auto-commit reservation owned by another partition incarnation" + ); + return; + } + // Taken by whichever arm answers: the deny paths, the NoAck fast path, + // or the pipeline entry that fires it at commit. Exactly one runs. + let mut reply = reply; + self.resynchronize_consumer_offset_reservations(); + let namespace = IggyNamespace::from_raw(message.header().group); + let client_id = message.header().client; + let request = message.header().request; let disposition = { let consensus = self.consensus(); @@ -2992,39 +3546,45 @@ where // code on this committed-shaped frame (op=commit_max) as success. if matches!(message.header().operation, Operation::StoreConsumerOffset) && let Some((_, _, Some(requested_offset), _)) = consumer_offset + && let Some(error) = self.store_offset_range_error(requested_offset) { - let current_offset = self.stats.current_offset(); - let partition_empty = - self.stats.messages_count_inconsistent() == 0 && current_offset == 0; - if partition_empty || requested_offset > current_offset { - emit_partition_diag( - tracing::Level::WARN, - &PartitionDiagEvent::new( - ReplicaLogContext::from_consensus(consensus, PlaneKind::Partitions), - "rejecting store_consumer_offset for out-of-range offset", - ) - .with_operation(message.header().operation) - .with_error(IggyError::InvalidOffset(requested_offset).to_string()), - ); - Self::send_partition_deny_or_log( - consensus, - message.header(), - IggyError::InvalidOffset(requested_offset).as_code(), - "store_consumer_offset deny reply send failed", - reply.take(), + emit_partition_diag( + tracing::Level::WARN, + &PartitionDiagEvent::new( + ReplicaLogContext::from_consensus(consensus, PlaneKind::Partitions), + "rejecting store_consumer_offset for out-of-range offset", ) - .await; - return; - } + .with_operation(message.header().operation) + .with_error(error.to_string()), + ); + Self::send_partition_deny_or_log( + consensus, + message.header(), + error.as_code(), + "store_consumer_offset deny reply send failed", + reply.take(), + ) + .await; + return; } - // NoAck -> fast path. Quorum -> VSR pipeline. + // The node-local fast path is safe only for a single-replica + // partition. Every mutation in a replicated partition enters VSR + // regardless of the acknowledgement byte. if let Some((kind, consumer_id, offset, AckLevel::NoAck)) = consumer_offset + && consensus.replica_count() == 1 && matches!( message.header().operation, Operation::StoreConsumerOffset | Operation::DeleteConsumerOffset, ) { + if offset.is_some() + && !self + .admit_consumer_offset_key(message.header(), kind, consumer_id, &mut reply) + .await + { + return; + } Disposition::NoAck { request_header: Box::new(*message.header()), kind, @@ -3071,10 +3631,24 @@ where waiter, ) .await; + } else if let Some(reservation) = reservation.take() { + self.queued_auto_commit_reservations + .borrow_mut() + .entry((reservation.kind, reservation.consumer_id)) + .or_default() + .push(reservation); } return; } + if let Some((kind, consumer_id, Some(_), _)) = consumer_offset + && !self + .admit_consumer_offset_key(message.header(), kind, consumer_id, &mut reply) + .await + { + return; + } + let prepare = message.project(consensus); consensus.verify_pipeline(); match reply.take() { @@ -3135,11 +3709,40 @@ where /// # Panics /// On mid-iteration status flip. Reachable only if `clear_request_queue` /// is bypassed at view-change reset. - #[allow(clippy::future_not_send)] + #[allow(clippy::future_not_send, clippy::too_many_lines)] pub async fn drain_request_queue_into_prepares(&mut self, slots_freed: usize) { - for _ in 0..slots_freed { + self.resynchronize_consumer_offset_reservations(); + let mut promoted = 0; + // Denials do not consume a slot, so without a budget one drain could + // answer the whole queue, each with an awaited reply send, inside one + // commit turn. Consecutive denials end the drain. The shard tick + // resumes parked work even if no further operation commits. + let mut consecutive_denials = 0usize; + while promoted < slots_freed { let req = self.consensus().pop_queued_request(); let Some(mut req) = req else { break }; + let parsed_store = (req.message.header().operation == Operation::StoreConsumerOffset) + .then(|| { + Self::parse_consumer_offset_request( + Operation::StoreConsumerOffset, + &req.message, + ) + }); + let _reservation = parsed_store + .as_ref() + .and_then(|parsed| parsed.as_ref().ok()) + .and_then(|(kind, id, _, _)| { + if !is_auto_commit_client(req.message.header().client) { + return None; + } + let mut queued = self.queued_auto_commit_reservations.borrow_mut(); + let reservations = queued.get_mut(&(*kind, *id))?; + let reservation = reservations.pop(); + if reservations.is_empty() { + queued.remove(&(*kind, *id)); + } + reservation + }); // Taken before the preflight so a refusal answers the parked waiter // instead of waking it with `Canceled`. @@ -3151,6 +3754,55 @@ where break; } + if let Some(parsed) = parsed_store { + let Ok((kind, consumer_id, Some(offset), _)) = parsed else { + Self::send_partition_deny_or_log( + self.consensus(), + req.message.header(), + IggyError::InvalidCommand.as_code(), + "queued consumer offset parse deny reply send failed", + reply_sender.take(), + ) + .await; + consecutive_denials += 1; + if consecutive_denials >= PROMOTION_DENIALS_MAX { + break; + } + continue; + }; + if let Some(error) = self.store_offset_range_error(offset) { + Self::send_partition_deny_or_log( + self.consensus(), + req.message.header(), + error.as_code(), + "queued offset range deny reply send failed", + reply_sender.take(), + ) + .await; + consecutive_denials += 1; + if consecutive_denials >= PROMOTION_DENIALS_MAX { + break; + } + continue; + } + if !self + .admit_consumer_offset_key( + req.message.header(), + kind, + consumer_id, + &mut reply_sender, + ) + .await + { + consecutive_denials += 1; + if consecutive_denials >= PROMOTION_DENIALS_MAX { + break; + } + continue; + } + } + consecutive_denials = 0; + let prepare = { let consensus = self.consensus(); assert!( @@ -3179,17 +3831,35 @@ where } prepare }; + promoted += 1; self.on_replicate(prepare).await; } } + #[must_use] + pub fn queued_requests_ready(&self) -> bool { + self.fatal.is_none() + && self.consensus.is_primary() + && self.consensus.is_normal() + && !self.consensus.is_transferring() + && !self.consensus.pipeline_is_full() + && self.consensus.request_queue_len() > 0 + } + + /// Resume a bounded promotion turn without waiting for another commit. + pub async fn resume_queued_requests(&mut self) { + if self.queued_requests_ready() { + self.drain_request_queue_into_prepares(1).await; + } + } + /// # Panics /// Panics on a primary when a prepare's op is ahead of the local /// sequencer: journaling it would make the next op assignment collide, /// which is unrecoverable in place. #[allow(clippy::future_not_send, clippy::too_many_lines)] pub async fn on_replicate(&mut self, message: Message) { - self.clear_pending_consumer_offset_commits_if_view_changed(); + self.resynchronize_consumer_offset_reservations(); let header = *message.header(); // Same reason as the metadata plane: `checksum` is compared as an opaque token // downstream, so a corrupted frame passes whenever its flipped value satisfies @@ -3525,7 +4195,7 @@ where if self.fatal.is_some() { return; } - self.clear_pending_consumer_offset_commits_if_view_changed(); + self.resynchronize_consumer_offset_reservations(); let header = *message.header(); { let consensus = self.consensus(); @@ -3600,7 +4270,7 @@ where if self.fatal.is_some() { return; } - self.clear_pending_consumer_offset_commits_if_view_changed(); + self.resynchronize_consumer_offset_reservations(); // The primary commits inline via `on_ack` (it drains its own pipeline). // Backups never populate the pipeline - they journal replicated prepares @@ -3673,7 +4343,8 @@ where Ok(frozen) } Operation::StoreConsumerOffset | Operation::DeleteConsumerOffset => { - // Replicated path is Quorum-only by construction; ack ignored. + // Replication semantics do not depend on the acknowledgement + // byte. Multi-replica `NoAck` requests enter this same path. let (kind, consumer_id, offset, _ack) = Self::parse_staged_consumer_offset_commit(header.operation, &message)?; let write_lock = self.write_lock.clone(); @@ -3948,6 +4619,10 @@ where } } self.consensus.invalidate_local_dvc_suffix(); + let commit_max = self.consensus.commit_max(); + self.pending_consumer_offset_commits + .retain(|op, _| *op < from_op || *op <= commit_max); + self.offset_reservations_need_resync.set(true); Ok(removed) } @@ -4075,6 +4750,12 @@ where let (frozen_batches, index_bytes, flush_index, batch_count, committed_info, chunk_len) = { let segment = self.log.active_segment(); let mut file_position = segment.size.as_bytes_u64(); + let persisted_end = if file_position == 0 { + segment.start_offset.checked_sub(1) + } else { + Some(segment.end_offset) + } + .max(self.recovered_durable_offset); let mut flush_index = None; let mut frozen = Vec::with_capacity(entries.len()); let mut batch_count = 0u32; @@ -4123,13 +4804,13 @@ where if message_count == 0 { continue; } - // A repaired batch at or below the boot-time recovered - // durable offset is already IN the segments this replica - // recovered; persisting it again would append duplicate - // bytes past the segment end. Evict it without writing. - // Live traffic always sits above the (immutable) line. + // Flush can run ahead of the bounded commit walk. Repair + // may re-journal an evicted batch above commit_min even + // after this process persisted it. Include the current + // segment frontier, not just the boot recovery frontier, + // so that replay cannot append a second copy. let batch_end = batch.header.base_offset + u64::from(message_count) - 1; - if let Some(durable) = self.recovered_durable_offset + if let Some(durable) = persisted_end && batch_end <= durable { continue; @@ -4320,13 +5001,22 @@ where let committed_batch_stats = self.resolve_committed_visible_offsets(&drained); let mut messages_committed = false; - for (mut entry, batch_stats) in drained.into_iter().zip(committed_batch_stats) { + // Apply the drained batch before advancing any op because directory + // durability is shared by every delete in the batch. Until the sync + // below succeeds, earlier applied entries intentionally remain above + // commit_min and their replies and dedup folds stay owned by `drained`. + // A failure at entry K fences the replica. Recovery replays the whole + // unadvanced prefix idempotently, including entries applied before K. + for cell in &self.consumer_offset_dirs_touched { + cell.set(None); + } + for (entry, batch_stats) in drained.iter().zip(&committed_batch_stats) { let prepare_header = entry.header; if !self .commit_partition_entry( prepare_header, &mut messages_committed, - batch_stats, + *batch_stats, &mut failed_commit, config, ) @@ -4367,7 +5057,49 @@ where }); return; } + } + + // Commit replies and the applied frontier must follow directory + // durability. One sync per dirty kind covers the whole walk. A sync + // failure is attributed to an operation of that kind in this walk. + // Covered stores depend on previously dirty directory entries too. + // Only a kind THIS walk uses can fence it: dirt left by a NoAck + // request belongs to that request's kind, and fencing a walk that wrote + // consumer offsets over the groups directory would take the node down + // for a failure none of its ops caused. + let touched = [ + self.consumer_offset_dirs_touched[0].replace(None), + self.consumer_offset_dirs_touched[1].replace(None), + ]; + let failed = self + .flush_consumer_offset_directories_for(touched.map(|entry| entry.is_some())) + .await; + if failed.iter().any(|failed| *failed) { + let failed_entry = failed + .iter() + .zip(touched) + .find_map(|(failed, entry)| if *failed { entry } else { None }); + error!( + namespace_raw, + failed_consumer = failed[0], + failed_consumer_group = failed[1], + touched_consumer = touched[0].is_some(), + touched_consumer_group = touched[1].is_some(), + fences = failed_entry.is_some(), + "consumer offset directory sync failed after committed operations" + ); + if let Some((op, operation)) = failed_entry { + self.fatal = Some(FatalCommit { + namespace_raw, + op, + operation, + }); + return; + } + } + for (mut entry, batch_stats) in drained.into_iter().zip(committed_batch_stats) { + let prepare_header = entry.header; self.consensus.advance_commit_min(prepare_header.op); // Fold the committed request into this group's dedup slice. Runs on @@ -4412,7 +5144,14 @@ where // `AUTO_COMMIT_CLIENT_ID`: no client ever waits on it, so skip the // reply. Emitting it would push an unrequested frame onto a real // client's lockstep reply stream if the sentinel ever routed there. - if send_client_replies && !is_auto_commit_client(prepare_header.client) { + if is_auto_commit_client(prepare_header.client) { + if let Some(sender) = entry.take_reply_sender() { + let _ = sender.send(build_reply_message( + &prepare_header, + &committed_reply_body(prepare_header.operation), + )); + } + } else if send_client_replies { let body = match prepare_header.operation { Operation::SendMessages => { send_messages_reply_body(prepare_header.group, batch_stats) @@ -4470,6 +5209,69 @@ where self.drain_request_queue_into_prepares(drained_count).await; } + /// Sync the dirty offset directories selected by `kinds`, indexed by + /// `consumer_kind_index`. Returns which of them failed. A failed directory + /// stays dirty for the next attempt. + async fn flush_consumer_offset_directories_for(&self, kinds: [bool; 2]) -> [bool; 2] { + let mut failed = [false; 2]; + for (index, dir) in [ + self.consumer_offsets_path.as_deref(), + self.consumer_group_offsets_path.as_deref(), + ] + .into_iter() + .enumerate() + { + if !kinds[index] || !self.consumer_offset_dirs_dirty[index].get() { + continue; + } + #[cfg(test)] + if self.consumer_offset_dir_sync_fault.get() == Some(index) { + failed[index] = true; + continue; + } + if let Some(dir) = dir { + match crate::state_transfer::fsync_dir(dir).await { + Ok(()) => {} + Err(error) if error.kind() == std::io::ErrorKind::NotFound => { + // The directory disappeared after the final unlink. + // There is no remaining dirent whose durability needs + // proving. + } + Err(error) => { + warn!( + target: "iggy.partitions.diag", + plane = "partitions", + replica_id = self.consensus.replica(), + namespace_raw = self.namespace().inner(), + path = dir, + error_kind = ?error.kind(), + %error, + "consumer offset directory sync failed" + ); + failed[index] = true; + continue; + } + } + #[cfg(test)] + self.offset_dir_sync_count + .set(self.offset_dir_sync_count.get() + 1); + } + self.consumer_offset_dirs_dirty[index].set(false); + } + failed + } + + fn mark_consumer_offset_dir_dirty(&self, kind: ConsumerKind) { + let index = crate::state_transfer::consumer_kind_index(kind); + self.consumer_offset_dirs_dirty[index].set(true); + } + + /// Keys of `kind` whose offset file could not be loaded or unlinked. + #[must_use] + pub fn stranded_consumer_offset_count(&self, kind: ConsumerKind) -> usize { + self.consumer_offset_capacity_for(kind).stranded_count() + } + /// Batch stats for each drained entry, positionally parallel to `drained`. /// Every entry contributes exactly one slot (`None` for the operations that /// carry no batch), which is what makes the pairing correct by @@ -4672,6 +5474,9 @@ where send_fail_label: &'static str, waiter: Option>>, ) { + if waiter.is_none() && is_auto_commit_client(header.client) { + return; + } let reply = build_deny_reply_from_request(consensus, header, status); Self::deliver_reply_or_log(consensus, header, reply, waiter, send_fail_label).await; } @@ -4718,17 +5523,17 @@ where .inner .repair_entry(op) .ok_or(IggyError::InvalidCommand)?; - // Deep copy: the journal buffer is shared and `Message::try_from` - // wants an `Owned`; this path only runs on the post-view-change - // fallback, never per-commit. - let owned = Owned::::copy_from_slice(entry.as_slice()); - let message = Message::::try_from(owned) - .map_err(|_| IggyError::InvalidCommand)? - .try_into_typed::() + let bytes = entry.as_slice(); + let header_bytes = bytes + .get(..size_of::()) + .ok_or(IggyError::InvalidCommand)?; + let header = bytemuck::checked::try_from_bytes::(header_bytes) .map_err(|_| IggyError::InvalidCommand)?; - let header = *message.header(); + let body = bytes + .get(size_of::()..header.size as usize) + .ok_or(IggyError::InvalidCommand)?; let (kind, consumer_id, offset, _ack) = - Self::parse_staged_consumer_offset_commit(header.operation, &message)?; + Self::parse_consumer_offset_payload(header.operation, body)?; match header.operation { Operation::StoreConsumerOffset => { let offset = offset.ok_or(IggyError::InvalidCommand)?; @@ -4806,8 +5611,13 @@ where // from the kept journal entries, so a cleared table alone would be // repopulated from the entry this guard is fencing. if prepare_header.op <= self.purge_floor_op { - self.pending_consumer_offset_commits - .remove(&prepare_header.op); + if let Some(pending) = self + .pending_consumer_offset_commits + .remove(&prepare_header.op) + && matches!(pending.mutation, PendingConsumerOffsetMutation::Upsert(_)) + { + self.release_consumer_offset_reservation(pending.kind, pending.consumer_id); + } return true; } @@ -5680,27 +6490,29 @@ where // Clear consumer + consumer-group offsets (memory + disk). Collect the // file paths before deleting so the map guard is not held across an // await. - let consumer_paths: Vec = { + let consumer_paths: Vec<(ConsumerKind, u32, String)> = { let guard = self.consumer_offsets.pin(); let paths = guard .iter() .filter_map(|(key, _)| { - u32::try_from(*key) - .ok() - .and_then(|id| self.persisted_offset_path(ConsumerKind::Consumer, id)) + u32::try_from(*key).ok().and_then(|id| { + self.persisted_offset_path(ConsumerKind::Consumer, id) + .map(|path| (ConsumerKind::Consumer, id, path)) + }) }) .collect(); guard.clear(); paths }; - let group_paths: Vec = { + let group_paths: Vec<(ConsumerKind, u32, String)> = { let guard = self.consumer_group_offsets.pin(); let paths = guard .iter() .filter_map(|(key, _)| { - u32::try_from(key.0) - .ok() - .and_then(|id| self.persisted_offset_path(ConsumerKind::ConsumerGroup, id)) + u32::try_from(key.0).ok().and_then(|id| { + self.persisted_offset_path(ConsumerKind::ConsumerGroup, id) + .map(|path| (ConsumerKind::ConsumerGroup, id, path)) + }) }) .collect(); guard.clear(); @@ -5710,15 +6522,43 @@ where // full reset, and an offset file the live map never held -- a pre-purge // op re-persisted by journal repair on a restarted replica -- would // otherwise survive for boot to hydrate back. - let strayed = - crate::state_transfer::strayed_offset_files(self.consumer_offsets_path.as_deref(), &[]) + let strayed_consumers = + crate::state_transfer::strayed_offset_files(self.consumer_offsets_path.as_deref()) .into_iter() - .chain(crate::state_transfer::strayed_offset_files( - self.consumer_group_offsets_path.as_deref(), - &[], - )); - for path in consumer_paths.into_iter().chain(group_paths).chain(strayed) { - let _ = delete_persisted_offset(&path).await; + .filter_map(|path| { + crate::state_transfer::numeric_offset_id(&path) + .map(|id| (ConsumerKind::Consumer, id, path)) + }); + let strayed_groups = crate::state_transfer::strayed_offset_files( + self.consumer_group_offsets_path.as_deref(), + ) + .into_iter() + .filter_map(|path| { + crate::state_transfer::numeric_offset_id(&path) + .map(|id| (ConsumerKind::ConsumerGroup, id, path)) + }); + for (kind, consumer_id, path) in consumer_paths + .into_iter() + .chain(group_paths) + .chain(strayed_consumers) + .chain(strayed_groups) + { + if let Err(error) = delete_persisted_offset(&path).await { + self.consumer_offset_capacity_for(kind) + .record_stranded(consumer_id); + warn!( + target: "iggy.partitions.diag", + plane = "partitions", + namespace_raw = namespace.inner(), + generation, + path, + %error, + "purge could not remove a consumer offset file" + ); + } else { + self.consumer_offset_capacity_for(kind) + .clear_stranded(consumer_id); + } } // Directory fsync so those unlinks stick, mirroring the install path: a // crash right after the purge otherwise resurrects the offset files at @@ -5747,10 +6587,13 @@ where ); } } - // The persisted-offset tracker mirrors the files unlinked above; a - // stale entry would make a post-purge auto-commit skip its write and - // lose the offset on restart. - self.persisted_offsets.borrow_mut().clear(); + self.durable_consumer_offsets.clear(); + self.pending_consumer_offset_commits.clear(); + self.queued_auto_commit_reservations.borrow_mut().clear(); + self.consumer_offset_capacity + .rebuild(&self.durable_consumer_offsets, std::iter::empty()); + self.consumer_group_offset_capacity + .rebuild(&self.durable_consumer_offsets, std::iter::empty()); // Clear the ephemeral cooperative-rebalance tracking too: after the // reset to offset 0 a stale `last_polled` (a high pre-purge offset) @@ -5944,6 +6787,7 @@ where consensus.sequencer().set_sequence(frontier); consensus.set_last_prepare_checksum(frontier_checksum); } + self.offset_reservations_need_resync.set(true); } /// Conclude a repair stream: settle the commit floor at the serving @@ -6399,6 +7243,11 @@ fn accumulate_committed_info( /// it -- only a backlog does, and that one drains over several passes. const SEGMENT_REMOVAL_BUDGET_PER_PASS: usize = 16; +/// Consecutive denials that end one promotion drain. Denials free no slot, so +/// without this bound a single commit turn could answer every queued request, +/// each with an awaited reply send. +const PROMOTION_DENIALS_MAX: usize = 4; + /// What one call to [`IggyPartition::remove_sealed_segments_up_to`] reclaimed. /// /// `budget_spent` reports that the pass stopped on @@ -6493,6 +7342,7 @@ mod tests { use iggy_binary_protocol::{Command, ReplyHeader, WireConsumer, WireEncode}; use message_bus::{BusMessage, SendError}; use server_common::MESSAGE_ALIGN; + use server_common::iobuf::Owned; use server_common::send_messages::{ COMMAND_HEADER_SIZE, IggyMessage, IggyMessageHeader, IggyMessages, SendMessagesOwned, decode_batch_slice, @@ -7748,69 +8598,983 @@ mod tests { }) } - /// Deleting a consumer offset that was never stored must answer with a - /// typed deny reply (empty body, `status` = `ConsumerOffsetNotFound`, - /// `op` 0) before consensus: nothing may enter the pipeline, and an - /// awaited client write must fail fast instead of waiting out its reply - /// timeout. Once the offset exists, the same request must pass the gate - /// into the pipeline without a deny. - #[compio::test] - async fn on_request_delete_of_missing_offset_replies_typed_deny() { - let (mut partition, sent_to_clients) = recording_partition(); - let client_id: u128 = 42; - let consumer_id: u32 = 5; + fn store_offset_request( + client_id: u128, + request_id: u64, + kind: ConsumerKind, + consumer_id: u32, + offset: u64, + ack: AckLevel, + ) -> Message { + let body = StoreConsumerOffsetRequest { + consumer: WireConsumer { + kind: kind.as_code(), + id: WireIdentifier::Numeric(consumer_id), + }, + stream_id: WireIdentifier::Numeric(1), + topic_id: WireIdentifier::Numeric(1), + partition_id: Some(0), + offset, + ack, + } + .to_bytes(); + let header_size = std::mem::size_of::(); + let total = header_size + body.len(); + let mut message = Message::::new(total); + message.as_mut_slice()[header_size..].copy_from_slice(&body); + message.transmute_header(|_, header: &mut RoutedRequestHeader| { + header.command = Command::Request; + header.operation = Operation::StoreConsumerOffset; + header.client = client_id; + header.session = 1; + header.request = request_id; + header.group = IggyNamespace::new(1, 1, 0).inner(); + header.size = u32::try_from(total).expect("request size fits u32"); + }) + } + #[compio::test] + async fn given_transfer_above_cap_when_file_write_fails_should_refuse_until_retry_succeeds() { + let dir = tempfile::tempdir().expect("transfer directory"); + let consumers = dir.path().join("consumers"); + let groups = dir.path().join("groups"); + std::fs::File::create(&groups).expect("block group-file creation"); + let mut partition = test_partition(); + partition.set_partition_dir(dir.path().to_string_lossy().into_owned()); + partition.set_consumer_offsets_max(1); + partition.consumer_offsets_path = Some(consumers.to_string_lossy().into_owned()); + partition.consumer_group_offsets_path = Some(groups.to_string_lossy().into_owned()); + let wire = crate::state_transfer::ConsumerOffsetsWire { + purge_generation: 0, + next_offset: 10, + consumers: vec![(7, 1), (8, 2)], + groups: vec![(9, 3)], + dedup: Vec::new(), + }; partition - .on_request(delete_offset_request(client_id, 7, consumer_id), None) - .await; - - { - let sent = sent_to_clients.borrow(); - assert_eq!(sent.len(), 1, "exactly one deny reply"); - let (reply_client, frame) = &sent[0]; - assert_eq!(*reply_client, client_id); - let header = bytemuck::checked::try_from_bytes::( - &frame.as_slice()[..std::mem::size_of::()], - ) - .expect("deny frame starts with a valid reply header"); - assert_eq!(header.command, Command::Reply); - assert_eq!( - header.status, - IggyError::ConsumerOffsetNotFound(0).as_code() - ); - assert_eq!(header.op, 0, "a deny commits nothing"); - assert_eq!(header.request, 7); - assert_eq!( - header.size as usize, - std::mem::size_of::(), - "deny reply body must be empty" - ); - } + .install_state_transfer(&repair_config(), 12, Vec::new(), &wire.encode(), 0) + .await + .expect_err("incomplete offset state must not advance the commit floor"); + assert_eq!(partition.consensus.commit_min(), 0); assert_eq!( - partition.consensus().pipeline_len(), - 0, - "denied delete must not replicate" + partition.log.segments().len(), + 1, + "offset staging failure must preserve the existing segment chain" ); - assert!(partition.pending_consumer_offset_commits.is_empty()); - - // Existing offset: the gate passes and the delete enters the pipeline. - partition.consumer_offsets.pin().insert( - consumer_id as usize, - ConsumerOffset::new(ConsumerKind::Consumer, consumer_id, 3, String::new()), + assert_eq!( + partition.durable_consumer_offset_count(ConsumerKind::Consumer), + 0 ); - partition - .on_request(delete_offset_request(client_id, 8, consumer_id), None) - .await; assert_eq!( - partition.consensus().pipeline_len(), - 1, - "existing offset delete must replicate" + partition.durable_consumer_offset_count(ConsumerKind::ConsumerGroup), + 0 ); - } - - #[compio::test] - async fn on_request_delete_on_stale_backup_replies_transient_before_not_found() { - let (mut partition, sent_to_clients) = recording_partition_at(1, 3); + std::fs::remove_file(&groups).expect("remove file-write fault"); + let outcome = partition + .install_state_transfer(&repair_config(), 12, Vec::new(), &wire.encode(), 0) + .await + .expect("retry must install accepted state above the local cap"); + assert!(outcome.purge_generation_recorded); + assert_eq!(partition.consensus.commit_min(), 12); + assert_eq!( + partition.durable_consumer_offset_count(ConsumerKind::Consumer), + 2 + ); + assert_eq!( + partition.durable_consumer_offset_count(ConsumerKind::ConsumerGroup), + 1 + ); + assert!( + partition + .reserve_consumer_offset(ConsumerKind::Consumer, 10) + .is_err() + ); + assert!( + partition + .durable_consumer_offsets + .covers(ConsumerKind::ConsumerGroup, 9, 3) + ); + assert_eq!( + partition + .offsets_wire_snapshot_for_test() + .expect("durable snapshot"), + vec![(7, 1), (8, 2)] + ); + } + + #[compio::test] + async fn given_committed_deletes_when_drained_together_should_sync_directory_once_before_replies() + { + let dir = tempfile::tempdir().unwrap(); + let (mut partition, sent) = recording_partition_at(0, 3); + partition.consumer_offsets_path = Some(dir.path().to_string_lossy().into_owned()); + partition.consumer_offset_enforce_fsync = true; + let mut drained = Vec::new(); + for id in 1..=2 { + partition + .persist_consumer_offset_commit(PendingConsumerOffsetCommit::upsert( + ConsumerKind::Consumer, + id, + 0, + )) + .await + .unwrap(); + partition.stage_consumer_offset_delete(u64::from(id), ConsumerKind::Consumer, id); + let header = PrepareHeader { + op: u64::from(id), + operation: Operation::DeleteConsumerOffset, + client: 42, + request: u64::from(id), + ..Default::default() + }; + drained.push(PipelineEntry::new(header)); + } + partition.consensus.restore_commit_state(0, 2); + partition + .handle_committed_entries(drained, &repair_config(), true) + .await; + assert_eq!(partition.offset_dir_sync_count.get(), 1); + assert_eq!(partition.consensus.commit_min(), 2); + assert_eq!(sent.borrow().len(), 2); + assert!(!dir.path().join("1").exists()); + assert!(!dir.path().join("2").exists()); + } + + #[compio::test] + async fn given_absent_offset_file_when_delete_commits_should_skip_directory_sync() { + let dir = tempfile::tempdir().unwrap(); + let (mut partition, sent) = recording_partition_at(0, 3); + partition.consumer_offsets_path = + Some(dir.path().join("missing").to_string_lossy().into_owned()); + partition.stage_consumer_offset_delete(1, ConsumerKind::Consumer, 7); + partition.consensus.restore_commit_state(0, 1); + let header = PrepareHeader { + op: 1, + operation: Operation::DeleteConsumerOffset, + client: 42, + request: 1, + ..Default::default() + }; + partition + .handle_committed_entries(vec![PipelineEntry::new(header)], &repair_config(), true) + .await; + assert!(partition.fatal.is_none()); + assert_eq!(partition.consensus.commit_min(), 1); + assert_eq!(partition.offset_dir_sync_count.get(), 0); + assert_eq!(sent.borrow().len(), 1); + } + + #[compio::test] + async fn given_one_offset_directory_sync_failure_when_flushing_should_sync_the_other_directory() + { + let dir = tempfile::tempdir().unwrap(); + let invalid = dir.path().join("file"); + std::fs::write(&invalid, b"not a directory").unwrap(); + let (mut partition, _) = recording_partition(); + partition.consumer_offsets_path = + Some(invalid.join("child").to_string_lossy().into_owned()); + partition.consumer_group_offsets_path = Some(dir.path().to_string_lossy().into_owned()); + partition.mark_consumer_offset_dir_dirty(ConsumerKind::Consumer); + partition.mark_consumer_offset_dir_dirty(ConsumerKind::ConsumerGroup); + assert_eq!( + partition + .flush_consumer_offset_directories_for([true, true]) + .await, + [true, false] + ); + assert!(partition.consumer_offset_dirs_dirty[0].get()); + assert!(!partition.consumer_offset_dirs_dirty[1].get()); + assert_eq!(partition.offset_dir_sync_count.get(), 1); + } + + #[compio::test] + async fn given_covered_auto_commit_when_persisting_should_not_sync_directory() { + let dir = tempfile::tempdir().unwrap(); + let (mut partition, _) = recording_partition(); + partition.consumer_offsets_path = Some(dir.path().to_string_lossy().into_owned()); + partition.consumer_offset_enforce_fsync = true; + let path = dir.path().join("7"); + persist_offset(path.to_str().unwrap(), 10, true) + .await + .unwrap(); + for offset in [5, 7] { + partition + .persist_consumer_offset_commit(PendingConsumerOffsetCommit::upsert_auto_commit( + ConsumerKind::Consumer, + 7, + offset, + )) + .await + .unwrap(); + assert!(!partition.consumer_offset_dirs_dirty[0].get()); + } + partition + .persist_consumer_offset_commit(PendingConsumerOffsetCommit::upsert_auto_commit( + ConsumerKind::Consumer, + 7, + 11, + )) + .await + .unwrap(); + assert!(partition.consumer_offset_dirs_dirty[0].get()); + } + + #[compio::test] + async fn given_empty_transfer_frontier_when_incoming_offsets_are_clamped_out_should_remove_old_files() + { + let dir = tempfile::tempdir().unwrap(); + let mut partition = test_partition(); + partition.set_partition_dir(dir.path().to_string_lossy().into_owned()); + let consumers = dir.path().join("offsets/consumers"); + let groups = dir.path().join("offsets/groups"); + partition.consumer_offsets_path = Some(consumers.to_string_lossy().into_owned()); + partition.consumer_group_offsets_path = Some(groups.to_string_lossy().into_owned()); + for path in [consumers.join("7"), groups.join("8")] { + persist_offset(path.to_str().unwrap(), 5, true) + .await + .unwrap(); + } + let wire = crate::state_transfer::ConsumerOffsetsWire { + purge_generation: 1, + next_offset: 0, + consumers: vec![(7, 5)], + groups: vec![(8, 5)], + dedup: Vec::new(), + }; + partition + .install_state_transfer(&repair_config(), 1, Vec::new(), &wire.encode(), 0) + .await + .unwrap(); + assert!(!consumers.join("7").exists()); + assert!(!groups.join("8").exists()); + assert_eq!( + partition.durable_consumer_offset_count(ConsumerKind::Consumer), + 0 + ); + assert_eq!( + partition.durable_consumer_offset_count(ConsumerKind::ConsumerGroup), + 0 + ); + } + + #[compio::test] + async fn given_committed_store_when_offset_persist_fails_should_set_fatal_commit() { + let dir = tempfile::tempdir().expect("temporary offset parent"); + let parent = dir.path().join("not-a-directory"); + std::fs::File::create(&parent).expect("create invalid directory fixture"); + let (mut partition, _) = recording_partition_at(0, 3); + partition.consumer_offsets_path = Some(parent.to_string_lossy().into_owned()); + partition.stats.increment_messages_count(1); + partition + .on_request( + store_offset_request(42, 1, ConsumerKind::Consumer, 7, 0, AckLevel::Quorum), + None, + ) + .await; + let header = partition + .log + .journal() + .inner + .header_by_op(1) + .expect("prepared offset op"); + partition.consensus.restore_commit_state(0, 1); + partition + .handle_committed_entries(vec![PipelineEntry::new(header)], &repair_config(), true) + .await; + let fatal = partition + .fatal + .as_ref() + .expect("committed persistence failure fences partition"); + assert_eq!(fatal.operation, Operation::StoreConsumerOffset); + assert_eq!(fatal.op, 1); + assert_eq!( + partition.durable_consumer_offset_count(ConsumerKind::Consumer), + 0 + ); + } + + #[compio::test] + async fn given_queued_capacity_denial_when_promoting_should_use_the_slot_for_next_existing_key() + { + let (mut partition, _) = recording_partition_at(0, 3); + partition.set_consumer_offsets_max(1); + partition.seed_recovered_consumer_offset(ConsumerKind::Consumer, 7, 0, 0); + partition.consumer_offsets.pin().insert( + 7, + ConsumerOffset::new(ConsumerKind::Consumer, 7, 0, String::new()), + ); + partition.stats.increment_messages_count(1); + for (request_id, consumer_id) in [(1, 8), (2, 7)] { + assert!( + partition + .consensus + .push_queued_request(consensus::RequestEntry::with_sender( + store_offset_request( + 42, + request_id, + ConsumerKind::Consumer, + consumer_id, + 0, + AckLevel::Quorum + ), + None, + )) + .is_ok() + ); + } + partition.drain_request_queue_into_prepares(1).await; + assert_eq!(partition.consensus.pipeline_len(), 1); + assert_eq!( + partition + .pending_consumer_offset_commits + .get(&1) + .expect("existing update projected") + .consumer_id, + 7 + ); + assert!(partition.consensus.pop_queued_request().is_none()); + } + + #[compio::test] + async fn given_same_view_truncation_when_resynchronizing_should_release_discarded_reservations() + { + let mut partition = test_partition(); + journal_prepare(&partition, 1, Operation::StoreConsumerOffset).await; + journal_prepare(&partition, 2, Operation::StoreConsumerOffset).await; + partition.consensus.sequencer().set_sequence(2); + partition.stage_consumer_offset_upsert(1, ConsumerKind::Consumer, 7, 0, false); + partition.stage_consumer_offset_upsert(2, ConsumerKind::Consumer, 8, 0, false); + partition.truncate_uncommitted_from(2).await.unwrap(); + partition.resynchronize_consumer_offset_reservations(); + assert!(partition.pending_consumer_offset_commits.contains_key(&1)); + assert!(!partition.pending_consumer_offset_commits.contains_key(&2)); + assert_eq!( + partition.occupied_consumer_offset_count(ConsumerKind::Consumer), + 1 + ); + assert!(!partition.consumer_offset_capacity.is_uncertain()); + } + + #[compio::test] + async fn given_promotion_denial_budget_when_no_more_commits_arrive_should_resume_queued_work() { + let (mut partition, sent) = recording_partition_at(0, 3); + partition.set_consumer_offsets_max(1); + partition.seed_recovered_consumer_offset(ConsumerKind::Consumer, 7, 0, 0); + partition.consumer_offsets.pin().insert( + 7, + ConsumerOffset::new(ConsumerKind::Consumer, 7, 0, String::new()), + ); + partition.stats.increment_messages_count(1); + for (request, id) in [8, 9, 10, 11, 12, 7].into_iter().enumerate() { + partition + .consensus + .push_queued_request(consensus::RequestEntry::with_sender( + store_offset_request( + 42, + request as u64 + 1, + ConsumerKind::Consumer, + id, + 0, + AckLevel::Quorum, + ), + None, + )) + .unwrap(); + } + partition.drain_request_queue_into_prepares(1).await; + assert_eq!(sent.borrow().len(), PROMOTION_DENIALS_MAX); + assert_eq!(partition.consensus.request_queue_len(), 2); + assert_eq!(partition.consensus.pipeline_len(), 0); + assert!(partition.queued_requests_ready()); + partition.resume_queued_requests().await; + assert_eq!(partition.consensus.request_queue_len(), 0); + assert_eq!(partition.consensus.pipeline_len(), 1); + assert!(!partition.queued_requests_ready()); + } + + #[compio::test] + async fn given_covered_store_with_dirty_directory_when_sync_fails_should_fence_its_own_operation() + { + let dir = tempfile::tempdir().unwrap(); + let (mut partition, sent) = recording_partition_at(0, 3); + partition.consumer_offsets_path = Some(dir.path().to_string_lossy().into_owned()); + partition.consumer_offset_enforce_fsync = true; + let pending = + PendingConsumerOffsetCommit::upsert_auto_commit(ConsumerKind::Consumer, 7, 10); + partition + .persist_consumer_offset_commit(pending) + .await + .unwrap(); + partition.apply_consumer_offset_commit(pending); + partition.consumer_offset_dir_sync_fault.set(Some(0)); + partition.stage_consumer_offset_upsert(1, ConsumerKind::Consumer, 7, 5, true); + partition.consensus.restore_commit_state(0, 1); + let header = PrepareHeader { + op: 1, + operation: Operation::StoreConsumerOffset, + client: 42, + request: 1, + ..Default::default() + }; + partition + .handle_committed_entries(vec![PipelineEntry::new(header)], &repair_config(), true) + .await; + assert_eq!(partition.fatal.as_ref().unwrap().op, 1); + assert_eq!( + partition.fatal.as_ref().unwrap().operation, + Operation::StoreConsumerOffset + ); + assert_eq!(partition.consensus.commit_min(), 0); + assert!(sent.borrow().is_empty()); + assert!(partition.consumer_offset_dirs_dirty[0].get()); + } + + #[compio::test] + async fn given_repeated_admitted_deletes_when_committed_should_be_idempotent() { + let (mut partition, _) = recording_partition(); + partition.consumer_offsets.pin().insert( + 7, + ConsumerOffset::new(ConsumerKind::Consumer, 7, 0, String::new()), + ); + partition.seed_recovered_consumer_offset(ConsumerKind::Consumer, 7, 0, 0); + for op in 1..=2 { + partition.stage_consumer_offset_delete(op, ConsumerKind::Consumer, 7); + } + for op in 1..=2 { + partition + .apply_staged_consumer_offset_commit(op) + .await + .expect("admitted delete converges"); + } + assert_eq!( + partition.occupied_consumer_offset_count(ConsumerKind::Consumer), + 0 + ); + } + + #[compio::test] + async fn given_queued_auto_commit_when_view_changes_should_release_its_provisional_slot() { + let namespace = IggyNamespace::new(1, 1, 0); + let consensus = VsrConsensus::new( + TEST_CLUSTER, + 0, + 3, + namespace.inner(), + RecordingBus::default(), + LocalPipeline::with_capacities(1, 2), + ); + consensus.init(); + let mut partition: IggyPartition = IggyPartition::with_in_memory_storage( + Arc::new(PartitionStats::default()), + consensus, + IggyByteSize::from(1024 * 1024), + false, + ); + partition.stats.increment_messages_count(1); + partition.set_consumer_offsets_max(2); + partition + .on_request( + store_offset_request(42, 1, ConsumerKind::Consumer, 7, 0, AckLevel::Quorum), + None, + ) + .await; + let reservation = partition + .consumer_offset_capacity + .reserve_provisional(8, &partition.durable_consumer_offsets) + .unwrap(); + partition + .on_request_with_reservation( + store_offset_request( + message_bus::AUTO_COMMIT_CLIENT_ID, + 1, + ConsumerKind::Consumer, + 8, + 0, + AckLevel::Quorum, + ), + None, + Some(reservation), + ) + .await; + assert_eq!( + partition.occupied_consumer_offset_count(ConsumerKind::Consumer), + 2 + ); + assert_eq!(partition.queued_auto_commit_reservations.borrow().len(), 1); + partition.consensus.set_view(3); + partition.resynchronize_consumer_offset_reservations(); + assert!( + partition + .queued_auto_commit_reservations + .borrow() + .is_empty() + ); + assert_eq!( + partition.occupied_consumer_offset_count(ConsumerKind::Consumer), + 1 + ); + } + + #[compio::test] + async fn given_auto_commit_guard_when_pump_admits_should_transfer_to_journal_without_reply() { + let (mut partition, sent) = recording_partition_at(0, 3); + partition.stats.increment_messages_count(1); + partition.set_consumer_offsets_max(1); + let reservation = partition + .consumer_offset_capacity + .reserve_provisional(7, &partition.durable_consumer_offsets) + .unwrap(); + partition + .on_request_with_reservation( + store_offset_request( + message_bus::AUTO_COMMIT_CLIENT_ID, + 1, + ConsumerKind::Consumer, + 7, + 0, + AckLevel::Quorum, + ), + None, + Some(reservation), + ) + .await; + assert_eq!(partition.pending_consumer_offset_commits.len(), 1); + assert_eq!( + partition.occupied_consumer_offset_count(ConsumerKind::Consumer), + 1 + ); + assert!(sent.borrow().is_empty()); + partition + .apply_staged_consumer_offset_commit(1) + .await + .unwrap(); + assert_eq!( + partition.durable_consumer_offset_count(ConsumerKind::Consumer), + 1 + ); + assert!(sent.borrow().is_empty()); + } + + #[compio::test] + async fn given_group_offset_updates_when_key_already_exists_should_keep_reconciliation_idle() { + let (mut partition, _) = recording_partition(); + let epoch = partition.consumer_group_offsets_reconcile_epoch.clone(); + let initial = epoch.get(); + partition.stage_consumer_offset_upsert(1, ConsumerKind::ConsumerGroup, 7, 1, true); + partition + .apply_staged_consumer_offset_commit(1) + .await + .unwrap(); + assert!(epoch.get() > initial); + let after_create = epoch.get(); + assert!( + partition + .dead_consumer_group_offset_ids(|_| true) + .is_empty() + ); + assert_eq!(epoch.get(), after_create); + partition.stage_consumer_offset_upsert(2, ConsumerKind::ConsumerGroup, 7, 2, true); + partition + .apply_staged_consumer_offset_commit(2) + .await + .unwrap(); + assert_eq!(epoch.get(), after_create); + partition.consumer_group_offsets.pin().insert( + ConsumerGroupId(8), + ConsumerOffset::new(ConsumerKind::ConsumerGroup, 8, 3, String::new()), + ); + partition.stage_consumer_offset_upsert(3, ConsumerKind::ConsumerGroup, 8, 3, true); + partition + .apply_staged_consumer_offset_commit(3) + .await + .unwrap(); + assert!( + epoch.get() > after_create, + "first durable commit must discover an eager map key" + ); + } + + #[compio::test] + async fn given_primary_view_change_when_map_pressure_arrives_should_reclaim_only_unprotected_key() + { + let (mut partition, _) = recording_partition_at(0, 3); + partition.set_consumer_offsets_max(4); + partition.stats.increment_messages_count(1); + partition.seed_recovered_consumer_offset(ConsumerKind::Consumer, 7, 0, 0); + partition + .on_request( + store_offset_request(42, 1, ConsumerKind::Consumer, 8, 0, AckLevel::Quorum), + None, + ) + .await; + let held = partition + .consumer_offset_capacity + .reserve_provisional(9, &partition.durable_consumer_offsets) + .unwrap(); + for id in 7..=10 { + partition.consumer_offsets.pin().insert( + id as usize, + ConsumerOffset::new(ConsumerKind::Consumer, id, 0, String::new()), + ); + } + partition.consensus.set_view(3); + partition.resynchronize_consumer_offset_reservations(); + assert_eq!(partition.consumer_offsets.len(), 4); + partition.reclaim_phantom_offsets(ConsumerKind::Consumer, 4); + assert_eq!(partition.consumer_offsets.len(), 3); + assert!(!partition.consumer_offsets.pin().contains_key(&10)); + for id in 7..=9 { + assert!(partition.consumer_offsets.pin().contains_key(&id)); + } + drop(held); + } + + #[test] + fn given_full_live_map_when_polling_existing_key_should_reclaim_only_for_new_keys() { + let (mut partition, _) = recording_partition(); + partition.set_consumer_offsets_max(2); + partition.offset_space.committed_seeded = true; + partition.seed_recovered_consumer_offset(ConsumerKind::Consumer, 7, 0, 0); + for id in 7..=8 { + partition.consumer_offsets.pin().insert( + id as usize, + ConsumerOffset::new(ConsumerKind::Consumer, id, 0, String::new()), + ); + } + let args = PollingArgs::new(iggy_common::PollingStrategy::first(), 1, true); + let _ = partition.build_poll_plan(PollingConsumer::Consumer(7, 0), &args, false); + assert_eq!(partition.consumer_offsets.len(), 2); + let _ = partition.build_poll_plan(PollingConsumer::Consumer(9, 0), &args, false); + assert_eq!(partition.consumer_offsets.len(), 1); + assert!(partition.consumer_offsets.pin().contains_key(&7)); + } + + #[compio::test] + async fn given_missing_retained_header_when_journal_progresses_should_retry_without_view_change() + { + let mut partition = test_partition(); + journal_prepare(&partition, 2, Operation::SendMessages).await; + partition.consensus.sequencer().set_sequence(2); + partition.consensus.set_view(1); + partition.resynchronize_consumer_offset_reservations(); + assert!(partition.consumer_offset_capacity.is_uncertain()); + assert!(!partition.offset_reservations_need_resync.get()); + let failed_scan = partition.offset_reservations_scan_state; + for op in 3..=20 { + journal_prepare(&partition, op, Operation::SendMessages).await; + partition.consensus.sequencer().set_sequence(op); + partition.offset_reservations_need_resync.set(true); + partition.resynchronize_consumer_offset_reservations(); + assert_eq!(partition.offset_reservations_scan_state, failed_scan); + } + partition.retry_consumer_offset_reservations(); + assert_ne!(partition.offset_reservations_scan_state, failed_scan); + assert!(partition.consumer_offset_capacity.is_uncertain()); + partition.seed_recovered_consumer_offset(ConsumerKind::Consumer, 7, 0, 0); + let existing = partition + .auto_commit_ctx(PollingConsumer::Consumer(7, 0), true) + .unwrap() + .apply(0) + .unwrap(); + assert!(partition.auto_commit_admission_ready(&existing)); + let new = partition + .auto_commit_ctx(PollingConsumer::Consumer(8, 0), true) + .unwrap() + .apply(0) + .unwrap(); + assert!(!partition.auto_commit_admission_ready(&new)); + partition.log.journal().inner.clear_all(); + for op in 1..=3 { + journal_prepare(&partition, op, Operation::SendMessages).await; + } + partition.consensus.sequencer().set_sequence(3); + partition.resynchronize_consumer_offset_reservations(); + assert!(partition.consumer_offset_capacity.is_uncertain()); + partition.retry_consumer_offset_reservations(); + assert!(!partition.consumer_offset_capacity.is_uncertain()); + } + + #[compio::test] + async fn given_evicted_committed_prefix_when_promoted_should_preserve_staging_without_latching() + { + let mut partition = test_partition(); + partition.consensus.restore_commit_state(0, 5000); + partition.consensus.sequencer().set_sequence(5001); + partition.stage_consumer_offset_upsert(4999, ConsumerKind::Consumer, 7, 0, false); + journal_prepare(&partition, 5001, Operation::SendMessages).await; + partition.consensus.set_view(1); + partition.resynchronize_consumer_offset_reservations(); + assert!(!partition.consumer_offset_capacity.is_uncertain()); + assert!( + partition + .pending_consumer_offset_commits + .contains_key(&4999) + ); + assert_eq!( + partition.occupied_consumer_offset_count(ConsumerKind::Consumer), + 1 + ); + } + + #[test] + fn given_truncated_journal_when_sequencer_is_ahead_should_not_latch_capacity() { + let (mut partition, _) = recording_partition(); + partition.consensus.sequencer().set_sequence(100); + partition.offset_reservations_need_resync.set(true); + partition.resynchronize_consumer_offset_reservations(); + assert!(!partition.consumer_offset_capacity.is_uncertain()); + assert!( + partition + .reserve_consumer_offset(ConsumerKind::Consumer, 7) + .is_ok() + ); + } + + #[test] + fn given_full_backup_phantom_map_when_new_consumer_polls_should_reclaim_without_promotion() { + let (mut partition, _) = recording_partition_at(1, 3); + partition.set_consumer_offsets_max(2); + partition.offset_space.committed_seeded = true; + for id in 1..=2 { + partition + .auto_commit_ctx(PollingConsumer::Consumer(id, 0), true) + .unwrap() + .apply(0) + .unwrap(); + } + let args = PollingArgs::new(iggy_common::PollingStrategy::first(), 1, true); + let _ = partition.build_poll_plan(PollingConsumer::Consumer(3, 0), &args, false); + assert_eq!(partition.consumer_offsets.len(), 1); + let retained_id = *partition.consumer_offsets.pin().keys().next().unwrap(); + assert!(retained_id == 1 || retained_id == 2); + assert!( + partition + .auto_commit_ctx(PollingConsumer::Consumer(3, 0), true) + .unwrap() + .apply(0) + .is_ok() + ); + assert!(partition.consumer_offsets.pin().contains_key(&retained_id)); + assert_eq!(partition.consumer_offsets.len(), 2); + assert!(!partition.consensus.is_primary()); + } + + #[compio::test] + async fn given_full_offset_table_when_storing_new_id_should_reject_before_replication() { + let (mut partition, sent_to_clients) = recording_partition(); + partition.set_consumer_offsets_max(1); + partition.stats.increment_messages_count(1); + + partition + .on_request( + store_offset_request(42, 1, ConsumerKind::Consumer, 7, 0, AckLevel::NoAck), + None, + ) + .await; + assert_eq!( + partition.durable_consumer_offset_count(ConsumerKind::Consumer), + 1 + ); + + partition + .on_request( + store_offset_request(42, 2, ConsumerKind::Consumer, 8, 0, AckLevel::NoAck), + None, + ) + .await; + let sent = sent_to_clients.borrow(); + let denied = sent + .iter() + .find_map(|(_, frame)| { + let header = bytemuck::checked::try_from_bytes::( + &frame.as_slice()[..std::mem::size_of::()], + ) + .ok()?; + (header.request == 2).then_some(*header) + }) + .expect("capacity denial reply"); + assert_eq!(denied.status, IggyError::TooManyConsumerOffsets.as_code()); + assert_eq!(denied.op, 0); + assert!(partition.consumer_offsets.pin().get(&8).is_none()); + assert_eq!(partition.consensus().pipeline_len(), 0); + } + + #[compio::test] + async fn given_full_offset_table_when_updating_existing_id_should_succeed() { + let (mut partition, _) = recording_partition(); + partition.set_consumer_offsets_max(1); + partition.stats.increment_messages_count(1); + for request_id in 1..=2 { + partition + .on_request( + store_offset_request( + 42, + request_id, + ConsumerKind::Consumer, + 7, + 0, + AckLevel::NoAck, + ), + None, + ) + .await; + } + assert_eq!( + partition.durable_consumer_offset_count(ConsumerKind::Consumer), + 1 + ); + assert_eq!(partition.consumer_offsets.pin().len(), 1); + } + + #[compio::test] + async fn given_replicated_partition_when_no_ack_offset_is_stored_should_enter_vsr() { + let (mut partition, sent_to_clients) = recording_partition_at(0, 3); + partition.stats.increment_messages_count(1); + partition + .on_request( + store_offset_request(42, 1, ConsumerKind::Consumer, 7, 0, AckLevel::NoAck), + None, + ) + .await; + + assert_eq!(partition.consensus().pipeline_len(), 1); + assert_eq!(partition.pending_consumer_offset_commits.len(), 1); + assert_eq!( + partition.durable_consumer_offset_count(ConsumerKind::Consumer), + 0, + "durable membership begins at commit" + ); + assert!(sent_to_clients.borrow().is_empty()); + } + + #[compio::test] + async fn given_pending_offset_prepare_when_view_changes_should_rebuild_capacity_from_journal() { + let (mut partition, _) = recording_partition_at(0, 3); + partition.set_consumer_offsets_max(1); + partition.stats.increment_messages_count(1); + partition + .on_request( + store_offset_request(42, 1, ConsumerKind::Consumer, 7, 0, AckLevel::Quorum), + None, + ) + .await; + assert_eq!(partition.pending_consumer_offset_commits.len(), 1); + + partition.pending_consumer_offset_commits.clear(); + partition.consumer_offset_capacity.release_reservation(7); + partition.consensus.set_view(1); + partition.resynchronize_consumer_offset_reservations(); + + assert_eq!(partition.pending_consumer_offset_commits.len(), 1); + assert!( + partition + .reserve_consumer_offset(ConsumerKind::Consumer, 8) + .is_err(), + "the retained prepare must keep the only slot reserved" + ); + } + + #[test] + fn given_phantom_and_eager_offsets_when_snapshotting_should_export_committed_durable_value_only() + { + let (partition, _) = recording_partition(); + partition.consumer_offsets.pin().insert( + 7, + ConsumerOffset::new(ConsumerKind::Consumer, 7, 9, String::new()), + ); + partition.consumer_offsets.pin().insert( + 8, + ConsumerOffset::new(ConsumerKind::Consumer, 8, 11, String::new()), + ); + partition.seed_recovered_consumer_offset(ConsumerKind::Consumer, 7, 5, 5); + + assert_eq!( + partition + .offsets_wire_snapshot_for_test() + .expect("durable key exists in live map"), + vec![(7, 5)], + "the eager value and the follower-local phantom must not enter the artifact" + ); + } + + #[test] + fn given_durable_offset_missing_from_live_map_when_snapshotting_should_refuse() { + let (partition, _) = recording_partition(); + partition.seed_recovered_consumer_offset(ConsumerKind::Consumer, 7, 5, 5); + assert!(matches!( + partition.offsets_wire_snapshot_for_test(), + Err( + crate::state_transfer::PartitionTransferUnavailable::ConsumerOffsetStateInconsistent { + kind: ConsumerKind::Consumer, + consumer_id: 7, + } + ) + )); + } + + /// Deleting a consumer offset that was never stored must answer with a + /// typed deny reply (empty body, `status` = `ConsumerOffsetNotFound`, + /// `op` 0) before consensus: nothing may enter the pipeline, and an + /// awaited client write must fail fast instead of waiting out its reply + /// timeout. Once the offset exists, the same request must pass the gate + /// into the pipeline without a deny. + #[compio::test] + async fn on_request_delete_of_missing_offset_replies_typed_deny() { + let (mut partition, sent_to_clients) = recording_partition(); + let client_id: u128 = 42; + let consumer_id: u32 = 5; + + partition + .on_request(delete_offset_request(client_id, 7, consumer_id), None) + .await; + + { + let sent = sent_to_clients.borrow(); + assert_eq!(sent.len(), 1, "exactly one deny reply"); + let (reply_client, frame) = &sent[0]; + assert_eq!(*reply_client, client_id); + let header = bytemuck::checked::try_from_bytes::( + &frame.as_slice()[..std::mem::size_of::()], + ) + .expect("deny frame starts with a valid reply header"); + assert_eq!(header.command, Command::Reply); + assert_eq!( + header.status, + IggyError::ConsumerOffsetNotFound(0).as_code() + ); + assert_eq!(header.op, 0, "a deny commits nothing"); + assert_eq!(header.request, 7); + assert_eq!( + header.size as usize, + std::mem::size_of::(), + "deny reply body must be empty" + ); + } + assert_eq!( + partition.consensus().pipeline_len(), + 0, + "denied delete must not replicate" + ); + assert!(partition.pending_consumer_offset_commits.is_empty()); + + // Existing offset: the gate passes and the delete enters the pipeline. + partition.consumer_offsets.pin().insert( + consumer_id as usize, + ConsumerOffset::new(ConsumerKind::Consumer, consumer_id, 3, String::new()), + ); + partition + .on_request(delete_offset_request(client_id, 8, consumer_id), None) + .await; + assert_eq!( + partition.consensus().pipeline_len(), + 1, + "existing offset delete must replicate" + ); + } + + #[compio::test] + async fn on_request_delete_on_stale_backup_replies_transient_before_not_found() { + let (mut partition, sent_to_clients) = recording_partition_at(1, 3); let client_id = 42; partition @@ -7891,11 +9655,15 @@ mod tests { ); assert!( - partition.is_auto_commit_offset_covered(ConsumerKind::Consumer, consumer_id, 114), + partition + .durable_consumer_offsets + .covers(ConsumerKind::Consumer, consumer_id, 114), "committed high-water covers the persisted offset" ); assert!( - !partition.is_auto_commit_offset_covered(ConsumerKind::Consumer, consumer_id, 115), + !partition + .durable_consumer_offsets + .covers(ConsumerKind::Consumer, consumer_id, 115), "an advancing offset is not covered and must submit" ); @@ -7910,7 +9678,9 @@ mod tests { .expect("explicit store persist 109"); assert_eq!(read_disk(&path), 109, "explicit store may rewind the file"); assert!( - !partition.is_auto_commit_offset_covered(ConsumerKind::Consumer, consumer_id, 114), + !partition + .durable_consumer_offsets + .covers(ConsumerKind::Consumer, consumer_id, 114), "explicit rewind lowers the high-water so a later auto-commit may re-advance" ); @@ -7958,7 +9728,9 @@ mod tests { .await .expect("seed offset file"); assert!( - !partition.is_auto_commit_offset_covered(ConsumerKind::Consumer, consumer_id, 1), + !partition + .durable_consumer_offsets + .covers(ConsumerKind::Consumer, consumer_id, 1), "a cold key is never covered; the first submit must go through" ); @@ -7976,7 +9748,9 @@ mod tests { "cold-key fold must not rewind the pre-existing on-disk value" ); assert!( - partition.is_auto_commit_offset_covered(ConsumerKind::Consumer, consumer_id, 114), + partition + .durable_consumer_offsets + .covers(ConsumerKind::Consumer, consumer_id, 114), "tracker warms with the on-disk value, not the trailing op offset" ); @@ -7989,7 +9763,9 @@ mod tests { .expect("delete persisted offset"); assert!(!std::path::Path::new(&path).exists(), "file unlinked"); assert!( - !partition.is_auto_commit_offset_covered(ConsumerKind::Consumer, consumer_id, 1), + !partition + .durable_consumer_offsets + .covers(ConsumerKind::Consumer, consumer_id, 1), "delete drops the tracker entry with the file" ); @@ -8006,49 +9782,369 @@ mod tests { let _ = std::fs::remove_dir_all(&dir); } - /// `reclaim_dead_group_offsets` must drop exactly the not-`is_live` groups - /// from the in-memory map and hand back their owned persisted-file paths, - /// leaving live groups untouched. The returned `Vec` is what the - /// reconciler unlinks off-borrow, so it carries no partition reference. - /// - /// Scope: the synchronous removal contract the off-borrow split relies on. - /// The cross-task interleave it enables -- a pump mutating the partitions vec - /// while a sibling task is parked mid-await -- is covered on the simulator's - /// deterministic executor, against the debug borrow tripwire, by - /// `simulator::tests::shell_detects_partition_borrow_held_across_await` - /// (`swap_remove`) and - /// `shell_detects_partition_borrow_held_across_a_pump_realloc` (a growing - /// `push`, which relocates every element). #[compio::test] - async fn reclaim_dead_group_offsets_drops_dead_keeps_live() { - let mut partition = test_partition(); - let group_offsets_path = "/iggy-test-cg-offsets".to_owned(); - partition.consumer_group_offsets_path = Some(group_offsets_path.clone()); + async fn given_dead_group_when_reclaimed_should_keep_capacity_until_replicated_delete() { + let (mut partition, _) = recording_partition(); + partition.consumer_group_offsets.pin().insert( + ConsumerGroupId(7), + ConsumerOffset::new(ConsumerKind::ConsumerGroup, 7, 11, String::new()), + ); + partition.seed_recovered_consumer_offset(ConsumerKind::ConsumerGroup, 7, 11, 11); + assert_eq!(partition.dead_consumer_group_offset_ids(|_| false), vec![7]); + assert_eq!(partition.consumer_group_offset_ids(), vec![7]); + assert_eq!( + partition.occupied_consumer_offset_count(ConsumerKind::ConsumerGroup), + 1 + ); + partition.stage_consumer_offset_delete(1, ConsumerKind::ConsumerGroup, 7); + partition + .apply_staged_consumer_offset_commit(1) + .await + .expect("delete commits"); + assert_eq!( + partition.occupied_consumer_offset_count(ConsumerKind::ConsumerGroup), + 0 + ); + assert!(partition.consumer_group_offset_ids().is_empty()); + } - let dead: u32 = 1; - let live: u32 = 2; + #[compio::test] + async fn given_known_stranded_group_file_when_reconciling_should_not_submit_delete_loop() { + let dir = tempfile::tempdir().unwrap(); + let group_dir = dir.path().join("groups"); + std::fs::create_dir_all(group_dir.join("7")).unwrap(); + let (mut partition, _) = recording_partition(); + partition.consumer_group_offsets_path = Some(group_dir.to_string_lossy().into_owned()); + // In the live map, so the reclaim walk actually meets the key and the + // stranded filter is what keeps it out of the delete log. + partition.seed_recovered_consumer_offset(ConsumerKind::ConsumerGroup, 7, 11, 11); partition.consumer_group_offsets.pin().insert( - ConsumerGroupId(dead as usize), - ConsumerOffset::new(ConsumerKind::ConsumerGroup, dead, 7, String::new()), + ConsumerGroupId(7), + ConsumerOffset::new(ConsumerKind::ConsumerGroup, 7, 11, String::new()), + ); + partition.consumer_group_offset_capacity.record_stranded(7); + assert!(partition.consumer_group_offset_capacity.is_stranded(7)); + assert_eq!( + partition.dead_consumer_group_offset_ids(|_| true), + Vec::::new() + ); + assert!( + partition + .dead_consumer_group_offset_ids(|_| false) + .is_empty(), + "automatic cleanup must not resubmit a known failed unlink" + ); + partition.consumer_group_offset_capacity.clear_stranded(7); + assert_eq!( + partition.dead_consumer_group_offset_ids(|_| false), + vec![7], + "the same dead key is reclaimed once it is no longer stranded" + ); + assert!( + partition + .offsets_wire_snapshot_for_test() + .unwrap() + .is_empty() + ); + } + + #[test] + fn given_local_only_key_when_replicated_should_require_durable_membership_to_delete() { + let (partition, _) = recording_partition_at(0, 3); + partition.seed_stranded_consumer_offset(ConsumerKind::ConsumerGroup, 7); + assert!( + partition + .ensure_consumer_offset_exists(ConsumerKind::ConsumerGroup, 7) + .is_err(), + "a stranded file is not a replicated key" ); partition.consumer_group_offsets.pin().insert( - ConsumerGroupId(live as usize), - ConsumerOffset::new(ConsumerKind::ConsumerGroup, live, 9, String::new()), + ConsumerGroupId(8), + ConsumerOffset::new(ConsumerKind::ConsumerGroup, 8, 0, String::new()), + ); + assert!( + partition + .ensure_consumer_offset_exists(ConsumerKind::ConsumerGroup, 8) + .is_err(), + "a local poll cursor must not produce a missing-key delete on an older peer" + ); + partition + .durable_consumer_offsets + .record_explicit(ConsumerKind::ConsumerGroup, 9, 0, 0); + assert!( + partition + .ensure_consumer_offset_exists(ConsumerKind::ConsumerGroup, 9) + .is_ok() ); + assert!( + partition + .ensure_consumer_offset_exists(ConsumerKind::ConsumerGroup, 10) + .is_err(), + "an unknown key still answers not found" + ); + + let (single, _) = recording_partition(); + single.seed_stranded_consumer_offset(ConsumerKind::ConsumerGroup, 7); + assert!( + single + .ensure_consumer_offset_exists(ConsumerKind::ConsumerGroup, 7) + .is_ok() + ); + } - let paths = partition.reclaim_dead_group_offsets(|group_id| group_id == u64::from(live)); + #[compio::test] + async fn given_committed_delete_when_unlink_fails_should_preserve_state_and_report_failure() { + let dir = tempfile::tempdir().unwrap(); + let group_dir = dir.path().join("groups"); + std::fs::create_dir_all(group_dir.join("7")).unwrap(); + let (mut partition, _) = recording_partition(); + partition.consumer_group_offsets_path = Some(group_dir.to_string_lossy().into_owned()); + partition.consumer_group_offsets.pin().insert( + ConsumerGroupId(7), + ConsumerOffset::new(ConsumerKind::ConsumerGroup, 7, 11, String::new()), + ); + partition.seed_recovered_consumer_offset(ConsumerKind::ConsumerGroup, 7, 11, 11); + partition.stage_consumer_offset_delete(1, ConsumerKind::ConsumerGroup, 7); + // A failed file mutation cannot become a successful logical delete. + assert!( + partition + .apply_staged_consumer_offset_commit(1) + .await + .is_err() + ); + assert_eq!(partition.consumer_group_offset_ids(), vec![7]); assert_eq!( - paths, - vec![format!("{group_offsets_path}/{dead}")], - "only the dead group's persisted path is returned for unlink" + partition.durable_consumer_offset_count(ConsumerKind::ConsumerGroup), + 1 + ); + assert!(partition.pending_consumer_offset_commits.contains_key(&1)); + assert!(partition.consumer_group_offset_capacity.is_stranded(7)); + assert!(partition.fatal.is_none()); + } + + #[compio::test] + async fn given_no_ack_store_when_unrelated_dirty_directory_is_gone_should_succeed() { + let dir = tempfile::tempdir().unwrap(); + let (mut partition, sent) = recording_partition(); + partition.consumer_offsets_path = + Some(dir.path().join("consumers").to_string_lossy().into_owned()); + partition.consumer_group_offsets_path = Some( + dir.path() + .join("missing-groups") + .to_string_lossy() + .into_owned(), + ); + partition.consumer_offset_dirs_dirty[1].set(true); + partition.stats.increment_messages_count(1); + + partition + .on_request( + store_offset_request(42, 1, ConsumerKind::Consumer, 7, 0, AckLevel::NoAck), + None, + ) + .await; + + assert!(partition.consumer_offsets.pin().contains_key(&7)); + assert!( + partition + .durable_consumer_offsets + .covers(ConsumerKind::Consumer, 7, 0) + ); + let reply = sent.borrow(); + let header = reply + .first() + .and_then(|(_, message)| { + bytemuck::checked::try_from_bytes::( + &message.as_slice()[..size_of::()], + ) + .ok() + }) + .expect("success reply"); + assert_eq!(header.status, 0); + } + + #[compio::test] + async fn given_no_ack_store_when_its_directory_sync_fails_should_report_failure_and_stay_dirty() + { + let dir = tempfile::tempdir().unwrap(); + let (mut partition, sent) = recording_partition(); + partition.consumer_offsets_path = + Some(dir.path().join("consumers").to_string_lossy().into_owned()); + partition.consumer_group_offsets_path = + Some(dir.path().join("groups").to_string_lossy().into_owned()); + partition.consumer_offset_enforce_fsync = true; + // Visibility alone cannot satisfy the explicitly requested barrier. + partition.consumer_offset_dir_sync_fault.set(Some(0)); + partition.stats.increment_messages_count(1); + + partition + .on_request( + store_offset_request(42, 1, ConsumerKind::Consumer, 7, 0, AckLevel::NoAck), + None, + ) + .await; + + assert!(partition.consumer_offsets.pin().contains_key(&7)); + assert!( + partition + .durable_consumer_offsets + .covers(ConsumerKind::Consumer, 7, 0) + ); + let reply = sent.borrow(); + let header = reply + .first() + .and_then(|(_, message)| { + bytemuck::checked::try_from_bytes::( + &message.as_slice()[..size_of::()], + ) + .ok() + }) + .expect("failure reply"); + assert_eq!(header.status, IggyError::CannotSyncFile.as_code()); + assert!( + partition.consumer_offset_dirs_dirty[0].get(), + "the failed directory keeps its dirt for the next walk" + ); + } + + #[compio::test] + async fn given_stale_dirt_on_the_other_kind_when_its_sync_fails_should_not_fence_the_walk() { + let dir = tempfile::tempdir().unwrap(); + let (mut partition, sent) = recording_partition_at(0, 3); + partition.consumer_offsets_path = + Some(dir.path().join("consumers").to_string_lossy().into_owned()); + partition.consumer_group_offsets_path = + Some(dir.path().join("groups").to_string_lossy().into_owned()); + partition.consumer_offset_enforce_fsync = true; + // Dirt a NoAck request left on the groups directory, whose sync now + // fails. This walk writes consumer offsets only. + partition.consumer_offset_dirs_dirty[1].set(true); + partition.consumer_offset_dir_sync_fault.set(Some(1)); + partition + .persist_consumer_offset_commit(PendingConsumerOffsetCommit::upsert( + ConsumerKind::Consumer, + 1, + 0, + )) + .await + .unwrap(); + partition.stage_consumer_offset_delete(1, ConsumerKind::Consumer, 1); + let header = PrepareHeader { + op: 1, + operation: Operation::DeleteConsumerOffset, + client: 42, + request: 1, + ..Default::default() + }; + partition.consensus.restore_commit_state(0, 1); + partition + .handle_committed_entries(vec![PipelineEntry::new(header)], &repair_config(), true) + .await; + assert!( + partition.fatal.is_none(), + "a failure on a kind this walk did not write must not fence it" + ); + assert_eq!(partition.consensus.commit_min(), 1); + assert_eq!(sent.borrow().len(), 1); + assert!(!partition.consumer_offset_dirs_dirty[0].get()); + assert!(partition.consumer_offset_dirs_dirty[1].get()); + + // The same failure on the kind the walk wrote fences it. + partition.consumer_offset_dir_sync_fault.set(Some(0)); + partition + .persist_consumer_offset_commit(PendingConsumerOffsetCommit::upsert( + ConsumerKind::Consumer, + 2, + 0, + )) + .await + .unwrap(); + partition.stage_consumer_offset_delete(2, ConsumerKind::Consumer, 2); + let header = PrepareHeader { + op: 2, + operation: Operation::DeleteConsumerOffset, + client: 42, + request: 2, + ..Default::default() + }; + partition.consensus.advance_commit_max(2); + partition + .handle_committed_entries(vec![PipelineEntry::new(header)], &repair_config(), true) + .await; + assert!( + partition.fatal.is_some(), + "a sync failure on a written kind fences the walk" + ); + assert_eq!(partition.consensus.commit_min(), 1); + } + + #[compio::test] + async fn given_no_ack_delete_sync_failure_when_retried_should_retry_barrier_and_succeed() { + let dir = tempfile::tempdir().unwrap(); + let (mut partition, sent) = recording_partition(); + partition.consumer_offsets_path = Some(dir.path().to_string_lossy().into_owned()); + partition.consumer_offset_enforce_fsync = true; + let stored = PendingConsumerOffsetCommit::upsert(ConsumerKind::Consumer, 7, 0); + partition + .persist_consumer_offset_commit(stored) + .await + .unwrap(); + partition.apply_consumer_offset_commit(stored); + partition.consumer_offset_dir_sync_fault.set(Some(0)); + partition + .apply_consumer_offset_no_ack( + Box::new(*delete_offset_request(42, 1, 7).header()), + ConsumerKind::Consumer, + 7, + None, + None, + ) + .await; + let status = |index: usize| { + let frames = sent.borrow(); + let bytes = frames[index].1.as_slice(); + let start = std::mem::offset_of!(ReplyHeader, status); + u32::from_le_bytes(bytes[start..start + 4].try_into().unwrap()) + }; + assert_eq!(status(0), IggyError::CannotSyncFile.as_code()); + assert!(!dir.path().join("7").exists()); + assert!( + partition + .ensure_consumer_offset_exists(ConsumerKind::Consumer, 7) + .is_ok() + ); + partition.consumer_offset_dir_sync_fault.set(None); + partition + .apply_consumer_offset_no_ack( + Box::new(*delete_offset_request(42, 2, 7).header()), + ConsumerKind::Consumer, + 7, + None, + None, + ) + .await; + assert_eq!(status(1), 0); + assert!(!partition.consumer_offset_dirs_dirty[0].get()); + assert!(!partition.consumer_offset_capacity.is_stranded(7)); + } + + #[test] + fn given_follower_when_group_is_deleted_should_wait_for_ordered_reclamation() { + let (partition, _) = recording_partition_at(1, 3); + partition.consumer_group_offsets.pin().insert( + ConsumerGroupId(7), + ConsumerOffset::new(ConsumerKind::ConsumerGroup, 7, 11, String::new()), + ); + partition.seed_recovered_consumer_offset(ConsumerKind::ConsumerGroup, 7, 11, 11); + assert!( + partition + .dead_consumer_group_offset_ids(|_| false) + .is_empty() ); - let mut remaining = partition.consumer_group_offset_ids(); - remaining.sort_unstable(); assert_eq!( - remaining, - vec![u64::from(live)], - "dead group removed in-memory; live group retained" + partition.durable_consumer_offset_count(ConsumerKind::ConsumerGroup), + 1 ); } @@ -8853,6 +10949,7 @@ mod tests { messages_required_to_save: 1, size_of_messages_required_to_save: IggyByteSize::from(1024 * 1024), enforce_fsync: false, + consumer_offset_enforce_fsync: false, validate_checksum: true, segment_size: IggyByteSize::from(1024 * 1024), preallocate_segments: false, @@ -9820,6 +11917,72 @@ mod tests { ); } + #[compio::test] + async fn given_flushed_repair_ahead_of_commit_min_when_replayed_should_persist_only_new_batches() + { + let dir = tempfile::tempdir().expect("temp dir"); + let log_path = dir.path().join("segment.log"); + let index_path = dir.path().join("segment.index"); + let mut fixture = PersistFixture::new( + log_path.to_str().expect("utf-8 path"), + index_path.to_str().expect("utf-8 path"), + ) + .await; + let partition = &mut fixture.partition; + partition.log.journal().inner.set_repair_retention(true); + partition.repair = Some(armed_session(4, 0, None)); + let prepares: Vec<_> = (1..=4) + .map(|op| repaired_send_prepare(op, 0, u128::from(op)).into_frozen()) + .collect(); + let replay = |op: usize| { + let bytes = prepares[op - 1].as_slice(); + let mut message = Message::::new(bytes.len()); + message.as_mut_slice().copy_from_slice(bytes); + message + }; + for op in 1..=3 { + partition.apply_repaired_prepare(replay(op)).await; + } + partition.consensus().advance_commit_max(3); + partition + .flush_committed_messages(&repair_config()) + .await + .expect("flush before the commit walk catches up"); + assert_eq!(partition.consensus().commit_min(), 0); + assert_eq!(partition.recovered_durable_offset, None); + let original = std::fs::read(&log_path).expect("read initial segment"); + let original_index = std::fs::read(&index_path).expect("read initial index"); + + for op in 2..=3 { + partition.apply_repaired_prepare(replay(op)).await; + } + partition + .flush_committed_messages(&repair_config()) + .await + .expect("flush replayed batches"); + assert_eq!(std::fs::read(&log_path).unwrap(), original); + assert_eq!(std::fs::read(&index_path).unwrap(), original_index); + + let next = replay(4); + let mut expected = original; + expected.extend_from_slice(&next.as_slice()[size_of::()..]); + partition.apply_repaired_prepare(next).await; + partition.consensus().advance_commit_max(4); + partition + .flush_committed_messages(&repair_config()) + .await + .expect("flush the new batch"); + assert_eq!(std::fs::read(&log_path).unwrap(), expected); + // The commit walk needs resident headers after the earlier flush. + for op in 1..=4 { + partition.apply_repaired_prepare(replay(op)).await; + } + partition.commit_journal(&repair_config()).await; + assert_eq!(partition.consensus().commit_min(), 4); + assert!(partition.fatal.is_none()); + assert_eq!(std::fs::read(&log_path).unwrap(), expected); + } + #[cfg(target_os = "linux")] #[compio::test] async fn given_a_persist_failure_on_a_committed_op_should_fence_the_partition_not_panic() { diff --git a/core/partitions/src/iggy_partitions.rs b/core/partitions/src/iggy_partitions.rs index 2a2ee6e1e5..d71819efd4 100644 --- a/core/partitions/src/iggy_partitions.rs +++ b/core/partitions/src/iggy_partitions.rs @@ -34,10 +34,10 @@ use message_bus::MessageBus; use server_common::Message; use server_common::send_messages::{ChecksumMode, convert_request_message, encrypt_batch_request}; use server_common::sharding::{IggyNamespace, LocalIdx, ShardId}; -#[cfg(debug_assertions)] use std::cell::Cell; use std::cell::{RefCell, UnsafeCell}; use std::collections::BTreeMap; +use std::rc::Rc; use tracing::warn; /// RAII counter for live [`IggyPartitions::with_partition`] borrows. The @@ -108,6 +108,7 @@ where /// per-shard runtime is single-threaded, so runtime borrow checks /// suffice; callers must not hold a borrow across `.await`. tombstoned: RefCell>, + consumer_group_offsets_reconcile_epoch: Rc>, /// Debug-only tripwire: counts live [`Self::with_partition`] borrows so /// `insert` / `remove` can assert the partitions vec is never mutated /// while a sanctioned non-pump read borrow is outstanding. Cannot fire for @@ -131,6 +132,7 @@ where partitions: UnsafeCell::new(Vec::new()), namespace_to_local: UnsafeCell::new(BTreeMap::new()), tombstoned: RefCell::new(AHashSet::new()), + consumer_group_offsets_reconcile_epoch: Rc::new(Cell::new(0)), #[cfg(debug_assertions)] borrow_active: Cell::new(0), } @@ -145,6 +147,7 @@ where // BTreeMap has no capacity hint; the Vec above absorbs the sizing. namespace_to_local: UnsafeCell::new(BTreeMap::new()), tombstoned: RefCell::new(AHashSet::new()), + consumer_group_offsets_reconcile_epoch: Rc::new(Cell::new(0)), #[cfg(debug_assertions)] borrow_active: Cell::new(0), } @@ -224,7 +227,11 @@ where /// [`Self::with_partition`]; the `&mut` path above is uncounted (it is /// pump-only, so it cannot alias this same-task mutation). #[doc(hidden)] - pub fn insert(&self, namespace: IggyNamespace, partition: IggyPartition) -> LocalIdx { + pub fn insert( + &self, + namespace: IggyNamespace, + mut partition: IggyPartition, + ) -> LocalIdx { #[cfg(debug_assertions)] debug_assert_eq!( self.borrow_active.get(), @@ -232,6 +239,9 @@ where "IggyPartitions::insert while a with_partition borrow is live" ); partition.publish_current_offset(); + partition.set_consumer_group_offsets_reconcile_epoch(Rc::clone( + &self.consumer_group_offsets_reconcile_epoch, + )); // Safety: pump-only invariant, caller responsibility. let partitions = unsafe { &mut *self.partitions.get() }; let local_idx = LocalIdx::new(partitions.len()); @@ -240,6 +250,18 @@ where local_idx } + pub fn consumer_group_offsets_reconcile_epoch(&self) -> u64 { + self.consumer_group_offsets_reconcile_epoch.get() + } + + pub fn note_consumer_group_offsets_reconcile_needed(&self) { + self.consumer_group_offsets_reconcile_epoch.set( + self.consumer_group_offsets_reconcile_epoch + .get() + .wrapping_add(1), + ); + } + /// Check if a namespace exists. pub fn contains(&self, namespace: &IggyNamespace) -> bool { self.namespace_map().contains_key(namespace) @@ -628,6 +650,31 @@ where let _ = waiter.send(build_deny_reply_from_request_header(header, status)); } } + + pub async fn on_auto_commit_request( + &self, + request: Message, + reservation: crate::AutoCommitReservation, + ) { + let namespace = IggyNamespace::from_raw(request.header().group); + if self.is_tombstoned(&namespace) { + tracing::debug!( + namespace_raw = namespace.inner(), + "dropping auto-commit for tombstoned partition" + ); + return; + } + if let Some(partition) = self.get_mut_by_ns(&namespace) { + partition + .on_request_with_reservation(request, None, Some(reservation)) + .await; + } else { + tracing::debug!( + namespace_raw = namespace.inner(), + "dropping auto-commit for missing partition" + ); + } + } } impl Plane> for IggyPartitions diff --git a/core/partitions/src/lib.rs b/core/partitions/src/lib.rs index 059c795a0f..36be8081b2 100644 --- a/core/partitions/src/lib.rs +++ b/core/partitions/src/lib.rs @@ -17,6 +17,7 @@ #![allow(clippy::future_not_send)] +mod consumer_offset_capacity; mod iggy_index; mod iggy_index_reader; mod iggy_index_writer; @@ -32,6 +33,7 @@ pub mod segment_anchor; pub mod state_transfer; mod types; +pub use consumer_offset_capacity::{AutoCommitReservation, ConsumerOffsetCapacityError}; use iggy_binary_protocol::PrepareHeader; use iggy_common::IggyError; pub use iggy_index::IggyIndex; @@ -54,12 +56,16 @@ pub use journal::{EVICTED_RING_BYTES_MAX, EVICTED_RING_CAPACITY}; /// Both consumers -- the fallback in [`IggyPartition`] and the `[partition]` /// config default boot installs -- already depend on this crate. pub const DEFAULT_OFFSET_RESERVATION_LEASE: u32 = 64 * 1024; + +/// Shipped per-kind durable consumer-offset limit for one partition. +pub const DEFAULT_CONSUMER_OFFSETS_MAX: usize = 4096; pub use messages_writer::MessagesWriter; pub use offset_storage::delete_persisted_offset; pub use poll_plan::{AutoCommitApplied, PollPlan}; pub use segment::Segment; use server_common::Message; pub use server_common::send_messages::{IggyMessage, IggyMessageHeader, IggyMessages}; +pub use state_transfer::CONSUMER_OFFSETS_ENTRIES_MAX; pub use types::{ AppendResult, COMMIT_WALK_OPS_MAX, FatalCommit, Fragment, PartitionOffsets, PartitionPathLayout, PartitionsConfig, PollFragments, PollQueryResult, PollingArgs, @@ -89,6 +95,8 @@ pub struct RetainedPartitionState { /// Whether that incarnation ever stamped an offset, i.e. whether the two /// numbers above describe an offset space at all. pub offset_space_used: bool, + pub consumer_offsets: Vec<(u32, u64)>, + pub consumer_group_offsets: Vec<(u32, u64)>, } /// Partition-level data plane operations. @@ -101,17 +109,6 @@ pub trait Partition { message: Message, ) -> impl Future>; - /// # Errors - /// Returns `IggyError::FeatureUnavailable` by default. - fn store_consumer_offset( - &self, - consumer: PollingConsumer, - offset: u64, - ) -> Result<(), IggyError> { - let _ = (consumer, offset); - Err(IggyError::FeatureUnavailable) - } - fn get_consumer_offset(&self, consumer: PollingConsumer) -> Option { let _ = consumer; None diff --git a/core/partitions/src/offset_storage.rs b/core/partitions/src/offset_storage.rs index 618f987eb1..f750b13459 100644 --- a/core/partitions/src/offset_storage.rs +++ b/core/partitions/src/offset_storage.rs @@ -16,7 +16,7 @@ // under the License. use compio::{ - fs::{OpenOptions, create_dir_all, remove_file}, + fs::{OpenOptions, create_dir_all, remove_file, rename}, io::{AsyncReadAt, AsyncReadAtExt, AsyncWriteAtExt}, }; use iggy_common::{IggyError, calculate_checksum}; @@ -39,6 +39,9 @@ pub const OFFSET_RECORD_SIZE: usize = OFFSET_SIZE + CHECKSUM_SIZE; /// partition incarnation it was applied for. pub const PURGE_GENERATION_FILE: &str = "purge.gen"; +/// Sibling name an atomic offset replacement writes before its rename lands. +const OFFSET_REPLACEMENT_SUFFIX: &str = ".tmp"; + /// `[generation][created_revision]`, both LE u64. const PURGE_GENERATION_RECORD_SIZE: usize = 2 * OFFSET_SIZE; @@ -48,7 +51,9 @@ pub enum OffsetRecord { /// A usable offset. `checksummed` is false for a bare offset predating the /// checksum, read as-is and upgraded by the next write. Value { offset: u64, checksummed: bool }, - /// Shorter than the value: a crash between `persist_offset`'s truncate and write. + /// Shorter than the value: a crash between the truncate and the write of an + /// in-place update, the default path while `consumer_offset_enforce_fsync` + /// is off. Torn, /// The checksum does not describe the value stored beside it. Corrupt { @@ -106,9 +111,41 @@ pub fn decode_offset_record(bytes: &[u8]) -> OffsetRecord { /// Overwrite a consumer-offset file with `offset` and a checksum over it. /// +/// Without `enforce_fsync` the file is rewritten in place and no directory is +/// synced. With it, the record goes to a sibling inode, is data-synced and +/// renamed over the prior file, so a failed write leaves the prior cursor +/// intact, and the caller marks the parent directory for a sync on the next +/// commit walk. The replacement is tied to the same knob as the sync: without +/// the sync neither the write nor the rename is ordered against a crash, so +/// the extra inode and rename buy nothing. +/// /// # Errors /// [`IggyError`] when the directory, file, or write cannot be created or completed. pub async fn persist_offset(path: &str, offset: u64, enforce_fsync: bool) -> Result<(), IggyError> { + let record = encode_offset_record(offset); + if enforce_fsync { + replace_file(path, record, true, false).await + } else { + write_in_place(path, record).await + } +} + +async fn write_in_place(path: &str, record: [u8; N]) -> Result<(), IggyError> { + create_parent_dir(path).await?; + let mut file = OpenOptions::new() + .write(true) + .create(true) + .truncate(true) + .open(path) + .await + .map_err(|_| IggyError::CannotOpenConsumerOffsetsFile(path.to_owned()))?; + file.write_all_at(record, 0) + .await + .0 + .map_err(|_| IggyError::CannotWriteToFile) +} + +async fn create_parent_dir(path: &str) -> Result<(), IggyError> { // No `exists()` probe first: that is a BLOCKING `std::path` stat on the pump // in front of every write, which serialises a batched fan-out on stats // before it can submit any I/O. `create_dir_all` is already a no-op on an @@ -118,26 +155,96 @@ pub async fn persist_offset(path: &str, offset: u64, enforce_fsync: bool) -> Res IggyError::CannotCreateConsumerOffsetsDirectory(parent.display().to_string()) })?; } + Ok(()) +} + +pub(crate) async fn stage_offset_replacement(path: &str, offset: u64) -> Result<(), IggyError> { + // Install can remove old files before publishing replacements. Staging + // must survive a crash regardless of the normal consumer-offset fsync knob. + write_replacement(path, encode_offset_record(offset), true) + .await + .map(|_| ()) +} + +pub(crate) async fn commit_offset_replacement(path: &str) -> Result<(), IggyError> { + rename(replacement_path(path), path) + .await + .map_err(|_| IggyError::CannotWriteToFile) +} + +pub(crate) async fn discard_offset_replacement(path: &str) { + let _ = remove_file(replacement_path(path)).await; +} + +async fn replace_file( + path: &str, + record: [u8; N], + enforce_fsync: bool, + sync_parent: bool, +) -> Result<(), IggyError> { + let temporary = write_replacement(path, record, enforce_fsync).await?; + if rename(&temporary, path).await.is_err() { + let _ = remove_file(&temporary).await; + return Err(IggyError::CannotWriteToFile); + } + if sync_parent && let Some(parent) = Path::new(path).parent() { + let parent = compio::fs::File::open(parent) + .await + .map_err(|_| IggyError::CannotSyncFile)?; + parent + .sync_all() + .await + .map_err(|_| IggyError::CannotSyncFile)?; + } + Ok(()) +} + +async fn write_replacement( + path: &str, + record: [u8; N], + enforce_fsync: bool, +) -> Result { + create_parent_dir(path).await?; + + // Keep the previous cursor intact until the complete replacement exists. + // A failed truncate-and-write otherwise turns a valid cursor into a torn + // file that boot discards. The fixed sibling is safe because writes to one + // consumer key are serialized by the partition pump. + let temporary = replacement_path(path); let mut file = OpenOptions::new() .write(true) .create(true) .truncate(true) - .open(path) + .open(&temporary) .await .map_err(|_| IggyError::CannotOpenConsumerOffsetsFile(path.to_owned()))?; - file.write_all_at(encode_offset_record(offset), 0) - .await - .0 - .map_err(|_| IggyError::CannotWriteToFile)?; + if file.write_all_at(record, 0).await.0.is_err() { + let _ = remove_file(&temporary).await; + return Err(IggyError::CannotWriteToFile); + } - if enforce_fsync { - file.sync_data() - .await - .map_err(|_| IggyError::CannotWriteToFile)?; + if enforce_fsync && file.sync_data().await.is_err() { + let _ = remove_file(&temporary).await; + return Err(IggyError::CannotWriteToFile); } + drop(file); + Ok(temporary) +} - Ok(()) +fn replacement_path(path: &str) -> String { + format!("{path}{OFFSET_REPLACEMENT_SUFFIX}") +} + +#[must_use] +pub fn offset_replacement_id(name: &str) -> Option { + name.strip_suffix(OFFSET_REPLACEMENT_SUFFIX)?.parse().ok() +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct PersistedOffset { + pub offset: u64, + pub written: bool, } /// Monotone counterpart of [`persist_offset`] for a server auto-commit op. @@ -165,7 +272,7 @@ pub async fn persist_offset_max( path: &str, offset: u64, enforce_fsync: bool, -) -> Result { +) -> Result { let on_disk = match read_offset_record(path).await? { Some(OffsetRecord::Value { offset, .. }) => Some(offset), Some(OffsetRecord::Corrupt { @@ -187,16 +294,20 @@ pub async fn persist_offset_max( Some(OffsetRecord::Torn) | None => None, }; let effective = on_disk.map_or(offset, |current| current.max(offset)); - if on_disk != Some(effective) { + let written = on_disk != Some(effective); + if written { persist_offset(path, effective, enforce_fsync).await?; } - Ok(effective) + Ok(PersistedOffset { + offset: effective, + written, + }) } /// Durably record the purge generation a partition has locally applied, keyed /// to the incarnation (`created_revision`) it was applied for. /// -/// Truncate+write like [`persist_offset`] but ALWAYS data-synced, regardless of +/// Atomic replacement like [`persist_offset`] but ALWAYS data-synced, regardless of /// the consumer-offset fsync knob: purges are rare, the record is 16 bytes, and /// a generation lost from the page cache in a crash makes the reconciler /// re-purge on restart, wiping messages appended after the purge. A failure @@ -210,30 +321,10 @@ pub async fn persist_purge_generation( generation: u64, created_revision: u64, ) -> Result<(), IggyError> { - if let Some(parent) = Path::new(path).parent() { - create_dir_all(parent).await.map_err(|_| { - IggyError::CannotCreateConsumerOffsetsDirectory(parent.display().to_string()) - })?; - } - - let mut file = OpenOptions::new() - .write(true) - .create(true) - .truncate(true) - .open(path) - .await - .map_err(|_| IggyError::CannotOpenConsumerOffsetsFile(path.to_owned()))?; let mut record = [0u8; PURGE_GENERATION_RECORD_SIZE]; record[..OFFSET_SIZE].copy_from_slice(&generation.to_le_bytes()); record[OFFSET_SIZE..].copy_from_slice(&created_revision.to_le_bytes()); - file.write_all_at(record, 0) - .await - .0 - .map_err(|_| IggyError::CannotWriteToFile)?; - file.sync_data() - .await - .map_err(|_| IggyError::CannotWriteToFile)?; - Ok(()) + replace_file(path, record, true, true).await } /// Read the purge generation this replica applied for the `created_revision` @@ -323,16 +414,17 @@ async fn read_offset_record(path: &str) -> Result, IggyErro } /// Unlink a persisted consumer-offset file. A no-op if the file is absent. +/// Returns whether a file was removed. An absent file is `Ok(false)`. /// /// # Errors /// Returns [`IggyError::CannotDeleteConsumerOffsetFile`] if the unlink fails. -pub async fn delete_persisted_offset(path: &str) -> Result<(), IggyError> { +pub async fn delete_persisted_offset(path: &str) -> Result { // NotFound is tolerated on the result instead of probed for: the probe was // a blocking stat on the pump before every unlink, and "already gone" is // exactly the outcome this wants anyway. match remove_file(path).await { - Ok(()) => Ok(()), - Err(error) if error.kind() == std::io::ErrorKind::NotFound => Ok(()), + Ok(()) => Ok(true), + Err(error) if error.kind() == std::io::ErrorKind::NotFound => Ok(false), Err(_) => Err(IggyError::CannotDeleteConsumerOffsetFile(path.to_owned())), } } @@ -479,6 +571,29 @@ mod tests { let _ = std::fs::remove_dir_all(&dir); } + #[compio::test] + async fn failed_replacement_keeps_the_previous_offset_file_intact() { + let dir = unique_temp_dir(); + let path = dir.join("42").to_string_lossy().into_owned(); + persist_offset(&path, 114, true) + .await + .expect("initial persist"); + std::fs::create_dir(format!("{path}{OFFSET_REPLACEMENT_SUFFIX}")) + .expect("block temporary file creation"); + + assert!(persist_offset(&path, 115, true).await.is_err()); + let bytes = std::fs::read(&path).expect("previous offset survives"); + assert_eq!( + decode_offset_record(&bytes), + OffsetRecord::Value { + offset: 114, + checksummed: true, + } + ); + + let _ = std::fs::remove_dir_all(&dir); + } + #[compio::test] async fn read_offset_record_torn_file_is_torn_not_error() { let dir = unique_temp_dir(); @@ -635,7 +750,7 @@ mod tests { .await .expect("a corrupt file must not fail the commit"); assert_eq!( - folded, 7, + folded.offset, 7, "the untrusted stored value must not win the fold" ); diff --git a/core/partitions/src/poll_plan.rs b/core/partitions/src/poll_plan.rs index 7d3fae869d..6b8a01e009 100644 --- a/core/partitions/src/poll_plan.rs +++ b/core/partitions/src/poll_plan.rs @@ -31,6 +31,10 @@ //! sound on a detached task concurrently with the pump's own writes. use crate::PollFragments; +use crate::consumer_offset_capacity::{ + AutoCommitReservation, ConsumerOffsetCapacity, ConsumerOffsetCapacityError, + DurableConsumerOffsets, +}; use crate::iggy_index::{IGGY_INDEX_SIZE, IggyIndexCache}; use crate::iggy_index_reader::IggyIndexReader; use crate::journal::{ @@ -174,6 +178,8 @@ pub struct DiskSegment { /// node-local only and diverge on failover. pub struct AutoCommitCtx { pub(crate) target: AutoCommitTarget, + pub(crate) capacity: Rc, + pub(crate) durable: Rc, } /// The offset an `auto_commit` poll applied in memory, surfaced for replication. @@ -185,6 +191,11 @@ pub struct AutoCommitApplied { pub kind: ConsumerKind, pub consumer_id: u32, pub offset: u64, + previous_offset: Option, + target: AutoCommitTarget, + capacity: Rc, + durable: Rc, + last_polled: Option, } /// The lock-free offset map this auto-commit updates, captured as an owned @@ -296,7 +307,17 @@ impl PollPlan { /// docs), so it is safe on a detached task. Returns the served fragments, /// the poll's high-water offset, and the auto-committed offset (if any) for /// the serving shard to replicate through consensus. - pub async fn execute(self) -> (PollFragments<4096>, u64, Option) { + /// + /// # Errors + /// Returns a capacity error when auto-commit would create a new live-map + /// entry after the configured per-kind bound has been reached. + /// The serving shard also rejects completion with `TransientNotAccepted` + /// if the partition disappeared or changed incarnation during disk I/O. + /// No fragments are returned for that rejected completion. + pub async fn execute( + self, + ) -> Result<(PollFragments<4096>, u64, Option), ConsumerOffsetCapacityError> + { let commit_offset = self.commit_offset; let (fragments, last_matching_offset) = match self.tier { PollTier::Empty => (PollFragments::new(), None), @@ -362,7 +383,7 @@ impl PollPlan { }; finish( - self.last_polled.as_ref(), + self.last_polled, self.auto_commit, commit_offset, fragments, @@ -374,8 +395,14 @@ impl PollPlan { /// ([`Self::needs_off_pump_io`] is `false`): no disk read, so the pump /// applies the auto-commit in memory and replies inline without spawning. /// The auto-committed offset is returned for the serving shard to replicate. - #[must_use] - pub fn execute_resident(self) -> (PollFragments<4096>, u64, Option) { + /// + /// # Errors + /// Returns a capacity error when auto-commit would create a new live-map + /// entry after the configured per-kind bound has been reached. + pub fn execute_resident( + self, + ) -> Result<(PollFragments<4096>, u64, Option), ConsumerOffsetCapacityError> + { let commit_offset = self.commit_offset; let (fragments, last_matching_offset) = match self.tier { PollTier::Empty => (PollFragments::new(), None), @@ -390,7 +417,7 @@ impl PollPlan { } }; finish( - self.last_polled.as_ref(), + self.last_polled, self.auto_commit, commit_offset, fragments, @@ -403,17 +430,19 @@ impl PollPlan { /// factored out so the high-water record, auto-commit, and returned triple /// stay identical across both. fn finish( - last_polled: Option<&LastPolledCtx>, + last_polled: Option, auto_commit: Option, commit_offset: u64, fragments: PollFragments<4096>, last_matching_offset: Option, -) -> (PollFragments<4096>, u64, Option) { - if let (Some(last_polled), Some(last_offset)) = (last_polled, last_matching_offset) { +) -> Result<(PollFragments<4096>, u64, Option), ConsumerOffsetCapacityError> { + let mut auto_commit_applied = apply_auto_commit(auto_commit, &fragments, last_matching_offset)?; + if let Some(applied) = &mut auto_commit_applied { + applied.last_polled = last_polled; + } else if let (Some(last_polled), Some(last_offset)) = (last_polled, last_matching_offset) { last_polled.record(last_offset); } - let auto_commit_applied = apply_auto_commit(auto_commit, &fragments, last_matching_offset); - (fragments, commit_offset, auto_commit_applied) + Ok((fragments, commit_offset, auto_commit_applied)) } /// Apply an `auto_commit` to the in-memory offset map (monotone) and surface @@ -429,19 +458,17 @@ fn apply_auto_commit( auto_commit: Option, fragments: &PollFragments<4096>, last_matching_offset: Option, -) -> Option { - let auto_commit = auto_commit?; +) -> Result, ConsumerOffsetCapacityError> { + let Some(auto_commit) = auto_commit else { + return Ok(None); + }; if fragments.is_empty() { - return None; + return Ok(None); } - let last_offset = last_matching_offset?; - auto_commit.apply(last_offset); - let (kind, consumer_id) = auto_commit.kind_and_id(); - Some(AutoCommitApplied { - kind, - consumer_id, - offset: last_offset, - }) + let Some(last_offset) = last_matching_offset else { + return Ok(None); + }; + auto_commit.apply(last_offset).map(Some) } pub enum PollTier { @@ -861,53 +888,168 @@ impl AutoCommitCtx { /// newer explicit store; the maps are lock-free (`papaya`), so this is /// sound off the pump task. #[allow(clippy::cast_possible_truncation)] - pub(crate) fn apply(&self, offset: u64) { - match &self.target { + pub(crate) fn apply( + self, + offset: u64, + ) -> Result { + let (kind, consumer_id) = self.kind_and_id(); + let create = |path: Option<&str>| { + ConsumerOffset::new( + kind, + consumer_id, + offset, + path.map_or_else(String::new, |path| format!("{path}/{consumer_id}")), + ) + }; + let previous_offset = match &self.target { AutoCommitTarget::Consumer { offsets, consumer_id, create_path, - } => { - let consumer_id = *consumer_id; - let map: &ConsumerOffsets = offsets; - upsert_offset_max(map, consumer_id as usize, offset, || { - create_path.as_deref().map_or_else( - || { - ConsumerOffset::new( - ConsumerKind::Consumer, - consumer_id, - 0, - String::new(), - ) - }, - |path| ConsumerOffset::default_for_consumer(consumer_id, path), - ) - }); - } + } => apply_local_offset( + offsets, + *consumer_id as usize, + offset, + &self.capacity, + self.durable.count(kind) >= self.capacity.limit(), + || create(create_path.as_deref()), + )?, AutoCommitTarget::ConsumerGroup { offsets, group_id, create_path, - } => { - let group_id = *group_id; - let key = ConsumerGroupId(group_id as usize); - let map: &ConsumerGroupOffsets = offsets; - upsert_offset_max(map, key, offset, || { - create_path.as_deref().map_or_else( - || { - ConsumerOffset::new( - ConsumerKind::ConsumerGroup, - group_id, - 0, - String::new(), - ) - }, - |path| ConsumerOffset::default_for_consumer_group(key, path), - ) - }); + } => apply_local_offset( + offsets, + ConsumerGroupId(*group_id as usize), + offset, + &self.capacity, + self.durable.count(kind) >= self.capacity.limit(), + || create(create_path.as_deref()), + )?, + }; + Ok(AutoCommitApplied { + kind, + consumer_id, + offset, + previous_offset, + target: self.target, + capacity: self.capacity, + durable: self.durable, + last_polled: None, + }) + } +} + +impl AutoCommitApplied { + /// Record the group handoff frontier only after poll admission succeeds. + fn mark_served(&self) { + if let Some(last_polled) = &self.last_polled { + last_polled.record(self.offset); + } + } + /// Reserve a durable key before the synthetic store is submitted. + /// Returns `None` when committed state already covers this offset. + /// + /// # Errors + /// Returns a capacity error when this is a new durable key and the + /// partition's per-kind limit has been reached. + pub fn reserve_durable( + &self, + ) -> Result, ConsumerOffsetCapacityError> { + if self + .durable + .covers(self.kind, self.consumer_id, self.offset) + { + return Ok(None); + } + self.capacity + .reserve_provisional(self.consumer_id, &self.durable) + .map(Some) + } + + pub(crate) fn belongs_to(&self, durable: &Rc) -> bool { + Rc::ptr_eq(&self.durable, durable) + } + + /// Run the serving shard's synchronous admission and settle this apply in + /// the same call: `Ok` marks the cursor served, `Err` rolls the eager + /// local update back and returns the error. The rollback is a bare store of + /// the previous offset, so nothing may yield between the decision and it. + /// Keeping both inside one synchronous method is what makes that hold for + /// every caller. + /// + /// # Errors + /// Whatever `decide` returned, after the rollback. + pub fn admit(self, decide: impl FnOnce(&Self) -> Result<(), E>) -> Result<(), E> { + match decide(&self) { + Ok(()) => { + self.mark_served(); + Ok(()) } + Err(error) => { + self.rollback_created(); + Err(error) + } + } + } + + /// Undo this poll's eager update after synchronous admission fails. + fn rollback_created(&self) { + match &self.target { + AutoCommitTarget::Consumer { + offsets, + consumer_id, + .. + } => rollback_local_offset(offsets, *consumer_id as usize, self.previous_offset), + AutoCommitTarget::ConsumerGroup { + offsets, group_id, .. + } => rollback_local_offset( + offsets, + ConsumerGroupId(*group_id as usize), + self.previous_offset, + ), + } + if self.previous_offset.is_none() { + self.capacity.note_local_key_change(); + self.capacity.forget_inactive_provisional(self.consumer_id); + } + } +} + +fn rollback_local_offset( + map: &papaya::HashMap, + key: K, + previous: Option, +) { + let guard = map.pin(); + if let Some(previous) = previous { + if let Some(entry) = guard.get(&key) { + entry.offset.store(previous, Ordering::Relaxed); } + } else { + guard.remove(&key); + } +} + +fn apply_local_offset( + map: &papaya::HashMap, + key: K, + offset: u64, + capacity: &ConsumerOffsetCapacity, + durable_full: bool, + create: impl FnOnce() -> ConsumerOffset, +) -> Result, ConsumerOffsetCapacityError> { + let guard = map.pin(); + if let Some(existing) = guard.get(&key) { + return Ok(Some(existing.offset.fetch_max(offset, Ordering::Relaxed))); } + // The `len()` read and the insert are not atomic on this lock-free map. + // The bound holds because every poll of one partition runs on that + // partition's own shard thread, so no second inserter exists. + capacity.admit_local_map_key(guard.len(), durable_full)?; + guard.insert(key, create()); + capacity.note_local_key_change(); + Ok(None) } /// Upsert a committed offset into a lock-free `papaya` offset map: bump an @@ -1141,15 +1283,95 @@ mod tests { } fn consumer_auto_commit(offsets: Arc, consumer_id: u32) -> AutoCommitCtx { + consumer_auto_commit_with_limit(offsets, consumer_id, crate::DEFAULT_CONSUMER_OFFSETS_MAX) + } + + fn consumer_auto_commit_with_limit( + offsets: Arc, + consumer_id: u32, + limit: usize, + ) -> AutoCommitCtx { AutoCommitCtx { target: AutoCommitTarget::Consumer { offsets, consumer_id, create_path: None, }, + capacity: Rc::new(ConsumerOffsetCapacity::new(ConsumerKind::Consumer, limit)), + durable: Rc::new(DurableConsumerOffsets::default()), } } + #[test] + fn given_existing_phantom_when_primary_reservation_is_denied_should_restore_previous_offset() { + let offsets = Arc::new(ConsumerOffsets::with_capacity(2)); + offsets.pin().insert( + 7, + ConsumerOffset::new(ConsumerKind::Consumer, 7, 4, String::new()), + ); + let context = consumer_auto_commit_with_limit(Arc::clone(&offsets), 7, 1); + context + .durable + .record_explicit(ConsumerKind::Consumer, 8, 0, 0); + let applied = context + .apply(9) + .expect("existing phantom is locally writable"); + assert!(applied.reserve_durable().is_err()); + applied.rollback_created(); + assert_eq!( + offsets + .pin() + .get(&7) + .expect("phantom retained") + .offset + .load(Ordering::Relaxed), + 4 + ); + } + + #[test] + fn given_new_auto_commit_when_provisional_guard_is_dropped_should_reopen_capacity() { + let offsets = Arc::new(ConsumerOffsets::with_capacity(1)); + let context = consumer_auto_commit_with_limit(Arc::clone(&offsets), 7, 1); + let capacity = Rc::clone(&context.capacity); + let durable = Rc::clone(&context.durable); + let applied = context.apply(9).expect("new poll fits"); + let reservation = applied + .reserve_durable() + .expect("primary admits") + .expect("new slot"); + assert!(capacity.check(8, &durable).is_err()); + drop(reservation); + applied.rollback_created(); + assert!(capacity.check(8, &durable).is_ok()); + assert!(offsets.pin().is_empty()); + } + + #[test] + fn given_full_auto_commit_map_when_creating_key_should_reject_without_insertion() { + let offsets = Arc::new(ConsumerOffsets::with_capacity(1)); + offsets.pin().insert( + 1, + ConsumerOffset::new(ConsumerKind::Consumer, 1, 0, String::new()), + ); + let plan = PollPlan { + commit_offset: 42, + auto_commit: Some(consumer_auto_commit_with_limit(offsets.clone(), 2, 1)), + last_polled: None, + tier: PollTier::Resident { + fragments: non_empty_fragments(), + last_matching_offset: Some(5), + }, + }; + + let Err(error) = plan.execute_resident() else { + panic!("a missing key at the map limit must be rejected"); + }; + assert_eq!(error.kind, ConsumerKind::Consumer); + assert_eq!(error.occupied, 1); + assert!(offsets.pin().get(&2).is_none()); + } + #[test] fn resident_auto_commit_applies_in_memory_and_surfaces_offset() { // A resident auto_commit poll stays on the inline fast path (no detached @@ -1172,7 +1394,8 @@ mod tests { "a resident auto_commit no longer persists on the poll path; the pump must not spawn", ); - let (fragments, commit_offset, applied) = plan.execute_resident(); + let (fragments, commit_offset, applied) = + plan.execute_resident().expect("auto-commit is admitted"); assert!(!fragments.is_empty(), "resident fragments must be returned"); assert_eq!(commit_offset, 42, "commit offset is forwarded verbatim"); @@ -1203,7 +1426,8 @@ mod tests { last_polled: None, tier: PollTier::Empty, }; - let (fragments, _commit_offset, applied) = plan.execute_resident(); + let (fragments, _commit_offset, applied) = + plan.execute_resident().expect("empty poll needs no slot"); assert!(fragments.is_empty()); assert!( applied.is_none(), @@ -1220,9 +1444,9 @@ mod tests { // Auto-commit must never rewind a newer offset (anti-rewind via // fetch_max); an explicit StoreConsumerOffset may legitimately rewind. let offsets = Arc::new(ConsumerOffsets::with_capacity(1)); - let auto_commit = consumer_auto_commit(offsets.clone(), 7); - - auto_commit.apply(10); + consumer_auto_commit(offsets.clone(), 7) + .apply(10) + .expect("first auto-commit is admitted"); let after_high = offsets .pin() .get(&7usize) @@ -1230,7 +1454,9 @@ mod tests { assert_eq!(after_high, Some(10)); // A stale auto-commit with a smaller offset must not rewind. - auto_commit.apply(4); + consumer_auto_commit(offsets.clone(), 7) + .apply(4) + .expect("existing key remains admitted"); let after_stale = offsets .pin() .get(&7usize) diff --git a/core/partitions/src/state_transfer.rs b/core/partitions/src/state_transfer.rs index 1872d52165..94a62b60e4 100644 --- a/core/partitions/src/state_transfer.rs +++ b/core/partitions/src/state_transfer.rs @@ -29,7 +29,9 @@ use crate::messages_writer::MessagesWriter; use crate::offset_storage::{ - PURGE_GENERATION_FILE, delete_persisted_offset, persist_offset, persist_purge_generation, + PURGE_GENERATION_FILE, commit_offset_replacement, delete_persisted_offset, + discard_offset_replacement, offset_replacement_id, persist_purge_generation, + stage_offset_replacement, }; use crate::segment::Segment; use crate::segment_anchor::ANCHOR_SUFFIX; @@ -46,8 +48,9 @@ use journal::superblock::SuperblockStore; use message_bus::MessageBus; use server_common::send_messages::decode_batch_slice; use server_common::{SegmentStorage, yield_to_reactor}; -use std::collections::HashSet; +use std::collections::{HashMap, HashSet}; use std::fmt; +use std::future::Future; use std::mem::size_of; use std::path::{Path, PathBuf}; use std::rc::Rc; @@ -79,7 +82,7 @@ const CONSUMER_OFFSETS_VERSION_V1: u8 = 1; /// A corruption guard, not a target: it bounds the allocation `decode` /// makes from a length field a peer sent, exactly like the manifest's own /// entry ceiling. -pub(crate) const CONSUMER_OFFSETS_ENTRIES_MAX: u32 = 1 << 20; +pub const CONSUMER_OFFSETS_ENTRIES_MAX: u32 = 1 << 20; /// Wire stride of one dedup entry: client u128 + watermark u64 + commit u64 + /// user u32 + committed window u128. @@ -612,10 +615,46 @@ impl fmt::Display for ConsumerOffsetsWireError { impl std::error::Error for ConsumerOffsetsWireError {} +const fn validate_consumer_offset_transfer_count( + kind: ConsumerKind, + count: usize, + max: usize, +) -> Result<(), PartitionTransferUnavailable> { + if count <= max { + return Ok(()); + } + Err(PartitionTransferUnavailable::ConsumerOffsetsTooLarge { kind, count, max }) +} + #[cfg(test)] mod tests { use super::*; + #[compio::test] + async fn given_transient_offset_io_failure_when_retried_should_succeed_without_exhausting_budget() + { + let attempts = std::cell::Cell::new(0); + let result = retry_offset_mutation(|| { + attempts.set(attempts.get() + 1); + std::future::ready(if attempts.get() == 1 { Err(()) } else { Ok(7) }) + }) + .await; + assert_eq!(result, Ok(7)); + assert_eq!(attempts.get(), 2); + } + + #[compio::test] + async fn given_persistent_offset_io_failure_when_retried_should_stop_at_attempt_limit() { + let attempts = std::cell::Cell::new(0); + let result = retry_offset_mutation(|| { + attempts.set(attempts.get() + 1); + std::future::ready(Err::<(), _>(7)) + }) + .await; + assert_eq!(result, Err(7)); + assert_eq!(attempts.get(), OFFSET_IO_ATTEMPTS); + } + fn table() -> ConsumerOffsetsWire { ConsumerOffsetsWire { purge_generation: 3, @@ -903,6 +942,25 @@ mod tests { "bytes past the last section must fail closed" ); } + + #[test] + fn given_offset_count_above_transfer_ceiling_when_validated_should_reject() { + assert!(validate_consumer_offset_transfer_count(ConsumerKind::Consumer, 4, 4).is_ok()); + assert!(matches!( + validate_consumer_offset_transfer_count(ConsumerKind::ConsumerGroup, 5, 4), + Err(PartitionTransferUnavailable::ConsumerOffsetsTooLarge { + kind: ConsumerKind::ConsumerGroup, + count: 5, + max: 4, + }) + )); + let error = validate_consumer_offset_transfer_count(ConsumerKind::Consumer, 5, 4) + .expect_err("count above ceiling"); + assert!( + !error.transient(), + "an artifact that cannot fit the decoder will not heal by retrying" + ); + } } /// What a full validation walk over one segment payload derived. @@ -1171,6 +1229,15 @@ pub enum PartitionTransferUnavailable { /// In-memory / simulated partition: nothing on disk to serve. NoPartitionDir, RepairInProgress, + ConsumerOffsetsTooLarge { + kind: ConsumerKind, + count: usize, + max: usize, + }, + ConsumerOffsetStateInconsistent { + kind: ConsumerKind, + consumer_id: u32, + }, /// Primary-by-index of a group that has committed nothing: an empty group /// is trivially "caught up", so this is the only thing separating a real /// primary from a view-0 phantom whose directory vanished. @@ -1220,6 +1287,8 @@ impl PartitionTransferUnavailable { | Self::SegmentSetChanged | Self::OfferBuildInProgress { .. } => true, Self::NoPartitionDir + | Self::ConsumerOffsetsTooLarge { .. } + | Self::ConsumerOffsetStateInconsistent { .. } | Self::ManifestTooLarge { .. } | Self::FlushFailed(_) | Self::SegmentUnreadable { .. } => false, @@ -1233,6 +1302,14 @@ impl fmt::Display for PartitionTransferUnavailable { Self::NotCaughtUpPrimary => write!(f, "not the caught-up primary of this group"), Self::NoPartitionDir => write!(f, "partition has no on-disk directory"), Self::RepairInProgress => write!(f, "partition is itself mid-repair"), + Self::ConsumerOffsetsTooLarge { kind, count, max } => write!( + f, + "partition has {count} {kind:?} offset entries, past the {max} transfer ceiling" + ), + Self::ConsumerOffsetStateInconsistent { kind, consumer_id } => write!( + f, + "durable {kind:?} offset {consumer_id} is missing from the live map" + ), Self::NothingCommitted => write!( f, "primary by index at view 0 with nothing committed; refusing to serve an empty offer" @@ -1262,9 +1339,8 @@ impl fmt::Display for PartitionTransferUnavailable { impl std::error::Error for PartitionTransferUnavailable {} -/// Outcome of a completed install. A degraded install is a SUCCESS: the -/// segments and floor landed; only some consumer-offset file writes failed, -/// and the next offset commit blind-writes those files. +/// Outcome of a completed install. Offset files must land before the installed +/// commit floor advances. A failed purge-generation record remains retryable. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub struct PartitionInstallOutcome { /// The consensus op the install applied. Named for what it holds: every @@ -1273,19 +1349,22 @@ pub struct PartitionInstallOutcome { /// `installed_frontier`), and op-vs-offset confusion is what produced this /// PR's durability defects. pub applied_commit_op: u64, - /// Every transferred offset file was WRITTEN (and the offset - /// directories fsynced, so the old files' unlinks stick). Not a - /// durability claim for the file contents: `persist_offset` fsyncs only - /// under `consumer_offset_enforce_fsync`, matching the normal - /// offset-commit path -- shipped default off. - pub offsets_written: bool, + /// The offered purge generation was already recorded or persisted during + /// this install. False means a restart may repeat the purge and transfer. + pub purge_generation_recorded: bool, } -/// Failure installing a transferred partition state. Every `check`-phase -/// variant means NOTHING was mutated. +/// Failure installing a transferred partition state. +/// +/// Validation failures mutate nothing. Pre-swap offset failures can leave +/// ignored replacement siblings when best-effort cleanup also fails, but never +/// alter live files. #[derive(Debug)] pub enum PartitionInstallError { NoPartitionDir, + NoOffsetDir { + kind: ConsumerKind, + }, /// `commit_op` fell below this replica's commit frontier; installing /// would rewind `commit_min` (the anti-rewind assert, as a refusal). StaleTransfer { @@ -1307,6 +1386,12 @@ pub enum PartitionInstallError { offer_next_offset: u64, local_next_offset: u64, }, + /// Consumer-offset staging failed before the segment swap, or finalizing a + /// staged offset failed during it. + OffsetPersistence { + path: String, + source: iggy_common::IggyError, + }, Offsets(ConsumerOffsetsWireError), /// Duplicate base offset in the staged set. DuplicateSegment { @@ -1353,6 +1438,9 @@ impl fmt::Display for PartitionInstallError { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { match self { Self::NoPartitionDir => write!(f, "partition has no on-disk directory"), + Self::NoOffsetDir { kind } => { + write!(f, "partition has no {kind:?} offset directory configured") + } Self::StaleTransfer { commit_op, commit_min, @@ -1372,6 +1460,9 @@ impl fmt::Display for PartitionInstallError { "offer frontier {offer_next_offset} is below this replica's own next offset \ {local_next_offset}; installing it would rewind the offset space" ), + Self::OffsetPersistence { path, source } => { + write!(f, "consumer offset persistence failed at {path}: {source}") + } Self::Offsets(source) => write!(f, "consumer-offsets artifact rejected: {source}"), Self::DuplicateSegment { start_offset } => { write!(f, "duplicate staged segment at base offset {start_offset}") @@ -1551,6 +1642,34 @@ fn final_paths(partition_dir: &str, start_offset: u64) -> (String, String) { /// depth against how long one partition monopolises it; matches the tick's /// superblock pre-pass. const OFFSET_PERSIST_CONCURRENCY: usize = 16; +const OFFSET_IO_ATTEMPTS: usize = 3; +/// First retry delay of [`retry_offset_mutation`]; each further retry doubles it. +const OFFSET_IO_BACKOFF_BASE: std::time::Duration = std::time::Duration::from_millis(10); + +async fn retry_offset_mutation>>( + mut operation: impl FnMut() -> F, +) -> Result { + for attempt in 1..OFFSET_IO_ATTEMPTS { + match operation().await { + Ok(value) => return Ok(value), + Err(error) => { + tracing::debug!( + attempt, + ?error, + "offset mutation failed, retrying after backoff" + ); + compio::time::sleep(OFFSET_IO_BACKOFF_BASE * (1 << (attempt - 1))).await; + } + } + } + operation().await.inspect_err(|error| { + tracing::debug!( + attempt = OFFSET_IO_ATTEMPTS, + ?error, + "offset mutation failed on the last attempt" + ); + }) +} /// One consumer-offset file the install is about to write. Collected before any /// write is issued so the offset maps and the persisted-offset tracker are never @@ -1562,6 +1681,40 @@ struct PlannedOffsetWrite { value: u64, } +pub(crate) const fn consumer_kind_index(kind: ConsumerKind) -> usize { + match kind { + ConsumerKind::Consumer => 0, + ConsumerKind::ConsumerGroup => 1, + } +} + +async fn stage_offset_writes(planned: &[PlannedOffsetWrite]) -> Result<(), PartitionInstallError> { + for batch in planned.chunks(OFFSET_PERSIST_CONCURRENCY) { + let writes = batch.iter().map(|write| async move { + ( + &write.path, + retry_offset_mutation(|| stage_offset_replacement(&write.path, write.value)).await, + ) + }); + for (path, result) in futures::future::join_all(writes).await { + if let Err(source) = result { + discard_offset_writes(planned).await; + return Err(PartitionInstallError::OffsetPersistence { + path: path.clone(), + source, + }); + } + } + } + Ok(()) +} + +async fn discard_offset_writes(planned: &[PlannedOffsetWrite]) { + for write in planned { + discard_offset_replacement(&write.path).await; + } +} + /// fsync the partition directory so a rename made durable stays durable. /// Async so the wait parks the task instead of the whole shard reactor; /// every other future on the pump keeps running through it. @@ -1577,6 +1730,45 @@ where B: MessageBus, SB: SuperblockStore, { + fn plan_transfer_offset_writes( + &self, + offsets_wire: &ConsumerOffsetsWire, + next_offset: u64, + ) -> Result, PartitionInstallError> { + let consumer_dir = + self.consumer_offsets_path + .as_deref() + .ok_or(PartitionInstallError::NoOffsetDir { + kind: ConsumerKind::Consumer, + })?; + let group_dir = self.consumer_group_offsets_path.as_deref().ok_or( + PartitionInstallError::NoOffsetDir { + kind: ConsumerKind::ConsumerGroup, + }, + )?; + let clamp = |offset: u64| next_offset.checked_sub(1).map(|last| offset.min(last)); + let mut planned = + Vec::with_capacity(offsets_wire.consumers.len() + offsets_wire.groups.len()); + for (kind, dir, offsets) in [ + ( + ConsumerKind::Consumer, + consumer_dir, + &offsets_wire.consumers, + ), + (ConsumerKind::ConsumerGroup, group_dir, &offsets_wire.groups), + ] { + planned.extend(offsets.iter().filter_map(|(id, offset)| { + clamp(*offset).map(|value| PlannedOffsetWrite { + kind, + id: *id, + path: format!("{dir}/{id}"), + value, + }) + })); + } + Ok(planned) + } + /// Build (or serve from cache) this group's state-transfer offer. /// /// Force-flushes the committed prefix first so the segments cover every @@ -1770,7 +1962,7 @@ where // serve it only when a recorded purge says the emptiness is the truth. // `install_state_transfer`'s `purge_advances` check re-decides that // against the metadata plane and refuses the rest. - let offsets_wire = self.offsets_wire_snapshot(); + let offsets_wire = self.offsets_wire_snapshot()?; if segments.is_empty() && offsets_wire.next_offset == 0 && offsets_wire.purge_generation == 0 @@ -1919,53 +2111,84 @@ where self.transfer_offer_cache.borrow_mut().take(); } - /// Snapshot the live offset maps + purge generation into the wire shape. - /// Eagerly auto-committed offsets can run slightly ahead of committed - /// state; that is safe because their covering ops sit in - /// `(commit_op, commit_max]`, which the receiver's tail repair replays, - /// and offset applies converge (monotone auto-commit, verbatim stores). - fn offsets_wire_snapshot(&self) -> ConsumerOffsetsWire { - // Every key is minted from a u32 wire id, so the narrowing filter is - // an invariant, not a policy: say so out loud instead of silently - // shrinking the snapshot when it ever breaks. - let mut consumers: Vec<(u32, u64)> = self - .consumer_offsets - .pin() - .iter() - .filter_map(|(id, offset)| { - let narrowed = u32::try_from(*id).ok(); - debug_assert!(narrowed.is_some(), "consumer offset key {id} exceeds u32"); - narrowed.map(|id| (id, offset.offset.load(Ordering::Acquire))) - }) - .collect(); - consumers.sort_unstable_by_key(|(id, _)| *id); - let mut groups: Vec<(u32, u64)> = self - .consumer_group_offsets - .pin() - .iter() - .filter_map(|(id, offset)| { - let narrowed = u32::try_from(id.0).ok(); - debug_assert!( - narrowed.is_some(), - "consumer group offset key {} exceeds u32", - id.0 + fn validate_consumer_offset_transfer_counts(&self) -> Result<(), PartitionTransferUnavailable> { + for kind in [ConsumerKind::Consumer, ConsumerKind::ConsumerGroup] { + let count = self.durable_consumer_offsets.count(kind); + if let Err(error) = validate_consumer_offset_transfer_count( + kind, + count, + CONSUMER_OFFSETS_ENTRIES_MAX as usize, + ) { + tracing::error!( + target: "iggy.partitions.diag", + plane = "partitions", + namespace_raw = self.consensus().group(), + ?kind, + count, + max = CONSUMER_OFFSETS_ENTRIES_MAX, + "consumer offset state exceeds the transfer ceiling" ); - narrowed.map(|id| (id, offset.offset.load(Ordering::Acquire))) - }) - .collect(); - groups.sort_unstable_by_key(|(id, _)| *id); + return Err(error); + } + } + Ok(()) + } + + /// Snapshot committed durable offsets only. Eager auto-commit progress and + /// follower-local cursor entries stay in the live maps until a replicated + /// store commits them, so neither can be promoted by state transfer. + fn offsets_wire_snapshot(&self) -> Result { + self.validate_consumer_offset_transfer_counts()?; + let consumer_map = self.consumer_offsets.pin(); + let consumers = self.snapshot_offset_kind(ConsumerKind::Consumer, |id| { + consumer_map.contains_key(&(id as usize)) + })?; + let group_map = self.consumer_group_offsets.pin(); + let groups = self.snapshot_offset_kind(ConsumerKind::ConsumerGroup, |id| { + group_map.contains_key(&ConsumerGroupId(id as usize)) + })?; // The append counter, not the segment end: retention can GC every // sealed segment while the counter stands at N, and the receiver // must resume minting at N either way. let next_offset = self.offset_frontier(); let dedup = self.dedup().watermarks_sorted(); - ConsumerOffsetsWire { + Ok(ConsumerOffsetsWire { purge_generation: self.applied_purge_generation, next_offset, consumers, groups, dedup, - } + }) + } + + fn snapshot_offset_kind( + &self, + kind: ConsumerKind, + map_contains: impl Fn(u32) -> bool, + ) -> Result, PartitionTransferUnavailable> { + self.durable_consumer_offsets.with_entries(kind, |entries| { + let mut snapshot = Vec::with_capacity(entries.len()); + for (&consumer_id, state) in entries { + if !map_contains(consumer_id) { + return Err( + PartitionTransferUnavailable::ConsumerOffsetStateInconsistent { + kind, + consumer_id, + }, + ); + } + snapshot.push((consumer_id, state.committed_offset)); + } + snapshot.sort_unstable_by_key(|(id, _)| *id); + Ok(snapshot) + }) + } + + #[cfg(test)] + pub(crate) fn offsets_wire_snapshot_for_test( + &self, + ) -> Result, PartitionTransferUnavailable> { + self.offsets_wire_snapshot().map(|wire| wire.consumers) } /// Validate one completed `SEGMENT_LOG` artifact and spill it to staging @@ -2169,10 +2392,10 @@ where /// `commit_op`. The live tail `(commit_op, commit_max]` is left to /// ordinary journal repair. /// - /// Two-phase: every validation runs before any mutation. The mutate - /// phase's crash windows all recover as an honestly-shorter partition - /// (see the swap ordering comments); no durable completeness claim - /// exists anywhere, so boot re-derives from whatever files survive. + /// Validation runs before live-state mutation. Offset replacement siblings + /// are then written and synced while the old partition remains intact. The + /// segment swap's crash windows recover as an honestly-shorter partition + /// (see the swap ordering comments); boot re-derives from surviving files. /// /// # Errors /// [`PartitionInstallError`]; check-phase variants mutate nothing. @@ -2185,7 +2408,8 @@ where offsets_bytes: &[u8], committed_purge_generation: u64, ) -> Result { - // ---- check phase: nothing below may mutate ---- + // ---- check phase: nothing below may mutate live state. Staging + // writes only sibling files the install can abandon. ---- let Some(partition_dir) = self.partition_dir.clone() else { return Err(PartitionInstallError::NoPartitionDir); }; @@ -2280,6 +2504,16 @@ where } } + let installed_end = staged.last().map(|meta| meta.end_offset); + let next_offset = offsets_wire + .next_offset + .max(installed_end.map_or(0, |end| end + 1)); + let planned_offsets = self.plan_transfer_offset_writes(&offsets_wire, next_offset)?; + // Write and data-sync every small offset record before the destructive + // segment swap. Only ignored, nonnumeric replacement siblings exist at + // this point, so a write fault leaves the old partition serviceable. + stage_offset_writes(&planned_offsets).await?; + // ---- mutate phase ---- // Record the INCOMING frontier before anything destructive: the swap // below unlinks the old chain and makes that durable before the first @@ -2316,6 +2550,7 @@ where .await }; if !frontier_durable { + discard_offset_writes(&planned_offsets).await; return Err(PartitionInstallError::FrontierNotDurable { frontier: offsets_wire.next_offset, }); @@ -2331,9 +2566,18 @@ where // below it. let staged_was_empty = staged.is_empty(); let outcome = self - .apply_checked_install(config, commit_op, staged, &offsets_wire, &partition_dir) + .apply_checked_install( + config, + commit_op, + staged, + &offsets_wire, + &planned_offsets, + &partition_dir, + next_offset, + ) .await; if outcome.is_err() { + discard_offset_writes(&planned_offsets).await; // A mutate-phase failure can leave the log drained or half // rebuilt while the disk already holds any prefix of the new // chain. Converge BOTH the live state and the disk to an empty, @@ -2382,14 +2626,16 @@ where /// caller converges from. Split out so the convergence handling cannot /// be forgotten on a new error path. Runs under the caller's write-lock /// guard. - #[allow(clippy::too_many_lines)] + #[allow(clippy::too_many_lines, clippy::too_many_arguments)] async fn apply_checked_install( &mut self, config: &PartitionsConfig, commit_op: u64, staged: Vec, offsets_wire: &ConsumerOffsetsWire, + planned_offsets: &[PlannedOffsetWrite], partition_dir: &str, + next_offset: u64, ) -> Result { // Sweep staging strays a dead earlier attempt left behind, keeping // only what THIS install is about to rename. Bounded disk hygiene; @@ -2646,21 +2892,20 @@ where // must not rewind every transferred offset to 0 -- a durable, // client-visible rewind the replicas would then disagree on. let installed_end = staged.last().map(|meta| meta.end_offset); - let next_offset = offsets_wire - .next_offset - .max(installed_end.map_or(0, |end| end + 1)); - let mut offsets_written = true; // A key that fails the u32 narrowing would strand its old offset // file's delete, which boot can then resurrect: unreachable while // keys are minted from u32 wire ids, so assert it. - let old_consumer_paths: Vec = { + let old_consumer_paths: Vec<(ConsumerKind, u32, String)> = { let guard = self.consumer_offsets.pin(); - let mut paths: Vec = guard + let mut paths: Vec<(ConsumerKind, u32, String)> = guard .iter() .filter_map(|(key, _)| { let narrowed = u32::try_from(*key).ok(); debug_assert!(narrowed.is_some(), "consumer offset key {key} exceeds u32"); - narrowed.and_then(|id| self.persisted_offset_path(ConsumerKind::Consumer, id)) + narrowed.and_then(|id| { + self.persisted_offset_path(ConsumerKind::Consumer, id) + .map(|path| (ConsumerKind::Consumer, id, path)) + }) }) .collect(); guard.clear(); @@ -2669,15 +2914,18 @@ where // and a purged origin offering `next_offset = 0` drops every // incoming entry, so a map-only sweep leaves the old table for boot // to resurrect. - paths.extend(strayed_offset_files( - self.consumer_offsets_path.as_deref(), - &offsets_wire.consumers, - )); + paths.extend( + strayed_offset_files(self.consumer_offsets_path.as_deref()) + .into_iter() + .filter_map(|path| { + numeric_offset_id(&path).map(|id| (ConsumerKind::Consumer, id, path)) + }), + ); paths }; - let old_group_paths: Vec = { + let old_group_paths: Vec<(ConsumerKind, u32, String)> = { let guard = self.consumer_group_offsets.pin(); - let mut paths: Vec = guard + let mut paths: Vec<(ConsumerKind, u32, String)> = guard .iter() .filter_map(|(key, _)| { let narrowed = u32::try_from(key.0).ok(); @@ -2686,130 +2934,125 @@ where "consumer group offset key {} exceeds u32", key.0 ); - narrowed - .and_then(|id| self.persisted_offset_path(ConsumerKind::ConsumerGroup, id)) + narrowed.and_then(|id| { + self.persisted_offset_path(ConsumerKind::ConsumerGroup, id) + .map(|path| (ConsumerKind::ConsumerGroup, id, path)) + }) }) .collect(); guard.clear(); - paths.extend(strayed_offset_files( - self.consumer_group_offsets_path.as_deref(), - &offsets_wire.groups, - )); + paths.extend( + strayed_offset_files(self.consumer_group_offsets_path.as_deref()) + .into_iter() + .filter_map(|path| { + numeric_offset_id(&path).map(|id| (ConsumerKind::ConsumerGroup, id, path)) + }), + ); paths }; - for path in old_consumer_paths.into_iter().chain(old_group_paths) { - if let Err(error) = delete_persisted_offset(&path).await { - // Not fatal, but not silent either: a stranded file is an id - // absent from the NEW table (matching ids get overwritten at - // the same path), and boot resurrects it. Sharpest after a - // purged origin ships `next_offset = 0`, where the clamp drops - // every incoming entry and the whole old table survives while - // the install still reports success. - tracing::warn!( - target: "iggy.partitions.diag", - plane = "partitions", - namespace_raw = self.consensus().group(), - path = %path, - %error, - "failed to unlink a superseded consumer-offset file during install" - ); + let mut offset_dirs_changed = [false; 2]; + let planned_ids: HashSet<_> = planned_offsets + .iter() + .map(|write| (write.kind, write.id)) + .collect(); + // Replacements atomically overwrite their old files. Only keys absent + // from the clamped plan require unlinking, including every key when + // the incoming message frontier is empty. + let old_paths: HashMap<_, _> = old_consumer_paths + .into_iter() + .chain(old_group_paths) + .filter(|(kind, id, _)| !planned_ids.contains(&(*kind, *id))) + .map(|(kind, id, path)| ((kind, id), path)) + .collect(); + for ((kind, consumer_id), path) in old_paths { + // An obsolete authoritative file must not survive a successful + // install because boot would reload it outside the incoming table. + match delete_persisted_offset(&path).await { + Ok(removed) => { + if removed { + offset_dirs_changed[consumer_kind_index(kind)] = true; + } + self.consumer_offset_capacity_for(kind) + .clear_stranded(consumer_id); + } + Err(error) => { + self.consumer_offset_capacity_for(kind) + .record_stranded(consumer_id); + tracing::warn!( + target: "iggy.partitions.diag", + plane = "partitions", + namespace_raw = self.consensus().group(), + path, + consumer_id, + %error, + "install could not remove a superseded consumer offset file" + ); + return Err(PartitionInstallError::OffsetPersistence { + path, + source: error, + }); + } } } - self.persisted_offsets.borrow_mut().clear(); + self.durable_consumer_offsets.clear(); self.pending_consumer_offset_commits.clear(); + self.queued_auto_commit_reservations.borrow_mut().clear(); + self.consumer_offset_capacity + .rebuild(&self.durable_consumer_offsets, std::iter::empty()); + self.consumer_group_offset_capacity + .rebuild(&self.durable_consumer_offsets, std::iter::empty()); self.last_polled_offsets.pin().clear(); - // `None` when the group's offset space is empty (`next_offset == 0`, - // a purged origin): clamping every transferred offset to 0 would tell - // each consumer it consumed offset 0 on a partition that never minted - // one, so a `Next` poll skips the first message. Dropping the entries - // is what "no offsets yet" means. - let clamp = |offset: u64| next_offset.checked_sub(1).map(|last| offset.min(last)); - if self.consumer_offsets_path.is_none() || self.consumer_group_offsets_path.is_none() { - // Nothing to write the transferred table into: unreachable via - // the server boot paths (they always configure storage), but if - // it ever fires the table was dropped and the flag must say so. - offsets_written = false; - } - // Both maps are populated first (no await, so nothing borrows across - // one), then the files are written in capped batches. One await per - // file put a rejoin carrying thousands of consumers on the pump for - // thousands of sequential open + write + optional fsync round trips; - // the tick's superblock pre-pass sets the precedent for the width. - // The dedup slice is memory-only, so it installs here with the maps - // rather than being written anywhere. No frontier fence is needed: the - // install lifts `commit_min` to the offer's `commit_op`, so the commit - // walk that follows starts strictly above everything this artifact - // covers, and `record_commit` is idempotent besides. + // The replacement siblings were written and data-synced before any + // segment mutation. Finalize only their directory entries here, then + // publish the matching maps and durable membership. self.dedup_mut() .install_watermarks(offsets_wire.dedup.iter().copied()); - let mut planned: Vec = - Vec::with_capacity(offsets_wire.consumers.len() + offsets_wire.groups.len()); - if let Some(dir) = self.consumer_offsets_path.clone() { - for (id, offset) in &offsets_wire.consumers { - let Some(value) = clamp(*offset) else { - continue; - }; - let entry = ConsumerOffset::default_for_consumer(*id, &dir); - entry.offset.store(value, Ordering::Release); - let path = entry.path.clone(); - self.consumer_offsets.pin().insert(*id as usize, entry); - planned.push(PlannedOffsetWrite { - kind: ConsumerKind::Consumer, - id: *id, - path, - value, - }); - } - } - if let Some(dir) = self.consumer_group_offsets_path.clone() { - for (id, offset) in &offsets_wire.groups { - let Some(value) = clamp(*offset) else { - continue; - }; - let group_id = ConsumerGroupId(*id as usize); - let entry = ConsumerOffset::default_for_consumer_group(group_id, &dir); - entry.offset.store(value, Ordering::Release); - let path = entry.path.clone(); - self.consumer_group_offsets.pin().insert(group_id, entry); - planned.push(PlannedOffsetWrite { - kind: ConsumerKind::ConsumerGroup, - id: *id, - path, - value, - }); - } - } - let enforce_fsync = self.consumer_offset_enforce_fsync; - for batch in planned.chunks(OFFSET_PERSIST_CONCURRENCY) { - let writes = batch.iter().map(|write| async move { - let written = persist_offset(&write.path, write.value, enforce_fsync) - .await - .is_ok(); - (written, write.kind, write.id, write.value) - }); - for (written, kind, id, value) in futures::future::join_all(writes).await { - if written { - self.persisted_offsets - .borrow_mut() - .insert((kind, id), value); - } else { - offsets_written = false; + for write in planned_offsets { + // A rename failure after the segment swap leaves an incomplete + // install. Propagate it to convergence rather than acknowledge + // mixed state or retry a standalone directory barrier. + commit_offset_replacement(&write.path) + .await + .map_err(|source| PartitionInstallError::OffsetPersistence { + path: write.path.clone(), + source, + })?; + offset_dirs_changed[consumer_kind_index(write.kind)] = true; + let entry = ConsumerOffset::new(write.kind, write.id, write.value, write.path.clone()); + match write.kind { + ConsumerKind::Consumer => { + self.consumer_offsets.pin().insert(write.id as usize, entry); + } + ConsumerKind::ConsumerGroup => { + self.consumer_group_offsets + .pin() + .insert(ConsumerGroupId(write.id as usize), entry); } } - } - // Directory fsync so the OLD files' unlinks stick: without it a - // crash right after install resurrects the pre-transfer offset - // files at boot. The per-file content durability stays governed by - // `consumer_offset_enforce_fsync` like every other offset commit. - for dir in self - .consumer_offsets_path - .clone() - .into_iter() - .chain(self.consumer_group_offsets_path.clone()) - { - if fsync_dir(&dir).await.is_err() { - offsets_written = false; + self.durable_consumer_offsets.record_explicit( + write.kind, + write.id, + write.value, + write.value, + ); + self.consumer_offset_capacity_for(write.kind) + .clear_stranded(write.id); + } + // One observation per changed directory. Retrying with a newly opened + // handle can mask the writeback error that the first fsync consumed. + for (changed, dir) in offset_dirs_changed.into_iter().zip([ + self.consumer_offsets_path.as_deref(), + self.consumer_group_offsets_path.as_deref(), + ]) { + if changed { + let dir = dir.expect("planned offset directory was validated before mutation"); + fsync_dir(dir) + .await + .map_err(|source| PartitionInstallError::SwapIo { + path: dir.to_owned(), + source, + })?; } } @@ -2862,8 +3105,8 @@ where // would make a restart re-purge the just-installed data and pull it // all over again. A write failure only re-opens that restart window // (the wipe-then-retransfer is self-healing, peers keep the data), - // so it degrades like the offset writes above instead of failing the - // install. + // so it is reported separately from the mandatory offset writes. + let mut purge_generation_recorded = true; if offsets_wire.purge_generation > self.applied_purge_generation && let Some(dir) = self.partition_dir.clone() { @@ -2885,7 +3128,7 @@ where generation; a restart before the next purge records it will \ re-purge and re-transfer this partition" ); - offsets_written = false; + purge_generation_recorded = false; } } self.applied_purge_generation = self @@ -2926,7 +3169,7 @@ where Ok(PartitionInstallOutcome { applied_commit_op: commit_op, - offsets_written, + purge_generation_recorded, }) } @@ -2944,6 +3187,7 @@ where /// [`iggy_common::IggyError`] when the sweep or the empty plant fails; /// the partition then has no serviceable chain and the caller must /// fence it (see [`PartitionInstallError::ConvergeFailed`]). + #[allow(clippy::too_many_lines)] async fn converge_to_empty_after_failed_install( &mut self, config: &PartitionsConfig, @@ -2959,6 +3203,62 @@ where } self.log.journal().inner.clear_all(); self.log.journal_mut().info = crate::log::JournalInfo::default(); + self.consumer_offsets.pin().clear(); + self.consumer_group_offsets.pin().clear(); + self.last_polled_offsets.pin().clear(); + self.durable_consumer_offsets.clear(); + self.pending_consumer_offset_commits.clear(); + self.queued_auto_commit_reservations.borrow_mut().clear(); + for kind in [ConsumerKind::Consumer, ConsumerKind::ConsumerGroup] { + self.consumer_offset_capacity_for(kind) + .rebuild(&self.durable_consumer_offsets, std::iter::empty()); + } + for (kind, dir) in [ + (ConsumerKind::Consumer, self.consumer_offsets_path.as_ref()), + ( + ConsumerKind::ConsumerGroup, + self.consumer_group_offsets_path.as_ref(), + ), + ] { + let Some(dir) = dir else { continue }; + for entry in offset_dir_entries(dir) { + match entry { + OffsetDirEntry::Replacement(path) => { + if let Err(error) = compio::fs::remove_file(&path).await { + tracing::warn!( + path, + %error, + "could not remove an abandoned offset replacement" + ); + } + } + OffsetDirEntry::Offset { id, path } => { + // A remaining authoritative file prevents convergence. + match retry_offset_mutation(|| delete_persisted_offset(&path)).await { + Ok(_) => self.consumer_offset_capacity_for(kind).clear_stranded(id), + Err(error) => { + self.consumer_offset_capacity_for(kind).record_stranded(id); + tracing::warn!( + path, + consumer_id = id, + %error, + "converge could not remove a consumer offset file" + ); + return Err(error); + } + } + } + } + } + // No `exists()` probe: that is a blocking stat on the pump. A + // directory the unlinks emptied and removed has nothing left to + // make durable. + match fsync_dir(dir).await { + Ok(()) => {} + Err(error) if error.kind() == std::io::ErrorKind::NotFound => {} + Err(_) => return Err(iggy_common::IggyError::CannotSyncFile), + } + } // Every segment and staging file this partition had is about to be // unlinked, so neither memo can describe anything real afterwards. The // checksum map's own doc promises the clear happens here; without it the @@ -3295,35 +3595,60 @@ pub fn offered_purge_generation(offsets_bytes: &[u8]) -> u64 { .unwrap_or_default() } -/// Offset files under `dir` whose id is absent from `incoming`. -/// -/// The install's own map cannot name these: a pre-purge offset op replayed by -/// journal repair persists a file this incarnation never held, and a purged -/// origin offers `next_offset = 0`, which drops every incoming entry. Left -/// behind, boot hydrates them back. +/// One regular file of a consumer-offset directory, classified by name. +enum OffsetDirEntry { + /// A sibling an atomic replacement left behind. + Replacement(String), + /// A bare-u32 offset file and its id. + Offset { id: u32, path: String }, +} + +/// The offset files and abandoned replacements under `dir`, in one pass. /// -/// A file whose name is not a bare u32 is left alone rather than guessed at: -/// every offset file is named by its id, so anything else is not ours. -pub(crate) fn strayed_offset_files(dir: Option<&str>, incoming: &[(u32, u64)]) -> Vec { - let Some(dir) = dir else { - return Vec::new(); - }; +/// A file whose name is neither a bare u32 nor a replacement sibling is left +/// alone rather than guessed at: every offset file is named by its id, so +/// anything else is not ours. +fn offset_dir_entries(dir: &str) -> Vec { let Ok(entries) = std::fs::read_dir(dir) else { return Vec::new(); }; entries .filter_map(Result::ok) .filter_map(|entry| { + if !entry.file_type().ok()?.is_file() { + return None; + } let name = entry.file_name().into_string().ok()?; + let path = format!("{dir}/{name}"); + if offset_replacement_id(&name).is_some() { + return Some(OffsetDirEntry::Replacement(path)); + } let id: u32 = name.parse().ok()?; - incoming - .iter() - .all(|(incoming_id, _)| *incoming_id != id) - .then(|| format!("{dir}/{name}")) + Some(OffsetDirEntry::Offset { id, path }) + }) + .collect() +} + +/// Every offset file under `dir`. +/// +/// The live map cannot name these: a pre-purge offset op replayed by journal +/// repair persists a file this incarnation never held. Left behind, boot +/// hydrates them back. +pub(crate) fn strayed_offset_files(dir: Option<&str>) -> Vec { + dir.map(offset_dir_entries) + .unwrap_or_default() + .into_iter() + .filter_map(|entry| match entry { + OffsetDirEntry::Offset { path, .. } => Some(path), + OffsetDirEntry::Replacement(_) => None, }) .collect() } +pub(crate) fn numeric_offset_id(path: &str) -> Option { + Path::new(path).file_name()?.to_str()?.parse().ok() +} + /// Stamp over every `SEGMENT_LOG` entry of a manifest, keying /// [`ReuseScanMemo`]. Equal digests mean the two offers expect byte-identical /// staged files, so a scan already done for one answers the other; the offsets diff --git a/core/partitions/src/types.rs b/core/partitions/src/types.rs index f2a3bd7918..0c720dd717 100644 --- a/core/partitions/src/types.rs +++ b/core/partitions/src/types.rs @@ -366,6 +366,7 @@ impl Default for PartitionPathLayout { /// Mirrors the relevant fields from the server's `PartitionConfig` and /// `SegmentConfig` (`core/server/src/configs/system.rs`). #[derive(Debug, Clone)] +#[allow(clippy::struct_excessive_bools)] pub struct PartitionsConfig { /// Flush journal to disk when it accumulates this many messages. pub messages_required_to_save: u32, @@ -373,6 +374,10 @@ pub struct PartitionsConfig { pub size_of_messages_required_to_save: IggyByteSize, /// Whether to enforce fsync after writes. pub enforce_fsync: bool, + /// Whether consumer-offset files are written crash-safe (data-synced, + /// renamed, directory synced). Independent of `enforce_fsync`, which + /// governs message and index files. + pub consumer_offset_enforce_fsync: bool, /// Whether a disk poll verifies each batch's `batch_checksum` against the bytes /// it just read. /// diff --git a/core/sdk/Cargo.toml b/core/sdk/Cargo.toml index 09ef84d03c..93993589c7 100644 --- a/core/sdk/Cargo.toml +++ b/core/sdk/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "iggy" -version = "0.11.0-edge.6" +version = "0.11.0-edge.7" description = "Iggy is the persistent message streaming platform written in Rust, supporting QUIC, TCP and HTTP transport protocols, capable of processing millions of messages per second." edition = "2024" rust-version.workspace = true diff --git a/core/sdk/src/clients/consumer.rs b/core/sdk/src/clients/consumer.rs index 1f38f2c76e..e6f66ba497 100644 --- a/core/sdk/src/clients/consumer.rs +++ b/core/sdk/src/clients/consumer.rs @@ -466,10 +466,12 @@ unsafe impl Sync for IggyConsumer {} /// comes back empty is not an error and not the end of the stream, it just means nothing new has /// arrived yet. /// -/// A failed request is yielded as `Some(Err(..))` and leaves the consumer usable, while the next call -/// retries. Connection and authentication failures pause polling until the client has reconnected -/// and signed in again, which the consumer handles automatically. Hence, deciding when to give up on -/// repeated errors is up to you. +/// A failed poll request waits [`polling_retry_interval()`] before yielding `Some(Err(..))` and leaves +/// the consumer usable for the next call. This delay also applies to terminal server errors and when +/// the ordinary poll interval is disabled. Connection and authentication failures are yielded at once +/// and pause polling until the client has reconnected and signed in again, which the consumer handles +/// automatically; the next call parks for [`polling_retry_interval()`] while polling is paused. Hence, +/// deciding when to give up on repeated errors is up to you. /// /// For a boilerplate implementation of such a loop Iggy provides [`IggyConsumerMessageExt::consume_messages`]. /// @@ -548,7 +550,7 @@ unsafe impl Sync for IggyConsumer {} /// | [`allow_replay()`] | off | whether a message can be handed over again | /// | [`auto_join_consumer_group()`] | on | joining the group during [`init()`](Self::init) and again whenever the membership is lost. With [`do_not_auto_join_consumer_group()`] joining is up to the caller, and a poll without a membership fails with [`IggyError::ConsumerGroupMemberNotFound`] | /// | [`create_consumer_group_if_not_exists()`] | on | creating the group when it is missing | -/// | [`polling_retry_interval()`] | one second | wait between attempts while polling is blocked or the member holds no partitions | +/// | [`polling_retry_interval()`] | one second | delay before yielding poll errors other than connection and authentication failures, and between attempts while polling is blocked or the member holds no partitions | /// | [`init_retries()`] | none, one second apart | retries when the stream or topic is missing at [`init()`](Self::init) | /// | [`offset_drain_timeout()`] | five seconds | how long [`shutdown()`](Self::shutdown) waits for pending commits | /// | [`encryptor()`] | inherited from the client | decrypting payloads and user headers, see [Encryption](#encryption) | @@ -618,6 +620,14 @@ unsafe impl Sync for IggyConsumer {} /// [`topic()`]: crate::prelude::IggyConsumerBuilder::topic /// [`without_encryptor()`]: crate::prelude::IggyConsumerBuilder::without_encryptor /// [`without_poll_interval()`]: crate::prelude::IggyConsumerBuilder::without_poll_interval +/// +/// A server-side auto-commit poll can fail with `TooManyConsumerOffsets` when +/// its consumer needs a new offset key at the partition's configured limit. +/// The rejected poll returns no messages. Other auto-commit modes store the +/// same key through a client request. Background and shutdown stores log a +/// capacity failure but do not yield it through this stream. Only +/// [`AutoCommit::Disabled`] avoids automatic key allocation. Existing keys +/// remain usable. pub struct IggyConsumer { initialized: bool, shutdown: Arc, @@ -1351,8 +1361,11 @@ impl IggyConsumer { return Ok(PolledMessages::empty()); } - // Handle connection/auth errors - disable polling until event task re-enables - // it after reconnection and rejoin complete + // Connection and auth errors: disable polling until the event task + // re-enables it after reconnection and rejoin complete. Yielded at + // once: the next poll already parks on `can_poll`, so a retry sleep + // here would only delay the caller's view of an error it does not + // act on. if matches!( error, IggyError::Disconnected | IggyError::Unauthenticated | IggyError::StaleClient @@ -1361,9 +1374,10 @@ impl IggyConsumer { if is_consumer_group { joined_consumer_group.store(false, ORDERING); } - trace!("Retrying to poll messages in {retry_interval}..."); - sleep(retry_interval.get_duration()).await; + return Err(error); } + trace!("Retrying to poll messages in {retry_interval}..."); + sleep(retry_interval.get_duration()).await; Err(error) } } diff --git a/core/sdk/src/leader_aware.rs b/core/sdk/src/leader_aware.rs index 2d1c61be80..0319451e22 100644 --- a/core/sdk/src/leader_aware.rs +++ b/core/sdk/src/leader_aware.rs @@ -355,6 +355,8 @@ fn normalize_address(addr: &str) -> String { /// failed dial cannot cycle the request back through nodes it already tried. #[derive(Debug)] pub(crate) struct RosterWalk { + /// Whether the roster named at least one node when the walk was built. + roster_known: bool, remaining: VecDeque, attempted: Vec, } @@ -382,6 +384,7 @@ impl RosterWalk { Self { remaining: ordered, attempted: vec![current.to_owned()], + roster_known: !roster.is_empty(), } } @@ -406,6 +409,13 @@ impl RosterWalk { self.attempted.push(endpoint.clone()); Some(endpoint) } + + /// True only when the roster itself names one node. An empty roster (its + /// discovery failed) also leaves nothing to walk, but replaying that one + /// address would be a guess, not a decision. + pub(crate) fn is_single_endpoint(&self) -> bool { + self.roster_known && self.remaining.is_empty() && self.attempted.len() == 1 + } } /// Coordinates callers of one client's complete connect and authentication @@ -777,6 +787,10 @@ mod tests { assert_eq!(walk.next().as_deref(), Some("10.0.0.2:8090")); assert_eq!(walk.next().as_deref(), Some("10.0.0.3:8090")); assert_eq!(walk.next(), None); + assert!( + !walk.is_single_endpoint(), + "exhausting a cluster does not make it a single node" + ); let mut from_last = RosterWalk::new("10.0.0.3:8090", &roster); assert_eq!(from_last.next().as_deref(), Some("10.0.0.1:8090")); @@ -784,6 +798,17 @@ mod tests { assert_eq!(from_last.next(), None); } + #[test] + fn given_single_endpoint_roster_when_exhausted_should_allow_local_retry() { + assert!(!RosterWalk::new("127.0.0.1:8090", &[]).is_single_endpoint()); + let mut walk = RosterWalk::new("127.0.0.1:8090", &["localhost:8090".to_owned()]); + assert!(walk.is_single_endpoint()); + assert_eq!(walk.next(), None); + assert!(walk.is_single_endpoint()); + walk.record_attempt("127.0.0.2:8090"); + assert!(!walk.is_single_endpoint()); + } + #[test] fn a_roster_walk_never_revisits_redirects_or_duplicate_spellings() { let mut walk = RosterWalk::new( diff --git a/core/sdk/src/quic/quic_client.rs b/core/sdk/src/quic/quic_client.rs index 5dcadc1d60..e0493c9469 100644 --- a/core/sdk/src/quic/quic_client.rs +++ b/core/sdk/src/quic/quic_client.rs @@ -162,11 +162,12 @@ impl BinaryTransport for QuicClient { && self.config.reconnection.enabled && !matches!(self.config.auto_login, AutoLogin::Disabled) { - let _routing_guard = + let routing_guard = match tokio::time::timeout_at(roster_deadline, self.routing_lock.lock()).await { Ok(guard) => guard, Err(_) => return Err(IggyError::TransientNotAccepted), }; + let mut routing_guard = Some(routing_guard); let overall_deadline = roster_deadline; // A concurrent refused request may have completed the movement // while this request waited for the gate. @@ -217,8 +218,17 @@ impl BinaryTransport for QuicClient { (target, false) } else if let Some(next) = roster_walk.as_mut().and_then(RosterWalk::next) { (next, true) + } else if roster_walk + .as_ref() + .is_some_and(RosterWalk::is_single_endpoint) + { + // A single-node roster can still be converging a newly + // committed partition. Retry this explicitly unadmitted + // request on the current endpoint within the same budget. + drop(routing_guard.take()); + (current, false) } else { - break; + return Err(IggyError::TransientNotAccepted); }; loop { diff --git a/core/sdk/src/tcp/tcp_client.rs b/core/sdk/src/tcp/tcp_client.rs index 1546708330..b559f81ba4 100644 --- a/core/sdk/src/tcp/tcp_client.rs +++ b/core/sdk/src/tcp/tcp_client.rs @@ -1280,8 +1280,23 @@ impl TcpClient { return Err(IggyError::TransientNotAccepted); } + if roster_walk + .as_ref() + .is_some_and(RosterWalk::is_single_endpoint) + { + // Discovery already proved there is no routing choice. + // Retry without reacquiring the lock so reconnects and + // unrelated requests do not wait out this deadline. + drop(routing_guard.take()); + continue; + } + if routing_guard.is_none() { - routing_guard = Some(self.routing_lock.lock().await); + routing_guard = Some( + tokio::time::timeout_at(overall_deadline, self.routing_lock.lock()) + .await + .map_err(|_| IggyError::TransientNotAccepted)?, + ); // A concurrent refused request may have moved the // shared client while this request waited. continue; @@ -1323,6 +1338,16 @@ impl TcpClient { // hops would bounce the request between two nodes and // never reach the rest of the roster. (next, true) + } else if roster_walk + .as_ref() + .is_some_and(RosterWalk::is_single_endpoint) + { + // Partition materialisation can outlast the short retry + // window on a single node. No routing change is needed. + // Subsequent refusals take the lock-free local retry + // branch above rather than reacquiring this guard. + drop(routing_guard.take()); + continue; } else { return Err(IggyError::TransientNotAccepted); }; diff --git a/core/sdk/src/websocket/websocket_client.rs b/core/sdk/src/websocket/websocket_client.rs index 2a22683e02..1ecc2997bd 100644 --- a/core/sdk/src/websocket/websocket_client.rs +++ b/core/sdk/src/websocket/websocket_client.rs @@ -155,11 +155,12 @@ impl BinaryTransport for WebSocketClient { && self.config.reconnection.enabled && !matches!(self.config.auto_login, AutoLogin::Disabled) { - let _routing_guard = + let routing_guard = match tokio::time::timeout_at(roster_deadline, self.routing_lock.lock()).await { Ok(guard) => guard, Err(_) => return Err(IggyError::TransientNotAccepted), }; + let mut routing_guard = Some(routing_guard); let overall_deadline = roster_deadline; // A concurrent refused request may have completed the movement // while this request waited for the gate. @@ -210,8 +211,17 @@ impl BinaryTransport for WebSocketClient { (target, false) } else if let Some(next) = roster_walk.as_mut().and_then(RosterWalk::next) { (next, true) + } else if roster_walk + .as_ref() + .is_some_and(RosterWalk::is_single_endpoint) + { + // A single-node roster can still be converging a newly + // committed partition. Retry this explicitly unadmitted + // request on the current endpoint within the same budget. + drop(routing_guard.take()); + (current, false) } else { - break; + return Err(IggyError::TransientNotAccepted); }; loop { diff --git a/core/server/Cargo.toml b/core/server/Cargo.toml index 1d59114880..553d8d9ff2 100644 --- a/core/server/Cargo.toml +++ b/core/server/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "server" -version = "0.9.0-edge.6" +version = "0.9.0-edge.7" edition = "2024" license = "Apache-2.0" publish = false diff --git a/core/server/config.toml b/core/server/config.toml index e369125132..55a3ad5060 100644 --- a/core/server/config.toml +++ b/core/server/config.toml @@ -1025,6 +1025,36 @@ prepare_queue_depth = 32 # actually sees, not the node's client total. dedup_clients_max = 4096 +# Distinct durable consumer-offset keys a partition primary admits per kind. +# Standalone consumers and consumer groups are counted separately. Existing +# keys remain writable at the limit. A new key is rejected before consensus. +# A non-empty auto_commit poll that needs a new offset key is also rejected +# with TooManyConsumerOffsets and returns no messages. A poll whose auto_commit +# cannot be submitted (the owning shard's inbox is full, or the partition +# changed primary during the read) is rejected with TransientNotAccepted, also +# without messages; both are retriable. Polls with auto_commit disabled do not +# allocate offset keys and remain available. +# UPGRADE: existing files are loaded even above this limit, but new keys then +# remain blocked until offsets are explicitly deleted or this limit is raised. +# Standalone offsets have no automatic expiry. Reuse stable consumer ids and +# count existing numeric offset files per partition and kind before upgrading. +# The environment override is IGGY_PARTITION_CONSUMER_OFFSETS_MAX. +# UPGRADE: on replicated partitions, offset stores and deletes carrying NoAck +# wait for VSR commit. First-party SDK offset methods request Quorum. +# Must be > 0 and <= 262144. +consumer_offsets_max = 4096 + +# Whether consumer-offset files are written crash-safe. On, every committed +# offset store writes a sibling file, fdatasyncs it and renames it over the +# prior cursor, and the offsets directory is fsynced once per commit walk before +# any reply in that walk is sent. Independent of the topic's enforce_fsync, +# which governs message and index files. Off, the file is rewritten in place +# with no sync. A lost or torn cursor can cause replay from the earliest retained +# data. On, the filesystem must support both file and directory sync. A failed +# required barrier is reported as a failure, even if the mutation is visible. +# The environment override is IGGY_PARTITION_CONSUMER_OFFSET_ENFORCE_FSYNC. +consumer_offset_enforce_fsync = false + # How many offsets a partition claims in its superblock ahead of the mint # counter before it will append, so a crash-restarted replica resumes above # every offset it confirmed to a client instead of re-minting it for a different diff --git a/core/server/src/boot/recovery.rs b/core/server/src/boot/recovery.rs index 46558ffc7d..812e4ee932 100644 --- a/core/server/src/boot/recovery.rs +++ b/core/server/src/boot/recovery.rs @@ -122,6 +122,7 @@ pub(in crate::boot) async fn build_shard_for_thread( iggy_common::DEFAULT_SIZE_OF_MESSAGES_REQUIRED_TO_SAVE, ), enforce_fsync: iggy_common::DEFAULT_ENFORCE_FSYNC, + consumer_offset_enforce_fsync: config.partition.consumer_offset_enforce_fsync, validate_checksum: config.system.partition.validate_checksum, segment_size: IggyByteSize::from(iggy_common::DEFAULT_SEGMENT_SIZE), preallocate_segments: iggy_common::DEFAULT_PREALLOCATE_SEGMENTS, @@ -331,6 +332,16 @@ const _: () = assert!( const _: () = assert!( configs::partition::PARTITION_DEDUP_CLIENTS_DEFAULT == consensus::PARTITION_DEDUP_CLIENTS_MAX ); +const _: () = assert!( + configs::partition::PARTITION_CONSUMER_OFFSETS_DEFAULT + == partitions::DEFAULT_CONSUMER_OFFSETS_MAX +); +const _: () = assert!( + 4 * configs::partition::PARTITION_CONSUMER_OFFSETS_CEILING + <= partitions::CONSUMER_OFFSETS_ENTRIES_MAX as usize, + "four ceilings fill the transfer decoder's entry budget exactly, with no headroom left; raise \ + CONSUMER_OFFSETS_ENTRIES_MAX before raising PARTITION_CONSUMER_OFFSETS_CEILING" +); const _: () = assert!(configs::metadata::DEFAULT_METADATA_CLIENTS_TABLE_MAX == consensus::CLIENTS_TABLE_MAX); const _: () = diff --git a/core/server/src/consumer_group.rs b/core/server/src/consumer_group.rs index d4b77eb954..abeca570f0 100644 --- a/core/server/src/consumer_group.rs +++ b/core/server/src/consumer_group.rs @@ -25,13 +25,12 @@ //! primary enriches the op here before replication, mirroring the PAT mint //! in [`crate::pat`] and the password hash in [`crate::users`]. -use crate::responses::resolve_partition_namespace; +use crate::responses::{resolve_offset_group_id, resolve_partition_namespace}; use crate::shell::{ShellBus, ShellShard}; use crate::wire::{request_body, rewrite_request_body}; use consensus::MetadataHandle; use iggy_binary_protocol::PrepareHeader; use iggy_binary_protocol::codec::{WireDecode, WireEncode}; -use iggy_binary_protocol::primitives::consumer::WireConsumer; use iggy_binary_protocol::requests::consumer_groups::{ JoinConsumerGroupRequest as WireJoinConsumerGroupRequest, LeaveConsumerGroupRequest as WireLeaveConsumerGroupRequest, @@ -226,16 +225,21 @@ where let body = request_body(&request); // The store/delete ops differ only in the decode type; this collapses // their identical decode -> resolve group id -> rewrite consumer id -> - // re-encode bodies. A non-group consumer or unresolved group returns the - // request untouched (the apply/read path handles the miss). + // re-encode bodies. Individual consumers pass through. A group identifier + // that metadata cannot resolve is rejected before it can create a raw file + // in the group-offset directory. macro_rules! rewrite_group_offset { ($ty:ty) => {{ let mut wire = <$ty>::decode_from(body).map_err(|_| IggyError::InvalidCommand)?; - let Some(group_id) = - resolve_group_offset_id(shard, &wire.consumer, (&wire.stream_id, &wire.topic_id)) - else { + if wire.consumer.kind != KIND_CONSUMER_GROUP { return Ok(request); - }; + } + let group_id = resolve_offset_group_id( + shard.plane.metadata().mux_stm.streams(), + &wire.stream_id, + &wire.topic_id, + &wire.consumer.id, + )?; // The partition-plane group-offset key is u32 (see the documented // ceiling on `Topic::next_consumer_group_id`). Clamp on the // ~4-billion-creates overflow rather than panic this live @@ -254,29 +258,3 @@ where rewrite_request_body(&request, &rewritten) } - -/// Resolve the monotonic group id for a group consumer-offset op, or `None` for -/// an individual consumer (kind != 2) / unresolved group (leave the body as-is; -/// the apply / read path handle the miss). -fn resolve_group_offset_id( - shard: &Rc>, - consumer: &WireConsumer, - namespace: (&WireIdentifier, &WireIdentifier), -) -> Option -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - if consumer.kind != KIND_CONSUMER_GROUP { - return None; - } - shard - .plane - .metadata() - .mux_stm - .streams() - .resolve_consumer_group_id(namespace.0, namespace.1, &consumer.id) -} diff --git a/core/server/src/dispatch/mod.rs b/core/server/src/dispatch/mod.rs index 50b40dcd25..662c25038d 100644 --- a/core/server/src/dispatch/mod.rs +++ b/core/server/src/dispatch/mod.rs @@ -909,6 +909,7 @@ mod tests { messages_required_to_save: 1, size_of_messages_required_to_save: iggy_common::IggyByteSize::from(1024_u64), enforce_fsync: false, + consumer_offset_enforce_fsync: false, validate_checksum: true, segment_size: iggy_common::IggyByteSize::from(1_048_576_u64), preallocate_segments: false, diff --git a/core/server/src/dispatch/partition.rs b/core/server/src/dispatch/partition.rs index ac20090abc..76fcd985c1 100644 --- a/core/server/src/dispatch/partition.rs +++ b/core/server/src/dispatch/partition.rs @@ -59,7 +59,7 @@ use iggy_binary_protocol::{ AckLevel, Command, KIND_CONSUMER_GROUP, Operation, RoutedRequestHeader, WireDecode, WireEncode, WireIdentifier, }; -use iggy_common::{IggyError, PollingStrategy, RESYNC_REQUIRED_PARTITION_SENTINEL}; +use iggy_common::{ConsumerKind, IggyError, PollingStrategy, RESYNC_REQUIRED_PARTITION_SENTINEL}; use journal::superblock::SuperblockStore; use journal::{Journal, JournalHandle}; use message_bus::{AUTO_COMMIT_CLIENT_ID, BusMessage}; @@ -67,7 +67,10 @@ use metadata::impls::metadata::{ StreamsFrontend, build_truncate_partition_client_message, build_truncate_partition_client_message_with_identifiers, }; -use partitions::{AutoCommitApplied, PollPlan, PollingArgs, PollingConsumer}; +use partitions::{ + AutoCommitApplied, ConsumerOffsetCapacityError, PollFragments, PollPlan, PollingArgs, + PollingConsumer, +}; use server_common::Message; use server_common::sharding::IggyNamespace; use shard::shards_table::ShardsTable; @@ -111,14 +114,8 @@ where spawn_poll_io(Rc::clone(&shard), namespace, plan, reply); } Some(plan) => { - let (fragments, current_offset, auto_commit) = plan.execute_resident(); - if let Some(applied) = auto_commit { - submit_auto_commit(&shard, namespace, &applied); - } - let _ = reply.try_send(PartitionReadReply::Poll { - fragments, - current_offset, - }); + let result = poll_reply(&shard, namespace, plan.execute_resident()); + let _ = reply.try_send(result); } } } @@ -190,7 +187,7 @@ fn spawn_poll_io( // measures near-zero real time and never fires). Do not derive any // replicated or reply value from it, or replay determinism breaks. let poll_started = std::time::Instant::now(); - let (fragments, current_offset, auto_commit) = plan.execute().await; + let result = plan.execute().await; let elapsed = poll_started.elapsed(); if elapsed > std::time::Duration::from_secs(1) { warn!( @@ -199,28 +196,67 @@ fn spawn_poll_io( "slow partition poll; gather side may have timed out" ); } - // Fire-and-forget: the poll reply is not gated on the offset commit. - if let Some(applied) = auto_commit { - submit_auto_commit(&shard, namespace, &applied); - } - let _ = reply.try_send(PartitionReadReply::Poll { - fragments, - current_offset, - }); + let result = poll_reply(&shard, namespace, result); + let _ = reply.try_send(result); }); } +fn poll_reply( + shard: &Rc>, + namespace: IggyNamespace, + result: Result<(PollFragments, u64, Option), ConsumerOffsetCapacityError>, +) -> PartitionReadReply +where + B: ShellBus, + MJ: JournalHandle + 'static, + MJ::Target: Journal, Header = PrepareHeader>, + S: 'static, + SB: SuperblockStore + 'static, +{ + match result { + Ok((fragments, current_offset, auto_commit)) => { + if let Some(applied) = auto_commit + && let Err(error) = submit_auto_commit(shard, namespace, applied) + { + PartitionReadReply::Rejected(error) + } else { + PartitionReadReply::Poll { + fragments, + current_offset, + } + } + } + Err(error) => { + if !error.uncertain { + shard.metrics().record_consumer_offset_denied(error.kind); + } + warn_auto_commit_capacity(namespace, error); + PartitionReadReply::Rejected(error.into()) + } + } +} + /// Replicate a poll's auto-committed offset through the partition consensus so /// it survives failover, mirroring the explicit `StoreConsumerOffset` path: the -/// same op code, submitted onto the owning shard's own pipeline. Best-effort and -/// fire-and-forget -- the poll reply never waits on it, and a full inbox drops -/// the op at WARN rather than backpressuring the reply. +/// same op code, submitted onto the owning shard's own pipeline. The poll reply +/// does not wait for commit, but a primary reserves cardinality before this +/// submission and rejects the poll if the local inbox cannot accept it. /// /// The partition plane admits writes on the primary only (it asserts so), and a /// poll is served on whichever node owns the namespace locally, which may be a -/// backup. So gate on primary status here and drop at WARN otherwise; auto-commit +/// backup. So gate on primary status here. Auto-commit /// is server-managed best-effort (at-least-once delivery), so a follower-served -/// poll simply does not advance the durable offset. +/// poll simply does not advance the durable offset. The same contract covers a +/// local cursor that never became durable: when the per-kind live map is over +/// its limit the partition evicts such a cursor, and that consumer's next +/// `Next` poll restarts from offset 0. +/// +/// A poll whose auto-commit cannot be submitted is answered +/// `TransientNotAccepted` and returns no messages, even though the fragments +/// were already read: the owning shard's inbox refused the frame, or the +/// partition changed primary or incarnation during the read. Like the +/// `TooManyConsumerOffsets` refusal at the key limit, it returns no batch. +/// Transient refusals permit retry. Capacity refusals need available capacity. /// /// Coalescing: an offset the partition's committed high-water already covers is /// dropped without a consensus op (the steady state for a re-poll of committed @@ -230,61 +266,86 @@ fn spawn_poll_io( fn submit_auto_commit( shard: &Rc>, namespace: IggyNamespace, - applied: &AutoCommitApplied, -) where + applied: AutoCommitApplied, +) -> Result<(), IggyError> +where B: ShellBus, MJ: JournalHandle + 'static, MJ::Target: Journal, Header = PrepareHeader>, S: 'static, SB: SuperblockStore + 'static, { - enum AutoCommitGate { - Submit, - Covered, - NotPrimary, - } - let gate = shard - .plane - .partitions() - .with_partition(&namespace, |partition| { - let consensus = partition.consensus(); - if !(consensus.is_primary() && consensus.is_normal() && !consensus.is_transferring()) { - AutoCommitGate::NotPrimary - } else if partition.is_auto_commit_offset_covered( - applied.kind, - applied.consumer_id, - applied.offset, - ) { - AutoCommitGate::Covered - } else { - AutoCommitGate::Submit - } - }); - match gate { - Some(AutoCommitGate::Submit) => {} - Some(AutoCommitGate::Covered) => return, - Some(AutoCommitGate::NotPrimary) | None => { - warn!( + // Everything inside is synchronous: `admit` rolls the eager cursor update + // back on `Err` in the same call, and no await may sit between the update + // and that rollback. + applied.admit(|applied| { + let primary = shard + .plane + .partitions() + .with_partition(&namespace, |partition| { + let consensus = partition.consensus(); + if !partition.auto_commit_admission_ready(applied) { + return Err(IggyError::TransientNotAccepted); + } + Ok(consensus.is_primary() && consensus.is_normal() && !consensus.is_transferring()) + }); + if matches!(primary, None | Some(Err(_))) { + return Err(IggyError::TransientNotAccepted); + } + if primary == Some(Ok(false)) { + debug!( namespace_raw = namespace.inner(), "auto-commit offset not replicated: partition not primary on this node (best-effort)" ); - return; + return Ok(()); } - } - let message = match build_auto_commit_request(namespace, applied) { - Ok(message) => message, - Err(error) => { + let reservation = match applied.reserve_durable() { + Ok(Some(reservation)) => reservation, + Ok(None) => return Ok(()), + Err(error) => { + if !error.uncertain { + shard.metrics().record_consumer_offset_denied(applied.kind); + } + warn_auto_commit_capacity(namespace, error); + return Err(error.into()); + } + }; + let message = build_auto_commit_request(namespace, applied).inspect_err(|error| { warn!( namespace_raw = namespace.inner(), error = %error, "failed to build auto-commit store-offset request" ); - return; - } - }; - // Routes by namespace to this same (owning, primary) shard's inbox; the pump - // admits it next turn exactly like a client store. `dispatch` never blocks. - shard.dispatch(message.into_generic()); + })?; + // Routes by namespace to this same owning primary shard's inbox. The + // pump admits it next turn exactly like a client store. `dispatch` + // never blocks. + shard + .submit_auto_commit_offset(message, reservation) + .map_err(|_| IggyError::TransientNotAccepted) + }) +} + +fn warn_auto_commit_capacity( + namespace: IggyNamespace, + error: partitions::ConsumerOffsetCapacityError, +) { + if error.first_in_episode && error.uncertain { + warn!( + namespace_raw = namespace.inner(), + kind = ?error.kind, + "consumer offset accounting unavailable during auto-commit" + ); + } else if error.first_in_episode { + warn!( + namespace_raw = namespace.inner(), + kind = ?error.kind, + occupied = error.occupied, + limit = error.limit, + config = "[partition] consumer_offsets_max", + "consumer offset map limit reached during auto-commit" + ); + } } /// Build the synthetic `StoreConsumerOffset` request for an auto-commit, keyed @@ -395,13 +456,15 @@ pub async fn dispatch_partition_request( operation = ?header.operation, "partition request with unresolved namespace; replying denied" ); - send_deny_reply( - shard, - transport_client_id, - &header, - IggyError::ResourceNotFound(String::new()).as_code(), - ) - .await; + let status = if matches!( + header.operation, + Operation::StoreConsumerOffset | Operation::DeleteConsumerOffset + ) { + error.as_code() + } else { + IggyError::ResourceNotFound(String::new()).as_code() + }; + send_deny_reply(shard, transport_client_id, &header, status).await; return; } }; @@ -471,11 +534,8 @@ pub async fn dispatch_partition_request( // metadata access to resolve it. let request = match maybe_rewrite_consumer_offset_request(shard, request) { Ok(rewritten) => rewritten, - // Not reachable through the wire path: the same body already decoded in - // `resolve_partition_request_namespace` above, and re-encoding it can - // only fail past `u32::MAX` bytes against a 64 MiB request cap. Denying - // typed keeps a future re-encode failure from acking work the partition - // plane never saw. + // Metadata can change during the routing wait, so a previously valid + // group may be missing now. Preserve the typed failure before submit. Err(error) => { warn!( transport_client_id, @@ -544,6 +604,7 @@ async fn relay_partition_reply( S: 'static, SB: SuperblockStore + 'static, { + let consumer_kind = consumer_offset_kind(&request); let Ok(ticket) = shard.partition_submit(namespace, request) else { // `PartitionSubmitRefused`: the frame never reached the owning shard, // so this is a known outcome and the client can be told now rather @@ -569,6 +630,18 @@ async fn relay_partition_reply( // commits moments later. The client's read-timeout is the recovery. return; }; + if let Some(kind) = consumer_kind + && reply + .as_slice() + .get(..size_of::()) + .and_then(|bytes| { + bytemuck::checked::try_from_bytes::(bytes) + .ok() + }) + .is_some_and(|header| header.status == IggyError::TooManyConsumerOffsets.as_code()) + { + shard.metrics().record_consumer_offset_denied(kind); + } if let Err(error) = shard .bus .send_to_client(transport_client_id, reply.into_frozen()) @@ -584,6 +657,15 @@ async fn relay_partition_reply( }); } +fn consumer_offset_kind(request: &Message) -> Option { + if request.header().operation != Operation::StoreConsumerOffset { + return None; + } + // WireConsumer starts with its kind byte. The dispatch path has already + // decoded and validated the complete request. + ConsumerKind::from_code(*request_body(request).first()?).ok() +} + /// Serve `poll_messages`: resolve the partition namespace, run the read on /// the owning shard ([`shard::IggyShard::partition_read`]), and re-encode /// the stored batches into the legacy wire `PolledMessages` body. @@ -651,7 +733,12 @@ pub(in crate::dispatch) async fn handle_poll_messages( .await; return; } - Err(fallback) => fallback, + Err(ReadPolledMessagesError::Fallback(fallback)) => fallback, + Err(ReadPolledMessagesError::Rejected(error)) => { + send_non_replicated_deny(shard, request, transport_client_id, error.as_code()) + .await; + return; + } } } // A generation fence: the client's cached assignment went stale after a @@ -701,7 +788,7 @@ async fn read_polled_messages( transport_client_id: u128, request: &Message, (namespace, partition_id, consumer, args): DecodedPollRequest, -) -> Result +) -> Result where B: ShellBus, MJ: JournalHandle + 'static, @@ -730,8 +817,9 @@ where error = %error, "failed to re-encode polled batches; replying empty poll" ); - empty_poll_fallback(partition_id) + ReadPolledMessagesError::Fallback(empty_poll_fallback(partition_id)) }), + Some(PartitionReadReply::Rejected(error)) => Err(ReadPolledMessagesError::Rejected(error)), other => { warn!( transport_client_id, @@ -739,11 +827,18 @@ where reply_was_none = other.is_none(), "partition read failed; replying empty poll" ); - Err(empty_poll_fallback(partition_id)) + Err(ReadPolledMessagesError::Fallback(empty_poll_fallback( + partition_id, + ))) } } } +enum ReadPolledMessagesError { + Fallback((Bytes, FrameChannel)), + Rejected(IggyError), +} + /// The fail-fast poll reply for a partition that could not answer: the /// 16-byte empty poll for `partition_id`, riding the re-sync sentinel /// channel when the id is the sentinel and the empty-frame channel @@ -1300,6 +1395,7 @@ mod tests { }; use iggy_binary_protocol::ReplyHeader; use iggy_binary_protocol::primitives::partition_assignment::CreatedPartitionAssignment; + use iggy_binary_protocol::requests::consumer_offsets::DeleteConsumerOffsetRequest; use iggy_binary_protocol::requests::messages::SendMessagesHeader; use iggy_binary_protocol::requests::streams::CreateStreamRequest; use iggy_binary_protocol::requests::topics::{ @@ -1320,6 +1416,146 @@ mod tests { ShardIdentity, shard_channel, }; + #[compio::test] + async fn given_invalid_partition_writes_when_resolving_should_preserve_offset_error_codes() { + const VSR_CLIENT: u128 = 1; + let bus = SpyBus::default(); + let shard = Rc::new(test_shard(&bus, 0, 1, 1)); + // Stream 0 / topic 0 / partition 0 committed straight into the STM, so a + // group op resolves its namespace and reaches the group fence, which + // runs before the routable wait. + let md = shard.plane.metadata(); + md.mux_stm.users().ensure_root_user("iggy", "hash"); + let create_stream = CreateStreamRequest { + name: WireName::new("stream").unwrap(), + options: WireOptions::empty(), + }; + md.mux_stm + .update(prepare_message( + Operation::CreateStream, + VSR_CLIENT, + 1, + &create_stream.to_bytes(), + )) + .unwrap(); + let create_topic = CreateTopicWithAssignmentsRequest { + request: CreateTopicRequest { + stream_id: WireIdentifier::numeric(0), + partitions_count: 1, + name: WireName::new("topic").unwrap(), + options: WireOptions::empty(), + }, + derived_options: WireOptions::empty(), + partitions: vec![CreatedPartitionAssignment { + partition_id: 0, + consensus_group_id: 1, + }], + created_view: 0, + }; + md.mux_stm + .update(prepare_message( + Operation::CreateTopicWithAssignments, + VSR_CLIENT, + 2, + &create_topic.to_bytes(), + )) + .unwrap(); + + let offset_body = |operation: Operation, + consumer: WireConsumer, + stream_id: WireIdentifier, + topic_id: WireIdentifier| { + if operation == Operation::StoreConsumerOffset { + StoreConsumerOffsetRequest { + consumer, + stream_id, + topic_id, + partition_id: Some(0), + offset: 0, + ack: iggy_binary_protocol::AckLevel::Quorum, + } + .to_bytes() + } else { + DeleteConsumerOffsetRequest { + consumer, + stream_id, + topic_id, + partition_id: Some(0), + ack: iggy_binary_protocol::AckLevel::Quorum, + } + .to_bytes() + } + }; + let mut cases = Vec::new(); + for operation in [ + Operation::StoreConsumerOffset, + Operation::DeleteConsumerOffset, + ] { + cases.push((operation, vec![1], IggyError::InvalidCommand)); + cases.push(( + operation, + offset_body( + operation, + WireConsumer::consumer(WireIdentifier::Numeric(1)), + WireIdentifier::Numeric(404), + WireIdentifier::Numeric(1), + ) + .to_vec(), + IggyError::StreamIdNotFound(Identifier::numeric(404).unwrap()), + )); + // The group fence: an unknown group on a known topic answers the + // group's own not-found code, numeric and named, not a bare + // ResourceNotFound. + cases.push(( + operation, + offset_body( + operation, + WireConsumer::consumer_group(WireIdentifier::Numeric(999)), + WireIdentifier::Numeric(0), + WireIdentifier::Numeric(0), + ) + .to_vec(), + IggyError::ConsumerGroupIdNotFound( + Identifier::numeric(999).unwrap(), + Identifier::numeric(0).unwrap(), + ), + )); + cases.push(( + operation, + offset_body( + operation, + WireConsumer::consumer_group(WireIdentifier::named("ghost").unwrap()), + WireIdentifier::Numeric(0), + WireIdentifier::Numeric(0), + ) + .to_vec(), + IggyError::ConsumerGroupNameNotFound( + "ghost".to_owned(), + Identifier::numeric(0).unwrap(), + ), + )); + } + cases.push(( + Operation::SendMessages, + vec![1], + IggyError::ResourceNotFound(String::new()), + )); + for (index, (operation, body, expected)) in cases.into_iter().enumerate() { + let request = request_message(operation, 1, 1, index as u64 + 1, &body); + dispatch_partition_request(&shard, request, 1, 1, 91, Some(DEFAULT_ROOT_USER_ID)).await; + let replies = bus.client_replies.borrow(); + assert_eq!( + replies.len(), + index + 1, + "{operation:?} must reply exactly once" + ); + let (_, frame) = replies.last().unwrap(); + let start = std::mem::offset_of!(ReplyHeader, status); + let status = u32::from_le_bytes(frame[start..start + 4].try_into().unwrap()); + assert_eq!(status, expected.as_code(), "{operation:?}"); + } + } + /// An undecodable request body is a PERMANENT client error, so every read /// on this path must answer a nonzero status. The fail-fast shapes these /// used to borrow all decode as success: the 16-byte empty poll reads as a @@ -1480,6 +1716,7 @@ mod tests { messages_required_to_save: 1, size_of_messages_required_to_save: iggy_common::IggyByteSize::from(1024_u64), enforce_fsync: false, + consumer_offset_enforce_fsync: false, validate_checksum: true, segment_size: iggy_common::IggyByteSize::from(1_048_576_u64), preallocate_segments: false, @@ -1606,6 +1843,7 @@ mod tests { messages_required_to_save: 1, size_of_messages_required_to_save: iggy_common::IggyByteSize::from(1024_u64), enforce_fsync: false, + consumer_offset_enforce_fsync: false, validate_checksum: true, segment_size: iggy_common::IggyByteSize::from(1_048_576_u64), preallocate_segments: false, @@ -1671,6 +1909,7 @@ mod tests { messages_required_to_save: 1, size_of_messages_required_to_save: iggy_common::IggyByteSize::from(1024_u64), enforce_fsync: false, + consumer_offset_enforce_fsync: false, validate_checksum: true, segment_size: iggy_common::IggyByteSize::from(1_048_576_u64), preallocate_segments: false, diff --git a/core/server/src/dispatch/session_ops.rs b/core/server/src/dispatch/session_ops.rs index a8d2e62db3..0dde38649a 100644 --- a/core/server/src/dispatch/session_ops.rs +++ b/core/server/src/dispatch/session_ops.rs @@ -1472,6 +1472,7 @@ mod tests { messages_required_to_save: 1, size_of_messages_required_to_save: iggy_common::IggyByteSize::from(1024_u64), enforce_fsync: false, + consumer_offset_enforce_fsync: false, validate_checksum: true, segment_size: iggy_common::IggyByteSize::from(1_048_576_u64), preallocate_segments: false, diff --git a/core/server/src/dispatch/test_support.rs b/core/server/src/dispatch/test_support.rs index c0a0057345..bbfded15ba 100644 --- a/core/server/src/dispatch/test_support.rs +++ b/core/server/src/dispatch/test_support.rs @@ -184,6 +184,7 @@ pub fn test_shard(bus: &SpyBus, replica: u8, replica_count: u8, incarnation: u12 messages_required_to_save: 1, size_of_messages_required_to_save: iggy_common::IggyByteSize::from(1024_u64), enforce_fsync: false, + consumer_offset_enforce_fsync: false, validate_checksum: true, segment_size: iggy_common::IggyByteSize::from(1_048_576_u64), preallocate_segments: false, diff --git a/core/server/src/http/handlers.rs b/core/server/src/http/handlers.rs index fc877054bb..cdd8f638ce 100644 --- a/core/server/src/http/handlers.rs +++ b/core/server/src/http/handlers.rs @@ -1296,6 +1296,7 @@ pub(in crate::http) async fn poll_messages( )) } Some(PartitionReadReply::NotFound) => Err(ReadError::NotFound), + Some(PartitionReadReply::Rejected(error)) => Err(ReadError::Rejected(error)), Some(_) => Err(ReadError::Rejected(IggyError::InvalidCommand)), None => Err(ReadError::Timeout), } @@ -1466,13 +1467,26 @@ pub(in crate::http) async fn store_consumer_offset( let request = store_offset_wire_request(&stream_id, &topic_id, &command) .map_err(PartitionWriteError::Rejected)?; let body = request.to_bytes(); - SendWrapper::new(partition_write_replicated( + let consumer_kind = command.consumer.kind; + let result = SendWrapper::new(partition_write_replicated( &state, &identity.session, Operation::StoreConsumerOffset, &body, )) - .await?; + .await; + if matches!( + &result, + Err(PartitionWriteError::Rejected( + IggyError::TooManyConsumerOffsets + )) + ) { + state + .shard + .metrics() + .record_consumer_offset_denied(consumer_kind); + } + result?; Ok(StatusCode::NO_CONTENT) } diff --git a/core/server/src/offset_recovery.rs b/core/server/src/offset_recovery.rs index afb60c8209..40dca3ff81 100644 --- a/core/server/src/offset_recovery.rs +++ b/core/server/src/offset_recovery.rs @@ -27,168 +27,200 @@ //! here as unchecksummed. use iggy_common::{ConsumerGroupId, ConsumerKind, ConsumerOffset, IggyError}; -use partitions::offset_storage::{OffsetRecord, decode_offset_record}; +use partitions::offset_storage::{OffsetRecord, decode_offset_record, offset_replacement_id}; +use std::path::PathBuf; use std::sync::atomic::AtomicU64; +use tokio::sync::{Semaphore, mpsc}; use tracing::{error, trace, warn}; const COMPONENT: &str = "STREAMING_PARTITIONS"; +const OFFSET_DIRECTORY_BUFFER: usize = 64; +static OFFSET_DIRECTORY_READERS: Semaphore = Semaphore::const_new(4); +type OffsetDirectoryEntries = mpsc::Receiver>>; -pub fn load_consumer_offsets(path: &str) -> Result, IggyError> { - trace!("Loading consumer offsets from path: {path}..."); - let Ok(dir_entries) = std::fs::read_dir(path) else { - return Err(IggyError::CannotReadConsumerOffsets(path.to_owned())); - }; - - let mut consumer_offsets = Vec::new(); - for dir_entry in dir_entries { - let dir_entry = match dir_entry { - Ok(entry) => entry, - Err(e) => { - warn!( - "Failed to read directory entry in consumer offsets path: {path}, \ - error: {e}, skipping." - ); - continue; - } - }; +pub struct RecoveredOffsets { + pub entries: Vec, + pub stranded_ids: Vec, +} - let metadata = match dir_entry.metadata() { - Ok(m) => m, - Err(e) => { - warn!( - "Failed to read metadata for entry in consumer offsets path: {path}, \ - error: {e}, skipping." - ); - continue; - } - }; +enum OffsetFileLoad { + Loaded(AtomicU64), + Removed, + Stranded, +} - if metadata.is_dir() { - continue; +impl Default for RecoveredOffsets { + fn default() -> Self { + Self { + entries: Vec::new(), + stranded_ids: Vec::new(), } - - let name = dir_entry.file_name().to_string_lossy().to_string(); - let Ok(consumer_id) = name.parse::() else { - warn!( - "Unexpected non-numeric consumer offset file: '{}', skipping.", - name - ); - continue; - }; - - let path = dir_entry.path(); - let Some(path) = path.to_str().map(str::to_owned) else { - error!("Invalid consumer ID path for file with name: '{}'.", name); - continue; - }; - - let Some(offset) = read_offset_file(&path, "consumer offset") else { - continue; - }; - - consumer_offsets.push(ConsumerOffset { - kind: ConsumerKind::Consumer, - consumer_id, - offset, - path, - }); } - - consumer_offsets.sort_by_key(|consumer_offset| consumer_offset.consumer_id); - Ok(consumer_offsets) } -pub fn load_consumer_group_offsets( +pub async fn load_consumer_offsets( path: &str, -) -> Result, IggyError> { - trace!("Loading consumer group offsets from path: {path}..."); - let Ok(dir_entries) = std::fs::read_dir(path) else { - return Err(IggyError::CannotReadConsumerOffsets(path.to_owned())); - }; +) -> Result, IggyError> { + let mut recovered = load_offsets(path, ConsumerKind::Consumer, |offset| offset).await?; + recovered.entries.sort_by_key(|offset| offset.consumer_id); + Ok(recovered) +} - let mut consumer_group_offsets = Vec::new(); - for dir_entry in dir_entries { - let dir_entry = match dir_entry { - Ok(entry) => entry, - Err(e) => { - warn!( - "Failed to read directory entry in consumer group offsets path: {path}, \ - error: {e}, skipping." - ); - continue; - } - }; +pub async fn load_consumer_group_offsets( + path: &str, +) -> Result, IggyError> { + load_offsets(path, ConsumerKind::ConsumerGroup, |offset| { + (ConsumerGroupId(offset.consumer_id as usize), offset) + }) + .await +} - let metadata = match dir_entry.metadata() { - Ok(m) => m, - Err(e) => { - warn!( - "Failed to read metadata for entry in consumer group offsets path: {path}, \ - error: {e}, skipping." - ); - continue; +async fn load_offsets( + path: &str, + kind: ConsumerKind, + construct: impl Fn(ConsumerOffset) -> T, +) -> Result, IggyError> { + trace!(?kind, path, "loading consumer offsets"); + let mut dir_entries = offset_directory_entries(path).await?; + let mut recovered = RecoveredOffsets::default(); + loop { + let entry_path = match dir_entries.recv().await { + Some(Ok(Some(path))) => path, + Some(Ok(None)) => break, + Some(Err(error)) => { + warn!(?kind, path, %error, "failed to enumerate offset directory"); + return Err(IggyError::CannotReadConsumerOffsets(path.to_owned())); } + None => return Err(IggyError::CannotReadConsumerOffsets(path.to_owned())), }; - - if metadata.is_dir() { + let name = entry_path + .file_name() + .unwrap_or_default() + .to_string_lossy() + .into_owned(); + if offset_replacement_id(&name).is_some() { + remove_stale_replacement(&entry_path, &name).await; continue; } - - let name = dir_entry.file_name().to_string_lossy().to_string(); - let Ok(raw_consumer_group_id) = name.parse::() else { + let Ok(consumer_id) = name.parse::() else { warn!( - "Unexpected non-numeric consumer group offset file: '{}', skipping.", - name + ?kind, + name, "unexpected non-numeric consumer offset file, skipping" ); continue; }; - let consumer_group_id = ConsumerGroupId(raw_consumer_group_id as usize); - - let path = dir_entry.path(); - let Some(path) = path.to_str().map(str::to_owned) else { - error!( - "Invalid consumer group offset path for file with name: '{}'.", - name - ); + let Some(path) = entry_path.to_str().map(str::to_owned) else { + error!(?kind, name, "invalid consumer offset path"); continue; }; - - let Some(offset) = read_offset_file(&path, "consumer group offset") else { - continue; + let offset = match read_offset_file(&path, offset_kind_label(kind)).await { + OffsetFileLoad::Loaded(offset) => offset, + OffsetFileLoad::Removed => continue, + OffsetFileLoad::Stranded => { + recovered.stranded_ids.push(consumer_id); + continue; + } }; - - let consumer_offset = ConsumerOffset { - kind: ConsumerKind::ConsumerGroup, - consumer_id: raw_consumer_group_id, + recovered.entries.push(construct(ConsumerOffset { + kind, + consumer_id, offset, path, - }; + })); + } + Ok(recovered) +} + +async fn offset_directory_entries(path: &str) -> Result { + // Compio has no asynchronous directory iterator and the shard's blocking + // pool is disabled. Bound both OS threads and buffered paths. The worker + // owns the permit so cancellation cannot exceed the concurrency bound. + let permit = OFFSET_DIRECTORY_READERS + .acquire() + .await + .map_err(|_| IggyError::CannotReadConsumerOffsets(path.to_owned()))?; + let (sender, receiver) = mpsc::channel(OFFSET_DIRECTORY_BUFFER); + let directory = path.to_owned(); + std::thread::Builder::new() + .name("iggy-offset-recovery".to_owned()) + .spawn(move || { + let _permit = permit; + let result = (|| { + // Only the directory read itself is fatal. One unreadable + // entry is skipped with a warning, as the on-reactor loader + // did, so a single bad dirent cannot keep a partition from + // booting. + for entry in std::fs::read_dir(&directory)? { + let entry = match entry { + Ok(entry) => entry, + Err(error) => { + warn!(path = directory, %error, "failed to read offset directory entry"); + continue; + } + }; + let is_file = match entry.file_type() { + Ok(file_type) => file_type.is_file(), + Err(error) => { + warn!(path = directory, %error, "failed to read offset entry type"); + continue; + } + }; + if is_file && sender.blocking_send(Ok(Some(entry.path()))).is_err() { + return Ok(()); + } + } + Ok(()) + })(); + // Explicit completion distinguishes an empty directory from an + // interrupted worker. Closed receivers simply abandon enumeration. + let _ = sender.blocking_send(result.map(|()| None)); + }) + .map_err(|error| { + error!(path, %error, "failed to start offset directory reader"); + IggyError::CannotReadConsumerOffsets(path.to_owned()) + })?; + Ok(receiver) +} - consumer_group_offsets.push((consumer_group_id, consumer_offset)); +/// A crashed atomic replacement leaves its sibling behind. The rename never +/// landed, so the sibling is never authoritative. Removal needs no directory +/// sync because a resurrected sibling is still ignored on the next load. +async fn remove_stale_replacement(path: &std::path::Path, name: &str) { + match compio::fs::remove_file(path).await { + Ok(()) => trace!("Removed stale offset replacement file: '{name}'."), + Err(e) => warn!( + "{COMPONENT} (error: {e}) - could not remove stale offset replacement \ + file: '{name}', skipping." + ), } +} - Ok(consumer_group_offsets) +const fn offset_kind_label(kind: ConsumerKind) -> &'static str { + match kind { + ConsumerKind::Consumer => "consumer offset", + ConsumerKind::ConsumerGroup => "consumer group offset", + } } -fn read_offset_file(path: &str, offset_kind: &'static str) -> Option { - let bytes = match std::fs::read(path) { +async fn read_offset_file(path: &str, offset_kind: &'static str) -> OffsetFileLoad { + let bytes = match compio::fs::read(path).await { Ok(bytes) => bytes, Err(e) => { warn!( "{COMPONENT} (error: {e}) - failed to read offset file, \ path: {path}, skipping." ); - return None; + return OffsetFileLoad::Stranded; } }; match decode_offset_record(&bytes) { - OffsetRecord::Value { offset, .. } => Some(AtomicU64::new(offset)), + OffsetRecord::Value { offset, .. } => OffsetFileLoad::Loaded(AtomicU64::new(offset)), OffsetRecord::Torn => { warn!( "{COMPONENT} - failed to read {offset_kind} from file (truncated), \ - path: {path}, skipping." + path: {path}, removing invalid file." ); - None + remove_invalid_offset_file(path, offset_kind).await } // Skipped rather than loaded: resuming from a cursor provably not the one // written reads as ordinary redelivery or a gap, never as corruption. @@ -206,13 +238,87 @@ fn read_offset_file(path: &str, offset_kind: &'static str) -> Option (offset: {offset}, expected: {expected}, found: {found}), \ path: {path}, removing it and resuming this consumer from the start." ); - if let Err(e) = std::fs::remove_file(path) { - error!( - "{COMPONENT} (error: {e}) - could not remove the corrupt \ - {offset_kind} file, path: {path}; remove it manually." - ); - } - None + remove_invalid_offset_file(path, offset_kind).await } } } + +async fn remove_invalid_offset_file(path: &str, offset_kind: &'static str) -> OffsetFileLoad { + if let Err(error) = compio::fs::remove_file(path).await { + error!( + "{COMPONENT} (error: {error}) - could not remove the invalid \ + {offset_kind} file, path: {path}; remove it manually." + ); + return OffsetFileLoad::Stranded; + } + let Some(parent) = std::path::Path::new(path).parent() else { + return OffsetFileLoad::Removed; + }; + match async { + let directory = compio::fs::File::open(parent).await?; + directory.sync_all().await + } + .await + { + Ok(()) => OffsetFileLoad::Removed, + Err(error) => { + error!( + "{COMPONENT} (error: {error}) - removed invalid {offset_kind} file but \ + could not sync its directory, path: {path}; retaining its capacity slot." + ); + OffsetFileLoad::Stranded + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[compio::test] + async fn given_missing_directory_when_loading_should_report_error_instead_of_empty_state() { + let dir = tempfile::tempdir().unwrap(); + let missing = dir.path().join("missing"); + assert!(matches!( + load_consumer_offsets(missing.to_str().unwrap()).await, + Err(IggyError::CannotReadConsumerOffsets(_)) + )); + } + + #[compio::test] + async fn given_full_directory_buffer_when_loader_is_cancelled_should_release_worker_capacity() { + let dir = tempfile::tempdir().unwrap(); + for id in 0..OFFSET_DIRECTORY_BUFFER * 2 { + std::fs::write(dir.path().join(id.to_string()), 0_u64.to_le_bytes()).unwrap(); + } + let path = dir.path().to_str().unwrap(); + for _ in 0..8 { + let entries = offset_directory_entries(path).await.unwrap(); + drop(entries); + } + let loaded = load_consumer_offsets(path).await.unwrap(); + assert_eq!(loaded.entries.len(), OFFSET_DIRECTORY_BUFFER * 2); + } + + #[compio::test] + async fn given_numeric_directory_and_torn_file_when_loading_should_remove_only_invalid_file() { + let dir = tempfile::tempdir().unwrap(); + std::fs::create_dir(dir.path().join("7")).unwrap(); + std::fs::write(dir.path().join("8"), [1, 2]).unwrap(); + std::fs::write(dir.path().join("9"), 12_u64.to_le_bytes()).unwrap(); + std::fs::write(dir.path().join("9.tmp"), [0_u8; 4]).unwrap(); + std::fs::write(dir.path().join("notes.tmp"), b"unrelated").unwrap(); + let path = dir.path().to_str().unwrap(); + let consumers = load_consumer_offsets(path).await.unwrap(); + assert!(!dir.path().join("9.tmp").exists()); + assert!(dir.path().join("notes.tmp").exists()); + assert_eq!(consumers.entries.len(), 1); + assert_eq!(consumers.entries[0].consumer_id, 9); + assert!(consumers.stranded_ids.is_empty()); + assert!(!dir.path().join("8").exists()); + let groups = load_consumer_group_offsets(path).await.unwrap(); + assert_eq!(groups.entries.len(), 1); + assert_eq!(groups.entries[0].0, ConsumerGroupId(9)); + assert!(groups.stranded_ids.is_empty()); + } +} diff --git a/core/server/src/partition_helpers.rs b/core/server/src/partition_helpers.rs index 15a519a223..af6e19028a 100644 --- a/core/server/src/partition_helpers.rs +++ b/core/server/src/partition_helpers.rs @@ -29,7 +29,9 @@ //! `CreateTopic` / `CreatePartitions` event has no matching local //! partition yet. -use crate::offset_recovery::{load_consumer_group_offsets, load_consumer_offsets}; +use crate::offset_recovery::{ + RecoveredOffsets, load_consumer_group_offsets, load_consumer_offsets, +}; use crate::segment_recovery::{RecoveredSegment, load_persisted_segments}; use crate::server_error::{PartitionRecoveryRefusal, ServerError}; use crate::shell::consensus_timers; @@ -39,8 +41,8 @@ use consensus::{ FreshGroupStart, JoinMode, LocalPipeline, VsrConsensus, VsrRestore, VsrState, fresh_group_start, }; use iggy_common::{ - ConsumerGroupOffsets, ConsumerOffsets, IggyByteSize, IggyError, IggyTimestamp, PartitionStats, - TopicRuntimeOptions, + ConsumerGroupOffsets, ConsumerKind, ConsumerOffsets, IggyByteSize, IggyError, IggyTimestamp, + PartitionStats, TopicRuntimeOptions, }; use journal::superblock::{PingPongSuperblock, SuperblockContents}; use message_bus::IggyMessageBus; @@ -155,7 +157,8 @@ pub async fn create_partition_file_hierarchy( /// Returns [`ServerError::ConsumerOffsetsLoad`] when the on-disk files /// exist but fail to decode. A stored offset past the offset space is clamped /// to `current_offset` (with a warning), not an error. -pub fn configure_consumer_offsets( +#[allow(clippy::too_many_lines)] +pub async fn configure_consumer_offsets( partition: &mut IggyPartition>, config: &ServerConfig, namespace: IggyNamespace, @@ -184,17 +187,18 @@ pub fn configure_consumer_offsets( // path, where the max leaves `current_offset` in charge as before. let offset_space_ceiling = current_offset.max(partition.mint_frontier().saturating_sub(1)); - let loaded_consumer_offsets = load_partition_consumer_offsets( + let recovered_consumers = load_partition_consumer_offsets( &consumer_offsets_path, "consumer", stream_id, topic_id, partition_id, - )?; - let consumer_offsets = ConsumerOffsets::with_capacity(loaded_consumer_offsets.len()); + ) + .await?; + let consumer_offsets = ConsumerOffsets::with_capacity(recovered_consumers.entries.len()); { let guard = consumer_offsets.pin(); - for offset in loaded_consumer_offsets { + for offset in recovered_consumers.entries { let recovered_offset = offset.offset.load(Ordering::Relaxed); if recovered_offset > offset_space_ceiling { // A crash can persist an offset ahead of the flushed data @@ -213,20 +217,30 @@ pub fn configure_consumer_offsets( ); offset.offset.store(current_offset, Ordering::Relaxed); } - guard.insert(offset.consumer_id as usize, offset); + let consumer_id = offset.consumer_id; + let committed_offset = offset.offset.load(Ordering::Relaxed); + partition.seed_recovered_consumer_offset( + ConsumerKind::Consumer, + consumer_id, + committed_offset, + recovered_offset, + ); + guard.insert(consumer_id as usize, offset); } } - let loaded_group_offsets = load_partition_consumer_group_offsets( + let recovered_groups = load_partition_consumer_group_offsets( &consumer_group_offsets_path, stream_id, topic_id, partition_id, - )?; - let consumer_group_offsets = ConsumerGroupOffsets::with_capacity(loaded_group_offsets.len()); + ) + .await?; + let consumer_group_offsets = + ConsumerGroupOffsets::with_capacity(recovered_groups.entries.len()); { let guard = consumer_group_offsets.pin(); - for (group_id, offset) in loaded_group_offsets { + for (group_id, offset) in recovered_groups.entries { let recovered_offset = offset.offset.load(Ordering::Relaxed); if recovered_offset > offset_space_ceiling { warn!( @@ -241,42 +255,71 @@ pub fn configure_consumer_offsets( ); offset.offset.store(current_offset, Ordering::Relaxed); } + let committed_offset = offset.offset.load(Ordering::Relaxed); + partition.seed_recovered_consumer_offset( + ConsumerKind::ConsumerGroup, + u32::try_from(group_id.0).expect("recovered group id originated as u32"), + committed_offset, + recovered_offset, + ); guard.insert(group_id, offset); } } - // Offset files follow the topic's own `enforce_fsync`: they are part of the - // same partition's durability story, and the global knob they used to read - // is gone. - let enforce_fsync = partition - .runtime_options() - .enforce_fsync - .unwrap_or(iggy_common::DEFAULT_ENFORCE_FSYNC); + // Offset files have their own knob, not the topic's `enforce_fsync`: that + // one gates message and index writes, and syncing a 16-byte cursor on every + // commit costs milliseconds per commit for a file whose loss is a redelivery. partition.configure_consumer_offset_storage( - consumer_offsets_path, - consumer_group_offsets_path, + consumer_offsets_path.clone(), + consumer_group_offsets_path.clone(), consumer_offsets, consumer_group_offsets, - enforce_fsync, + config.partition.consumer_offset_enforce_fsync, ); + for consumer_id in recovered_consumers.stranded_ids { + if partition.seed_stranded_consumer_offset(ConsumerKind::Consumer, consumer_id) { + warn!(stream_id, topic_id, partition_id, consumer_id, path = %consumer_offsets_path, + "unloaded consumer offset file retains its capacity slot until updated or deleted"); + } + } + for group_id in recovered_groups.stranded_ids { + if partition.seed_stranded_consumer_offset(ConsumerKind::ConsumerGroup, group_id) { + warn!(stream_id, topic_id, partition_id, group_id, path = %consumer_group_offsets_path, + "unloaded group offset file retains its capacity slot until repaired or reclaimed"); + } + } + for kind in [ConsumerKind::Consumer, ConsumerKind::ConsumerGroup] { + let count = partition.occupied_consumer_offset_count(kind); + if count > config.partition.consumer_offsets_max { + warn!( + stream_id, + topic_id, + partition_id, + ?kind, + count, + limit = config.partition.consumer_offsets_max, + "recovered consumer offsets exceed the configured admission limit" + ); + } + } Ok(()) } -fn load_partition_consumer_offsets( +async fn load_partition_consumer_offsets( path: &str, consumer_kind: &'static str, stream_id: usize, topic_id: usize, partition_id: usize, -) -> Result, ServerError> { +) -> Result, ServerError> { if !Path::new(path).exists() { - return Ok(Vec::new()); + return Ok(RecoveredOffsets::default()); } - load_consumer_offsets(path).or_else(|source| { + load_consumer_offsets(path).await.or_else(|source| { if matches!(&source, IggyError::CannotReadConsumerOffsets(missing_path) if !Path::new(missing_path).exists()) { - return Ok(Vec::new()); + return Ok(RecoveredOffsets::default()); } Err(ServerError::ConsumerOffsetsLoad { @@ -290,20 +333,23 @@ fn load_partition_consumer_offsets( }) } -fn load_partition_consumer_group_offsets( +async fn load_partition_consumer_group_offsets( path: &str, stream_id: usize, topic_id: usize, partition_id: usize, -) -> Result, ServerError> { +) -> Result< + RecoveredOffsets<(iggy_common::ConsumerGroupId, iggy_common::ConsumerOffset)>, + ServerError, +> { if !Path::new(path).exists() { - return Ok(Vec::new()); + return Ok(RecoveredOffsets::default()); } - load_consumer_group_offsets(path).or_else(|source| { + load_consumer_group_offsets(path).await.or_else(|source| { if matches!(&source, IggyError::CannotReadConsumerOffsets(missing_path) if !Path::new(missing_path).exists()) { - return Ok(Vec::new()); + return Ok(RecoveredOffsets::default()); } Err(ServerError::ConsumerOffsetsLoad { @@ -816,6 +862,7 @@ async fn load_partition( config.partition.evicted_ring_bytes_max.as_bytes_u64(), ); partition.set_dedup_clients_max(config.partition.dedup_clients_max); + partition.set_consumer_offsets_max(config.partition.consumer_offsets_max); partition.set_offset_reservation_lease(config.partition.offset_reservation_lease); partition.set_partition_dir(partition_dir.clone()); // Before the hydrate: the durable record is keyed by incarnation, so a @@ -836,7 +883,7 @@ async fn load_partition( restore_partition_offsets(&mut partition, partitions_config, recovered_state.as_ref()).await?; let current_offset = partition.offset.load(Ordering::Acquire); - configure_consumer_offsets(&mut partition, config, namespace, current_offset)?; + configure_consumer_offsets(&mut partition, config, namespace, current_offset).await?; ensure_initial_segment(&mut partition, config, stream_id, topic_id, partition_id).await?; Ok(partition) @@ -1264,6 +1311,7 @@ pub async fn build_partition_fresh( config.partition.evicted_ring_bytes_max.as_bytes_u64(), ); partition.set_dedup_clients_max(config.partition.dedup_clients_max); + partition.set_consumer_offsets_max(config.partition.consumer_offsets_max); partition.set_offset_reservation_lease(config.partition.offset_reservation_lease); partition.set_partition_dir(partition_dir); // Fresh dirs read generation 0; a dir surviving from a crashed process @@ -1301,7 +1349,7 @@ pub async fn build_partition_fresh( let current_offset = partition.offset.load(Ordering::Acquire); - configure_consumer_offsets(&mut partition, config, namespace, current_offset)?; + configure_consumer_offsets(&mut partition, config, namespace, current_offset).await?; ensure_initial_segment(&mut partition, config, stream_id, topic_id, partition_id).await?; // Claim the first offset-reservation block HERE so no send ever pays the @@ -1508,6 +1556,7 @@ mod tests { messages_required_to_save: 1, size_of_messages_required_to_save: IggyByteSize::from(1024_u64), enforce_fsync: false, + consumer_offset_enforce_fsync: false, validate_checksum: true, segment_size: IggyByteSize::from(1_048_576_u64), preallocate_segments: false, diff --git a/core/server/src/partition_reconciler.rs b/core/server/src/partition_reconciler.rs index e35f623784..6ce661ee72 100644 --- a/core/server/src/partition_reconciler.rs +++ b/core/server/src/partition_reconciler.rs @@ -170,11 +170,17 @@ use ahash::{AHashMap, AHashSet}; use configs::server::ServerConfig; use consensus::{MetadataHandle, PartitionsHandle}; use futures::FutureExt; +use iggy_binary_protocol::requests::consumer_offsets::DeleteConsumerOffsetRequest; +use iggy_binary_protocol::{ + AckLevel, Command, Operation, ReplyHeader, RoutedRequestHeader, WireConsumer, WireEncode, + WireIdentifier, +}; use iggy_common::{ConsumerGroupId, IggyTimestamp}; +use message_bus::AUTO_COMMIT_CLIENT_ID; use message_bus::MessageBus; use metadata::impls::metadata::StreamsFrontend; use metadata::stm::stream::Partition; -use partitions::delete_persisted_offset; +use server_common::Message; use server_common::sharding::{IggyNamespace, ShardId}; use shard::MetadataSubmit; use shard::ReconcileOp; @@ -184,10 +190,11 @@ use std::cell::{Cell, RefCell}; use std::rc::Rc; use std::sync::Arc; use std::time::{Duration, Instant}; -use tracing::{debug, error, trace, warn}; +use tracing::{debug, error, trace}; const BACKOFF_BASE: Duration = Duration::from_secs(1); const BACKOFF_MAX: Duration = Duration::from_mins(1); +const GROUP_OFFSET_DELETES_PER_PASS: usize = 32; /// Consecutive same-cause failures before [`ReconcilerCtx::record_failure`] /// escalates to an operator-visible error (the backoff is capped, so @@ -229,6 +236,9 @@ pub struct ReconcilerCtx { /// `true` when the previous pass made no changes. Only then is a /// same-`revision` pass safe to skip. last_pass_noop: Cell, + group_offset_cleanup_inflight: Rc>>, + group_offset_cleanup_completed: Rc>, + last_group_offset_reconcile_epoch: Cell, } impl ReconcilerCtx { @@ -251,6 +261,9 @@ impl ReconcilerCtx { failure_state: RefCell::new(AHashMap::new()), last_revision: Cell::new(None), last_pass_noop: Cell::new(false), + group_offset_cleanup_inflight: Rc::new(RefCell::new(AHashSet::new())), + group_offset_cleanup_completed: Rc::new(Cell::new(0)), + last_group_offset_reconcile_epoch: Cell::new(0), } } @@ -383,9 +396,12 @@ struct PassCounters { backoff_skipped: usize, /// Stale incarnations (slab-key reuse) torn down for rebuild. stale: usize, - /// Consumer-group offsets reclaimed for groups deleted while their topic - /// survived (a bare `DeleteConsumerGroup`, not a topic/stream delete). - cg_offsets_purged: usize, + /// Group-offset deletes successfully handed to the pump this pass. + cg_offsets_submitted: usize, + /// Successful replicated deletes reported by detached completion tasks. + cg_offsets_completed: usize, + /// Group-offset deletes refused by the inbox and needing another pass. + cg_offsets_pending: usize, /// Committed delete watermarks not yet fully enforced on local segments. /// Counted so the pass does not arm the fast-skip: the pump can be /// blocked by a consumer barrier or by a rejoin whose offsets land via @@ -423,7 +439,9 @@ impl PassCounters { + self.removed_routed + self.backoff_skipped + self.stale - + self.cg_offsets_purged + + self.cg_offsets_submitted + + self.cg_offsets_completed + + self.cg_offsets_pending + self.trims_pending + self.purges_staged + self.deferred @@ -439,6 +457,14 @@ impl PassCounters { async fn reconcile_once(ctx: &ReconcilerCtx) -> bool { let shard_id = ctx.shard.id; let revision = current_revision(ctx); + // Snapshotted with `revision` before the pass, so a partition that arms + // itself during one of the awaits below bumps past this value and forces + // the next pass instead of being absorbed by an end-of-pass read. + let group_offset_epoch = ctx + .shard + .plane + .partitions() + .consumer_group_offsets_reconcile_epoch(); // Cooperative-revocation completion runs every tick, before the fast-skip: // a timeout fires on wall-clock and a drain on partition-offset state, and @@ -464,7 +490,9 @@ async fn reconcile_once(ctx: &ReconcilerCtx) -> bool { if ctx.last_revision.get() == Some(revision) && ctx.last_pass_noop.get() && ctx.failure_state.borrow().is_empty() + && ctx.group_offset_cleanup_completed.get() == 0 && !ctx.shard.has_parked_partition_frames() + && ctx.last_group_offset_reconcile_epoch.get() == group_offset_epoch { trace!( shard = shard_id, @@ -475,12 +503,15 @@ async fn reconcile_once(ctx: &ReconcilerCtx) -> bool { let target = snapshot_target_namespaces(ctx); let target_set: AHashSet = target.iter().map(|partition| partition.ns).collect(); - let mut counters = PassCounters::default(); + let mut counters = PassCounters { + cg_offsets_completed: ctx.group_offset_cleanup_completed.replace(0), + ..PassCounters::default() + }; reconcile_additions(ctx, target, &mut counters).await; reconcile_removals(ctx, &target_set, &mut counters).await; reconcile_parked_frames(ctx, &mut counters); - reconcile_consumer_group_offsets(ctx, &mut counters).await; + reconcile_consumer_group_offsets(ctx, &mut counters); reconcile_segment_truncations(ctx, &mut counters); reconcile_partition_purges(ctx, &mut counters); @@ -494,6 +525,8 @@ async fn reconcile_once(ctx: &ReconcilerCtx) -> bool { // still runs even though `revision` did not change. let did_work = counters.total() > 0; ctx.last_revision.set(Some(revision)); + ctx.last_group_offset_reconcile_epoch + .set(group_offset_epoch); ctx.last_pass_noop.set(!did_work); if did_work { @@ -511,6 +544,7 @@ async fn reconcile_once(ctx: &ReconcilerCtx) -> bool { parked_reclaimed = counters.parked_reclaimed, purges_staged = counters.purges_staged, trims_pending = counters.trims_pending, + cg_offsets_pending = counters.cg_offsets_pending, "partition reconciler pass complete" ); } else { @@ -960,47 +994,118 @@ async fn tear_down_owned_partition( counters.removed_local += 1; } -/// Reclaim consumer-group offsets left behind by a `DeleteConsumerGroup` whose -/// topic still exists (a topic/stream delete already drops the whole partition -/// directory, offsets included). For each owned partition, any stored -/// consumer-group offset whose group id is no longer present in the topic's -/// committed metadata is removed (in-memory entry + persisted file). Monotonic, -/// never-reused group ids make this purely reclamation -- a recreated group -/// gets a fresh id and never reads a dead group's offset -- so it is safe to do -/// lazily on the reconcile pass rather than synchronously on delete. -async fn reconcile_consumer_group_offsets(ctx: &ReconcilerCtx, counters: &mut PassCounters) { +/// Reclaim deleted groups through ordered offset deletes. Replicas must see +/// each delete before a replacement store can reuse its durable slot. +fn reconcile_consumer_group_offsets(ctx: &ReconcilerCtx, counters: &mut PassCounters) { let live_groups = snapshot_topic_live_groups(ctx); let partitions = ctx.shard.plane.partitions(); let owned: Vec = partitions.namespaces().copied().collect(); - for ns in owned { - let live = live_groups.get(&(ns.stream_id(), ns.topic_id())); - // Take the in-memory removes + owned unlink paths under a closure-scoped - // borrow that cannot escape into the await below. Holding a raw - // `&IggyPartition` across `delete_persisted_offset().await` would let the - // pump task realloc the partitions vec underneath us (a UAF). - let paths = partitions.with_partition(&ns, |partition| { - partition.reclaim_dead_group_offsets(|group_id| { - live.is_some_and(|set| set.contains(&group_id)) - }) - }); - let Some(paths) = paths else { + for namespace in owned { + if ctx + .group_offset_cleanup_inflight + .borrow() + .contains(&namespace) + { + continue; + } + let Some((next_group_id, live)) = + live_groups.get(&(namespace.stream_id(), namespace.topic_id())) + else { continue; }; - for path in paths { - if let Err(err) = delete_persisted_offset(&path).await { - warn!( - shard = ctx.shard.id, - ns_raw = ns.inner(), - error = %err, - "reconciler failed to reclaim deleted consumer-group offset" - ); - continue; + let dead = partitions + .with_partition(&namespace, |partition| { + partition.dead_consumer_group_offset_ids(|group_id| { + // A lagging metadata replica cannot prove a group deleted + // until it has applied the allocation of that group's id. + group_id >= *next_group_id || live.contains(&group_id) + }) + }) + .unwrap_or_default(); + // Bound work per pass so a historical directory cannot monopolize the + // reconciler. Unprocessed keys keep the partition's dirty flag armed. + let mut tickets = Vec::with_capacity(dead.len().min(GROUP_OFFSET_DELETES_PER_PASS)); + for consumer_id in dead.into_iter().take(GROUP_OFFSET_DELETES_PER_PASS) { + let request = group_offset_delete_request(namespace, consumer_id); + if let Ok(ticket) = ctx.shard.partition_submit(namespace, request) { + counters.cg_offsets_submitted += 1; + tickets.push(ticket); + } else { + counters.cg_offsets_pending += 1; } - counters.cg_offsets_purged += 1; } + if tickets.is_empty() { + continue; + } + ctx.group_offset_cleanup_inflight + .borrow_mut() + .insert(namespace); + let inflight = Rc::clone(&ctx.group_offset_cleanup_inflight); + let completed = Rc::clone(&ctx.group_offset_cleanup_completed); + let shard = Rc::clone(&ctx.shard); + shard.bus.clone().spawn(async move { + let replies = futures::future::join_all( + tickets + .into_iter() + .map(|ticket| shard.await_partition_submit(ticket)), + ) + .await; + let count = replies + .into_iter() + .flatten() + .filter(|reply| { + reply + .as_slice() + .get(..size_of::()) + .and_then(|bytes| { + bytemuck::checked::try_from_bytes::(bytes).ok() + }) + .is_some_and(|header| header.status == 0) + }) + .count(); + completed.set(completed.get() + count); + inflight.borrow_mut().remove(&namespace); + shard + .plane + .partitions() + .note_consumer_group_offsets_reconcile_needed(); + }); } } +fn group_offset_delete_request( + namespace: IggyNamespace, + consumer_id: u32, +) -> Message { + let body = DeleteConsumerOffsetRequest { + consumer: WireConsumer::consumer_group(WireIdentifier::Numeric(consumer_id)), + stream_id: WireIdentifier::Numeric( + u32::try_from(namespace.stream_id()).expect("stream id fits u32"), + ), + topic_id: WireIdentifier::Numeric( + u32::try_from(namespace.topic_id()).expect("topic id fits u32"), + ), + partition_id: Some(u32::try_from(namespace.partition_id()).expect("partition id fits u32")), + ack: AckLevel::Quorum, + } + .to_bytes(); + let size = size_of::() + body.len(); + let mut request = Message::::new(size); + request.as_mut_slice()[size_of::()..].copy_from_slice(&body); + request.transmute_header(|_, header: &mut RoutedRequestHeader| { + *header = RoutedRequestHeader { + command: Command::Request, + operation: Operation::DeleteConsumerOffset, + size: u32::try_from(size).expect("offset request fits u32"), + client: AUTO_COMMIT_CLIENT_ID, + session: 1, + request: 1, + group: namespace.inner(), + ..Default::default() + }; + }) +} + /// Complete cooperative consumer-group revocations whose source member has /// drained the partition (`committed >= last_polled`), was never polled, or /// timed out. Reads pending revocations from metadata + local partition offset @@ -1068,21 +1173,23 @@ fn reconcile_pending_revocations(ctx: &ReconcilerCtx) { /// id (the store path is rewritten to it; the read path resolves it), so the /// live-set carries those ids too -- otherwise the reconciler would treat every /// live offset as orphaned and purge it. -fn snapshot_topic_live_groups(ctx: &ReconcilerCtx) -> AHashMap<(usize, usize), AHashSet> { +fn snapshot_topic_live_groups( + ctx: &ReconcilerCtx, +) -> AHashMap<(usize, usize), (u64, AHashSet)> { ctx.shard.plane.metadata().mux_stm.streams().read(|inner| { - let mut map: AHashMap<(usize, usize), AHashSet> = AHashMap::new(); + let mut map = AHashMap::new(); for (_, stream) in &inner.items { for (topic_id, topic) in &stream.topics { - if topic.consumer_groups.is_empty() { - continue; - } map.insert( (stream.id, topic_id), - topic - .consumer_groups - .values() - .map(|group| group.id) - .collect(), + ( + topic.next_consumer_group_id, + topic + .consumer_groups + .values() + .map(|group| group.id) + .collect(), + ), ); } } @@ -1273,8 +1380,9 @@ pub fn install_tick_handler(shard: &Rc, wake_tx: WakeTx) { #[cfg(test)] mod tests { use super::{ - FailureCause, FailureRecord, ReconcilerCtx, build_partition_fresh, - delete_partitions_from_disk, fetch_partition_stats, reconcile_once, + FailureCause, FailureRecord, PassCounters, ReconcilerCtx, build_partition_fresh, + current_revision, delete_partitions_from_disk, fetch_partition_stats, + reconcile_consumer_group_offsets, reconcile_once, }; use configs::server::{ServerConfig, ServerSystemConfig}; use consensus::{MetadataHandle, PartitionsHandle}; @@ -1621,6 +1729,7 @@ mod tests { messages_required_to_save: 1, size_of_messages_required_to_save: iggy_common::IggyByteSize::from(1024_u64), enforce_fsync: false, + consumer_offset_enforce_fsync: false, validate_checksum: true, segment_size: iggy_common::IggyByteSize::from(iggy_common::DEFAULT_SEGMENT_SIZE), preallocate_segments: false, @@ -2410,6 +2519,21 @@ mod tests { "unchanged revision after convergence must fast-skip the diff" ); + let revision = current_revision(&ctx); + shard + .plane + .partitions() + .note_consumer_group_offsets_reconcile_needed(); + assert_eq!( + current_revision(&ctx), + revision, + "partition-local offset work does not change metadata revision" + ); + assert!( + reconcile_once(&ctx).await, + "the shard offset epoch must defeat the O(1) fast-skip" + ); + // A new partition-shaping commit bumps the revision → next pass runs. seed_topic( &shard.plane.metadata().mux_stm, @@ -3181,11 +3305,146 @@ mod tests { ); } - /// A bare `DeleteConsumerGroup` (topic survives) leaves the group's offsets - /// on the partition. The reconciler must reclaim a deleted group's offset - /// while leaving a still-live group's offset untouched. + /// Failed delivery must leave the offset and its quota slot intact for a + /// later replicated delete. This fixture deliberately has no running pump. #[compio::test] - async fn reconcile_reclaims_offsets_of_deleted_consumer_group() { + async fn given_pending_group_cleanup_when_another_topic_arrives_should_reconcile_without_waiting() + { + let tmp = TempDir::new().unwrap(); + let config = test_config(&tmp); + let mux = TestMux::default(); + seed_stream(&mux, 1, "cleanup-stream"); + seed_topic(&mux, 2, 0, "cleanup-topic", vec![assignment(0, 1)]); + seed_create_consumer_group(&mux, 3, 0, 0, "deleted"); + seed_delete_consumer_group(&mux, 4, 0, 0, 0); + let (shard, _inbox) = build_test_shard_with_inbox(0, &config, mux, 32); + let ctx = make_ctx(Rc::clone(&shard), 1, Rc::new(config)); + reconcile_pass(&ctx).await; + let namespace = IggyNamespace::new(0, 0, 0); + shard + .plane + .partitions() + .with_partition(&namespace, |partition| { + partition.seed_recovered_consumer_offset( + iggy_common::ConsumerKind::ConsumerGroup, + 0, + 0, + 0, + ); + partition.consumer_group_offsets.pin().insert( + iggy_common::ConsumerGroupId(0), + iggy_common::ConsumerOffset::new( + iggy_common::ConsumerKind::ConsumerGroup, + 0, + 0, + String::new(), + ), + ); + }); + reconcile_consumer_group_offsets(&ctx, &mut PassCounters::default()); + assert!( + ctx.group_offset_cleanup_inflight + .borrow() + .contains(&namespace) + ); + seed_topic( + &shard.plane.metadata().mux_stm, + 5, + 0, + "unrelated-topic", + vec![assignment(0, 1)], + ); + reconcile_pass(&ctx).await; + assert!( + shard + .plane + .partitions() + .contains(&IggyNamespace::new(0, 1, 0)), + "an unanswered cleanup must not block partition creation" + ); + assert!( + ctx.group_offset_cleanup_inflight + .borrow() + .contains(&namespace) + ); + } + + #[compio::test] + async fn given_lagging_group_metadata_when_reconciling_should_wait_for_proven_deletion() { + let tmp = TempDir::new().unwrap(); + let config = test_config(&tmp); + let mux = TestMux::default(); + seed_stream(&mux, 1, "stream"); + seed_topic(&mux, 2, 0, "topic", vec![assignment(0, 1)]); + let (shard, _inbox) = build_test_shard_with_inbox(0, &config, mux, 32); + let ctx = make_ctx(Rc::clone(&shard), 1, Rc::new(config)); + reconcile_pass(&ctx).await; + let namespace = IggyNamespace::new(0, 0, 0); + shard + .plane + .partitions() + .with_partition(&namespace, |partition| { + partition.seed_recovered_consumer_offset( + iggy_common::ConsumerKind::ConsumerGroup, + 0, + 0, + 0, + ); + partition.consumer_group_offsets.pin().insert( + iggy_common::ConsumerGroupId(0), + iggy_common::ConsumerOffset::new( + iggy_common::ConsumerKind::ConsumerGroup, + 0, + 0, + String::new(), + ), + ); + }); + let mut counters = PassCounters::default(); + reconcile_consumer_group_offsets(&ctx, &mut counters); + assert_eq!(counters.cg_offsets_submitted, 0); + let stm = &shard.plane.metadata().mux_stm; + seed_create_consumer_group(stm, 3, 0, 0, "not-replayed-yet"); + reconcile_consumer_group_offsets(&ctx, &mut counters); + assert_eq!(counters.cg_offsets_submitted, 0); + seed_delete_consumer_group(stm, 4, 0, 0, 0); + reconcile_consumer_group_offsets(&ctx, &mut counters); + assert_eq!(counters.cg_offsets_submitted, 1); + } + + #[compio::test] + async fn given_missing_topic_snapshot_when_group_offset_exists_should_not_submit_delete() { + let tmp = TempDir::new().unwrap(); + let config = test_config(&tmp); + let mux = TestMux::default(); + seed_stream(&mux, 1, "stream"); + seed_topic(&mux, 2, 0, "topic", vec![assignment(0, 1)]); + let (shard, _inbox) = build_test_shard_with_inbox(0, &config, mux, 32); + let ctx = make_ctx(Rc::clone(&shard), 1, Rc::new(config)); + reconcile_pass(&ctx).await; + shard + .plane + .partitions() + .with_partition(&IggyNamespace::new(0, 0, 0), |partition| { + partition.consumer_group_offsets.pin().insert( + iggy_common::ConsumerGroupId(0), + iggy_common::ConsumerOffset::new( + iggy_common::ConsumerKind::ConsumerGroup, + 0, + 0, + String::new(), + ), + ); + }); + seed_delete_topic(&shard.plane.metadata().mux_stm, 3, 0, 0); + let mut counters = PassCounters::default(); + reconcile_consumer_group_offsets(&ctx, &mut counters); + assert_eq!(counters.cg_offsets_submitted, 0); + assert!(ctx.group_offset_cleanup_inflight.borrow().is_empty()); + } + + #[compio::test] + async fn given_deleted_group_when_cleanup_submit_fails_should_preserve_state_for_retry() { use iggy_common::{ConsumerGroupId, ConsumerKind, ConsumerOffset}; let tmp = TempDir::new().expect("tempdir for system path"); @@ -3213,6 +3472,8 @@ mod tests { { let partitions = shard.plane.partitions(); let partition = partitions.get_by_ns(&ns).expect("partition materialised"); + partition.seed_recovered_consumer_offset(ConsumerKind::ConsumerGroup, dead_key, 7, 7); + partition.seed_recovered_consumer_offset(ConsumerKind::ConsumerGroup, live_key, 9, 9); partition.consumer_group_offsets.pin().insert( ConsumerGroupId(dead_key as usize), ConsumerOffset::new(ConsumerKind::ConsumerGroup, dead_key, 7, String::new()), @@ -3235,9 +3496,14 @@ mod tests { ids.sort_unstable(); assert_eq!( ids, - vec![u64::from(live_key)], - "deleted group's offset reclaimed; live group's offset retained" + vec![u64::from(dead_key), u64::from(live_key)], + "failed submission must not unlink or remove either offset" + ); + assert_eq!( + partition.dead_consumer_group_offset_ids(|id| id == u64::from(live_key)), + vec![dead_key] ); + assert!(!ctx.last_pass_noop.get(), "failed cleanup must be retried"); } /// A partition-count change must re-run consumer-group assignment: a new diff --git a/core/server/src/responses.rs b/core/server/src/responses.rs index 6cea586424..cd3460a331 100644 --- a/core/server/src/responses.rs +++ b/core/server/src/responses.rs @@ -90,6 +90,7 @@ use journal::superblock::SuperblockStore; use journal::{Journal, JournalHandle}; use message_bus::BusMessage; use metadata::impls::metadata::StreamsFrontend; +use metadata::stm::stream::Streams; use partitions::{Fragment, PollFragments}; use server_common::iobuf::{Frozen, Owned}; use server_common::send_messages; @@ -259,6 +260,7 @@ where /// touch the offset of a partition it currently owns. `Ok` for individual /// consumers (no fence) and for owned group partitions; `Err` otherwise so a /// stale client re-syncs instead of corrupting the shared group offset. +#[allow(clippy::cast_possible_truncation)] fn fence_group_offset( shard: &Rc>, consumer: &WireConsumer, @@ -278,12 +280,8 @@ where return Ok(()); } let partition_id = partition_id.ok_or(IggyError::InvalidIdentifier)?; - #[allow(clippy::cast_possible_truncation)] - shard - .plane - .metadata() - .mux_stm - .streams() + let streams = shard.plane.metadata().mux_stm.streams(); + let Some(_) = streams // Commit fence: allow a pending-revoked partition (the source commits it // to drain the cooperative handoff), so `require_pollable = false`. .consumer_group_fence( @@ -294,11 +292,46 @@ where partition_id, false, ) - .map(|_| ()) - .ok_or(IggyError::ConsumerGroupPartitionNotOwned( + else { + resolve_offset_group_id(streams, stream_id, topic_id, &consumer.id)?; + return Err(IggyError::ConsumerGroupPartitionNotOwned( client_id as u32, partition_id, - )) + )); + }; + Ok(()) +} + +pub fn resolve_offset_group_id( + streams: &Streams, + stream_id: &WireIdentifier, + topic_id: &WireIdentifier, + group: &WireIdentifier, +) -> Result { + streams + .resolve_consumer_group_id(stream_id, topic_id, group) + .ok_or_else(|| { + if streams + .topic_partitions_count(stream_id, topic_id) + .is_some() + { + missing_consumer_group_error(group, topic_id) + } else { + IggyError::ResourceNotFound(String::new()) + } + }) +} + +pub fn missing_consumer_group_error(group: &WireIdentifier, topic: &WireIdentifier) -> IggyError { + let topic = wire_identifier_for_display(topic); + match group { + WireIdentifier::Numeric(_) => { + IggyError::ConsumerGroupIdNotFound(wire_identifier_for_display(group), topic) + } + WireIdentifier::String(name) => { + IggyError::ConsumerGroupNameNotFound(name.as_str().to_owned(), topic) + } + } } /// Fence a consumer-group offset op then resolve its target partition diff --git a/core/shard/src/lib.rs b/core/shard/src/lib.rs index 008db6fbc8..ac4c2b0185 100644 --- a/core/shard/src/lib.rs +++ b/core/shard/src/lib.rs @@ -49,7 +49,7 @@ use iggy_binary_protocol::{ #[cfg(feature = "simulator")] use iggy_common::PartitionStats; use iggy_common::variadic; -use iggy_common::{IggyError, IggyExpiry, IggyTimestamp}; +use iggy_common::{ConsumerKind, IggyError, IggyExpiry, IggyTimestamp}; use journal::superblock::{PingPongSuperblock, SuperblockStore}; use journal::{Journal, JournalHandle}; use message_bus::client_listener::RequestHandler; @@ -114,6 +114,7 @@ where pub struct PartitionMaterialisation { epoch: u64, created_view: u32, + consumer_offsets_max: usize, } #[cfg(feature = "simulator")] @@ -123,8 +124,15 @@ impl PartitionMaterialisation { Self { epoch, created_view, + consumer_offsets_max: partitions::DEFAULT_CONSUMER_OFFSETS_MAX, } } + + #[must_use] + pub const fn with_consumer_offsets_max(mut self, consumer_offsets_max: usize) -> Self { + self.consumer_offsets_max = consumer_offsets_max; + self + } } /// Replica id + count bundle. @@ -358,6 +366,15 @@ pub enum PartitionReadReply { stored: Option, current_offset: u64, }, + /// The read was refused and returns no messages, even where fragments were + /// already gathered. For a poll with `auto_commit`, `TooManyConsumerOffsets` + /// when the poll needed a new offset key past `[partition] + /// consumer_offsets_max`, and `TransientNotAccepted` when the auto-commit + /// could not be submitted: the owning shard's inbox was full, or the + /// partition changed primary or incarnation during the read. Transient + /// refusal permits re-polling. A capacity refusal needs a slot reclaimed + /// or a higher configured limit before a new key can succeed. + Rejected(IggyError), /// Reply to [`PartitionRead::GroupOffsetState`]: the group's last-polled and /// committed offsets on this partition (each `None` if absent). GroupOffsetState { @@ -735,6 +752,12 @@ pub enum LifecycleFrame { request: Message, reply: Sender>>, }, + /// Local auto-commit submission. The guard travels with the frame so an + /// inbox drop or admission refusal releases its provisional key directly. + AutoCommitSubmit { + request: Message, + reservation: partitions::AutoCommitReservation, + }, /// Shard 0 broadcasts after a partition-shaped metadata commit; wakes /// the per-shard reconciler. No payload: reconciler re-reads target /// state. Drops covered by the periodic safety tick. @@ -2038,6 +2061,32 @@ where }) } + /// Submit an auto-commit back to the partition-owning shard's pump. + /// + /// # Errors + /// Returns a refusal if the local inbox cannot accept the frame. + pub fn submit_auto_commit_offset( + &self, + request: Message, + reservation: partitions::AutoCommitReservation, + ) -> Result<(), PartitionSubmitRefused> { + let frame = ShardFrame::lifecycle(LifecycleFrame::AutoCommitSubmit { + request, + reservation, + }); + let sender = self + .senders + .get(usize::from(self.id)) + .ok_or(PartitionSubmitRefused)?; + sender.try_send(frame).map_err(|error| { + self.metrics.record_frame_drop( + crate::metrics::frame_drop_variant::PARTITION_AUTO_COMMIT, + crate::coordinator::classify_try_send_err(&error), + ); + PartitionSubmitRefused + }) + } + /// Wait out a submitted write's committed reply. /// /// `None` = reply channel dropped before a reply (view-change reset, park @@ -4012,6 +4061,7 @@ where let PartitionMaterialisation { epoch, created_view, + consumer_offsets_max, } = materialisation; let partitions = self.plane.partitions(); if partitions.contains(&namespace) { @@ -4070,8 +4120,9 @@ where stats, consensus, partitions.config().segment_size, - partitions.config().enforce_fsync, + partitions.config().consumer_offset_enforce_fsync, ); + partition.set_consumer_offsets_max(consumer_offsets_max); if let Some(superblock) = superblock { partition.set_superblock(superblock, recovered_state.as_ref()); } @@ -7204,6 +7255,15 @@ where let Some(partition) = partitions.get_mut_by_ns(&namespace) else { continue; }; + partition.retry_consumer_offset_reservations(); + if partition.queued_requests_ready() { + if walks < PARTITION_WALKS_PER_TICK_MAX { + walks += 1; + partition.resume_queued_requests().await; + } else { + walk_cursor.get_or_insert(namespace); + } + } let consensus_normal = partition.consensus().is_normal(); let consensus_view = partition.consensus().view(); let commit_min = partition.consensus().commit_min(); @@ -7550,6 +7610,21 @@ where // retires whatever completed. self.partition_repairs_inflight .set(repairs_live + repair_arms); + // Republished per sweep like the repair count: a stranded key is + // permanent until its own store or delete succeeds, so a gauge that + // never falls is the operator's only signal. + let mut stranded = [0usize; 2]; + for namespace in partitions.namespaces() { + if let Some(partition) = partitions.get_by_ns(namespace) { + stranded[0] += partition.stranded_consumer_offset_count(ConsumerKind::Consumer); + stranded[1] += + partition.stranded_consumer_offset_count(ConsumerKind::ConsumerGroup); + } + } + self.metrics + .set_consumer_offsets_stranded(ConsumerKind::Consumer, stranded[0]); + self.metrics + .set_consumer_offsets_stranded(ConsumerKind::ConsumerGroup, stranded[1]); fatal } @@ -8445,24 +8520,16 @@ where if !commit_lag && head <= commit_to_op { return false; } - let missing_suffix = partition_missing_suffix(partition); - if !commit_lag && !missing_suffix { + let missing_suffix = partition_missing_suffix_through(partition); + // Fetch the adopted suffix even while committed operations lag. Later + // live prepares can advance the head while an adopted body is missing. + let Some(fetch_to_op) = + partition_repair_fetch_to_op(consensus.commit_min(), commit_to_op, missing_suffix) + else { return false; - } + }; let nonce = iggy_common::random_id::get_uuid(); let from_op = consensus.commit_min() + 1; - // Capping at `commit_to_op` while a commit lag stands avoids - // re-asking for `(commit_min, commit_to_op]`, the committed prefix - // this replica already holds. But a restarted node has a commit lag - // by construction, and if it also adopted a StartView suffix, the - // missing bodies sit above `commit_to_op`, not within it -- so the - // cap only holds when no suffix is missing; otherwise it must widen - // to `head` to ever reach those bodies. - let fetch_to_op = if commit_lag && !missing_suffix { - commit_to_op - } else { - head - }; let cluster = consensus.cluster(); let self_id = consensus.replica(); let namespace = consensus.group(); @@ -9002,7 +9069,7 @@ where partition.note_transfer_progress(); partition.note_transfer_installed(); partition.transfer_rearm = None; - if outcome.offsets_written { + if outcome.purge_generation_recorded { tracing::info!( shard = self.id, namespace_raw = namespace, @@ -9017,8 +9084,8 @@ where shard = self.id, namespace_raw = namespace, applied_commit_op = outcome.applied_commit_op, - "partition state transfer landed WITHOUT fully written consumer \ - offsets; the next offset commit rewrites the files" + "partition state transfer landed without a durable purge generation. \ + A restart may repeat the purge and transfer" ); } partition.commit_journal(&config).await; @@ -10067,7 +10134,7 @@ struct GapProbe { /// bodies never arrived. Its own recovery shape, disjoint from the lag /// below the frontier: the group cannot gather quorum for that suffix until /// the bodies land, and the only other site that notices is the single - /// `on_start_view` edge that adopted them. See [`partition_missing_suffix`]. + /// `on_start_view` edge that adopted them. See [`partition_missing_suffix_through`]. /// /// Always `false` on a metadata probe: the shape it names is read off the /// partition's own journal window, and the metadata plane's equivalent is @@ -10259,8 +10326,16 @@ fn rotate_sweep_to_cursor(namespaces: &mut [IggyNamespace], cursor: Option, +) -> Option { + (commit_min < commit_max || missing_suffix.is_some()) + .then(|| missing_suffix.unwrap_or(commit_max)) +} + +/// Highest adopted suffix op whose bodies are not all present above `commit_max`. /// /// The shape `maybe_request_partition_repair` widens its window for, read here /// so the sweep's detector and the arm agree by construction. A backup that @@ -10271,8 +10346,9 @@ fn rotate_sweep_to_cursor(namespaces: &mut [IggyNamespace], cursor: Option(partition: &IggyPartition) -> bool +/// group that has both pays the header-vec walk. Later live prepares can raise +/// the sequencer without extending the adopted canonical header list. +fn partition_missing_suffix_through(partition: &IggyPartition) -> Option where B: MessageBus, SB: SuperblockStore, @@ -10281,18 +10357,24 @@ where let commit_max = consensus.commit_max(); let head = consensus.sequencer().current_sequence(); if head <= commit_max { - return false; + return None; } - let canonical_suffix = consensus - .with_pending_view_log(|pending| pending_covers_suffix(pending, commit_max, head)) - .unwrap_or(false); - canonical_suffix - && !partition - .log - .journal() - .inner - .repaired_window_shape(commit_max, head) - .complete + let adopted_head = consensus + .with_pending_view_log(|pending| adopted_suffix_head(pending, commit_max, head)) + .flatten()?; + (!partition + .log + .journal() + .inner + .repaired_window_shape(commit_max, adopted_head) + .complete) + .then_some(adopted_head) +} + +fn adopted_suffix_head(pending: &MergedLog, commit_max: u64, current_head: u64) -> Option { + let adopted_head = pending.op_head.min(current_head); + (adopted_head > commit_max && pending_covers_suffix(pending, commit_max, adopted_head)) + .then_some(adopted_head) } /// Read the gap probe off a live partition. @@ -10324,8 +10406,10 @@ where // Same discipline, one guard deeper: the suffix test walks the header vec, // so it runs only for a group that HAS an unfinished suffix and already // owes nothing else. - let missing_suffix = - normal && !transferring && !recovery_owned && partition_missing_suffix(partition); + let missing_suffix = normal + && !transferring + && !recovery_owned + && partition_missing_suffix_through(partition).is_some(); GapProbe { normal, transferring, @@ -10341,9 +10425,11 @@ where /// suffix `(commit_max, head]`, in descending order. Only this canonical list /// makes fetching bodies above the commit point safe. fn pending_covers_suffix(pending: &MergedLog, commit_max: u64, head: u64) -> bool { - if head <= commit_max || pending.commit_max != commit_max || pending.op_head != head { + if head <= commit_max || pending.commit_max > commit_max || pending.op_head != head { return false; } + // Live commits can advance inside an adopted suffix. Its remaining + // canonical headers still authorize repair above the new commit point. let mut expected = head; for header in pending .headers @@ -11129,7 +11215,10 @@ mod repair_scope_tests { use iggy_binary_protocol::{Command, PrepareHeader}; - use super::{MergedLog, pending_covers_suffix, repair_op_in_scope, repair_serve_ceiling}; + use super::{ + MergedLog, adopted_suffix_head, pending_covers_suffix, repair_op_in_scope, + repair_serve_ceiling, + }; fn header(op: u64) -> PrepareHeader { PrepareHeader { @@ -11201,6 +11290,29 @@ mod repair_scope_tests { assert_eq!(repair_serve_ceiling(u64::MAX, 120, 90), 120); } + #[test] + fn given_an_adopted_suffix_when_live_head_advances_should_preserve_its_repair_boundary() { + let pending = parked(); + assert_eq!(adopted_suffix_head(&pending, 98, 100), Some(100)); + assert_eq!(adopted_suffix_head(&pending, 98, 101), Some(100)); + assert_eq!(adopted_suffix_head(&pending, 99, 101), Some(100)); + assert_eq!(adopted_suffix_head(&pending, 100, 101), None); + let suffix = adopted_suffix_head(&pending, 98, 101); + assert_eq!( + super::partition_repair_fetch_to_op(0, 98, suffix), + Some(100) + ); + assert_eq!( + super::partition_repair_fetch_to_op(98, 98, suffix), + Some(100) + ); + assert_eq!(super::partition_repair_fetch_to_op(0, 98, None), Some(98)); + assert_eq!(super::partition_repair_fetch_to_op(98, 98, None), None); + let mut missing = pending; + missing.headers.retain(|header| header.op != 99); + assert_eq!(adopted_suffix_head(&missing, 98, 101), None); + } + #[test] fn given_a_parked_view_when_fetching_above_commit_should_require_dense_canonical_suffix() { let pending = parked(); @@ -11211,7 +11323,7 @@ mod repair_scope_tests { assert!(!pending_covers_suffix(&missing, 98, 100)); let mut wrong_frontier = pending; - wrong_frontier.commit_max = 97; + wrong_frontier.commit_max = 99; assert!(!pending_covers_suffix(&wrong_frontier, 98, 100)); } } diff --git a/core/shard/src/metrics.rs b/core/shard/src/metrics.rs index 2d9535c25e..b5041379d7 100644 --- a/core/shard/src/metrics.rs +++ b/core/shard/src/metrics.rs @@ -39,9 +39,12 @@ use prometheus_client::encoding::EncodeLabelSet; use prometheus_client::metrics::counter::Counter; use prometheus_client::metrics::family::Family; +use prometheus_client::metrics::gauge::Gauge; use prometheus_client::registry::Registry; use std::sync::{Arc, OnceLock}; +use iggy_common::ConsumerKind; + /// Label for `frame_drops_total`. /// /// `variant` describes the dropped frame class; `reason` is `"full"` or @@ -62,6 +65,11 @@ pub struct FrameDropLabel { pub reason: &'static str, } +#[derive(Clone, Hash, Eq, PartialEq, EncodeLabelSet, Debug)] +pub struct ConsumerOffsetKindLabel { + pub kind: &'static str, +} + /// Variant labels used in `frame_drops_total`. Exposed as constants to /// catch typos at compile time and to keep the cardinality bounded. /// @@ -99,6 +107,12 @@ pub mod frame_drop_variant { /// dropped; the shard-0 deadline expiry recovers the slot / pending /// entry, so this stays informational. pub const REPLICA_HANDSHAKE_ACK: &str = "replica_handshake_ack"; + /// A poll's auto-commit submit refused by the owning shard's own inbox. + /// + /// Its own series, not `PARTITION`: the poll is answered with a retriable + /// status and no frame of the client's was dropped, so counting it with + /// shed frames would read as a routing loss. + pub const PARTITION_AUTO_COMMIT: &str = "partition_auto_commit"; } /// Reason labels used in `frame_drops_total`. @@ -147,7 +161,7 @@ pub mod frame_drop_reason { // pair enters the `Family` (and therefore the scrape) the first time a drop // site actually produces it, so the unreachable corners of the 7 x 9 cross // product never appear as permanent zero-valued series. -const VARIANT_COUNT: usize = 7; +const VARIANT_COUNT: usize = 8; const REASON_COUNT: usize = 11; const VARIANTS: [&str; VARIANT_COUNT] = [ @@ -158,6 +172,7 @@ const VARIANTS: [&str; VARIANT_COUNT] = [ frame_drop_variant::FORWARD_REPLICA_SEND, frame_drop_variant::METADATA_COMMIT_TICK, frame_drop_variant::REPLICA_HANDSHAKE_ACK, + frame_drop_variant::PARTITION_AUTO_COMMIT, ]; const REASONS: [&str; REASON_COUNT] = [ @@ -182,6 +197,13 @@ fn reason_index(s: &str) -> Option { REASONS.iter().position(|r| *r == s) } +const fn consumer_kind_index(kind: ConsumerKind) -> usize { + match kind { + ConsumerKind::Consumer => 0, + ConsumerKind::ConsumerGroup => 1, + } +} + /// Per-shard metric handles. /// /// Cheap to clone (`Arc` of a `Family` under the hood). Each shard owns @@ -215,6 +237,10 @@ pub struct ShardMetrics { metadata_prepare_gap_drops_total: Counter, metadata_read_frontier_refusals_total: Counter, client_requests_denied_queue_full_total: Counter, + partition_consumer_offsets_denied_total: Family, + consumer_offset_denied_counters: [Counter; 2], + partition_consumer_offsets_stranded: Family, + consumer_offset_stranded_gauges: [Gauge; 2], } impl ShardMetrics { @@ -227,6 +253,34 @@ impl ShardMetrics { let cached_counters = Arc::new(std::array::from_fn(|_| { std::array::from_fn(|_| OnceLock::new()) })); + let partition_consumer_offsets_denied_total: Family = + Family::default(); + let consumer_denied = { + partition_consumer_offsets_denied_total + .get_or_create(&ConsumerOffsetKindLabel { kind: "consumer" }) + .clone() + }; + let consumer_group_denied = { + partition_consumer_offsets_denied_total + .get_or_create(&ConsumerOffsetKindLabel { + kind: "consumer_group", + }) + .clone() + }; + let consumer_offset_denied_counters = [consumer_denied, consumer_group_denied]; + let partition_consumer_offsets_stranded: Family = + Family::default(); + // End each Family read guard before creating the next series, which + // needs the same family's write lock on a miss. + let consumer_stranded = partition_consumer_offsets_stranded + .get_or_create(&ConsumerOffsetKindLabel { kind: "consumer" }) + .clone(); + let group_stranded = partition_consumer_offsets_stranded + .get_or_create(&ConsumerOffsetKindLabel { + kind: "consumer_group", + }) + .clone(); + let consumer_offset_stranded_gauges = [consumer_stranded, group_stranded]; Self { frame_drops_total, cached_counters, @@ -243,9 +297,36 @@ impl ShardMetrics { metadata_prepare_gap_drops_total: Counter::default(), metadata_read_frontier_refusals_total: Counter::default(), client_requests_denied_queue_full_total: Counter::default(), + partition_consumer_offsets_denied_total, + consumer_offset_denied_counters, + partition_consumer_offsets_stranded, + consumer_offset_stranded_gauges, } } + /// Best effort: counts explicit client denials read off the reply status + /// and the poll-side reservation refusals. A denial the pump answers to an + /// auto-commit submit has no client reply to read and is not counted. + pub fn record_consumer_offset_denied(&self, kind: ConsumerKind) { + self.consumer_offset_denied_counters[consumer_kind_index(kind)].inc(); + } + + /// Republished by every partition sweep: the sum over this shard's + /// partitions of offset keys whose file could not be loaded or unlinked. + /// Such a key stays counted against `consumer_offsets_max` until a later + /// store or delete of it succeeds, so a non-zero value that never falls is + /// an offsets directory an operator has to repair. + pub fn set_consumer_offsets_stranded(&self, kind: ConsumerKind, count: usize) { + self.consumer_offset_stranded_gauges[consumer_kind_index(kind)] + .set(i64::try_from(count).unwrap_or(i64::MAX)); + } + + #[cfg(test)] + #[must_use] + pub fn consumer_offset_denied_value(&self, kind: ConsumerKind) -> u64 { + self.consumer_offset_denied_counters[consumer_kind_index(kind)].get() + } + /// Bumped every time a client request is answered with a retryable denial /// because that client already has the maximum number of requests queued /// behind one the shard has not answered yet. @@ -625,6 +706,18 @@ impl ShardMetrics { "client requests denied retryable because that client's request queue was full", self.client_requests_denied_queue_full_total.clone(), ); + registry.register( + "partition_consumer_offsets_denied", + "consumer offset creations denied at the per-partition admission limit (best effort: \ + explicit client denials and poll-side reservation refusals)", + self.partition_consumer_offsets_denied_total.clone(), + ); + registry.register( + "partition_consumer_offsets_stranded", + "consumer offset keys whose file could not be loaded or unlinked, still counted \ + against the limit", + self.partition_consumer_offsets_stranded.clone(), + ); } } @@ -706,6 +799,35 @@ mod tests { assert_eq!(from_family, 5); } + #[test] + fn consumer_offset_denials_use_two_cached_kind_series() { + let metrics = ShardMetrics::for_shard(); + metrics.record_consumer_offset_denied(ConsumerKind::Consumer); + metrics.record_consumer_offset_denied(ConsumerKind::Consumer); + metrics.record_consumer_offset_denied(ConsumerKind::ConsumerGroup); + + assert_eq!( + metrics.consumer_offset_denied_value(ConsumerKind::Consumer), + 2 + ); + assert_eq!( + metrics.consumer_offset_denied_value(ConsumerKind::ConsumerGroup), + 1 + ); + let mut registry = Registry::default(); + metrics.register(&mut registry); + let mut buffer = String::new(); + prometheus_client::encoding::text::encode(&mut buffer, ®istry) + .expect("scrape encoding succeeds"); + assert_eq!( + buffer + .lines() + .filter(|line| line.starts_with("partition_consumer_offsets_denied_total")) + .count(), + 2 + ); + } + #[test] fn unknown_label_set_falls_back_to_family() { // A label outside the const tables must still record via the slow diff --git a/core/shard/src/router.rs b/core/shard/src/router.rs index 4a3a0d2149..b7465d2b7c 100644 --- a/core/shard/src/router.rs +++ b/core/shard/src/router.rs @@ -728,6 +728,15 @@ where // already made. self.on_partition_submit(request, reply).await; } + LifecycleFrame::AutoCommitSubmit { + request, + reservation, + } => { + self.plane + .partitions() + .on_auto_commit_request(request, reservation) + .await; + } LifecycleFrame::MetadataCommitTick => { // Reconciler may not yet be wired (e.g. mid-bootstrap, or // single-shard tests that never enable the reconciler loop). diff --git a/core/simulator/src/client.rs b/core/simulator/src/client.rs index 99fe11a7f4..1c78d8b050 100644 --- a/core/simulator/src/client.rs +++ b/core/simulator/src/client.rs @@ -682,8 +682,8 @@ impl SimClient { self.non_replicated_request(GET_STREAM_CODE, METADATA_GROUP, &body) } - /// Store offset with explicit `AckLevel`. `NoAck` takes the primary's - /// fast path (no replication); `Quorum` goes through VSR. + /// Store offset with explicit `AckLevel`. A multi-replica partition routes + /// both values through VSR. A single replica retains the `NoAck` fast path. /// /// # Panics /// Panics on payload too large for `Owned::<4096>` or invalid diff --git a/core/simulator/src/lib.rs b/core/simulator/src/lib.rs index 373e9fd23c..d5c554023e 100644 --- a/core/simulator/src/lib.rs +++ b/core/simulator/src/lib.rs @@ -187,6 +187,7 @@ pub struct Simulator { /// [`shard::IggyShard::deliver_client_request`]) instead of raw `dispatch` /// routing. Set at construction by [`Simulator::with_shards_shell`]. shell: bool, + consumer_offsets_max: usize, } impl Simulator { @@ -487,9 +488,44 @@ impl Simulator { deferred_client_replies: Vec::new(), seed, shell, + consumer_offsets_max: partitions::DEFAULT_CONSUMER_OFFSETS_MAX, } } + /// Configure the per-kind offset limit used by partitions materialised + /// after this call. + /// + /// # Panics + /// Panics when `consumer_offsets_max` is zero. + pub fn set_consumer_offsets_max(&mut self, consumer_offsets_max: usize) { + assert!( + consumer_offsets_max > 0, + "consumer offset limit must be nonzero" + ); + self.consumer_offsets_max = consumer_offsets_max; + } + + #[must_use] + /// # Panics + /// Panics when `replica_idx` is outside the simulated roster. + pub fn partition_consumer_offset_counts( + &self, + replica_idx: usize, + namespace: IggyNamespace, + kind: iggy_common::ConsumerKind, + ) -> Option<(usize, usize)> { + self.replicas[replica_idx] + .partition_shard(namespace) + .plane + .partitions() + .with_partition(&namespace, |partition| { + ( + partition.durable_consumer_offset_count(kind), + partition.consumer_offset_map_count(kind), + ) + }) + } + /// Init a partition with its own consensus group on every live replica. /// /// The reconciler's outcome without running it: the namespace is committed to @@ -518,6 +554,7 @@ impl Simulator { namespace, self.restore_partition_frontier, created_view, + self.consumer_offsets_max, ); } } @@ -1212,6 +1249,7 @@ impl Simulator { namespace, self.restore_partition_frontier, created_view, + self.consumer_offsets_max, ); } @@ -1249,6 +1287,10 @@ impl Simulator { retained.insert( namespace, RetainedPartitionState { + consumer_offsets: partition + .retained_consumer_offsets(iggy_common::ConsumerKind::Consumer), + consumer_group_offsets: partition + .retained_consumer_offsets(iggy_common::ConsumerKind::ConsumerGroup), log: std::mem::take(&mut partition.log), durable_offset: offsets.commit_offset, write_offset: offsets.write_offset, @@ -1303,7 +1345,10 @@ impl Simulator { }; // Partitions are driven directly, so a poll's auto-commit is never // replicated (the serving shard's job in the real server). Offset discarded. - let (fragments, _commit_offset, _auto_commit) = futures::executor::block_on(plan.execute()); + let (fragments, _commit_offset, auto_commit) = futures::executor::block_on(plan.execute())?; + if let Some(applied) = auto_commit { + applied.admit(|_| Ok::<(), IggyError>(()))?; + } Ok(fragments) } @@ -1431,6 +1476,7 @@ fn materialise_partition( namespace: IggyNamespace, restore_frontier: bool, created_view: u32, + consumer_offsets_max: usize, ) { let shard_count = u32::try_from(replica.shards.len()).expect("shard count fits u32"); let owner = calculate_shard_assignment(&namespace, shard_count); @@ -1478,7 +1524,8 @@ fn materialise_partition( recovered_state, retained, restore_frontier, - PartitionMaterialisation::new(epoch, created_view), + PartitionMaterialisation::new(epoch, created_view) + .with_consumer_offsets_max(consumer_offsets_max), ); for shard in &replica.shards { shard.shards_table().insert( @@ -1495,8 +1542,166 @@ mod tests { use crate::workload::apply_sim_commands; use bytes::Bytes; use consensus::Status; + use iggy_binary_protocol::{AckLevel, RoutedRequestHeader}; + use iggy_common::ConsumerKind; use server_common::sharding::IggyNamespace; + fn submit_and_wait_for_reply( + sim: &mut Simulator, + client_id: u128, + target: u8, + request: Message, + ) -> Message { + let request_id = request.header().request; + sim.submit_request(client_id, target, request.into_generic()); + for _ in 0..100 { + if let Some(reply) = sim + .step() + .into_iter() + .find(|reply| reply.header().request == request_id) + { + return reply; + } + } + panic!("request {request_id} did not receive a reply"); + } + + #[test] + #[allow(clippy::too_many_lines)] + fn given_small_offset_limit_when_using_quorum_and_no_ack_should_bound_every_replica() { + server_common::MemoryPool::init_pool(&server_common::MemoryPoolSettings { + enabled: false, + size: iggy_common::IggyByteSize::from(0u64), + bucket_capacity: 1, + }); + let replica_count = 3u8; + let client_id = 1u128; + let namespace = IggyNamespace::new(1, 1, 0); + let mut sim = Simulator::new( + usize::from(replica_count), + std::iter::once(client_id), + packet::PacketSimulatorOptions { + node_count: replica_count, + client_count: 1, + seed: 0xC0FF_EE01, + ..packet::PacketSimulatorOptions::default() + }, + ); + sim.set_consumer_offsets_max(2); + sim.init_partition(namespace); + let client = SimClient::new(client_id); + sim.register_client_with_primary(&client); + + let produced = submit_and_wait_for_reply( + &mut sim, + client_id, + 0, + client.send_messages(namespace, &[Bytes::from_static(b"offset-cap")]), + ); + assert_eq!(produced.header().status, 0); + + for (consumer_id, ack) in [(1, AckLevel::Quorum), (2, AckLevel::NoAck)] { + let reply = submit_and_wait_for_reply( + &mut sim, + client_id, + 0, + client.store_consumer_offset(namespace, 1, consumer_id, 0, ack), + ); + assert_eq!(reply.header().status, 0); + } + + let denied = submit_and_wait_for_reply( + &mut sim, + client_id, + 0, + client.store_consumer_offset(namespace, 1, 3, 0, AckLevel::NoAck), + ); + assert_eq!( + denied.header().status, + IggyError::TooManyConsumerOffsets.as_code() + ); + + let deleted = submit_and_wait_for_reply( + &mut sim, + client_id, + 0, + client.delete_consumer_offset(namespace, 1, 1, AckLevel::NoAck), + ); + assert_eq!(deleted.header().status, 0); + let replacement = submit_and_wait_for_reply( + &mut sim, + client_id, + 0, + client.store_consumer_offset(namespace, 1, 3, 0, AckLevel::Quorum), + ); + assert_eq!(replacement.header().status, 0); + + sim.replica_crash(0); + let new_primary = (0..800) + .find_map(|_| { + sim.step(); + (1..replica_count).find(|replica_id| { + sim.partition_consensus_state(usize::from(*replica_id), namespace) + .is_some_and(|state| { + state.is_primary + && state.status == Status::Normal + && sim + .partition_consumer_offset_counts( + usize::from(*replica_id), + namespace, + ConsumerKind::Consumer, + ) + .is_some_and(|counts| counts.0 == 2) + }) + }) + }) + .expect("surviving replicas elect a new partition primary"); + let denied_after_failover = submit_and_wait_for_reply( + &mut sim, + client_id, + new_primary, + client.store_consumer_offset(namespace, 1, 4, 0, AckLevel::Quorum), + ); + assert_eq!( + denied_after_failover.header().status, + IggyError::TooManyConsumerOffsets.as_code(), + "the promoted primary must preserve the durable admission bound" + ); + + for _ in 0..100 { + sim.step(); + } + for replica_id in 0..sim.replicas.len() { + let counts = sim + .partition_consumer_offset_counts(replica_id, namespace, ConsumerKind::Consumer) + .expect("partition is materialised"); + assert!( + counts.0 <= 2, + "replica {replica_id} durable count {counts:?}" + ); + assert!(counts.1 <= 4, "replica {replica_id} map count {counts:?}"); + } + + sim.replica_restart(0); + let recovered = sim + .partition_consumer_offset_counts(0, namespace, ConsumerKind::Consumer) + .expect("restarted replica rematerialises the partition"); + assert_eq!(recovered.0, 2, "restart must retain durable offset keys"); + for _ in 0..200 { + sim.step(); + } + for replica_id in 0..sim.replicas.len() { + let counts = sim + .partition_consumer_offset_counts(replica_id, namespace, ConsumerKind::Consumer) + .expect("partition remains materialised after rejoin"); + assert!( + counts.0 <= 2, + "replica {replica_id} durable count {counts:?}" + ); + assert!(counts.1 <= 4, "replica {replica_id} map count {counts:?}"); + } + } + /// Crashing the primary in a 5-node cluster: 4 survivors detect via /// heartbeat timeout and elect a new primary via view change. #[test] @@ -2053,8 +2258,11 @@ mod tests { // together remap every stream; and partition ops drawing from the one shared // request counter instead of a separate sequence based at `1<<63`, which // renumbers every partition request id and so every reply header in the trace. + // Multi-replica consumer-offset requests carrying `NoAck` now enter + // VSR, so their scheduling and committed replies contribute to the + // deterministic trace instead of taking the primary-local fast path. assert_eq!( - h1, 0x5C2B_6057_2DA9_908B, + h1, 0x31C2_ADA9_9411_FCD4, "workload reply hash drifted from locked baseline" ); } @@ -2482,6 +2690,7 @@ mod tests { /// prepares. A third prepare is withheld from it, then placed on its inbox /// after simulator materialisation stages the parked prefix. Running the pump /// without advancing its tick forces the inbox arm to drain the prefix first. + #[allow(clippy::too_many_lines)] fn parked_prepare_redispatch_trace(seed: u64) -> (usize, Vec, u64) { const CLIENT_ID: u128 = 1; @@ -2504,8 +2713,20 @@ mod tests { // Replica 0 is the view-0 primary. Replica 2 supplies quorum while // replica 1 has committed metadata but no local partition yet. - materialise_partition(&sim.replicas[0], namespace, false, created_view); - materialise_partition(&sim.replicas[2], namespace, false, created_view); + materialise_partition( + &sim.replicas[0], + namespace, + false, + created_view, + sim.consumer_offsets_max, + ); + materialise_partition( + &sim.replicas[2], + namespace, + false, + created_view, + sim.consumer_offsets_max, + ); let client = SimClient::new(CLIENT_ID); sim.shell_login(&client); @@ -2557,7 +2778,13 @@ mod tests { let later_prepare = retained_prepare(&sim, 0, namespace, 3); sim.network.process_enable(ProcessId::Replica(1)); - materialise_partition(&sim.replicas[1], namespace, false, created_view); + materialise_partition( + &sim.replicas[1], + namespace, + false, + created_view, + sim.consumer_offsets_max, + ); assert_eq!(lagging_shard.parked_frame_count(namespace), 0); assert_eq!( lagging_shard.redispatched_frame_count(), diff --git a/core/simulator/src/replica.rs b/core/simulator/src/replica.rs index 28816aa465..09096c0df7 100644 --- a/core/simulator/src/replica.rs +++ b/core/simulator/src/replica.rs @@ -380,6 +380,7 @@ pub fn new_shard( messages_required_to_save: 1000, size_of_messages_required_to_save: IggyByteSize::from(4 * 1024 * 1024), enforce_fsync: false, //Disable fsync for simulation + consumer_offset_enforce_fsync: false, validate_checksum: true, segment_size: IggyByteSize::from(1024 * 1024 * 1024), preallocate_segments: false, diff --git a/core/simulator/src/workload/effect.rs b/core/simulator/src/workload/effect.rs index a9d8c868c2..e8df466e92 100644 --- a/core/simulator/src/workload/effect.rs +++ b/core/simulator/src/workload/effect.rs @@ -17,7 +17,7 @@ //! Predicted server-state mutations emitted by op modules on commit. //! -//! Name-keyed throughout. Server-ng emits empty reply bodies, so the +//! Name-keyed throughout. Server emits empty reply bodies, so the //! workload cannot recover server-assigned numeric ids; shadow lookups //! address entities by name (`WireIdentifier::named`). Id-keyed effects //! return once reply-body parsing lands. diff --git a/core/simulator/src/workload/shadow.rs b/core/simulator/src/workload/shadow.rs index 2692ff1684..b0e0ea8cb0 100644 --- a/core/simulator/src/workload/shadow.rs +++ b/core/simulator/src/workload/shadow.rs @@ -17,7 +17,7 @@ //! Shadow state: the workload's prediction of server-side entity state. //! -//! Name-keyed throughout. Server-ng does not yet ship reply bodies, so +//! Name-keyed throughout. Server does not yet ship reply bodies, so //! the workload cannot observe server-assigned numeric ids. Lookups are //! by name; requests route via `WireIdentifier::named(...)`. When typed //! response bodies land, id-keyed maps return as a parallel index; diff --git a/examples/node/package-lock.json b/examples/node/package-lock.json index fdf8f7883a..2847ff873c 100644 --- a/examples/node/package-lock.json +++ b/examples/node/package-lock.json @@ -23,7 +23,7 @@ }, "../../foreign/node": { "name": "apache-iggy", - "version": "0.10.0-edge.5", + "version": "0.10.0-edge.6", "license": "Apache-2.0", "dependencies": { "@node-rs/xxhash": "1.7.7", diff --git a/examples/python/uv.lock b/examples/python/uv.lock index bde592d6b1..469565c7ed 100644 --- a/examples/python/uv.lock +++ b/examples/python/uv.lock @@ -8,7 +8,7 @@ exclude-newer-span = "P7D" [[package]] name = "apache-iggy" -version = "0.9.0.dev6" +version = "0.9.0.dev7" source = { directory = "../../foreign/python" } [package.metadata] diff --git a/foreign/go/contracts/version.go b/foreign/go/contracts/version.go index 32ff7a2821..f3799086d4 100644 --- a/foreign/go/contracts/version.go +++ b/foreign/go/contracts/version.go @@ -17,4 +17,4 @@ package iggcon -const Version = "0.9.0-edge.5" +const Version = "0.9.0-edge.6" diff --git a/foreign/go/errors/errors.yaml b/foreign/go/errors/errors.yaml index c1f96ea4a9..145cd569f8 100644 --- a/foreign/go/errors/errors.yaml +++ b/foreign/go/errors/errors.yaml @@ -775,6 +775,10 @@ fields: - name: Path type: string +- name: TooManyConsumerOffsets + code: 3024 + format: "consumer offset limit reached for partition, raise [partition] consumer_offsets_max" + fields: [] - name: PartitionIdSpaceExhausted code: 3013 format: "partition id space exhausted for this topic" diff --git a/foreign/go/errors/errors_gen.go b/foreign/go/errors/errors_gen.go index ca981aa126..2f80a8a781 100644 --- a/foreign/go/errors/errors_gen.go +++ b/foreign/go/errors/errors_gen.go @@ -1584,6 +1584,17 @@ func (e CannotOpenConsumerOffsetsFile) Is(target error) bool { return ok } +type TooManyConsumerOffsets struct{} + +func (e TooManyConsumerOffsets) Error() string { + return "consumer offset limit reached for partition, raise [partition] consumer_offsets_max" +} +func (e TooManyConsumerOffsets) Code() Code { return 3024 } +func (e TooManyConsumerOffsets) Is(target error) bool { + _, ok := target.(TooManyConsumerOffsets) + return ok +} + type PartitionIdSpaceExhausted struct{} func (e PartitionIdSpaceExhausted) Error() string { @@ -2784,6 +2795,7 @@ var ( ErrConsumerOffsetNotFound = ConsumerOffsetNotFound{} ErrNotResolvedConsumer = NotResolvedConsumer{} ErrCannotOpenConsumerOffsetsFile = CannotOpenConsumerOffsetsFile{} + ErrTooManyConsumerOffsets = TooManyConsumerOffsets{} ErrPartitionIdSpaceExhausted = PartitionIdSpaceExhausted{} ErrSegmentNotFound = SegmentNotFound{} ErrSegmentClosed = SegmentClosed{} @@ -3027,6 +3039,7 @@ const ( ConsumerOffsetNotFoundCode Code = 3021 NotResolvedConsumerCode Code = 3022 CannotOpenConsumerOffsetsFileCode Code = 3023 + TooManyConsumerOffsetsCode Code = 3024 PartitionIdSpaceExhaustedCode Code = 3013 SegmentNotFoundCode Code = 4000 SegmentClosedCode Code = 4001 @@ -3409,6 +3422,8 @@ func (c Code) String() string { return "NotResolvedConsumer" case CannotOpenConsumerOffsetsFileCode: return "CannotOpenConsumerOffsetsFile" + case TooManyConsumerOffsetsCode: + return "TooManyConsumerOffsets" case PartitionIdSpaceExhaustedCode: return "PartitionIdSpaceExhausted" case SegmentNotFoundCode: @@ -3892,6 +3907,8 @@ func FromCode(code Code) IggyError { return ErrNotResolvedConsumer case CannotOpenConsumerOffsetsFileCode: return ErrCannotOpenConsumerOffsetsFile + case TooManyConsumerOffsetsCode: + return ErrTooManyConsumerOffsets case PartitionIdSpaceExhaustedCode: return ErrPartitionIdSpaceExhausted case SegmentNotFoundCode: diff --git a/foreign/go/errors/errors_test.go b/foreign/go/errors/errors_test.go index cacb8f90b5..cb9719983f 100644 --- a/foreign/go/errors/errors_test.go +++ b/foreign/go/errors/errors_test.go @@ -95,6 +95,25 @@ func TestIggyError_ConsensusErrors(t *testing.T) { } } +func TestIggyError_TooManyConsumerOffsets(t *testing.T) { + err := TooManyConsumerOffsets{} + if err.Code() != TooManyConsumerOffsetsCode { + t.Errorf("Code() = %v, want %v", err.Code(), TooManyConsumerOffsetsCode) + } + if err.Error() != "consumer offset limit reached for partition, raise [partition] consumer_offsets_max" { + t.Errorf("Error() = %q", err.Error()) + } + if !errors.Is(err, ErrTooManyConsumerOffsets) { + t.Errorf("errors.Is(%v, ErrTooManyConsumerOffsets) = false", err) + } + if errors.Is(err, ErrConsumerOffsetNotFound) { + t.Errorf("errors.Is(%v, ErrConsumerOffsetNotFound) = true", err) + } + if resolved := FromCode(TooManyConsumerOffsetsCode); !errors.Is(resolved, ErrTooManyConsumerOffsets) { + t.Errorf("FromCode(%d) = %v", TooManyConsumerOffsetsCode, resolved) + } +} + func TestFromCode_FallsBackToTheGenericError(t *testing.T) { if resolved := FromCode(Code(0xFFFFFF)); !errors.Is(resolved, ErrError) { t.Errorf("FromCode(unknown) = %v, want ErrError", resolved) diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/exception/IggyErrorCode.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/exception/IggyErrorCode.java index 80e729572a..5d776f8276 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/exception/IggyErrorCode.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/exception/IggyErrorCode.java @@ -94,6 +94,9 @@ public enum IggyErrorCode { PARTITION_NOT_FOUND(3007), PARTITION_ID_SPACE_EXHAUSTED(3013), + // Consumer offset errors + TOO_MANY_CONSUMER_OFFSETS(3024), + // Segment errors SEGMENT_NOT_FOUND(4000), SEGMENT_CLOSED(4001), diff --git a/foreign/java/java-sdk/src/test/java/org/apache/iggy/exception/IggyErrorCodeTest.java b/foreign/java/java-sdk/src/test/java/org/apache/iggy/exception/IggyErrorCodeTest.java index 3d87877c63..45feef08b8 100644 --- a/foreign/java/java-sdk/src/test/java/org/apache/iggy/exception/IggyErrorCodeTest.java +++ b/foreign/java/java-sdk/src/test/java/org/apache/iggy/exception/IggyErrorCodeTest.java @@ -215,6 +215,7 @@ void shouldParseCodeOne() { // Partition errors "3007, PARTITION_NOT_FOUND", "3013, PARTITION_ID_SPACE_EXHAUSTED", + "3024, TOO_MANY_CONSUMER_OFFSETS", // Segment errors "4000, SEGMENT_NOT_FOUND", diff --git a/foreign/node/package-lock.json b/foreign/node/package-lock.json index 69b394e60f..10bbf5baf4 100644 --- a/foreign/node/package-lock.json +++ b/foreign/node/package-lock.json @@ -1,12 +1,12 @@ { "name": "apache-iggy", - "version": "0.10.0-edge.5", + "version": "0.10.0-edge.6", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "apache-iggy", - "version": "0.10.0-edge.5", + "version": "0.10.0-edge.6", "license": "Apache-2.0", "dependencies": { "@node-rs/xxhash": "1.7.7", diff --git a/foreign/node/package.json b/foreign/node/package.json index 6722c71616..ea4ec4dd04 100644 --- a/foreign/node/package.json +++ b/foreign/node/package.json @@ -1,7 +1,7 @@ { "name": "apache-iggy", "type": "module", - "version": "0.10.0-edge.5", + "version": "0.10.0-edge.6", "description": "Official Apache Iggy NodeJS SDK", "keywords": [ "iggy", diff --git a/foreign/node/src/wire/error.code.test.ts b/foreign/node/src/wire/error.code.test.ts index 680d91be21..dc768e2d20 100644 --- a/foreign/node/src/wire/error.code.test.ts +++ b/foreign/node/src/wire/error.code.test.ts @@ -19,6 +19,10 @@ import assert from 'node:assert/strict'; import { it } from 'node:test'; import { translateErrorCode } from './error.code.js'; +it('translates the consumer-offset capacity error', () => { + assert.equal(translateErrorCode(3024), 'Consumer offset limit reached for partition, raise [partition] consumer_offsets_max'); +}); + it('translates the consumer-group error range', () => { assert.equal( translateErrorCode(5000), diff --git a/foreign/node/src/wire/error.code.ts b/foreign/node/src/wire/error.code.ts index efec357035..58773d957a 100644 --- a/foreign/node/src/wire/error.code.ts +++ b/foreign/node/src/wire/error.code.ts @@ -165,6 +165,7 @@ export const translateErrorCode = (code: number): string => { case '3021': return "Consumer offset for consumer with ID: {0} was not found."; case '3022': return "Failed to resolve consumer with ID: {0}"; case '3023': return "Cannot open consumer offsets file for path: {0}"; + case '3024': return "Consumer offset limit reached for partition, raise [partition] consumer_offsets_max"; case '3013': return "Partition id space exhausted for this topic"; // MESSAGE diff --git a/foreign/python/Cargo.toml b/foreign/python/Cargo.toml index f3965a16e3..1430a98fc8 100644 --- a/foreign/python/Cargo.toml +++ b/foreign/python/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "apache-iggy" -version = "0.9.0-dev6" +version = "0.9.0-dev7" edition = "2024" authors = ["Iggy Committers "] license = "Apache-2.0" @@ -37,7 +37,7 @@ doc = false [dependencies] bytes = "1.12.1" futures = "0.3.34" -iggy = { path = "../../core/sdk", version = "0.11.0-edge.6" } +iggy = { path = "../../core/sdk", version = "0.11.0-edge.7" } paste = "1" pyo3 = "0.29.2" pyo3-async-runtimes = { version = "0.29.0", features = [ diff --git a/foreign/python/pyproject.toml b/foreign/python/pyproject.toml index 54d53ec901..1434010dc6 100644 --- a/foreign/python/pyproject.toml +++ b/foreign/python/pyproject.toml @@ -22,7 +22,7 @@ build-backend = "maturin" [project] name = "apache-iggy" requires-python = ">=3.10" -version = "0.9.0.dev6" +version = "0.9.0.dev7" description = "Apache Iggy is the persistent message streaming platform written in Rust, supporting QUIC, TCP and HTTP transport protocols, capable of processing millions of messages per second." readme = "README.md" license = { file = "LICENSE" } diff --git a/foreign/python/uv.lock b/foreign/python/uv.lock index d6f3eb5067..f4fce79c96 100644 --- a/foreign/python/uv.lock +++ b/foreign/python/uv.lock @@ -8,7 +8,7 @@ exclude-newer-span = "P7D" [[package]] name = "apache-iggy" -version = "0.9.0.dev6" +version = "0.9.0.dev7" source = { editable = "." } [package.optional-dependencies] From add0bd9676b2b8868f49cf44e320ac9ee263b6cd Mon Sep 17 00:00:00 2001 From: saie-ch <132209179+saie-ch@users.noreply.github.com> Date: Mon, 7 Sep 2026 16:47:29 +0530 Subject: [PATCH 077/182] feat(python): add QuicConfig transport configuration (#3991) Relates to #2835 --- .../python-maturin/pre-merge/action.yml | 1 + .github/workflows/coverage-baseline.yml | 1 + core/sdk/src/prelude.rs | 1 + examples/python/getting-started/consumer.py | 17 +- examples/python/getting-started/producer.py | 17 +- foreign/python/README.md | 5 +- foreign/python/apache_iggy.pyi | 170 ++++++- foreign/python/src/client.rs | 81 +-- foreign/python/src/config.rs | 465 +++++++++++++++++- foreign/python/src/duration.rs | 33 ++ foreign/python/src/lib.rs | 4 +- foreign/python/tests/test_connectivity.py | 2 +- foreign/python/tests/test_quic_config.py | 434 ++++++++++++++++ foreign/python/tests/utils.py | 30 +- 14 files changed, 1211 insertions(+), 50 deletions(-) create mode 100644 foreign/python/tests/test_quic_config.py diff --git a/.github/actions/python-maturin/pre-merge/action.yml b/.github/actions/python-maturin/pre-merge/action.yml index 234c36efed..3cca87fceb 100644 --- a/.github/actions/python-maturin/pre-merge/action.yml +++ b/.github/actions/python-maturin/pre-merge/action.yml @@ -180,6 +180,7 @@ runs: # overwrite the coverage-instrumented .so with a non-instrumented one IGGY_SERVER_HOST=127.0.0.1 \ IGGY_SERVER_TCP_PORT=8090 \ + IGGY_SERVER_QUIC_PORT=8080 \ IGGY_SERVER_DOCKER_IMAGE=iggy-server:local \ uv run --no-sync pytest tests/ -v \ --junitxml=../../reports/python-junit.xml \ diff --git a/.github/workflows/coverage-baseline.yml b/.github/workflows/coverage-baseline.yml index 9daa683559..d8f070ba25 100644 --- a/.github/workflows/coverage-baseline.yml +++ b/.github/workflows/coverage-baseline.yml @@ -349,6 +349,7 @@ jobs: cd foreign/python IGGY_SERVER_HOST=127.0.0.1 \ IGGY_SERVER_TCP_PORT=8090 \ + IGGY_SERVER_QUIC_PORT=8080 \ IGGY_SERVER_DOCKER_IMAGE=iggy-server:local \ uv run --no-sync pytest tests/ -v \ --junitxml=../../reports/python-junit.xml \ diff --git a/core/sdk/src/prelude.rs b/core/sdk/src/prelude.rs index 72e2503d16..10cfbc685c 100644 --- a/core/sdk/src/prelude.rs +++ b/core/sdk/src/prelude.rs @@ -41,6 +41,7 @@ pub use crate::clients::producer_builder::IggyProducerBuilder; pub use crate::clients::producer_config::{BackgroundConfig, DirectConfig}; pub use crate::clients::producer_sharding::{BalancedSharding, OrderedSharding, Sharding}; pub use crate::consumer_ext::IggyConsumerMessageExt; +pub use crate::quic::quic_client::QuicClient; pub use crate::stream_builder::IggyConsumerConfig; pub use crate::stream_builder::IggyStreamConsumer; pub use crate::stream_builder::{IggyProducerConfig, IggyStreamProducer}; diff --git a/examples/python/getting-started/consumer.py b/examples/python/getting-started/consumer.py index 6510841046..f3f7886409 100755 --- a/examples/python/getting-started/consumer.py +++ b/examples/python/getting-started/consumer.py @@ -102,7 +102,22 @@ def parse_args() -> ArgNamespace: def build_config(args: ArgNamespace) -> TcpConfig: - """Build a TCP client configuration with auto-login and reconnection.""" + """Build the TCP client configuration with auto-login and reconnection.""" + + # IggyClient(...) also accepts a QuicConfig for the QUIC transport. To use + # it, import QuicConfig and QuicReconnectionConfig above, change the return + # annotation to QuicConfig, and replace the return statement with: + # + # return QuicConfig( + # server_address="127.0.0.1:8080", + # server_name="localhost", + # auto_login=AutoLogin.username_password(args.username, args.password), + # reconnection=QuicReconnectionConfig( + # enabled=True, interval=timedelta(seconds=1) + # ), + # ) + # + # main() logs args.tcp_server_address, so change that line too. return TcpConfig( server_address=args.tcp_server_address, diff --git a/examples/python/getting-started/producer.py b/examples/python/getting-started/producer.py index 80ab7c6a87..113bee858d 100755 --- a/examples/python/getting-started/producer.py +++ b/examples/python/getting-started/producer.py @@ -101,7 +101,22 @@ def parse_args() -> ArgNamespace: def build_config(args: ArgNamespace) -> TcpConfig: - """Build a TCP client configuration with auto-login and reconnection.""" + """Build the TCP client configuration with auto-login and reconnection.""" + + # IggyClient(...) also accepts a QuicConfig for the QUIC transport. To use + # it, import QuicConfig and QuicReconnectionConfig above, change the return + # annotation to QuicConfig, and replace the return statement with: + # + # return QuicConfig( + # server_address="127.0.0.1:8080", + # server_name="localhost", + # auto_login=AutoLogin.username_password(args.username, args.password), + # reconnection=QuicReconnectionConfig( + # enabled=True, interval=timedelta(seconds=1) + # ), + # ) + # + # main() logs args.tcp_server_address, so change that line too. return TcpConfig( server_address=args.tcp_server_address, diff --git a/foreign/python/README.md b/foreign/python/README.md index e2a4fffc2d..df62cfc77e 100644 --- a/foreign/python/README.md +++ b/foreign/python/README.md @@ -134,7 +134,7 @@ running prek / committing / pushing. This list is not exhaustive and other hook ## Client Configuration -`IggyClient` takes either a server address or a `TcpConfig`: +`IggyClient` takes a server address, a `TcpConfig`, or a `QuicConfig`: ```python import asyncio @@ -168,6 +168,9 @@ async def main(): asyncio.run(main()) ``` +`IggyClient(...)` also accepts a `QuicConfig` for the QUIC transport; see +`examples/python/getting-started/producer.py` for a config swap example. + ## Examples Refer to the [examples/python/](https://github.com/apache/iggy/tree/master/examples/python) directory for usage examples. diff --git a/foreign/python/apache_iggy.pyi b/foreign/python/apache_iggy.pyi index ed15ed2f54..814549ffd1 100644 --- a/foreign/python/apache_iggy.pyi +++ b/foreign/python/apache_iggy.pyi @@ -47,6 +47,8 @@ __all__ = [ "Partition", "Permissions", "PollingStrategy", + "QuicConfig", + "QuicReconnectionConfig", "ReceiveMessage", "SendMessage", "SendMessagesConfirmation", @@ -897,23 +899,26 @@ class IggyClient: A Python class representing the Iggy client. It provides asynchronous functionality through the contained runtime. """ - def __new__(cls, conn: TcpConfig | builtins.str | None = None) -> IggyClient: + def __new__( + cls, conn: TcpConfig | QuicConfig | builtins.str | None = None + ) -> IggyClient: r""" - Constructs a new IggyClient from a TCP server address or a `TcpConfig`. - This initializes a new runtime for asynchronous operations. + Constructs a new IggyClient from a TCP server address, a `TcpConfig`, or a + `QuicConfig`. This initializes a new runtime for asynchronous operations. Future versions might utilize asyncio for more Pythonic async. Args: - conn: Either a `host:port` address, or a `TcpConfig` carrying the full - transport configuration. Defaults to `127.0.0.1:8090` with auto-login - disabled. A malformed address is reported differently by the two - forms: the string form raises `RuntimeError` here, while `TcpConfig` - raises `ValueError` when it is constructed, before it ever reaches - this call. Neither exception is a subclass of the other. + conn: A `host:port` address, a `TcpConfig`, or a `QuicConfig`. Defaults + to `127.0.0.1:8090` over TCP with auto-login disabled. A malformed + address is reported differently depending on the form: the string + form raises `RuntimeError` here, while `TcpConfig`/`QuicConfig` + raise `ValueError` when they are constructed, before either ever + reaches this call. Neither exception is a subclass of the other. Raises: RuntimeError: If the address passed as a string is not a valid - `host:port` pair. + `host:port` pair, or if a `QuicConfig` client cannot bind its + local UDP socket (for example the port is already in use). """ @classmethod def from_connection_string(cls, connection_string: builtins.str) -> IggyClient: @@ -1775,6 +1780,151 @@ class PollingStrategy: ... +@typing.final +class QuicConfig: + r""" + Configuration for the QUIC transport, accepted by `IggyClient(...)`. + + Every field is keyword-only and optional. + """ + @property + def server_address(self) -> builtins.str: ... + @property + def client_address(self) -> builtins.str: ... + @property + def server_name(self) -> builtins.str: ... + @property + def auto_login(self) -> AutoLogin: ... + @property + def reconnection(self) -> QuicReconnectionConfig: ... + @property + def heartbeat_interval(self) -> datetime.timedelta: ... + @property + def response_buffer_size(self) -> builtins.int: ... + @property + def max_concurrent_bidi_streams(self) -> builtins.int: ... + @property + def datagram_send_buffer_size(self) -> builtins.int: ... + @property + def initial_mtu(self) -> builtins.int: ... + @property + def send_window(self) -> builtins.int: ... + @property + def receive_window(self) -> builtins.int: ... + @property + def keep_alive_interval(self) -> datetime.timedelta: ... + @property + def max_idle_timeout(self) -> datetime.timedelta: ... + @property + def validate_certificate(self) -> builtins.bool: ... + def __new__( + cls, + *, + server_address: builtins.str | None = None, + client_address: builtins.str | None = None, + server_name: builtins.str | None = None, + auto_login: AutoLogin | None = None, + reconnection: QuicReconnectionConfig | None = None, + heartbeat_interval: datetime.timedelta | None = None, + response_buffer_size: builtins.int | None = None, + max_concurrent_bidi_streams: builtins.int | None = None, + datagram_send_buffer_size: builtins.int | None = None, + initial_mtu: builtins.int | None = None, + send_window: builtins.int | None = None, + receive_window: builtins.int | None = None, + keep_alive_interval: datetime.timedelta | None = None, + max_idle_timeout: datetime.timedelta | None = None, + validate_certificate: builtins.bool | None = None, + ) -> QuicConfig: + r""" + Constructs a QUIC configuration. + + Args: + server_address: `host:port` of the Iggy server. Defaults to `127.0.0.1:8080`. + client_address: `host:port` to bind the local UDP socket to. Defaults to + `127.0.0.1:0`, which binds to any available port. That exact value, + passed or defaulted, binds `[::1]:0` instead when `server_address` + resolves to IPv6, so the socket in use may not be the address read + back here. Any other value binds as given. + server_name: Server name used for the QUIC/TLS handshake. Defaults to + `localhost`. + auto_login: Credentials replayed on every connect. Defaults to `AutoLogin.disabled()`. + reconnection: Reconnection policy. Defaults to `QuicReconnectionConfig()`. + heartbeat_interval: Interval of heartbeats sent by the client. Defaults to 5 seconds. + response_buffer_size: Size of the response buffer in bytes. Defaults to 10 MB. + max_concurrent_bidi_streams: Maximum number of concurrent bidirectional + streams. Defaults to 10,000. + datagram_send_buffer_size: Size of the datagram send buffer in bytes. + Defaults to 100,000. + initial_mtu: Initial MTU in bytes. Defaults to 1200. + send_window: Send window size in bytes. Defaults to 100,000. + receive_window: Receive window size in bytes. Defaults to 100,000. + keep_alive_interval: Interval between QUIC keep-alive pings, or a zero + duration to disable them. Defaults to 5 seconds. + max_idle_timeout: How long the connection tolerates silence before it is + considered dead, or a zero duration to use quinn's own default (30 + seconds) instead, since `configure()` skips the setter entirely when + zero. Defaults to 10 seconds. + validate_certificate: Whether to validate the server certificate. Defaults + to disabled, unlike the TCP and WebSocket transports. + + Raises: + ValueError: If `server_address` or `client_address` is not a valid + `host:port` pair, if a duration is negative, if + `heartbeat_interval` is zero, if `keep_alive_interval` or + `max_idle_timeout` is not a whole number of milliseconds, if + `initial_mtu` is below quinn's minimum of 1200, or if a numeric + field is outside the range of its underlying wire type. + """ + def __repr__(self) -> builtins.str: ... + +@typing.final +class QuicReconnectionConfig: + r""" + How the QUIC client reconnects after the connection to the server is lost. + """ + @property + def enabled(self) -> builtins.bool: ... + @property + def max_retries(self) -> builtins.int | None: ... + @property + def interval(self) -> datetime.timedelta: ... + @property + def reestablish_after(self) -> datetime.timedelta: ... + def __new__( + cls, + *, + enabled: builtins.bool | None = None, + max_retries: builtins.int | None = None, + interval: datetime.timedelta | None = None, + reestablish_after: datetime.timedelta | None = None, + ) -> QuicReconnectionConfig: + r""" + Constructs a reconnection policy. + + Args: + enabled: Whether to reconnect at all. Defaults to enabled. + max_retries: Redials of the configured server address after the first + attempt, or `None` for unlimited; `0` still makes that first + attempt. Unlike the TCP transport, QUIC redials the one address + it was configured with rather than walking a cluster roster, so + this counts dials. Defaults to unlimited, which means a call + awaited while the server is down never returns: `connect()` + waits inside the retry loop, as do `send_messages()` and + `poll_messages()` once auto-login is configured. Set a finite + number for request/reply style usage, so a call fails instead. + interval: Delay before each redial. Defaults to 1 second. + reestablish_after: Cooldown before redialing after a previously + successful connection, measured from when it was established, so + a session that outlived the interval is redialed at once. + Defaults to 5 seconds. + + Raises: + ValueError: If a duration is negative, if `max_retries` is outside the + range of an unsigned 32-bit integer, or if `interval` is zero. + """ + def __repr__(self) -> builtins.str: ... + @typing.final class ReceiveMessage: r""" diff --git a/foreign/python/src/client.rs b/foreign/python/src/client.rs index 8252156a36..f15b6e2d20 100644 --- a/foreign/python/src/client.rs +++ b/foreign/python/src/client.rs @@ -28,6 +28,7 @@ use pyo3_async_runtimes::tokio::future_into_py; use pyo3_stub_gen::define_stub_info_gatherer; use pyo3_stub_gen::derive::{gen_stub_pyclass, gen_stub_pymethods}; use std::collections::BTreeMap; +use std::fmt::Display; use std::str::FromStr; use std::sync::Arc; @@ -58,6 +59,11 @@ pub struct IggyClient { inner: Arc, } +/// Keeps the SDK's own message on the `RuntimeError` the Python surface raises. +fn to_runtime_error(error: E) -> PyErr { + PyErr::new::(error.to_string()) +} + /// Resolves the shared `create_topic`/`update_topic` parameters, applying /// server defaults where the caller left them unset. fn resolve_topic_params( @@ -66,8 +72,7 @@ fn resolve_topic_params( max_topic_size: Option<&MaxTopicSize>, ) -> PyResult<(CompressionAlgorithm, RustIggyExpiry, RustMaxTopicSize)> { let compression_algorithm = match compression_algorithm { - Some(algo) => CompressionAlgorithm::from_str(&algo) - .map_err(|e| PyErr::new::(e.to_string()))?, + Some(algo) => CompressionAlgorithm::from_str(&algo).map_err(to_runtime_error)?, None => CompressionAlgorithm::default(), }; @@ -87,44 +92,59 @@ fn resolve_topic_params( #[gen_stub_pymethods] #[pymethods] impl IggyClient { - /// Constructs a new IggyClient from a TCP server address or a `TcpConfig`. - /// This initializes a new runtime for asynchronous operations. + /// Constructs a new IggyClient from a TCP server address, a `TcpConfig`, or a + /// `QuicConfig`. This initializes a new runtime for asynchronous operations. /// Future versions might utilize asyncio for more Pythonic async. /// /// Args: - /// conn: Either a `host:port` address, or a `TcpConfig` carrying the full - /// transport configuration. Defaults to `127.0.0.1:8090` with auto-login - /// disabled. A malformed address is reported differently by the two - /// forms: the string form raises `RuntimeError` here, while `TcpConfig` - /// raises `ValueError` when it is constructed, before it ever reaches - /// this call. Neither exception is a subclass of the other. + /// conn: A `host:port` address, a `TcpConfig`, or a `QuicConfig`. Defaults + /// to `127.0.0.1:8090` over TCP with auto-login disabled. A malformed + /// address is reported differently depending on the form: the string + /// form raises `RuntimeError` here, while `TcpConfig`/`QuicConfig` + /// raise `ValueError` when they are constructed, before either ever + /// reaches this call. Neither exception is a subclass of the other. /// /// Raises: /// RuntimeError: If the address passed as a string is not a valid - /// `host:port` pair. + /// `host:port` pair, or if a `QuicConfig` client cannot bind its + /// local UDP socket (for example the port is already in use). #[new] #[pyo3(signature = (conn=None))] fn new( - #[gen_stub(override_type(type_repr = "TcpConfig | builtins.str | None"))] conn: Option< - PyClientConfig, - >, + #[gen_stub(override_type(type_repr = "TcpConfig | QuicConfig | builtins.str | None"))] + conn: Option, ) -> PyResult { - let config = match conn { - Some(PyClientConfig::Config(config)) => config.client_config(), - Some(PyClientConfig::ServerAddress(server_address)) => Arc::new( - TcpClientConfigBuilder::new() - .with_server_address(server_address) - .build() - .map_err(|e| { - PyErr::new::(e.to_string()) - })?, + let wrapper = match conn { + Some(PyClientConfig::Tcp(config)) => ClientWrapper::Tcp( + TcpClient::create(config.client_config()).map_err(to_runtime_error)?, + ), + Some(PyClientConfig::ServerAddress(server_address)) => { + let config = Arc::new( + TcpClientConfigBuilder::new() + .with_server_address(server_address) + .build() + .map_err(to_runtime_error)?, + ); + ClientWrapper::Tcp(TcpClient::create(config).map_err(to_runtime_error)?) + } + Some(PyClientConfig::Quic(config)) => { + // `quinn::Endpoint::client` (invoked eagerly by `QuicClient::create`) looks + // up the current Tokio runtime via `Handle::try_current()` and fails with + // `CannotCreateEndpoint` if none is active. This method runs synchronously + // from Python without one, so enter the runtime pyo3-async-runtimes uses + // for our own async methods before building the endpoint. + let _guard = pyo3_async_runtimes::tokio::get_runtime().enter(); + ClientWrapper::Quic( + QuicClient::create(config.client_config()).map_err(to_runtime_error)?, + ) + } + None => ClientWrapper::Tcp( + TcpClient::create(Arc::new(TcpClientConfig::default())) + .map_err(to_runtime_error)?, ), - None => Arc::new(TcpClientConfig::default()), }; - let tcp_client = TcpClient::create(config) - .map_err(|e| PyErr::new::(e.to_string()))?; - Ok(IggyClient { - inner: Arc::new(RustIggyClient::new(ClientWrapper::Tcp(tcp_client))), + Ok(Self { + inner: Arc::new(RustIggyClient::new(wrapper)), }) } @@ -138,6 +158,11 @@ impl IggyClient { _cls: &Bound<'_, PyType>, connection_string: String, ) -> PyResult { + // The QUIC transport builds its endpoint eagerly and needs a Tokio runtime context + // to do so (see the `QuicConfig` arm of `new()` above for details); entering it here + // is a no-op for the other transports since the protocol isn't known until the + // connection string is parsed. + let _guard = pyo3_async_runtimes::tokio::get_runtime().enter(); let client = RustIggyClient::from_connection_string(&connection_string) .map_err(|e| PyErr::new::(e.to_string()))?; Ok(Self { diff --git a/foreign/python/src/config.rs b/foreign/python/src/config.rs index ed6e2a14f0..cd98ef626d 100644 --- a/foreign/python/src/config.rs +++ b/foreign/python/src/config.rs @@ -17,6 +17,8 @@ use iggy::prelude::{ AutoLogin as RustAutoLogin, Credentials as RustCredentials, + QuicClientConfig as RustQuicClientConfig, QuicClientConfigBuilder, + QuicClientReconnectionConfig as RustQuicClientReconnectionConfig, TcpClientConfig as RustTcpClientConfig, TcpClientConfigBuilder, TcpClientReconnectionConfig as RustTcpClientReconnectionConfig, }; @@ -26,10 +28,12 @@ use pyo3::types::PyDelta; use pyo3_stub_gen::derive::{gen_stub_pyclass, gen_stub_pymethods}; use pyo3_stub_gen::impl_stub_type; use secrecy::SecretString; +use std::net::SocketAddr; use std::sync::Arc; use crate::duration::{ - duration_repr, iggy_duration_to_py_delta, py_delta_to_iggy_duration, reject_zero, + duration_repr, iggy_duration_to_py_delta, millis_repr, millis_to_py_delta, + py_delta_to_iggy_duration, py_delta_to_millis, reject_zero, }; /// The credentials replayed by the client every time it (re)connects. @@ -408,16 +412,469 @@ impl TcpConfig { } } +/// How the QUIC client reconnects after the connection to the server is lost. +#[gen_stub_pyclass] +#[pyclass(from_py_object)] +#[derive(Clone)] +pub struct QuicReconnectionConfig { + pub(crate) inner: RustQuicClientReconnectionConfig, +} + +#[gen_stub_pymethods] +#[pymethods] +impl QuicReconnectionConfig { + /// Constructs a reconnection policy. + /// + /// Args: + /// enabled: Whether to reconnect at all. Defaults to enabled. + /// max_retries: Redials of the configured server address after the first + /// attempt, or `None` for unlimited; `0` still makes that first + /// attempt. Unlike the TCP transport, QUIC redials the one address + /// it was configured with rather than walking a cluster roster, so + /// this counts dials. Defaults to unlimited, which means a call + /// awaited while the server is down never returns: `connect()` + /// waits inside the retry loop, as do `send_messages()` and + /// `poll_messages()` once auto-login is configured. Set a finite + /// number for request/reply style usage, so a call fails instead. + /// interval: Delay before each redial. Defaults to 1 second. + /// reestablish_after: Cooldown before redialing after a previously + /// successful connection, measured from when it was established, so + /// a session that outlived the interval is redialed at once. + /// Defaults to 5 seconds. + /// + /// Raises: + /// ValueError: If a duration is negative, if `max_retries` is outside the + /// range of an unsigned 32-bit integer, or if `interval` is zero. + #[new] + #[pyo3(signature = (*, enabled=None, max_retries=None, interval=None, reestablish_after=None))] + fn new( + #[gen_stub(override_type(type_repr = "builtins.bool | None"))] enabled: Option, + #[gen_stub(override_type(type_repr = "builtins.int | None"))] max_retries: Option, + #[gen_stub(override_type(type_repr = "datetime.timedelta | None", imports=("datetime")))] + interval: Option>, + #[gen_stub(override_type(type_repr = "datetime.timedelta | None", imports=("datetime")))] + reestablish_after: Option>, + ) -> PyResult { + let defaults = RustQuicClientReconnectionConfig::default(); + let enabled = enabled.unwrap_or(defaults.enabled); + let max_retries = max_retries + .map(|max_retries| { + u32::try_from(max_retries).map_err(|_| { + PyValueError::new_err(format!( + "'max_retries' must be between 0 and {}", + u32::MAX + )) + }) + }) + .transpose()?; + let interval = interval + .as_ref() + .map(py_delta_to_iggy_duration) + .transpose()? + .map(|interval| reject_zero(interval, "interval")) + .transpose()? + .unwrap_or(defaults.interval); + Ok(Self { + inner: RustQuicClientReconnectionConfig { + enabled, + max_retries, + interval, + reestablish_after: reestablish_after + .as_ref() + .map(py_delta_to_iggy_duration) + .transpose()? + .unwrap_or(defaults.reestablish_after), + }, + }) + } + + #[getter] + fn enabled(&self) -> bool { + self.inner.enabled + } + + #[gen_stub(override_return_type(type_repr = "builtins.int | None"))] + #[getter] + fn max_retries(&self) -> Option { + self.inner.max_retries + } + + #[gen_stub(override_return_type(type_repr = "datetime.timedelta", imports=("datetime")))] + #[getter] + fn interval<'a>(&self, py: Python<'a>) -> PyResult> { + iggy_duration_to_py_delta(py, self.inner.interval.get()) + } + + #[gen_stub(override_return_type(type_repr = "datetime.timedelta", imports=("datetime")))] + #[getter] + fn reestablish_after<'a>(&self, py: Python<'a>) -> PyResult> { + iggy_duration_to_py_delta(py, self.inner.reestablish_after) + } + + fn __repr__(&self) -> String { + let max_retries = match self.inner.max_retries { + Some(max_retries) => max_retries.to_string(), + None => "None".to_owned(), + }; + format!( + "QuicReconnectionConfig(enabled={}, max_retries={max_retries}, interval={}, reestablish_after={})", + python_bool(self.inner.enabled), + duration_repr(self.inner.interval.get()), + duration_repr(self.inner.reestablish_after), + ) + } +} + +/// quinn clamps `TransportConfig::initial_mtu` up to this floor rather than +/// rejecting a smaller value, so `QuicConfig` rejects it instead: otherwise the +/// getter would read back a value that is not the one actually in effect. +const QUINN_MIN_INITIAL_MTU: u16 = 1200; + +/// Configuration for the QUIC transport, accepted by `IggyClient(...)`. +/// +/// Every field is keyword-only and optional. +#[gen_stub_pyclass] +#[pyclass(from_py_object)] +#[derive(Clone)] +pub struct QuicConfig { + inner: Arc, +} + +impl QuicConfig { + /// The configuration in the shape `QuicClient::create` expects. + pub(crate) fn client_config(&self) -> Arc { + self.inner.clone() + } +} + +#[gen_stub_pymethods] +#[pymethods] +impl QuicConfig { + /// Constructs a QUIC configuration. + /// + /// Args: + /// server_address: `host:port` of the Iggy server. Defaults to `127.0.0.1:8080`. + /// client_address: `host:port` to bind the local UDP socket to. Defaults to + /// `127.0.0.1:0`, which binds to any available port. That exact value, + /// passed or defaulted, binds `[::1]:0` instead when `server_address` + /// resolves to IPv6, so the socket in use may not be the address read + /// back here. Any other value binds as given. + /// server_name: Server name used for the QUIC/TLS handshake. Defaults to + /// `localhost`. + /// auto_login: Credentials replayed on every connect. Defaults to `AutoLogin.disabled()`. + /// reconnection: Reconnection policy. Defaults to `QuicReconnectionConfig()`. + /// heartbeat_interval: Interval of heartbeats sent by the client. Defaults to 5 seconds. + /// response_buffer_size: Size of the response buffer in bytes. Defaults to 10 MB. + /// max_concurrent_bidi_streams: Maximum number of concurrent bidirectional + /// streams. Defaults to 10,000. + /// datagram_send_buffer_size: Size of the datagram send buffer in bytes. + /// Defaults to 100,000. + /// initial_mtu: Initial MTU in bytes. Defaults to 1200. + /// send_window: Send window size in bytes. Defaults to 100,000. + /// receive_window: Receive window size in bytes. Defaults to 100,000. + /// keep_alive_interval: Interval between QUIC keep-alive pings, or a zero + /// duration to disable them. Defaults to 5 seconds. + /// max_idle_timeout: How long the connection tolerates silence before it is + /// considered dead, or a zero duration to use quinn's own default (30 + /// seconds) instead, since `configure()` skips the setter entirely when + /// zero. Defaults to 10 seconds. + /// validate_certificate: Whether to validate the server certificate. Defaults + /// to disabled, unlike the TCP and WebSocket transports. + /// + /// Raises: + /// ValueError: If `server_address` or `client_address` is not a valid + /// `host:port` pair, if a duration is negative, if + /// `heartbeat_interval` is zero, if `keep_alive_interval` or + /// `max_idle_timeout` is not a whole number of milliseconds, if + /// `initial_mtu` is below quinn's minimum of 1200, or if a numeric + /// field is outside the range of its underlying wire type. + #[new] + #[pyo3(signature = ( + *, + server_address=None, + client_address=None, + server_name=None, + auto_login=None, + reconnection=None, + heartbeat_interval=None, + response_buffer_size=None, + max_concurrent_bidi_streams=None, + datagram_send_buffer_size=None, + initial_mtu=None, + send_window=None, + receive_window=None, + keep_alive_interval=None, + max_idle_timeout=None, + validate_certificate=None, + ))] + #[allow(clippy::too_many_arguments)] + fn new( + #[gen_stub(override_type(type_repr = "builtins.str | None"))] server_address: Option< + String, + >, + #[gen_stub(override_type(type_repr = "builtins.str | None"))] client_address: Option< + String, + >, + #[gen_stub(override_type(type_repr = "builtins.str | None"))] server_name: Option, + #[gen_stub(override_type(type_repr = "AutoLogin | None"))] auto_login: Option, + #[gen_stub(override_type(type_repr = "QuicReconnectionConfig | None"))] + reconnection: Option, + #[gen_stub(override_type(type_repr = "datetime.timedelta | None", imports=("datetime")))] + heartbeat_interval: Option>, + #[gen_stub(override_type(type_repr = "builtins.int | None"))] response_buffer_size: Option< + i64, + >, + #[gen_stub(override_type(type_repr = "builtins.int | None"))] + max_concurrent_bidi_streams: Option, + #[gen_stub(override_type(type_repr = "builtins.int | None"))] + datagram_send_buffer_size: Option, + #[gen_stub(override_type(type_repr = "builtins.int | None"))] initial_mtu: Option, + #[gen_stub(override_type(type_repr = "builtins.int | None"))] send_window: Option, + #[gen_stub(override_type(type_repr = "builtins.int | None"))] receive_window: Option, + #[gen_stub(override_type(type_repr = "datetime.timedelta | None", imports=("datetime")))] + keep_alive_interval: Option>, + #[gen_stub(override_type(type_repr = "datetime.timedelta | None", imports=("datetime")))] + max_idle_timeout: Option>, + #[gen_stub(override_type(type_repr = "builtins.bool | None"))] validate_certificate: Option< + bool, + >, + ) -> PyResult { + // The builder starts from `QuicClientConfig::default()`, and its `build()` + // trims and validates the server address whether or not one was set here. + let mut builder = QuicClientConfigBuilder::new(); + if let Some(server_address) = server_address { + builder = builder.with_server_address(server_address); + } + let mut inner = builder + .build() + .map_err(|e| PyValueError::new_err(e.to_string()))?; + if let Some(client_address) = client_address { + // Kept verbatim rather than normalized: `QuicClient::create` compares + // this against the literal default to decide whether to bind an IPv6 + // socket for an IPv6 server, and a rewritten string would not match. + client_address.parse::().map_err(|e| { + PyValueError::new_err(format!("'client_address' is not a valid 'host:port': {e}")) + })?; + inner.client_address = client_address; + } + if let Some(server_name) = server_name { + inner.server_name = server_name; + } + if let Some(auto_login) = auto_login { + inner.auto_login = auto_login.inner; + } + if let Some(reconnection) = reconnection { + inner.reconnection = reconnection.inner; + } + if let Some(heartbeat_interval) = heartbeat_interval { + inner.heartbeat_interval = reject_zero( + py_delta_to_iggy_duration(&heartbeat_interval)?, + "heartbeat_interval", + )?; + } + if let Some(response_buffer_size) = response_buffer_size { + inner.response_buffer_size = u64_param(response_buffer_size, "response_buffer_size")?; + } + if let Some(max_concurrent_bidi_streams) = max_concurrent_bidi_streams { + inner.max_concurrent_bidi_streams = + varint_param(max_concurrent_bidi_streams, "max_concurrent_bidi_streams")?; + } + if let Some(datagram_send_buffer_size) = datagram_send_buffer_size { + inner.datagram_send_buffer_size = + u64_param(datagram_send_buffer_size, "datagram_send_buffer_size")?; + } + if let Some(initial_mtu) = initial_mtu { + let initial_mtu = u16_param(initial_mtu, "initial_mtu")?; + if initial_mtu < QUINN_MIN_INITIAL_MTU { + return Err(PyValueError::new_err(format!( + "'initial_mtu' must be at least {QUINN_MIN_INITIAL_MTU}; quinn silently \ + raises anything smaller to that floor, so the getter would no longer \ + match the value actually in effect" + ))); + } + inner.initial_mtu = initial_mtu; + } + if let Some(send_window) = send_window { + inner.send_window = u64_param(send_window, "send_window")?; + } + if let Some(receive_window) = receive_window { + inner.receive_window = varint_param(receive_window, "receive_window")?; + } + if let Some(keep_alive_interval) = keep_alive_interval { + inner.keep_alive_interval = + py_delta_to_millis(&keep_alive_interval, "keep_alive_interval")?; + } + if let Some(max_idle_timeout) = max_idle_timeout { + inner.max_idle_timeout = py_delta_to_millis(&max_idle_timeout, "max_idle_timeout")?; + } + if let Some(validate_certificate) = validate_certificate { + inner.validate_certificate = validate_certificate; + } + + Ok(Self { + inner: Arc::new(inner), + }) + } + + #[getter] + fn server_address(&self) -> String { + self.inner.server_address.clone() + } + + #[getter] + fn client_address(&self) -> String { + self.inner.client_address.clone() + } + + #[getter] + fn server_name(&self) -> String { + self.inner.server_name.clone() + } + + #[getter] + fn auto_login(&self) -> AutoLogin { + AutoLogin { + inner: self.inner.auto_login.clone(), + } + } + + #[getter] + fn reconnection(&self) -> QuicReconnectionConfig { + QuicReconnectionConfig { + inner: self.inner.reconnection.clone(), + } + } + + #[gen_stub(override_return_type(type_repr = "datetime.timedelta", imports=("datetime")))] + #[getter] + fn heartbeat_interval<'a>(&self, py: Python<'a>) -> PyResult> { + iggy_duration_to_py_delta(py, self.inner.heartbeat_interval.get()) + } + + #[gen_stub(override_return_type(type_repr = "builtins.int"))] + #[getter] + fn response_buffer_size(&self) -> u64 { + self.inner.response_buffer_size + } + + #[gen_stub(override_return_type(type_repr = "builtins.int"))] + #[getter] + fn max_concurrent_bidi_streams(&self) -> u64 { + self.inner.max_concurrent_bidi_streams + } + + #[gen_stub(override_return_type(type_repr = "builtins.int"))] + #[getter] + fn datagram_send_buffer_size(&self) -> u64 { + self.inner.datagram_send_buffer_size + } + + #[gen_stub(override_return_type(type_repr = "builtins.int"))] + #[getter] + fn initial_mtu(&self) -> u16 { + self.inner.initial_mtu + } + + #[gen_stub(override_return_type(type_repr = "builtins.int"))] + #[getter] + fn send_window(&self) -> u64 { + self.inner.send_window + } + + #[gen_stub(override_return_type(type_repr = "builtins.int"))] + #[getter] + fn receive_window(&self) -> u64 { + self.inner.receive_window + } + + #[gen_stub(override_return_type(type_repr = "datetime.timedelta", imports=("datetime")))] + #[getter] + fn keep_alive_interval<'a>(&self, py: Python<'a>) -> PyResult> { + millis_to_py_delta(py, self.inner.keep_alive_interval) + } + + #[gen_stub(override_return_type(type_repr = "datetime.timedelta", imports=("datetime")))] + #[getter] + fn max_idle_timeout<'a>(&self, py: Python<'a>) -> PyResult> { + millis_to_py_delta(py, self.inner.max_idle_timeout) + } + + #[getter] + fn validate_certificate(&self) -> bool { + self.inner.validate_certificate + } + + fn __repr__(&self) -> String { + format!( + "QuicConfig(server_address={:?}, client_address={:?}, server_name={:?}, auto_login={}, reconnection={}, heartbeat_interval={}, response_buffer_size={}, max_concurrent_bidi_streams={}, datagram_send_buffer_size={}, initial_mtu={}, send_window={}, receive_window={}, keep_alive_interval={}, max_idle_timeout={}, validate_certificate={})", + self.inner.server_address, + self.inner.client_address, + self.inner.server_name, + self.auto_login().__repr__(), + self.reconnection().__repr__(), + duration_repr(self.inner.heartbeat_interval.get()), + self.inner.response_buffer_size, + self.inner.max_concurrent_bidi_streams, + self.inner.datagram_send_buffer_size, + self.inner.initial_mtu, + self.inner.send_window, + self.inner.receive_window, + millis_repr(self.inner.keep_alive_interval), + millis_repr(self.inner.max_idle_timeout), + python_bool(self.inner.validate_certificate), + ) + } +} + fn python_bool(value: bool) -> &'static str { if value { "True" } else { "False" } } -/// What `IggyClient(...)` accepts: a bare `host:port` or a full `TcpConfig`. +/// Converts a Python int to the unsigned 64-bit integer a QUIC transport +/// field expects, naming the parameter in the error so a caller can tell +/// which argument was out of range. The bound in the message is `i64::MAX` +/// rather than `u64::MAX` because pyo3 extracts the argument as an `i64` +/// first: anything above that never reaches here, raising `OverflowError` +/// on the way in. Every one of these fields is a buffer or window size, so +/// the unreachable half of the range has no practical use. +fn u64_param(value: i64, parameter: &str) -> PyResult { + u64::try_from(value).map_err(|_| { + PyValueError::new_err(format!("'{parameter}' must be between 0 and {}", i64::MAX)) + }) +} + +/// Converts a Python int to the unsigned 16-bit integer `initial_mtu` expects. +fn u16_param(value: i64, parameter: &str) -> PyResult { + u16::try_from(value).map_err(|_| { + PyValueError::new_err(format!("'{parameter}' must be between 0 and {}", u16::MAX)) + }) +} + +/// Converts a Python int to a `u64` that also fits `quinn::VarInt` (max +/// `2^62 - 1`), which `max_concurrent_bidi_streams` and `receive_window` are +/// narrowed into when the connection is configured. A `u64` in range for +/// `u64::MAX` but not `VarInt::MAX` would otherwise only fail there, as an +/// opaque `RuntimeError` instead of a `ValueError` naming the argument. +fn varint_param(value: i64, parameter: &str) -> PyResult { + const VARINT_MAX: u64 = (1u64 << 62) - 1; + let value = u64_param(value, parameter)?; + if value > VARINT_MAX { + return Err(PyValueError::new_err(format!( + "'{parameter}' must be between 0 and {VARINT_MAX}" + ))); + } + Ok(value) +} + +/// What `IggyClient(...)` accepts: a bare `host:port`, a full `TcpConfig`, or a +/// `QuicConfig` for the QUIC transport. #[derive(FromPyObject)] pub enum PyClientConfig { #[pyo3(transparent)] - Config(TcpConfig), + Tcp(TcpConfig), + #[pyo3(transparent)] + Quic(QuicConfig), #[pyo3(transparent, annotation = "str")] ServerAddress(String), } -impl_stub_type!(PyClientConfig = TcpConfig | String); +impl_stub_type!(PyClientConfig = TcpConfig | QuicConfig | String); diff --git a/foreign/python/src/duration.rs b/foreign/python/src/duration.rs index 1d448ad254..776df6604d 100644 --- a/foreign/python/src/duration.rs +++ b/foreign/python/src/duration.rs @@ -40,6 +40,34 @@ pub fn iggy_duration_to_py_delta( duration.get_duration().into_pyobject(py) } +/// Converts a Python timedelta to milliseconds, for fields the Rust SDK +/// stores as a raw millisecond count rather than an `IggyDuration` (e.g. +/// QUIC's `keep_alive_interval`/`max_idle_timeout`). Anything finer than a +/// millisecond is rejected rather than truncated, so the getter always reads +/// back the duration that is actually in effect. Zero is accepted for both of +/// those fields as a magic value (disables the keep-alive, or falls back to +/// quinn's own default), so a non-zero duration below 1ms is rejected with its +/// own message rather than collapsing into it. +pub fn py_delta_to_millis(delta: &Py, parameter: &str) -> PyResult { + let duration = py_delta_to_iggy_duration(delta)?.get_duration(); + if !duration.is_zero() && duration.as_millis() == 0 { + return Err(PyValueError::new_err(format!( + "'{parameter}' is non-zero but rounds down to 0ms; use a duration of at least 1ms, or exactly zero" + ))); + } + if duration.subsec_nanos() % 1_000_000 != 0 { + return Err(PyValueError::new_err(format!( + "'{parameter}' must be a whole number of milliseconds; anything finer is dropped by the QUIC transport, which stores it as a millisecond count" + ))); + } + Ok(duration.as_millis() as u64) +} + +/// The inverse of `py_delta_to_millis`. +pub fn millis_to_py_delta(py: Python<'_>, millis: u64) -> PyResult> { + Duration::from_millis(millis).into_pyobject(py) +} + /// Renders a duration the way it would be written in Python, so that a `__repr__` /// built from it can be pasted back into a constructor. pub fn duration_repr(duration: IggyDuration) -> String { @@ -53,6 +81,11 @@ pub fn duration_repr(duration: IggyDuration) -> String { } } +/// The `duration_repr` equivalent for a raw millisecond count. +pub fn millis_repr(millis: u64) -> String { + duration_repr(IggyDuration::new(Duration::from_millis(millis))) +} + /// Converts a duration for parameters that pace a loop, where zero means an /// unthrottled loop rather than "disabled". pub fn reject_zero(duration: IggyDuration, parameter: &str) -> PyResult { diff --git a/foreign/python/src/lib.rs b/foreign/python/src/lib.rs index d4397d5ab0..5da34876c5 100644 --- a/foreign/python/src/lib.rs +++ b/foreign/python/src/lib.rs @@ -31,7 +31,7 @@ mod user; mod user_headers; use client::IggyClient; -use config::{AutoLogin, TcpConfig, TcpReconnectionConfig}; +use config::{AutoLogin, QuicConfig, QuicReconnectionConfig, TcpConfig, TcpReconnectionConfig}; use consumer::{ AutoCommit, AutoCommitAfter, AutoCommitWhen, Consumer, ConsumerGroup, ConsumerGroupDetails, ConsumerGroupMember, IggyConsumer, ReceiveMessageIterator, @@ -58,6 +58,8 @@ fn apache_iggy(_py: Python, m: &Bound<'_, PyModule>) -> PyResult<()> { m.add_class::()?; m.add_class::()?; m.add_class::()?; + m.add_class::()?; + m.add_class::()?; m.add_class::()?; m.add_class::()?; m.add_class::()?; diff --git a/foreign/python/tests/test_connectivity.py b/foreign/python/tests/test_connectivity.py index 69516d74c1..3b4230705d 100644 --- a/foreign/python/tests/test_connectivity.py +++ b/foreign/python/tests/test_connectivity.py @@ -40,6 +40,7 @@ async def test_client_not_none(self, iggy_client: IggyClient): "iggy+http://iggy:iggy@127.0.0.1:3000?heartbeat_interval=5s&retries=3", "iggy+ws://iggy:iggy@127.0.0.1:8092", "iggy+ws://iggy:iggy@127.0.0.1:8092?heartbeat_interval=5s&reconnection_retries=3&reconnection_interval=1s&reestablish_after=5s&read_buffer_size=4096&write_buffer_size=4096&max_write_buffer_size=8192&max_message_size=16384&max_frame_size=16384&accept_unmasked_frames=false&tls_domain=localhost&tls_ca_file=unused.pem&tls_validate_certificate=false&tls=false", + "iggy+quic://iggy:iggy@127.0.0.1:8080?reconnection_max_retries=0", ], ) @pytest.mark.asyncio @@ -77,7 +78,6 @@ async def test_valid_connection_string(self, connection_string: str): "iggy+tcp://iggy:iggy@{host}:{port}?invalid_option=value", "Invalid connection string", ), - ("iggy+quic://iggy:iggy@127.0.0.1:8080", "Cannot create endpoint"), ], ) def test_invalid_connection_string(self, invalid_value: str, expected_error: str): diff --git a/foreign/python/tests/test_quic_config.py b/foreign/python/tests/test_quic_config.py new file mode 100644 index 0000000000..d0fdc3f7b6 --- /dev/null +++ b/foreign/python/tests/test_quic_config.py @@ -0,0 +1,434 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +""" +Tests for the QUIC client configuration surface. + +`QuicConfig` and `QuicReconnectionConfig` mirror the Rust SDK types the same +way `TcpConfig`/`TcpReconnectionConfig` do, so most of these assert that a +value set from Python survives to the getters and that unset fields fall +back to the Rust defaults. `AutoLogin` is transport-agnostic and already +covered by `test_client_config.py`. +""" + +import ast +from collections.abc import Callable +from datetime import timedelta + +import pytest + +from apache_iggy import AutoLogin, IggyClient, QuicConfig, QuicReconnectionConfig + +from .utils import get_quic_server_config, wait_for_ping + + +@pytest.mark.unit +class TestQuicReconnectionConfig: + """Test the reconnection policy.""" + + def test_defaults_match_the_rust_sdk(self): + """Test that an unconfigured policy reconnects forever, one second apart.""" + reconnection = QuicReconnectionConfig() + + assert reconnection.enabled is True + assert reconnection.max_retries is None + assert reconnection.interval == timedelta(seconds=1) + assert reconnection.reestablish_after == timedelta(seconds=5) + + def test_every_field_round_trips(self): + """Test that each configured field is readable back unchanged.""" + reconnection = QuicReconnectionConfig( + enabled=False, + max_retries=10, + interval=timedelta(milliseconds=250), + reestablish_after=timedelta(seconds=30), + ) + + assert reconnection.enabled is False + assert reconnection.max_retries == 10 + assert reconnection.interval == timedelta(milliseconds=250) + assert reconnection.reestablish_after == timedelta(seconds=30) + + def test_arguments_are_keyword_only(self): + """Test that the adjacent flags cannot be passed positionally.""" + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-count + QuicReconnectionConfig(True) + + @pytest.mark.parametrize( + "construct", + [ + lambda duration: QuicReconnectionConfig(interval=duration), + lambda duration: QuicReconnectionConfig(reestablish_after=duration), + ], + ids=["interval", "reestablish_after"], + ) + @pytest.mark.parametrize( + "negative", + [timedelta(microseconds=-1), timedelta(seconds=-1), timedelta(days=-1)], + ) + def test_negative_duration_is_rejected( + self, + construct: Callable[[timedelta], QuicReconnectionConfig], + negative: timedelta, + ): + """Test that a negative duration fails at construction, not at connect.""" + with pytest.raises(ValueError, match="negative"): + construct(negative) + + @pytest.mark.parametrize("out_of_range", [-1, 2**32]) + def test_out_of_range_max_retries_is_rejected(self, out_of_range: int): + """Test that a retry count outside the wire range names the argument. + + The conversion pyo3 does on its own raises OverflowError, which is not a + ValueError and so escapes the handler a caller wraps construction in. + """ + with pytest.raises(ValueError, match="max_retries"): + QuicReconnectionConfig(max_retries=out_of_range) + + def test_zero_reestablish_after_is_allowed(self): + """Test that a zero cooldown is legal and readable back.""" + reconnection = QuicReconnectionConfig(reestablish_after=timedelta(0)) + + assert reconnection.reestablish_after == timedelta(0) + + @pytest.mark.parametrize( + "kwargs", + [ + {}, + {"max_retries": 5}, + {"enabled": False}, + ], + ids=["unlimited_retries", "bounded_retries", "reconnection_disabled"], + ) + def test_zero_interval_is_rejected(self, kwargs: dict): + """Test that a zero interval fails whatever the retry policy is. + + The interval is a delay between passes, so zero reconnects in a + continuous loop. + """ + with pytest.raises(ValueError, match=r"interval.*must not be zero"): + QuicReconnectionConfig(interval=timedelta(0), **kwargs) + + def test_very_long_interval_round_trips(self): + """Test that an interval beyond 68 years survives the i32 boundary.""" + reconnection = QuicReconnectionConfig(interval=timedelta(days=30_000)) + + assert reconnection.interval == timedelta(days=30_000) + + def test_maximum_interval_round_trips(self): + """Test that the largest timedelta survives the day conversion.""" + reconnection = QuicReconnectionConfig(interval=timedelta(days=999_999_999)) + + assert reconnection.interval == timedelta(days=999_999_999) + + +@pytest.mark.unit +class TestQuicConfig: + """Test the transport configuration.""" + + def test_defaults_match_the_rust_sdk(self): + """Test that an unconfigured transport matches the Rust SDK defaults.""" + config = QuicConfig() + + assert config.server_address == "127.0.0.1:8080" + assert config.client_address == "127.0.0.1:0" + assert config.server_name == "localhost" + assert config.auto_login.enabled is False + assert config.reconnection.enabled is True + assert config.heartbeat_interval == timedelta(seconds=5) + assert config.response_buffer_size == 10_000_000 + assert config.max_concurrent_bidi_streams == 10_000 + assert config.datagram_send_buffer_size == 100_000 + assert config.initial_mtu == 1200 + assert config.send_window == 100_000 + assert config.receive_window == 100_000 + assert config.keep_alive_interval == timedelta(milliseconds=5000) + assert config.max_idle_timeout == timedelta(milliseconds=10_000) + assert config.validate_certificate is False + + def test_every_field_round_trips(self): + """Test that each configured field is readable back unchanged.""" + config = QuicConfig( + server_address="127.0.0.1:8081", + client_address="127.0.0.1:9000", + server_name="example.com", + auto_login=AutoLogin.username_password("iggy", "iggy"), + reconnection=QuicReconnectionConfig(max_retries=3), + heartbeat_interval=timedelta(seconds=15), + response_buffer_size=5_000_000, + max_concurrent_bidi_streams=500, + datagram_send_buffer_size=50_000, + initial_mtu=1400, + send_window=200_000, + receive_window=200_000, + keep_alive_interval=timedelta(seconds=2), + max_idle_timeout=timedelta(seconds=20), + validate_certificate=True, + ) + + assert config.server_address == "127.0.0.1:8081" + assert config.client_address == "127.0.0.1:9000" + assert config.server_name == "example.com" + assert config.auto_login.username == "iggy" + assert config.reconnection.max_retries == 3 + assert config.heartbeat_interval == timedelta(seconds=15) + assert config.response_buffer_size == 5_000_000 + assert config.max_concurrent_bidi_streams == 500 + assert config.datagram_send_buffer_size == 50_000 + assert config.initial_mtu == 1400 + assert config.send_window == 200_000 + assert config.receive_window == 200_000 + assert config.keep_alive_interval == timedelta(seconds=2) + assert config.max_idle_timeout == timedelta(seconds=20) + assert config.validate_certificate is True + + def test_arguments_are_keyword_only(self): + """Test that the address cannot be passed positionally.""" + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-count + QuicConfig("127.0.0.1:8080") + + def test_repr_hides_the_password(self): + """Test that the password does not leak through repr.""" + config = QuicConfig(auto_login=AutoLogin.username_password("iggy", "secret")) + + assert "secret" not in repr(config) + + def test_repr_shows_every_field_as_python(self): + """Test that repr covers the QUIC-specific fields and parses as Python.""" + config = QuicConfig( + heartbeat_interval=timedelta(seconds=15), + keep_alive_interval=timedelta(seconds=2), + max_idle_timeout=timedelta(seconds=20), + validate_certificate=True, + ) + + printed = repr(config) + + assert "validate_certificate=True" in printed + assert "heartbeat_interval=datetime.timedelta(seconds=15)" in printed + assert "keep_alive_interval=datetime.timedelta(seconds=2)" in printed + assert "max_idle_timeout=datetime.timedelta(seconds=20)" in printed + ast.parse(printed) + + @pytest.mark.parametrize( + "invalid_address", + ["", "127.0.0.1", "127.0.0.1:not-a-port", "127.0.0.1:70000", "::1:8080"], + ) + def test_invalid_server_address_is_rejected(self, invalid_address: str): + """Test that a malformed address fails at construction, not at connect.""" + with pytest.raises(ValueError): + QuicConfig(server_address=invalid_address) + + @pytest.mark.parametrize( + "invalid_address", + ["", "127.0.0.1", "127.0.0.1:not-a-port", "127.0.0.1:70000", "localhost:0"], + ) + def test_invalid_client_address_is_rejected(self, invalid_address: str): + """Test that a malformed bind address fails at construction. + + `QuicClient::create` parses this as a `SocketAddr`, so a hostname is + rejected alongside the malformed forms: without the eager check the + failure would surface as a `RuntimeError` from `IggyClient(...)` + instead, which is not a `ValueError` and so escapes the handler a + caller wraps construction in. + """ + with pytest.raises(ValueError, match="client_address"): + QuicConfig(client_address=invalid_address) + + def test_negative_heartbeat_interval_is_rejected(self): + """Test that a negative heartbeat interval fails at construction.""" + with pytest.raises(ValueError, match="negative"): + QuicConfig(heartbeat_interval=timedelta(seconds=-3)) + + def test_zero_heartbeat_interval_is_rejected(self): + """Test that a zero heartbeat interval fails at construction. + + Nothing downstream reads zero as "disabled"; it heartbeats in a + continuous loop for as long as the client lives. + """ + with pytest.raises(ValueError, match=r"heartbeat_interval.*must not be zero"): + QuicConfig(heartbeat_interval=timedelta(0)) + + @pytest.mark.parametrize( + ("field", "out_of_range"), + [ + ("response_buffer_size", -1), + ("max_concurrent_bidi_streams", -1), + ("datagram_send_buffer_size", -1), + ("send_window", -1), + ("receive_window", -1), + ("initial_mtu", -1), + ("initial_mtu", 2**16), + ("max_concurrent_bidi_streams", 2**62), + ("receive_window", 2**62), + ], + ) + def test_out_of_range_numeric_field_is_rejected( + self, field: str, out_of_range: int + ): + """Test that a numeric field outside its wire type's range names itself. + + `max_concurrent_bidi_streams` and `receive_window` fit `u64`, but + quinn narrows them further into a `VarInt` (max `2**62 - 1`), so + `2**62` fits the wire type and must still be rejected. + """ + with pytest.raises(ValueError, match=field): + # pyrefly: ignore # bad-argument-type + QuicConfig(**{field: out_of_range}) + + @pytest.mark.parametrize("field", ["keep_alive_interval", "max_idle_timeout"]) + def test_duration_rounding_down_to_zero_millis_is_rejected(self, field: str): + """Test that a non-zero sub-millisecond duration names itself. + + Both fields are raw millisecond counts to the Rust SDK where zero is a + magic value (disables the keep-alive, or falls back to quinn's own + default), so a duration that rounds down to zero would silently mean + something other than what was asked for. + """ + with pytest.raises(ValueError, match=rf"{field}.*rounds down to 0ms"): + # pyrefly: ignore # bad-argument-type + QuicConfig(**{field: timedelta(microseconds=500)}) + + @pytest.mark.parametrize("field", ["keep_alive_interval", "max_idle_timeout"]) + def test_sub_millisecond_precision_is_rejected(self, field: str): + """Test that a duration with a sub-millisecond remainder is refused. + + The Rust SDK stores both fields as a millisecond count, so the + remainder would be dropped and the getter would read back a different + duration than the one that was passed in. + """ + with pytest.raises(ValueError, match=rf"{field}.*whole number of milliseconds"): + # pyrefly: ignore # bad-argument-type + QuicConfig(**{field: timedelta(milliseconds=1, microseconds=500)}) + + @pytest.mark.parametrize("field", ["keep_alive_interval", "max_idle_timeout"]) + def test_exact_zero_duration_is_allowed(self, field: str): + """Test that an exact zero duration is still legal for these fields.""" + # pyrefly: ignore # bad-argument-type + config = QuicConfig(**{field: timedelta(0)}) + + assert getattr(config, field) == timedelta(0) + + @pytest.mark.parametrize("field", ["keep_alive_interval", "max_idle_timeout"]) + def test_whole_millisecond_duration_is_allowed(self, field: str): + """Test that a duration that is not a whole number of seconds round-trips. + + Every other duration these fields accept here is second-aligned, so a + check tightened to whole seconds would otherwise pass the suite. + """ + # pyrefly: ignore # bad-argument-type + config = QuicConfig(**{field: timedelta(milliseconds=1500)}) + + assert getattr(config, field) == timedelta(milliseconds=1500) + + def test_initial_mtu_below_quinns_minimum_is_rejected(self): + """Test that an initial_mtu below 1200 fails at construction. + + quinn silently raises anything smaller to that floor instead of + rejecting it, so accepting it here would let the getter read back a + value that is not the one actually in effect on the connection. + """ + with pytest.raises(ValueError, match="initial_mtu"): + QuicConfig(initial_mtu=1199) + + def test_initial_mtu_at_quinns_minimum_is_allowed(self): + """Test that exactly 1200, quinn's own floor, is accepted.""" + config = QuicConfig(initial_mtu=1200) + + assert config.initial_mtu == 1200 + + +@pytest.mark.unit +class TestQuicClientConstruction: + """Test that `IggyClient(...)` accepts a `QuicConfig`.""" + + def test_accepts_a_config(self): + """Test that a client can be built from a config object.""" + assert IggyClient(QuicConfig(server_address="127.0.0.1:8080")) is not None + + def test_accepts_the_default_config(self): + """Test that an explicit default `QuicConfig` is accepted.""" + assert IggyClient(QuicConfig()) is not None + + +@pytest.mark.integration +class TestAutoLoginAgainstServer: + """Test that configured credentials are actually replayed on connect.""" + + @pytest.mark.asyncio + async def test_auto_login_authenticates_without_login_user(self, unique_name): + """Test that a privileged call succeeds without a manual login_user().""" + host, port = get_quic_server_config() + + client = IggyClient( + QuicConfig( + server_address=f"{host}:{port}", + auto_login=AutoLogin.username_password("iggy", "iggy"), + # The default reconnection policy retries forever: a missing + # listener would stall this test for the full 30s pytest + # timeout instead of failing fast. + reconnection=QuicReconnectionConfig(enabled=False), + ) + ) + await client.connect() + await wait_for_ping(client) + + stream_name = unique_name() + await client.create_stream(stream_name) + assert await client.get_stream(stream_name) is not None + + @pytest.mark.asyncio + async def test_without_auto_login_a_privileged_call_is_unauthenticated( + self, unique_name + ): + """Test that the same call fails when no credentials are configured.""" + host, port = get_quic_server_config() + + client = IggyClient( + QuicConfig( + server_address=f"{host}:{port}", + # The default reconnection policy retries forever: a missing + # listener would stall this test for the full 30s pytest + # timeout instead of failing fast. + reconnection=QuicReconnectionConfig(enabled=False), + ) + ) + await client.connect() + await wait_for_ping(client) + + with pytest.raises(RuntimeError): + await client.create_stream(unique_name()) + + @pytest.mark.asyncio + async def test_wrong_auto_login_credentials_fail(self): + """Test that bad configured credentials surface as a connect failure.""" + host, port = get_quic_server_config() + + client = IggyClient( + QuicConfig( + server_address=f"{host}:{port}", + auto_login=AutoLogin.username_password("iggy", "invalid-password"), + reconnection=QuicReconnectionConfig(enabled=False), + ) + ) + + # A bare RuntimeError would also match "Cannot establish connection", + # which is what a missing listener raises on this reconnection policy. + with pytest.raises(RuntimeError, match="Invalid credentials"): + await client.connect() diff --git a/foreign/python/tests/utils.py b/foreign/python/tests/utils.py index b37a53831e..cdede9f32f 100644 --- a/foreign/python/tests/utils.py +++ b/foreign/python/tests/utils.py @@ -33,15 +33,19 @@ MAX_PASSWORD_BYTES = 100 -def get_server_config() -> tuple[str, int]: +def get_transport_config(port_env_var: str, default_port: int) -> tuple[str, int]: """ - Get server configuration from environment variables or defaults. + Get transport-specific server configuration from environment variables or defaults. + + Args: + port_env_var: Name of the environment variable holding the port. + default_port: Port to use if the environment variable is not set. Returns: tuple: (host, port) for the Iggy server """ host = os.environ.get("IGGY_SERVER_HOST", "127.0.0.1") - port = int(os.environ.get("IGGY_SERVER_TCP_PORT", "8090")) + port = int(os.environ.get(port_env_var, str(default_port))) # Convert hostname to IP address for the Rust client if host not in ("127.0.0.1", "localhost"): @@ -58,6 +62,26 @@ def get_server_config() -> tuple[str, int]: return host, port +def get_server_config() -> tuple[str, int]: + """ + Get TCP server configuration from environment variables or defaults. + + Returns: + tuple: (host, port) for the Iggy server + """ + return get_transport_config("IGGY_SERVER_TCP_PORT", 8090) + + +def get_quic_server_config() -> tuple[str, int]: + """ + Get QUIC server configuration from environment variables or defaults. + + Returns: + tuple: (host, port) for the Iggy server + """ + return get_transport_config("IGGY_SERVER_QUIC_PORT", 8080) + + def wait_for_server(host: str, port: int, timeout: int = 60, interval: int = 2) -> None: """ Wait for the server to become available. From 03ca396bb33f8dfdc54c5b390d073b548d439de5 Mon Sep 17 00:00:00 2001 From: Faisal Ahmed <42486737+felixfaisal@users.noreply.github.com> Date: Mon, 7 Sep 2026 18:48:26 +0530 Subject: [PATCH 078/182] feat(python): add stream listing, update, delete, and purge (#3701) Closes #3520 --------- Co-authored-by: spetz --- bdd/python/pylock.toml | 346 +++++++++++----------- bdd/python/uv.lock | 2 - examples/python/pylock.toml | 81 +++-- examples/python/uv.lock | 2 - foreign/python/apache_iggy.pyi | 142 +++++++++ foreign/python/pylock.toml | 37 +-- foreign/python/pyproject.toml | 2 - foreign/python/src/client.rs | 138 ++++++++- foreign/python/src/lib.rs | 3 +- foreign/python/src/options.rs | 1 + foreign/python/src/stream.rs | 78 ++++- foreign/python/src/topic.rs | 19 ++ foreign/python/tests/conftest.py | 27 +- foreign/python/tests/test_stream.py | 444 +++++++++++++++++++++++++++- foreign/python/uv.lock | 26 -- licenserc.toml | 1 + 16 files changed, 1053 insertions(+), 296 deletions(-) diff --git a/bdd/python/pylock.toml b/bdd/python/pylock.toml index 4d2968d096..e9f2d61c2c 100644 --- a/bdd/python/pylock.toml +++ b/bdd/python/pylock.toml @@ -1,22 +1,5 @@ -# Licensed to the Apache Software Foundation (ASF) under one -# or more contributor license agreements. See the NOTICE file -# distributed with this work for additional information -# regarding copyright ownership. The ASF licenses this file -# to you under the Apache License, Version 2.0 (the -# "License"); you may not use this file except in compliance -# with the License. You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, -# software distributed under the License is distributed on an -# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY -# KIND, either express or implied. See the License for the -# specific language governing permissions and limitations -# under the License. - # This file was autogenerated by uv via the following command: -# uv export -o pylock.toml +# uv export --locked --format pylock.toml --output-file pylock.toml --all-extras lock-version = "1.0" created-by = "uv" requires-python = ">=3.10" @@ -47,115 +30,130 @@ wheels = [ [[packages]] name = "coverage" -version = "7.14.0" +version = "7.15.4" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/23/7f/d0720730a397a999ffc0fd3f5bebef347338e3a47b727da66fbb228e2ff2/coverage-7.14.0.tar.gz", upload-time = 2026-05-10T18:02:31Z, size = 919489, hashes = { sha256 = "057a6af2f160a85384cde4ab36f0d2777bae1057bae255f95413cdd382aa5c74" } } +sdist = { url = "https://files.pythonhosted.org/packages/be/c3/4f2195f512fb172aa425a8803a874b2baa9ba7f80ff7b6080998761fc701/coverage-7.15.4.tar.gz", upload-time = 2026-08-06T13:50:24Z, size = 936952, hashes = { sha256 = "0548198fff07ccf4faf469520bce1c2eceb1ce3e62891921138dec10907f9d00" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/59/9d/7c83ef51c3eb495f10010094e661833588b7709946da634c8b66520b97c7/coverage-7.14.0-cp310-cp310-macosx_10_9_x86_64.whl", upload-time = 2026-05-10T17:59:23Z, size = 219668, hashes = { sha256 = "84c32d90bf4537f0e7b4dec9aaa9a938fb8205136b9d2ecf4d7629d5262dc075" } }, - { url = "https://files.pythonhosted.org/packages/24/34/898546aefbd28f0af131201d0dc852c9e976f817bd7d5bfb8dc4e02863bb/coverage-7.14.0-cp310-cp310-macosx_11_0_arm64.whl", upload-time = 2026-05-10T17:59:26Z, size = 220192, hashes = { sha256 = "7c843572c605ab51cfdb5c6b5f2586e2a8467c0d28eca4bdef4ec70c5fecbd82" } }, - { url = "https://files.pythonhosted.org/packages/df/4a/b457c88aca72b0df13a98167ebd5d947135ccd9881ea88ce6a570e13aa9b/coverage-7.14.0-cp310-cp310-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-05-10T17:59:27Z, size = 246932, hashes = { sha256 = "0c451757d3fa2603354fdc789b5e58a0e327a117c370a40e3476ba4eabab228c" } }, - { url = "https://files.pythonhosted.org/packages/b5/d9/92600e89486fd074c50f0117422b2c9592c3e144e2f25bd5ac0bc62bc7a0/coverage-7.14.0-cp310-cp310-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-05-10T17:59:29Z, size = 248762, hashes = { sha256 = "3fd43f0616e765ab78d069cf8358def7363957a45cee446d65c502dcfeea7893" } }, - { url = "https://files.pythonhosted.org/packages/0d/e1/9ea1eb9c311da7f15853559dc1d9d82bef88ecd3e59fbeb51f16bc2ffa91/coverage-7.14.0-cp310-cp310-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-05-10T17:59:31Z, size = 250625, hashes = { sha256 = "731e535b1498b27d13594a0527a79b0510867b0ad891532be41cb883f2128e20" } }, - { url = "https://files.pythonhosted.org/packages/a5/03/57afca1b8106f8549a5329139315041fe166d6099bd9381346b9430dfbd1/coverage-7.14.0-cp310-cp310-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-05-10T17:59:32Z, size = 252539, hashes = { sha256 = "c7492f2d493b976941c7ca050f273cbda2f43c381124f7586a3e3c16d1804fec" } }, - { url = "https://files.pythonhosted.org/packages/57/5e/2e9fc63c9928119c1dbae02222be51407d3e7ebac5811ebbda4af3557795/coverage-7.14.0-cp310-cp310-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-05-10T17:59:34Z, size = 247636, hashes = { sha256 = "dc38367eaa2abb1b766ac333142bce7655335a73537f5c8b75aaa89c2b987757" } }, - { url = "https://files.pythonhosted.org/packages/f0/e2/0b7898cda21041cc67546e19b80ba66cbbb47cbece52a76a5904de6a3aaf/coverage-7.14.0-cp310-cp310-musllinux_1_2_aarch64.whl", upload-time = 2026-05-10T17:59:36Z, size = 248666, hashes = { sha256 = "0a951308cde22cf77f953955a754d04dccb57fe3bb8e345d685778ed9fc1632a" } }, - { url = "https://files.pythonhosted.org/packages/d6/e3/d33662a2fdaef23229c15921f39c84ec38441f3069ba26e134ed402c833b/coverage-7.14.0-cp310-cp310-musllinux_1_2_i686.whl", upload-time = 2026-05-10T17:59:38Z, size = 246670, hashes = { sha256 = "fab3877e4ebb06bd9d4d4d00ee53309ee5478e66873c66a382272e3ee33eb7ea" } }, - { url = "https://files.pythonhosted.org/packages/99/b2/533942c3bfbf6770b5c32d7f2ff029fe013dba31f3fe8b45cabbb250365e/coverage-7.14.0-cp310-cp310-musllinux_1_2_ppc64le.whl", upload-time = 2026-05-10T17:59:39Z, size = 250484, hashes = { sha256 = "b812eb847b19876ebf33fb6c4f11819af05ab6050b0bfa1bc53412ae81779adb" } }, - { url = "https://files.pythonhosted.org/packages/d8/00/15acbad83a96de13c73831486c7627bfed73dfaec53b04e4a6315edf3fd8/coverage-7.14.0-cp310-cp310-musllinux_1_2_riscv64.whl", upload-time = 2026-05-10T17:59:41Z, size = 246942, hashes = { sha256 = "d9c8ef6ed820c433de075657d72dda1f89a2984955e58b8a75feb3f184250218" } }, - { url = "https://files.pythonhosted.org/packages/70/db/cef0228de493f2c740c760a9057a61d00c6849480073b70a75b87c7d4bab/coverage-7.14.0-cp310-cp310-musllinux_1_2_x86_64.whl", upload-time = 2026-05-10T17:59:43Z, size = 247544, hashes = { sha256 = "d128b1bba9361fbaaf6a19e179e6cfd6a9103ce0c0555876f72780acc93efd85" } }, - { url = "https://files.pythonhosted.org/packages/77/a0/d9ef8e148f3025c2ae8401d77cda1502b6d2a4d8102603a8af31460aedb6/coverage-7.14.0-cp310-cp310-win32.whl", upload-time = 2026-05-10T17:59:44Z, size = 222285, hashes = { sha256 = "65f267ca1370726ec2c1aa38bbe4df9a71a740f22878d2d4bf59d71a4cd8d323" } }, - { url = "https://files.pythonhosted.org/packages/85/c0/30c454c7d3cf47b2805d4e06f12443f5eece8a5d030d3b0350e7b74ecb49/coverage-7.14.0-cp310-cp310-win_amd64.whl", upload-time = 2026-05-10T17:59:46Z, size = 223215, hashes = { sha256 = "b34ece8065914f938ed7f2c5872bb865336977a52919149846eac3744327267a" } }, - { url = "https://files.pythonhosted.org/packages/fc/e4/649c8d4f7f1709b6dbfc474358aa1bba02f67bcd52e2fec291a5014006cd/coverage-7.14.0-cp311-cp311-macosx_10_9_x86_64.whl", upload-time = 2026-05-10T17:59:48Z, size = 219795, hashes = { sha256 = "6a78e2a9d9c5e3b8d4ab9b9d28c985ea66fced0a7d7c2aec1f216e03a2011480" } }, - { url = "https://files.pythonhosted.org/packages/7f/8d/46692d24b3f395d4cbf17bfcc57136b4f2f9c0c0df864b0bddfc1d71a014/coverage-7.14.0-cp311-cp311-macosx_11_0_arm64.whl", upload-time = 2026-05-10T17:59:49Z, size = 220299, hashes = { sha256 = "a1816c505187592dcd1c5a5f226601a549f70365fbd00930ac88b0c225b76bb4" } }, - { url = "https://files.pythonhosted.org/packages/12/c2/a40f5cb295bbcbb697a76947a56081c494c61950366294ee426ffe261099/coverage-7.14.0-cp311-cp311-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-05-10T17:59:51Z, size = 250721, hashes = { sha256 = "d8e1762f0e9cbc26ec315471e7b47855218e833cd5a032d706fbf43845d878c7" } }, - { url = "https://files.pythonhosted.org/packages/fd/35/202235eb5c3c14c212462cd91d61b7386bf8fc44bc7a77f4742d2a69174b/coverage-7.14.0-cp311-cp311-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-05-10T17:59:53Z, size = 252633, hashes = { sha256 = "9336e23e8bb3a3925398261385e2a1533957d3e760e91070dcb0e98bfa514eed" } }, - { url = "https://files.pythonhosted.org/packages/bb/80/5f596e8995785124ee191c42535664c5e62c65995b66f4ca21e28ae04c81/coverage-7.14.0-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-05-10T17:59:55Z, size = 254743, hashes = { sha256 = "9cd1169b2230f9cbe9c638ba38022ed7a2b1e641cc07f7cea0365e4be2a74980" } }, - { url = "https://files.pythonhosted.org/packages/1e/6d/0d178825be2350f0adb27984d0aa7cf84bbdab201f6fb926b535d23a8f5f/coverage-7.14.0-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-05-10T17:59:56Z, size = 256700, hashes = { sha256 = "d1bb3543b58fea74d2cd1abc4054cc927e4724687cb4560cd2ed88d2c7d820c0" } }, - { url = "https://files.pythonhosted.org/packages/19/5b/9e549c2f6e9dfea472adadba06c294e64735dabc2dd19015fac082095013/coverage-7.14.0-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-05-10T17:59:57Z, size = 250854, hashes = { sha256 = "a93bac2cb577ef60074999ed56d8a1535894398e2ed920d4185c3ec0c8864742" } }, - { url = "https://files.pythonhosted.org/packages/3d/1c/b94f9f5f36396021ee2f62c5834b12e6a3d31f0bed5d6fc6d1c3caec087c/coverage-7.14.0-cp311-cp311-musllinux_1_2_aarch64.whl", upload-time = 2026-05-10T17:59:59Z, size = 252433, hashes = { sha256 = "5904abf7e18cddc463219b17552229650c6b79e061d31a1059283051169cf7d5" } }, - { url = "https://files.pythonhosted.org/packages/b5/cb/d192cd8e1345eccabc32016f2d39072ecd10cb4f4b983ed8d0ebdeaf00dc/coverage-7.14.0-cp311-cp311-musllinux_1_2_i686.whl", upload-time = 2026-05-10T18:00:01Z, size = 250494, hashes = { sha256 = "741f57cddc9004a8c81b084660215f33a6b597dbe62c31386b983ee26310e327" } }, - { url = "https://files.pythonhosted.org/packages/53/c5/aac9f460a41d835dbddef1d377f105f6ac2311d0f3c1588e9f51046d8813/coverage-7.14.0-cp311-cp311-musllinux_1_2_ppc64le.whl", upload-time = 2026-05-10T18:00:03Z, size = 254261, hashes = { sha256 = "664123feb0929d7affc135717dbd70d61d98688a08ab1e5ba464739620c6252d" } }, - { url = "https://files.pythonhosted.org/packages/23/aa/7af7c0081980a9cb3d289c5a435a4b7657dcecbd128e25c580e6a50389b5/coverage-7.14.0-cp311-cp311-musllinux_1_2_riscv64.whl", upload-time = 2026-05-10T18:00:05Z, size = 250216, hashes = { sha256 = "c83d2399a51bbec8429266905d33616f04bc5726b1138c35844d5fcd896b2e20" } }, - { url = "https://files.pythonhosted.org/packages/35/60/a4257538ce2f6b978aeb51870d6c4208c510928a03db7e0339bb625dccb7/coverage-7.14.0-cp311-cp311-musllinux_1_2_x86_64.whl", upload-time = 2026-05-10T18:00:06Z, size = 251125, hashes = { sha256 = "bcb2e855b87321259a037429288ae85216d191c74de3e79bf57cd2bc0761992c" } }, - { url = "https://files.pythonhosted.org/packages/a1/ab/f91af47642ec1aa53490e835a95847168d9c77fc39aa58527604c051e145/coverage-7.14.0-cp311-cp311-win32.whl", upload-time = 2026-05-10T18:00:08Z, size = 222300, hashes = { sha256 = "731dc15b385ac52289743d476245b61e1a2927e803bef655b52bc3b2a75a21f3" } }, - { url = "https://files.pythonhosted.org/packages/f0/f0/a71ddbd874431e7a7cd96071f0c331cfbbad07704833c765d24ffbab8a67/coverage-7.14.0-cp311-cp311-win_amd64.whl", upload-time = 2026-05-10T18:00:10Z, size = 223241, hashes = { sha256 = "bfb0ed8ec5d25e93face268115d7964db9df8b9aae8edcde9ec6b16c726a7cc1" } }, - { url = "https://files.pythonhosted.org/packages/d8/6e/d9d312a5151a96cd110efee32efc3fc97b01ebd86203fe618ccb29cf4c92/coverage-7.14.0-cp311-cp311-win_arm64.whl", upload-time = 2026-05-10T18:00:12Z, size = 221908, hashes = { sha256 = "7ebb1c6df9f78046a1b1e0a89674cd4bf73b7c648914eebcf976a57fd99a5627" } }, - { url = "https://files.pythonhosted.org/packages/09/1e/2f996b2c8415cbb6f54b0f5ec1ee850c96d7911961afb4fc05f4a89d8c58/coverage-7.14.0-cp312-cp312-macosx_10_13_x86_64.whl", upload-time = 2026-05-10T18:00:13Z, size = 219967, hashes = { sha256 = "7ffd19fc8aed057fd686a17a4935eef5f9859d69208f96310e893e64b9b6ccf5" } }, - { url = "https://files.pythonhosted.org/packages/34/23/35c7aea1274aef7525bdd2dc92f710bdde6d11652239d71d1ec450067939/coverage-7.14.0-cp312-cp312-macosx_11_0_arm64.whl", upload-time = 2026-05-10T18:00:15Z, size = 220329, hashes = { sha256 = "829994cfe1aeb773ca27bf246d4badc1e764893e3bfb98fff820fcecd1ca4662" } }, - { url = "https://files.pythonhosted.org/packages/75/cf/a8f4b43a16e194b0261257ad28ded5853ec052570afef4a84e1d81189f3b/coverage-7.14.0-cp312-cp312-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-05-10T18:00:17Z, size = 251839, hashes = { sha256 = "b4f07cf7edcb7ec39431a5074d7ea83b29a9f71fcfc494f0f40af4e65180420f" } }, - { url = "https://files.pythonhosted.org/packages/69/ff/6699e7b71e60d3049eb2bdcbc95ee3f35707b2b0e48f32e9e63d3ce30c08/coverage-7.14.0-cp312-cp312-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-05-10T18:00:18Z, size = 254576, hashes = { sha256 = "ca3d9cf2c32b521bd9518385608787fa86f38daf993695307531822c3430ed67" } }, - { url = "https://files.pythonhosted.org/packages/22/ec/c936d495fcd67f48f03a9c4ad3297ff80d1f222a5df3980f15b34c186c21/coverage-7.14.0-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-05-10T18:00:20Z, size = 255690, hashes = { sha256 = "92af52828e7f29d827346b0294e5a0853fa206db77db0395b282918d41e28db9" } }, - { url = "https://files.pythonhosted.org/packages/5c/42/5af63f636cc62a4a2b1b3ba9146f6ee6f53a35a50d5cefc54d5670f60999/coverage-7.14.0-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-05-10T18:00:22Z, size = 257949, hashes = { sha256 = "7b2bb6c9d7e769360d0f20a0f219603fd64f0c8f97de17ab25853261602be0fb" } }, - { url = "https://files.pythonhosted.org/packages/26/d3/a225317bd2012132a27e1176d51660b826f99bb975876463c44ea0d7ee5a/coverage-7.14.0-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-05-10T18:00:24Z, size = 252242, hashes = { sha256 = "1c9ed6ef99f88fb8c14aa8e2bf8eb0fe55fa2edfea68f8675d78741df1a5ac0e" } }, - { url = "https://files.pythonhosted.org/packages/f1/7f/9e65495298c3ea414742998539c37d048b5e81cc818fb1828cc6b51d10bf/coverage-7.14.0-cp312-cp312-musllinux_1_2_aarch64.whl", upload-time = 2026-05-10T18:00:25Z, size = 253608, hashes = { sha256 = "8231ade007f37959fbf58acc677f26b922c02eda6f0428ea307da0fd39681bf3" } }, - { url = "https://files.pythonhosted.org/packages/94/46/1522b524a35bdad22b2b8c4f9d32d0a104b524726ec380b2db68db1746f5/coverage-7.14.0-cp312-cp312-musllinux_1_2_i686.whl", upload-time = 2026-05-10T18:00:27Z, size = 251753, hashes = { sha256 = "d8b013632cc1ce1d09dbe4f32667b4d320ec2f54fc326ebeffcd0b0bcc2bb6c4" } }, - { url = "https://files.pythonhosted.org/packages/f3/e9/cdf00d38817742c541ade405e115a3f7bf36e6f2a8b99d4f209861b85a2d/coverage-7.14.0-cp312-cp312-musllinux_1_2_ppc64le.whl", upload-time = 2026-05-10T18:00:29Z, size = 255823, hashes = { sha256 = "1733198802d71ec4c524f322e2867ee05c62e9e75df86bdca545407a221827d1" } }, - { url = "https://files.pythonhosted.org/packages/38/fc/5e7877cf5f902d08a17ff1c532511476d87e1bea355bd5028cb97f902e79/coverage-7.14.0-cp312-cp312-musllinux_1_2_riscv64.whl", upload-time = 2026-05-10T18:00:30Z, size = 251323, hashes = { sha256 = "72a305291fa8ee01332f1aaf38b348ca34097f6aa0b0ef627eef2837e57bbba5" } }, - { url = "https://files.pythonhosted.org/packages/18/9d/50f05a72dff8487464fdd4178dda5daed642a060e60afb644e3d45123559/coverage-7.14.0-cp312-cp312-musllinux_1_2_x86_64.whl", upload-time = 2026-05-10T18:00:32Z, size = 253197, hashes = { sha256 = "fcaba850dd317c65423a9d63d88f9573c53b00354d6dd95724576cc98a131595" } }, - { url = "https://files.pythonhosted.org/packages/00/3f/6f61ffe6439df266c3cf60f5c99cfaa21103d0210d706a42fc6c30683ff8/coverage-7.14.0-cp312-cp312-win32.whl", upload-time = 2026-05-10T18:00:33Z, size = 222515, hashes = { sha256 = "5ac83957a80d0701310e96d8bec68cdcf4f90a7674b7d13f15a344315b41ab27" } }, - { url = "https://files.pythonhosted.org/packages/85/19/93853133df2cb371083285ef6a93982a0173e7a233b0f61373ba9fd30eb2/coverage-7.14.0-cp312-cp312-win_amd64.whl", upload-time = 2026-05-10T18:00:35Z, size = 223324, hashes = { sha256 = "70390b0da32cb90b501953716302906e8bcce087cb283e70d8c97729f22e92b2" } }, - { url = "https://files.pythonhosted.org/packages/74/18/9f7fe62f659f24b7a82a0be56bf94c1bd0a89e0ae7ab4c668f6e82404294/coverage-7.14.0-cp312-cp312-win_arm64.whl", upload-time = 2026-05-10T18:00:37Z, size = 221944, hashes = { sha256 = "91b993743d959b8be85b4abf9d5478216a69329c321efe5be0433c1a841d691d" } }, - { url = "https://files.pythonhosted.org/packages/6b/76/b7c66ee3c66e1b0f9d894c8125983aa0c03fb2336f2fd16559f9c966157f/coverage-7.14.0-cp313-cp313-macosx_10_13_x86_64.whl", upload-time = 2026-05-10T18:00:38Z, size = 219990, hashes = { sha256 = "f2bbb8254370eb4c628ff3d6fa8a7f74ddc40565394d4f7ab791d1fe568e37ef" } }, - { url = "https://files.pythonhosted.org/packages/b3/af/e567cbad5ba69c013a50146dfa886dc7193361fda77521f51274ff620e1b/coverage-7.14.0-cp313-cp313-macosx_11_0_arm64.whl", upload-time = 2026-05-10T18:00:40Z, size = 220365, hashes = { sha256 = "23b81107f46d3f21d0cbce30664fcec0f5d9f585638a67081750f99738f6bf66" } }, - { url = "https://files.pythonhosted.org/packages/44/6f/9ad575d505b4d805b254febc8a5b338a2efe278f8786e56ff1cb8413f9c3/coverage-7.14.0-cp313-cp313-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-05-10T18:00:42Z, size = 251363, hashes = { sha256 = "22a7e06a5f11a757cdfe79018e9095f9f69ae283c5cd8123774c788deec8717b" } }, - { url = "https://files.pythonhosted.org/packages/6f/5f/b5370068b2f57787454592ed7dcd1002f0f1703b7db1fa30f6a325a4ca6e/coverage-7.14.0-cp313-cp313-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-05-10T18:00:44Z, size = 253961, hashes = { sha256 = "9d1aa57a1dc8e05bdc42e81c5d671d849577aeedf279f4c449d6d286f9ed88ca" } }, - { url = "https://files.pythonhosted.org/packages/29/1e/51adf17738976e8f2b85ddef7b7aa12a0838b056c92f175941d8862767c1/coverage-7.14.0-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-05-10T18:00:45Z, size = 255193, hashes = { sha256 = "90c1a51bcfddf645b3bb7ec333d9e94393a8e94f55642380fa8a9a5a9e636cb7" } }, - { url = "https://files.pythonhosted.org/packages/9e/7b/5bfd7ac1df3b881c2ac7a5cbc99c7609e6296c402f5ef587cd81c6f355b3/coverage-7.14.0-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-05-10T18:00:47Z, size = 257326, hashes = { sha256 = "a841fae2fadcae4f438d43b6ccc4aac2ad609f47cdb6cfdce60cbb3fe5ca7bc2" } }, - { url = "https://files.pythonhosted.org/packages/7d/38/1d37d316b174fad3843a1d76dbdfe4398771c9ecd0515935dd9ece9cd627/coverage-7.14.0-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-05-10T18:00:49Z, size = 251582, hashes = { sha256 = "c79d2319cabef1fe8e86df73371126931550804738f78ad7d31e3aad85a67367" } }, - { url = "https://files.pythonhosted.org/packages/34/46/746704f95980ba220214e1a41e18cec5aea80a898eaa53c51bf2d645ff36/coverage-7.14.0-cp313-cp313-musllinux_1_2_aarch64.whl", upload-time = 2026-05-10T18:00:51Z, size = 253325, hashes = { sha256 = "1b23b0c6f0b1db6ad769b7050c8b641c0bf215ded26c1816955b17b7f26edfa9" } }, - { url = "https://files.pythonhosted.org/packages/e1/b9/bbe87206d9687b192352f893797825b5f5b15ecd3aa9c68fbff0c074d77b/coverage-7.14.0-cp313-cp313-musllinux_1_2_i686.whl", upload-time = 2026-05-10T18:00:52Z, size = 251291, hashes = { sha256 = "55d3089079ce181a4566b1065ab28d2575eb76d8ac8f81f4fcda2bf037fee087" } }, - { url = "https://files.pythonhosted.org/packages/46/57/b8cdb12ac0d73ef0243218bd5e22c9df8f92edab8018213a86aec67c5324/coverage-7.14.0-cp313-cp313-musllinux_1_2_ppc64le.whl", upload-time = 2026-05-10T18:00:54Z, size = 255448, hashes = { sha256 = "49c005cba1e2f9677fb2845dcdf9a2e72a52a17d63e8231aaaae35d9f50215ef" } }, - { url = "https://files.pythonhosted.org/packages/1f/d4/5002019538b2036ce3c84340f54d2fd5100d55b0a6b0894eee56128d03c7/coverage-7.14.0-cp313-cp313-musllinux_1_2_riscv64.whl", upload-time = 2026-05-10T18:00:56Z, size = 251110, hashes = { sha256 = "9117377b823daa28aa8635fbb08cda1cd6be3d7143257345459559aeef852d52" } }, - { url = "https://files.pythonhosted.org/packages/37/53/20c5009477660f084e6ed60bc02a91894b8e234e617e86ecfd9aaf78e27b/coverage-7.14.0-cp313-cp313-musllinux_1_2_x86_64.whl", upload-time = 2026-05-10T18:00:57Z, size = 252885, hashes = { sha256 = "7b79d646cf46d5cf9a9f40281d4441df5849e445726e369006d2b117710b33fe" } }, - { url = "https://files.pythonhosted.org/packages/ae/ab/3cf6427ac9c1f1db747dbb1ce71dde47984876d4c2cfd018a3fef0a78d4d/coverage-7.14.0-cp313-cp313-win32.whl", upload-time = 2026-05-10T18:00:59Z, size = 222539, hashes = { sha256 = "fb609b3658479e33f9516d46f1a89dbb9b6c261366e3a11844a96ec487533dae" } }, - { url = "https://files.pythonhosted.org/packages/8f/b8/9228523e80321c2cb4880d1f589bc0171f2f71432c35118ad04dc01decce/coverage-7.14.0-cp313-cp313-win_amd64.whl", upload-time = 2026-05-10T18:01:01Z, size = 223344, hashes = { sha256 = "0773d8329cf32b6fd222e4b52622c61fe8d503eb966cfc8d3c3c10c96266d50e" } }, - { url = "https://files.pythonhosted.org/packages/a3/99/118daa192f95e3a6cb2740100fbf8797cda1734b4134ef0b5d501a7fa8f3/coverage-7.14.0-cp313-cp313-win_arm64.whl", upload-time = 2026-05-10T18:01:03Z, size = 221966, hashes = { sha256 = "b4e26a0f1b696faf283bffe5b8569e44e336c582439df5d53281ab89ee0cba96" } }, - { url = "https://files.pythonhosted.org/packages/e6/f1/a46cc0c013be170216253184a32366d7cbdb9252feaec866b05c2d12a894/coverage-7.14.0-cp313-cp313t-macosx_10_13_x86_64.whl", upload-time = 2026-05-10T18:01:05Z, size = 220679, hashes = { sha256 = "953f521ca9445300397e65fda3dca58b2dbd68fee983777420b57ac3c77e9f90" } }, - { url = "https://files.pythonhosted.org/packages/64/8c/9c30a3d311a34177fa432995be7fbfc64477d8bac5630bd38055b1c9b424/coverage-7.14.0-cp313-cp313t-macosx_11_0_arm64.whl", upload-time = 2026-05-10T18:01:07Z, size = 221033, hashes = { sha256 = "98af83fd65ae24b1fdd03aaead967a9f523bcd2f1aab2d4f3ffda65bb568a6f1" } }, - { url = "https://files.pythonhosted.org/packages/9a/cd/3fb5e06c3badefd0c1b47e2044fdca67f8220a4ec2e7fcfb476aa0a67c6c/coverage-7.14.0-cp313-cp313t-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-05-10T18:01:08Z, size = 262333, hashes = { sha256 = "668b92e6958c4db7cf92e81caac328dfbbdbb215db2850ad28f0cbe1eea0bfbd" } }, - { url = "https://files.pythonhosted.org/packages/a8/e6/fbc322325c7294d3e22c1ad6b79e45d0806b25228c8e5842aed6d8169aa7/coverage-7.14.0-cp313-cp313t-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-05-10T18:01:10Z, size = 264410, hashes = { sha256 = "9fbd898551762dea00d3fef2b1c4f99afd2c6a3ff952ea07d60a9bd5ed4f34bc" } }, - { url = "https://files.pythonhosted.org/packages/08/92/c497b264bec1673c47cc77e26f760fcda4654cabf1f39546d1a23a3b8c35/coverage-7.14.0-cp313-cp313t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-05-10T18:01:12Z, size = 266836, hashes = { sha256 = "68af363c07ecd8d4b7d4043d85cb376d7d227eceb54e5323ee45da73dbd3e426" } }, - { url = "https://files.pythonhosted.org/packages/78/fc/045da320987f401af5d2815d351e8aa799aec859f60e29f445e3089eeedb/coverage-7.14.0-cp313-cp313t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-05-10T18:01:13Z, size = 267974, hashes = { sha256 = "6e57054a583da8ac55edf24117ea4c9133032cfc4cf72aa2d48c1e5d4b52f899" } }, - { url = "https://files.pythonhosted.org/packages/1b/ae/227b1e379497fb7a4fc3286e620f80c8a1e7cec66d45695a01639eb1af65/coverage-7.14.0-cp313-cp313t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-05-10T18:01:15Z, size = 261578, hashes = { sha256 = "cc3499459bbcdd51a65b64c35ab7ed2764eaf3cba826e0df3f1d7fe2e102b70b" } }, - { url = "https://files.pythonhosted.org/packages/a0/f5/3570342900f2acea31d33ff1590c5d8bac1a8e1a2e1c6d34a5d5e61de681/coverage-7.14.0-cp313-cp313t-musllinux_1_2_aarch64.whl", upload-time = 2026-05-10T18:01:17Z, size = 264394, hashes = { sha256 = "45899ec2138a4346ed34d601dedf5076fb74edf2d1dd9dc76a78e82397edee90" } }, - { url = "https://files.pythonhosted.org/packages/16/29/de1bbc01c935b28f89b1dc3db85b011c055e843a8e5e3b83141c3f80af7f/coverage-7.14.0-cp313-cp313t-musllinux_1_2_i686.whl", upload-time = 2026-05-10T18:01:19Z, size = 262022, hashes = { sha256 = "8767486808c436f05b23ab98eb963fb29185e32a9357a166971685cb3459900f" } }, - { url = "https://files.pythonhosted.org/packages/35/95/f53890b0bf2fc10ab168e05d38869215e73ca24c4cb521c3bb0eb62fe16b/coverage-7.14.0-cp313-cp313t-musllinux_1_2_ppc64le.whl", upload-time = 2026-05-10T18:01:21Z, size = 265732, hashes = { sha256 = "a3b5ddfd6aa7ddad53ee3edb231e88a2151507a43229b7d71b953916deca127d" } }, - { url = "https://files.pythonhosted.org/packages/ed/ea/c919e259081dd2bdf0e43b87209709ba7ec2e4117c2a7f5185379c43463c/coverage-7.14.0-cp313-cp313t-musllinux_1_2_riscv64.whl", upload-time = 2026-05-10T18:01:23Z, size = 260921, hashes = { sha256 = "63df0fe568e698e1045792399f8ab6da3a6c2dce3182813fb92afa2641087b47" } }, - { url = "https://files.pythonhosted.org/packages/1a/2c/c2831889705a81dc5d1c6ca12e4d8e9b95dfc146d153488a6c0ea685d28e/coverage-7.14.0-cp313-cp313t-musllinux_1_2_x86_64.whl", upload-time = 2026-05-10T18:01:25Z, size = 263109, hashes = { sha256 = "827d6397dbd95144939b18f89edf31f63e1f99633e8d5f32f22ba8bdda567477" } }, - { url = "https://files.pythonhosted.org/packages/5a/a9/2fcae5003cac3d63fe344d2166243c2756935f48420863c5272b240d550b/coverage-7.14.0-cp313-cp313t-win32.whl", upload-time = 2026-05-10T18:01:27Z, size = 223212, hashes = { sha256 = "7bf43e000d24012599b879791cff41589af90674722421ef11b11a5431920bab" } }, - { url = "https://files.pythonhosted.org/packages/3f/bb/18e94d7b14b9b398164197114a587a04ab7c9fdbe1d237eef57311c5e883/coverage-7.14.0-cp313-cp313t-win_amd64.whl", upload-time = 2026-05-10T18:01:29Z, size = 224272, hashes = { sha256 = "3f5549365af25d770e06b1f8f5682d9a5637d06eb494db91c6fa75d3950cc917" } }, - { url = "https://files.pythonhosted.org/packages/db/56/4f14fad782b035c81c4ffd09159e7103d42bb1d93ac8496d04b90a11b7da/coverage-7.14.0-cp313-cp313t-win_arm64.whl", upload-time = 2026-05-10T18:01:31Z, size = 222530, hashes = { sha256 = "6d160217ec6fe890f16ad3a9531761589443749e448f91986c972714fad361c8" } }, - { url = "https://files.pythonhosted.org/packages/1c/18/b9a6586d73992807c26f9a5f274131be3d76b56b18a82b9392e2a25d2e45/coverage-7.14.0-cp314-cp314-macosx_10_15_x86_64.whl", upload-time = 2026-05-10T18:01:33Z, size = 220036, hashes = { sha256 = "9aed9fa983514ca032790f3fe0d1c0e42ca7e16b42432af1706b50a9a46bef5d" } }, - { url = "https://files.pythonhosted.org/packages/f3/9b/4165a1d56ddc302a0e2d518fd9d412a4fd0b57562618c78c5f21c57194f5/coverage-7.14.0-cp314-cp314-macosx_11_0_arm64.whl", upload-time = 2026-05-10T18:01:34Z, size = 220368, hashes = { sha256 = "ba3b8390db29296dbbf49e91b6fe08f990743a90c8f447ba4c2ffc29670dfa63" } }, - { url = "https://files.pythonhosted.org/packages/69/aa/c12e52a5ba148d9995229d557e3be6e554fe469addc0e9241b2f0956d8ea/coverage-7.14.0-cp314-cp314-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-05-10T18:01:36Z, size = 251417, hashes = { sha256 = "3a5d8e876dfa2f102e970b183863d6dedd023d3c0eeca1fe7a9787bc5f28b212" } }, - { url = "https://files.pythonhosted.org/packages/d7/51/ec641c26e6dca1b25a7d2035ba6ecb7c884ef1a100a9e42fbe4ce4405139/coverage-7.14.0-cp314-cp314-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-05-10T18:01:38Z, size = 253924, hashes = { sha256 = "5ebb8f4614a3787d567e610bbfdf96a4798dd69a1afb1bd8ad228d4111fe6ff3" } }, - { url = "https://files.pythonhosted.org/packages/33/c4/59c3de0bd1b538824173fd518fed51c1ce740ca5ed68e74545983f4053a9/coverage-7.14.0-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-05-10T18:01:40Z, size = 255269, hashes = { sha256 = "6b9bf47223dd8db3d4c4b2e443b02bace480d428f0822c3f991600448a176c97" } }, - { url = "https://files.pythonhosted.org/packages/7b/a9/36dfa153a62040296f6e7febfdb20a5720622f6ef5a81a41e8237b9a5344/coverage-7.14.0-cp314-cp314-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-05-10T18:01:42Z, size = 257583, hashes = { sha256 = "3485a836550b303d006d57cc06e3d5afaabc642c77050b7c985a97b13e3776b8" } }, - { url = "https://files.pythonhosted.org/packages/26/7b/cc2c048d4114d9ab1c2409e9ee365e5ae10736df6dffcfc9444effa6c708/coverage-7.14.0-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-05-10T18:01:44Z, size = 251434, hashes = { sha256 = "3e7e88110bae996d199d1693ca8ec3fd52441d426401ae963437598667b4c5eb" } }, - { url = "https://files.pythonhosted.org/packages/ee/df/6770eaa576e604575e9a78055313250faef5faa84bd6f71a39fece519c43/coverage-7.14.0-cp314-cp314-musllinux_1_2_aarch64.whl", upload-time = 2026-05-10T18:01:46Z, size = 253280, hashes = { sha256 = "15228a6800ce7bdf1b74800595e56db7138cecb338fdbf044806e10dcf182dfe" } }, - { url = "https://files.pythonhosted.org/packages/ad/9e/1c0264514a3f98259a6d64765a397b2c8373e3ba59ee722a4802d3ec0c61/coverage-7.14.0-cp314-cp314-musllinux_1_2_i686.whl", upload-time = 2026-05-10T18:01:48Z, size = 251241, hashes = { sha256 = "9d26ac7f5398bafc5b57421ad994e8a4749e8a7a0e62d05ec7d53014d5963bfa" } }, - { url = "https://files.pythonhosted.org/packages/64/16/4efdf3e3c4079cdbf0ece56a2fea872df9e8a3e15a13a0af4400e1075944/coverage-7.14.0-cp314-cp314-musllinux_1_2_ppc64le.whl", upload-time = 2026-05-10T18:01:50Z, size = 255516, hashes = { sha256 = "2fb73254ff43c911c967a899e1359bc5049b4b115d6e8fbdde4937d0a2246cd5" } }, - { url = "https://files.pythonhosted.org/packages/93/69/b1de96346603881b3d1bc8d6447c83200e1c9700ffbaff926ba01ff5724c/coverage-7.14.0-cp314-cp314-musllinux_1_2_riscv64.whl", upload-time = 2026-05-10T18:01:52Z, size = 251059, hashes = { sha256 = "454a380af72c6adada298ed270d38c7a391288198dbfb8467f786f588751a90c" } }, - { url = "https://files.pythonhosted.org/packages/a4/66/2881853e0363a5e0a724d1103e53650795367471b6afb234f8b49e713bc6/coverage-7.14.0-cp314-cp314-musllinux_1_2_x86_64.whl", upload-time = 2026-05-10T18:01:54Z, size = 252716, hashes = { sha256 = "65c86fb646d2bd2972e96bd1a8b45817ed907cee68655d6295fe7ec031d04cca" } }, - { url = "https://files.pythonhosted.org/packages/55/5c/0d3305d002c41dcde873dbe456491e663dc55152ca526b630b5c47efd62f/coverage-7.14.0-cp314-cp314-win32.whl", upload-time = 2026-05-10T18:01:56Z, size = 222788, hashes = { sha256 = "6a6516b02a6101398e19a3f44820f69bab2590697f7def4331f668b14adaf828" } }, - { url = "https://files.pythonhosted.org/packages/f9/58/6e1b8f52fdc3184b47dc5037f5070d83a3d11042db1594b02d2a44d786c8/coverage-7.14.0-cp314-cp314-win_amd64.whl", upload-time = 2026-05-10T18:01:58Z, size = 223600, hashes = { sha256 = "45e0f79d8351fa76e256716df91eab12890d32678b9590df7ae1042e4bd4cf5d" } }, - { url = "https://files.pythonhosted.org/packages/00/70/a18c408e674bc26281cadaedc7351f929bd2094e191e4b15271c30b084cc/coverage-7.14.0-cp314-cp314-win_arm64.whl", upload-time = 2026-05-10T18:02:00Z, size = 222168, hashes = { sha256 = "4b899594a8b2d81e5cc064a0d7f9cac2081fed91049456cae7676787e41549c9" } }, - { url = "https://files.pythonhosted.org/packages/3d/89/2681f071d238b62aff8dfc2ab44fc24cfdb38d1c01f391a80522ff5d3a16/coverage-7.14.0-cp314-cp314t-macosx_10_15_x86_64.whl", upload-time = 2026-05-10T18:02:02Z, size = 220766, hashes = { sha256 = "f580f8c80acd94ac72e863efe2cab791d8c38d153e0b463b92dfa000d5c84cd1" } }, - { url = "https://files.pythonhosted.org/packages/bd/c7/c987babafd9207ffa1995e1ef1f9b26762cf4963aa768a66b6f0501e4616/coverage-7.14.0-cp314-cp314t-macosx_11_0_arm64.whl", upload-time = 2026-05-10T18:02:04Z, size = 221035, hashes = { sha256 = "a2bd259c442cd43c49b30fbafc51776eb19ea396faf159d26a83e6a0a5f13b0c" } }, - { url = "https://files.pythonhosted.org/packages/5a/e9/d6a5ac3b333088143d6fc877d398a9a674dc03124a2f776e131f03864823/coverage-7.14.0-cp314-cp314t-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-05-10T18:02:05Z, size = 262405, hashes = { sha256 = "a706b908dfa85538863504c624b237a3cc34232bf403c057414ebfdb3b4d9f84" } }, - { url = "https://files.pythonhosted.org/packages/38/b1/e70838d29a7c08e22d44398a46db90815bbcbf28de06992bd9210d1a8d8e/coverage-7.14.0-cp314-cp314t-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-05-10T18:02:07Z, size = 264530, hashes = { sha256 = "7333cd944ee4393b9b3d3c1b598c936d4fc8d70573a4c7dacfec5590dd50e436" } }, - { url = "https://files.pythonhosted.org/packages/6b/73/5c31ef97763288d03d9995152b96d5475b527c63d91c84b01caea894b83a/coverage-7.14.0-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-05-10T18:02:09Z, size = 266932, hashes = { sha256 = "0f162bc9a15b82d947b02651b0c7e1609d6f7a8735ca330cfadec8481dd97d5a" } }, - { url = "https://files.pythonhosted.org/packages/e1/76/dd56d80f29c5f05b4d76f7e7c6d47cafacae017189c75c5759d24f9ff0cc/coverage-7.14.0-cp314-cp314t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-05-10T18:02:11Z, size = 268062, hashes = { sha256 = "362cb78e01a5dc82009d88004cf60f2e6b6d6fcbfdec05b05af73b0abf40118f" } }, - { url = "https://files.pythonhosted.org/packages/6e/c7/27ba85cd5b95614f159ff93ebff1901584a8d192e2e5e24c4943a7453f59/coverage-7.14.0-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-05-10T18:02:13Z, size = 261504, hashes = { sha256 = "acebd068fca5512c3a6fde9c045f901613478781a73f0e82b307b214daef23fb" } }, - { url = "https://files.pythonhosted.org/packages/13/2e/e8149f60ab5d5684c6eee881bdf34b127115cddbb958b196768dd9d63473/coverage-7.14.0-cp314-cp314t-musllinux_1_2_aarch64.whl", upload-time = 2026-05-10T18:02:15Z, size = 264398, hashes = { sha256 = "29fe3da551dface75deb2ccbf87b6b66e2e7ef38f6d89050b428be94afff3490" } }, - { url = "https://files.pythonhosted.org/packages/d9/7f/1261b025285323225f4b4abffa5a643649dfd67e25ddca7ebcbdea3b7cb3/coverage-7.14.0-cp314-cp314t-musllinux_1_2_i686.whl", upload-time = 2026-05-10T18:02:16Z, size = 262000, hashes = { sha256 = "b4cc4fce8672fffcb09b0eafc167b396b3ba53c4a7230f54b7aaffbf6c835fa9" } }, - { url = "https://files.pythonhosted.org/packages/d3/dc/829c54f60b9d08389439c00f813c752781c496fc5788c78d8006db4b4f2b/coverage-7.14.0-cp314-cp314t-musllinux_1_2_ppc64le.whl", upload-time = 2026-05-10T18:02:18Z, size = 265732, hashes = { sha256 = "5d4a51aad8ba8bdcd2b8bd8f03d4aca19693fa2327a3470e4718a25b03481020" } }, - { url = "https://files.pythonhosted.org/packages/ed/b0/70bd1419941652fa062689cba9c3eeafb8f5e6fbb890bce41c3bdda5dbd6/coverage-7.14.0-cp314-cp314t-musllinux_1_2_riscv64.whl", upload-time = 2026-05-10T18:02:20Z, size = 260847, hashes = { sha256 = "9f323af3e1e4f68b60b7b247e37b8515563a61375518fa59de1af48ba28a3db6" } }, - { url = "https://files.pythonhosted.org/packages/f2/73/be40b2390656c654d35ea0015ea7ba3d945769cf80790ad5e0bb2d56d2ba/coverage-7.14.0-cp314-cp314t-musllinux_1_2_x86_64.whl", upload-time = 2026-05-10T18:02:22Z, size = 263166, hashes = { sha256 = "1a0abc7342ea9711c469dd8b821c6c311e6bc6aac1442e5fbd6b27fae0a8f3db" } }, - { url = "https://files.pythonhosted.org/packages/29/55/4a643f712fcf7cf2881f8ec1e0ccb7b164aff3108f69b51801246c8799f2/coverage-7.14.0-cp314-cp314t-win32.whl", upload-time = 2026-05-10T18:02:24Z, size = 223573, hashes = { sha256 = "a9f864ef57b7172e2db87a096642dd51e179e085ab6b2c371c29e885f65c8fb2" } }, - { url = "https://files.pythonhosted.org/packages/27/96/3acae5da0953be042c0b4dea6d6789d2f080701c77b88e44d5bd41b9219b/coverage-7.14.0-cp314-cp314t-win_amd64.whl", upload-time = 2026-05-10T18:02:25Z, size = 224680, hashes = { sha256 = "29943e552fdc08e082eb51400fb2f58e118a83b5542bd06531214e084399b644" } }, - { url = "https://files.pythonhosted.org/packages/93/3d/6ab5d2dd8325d838737c6f8d83d62eb6230e0d70b87b51b57bbfd08fa767/coverage-7.14.0-cp314-cp314t-win_arm64.whl", upload-time = 2026-05-10T18:02:27Z, size = 222703, hashes = { sha256 = "742a73ea621953b012f2c4c2219b512180dd84489acf5b1596b0aafc55b9100b" } }, - { url = "https://files.pythonhosted.org/packages/61/e8/cb8e80d6f9f55b99588625062822bf946cf03ed06315df4bd8397f5632a1/coverage-7.14.0-py3-none-any.whl", upload-time = 2026-05-10T18:02:29Z, size = 211764, hashes = { sha256 = "8de5b61163aee3d05c8a2beab6f47913df7981dad1baf82c414d99158c286ab1" } }, + { url = "https://files.pythonhosted.org/packages/30/70/b052a519a584663a7bd052841a2debe11c8309ec49a7786340003f9c0a02/coverage-7.15.4-cp310-cp310-macosx_10_9_x86_64.whl", upload-time = 2026-08-06T13:46:55Z, size = 222245, hashes = { sha256 = "d0be6daac4cce6b8c8dc65886bae1b082ddbca4da8e5cbb5e15166acf253e264" } }, + { url = "https://files.pythonhosted.org/packages/67/39/892fa511aba3d1c3c8f49509a0ff5c71eab9f9f88d08e1a38da395821660/coverage-7.15.4-cp310-cp310-macosx_11_0_arm64.whl", upload-time = 2026-08-06T13:46:57Z, size = 222762, hashes = { sha256 = "b24e078eabcd6a9caa8b0713f9bc1eeb310bcc960a29d45a3b4fcd4b16d5b11d" } }, + { url = "https://files.pythonhosted.org/packages/9f/95/b2c724ce1e64bc23cb5b1d7eeffa9548dc3d811f7a6297b2d01607f4e062/coverage-7.15.4-cp310-cp310-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-08-06T13:46:59Z, size = 249498, hashes = { sha256 = "cfe20cc8cf8821d4fe54f89106cbf06aa27f37b5bbe3535568065a81539b4150" } }, + { url = "https://files.pythonhosted.org/packages/0b/4f/b1973f67a1382af65b572a31ed692f8e490a6ad707191eab59148376832a/coverage-7.15.4-cp310-cp310-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-08-06T13:47:00Z, size = 251328, hashes = { sha256 = "83cf06cdd687677742caff1a9134833b7a8b75f111519d2cb0e0ba1b9a851e15" } }, + { url = "https://files.pythonhosted.org/packages/a2/09/03efa6722a132abcac91b32a60b64b240dd707c189c64eee697e48992c96/coverage-7.15.4-cp310-cp310-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-06T13:47:01Z, size = 253194, hashes = { sha256 = "8fa4de68e2a752468ff14b4e15db7def689a71be759e826a31ccecbef69c5fd0" } }, + { url = "https://files.pythonhosted.org/packages/45/63/8299201d9c80fb65551ce99c966cab83d706ec4066ac999bef08201346de/coverage-7.15.4-cp310-cp310-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-06T13:47:03Z, size = 255106, hashes = { sha256 = "4dff9daa47d83120c3ec38ce921214242944a832aa04e903e50b5b7ebac8972d" } }, + { url = "https://files.pythonhosted.org/packages/ee/16/26fd8a691eb8d9a230128685f6d23309d7402cb030aa553001788c8c50fc/coverage-7.15.4-cp310-cp310-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-06T13:47:04Z, size = 250177, hashes = { sha256 = "a093fd37229918976f602aa07aa59e0973cde82186f220c8e197f721f5be0ce4" } }, + { url = "https://files.pythonhosted.org/packages/ad/ef/3c7556f33783a0a566e01443ca62bd8eb2cdfe22d271efdc02e08beb5654/coverage-7.15.4-cp310-cp310-musllinux_1_2_aarch64.whl", upload-time = 2026-08-06T13:47:06Z, size = 251234, hashes = { sha256 = "317db01a2cb02552fd67e2b1cca77a4b528a2a277176c5e0bf2cecbb639d3f54" } }, + { url = "https://files.pythonhosted.org/packages/29/49/640a34043edac950738f36a3567832db5731d4cb2ed84b59cdb89c6bccbf/coverage-7.15.4-cp310-cp310-musllinux_1_2_i686.whl", upload-time = 2026-08-06T13:47:07Z, size = 249237, hashes = { sha256 = "8ee3838dcb656602c3b51e16aed9bfb0822f8d8d6d1c5966d32ec8c104be8e20" } }, + { url = "https://files.pythonhosted.org/packages/48/f5/e80f212669dd1be954ff844f883ef11a437ef4fd0089c6e0effc7b66b15d/coverage-7.15.4-cp310-cp310-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-06T13:47:08Z, size = 253050, hashes = { sha256 = "425920379052ff1fe465268f3361d35804a241bbdd5a1b592c8cb60df4c52325" } }, + { url = "https://files.pythonhosted.org/packages/c7/e9/e5da0fe39f7fde1bca9edc09c60921bb5fdba4cec7db5bbad41ddfd8c230/coverage-7.15.4-cp310-cp310-musllinux_1_2_riscv64.whl", upload-time = 2026-08-06T13:47:10Z, size = 249508, hashes = { sha256 = "69bb2400abef928e365ea7d4d9925169ada78ed2295546780002d4b65de3df88" } }, + { url = "https://files.pythonhosted.org/packages/7d/38/41bf25774a0c8bba6b467f917cb1c9a0a2605e02dc93aad489fc7050ed59/coverage-7.15.4-cp310-cp310-musllinux_1_2_x86_64.whl", upload-time = 2026-08-06T13:47:11Z, size = 250110, hashes = { sha256 = "81661f82d302484e3119e7c80c519c02fa9bcc2a6b339baf67d67bc89c580f04" } }, + { url = "https://files.pythonhosted.org/packages/89/6e/26f2e54b79acc29d179ee4272922625aedb69198c4eb61f7ff4f098f3c78/coverage-7.15.4-cp310-cp310-win32.whl", upload-time = 2026-08-06T13:47:12Z, size = 224294, hashes = { sha256 = "cb476b2e828ecb71cb6b6a928d23fd20a7ddb501188022dae1c37499149cc338" } }, + { url = "https://files.pythonhosted.org/packages/7b/06/9a318fc3ae040d4d6cb2d86101c6aa963fab20899a5c58666adf52cde0ca/coverage-7.15.4-cp310-cp310-win_amd64.whl", upload-time = 2026-08-06T13:47:14Z, size = 224919, hashes = { sha256 = "3fc2130bf37df31852a8384f12601563a45a0024bccc6624f38355cba7a8d360" } }, + { url = "https://files.pythonhosted.org/packages/2a/66/edcec7d7a0b524aa8923e22925fde6fe50ce005a113dca13ae1581455c4c/coverage-7.15.4-cp311-cp311-macosx_10_9_x86_64.whl", upload-time = 2026-08-06T13:47:15Z, size = 222367, hashes = { sha256 = "bbac5abad70df71019988f83f26ac7092ff2642975def4429e98dc7585ef3490" } }, + { url = "https://files.pythonhosted.org/packages/e6/c6/ab8de429e2e8548faf58ec7e1674a4ce00414b4113942d3fe87109cf0f68/coverage-7.15.4-cp311-cp311-macosx_11_0_arm64.whl", upload-time = 2026-08-06T13:47:16Z, size = 222874, hashes = { sha256 = "357a173465c7ce028d07a95cc2b63b5bf59f50ecdd5ad75c5cbb78ada984048e" } }, + { url = "https://files.pythonhosted.org/packages/be/c4/3b7b49587e8a6b9af79b3eb468d443d6042b6d65b47aa26586846a0d6566/coverage-7.15.4-cp311-cp311-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-08-06T13:47:18Z, size = 253287, hashes = { sha256 = "21b803935e2efc3acebe9697197a294fccf5dc4e5382bd6369542ff7a7d2a1d7" } }, + { url = "https://files.pythonhosted.org/packages/fb/65/ec03b743a2a229c72cc1eff3e57be9d3564e9c6b4d5aba2d70744a3fc0d8/coverage-7.15.4-cp311-cp311-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-08-06T13:47:19Z, size = 255199, hashes = { sha256 = "7a2b580774a4786c1053157c0165e04476e03ff293993d7c148eee784a94bae6" } }, + { url = "https://files.pythonhosted.org/packages/41/4b/5163729e4b6582d61975cfd3ccab45b4ec53e21cf156d9941cb025188468/coverage-7.15.4-cp311-cp311-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-06T13:47:21Z, size = 257308, hashes = { sha256 = "a9464451c4efffe8d47ace5a540b10b0dc10e879066290f8600872b7f54a419d" } }, + { url = "https://files.pythonhosted.org/packages/86/08/2167a0f08fb87d702fa423a48578a32865464b7c9e1db3911ad7812ab414/coverage-7.15.4-cp311-cp311-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-06T13:47:22Z, size = 259268, hashes = { sha256 = "de602f34123c2f4af1c1869c6dbbbd60da6d5983bf01937367295d135cccbfce" } }, + { url = "https://files.pythonhosted.org/packages/1e/e5/68eebae3053dbd48508edea559c21b23fbdf3460784f91370c83a86a6acd/coverage-7.15.4-cp311-cp311-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-06T13:47:23Z, size = 253392, hashes = { sha256 = "6879ded16a27f3eeca19b900c147e81616e7054db451471a611b2755ee5249f7" } }, + { url = "https://files.pythonhosted.org/packages/1a/46/fd4ced40a2b691c774e515c9b69500bfa64c7960b67fcee4b2f6fad97fc3/coverage-7.15.4-cp311-cp311-musllinux_1_2_aarch64.whl", upload-time = 2026-08-06T13:47:25Z, size = 255001, hashes = { sha256 = "986be58c3ab54aae8d3496a6225eea74f760fdbe739b38bd442c7e8d133aa53b" } }, + { url = "https://files.pythonhosted.org/packages/53/25/ae2e5fa710bb6957a9aadeb9e3598d3b3e4af6587ce857ad42e8639a3f30/coverage-7.15.4-cp311-cp311-musllinux_1_2_i686.whl", upload-time = 2026-08-06T13:47:26Z, size = 253061, hashes = { sha256 = "c6103639613fe6c1e989082948419bc77a2d26b6c825c99d7fad25f7d3d87afc" } }, + { url = "https://files.pythonhosted.org/packages/d7/31/67ddc0365db2c6e93ac8580bc4bbc50f65273262f973f63ebcdbc15c0495/coverage-7.15.4-cp311-cp311-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-06T13:47:28Z, size = 256831, hashes = { sha256 = "d3af93dddb5659276c63bc16ac6466ac2033a70ca816097bbc06345b8ccdf571" } }, + { url = "https://files.pythonhosted.org/packages/f6/78/82b8fd18f57fb13f12d98fe874995bb2c4f9f17be8aff762c426323fdb96/coverage-7.15.4-cp311-cp311-musllinux_1_2_riscv64.whl", upload-time = 2026-08-06T13:47:29Z, size = 252781, hashes = { sha256 = "b10075e5421d04265766a6d1dac809bbeb8a946fbb23c8f82c227409b2190719" } }, + { url = "https://files.pythonhosted.org/packages/0a/eb/6c74ef4dd12b252e573c49bdef9e2ac265bf3dbb79b8d7feb3266e084e9e/coverage-7.15.4-cp311-cp311-musllinux_1_2_x86_64.whl", upload-time = 2026-08-06T13:47:31Z, size = 253692, hashes = { sha256 = "a67a9f78b2942d87ba8ce3059c642164d2aedd65337377fb52fe9803656bc5c7" } }, + { url = "https://files.pythonhosted.org/packages/5a/66/eb9aed1c3fd2d36ee00eb173f434b14fa607fc056739c9a89ff4244010ea/coverage-7.15.4-cp311-cp311-win32.whl", upload-time = 2026-08-06T13:47:32Z, size = 224461, hashes = { sha256 = "69484d1aca26e322e1c3ce03f09341e84524ababad2d7202161738d83cc9f82e" } }, + { url = "https://files.pythonhosted.org/packages/e2/6d/81fa4161dfb3ed9d74e40d58647eff83a56b7612e78352581280fce2f477/coverage-7.15.4-cp311-cp311-win_amd64.whl", upload-time = 2026-08-06T13:47:34Z, size = 224937, hashes = { sha256 = "63fd6fcd1dd6e158f7eb78606e72933b3f6d01e7b747f99c6c12d764307a0fdc" } }, + { url = "https://files.pythonhosted.org/packages/5b/c1/d8dacf683c6cad3cf85ce68fd3774a6774ec402128822fdfaed920f11e6a/coverage-7.15.4-cp311-cp311-win_arm64.whl", upload-time = 2026-08-06T13:47:36Z, size = 224479, hashes = { sha256 = "ea82116c9893fa89e929b7f197ee5a1950a76e91cc5c85ba503fc02379d04890" } }, + { url = "https://files.pythonhosted.org/packages/1d/48/bc8d4ba7b37551a767bd863f15b3f80182b271c2f55975356f5f7dbe94c2/coverage-7.15.4-cp312-cp312-macosx_10_13_x86_64.whl", upload-time = 2026-08-06T13:47:37Z, size = 222543, hashes = { sha256 = "d4fedd1f7f428f9fe83b1ead5e7cc87a43427be31aadafbac3ac0636dc7abb22" } }, + { url = "https://files.pythonhosted.org/packages/20/dd/88d6f83f1fffc974a3691a34a97951c5b12df7512a6782c5963883cbc058/coverage-7.15.4-cp312-cp312-macosx_11_0_arm64.whl", upload-time = 2026-08-06T13:47:38Z, size = 222905, hashes = { sha256 = "37e2f0cdf58e2e1fed4e4d5a8f8786ae2f7eb80b478016876667dc4a01d60a97" } }, + { url = "https://files.pythonhosted.org/packages/bd/5c/54ee0d4748585bb0acab9891cd8d92f2d3593165b4e59fc9de113bfb3140/coverage-7.15.4-cp312-cp312-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-08-06T13:47:40Z, size = 254407, hashes = { sha256 = "fb55d0e70bb15f2e81477613627286581414693d74ac7963c93a790dd453ca9d" } }, + { url = "https://files.pythonhosted.org/packages/8c/3f/f0642a372f494bd0d7dad3b497083b910194a5f1c88be2c94fef707c3b59/coverage-7.15.4-cp312-cp312-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-08-06T13:47:41Z, size = 257145, hashes = { sha256 = "899b9da30f3c6c336566e3707495bb23e8302d39d862f01fa78c48b99b9437e2" } }, + { url = "https://files.pythonhosted.org/packages/71/17/8b46d0ed68251016002ec972c8fc0119961a765d0984cafb8bf317c43758/coverage-7.15.4-cp312-cp312-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-06T13:47:43Z, size = 258257, hashes = { sha256 = "d15715e8c46552827e5e4f30a35575a2dbcad14454cf3284c54483946bd16931" } }, + { url = "https://files.pythonhosted.org/packages/30/b8/8498a0e72d0adbe15477dd07463d2b3bb2c9f6a4815e8589e50939e2c3ae/coverage-7.15.4-cp312-cp312-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-06T13:47:45Z, size = 260517, hashes = { sha256 = "002a438859f7b430bc99afeaf01a6d187dad1d0dc907b64cdeffc632a5db8fd8" } }, + { url = "https://files.pythonhosted.org/packages/41/e1/7dce19c3bdb1e3dd63e769508216500edad81bd5f69a26d724e32aceaf78/coverage-7.15.4-cp312-cp312-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-06T13:47:46Z, size = 254785, hashes = { sha256 = "e4193a04b518f7968f3099755f5509ee7cccc6dc2b92a6b14841934d22e222c9" } }, + { url = "https://files.pythonhosted.org/packages/dd/b1/e1494703c675a2561723cd9b89f45c9168782c31280c611b1f767851e57c/coverage-7.15.4-cp312-cp312-musllinux_1_2_aarch64.whl", upload-time = 2026-08-06T13:47:48Z, size = 256176, hashes = { sha256 = "e98dcc55d572b38e69d117da7e8e8efb8500f1f5eaf81ecd460a63220790b839" } }, + { url = "https://files.pythonhosted.org/packages/73/76/a5629d270fb638a43a4b10466f51e2f49d532c1aa4da2913cbbb150bbe0a/coverage-7.15.4-cp312-cp312-musllinux_1_2_i686.whl", upload-time = 2026-08-06T13:47:49Z, size = 254321, hashes = { sha256 = "af6c538498ce66c10d3fd541c2a8d5b03da5850355add34e6cba564210cb9e72" } }, + { url = "https://files.pythonhosted.org/packages/ff/4f/9c44447218435d5766b911534f9d798144a5560f85e9a54ebe5f3f5d19f9/coverage-7.15.4-cp312-cp312-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-06T13:47:51Z, size = 258390, hashes = { sha256 = "1d10025d96ea89fc2f73714dbc4cbd433fe012c1ac9e23f895d7728b238b6e52" } }, + { url = "https://files.pythonhosted.org/packages/de/36/c1e127616fb3fa18a9ff71e76c417f2fd7424332a4870015ac224ef4c039/coverage-7.15.4-cp312-cp312-musllinux_1_2_riscv64.whl", upload-time = 2026-08-06T13:47:52Z, size = 253894, hashes = { sha256 = "d802e1947603162ded419bff83ac7489820355d2b856dfb09206574e3a37ac0c" } }, + { url = "https://files.pythonhosted.org/packages/e9/b9/fdb92c8ae7a8bb9b850cc253b7b3b9c8526f68130002048b5671cd510d09/coverage-7.15.4-cp312-cp312-musllinux_1_2_x86_64.whl", upload-time = 2026-08-06T13:47:54Z, size = 255763, hashes = { sha256 = "c2de40895718f91951b86712b4c5b694acaf9a0a49be13874896f599a1eed3f4" } }, + { url = "https://files.pythonhosted.org/packages/6f/c0/a7d51b2587c7bdb76e71b0896d2565bf7d60436b5122fc83e511adb1f7cd/coverage-7.15.4-cp312-cp312-win32.whl", upload-time = 2026-08-06T13:47:56Z, size = 224597, hashes = { sha256 = "5c3431b2161279b7db5c2a1aa58ae02e5cb8c3c42d93a5094be3f5537bd5b11b" } }, + { url = "https://files.pythonhosted.org/packages/49/b9/5c5f80cc55f5acaaca6dee677626bfcec8c87204a7809b438b08e84f4571/coverage-7.15.4-cp312-cp312-win_amd64.whl", upload-time = 2026-08-06T13:47:57Z, size = 225135, hashes = { sha256 = "6befeab5fb2b51c958ca4ac6c5d141a1e8240f4f76e46350f1911963deda49cd" } }, + { url = "https://files.pythonhosted.org/packages/47/e4/2a4561f89ff6bf7c925c287d0f2cce8bdf139c3a33735c87e3203401cf94/coverage-7.15.4-cp312-cp312-win_arm64.whl", upload-time = 2026-08-06T13:47:58Z, size = 224515, hashes = { sha256 = "67bc345491ab55b837277d76f5775d057e8c7f1ac44d890d8c2c82adde258c6f" } }, + { url = "https://files.pythonhosted.org/packages/f1/84/651a9310859673aaa3b3203f1aa1641ca60fcf2494683e1c9474c7172780/coverage-7.15.4-cp313-cp313-macosx_10_13_x86_64.whl", upload-time = 2026-08-06T13:48:00Z, size = 222565, hashes = { sha256 = "c705b28feb2775dc82a25f1d473a370bc37ff93f5177f4e29ce2425f560f6921" } }, + { url = "https://files.pythonhosted.org/packages/82/f9/4dcf700137e8af550670f4d74d1b63828ce93e1e2b05e5f10710eb2ea987/coverage-7.15.4-cp313-cp313-macosx_11_0_arm64.whl", upload-time = 2026-08-06T13:48:02Z, size = 222936, hashes = { sha256 = "3ff205ab5e3ecc670f6a4dd19d9cbf12ede53dd41cfc1e15716ec961ea6d314e" } }, + { url = "https://files.pythonhosted.org/packages/07/4a/612ff1e780b3fbfd637486f542f84adc5503873d8b5d279dec1ffeef9414/coverage-7.15.4-cp313-cp313-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-08-06T13:48:04Z, size = 253926, hashes = { sha256 = "5172326e861a38b48b48befca15e0f477a26b283337a33a739c8fed229934e36" } }, + { url = "https://files.pythonhosted.org/packages/b0/04/d1cff1c2ead4708a6a79c01d3736b6a25bd38a36678398f72a8dd33dfad9/coverage-7.15.4-cp313-cp313-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-08-06T13:48:05Z, size = 256523, hashes = { sha256 = "12b59c90084e3234fb11184886bf4a40f4f16a8c8f867be2e087b81f8e8868d4" } }, + { url = "https://files.pythonhosted.org/packages/b9/80/d34e13fb4b293cbdb9665838cf5522077b8ad14ef947550631a4bced36a5/coverage-7.15.4-cp313-cp313-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-06T13:48:08Z, size = 257759, hashes = { sha256 = "349062d66f00b40fa2c1c222438bad25fabf755631b5d82937fe985c8008615c" } }, + { url = "https://files.pythonhosted.org/packages/0f/e7/2c5fe7636fdb0732fe0f09f308a5b066864078b7fc61f6678e8478554f2e/coverage-7.15.4-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-06T13:48:09Z, size = 259890, hashes = { sha256 = "4256ced708e598e05209bc1a8ab4074e04a51dba4c62fb45926a229af675ace7" } }, + { url = "https://files.pythonhosted.org/packages/92/28/9689f0858dfff59c2ea688938ab9fa2925631235df67126a42b6c5c70ae1/coverage-7.15.4-cp313-cp313-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-06T13:48:11Z, size = 254121, hashes = { sha256 = "d80f974b20782d9612c8b4c9beeca867074c7cf4079d1419843fa25a26428b25" } }, + { url = "https://files.pythonhosted.org/packages/f9/e2/785077c230c157243eb5aa9a26c3be260ecd02001bead54a3cada3df8e03/coverage-7.15.4-cp313-cp313-musllinux_1_2_aarch64.whl", upload-time = 2026-08-06T13:48:13Z, size = 255891, hashes = { sha256 = "2e179f19bfe1d31f8eeeaa12990194d761c4f62f0759661000bca6cd8729f40b" } }, + { url = "https://files.pythonhosted.org/packages/d4/90/e20371b17b40f912f21305c2db2f30efa3de306f7320fc916804872c85a4/coverage-7.15.4-cp313-cp313-musllinux_1_2_i686.whl", upload-time = 2026-08-06T13:48:14Z, size = 253859, hashes = { sha256 = "8bc16bb47b7679670eceff71d78bfb7d6e5b143f6c2cd117487ec7c75e0d4b78" } }, + { url = "https://files.pythonhosted.org/packages/05/49/25371987ee459a5f67c0427fb75c74f9358e65f2c71fe75bf41c1b6c5fcb/coverage-7.15.4-cp313-cp313-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-06T13:48:16Z, size = 258011, hashes = { sha256 = "1cd685005cd2c4200adfc14cf39a603b9320efab3f18a8f7f156d20c9cc3345f" } }, + { url = "https://files.pythonhosted.org/packages/30/6e/32e67467f6154bf4f1c4f63b05acc5097cba4237d45bbeeea446b52e8ac1/coverage-7.15.4-cp313-cp313-musllinux_1_2_riscv64.whl", upload-time = 2026-08-06T13:48:18Z, size = 253676, hashes = { sha256 = "337399ad2c93b3acd2a937627dae8b3e86b66707cd3d3e856347999aadf1ef8d" } }, + { url = "https://files.pythonhosted.org/packages/03/c1/8b24192e89286399765155251f99ee9f070a9d637109018ac23d99b99f6f/coverage-7.15.4-cp313-cp313-musllinux_1_2_x86_64.whl", upload-time = 2026-08-06T13:48:20Z, size = 255453, hashes = { sha256 = "96e257121228ec5cd2bb919276e94ac11074471bc37d68dbae0e8308cce15fff" } }, + { url = "https://files.pythonhosted.org/packages/16/6f/8b41ebdf67c87854e17c035336a90f1cfbad0c14c2a584301be6ff148718/coverage-7.15.4-cp313-cp313-win32.whl", upload-time = 2026-08-06T13:48:21Z, size = 224605, hashes = { sha256 = "c65a9e0dfc6143491879da4e13b5e30f8be192055de508d737fb14601edbd22c" } }, + { url = "https://files.pythonhosted.org/packages/e0/e2/2946c7f0b42b152ecb21ff1bdad72e3d301e790c0c487e4a86e8c9f69347/coverage-7.15.4-cp313-cp313-win_amd64.whl", upload-time = 2026-08-06T13:48:23Z, size = 225148, hashes = { sha256 = "2ff8f5e9b8f7a94f0c11c45631eee103dbcb7d63274edd12c56efe1be690b3b4" } }, + { url = "https://files.pythonhosted.org/packages/9e/83/3f4a69957f48ae7a0aba76c34743f88963d607b19e03f3f8e66f91cae0f9/coverage-7.15.4-cp313-cp313-win_arm64.whl", upload-time = 2026-08-06T13:48:25Z, size = 224536, hashes = { sha256 = "6e0a8a5083b096487d6cfced94cdd514d8f5db6f113610fb36c0620edb1028cf" } }, + { url = "https://files.pythonhosted.org/packages/ea/ac/748cf29eeb2d6be34a3176ce26a4f49e38085ee08e8935f05f6f26ed7e0f/coverage-7.15.4-cp314-cp314-macosx_10_15_x86_64.whl", upload-time = 2026-08-06T13:48:26Z, size = 222608, hashes = { sha256 = "770e9325ab5ea6d56f77e59b29ecfe0ac20b57a82a601876f90494a4dda0386f" } }, + { url = "https://files.pythonhosted.org/packages/0b/02/1abbf5c984677b0aa439cdacaccbf38d248939d8ef8fe1cc7a50d73edb77/coverage-7.15.4-cp314-cp314-macosx_11_0_arm64.whl", upload-time = 2026-08-06T13:48:28Z, size = 222940, hashes = { sha256 = "d12b33a3a50a1676b7784dc8d00a0c6d66a9f2add4b85a041c19b6a7e53ef23c" } }, + { url = "https://files.pythonhosted.org/packages/eb/e1/ff8f9f53d9fcf586125b55d0b1f04ec1c14955fee41e83d5814bee141bb5/coverage-7.15.4-cp314-cp314-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-08-06T13:48:29Z, size = 253985, hashes = { sha256 = "5669c8378ebde86f5def7a25d29586631b58acc27ffde04399f678f3dfc6e082" } }, + { url = "https://files.pythonhosted.org/packages/a1/26/595759762e514e81be1d7d01ed03444303bcd152226a6529998d253f9201/coverage-7.15.4-cp314-cp314-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-08-06T13:48:31Z, size = 256492, hashes = { sha256 = "ff97a14362eef486483ed44042ca2027ea257df6ff768e62358ee0c9776925ac" } }, + { url = "https://files.pythonhosted.org/packages/24/68/b79aabac54d482be23b5fcdd4f4662bff24a78edc4ee29201726929936d5/coverage-7.15.4-cp314-cp314-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-06T13:48:33Z, size = 257837, hashes = { sha256 = "5a325e815318638aed1655d9c06e6d7c2d3d46c09231ce988070428a8762d734" } }, + { url = "https://files.pythonhosted.org/packages/09/0f/bf7f297885a5bf6fd71e5782404e0ff059ca09e8711ceb3a08544abde45a/coverage-7.15.4-cp314-cp314-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-06T13:48:34Z, size = 260152, hashes = { sha256 = "474223409d88eb20d2d6a0d37ea60e8647a65a90cc008dc1f0410af5f64f1e0d" } }, + { url = "https://files.pythonhosted.org/packages/fd/f1/296744e854ff8368542343457414380465e9ceefb9192342feb9d3bc461d/coverage-7.15.4-cp314-cp314-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-06T13:48:36Z, size = 253978, hashes = { sha256 = "7f2f62ae3cd189dd2e13aece758c57b3eecbd27be070dbd4cbd10936049e5dbf" } }, + { url = "https://files.pythonhosted.org/packages/55/b0/bbdb2e9057493e66220a2e149ca2d301ba0e3a58a83bd6b90de9826d16f3/coverage-7.15.4-cp314-cp314-musllinux_1_2_aarch64.whl", upload-time = 2026-08-06T13:48:38Z, size = 255846, hashes = { sha256 = "39ece820e29e0a2ba34b3ecb3be83c27e997eed8926f2ba6fe7ce7a0bda5843b" } }, + { url = "https://files.pythonhosted.org/packages/96/e4/38015b2b6d21258713bd17e76b59d033b191efb5703589cffd037dfbca20/coverage-7.15.4-cp314-cp314-musllinux_1_2_i686.whl", upload-time = 2026-08-06T13:48:39Z, size = 253808, hashes = { sha256 = "f21b56dcace11dfe013014201f577dcd592b2a9b72182d930361b47cf6f73f25" } }, + { url = "https://files.pythonhosted.org/packages/0b/64/0d515c1e60ee6fbfd1a0e79c07cd87d388a233b7adc37758735677203808/coverage-7.15.4-cp314-cp314-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-06T13:48:41Z, size = 258081, hashes = { sha256 = "93a3a0b662abcc10c73a47cbc72cd60f63618d6989fb2d1286e50eacd974f303" } }, + { url = "https://files.pythonhosted.org/packages/91/71/04d9e7a3642146c6351338aef4ef85ab11dbbb54744c13245caba1aad1c0/coverage-7.15.4-cp314-cp314-musllinux_1_2_riscv64.whl", upload-time = 2026-08-06T13:48:43Z, size = 253624, hashes = { sha256 = "141fae2cabf5569b782c10afc4c850ce10f618c13f8db54765cba99cc839da1f" } }, + { url = "https://files.pythonhosted.org/packages/b4/a7/6c28b74c81ebff66987b0e2522ba5cffa3e90b0c33cb6a2eb264d4ee8cf1/coverage-7.15.4-cp314-cp314-musllinux_1_2_x86_64.whl", upload-time = 2026-08-06T13:48:45Z, size = 255280, hashes = { sha256 = "81294c7e6ab30c5f74c0353b11b2fd6320e72d9bee6ac73b357caa8b916323a5" } }, + { url = "https://files.pythonhosted.org/packages/52/af/bc19996a7014b98d7bbb0f0939453c67074af65784a3aa16a789a07381fa/coverage-7.15.4-cp314-cp314-win32.whl", upload-time = 2026-08-06T13:48:47Z, size = 224768, hashes = { sha256 = "7bbd7d6418e0dab31a206af5203bd43ae36edb8e7fba1940b055d3e9249290d7" } }, + { url = "https://files.pythonhosted.org/packages/ee/90/219484e476d6e101ba0a444852579e05f5b75c37c611a42ed1190f73ef62/coverage-7.15.4-cp314-cp314-win_amd64.whl", upload-time = 2026-08-06T13:48:49Z, size = 225259, hashes = { sha256 = "f0204ed122758782970526057093f448051a39db9d810d4e344bb87a3546f425" } }, + { url = "https://files.pythonhosted.org/packages/b7/66/fa77daf4e383e5f776dac62c2409b6af81910ae6fe326bd5170dba74cc63/coverage-7.15.4-cp314-cp314-win_arm64.whl", upload-time = 2026-08-06T13:48:51Z, size = 224684, hashes = { sha256 = "9e71e7bc71c686a123347ae47a0de33a175e797a85bb57b791492adf4eec8ed8" } }, + { url = "https://files.pythonhosted.org/packages/58/5b/f03bf0ce362bbf3f785fa5219620d00778d4ac6fc9e407734828e9c672f6/coverage-7.15.4-cp314-cp314t-macosx_10_15_x86_64.whl", upload-time = 2026-08-06T13:48:52Z, size = 223338, hashes = { sha256 = "7c922735321eef3f87c280a3d39afff6b646723a2880b862cda4ac7a093b8aa8" } }, + { url = "https://files.pythonhosted.org/packages/0f/76/e77d0ae22501831cc9f92193e8a957a5caa1dd177f90a6d1d9b106242d92/coverage-7.15.4-cp314-cp314t-macosx_11_0_arm64.whl", upload-time = 2026-08-06T13:48:54Z, size = 223609, hashes = { sha256 = "f41c17c4668a655ce96d090d8d5ffdc24ef64b5a02f9753884d08483e8a4a41a" } }, + { url = "https://files.pythonhosted.org/packages/82/1a/b1f089da8d38ac612fa2dd6dc7f4a1a7657d12f3e261d2996edd3a838d0b/coverage-7.15.4-cp314-cp314t-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-08-06T13:48:56Z, size = 264970, hashes = { sha256 = "46822e9b6ff1c6a72b518c162c44a8f45a61a1d609c51084bf5b16c023c5037b" } }, + { url = "https://files.pythonhosted.org/packages/bf/31/e66d98d6e9c7fcc88470f1e234eaf6b1950dc0dfbf797f7282c1c861da24/coverage-7.15.4-cp314-cp314t-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-08-06T13:48:58Z, size = 267088, hashes = { sha256 = "3d6f4955b73b5445271379a59e3792b0d978f42d4a01e0cf7a67d9c33a3bb0a5" } }, + { url = "https://files.pythonhosted.org/packages/59/a1/ae94eb2c541add426378408379f233591e069040b1e2cdb33df9498a0682/coverage-7.15.4-cp314-cp314t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-06T13:49:00Z, size = 269508, hashes = { sha256 = "3fc9e047706fb4a9abb54f719d3aa643e80e5bb3818182c40aee01ac0f0247ba" } }, + { url = "https://files.pythonhosted.org/packages/9c/c7/88a10694a1c6a213569766aba9f25847b28155d4ac731b13226db216356d/coverage-7.15.4-cp314-cp314t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-06T13:49:02Z, size = 270629, hashes = { sha256 = "05e491d4f3165d62d4f5c8fd48dfeabf2ae8f42cbbd484319af33ea851b78982" } }, + { url = "https://files.pythonhosted.org/packages/b3/34/d8b8232e5e55169933b59aabcef2fedfa4b9d8897361bb80fcbda146505f/coverage-7.15.4-cp314-cp314t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-06T13:49:04Z, size = 264043, hashes = { sha256 = "226c66e80ec0598d3b9b4874123df167ccca342aca8714f77cac6829688ee09c" } }, + { url = "https://files.pythonhosted.org/packages/7e/35/58b009dbf8c471c7224716478b9fed4a7e1af15320e1ed41660978504663/coverage-7.15.4-cp314-cp314t-musllinux_1_2_aarch64.whl", upload-time = 2026-08-06T13:49:05Z, size = 266963, hashes = { sha256 = "ac41cc14bebda0dbfb0628036b7f75706935c95bcc07fefe9a0f93614aa60a57" } }, + { url = "https://files.pythonhosted.org/packages/62/aa/57fbda1b42c892968273c56b6ee9dc0f1310850859230a507bc7873b1f65/coverage-7.15.4-cp314-cp314t-musllinux_1_2_i686.whl", upload-time = 2026-08-06T13:49:07Z, size = 264569, hashes = { sha256 = "8af623e5cd92080acddd02b38f2f406a2c3a0893c38950b211890361448fbf26" } }, + { url = "https://files.pythonhosted.org/packages/98/8a/360e6e7f24d477b7e889703af0afa878d15b6d4d8d2a822b2835c169a879/coverage-7.15.4-cp314-cp314t-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-06T13:49:09Z, size = 268299, hashes = { sha256 = "07545711d4f0f32852a18f18ad11f76f0109909d09e78b9008b4cfc67e829429" } }, + { url = "https://files.pythonhosted.org/packages/4e/89/6f701261aee21b6b5fa8f7872229406dc917e125069448292223bf213606/coverage-7.15.4-cp314-cp314t-musllinux_1_2_riscv64.whl", upload-time = 2026-08-06T13:49:11Z, size = 263413, hashes = { sha256 = "a0865421cfdc53654b342d515e5a233187590882d20b95752150e53f65460017" } }, + { url = "https://files.pythonhosted.org/packages/3f/0f/6f04036edc260ed425af83e834f627fad48941ce97b50bfe6edd8b6fa623/coverage-7.15.4-cp314-cp314t-musllinux_1_2_x86_64.whl", upload-time = 2026-08-06T13:49:13Z, size = 265725, hashes = { sha256 = "460115e32ee40566476db5048f9bec1e842c127ad8e6f8be745aad3ac9cbc839" } }, + { url = "https://files.pythonhosted.org/packages/c4/ce/d19b5d4d5c49a7bfb925fd74310fee7d28bc99520ac3367ccbc54e662518/coverage-7.15.4-cp314-cp314t-win32.whl", upload-time = 2026-08-06T13:49:15Z, size = 225079, hashes = { sha256 = "cbde877ef9dd7baf272b9bfef2b8a25edd45d9170fc326951dd20eb480335e85" } }, + { url = "https://files.pythonhosted.org/packages/26/bb/7aa1b3b173faee0679037ca950bbbe1247273656697994d8d13f80f8d4b4/coverage-7.15.4-cp314-cp314t-win_amd64.whl", upload-time = 2026-08-06T13:49:17Z, size = 225911, hashes = { sha256 = "3da9e92d1c551fd7563833e9ade686efb0c4b7363ab7681a94283958c950bf5e" } }, + { url = "https://files.pythonhosted.org/packages/81/1c/4ea9e47426d80038d9222db3c4534cb6021a74b237d3ff97ffd33b6600dd/coverage-7.15.4-cp314-cp314t-win_arm64.whl", upload-time = 2026-08-06T13:49:19Z, size = 225219, hashes = { sha256 = "3a54f5a0d85050c73a38f6793090ee83974531e67fe5e57a1da9bee11398aa5e" } }, + { url = "https://files.pythonhosted.org/packages/2b/c4/dc5d2ac8f9142e7ec7de66e7bf0591db29d78955a040bd915870d9c0e657/coverage-7.15.4-cp315-cp315-macosx_10_15_x86_64.whl", upload-time = 2026-08-06T13:49:21Z, size = 222604, hashes = { sha256 = "2c9872e4d9dc5d3cf616bf4b382f5a00359305a5be666a3dd0b5cdb4e49597f9" } }, + { url = "https://files.pythonhosted.org/packages/70/39/33e63df81fe2ee100897451841c821467635923e58e37c6bd4b46dd8106c/coverage-7.15.4-cp315-cp315-macosx_11_0_arm64.whl", upload-time = 2026-08-06T13:49:23Z, size = 222944, hashes = { sha256 = "e101dbb4b9b72f0cddd8cdc8c9c5b47f456766f5e0ac82dbfb75e5c55409b78a" } }, + { url = "https://files.pythonhosted.org/packages/99/1f/ef3ffb5557febc75a0d97aa459d0266d7d741110265121cc6d8539343d44/coverage-7.15.4-cp315-cp315-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-08-06T13:49:25Z, size = 254050, hashes = { sha256 = "7d1abebdb047729e852b9c77a00497dfbeb11eb3a117e037d7dbc3ac8e5f5c54" } }, + { url = "https://files.pythonhosted.org/packages/6f/f5/1f0f6f77698c3601ca0ae7431e34b24c62ca2f06fecb23b73ed1f651d2be/coverage-7.15.4-cp315-cp315-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-08-06T13:49:26Z, size = 256967, hashes = { sha256 = "d28a4a899354d0ea6214cc59b4fa19eefbce1b9ff1688ab579acf49e894bd3fb" } }, + { url = "https://files.pythonhosted.org/packages/03/7a/2ed9bed79925f4367c83c77f66a89e5ca7229c288d2d19ad5f36d1ca0070/coverage-7.15.4-cp315-cp315-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-06T13:49:28Z, size = 258587, hashes = { sha256 = "ffb3c2aacea411cc7e1d27712490c11108e2de1d39019ae32915493a59a8b9ed" } }, + { url = "https://files.pythonhosted.org/packages/45/8c/fa34044f71b7cc4ecb6da9c2408770959b0591fa9b5fb6fb6bca38f94298/coverage-7.15.4-cp315-cp315-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-06T13:49:30Z, size = 260785, hashes = { sha256 = "a9447978a92f405d301123cfd39ff49895490efb769a758fe2734c7f631bf8ce" } }, + { url = "https://files.pythonhosted.org/packages/4f/54/d5727ce36b4524a7394ab9f5f1df378e1f23affcdab01037dc8655185cc7/coverage-7.15.4-cp315-cp315-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-06T13:49:32Z, size = 254545, hashes = { sha256 = "050467a7983b8e2fe7dd41a78bb30c3e7f8c0b8cafda14b1c46f8b5e3cf2dd3c" } }, + { url = "https://files.pythonhosted.org/packages/dc/e6/6e3783e576719590194bdffb6dd6d85490801785b7c331e35a245d8cb8b5/coverage-7.15.4-cp315-cp315-musllinux_1_2_aarch64.whl", upload-time = 2026-08-06T13:49:34Z, size = 256682, hashes = { sha256 = "d003b7a5708ddad5c206c79607a6b92abb6fc13c57d99d8a4468cc03a2941ced" } }, + { url = "https://files.pythonhosted.org/packages/dc/f2/bacdbde18b69ed2de424fcf64d9fb0a4913753d4f0eca8bae9daad69f4bd/coverage-7.15.4-cp315-cp315-musllinux_1_2_i686.whl", upload-time = 2026-08-06T13:49:36Z, size = 254560, hashes = { sha256 = "c38efe30fd74e5c19e9433f11fb1f5dc9c6522770971b7c6145bbaa413dc8800" } }, + { url = "https://files.pythonhosted.org/packages/6c/a3/1fb927196e3477c1b48831169ab58ba08f451ba87ae311ff1de68b26a616/coverage-7.15.4-cp315-cp315-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-06T13:49:38Z, size = 258792, hashes = { sha256 = "1f4f826d70f772ab8b0c052329580d7fe8b8abd191e4ce0c8f81aec6614665d3" } }, + { url = "https://files.pythonhosted.org/packages/41/58/30d4c149c69053de0edfe325614c1d28d508f62b1783e0e4a234d2e49136/coverage-7.15.4-cp315-cp315-musllinux_1_2_riscv64.whl", upload-time = 2026-08-06T13:49:39Z, size = 253968, hashes = { sha256 = "4a4bf917c9953f57c957be31c1cd504e3bd2f34d4a352b9d391a3025336f6768" } }, + { url = "https://files.pythonhosted.org/packages/89/e4/77f639371b918aad30dda4051f95404b43578f7f2e2f87ba73e02ed1ff37/coverage-7.15.4-cp315-cp315-musllinux_1_2_x86_64.whl", upload-time = 2026-08-06T13:49:41Z, size = 255893, hashes = { sha256 = "1c9bf40ebef178a45192c75c4964760bb261b0e6ad725da5fc4c93f674f19753" } }, + { url = "https://files.pythonhosted.org/packages/5c/62/13be29b3ddab35f14c87967a4820a05106d2a3eccb4fa4ff550bf30b75e0/coverage-7.15.4-cp315-cp315-win32.whl", upload-time = 2026-08-06T13:49:44Z, size = 224768, hashes = { sha256 = "43619d04c3671792d2c4706ae8bf45e265dc87bbd4078189ef8b847ea1e74be2" } }, + { url = "https://files.pythonhosted.org/packages/a1/70/af0c6be0f964af6954f6b74bc109b0dbca02824696d2520fb17fe1ab06e3/coverage-7.15.4-cp315-cp315-win_amd64.whl", upload-time = 2026-08-06T13:49:45Z, size = 225242, hashes = { sha256 = "be619439dbcd31a2eab10b32de9fff62c26ed4bab69dc32b8363fdaaa0882809" } }, + { url = "https://files.pythonhosted.org/packages/4f/2d/f3bd3aab899fc9efc18b53133ee68f5f98574ef480649b23e12962226387/coverage-7.15.4-cp315-cp315-win_arm64.whl", upload-time = 2026-08-06T13:49:48Z, size = 224674, hashes = { sha256 = "def597967dafc2e8d97c9097ea453c464e0bb8ed38f193a43070f10dc623bb6d" } }, + { url = "https://files.pythonhosted.org/packages/f5/ca/f69251cd63eabc6438321aea22148754cce758a26bde07dd490e3fe7cfc5/coverage-7.15.4-cp315-cp315t-macosx_10_15_x86_64.whl", upload-time = 2026-08-06T13:49:50Z, size = 223333, hashes = { sha256 = "c7dbc748ac8a1e3e59a2b28bea47675e6e778081dbbf081bde0d75def2fcbe1d" } }, + { url = "https://files.pythonhosted.org/packages/a7/a7/037b53b2885b0d8447064432491a4d5a1014cd9f97a594d53acd0c04541a/coverage-7.15.4-cp315-cp315t-macosx_11_0_arm64.whl", upload-time = 2026-08-06T13:49:52Z, size = 223630, hashes = { sha256 = "2413074a5ecbb61a01a7888fc72db0ca324d13588c5b38bc0dd8564cdcdfea26" } }, + { url = "https://files.pythonhosted.org/packages/80/4f/152b8a4779ae90da11bb24f7467df8a59f0be48a5c52acb856325ca48289/coverage-7.15.4-cp315-cp315t-manylinux1_i686.manylinux_2_28_i686.manylinux_2_5_i686.whl", upload-time = 2026-08-06T13:49:54Z, size = 264489, hashes = { sha256 = "4e6f6f632b7b2f714bf7a1346e8f97b650ee71f3c298aaad42a2ab60f0f07645" } }, + { url = "https://files.pythonhosted.org/packages/10/2d/84b4b9e0e1dd6528a51920ff7031f35b789382e467a28ec6a5a578cb8812/coverage-7.15.4-cp315-cp315t-manylinux1_x86_64.manylinux_2_28_x86_64.manylinux_2_5_x86_64.whl", upload-time = 2026-08-06T13:49:56Z, size = 267567, hashes = { sha256 = "8df457da2249d3c75ca2e5e835d59c725abfe92d27fdff6cd99eed85b51d5e9a" } }, + { url = "https://files.pythonhosted.org/packages/53/fc/ba01cc25299f9f8a2c8b02d3b28c53f3543d9fbfbe4e74fa2760b48f163e/coverage-7.15.4-cp315-cp315t-manylinux2014_aarch64.manylinux_2_17_aarch64.manylinux_2_28_aarch64.whl", upload-time = 2026-08-06T13:49:58Z, size = 270123, hashes = { sha256 = "050f66a08805acb5b8a23c6d4a517b1ecf82c08e81ed0e4bd727df065e5c6624" } }, + { url = "https://files.pythonhosted.org/packages/cf/d0/db2647cbf40b14f8c308f94ff7bf89c06d564e59f396906edf50086ec788/coverage-7.15.4-cp315-cp315t-manylinux2014_ppc64le.manylinux_2_17_ppc64le.manylinux_2_28_ppc64le.whl", upload-time = 2026-08-06T13:50:00Z, size = 271107, hashes = { sha256 = "1587fb771d1ccceef708fdde1e5af8c7ed24b486b61d13a321acb7d8145390aa" } }, + { url = "https://files.pythonhosted.org/packages/70/ff/4d2d17924552c458bb4f77dd631f0e3bc92fbbdf2d2d916cd4b33bbfd5b1/coverage-7.15.4-cp315-cp315t-manylinux_2_31_riscv64.manylinux_2_39_riscv64.whl", upload-time = 2026-08-06T13:50:03Z, size = 264955, hashes = { sha256 = "8b4f1c3a69ca580f3fbd6b2046915f536d7f586874f25c1bb23add2a3c88d50f" } }, + { url = "https://files.pythonhosted.org/packages/ee/de/dc010c7a3691f396d93bbc26bfcafa1c2a3a351cd520470f15faf5795bd5/coverage-7.15.4-cp315-cp315t-musllinux_1_2_aarch64.whl", upload-time = 2026-08-06T13:50:05Z, size = 267949, hashes = { sha256 = "ffb58d7eff5b7f6ecc6fa21d6288ab7f968a212cb67d682c269c09b9eba3b66f" } }, + { url = "https://files.pythonhosted.org/packages/78/ea/dc96a11375e83c045c2f7c61fb6918277cfe9401db7c0f7b1d111a84b2e5/coverage-7.15.4-cp315-cp315t-musllinux_1_2_i686.whl", upload-time = 2026-08-06T13:50:07Z, size = 264421, hashes = { sha256 = "d9df165544774574ee004b953023d1bebada1894a80b1052a43d798b0f676e67" } }, + { url = "https://files.pythonhosted.org/packages/c8/86/b77131a0f9503ce461cd577076147d7a9040f0c5dda772686f729e2cc9cb/coverage-7.15.4-cp315-cp315t-musllinux_1_2_ppc64le.whl", upload-time = 2026-08-06T13:50:09Z, size = 269121, hashes = { sha256 = "f9de0a24a4079b53e523b5c5e2c5945ec251ab486652659955187cf255a259bc" } }, + { url = "https://files.pythonhosted.org/packages/24/24/944bc35007862955e7ebf05754e645419dcf5d7526c52735cfa2715e8ebf/coverage-7.15.4-cp315-cp315t-musllinux_1_2_riscv64.whl", upload-time = 2026-08-06T13:50:11Z, size = 264565, hashes = { sha256 = "150089274bdc9f940628552cb92844e0223c987f1902ab8efe9f45a2ec758d88" } }, + { url = "https://files.pythonhosted.org/packages/c7/cc/a3bb9f93e7e740659163e2ea584f8196ddcd2c456a5dbe15f6c50105fec1/coverage-7.15.4-cp315-cp315t-musllinux_1_2_x86_64.whl", upload-time = 2026-08-06T13:50:13Z, size = 266522, hashes = { sha256 = "a58a94fed5da6997d258e8f7668c1e195fbd04a691d781b7558f1e468f9e68bc" } }, + { url = "https://files.pythonhosted.org/packages/49/dd/e0e40f3560d878d888c580698ff5ad1179f5e1c3ac949684ef66b41a3817/coverage-7.15.4-cp315-cp315t-win32.whl", upload-time = 2026-08-06T13:50:15Z, size = 225068, hashes = { sha256 = "ebd5a6d8466ff30836572f3ba2cae8a5e8f85029b1c6d5e2ed338dc472a5166a" } }, + { url = "https://files.pythonhosted.org/packages/c6/7e/37732ea80eebc30e976e4cdab15c190bc42d96959a42e38ddf6f8c60468f/coverage-7.15.4-cp315-cp315t-win_amd64.whl", upload-time = 2026-08-06T13:50:17Z, size = 225895, hashes = { sha256 = "288bde2a2d7ab6b6c2d7252fcde8b524387f2d970bdba9658fc6f8bbcaef0f9b" } }, + { url = "https://files.pythonhosted.org/packages/c6/08/1e00f7923eaaba45fb3d51dd794125fc766304b1df264f3a9c6557bfb30e/coverage-7.15.4-cp315-cp315t-win_arm64.whl", upload-time = 2026-08-06T13:50:19Z, size = 225213, hashes = { sha256 = "68be5e1de60ff13c9095bbec0e5a7fa45b33b101752215b91345ea1f61c4a278" } }, + { url = "https://files.pythonhosted.org/packages/b4/d9/e70c286c979378f061d8266e279b686ab0b0b688e1fe0af864684f23a77d/coverage-7.15.4-py3-none-any.whl", upload-time = 2026-08-06T13:50:22Z, size = 214332, hashes = { sha256 = "964730a1e9de9c0cf11be6a1a3c79ce419c34882842abd256086ba4698705e84" } }, ] [[packages]] @@ -188,11 +186,11 @@ wheels = [ [[packages]] name = "mako" -version = "1.3.12" +version = "1.4.1" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/00/62/791b31e69ae182791ec67f04850f2f062716bbd205483d63a215f3e062d3/mako-1.3.12.tar.gz", upload-time = 2026-04-28T19:01:08Z, size = 400219, hashes = { sha256 = "9f778e93289bd410bb35daadeb4fc66d95a746f0b75777b942088b7fd7af550a" } } +sdist = { url = "https://files.pythonhosted.org/packages/2a/12/b5fa2353e2754cd67fb9f83793fa48ff42c213a5da7e719869d2301f6ab8/mako-1.4.1.tar.gz", upload-time = 2026-08-05T06:10:56Z, size = 410165, hashes = { sha256 = "d7904710b662996425a21627710c4777c45053146942cf8a7aebf757c92b8c27" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/bc/b1/a0ec7a5a9db730a08daef1fdfb8090435b82465abbf758a596f0ea88727e/mako-1.3.12-py3-none-any.whl", upload-time = 2026-04-28T19:01:10Z, size = 78521, hashes = { sha256 = "8f61569480282dbf557145ce441e4ba888be453c30989f879f0d652e39f53ea9" } }, + { url = "https://files.pythonhosted.org/packages/a5/54/12ed58d458474aaab5c3d180173e745a4fe131bb330370596876d19ff60f/mako-1.4.1-py3-none-any.whl", upload-time = 2026-08-05T06:10:58Z, size = 80010, hashes = { sha256 = "a359d9a94a541213958742b2698d0a7757bb83551767bc468a74b9905aba9617" } }, ] [[packages]] @@ -282,20 +280,20 @@ wheels = [ [[packages]] name = "packaging" -version = "26.2" +version = "26.3" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/d7/f1/e7a6dd94a8d4a5626c03e4e99c87f241ba9e350cd9e6d75123f992427270/packaging-26.2.tar.gz", upload-time = 2026-04-24T20:15:23Z, size = 228134, hashes = { sha256 = "ff452ff5a3e828ce110190feff1178bb1f2ea2281fa2075aadb987c2fb221661" } } +sdist = { url = "https://files.pythonhosted.org/packages/7d/fa/3944b40b07da9ce895c0e6303a5ab7d53da063554f534556b134a54d6093/packaging-26.3.tar.gz", upload-time = 2026-08-04T18:15:28Z, size = 313412, hashes = { sha256 = "94edc256424af38762eb31306eed28beb9f0efc50a8837492c9d6fd6004aed79" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/df/b2/87e62e8c3e2f4b32e5fe99e0b86d576da1312593b39f47d8ceef365e95ed/packaging-26.2-py3-none-any.whl", upload-time = 2026-04-24T20:15:22Z, size = 100195, hashes = { sha256 = "5fc45236b9446107ff2415ce77c807cee2862cb6fac22b8a73826d0693b0980e" } }, + { url = "https://files.pythonhosted.org/packages/63/34/ba1c580383c9eada3711951fef0795c80b829a078d72188184bcab9dd527/packaging-26.3-py3-none-any.whl", upload-time = 2026-08-04T18:15:27Z, size = 129956, hashes = { sha256 = "d7193f7c8e4e93f444fde0262bf90af30e16fa0ad0ad44cb553c87339b23cd1c" } }, ] [[packages]] name = "parse" -version = "1.22.0" +version = "1.22.1" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/7b/a2/dd269daedd5ac3a244ca7855b4878d8655393fd4554d5c24a56bc31e302a/parse-1.22.0.tar.gz", upload-time = 2026-05-02T01:36:25Z, size = 36767, hashes = { sha256 = "d4987d68ccf08b6ba3bf80b5004ff7de61c4337cba2d8350ae5c9925794979d9" } } +sdist = { url = "https://files.pythonhosted.org/packages/a4/f2/0b504486c2a5564798607d3860e48ed19c6443d5e9cc3ec61cc6b8b4ef58/parse-1.22.1.tar.gz", upload-time = 2026-05-26T03:44:52Z, size = 36970, hashes = { sha256 = "d3a4740ec3da338e2b258b2d69741b731eadfddca59e24a14bc4ee5fce38c911" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/69/3a/0c2cf5922c6133b74c1cebe4b66f6949818e2cf8121aa59e3ebcd64ac6ac/parse-1.22.0-py2.py3-none-any.whl", upload-time = 2026-05-02T01:36:24Z, size = 20839, hashes = { sha256 = "eea8ed34e2614cea65d9c1d4af9cb68cce26aea13d44bdcaf83c1b40884fe945" } }, + { url = "https://files.pythonhosted.org/packages/6f/c5/7c16e99869e1f422629092cfd23e3b58e461988c3f9c36fd3624bb4142e6/parse-1.22.1-py2.py3-none-any.whl", upload-time = 2026-05-26T03:44:51Z, size = 20925, hashes = { sha256 = "20f0925a46f06602485ac90d751764d0697fd8455aaa97489ba8953a4b66de32" } }, ] [[packages]] @@ -327,37 +325,39 @@ wheels = [ [[packages]] name = "pyrefly" -version = "1.0.0" +version = "1.2.0" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/9f/3a/9045b0097ac58979c7c30a4fa0e673db942d4adbc7b6d439bd54ae58c441/pyrefly-1.0.0.tar.gz", upload-time = 2026-05-12T20:12:46Z, size = 5677995, hashes = { sha256 = "5c2b810ffcebd84be71de5df1223651edee951653a66935c6f091e957c452455" } } +sdist = { url = "https://files.pythonhosted.org/packages/89/01/a86e9f24722b095c3f88e3616132b75a21b0df53804bdc6a45314dd4d93c/pyrefly-1.2.0.tar.gz", upload-time = 2026-08-01T02:56:27Z, size = 6243654, hashes = { sha256 = "5485f960fc2481617068c918335c39ab1507ef90b6b5bd35bf57726e60e73185" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/f4/c6/90788819bac9c61dd7bacba53b79f3c12d47ccbe5e51b3d6d89f2387e1d2/pyrefly-1.0.0-py3-none-macosx_10_12_x86_64.whl", upload-time = 2026-05-12T20:12:20Z, size = 13122950, hashes = { sha256 = "e355a0908555348ed4b9585ef25c76ff566673e345c866c325f1633f44d890b6" } }, - { url = "https://files.pythonhosted.org/packages/82/91/a3cf2a1e87d336eaa804a1e6fc93266faf6dc2a97eecdbc7eae289628022/pyrefly-1.0.0-py3-none-macosx_11_0_arm64.whl", upload-time = 2026-05-12T20:12:23Z, size = 12599494, hashes = { sha256 = "a7038efc3a40f8294edee339895633cf22db268c0d434cdbcbefc34f78a9ecc3" } }, - { url = "https://files.pythonhosted.org/packages/cd/ab/74d1e11e737e99b1c003ecc5d7d2e846c4ea1f328966bfdbbd0ac63fad0a/pyrefly-1.0.0-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", upload-time = 2026-05-12T20:12:25Z, size = 12995507, hashes = { sha256 = "da331ca515ed1c08791da2b5f664cf9c1294c48fd802133262e7d5d51e0f4416" } }, - { url = "https://files.pythonhosted.org/packages/7c/ac/2df0899f8464c97e5d995f994c97c5cb5b0f58610432aa90d26d924e1db5/pyrefly-1.0.0-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", upload-time = 2026-05-12T20:12:29Z, size = 13947693, hashes = { sha256 = "c74219d8f3e63cdaa5501a0b21d1c9d37011820f9606728d0ed06f09ae86a878" } }, - { url = "https://files.pythonhosted.org/packages/6b/3e/b247c24321e36f04b7d51f9ccf3df93e5009e4b29939524b36ec2e17dc2a/pyrefly-1.0.0-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", upload-time = 2026-05-12T20:12:31Z, size = 13925803, hashes = { sha256 = "c0d05543b1bb6ee6d64149eb5d6b2fb15aa72d3962d6a97abca0afaca8b0c131" } }, - { url = "https://files.pythonhosted.org/packages/61/16/cfa2d61a4aa1e1f7bca48bb37acd01c6a09db4864b16a54f9587092765ff/pyrefly-1.0.0-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", upload-time = 2026-05-12T20:12:35Z, size = 13470398, hashes = { sha256 = "1382d5b1fcdb49a4de9f34d112d2bddf290a78ff93ee8149492ad5f1077ddffc" } }, - { url = "https://files.pythonhosted.org/packages/cb/2b/6372c7dddb326223e24a46b17efd0d4bd7b4fe22c821e523157577eed2d2/pyrefly-1.0.0-py3-none-win32.whl", upload-time = 2026-05-12T20:12:38Z, size = 12222643, hashes = { sha256 = "aa8b5d0e47080e3202a2547b39f7a5a61d2c781c712b3b67884f745ca2c759d2" } }, - { url = "https://files.pythonhosted.org/packages/be/ad/1d23be700b6b2ddaeb362360c7145917a8edbbf7240ae428d40541772fce/pyrefly-1.0.0-py3-none-win_amd64.whl", upload-time = 2026-05-12T20:12:41Z, size = 13146369, hashes = { sha256 = "c8abcb0f2082e83c890375128f9cff4aa4d3f210b85eea7b3046c1ae764e77f5" } }, - { url = "https://files.pythonhosted.org/packages/8c/38/16589134f3012fd097a10dcc85771555f1a5fb76e04b682597180743af30/pyrefly-1.0.0-py3-none-win_arm64.whl", upload-time = 2026-05-12T20:12:43Z, size = 12538326, hashes = { sha256 = "d150fa9e40e8392832be81c3bcfc0497c146674ce4d0f8e04e1ec29e775ffb8c" } }, + { url = "https://files.pythonhosted.org/packages/7d/9d/3c0ef1d4843987b22f996ed381ec9cf5a3b1273e29804db276252e4c95eb/pyrefly-1.2.0-py3-none-macosx_10_12_x86_64.whl", upload-time = 2026-08-01T02:56:02Z, size = 14026305, hashes = { sha256 = "7f46d983ac49ddd2b043694960a01dc6a19a5cfd8eec609d6bd9c42866f91b4e" } }, + { url = "https://files.pythonhosted.org/packages/0a/06/03bbb78fbea54cdc65b626619f3597d5611aca4fdef11e72a4e8360e7e63/pyrefly-1.2.0-py3-none-macosx_11_0_arm64.whl", upload-time = 2026-08-01T02:56:04Z, size = 13463880, hashes = { sha256 = "756f669b5555090f5c1a4fef30db1785fabe657764f7e4e6dc88994dfb8ca82d" } }, + { url = "https://files.pythonhosted.org/packages/13/5a/7d8bc00a38e93bbc9c3e7bd14d305f7948717e667c9bcddeab9dd42fd255/pyrefly-1.2.0-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", upload-time = 2026-08-01T02:56:07Z, size = 13907329, hashes = { sha256 = "e3465812ce5ef4781fb592edbf2724547296f0a3124be115d73c7e8b2401862d" } }, + { url = "https://files.pythonhosted.org/packages/be/94/9e08b4bf799d0b8f36b55a2783c7ba5f51730cf0632a85a67b5b5ed876cd/pyrefly-1.2.0-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", upload-time = 2026-08-01T02:56:09Z, size = 15039020, hashes = { sha256 = "5de7b2ad2bba5c8055181681a84b74143eac2234a48ba5d1b7ed7e7a722b02bd" } }, + { url = "https://files.pythonhosted.org/packages/5b/bd/bca5fd0c80f4daf8ee6903a29df9f3de1feb05ff0946b8f35ec8c5096b13/pyrefly-1.2.0-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", upload-time = 2026-08-01T02:56:11Z, size = 14986199, hashes = { sha256 = "25822ea9505f589ea8a725e4268b475132fb89e038fbf092e446510443ac142a" } }, + { url = "https://files.pythonhosted.org/packages/97/f7/f07087f3d185ad2eced0c56cef89ca5474dfb4ff25f146cd50a861c97553/pyrefly-1.2.0-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", upload-time = 2026-08-01T02:56:14Z, size = 14393715, hashes = { sha256 = "90efe75e17491ef5d636e10469e9278d7d0256b3b4c5e1f4750069bf3ae0f5d1" } }, + { url = "https://files.pythonhosted.org/packages/d3/70/0d142c320e284b9e3ce35e9b1e58b8ce2ee1f578f2a7234bc30e5022b94f/pyrefly-1.2.0-py3-none-musllinux_1_2_aarch64.whl", upload-time = 2026-08-01T02:56:16Z, size = 13933008, hashes = { sha256 = "368aaf7eee4f511ddc0f8e564cf14e01ab2f10b0db9105c6d5b153bf498d07bf" } }, + { url = "https://files.pythonhosted.org/packages/5d/e8/e84f11b6e1f63fd453ad3654213b9a0f6f4de8cef6b58038eef2d0d5955d/pyrefly-1.2.0-py3-none-musllinux_1_2_x86_64.whl", upload-time = 2026-08-01T02:56:18Z, size = 14431827, hashes = { sha256 = "d52d5da7bc65fb7675fbaa80eda879d4f8787c494f04cac21603330d3abbdbbe" } }, + { url = "https://files.pythonhosted.org/packages/0f/06/810d31380f66c75e1c0779a408d3b16117b1b368b57894f6aa66bef21686/pyrefly-1.2.0-py3-none-win32.whl", upload-time = 2026-08-01T02:56:20Z, size = 13229447, hashes = { sha256 = "8c90751de8506d938e8f802659c74cf35bd7a0036510ee6c634a38eebb280bfa" } }, + { url = "https://files.pythonhosted.org/packages/ed/98/4dafa3c7a1caed2dc8cc708dde09ba27963c7736508f55b626fff3024113/pyrefly-1.2.0-py3-none-win_amd64.whl", upload-time = 2026-08-01T02:56:23Z, size = 14087387, hashes = { sha256 = "8a8964c224ccc4882730130955815de21ff443c1ac3f0b90685b19bf63848170" } }, + { url = "https://files.pythonhosted.org/packages/1b/1c/df3cb0a2e5591660ded7a1836cd2f29dc48c91adb1c0a3a700a96f6d09e1/pyrefly-1.2.0-py3-none-win_arm64.whl", upload-time = 2026-08-01T02:56:25Z, size = 13430873, hashes = { sha256 = "3a90bb8df39dfbac74b1f3b2e9d7c526b8f80568884c3944d955023a73ebf61e" } }, ] [[packages]] name = "pytest" -version = "9.0.3" +version = "9.1.1" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/7d/0d/549bd94f1a0a402dc8cf64563a117c0f3765662e2e668477624baeec44d5/pytest-9.0.3.tar.gz", upload-time = 2026-04-07T17:16:18Z, size = 1572165, hashes = { sha256 = "b86ada508af81d19edeb213c681b1d48246c1a91d304c6c81a427674c17eb91c" } } +sdist = { url = "https://files.pythonhosted.org/packages/e4/47/b9efed96c114afcfa3c9d3fe98a76a1d14c74a9e266d397cf6eb64be5e01/pytest-9.1.1.tar.gz", upload-time = 2026-06-19T10:58:32Z, size = 1636369, hashes = { sha256 = "1088fbde8f2b49d95a549a195707afa7a76a3ce9bcadc26b6d71f0ffda5fe313" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/d4/24/a372aaf5c9b7208e7112038812994107bc65a84cd00e0354a88c2c77a617/pytest-9.0.3-py3-none-any.whl", upload-time = 2026-04-07T17:16:16Z, size = 375249, hashes = { sha256 = "2c5efc453d45394fdd706ade797c0a81091eccd1d6e4bccfcd476e2b8e0ab5d9" } }, + { url = "https://files.pythonhosted.org/packages/24/25/1de2678b631f5a49215c6c96fff41ba892b0a34df68d6d80292b1b48aa7f/pytest-9.1.1-py3-none-any.whl", upload-time = 2026-06-19T10:58:31Z, size = 386536, hashes = { sha256 = "37a86b45efb9a47a61a36449063e8e18d0cab3161329fc099eb21783169c4f0c" } }, ] [[packages]] name = "pytest-asyncio" -version = "1.3.0" +version = "1.4.0" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/90/2c/8af215c0f776415f3590cac4f9086ccefd6fd463befeae41cd4d3f193e5a/pytest_asyncio-1.3.0.tar.gz", upload-time = 2025-11-10T16:07:47Z, size = 50087, hashes = { sha256 = "d7f52f36d231b80ee124cd216ffb19369aa168fc10095013c6b014a34d3ee9e5" } } +sdist = { url = "https://files.pythonhosted.org/packages/43/7c/d36d04db312ecf4298932ef77e6e4a9e8ad017906e24e34f0b0c361a2473/pytest_asyncio-1.4.0.tar.gz", upload-time = 2026-05-26T09:56:04Z, size = 58514, hashes = { sha256 = "c6c0d2259945122819f171a32ecea2c349ead889ee28176caaf492143424be42" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/e5/35/f8b19922b6a25bc0880171a2f1a003eaeb93657475193ab516fd87cac9da/pytest_asyncio-1.3.0-py3-none-any.whl", upload-time = 2025-11-10T16:07:45Z, size = 15075, hashes = { sha256 = "611e26147c7f77640e6d0a92a38ed17c3e9848063698d5c93d5aa7aa11cebff5" } }, + { url = "https://files.pythonhosted.org/packages/03/e2/08a497ef684b88559c9cc5f4ad53a37e7b99e727094a86d6ea32536d5d3c/pytest_asyncio-1.4.0-py3-none-any.whl", upload-time = 2026-05-26T09:56:02Z, size = 16930, hashes = { sha256 = "933ca923a23075a87fb7070c0ec272a6848489824d887c85c812670932835aa1" } }, ] [[packages]] @@ -380,27 +380,27 @@ wheels = [ [[packages]] name = "ruff" -version = "0.15.13" +version = "0.16.4" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/24/21/a7d5c126d5b557715ef81098f3db2fe20f622a039ff2e626af28d674ab80/ruff-0.15.13.tar.gz", upload-time = 2026-05-14T13:44:37Z, size = 4678180, hashes = { sha256 = "f9d89f17f7ba7fb2ed42921f0df75da797a9a5d71bc39049e2c687cf2baf44b7" } } +sdist = { url = "https://files.pythonhosted.org/packages/00/8f/d8074b1f25e003164087a8bfe79a0f1a3945135764dbb6aaab04103dcaf9/ruff-0.16.4.tar.gz", upload-time = 2026-08-20T17:43:59Z, size = 4899731, hashes = { sha256 = "13171aa9d9af2240ee3504e639de73122c67e74036de5ba2e1d01422cd17e3dc" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/c6/61/11d458dc6ac22504fd8e237b29dfd40504c7fbbcc8930402cfe51a8e63ed/ruff-0.15.13-py3-none-linux_armv6l.whl", upload-time = 2026-05-14T13:44:18Z, size = 10738279, hashes = { sha256 = "444b580fc72fd6887e650acd3e575e18cdc79dbcf42fb4030b491057921f61f8" } }, - { url = "https://files.pythonhosted.org/packages/86/ca/caa871ee7be718c45256fada4e16a218ee3e33f0c4a46b729a60a24912e6/ruff-0.15.13-py3-none-macosx_10_12_x86_64.whl", upload-time = 2026-05-14T13:44:06Z, size = 11124798, hashes = { sha256 = "6590d009e7cb7ebf36f83dbdd44a3fa48a0994ff6f1cdc1b08006abe58f98dc7" } }, - { url = "https://files.pythonhosted.org/packages/d3/19/43f5f2e568dddde567fc41f8471f9432c09563e19d3e617a48cfa52f8f0a/ruff-0.15.13-py3-none-macosx_11_0_arm64.whl", upload-time = 2026-05-14T13:44:04Z, size = 10460761, hashes = { sha256 = "1c26d2f66163deeb6e08d8b39fbbe983ce3c71cea06a6d7591cfd1421793c629" } }, - { url = "https://files.pythonhosted.org/packages/99/df/cf938cd6de3003178f03ad7c1ea2a6c099468c03a35037985070b37e76be/ruff-0.15.13-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", upload-time = 2026-05-14T13:44:25Z, size = 10804451, hashes = { sha256 = "9dbd6f94b434f896308e4d57fb7bfde0d02b99f7a64b3bdab0fdfa6a864203a5" } }, - { url = "https://files.pythonhosted.org/packages/c7/7d/5d0973129b154ded2225729169d7068f26b467760b146493fde138415f23/ruff-0.15.13-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", upload-time = 2026-05-14T13:44:08Z, size = 10534285, hashes = { sha256 = "bf3259f3be4d181bda591da5db2571aed6853c6a048157756448020bc6c5cd22" } }, - { url = "https://files.pythonhosted.org/packages/1f/e3/6b999bbc66cd51e5f073842bc2a3995e99c5e0e72e16b15e7261f7abf57a/ruff-0.15.13-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", upload-time = 2026-05-14T13:44:11Z, size = 11312063, hashes = { sha256 = "ae9c17e5eb4430c154e76abc25d79a318190f5a997f38fb6b114416c5319ffc9" } }, - { url = "https://files.pythonhosted.org/packages/af/5a/642639e9f5db04f1e97fbd6e091c6fd20725bdf072fb114d00eefb9e6eb8/ruff-0.15.13-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", upload-time = 2026-05-14T13:44:01Z, size = 12183079, hashes = { sha256 = "2e2e39bff6c341f4b577a21b801326fab0b11847f48fcaa83f00a113c9b3cb55" } }, - { url = "https://files.pythonhosted.org/packages/19/4c/7585735f6b53b0f12de13618b2f7d250a844f018822efc899df2e7b8295f/ruff-0.15.13-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", upload-time = 2026-05-14T13:43:59Z, size = 11440833, hashes = { sha256 = "e8d9a8e08013542e94d3220bc5b62cc3e5ef87c5f74bff367d3fac14fab013e6" } }, - { url = "https://files.pythonhosted.org/packages/e8/31/bf1a0803d077e679cfeee5f2f67290a0fa79c7385b5d9a8c17b9db2c48f0/ruff-0.15.13-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", upload-time = 2026-05-14T13:44:27Z, size = 11434486, hashes = { sha256 = "cc411dfebe5eebe55ce041c6ae080eb7668955e866daa2fbb16692a784f1c4ca" } }, - { url = "https://files.pythonhosted.org/packages/e1/4e/62c9b999875d4f14db80f277c030578f5e249c9852d65b7ac7ad0b43c041/ruff-0.15.13-py3-none-manylinux_2_31_riscv64.whl", upload-time = 2026-05-14T13:44:13Z, size = 11385189, hashes = { sha256 = "768494eb08b9cee54e2fd27969966f74db5a57f6eaa7a90fcb3306af34dfc4bd" } }, - { url = "https://files.pythonhosted.org/packages/fc/89/7e959047a104df3eb12863447c110140191fc5b6c4f379ea2e803fcdb0e4/ruff-0.15.13-py3-none-musllinux_1_2_aarch64.whl", upload-time = 2026-05-14T13:43:56Z, size = 10781380, hashes = { sha256 = "fb75f9a3a7e42ffe117d734494e6c5e5cb3565d66e12612cb63d0e572a41a5b6" } }, - { url = "https://files.pythonhosted.org/packages/ff/52/5fd18f3b88cab63e88aa11516b3b4e1e5f720e5c330f8dbe5c26210f41f8/ruff-0.15.13-py3-none-musllinux_1_2_armv7l.whl", upload-time = 2026-05-14T13:44:20Z, size = 10540605, hashes = { sha256 = "8cb74dd33bb2f6613faf7fc03b660053b5ac4f80e706d5788c6335e2a8048d51" } }, - { url = "https://files.pythonhosted.org/packages/e8/e0/9e35f338990d3e41a82875ff7053ffe97541dae81c9d02143177f381d572/ruff-0.15.13-py3-none-musllinux_1_2_i686.whl", upload-time = 2026-05-14T13:44:16Z, size = 11036554, hashes = { sha256 = "7ef823f817fcd191dc934e984be9cf4094f808effa16f2542ad8e821ba02bbf2" } }, - { url = "https://files.pythonhosted.org/packages/c2/13/070fb048c24080fba188f66371e2a92785be257ad02242066dc7255ac6e9/ruff-0.15.13-py3-none-musllinux_1_2_x86_64.whl", upload-time = 2026-05-14T13:44:22Z, size = 11528133, hashes = { sha256 = "f345a13937bd7f09f6f5d19fa0721b0c103e00e7f62bc67089a8e5e037719e0b" } }, - { url = "https://files.pythonhosted.org/packages/6b/8c/b1e1666aef7fc6555094d73ae6cd981701781ae85b97ceefc0eebd0b4668/ruff-0.15.13-py3-none-win32.whl", upload-time = 2026-05-14T13:44:35Z, size = 10721455, hashes = { sha256 = "4044f94208b3b05ba0fc4a4abd0558cf4d6459bd18325eead7fd8cc66f909b41" } }, - { url = "https://files.pythonhosted.org/packages/ab/a6/870a3e8a50590bb92be184ad928c2922f088b00d9dc5c5ec7b924ee08c22/ruff-0.15.13-py3-none-win_amd64.whl", upload-time = 2026-05-14T13:44:30Z, size = 11900409, hashes = { sha256 = "7064884d442b7d477b4e7473d12da7f08851d2b1982763c5d3f388a19468a1a4" } }, - { url = "https://files.pythonhosted.org/packages/9b/36/9c015cd052fca743dae8cb2aeb16b551444787467db42ceab0fc968865af/ruff-0.15.13-py3-none-win_arm64.whl", upload-time = 2026-05-14T13:44:33Z, size = 11179336, hashes = { sha256 = "2471da9bd1068c8c064b5fd9c0c4b6dddffd6369cb1cd68b29993b1709ff1b21" } }, + { url = "https://files.pythonhosted.org/packages/ff/80/779895ef584e089d22f2c6df0d0e99a65ec2df0805f1fffd439415b8c1f0/ruff-0.16.4-py3-none-linux_armv6l.whl", upload-time = 2026-08-20T17:43:16Z, size = 10006909, hashes = { sha256 = "df4075f71ddac40b9934af60c3ec8a53047dd5a5fdc43224e6e4e8e9a27cb6f7" } }, + { url = "https://files.pythonhosted.org/packages/a9/e6/f553199b5e8927a05cb5c422d921fd0656b29ab976e91c44802107c6b0da/ruff-0.16.4-py3-none-macosx_10_12_x86_64.whl", upload-time = 2026-08-20T17:43:19Z, size = 10240201, hashes = { sha256 = "0c95538517af68004306b0fb3214ff2f2af67a65092aee77cd9eb86db6656604" } }, + { url = "https://files.pythonhosted.org/packages/1c/70/4a6dc4bb34da4dee35e30f09bbd1bfbdd26f33b62fb9b8df31f08a199cd2/ruff-0.16.4-py3-none-macosx_11_0_arm64.whl", upload-time = 2026-08-20T17:43:21Z, size = 9835122, hashes = { sha256 = "963f83df8e69e575b64d67dd447ebbc917db41a14bf38d4593a4183e7aaa8255" } }, + { url = "https://files.pythonhosted.org/packages/24/12/c6e22d686372c15bcb7af99831f1a1be96df696491babf4f24e4f942c527/ruff-0.16.4-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", upload-time = 2026-08-20T17:43:24Z, size = 9977162, hashes = { sha256 = "32a5057c7ff3f6e6480a48fccfb3a412a690f48a3d03ac5cf08177d6c2da3ade" } }, + { url = "https://files.pythonhosted.org/packages/46/49/72b10ec912f5ab5854992eaf7aa7cd36729b6937d9dc4e0fb41b3bf428ec/ruff-0.16.4-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", upload-time = 2026-08-20T17:43:26Z, size = 9829789, hashes = { sha256 = "b3dce8d9b0c57c265b91885a66a567d8ea1372e8eb4e250fa8e5e3f579e99cff" } }, + { url = "https://files.pythonhosted.org/packages/fa/80/0f30e32e7f6ee26edc39075502db9d368d788a44a79b55f763eb4ab03796/ruff-0.16.4-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", upload-time = 2026-08-20T17:43:29Z, size = 10527949, hashes = { sha256 = "7dc651db49283c69f8e72c834eec4fe5573e4c646856aebece0ce385dceb2a80" } }, + { url = "https://files.pythonhosted.org/packages/52/3d/86e8ad3542169e56cac3859a343afdb9df2ad54d35a59ce1e67baee83421/ruff-0.16.4-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", upload-time = 2026-08-20T17:43:31Z, size = 11333695, hashes = { sha256 = "3817b87dbcabc92f13b05019257c5b89b5b4d51b5fb20f56fb5235ceb723cd07" } }, + { url = "https://files.pythonhosted.org/packages/d0/16/481c29b380c20a0054a8261066665e1b3488e23636c49d0a43e75975b9bb/ruff-0.16.4-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", upload-time = 2026-08-20T17:43:34Z, size = 10727741, hashes = { sha256 = "e9fce1499134b2c8c68e5166f95705a5812062bb93aacc5f9873bb1a27084bc7" } }, + { url = "https://files.pythonhosted.org/packages/5e/b6/56bc0b8cf45b54b28b3a5e6381c8945d51b5b18adf659454c32295209a31/ruff-0.16.4-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", upload-time = 2026-08-20T17:43:37Z, size = 10286522, hashes = { sha256 = "f2d812e482f5a7e02eee26cd73d2a37ebbdf47d795ea63ba1b89110ae93e9fb3" } }, + { url = "https://files.pythonhosted.org/packages/e8/8b/b345b4fb110f2fbe2bd31eabd271e5e8b3b7e4ee6c0e02f2dc6be78db000/ruff-0.16.4-py3-none-manylinux_2_31_riscv64.whl", upload-time = 2026-08-20T17:43:39Z, size = 10584182, hashes = { sha256 = "6baaf984aa7976edf93d3b627fe2d1d22ee94bbca05fa6f90fc76d73924e3454" } }, + { url = "https://files.pythonhosted.org/packages/29/e5/827b34041c35f58774a9681a4213994c164fc987800f4dddabcf451da0bf/ruff-0.16.4-py3-none-musllinux_1_2_aarch64.whl", upload-time = 2026-08-20T17:43:42Z, size = 10134195, hashes = { sha256 = "bdfcf0b28662eb890372d50f92c283bb94e67e7635ed93c7fd533970acff7b2b" } }, + { url = "https://files.pythonhosted.org/packages/0f/10/d0bffcdd6729b87afc82ba0ef377173356a7dc8e972f5179968cf2fdf98c/ruff-0.16.4-py3-none-musllinux_1_2_armv7l.whl", upload-time = 2026-08-20T17:43:44Z, size = 9825821, hashes = { sha256 = "b66b02cb9b04f537643cadf5768e5f98dc461890d530cb67113d71c8c76e605d" } }, + { url = "https://files.pythonhosted.org/packages/f5/32/0db2a863b796ca62d83e92a07a3ccf00921b14db02059347576a2fda3d4b/ruff-0.16.4-py3-none-musllinux_1_2_i686.whl", upload-time = 2026-08-20T17:43:46Z, size = 10267658, hashes = { sha256 = "8528bf9a4b291a60bf02ea453511e8ce6215bd2b982ee80405b66b008b6c30a0" } }, + { url = "https://files.pythonhosted.org/packages/b2/a0/fbdeb59e48c6261f523e56c8f12e9c08fbe693786595cc7e3959207a9232/ruff-0.16.4-py3-none-musllinux_1_2_x86_64.whl", upload-time = 2026-08-20T17:43:49Z, size = 10697071, hashes = { sha256 = "fbd85d2875fdd67e833213a651f613bbf25303abf6aa822a5121f4531195678d" } }, + { url = "https://files.pythonhosted.org/packages/aa/28/0c6dd865859c6d17bc8ccc34cb72b0e02d6c7eb25e8a1e22b5bea681e2c0/ruff-0.16.4-py3-none-win32.whl", upload-time = 2026-08-20T17:43:52Z, size = 10021687, hashes = { sha256 = "312769988007aaeb8e189b443ccdd03c0e6374489e053467be6d96518ebff76e" } }, + { url = "https://files.pythonhosted.org/packages/a3/03/e724450f621698117f9aa6dd241c94d0274ae96781378dc86745ae29f0e7/ruff-0.16.4-py3-none-win_amd64.whl", upload-time = 2026-08-20T17:43:54Z, size = 10567657, hashes = { sha256 = "05d9d27a18c4bcbefada602480ec9e01e0bc949d432e0ced5df77edac195919c" } }, + { url = "https://files.pythonhosted.org/packages/0e/fe/da8b9e1347696bb22120b77280ec5ce25d500ca5cb39d5ad6e5c18de19c1/ruff-0.16.4-py3-none-win_arm64.whl", upload-time = 2026-08-20T17:43:57Z, size = 10451579, hashes = { sha256 = "a3a61621c9b6f6a89573e938a080e648f1695baa3f58570a3a707bc51ff65a21" } }, ] [[packages]] @@ -469,9 +469,9 @@ wheels = [ [[packages]] name = "typing-extensions" -version = "4.15.0" +version = "4.16.0" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/72/94/1a15dd82efb362ac84269196e94cf00f187f7ed21c242792a923cdb1c61f/typing_extensions-4.15.0.tar.gz", upload-time = 2025-08-25T13:49:26Z, size = 109391, hashes = { sha256 = "0cea48d173cc12fa28ecabc3b837ea3cf6f38c6d1136f85cbaaf598984861466" } } +sdist = { url = "https://files.pythonhosted.org/packages/f6/cc/6253133b5bb138fc3306cebfbda2c520f545d36b5be2c7255cc528bb45d6/typing_extensions-4.16.0.tar.gz", upload-time = 2026-07-02T08:40:05Z, size = 113555, hashes = { sha256 = "dc983d19a509c94dba722ee6abd33940f7c05a89e243c47e907eb4db6f1a43e5" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/18/67/36e9267722cc04a6b9f15c7f3441c2363321a3ea07da7ae0c0707beb2a9c/typing_extensions-4.15.0-py3-none-any.whl", upload-time = 2025-08-25T13:49:24Z, size = 44614, hashes = { sha256 = "f0fa19c6845758ab08074a0cfa8b7aecb71c999ca73d62883bc25cc018c4e548" } }, + { url = "https://files.pythonhosted.org/packages/49/d3/b8441a820a491ddfc024b0b0cf0393375b75ea13866d9c66727e54c2fc80/typing_extensions-4.16.0-py3-none-any.whl", upload-time = 2026-07-02T08:40:04Z, size = 45571, hashes = { sha256 = "481caa481374e813c1b176ada14e97f1f67a4539ce9cfeb3f350d78d6370c2e8" } }, ] diff --git a/bdd/python/uv.lock b/bdd/python/uv.lock index b10e81b848..fad1a3fac5 100644 --- a/bdd/python/uv.lock +++ b/bdd/python/uv.lock @@ -25,8 +25,6 @@ requires-dist = [ { name = "pytest-cov", marker = "extra == 'testing'", specifier = ">=4.0,<8.0" }, { name = "pytest-timeout", marker = "extra == 'all'", specifier = ">=2.0,<3.0" }, { name = "pytest-timeout", marker = "extra == 'testing'", specifier = ">=2.0,<3.0" }, - { name = "pytest-xdist", marker = "extra == 'all'", specifier = ">=3.0,<4.0" }, - { name = "pytest-xdist", marker = "extra == 'testing'", specifier = ">=3.0,<4.0" }, { name = "ruff", marker = "extra == 'all'", specifier = ">=0.1.0,<1.0" }, { name = "ruff", marker = "extra == 'dev'", specifier = ">=0.1.0,<1.0" }, { name = "testcontainers", marker = "extra == 'all'", specifier = ">=4.15.0,<5.0" }, diff --git a/examples/python/pylock.toml b/examples/python/pylock.toml index 082a913f96..f653d94a95 100644 --- a/examples/python/pylock.toml +++ b/examples/python/pylock.toml @@ -1,22 +1,5 @@ -# Licensed to the Apache Software Foundation (ASF) under one -# or more contributor license agreements. See the NOTICE file -# distributed with this work for additional information -# regarding copyright ownership. The ASF licenses this file -# to you under the Apache License, Version 2.0 (the -# "License"); you may not use this file except in compliance -# with the License. You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, -# software distributed under the License is distributed on an -# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY -# KIND, either express or implied. See the License for the -# specific language governing permissions and limitations -# under the License. - # This file was autogenerated by uv via the following command: -# uv export -o pylock.toml +# uv export --locked --format pylock.toml --output-file pylock.toml --all-extras lock-version = "1.0" created-by = "uv" requires-python = ">=3.10" @@ -55,44 +38,46 @@ wheels = [ [[packages]] name = "pyrefly" -version = "1.0.0" +version = "1.2.0" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/9f/3a/9045b0097ac58979c7c30a4fa0e673db942d4adbc7b6d439bd54ae58c441/pyrefly-1.0.0.tar.gz", upload-time = 2026-05-12T20:12:46Z, size = 5677995, hashes = { sha256 = "5c2b810ffcebd84be71de5df1223651edee951653a66935c6f091e957c452455" } } +sdist = { url = "https://files.pythonhosted.org/packages/89/01/a86e9f24722b095c3f88e3616132b75a21b0df53804bdc6a45314dd4d93c/pyrefly-1.2.0.tar.gz", upload-time = 2026-08-01T02:56:27Z, size = 6243654, hashes = { sha256 = "5485f960fc2481617068c918335c39ab1507ef90b6b5bd35bf57726e60e73185" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/f4/c6/90788819bac9c61dd7bacba53b79f3c12d47ccbe5e51b3d6d89f2387e1d2/pyrefly-1.0.0-py3-none-macosx_10_12_x86_64.whl", upload-time = 2026-05-12T20:12:20Z, size = 13122950, hashes = { sha256 = "e355a0908555348ed4b9585ef25c76ff566673e345c866c325f1633f44d890b6" } }, - { url = "https://files.pythonhosted.org/packages/82/91/a3cf2a1e87d336eaa804a1e6fc93266faf6dc2a97eecdbc7eae289628022/pyrefly-1.0.0-py3-none-macosx_11_0_arm64.whl", upload-time = 2026-05-12T20:12:23Z, size = 12599494, hashes = { sha256 = "a7038efc3a40f8294edee339895633cf22db268c0d434cdbcbefc34f78a9ecc3" } }, - { url = "https://files.pythonhosted.org/packages/cd/ab/74d1e11e737e99b1c003ecc5d7d2e846c4ea1f328966bfdbbd0ac63fad0a/pyrefly-1.0.0-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", upload-time = 2026-05-12T20:12:25Z, size = 12995507, hashes = { sha256 = "da331ca515ed1c08791da2b5f664cf9c1294c48fd802133262e7d5d51e0f4416" } }, - { url = "https://files.pythonhosted.org/packages/7c/ac/2df0899f8464c97e5d995f994c97c5cb5b0f58610432aa90d26d924e1db5/pyrefly-1.0.0-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", upload-time = 2026-05-12T20:12:29Z, size = 13947693, hashes = { sha256 = "c74219d8f3e63cdaa5501a0b21d1c9d37011820f9606728d0ed06f09ae86a878" } }, - { url = "https://files.pythonhosted.org/packages/6b/3e/b247c24321e36f04b7d51f9ccf3df93e5009e4b29939524b36ec2e17dc2a/pyrefly-1.0.0-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", upload-time = 2026-05-12T20:12:31Z, size = 13925803, hashes = { sha256 = "c0d05543b1bb6ee6d64149eb5d6b2fb15aa72d3962d6a97abca0afaca8b0c131" } }, - { url = "https://files.pythonhosted.org/packages/61/16/cfa2d61a4aa1e1f7bca48bb37acd01c6a09db4864b16a54f9587092765ff/pyrefly-1.0.0-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", upload-time = 2026-05-12T20:12:35Z, size = 13470398, hashes = { sha256 = "1382d5b1fcdb49a4de9f34d112d2bddf290a78ff93ee8149492ad5f1077ddffc" } }, - { url = "https://files.pythonhosted.org/packages/cb/2b/6372c7dddb326223e24a46b17efd0d4bd7b4fe22c821e523157577eed2d2/pyrefly-1.0.0-py3-none-win32.whl", upload-time = 2026-05-12T20:12:38Z, size = 12222643, hashes = { sha256 = "aa8b5d0e47080e3202a2547b39f7a5a61d2c781c712b3b67884f745ca2c759d2" } }, - { url = "https://files.pythonhosted.org/packages/be/ad/1d23be700b6b2ddaeb362360c7145917a8edbbf7240ae428d40541772fce/pyrefly-1.0.0-py3-none-win_amd64.whl", upload-time = 2026-05-12T20:12:41Z, size = 13146369, hashes = { sha256 = "c8abcb0f2082e83c890375128f9cff4aa4d3f210b85eea7b3046c1ae764e77f5" } }, - { url = "https://files.pythonhosted.org/packages/8c/38/16589134f3012fd097a10dcc85771555f1a5fb76e04b682597180743af30/pyrefly-1.0.0-py3-none-win_arm64.whl", upload-time = 2026-05-12T20:12:43Z, size = 12538326, hashes = { sha256 = "d150fa9e40e8392832be81c3bcfc0497c146674ce4d0f8e04e1ec29e775ffb8c" } }, + { url = "https://files.pythonhosted.org/packages/7d/9d/3c0ef1d4843987b22f996ed381ec9cf5a3b1273e29804db276252e4c95eb/pyrefly-1.2.0-py3-none-macosx_10_12_x86_64.whl", upload-time = 2026-08-01T02:56:02Z, size = 14026305, hashes = { sha256 = "7f46d983ac49ddd2b043694960a01dc6a19a5cfd8eec609d6bd9c42866f91b4e" } }, + { url = "https://files.pythonhosted.org/packages/0a/06/03bbb78fbea54cdc65b626619f3597d5611aca4fdef11e72a4e8360e7e63/pyrefly-1.2.0-py3-none-macosx_11_0_arm64.whl", upload-time = 2026-08-01T02:56:04Z, size = 13463880, hashes = { sha256 = "756f669b5555090f5c1a4fef30db1785fabe657764f7e4e6dc88994dfb8ca82d" } }, + { url = "https://files.pythonhosted.org/packages/13/5a/7d8bc00a38e93bbc9c3e7bd14d305f7948717e667c9bcddeab9dd42fd255/pyrefly-1.2.0-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", upload-time = 2026-08-01T02:56:07Z, size = 13907329, hashes = { sha256 = "e3465812ce5ef4781fb592edbf2724547296f0a3124be115d73c7e8b2401862d" } }, + { url = "https://files.pythonhosted.org/packages/be/94/9e08b4bf799d0b8f36b55a2783c7ba5f51730cf0632a85a67b5b5ed876cd/pyrefly-1.2.0-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", upload-time = 2026-08-01T02:56:09Z, size = 15039020, hashes = { sha256 = "5de7b2ad2bba5c8055181681a84b74143eac2234a48ba5d1b7ed7e7a722b02bd" } }, + { url = "https://files.pythonhosted.org/packages/5b/bd/bca5fd0c80f4daf8ee6903a29df9f3de1feb05ff0946b8f35ec8c5096b13/pyrefly-1.2.0-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", upload-time = 2026-08-01T02:56:11Z, size = 14986199, hashes = { sha256 = "25822ea9505f589ea8a725e4268b475132fb89e038fbf092e446510443ac142a" } }, + { url = "https://files.pythonhosted.org/packages/97/f7/f07087f3d185ad2eced0c56cef89ca5474dfb4ff25f146cd50a861c97553/pyrefly-1.2.0-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", upload-time = 2026-08-01T02:56:14Z, size = 14393715, hashes = { sha256 = "90efe75e17491ef5d636e10469e9278d7d0256b3b4c5e1f4750069bf3ae0f5d1" } }, + { url = "https://files.pythonhosted.org/packages/d3/70/0d142c320e284b9e3ce35e9b1e58b8ce2ee1f578f2a7234bc30e5022b94f/pyrefly-1.2.0-py3-none-musllinux_1_2_aarch64.whl", upload-time = 2026-08-01T02:56:16Z, size = 13933008, hashes = { sha256 = "368aaf7eee4f511ddc0f8e564cf14e01ab2f10b0db9105c6d5b153bf498d07bf" } }, + { url = "https://files.pythonhosted.org/packages/5d/e8/e84f11b6e1f63fd453ad3654213b9a0f6f4de8cef6b58038eef2d0d5955d/pyrefly-1.2.0-py3-none-musllinux_1_2_x86_64.whl", upload-time = 2026-08-01T02:56:18Z, size = 14431827, hashes = { sha256 = "d52d5da7bc65fb7675fbaa80eda879d4f8787c494f04cac21603330d3abbdbbe" } }, + { url = "https://files.pythonhosted.org/packages/0f/06/810d31380f66c75e1c0779a408d3b16117b1b368b57894f6aa66bef21686/pyrefly-1.2.0-py3-none-win32.whl", upload-time = 2026-08-01T02:56:20Z, size = 13229447, hashes = { sha256 = "8c90751de8506d938e8f802659c74cf35bd7a0036510ee6c634a38eebb280bfa" } }, + { url = "https://files.pythonhosted.org/packages/ed/98/4dafa3c7a1caed2dc8cc708dde09ba27963c7736508f55b626fff3024113/pyrefly-1.2.0-py3-none-win_amd64.whl", upload-time = 2026-08-01T02:56:23Z, size = 14087387, hashes = { sha256 = "8a8964c224ccc4882730130955815de21ff443c1ac3f0b90685b19bf63848170" } }, + { url = "https://files.pythonhosted.org/packages/1b/1c/df3cb0a2e5591660ded7a1836cd2f29dc48c91adb1c0a3a700a96f6d09e1/pyrefly-1.2.0-py3-none-win_arm64.whl", upload-time = 2026-08-01T02:56:25Z, size = 13430873, hashes = { sha256 = "3a90bb8df39dfbac74b1f3b2e9d7c526b8f80568884c3944d955023a73ebf61e" } }, ] [[packages]] name = "ruff" -version = "0.15.13" +version = "0.16.4" index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/24/21/a7d5c126d5b557715ef81098f3db2fe20f622a039ff2e626af28d674ab80/ruff-0.15.13.tar.gz", upload-time = 2026-05-14T13:44:37Z, size = 4678180, hashes = { sha256 = "f9d89f17f7ba7fb2ed42921f0df75da797a9a5d71bc39049e2c687cf2baf44b7" } } +sdist = { url = "https://files.pythonhosted.org/packages/00/8f/d8074b1f25e003164087a8bfe79a0f1a3945135764dbb6aaab04103dcaf9/ruff-0.16.4.tar.gz", upload-time = 2026-08-20T17:43:59Z, size = 4899731, hashes = { sha256 = "13171aa9d9af2240ee3504e639de73122c67e74036de5ba2e1d01422cd17e3dc" } } wheels = [ - { url = "https://files.pythonhosted.org/packages/c6/61/11d458dc6ac22504fd8e237b29dfd40504c7fbbcc8930402cfe51a8e63ed/ruff-0.15.13-py3-none-linux_armv6l.whl", upload-time = 2026-05-14T13:44:18Z, size = 10738279, hashes = { sha256 = "444b580fc72fd6887e650acd3e575e18cdc79dbcf42fb4030b491057921f61f8" } }, - { url = "https://files.pythonhosted.org/packages/86/ca/caa871ee7be718c45256fada4e16a218ee3e33f0c4a46b729a60a24912e6/ruff-0.15.13-py3-none-macosx_10_12_x86_64.whl", upload-time = 2026-05-14T13:44:06Z, size = 11124798, hashes = { sha256 = "6590d009e7cb7ebf36f83dbdd44a3fa48a0994ff6f1cdc1b08006abe58f98dc7" } }, - { url = "https://files.pythonhosted.org/packages/d3/19/43f5f2e568dddde567fc41f8471f9432c09563e19d3e617a48cfa52f8f0a/ruff-0.15.13-py3-none-macosx_11_0_arm64.whl", upload-time = 2026-05-14T13:44:04Z, size = 10460761, hashes = { sha256 = "1c26d2f66163deeb6e08d8b39fbbe983ce3c71cea06a6d7591cfd1421793c629" } }, - { url = "https://files.pythonhosted.org/packages/99/df/cf938cd6de3003178f03ad7c1ea2a6c099468c03a35037985070b37e76be/ruff-0.15.13-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", upload-time = 2026-05-14T13:44:25Z, size = 10804451, hashes = { sha256 = "9dbd6f94b434f896308e4d57fb7bfde0d02b99f7a64b3bdab0fdfa6a864203a5" } }, - { url = "https://files.pythonhosted.org/packages/c7/7d/5d0973129b154ded2225729169d7068f26b467760b146493fde138415f23/ruff-0.15.13-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", upload-time = 2026-05-14T13:44:08Z, size = 10534285, hashes = { sha256 = "bf3259f3be4d181bda591da5db2571aed6853c6a048157756448020bc6c5cd22" } }, - { url = "https://files.pythonhosted.org/packages/1f/e3/6b999bbc66cd51e5f073842bc2a3995e99c5e0e72e16b15e7261f7abf57a/ruff-0.15.13-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", upload-time = 2026-05-14T13:44:11Z, size = 11312063, hashes = { sha256 = "ae9c17e5eb4430c154e76abc25d79a318190f5a997f38fb6b114416c5319ffc9" } }, - { url = "https://files.pythonhosted.org/packages/af/5a/642639e9f5db04f1e97fbd6e091c6fd20725bdf072fb114d00eefb9e6eb8/ruff-0.15.13-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", upload-time = 2026-05-14T13:44:01Z, size = 12183079, hashes = { sha256 = "2e2e39bff6c341f4b577a21b801326fab0b11847f48fcaa83f00a113c9b3cb55" } }, - { url = "https://files.pythonhosted.org/packages/19/4c/7585735f6b53b0f12de13618b2f7d250a844f018822efc899df2e7b8295f/ruff-0.15.13-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", upload-time = 2026-05-14T13:43:59Z, size = 11440833, hashes = { sha256 = "e8d9a8e08013542e94d3220bc5b62cc3e5ef87c5f74bff367d3fac14fab013e6" } }, - { url = "https://files.pythonhosted.org/packages/e8/31/bf1a0803d077e679cfeee5f2f67290a0fa79c7385b5d9a8c17b9db2c48f0/ruff-0.15.13-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", upload-time = 2026-05-14T13:44:27Z, size = 11434486, hashes = { sha256 = "cc411dfebe5eebe55ce041c6ae080eb7668955e866daa2fbb16692a784f1c4ca" } }, - { url = "https://files.pythonhosted.org/packages/e1/4e/62c9b999875d4f14db80f277c030578f5e249c9852d65b7ac7ad0b43c041/ruff-0.15.13-py3-none-manylinux_2_31_riscv64.whl", upload-time = 2026-05-14T13:44:13Z, size = 11385189, hashes = { sha256 = "768494eb08b9cee54e2fd27969966f74db5a57f6eaa7a90fcb3306af34dfc4bd" } }, - { url = "https://files.pythonhosted.org/packages/fc/89/7e959047a104df3eb12863447c110140191fc5b6c4f379ea2e803fcdb0e4/ruff-0.15.13-py3-none-musllinux_1_2_aarch64.whl", upload-time = 2026-05-14T13:43:56Z, size = 10781380, hashes = { sha256 = "fb75f9a3a7e42ffe117d734494e6c5e5cb3565d66e12612cb63d0e572a41a5b6" } }, - { url = "https://files.pythonhosted.org/packages/ff/52/5fd18f3b88cab63e88aa11516b3b4e1e5f720e5c330f8dbe5c26210f41f8/ruff-0.15.13-py3-none-musllinux_1_2_armv7l.whl", upload-time = 2026-05-14T13:44:20Z, size = 10540605, hashes = { sha256 = "8cb74dd33bb2f6613faf7fc03b660053b5ac4f80e706d5788c6335e2a8048d51" } }, - { url = "https://files.pythonhosted.org/packages/e8/e0/9e35f338990d3e41a82875ff7053ffe97541dae81c9d02143177f381d572/ruff-0.15.13-py3-none-musllinux_1_2_i686.whl", upload-time = 2026-05-14T13:44:16Z, size = 11036554, hashes = { sha256 = "7ef823f817fcd191dc934e984be9cf4094f808effa16f2542ad8e821ba02bbf2" } }, - { url = "https://files.pythonhosted.org/packages/c2/13/070fb048c24080fba188f66371e2a92785be257ad02242066dc7255ac6e9/ruff-0.15.13-py3-none-musllinux_1_2_x86_64.whl", upload-time = 2026-05-14T13:44:22Z, size = 11528133, hashes = { sha256 = "f345a13937bd7f09f6f5d19fa0721b0c103e00e7f62bc67089a8e5e037719e0b" } }, - { url = "https://files.pythonhosted.org/packages/6b/8c/b1e1666aef7fc6555094d73ae6cd981701781ae85b97ceefc0eebd0b4668/ruff-0.15.13-py3-none-win32.whl", upload-time = 2026-05-14T13:44:35Z, size = 10721455, hashes = { sha256 = "4044f94208b3b05ba0fc4a4abd0558cf4d6459bd18325eead7fd8cc66f909b41" } }, - { url = "https://files.pythonhosted.org/packages/ab/a6/870a3e8a50590bb92be184ad928c2922f088b00d9dc5c5ec7b924ee08c22/ruff-0.15.13-py3-none-win_amd64.whl", upload-time = 2026-05-14T13:44:30Z, size = 11900409, hashes = { sha256 = "7064884d442b7d477b4e7473d12da7f08851d2b1982763c5d3f388a19468a1a4" } }, - { url = "https://files.pythonhosted.org/packages/9b/36/9c015cd052fca743dae8cb2aeb16b551444787467db42ceab0fc968865af/ruff-0.15.13-py3-none-win_arm64.whl", upload-time = 2026-05-14T13:44:33Z, size = 11179336, hashes = { sha256 = "2471da9bd1068c8c064b5fd9c0c4b6dddffd6369cb1cd68b29993b1709ff1b21" } }, + { url = "https://files.pythonhosted.org/packages/ff/80/779895ef584e089d22f2c6df0d0e99a65ec2df0805f1fffd439415b8c1f0/ruff-0.16.4-py3-none-linux_armv6l.whl", upload-time = 2026-08-20T17:43:16Z, size = 10006909, hashes = { sha256 = "df4075f71ddac40b9934af60c3ec8a53047dd5a5fdc43224e6e4e8e9a27cb6f7" } }, + { url = "https://files.pythonhosted.org/packages/a9/e6/f553199b5e8927a05cb5c422d921fd0656b29ab976e91c44802107c6b0da/ruff-0.16.4-py3-none-macosx_10_12_x86_64.whl", upload-time = 2026-08-20T17:43:19Z, size = 10240201, hashes = { sha256 = "0c95538517af68004306b0fb3214ff2f2af67a65092aee77cd9eb86db6656604" } }, + { url = "https://files.pythonhosted.org/packages/1c/70/4a6dc4bb34da4dee35e30f09bbd1bfbdd26f33b62fb9b8df31f08a199cd2/ruff-0.16.4-py3-none-macosx_11_0_arm64.whl", upload-time = 2026-08-20T17:43:21Z, size = 9835122, hashes = { sha256 = "963f83df8e69e575b64d67dd447ebbc917db41a14bf38d4593a4183e7aaa8255" } }, + { url = "https://files.pythonhosted.org/packages/24/12/c6e22d686372c15bcb7af99831f1a1be96df696491babf4f24e4f942c527/ruff-0.16.4-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", upload-time = 2026-08-20T17:43:24Z, size = 9977162, hashes = { sha256 = "32a5057c7ff3f6e6480a48fccfb3a412a690f48a3d03ac5cf08177d6c2da3ade" } }, + { url = "https://files.pythonhosted.org/packages/46/49/72b10ec912f5ab5854992eaf7aa7cd36729b6937d9dc4e0fb41b3bf428ec/ruff-0.16.4-py3-none-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", upload-time = 2026-08-20T17:43:26Z, size = 9829789, hashes = { sha256 = "b3dce8d9b0c57c265b91885a66a567d8ea1372e8eb4e250fa8e5e3f579e99cff" } }, + { url = "https://files.pythonhosted.org/packages/fa/80/0f30e32e7f6ee26edc39075502db9d368d788a44a79b55f763eb4ab03796/ruff-0.16.4-py3-none-manylinux_2_17_i686.manylinux2014_i686.whl", upload-time = 2026-08-20T17:43:29Z, size = 10527949, hashes = { sha256 = "7dc651db49283c69f8e72c834eec4fe5573e4c646856aebece0ce385dceb2a80" } }, + { url = "https://files.pythonhosted.org/packages/52/3d/86e8ad3542169e56cac3859a343afdb9df2ad54d35a59ce1e67baee83421/ruff-0.16.4-py3-none-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", upload-time = 2026-08-20T17:43:31Z, size = 11333695, hashes = { sha256 = "3817b87dbcabc92f13b05019257c5b89b5b4d51b5fb20f56fb5235ceb723cd07" } }, + { url = "https://files.pythonhosted.org/packages/d0/16/481c29b380c20a0054a8261066665e1b3488e23636c49d0a43e75975b9bb/ruff-0.16.4-py3-none-manylinux_2_17_s390x.manylinux2014_s390x.whl", upload-time = 2026-08-20T17:43:34Z, size = 10727741, hashes = { sha256 = "e9fce1499134b2c8c68e5166f95705a5812062bb93aacc5f9873bb1a27084bc7" } }, + { url = "https://files.pythonhosted.org/packages/5e/b6/56bc0b8cf45b54b28b3a5e6381c8945d51b5b18adf659454c32295209a31/ruff-0.16.4-py3-none-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", upload-time = 2026-08-20T17:43:37Z, size = 10286522, hashes = { sha256 = "f2d812e482f5a7e02eee26cd73d2a37ebbdf47d795ea63ba1b89110ae93e9fb3" } }, + { url = "https://files.pythonhosted.org/packages/e8/8b/b345b4fb110f2fbe2bd31eabd271e5e8b3b7e4ee6c0e02f2dc6be78db000/ruff-0.16.4-py3-none-manylinux_2_31_riscv64.whl", upload-time = 2026-08-20T17:43:39Z, size = 10584182, hashes = { sha256 = "6baaf984aa7976edf93d3b627fe2d1d22ee94bbca05fa6f90fc76d73924e3454" } }, + { url = "https://files.pythonhosted.org/packages/29/e5/827b34041c35f58774a9681a4213994c164fc987800f4dddabcf451da0bf/ruff-0.16.4-py3-none-musllinux_1_2_aarch64.whl", upload-time = 2026-08-20T17:43:42Z, size = 10134195, hashes = { sha256 = "bdfcf0b28662eb890372d50f92c283bb94e67e7635ed93c7fd533970acff7b2b" } }, + { url = "https://files.pythonhosted.org/packages/0f/10/d0bffcdd6729b87afc82ba0ef377173356a7dc8e972f5179968cf2fdf98c/ruff-0.16.4-py3-none-musllinux_1_2_armv7l.whl", upload-time = 2026-08-20T17:43:44Z, size = 9825821, hashes = { sha256 = "b66b02cb9b04f537643cadf5768e5f98dc461890d530cb67113d71c8c76e605d" } }, + { url = "https://files.pythonhosted.org/packages/f5/32/0db2a863b796ca62d83e92a07a3ccf00921b14db02059347576a2fda3d4b/ruff-0.16.4-py3-none-musllinux_1_2_i686.whl", upload-time = 2026-08-20T17:43:46Z, size = 10267658, hashes = { sha256 = "8528bf9a4b291a60bf02ea453511e8ce6215bd2b982ee80405b66b008b6c30a0" } }, + { url = "https://files.pythonhosted.org/packages/b2/a0/fbdeb59e48c6261f523e56c8f12e9c08fbe693786595cc7e3959207a9232/ruff-0.16.4-py3-none-musllinux_1_2_x86_64.whl", upload-time = 2026-08-20T17:43:49Z, size = 10697071, hashes = { sha256 = "fbd85d2875fdd67e833213a651f613bbf25303abf6aa822a5121f4531195678d" } }, + { url = "https://files.pythonhosted.org/packages/aa/28/0c6dd865859c6d17bc8ccc34cb72b0e02d6c7eb25e8a1e22b5bea681e2c0/ruff-0.16.4-py3-none-win32.whl", upload-time = 2026-08-20T17:43:52Z, size = 10021687, hashes = { sha256 = "312769988007aaeb8e189b443ccdd03c0e6374489e053467be6d96518ebff76e" } }, + { url = "https://files.pythonhosted.org/packages/a3/03/e724450f621698117f9aa6dd241c94d0274ae96781378dc86745ae29f0e7/ruff-0.16.4-py3-none-win_amd64.whl", upload-time = 2026-08-20T17:43:54Z, size = 10567657, hashes = { sha256 = "05d9d27a18c4bcbefada602480ec9e01e0bc949d432e0ced5df77edac195919c" } }, + { url = "https://files.pythonhosted.org/packages/0e/fe/da8b9e1347696bb22120b77280ec5ce25d500ca5cb39d5ad6e5c18de19c1/ruff-0.16.4-py3-none-win_arm64.whl", upload-time = 2026-08-20T17:43:57Z, size = 10451579, hashes = { sha256 = "a3a61621c9b6f6a89573e938a080e648f1695baa3f58570a3a707bc51ff65a21" } }, ] [[packages]] diff --git a/examples/python/uv.lock b/examples/python/uv.lock index 469565c7ed..f27d0c65b3 100644 --- a/examples/python/uv.lock +++ b/examples/python/uv.lock @@ -25,8 +25,6 @@ requires-dist = [ { name = "pytest-cov", marker = "extra == 'testing'", specifier = ">=4.0,<8.0" }, { name = "pytest-timeout", marker = "extra == 'all'", specifier = ">=2.0,<3.0" }, { name = "pytest-timeout", marker = "extra == 'testing'", specifier = ">=2.0,<3.0" }, - { name = "pytest-xdist", marker = "extra == 'all'", specifier = ">=3.0,<4.0" }, - { name = "pytest-xdist", marker = "extra == 'testing'", specifier = ">=3.0,<4.0" }, { name = "ruff", marker = "extra == 'all'", specifier = ">=0.1.0,<1.0" }, { name = "ruff", marker = "extra == 'dev'", specifier = ">=0.1.0,<1.0" }, { name = "testcontainers", marker = "extra == 'all'", specifier = ">=4.15.0,<5.0" }, diff --git a/foreign/python/apache_iggy.pyi b/foreign/python/apache_iggy.pyi index 814549ffd1..e543869e9c 100644 --- a/foreign/python/apache_iggy.pyi +++ b/foreign/python/apache_iggy.pyi @@ -54,6 +54,7 @@ __all__ = [ "SendMessagesConfirmation", "SendMessagesResponse", "Stats", + "Stream", "StreamDetails", "StreamPermissions", "TcpConfig", @@ -1129,6 +1130,93 @@ class IggyClient: Returns the stream details, or `None` if the stream does not exist. Raises `RuntimeError` on failure. """ + def get_streams(self) -> collections.abc.Awaitable[list[Stream]]: + r""" + Return all streams. + + Returns: + A list of `Stream` summaries. + + Raises: + RuntimeError: If the client is not authenticated, the user lacks global + `read_streams` or `manage_streams` permission, or the request fails. + """ + def update_stream( + self, + stream_id: builtins.str | builtins.int, + name: builtins.str, + options: builtins.dict[builtins.str, builtins.str] | None = None, + ) -> collections.abc.Awaitable[None]: + r""" + Rename a stream selected by name or numeric ID. + + `stream_id` accepts a stream name as `str` or numeric ID as `int`. A + decimal-only string is interpreted as a numeric ID. `name` must be unique + and contain between 1 and 255 UTF-8 bytes. Renaming a stream to its current + name succeeds without changing it. + + Args: + stream_id: Stream identifier as `str | int`. + name: New stream name as `str`. + options: Additional option keys as `dict[str, str] | None`, forwarded + to the server. Current server versions reject all stream update + option keys. + + Returns: + None. + + Raises: + TypeError: If `stream_id` is not `str` or an integer in + `0..=2**32 - 1`, or `name` is not `str`. + ValueError: If a string identifier is empty or exceeds 255 UTF-8 bytes. + RuntimeError: If the client is not authenticated, the user lacks global + `manage_streams` or per-stream `manage_stream` permission, the + stream does not exist, the new name is invalid or already used, or + the request fails. + """ + def delete_stream( + self, stream_id: builtins.str | builtins.int + ) -> collections.abc.Awaitable[None]: + r""" + Delete a stream selected by name or numeric ID. + + Deletion removes the stream and all of its topics, partitions, and messages. + `stream_id` accepts a stream name as `str` or numeric ID as `int`. A + decimal-only string is interpreted as a numeric ID. + + Returns: + None. + + Raises: + TypeError: If `stream_id` is not `str` or an integer in + `0..=2**32 - 1`. + ValueError: If a string identifier is empty or exceeds 255 UTF-8 bytes. + RuntimeError: If the client is not authenticated, the user lacks global + `manage_streams` or per-stream `manage_stream` permission, the + stream does not exist, or the request fails. + """ + def purge_stream( + self, stream_id: builtins.str | builtins.int + ) -> collections.abc.Awaitable[None]: + r""" + Delete all messages from every topic in a stream. + + The stream, topics, and partitions remain available. Repeated purges of an + existing empty stream succeed. `stream_id` accepts a stream name as `str` + or numeric ID as `int`. A decimal-only string is interpreted as a numeric + ID. + + Returns: + None. + + Raises: + TypeError: If `stream_id` is not `str` or an integer in + `0..=2**32 - 1`. + ValueError: If a string identifier is empty or exceeds 255 UTF-8 bytes. + RuntimeError: If the client is not authenticated, the user lacks global + `manage_streams` or per-stream `manage_stream` permission, the + stream does not exist, or the request fails. + """ def create_topic( self, stream: builtins.str | builtins.int, @@ -2231,16 +2319,70 @@ class Stats: """ def __repr__(self) -> builtins.str: ... +@typing.final +class Stream: + r""" + Summary information returned by `IggyClient.get_streams()`. + + `created_at` is Unix time in microseconds. `size` is the stream's current + stored size in bytes. + """ + @property + def id(self) -> builtins.int: + r""" + Numeric stream identifier. + """ + @property + def created_at(self) -> builtins.int: + r""" + Stream creation time as Unix time in microseconds. + """ + @property + def name(self) -> builtins.str: + r""" + Unique stream name. + """ + @property + def size(self) -> builtins.int: + r""" + Current stored stream size in bytes. + """ + @property + def messages_count(self) -> builtins.int: + r""" + Total messages across all topics in the stream. + """ + @property + def topics_count(self) -> builtins.int: + r""" + Number of topics in the stream. + """ + @typing.final class StreamDetails: @property + def created_at(self) -> builtins.int: + r""" + Stream creation time as Unix time in microseconds. + """ + @property def id(self) -> builtins.int: ... @property def name(self) -> builtins.str: ... @property + def size(self) -> builtins.int: + r""" + Current stored stream size in bytes. + """ + @property def messages_count(self) -> builtins.int: ... @property def topics_count(self) -> builtins.int: ... + @property + def topics(self) -> builtins.list[Topic]: + r""" + Returns the topics in the stream. + """ @typing.final class StreamPermissions: diff --git a/foreign/python/pylock.toml b/foreign/python/pylock.toml index 37ced59acf..d88e4e29c6 100644 --- a/foreign/python/pylock.toml +++ b/foreign/python/pylock.toml @@ -1,22 +1,5 @@ -# Licensed to the Apache Software Foundation (ASF) under one -# or more contributor license agreements. See the NOTICE file -# distributed with this work for additional information -# regarding copyright ownership. The ASF licenses this file -# to you under the Apache License, Version 2.0 (the -# "License"); you may not use this file except in compliance -# with the License. You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, -# software distributed under the License is distributed on an -# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY -# KIND, either express or implied. See the License for the -# specific language governing permissions and limitations -# under the License. - # This file was autogenerated by uv via the following command: -# uv export --format pylock.toml --output-file pylock.toml --all-extras +# uv export --locked --format pylock.toml --output-file pylock.toml --all-extras lock-version = "1.0" created-by = "uv" requires-python = ">=3.10" @@ -364,15 +347,6 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/8a/0e/97c33bf5009bdbac74fd2beace167cab3f978feb69cc36f1ef79360d6c4e/exceptiongroup-1.3.1-py3-none-any.whl", upload-time = 2025-11-21T23:01:53Z, size = 16740, hashes = { sha256 = "a7a39a3bd276781e98394987d3a5701d0c4edffb633bb7a5144577f82c773598" } }, ] -[[packages]] -name = "execnet" -version = "2.1.2" -index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/bf/89/780e11f9588d9e7128a3f87788354c7946a9cbb1401ad38a48c4db9a4f07/execnet-2.1.2.tar.gz", upload-time = 2025-11-12T09:56:37Z, size = 166622, hashes = { sha256 = "63d83bfdd9a23e35b9c6a3261412324f964c2ec8dcd8d3c6916ee9373e0befcd" } } -wheels = [ - { url = "https://files.pythonhosted.org/packages/ab/84/02fc1827e8cdded4aa65baef11296a9bbe595c474f0d6d758af082d849fd/execnet-2.1.2-py3-none-any.whl", upload-time = 2025-11-12T09:56:36Z, size = 40708, hashes = { sha256 = "67fba928dd5a544b783f6056f449e5e3931a5c378b128bc18501f7ea79e296ec" } }, -] - [[packages]] name = "idna" version = "3.19" @@ -494,15 +468,6 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/fa/b6/3127540ecdf1464a00e5a01ee60a1b09175f6913f0644ac748494d9c4b21/pytest_timeout-2.4.0-py3-none-any.whl", upload-time = 2025-05-05T19:44:33Z, size = 14382, hashes = { sha256 = "c42667e5cdadb151aeb5b26d114aff6bdf5a907f176a007a30b940d3d865b5c2" } }, ] -[[packages]] -name = "pytest-xdist" -version = "3.8.0" -index = "https://pypi.org/simple" -sdist = { url = "https://files.pythonhosted.org/packages/78/b4/439b179d1ff526791eb921115fca8e44e596a13efeda518b9d845a619450/pytest_xdist-3.8.0.tar.gz", upload-time = 2025-07-01T13:30:59Z, size = 88069, hashes = { sha256 = "7e578125ec9bc6050861aa93f2d59f1d8d085595d6551c2c90b6f4fad8d3a9f1" } } -wheels = [ - { url = "https://files.pythonhosted.org/packages/ca/31/d4e37e9e550c2b92a9cbc2e4d0b7420a27224968580b5a447f420847c975/pytest_xdist-3.8.0-py3-none-any.whl", upload-time = 2025-07-01T13:30:56Z, size = 46396, hashes = { sha256 = "202ca578cfeb7370784a8c33d6d05bc6e13b4f25b5053c30a152269fd10f0b88" } }, -] - [[packages]] name = "python-dotenv" version = "1.2.3" diff --git a/foreign/python/pyproject.toml b/foreign/python/pyproject.toml index 1434010dc6..0ed0dae312 100644 --- a/foreign/python/pyproject.toml +++ b/foreign/python/pyproject.toml @@ -102,7 +102,6 @@ testing = [ "pytest>=9.1.1,<10.0", "pytest-asyncio>=0.24.0,<2.0", "pytest-cov>=4.0,<8.0", - "pytest-xdist>=3.0,<4.0", "pytest-timeout>=2.0,<3.0", ] @@ -123,7 +122,6 @@ all = [ "pytest>=9.1.1,<10.0", "pytest-asyncio>=0.24.0,<2.0", "pytest-cov>=4.0,<8.0", - "pytest-xdist>=3.0,<4.0", "pytest-timeout>=2.0,<3.0", "ruff>=0.1.0,<1.0", "testcontainers>=4.15.0,<5.0", diff --git a/foreign/python/src/client.rs b/foreign/python/src/client.rs index f15b6e2d20..fa0e9e3ac0 100644 --- a/foreign/python/src/client.rs +++ b/foreign/python/src/client.rs @@ -44,7 +44,7 @@ use crate::permissions::Permissions as PyPermissions; use crate::receive_message::{PollingStrategy, ReceiveMessage}; use crate::send_message::{SendMessage, SendMessagesResponse as PySendMessagesResponse}; use crate::stats::Stats as PyStats; -use crate::stream::StreamDetails; +use crate::stream::{Stream, StreamDetails}; use crate::topic::{IggyExpiry, MaxTopicSize, Topic, TopicDetails}; use crate::user::{ UserInfo as PyUserInfo, UserInfoDetails as PyUserInfoDetails, UserStatus as PyUserStatus, @@ -550,6 +550,142 @@ impl IggyClient { }) } + /// Return all streams. + /// + /// Returns: + /// A list of `Stream` summaries. + /// + /// Raises: + /// RuntimeError: If the client is not authenticated, the user lacks global + /// `read_streams` or `manage_streams` permission, or the request fails. + #[gen_stub(override_return_type(type_repr="collections.abc.Awaitable[list[Stream]]", imports=("collections.abc")))] + fn get_streams<'a>(&self, py: Python<'a>) -> PyResult> { + let inner = self.inner.clone(); + future_into_py(py, async move { + let streams = inner + .get_streams() + .await + .map_err(|e| PyErr::new::(e.to_string()))?; + Ok(streams.into_iter().map(Stream::from).collect::>()) + }) + } + + /// Rename a stream selected by name or numeric ID. + /// + /// `stream_id` accepts a stream name as `str` or numeric ID as `int`. A + /// decimal-only string is interpreted as a numeric ID. `name` must be unique + /// and contain between 1 and 255 UTF-8 bytes. Renaming a stream to its current + /// name succeeds without changing it. + /// + /// Args: + /// stream_id: Stream identifier as `str | int`. + /// name: New stream name as `str`. + /// options: Additional option keys as `dict[str, str] | None`, forwarded + /// to the server. Current server versions reject all stream update + /// option keys. + /// + /// Returns: + /// None. + /// + /// Raises: + /// TypeError: If `stream_id` is not `str` or an integer in + /// `0..=2**32 - 1`, or `name` is not `str`. + /// ValueError: If a string identifier is empty or exceeds 255 UTF-8 bytes. + /// RuntimeError: If the client is not authenticated, the user lacks global + /// `manage_streams` or per-stream `manage_stream` permission, the + /// stream does not exist, the new name is invalid or already used, or + /// the request fails. + #[pyo3(signature = (stream_id, name, options = None))] + #[gen_stub(override_return_type(type_repr="collections.abc.Awaitable[None]", imports=("collections.abc")))] + fn update_stream<'a>( + &self, + py: Python<'a>, + stream_id: PyIdentifier, + name: String, + #[gen_stub(override_type(type_repr = "builtins.dict[builtins.str, builtins.str] | None"))] + options: Option>, + ) -> PyResult> { + let stream_id = Identifier::try_from(stream_id)?; + let update_options = StreamUpdateOptions { + raw: options.unwrap_or_default(), + }; + let inner = self.inner.clone(); + future_into_py(py, async move { + inner + .update_stream(&stream_id, &name, &update_options) + .await + .map_err(|e| PyErr::new::(e.to_string()))?; + Ok(()) + }) + } + + /// Delete a stream selected by name or numeric ID. + /// + /// Deletion removes the stream and all of its topics, partitions, and messages. + /// `stream_id` accepts a stream name as `str` or numeric ID as `int`. A + /// decimal-only string is interpreted as a numeric ID. + /// + /// Returns: + /// None. + /// + /// Raises: + /// TypeError: If `stream_id` is not `str` or an integer in + /// `0..=2**32 - 1`. + /// ValueError: If a string identifier is empty or exceeds 255 UTF-8 bytes. + /// RuntimeError: If the client is not authenticated, the user lacks global + /// `manage_streams` or per-stream `manage_stream` permission, the + /// stream does not exist, or the request fails. + #[gen_stub(override_return_type(type_repr="collections.abc.Awaitable[None]", imports=("collections.abc")))] + fn delete_stream<'a>( + &self, + py: Python<'a>, + stream_id: PyIdentifier, + ) -> PyResult> { + let stream_id = Identifier::try_from(stream_id)?; + let inner = self.inner.clone(); + future_into_py(py, async move { + inner + .delete_stream(&stream_id) + .await + .map_err(|e| PyErr::new::(e.to_string()))?; + Ok(()) + }) + } + + /// Delete all messages from every topic in a stream. + /// + /// The stream, topics, and partitions remain available. Repeated purges of an + /// existing empty stream succeed. `stream_id` accepts a stream name as `str` + /// or numeric ID as `int`. A decimal-only string is interpreted as a numeric + /// ID. + /// + /// Returns: + /// None. + /// + /// Raises: + /// TypeError: If `stream_id` is not `str` or an integer in + /// `0..=2**32 - 1`. + /// ValueError: If a string identifier is empty or exceeds 255 UTF-8 bytes. + /// RuntimeError: If the client is not authenticated, the user lacks global + /// `manage_streams` or per-stream `manage_stream` permission, the + /// stream does not exist, or the request fails. + #[gen_stub(override_return_type(type_repr="collections.abc.Awaitable[None]", imports=("collections.abc")))] + fn purge_stream<'a>( + &self, + py: Python<'a>, + stream_id: PyIdentifier, + ) -> PyResult> { + let stream_id = Identifier::try_from(stream_id)?; + let inner = self.inner.clone(); + future_into_py(py, async move { + inner + .purge_stream(&stream_id) + .await + .map_err(|e| PyErr::new::(e.to_string()))?; + Ok(()) + }) + } + /// Creates a new topic with the given parameters. /// /// Args: diff --git a/foreign/python/src/lib.rs b/foreign/python/src/lib.rs index 5da34876c5..e39b3339e8 100644 --- a/foreign/python/src/lib.rs +++ b/foreign/python/src/lib.rs @@ -42,7 +42,7 @@ use pyo3::prelude::*; use receive_message::{PollingStrategy, ReceiveMessage}; use send_message::{SendMessage, SendMessagesConfirmation, SendMessagesResponse}; use stats::{CacheMetrics, CacheMetricsKey, Stats}; -use stream::StreamDetails; +use stream::{Stream, StreamDetails}; use topic::{IggyExpiry, MaxTopicSize, Partition, Topic, TopicDetails}; use user::{UserInfo, UserInfoDetails, UserStatus}; use user_headers::{HeaderKey, HeaderValue, UserHeaders}; @@ -61,6 +61,7 @@ fn apache_iggy(_py: Python, m: &Bound<'_, PyModule>) -> PyResult<()> { m.add_class::()?; m.add_class::()?; m.add_class::()?; + m.add_class::()?; m.add_class::()?; m.add_class::()?; m.add_class::()?; diff --git a/foreign/python/src/options.rs b/foreign/python/src/options.rs index c3eee8cdd6..fa489c5503 100644 --- a/foreign/python/src/options.rs +++ b/foreign/python/src/options.rs @@ -62,6 +62,7 @@ impl OptionSpec { /// /// The same type message user headers use, so the usual accessors read it; /// options ride that codec. + #[gen_stub(override_return_type(type_repr = "HeaderValue | None"))] #[getter] pub fn default_value<'a>(&self, py: Python<'a>) -> PyResult>> { if self.inner.default_value.is_empty() { diff --git a/foreign/python/src/stream.rs b/foreign/python/src/stream.rs index 7343d2c53f..d5f6a8e444 100644 --- a/foreign/python/src/stream.rs +++ b/foreign/python/src/stream.rs @@ -15,10 +15,12 @@ // specific language governing permissions and limitations // under the License. -use iggy::prelude::StreamDetails as RustStreamDetails; +use iggy::prelude::{Stream as RustStream, StreamDetails as RustStreamDetails}; use pyo3::prelude::*; use pyo3_stub_gen::derive::{gen_stub_pyclass, gen_stub_pymethods}; +use crate::topic::Topic; + #[pyclass] #[gen_stub_pyclass] pub struct StreamDetails { @@ -36,21 +38,95 @@ impl From for StreamDetails { #[gen_stub_pymethods] #[pymethods] impl StreamDetails { + /// Stream creation time as Unix time in microseconds. + #[getter] + pub fn created_at(&self) -> u64 { + self.inner.created_at.as_micros() + } + + #[getter] + pub fn id(&self) -> u32 { + self.inner.id + } + + #[getter] + pub fn name(&self) -> String { + self.inner.name.to_string() + } + + /// Current stored stream size in bytes. + #[getter] + pub fn size(&self) -> u64 { + self.inner.size.as_bytes_u64() + } + + #[getter] + pub fn messages_count(&self) -> u64 { + self.inner.messages_count + } + + #[getter] + pub fn topics_count(&self) -> u32 { + self.inner.topics_count + } + + /// Returns the topics in the stream. + #[getter] + pub fn topics(&self) -> Vec { + self.inner.topics.iter().map(Topic::from).collect() + } +} + +/// Summary information returned by `IggyClient.get_streams()`. +/// +/// `created_at` is Unix time in microseconds. `size` is the stream's current +/// stored size in bytes. +#[gen_stub_pyclass] +#[pyclass] +pub struct Stream { + pub(crate) inner: RustStream, +} + +impl From for Stream { + fn from(stream: RustStream) -> Self { + Self { inner: stream } + } +} + +#[gen_stub_pymethods] +#[pymethods] +impl Stream { + /// Numeric stream identifier. #[getter] pub fn id(&self) -> u32 { self.inner.id } + /// Stream creation time as Unix time in microseconds. + #[getter] + pub fn created_at(&self) -> u64 { + self.inner.created_at.as_micros() + } + + /// Unique stream name. #[getter] pub fn name(&self) -> String { self.inner.name.to_string() } + /// Current stored stream size in bytes. + #[getter] + pub fn size(&self) -> u64 { + self.inner.size.as_bytes_u64() + } + + /// Total messages across all topics in the stream. #[getter] pub fn messages_count(&self) -> u64 { self.inner.messages_count } + /// Number of topics in the stream. #[getter] pub fn topics_count(&self) -> u32 { self.inner.topics_count diff --git a/foreign/python/src/topic.rs b/foreign/python/src/topic.rs index 0d6e35c803..fe2be86d61 100644 --- a/foreign/python/src/topic.rs +++ b/foreign/python/src/topic.rs @@ -185,6 +185,25 @@ impl From for Topic { } } +impl From<&RustTopic> for Topic { + fn from(topic: &RustTopic) -> Self { + Self { + inner: RustTopic { + id: topic.id, + created_at: topic.created_at, + name: topic.name.clone(), + size: topic.size, + message_expiry: topic.message_expiry, + compression_algorithm: topic.compression_algorithm, + max_topic_size: topic.max_topic_size, + messages_count: topic.messages_count, + partitions_count: topic.partitions_count, + options: topic.options.clone(), + }, + } + } +} + #[gen_stub_pymethods] #[pymethods] impl Topic { diff --git a/foreign/python/tests/conftest.py b/foreign/python/tests/conftest.py index 3ab97065f5..ac9acdec4f 100644 --- a/foreign/python/tests/conftest.py +++ b/foreign/python/tests/conftest.py @@ -22,13 +22,12 @@ and connecting to Iggy servers in various configurations. """ -# TODO(slbotbm): Create text fixture for clean up after -# delete_stream() has been implemented. - import asyncio +import contextlib import os import secrets import string +from collections.abc import AsyncGenerator from pathlib import Path import pytest @@ -108,6 +107,28 @@ def make_name( return make_name +@pytest.fixture(scope="function", autouse=True) +async def cleanup_streams( + request: pytest.FixtureRequest, +) -> AsyncGenerator[None, None]: + """Delete streams created by the current test.""" + if "iggy_client" not in request.fixturenames: + yield + return + + client: IggyClient = request.getfixturevalue("iggy_client") + existing_streams = { + (stream.id, stream.created_at) for stream in await client.get_streams() + } + + yield + + for stream in await client.get_streams(): + if (stream.id, stream.created_at) not in existing_streams: + with contextlib.suppress(RuntimeError): + await client.delete_stream(stream.id) + + @pytest.fixture(scope="session", autouse=True) def configure_asyncio(): """Configure asyncio settings for tests.""" diff --git a/foreign/python/tests/test_stream.py b/foreign/python/tests/test_stream.py index 5e71a543a4..c6e6bdc75d 100644 --- a/foreign/python/tests/test_stream.py +++ b/foreign/python/tests/test_stream.py @@ -17,7 +17,7 @@ import pytest -from apache_iggy import IggyClient +from apache_iggy import IggyClient, SendMessage from .utils import get_server_config, wait_for_ping, wait_for_server @@ -62,7 +62,9 @@ async def test_create_and_get_stream( stream = await iggy_client.get_stream(stream_name) assert stream is not None + assert stream.created_at > 0 assert stream.name == stream_name + assert stream.size == 0 assert stream.topics_count == 0 @pytest.mark.asyncio @@ -203,3 +205,443 @@ async def test_create_stream_before_login_fails(self, unique_name): with pytest.raises(RuntimeError): await client.create_stream(unique_name()) + + +class TestGetStreams: + """Test listing streams via get_streams.""" + + @pytest.mark.asyncio + async def test_get_streams_returns_created_streams( + self, iggy_client: IggyClient, unique_name + ): + """Test get_streams returns every stream created during the test.""" + # Stream IDs can be reused after deletion, so creation order does not + # imply numeric ID order. The client fixture is session-scoped, so + # other tests may have created streams; assert on the ones created + # here instead of the full server view. + created = [unique_name(f"z{index}") for index in range(3, 0, -1)] + for name in created: + await iggy_client.create_stream(name) + + streams = await iggy_client.get_streams() + created_names = set(created) + mine = [stream for stream in streams if stream.name in created_names] + + assert {stream.name for stream in mine} == created_names + assert [stream.id for stream in mine] == sorted(stream.id for stream in mine) + assert all(stream.created_at > 0 for stream in mine) + assert all(stream.size == 0 for stream in mine) + assert all(stream.messages_count == 0 for stream in mine) + assert all(stream.topics_count == 0 for stream in mine) + + @pytest.mark.asyncio + async def test_get_streams_reflects_topic_count( + self, iggy_client: IggyClient, unique_name + ): + """Test get_streams reports the topic count for a listed stream.""" + stream_name = unique_name() + topic_name = unique_name() + + await iggy_client.create_stream(stream_name) + await iggy_client.create_topic( + stream=stream_name, name=topic_name, partitions_count=1 + ) + + streams = await iggy_client.get_streams() + listed = next( + (stream for stream in streams if stream.name == stream_name), None + ) + assert listed is not None + assert listed.topics_count == 1 + + stream = await iggy_client.get_stream(stream_name) + assert stream is not None + assert [topic.name for topic in stream.topics] == [topic_name] + + @pytest.mark.asyncio + async def test_get_streams_returns_same_result_when_called_repeatedly( + self, iggy_client: IggyClient, unique_name + ): + """Test repeated get_streams calls return an identically ordered view.""" + await iggy_client.create_stream(unique_name()) + + first = await iggy_client.get_streams() + second = await iggy_client.get_streams() + assert [stream.id for stream in first] == [stream.id for stream in second] + assert [stream.name for stream in first] == [stream.name for stream in second] + + @pytest.mark.asyncio + async def test_get_streams_requires_connection_and_auth(self): + """Test get_streams fails both before connecting and before logging in.""" + host, port = get_server_config() + wait_for_server(host, port) + + client = IggyClient(f"{host}:{port}") + with pytest.raises(RuntimeError): + await client.get_streams() + + await client.connect() + with pytest.raises(RuntimeError): + await client.get_streams() + + +class TestUpdateStream: + """Test updating streams via update_stream.""" + + @pytest.mark.asyncio + async def test_update_stream_renames_stream( + self, iggy_client: IggyClient, unique_name + ): + """Test update_stream renames a stream; old name no longer resolves.""" + stream_name = unique_name() + new_name = unique_name() + + await iggy_client.create_stream(stream_name) + before = await iggy_client.get_stream(stream_name) + assert before is not None + + await iggy_client.update_stream(stream_id=stream_name, name=new_name) + + renamed = await iggy_client.get_stream(new_name) + assert renamed is not None + assert renamed.name == new_name + assert renamed.id == before.id + + old = await iggy_client.get_stream(stream_name) + assert old is None + + @pytest.mark.asyncio + async def test_update_stream_by_numeric_id( + self, iggy_client: IggyClient, unique_name + ): + """Test update_stream accepts a numeric stream id.""" + stream_name = unique_name() + new_name = unique_name() + + await iggy_client.create_stream(stream_name) + stream = await iggy_client.get_stream(stream_name) + assert stream is not None + + await iggy_client.update_stream(stream_id=stream.id, name=new_name) + + renamed = await iggy_client.get_stream(new_name) + assert renamed is not None + assert renamed.name == new_name + + @pytest.mark.asyncio + async def test_update_stream_forwards_options( + self, iggy_client: IggyClient, unique_name + ): + """Test update_stream forwards option keys to the server.""" + stream_name = unique_name() + await iggy_client.create_stream(stream_name) + + with pytest.raises(RuntimeError): + await iggy_client.update_stream( + stream_id=stream_name, + name=stream_name, + options={"unknown": "value"}, + ) + + @pytest.mark.asyncio + @pytest.mark.parametrize( + "new_name", + [ + pytest.param("a", id="one-byte"), + pytest.param("a" * 255, id="255-byte-ascii"), + pytest.param(("é" * 127) + "a", id="255-byte-utf8"), + ], + ) + async def test_update_stream_accepts_name_boundaries( + self, iggy_client: IggyClient, unique_name, new_name: str + ): + """Test update_stream accepts names at the UTF-8 byte boundaries.""" + stream_name = unique_name() + await iggy_client.create_stream(stream_name) + + await iggy_client.update_stream(stream_id=stream_name, name=new_name) + renamed = await iggy_client.get_stream(new_name) + assert renamed is not None + assert renamed.name == new_name + + @pytest.mark.asyncio + @pytest.mark.parametrize( + "new_name", + [ + pytest.param("", id="empty"), + pytest.param("a" * 256, id="256-byte-ascii"), + pytest.param("é" * 128, id="256-byte-utf8"), + pytest.param(("😀" * 63) + "aaaa", id="256-byte-four-byte-utf8"), + ], + ) + async def test_update_stream_rejects_invalid_name_boundaries( + self, iggy_client: IggyClient, unique_name, new_name: str + ): + """Test update_stream rejects empty and 256-byte names.""" + stream_name = unique_name() + await iggy_client.create_stream(stream_name) + + with pytest.raises(RuntimeError): + await iggy_client.update_stream(stream_id=stream_name, name=new_name) + + @pytest.mark.asyncio + async def test_update_stream_applies_repeated_updates( + self, iggy_client: IggyClient, unique_name + ): + """Test successive update_stream calls each take effect.""" + stream_name = unique_name() + first_rename = unique_name() + second_rename = unique_name() + + await iggy_client.create_stream(stream_name) + + await iggy_client.update_stream(stream_id=stream_name, name=first_rename) + after_first = await iggy_client.get_stream(first_rename) + assert after_first is not None + assert after_first.name == first_rename + assert await iggy_client.get_stream(stream_name) is None + + await iggy_client.update_stream(stream_id=first_rename, name=second_rename) + after_second = await iggy_client.get_stream(second_rename) + assert after_second is not None + assert after_second.name == second_rename + assert await iggy_client.get_stream(first_rename) is None + + @pytest.mark.asyncio + async def test_update_nonexistent_stream_fails( + self, iggy_client: IggyClient, unique_name + ): + """Test update_stream raises for a non-existent stream.""" + with pytest.raises(RuntimeError): + await iggy_client.update_stream(stream_id=unique_name(), name=unique_name()) + + @pytest.mark.asyncio + async def test_update_stream_to_existing_name_fails_and_current_name_is_a_noop( + self, iggy_client: IggyClient, unique_name + ): + """Test update_stream rejects conflicts and preserves a self-rename.""" + first_stream = unique_name() + second_stream = unique_name() + + await iggy_client.create_stream(first_stream) + await iggy_client.create_stream(second_stream) + streams = await iggy_client.get_streams() + before = next(stream for stream in streams if stream.name == second_stream) + before_metadata = ( + before.id, + before.created_at, + before.name, + before.size, + before.messages_count, + before.topics_count, + ) + + with pytest.raises(RuntimeError): + await iggy_client.update_stream(stream_id=second_stream, name=first_stream) + + await iggy_client.update_stream(stream_id=second_stream, name=second_stream) + + streams = await iggy_client.get_streams() + after = next(stream for stream in streams if stream.id == before.id) + assert ( + after.id, + after.created_at, + after.name, + after.size, + after.messages_count, + after.topics_count, + ) == before_metadata + + @pytest.mark.asyncio + async def test_update_stream_requires_connection_and_auth(self, unique_name): + """Test update_stream fails both before connecting and before logging in.""" + host, port = get_server_config() + wait_for_server(host, port) + + client = IggyClient(f"{host}:{port}") + with pytest.raises(RuntimeError): + await client.update_stream(stream_id=unique_name(), name=unique_name()) + + await client.connect() + with pytest.raises(RuntimeError): + await client.update_stream(stream_id=unique_name(), name=unique_name()) + + +class TestDeleteStream: + """Test deleting streams via delete_stream.""" + + @pytest.mark.asyncio + async def test_delete_stream_removes_stream( + self, iggy_client: IggyClient, unique_name + ): + """Test delete_stream removes the stream so it no longer resolves.""" + stream_name = unique_name() + + await iggy_client.create_stream(stream_name) + + await iggy_client.delete_stream(stream_name) + + assert await iggy_client.get_stream(stream_name) is None + + @pytest.mark.asyncio + async def test_delete_stream_by_numeric_id( + self, iggy_client: IggyClient, unique_name + ): + """Test delete_stream accepts a numeric stream id.""" + stream_name = unique_name() + + await iggy_client.create_stream(stream_name) + stream = await iggy_client.get_stream(stream_name) + assert stream is not None + + await iggy_client.delete_stream(stream.id) + + assert await iggy_client.get_stream(stream_name) is None + + @pytest.mark.asyncio + async def test_delete_stream_leaves_other_streams( + self, iggy_client: IggyClient, unique_name + ): + """Test delete_stream removes only the targeted stream.""" + stream_to_delete = unique_name() + stream_to_keep = unique_name() + + await iggy_client.create_stream(stream_to_delete) + await iggy_client.create_stream(stream_to_keep) + + await iggy_client.delete_stream(stream_to_delete) + + assert await iggy_client.get_stream(stream_to_delete) is None + kept = await iggy_client.get_stream(stream_to_keep) + assert kept is not None + assert kept.name == stream_to_keep + + @pytest.mark.asyncio + async def test_delete_nonexistent_stream_fails( + self, iggy_client: IggyClient, unique_name + ): + """Test delete_stream raises for a non-existent stream.""" + with pytest.raises(RuntimeError): + await iggy_client.delete_stream(unique_name()) + + @pytest.mark.asyncio + async def test_delete_stream_twice_fails_second_time( + self, iggy_client: IggyClient, unique_name + ): + """Test deleting an already-deleted stream raises on the second call.""" + stream_name = unique_name() + + await iggy_client.create_stream(stream_name) + + await iggy_client.delete_stream(stream_name) + with pytest.raises(RuntimeError): + await iggy_client.delete_stream(stream_name) + + @pytest.mark.asyncio + async def test_delete_stream_requires_connection_and_auth(self, unique_name): + """Test delete_stream fails both before connecting and before logging in.""" + host, port = get_server_config() + wait_for_server(host, port) + + client = IggyClient(f"{host}:{port}") + with pytest.raises(RuntimeError): + await client.delete_stream(unique_name()) + + await client.connect() + with pytest.raises(RuntimeError): + await client.delete_stream(unique_name()) + + +class TestPurgeStream: + """Test purging stream messages via purge_stream.""" + + @pytest.mark.asyncio + async def test_purge_stream_clears_messages_but_keeps_stream( + self, iggy_client: IggyClient, unique_name + ): + """Test purge_stream empties the stream while leaving it in place.""" + stream_name = unique_name() + topic_name = unique_name() + + await iggy_client.create_stream(stream_name) + await iggy_client.create_topic( + stream=stream_name, name=topic_name, partitions_count=1 + ) + + messages = [SendMessage(f"payload-{index}") for index in range(5)] + await iggy_client.send_messages(stream_name, topic_name, 0, messages) + + before = await iggy_client.get_stream(stream_name) + assert before is not None + assert before.messages_count == 5 + + await iggy_client.purge_stream(stream_name) + + after = await iggy_client.get_stream(stream_name) + # Purging clears messages only; the stream itself survives (purge is + # not delete) and keeps its identity and topics. + assert after is not None + assert after.messages_count == 0 + assert after.id == before.id + assert after.name == before.name + assert after.topics_count == before.topics_count + + @pytest.mark.asyncio + async def test_purge_empty_stream_succeeds( + self, iggy_client: IggyClient, unique_name + ): + """Test purge_stream is a no-op on a stream with no messages.""" + stream_name = unique_name() + + await iggy_client.create_stream(stream_name) + + await iggy_client.purge_stream(stream_name) + + stream = await iggy_client.get_stream(stream_name) + assert stream is not None + assert stream.messages_count == 0 + + @pytest.mark.asyncio + async def test_purge_stream_is_idempotent_when_called_repeatedly( + self, iggy_client: IggyClient, unique_name + ): + """Test purge_stream succeeds when called repeatedly on the same stream.""" + stream_name = unique_name() + topic_name = unique_name() + + await iggy_client.create_stream(stream_name) + await iggy_client.create_topic( + stream=stream_name, name=topic_name, partitions_count=1 + ) + + messages = [SendMessage(f"payload-{index}") for index in range(5)] + await iggy_client.send_messages(stream_name, topic_name, 0, messages) + + await iggy_client.purge_stream(stream_name) + await iggy_client.purge_stream(stream_name) + + stream = await iggy_client.get_stream(stream_name) + assert stream is not None + assert stream.messages_count == 0 + + @pytest.mark.asyncio + async def test_purge_nonexistent_stream_fails( + self, iggy_client: IggyClient, unique_name + ): + """Test purge_stream raises for a non-existent stream.""" + with pytest.raises(RuntimeError): + await iggy_client.purge_stream(unique_name()) + + @pytest.mark.asyncio + async def test_purge_stream_requires_connection_and_auth(self, unique_name): + """Test purge_stream fails both before connecting and before logging in.""" + host, port = get_server_config() + wait_for_server(host, port) + + client = IggyClient(f"{host}:{port}") + with pytest.raises(RuntimeError): + await client.purge_stream(unique_name()) + + await client.connect() + with pytest.raises(RuntimeError): + await client.purge_stream(unique_name()) diff --git a/foreign/python/uv.lock b/foreign/python/uv.lock index f4fce79c96..0c1d257fa5 100644 --- a/foreign/python/uv.lock +++ b/foreign/python/uv.lock @@ -19,7 +19,6 @@ all = [ { name = "pytest-asyncio" }, { name = "pytest-cov" }, { name = "pytest-timeout" }, - { name = "pytest-xdist" }, { name = "ruff" }, { name = "testcontainers" }, ] @@ -33,7 +32,6 @@ testing = [ { name = "pytest-asyncio" }, { name = "pytest-cov" }, { name = "pytest-timeout" }, - { name = "pytest-xdist" }, ] testing-docker = [ { name = "testcontainers" }, @@ -53,8 +51,6 @@ requires-dist = [ { name = "pytest-cov", marker = "extra == 'testing'", specifier = ">=4.0,<8.0" }, { name = "pytest-timeout", marker = "extra == 'all'", specifier = ">=2.0,<3.0" }, { name = "pytest-timeout", marker = "extra == 'testing'", specifier = ">=2.0,<3.0" }, - { name = "pytest-xdist", marker = "extra == 'all'", specifier = ">=3.0,<4.0" }, - { name = "pytest-xdist", marker = "extra == 'testing'", specifier = ">=3.0,<4.0" }, { name = "ruff", marker = "extra == 'all'", specifier = ">=0.1.0,<1.0" }, { name = "ruff", marker = "extra == 'dev'", specifier = ">=0.1.0,<1.0" }, { name = "testcontainers", marker = "extra == 'all'", specifier = ">=4.15.0,<5.0" }, @@ -411,15 +407,6 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/8a/0e/97c33bf5009bdbac74fd2beace167cab3f978feb69cc36f1ef79360d6c4e/exceptiongroup-1.3.1-py3-none-any.whl", hash = "sha256:a7a39a3bd276781e98394987d3a5701d0c4edffb633bb7a5144577f82c773598", size = 16740, upload-time = "2025-11-21T23:01:53.443Z" }, ] -[[package]] -name = "execnet" -version = "2.1.2" -source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/bf/89/780e11f9588d9e7128a3f87788354c7946a9cbb1401ad38a48c4db9a4f07/execnet-2.1.2.tar.gz", hash = "sha256:63d83bfdd9a23e35b9c6a3261412324f964c2ec8dcd8d3c6916ee9373e0befcd", size = 166622, upload-time = "2025-11-12T09:56:37.75Z" } -wheels = [ - { url = "https://files.pythonhosted.org/packages/ab/84/02fc1827e8cdded4aa65baef11296a9bbe595c474f0d6d758af082d849fd/execnet-2.1.2-py3-none-any.whl", hash = "sha256:67fba928dd5a544b783f6056f449e5e3931a5c378b128bc18501f7ea79e296ec", size = 40708, upload-time = "2025-11-12T09:56:36.333Z" }, -] - [[package]] name = "idna" version = "3.19" @@ -566,19 +553,6 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/fa/b6/3127540ecdf1464a00e5a01ee60a1b09175f6913f0644ac748494d9c4b21/pytest_timeout-2.4.0-py3-none-any.whl", hash = "sha256:c42667e5cdadb151aeb5b26d114aff6bdf5a907f176a007a30b940d3d865b5c2", size = 14382, upload-time = "2025-05-05T19:44:33.502Z" }, ] -[[package]] -name = "pytest-xdist" -version = "3.8.0" -source = { registry = "https://pypi.org/simple" } -dependencies = [ - { name = "execnet" }, - { name = "pytest" }, -] -sdist = { url = "https://files.pythonhosted.org/packages/78/b4/439b179d1ff526791eb921115fca8e44e596a13efeda518b9d845a619450/pytest_xdist-3.8.0.tar.gz", hash = "sha256:7e578125ec9bc6050861aa93f2d59f1d8d085595d6551c2c90b6f4fad8d3a9f1", size = 88069, upload-time = "2025-07-01T13:30:59.346Z" } -wheels = [ - { url = "https://files.pythonhosted.org/packages/ca/31/d4e37e9e550c2b92a9cbc2e4d0b7420a27224968580b5a447f420847c975/pytest_xdist-3.8.0-py3-none-any.whl", hash = "sha256:202ca578cfeb7370784a8c33d6d05bc6e13b4f25b5053c30a152269fd10f0b88", size = 46396, upload-time = "2025-07-01T13:30:56.632Z" }, -] - [[package]] name = "python-dotenv" version = "1.2.3" diff --git a/licenserc.toml b/licenserc.toml index 1005808aa7..fdcf10f3a1 100644 --- a/licenserc.toml +++ b/licenserc.toml @@ -37,6 +37,7 @@ excludes = [ "**/local_data*/**", "**/performance_results*/**", "**/*.lock", + "**/pylock.toml", "**/*.txt", "**/*.tpl", "**/LICENSE", From b87b6a6f77dfa650753cc1998f75fc7641f00463 Mon Sep 17 00:00:00 2001 From: Maciej Modzelewski Date: Mon, 7 Sep 2026 15:59:49 +0200 Subject: [PATCH 079/182] fix(java): size wire strings by UTF-8 byte length, not char count (#4068) Names, keys and passwords with non-ASCII characters went out with a length prefix counted in UTF-16 chars while the bytes came from the platform default charset, so the prefix and payload disagreed and the server misread the frame. The 255 cap was also checked in chars, so a short name could still overflow the u8 prefix, and uncapped strings wrapped it silently. Every wire string is now encoded as UTF-8 once and prefixed with its byte length. The serializer rejects empty or over-long strings and names the field. Identifier caches the encoded name, and Partitioning checks the value shape the server expects for each kind. The SDK version is cut at a code point boundary. Message.of, the example consumer and the javadoc all use UTF-8 explicitly. The login payload no longer carries version and context strings; the VSR codec drops them and sends the SDK version itself. The codec now validates fields before allocating, closing a pooled-buffer leak on an empty username. Usernames, passwords and token names are checked against the server's own bounds before the round trip. Message keys are hashed as UTF-8 bytes, so on a JVM whose default charset is not UTF-8 a non-ASCII key may land on a different partition than before. Needs a 0.9.0 release note. Non-ASCII names are tested end to end over TCP only; the HTTP client does not percent-encode paths yet. Refs #4056 --- .../iggy/examples/async/AsyncConsumer.java | 3 +- .../iggy/client/async/MessagesClient.java | 2 +- .../async/tcp/ConsumerGroupsTcpClient.java | 10 +- .../tcp/PersonalAccessTokensTcpClient.java | 14 +- .../client/async/tcp/StreamsTcpClient.java | 22 +-- .../client/async/tcp/TopicsTcpClient.java | 4 +- .../iggy/client/async/tcp/UsersTcpClient.java | 46 +++-- .../client/async/tcp/vsr/VsrLoginCodec.java | 58 +++++-- .../apache/iggy/identifier/Identifier.java | 32 +++- .../java/org/apache/iggy/message/Message.java | 3 +- .../org/apache/iggy/message/Partitioning.java | 42 ++++- .../apache/iggy/serde/BytesSerializer.java | 37 ++-- .../async/tcp/vsr/VsrLoginCodecTest.java | 159 ++++++++++++++++++ .../blocking/tcp/BytesSerializerTest.java | 109 ++++++++++-- .../blocking/tcp/MessagesTcpClientTest.java | 35 ++++ .../blocking/tcp/StreamTcpClientTest.java | 24 +++ .../iggy/identifier/IdentifierTest.java | 49 ++++++ .../org/apache/iggy/message/MessageTest.java | 9 + .../apache/iggy/message/PartitioningTest.java | 60 +++++++ 19 files changed, 616 insertions(+), 102 deletions(-) create mode 100644 foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/vsr/VsrLoginCodecTest.java diff --git a/examples/java/src/main/java/org/apache/iggy/examples/async/AsyncConsumer.java b/examples/java/src/main/java/org/apache/iggy/examples/async/AsyncConsumer.java index 99e59c4d21..885a181862 100644 --- a/examples/java/src/main/java/org/apache/iggy/examples/async/AsyncConsumer.java +++ b/examples/java/src/main/java/org/apache/iggy/examples/async/AsyncConsumer.java @@ -31,6 +31,7 @@ import org.slf4j.LoggerFactory; import java.math.BigInteger; +import java.nio.charset.StandardCharsets; import java.util.Optional; import java.util.concurrent.CompletableFuture; import java.util.concurrent.ExecutorService; @@ -312,7 +313,7 @@ private static CompletableFuture processMessages( int messageCount = polled.messages().size(); for (Message message : polled.messages()) { - String payload = new String(message.payload()); + String payload = new String(message.payload(), StandardCharsets.UTF_8); // Simulate message processing (in real app: parse, validate, store, etc.) // This could be CPU-intensive or involve blocking I/O (database, HTTP calls) diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/MessagesClient.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/MessagesClient.java index f3b9ad115b..7c69be7c6d 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/MessagesClient.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/MessagesClient.java @@ -59,7 +59,7 @@ * Consumer.of(1L), PollingStrategy.first(), 100L, true) * .thenAccept(polled -> { * for (var msg : polled.messages()) { - * System.out.println(new String(msg.payload())); + * System.out.println(new String(msg.payload(), StandardCharsets.UTF_8)); * } * }); * } diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/ConsumerGroupsTcpClient.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/ConsumerGroupsTcpClient.java index 786bcc766c..d763335eb4 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/ConsumerGroupsTcpClient.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/ConsumerGroupsTcpClient.java @@ -104,13 +104,9 @@ public CompletableFuture> getConsumerGroups(StreamId streamI @Override public CompletableFuture createConsumerGroup( StreamId streamId, TopicId topicId, String name) { - var streamIdBytes = BytesSerializer.toBytes(streamId); - var topicIdBytes = BytesSerializer.toBytes(topicId); - var payload = Unpooled.buffer(1 + streamIdBytes.readableBytes() + topicIdBytes.readableBytes() + name.length()); - - payload.writeBytes(streamIdBytes); - payload.writeBytes(topicIdBytes); - payload.writeBytes(BytesSerializer.toBytes(name)); + var payload = BytesSerializer.toBytes(streamId); + payload.writeBytes(BytesSerializer.toBytes(topicId)); + payload.writeBytes(BytesSerializer.toBytes(name, "name")); log.debug("Creating consumer group - Stream: {}, Topic: {}, Name: {}", streamId, topicId, name); diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/PersonalAccessTokensTcpClient.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/PersonalAccessTokensTcpClient.java index 6803843796..37671a8738 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/PersonalAccessTokensTcpClient.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/PersonalAccessTokensTcpClient.java @@ -44,6 +44,14 @@ public class PersonalAccessTokensTcpClient implements PersonalAccessTokensClient { private static final Logger log = LoggerFactory.getLogger(PersonalAccessTokensTcpClient.class); + /** + * Token name bounds the server enforces, in UTF-8 bytes. Checked here so a bad name fails + * before the round trip instead of as an opaque server error. + */ + private static final int MIN_NAME_LENGTH = 3; + + private static final int MAX_NAME_LENGTH = 30; + private final Supplier connectionSupplier; private final LoginRoutingHook routingHook; @@ -63,7 +71,7 @@ private AsyncTcpConnection connection() { @Override public CompletableFuture createPersonalAccessToken(String name, BigInteger expiry) { var payload = Unpooled.buffer(); - payload.writeBytes(BytesSerializer.toBytes(name)); + payload.writeBytes(BytesSerializer.toBytes(name, "name", MIN_NAME_LENGTH, MAX_NAME_LENGTH)); payload.writeBytes(BytesSerializer.toBytesAsU64(expiry)); log.debug("Creating personal access token: {}", name); @@ -102,7 +110,7 @@ public CompletableFuture> getPersonalAccessTokens( @Override public CompletableFuture deletePersonalAccessToken(String name) { - var payload = BytesSerializer.toBytes(name); + var payload = BytesSerializer.toBytes(name, "name", MIN_NAME_LENGTH, MAX_NAME_LENGTH); log.debug("Deleting personal access token: {}", name); @@ -119,7 +127,7 @@ public CompletableFuture loginWithPersonalAccessToken(String token } private CompletableFuture loginWithoutRedirect(String token) { - var payload = BytesSerializer.toBytes(token); + var payload = BytesSerializer.toBytes(token, "token"); log.debug("Logging in with personal access token"); diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/StreamsTcpClient.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/StreamsTcpClient.java index 77957a5b47..ec246ef41d 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/StreamsTcpClient.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/StreamsTcpClient.java @@ -23,8 +23,6 @@ import io.netty.util.ReferenceCounted; import org.apache.iggy.client.async.StreamsClient; import org.apache.iggy.identifier.StreamId; -import org.apache.iggy.message.HeaderKey; -import org.apache.iggy.message.HeaderValue; import org.apache.iggy.serde.BytesSerializer; import org.apache.iggy.serde.CommandCode; import org.apache.iggy.stream.StreamBase; @@ -32,7 +30,6 @@ import java.util.ArrayList; import java.util.List; -import java.util.Map; import java.util.Optional; import java.util.concurrent.CompletableFuture; import java.util.function.Supplier; @@ -58,10 +55,7 @@ private AsyncTcpConnection connection() { @Override public CompletableFuture createStream(String name) { - var payloadSize = 1 + name.length(); - var payload = Unpooled.buffer(payloadSize); - - payload.writeBytes(BytesSerializer.toBytes(name)); + var payload = BytesSerializer.toBytes(name, "name"); return connection().send(CommandCode.Stream.CREATE.getValue(), payload).thenApply(response -> { StreamDetails details = readStreamDetails(response); @@ -102,15 +96,11 @@ public CompletableFuture> getStreams() { @Override public CompletableFuture updateStream(StreamId streamId, String name) { - var payloadSize = 1 + name.length(); - var idBytes = toBytes(streamId); - var payload = Unpooled.buffer(payloadSize + idBytes.capacity()); - - payload.writeBytes(idBytes); - payload.writeBytes(BytesSerializer.toBytes(name)); - // Trailing options block. Streams have no catalog keys yet, so the - // server rejects every key; the empty block is the extension point. - payload.writeBytes(BytesSerializer.toBytes(Map.of())); + var payload = toBytes(streamId); + payload.writeBytes(BytesSerializer.toBytes(name, "name")); + // No trailing options block: streams have no catalog keys yet and the + // server reads an absent block as empty. Settings will ride one here, + // as topics do. return connection().send(CommandCode.Stream.UPDATE.getValue(), payload).thenAccept(ReferenceCounted::release); } diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/TopicsTcpClient.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/TopicsTcpClient.java index ff06d86b9b..a25371ae4e 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/TopicsTcpClient.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/TopicsTcpClient.java @@ -134,7 +134,7 @@ static ByteBuf createTopicPayload( var payload = Unpooled.buffer(); payload.writeBytes(toBytes(streamId)); payload.writeIntLE(partitionsCount.intValue()); - payload.writeBytes(BytesSerializer.toBytes(name)); + payload.writeBytes(BytesSerializer.toBytes(name, "name")); payload.writeBytes(BytesSerializer.toBytes( createTopicOptions(compressionAlgorithm, messageExpiry, maxTopicSize, options))); return payload; @@ -181,7 +181,7 @@ public CompletableFuture updateTopic( var payload = Unpooled.buffer(); payload.writeBytes(toBytes(streamId)); payload.writeBytes(toBytes(topicId)); - payload.writeBytes(BytesSerializer.toBytes(name)); + payload.writeBytes(BytesSerializer.toBytes(name, "name")); // Settings ride the options block. A default value means the caller did // not set the key, so it is omitted and the server leaves the topic's // current value alone. diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/UsersTcpClient.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/UsersTcpClient.java index 1bf1e7ceda..6a4979972c 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/UsersTcpClient.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/UsersTcpClient.java @@ -20,11 +20,8 @@ package org.apache.iggy.client.async.tcp; import io.netty.buffer.Unpooled; -import org.apache.iggy.IggyVersion; import org.apache.iggy.client.async.UsersClient; import org.apache.iggy.identifier.UserId; -import org.apache.iggy.message.HeaderKey; -import org.apache.iggy.message.HeaderValue; import org.apache.iggy.serde.BytesDeserializer; import org.apache.iggy.serde.CommandCode; import org.apache.iggy.user.IdentityInfo; @@ -36,7 +33,6 @@ import org.slf4j.LoggerFactory; import java.util.List; -import java.util.Map; import java.util.Optional; import java.util.concurrent.CompletableFuture; import java.util.function.Supplier; @@ -50,6 +46,16 @@ public class UsersTcpClient implements UsersClient { private static final Logger log = LoggerFactory.getLogger(UsersTcpClient.class); + /** + * Credential bounds the server enforces, in UTF-8 bytes. Checked here so a bad value fails + * before the round trip instead of as an opaque server error. + */ + private static final int MIN_USERNAME_LENGTH = 3; + + private static final int MAX_USERNAME_LENGTH = 50; + private static final int MIN_PASSWORD_LENGTH = 3; + private static final int MAX_PASSWORD_LENGTH = 100; + private final Supplier connectionSupplier; private final LoginRoutingHook routingHook; @@ -82,8 +88,8 @@ public CompletableFuture> getUsers() { public CompletableFuture createUser( String username, String password, UserStatus status, Optional permissions) { var payload = Unpooled.buffer(); - payload.writeBytes(toBytes(username)); - payload.writeBytes(toBytes(password)); + payload.writeBytes(toBytes(username, "username", MIN_USERNAME_LENGTH, MAX_USERNAME_LENGTH)); + payload.writeBytes(toBytes(password, "password", MIN_PASSWORD_LENGTH, MAX_PASSWORD_LENGTH)); payload.writeByte(status.asCode()); permissions.ifPresentOrElse( perms -> { @@ -109,7 +115,7 @@ public CompletableFuture updateUser(UserId userId, Optional userna username.ifPresentOrElse( un -> { payload.writeByte(1); - payload.writeBytes(toBytes(un)); + payload.writeBytes(toBytes(un, "username", MIN_USERNAME_LENGTH, MAX_USERNAME_LENGTH)); }, () -> payload.writeByte(0)); status.ifPresentOrElse( @@ -118,9 +124,9 @@ public CompletableFuture updateUser(UserId userId, Optional userna payload.writeByte(s.asCode()); }, () -> payload.writeByte(0)); - // Trailing options block. Users have no catalog keys yet, so the - // server rejects every key; the empty block is the extension point. - payload.writeBytes(toBytes(Map.of())); + // No trailing options block: users have no catalog keys yet and the + // server reads an absent block as empty. Settings will ride one here, + // as topics do. return connection().sendAndRelease(CommandCode.User.UPDATE, payload); } @@ -144,8 +150,8 @@ public CompletableFuture updatePermissions(UserId userId, Optional changePassword(UserId userId, String currentPassword, String newPassword) { var payload = toBytes(userId); - payload.writeBytes(toBytes(currentPassword)); - payload.writeBytes(toBytes(newPassword)); + payload.writeBytes(toBytes(currentPassword, "current password", MIN_PASSWORD_LENGTH, MAX_PASSWORD_LENGTH)); + payload.writeBytes(toBytes(newPassword, "new password", MIN_PASSWORD_LENGTH, MAX_PASSWORD_LENGTH)); return connection().sendAndRelease(CommandCode.User.CHANGE_PASSWORD, payload); } @@ -156,19 +162,11 @@ public CompletableFuture login(String username, String password) { } private CompletableFuture loginWithoutRedirect(String username, String password) { - String version = IggyVersion.getInstance().getUserAgent(); - String context = IggyVersion.getInstance().toString(); - + // The VSR codec re-frames this into a Register and carries the SDK + // version itself, so the payload is only the two credentials. var payload = Unpooled.buffer(); - var usernameBytes = toBytes(username); - var passwordBytes = toBytes(password); - - payload.writeBytes(usernameBytes); - payload.writeBytes(passwordBytes); - payload.writeIntLE(version.length()); - payload.writeBytes(version.getBytes()); - payload.writeIntLE(context.length()); - payload.writeBytes(context.getBytes()); + payload.writeBytes(toBytes(username, "username", MIN_USERNAME_LENGTH, MAX_USERNAME_LENGTH)); + payload.writeBytes(toBytes(password, "password", MIN_PASSWORD_LENGTH, MAX_PASSWORD_LENGTH)); log.debug("Logging in user: {}", username); diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/vsr/VsrLoginCodec.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/vsr/VsrLoginCodec.java index 36c856c3a2..5a6e9a1ab8 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/vsr/VsrLoginCodec.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/vsr/VsrLoginCodec.java @@ -25,6 +25,7 @@ import org.apache.iggy.exception.IggyInvalidArgumentException; import java.nio.charset.StandardCharsets; +import java.util.Arrays; /** * Rewrites serialized login payloads into the @@ -46,18 +47,27 @@ final class VsrLoginCodec { static final String SDK_NAME = "java-sdk"; + /** Bound on a u8-length-prefixed field, in encoded bytes. */ + static final int MAX_SHORT_FIELD_LENGTH = 255; + + private static final String UNKNOWN_SDK_VERSION = "unknown"; + + private static final byte[] SDK_NAME_FIELD = SDK_NAME.getBytes(StandardCharsets.UTF_8); + private static final byte[] SDK_VERSION_FIELD = + sdkVersionField(IggyVersion.getInstance().getVersion()); + private VsrLoginCodec() {} /** * {@code LoginUser} (code 38) payload in: - * {@code [username:u8-len][password:u8-len][version:u32-len][context:u32-len]}. - * The trailing version/context strings are superseded by the - * {@code ClientVersionInfo} prefix and dropped. + * {@code [username:u8-len][password:u8-len]}. Anything after the password + * is ignored. */ static ByteBuf rewriteUserLogin(ByteBufAllocator alloc, ByteBuf loginPayload) { ByteBuf in = loginPayload.slice(); byte[] username = readShortField(in, "username"); byte[] password = readShortField(in, "password"); + requireShortField(username, "username"); ByteBuf body = alloc.buffer(); writeVersionInfo(body); @@ -75,6 +85,7 @@ static ByteBuf rewriteUserLogin(ByteBufAllocator alloc, ByteBuf loginPayload) { static ByteBuf rewritePatLogin(ByteBufAllocator alloc, ByteBuf loginPayload) { ByteBuf in = loginPayload.slice(); byte[] token = readShortField(in, "token"); + requireShortField(token, "token"); ByteBuf body = alloc.buffer(); writeVersionInfo(body); @@ -93,16 +104,29 @@ static long readSessionEpoch(ByteBuf registerBody) { private static void writeVersionInfo(ByteBuf body) { body.writeIntLE(PROTOCOL_VERSION); - writeShortField(body, SDK_NAME.getBytes(StandardCharsets.UTF_8)); - writeShortField(body, sdkVersion().getBytes(StandardCharsets.UTF_8)); + writeShortField(body, SDK_NAME_FIELD); + writeShortField(body, SDK_VERSION_FIELD); } - private static String sdkVersion() { - String version = IggyVersion.getInstance().getVersion(); - if (version == null || version.isEmpty()) { - return "unknown"; + /** + * An over-long version is cut on the encoded bytes at a code point boundary, so the field + * always fits its u8 prefix and still decodes as UTF-8 on the server. + */ + static byte[] sdkVersionField(String version) { + String value = version == null || version.isEmpty() ? UNKNOWN_SDK_VERSION : version; + byte[] encoded = value.getBytes(StandardCharsets.UTF_8); + if (encoded.length <= MAX_SHORT_FIELD_LENGTH) { + return encoded; + } + int end = MAX_SHORT_FIELD_LENGTH; + while (isContinuationByte(encoded[end])) { + end--; } - return version.length() > 255 ? version.substring(0, 255) : version; + return Arrays.copyOf(encoded, end); + } + + private static boolean isContinuationByte(byte value) { + return (value & 0xC0) == 0x80; } private static byte[] readShortField(ByteBuf in, String field) { @@ -118,10 +142,18 @@ private static byte[] readShortField(ByteBuf in, String field) { return value; } - private static void writeShortField(ByteBuf out, byte[] value) { - if (value.length == 0 || value.length > 255) { - throw new IggyInvalidArgumentException("Wire name fields must be 1..255 bytes, got " + value.length); + /** + * Runs before {@code alloc.buffer()}: the encoder releases the body only once the codec + * returns, so a throw after allocation would leak the pooled buffer. + */ + private static void requireShortField(byte[] value, String field) { + if (value.length == 0 || value.length > MAX_SHORT_FIELD_LENGTH) { + throw new IggyInvalidArgumentException( + "Login payload " + field + " must be 1.." + MAX_SHORT_FIELD_LENGTH + " bytes, got " + value.length); } + } + + private static void writeShortField(ByteBuf out, byte[] value) { out.writeByte(value.length); out.writeBytes(value); } diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/identifier/Identifier.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/identifier/Identifier.java index 945d35696f..ddf94b0f7f 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/identifier/Identifier.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/identifier/Identifier.java @@ -19,14 +19,21 @@ package org.apache.iggy.identifier; +import io.netty.buffer.ByteBuf; +import io.netty.buffer.Unpooled; import org.apache.commons.lang3.StringUtils; import org.apache.iggy.exception.IggyInvalidArgumentException; import javax.annotation.Nullable; +import java.nio.charset.StandardCharsets; public abstract class Identifier { + /** Server-side cap on a wire name, in UTF-8 bytes, matching its u8 length prefix. */ + private static final int MAX_NAME_LENGTH = 255; + private final String name; + private final byte[] encodedName; private final Long id; protected Identifier(@Nullable String name, @Nullable Long id) { @@ -37,10 +44,17 @@ protected Identifier(@Nullable String name, @Nullable Long id) { throw new IggyInvalidArgumentException("Name and id cannot be both present"); } if (StringUtils.isNotBlank(name)) { + byte[] encoded = name.getBytes(StandardCharsets.UTF_8); + if (encoded.length > MAX_NAME_LENGTH) { + throw new IggyInvalidArgumentException( + "Name must be at most " + MAX_NAME_LENGTH + " bytes, got " + encoded.length); + } this.name = name; + this.encodedName = encoded; this.id = null; } else { this.name = null; + this.encodedName = null; this.id = id; } } @@ -68,13 +82,27 @@ public String getName() { return name; } + /** Wire encoding: kind, u8 length, then the u32 little-endian id or the UTF-8 name. */ + public ByteBuf toBytes() { + ByteBuf buffer = Unpooled.buffer(getSize()); + buffer.writeByte(getKind()); + if (id != null) { + buffer.writeByte(4); + buffer.writeIntLE(id.intValue()); + } else { + buffer.writeByte(encodedName.length); + buffer.writeBytes(encodedName); + } + return buffer; + } + public int getSize() { if (id != null) { // kind, 1 byte + length, 1 byte + id, 4 bytes return 6; } else { - // kind, 1 byte + length, 1 byte + name.length() - return 2 + name.length(); + // kind, 1 byte + length, 1 byte + encoded name bytes + return 2 + encodedName.length; } } } diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/message/Message.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/message/Message.java index aa5389f261..d7249df158 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/message/Message.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/message/Message.java @@ -21,6 +21,7 @@ import javax.annotation.Nullable; import java.math.BigInteger; +import java.nio.charset.StandardCharsets; import java.util.Collections; import java.util.HashMap; import java.util.List; @@ -46,7 +47,7 @@ public static Message of(String payload) { } public static Message of(String payload, Map userHeaders) { - final byte[] payloadBytes = payload.getBytes(); + final byte[] payloadBytes = payload.getBytes(StandardCharsets.UTF_8); final long userHeadersLength = getUserHeadersSize(userHeaders); final MessageHeader msgHeader = new MessageHeader( BigInteger.ZERO, diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/message/Partitioning.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/message/Partitioning.java index 51f071fc3b..93e9792392 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/message/Partitioning.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/message/Partitioning.java @@ -23,14 +23,40 @@ import org.apache.iggy.exception.IggyInvalidArgumentException; import java.nio.ByteBuffer; +import java.nio.charset.StandardCharsets; public record Partitioning(PartitioningKind kind, byte[] value) { + + private static final int PARTITION_ID_LENGTH = 4; + + /** Server-side cap on a messages key, in encoded bytes, matching its u8 length prefix. */ + private static final int MAX_MESSAGES_KEY_LENGTH = 255; + + public Partitioning { + if (kind == null || value == null) { + throw new IggyInvalidArgumentException("Partitioning kind and value cannot be null"); + } + boolean valid = + switch (kind) { + case Balanced -> value.length == 0; + case PartitionId -> value.length == PARTITION_ID_LENGTH; + case MessagesKey -> value.length >= 1 && value.length <= MAX_MESSAGES_KEY_LENGTH; + }; + if (!valid) { + throw new IggyInvalidArgumentException( + kind + " partitioning value must be " + expectedLength(kind) + " bytes, got " + value.length); + } + } + public static Partitioning balanced() { return new Partitioning(PartitioningKind.Balanced, new byte[] {}); } public static Partitioning partitionId(Long id) { - ByteBuffer buffer = ByteBuffer.allocate(4); + if (id == null) { + throw new IggyInvalidArgumentException("Partition id cannot be null"); + } + ByteBuffer buffer = ByteBuffer.allocate(PARTITION_ID_LENGTH); buffer.putInt(id.intValue()); byte[] partitionId = buffer.array(); ArrayUtils.reverse(partitionId); @@ -38,14 +64,22 @@ public static Partitioning partitionId(Long id) { } public static Partitioning messagesKey(String key) { - if (key == null || key.isBlank() || key.length() > 255) { - throw new IggyInvalidArgumentException("Key must be non-empty and less than 255 characters long"); + if (key == null || key.isBlank()) { + throw new IggyInvalidArgumentException("Key must be non-empty"); } - return new Partitioning(PartitioningKind.MessagesKey, key.getBytes()); + return new Partitioning(PartitioningKind.MessagesKey, key.getBytes(StandardCharsets.UTF_8)); } public int getSize() { // kind, 1 byte + length, 1 byte + value.length() return 2 + value.length; } + + private static String expectedLength(PartitioningKind kind) { + return switch (kind) { + case Balanced -> "0"; + case PartitionId -> String.valueOf(PARTITION_ID_LENGTH); + case MessagesKey -> "1.." + MAX_MESSAGES_KEY_LENGTH; + }; + } } diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/serde/BytesSerializer.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/serde/BytesSerializer.java index 7a20e27fa1..51dcaed938 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/serde/BytesSerializer.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/serde/BytesSerializer.java @@ -61,6 +61,9 @@ public final class BytesSerializer { */ private static final int MAX_HEADER_FIELD_LENGTH = 255; + /** Bound on a u8-length-prefixed wire string, in encoded bytes. */ + private static final int MAX_U8_STRING_LENGTH = 255; + /** The timestamp delta is a u32 microsecond offset from the batch origin timestamp. */ private static final BigInteger MAX_TIMESTAMP_DELTA_MICROS = BigInteger.valueOf(0xFFFF_FFFFL); @@ -77,21 +80,7 @@ public static ByteBuf toBytes(Consumer consumer) { } public static ByteBuf toBytes(Identifier identifier) { - if (identifier.getKind() == 1) { - ByteBuf buffer = Unpooled.buffer(6); - buffer.writeByte(1); - buffer.writeByte(4); - buffer.writeIntLE(identifier.getId().intValue()); - return buffer; - } else if (identifier.getKind() == 2) { - ByteBuf buffer = Unpooled.buffer(2 + identifier.getName().length()); - buffer.writeByte(2); - buffer.writeByte(identifier.getName().length()); - buffer.writeBytes(identifier.getName().getBytes()); - return buffer; - } else { - throw new IggyInvalidArgumentException("Unknown identifier kind: " + identifier.getKind()); - } + return identifier.toBytes(); } public static ByteBuf toBytes(Partitioning partitioning) { @@ -207,10 +196,22 @@ public static ByteBuf toBytes(TopicPermissions permissions) { return buffer; } - public static ByteBuf toBytes(String value) { - int bufferLength = 1 + value.length(); - ByteBuf buffer = Unpooled.buffer(bufferLength); + /** A u8-length-prefixed wire string; {@code field} names it in the error when it does not fit. */ + public static ByteBuf toBytes(String value, String field) { + return toBytes(value, field, 1, MAX_U8_STRING_LENGTH); + } + + /** + * A u8-length-prefixed wire string bounded to {@code [minLength, maxLength]} UTF-8 bytes, for + * fields the server holds to a tighter range than the prefix allows. + */ + public static ByteBuf toBytes(String value, String field, int minLength, int maxLength) { byte[] stringBytes = value.getBytes(StandardCharsets.UTF_8); + if (stringBytes.length < minLength || stringBytes.length > maxLength) { + throw new IggyInvalidArgumentException("Invalid " + field + " length: " + stringBytes.length + + " bytes when UTF-8 encoded, must be between " + minLength + " and " + maxLength); + } + ByteBuf buffer = Unpooled.buffer(1 + stringBytes.length); buffer.writeByte(stringBytes.length); buffer.writeBytes(stringBytes); return buffer; diff --git a/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/vsr/VsrLoginCodecTest.java b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/vsr/VsrLoginCodecTest.java new file mode 100644 index 0000000000..085da287d0 --- /dev/null +++ b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/vsr/VsrLoginCodecTest.java @@ -0,0 +1,159 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.iggy.client.async.tcp.vsr; + +import io.netty.buffer.AbstractByteBufAllocator; +import io.netty.buffer.ByteBuf; +import io.netty.buffer.Unpooled; +import io.netty.buffer.UnpooledByteBufAllocator; +import org.apache.iggy.exception.IggyInvalidArgumentException; +import org.apache.iggy.serde.BytesSerializer; +import org.junit.jupiter.api.Test; +import org.junit.jupiter.params.ParameterizedTest; +import org.junit.jupiter.params.provider.EmptySource; +import org.junit.jupiter.params.provider.NullSource; + +import java.nio.charset.StandardCharsets; + +import static org.assertj.core.api.Assertions.assertThat; +import static org.assertj.core.api.Assertions.assertThatThrownBy; + +class VsrLoginCodecTest { + + @Test + void sdkVersionFieldEncodesVersionAsUtf8() { + assertThat(VsrLoginCodec.sdkVersionField("0.9.0-SNAPSHOT")) + .isEqualTo("0.9.0-SNAPSHOT".getBytes(StandardCharsets.UTF_8)); + } + + @ParameterizedTest + @NullSource + @EmptySource + void sdkVersionFieldFallsBackToUnknownWhenVersionIsMissing(String version) { + assertThat(VsrLoginCodec.sdkVersionField(version)).isEqualTo("unknown".getBytes(StandardCharsets.UTF_8)); + } + + @Test + void sdkVersionFieldTruncatesOnCodePointBoundaryWithinU8Prefix() { + String version = "é".repeat(300); + + byte[] field = VsrLoginCodec.sdkVersionField(version); + + assertThat(field).hasSize(254); + assertThat(new String(field, StandardCharsets.UTF_8)).isEqualTo("é".repeat(127)); + } + + @Test + void sdkVersionFieldKeepsSurrogatePairThatEndsExactlyAtU8Prefix() { + String version = "a".repeat(251) + "\uD83D\uDE00"; + assertThat(version.getBytes(StandardCharsets.UTF_8)).hasSize(255); + + byte[] field = VsrLoginCodec.sdkVersionField(version); + + assertThat(field).hasSize(255); + assertThat(new String(field, StandardCharsets.UTF_8)).isEqualTo(version); + } + + @Test + void sdkVersionFieldDropsWholeSurrogatePairThatWouldStraddleU8Prefix() { + String version = "a".repeat(252) + "\uD83D\uDE00"; + + byte[] field = VsrLoginCodec.sdkVersionField(version); + + assertThat(field).hasSize(252); + assertThat(new String(field, StandardCharsets.UTF_8)).isEqualTo("a".repeat(252)); + } + + @Test + void rewriteUserLoginPrefixesCredentialsWithUtf8ByteLength() { + String username = "użytkownik"; + String password = "hasło"; + ByteBuf loginPayload = BytesSerializer.toBytes(username, "username"); + loginPayload.writeBytes(BytesSerializer.toBytes(password, "password")); + + ByteBuf body = VsrLoginCodec.rewriteUserLogin(UnpooledByteBufAllocator.DEFAULT, loginPayload); + try { + assertThat(body.readIntLE()).isEqualTo(VsrLoginCodec.PROTOCOL_VERSION); + assertThat(readShortField(body)).isEqualTo(VsrLoginCodec.SDK_NAME); + assertThat(readShortField(body)).isNotEmpty(); + assertThat(readShortField(body)).isEqualTo(username); + assertThat(readShortField(body)).isEqualTo(password); + assertThat(body.readIntLE()).isZero(); + assertThat(body.isReadable()).isFalse(); + } finally { + body.release(); + loginPayload.release(); + } + } + + @Test + void rewriteUserLoginRejectsEmptyUsernameBeforeAllocating() { + ByteBuf loginPayload = Unpooled.buffer(); + loginPayload.writeByte(0); + loginPayload.writeBytes(BytesSerializer.toBytes("secret", "password")); + CountingAllocator alloc = new CountingAllocator(); + + assertThatThrownBy(() -> VsrLoginCodec.rewriteUserLogin(alloc, loginPayload)) + .isInstanceOf(IggyInvalidArgumentException.class) + .hasMessageContaining("username"); + assertThat(alloc.allocations).isZero(); + loginPayload.release(); + } + + @Test + void rewritePatLoginRejectsEmptyTokenBeforeAllocating() { + ByteBuf loginPayload = Unpooled.buffer(); + loginPayload.writeByte(0); + CountingAllocator alloc = new CountingAllocator(); + + assertThatThrownBy(() -> VsrLoginCodec.rewritePatLogin(alloc, loginPayload)) + .isInstanceOf(IggyInvalidArgumentException.class) + .hasMessageContaining("token"); + assertThat(alloc.allocations).isZero(); + loginPayload.release(); + } + + private static String readShortField(ByteBuf buffer) { + byte[] bytes = new byte[buffer.readUnsignedByte()]; + buffer.readBytes(bytes); + return new String(bytes, StandardCharsets.UTF_8); + } + + private static final class CountingAllocator extends AbstractByteBufAllocator { + private int allocations; + + @Override + protected ByteBuf newHeapBuffer(int initialCapacity, int maxCapacity) { + allocations++; + return Unpooled.buffer(initialCapacity, maxCapacity); + } + + @Override + protected ByteBuf newDirectBuffer(int initialCapacity, int maxCapacity) { + allocations++; + return Unpooled.directBuffer(initialCapacity, maxCapacity); + } + + @Override + public boolean isDirectBufferPooled() { + return false; + } + } +} diff --git a/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/blocking/tcp/BytesSerializerTest.java b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/blocking/tcp/BytesSerializerTest.java index 091ffc189c..346b63bc34 100644 --- a/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/blocking/tcp/BytesSerializerTest.java +++ b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/blocking/tcp/BytesSerializerTest.java @@ -20,6 +20,7 @@ package org.apache.iggy.client.blocking.tcp; import io.netty.buffer.ByteBuf; +import io.netty.buffer.ByteBufUtil; import io.netty.buffer.Unpooled; import org.apache.iggy.consumergroup.Consumer; import org.apache.iggy.exception.IggyInvalidArgumentException; @@ -36,6 +37,8 @@ import org.apache.iggy.user.TopicPermissions; import org.junit.jupiter.api.Nested; import org.junit.jupiter.api.Test; +import org.junit.jupiter.params.ParameterizedTest; +import org.junit.jupiter.params.provider.CsvSource; import java.math.BigInteger; import java.nio.charset.StandardCharsets; @@ -168,7 +171,7 @@ void shouldSerializeSimpleString() { String input = "test"; // when - ByteBuf result = BytesSerializer.toBytes(input); + ByteBuf result = BytesSerializer.toBytes(input, "name"); // then assertThat(result.readByte()).isEqualTo((byte) 4); // length @@ -178,16 +181,15 @@ void shouldSerializeSimpleString() { } @Test - void shouldSerializeEmptyString() { + void shouldRejectEmptyString() { // given String input = ""; - // when - ByteBuf result = BytesSerializer.toBytes(input); - - // then - assertThat(result.readByte()).isEqualTo((byte) 0); // length = 0 - assertThat(result.readableBytes()).isEqualTo(0); + // when / then + assertThatThrownBy(() -> BytesSerializer.toBytes(input, "name")) + .isInstanceOf(IggyInvalidArgumentException.class) + .hasMessageContaining("name") + .hasMessageContaining("0 bytes"); } @Test @@ -196,7 +198,7 @@ void shouldSerializeUtf8Characters() { String input = "Hello世界"; // when - ByteBuf result = BytesSerializer.toBytes(input); + ByteBuf result = BytesSerializer.toBytes(input, "name"); // then byte[] expectedBytes = input.getBytes(StandardCharsets.UTF_8); @@ -205,6 +207,55 @@ void shouldSerializeUtf8Characters() { result.readBytes(stringBytes); assertThat(stringBytes).isEqualTo(expectedBytes); } + + @Test + void shouldSerializeStringOfExactly255EncodedBytes() { + // given + String input = "世".repeat(85); + + // when + ByteBuf result = BytesSerializer.toBytes(input, "name"); + + // then + assertThat(result.readUnsignedByte()).isEqualTo((short) 255); + assertThat(result.readableBytes()).isEqualTo(255); + } + + @Test + void shouldRejectStringLongerThan255EncodedBytesEvenIfUnder255Chars() { + // given + String input = "あ".repeat(86); + assertThat(input.length()).isLessThan(255); + + // when / then + assertThatThrownBy(() -> BytesSerializer.toBytes(input, "name")) + .isInstanceOf(IggyInvalidArgumentException.class) + .hasMessageContaining("258"); + } + + @Test + void shouldNameTheRejectedField() { + assertThatThrownBy(() -> BytesSerializer.toBytes("", "username")) + .isInstanceOf(IggyInvalidArgumentException.class) + .hasMessageContaining("Invalid username length"); + } + + @Test + void shouldApplyCallerBoundsToEncodedLength() { + // given: three chars, six bytes + String input = "ééé"; + + // when / then + assertThat(BytesSerializer.toBytes(input, "password", 3, 6).readUnsignedByte()) + .isEqualTo((short) 6); + assertThatThrownBy(() -> BytesSerializer.toBytes(input, "password", 3, 5)) + .isInstanceOf(IggyInvalidArgumentException.class) + .hasMessageContaining("password") + .hasMessageContaining("between 3 and 5"); + assertThatThrownBy(() -> BytesSerializer.toBytes("ab", "password", 3, 5)) + .isInstanceOf(IggyInvalidArgumentException.class) + .hasMessageContaining("2 bytes"); + } } @Nested @@ -237,7 +288,45 @@ void shouldSerializeStringIdentifier() { assertThat(result.readByte()).isEqualTo((byte) 11); // length = "test-stream".length() byte[] nameBytes = new byte[11]; result.readBytes(nameBytes); - assertThat(new String(nameBytes)).isEqualTo("test-stream"); + assertThat(new String(nameBytes, StandardCharsets.UTF_8)).isEqualTo("test-stream"); + } + + @Test + void shouldSerializeUtf8StringIdentifierWithByteLength() { + // given + String name = "strumień-世界"; + byte[] expectedBytes = name.getBytes(StandardCharsets.UTF_8); + var identifier = StreamId.of(name); + + // when + ByteBuf result = BytesSerializer.toBytes(identifier); + + // then + assertThat(expectedBytes.length).isGreaterThan(name.length()); + assertThat(result.readableBytes()).isEqualTo(identifier.getSize()); + assertThat(result.readByte()).isEqualTo((byte) 2); // kind = 2 (string) + assertThat(result.readByte()).isEqualTo((byte) expectedBytes.length); + byte[] nameBytes = new byte[expectedBytes.length]; + result.readBytes(nameBytes); + assertThat(nameBytes).isEqualTo(expectedBytes); + assertThat(result.readableBytes()).isEqualTo(0); + } + + @ParameterizedTest + @CsvSource({ + "café, 02 05 63 61 66 C3 A9", + "naïve-café, 02 0C 6E 61 C3 AF 76 65 2D 63 61 66 C3 A9", + "日本語, 02 09 E6 97 A5 E6 9C AC E8 AA 9E", + }) + void shouldMatchExpectedWireLayoutForNonAsciiNames(String name, String expectedHex) { + // given + var identifier = StreamId.of(name); + + // when + ByteBuf result = BytesSerializer.toBytes(identifier); + + // then + assertThat(ByteBufUtil.hexDump(result).toUpperCase()).isEqualTo(expectedHex.replace(" ", "")); } } diff --git a/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/blocking/tcp/MessagesTcpClientTest.java b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/blocking/tcp/MessagesTcpClientTest.java index c644b50ac7..6fc24de500 100644 --- a/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/blocking/tcp/MessagesTcpClientTest.java +++ b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/blocking/tcp/MessagesTcpClientTest.java @@ -21,11 +21,19 @@ import org.apache.iggy.client.blocking.IggyBaseClient; import org.apache.iggy.client.blocking.MessagesClientBaseTest; +import org.apache.iggy.consumergroup.Consumer; +import org.apache.iggy.identifier.StreamId; +import org.apache.iggy.identifier.TopicId; import org.apache.iggy.message.Message; import org.apache.iggy.message.Partitioning; +import org.apache.iggy.message.PollingStrategy; +import org.apache.iggy.topic.CompressionAlgorithm; import org.junit.jupiter.api.Test; +import java.math.BigInteger; +import java.nio.charset.StandardCharsets; import java.util.List; +import java.util.Optional; import static org.apache.iggy.TestConstants.STREAM_NAME; import static org.apache.iggy.TestConstants.TOPIC_NAME; @@ -59,4 +67,31 @@ void shouldRouteSameMessageKeyToSamePartition() { assertThat(secondResponse.confirmations().get(0).partitionId()) .isEqualTo(firstResponse.confirmations().get(0).partitionId()); } + + /* + * TCP only: the HTTP client does not percent-encode path segments yet, so + * non-ASCII stream and topic names cannot be addressed over HTTP. + */ + @Test + void shouldRoundTripNonAsciiNamesKeyAndPayload() { + // given + var streamId = StreamId.of("strumień-世界"); + var topicId = TopicId.of("тема-日本語"); + var stream = client.streams().createStream(streamId.getName()); + trackStream(stream.id()); + client.topics() + .createTopic( + streamId, 1L, CompressionAlgorithm.None, BigInteger.ZERO, BigInteger.ZERO, topicId.getName()); + String text = "wiadomość 世界 😀"; + + // when + messagesClient.sendMessages(streamId, topicId, Partitioning.messagesKey("klucz-键"), List.of(Message.of(text))); + var polledMessages = messagesClient.pollMessages( + streamId, topicId, Optional.of(0L), Consumer.of(0L), PollingStrategy.first(), 10L, false); + + // then + assertThat(polledMessages.messages()).hasSize(1); + assertThat(new String(polledMessages.messages().get(0).payload(), StandardCharsets.UTF_8)) + .isEqualTo(text); + } } diff --git a/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/blocking/tcp/StreamTcpClientTest.java b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/blocking/tcp/StreamTcpClientTest.java index a3954f189a..007efc4ec2 100644 --- a/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/blocking/tcp/StreamTcpClientTest.java +++ b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/blocking/tcp/StreamTcpClientTest.java @@ -21,6 +21,10 @@ import org.apache.iggy.client.blocking.IggyBaseClient; import org.apache.iggy.client.blocking.StreamClientBaseTest; +import org.apache.iggy.identifier.StreamId; +import org.junit.jupiter.api.Test; + +import static org.assertj.core.api.Assertions.assertThat; class StreamTcpClientTest extends StreamClientBaseTest { @@ -28,4 +32,24 @@ class StreamTcpClientTest extends StreamClientBaseTest { protected IggyBaseClient getClient() { return TcpClientFactory.create(serverHost(), serverTcpPort()); } + + /* + * TCP only: the HTTP client does not percent-encode path segments yet, so + * a non-ASCII name cannot be looked up over HTTP. + */ + @Test + void shouldCreateAndFetchStreamWithNonAsciiName() { + // given + var name = "strumień-世界"; + + // when + var streamDetails = client.streams().createStream(name); + trackStream(streamDetails.id()); + var streamByName = client.streams().getStream(StreamId.of(name)); + + // then + assertThat(streamDetails.name()).isEqualTo(name); + assertThat(streamByName).isPresent(); + assertThat(streamByName.get().id()).isEqualTo(streamDetails.id()); + } } diff --git a/foreign/java/java-sdk/src/test/java/org/apache/iggy/identifier/IdentifierTest.java b/foreign/java/java-sdk/src/test/java/org/apache/iggy/identifier/IdentifierTest.java index cc7bee7a99..6a73dcf917 100644 --- a/foreign/java/java-sdk/src/test/java/org/apache/iggy/identifier/IdentifierTest.java +++ b/foreign/java/java-sdk/src/test/java/org/apache/iggy/identifier/IdentifierTest.java @@ -19,10 +19,14 @@ package org.apache.iggy.identifier; +import io.netty.buffer.ByteBuf; import org.apache.iggy.exception.IggyInvalidArgumentException; import org.jspecify.annotations.Nullable; import org.junit.jupiter.api.Test; +import java.nio.charset.StandardCharsets; + +import static org.assertj.core.api.Assertions.assertThat; import static org.assertj.core.api.Assertions.assertThatThrownBy; public class IdentifierTest { @@ -31,6 +35,51 @@ void constructorThrowsIggyInvalidArgumentExceptionWhenBothNameAndIdAreProvided() assertThatThrownBy(() -> new FakeIdentifier("foo", 123L)).isInstanceOf(IggyInvalidArgumentException.class); } + @Test + void getSizeCountsEncodedBytesOfName() { + assertThat(new FakeIdentifier("世界", null).getSize()).isEqualTo(2 + 6); + } + + @Test + void toBytesEncodesNameIdentifierAsKindLengthAndUtf8Bytes() { + byte[] expected = "世界".getBytes(StandardCharsets.UTF_8); + + ByteBuf result = new FakeIdentifier("世界", null).toBytes(); + + assertThat(result.readableBytes()).isEqualTo(2 + expected.length); + assertThat(result.readByte()).isEqualTo((byte) 2); + assertThat(result.readUnsignedByte()).isEqualTo((short) expected.length); + byte[] name = new byte[expected.length]; + result.readBytes(name); + assertThat(name).isEqualTo(expected); + } + + @Test + void toBytesEncodesNumericIdentifierAsKindLengthAndLittleEndianId() { + ByteBuf result = new FakeIdentifier(null, 7L).toBytes(); + + assertThat(result.readableBytes()).isEqualTo(6); + assertThat(result.readByte()).isEqualTo((byte) 1); + assertThat(result.readByte()).isEqualTo((byte) 4); + assertThat(result.readIntLE()).isEqualTo(7); + } + + @Test + void constructorAcceptsNameOfExactly255EncodedBytes() { + String name = "世".repeat(85); + assertThat(name.getBytes(StandardCharsets.UTF_8)).hasSize(255); + assertThat(new FakeIdentifier(name, null).getName()).isEqualTo(name); + } + + @Test + void constructorThrowsWhenNameExceeds255EncodedBytesEvenIfUnder255Chars() { + String name = "あ".repeat(200); + assertThat(name.length()).isLessThan(255); + assertThatThrownBy(() -> new FakeIdentifier(name, null)) + .isInstanceOf(IggyInvalidArgumentException.class) + .hasMessageContaining("600"); + } + static class FakeIdentifier extends Identifier { protected FakeIdentifier(@Nullable String name, @Nullable Long id) { super(name, id); diff --git a/foreign/java/java-sdk/src/test/java/org/apache/iggy/message/MessageTest.java b/foreign/java/java-sdk/src/test/java/org/apache/iggy/message/MessageTest.java index 48cb5a8be4..927dcc3ba8 100644 --- a/foreign/java/java-sdk/src/test/java/org/apache/iggy/message/MessageTest.java +++ b/foreign/java/java-sdk/src/test/java/org/apache/iggy/message/MessageTest.java @@ -22,6 +22,7 @@ import org.junit.jupiter.api.Test; import java.math.BigInteger; +import java.nio.charset.StandardCharsets; import java.util.List; import java.util.Map; @@ -72,6 +73,14 @@ void ofCreatesMessageWhenGivenMessageHeaderPayloadAndNullUserHeaders() { assertThat(message.userHeaders().size()).isEqualTo(0); } + @Test + void ofEncodesStringPayloadAsUtf8() { + var message = Message.of("世界"); + + assertThat(message.payload()).isEqualTo("世界".getBytes(StandardCharsets.UTF_8)); + assertThat(message.header().payloadLength()).isEqualTo(6L); + } + @Test void ofCreatesExpectedMessageWhenGivenPayloadOnly() { var message = Message.of("foo"); diff --git a/foreign/java/java-sdk/src/test/java/org/apache/iggy/message/PartitioningTest.java b/foreign/java/java-sdk/src/test/java/org/apache/iggy/message/PartitioningTest.java index 3fd35ac324..260fd89d75 100644 --- a/foreign/java/java-sdk/src/test/java/org/apache/iggy/message/PartitioningTest.java +++ b/foreign/java/java-sdk/src/test/java/org/apache/iggy/message/PartitioningTest.java @@ -25,6 +25,8 @@ import org.junit.jupiter.params.provider.NullSource; import org.junit.jupiter.params.provider.ValueSource; +import java.nio.charset.StandardCharsets; + import static org.assertj.core.api.Assertions.assertThat; import static org.assertj.core.api.Assertions.assertThatThrownBy; @@ -59,6 +61,64 @@ void messagesKeyThrowsIggyInvalidArgumentExceptionWhenGivenValueThatIsTooLong() assertThatThrownBy(() -> Partitioning.messagesKey(id)).isInstanceOf(IggyInvalidArgumentException.class); } + @Test + void messagesKeyThrowsIggyInvalidArgumentExceptionWhenEncodedValueExceeds255BytesEvenIfUnder255Chars() { + var id = "あ".repeat(100); + assertThat(id.length()).isLessThan(255); + assertThatThrownBy(() -> Partitioning.messagesKey(id)) + .isInstanceOf(IggyInvalidArgumentException.class) + .hasMessageContaining("300"); + } + + @Test + void constructorThrowsIggyInvalidArgumentExceptionWhenValueExceeds255Bytes() { + assertThatThrownBy(() -> new Partitioning(PartitioningKind.MessagesKey, new byte[256])) + .isInstanceOf(IggyInvalidArgumentException.class) + .hasMessageContaining("256"); + } + + @Test + void constructorThrowsIggyInvalidArgumentExceptionWhenMessagesKeyValueIsEmpty() { + assertThatThrownBy(() -> new Partitioning(PartitioningKind.MessagesKey, new byte[0])) + .isInstanceOf(IggyInvalidArgumentException.class) + .hasMessageContaining("1..255"); + } + + @Test + void constructorThrowsIggyInvalidArgumentExceptionWhenBalancedValueIsNotEmpty() { + assertThatThrownBy(() -> new Partitioning(PartitioningKind.Balanced, new byte[] {1})) + .isInstanceOf(IggyInvalidArgumentException.class) + .hasMessageContaining("Balanced"); + } + + @ParameterizedTest + @ValueSource(ints = {0, 3, 5}) + void constructorThrowsIggyInvalidArgumentExceptionWhenPartitionIdValueIsNotFourBytes(int length) { + assertThatThrownBy(() -> new Partitioning(PartitioningKind.PartitionId, new byte[length])) + .isInstanceOf(IggyInvalidArgumentException.class) + .hasMessageContaining("must be 4 bytes"); + } + + @Test + void constructorThrowsIggyInvalidArgumentExceptionWhenKindOrValueIsNull() { + assertThatThrownBy(() -> new Partitioning(null, new byte[0])).isInstanceOf(IggyInvalidArgumentException.class); + assertThatThrownBy(() -> new Partitioning(PartitioningKind.MessagesKey, null)) + .isInstanceOf(IggyInvalidArgumentException.class); + } + + @Test + void partitionIdThrowsIggyInvalidArgumentExceptionWhenIdIsNull() { + assertThatThrownBy(() -> Partitioning.partitionId(null)).isInstanceOf(IggyInvalidArgumentException.class); + } + + @Test + void messagesKeyEncodesValueAsUtf8() { + var result = Partitioning.messagesKey("世界"); + + assertThat(result.value()).isEqualTo("世界".getBytes(StandardCharsets.UTF_8)); + assertThat(result.getSize()).isEqualTo(2 + 6); + } + @Test void messagesKeyReturnsPartitioningWithMessagesKeyKindAndExpectedValueWhenGivenValidArguments() { var id = "the-key"; From ab734f17f849497d828667b5fba51820c3c3b531 Mon Sep 17 00:00:00 2001 From: Ryan Huang Date: Mon, 7 Sep 2026 22:57:18 +0800 Subject: [PATCH 080/182] fix(connectors): validate connector key before it becomes a path (#4083) Closes #4058 --- core/connectors/runtime/README.md | 15 ++ core/connectors/runtime/src/api/error.rs | 15 +- core/connectors/runtime/src/api/key.rs | 42 ++++ core/connectors/runtime/src/api/mod.rs | 1 + core/connectors/runtime/src/api/sink.rs | 44 +++-- core/connectors/runtime/src/api/source.rs | 44 +++-- .../runtime/src/configs/connectors.rs | 182 ++++++++++++++++-- .../src/configs/connectors/http_provider.rs | 12 +- .../src/configs/connectors/local_provider.rs | 50 ++++- core/connectors/runtime/src/error.rs | 6 + core/connectors/runtime/src/main.rs | 30 ++- .../tests/connectors/api/endpoints.rs | 104 ++++++++++ .../tests/connectors/api/key_validation.toml | 31 +++ 13 files changed, 505 insertions(+), 71 deletions(-) create mode 100644 core/connectors/runtime/src/api/key.rs create mode 100644 core/integration/tests/connectors/api/key_validation.toml diff --git a/core/connectors/runtime/README.md b/core/connectors/runtime/README.md index 23803704d0..ec824abda6 100644 --- a/core/connectors/runtime/README.md +++ b/core/connectors/runtime/README.md @@ -326,6 +326,21 @@ Currently, it does expose the following endpoints: - `POST /sources/{key}/restart`: stop the source and start it again from its highest stored configuration version, which on the local provider is not necessarily the active one ([#3848](https://github.com/apache/iggy/issues/3848)). - `GET /sources/{key}/transforms`: source transforms to be applied to the fields. +`{key}` is the connector key: at most 128 bytes of ASCII letters, digits, `-`, +`_` and `.`, starting with a letter or digit. A decoded segment outside that +rule, such as `..%2F..%2Fpwned`, is answered with `400 Bad Request` and the +error code `invalid_connector_key` before the request reaches the configuration +provider, because on the local provider the key becomes part of a filename under +`config_dir`, and on the HTTP provider part of a URL. An unencoded `/` splits +the path and matches no route, so it is a `404`. + +Keys loaded from configuration files or the HTTP provider are not rejected, so +existing deployments keep starting, but a key outside the rule is logged at +startup and cannot be addressed through the API. Keys are case-sensitive to the +runtime while the local provider maps them to filenames, so on a +case-insensitive filesystem two keys that differ only by case share one file; +prefer lowercase. + ## Telemetry The connector runtime supports OpenTelemetry for logs and traces. To enable telemetry, add the following configuration: diff --git a/core/connectors/runtime/src/api/error.rs b/core/connectors/runtime/src/api/error.rs index e22070c172..77125bad85 100644 --- a/core/connectors/runtime/src/api/error.rs +++ b/core/connectors/runtime/src/api/error.rs @@ -16,7 +16,7 @@ // under the License. use crate::error::RuntimeError; -use axum::{Json, http::StatusCode, response::IntoResponse}; +use axum::{Json, extract::rejection::PathRejection, http::StatusCode, response::IntoResponse}; use serde::Serialize; use thiserror::Error; use tracing::error; @@ -27,6 +27,8 @@ pub enum ApiError { Error(#[from] RuntimeError), #[error(transparent)] JsonError(#[from] serde_json::Error), + #[error(transparent)] + PathRejection(#[from] PathRejection), } #[derive(Debug, Serialize)] @@ -43,6 +45,7 @@ impl IntoResponse for ApiError { let status_code = match error { RuntimeError::MissingIggyCredentials => StatusCode::BAD_REQUEST, RuntimeError::InvalidConfiguration(_) => StatusCode::BAD_REQUEST, + RuntimeError::InvalidConnectorKey(_) => StatusCode::BAD_REQUEST, RuntimeError::CannotConvertConfiguration => StatusCode::BAD_REQUEST, RuntimeError::SinkNotFound(_) => StatusCode::NOT_FOUND, RuntimeError::SourceNotFound(_) => StatusCode::NOT_FOUND, @@ -66,6 +69,16 @@ impl IntoResponse for ApiError { }), ) } + ApiError::PathRejection(rejection) => { + error!("There was a path error: {rejection}"); + ( + rejection.status(), + Json(ErrorResponse { + code: "invalid_path".to_owned(), + reason: rejection.body_text(), + }), + ) + } } .into_response() } diff --git a/core/connectors/runtime/src/api/key.rs b/core/connectors/runtime/src/api/key.rs new file mode 100644 index 0000000000..a6438c2a7d --- /dev/null +++ b/core/connectors/runtime/src/api/key.rs @@ -0,0 +1,42 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use super::error::ApiError; +use crate::configs::connectors::ConnectorKey; +use axum::extract::{FromRequestParts, Path}; +use axum::http::request::Parts; +use serde::Deserialize; + +/// The `{key}` route segment parsed as a `ConnectorKey`. Deserializing +/// `Path` directly would answer a bad key with axum's plain-text +/// rejection; going through `ApiError` keeps the `{code, reason}` envelope the +/// rest of the API returns. +pub struct KeyPath(pub ConnectorKey); + +#[derive(Deserialize)] +struct KeyParam { + key: String, +} + +impl FromRequestParts for KeyPath { + type Rejection = ApiError; + + async fn from_request_parts(parts: &mut Parts, state: &S) -> Result { + let Path(KeyParam { key }) = Path::from_request_parts(parts, state).await?; + Ok(Self(ConnectorKey::try_from(key)?)) + } +} diff --git a/core/connectors/runtime/src/api/mod.rs b/core/connectors/runtime/src/api/mod.rs index 62950fa6e6..5669ede295 100644 --- a/core/connectors/runtime/src/api/mod.rs +++ b/core/connectors/runtime/src/api/mod.rs @@ -34,6 +34,7 @@ use tracing::{error, info, warn}; mod auth; pub mod config; mod error; +mod key; mod models; mod sink; mod source; diff --git a/core/connectors/runtime/src/api/sink.rs b/core/connectors/runtime/src/api/sink.rs index c70f85a57a..d99e89cf1d 100644 --- a/core/connectors/runtime/src/api/sink.rs +++ b/core/connectors/runtime/src/api/sink.rs @@ -18,6 +18,7 @@ use super::{ config::map_connector_config, error::ApiError, + key::KeyPath, models::{SinkDetailsResponse, SinkInfoResponse, TransformResponse}, }; use crate::api::models::SinkConfigResponse; @@ -69,10 +70,10 @@ async fn get_sinks( async fn get_sink( State(context): State>, - Path(key): Path, + KeyPath(key): KeyPath, ) -> Result, ApiError> { let Some(sink) = context.sinks.get(&key).await else { - return Err(ApiError::Error(RuntimeError::SinkNotFound(key))); + return Err(ApiError::Error(RuntimeError::SinkNotFound(key.into()))); }; let sink = sink.lock().await; Ok(Json(SinkDetailsResponse { @@ -83,11 +84,11 @@ async fn get_sink( async fn get_sink_plugin_config( State(context): State>, - Path(key): Path, + KeyPath(key): KeyPath, Query(query): Query, ) -> Result { let Some(sink) = context.sinks.get(&key).await else { - return Err(ApiError::Error(RuntimeError::SinkNotFound(key))); + return Err(ApiError::Error(RuntimeError::SinkNotFound(key.into()))); }; let sink = sink.lock().await; let Some(config) = sink.config.plugin_config.as_ref() else { @@ -110,10 +111,10 @@ struct GetSinkConfig { async fn get_sink_transforms( State(context): State>, - Path(key): Path, + KeyPath(key): KeyPath, ) -> Result>, ApiError> { let Some(sink) = context.sinks.get(&key).await else { - return Err(ApiError::Error(RuntimeError::SinkNotFound(key))); + return Err(ApiError::Error(RuntimeError::SinkNotFound(key.into()))); }; let sink = sink.lock().await; let Some(transforms) = sink.config.transforms.as_ref() else { @@ -134,13 +135,13 @@ async fn get_sink_transforms( async fn get_sink_configs( State(context): State>, - Path((key,)): Path<(String,)>, + KeyPath(key): KeyPath, ) -> Result>, ApiError> { let active_config = context .sinks .get_config(&key) .await - .ok_or(ApiError::Error(RuntimeError::SinkNotFound(key.clone())))?; + .ok_or_else(|| ApiError::Error(RuntimeError::SinkNotFound(key.to_string())))?; let configs = context.config_provider.get_sink_configs(&key).await?; let configs = configs .into_iter() @@ -154,12 +155,12 @@ async fn get_sink_configs( async fn create_sink_config( State(context): State>, - Path((key,)): Path<(String,)>, + KeyPath(key): KeyPath, Json(config): Json, ) -> Result, ApiError> { let created_config = context .config_provider - .create_sink_config(&key, config.clone()) + .create_sink_config(&key, config) .await?; Ok(Json(SinkConfigResponse { @@ -168,15 +169,21 @@ async fn create_sink_config( })) } +#[derive(Debug, Deserialize)] +struct ConfigVersion { + version: u64, +} + async fn get_sink_config( State(context): State>, - Path((key, version)): Path<(String, u64)>, + KeyPath(key): KeyPath, + Path(ConfigVersion { version }): Path, ) -> Result, ApiError> { let active_config = context .sinks .get_config(&key) .await - .ok_or(ApiError::Error(RuntimeError::SinkNotFound(key.clone())))?; + .ok_or_else(|| ApiError::Error(RuntimeError::SinkNotFound(key.to_string())))?; let config = context .config_provider @@ -192,20 +199,21 @@ async fn get_sink_config( })) } None => Err(ApiError::Error(RuntimeError::SinkConfigNotFound( - key, version, + key.into(), + version, ))), } } async fn get_sink_active_config( State(context): State>, - Path((key,)): Path<(String,)>, + KeyPath(key): KeyPath, ) -> Result, ApiError> { let config = context .sinks .get_config(&key) .await - .ok_or(ApiError::Error(RuntimeError::SinkNotFound(key)))?; + .ok_or(ApiError::Error(RuntimeError::SinkNotFound(key.into())))?; Ok(Json(SinkConfigResponse { config, active: true, @@ -219,7 +227,7 @@ struct UpdateSinkActiveConfig { async fn update_sink_active_config( State(context): State>, - Path((key,)): Path<(String,)>, + KeyPath(key): KeyPath, Json(update): Json, ) -> Result { context @@ -236,7 +244,7 @@ struct DeleteSinkConfig { async fn delete_sink_config( State(context): State>, - Path((key,)): Path<(String,)>, + KeyPath(key): KeyPath, Query(query): Query, ) -> Result { context @@ -248,7 +256,7 @@ async fn delete_sink_config( async fn restart_sink( State(context): State>, - Path(key): Path, + KeyPath(key): KeyPath, ) -> Result { context .sinks diff --git a/core/connectors/runtime/src/api/source.rs b/core/connectors/runtime/src/api/source.rs index a1c700edfc..480c7dffa9 100644 --- a/core/connectors/runtime/src/api/source.rs +++ b/core/connectors/runtime/src/api/source.rs @@ -18,6 +18,7 @@ use super::{ config::map_connector_config, error::ApiError, + key::KeyPath, models::{SourceDetailsResponse, SourceInfoResponse, TransformResponse}, }; use crate::api::models::SourceConfigResponse; @@ -72,10 +73,10 @@ async fn get_sources( async fn get_source( State(context): State>, - Path(key): Path, + KeyPath(key): KeyPath, ) -> Result, ApiError> { let Some(source) = context.sources.get(&key).await else { - return Err(ApiError::Error(RuntimeError::SourceNotFound(key))); + return Err(ApiError::Error(RuntimeError::SourceNotFound(key.into()))); }; let source = source.lock().await; Ok(Json(SourceDetailsResponse { @@ -86,11 +87,11 @@ async fn get_source( async fn get_source_plugin_config( State(context): State>, - Path(key): Path, + KeyPath(key): KeyPath, Query(query): Query, ) -> Result { let Some(source) = context.sources.get(&key).await else { - return Err(ApiError::Error(RuntimeError::SourceNotFound(key))); + return Err(ApiError::Error(RuntimeError::SourceNotFound(key.into()))); }; let source = source.lock().await; let Some(config) = source.config.plugin_config.as_ref() else { @@ -113,10 +114,10 @@ struct GetSourceConfig { async fn get_source_transforms( State(context): State>, - Path(key): Path, + KeyPath(key): KeyPath, ) -> Result>, ApiError> { let Some(source) = context.sources.get(&key).await else { - return Err(ApiError::Error(RuntimeError::SourceNotFound(key))); + return Err(ApiError::Error(RuntimeError::SourceNotFound(key.into()))); }; let source = source.lock().await; let Some(transforms) = source.config.transforms.as_ref() else { @@ -137,13 +138,13 @@ async fn get_source_transforms( async fn get_source_configs( State(context): State>, - Path((key,)): Path<(String,)>, + KeyPath(key): KeyPath, ) -> Result>, ApiError> { let active_config = context .sources .get_config(&key) .await - .ok_or(ApiError::Error(RuntimeError::SourceNotFound(key.clone())))?; + .ok_or_else(|| ApiError::Error(RuntimeError::SourceNotFound(key.to_string())))?; let configs = context.config_provider.get_source_configs(&key).await?; let configs = configs .into_iter() @@ -157,12 +158,12 @@ async fn get_source_configs( async fn create_source_config( State(context): State>, - Path((key,)): Path<(String,)>, + KeyPath(key): KeyPath, Json(config): Json, ) -> Result, ApiError> { let created_config = context .config_provider - .create_source_config(&key, config.clone()) + .create_source_config(&key, config) .await?; Ok(Json(SourceConfigResponse { @@ -171,15 +172,21 @@ async fn create_source_config( })) } +#[derive(Debug, Deserialize)] +struct ConfigVersion { + version: u64, +} + async fn get_source_config( State(context): State>, - Path((key, version)): Path<(String, u64)>, + KeyPath(key): KeyPath, + Path(ConfigVersion { version }): Path, ) -> Result, ApiError> { let active_config = context .sources .get_config(&key) .await - .ok_or(ApiError::Error(RuntimeError::SourceNotFound(key.clone())))?; + .ok_or_else(|| ApiError::Error(RuntimeError::SourceNotFound(key.to_string())))?; let config = context .config_provider @@ -195,20 +202,21 @@ async fn get_source_config( })) } None => Err(ApiError::Error(RuntimeError::SourceConfigNotFound( - key, version, + key.into(), + version, ))), } } async fn get_source_active_config( State(context): State>, - Path((key,)): Path<(String,)>, + KeyPath(key): KeyPath, ) -> Result, ApiError> { let config = context .sources .get_config(&key) .await - .ok_or(ApiError::Error(RuntimeError::SourceNotFound(key)))?; + .ok_or(ApiError::Error(RuntimeError::SourceNotFound(key.into())))?; Ok(Json(SourceConfigResponse { config, active: true, @@ -222,7 +230,7 @@ struct UpdateSourceActiveConfig { async fn update_source_active_config( State(context): State>, - Path((key,)): Path<(String,)>, + KeyPath(key): KeyPath, Json(update): Json, ) -> Result { context @@ -239,7 +247,7 @@ struct DeleteSourceConfig { async fn delete_source_config( State(context): State>, - Path((key,)): Path<(String,)>, + KeyPath(key): KeyPath, Query(query): Query, ) -> Result { context @@ -251,7 +259,7 @@ async fn delete_source_config( async fn restart_source( State(context): State>, - Path(key): Path, + KeyPath(key): KeyPath, ) -> Result { context .sources diff --git a/core/connectors/runtime/src/configs/connectors.rs b/core/connectors/runtime/src/configs/connectors.rs index 24647e8703..ad6a6c435e 100644 --- a/core/connectors/runtime/src/configs/connectors.rs +++ b/core/connectors/runtime/src/configs/connectors.rs @@ -30,7 +30,9 @@ use iggy_connector_sdk::transforms::TransformType; use serde::{Deserialize, Serialize}; use std::collections::HashMap; use std::fmt::Formatter; +use std::ops::Deref; use std::path::PathBuf; +use std::str::FromStr; use strum::Display; #[derive( @@ -49,6 +51,74 @@ pub enum ConfigFormat { Text, } +/// A connector key becomes part of a filename under the local provider's +/// `config_dir` and of a URL under the HTTP provider, so it must stay a single +/// path component. Requiring a leading letter or digit is what rules out `.`, +/// `..` and hidden-file names outright, instead of relying on the `sink_` / +/// `source_` filename prefix to neutralize them. +#[derive(Debug)] +pub struct ConnectorKey(String); + +impl ConnectorKey { + /// Leaves room for the `source_` prefix, the version suffix and the + /// `.toml` extension inside a 255-byte filename limit. + pub const MAX_LENGTH: usize = 128; + + pub fn as_str(&self) -> &str { + &self.0 + } + + fn is_valid(key: &str) -> bool { + key.len() <= Self::MAX_LENGTH + && key.as_bytes().split_first().is_some_and(|(first, rest)| { + first.is_ascii_alphanumeric() + && rest.iter().all(|byte| { + byte.is_ascii_alphanumeric() || matches!(*byte, b'-' | b'_' | b'.') + }) + }) + } +} + +impl TryFrom for ConnectorKey { + type Error = RuntimeError; + + fn try_from(key: String) -> Result { + if Self::is_valid(&key) { + Ok(Self(key)) + } else { + Err(RuntimeError::InvalidConnectorKey(key)) + } + } +} + +impl FromStr for ConnectorKey { + type Err = RuntimeError; + + fn from_str(key: &str) -> Result { + Self::try_from(key.to_owned()) + } +} + +impl Deref for ConnectorKey { + type Target = str; + + fn deref(&self) -> &Self::Target { + &self.0 + } +} + +impl From for String { + fn from(key: ConnectorKey) -> Self { + key.0 + } +} + +impl std::fmt::Display for ConnectorKey { + fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { + f.write_str(&self.0) + } +} + #[derive(Debug, Clone, Deserialize, Serialize)] #[serde(tag = "type", rename_all = "lowercase")] pub enum ConnectorConfig { @@ -87,17 +157,17 @@ pub struct CreateSinkConfig { } impl CreateSinkConfig { - fn to_sink_config(&self, key: &str, version: u64) -> SinkConfig { + fn into_sink_config(self, key: &ConnectorKey, version: u64) -> SinkConfig { SinkConfig { - key: key.to_owned(), + key: key.to_string(), enabled: self.enabled, version, - name: self.name.clone(), - path: self.path.clone(), - transforms: self.transforms.clone(), - streams: self.streams.clone(), + name: self.name, + path: self.path, + transforms: self.transforms, + streams: self.streams, plugin_config_format: self.plugin_config_format, - plugin_config: self.plugin_config.clone(), + plugin_config: self.plugin_config, verbose: self.verbose, benchmark: self.benchmark, } @@ -140,17 +210,17 @@ pub struct CreateSourceConfig { } impl CreateSourceConfig { - fn to_source_config(&self, key: &str, version: u64) -> SourceConfig { + fn into_source_config(self, key: &ConnectorKey, version: u64) -> SourceConfig { SourceConfig { - key: key.to_owned(), + key: key.to_string(), enabled: self.enabled, version, - name: self.name.clone(), - path: self.path.clone(), - transforms: self.transforms.clone(), - streams: self.streams.clone(), + name: self.name, + path: self.path, + transforms: self.transforms, + streams: self.streams, plugin_config_format: self.plugin_config_format, - plugin_config: self.plugin_config.clone(), + plugin_config: self.plugin_config, verbose: self.verbose, benchmark: self.benchmark, } @@ -222,16 +292,19 @@ pub struct ConnectorConfigVersions { pub sources: HashMap, } +/// Only the two `create_*` methods take a parsed key: they are where the local +/// provider turns the key into a filename, so the type carries the proof that +/// the API boundary already validated it. The other methods only compare keys. #[async_trait] pub trait ConnectorsConfigProvider: Send + Sync { async fn create_sink_config( &self, - key: &str, + key: &ConnectorKey, config: CreateSinkConfig, ) -> Result; async fn create_source_config( &self, - key: &str, + key: &ConnectorKey, config: CreateSourceConfig, ) -> Result; async fn get_active_configs(&self) -> Result; @@ -411,3 +484,80 @@ impl ConnectorsConfig { &self.sources } } + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn given_single_component_key_when_parsed_should_succeed() { + for key in ["postgres", "es-sink.v2_1", "A1", "9lives", "a.b-c_d"] { + let parsed: ConnectorKey = key + .parse() + .unwrap_or_else(|error| panic!("key {key:?} should be accepted, got: {error}")); + assert_eq!(parsed.as_str(), key); + assert_eq!(parsed.to_string(), key); + } + } + + #[test] + fn given_key_at_the_length_limit_when_parsed_should_succeed() { + let key = "k".repeat(ConnectorKey::MAX_LENGTH); + assert_eq!(key.parse::().unwrap().as_str(), key); + } + + #[test] + fn given_key_over_the_length_limit_when_parsed_should_fail() { + assert_rejected(&"k".repeat(ConnectorKey::MAX_LENGTH + 1)); + } + + #[test] + fn given_key_with_path_separator_when_parsed_should_fail() { + for key in ["../../pwned", "x/../../../tmp/pwn", "a/b", "a\\b", "/abs"] { + assert_rejected(key); + } + } + + #[test] + fn given_key_that_is_a_dot_segment_or_hidden_name_when_parsed_should_fail() { + for key in [".", "..", "..evil", ".hidden"] { + assert_rejected(key); + } + } + + #[test] + fn given_key_with_characters_outside_the_charset_when_parsed_should_fail() { + for key in [ + "", + "-leading-dash", + "_leading_underscore", + "with space", + "k\0ey", + "k\ney", + "ключ", + "a#b", + ] { + assert_rejected(key); + } + } + + #[test] + fn given_owned_key_when_converted_should_apply_the_same_rule() { + let accepted = ConnectorKey::try_from("random".to_owned()).unwrap(); + assert_eq!(accepted.as_str(), "random"); + + let rejected = ConnectorKey::try_from("../pwned".to_owned()).unwrap_err(); + assert!( + matches!(&rejected, RuntimeError::InvalidConnectorKey(key) if key == "../pwned"), + "unexpected error: {rejected}" + ); + } + + fn assert_rejected(key: &str) { + let result = key.parse::(); + assert!( + matches!(&result, Err(RuntimeError::InvalidConnectorKey(rejected)) if rejected == key), + "key {key:?} should be rejected, got: {result:?}" + ); + } +} diff --git a/core/connectors/runtime/src/configs/connectors/http_provider.rs b/core/connectors/runtime/src/configs/connectors/http_provider.rs index ea49e5e0df..8133719e9d 100644 --- a/core/connectors/runtime/src/configs/connectors/http_provider.rs +++ b/core/connectors/runtime/src/configs/connectors/http_provider.rs @@ -21,8 +21,8 @@ mod url_builder; use crate::configs::connectors::http_provider::response_extractor::ResponseExtractor; use crate::configs::connectors::http_provider::url_builder::{TemplateKeys, UrlBuilder}; use crate::configs::connectors::{ - ConnectorConfigVersions, ConnectorsConfig, ConnectorsConfigProvider, CreateSinkConfig, - CreateSourceConfig, SinkConfig, SourceConfig, + ConnectorConfigVersions, ConnectorKey, ConnectorsConfig, ConnectorsConfigProvider, + CreateSinkConfig, CreateSourceConfig, SinkConfig, SourceConfig, }; use crate::configs::runtime::{ResponseConfig, RetryConfig}; use crate::error::RuntimeError; @@ -131,11 +131,11 @@ impl HttpConnectorsConfigProvider { impl ConnectorsConfigProvider for HttpConnectorsConfigProvider { async fn create_sink_config( &self, - key: &str, + key: &ConnectorKey, config: CreateSinkConfig, ) -> Result { let mut vars = HashMap::new(); - vars.insert("key", key); + vars.insert("key", key.as_str()); let url = self.url_builder.build(TemplateKeys::CREATE_SINK, &vars); let response = self @@ -157,11 +157,11 @@ impl ConnectorsConfigProvider for HttpConnectorsConfigProvider { async fn create_source_config( &self, - key: &str, + key: &ConnectorKey, config: CreateSourceConfig, ) -> Result { let mut vars = HashMap::new(); - vars.insert("key", key); + vars.insert("key", key.as_str()); let url = self.url_builder.build(TemplateKeys::CREATE_SOURCE, &vars); let response = self diff --git a/core/connectors/runtime/src/configs/connectors/local_provider.rs b/core/connectors/runtime/src/configs/connectors/local_provider.rs index 566ca48b58..5df18e2386 100644 --- a/core/connectors/runtime/src/configs/connectors/local_provider.rs +++ b/core/connectors/runtime/src/configs/connectors/local_provider.rs @@ -16,8 +16,9 @@ // under the License. use crate::configs::connectors::{ - ConnectorConfig, ConnectorConfigVersionInfo, ConnectorConfigVersions, ConnectorsConfig, - ConnectorsConfigProvider, CreateSinkConfig, CreateSourceConfig, SinkConfig, SourceConfig, + ConnectorConfig, ConnectorConfigVersionInfo, ConnectorConfigVersions, ConnectorKey, + ConnectorsConfig, ConnectorsConfigProvider, CreateSinkConfig, CreateSourceConfig, SinkConfig, + SourceConfig, }; use crate::error::RuntimeError; use ::configs::{ConfigProvider, FileConfigProvider, TypedEnvProvider}; @@ -392,18 +393,18 @@ impl BaseConnectorConfig { impl ConnectorsConfigProvider for LocalConnectorsConfigProvider { async fn create_sink_config( &self, - key: &str, + key: &ConnectorKey, cmd: CreateSinkConfig, ) -> Result { let sinks = self.state.connectors_config.sinks(); let next_version = sinks .iter() - .filter(|entry| entry.key().key == key) + .filter(|entry| entry.key().key == key.as_str()) .max_by_key(|entry| entry.config.version) .map(|entry| entry.config.version + 1) .unwrap_or(0); - let config = cmd.to_sink_config(key, next_version); + let config = cmd.into_sink_config(key, next_version); let connector_config = ConnectorConfig::Sink(config.clone()); let connector_id: ConnectorId = (&connector_config).into(); @@ -418,7 +419,7 @@ impl ConnectorsConfigProvider for LocalConnectorsConfigProvider { SinkConfigFile { config: config.clone(), created_at: Utc::now(), - path: path.clone(), + path, }, ); @@ -427,18 +428,18 @@ impl ConnectorsConfigProvider for LocalConnectorsConfigProvider { async fn create_source_config( &self, - key: &str, + key: &ConnectorKey, cmd: CreateSourceConfig, ) -> Result { let sources = &self.state.connectors_config.sources; let next_version = sources .iter() - .filter(|entry| entry.key().key == key) + .filter(|entry| entry.key().key == key.as_str()) .max_by_key(|entry| entry.config.version) .map(|entry| entry.config.version + 1) .unwrap_or(0); - let config = cmd.to_source_config(key, next_version); + let config = cmd.into_source_config(key, next_version); let connector_config = ConnectorConfig::Source(config.clone()); let connector_id: ConnectorId = (&connector_config).into(); @@ -453,7 +454,7 @@ impl ConnectorsConfigProvider for LocalConnectorsConfigProvider { SourceConfigFile { config: config.clone(), created_at: Utc::now(), - path: path.clone(), + path, }, ); @@ -851,3 +852,32 @@ impl Provider for ConnectorEnvProvider { } } } + +#[cfg(test)] +mod tests { + use super::*; + use tempfile::TempDir; + + #[tokio::test] + async fn given_valid_key_when_creating_source_config_should_write_prefixed_file() { + let dir = TempDir::new().unwrap(); + let provider = LocalConnectorsConfigProvider::new(dir.path().to_str().unwrap()) + .init() + .await + .unwrap(); + let key: ConnectorKey = "random".parse().unwrap(); + + let config = provider + .create_source_config(&key, CreateSourceConfig::default()) + .await + .unwrap(); + + assert_eq!(config.key, "random"); + assert_eq!(config.version, 0); + let entries: Vec = std::fs::read_dir(dir.path()) + .unwrap() + .map(|entry| entry.unwrap().file_name().to_string_lossy().into_owned()) + .collect(); + assert_eq!(entries, vec!["source_random_0.toml"]); + } +} diff --git a/core/connectors/runtime/src/error.rs b/core/connectors/runtime/src/error.rs index 35be3bb17c..f0f5d11b08 100644 --- a/core/connectors/runtime/src/error.rs +++ b/core/connectors/runtime/src/error.rs @@ -21,6 +21,11 @@ use thiserror::Error; pub enum RuntimeError { #[error("Invalid configuration: {0}")] InvalidConfiguration(String), + #[error( + "Invalid connector key {0:?}: expected at most {max_length} bytes of ASCII letters, digits, '-', '_' or '.', starting with a letter or digit", + max_length = crate::configs::connectors::ConnectorKey::MAX_LENGTH + )] + InvalidConnectorKey(String), #[error("Failed to serialize topic metadata")] FailedToSerializeTopicMetadata, #[error("Failed to serialize messages metadata")] @@ -79,6 +84,7 @@ impl RuntimeError { RuntimeError::SourceConfigNotFound(_, _) => "source_config_not_found", RuntimeError::MissingIggyCredentials => "invalid_configuration", RuntimeError::InvalidConfiguration(_) => "invalid_configuration", + RuntimeError::InvalidConnectorKey(_) => "invalid_connector_key", RuntimeError::HttpRequestFailed(_) => "http_request_failed", RuntimeError::StateLoadFailed { .. } => "state_load_failed", RuntimeError::TokenFileNotFound(_) => "invalid_configuration", diff --git a/core/connectors/runtime/src/main.rs b/core/connectors/runtime/src/main.rs index 08311d7d1d..fcf8cafa91 100644 --- a/core/connectors/runtime/src/main.rs +++ b/core/connectors/runtime/src/main.rs @@ -15,7 +15,9 @@ // specific language governing permissions and limitations // under the License. -use crate::configs::connectors::{ConnectorsConfigProvider, create_connectors_config_provider}; +use crate::configs::connectors::{ + ConnectorKey, ConnectorsConfig, ConnectorsConfigProvider, create_connectors_config_provider, +}; use ::configs::ConfigProvider; use clap::Parser; use configs::connectors::ConfigFormat; @@ -40,7 +42,7 @@ use std::{ sync::{Arc, atomic::AtomicU32}, }; use system_stats::capture_allowed_cpus; -use tracing::{error, info}; +use tracing::{error, info, warn}; mod api; mod benchmark; @@ -155,6 +157,7 @@ async fn main() -> Result<(), RuntimeError> { connectors_config.sources().len(), connectors_config.sinks().len() ); + warn_on_unaddressable_keys(&connectors_config); let sources_config = connectors_config.sources(); let (sources, failed_sources) = source::init( sources_config.clone(), @@ -306,6 +309,29 @@ async fn main() -> Result<(), RuntimeError> { Ok(()) } +/// Keys loaded from a provider are not held to `ConnectorKey`, so existing +/// deployments keep starting, but the control API only routes keys that pass +/// it. Say so at startup instead of letting the operator discover a 400. +fn warn_on_unaddressable_keys(connectors_config: &ConnectorsConfig) { + let keys = connectors_config + .sinks() + .keys() + .map(|key| ("sink", key)) + .chain( + connectors_config + .sources() + .keys() + .map(|key| ("source", key)), + ); + for (connector_type, key) in keys { + if let Err(error) = key.parse::() { + warn!( + "Loaded {connector_type} connector with key {key:?} that the control API cannot address: {error}" + ); + } + } +} + /// Resolves a plugin shared library path from the connector config `path` field. /// /// Accepts both `plugin.so` and `plugin` (OS-specific extension appended if missing). diff --git a/core/integration/tests/connectors/api/endpoints.rs b/core/integration/tests/connectors/api/endpoints.rs index 97ebd72e4c..d1d8c75ab9 100644 --- a/core/integration/tests/connectors/api/endpoints.rs +++ b/core/integration/tests/connectors/api/endpoints.rs @@ -21,8 +21,15 @@ use iggy_connector_sdk::api::{ use integration::harness::seeds; use integration::iggy_harness; use reqwest::Client; +use serde_json::{Value, json}; +use std::fs; const API_KEY: &str = "test-api-key"; +/// `config_dir` of `key_validation.toml`, relative to the crate root, which is +/// the working directory of both the test process and the spawned runtime. It +/// lives under the gitignored `test_logs/` so a regression that writes a config +/// file cannot land in the source tree or be loaded by another test. +const CONNECTORS_CONFIG_DIR: &str = "../../test_logs/connectors_api_key_validation"; #[iggy_harness( server(connectors_runtime(config_path = "tests/connectors/api/config.toml")), @@ -243,3 +250,100 @@ async fn api_key_authentication_rejected_with_invalid_key(harness: &TestHarness) assert_eq!(response.status(), 401); } + +#[iggy_harness( + server(connectors_runtime(config_path = "tests/connectors/api/key_validation.toml")), + seed = seeds::connector_stream +)] +async fn key_endpoints_reject_key_that_is_not_a_single_path_component(harness: &TestHarness) { + let api_address = harness + .connectors_runtime() + .expect("connector runtime should be available") + .http_url(); + let client = Client::new(); + let config = json!({ + "enabled": false, + "name": "x", + "path": "/tmp/evil.so", + "streams": [] + }); + let config_dir_before = config_dir_entries(); + + // Each entry is a percent-encoded path segment: axum decodes it before + // handing it to the handler, so `..%2F..%2Fpwned` arrives as `../../pwned`. + // The charset itself is covered by unit tests; these are the two probes + // from the issue report plus a hidden-file name. + let keys = ["..%2F..%2Fpwned", "x%2F..%2F..%2F..%2Ftmp%2Fpwn", ".hidden"]; + for key in keys { + for kind in ["sources", "sinks"] { + let response = client + .post(format!("{api_address}/{kind}/{key}/configs")) + .header("api-key", API_KEY) + .json(&config) + .send() + .await + .unwrap(); + assert_eq!(response.status(), 400, "POST /{kind}/{key}/configs"); + let body: Value = response.json().await.unwrap(); + assert_eq!( + body["code"], "invalid_connector_key", + "POST /{kind}/{key}/configs body: {body}" + ); + + let response = client + .get(format!("{api_address}/{kind}/{key}")) + .header("api-key", API_KEY) + .send() + .await + .unwrap(); + assert_eq!(response.status(), 400, "GET /{kind}/{key}"); + let body: Value = response.json().await.unwrap(); + assert_eq!( + body["code"], "invalid_connector_key", + "GET /{kind}/{key} body: {body}" + ); + } + } + + assert_eq!( + config_dir_entries(), + config_dir_before, + "no config file should have been written" + ); +} + +#[iggy_harness( + server(connectors_runtime(config_path = "tests/connectors/api/config.toml")), + seed = seeds::connector_stream +)] +async fn key_endpoints_accept_single_component_key(harness: &TestHarness) { + let api_address = harness + .connectors_runtime() + .expect("connector runtime should be available") + .http_url(); + let client = Client::new(); + + for (kind, code) in [("sources", "source_not_found"), ("sinks", "sink_not_found")] { + let response = client + .get(format!("{api_address}/{kind}/postgres-cdc.v2_1")) + .header("api-key", API_KEY) + .send() + .await + .unwrap(); + assert_eq!(response.status(), 404, "GET /{kind}/postgres-cdc.v2_1"); + let body: Value = response.json().await.unwrap(); + assert_eq!( + body["code"], code, + "GET /{kind}/postgres-cdc.v2_1 body: {body}" + ); + } +} + +fn config_dir_entries() -> Vec { + let mut entries: Vec = fs::read_dir(CONNECTORS_CONFIG_DIR) + .unwrap() + .map(|entry| entry.unwrap().file_name().to_string_lossy().into_owned()) + .collect(); + entries.sort(); + entries +} diff --git a/core/integration/tests/connectors/api/key_validation.toml b/core/integration/tests/connectors/api/key_validation.toml new file mode 100644 index 0000000000..72c2fa69a7 --- /dev/null +++ b/core/integration/tests/connectors/api/key_validation.toml @@ -0,0 +1,31 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +[http] +enabled = true +address = "0.0.0.0:0" +api_key = "test-api-key" + +[http.metrics] +enabled = true +endpoint = "/metrics" + +[connectors] +config_type = "local" +# Gitignored: the key-validation test snapshots this directory and must +# never leave a config file in the source tree. +config_dir = "../../test_logs/connectors_api_key_validation" From 06a12e306cb6febc773bae6f4e358f3a34f3fbf6 Mon Sep 17 00:00:00 2001 From: Piotr Gankiewicz Date: Mon, 7 Sep 2026 18:36:45 +0200 Subject: [PATCH 081/182] chore: add Iggy banner to server startup (#4085) --- core/server/src/banner.rs | 80 +++++++++++++++++++++++++++++++++++++++ core/server/src/main.rs | 2 + 2 files changed, 82 insertions(+) create mode 100644 core/server/src/banner.rs diff --git a/core/server/src/banner.rs b/core/server/src/banner.rs new file mode 100644 index 0000000000..06a403d7e0 --- /dev/null +++ b/core/server/src/banner.rs @@ -0,0 +1,80 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use std::env; +use std::io::{self, IsTerminal, Write}; + +use figlet_rs::FIGlet; + +const WIDTH: usize = 64; + +const IGGY: &str = r" + ⣀⣠⣤⣤⣤⣤⣤⣾⣛⠛⠓⠶⣄ + ⣠⡴⠞⠛⠋⠛⠛⠲⢶⠶⠛⠉⠁ ⠉⠙⠻⣷⣜⣧ + ⣴⠋⣠⠴⣶⣿⣶⣶⣦⠄ ⢀⣈⣿⡽⣆ + ⣸⠇⠜⠁⡞⠉⠈⠹⣿⠃ ⢠⠴⣶⢦⡀ ⢪⣨⣆⢳⡹⣆ + ⣰⠏ ⢸⡇ ⠉ ⢿⣦⣾⠆⠇ ⠈⢻⡃⣼⣿⣿⣷⡄ + ⢀⣴⠏ ⢀⣠⡾⠛⠶⠶⣦⡀ ⠉⠉ ⠙⢿⣉⠉⠁ + ⠈⠙⠛⠛⠋⠁ ⢸⣷ ⠈⢳⣄ + ⢸⡏⢣⡀ ⢀⡴⢊⣁⣀⣬⣵ + ⣼⡇ ⠸⣿⣷⣦⣄ ⠘⣷⣿⣻⣿⣿⣿⡀ + ⣿⠁ ⠠⡀ ⢻⣜⢿⣿⣿⣶⣤⣄⣀⡀ ⠈⠙⣛⣿⡟ ⠙⠲⣄ + ⣿ ⢙⢦⣄⠙⣮⡻⣿⣿⣿⣿⣿⣿⣏⠉⠉⠉ ⠙⢦⡀ ⠈⠳⣄ + ⢀⣀⣀ ⣿ ⠙⣿⣷⣌⠛⢮⣿⣿⣿⣿⣿⣿⣧⡀ ⠙⢦⡀ ⠈⠳⡄ + ⠸⣄⡼ ⢰⡋⢹ ⣿ ⠈⢻⣿⣷⣄⠉⢻⡟⣿⣿⣿⣿⣷⡀ ⠙⢆ ⠘⢦ + ⣠⠶⢄ ⠱⡄ ⠉⣇ ⢰⡇ ⢻⣿⣿⣷⣦⣻⡄ ⣮⠙⠘⣇ ⠈⢳⡀ ⠈⢳⡀ + ⠙⠦⠞⠢⢄⡀ ⠘⢦⡀⠘⣆ ⢸⡇ ⢿⣿⣿⡏⠛⠻⣦⣘⣂⣠⡟ ⠹⡄ ⢷ + ⠈⠑⠢⢤⣀⠙⠦⣌⠑⠆⢠⣄⡀ ⣠⣤⣤⣄⣾⠁ ⠘⣿⢹⡇ ⠈⠉⠉⠉ ⢳ ⠘⡇ + ⡤⢤⣀⣀⣠⣤⣄⣀⣈⠙⠲⢬⠉⢠⠏ ⠉⠓⠦⣤⣶⠷⡾⠋⠁⠉⠻⣿⣶⣤⡀ ⢿⠈⢿⡀ ⢀⣤⣤⣤⣀ ⢸⡇ ⣷ + ⠈⠧⠴⠃ ⣀⣠⣬⣉⣓⠂⢠⠏ ⣰⡿⠁ ⡀ ⠈⠹⣷ ⢸⡇⠘⣷⣠⣾⠿⠻⠿⠋ ⠉⢻⣶⣶⣄ ⣸⠁ ⡏ + ⢰⡉⢹ ⠈⢠⣏⠠⢄⣀ ⣿⠃ ⡼ ⣰⠁ ⡀ ⣿⣇ ⢇ ⢸⣿⠃ ⠈⢻⣧⢀⡴⠃ ⣸⠃ + ⠉⠁ ⠉⠛⠶⢮⣄⣒⣿ ⢰⠇ ⢰⡇ ⣸⠁ ⣿⡏⠙⠛⠲⠶⢤⣤⣀⣀⣀⣀⣸⣿ ⡆ ⢣ ⢳ ⣿⡏ ⢀⡴⠃ + ⠈⠉⠻⣷⣾ ⣾ ⢀⡟ ⣸⡟ ⠉⠉⠉⠛⣿ ⣷ ⢸⡆ ⢸⡇ ⣿⡇⣀⠴⠋ + ⢹⣆⣠⣿ ⣼⡿⠿⠿⢤⣤⣄⣀⣀⡀ ⢿⣆⣀⣿ ⢸⡇ ⢀⣇⣠⣿⠋⠁ + ⠈⠙⠛⠻⣶⣾⠟ ⠈⠉⠉⠉⠛⠛⠛⠛⠒⠒⠚⠛⠿⢿⡄ ⣸⡇⢀⣼⠿⠟⠁ + ⠈⢿⣶⠟⠛⠿⠋ +"; + +pub fn print(version: &str) { + let stdout = io::stdout(); + let color = stdout.is_terminal() + && env::var_os("TERM").is_none_or(|term| term != "dumb") + && env::var_os("NO_COLOR").is_none_or(|value| value.is_empty()); + let _ = render(&mut stdout.lock(), version, color); +} + +fn render(output: &mut impl Write, version: &str, color: bool) -> io::Result<()> { + let wordmark = FIGlet::standard() + .ok() + .and_then(|font| font.convert("Iggy").map(|figure| figure.to_string())) + .unwrap_or_else(|| "Iggy".to_owned()); + let width = wordmark.lines().map(str::len).max().unwrap_or_default(); + let padding = 2 + WIDTH.saturating_sub(width) / 2; + let (orange, reset) = if color { + ("\x1b[38;5;208m", "\x1b[0m") + } else { + ("", "") + }; + + writeln!(output, "{IGGY}")?; + for line in wordmark.lines() { + writeln!(output, "{orange}{:padding$}{}{reset}", "", line.trim_end())?; + } + writeln!(output)?; + writeln!(output, " {:^WIDTH$}", format!("Apache Iggy v{version}"))?; + writeln!(output) +} diff --git a/core/server/src/main.rs b/core/server/src/main.rs index 5706a32601..14d5592720 100644 --- a/core/server/src/main.rs +++ b/core/server/src/main.rs @@ -18,6 +18,7 @@ #![allow(clippy::future_not_send)] mod args; +mod banner; use args::Args; use clap::Parser; @@ -37,6 +38,7 @@ fn main() -> Result<(), ServerError> { // visible. `create_shard_executor` also reads its capacity knob from the // environment, which is why the `.env` load has to precede it. let args = Args::parse(); + banner::print(server::VERSION); // `logging` owns the tracing appender worker guards; it must outlive the // shard threads or every log line after bootstrap is silently dropped. let mut logging = Logging::new(server::VERSION); From bd1a873fe12887798aa045aa9fa8e7e2bafd2ea3 Mon Sep 17 00:00:00 2001 From: saie-ch <132209179+saie-ch@users.noreply.github.com> Date: Mon, 7 Sep 2026 23:01:27 +0530 Subject: [PATCH 082/182] feat(python): add HttpConfig transport configuration (#3992) Relates to #2835 --------- Co-authored-by: Hubert Gruszecki --- .../python-maturin/pre-merge/action.yml | 14 +- .github/workflows/coverage-baseline.yml | 13 +- core/sdk/src/prelude.rs | 1 + examples/python/getting-started/consumer.py | 20 +- examples/python/getting-started/producer.py | 20 +- foreign/python/README.md | 38 +- foreign/python/apache_iggy.pyi | 114 +++++- foreign/python/src/client.rs | 40 +- foreign/python/src/config.rs | 212 ++++++++-- foreign/python/src/lib.rs | 5 +- foreign/python/tests/test_client_config.py | 4 +- foreign/python/tests/test_http_config.py | 363 ++++++++++++++++++ foreign/python/tests/test_quic_config.py | 76 +++- foreign/python/tests/utils.py | 18 +- 14 files changed, 855 insertions(+), 83 deletions(-) create mode 100644 foreign/python/tests/test_http_config.py diff --git a/.github/actions/python-maturin/pre-merge/action.yml b/.github/actions/python-maturin/pre-merge/action.yml index 3cca87fceb..53bf430dc0 100644 --- a/.github/actions/python-maturin/pre-merge/action.yml +++ b/.github/actions/python-maturin/pre-merge/action.yml @@ -173,14 +173,24 @@ runs: run: | cd foreign/python - echo "Running integration tests with Iggy server at ${{ steps.iggy.outputs.address }}..." + # TCP and HTTP ports come from server-start's own outputs rather than + # duplicated literals, so a future default change here can't silently + # desync from what the server actually started with. QUIC stays a + # literal below because server-start exposes no output for it. + tcp_port="${{ steps.iggy.outputs.tcp_address }}" + http_port="${{ steps.iggy.outputs.http_address }}" + tcp_port="${tcp_port##*:}" + http_port="${http_port##*:}" + + echo "Running integration tests with Iggy server at ${{ steps.iggy.outputs.tcp_address }}..." # Run all tests with server connection # --no-sync prevents uv from re-syncing the venv which would # overwrite the coverage-instrumented .so with a non-instrumented one IGGY_SERVER_HOST=127.0.0.1 \ - IGGY_SERVER_TCP_PORT=8090 \ + IGGY_SERVER_TCP_PORT="$tcp_port" \ IGGY_SERVER_QUIC_PORT=8080 \ + IGGY_SERVER_HTTP_PORT="$http_port" \ IGGY_SERVER_DOCKER_IMAGE=iggy-server:local \ uv run --no-sync pytest tests/ -v \ --junitxml=../../reports/python-junit.xml \ diff --git a/.github/workflows/coverage-baseline.yml b/.github/workflows/coverage-baseline.yml index d8f070ba25..76d12d1b01 100644 --- a/.github/workflows/coverage-baseline.yml +++ b/.github/workflows/coverage-baseline.yml @@ -347,9 +347,20 @@ jobs: - name: Run tests run: | cd foreign/python + + # TCP and HTTP ports come from server-start's own outputs rather than + # duplicated literals, so a future default change here can't silently + # desync from what the server actually started with. QUIC stays a + # literal below because server-start exposes no output for it. + tcp_port="${{ steps.iggy.outputs.tcp_address }}" + http_port="${{ steps.iggy.outputs.http_address }}" + tcp_port="${tcp_port##*:}" + http_port="${http_port##*:}" + IGGY_SERVER_HOST=127.0.0.1 \ - IGGY_SERVER_TCP_PORT=8090 \ + IGGY_SERVER_TCP_PORT="$tcp_port" \ IGGY_SERVER_QUIC_PORT=8080 \ + IGGY_SERVER_HTTP_PORT="$http_port" \ IGGY_SERVER_DOCKER_IMAGE=iggy-server:local \ uv run --no-sync pytest tests/ -v \ --junitxml=../../reports/python-junit.xml \ diff --git a/core/sdk/src/prelude.rs b/core/sdk/src/prelude.rs index 10cfbc685c..79c6774b2b 100644 --- a/core/sdk/src/prelude.rs +++ b/core/sdk/src/prelude.rs @@ -41,6 +41,7 @@ pub use crate::clients::producer_builder::IggyProducerBuilder; pub use crate::clients::producer_config::{BackgroundConfig, DirectConfig}; pub use crate::clients::producer_sharding::{BalancedSharding, OrderedSharding, Sharding}; pub use crate::consumer_ext::IggyConsumerMessageExt; +pub use crate::http::http_client::HttpClient; pub use crate::quic::quic_client::QuicClient; pub use crate::stream_builder::IggyConsumerConfig; pub use crate::stream_builder::IggyStreamConsumer; diff --git a/examples/python/getting-started/consumer.py b/examples/python/getting-started/consumer.py index f3f7886409..3a48f2a2e6 100755 --- a/examples/python/getting-started/consumer.py +++ b/examples/python/getting-started/consumer.py @@ -24,8 +24,10 @@ from apache_iggy import ( AutoLogin, Consumer, + HttpConfig, IggyClient, PollingStrategy, + QuicConfig, ReceiveMessage, TcpConfig, TcpReconnectionConfig, @@ -101,12 +103,12 @@ def parse_args() -> ArgNamespace: return ArgNamespace(**vars(args)) -def build_config(args: ArgNamespace) -> TcpConfig: - """Build the TCP client configuration with auto-login and reconnection.""" +def build_config(args: ArgNamespace) -> TcpConfig | QuicConfig | HttpConfig: + """Build the client configuration, TCP with auto-login and reconnection.""" # IggyClient(...) also accepts a QuicConfig for the QUIC transport. To use - # it, import QuicConfig and QuicReconnectionConfig above, change the return - # annotation to QuicConfig, and replace the return statement with: + # it, uncomment the return below and import QuicReconnectionConfig, which is + # left out above because only the commented block names it: # # return QuicConfig( # server_address="127.0.0.1:8080", @@ -116,8 +118,12 @@ def build_config(args: ArgNamespace) -> TcpConfig: # enabled=True, interval=timedelta(seconds=1) # ), # ) - # - # main() logs args.tcp_server_address, so change that line too. + + # IggyClient(...) also accepts an HttpConfig for the HTTP transport. HTTP + # has no AutoLogin or reconnection policy, so main() below would also need + # an explicit `await client.login_user(args.username, args.password)` + # after connecting: + # return HttpConfig(api_url="http://127.0.0.1:3000") return TcpConfig( server_address=args.tcp_server_address, @@ -138,7 +144,7 @@ async def main(): except ValueError as error: logger.error(f"Invalid client configuration: {error}") return - logger.info(f"Connecting to {args.tcp_server_address} (TLS: {args.tls})") + logger.info(f"Connecting with {config}") client = IggyClient(config) try: diff --git a/examples/python/getting-started/producer.py b/examples/python/getting-started/producer.py index 113bee858d..e8a8fea35b 100755 --- a/examples/python/getting-started/producer.py +++ b/examples/python/getting-started/producer.py @@ -23,7 +23,9 @@ from apache_iggy import ( AutoLogin, + HttpConfig, IggyClient, + QuicConfig, StreamDetails, TcpConfig, TcpReconnectionConfig, @@ -100,12 +102,12 @@ def parse_args() -> ArgNamespace: return ArgNamespace(**vars(args)) -def build_config(args: ArgNamespace) -> TcpConfig: - """Build the TCP client configuration with auto-login and reconnection.""" +def build_config(args: ArgNamespace) -> TcpConfig | QuicConfig | HttpConfig: + """Build the client configuration, TCP with auto-login and reconnection.""" # IggyClient(...) also accepts a QuicConfig for the QUIC transport. To use - # it, import QuicConfig and QuicReconnectionConfig above, change the return - # annotation to QuicConfig, and replace the return statement with: + # it, uncomment the return below and import QuicReconnectionConfig, which is + # left out above because only the commented block names it: # # return QuicConfig( # server_address="127.0.0.1:8080", @@ -115,8 +117,12 @@ def build_config(args: ArgNamespace) -> TcpConfig: # enabled=True, interval=timedelta(seconds=1) # ), # ) - # - # main() logs args.tcp_server_address, so change that line too. + + # IggyClient(...) also accepts an HttpConfig for the HTTP transport. HTTP + # has no AutoLogin or reconnection policy, so main() below would also need + # an explicit `await client.login_user(args.username, args.password)` + # after connecting: + # return HttpConfig(api_url="http://127.0.0.1:3000") return TcpConfig( server_address=args.tcp_server_address, @@ -137,7 +143,7 @@ async def main(): except ValueError as error: logger.error(f"Invalid client configuration: {error}") return - logger.info(f"Connecting to {args.tcp_server_address} (TLS: {args.tls})") + logger.info(f"Connecting with {config}") client = IggyClient(config) logger.info("Connecting to IggyClient") diff --git a/foreign/python/README.md b/foreign/python/README.md index df62cfc77e..293bbdfbc3 100644 --- a/foreign/python/README.md +++ b/foreign/python/README.md @@ -134,7 +134,8 @@ running prek / committing / pushing. This list is not exhaustive and other hook ## Client Configuration -`IggyClient` takes a server address, a `TcpConfig`, or a `QuicConfig`: +`IggyClient` takes a server address, a `TcpConfig`, a `QuicConfig`, or an +`HttpConfig`: ```python import asyncio @@ -168,8 +169,39 @@ async def main(): asyncio.run(main()) ``` -`IggyClient(...)` also accepts a `QuicConfig` for the QUIC transport; see -`examples/python/getting-started/producer.py` for a config swap example. +`IggyClient(...)` also accepts a `QuicConfig` for the QUIC transport and an +`HttpConfig` for the HTTP transport; +`examples/python/getting-started/producer.py` shows either swap in context. + +`HttpConfig` differs from TCP in two ways. There is no reconnection policy and no +`AutoLogin`: `connect()` does not dial over HTTP, but it does start the +heartbeat that `heartbeat_interval` configures, so call it and then +`login_user(...)`. And HTTP is single-consumer only: the `consumer_group(...)` +path always fails with `Feature is unavailable`, at the join by default and at +the returned consumer's first poll if you disable `auto_join_consumer_group`, +so disabling it is not a workaround. A direct +`poll_messages(consumer=Consumer.Group(...))` fails the same way unless you +pass an explicit `partition_id`, and with one it degrades silently instead: the +consumer kind is not carried on the HTTP wire, so the group is served as an +ordinary consumer named after it, with no membership or partition assignment +behind it. Use `Consumer.Single(...)` with `poll_messages(...)`. Delivery is +also at-least-once: the default `retries=3` replays the full request body, so a +send whose response was lost is applied twice, and only `retries=0` opts out. + +```python +import asyncio + +from apache_iggy import HttpConfig, IggyClient + + +async def main(): + client = IggyClient(HttpConfig(api_url="http://127.0.0.1:3000")) + await client.connect() + await client.login_user("iggy", "iggy") + + +asyncio.run(main()) +``` ## Examples diff --git a/foreign/python/apache_iggy.pyi b/foreign/python/apache_iggy.pyi index e543869e9c..0400a1a514 100644 --- a/foreign/python/apache_iggy.pyi +++ b/foreign/python/apache_iggy.pyi @@ -39,6 +39,7 @@ __all__ = [ "GlobalPermissions", "HeaderKey", "HeaderValue", + "HttpConfig", "IggyClient", "IggyConsumer", "IggyExpiry", @@ -894,6 +895,78 @@ class HeaderValue: def value(self) -> builtins.float: ... def __new__(cls, value: builtins.float) -> HeaderValue.Float64: ... +@typing.final +class HttpConfig: + r""" + Configuration for the HTTP transport, accepted by `IggyClient(...)`. + + Every field is keyword-only and optional. + + There is no `AutoLogin` and no reconnection policy, and `connect()` does not + dial: it only starts the heartbeat, so `login_user(...)` has to follow it. + + HTTP is single-consumer only. `consumer_group(...)` fails with + `Feature is unavailable`, and so does a `Consumer.Group(...)` poll unless it + names an explicit `partition_id`. With one, the consumer kind is not carried + on the HTTP wire, so the poll is served as an ordinary consumer named after + the group, with no membership, no partition assignment, and no rebalancing + behind it. Pass `Consumer.Single(...)` explicitly. + """ + @property + def api_url(self) -> builtins.str: ... + @property + def retries(self) -> builtins.int: ... + @property + def has_jwt(self) -> builtins.bool: + r""" + Whether a JWT is configured, without exposing the token itself. + """ + @property + def heartbeat_interval(self) -> datetime.timedelta: ... + def __new__( + cls, + *, + api_url: builtins.str | None = None, + retries: builtins.int | None = None, + jwt: builtins.str | None = None, + heartbeat_interval: datetime.timedelta | None = None, + ) -> HttpConfig: + r""" + Constructs an HTTP configuration. + + Args: + api_url: Base URL of the Iggy HTTP API, as `scheme://host[:port]` + only - no path, query, fragment, or credentials. Defaults to + `http://127.0.0.1:3000`. + retries: Number of retries to perform on transient errors, each one + replaying the full request (including its body) via automatic + middleware. Defaults to 3. Delivery is therefore at-least-once: + if the original request actually committed but its response + was lost (e.g. to a timeout), a retried call applies the same + operation again. Set to 0 to disable automatic replay and match + the other transports, which surface the failure instead of + silently resending. + jwt: JWT token for A2A (Agent-to-Agent) authentication. Defaults to + `None`. Stored trimmed, since a token read from a file carries a + trailing newline that the `Authorization` header value rejects. + Rejected if empty or whitespace-only: accepting it would make + `has_jwt` report `True` while every call still fails + `Unauthenticated`. + heartbeat_interval: Interval between the client's liveness probes + (a bare `GET /ping`). Defaults to 5 seconds. Unlike TCP/QUIC, + HTTP has no persistent connection or session for this to keep + alive; it only proves the server is reachable. + + Raises: + ValueError: If `api_url` is not a valid URL, if `retries` is outside + the range of an unsigned 32-bit integer, if `jwt` is empty or + whitespace-only, if a duration is negative, or if + `heartbeat_interval` is zero. + OverflowError: If `retries` does not fit a signed 64-bit integer, + raised by the underlying conversion before this constructor runs. + """ + def __repr__(self) -> builtins.str: ... + @typing.final class IggyClient: r""" @@ -901,20 +974,22 @@ class IggyClient: It provides asynchronous functionality through the contained runtime. """ def __new__( - cls, conn: TcpConfig | QuicConfig | builtins.str | None = None + cls, conn: TcpConfig | QuicConfig | HttpConfig | builtins.str | None = None ) -> IggyClient: r""" - Constructs a new IggyClient from a TCP server address, a `TcpConfig`, or a - `QuicConfig`. This initializes a new runtime for asynchronous operations. + Constructs a new IggyClient from a TCP server address, a `TcpConfig`, a + `QuicConfig`, or an `HttpConfig`. This initializes a new runtime for + asynchronous operations. Future versions might utilize asyncio for more Pythonic async. Args: - conn: A `host:port` address, a `TcpConfig`, or a `QuicConfig`. Defaults - to `127.0.0.1:8090` over TCP with auto-login disabled. A malformed - address is reported differently depending on the form: the string - form raises `RuntimeError` here, while `TcpConfig`/`QuicConfig` - raise `ValueError` when they are constructed, before either ever - reaches this call. Neither exception is a subclass of the other. + conn: A `host:port` address, a `TcpConfig`, a `QuicConfig`, or an + `HttpConfig`. Defaults to `127.0.0.1:8090` over TCP with auto-login + disabled. A malformed address is reported differently depending on + the form: the string form raises `RuntimeError` here, while + `TcpConfig`/`QuicConfig`/`HttpConfig` raise `ValueError` when they + are constructed, before any of them ever reaches this call. Neither + exception is a subclass of the other. Raises: RuntimeError: If the address passed as a string is not a valid @@ -1114,8 +1189,10 @@ class IggyClient: """ def connect(self) -> collections.abc.Awaitable[None]: r""" - Connects the IggyClient to its service. - Raises `RuntimeError` if the connection fails. + Connects the IggyClient to its service and starts the heartbeat task. + Raises `RuntimeError` if the connection fails. Over HTTP there is no + connection to establish, so only the heartbeat starts and this call + succeeds even against an unreachable server. """ def create_stream(self, name: builtins.str) -> collections.abc.Awaitable[None]: r""" @@ -1545,6 +1622,15 @@ class IggyClient: `poll_interval`, `polling_retry_interval`, `init_retry_interval` or an `AutoCommit` interval is negative, or if any of those except `poll_interval` is zero. + + Consumer groups are not available over HTTP. With `auto_join_consumer_group` + left on, this call fails at the join with `Feature is unavailable`. + Turning it off is not a workaround: the join is skipped, but a group + member always polls without a partition, so the first poll fails with + the same error. Use `Consumer.Single(...)` with `poll_messages(...)` + instead - a `Consumer.Group(...)` poll with an explicit `partition_id` + does reach the server, but is served as an ordinary consumer named + after the group. """ def send_binary_request( self, code: builtins.int, payload: builtins.bytes @@ -1963,6 +2049,8 @@ class QuicConfig: `max_idle_timeout` is not a whole number of milliseconds, if `initial_mtu` is below quinn's minimum of 1200, or if a numeric field is outside the range of its underlying wire type. + OverflowError: If a numeric field does not fit a signed 64-bit integer, + raised by the underlying conversion before this constructor runs. """ def __repr__(self) -> builtins.str: ... @@ -2010,6 +2098,8 @@ class QuicReconnectionConfig: Raises: ValueError: If a duration is negative, if `max_retries` is outside the range of an unsigned 32-bit integer, or if `interval` is zero. + OverflowError: If `max_retries` does not fit a signed 64-bit integer, + raised by the underlying conversion before this constructor runs. """ def __repr__(self) -> builtins.str: ... @@ -2578,6 +2668,8 @@ class TcpReconnectionConfig: Raises: ValueError: If a duration is negative, if `max_retries` is outside the range of an unsigned 32-bit integer, or if `interval` is zero. + OverflowError: If `max_retries` does not fit a signed 64-bit integer, + raised by the underlying conversion before this constructor runs. """ def __repr__(self) -> builtins.str: ... diff --git a/foreign/python/src/client.rs b/foreign/python/src/client.rs index fa0e9e3ac0..7d19de9954 100644 --- a/foreign/python/src/client.rs +++ b/foreign/python/src/client.rs @@ -92,17 +92,19 @@ fn resolve_topic_params( #[gen_stub_pymethods] #[pymethods] impl IggyClient { - /// Constructs a new IggyClient from a TCP server address, a `TcpConfig`, or a - /// `QuicConfig`. This initializes a new runtime for asynchronous operations. + /// Constructs a new IggyClient from a TCP server address, a `TcpConfig`, a + /// `QuicConfig`, or an `HttpConfig`. This initializes a new runtime for + /// asynchronous operations. /// Future versions might utilize asyncio for more Pythonic async. /// /// Args: - /// conn: A `host:port` address, a `TcpConfig`, or a `QuicConfig`. Defaults - /// to `127.0.0.1:8090` over TCP with auto-login disabled. A malformed - /// address is reported differently depending on the form: the string - /// form raises `RuntimeError` here, while `TcpConfig`/`QuicConfig` - /// raise `ValueError` when they are constructed, before either ever - /// reaches this call. Neither exception is a subclass of the other. + /// conn: A `host:port` address, a `TcpConfig`, a `QuicConfig`, or an + /// `HttpConfig`. Defaults to `127.0.0.1:8090` over TCP with auto-login + /// disabled. A malformed address is reported differently depending on + /// the form: the string form raises `RuntimeError` here, while + /// `TcpConfig`/`QuicConfig`/`HttpConfig` raise `ValueError` when they + /// are constructed, before any of them ever reaches this call. Neither + /// exception is a subclass of the other. /// /// Raises: /// RuntimeError: If the address passed as a string is not a valid @@ -111,7 +113,9 @@ impl IggyClient { #[new] #[pyo3(signature = (conn=None))] fn new( - #[gen_stub(override_type(type_repr = "TcpConfig | QuicConfig | builtins.str | None"))] + #[gen_stub(override_type( + type_repr = "TcpConfig | QuicConfig | HttpConfig | builtins.str | None" + ))] conn: Option, ) -> PyResult { let wrapper = match conn { @@ -138,6 +142,9 @@ impl IggyClient { QuicClient::create(config.client_config()).map_err(to_runtime_error)?, ) } + Some(PyClientConfig::Http(config)) => ClientWrapper::Http( + HttpClient::create(config.client_config()).map_err(to_runtime_error)?, + ), None => ClientWrapper::Tcp( TcpClient::create(Arc::new(TcpClientConfig::default())) .map_err(to_runtime_error)?, @@ -500,8 +507,10 @@ impl IggyClient { }) } - /// Connects the IggyClient to its service. - /// Raises `RuntimeError` if the connection fails. + /// Connects the IggyClient to its service and starts the heartbeat task. + /// Raises `RuntimeError` if the connection fails. Over HTTP there is no + /// connection to establish, so only the heartbeat starts and this call + /// succeeds even against an unreachable server. #[gen_stub(override_return_type(type_repr="collections.abc.Awaitable[None]", imports=("collections.abc")))] fn connect<'a>(&self, py: Python<'a>) -> PyResult> { let inner = self.inner.clone(); @@ -1279,6 +1288,15 @@ impl IggyClient { /// `poll_interval`, `polling_retry_interval`, `init_retry_interval` or an /// `AutoCommit` interval is negative, or if any of those except `poll_interval` /// is zero. + /// + /// Consumer groups are not available over HTTP. With `auto_join_consumer_group` + /// left on, this call fails at the join with `Feature is unavailable`. + /// Turning it off is not a workaround: the join is skipped, but a group + /// member always polls without a partition, so the first poll fails with + /// the same error. Use `Consumer.Single(...)` with `poll_messages(...)` + /// instead - a `Consumer.Group(...)` poll with an explicit `partition_id` + /// does reach the server, but is served as an ordinary consumer named + /// after the group. #[allow(clippy::too_many_arguments)] #[pyo3(signature = ( name, diff --git a/foreign/python/src/config.rs b/foreign/python/src/config.rs index cd98ef626d..f699a23590 100644 --- a/foreign/python/src/config.rs +++ b/foreign/python/src/config.rs @@ -17,6 +17,7 @@ use iggy::prelude::{ AutoLogin as RustAutoLogin, Credentials as RustCredentials, + HttpClientConfig as RustHttpClientConfig, HttpClientConfigBuilder, QuicClientConfig as RustQuicClientConfig, QuicClientConfigBuilder, QuicClientReconnectionConfig as RustQuicClientReconnectionConfig, TcpClientConfig as RustTcpClientConfig, TcpClientConfigBuilder, @@ -28,6 +29,7 @@ use pyo3::types::PyDelta; use pyo3_stub_gen::derive::{gen_stub_pyclass, gen_stub_pymethods}; use pyo3_stub_gen::impl_stub_type; use secrecy::SecretString; +use std::fmt::Display; use std::net::SocketAddr; use std::sync::Arc; @@ -144,6 +146,8 @@ impl TcpReconnectionConfig { /// Raises: /// ValueError: If a duration is negative, if `max_retries` is outside the /// range of an unsigned 32-bit integer, or if `interval` is zero. + /// OverflowError: If `max_retries` does not fit a signed 64-bit integer, + /// raised by the underlying conversion before this constructor runs. #[new] #[pyo3(signature = (*, enabled=None, max_retries=None, interval=None, reestablish_after=None))] fn new( @@ -157,14 +161,7 @@ impl TcpReconnectionConfig { let defaults = RustTcpClientReconnectionConfig::default(); let enabled = enabled.unwrap_or(defaults.enabled); let max_retries = max_retries - .map(|max_retries| { - u32::try_from(max_retries).map_err(|_| { - PyValueError::new_err(format!( - "'max_retries' must be between 0 and {}", - u32::MAX - )) - }) - }) + .map(|max_retries| u32_param(max_retries, "max_retries")) .transpose()?; let interval = interval .as_ref() @@ -308,7 +305,7 @@ impl TcpConfig { } let mut inner = builder .build() - .map_err(|e| PyValueError::new_err(e.to_string()))?; + .map_err(|e| invalid_address("server_address", e))?; if let Some(auto_login) = auto_login { inner.auto_login = auto_login.inner; } @@ -445,6 +442,8 @@ impl QuicReconnectionConfig { /// Raises: /// ValueError: If a duration is negative, if `max_retries` is outside the /// range of an unsigned 32-bit integer, or if `interval` is zero. + /// OverflowError: If `max_retries` does not fit a signed 64-bit integer, + /// raised by the underlying conversion before this constructor runs. #[new] #[pyo3(signature = (*, enabled=None, max_retries=None, interval=None, reestablish_after=None))] fn new( @@ -458,14 +457,7 @@ impl QuicReconnectionConfig { let defaults = RustQuicClientReconnectionConfig::default(); let enabled = enabled.unwrap_or(defaults.enabled); let max_retries = max_retries - .map(|max_retries| { - u32::try_from(max_retries).map_err(|_| { - PyValueError::new_err(format!( - "'max_retries' must be between 0 and {}", - u32::MAX - )) - }) - }) + .map(|max_retries| u32_param(max_retries, "max_retries")) .transpose()?; let interval = interval .as_ref() @@ -588,6 +580,8 @@ impl QuicConfig { /// `max_idle_timeout` is not a whole number of milliseconds, if /// `initial_mtu` is below quinn's minimum of 1200, or if a numeric /// field is outside the range of its underlying wire type. + /// OverflowError: If a numeric field does not fit a signed 64-bit integer, + /// raised by the underlying conversion before this constructor runs. #[new] #[pyo3(signature = ( *, @@ -647,15 +641,18 @@ impl QuicConfig { } let mut inner = builder .build() - .map_err(|e| PyValueError::new_err(e.to_string()))?; + .map_err(|e| invalid_address("server_address", e))?; if let Some(client_address) = client_address { - // Kept verbatim rather than normalized: `QuicClient::create` compares - // this against the literal default to decide whether to bind an IPv6 - // socket for an IPv6 server, and a rewritten string would not match. - client_address.parse::().map_err(|e| { - PyValueError::new_err(format!("'client_address' is not a valid 'host:port': {e}")) - })?; - inner.client_address = client_address; + // Trimmed like the server address, but otherwise kept verbatim rather + // than re-serialized from the parsed `SocketAddr`: `QuicClient::create` + // compares this against the literal default to decide whether to bind + // an IPv6 socket for an IPv6 server, and a rewritten string would not + // match. + let client_address = client_address.trim(); + client_address + .parse::() + .map_err(|e| invalid_address("client_address", e))?; + inner.client_address = client_address.to_owned(); } if let Some(server_name) = server_name { inner.server_name = server_name; @@ -826,10 +823,167 @@ impl QuicConfig { } } +/// Configuration for the HTTP transport, accepted by `IggyClient(...)`. +/// +/// Every field is keyword-only and optional. +/// +/// There is no `AutoLogin` and no reconnection policy, and `connect()` does not +/// dial: it only starts the heartbeat, so `login_user(...)` has to follow it. +/// +/// HTTP is single-consumer only. `consumer_group(...)` fails with +/// `Feature is unavailable`, and so does a `Consumer.Group(...)` poll unless it +/// names an explicit `partition_id`. With one, the consumer kind is not carried +/// on the HTTP wire, so the poll is served as an ordinary consumer named after +/// the group, with no membership, no partition assignment, and no rebalancing +/// behind it. Pass `Consumer.Single(...)` explicitly. +#[gen_stub_pyclass] +#[pyclass(from_py_object)] +#[derive(Clone)] +pub struct HttpConfig { + inner: Arc, +} + +impl HttpConfig { + /// The configuration in the shape `HttpClient::create` expects. + pub(crate) fn client_config(&self) -> Arc { + self.inner.clone() + } +} + +#[gen_stub_pymethods] +#[pymethods] +impl HttpConfig { + /// Constructs an HTTP configuration. + /// + /// Args: + /// api_url: Base URL of the Iggy HTTP API, as `scheme://host[:port]` + /// only - no path, query, fragment, or credentials. Defaults to + /// `http://127.0.0.1:3000`. + /// retries: Number of retries to perform on transient errors, each one + /// replaying the full request (including its body) via automatic + /// middleware. Defaults to 3. Delivery is therefore at-least-once: + /// if the original request actually committed but its response + /// was lost (e.g. to a timeout), a retried call applies the same + /// operation again. Set to 0 to disable automatic replay and match + /// the other transports, which surface the failure instead of + /// silently resending. + /// jwt: JWT token for A2A (Agent-to-Agent) authentication. Defaults to + /// `None`. Stored trimmed, since a token read from a file carries a + /// trailing newline that the `Authorization` header value rejects. + /// Rejected if empty or whitespace-only: accepting it would make + /// `has_jwt` report `True` while every call still fails + /// `Unauthenticated`. + /// heartbeat_interval: Interval between the client's liveness probes + /// (a bare `GET /ping`). Defaults to 5 seconds. Unlike TCP/QUIC, + /// HTTP has no persistent connection or session for this to keep + /// alive; it only proves the server is reachable. + /// + /// Raises: + /// ValueError: If `api_url` is not a valid URL, if `retries` is outside + /// the range of an unsigned 32-bit integer, if `jwt` is empty or + /// whitespace-only, if a duration is negative, or if + /// `heartbeat_interval` is zero. + /// OverflowError: If `retries` does not fit a signed 64-bit integer, + /// raised by the underlying conversion before this constructor runs. + #[new] + #[pyo3(signature = (*, api_url=None, retries=None, jwt=None, heartbeat_interval=None))] + fn new( + #[gen_stub(override_type(type_repr = "builtins.str | None"))] api_url: Option, + #[gen_stub(override_type(type_repr = "builtins.int | None"))] retries: Option, + #[gen_stub(override_type(type_repr = "builtins.str | None"))] jwt: Option, + #[gen_stub(override_type(type_repr = "datetime.timedelta | None", imports=("datetime")))] + heartbeat_interval: Option>, + ) -> PyResult { + // The builder starts from `HttpClientConfig::default()`, and its `build()` + // trims and validates the API URL whether or not one was set here. + let mut builder = HttpClientConfigBuilder::new(); + if let Some(api_url) = api_url { + builder = builder.with_api_url(api_url); + } + let mut inner = builder + .build() + .map_err(|e| PyValueError::new_err(format!("'api_url' is not a valid URL: {e}")))?; + if let Some(retries) = retries { + inner.retries = u32_param(retries, "retries")?; + } + if let Some(jwt) = jwt { + let jwt = jwt.trim(); + if jwt.is_empty() { + return Err(PyValueError::new_err( + "'jwt' must not be empty or whitespace-only", + )); + } + inner.jwt = Some(jwt.to_owned()); + } + if let Some(heartbeat_interval) = heartbeat_interval { + inner.heartbeat_interval = reject_zero( + py_delta_to_iggy_duration(&heartbeat_interval)?, + "heartbeat_interval", + )?; + } + + Ok(Self { + inner: Arc::new(inner), + }) + } + + #[getter] + fn api_url(&self) -> String { + self.inner.api_url.clone() + } + + #[getter] + fn retries(&self) -> u32 { + self.inner.retries + } + + /// Whether a JWT is configured, without exposing the token itself. + #[getter] + fn has_jwt(&self) -> bool { + self.inner.jwt.is_some() + } + + #[gen_stub(override_return_type(type_repr = "datetime.timedelta", imports=("datetime")))] + #[getter] + fn heartbeat_interval<'a>(&self, py: Python<'a>) -> PyResult> { + iggy_duration_to_py_delta(py, self.inner.heartbeat_interval.get()) + } + + fn __repr__(&self) -> String { + let jwt = if self.inner.jwt.is_some() { + "..." + } else { + "None" + }; + format!( + "HttpConfig(api_url={:?}, retries={}, jwt={jwt}, heartbeat_interval={})", + self.inner.api_url, + self.inner.retries, + duration_repr(self.inner.heartbeat_interval.get()), + ) + } +} + fn python_bool(value: bool) -> &'static str { if value { "True" } else { "False" } } +/// Rejects an address that is not a valid `host:port`, naming the argument it +/// came from: neither the builder's error nor `SocketAddr`'s mentions which one. +fn invalid_address(parameter: &str, error: impl Display) -> PyErr { + PyValueError::new_err(format!("'{parameter}' is not a valid 'host:port': {error}")) +} + +/// Converts a Python int to the unsigned 32-bit integer `max_retries`/`retries` +/// expect, naming the parameter in the error so a caller can tell which +/// argument was out of range. A value too large even for `i64` still raises +/// pyo3's own unnamed `OverflowError` before this ever runs. +fn u32_param(value: i64, parameter: &str) -> PyResult { + u32::try_from(value).map_err(|_| { + PyValueError::new_err(format!("'{parameter}' must be between 0 and {}", u32::MAX)) + }) +} + /// Converts a Python int to the unsigned 64-bit integer a QUIC transport /// field expects, naming the parameter in the error so a caller can tell /// which argument was out of range. The bound in the message is `i64::MAX` @@ -866,15 +1020,17 @@ fn varint_param(value: i64, parameter: &str) -> PyResult { Ok(value) } -/// What `IggyClient(...)` accepts: a bare `host:port`, a full `TcpConfig`, or a -/// `QuicConfig` for the QUIC transport. +/// What `IggyClient(...)` accepts: a bare `host:port`, a full `TcpConfig`, a +/// `QuicConfig` for the QUIC transport, or an `HttpConfig` for the HTTP transport. #[derive(FromPyObject)] pub enum PyClientConfig { #[pyo3(transparent)] Tcp(TcpConfig), #[pyo3(transparent)] Quic(QuicConfig), + #[pyo3(transparent)] + Http(HttpConfig), #[pyo3(transparent, annotation = "str")] ServerAddress(String), } -impl_stub_type!(PyClientConfig = TcpConfig | QuicConfig | String); +impl_stub_type!(PyClientConfig = TcpConfig | QuicConfig | HttpConfig | String); diff --git a/foreign/python/src/lib.rs b/foreign/python/src/lib.rs index e39b3339e8..80a123ee09 100644 --- a/foreign/python/src/lib.rs +++ b/foreign/python/src/lib.rs @@ -31,7 +31,9 @@ mod user; mod user_headers; use client::IggyClient; -use config::{AutoLogin, QuicConfig, QuicReconnectionConfig, TcpConfig, TcpReconnectionConfig}; +use config::{ + AutoLogin, HttpConfig, QuicConfig, QuicReconnectionConfig, TcpConfig, TcpReconnectionConfig, +}; use consumer::{ AutoCommit, AutoCommitAfter, AutoCommitWhen, Consumer, ConsumerGroup, ConsumerGroupDetails, ConsumerGroupMember, IggyConsumer, ReceiveMessageIterator, @@ -60,6 +62,7 @@ fn apache_iggy(_py: Python, m: &Bound<'_, PyModule>) -> PyResult<()> { m.add_class::()?; m.add_class::()?; m.add_class::()?; + m.add_class::()?; m.add_class::()?; m.add_class::()?; m.add_class::()?; diff --git a/foreign/python/tests/test_client_config.py b/foreign/python/tests/test_client_config.py index 26d9f4430e..7e80fe0659 100644 --- a/foreign/python/tests/test_client_config.py +++ b/foreign/python/tests/test_client_config.py @@ -257,8 +257,8 @@ def test_repr_shows_every_field_as_python(self): ["", "127.0.0.1", "127.0.0.1:not-a-port", "127.0.0.1:70000", "::1:8090"], ) def test_invalid_server_address_is_rejected(self, invalid_address: str): - """Test that a malformed address fails at construction, not at connect.""" - with pytest.raises(ValueError): + """Test that a malformed address fails at construction, naming itself.""" + with pytest.raises(ValueError, match="server_address"): TcpConfig(server_address=invalid_address) def test_negative_heartbeat_interval_is_rejected(self): diff --git a/foreign/python/tests/test_http_config.py b/foreign/python/tests/test_http_config.py new file mode 100644 index 0000000000..d2237aa35a --- /dev/null +++ b/foreign/python/tests/test_http_config.py @@ -0,0 +1,363 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +""" +Tests for the HTTP client configuration surface. + +`HttpConfig` mirrors the Rust SDK's `HttpClientConfig` the same way +`TcpConfig` does, so most of these assert that a value set from Python +survives to the getters and that unset fields fall back to the Rust +defaults. Unlike TCP there is no `AutoLogin` or reconnection policy to +configure. +""" + +import ast +import json +import urllib.request +from datetime import timedelta + +import pytest + +from apache_iggy import Consumer, HttpConfig, IggyClient, PollingStrategy +from apache_iggy import SendMessage as Message + +from .utils import get_http_server_config, wait_for_ping + + +@pytest.mark.unit +class TestHttpConfig: + """Test the transport configuration.""" + + def test_defaults_match_the_rust_sdk(self): + """Test that an unconfigured transport matches the Rust SDK defaults.""" + config = HttpConfig() + + assert config.api_url == "http://127.0.0.1:3000" + assert config.retries == 3 + assert config.has_jwt is False + assert config.heartbeat_interval == timedelta(seconds=5) + + def test_every_field_round_trips(self): + """Test that each configured field is readable back unchanged.""" + config = HttpConfig( + api_url="http://127.0.0.1:3001", + retries=5, + jwt="a-token", + heartbeat_interval=timedelta(seconds=15), + ) + + assert config.api_url == "http://127.0.0.1:3001" + assert config.retries == 5 + assert config.has_jwt is True + assert config.heartbeat_interval == timedelta(seconds=15) + + def test_arguments_are_keyword_only(self): + """Test that the API URL cannot be passed positionally.""" + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-count + HttpConfig("http://127.0.0.1:3000") + + def test_repr_hides_the_jwt(self): + """Test that the JWT does not leak through repr but still parses as Python.""" + config = HttpConfig(jwt="a-secret-token") + + printed = repr(config) + + assert "a-secret-token" not in printed + ast.parse(printed) + + def test_repr_shows_every_field_as_python(self): + """Test that repr covers the configured fields and parses as Python. + + `heartbeat_interval` is included: its repr is built from a duration, + the one format-fragile field here, and `ast.parse` alone would not + catch a regression that renders it as something other than a + `datetime.timedelta` call. + """ + config = HttpConfig( + api_url="http://127.0.0.1:3001", + retries=5, + heartbeat_interval=timedelta(seconds=15), + ) + + printed = repr(config) + + assert 'api_url="http://127.0.0.1:3001"' in printed + assert "retries=5" in printed + assert "heartbeat_interval=datetime.timedelta(seconds=15)" in printed + ast.parse(printed) + + @pytest.mark.parametrize( + "invalid_url", + [ + "", + "not-a-url", + "http://127.0.0.1:0", + "http://127.0.0.1:3000/iggy", + "http://user:pass@127.0.0.1:3000", + ], + ) + def test_invalid_api_url_is_rejected(self, invalid_url: str): + """Test that a malformed API URL fails at construction, not at connect. + + Only `scheme://host[:port]` is accepted: a path, query, fragment, or + embedded credentials are all rejected, not just a missing/zero port. + """ + with pytest.raises(ValueError, match="api_url"): + HttpConfig(api_url=invalid_url) + + @pytest.mark.parametrize("bad_jwt", ["", " ", "\t"]) + def test_empty_or_whitespace_jwt_is_rejected(self, bad_jwt: str): + """Test that an empty or whitespace-only JWT fails at construction. + + Accepting it would make `has_jwt` report `True` while every call + still fails `Unauthenticated`, since the stored token is blank. + """ + with pytest.raises(ValueError, match="jwt"): + HttpConfig(jwt=bad_jwt) + + @pytest.mark.parametrize("out_of_range", [-1, 2**32]) + def test_out_of_range_retries_is_rejected(self, out_of_range: int): + """Test that a retry count outside the wire range names the argument. + + The conversion pyo3 does on its own raises OverflowError, which is not a + ValueError and so escapes the handler a caller wraps construction in. + """ + with pytest.raises(ValueError, match="retries"): + HttpConfig(retries=out_of_range) + + @pytest.mark.parametrize( + "negative", + [timedelta(microseconds=-1), timedelta(seconds=-1), timedelta(days=-1)], + ) + def test_negative_heartbeat_interval_is_rejected(self, negative: timedelta): + """Test that a negative heartbeat interval fails at construction.""" + with pytest.raises(ValueError, match="negative"): + HttpConfig(heartbeat_interval=negative) + + def test_zero_heartbeat_interval_is_rejected(self): + """Test that a zero heartbeat interval fails at construction. + + Nothing downstream reads zero as "disabled"; it heartbeats in a + continuous loop for as long as the client lives. + """ + with pytest.raises(ValueError, match=r"heartbeat_interval.*must not be zero"): + HttpConfig(heartbeat_interval=timedelta(0)) + + def test_maximum_heartbeat_interval_round_trips(self): + """Test that the largest timedelta survives the duration conversion. + + The repr is asserted in seconds rather than days: it is rendered from + a microsecond count, so the maximum comes back as a whole-second + `timedelta` instead of the `days=` form it was constructed with. + """ + maximum = timedelta(days=999_999_999) + + config = HttpConfig(heartbeat_interval=maximum) + + printed = repr(config) + + assert config.heartbeat_interval == maximum + assert ( + "heartbeat_interval=datetime.timedelta(seconds=86399999913600)" in printed + ) + ast.parse(printed) + + +@pytest.mark.unit +class TestHttpClientConstruction: + """Test that `IggyClient(...)` builds an HTTP client from an `HttpConfig`.""" + + @pytest.mark.asyncio + async def test_accepts_a_config(self): + """Test that the resulting client is actually HTTP, not silently TCP. + + `IggyClient(...)` is not None for either union arm, so that alone + never pinned the transport, and a bare `RuntimeError` does not either: + a client that regressed to the TCP arm raises too, just after hanging + until the pytest timeout. `Invalid HTTP request` is the HTTP + transport's own send failure, so matching it pins the transport. + `retries=0` keeps the failure immediate instead of working through + the default retry/backoff first. + """ + client = IggyClient(HttpConfig(api_url="http://127.0.0.1:1", retries=0)) + + with pytest.raises(RuntimeError, match="Invalid HTTP request"): + await client.ping() + + def test_accepts_the_default_config(self): + """Test that an explicit default `HttpConfig` is accepted.""" + assert IggyClient(HttpConfig()) is not None + + @pytest.mark.asyncio + async def test_without_a_jwt_a_privileged_call_is_unauthenticated(self): + """Test that a privileged call fails when no token is configured. + + `HttpClient` seeds its access token from `jwt`, so with none configured + and no `login_user()` it stays empty and the client rejects the call + itself. The dead port is what proves that: nothing is dialled, so the + failure cannot be the server's. + """ + client = IggyClient(HttpConfig(api_url="http://127.0.0.1:1", retries=0)) + await client.connect() + + with pytest.raises(RuntimeError, match="Unauthenticated"): + await client.create_stream("never-created") + + +@pytest.mark.integration +class TestHttpConfigAgainstServer: + """Test that a client built from `HttpConfig` actually connects.""" + + @pytest.mark.asyncio + async def test_client_connects_and_pings(self): + """Test that a client built with a custom config reaches the server.""" + host, port = get_http_server_config() + + client = IggyClient(HttpConfig(api_url=f"http://{host}:{port}")) + await client.connect() + await wait_for_ping(client) + + @pytest.mark.asyncio + async def test_client_sends_and_polls_a_message(self, unique_name): + """Test a full round trip: login, create stream/topic, send, poll. + + This is the part `test_client_connects_and_pings` above does not + cover: that a client built from `HttpConfig` can carry a real + workload, not just answer a ping. + """ + host, port = get_http_server_config() + stream_name = unique_name() + topic_name = unique_name() + payload = f"payload-{unique_name()}" + + client = IggyClient(HttpConfig(api_url=f"http://{host}:{port}")) + await client.connect() + await wait_for_ping(client) + await client.login_user("iggy", "iggy") + + await client.create_stream(stream_name) + await client.create_topic( + stream=stream_name, name=topic_name, partitions_count=1 + ) + await client.send_messages( + stream=stream_name, + topic=topic_name, + partitioning=0, + messages=[Message(payload)], + ) + + polled_messages = await client.poll_messages( + stream=stream_name, + topic=topic_name, + consumer=Consumer.Single("http-round-trip"), + partition_id=0, + polling_strategy=PollingStrategy.First(), + count=1, + auto_commit=True, + ) + + assert [message.payload().decode() for message in polled_messages] == [payload] + + @pytest.mark.asyncio + async def test_jwt_config_actually_authenticates(self, unique_name): + """Test that a JWT passed to `HttpConfig` reaches `access_token`. + + `has_jwt` only proves a token is configured, not that it works. + `/users/login` is unauthenticated, so a token minted out-of-band via + stdlib `urllib` (bypassing `HttpConfig` and `login_user()` entirely) + proves the client actually authenticates with the token it was given. + + The trailing newline is deliberate, and is the only coverage of the + trim in `HttpConfig::new`: it reproduces a token read from a file, and + untrimmed it builds a `Bearer \\n` header value that + `HeaderValue` rejects, failing every call with `Invalid HTTP request`. + Do not remove it. + """ + host, port = get_http_server_config() + api_url = f"http://{host}:{port}" + + request = urllib.request.Request( # noqa: S310 + f"{api_url}/users/login", + data=json.dumps({"username": "iggy", "password": "iggy"}).encode(), + headers={"Content-Type": "application/json"}, + method="POST", + ) + with urllib.request.urlopen(request) as response: # noqa: S310 + identity = json.loads(response.read()) + token = identity["access_token"]["token"] + + client = IggyClient(HttpConfig(api_url=api_url, jwt=f"{token}\n")) + await client.connect() + await wait_for_ping(client) + + stream_name = unique_name() + await client.create_stream(stream_name) + assert await client.get_stream(stream_name) is not None + + @pytest.mark.asyncio + async def test_wrong_jwt_is_rejected_by_the_server(self, unique_name): + """Test that a JWT the server cannot decode fails the call, not connect. + + `connect()` does not dial over HTTP, so a bad token can only surface + later. `wait_for_ping` runs first because ping needs no credentials: + it proves the server is reachable, which pins the failure below on the + token rather than on a missing listener. The server answers an + undecodable token with 401, which the HTTP client maps back to + `Unauthenticated`. + """ + host, port = get_http_server_config() + + client = IggyClient( + HttpConfig(api_url=f"http://{host}:{port}", jwt="not-a-real-token") + ) + await client.connect() + await wait_for_ping(client) + + with pytest.raises(RuntimeError, match="Unauthenticated"): + await client.create_stream(unique_name()) + + @pytest.mark.asyncio + async def test_consumer_group_is_rejected(self, unique_name): + """Test that a consumer group fails loudly over HTTP, not silently. + + `join_consumer_group` answers `Feature is unavailable` over HTTP, and + `consumer_group(...)` awaits that join before returning, so the + failure surfaces at construction. `auto_join_consumer_group=False` + is not a way around it: the join is skipped, but the first poll + raises the same error, because a group member is pinned to no + partition and a consumer-group poll without one is rejected + client-side. + """ + host, port = get_http_server_config() + stream_name = unique_name() + topic_name = unique_name() + + client = IggyClient(HttpConfig(api_url=f"http://{host}:{port}")) + await client.connect() + await wait_for_ping(client) + await client.login_user("iggy", "iggy") + + await client.create_stream(stream_name) + await client.create_topic( + stream=stream_name, name=topic_name, partitions_count=1 + ) + + with pytest.raises(RuntimeError, match="Feature is unavailable"): + await client.consumer_group( + name=unique_name(), stream=stream_name, topic=topic_name + ) diff --git a/foreign/python/tests/test_quic_config.py b/foreign/python/tests/test_quic_config.py index d0fdc3f7b6..eb4089a154 100644 --- a/foreign/python/tests/test_quic_config.py +++ b/foreign/python/tests/test_quic_config.py @@ -26,6 +26,7 @@ """ import ast +import socket from collections.abc import Callable from datetime import timedelta @@ -131,10 +132,21 @@ def test_very_long_interval_round_trips(self): assert reconnection.interval == timedelta(days=30_000) def test_maximum_interval_round_trips(self): - """Test that the largest timedelta survives the day conversion.""" - reconnection = QuicReconnectionConfig(interval=timedelta(days=999_999_999)) + """Test that the largest timedelta survives the day conversion. - assert reconnection.interval == timedelta(days=999_999_999) + The repr is asserted in seconds rather than days: it is rendered from + a microsecond count, so the maximum comes back as a whole-second + `timedelta` instead of the `days=` form it was constructed with. + """ + maximum = timedelta(days=999_999_999) + + reconnection = QuicReconnectionConfig(interval=maximum) + + printed = repr(reconnection) + + assert reconnection.interval == maximum + assert "interval=datetime.timedelta(seconds=86399999913600)" in printed + ast.parse(printed) @pytest.mark.unit @@ -210,8 +222,15 @@ def test_repr_hides_the_password(self): assert "secret" not in repr(config) def test_repr_shows_every_field_as_python(self): - """Test that repr covers the QUIC-specific fields and parses as Python.""" + """Test that repr covers the configured fields and parses as Python. + + The three string fields are asserted too: `ast.parse` alone still + passes on a repr that dropped one from the format string. + """ config = QuicConfig( + server_address="127.0.0.1:8081", + client_address="127.0.0.1:9000", + server_name="example.com", heartbeat_interval=timedelta(seconds=15), keep_alive_interval=timedelta(seconds=2), max_idle_timeout=timedelta(seconds=20), @@ -220,6 +239,9 @@ def test_repr_shows_every_field_as_python(self): printed = repr(config) + assert 'server_address="127.0.0.1:8081"' in printed + assert 'client_address="127.0.0.1:9000"' in printed + assert 'server_name="example.com"' in printed assert "validate_certificate=True" in printed assert "heartbeat_interval=datetime.timedelta(seconds=15)" in printed assert "keep_alive_interval=datetime.timedelta(seconds=2)" in printed @@ -231,8 +253,8 @@ def test_repr_shows_every_field_as_python(self): ["", "127.0.0.1", "127.0.0.1:not-a-port", "127.0.0.1:70000", "::1:8080"], ) def test_invalid_server_address_is_rejected(self, invalid_address: str): - """Test that a malformed address fails at construction, not at connect.""" - with pytest.raises(ValueError): + """Test that a malformed address fails at construction, naming itself.""" + with pytest.raises(ValueError, match="server_address"): QuicConfig(server_address=invalid_address) @pytest.mark.parametrize( @@ -251,6 +273,20 @@ def test_invalid_client_address_is_rejected(self, invalid_address: str): with pytest.raises(ValueError, match="client_address"): QuicConfig(client_address=invalid_address) + def test_surrounding_whitespace_in_the_addresses_is_trimmed(self): + """Test that both addresses tolerate whitespace, like HTTP's `api_url`. + + `client_address` is stored as the string `QuicClient::create` compares + against the literal default to pick an IPv6 bind address, so storing + the trimmed form is what keeps a padded default matching that sentinel. + """ + config = QuicConfig( + server_address=" 127.0.0.1:8080 ", client_address=" 127.0.0.1:0 " + ) + + assert config.server_address == "127.0.0.1:8080" + assert config.client_address == "127.0.0.1:0" + def test_negative_heartbeat_interval_is_rejected(self): """Test that a negative heartbeat interval fails at construction.""" with pytest.raises(ValueError, match="negative"): @@ -359,8 +395,32 @@ class TestQuicClientConstruction: """Test that `IggyClient(...)` accepts a `QuicConfig`.""" def test_accepts_a_config(self): - """Test that a client can be built from a config object.""" - assert IggyClient(QuicConfig(server_address="127.0.0.1:8080")) is not None + """Test that the resulting client is actually QUIC, not silently TCP. + + `IggyClient(...)` is not None for either union arm, so that alone never + pinned the transport. `client_address` is a QUIC-only field that + `QuicClient::create` binds eagerly here, so a port already held fails + the bind synchronously with `Cannot create endpoint`, with no server + and no privileged port involved. A client that regressed to the TCP arm + has no such field and would construct fine, and no other error maps to + that message. + + The held port has to be a real one: an unroutable address would only + fail the bind while `net.ipv4.ip_nonlocal_bind` is 0, and a host + running keepalived or a container setting it alone would bind + successfully and assert nothing. Neither the socket below nor quinn + sets `SO_REUSEADDR`, so the second bind is EADDRINUSE regardless. + """ + # A bindable client_address of its own, so the failure below is the + # collision and not the field being set at all. + assert IggyClient(QuicConfig(client_address="127.0.0.1:0")) is not None + + with socket.socket(socket.AF_INET, socket.SOCK_DGRAM) as held: + held.bind(("127.0.0.1", 0)) + port = held.getsockname()[1] + + with pytest.raises(RuntimeError, match="Cannot create endpoint"): + IggyClient(QuicConfig(client_address=f"127.0.0.1:{port}")) def test_accepts_the_default_config(self): """Test that an explicit default `QuicConfig` is accepted.""" diff --git a/foreign/python/tests/utils.py b/foreign/python/tests/utils.py index cdede9f32f..5d7654cbf4 100644 --- a/foreign/python/tests/utils.py +++ b/foreign/python/tests/utils.py @@ -32,6 +32,10 @@ MIN_PASSWORD_BYTES = 3 MAX_PASSWORD_BYTES = 100 +DEFAULT_TCP_PORT = 8090 +DEFAULT_QUIC_PORT = 8080 +DEFAULT_HTTP_PORT = 3000 + def get_transport_config(port_env_var: str, default_port: int) -> tuple[str, int]: """ @@ -69,7 +73,7 @@ def get_server_config() -> tuple[str, int]: Returns: tuple: (host, port) for the Iggy server """ - return get_transport_config("IGGY_SERVER_TCP_PORT", 8090) + return get_transport_config("IGGY_SERVER_TCP_PORT", DEFAULT_TCP_PORT) def get_quic_server_config() -> tuple[str, int]: @@ -79,7 +83,17 @@ def get_quic_server_config() -> tuple[str, int]: Returns: tuple: (host, port) for the Iggy server """ - return get_transport_config("IGGY_SERVER_QUIC_PORT", 8080) + return get_transport_config("IGGY_SERVER_QUIC_PORT", DEFAULT_QUIC_PORT) + + +def get_http_server_config() -> tuple[str, int]: + """ + Get HTTP server configuration from environment variables or defaults. + + Returns: + tuple: (host, port) for the Iggy HTTP API + """ + return get_transport_config("IGGY_SERVER_HTTP_PORT", DEFAULT_HTTP_PORT) def wait_for_server(host: str, port: int, timeout: int = 60, interval: int = 2) -> None: From 1004eee2d76943373384098d10825335383f8fad Mon Sep 17 00:00:00 2001 From: Ethan Lin <103916325+ethanlin01x@users.noreply.github.com> Date: Tue, 8 Sep 2026 02:01:03 +0800 Subject: [PATCH 083/182] ci(connectors): run a plugin's integration suite when it changes (#4077) Relates to #3996 --- .github/actions/rust/pre-merge/action.yml | 51 +++++++++++++++++++---- core/integration/Cargo.toml | 4 ++ 2 files changed, 47 insertions(+), 8 deletions(-) diff --git a/.github/actions/rust/pre-merge/action.yml b/.github/actions/rust/pre-merge/action.yml index ab23317bef..f02712e0f0 100644 --- a/.github/actions/rust/pre-merge/action.yml +++ b/.github/actions/rust/pre-merge/action.yml @@ -112,6 +112,10 @@ runs: # plugins (cdylib) are loaded at runtime via dlopen by iggy-connectors. # Both must always be compiled alongside affected crates. echo "$METADATA_JSON" | jq -r '.packages[] | select(.targets[] | .kind[] | (. == "bin" or . == "cdylib")) | .name' 2>/dev/null > /tmp/bin-packages.txt || true + # Connector plugins on their own. The same dlopen edge that keeps them + # out of the DAG also hides the integration suites that exercise them, + # which the filter below puts back. + echo "$METADATA_JSON" | jq -r '.packages[] | select(.targets[] | .kind[] | . == "cdylib") | .name' 2>/dev/null > /tmp/plugin-packages.txt || true PLAN_JSON=$(cargo rail plan --since origin/master -f json 2>/tmp/affected-stderr.txt || echo "") @@ -119,9 +123,34 @@ runs: MODE=$(echo "$PLAN_JSON" | jq -r '.scope.mode') if [[ "$MODE" == "crates" ]]; then CRATES=$(echo "$PLAN_JSON" | jq -r '.scope.crates[]') + # `iggy_connector__{sink,source}` is exercised by + # `integration::connectors::`, but the plugin reaches the + # runtime through dlopen, so cargo records no edge and the DAG drops + # those tests. Name the suite of every affected plugin that has one. + PLUGIN_TESTS="" + PLUGIN_SUITES=0 + while read -r plugin; do + if [[ ! -d "core/integration/tests/connectors/$plugin" ]]; then + echo "::warning::No integration suite at core/integration/tests/connectors/${plugin}, leaving it out of the test scope" + continue + fi + PLUGIN_TESTS+=" | (package(integration) & test(/^connectors::${plugin}::/))" + PLUGIN_SUITES=$(( PLUGIN_SUITES + 1 )) + done < <(comm -12 <(sort -u /tmp/plugin-packages.txt) <(echo "$CRATES" | sort -u) \ + | sed -nE 's/^iggy_connector_(.*)_(sink|source)$/\1/p' | sort -u) + echo "$PLUGIN_SUITES" > /tmp/plugin-suites.txt + # Those suites live in the integration test binary, so it has to be + # built even when the DAG left it out of scope. Only the build set + # grows: the `cargo test` fallback below has no filter, and adding + # the package there would run the whole integration suite. + BUILD_CRATES="$CRATES" + if (( PLUGIN_SUITES > 0 )); then + BUILD_CRATES=$(printf '%s\nintegration\n' "$CRATES" | sort -u) + fi CRATE_COUNT=$(echo "$CRATES" | wc -l) # Build nextest filter expression for cargo nextest run (affected crates only) - echo "$CRATES" | sed 's/^/package(/; s/$/)/' | paste -sd '|' | sed 's/|/ | /g' > /tmp/nextest-filter.txt + FILTER=$(echo "$CRATES" | sed 's/^/package(/; s/$/)/' | paste -sd '|' - | sed 's/|/ | /g') + echo "${FILTER}${PLUGIN_TESTS}" > /tmp/nextest-filter.txt # Save affected-only -p flags for cargo test fallback (no nextest filter) echo "$CRATES" | sed 's/^/-p /' | tr '\n' ' ' > /tmp/test-packages.txt # Build -p flags: affected crates + packages with bin/cdylib targets. @@ -129,17 +158,17 @@ runs: # artifacts; cdylib packages: connector plugins loaded via dlopen. # Both must be in the build even if not directly in the DAG scope. BIN_PKGS=$(cat /tmp/bin-packages.txt 2>/dev/null || echo "") - ALL_BUILD_PKGS=$(printf '%s\n%s\n' "$CRATES" "$BIN_PKGS" | sort -u | grep -v '^$') + ALL_BUILD_PKGS=$(printf '%s\n%s\n' "$BUILD_CRATES" "$BIN_PKGS" | sort -u | grep -v '^$') echo "$ALL_BUILD_PKGS" | sed 's/^/-p /' | tr '\n' ' ' > /tmp/packages.txt BUILD_COUNT=$(echo "$ALL_BUILD_PKGS" | wc -l) - echo "::notice::DAG analysis: testing ${CRATE_COUNT} crates, building ${BUILD_COUNT} (of ${TOTAL_CRATES} total, +$(( BUILD_COUNT - CRATE_COUNT )) binary pkgs)" + echo "::notice::DAG analysis: testing ${CRATE_COUNT} crates + ${PLUGIN_SUITES} connector suites, building ${BUILD_COUNT} (of ${TOTAL_CRATES} total, +$(( BUILD_COUNT - CRATE_COUNT )) extra pkgs)" else echo "::notice::Full workspace affected (${TOTAL_CRATES} crates)" fi else STDERR=$(cat /tmp/affected-stderr.txt 2>/dev/null || echo "") echo "::warning::Could not compute affected crates, running full test suite. ${STDERR}" - rm -f /tmp/nextest-filter.txt /tmp/packages.txt + rm -f /tmp/nextest-filter.txt /tmp/packages.txt /tmp/plugin-suites.txt fi shell: bash @@ -241,6 +270,7 @@ runs: PACKAGE_FLAGS="" TEST_PACKAGE_FLAGS="" TOTAL_CRATES="?" + PLUGIN_SUITES=0 if [[ -f /tmp/nextest-filter.txt ]]; then NEXTEST_FILTER=$(cat /tmp/nextest-filter.txt) fi @@ -253,11 +283,16 @@ runs: if [[ -f /tmp/total-crates.txt ]]; then TOTAL_CRATES=$(cat /tmp/total-crates.txt) fi + if [[ -f /tmp/plugin-suites.txt ]]; then + PLUGIN_SUITES=$(cat /tmp/plugin-suites.txt) + fi if [[ -n "$PACKAGE_FLAGS" ]]; then - TEST_CRATE_COUNT=$(echo "$NEXTEST_FILTER" | grep -o 'package(' | wc -l) + # Each connector suite contributes its own `package(integration)`, so + # subtract them to keep the crate count a crate count. + TEST_CRATE_COUNT=$(( $(echo "$NEXTEST_FILTER" | grep -o 'package(' | wc -l) - PLUGIN_SUITES )) BUILD_CRATE_COUNT=$(echo "$PACKAGE_FLAGS" | grep -o '\-p ' | wc -l) - echo "::notice::DAG-scoped: testing ${TEST_CRATE_COUNT} crates, building ${BUILD_CRATE_COUNT} (cargo check/clippy cover full workspace separately)" + echo "::notice::DAG-scoped: testing ${TEST_CRATE_COUNT} crates + ${PLUGIN_SUITES} connector suites, building ${BUILD_CRATE_COUNT} (cargo check/clippy cover full workspace separately)" else echo "::notice::Full workspace build (no DAG filter available)" fi @@ -389,9 +424,9 @@ runs: echo "" echo "=========================================" if [[ -n "$PACKAGE_FLAGS" ]]; then - TEST_CRATE_COUNT=$(echo "$NEXTEST_FILTER" | grep -o 'package(' | wc -l) + TEST_CRATE_COUNT=$(( $(echo "$NEXTEST_FILTER" | grep -o 'package(' | wc -l) - PLUGIN_SUITES )) BUILD_CRATE_COUNT=$(echo "$PACKAGE_FLAGS" | grep -o '\-p ' | wc -l) - echo "DAG scope (test): ${TEST_CRATE_COUNT}/${TOTAL_CRATES} crates" + echo "DAG scope (test): ${TEST_CRATE_COUNT}/${TOTAL_CRATES} crates + ${PLUGIN_SUITES} connector suites" echo "DAG scope (build): ${BUILD_CRATE_COUNT}/${TOTAL_CRATES} crates" else echo "DAG scope: full workspace (${TOTAL_CRATES} crates)" diff --git a/core/integration/Cargo.toml b/core/integration/Cargo.toml index f53fc13697..0450fa306d 100644 --- a/core/integration/Cargo.toml +++ b/core/integration/Cargo.toml @@ -60,6 +60,10 @@ iggy_binary_protocol = { workspace = true } iggy_common = { workspace = true } # Path-dep only so the Doris integration test can reuse the connector's pure # `build_label` function — keeping the test and production label format in lock-step. +# The only connector crate `integration` can depend on. Every plugin exports the +# same no_mangle `iggy_sink_*` symbols, so a second one fails the test binary +# link with duplicate symbols. Scoping a plugin's tests into a CI run needs no +# dependency. iggy_connector_doris_sink = { workspace = true } iggy_connector_sdk = { workspace = true, features = ["api"] } # Locates and decodes the on-disk superblock slot files in the recovery test From be3663805829492a41c971b3056f2da8680e2035 Mon Sep 17 00:00:00 2001 From: Gunther Xing Date: Tue, 8 Sep 2026 02:23:55 +0800 Subject: [PATCH 084/182] feat(python): support message partitioning strategies (#3927) Closes #3896 --- examples/python/README.md | 2 +- examples/python/basic/producer.py | 8 +- foreign/python/apache_iggy.pyi | 66 ++++++++- foreign/python/src/client.rs | 37 +++-- foreign/python/src/lib.rs | 3 + foreign/python/src/options.rs | 1 + foreign/python/src/partitioning.rs | 105 ++++++++++++++ .../python/tests/test_message_operations.py | 137 ++++++++++++++++++ 8 files changed, 341 insertions(+), 18 deletions(-) create mode 100644 foreign/python/src/partitioning.rs diff --git a/examples/python/README.md b/examples/python/README.md index 4e5c0c740d..e108105928 100644 --- a/examples/python/README.md +++ b/examples/python/README.md @@ -12,7 +12,7 @@ docker run --rm -p 8080:8080 -p 3000:3000 -p 8090:8090 \ -e IGGY_NODE_ADVERTISED_ADDRESS=localhost apache/iggy:latest # Or build from source (recommended for development) -cd ../../ && cargo run --bin iggy-server +cd ../../ && cargo run --bin iggy-server -- --with-default-root-credentials --fresh ``` For server configuration options and help: diff --git a/examples/python/basic/producer.py b/examples/python/basic/producer.py index cf3e4b84a1..13a4ae34ba 100644 --- a/examples/python/basic/producer.py +++ b/examples/python/basic/producer.py @@ -19,7 +19,7 @@ import asyncio from typing import NamedTuple -from apache_iggy import IggyClient, StreamDetails, TopicDetails +from apache_iggy import IggyClient, Partitioning, StreamDetails, TopicDetails from apache_iggy import SendMessage as Message from loguru import logger @@ -106,7 +106,11 @@ async def produce_messages(client: IggyClient): await client.send_messages( stream=STREAM_NAME, topic=TOPIC_NAME, - partitioning=PARTITION_ID, + # A fixed strategy sends the whole batch to this partition. + # For topics with multiple partitions, Partitioning.balanced() + # distributes batches round-robin, while + # Partitioning.messages_key(key) keeps equal keys on the same partition. + partitioning=Partitioning.partition_id(PARTITION_ID), messages=messages, ) n_sent_batches += 1 diff --git a/foreign/python/apache_iggy.pyi b/foreign/python/apache_iggy.pyi index 0400a1a514..ffed2857d0 100644 --- a/foreign/python/apache_iggy.pyi +++ b/foreign/python/apache_iggy.pyi @@ -46,6 +46,7 @@ __all__ = [ "MaxTopicSize", "OptionSpec", "Partition", + "Partitioning", "Permissions", "PollingStrategy", "QuicConfig", @@ -1570,15 +1571,32 @@ class IggyClient: self, stream: builtins.str | builtins.int, topic: builtins.str | builtins.int, - partitioning: builtins.int, + partitioning: Partitioning | builtins.int, messages: list[SendMessage], ) -> collections.abc.Awaitable[SendMessagesResponse]: r""" - Sends a list of messages to the specified topic. - Returns a SendMessagesResponse carrying the per-partition commit - confirmations, or a PyRuntimeError on failure. The confirmation list is - empty when the server reports no offsets, and the legacy server never - reports any. + Sends a batch of messages to a topic using the selected partitioning strategy. + + Args: + stream: Stream identifier as `str | int`. + topic: Topic identifier as `str | int`. + partitioning: A `Partitioning` strategy or an integer partition ID. + Use `Partitioning.balanced()`, `Partitioning.partition_id(id)`, or + `Partitioning.messages_key(key)`. An integer is shorthand for + `Partitioning.partition_id(id)`. + messages: Messages to send as `list[SendMessage]`. + + Returns: + An awaitable that resolves to `SendMessagesResponse`. Its confirmations + report the committed partition and batch base offset. The list is empty + when the server reports no offsets, including on the legacy server. + + Raises: + ValueError: If a string stream or topic identifier is invalid. + TypeError: If `partitioning` or `messages` has an unsupported type. + OverflowError: If a numeric stream, topic, or partition ID is outside + the supported unsigned 32-bit range. + RuntimeError: If the request fails. """ def poll_messages( self, @@ -1889,6 +1907,42 @@ class Partition: The number of messages in the partition. """ +@typing.final +class Partitioning: + r""" + Defines how a batch of messages is assigned to a topic partition. + """ + @staticmethod + def balanced() -> Partitioning: + r""" + Routes the batch to one partition selected by round-robin. + """ + @staticmethod + def partition_id(partition_id: builtins.int) -> Partitioning: + r""" + Routes the batch to the specified partition. + + `partition_id` must be between 0 and `2**32 - 1`. The topic must contain + that partition when the batch is sent. + + Raises: + TypeError: If `partition_id` is not an integer. + OverflowError: If `partition_id` is outside the supported unsigned + 32-bit range. + """ + @staticmethod + def messages_key(key: builtins.str | bytes) -> Partitioning: + r""" + Routes the batch to one partition selected by hashing `key`. + + `key` may be `str` or `bytes`. Strings are encoded as UTF-8; the encoded + key must contain between 1 and 255 bytes. + + Raises: + ValueError: If the encoded key is empty or exceeds 255 bytes. + TypeError: If `key` is not `str` or `bytes`. + """ + @typing.final class Permissions: r""" diff --git a/foreign/python/src/client.rs b/foreign/python/src/client.rs index 7d19de9954..168ef2f6d2 100644 --- a/foreign/python/src/client.rs +++ b/foreign/python/src/client.rs @@ -19,7 +19,7 @@ use bytes::Bytes; use iggy::prelude::{ AutoCommit as RustAutoCommit, Consumer as RustConsumer, IggyClient as RustIggyClient, IggyExpiry as RustIggyExpiry, IggyMessage as RustMessage, MaxTopicSize as RustMaxTopicSize, - PollingStrategy as RustPollingStrategy, *, + Partitioning as RustPartitioning, PollingStrategy as RustPollingStrategy, *, }; use pyo3::PyRef; use pyo3::prelude::*; @@ -40,6 +40,7 @@ use crate::consumer::{ use crate::duration::{py_delta_to_iggy_duration, reject_zero}; use crate::identifier::PyIdentifier; use crate::options::OptionSpec as PyOptionSpec; +use crate::partitioning::PyPartitioning; use crate::permissions::Permissions as PyPermissions; use crate::receive_message::{PollingStrategy, ReceiveMessage}; use crate::send_message::{SendMessage, SendMessagesResponse as PySendMessagesResponse}; @@ -1190,18 +1191,36 @@ impl IggyClient { }) } - /// Sends a list of messages to the specified topic. - /// Returns a SendMessagesResponse carrying the per-partition commit - /// confirmations, or a PyRuntimeError on failure. The confirmation list is - /// empty when the server reports no offsets, and the legacy server never - /// reports any. + /// Sends a batch of messages to a topic using the selected partitioning strategy. + /// + /// Args: + /// stream: Stream identifier as `str | int`. + /// topic: Topic identifier as `str | int`. + /// partitioning: A `Partitioning` strategy or an integer partition ID. + /// Use `Partitioning.balanced()`, `Partitioning.partition_id(id)`, or + /// `Partitioning.messages_key(key)`. An integer is shorthand for + /// `Partitioning.partition_id(id)`. + /// messages: Messages to send as `list[SendMessage]`. + /// + /// Returns: + /// An awaitable that resolves to `SendMessagesResponse`. Its confirmations + /// report the committed partition and batch base offset. The list is empty + /// when the server reports no offsets, including on the legacy server. + /// + /// Raises: + /// ValueError: If a string stream or topic identifier is invalid. + /// TypeError: If `partitioning` or `messages` has an unsupported type. + /// OverflowError: If a numeric stream, topic, or partition ID is outside + /// the supported unsigned 32-bit range. + /// RuntimeError: If the request fails. #[gen_stub(override_return_type(type_repr="collections.abc.Awaitable[SendMessagesResponse]", imports=("collections.abc")))] fn send_messages<'a>( &self, py: Python<'a>, stream: PyIdentifier, topic: PyIdentifier, - partitioning: u32, + #[gen_stub(override_type(type_repr = "Partitioning | builtins.int"))] + partitioning: PyPartitioning, #[gen_stub(override_type(type_repr = "list[SendMessage]"))] messages: &Bound<'_, PyList>, ) -> PyResult> { let messages: Vec = messages @@ -1218,7 +1237,7 @@ impl IggyClient { let stream = Identifier::try_from(stream)?; let topic = Identifier::try_from(topic)?; - let partitioning = Partitioning::partition_id(partitioning); + let partitioning = RustPartitioning::from(partitioning); let inner = self.inner.clone(); future_into_py(py, async move { @@ -1246,7 +1265,7 @@ impl IggyClient { polling_strategy: &PollingStrategy, count: u32, auto_commit: bool, - partition_id: Option, + #[gen_stub(override_type(type_repr = "builtins.int | None"))] partition_id: Option, ) -> PyResult> { let consumer = RustConsumer::try_from(consumer)?; let stream = Identifier::try_from(stream)?; diff --git a/foreign/python/src/lib.rs b/foreign/python/src/lib.rs index 80a123ee09..78b777918c 100644 --- a/foreign/python/src/lib.rs +++ b/foreign/python/src/lib.rs @@ -21,6 +21,7 @@ mod consumer; mod duration; mod identifier; mod options; +mod partitioning; mod permissions; mod receive_message; mod send_message; @@ -39,6 +40,7 @@ use consumer::{ ConsumerGroupMember, IggyConsumer, ReceiveMessageIterator, }; use options::OptionSpec; +use partitioning::Partitioning; use permissions::{GlobalPermissions, Permissions, StreamPermissions, TopicPermissions}; use pyo3::prelude::*; use receive_message::{PollingStrategy, ReceiveMessage}; @@ -73,6 +75,7 @@ fn apache_iggy(_py: Python, m: &Bound<'_, PyModule>) -> PyResult<()> { m.add_class::()?; m.add_class::()?; m.add_class::()?; + m.add_class::()?; m.add_class::()?; m.add_class::()?; m.add_class::()?; diff --git a/foreign/python/src/options.rs b/foreign/python/src/options.rs index fa489c5503..a0d4fa8095 100644 --- a/foreign/python/src/options.rs +++ b/foreign/python/src/options.rs @@ -64,6 +64,7 @@ impl OptionSpec { /// options ride that codec. #[gen_stub(override_return_type(type_repr = "HeaderValue | None"))] #[getter] + #[gen_stub(override_return_type(type_repr = "HeaderValue | None"))] pub fn default_value<'a>(&self, py: Python<'a>) -> PyResult>> { if self.inner.default_value.is_empty() { return Ok(None); diff --git a/foreign/python/src/partitioning.rs b/foreign/python/src/partitioning.rs new file mode 100644 index 0000000000..83d9715d8c --- /dev/null +++ b/foreign/python/src/partitioning.rs @@ -0,0 +1,105 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use iggy::prelude::Partitioning as RustPartitioning; +use pyo3::{exceptions::PyValueError, prelude::*, types::PyBytes}; +use pyo3_stub_gen::{ + derive::{gen_stub_pyclass, gen_stub_pymethods}, + impl_stub_type, +}; + +/// Defines how a batch of messages is assigned to a topic partition. +#[derive(Clone)] +#[pyclass(from_py_object)] +#[gen_stub_pyclass] +pub struct Partitioning { + pub(crate) inner: RustPartitioning, +} + +#[gen_stub_pymethods] +#[pymethods] +impl Partitioning { + /// Routes the batch to one partition selected by round-robin. + #[staticmethod] + pub fn balanced() -> Self { + Self { + inner: RustPartitioning::balanced(), + } + } + + /// Routes the batch to the specified partition. + /// + /// `partition_id` must be between 0 and `2**32 - 1`. The topic must contain + /// that partition when the batch is sent. + /// + /// Raises: + /// TypeError: If `partition_id` is not an integer. + /// OverflowError: If `partition_id` is outside the supported unsigned + /// 32-bit range. + #[staticmethod] + pub fn partition_id(partition_id: u32) -> Self { + Self { + inner: RustPartitioning::partition_id(partition_id), + } + } + + /// Routes the batch to one partition selected by hashing `key`. + /// + /// `key` may be `str` or `bytes`. Strings are encoded as UTF-8; the encoded + /// key must contain between 1 and 255 bytes. + /// + /// Raises: + /// ValueError: If the encoded key is empty or exceeds 255 bytes. + /// TypeError: If `key` is not `str` or `bytes`. + #[staticmethod] + pub fn messages_key(py: Python<'_>, key: PyMessagesKey) -> PyResult { + let key = match key { + PyMessagesKey::String(key) => key.into_bytes(), + PyMessagesKey::Bytes(key) => key.extract::>(py)?, + }; + let inner = RustPartitioning::messages_key(&key) + .map_err(|error| PyValueError::new_err(error.to_string()))?; + Ok(Self { inner }) + } +} + +#[derive(FromPyObject)] +pub enum PyMessagesKey { + #[pyo3(transparent, annotation = "str")] + String(String), + #[pyo3(transparent, annotation = "bytes")] + Bytes(Py), +} +impl_stub_type!(PyMessagesKey = String | PyBytes); + +#[derive(FromPyObject)] +pub(crate) enum PyPartitioning { + #[pyo3(transparent, annotation = "Partitioning")] + Strategy(Partitioning), + #[pyo3(transparent, annotation = "int")] + PartitionId(u32), +} +impl_stub_type!(PyPartitioning = Partitioning | isize); + +impl From for RustPartitioning { + fn from(partitioning: PyPartitioning) -> Self { + match partitioning { + PyPartitioning::Strategy(partitioning) => partitioning.inner, + PyPartitioning::PartitionId(partition_id) => Self::partition_id(partition_id), + } + } +} diff --git a/foreign/python/tests/test_message_operations.py b/foreign/python/tests/test_message_operations.py index bbf5b1d0e8..af5924c4dc 100644 --- a/foreign/python/tests/test_message_operations.py +++ b/foreign/python/tests/test_message_operations.py @@ -25,12 +25,54 @@ HeaderKey, HeaderValue, IggyClient, + Partitioning, PollingStrategy, UserHeaders, ) from apache_iggy import SendMessage as Message +class TestPartitioning: + """Test message partitioning strategy construction.""" + + @pytest.mark.unit + def test_balanced_and_partition_id_strategies(self): + assert isinstance(Partitioning.balanced(), Partitioning) + assert isinstance(Partitioning.partition_id(1), Partitioning) + assert isinstance(Partitioning.partition_id(2**32 - 1), Partitioning) + + @pytest.mark.unit + @pytest.mark.parametrize("key", [b"customer-42", "customer-42", "客户-42"]) + def test_messages_key_accepts_bytes_and_strings(self, key): + assert isinstance(Partitioning.messages_key(key), Partitioning) + + @pytest.mark.unit + @pytest.mark.parametrize("key", [b"a" * 255, "a" * 255, "界" * 85]) + def test_messages_key_accepts_255_bytes(self, key): + assert isinstance(Partitioning.messages_key(key), Partitioning) + + @pytest.mark.unit + @pytest.mark.parametrize("key", [b"", "", b"a" * 256, "a" * 256, "界" * 86]) + def test_messages_key_rejects_invalid_encoded_length(self, key): + with pytest.raises(ValueError): + Partitioning.messages_key(key) + + @pytest.mark.unit + @pytest.mark.parametrize("partition_id", [-1, 2**32]) + def test_partition_id_rejects_values_outside_u32(self, partition_id): + with pytest.raises(OverflowError): + Partitioning.partition_id(partition_id) + + @pytest.mark.unit + def test_partitioning_rejects_invalid_types(self): + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-type + Partitioning.partition_id("0") + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-type + Partitioning.messages_key(1) + + class TestMessageOperations: """Test message sending, polling, and processing.""" @@ -112,6 +154,101 @@ async def test_send_messages_reports_committed_confirmation( assert confirmation.partition_id == partition_id assert confirmation.base_offset == polled_messages[0].offset() + @pytest.mark.asyncio + async def test_send_messages_with_partition_id_strategy( + self, iggy_client: IggyClient, unique_name + ): + stream_name = unique_name() + topic_name = unique_name() + partition_id = 2 + + await iggy_client.create_stream(stream_name) + await iggy_client.create_topic( + stream=stream_name, name=topic_name, partitions_count=3 + ) + + response = await iggy_client.send_messages( + stream=stream_name, + topic=topic_name, + partitioning=Partitioning.partition_id(partition_id), + messages=[Message("fixed partition 1"), Message("fixed partition 2")], + ) + + assert len(response.confirmations) == 1 + assert response.confirmations[0].partition_id == partition_id + + with pytest.raises(RuntimeError): + await iggy_client.send_messages( + stream=stream_name, + topic=topic_name, + partitioning=Partitioning.partition_id(3), + messages=[Message("missing partition")], + ) + + @pytest.mark.asyncio + async def test_send_messages_with_balanced_strategy( + self, iggy_client: IggyClient, unique_name + ): + stream_name = unique_name() + topic_name = unique_name() + partitions_count = 3 + + await iggy_client.create_stream(stream_name) + await iggy_client.create_topic( + stream=stream_name, + name=topic_name, + partitions_count=partitions_count, + ) + + responses = [ + await iggy_client.send_messages( + stream=stream_name, + topic=topic_name, + partitioning=Partitioning.balanced(), + messages=[Message(f"balanced {index}")], + ) + for index in range(partitions_count) + ] + + assert all(len(response.confirmations) == 1 for response in responses) + assert ( + len({response.confirmations[0].partition_id for response in responses}) + == partitions_count + ) + + @pytest.mark.asyncio + @pytest.mark.parametrize("key", [b"customer-42", "customer-42"]) + async def test_send_messages_with_same_key_uses_same_partition( + self, iggy_client: IggyClient, unique_name, key + ): + stream_name = unique_name() + topic_name = unique_name() + + await iggy_client.create_stream(stream_name) + await iggy_client.create_topic( + stream=stream_name, name=topic_name, partitions_count=3 + ) + partitioning = Partitioning.messages_key(key) + + first = await iggy_client.send_messages( + stream=stream_name, + topic=topic_name, + partitioning=partitioning, + messages=[Message("first")], + ) + second = await iggy_client.send_messages( + stream=stream_name, + topic=topic_name, + partitioning=partitioning, + messages=[Message("second")], + ) + + assert len(first.confirmations) == 1 + assert len(second.confirmations) == 1 + assert ( + first.confirmations[0].partition_id == second.confirmations[0].partition_id + ) + @pytest.mark.asyncio async def test_send_and_poll_messages_as_bytes( self, iggy_client: IggyClient, unique_name From 3c1a8ccf65eb67a57f4e92ffaeb92dd2df506765 Mon Sep 17 00:00:00 2001 From: saie-ch <132209179+saie-ch@users.noreply.github.com> Date: Tue, 8 Sep 2026 00:56:38 +0530 Subject: [PATCH 085/182] feat(python): add WebSocketConfig transport configuration (#4000) --- .../python-maturin/pre-merge/action.yml | 5 +- .github/workflows/coverage-baseline.yml | 5 +- .../websocket_client_config.rs | 75 ++- core/sdk/src/prelude.rs | 4 +- examples/python/getting-started/consumer.py | 17 +- examples/python/getting-started/producer.py | 17 +- foreign/python/README.md | 11 +- foreign/python/apache_iggy.pyi | 209 ++++++- foreign/python/docker-compose.test.yml | 2 + foreign/python/src/client.rs | 21 +- foreign/python/src/config.rs | 529 +++++++++++++++++- foreign/python/src/lib.rs | 4 + foreign/python/tests/test_websocket_config.py | 522 +++++++++++++++++ foreign/python/tests/utils.py | 11 + 14 files changed, 1388 insertions(+), 44 deletions(-) create mode 100644 foreign/python/tests/test_websocket_config.py diff --git a/.github/actions/python-maturin/pre-merge/action.yml b/.github/actions/python-maturin/pre-merge/action.yml index 53bf430dc0..d8ab37ad64 100644 --- a/.github/actions/python-maturin/pre-merge/action.yml +++ b/.github/actions/python-maturin/pre-merge/action.yml @@ -175,8 +175,8 @@ runs: # TCP and HTTP ports come from server-start's own outputs rather than # duplicated literals, so a future default change here can't silently - # desync from what the server actually started with. QUIC stays a - # literal below because server-start exposes no output for it. + # desync from what the server actually started with. QUIC and WebSocket + # stay literals below because server-start exposes no output for them. tcp_port="${{ steps.iggy.outputs.tcp_address }}" http_port="${{ steps.iggy.outputs.http_address }}" tcp_port="${tcp_port##*:}" @@ -191,6 +191,7 @@ runs: IGGY_SERVER_TCP_PORT="$tcp_port" \ IGGY_SERVER_QUIC_PORT=8080 \ IGGY_SERVER_HTTP_PORT="$http_port" \ + IGGY_SERVER_WS_PORT=8092 \ IGGY_SERVER_DOCKER_IMAGE=iggy-server:local \ uv run --no-sync pytest tests/ -v \ --junitxml=../../reports/python-junit.xml \ diff --git a/.github/workflows/coverage-baseline.yml b/.github/workflows/coverage-baseline.yml index 76d12d1b01..b4bd7c3156 100644 --- a/.github/workflows/coverage-baseline.yml +++ b/.github/workflows/coverage-baseline.yml @@ -350,8 +350,8 @@ jobs: # TCP and HTTP ports come from server-start's own outputs rather than # duplicated literals, so a future default change here can't silently - # desync from what the server actually started with. QUIC stays a - # literal below because server-start exposes no output for it. + # desync from what the server actually started with. QUIC and WebSocket + # stay literals below because server-start exposes no output for them. tcp_port="${{ steps.iggy.outputs.tcp_address }}" http_port="${{ steps.iggy.outputs.http_address }}" tcp_port="${tcp_port##*:}" @@ -361,6 +361,7 @@ jobs: IGGY_SERVER_TCP_PORT="$tcp_port" \ IGGY_SERVER_QUIC_PORT=8080 \ IGGY_SERVER_HTTP_PORT="$http_port" \ + IGGY_SERVER_WS_PORT=8092 \ IGGY_SERVER_DOCKER_IMAGE=iggy-server:local \ uv run --no-sync pytest tests/ -v \ --junitxml=../../reports/python-junit.xml \ diff --git a/core/common/src/types/configuration/websocket_config/websocket_client_config.rs b/core/common/src/types/configuration/websocket_config/websocket_client_config.rs index 4295ab5270..6fdd791796 100644 --- a/core/common/src/types/configuration/websocket_config/websocket_client_config.rs +++ b/core/common/src/types/configuration/websocket_config/websocket_client_config.rs @@ -55,9 +55,13 @@ pub struct WebSocketConfig { pub write_buffer_size: Option, /// Maximum write buffer size. pub max_write_buffer_size: Option, - /// Maximum message size. + /// Maximum message size, or `None` for no limit. Defaults to tungstenite's + /// own limit rather than `None`, so `None` always means the limit was lifted + /// deliberately. pub max_message_size: Option, - /// Maximum frame size. + /// Maximum frame size, or `None` for no limit. Defaults to tungstenite's own + /// limit rather than `None`, so `None` always means the limit was lifted + /// deliberately. pub max_frame_size: Option, /// Accept unmasked frames (client should typically keep as false). pub accept_unmasked_frames: bool, @@ -111,13 +115,13 @@ impl WebSocketConfig { config = config.max_write_buffer_size(max_write_buf_size); } - if let Some(max_msg_size) = self.max_message_size { - config = config.max_message_size(Some(max_msg_size)); - } - - if let Some(max_frame_size) = self.max_frame_size { - config = config.max_frame_size(Some(max_frame_size)); - } + // Set unconditionally, unlike the buffer sizes above: `None` here means + // "no limit", not "unset". `Default` seeds both from tungstenite's own + // values, so a `None` can only come from a caller that asked for the + // limit to be lifted, and skipping the setter would silently leave + // tungstenite's default in force instead. + config = config.max_message_size(self.max_message_size); + config = config.max_frame_size(self.max_frame_size); config = config.accept_unmasked_frames(self.accept_unmasked_frames); @@ -189,3 +193,56 @@ impl Display for WebSocketConfig { ) } } + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn default_should_carry_tungstenites_own_size_limits() { + let config = WebSocketConfig::default(); + let tungstenite = TungsteniteConfig::default(); + + assert_eq!(config.max_message_size, tungstenite.max_message_size); + assert_eq!(config.max_frame_size, tungstenite.max_frame_size); + assert!(config.max_message_size.is_some()); + assert!(config.max_frame_size.is_some()); + } + + #[test] + fn none_size_limits_should_reach_tungstenite_as_no_limit() { + let config = WebSocketConfig { + max_message_size: None, + max_frame_size: None, + ..Default::default() + }; + + let tungstenite = config.to_tungstenite_config(); + + assert_eq!(tungstenite.max_message_size, None); + assert_eq!(tungstenite.max_frame_size, None); + } + + #[test] + fn default_size_limits_should_reach_tungstenite_unchanged() { + let tungstenite = WebSocketConfig::default().to_tungstenite_config(); + let expected = TungsteniteConfig::default(); + + assert_eq!(tungstenite.max_message_size, expected.max_message_size); + assert_eq!(tungstenite.max_frame_size, expected.max_frame_size); + } + + #[test] + fn explicit_size_limits_should_reach_tungstenite_unchanged() { + let config = WebSocketConfig { + max_message_size: Some(1024), + max_frame_size: Some(512), + ..Default::default() + }; + + let tungstenite = config.to_tungstenite_config(); + + assert_eq!(tungstenite.max_message_size, Some(1024)); + assert_eq!(tungstenite.max_frame_size, Some(512)); + } +} diff --git a/core/sdk/src/prelude.rs b/core/sdk/src/prelude.rs index 79c6774b2b..f239eec672 100644 --- a/core/sdk/src/prelude.rs +++ b/core/sdk/src/prelude.rs @@ -67,8 +67,8 @@ pub use iggy_common::{ TcpClientConfigBuilder, TcpClientReconnectionConfig, Topic, TopicCreateOptions, TopicDetails, TopicPermissions, TopicUpdateOptions, TransportEndpoints, TransportProtocol, UserId, UserInfo, UserInfoDetails, UserStatus, UserUpdateOptions, Validatable, WebSocketClientConfig, - WebSocketClientConfigBuilder, WebSocketClientReconnectionConfig, defaults, locking, - topic_option_keys, + WebSocketClientConfigBuilder, WebSocketClientReconnectionConfig, WebSocketConfig, defaults, + locking, topic_option_keys, }; pub use iggy_common::{ Client, ClusterClient, ConsumerGroupClient, ConsumerOffsetClient, MessageClient, diff --git a/examples/python/getting-started/consumer.py b/examples/python/getting-started/consumer.py index 3a48f2a2e6..16405d9d42 100755 --- a/examples/python/getting-started/consumer.py +++ b/examples/python/getting-started/consumer.py @@ -31,6 +31,7 @@ ReceiveMessage, TcpConfig, TcpReconnectionConfig, + WebSocketConfig, ) from loguru import logger @@ -103,7 +104,9 @@ def parse_args() -> ArgNamespace: return ArgNamespace(**vars(args)) -def build_config(args: ArgNamespace) -> TcpConfig | QuicConfig | HttpConfig: +def build_config( + args: ArgNamespace, +) -> TcpConfig | QuicConfig | HttpConfig | WebSocketConfig: """Build the client configuration, TCP with auto-login and reconnection.""" # IggyClient(...) also accepts a QuicConfig for the QUIC transport. To use @@ -125,6 +128,18 @@ def build_config(args: ArgNamespace) -> TcpConfig | QuicConfig | HttpConfig: # after connecting: # return HttpConfig(api_url="http://127.0.0.1:3000") + # IggyClient(...) also accepts a WebSocketConfig for the WebSocket transport. + # To use it, uncomment the return below and import WebSocketReconnectionConfig, + # which is left out above because only the commented block names it: + # + # return WebSocketConfig( + # server_address="127.0.0.1:8092", + # auto_login=AutoLogin.username_password(args.username, args.password), + # reconnection=WebSocketReconnectionConfig( + # enabled=True, interval=timedelta(seconds=1) + # ), + # ) + return TcpConfig( server_address=args.tcp_server_address, auto_login=AutoLogin.username_password(args.username, args.password), diff --git a/examples/python/getting-started/producer.py b/examples/python/getting-started/producer.py index e8a8fea35b..2cde76e571 100755 --- a/examples/python/getting-started/producer.py +++ b/examples/python/getting-started/producer.py @@ -30,6 +30,7 @@ TcpConfig, TcpReconnectionConfig, TopicDetails, + WebSocketConfig, ) from apache_iggy import SendMessage as Message from loguru import logger @@ -102,7 +103,9 @@ def parse_args() -> ArgNamespace: return ArgNamespace(**vars(args)) -def build_config(args: ArgNamespace) -> TcpConfig | QuicConfig | HttpConfig: +def build_config( + args: ArgNamespace, +) -> TcpConfig | QuicConfig | HttpConfig | WebSocketConfig: """Build the client configuration, TCP with auto-login and reconnection.""" # IggyClient(...) also accepts a QuicConfig for the QUIC transport. To use @@ -124,6 +127,18 @@ def build_config(args: ArgNamespace) -> TcpConfig | QuicConfig | HttpConfig: # after connecting: # return HttpConfig(api_url="http://127.0.0.1:3000") + # IggyClient(...) also accepts a WebSocketConfig for the WebSocket transport. + # To use it, uncomment the return below and import WebSocketReconnectionConfig, + # which is left out above because only the commented block names it: + # + # return WebSocketConfig( + # server_address="127.0.0.1:8092", + # auto_login=AutoLogin.username_password(args.username, args.password), + # reconnection=WebSocketReconnectionConfig( + # enabled=True, interval=timedelta(seconds=1) + # ), + # ) + return TcpConfig( server_address=args.tcp_server_address, auto_login=AutoLogin.username_password(args.username, args.password), diff --git a/foreign/python/README.md b/foreign/python/README.md index 293bbdfbc3..b4a33a1167 100644 --- a/foreign/python/README.md +++ b/foreign/python/README.md @@ -134,8 +134,8 @@ running prek / committing / pushing. This list is not exhaustive and other hook ## Client Configuration -`IggyClient` takes a server address, a `TcpConfig`, a `QuicConfig`, or an -`HttpConfig`: +`IggyClient` takes a server address, a `TcpConfig`, a `QuicConfig`, an +`HttpConfig`, or a `WebSocketConfig`: ```python import asyncio @@ -169,9 +169,10 @@ async def main(): asyncio.run(main()) ``` -`IggyClient(...)` also accepts a `QuicConfig` for the QUIC transport and an -`HttpConfig` for the HTTP transport; -`examples/python/getting-started/producer.py` shows either swap in context. +`IggyClient(...)` also accepts a `QuicConfig` for the QUIC transport, an +`HttpConfig` for the HTTP transport, and a `WebSocketConfig` for the WebSocket +transport. `examples/python/getting-started/producer.py` shows each swap in +context. `HttpConfig` differs from TCP in two ways. There is no reconnection policy and no `AutoLogin`: `connect()` does not dial over HTTP, but it does start the diff --git a/foreign/python/apache_iggy.pyi b/foreign/python/apache_iggy.pyi index ffed2857d0..901af64aa4 100644 --- a/foreign/python/apache_iggy.pyi +++ b/foreign/python/apache_iggy.pyi @@ -68,6 +68,9 @@ __all__ = [ "UserInfo", "UserInfoDetails", "UserStatus", + "WebSocketConfig", + "WebSocketFramingConfig", + "WebSocketReconnectionConfig", ] class AutoCommit: @@ -975,21 +978,27 @@ class IggyClient: It provides asynchronous functionality through the contained runtime. """ def __new__( - cls, conn: TcpConfig | QuicConfig | HttpConfig | builtins.str | None = None + cls, + conn: TcpConfig + | QuicConfig + | HttpConfig + | WebSocketConfig + | builtins.str + | None = None, ) -> IggyClient: r""" Constructs a new IggyClient from a TCP server address, a `TcpConfig`, a - `QuicConfig`, or an `HttpConfig`. This initializes a new runtime for - asynchronous operations. + `QuicConfig`, an `HttpConfig`, or a `WebSocketConfig`. This initializes a + new runtime for asynchronous operations. Future versions might utilize asyncio for more Pythonic async. Args: - conn: A `host:port` address, a `TcpConfig`, a `QuicConfig`, or an - `HttpConfig`. Defaults to `127.0.0.1:8090` over TCP with auto-login - disabled. A malformed address is reported differently depending on - the form: the string form raises `RuntimeError` here, while - `TcpConfig`/`QuicConfig`/`HttpConfig` raise `ValueError` when they - are constructed, before any of them ever reaches this call. Neither + conn: A `host:port` address, a `TcpConfig`, a `QuicConfig`, an + `HttpConfig`, or a `WebSocketConfig`. Defaults to `127.0.0.1:8090` + over TCP with auto-login disabled. A malformed address is reported + differently depending on the form: the string form raises + `RuntimeError` here, while every config type raises `ValueError` + when it is constructed, before any of them reaches this call. Neither exception is a subclass of the other. Raises: @@ -2094,7 +2103,7 @@ class QuicConfig: seconds) instead, since `configure()` skips the setter entirely when zero. Defaults to 10 seconds. validate_certificate: Whether to validate the server certificate. Defaults - to disabled, unlike the TCP and WebSocket transports. + to disabled; only the TCP transport validates by default. Raises: ValueError: If `server_address` or `client_address` is not a valid @@ -3002,6 +3011,186 @@ class UserInfoDetails: The permissions of the user, or `None` when the user has none assigned. """ +@typing.final +class WebSocketConfig: + r""" + Configuration for the WebSocket transport, accepted by `IggyClient(...)`. + + Every field is keyword-only and optional. + """ + @property + def server_address(self) -> builtins.str: ... + @property + def auto_login(self) -> AutoLogin: ... + @property + def reconnection(self) -> WebSocketReconnectionConfig: ... + @property + def heartbeat_interval(self) -> datetime.timedelta: ... + @property + def framing(self) -> WebSocketFramingConfig: ... + @property + def tls_enabled(self) -> builtins.bool: ... + @property + def tls_domain(self) -> builtins.str: ... + @property + def tls_ca_file(self) -> builtins.str | None: ... + @property + def tls_validate_certificate(self) -> builtins.bool: ... + def __new__( + cls, + *, + server_address: builtins.str | None = None, + auto_login: AutoLogin | None = None, + reconnection: WebSocketReconnectionConfig | None = None, + heartbeat_interval: datetime.timedelta | None = None, + framing: WebSocketFramingConfig | None = None, + tls_enabled: builtins.bool | None = None, + tls_domain: builtins.str | None = None, + tls_ca_file: builtins.str | None = None, + tls_validate_certificate: builtins.bool | None = None, + ) -> WebSocketConfig: + r""" + Constructs a WebSocket configuration. + + Args: + server_address: `host:port` of the Iggy server. Defaults to `127.0.0.1:8092`. + auto_login: Credentials replayed on every connect. Defaults to `AutoLogin.disabled()`. + reconnection: Reconnection policy. Defaults to `WebSocketReconnectionConfig()`. + heartbeat_interval: Interval of heartbeats sent by the client. Defaults to 5 seconds. + framing: Frame- and buffer-level options. Defaults to `WebSocketFramingConfig()`. + tls_enabled: Whether to connect over TLS. Defaults to disabled. + tls_domain: Domain to validate the certificate against. Defaults to + `localhost`. Empty means it is taken from the IP `server_address` + resolves to. + tls_ca_file: Path to the CA file for TLS. Read only when `tls_enabled` + and `tls_validate_certificate` are both on; with either one off it + is kept but never consulted, so pairing it with + `tls_validate_certificate=False` pins nothing. + tls_validate_certificate: Whether to validate the server certificate. + Defaults to `False`; only the TCP transport validates by default. + Disabling this accepts any certificate the server presents, + including self-signed and mismatched ones, and takes precedence + over `tls_ca_file`. + + Raises: + ValueError: If `server_address` is not a valid `host:port` pair, if a + duration is negative, or if `heartbeat_interval` is zero. + """ + def __repr__(self) -> builtins.str: ... + +@typing.final +class WebSocketFramingConfig: + r""" + Frame- and buffer-level options passed through to the underlying WebSocket + implementation, accepted by `WebSocketConfig`'s `framing` argument. + + Every field is keyword-only and optional; unset fields fall back to the + underlying WebSocket library's own defaults. + """ + @property + def read_buffer_size(self) -> builtins.int | None: ... + @property + def write_buffer_size(self) -> builtins.int | None: ... + @property + def max_write_buffer_size(self) -> builtins.int | None: ... + @property + def max_message_size(self) -> builtins.int | None: ... + @property + def max_frame_size(self) -> builtins.int | None: ... + @property + def accept_unmasked_frames(self) -> builtins.bool: ... + def __new__( + cls, + *, + read_buffer_size: builtins.int | None = None, + write_buffer_size: builtins.int | None = None, + max_write_buffer_size: builtins.int | None = None, + max_message_size: builtins.int | None = 64 << 20, + max_frame_size: builtins.int | None = 16 << 20, + accept_unmasked_frames: builtins.bool | None = None, + ) -> WebSocketFramingConfig: + r""" + Constructs a WebSocket framing configuration. + + Args: + read_buffer_size: Read buffer size in bytes. Defaults to 128 KiB. + write_buffer_size: Write buffer size in bytes. Defaults to 128 KiB. + max_write_buffer_size: Maximum write buffer size in bytes. Defaults to + unbounded, which reads back as the largest value a pointer-sized + unsigned integer holds rather than as `None`. + max_message_size: Maximum message size in bytes, or an explicit `None` + to lift the limit entirely. Omitting the argument is not the same + as passing `None`: it keeps the underlying default of 64 MiB. + Lifting the limit lets a peer queue an arbitrarily large message + in memory, so prefer a finite value. + max_frame_size: Maximum frame size in bytes, or an explicit `None` to + lift the limit entirely. Omitting the argument keeps the + underlying default of 16 MiB, with the same caveat as + `max_message_size`. + accept_unmasked_frames: Whether to accept unmasked frames. Defaults to + `False`; clients should typically keep this off for RFC compliance. + + Raises: + ValueError: If a numeric field is outside the range of a pointer-sized + unsigned integer, or if `max_write_buffer_size` does not come out + greater than `write_buffer_size`. tungstenite enforces the same + invariant with an `assert!` at connect time, which would otherwise + surface as an unrecoverable Rust panic instead of a `ValueError`. + OverflowError: If a numeric field does not fit a signed 128-bit integer, + raised by the underlying conversion before this constructor runs. + """ + def __repr__(self) -> builtins.str: ... + +@typing.final +class WebSocketReconnectionConfig: + r""" + How the WebSocket client reconnects after the connection to the server is lost. + """ + @property + def enabled(self) -> builtins.bool: ... + @property + def max_retries(self) -> builtins.int | None: ... + @property + def interval(self) -> datetime.timedelta: ... + @property + def reestablish_after(self) -> datetime.timedelta: ... + def __new__( + cls, + *, + enabled: builtins.bool | None = None, + max_retries: builtins.int | None = None, + interval: datetime.timedelta | None = None, + reestablish_after: datetime.timedelta | None = None, + ) -> WebSocketReconnectionConfig: + r""" + Constructs a reconnection policy. + + Args: + enabled: Whether to reconnect at all. Defaults to enabled. + max_retries: Redials of the configured server address after the first + attempt, or `None` for unlimited; `0` still makes that first + attempt. Unlike the TCP transport, WebSocket redials the one + address it was configured with rather than walking a cluster + roster, so this counts dials. Defaults to unlimited, which means + a call awaited while the server is down never returns: + `connect()` waits inside the retry loop, as do `send_messages()` + and `poll_messages()` once auto-login is configured. Set a finite + number for request/reply style usage, so a call fails instead. + interval: Delay before each redial. Defaults to 1 second. + reestablish_after: Cooldown before redialing after a previously + successful connection, measured from when it was established, so + a session that outlived the interval is redialed at once. Applied + from the first redial onward, not to the initial connect. + Defaults to 5 seconds. + + Raises: + ValueError: If a duration is negative, if `max_retries` is outside the + range of an unsigned 32-bit integer, or if `interval` is zero. + OverflowError: If `max_retries` does not fit a signed 64-bit integer, + raised by the underlying conversion before this constructor runs. + """ + def __repr__(self) -> builtins.str: ... + @typing.final class UserStatus(enum.Enum): r""" diff --git a/foreign/python/docker-compose.test.yml b/foreign/python/docker-compose.test.yml index 665df9121c..c16eded3cc 100644 --- a/foreign/python/docker-compose.test.yml +++ b/foreign/python/docker-compose.test.yml @@ -33,6 +33,7 @@ services: - "3000:3000" - "8080:8080" - "8090:8090" + - "8092:8092" environment: - IGGY_HTTP_ADDRESS=0.0.0.0:3000 - IGGY_TCP_ADDRESS=0.0.0.0:8090 @@ -65,6 +66,7 @@ services: - IGGY_SERVER_TCP_PORT=8090 - IGGY_SERVER_HTTP_PORT=3000 - IGGY_SERVER_QUIC_PORT=8080 + - IGGY_SERVER_WS_PORT=8092 - PYTHONPATH=/workspace/foreign/python - PYTEST_ARGS=-v --tb=short volumes: diff --git a/foreign/python/src/client.rs b/foreign/python/src/client.rs index 168ef2f6d2..0fdef39d9c 100644 --- a/foreign/python/src/client.rs +++ b/foreign/python/src/client.rs @@ -94,17 +94,17 @@ fn resolve_topic_params( #[pymethods] impl IggyClient { /// Constructs a new IggyClient from a TCP server address, a `TcpConfig`, a - /// `QuicConfig`, or an `HttpConfig`. This initializes a new runtime for - /// asynchronous operations. + /// `QuicConfig`, an `HttpConfig`, or a `WebSocketConfig`. This initializes a + /// new runtime for asynchronous operations. /// Future versions might utilize asyncio for more Pythonic async. /// /// Args: - /// conn: A `host:port` address, a `TcpConfig`, a `QuicConfig`, or an - /// `HttpConfig`. Defaults to `127.0.0.1:8090` over TCP with auto-login - /// disabled. A malformed address is reported differently depending on - /// the form: the string form raises `RuntimeError` here, while - /// `TcpConfig`/`QuicConfig`/`HttpConfig` raise `ValueError` when they - /// are constructed, before any of them ever reaches this call. Neither + /// conn: A `host:port` address, a `TcpConfig`, a `QuicConfig`, an + /// `HttpConfig`, or a `WebSocketConfig`. Defaults to `127.0.0.1:8090` + /// over TCP with auto-login disabled. A malformed address is reported + /// differently depending on the form: the string form raises + /// `RuntimeError` here, while every config type raises `ValueError` + /// when it is constructed, before any of them reaches this call. Neither /// exception is a subclass of the other. /// /// Raises: @@ -115,7 +115,7 @@ impl IggyClient { #[pyo3(signature = (conn=None))] fn new( #[gen_stub(override_type( - type_repr = "TcpConfig | QuicConfig | HttpConfig | builtins.str | None" + type_repr = "TcpConfig | QuicConfig | HttpConfig | WebSocketConfig | builtins.str | None" ))] conn: Option, ) -> PyResult { @@ -146,6 +146,9 @@ impl IggyClient { Some(PyClientConfig::Http(config)) => ClientWrapper::Http( HttpClient::create(config.client_config()).map_err(to_runtime_error)?, ), + Some(PyClientConfig::WebSocket(config)) => ClientWrapper::WebSocket( + WebSocketClient::create(config.client_config()).map_err(to_runtime_error)?, + ), None => ClientWrapper::Tcp( TcpClient::create(Arc::new(TcpClientConfig::default())) .map_err(to_runtime_error)?, diff --git a/foreign/python/src/config.rs b/foreign/python/src/config.rs index f699a23590..e685d048c5 100644 --- a/foreign/python/src/config.rs +++ b/foreign/python/src/config.rs @@ -22,6 +22,9 @@ use iggy::prelude::{ QuicClientReconnectionConfig as RustQuicClientReconnectionConfig, TcpClientConfig as RustTcpClientConfig, TcpClientConfigBuilder, TcpClientReconnectionConfig as RustTcpClientReconnectionConfig, + WebSocketClientConfig as RustWebSocketClientConfig, WebSocketClientConfigBuilder, + WebSocketClientReconnectionConfig as RustWebSocketClientReconnectionConfig, + WebSocketConfig as RustWebSocketFramingConfig, }; use pyo3::exceptions::PyValueError; use pyo3::prelude::*; @@ -571,7 +574,7 @@ impl QuicConfig { /// seconds) instead, since `configure()` skips the setter entirely when /// zero. Defaults to 10 seconds. /// validate_certificate: Whether to validate the server certificate. Defaults - /// to disabled, unlike the TCP and WebSocket transports. + /// to disabled; only the TCP transport validates by default. /// /// Raises: /// ValueError: If `server_address` or `client_address` is not a valid @@ -964,6 +967,466 @@ impl HttpConfig { } } +/// How the WebSocket client reconnects after the connection to the server is lost. +#[gen_stub_pyclass] +#[pyclass(from_py_object)] +#[derive(Clone)] +pub struct WebSocketReconnectionConfig { + pub(crate) inner: RustWebSocketClientReconnectionConfig, +} + +#[gen_stub_pymethods] +#[pymethods] +impl WebSocketReconnectionConfig { + /// Constructs a reconnection policy. + /// + /// Args: + /// enabled: Whether to reconnect at all. Defaults to enabled. + /// max_retries: Redials of the configured server address after the first + /// attempt, or `None` for unlimited; `0` still makes that first + /// attempt. Unlike the TCP transport, WebSocket redials the one + /// address it was configured with rather than walking a cluster + /// roster, so this counts dials. Defaults to unlimited, which means + /// a call awaited while the server is down never returns: + /// `connect()` waits inside the retry loop, as do `send_messages()` + /// and `poll_messages()` once auto-login is configured. Set a finite + /// number for request/reply style usage, so a call fails instead. + /// interval: Delay before each redial. Defaults to 1 second. + /// reestablish_after: Cooldown before redialing after a previously + /// successful connection, measured from when it was established, so + /// a session that outlived the interval is redialed at once. Applied + /// from the first redial onward, not to the initial connect. + /// Defaults to 5 seconds. + /// + /// Raises: + /// ValueError: If a duration is negative, if `max_retries` is outside the + /// range of an unsigned 32-bit integer, or if `interval` is zero. + /// OverflowError: If `max_retries` does not fit a signed 64-bit integer, + /// raised by the underlying conversion before this constructor runs. + #[new] + #[pyo3(signature = (*, enabled=None, max_retries=None, interval=None, reestablish_after=None))] + fn new( + #[gen_stub(override_type(type_repr = "builtins.bool | None"))] enabled: Option, + #[gen_stub(override_type(type_repr = "builtins.int | None"))] max_retries: Option, + #[gen_stub(override_type(type_repr = "datetime.timedelta | None", imports=("datetime")))] + interval: Option>, + #[gen_stub(override_type(type_repr = "datetime.timedelta | None", imports=("datetime")))] + reestablish_after: Option>, + ) -> PyResult { + let defaults = RustWebSocketClientReconnectionConfig::default(); + let enabled = enabled.unwrap_or(defaults.enabled); + let max_retries = max_retries + .map(|max_retries| u32_param(max_retries, "max_retries")) + .transpose()?; + let interval = interval + .as_ref() + .map(py_delta_to_iggy_duration) + .transpose()? + .map(|interval| reject_zero(interval, "interval")) + .transpose()? + .unwrap_or(defaults.interval); + Ok(Self { + inner: RustWebSocketClientReconnectionConfig { + enabled, + max_retries, + interval, + reestablish_after: reestablish_after + .as_ref() + .map(py_delta_to_iggy_duration) + .transpose()? + .unwrap_or(defaults.reestablish_after), + }, + }) + } + + #[getter] + fn enabled(&self) -> bool { + self.inner.enabled + } + + #[gen_stub(override_return_type(type_repr = "builtins.int | None"))] + #[getter] + fn max_retries(&self) -> Option { + self.inner.max_retries + } + + #[gen_stub(override_return_type(type_repr = "datetime.timedelta", imports=("datetime")))] + #[getter] + fn interval<'a>(&self, py: Python<'a>) -> PyResult> { + iggy_duration_to_py_delta(py, self.inner.interval.get()) + } + + #[gen_stub(override_return_type(type_repr = "datetime.timedelta", imports=("datetime")))] + #[getter] + fn reestablish_after<'a>(&self, py: Python<'a>) -> PyResult> { + iggy_duration_to_py_delta(py, self.inner.reestablish_after) + } + + fn __repr__(&self) -> String { + let max_retries = match self.inner.max_retries { + Some(max_retries) => max_retries.to_string(), + None => "None".to_owned(), + }; + format!( + "WebSocketReconnectionConfig(enabled={}, max_retries={max_retries}, interval={}, reestablish_after={})", + python_bool(self.inner.enabled), + duration_repr(self.inner.interval.get()), + duration_repr(self.inner.reestablish_after), + ) + } +} + +/// Frame- and buffer-level options passed through to the underlying WebSocket +/// implementation, accepted by `WebSocketConfig`'s `framing` argument. +/// +/// Every field is keyword-only and optional; unset fields fall back to the +/// underlying WebSocket library's own defaults. +#[gen_stub_pyclass] +#[pyclass(from_py_object)] +#[derive(Clone)] +pub struct WebSocketFramingConfig { + pub(crate) inner: RustWebSocketFramingConfig, +} + +#[gen_stub_pymethods] +#[pymethods] +impl WebSocketFramingConfig { + /// Constructs a WebSocket framing configuration. + /// + /// Args: + /// read_buffer_size: Read buffer size in bytes. Defaults to 128 KiB. + /// write_buffer_size: Write buffer size in bytes. Defaults to 128 KiB. + /// max_write_buffer_size: Maximum write buffer size in bytes. Defaults to + /// unbounded, which reads back as the largest value a pointer-sized + /// unsigned integer holds rather than as `None`. + /// max_message_size: Maximum message size in bytes, or an explicit `None` + /// to lift the limit entirely. Omitting the argument is not the same + /// as passing `None`: it keeps the underlying default of 64 MiB. + /// Lifting the limit lets a peer queue an arbitrarily large message + /// in memory, so prefer a finite value. + /// max_frame_size: Maximum frame size in bytes, or an explicit `None` to + /// lift the limit entirely. Omitting the argument keeps the + /// underlying default of 16 MiB, with the same caveat as + /// `max_message_size`. + /// accept_unmasked_frames: Whether to accept unmasked frames. Defaults to + /// `False`; clients should typically keep this off for RFC compliance. + /// + /// Raises: + /// ValueError: If a numeric field is outside the range of a pointer-sized + /// unsigned integer, or if `max_write_buffer_size` does not come out + /// greater than `write_buffer_size`. tungstenite enforces the same + /// invariant with an `assert!` at connect time, which would otherwise + /// surface as an unrecoverable Rust panic instead of a `ValueError`. + /// OverflowError: If a numeric field does not fit a signed 128-bit integer, + /// raised by the underlying conversion before this constructor runs. + #[new] + #[pyo3(signature = ( + *, + read_buffer_size=None, + write_buffer_size=None, + max_write_buffer_size=None, + max_message_size=64 << 20, + max_frame_size=16 << 20, + accept_unmasked_frames=None, + ))] + fn new( + #[gen_stub(override_type(type_repr = "builtins.int | None"))] read_buffer_size: Option< + i128, + >, + #[gen_stub(override_type(type_repr = "builtins.int | None"))] write_buffer_size: Option< + i128, + >, + #[gen_stub(override_type(type_repr = "builtins.int | None"))] max_write_buffer_size: Option< + i128, + >, + #[gen_stub(override_type(type_repr = "builtins.int | None"))] max_message_size: Option< + i128, + >, + #[gen_stub(override_type(type_repr = "builtins.int | None"))] max_frame_size: Option, + #[gen_stub(override_type(type_repr = "builtins.bool | None"))] + accept_unmasked_frames: Option, + ) -> PyResult { + let mut inner = RustWebSocketFramingConfig::default(); + if let Some(read_buffer_size) = read_buffer_size { + inner.read_buffer_size = Some(usize_param(read_buffer_size, "read_buffer_size")?); + } + if let Some(write_buffer_size) = write_buffer_size { + inner.write_buffer_size = Some(usize_param(write_buffer_size, "write_buffer_size")?); + } + if let Some(max_write_buffer_size) = max_write_buffer_size { + inner.max_write_buffer_size = + Some(usize_param(max_write_buffer_size, "max_write_buffer_size")?); + } + // Assigned unconditionally, unlike the buffer sizes above: `None` here + // means "no limit", and pyo3 cannot tell an omitted argument from an + // explicit `None` on its own. The signature defaults carry the + // underlying limits instead, so omission lands on `Some(default)` and + // only an explicit `None` reaches this as `None`. + inner.max_message_size = max_message_size + .map(|max_message_size| usize_param(max_message_size, "max_message_size")) + .transpose()?; + inner.max_frame_size = max_frame_size + .map(|max_frame_size| usize_param(max_frame_size, "max_frame_size")) + .transpose()?; + if let Some(accept_unmasked_frames) = accept_unmasked_frames { + inner.accept_unmasked_frames = accept_unmasked_frames; + } + if let (Some(write_buffer_size), Some(max_write_buffer_size)) = + (inner.write_buffer_size, inner.max_write_buffer_size) + && max_write_buffer_size <= write_buffer_size + { + return Err(PyValueError::new_err(format!( + "'max_write_buffer_size' ({max_write_buffer_size}) must be greater than \ + 'write_buffer_size' ({write_buffer_size})" + ))); + } + + Ok(Self { inner }) + } + + #[gen_stub(override_return_type(type_repr = "builtins.int | None"))] + #[getter] + fn read_buffer_size(&self) -> Option { + self.inner.read_buffer_size + } + + #[gen_stub(override_return_type(type_repr = "builtins.int | None"))] + #[getter] + fn write_buffer_size(&self) -> Option { + self.inner.write_buffer_size + } + + #[gen_stub(override_return_type(type_repr = "builtins.int | None"))] + #[getter] + fn max_write_buffer_size(&self) -> Option { + self.inner.max_write_buffer_size + } + + #[gen_stub(override_return_type(type_repr = "builtins.int | None"))] + #[getter] + fn max_message_size(&self) -> Option { + self.inner.max_message_size + } + + #[gen_stub(override_return_type(type_repr = "builtins.int | None"))] + #[getter] + fn max_frame_size(&self) -> Option { + self.inner.max_frame_size + } + + #[getter] + fn accept_unmasked_frames(&self) -> bool { + self.inner.accept_unmasked_frames + } + + fn __repr__(&self) -> String { + let optional_usize = |value: Option| match value { + Some(value) => value.to_string(), + None => "None".to_owned(), + }; + format!( + "WebSocketFramingConfig(read_buffer_size={}, write_buffer_size={}, max_write_buffer_size={}, max_message_size={}, max_frame_size={}, accept_unmasked_frames={})", + optional_usize(self.inner.read_buffer_size), + optional_usize(self.inner.write_buffer_size), + optional_usize(self.inner.max_write_buffer_size), + optional_usize(self.inner.max_message_size), + optional_usize(self.inner.max_frame_size), + python_bool(self.inner.accept_unmasked_frames), + ) + } +} + +/// Configuration for the WebSocket transport, accepted by `IggyClient(...)`. +/// +/// Every field is keyword-only and optional. +#[gen_stub_pyclass] +#[pyclass(from_py_object)] +#[derive(Clone)] +pub struct WebSocketConfig { + inner: Arc, +} + +impl WebSocketConfig { + /// The configuration in the shape `WebSocketClient::create` expects. + pub(crate) fn client_config(&self) -> Arc { + self.inner.clone() + } +} + +#[gen_stub_pymethods] +#[pymethods] +impl WebSocketConfig { + /// Constructs a WebSocket configuration. + /// + /// Args: + /// server_address: `host:port` of the Iggy server. Defaults to `127.0.0.1:8092`. + /// auto_login: Credentials replayed on every connect. Defaults to `AutoLogin.disabled()`. + /// reconnection: Reconnection policy. Defaults to `WebSocketReconnectionConfig()`. + /// heartbeat_interval: Interval of heartbeats sent by the client. Defaults to 5 seconds. + /// framing: Frame- and buffer-level options. Defaults to `WebSocketFramingConfig()`. + /// tls_enabled: Whether to connect over TLS. Defaults to disabled. + /// tls_domain: Domain to validate the certificate against. Defaults to + /// `localhost`. Empty means it is taken from the IP `server_address` + /// resolves to. + /// tls_ca_file: Path to the CA file for TLS. Read only when `tls_enabled` + /// and `tls_validate_certificate` are both on; with either one off it + /// is kept but never consulted, so pairing it with + /// `tls_validate_certificate=False` pins nothing. + /// tls_validate_certificate: Whether to validate the server certificate. + /// Defaults to `False`; only the TCP transport validates by default. + /// Disabling this accepts any certificate the server presents, + /// including self-signed and mismatched ones, and takes precedence + /// over `tls_ca_file`. + /// + /// Raises: + /// ValueError: If `server_address` is not a valid `host:port` pair, if a + /// duration is negative, or if `heartbeat_interval` is zero. + #[new] + #[pyo3(signature = ( + *, + server_address=None, + auto_login=None, + reconnection=None, + heartbeat_interval=None, + framing=None, + tls_enabled=None, + tls_domain=None, + tls_ca_file=None, + tls_validate_certificate=None, + ))] + #[allow(clippy::too_many_arguments)] + fn new( + #[gen_stub(override_type(type_repr = "builtins.str | None"))] server_address: Option< + String, + >, + #[gen_stub(override_type(type_repr = "AutoLogin | None"))] auto_login: Option, + #[gen_stub(override_type(type_repr = "WebSocketReconnectionConfig | None"))] + reconnection: Option, + #[gen_stub(override_type(type_repr = "datetime.timedelta | None", imports=("datetime")))] + heartbeat_interval: Option>, + #[gen_stub(override_type(type_repr = "WebSocketFramingConfig | None"))] framing: Option< + WebSocketFramingConfig, + >, + #[gen_stub(override_type(type_repr = "builtins.bool | None"))] tls_enabled: Option, + #[gen_stub(override_type(type_repr = "builtins.str | None"))] tls_domain: Option, + #[gen_stub(override_type(type_repr = "builtins.str | None"))] tls_ca_file: Option, + #[gen_stub(override_type(type_repr = "builtins.bool | None"))] + tls_validate_certificate: Option, + ) -> PyResult { + // The builder starts from `WebSocketClientConfig::default()`, and its + // `build()` trims and validates the address whether or not one was set here. + let mut builder = WebSocketClientConfigBuilder::new(); + if let Some(server_address) = server_address { + builder = builder.with_server_address(server_address); + } + let mut inner = builder + .build() + .map_err(|e| invalid_address("server_address", e))?; + if let Some(auto_login) = auto_login { + inner.auto_login = auto_login.inner; + } + if let Some(reconnection) = reconnection { + inner.reconnection = reconnection.inner; + } + if let Some(heartbeat_interval) = heartbeat_interval { + inner.heartbeat_interval = reject_zero( + py_delta_to_iggy_duration(&heartbeat_interval)?, + "heartbeat_interval", + )?; + } + if let Some(framing) = framing { + inner.ws_config = framing.inner; + } + if let Some(tls_enabled) = tls_enabled { + inner.tls_enabled = tls_enabled; + } + if let Some(tls_domain) = tls_domain { + inner.tls_domain = tls_domain; + } + if tls_ca_file.is_some() { + inner.tls_ca_file = tls_ca_file; + } + if let Some(tls_validate_certificate) = tls_validate_certificate { + inner.tls_validate_certificate = tls_validate_certificate; + } + + Ok(Self { + inner: Arc::new(inner), + }) + } + + #[getter] + fn server_address(&self) -> String { + self.inner.server_address.clone() + } + + #[getter] + fn auto_login(&self) -> AutoLogin { + AutoLogin { + inner: self.inner.auto_login.clone(), + } + } + + #[getter] + fn reconnection(&self) -> WebSocketReconnectionConfig { + WebSocketReconnectionConfig { + inner: self.inner.reconnection.clone(), + } + } + + #[gen_stub(override_return_type(type_repr = "datetime.timedelta", imports=("datetime")))] + #[getter] + fn heartbeat_interval<'a>(&self, py: Python<'a>) -> PyResult> { + iggy_duration_to_py_delta(py, self.inner.heartbeat_interval.get()) + } + + #[getter] + fn framing(&self) -> WebSocketFramingConfig { + WebSocketFramingConfig { + inner: self.inner.ws_config.clone(), + } + } + + #[getter] + fn tls_enabled(&self) -> bool { + self.inner.tls_enabled + } + + #[getter] + fn tls_domain(&self) -> String { + self.inner.tls_domain.clone() + } + + #[gen_stub(override_return_type(type_repr = "builtins.str | None"))] + #[getter] + fn tls_ca_file(&self) -> Option { + self.inner.tls_ca_file.clone() + } + + #[getter] + fn tls_validate_certificate(&self) -> bool { + self.inner.tls_validate_certificate + } + + fn __repr__(&self) -> String { + let tls_ca_file = match &self.inner.tls_ca_file { + Some(tls_ca_file) => format!("{tls_ca_file:?}"), + None => "None".to_owned(), + }; + format!( + "WebSocketConfig(server_address={:?}, auto_login={}, reconnection={}, heartbeat_interval={}, framing={}, tls_enabled={}, tls_domain={:?}, tls_ca_file={tls_ca_file}, tls_validate_certificate={})", + self.inner.server_address, + self.auto_login().__repr__(), + self.reconnection().__repr__(), + duration_repr(self.inner.heartbeat_interval.get()), + self.framing().__repr__(), + python_bool(self.inner.tls_enabled), + self.inner.tls_domain, + python_bool(self.inner.tls_validate_certificate), + ) + } +} + fn python_bool(value: bool) -> &'static str { if value { "True" } else { "False" } } @@ -1020,8 +1483,24 @@ fn varint_param(value: i64, parameter: &str) -> PyResult { Ok(value) } +/// Converts a Python int to the unsigned pointer-sized integer a WebSocket +/// framing field expects, naming the parameter in the error so a caller can +/// tell which argument was out of range. Extracted as an `i128` rather than an +/// `i64` so the whole `usize` range survives the way in: `max_write_buffer_size` +/// defaults to `usize::MAX`, which an `i64` parameter would refuse to take back +/// with an unnamed `OverflowError`, breaking `eval(repr(config))`. +fn usize_param(value: i128, parameter: &str) -> PyResult { + usize::try_from(value).map_err(|_| { + PyValueError::new_err(format!( + "'{parameter}' must be between 0 and {}", + usize::MAX + )) + }) +} + /// What `IggyClient(...)` accepts: a bare `host:port`, a full `TcpConfig`, a -/// `QuicConfig` for the QUIC transport, or an `HttpConfig` for the HTTP transport. +/// `QuicConfig` for the QUIC transport, an `HttpConfig` for the HTTP transport, +/// or a `WebSocketConfig` for the WebSocket transport. #[derive(FromPyObject)] pub enum PyClientConfig { #[pyo3(transparent)] @@ -1030,7 +1509,51 @@ pub enum PyClientConfig { Quic(QuicConfig), #[pyo3(transparent)] Http(HttpConfig), + #[pyo3(transparent)] + WebSocket(WebSocketConfig), #[pyo3(transparent, annotation = "str")] ServerAddress(String), } -impl_stub_type!(PyClientConfig = TcpConfig | QuicConfig | HttpConfig | String); +impl_stub_type!(PyClientConfig = TcpConfig | QuicConfig | HttpConfig | WebSocketConfig | String); + +#[cfg(test)] +mod tests { + use super::*; + + /// Mirrors the literal in `WebSocketFramingConfig::new`'s signature. + const DEFAULT_MAX_MESSAGE_SIZE: usize = 64 << 20; + + /// Mirrors the literal in `WebSocketFramingConfig::new`'s signature. + const DEFAULT_MAX_FRAME_SIZE: usize = 16 << 20; + + /// The signature defaults have to be literals for the generated stub to stay + /// valid Python, so nothing but this test stops them drifting from the SDK + /// (and so from tungstenite) on a dependency bump. + #[test] + fn defaults_should_match_the_sdk() { + let defaults = RustWebSocketFramingConfig::default(); + + assert_eq!( + defaults.max_message_size, + Some(DEFAULT_MAX_MESSAGE_SIZE), + "'max_message_size' drifted from the SDK, update the literal in \ + WebSocketFramingConfig::new's signature too" + ); + assert_eq!( + defaults.max_frame_size, + Some(DEFAULT_MAX_FRAME_SIZE), + "'max_frame_size' drifted from the SDK, update the literal in \ + WebSocketFramingConfig::new's signature too" + ); + assert!( + defaults.write_buffer_size.is_some(), + "'write_buffer_size' lost its default, so the write buffer invariant \ + check in WebSocketFramingConfig::new would stop running" + ); + assert!( + defaults.max_write_buffer_size.is_some(), + "'max_write_buffer_size' lost its default, so the write buffer \ + invariant check in WebSocketFramingConfig::new would stop running" + ); + } +} diff --git a/foreign/python/src/lib.rs b/foreign/python/src/lib.rs index 78b777918c..dc238836e3 100644 --- a/foreign/python/src/lib.rs +++ b/foreign/python/src/lib.rs @@ -34,6 +34,7 @@ mod user_headers; use client::IggyClient; use config::{ AutoLogin, HttpConfig, QuicConfig, QuicReconnectionConfig, TcpConfig, TcpReconnectionConfig, + WebSocketConfig, WebSocketFramingConfig, WebSocketReconnectionConfig, }; use consumer::{ AutoCommit, AutoCommitAfter, AutoCommitWhen, Consumer, ConsumerGroup, ConsumerGroupDetails, @@ -65,6 +66,9 @@ fn apache_iggy(_py: Python, m: &Bound<'_, PyModule>) -> PyResult<()> { m.add_class::()?; m.add_class::()?; m.add_class::()?; + m.add_class::()?; + m.add_class::()?; + m.add_class::()?; m.add_class::()?; m.add_class::()?; m.add_class::()?; diff --git a/foreign/python/tests/test_websocket_config.py b/foreign/python/tests/test_websocket_config.py new file mode 100644 index 0000000000..555e758fa3 --- /dev/null +++ b/foreign/python/tests/test_websocket_config.py @@ -0,0 +1,522 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +""" +Tests for the WebSocket client configuration surface. + +`WebSocketConfig` and `WebSocketReconnectionConfig` mirror the Rust SDK +types the same way `TcpConfig`/`TcpReconnectionConfig` do, so most of these +assert that a value set from Python survives to the getters and that unset +fields fall back to the Rust defaults. `WebSocketFramingConfig` is new: it +wraps the tungstenite frame- and buffer-level options the Rust SDK nests +under `ws_config`, exposed here as a `framing` argument rather than flat +kwargs, mirroring the Rust struct's own nesting. `AutoLogin` is +transport-agnostic and already covered by `test_client_config.py`. +""" + +import ast +import socket +import sys +import threading +from collections.abc import Callable +from datetime import timedelta + +import pytest + +from apache_iggy import ( + AutoLogin, + IggyClient, + WebSocketConfig, + WebSocketFramingConfig, + WebSocketReconnectionConfig, +) + +from .utils import get_websocket_server_config, wait_for_ping, wait_for_server + +# tungstenite's own `WebSocketConfig::default()` values, mirrored here so a +# dependency bump that moves one fails loudly instead of silently redefining the +# Python surface. +TUNGSTENITE_READ_BUFFER_SIZE = 128 * 1024 +TUNGSTENITE_WRITE_BUFFER_SIZE = 128 * 1024 +# `usize::MAX`, tungstenite's "unbounded" write buffer. +TUNGSTENITE_MAX_WRITE_BUFFER_SIZE = sys.maxsize * 2 + 1 +TUNGSTENITE_MAX_MESSAGE_SIZE = 64 << 20 +TUNGSTENITE_MAX_FRAME_SIZE = 16 << 20 + +# The accept thread only has to come back from a connection the client already +# dialed, so anything beyond this is a hang, not slowness. +ACCEPT_JOIN_TIMEOUT_SECONDS = 5.0 + + +def _accept_and_close(listener: socket.socket) -> None: + """Accept one connection and drop it, failing any WebSocket handshake.""" + try: + connection, _ = listener.accept() + except OSError: + return + connection.close() + + +@pytest.mark.unit +class TestWebSocketReconnectionConfig: + """Test the reconnection policy.""" + + def test_defaults_match_the_rust_sdk(self): + """Test that an unconfigured policy reconnects forever, one second apart.""" + reconnection = WebSocketReconnectionConfig() + + assert reconnection.enabled is True + assert reconnection.max_retries is None + assert reconnection.interval == timedelta(seconds=1) + assert reconnection.reestablish_after == timedelta(seconds=5) + + def test_every_field_round_trips(self): + """Test that each configured field is readable back unchanged.""" + reconnection = WebSocketReconnectionConfig( + enabled=False, + max_retries=10, + interval=timedelta(milliseconds=250), + reestablish_after=timedelta(seconds=30), + ) + + assert reconnection.enabled is False + assert reconnection.max_retries == 10 + assert reconnection.interval == timedelta(milliseconds=250) + assert reconnection.reestablish_after == timedelta(seconds=30) + + def test_arguments_are_keyword_only(self): + """Test that the adjacent flags cannot be passed positionally.""" + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-count + WebSocketReconnectionConfig(True) + + @pytest.mark.parametrize( + "construct", + [ + lambda duration: WebSocketReconnectionConfig(interval=duration), + lambda duration: WebSocketReconnectionConfig(reestablish_after=duration), + ], + ids=["interval", "reestablish_after"], + ) + @pytest.mark.parametrize( + "negative", + [timedelta(microseconds=-1), timedelta(seconds=-1), timedelta(days=-1)], + ) + def test_negative_duration_is_rejected( + self, + construct: Callable[[timedelta], WebSocketReconnectionConfig], + negative: timedelta, + ): + """Test that a negative duration fails at construction, not at connect.""" + with pytest.raises(ValueError, match="negative"): + construct(negative) + + @pytest.mark.parametrize("out_of_range", [-1, 2**32]) + def test_out_of_range_max_retries_is_rejected(self, out_of_range: int): + """Test that a retry count outside the wire range names the argument. + + The conversion pyo3 does on its own raises OverflowError, which is not a + ValueError and so escapes the handler a caller wraps construction in. + """ + with pytest.raises(ValueError, match="max_retries"): + WebSocketReconnectionConfig(max_retries=out_of_range) + + def test_zero_reestablish_after_is_allowed(self): + """Test that a zero cooldown is legal and readable back.""" + reconnection = WebSocketReconnectionConfig(reestablish_after=timedelta(0)) + + assert reconnection.reestablish_after == timedelta(0) + + @pytest.mark.parametrize( + "kwargs", + [ + {}, + {"max_retries": 5}, + {"enabled": False}, + ], + ids=["unlimited_retries", "bounded_retries", "reconnection_disabled"], + ) + def test_zero_interval_is_rejected(self, kwargs: dict): + """Test that a zero interval fails whatever the retry policy is. + + The interval is a delay between passes, so zero reconnects in a + continuous loop. + """ + with pytest.raises(ValueError, match="zero"): + WebSocketReconnectionConfig(interval=timedelta(0), **kwargs) + + def test_very_long_interval_round_trips(self): + """Test that an interval beyond 68 years survives the i32 boundary.""" + reconnection = WebSocketReconnectionConfig(interval=timedelta(days=30_000)) + + assert reconnection.interval == timedelta(days=30_000) + + def test_maximum_interval_round_trips(self): + """Test that the largest timedelta survives the day conversion.""" + reconnection = WebSocketReconnectionConfig(interval=timedelta(days=999_999_999)) + + assert reconnection.interval == timedelta(days=999_999_999) + + +@pytest.mark.unit +class TestWebSocketFramingConfig: + """Test the frame- and buffer-level options.""" + + def test_defaults_match_tungstenite(self): + """Test that unconfigured sizes fall back to the tungstenite defaults. + + Every size defaults to a concrete value; an omitted argument keeps that + default, while an explicit `None` clears the limit. See + `test_explicit_none_clears_the_limit`. + """ + framing = WebSocketFramingConfig() + + assert framing.read_buffer_size == TUNGSTENITE_READ_BUFFER_SIZE + assert framing.write_buffer_size == TUNGSTENITE_WRITE_BUFFER_SIZE + assert framing.max_write_buffer_size == TUNGSTENITE_MAX_WRITE_BUFFER_SIZE + assert framing.max_message_size == TUNGSTENITE_MAX_MESSAGE_SIZE + assert framing.max_frame_size == TUNGSTENITE_MAX_FRAME_SIZE + assert framing.accept_unmasked_frames is False + + def test_every_field_round_trips(self): + """Test that each configured field is readable back unchanged.""" + framing = WebSocketFramingConfig( + read_buffer_size=4096, + write_buffer_size=4096, + max_write_buffer_size=8192, + max_message_size=16384, + max_frame_size=16384, + accept_unmasked_frames=True, + ) + + assert framing.read_buffer_size == 4096 + assert framing.write_buffer_size == 4096 + assert framing.max_write_buffer_size == 8192 + assert framing.max_message_size == 16384 + assert framing.max_frame_size == 16384 + assert framing.accept_unmasked_frames is True + + @pytest.mark.parametrize("field", ["max_message_size", "max_frame_size"]) + def test_explicit_none_clears_the_limit(self, field: str): + """Test that an explicit `None` lifts the size limit. + + An omitted argument and an explicit `None` both reach Rust as + `Option::None`, so without a sentinel the constructor cannot tell them + apart and a caller asking for no limit would silently keep the default. + """ + framing = WebSocketFramingConfig(**{field: None}) + + assert getattr(framing, field) is None + + @pytest.mark.parametrize("field", ["max_message_size", "max_frame_size"]) + def test_omitting_the_argument_keeps_the_default_limit(self, field: str): + """Test that omitting the argument is not the same as passing `None`.""" + framing = WebSocketFramingConfig() + + assert getattr(framing, field) is not None + + def test_clearing_one_limit_leaves_the_other_alone(self): + """Test that the two sentinels are independent.""" + framing = WebSocketFramingConfig(max_message_size=None) + + assert framing.max_message_size is None + assert framing.max_frame_size is not None + + def test_arguments_are_keyword_only(self): + """Test that the first field cannot be passed positionally.""" + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-count + WebSocketFramingConfig(4096) + + def test_repr_shows_every_field_as_python(self): + """Test that repr covers every field and parses as Python.""" + framing = WebSocketFramingConfig( + read_buffer_size=4096, + accept_unmasked_frames=True, + ) + + printed = repr(framing) + + assert "read_buffer_size=4096" in printed + assert "accept_unmasked_frames=True" in printed + ast.parse(printed) + + def test_default_repr_is_constructible(self): + """Test that the default repr survives being fed back to the constructor. + + `max_write_buffer_size` defaults to `usize::MAX`, so this only holds + because the constructor takes the sizes as `i128`. An `i64` parameter + rejects its own default with an `OverflowError`, which `ast.parse` + alone would not catch. + """ + printed = repr(WebSocketFramingConfig()) + + assert repr(eval(printed)) == printed # noqa: S307 + + @pytest.mark.parametrize( + "field", + [ + "read_buffer_size", + "write_buffer_size", + "max_write_buffer_size", + "max_message_size", + "max_frame_size", + ], + ) + def test_negative_size_is_rejected(self, field: str): + """Test that a negative size names the argument that caused it.""" + with pytest.raises(ValueError, match=field): + # pyrefly: ignore # bad-argument-type + WebSocketFramingConfig(**{field: -1}) + + def test_max_write_buffer_size_not_greater_than_write_buffer_size_is_rejected(self): + """Test that a non-increasing write buffer pair fails at construction. + + tungstenite enforces `max_write_buffer_size > write_buffer_size` with an + `assert!` when the connection is established, which would otherwise + surface as an unrecoverable Rust panic instead of a catchable error. + """ + with pytest.raises(ValueError, match="max_write_buffer_size"): + WebSocketFramingConfig(write_buffer_size=1000, max_write_buffer_size=1000) + + def test_max_write_buffer_size_below_the_default_write_buffer_size_is_rejected( + self, + ): + """Test that the invariant is checked against the default too. + + Setting only `max_write_buffer_size` below the untouched default + `write_buffer_size` (128 KiB) must fail the same way as setting both. + """ + with pytest.raises(ValueError, match="max_write_buffer_size"): + WebSocketFramingConfig(max_write_buffer_size=1000) + + +@pytest.mark.unit +class TestWebSocketConfig: + """Test the transport configuration.""" + + def test_defaults_match_the_rust_sdk(self): + """Test that an unconfigured transport matches the Rust SDK defaults.""" + config = WebSocketConfig() + + assert config.server_address == "127.0.0.1:8092" + assert config.auto_login.enabled is False + assert config.reconnection.enabled is True + assert config.heartbeat_interval == timedelta(seconds=5) + assert config.tls_enabled is False + assert config.tls_domain == "localhost" + assert config.tls_ca_file is None + # Unlike TCP and QUIC, WebSocket does not validate the server + # certificate by default. + assert config.tls_validate_certificate is False + + def test_every_field_round_trips(self): + """Test that each configured field is readable back unchanged.""" + config = WebSocketConfig( + server_address="127.0.0.1:8093", + auto_login=AutoLogin.username_password("iggy", "iggy"), + reconnection=WebSocketReconnectionConfig(max_retries=3), + heartbeat_interval=timedelta(seconds=15), + framing=WebSocketFramingConfig(read_buffer_size=4096), + tls_enabled=True, + tls_domain="example.com", + tls_ca_file="ca.pem", + tls_validate_certificate=True, + ) + + assert config.server_address == "127.0.0.1:8093" + assert config.auto_login.username == "iggy" + assert config.reconnection.max_retries == 3 + assert config.heartbeat_interval == timedelta(seconds=15) + assert config.framing.read_buffer_size == 4096 + assert config.tls_enabled is True + assert config.tls_domain == "example.com" + assert config.tls_ca_file == "ca.pem" + assert config.tls_validate_certificate is True + + def test_arguments_are_keyword_only(self): + """Test that the address cannot be passed positionally.""" + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-count + WebSocketConfig("127.0.0.1:8092") + + def test_repr_hides_the_password(self): + """Test that the password does not leak through repr.""" + config = WebSocketConfig( + auto_login=AutoLogin.username_password("iggy", "secret") + ) + + assert "secret" not in repr(config) + + def test_repr_shows_every_field_as_python(self): + """Test that repr covers the TLS fields and parses as Python.""" + config = WebSocketConfig( + heartbeat_interval=timedelta(seconds=15), + tls_enabled=True, + tls_domain="example.com", + tls_ca_file="ca.pem", + tls_validate_certificate=True, + ) + + printed = repr(config) + + assert 'tls_domain="example.com"' in printed + assert 'tls_ca_file="ca.pem"' in printed + assert "tls_validate_certificate=True" in printed + assert "heartbeat_interval=datetime.timedelta(seconds=15)" in printed + ast.parse(printed) + + @pytest.mark.parametrize( + "invalid_address", + ["", "127.0.0.1", "127.0.0.1:not-a-port", "127.0.0.1:70000", "::1:8092"], + ) + def test_invalid_server_address_is_rejected(self, invalid_address: str): + """Test that a malformed address fails at construction, naming itself.""" + with pytest.raises(ValueError, match="server_address"): + WebSocketConfig(server_address=invalid_address) + + def test_negative_heartbeat_interval_is_rejected(self): + """Test that a negative heartbeat interval fails at construction.""" + with pytest.raises(ValueError, match="negative"): + WebSocketConfig(heartbeat_interval=timedelta(seconds=-3)) + + def test_zero_heartbeat_interval_is_rejected(self): + """Test that a zero heartbeat interval fails at construction. + + Nothing downstream reads zero as "disabled"; it heartbeats in a + continuous loop for as long as the client lives. + """ + with pytest.raises(ValueError, match="zero"): + WebSocketConfig(heartbeat_interval=timedelta(0)) + + +@pytest.mark.unit +class TestWebSocketClientConstruction: + """Test that `IggyClient(...)` accepts a `WebSocketConfig`.""" + + @pytest.mark.asyncio + async def test_accepts_a_config(self): + """Test that the resulting client is actually WebSocket, not silently TCP. + + `IggyClient(...)` is not None for either union arm, so that alone never + pinned the transport, and neither does the error text: both arms report + a failed dial as `Cannot establish connection`. What separates them is + the outcome against a plain TCP listener. WebSocket has to complete an + HTTP upgrade handshake, which the listener below refuses by closing at + once, while a client that regressed to the TCP arm needs nothing beyond + the accepted socket and would connect. So the raise itself is the proof. + + The listener is a real socket on an ephemeral loopback port, so nothing + here depends on a sysctl, a privileged port, or a running server. + `max_retries=0` keeps the failure immediate instead of retrying + forever, which is the default. + + The accept thread is joined before the block ends so it is out of + `accept()` before the listener closes under it. + """ + with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as listener: + listener.bind(("127.0.0.1", 0)) + listener.listen(1) + port = listener.getsockname()[1] + accept_thread = threading.Thread( + target=lambda: _accept_and_close(listener), daemon=True + ) + accept_thread.start() + + client = IggyClient( + WebSocketConfig( + server_address=f"127.0.0.1:{port}", + reconnection=WebSocketReconnectionConfig(max_retries=0), + ) + ) + + with pytest.raises(RuntimeError, match="Cannot establish connection"): + await client.connect() + + accept_thread.join(timeout=ACCEPT_JOIN_TIMEOUT_SECONDS) + + def test_accepts_the_default_config(self): + """Test that an explicit default `WebSocketConfig` is accepted.""" + assert IggyClient(WebSocketConfig()) is not None + + +@pytest.mark.integration +class TestAutoLoginAgainstServer: + """Test that configured credentials are actually replayed on connect.""" + + @pytest.mark.asyncio + async def test_auto_login_authenticates_without_login_user(self, unique_name): + """Test that a privileged call succeeds without a manual login_user().""" + host, port = get_websocket_server_config() + wait_for_server(host, port) + + client = IggyClient( + WebSocketConfig( + server_address=f"{host}:{port}", + auto_login=AutoLogin.username_password("iggy", "iggy"), + # The default reconnection policy retries forever: a missing + # listener would hang this test until the CI timeout instead + # of failing. + reconnection=WebSocketReconnectionConfig(enabled=False), + ) + ) + await client.connect() + await wait_for_ping(client) + + stream_name = unique_name() + await client.create_stream(stream_name) + assert await client.get_stream(stream_name) is not None + + @pytest.mark.asyncio + async def test_without_auto_login_a_privileged_call_is_unauthenticated( + self, unique_name + ): + """Test that the same call fails when no credentials are configured.""" + host, port = get_websocket_server_config() + wait_for_server(host, port) + + client = IggyClient( + WebSocketConfig( + server_address=f"{host}:{port}", + # The default reconnection policy retries forever: a missing + # listener would hang this test until the CI timeout instead + # of failing. + reconnection=WebSocketReconnectionConfig(enabled=False), + ) + ) + await client.connect() + await wait_for_ping(client) + + with pytest.raises(RuntimeError): + await client.create_stream(unique_name()) + + @pytest.mark.asyncio + async def test_wrong_auto_login_credentials_fail(self): + """Test that bad configured credentials surface as a connect failure.""" + host, port = get_websocket_server_config() + wait_for_server(host, port) + + client = IggyClient( + WebSocketConfig( + server_address=f"{host}:{port}", + auto_login=AutoLogin.username_password("iggy", "invalid-password"), + reconnection=WebSocketReconnectionConfig(enabled=False), + ) + ) + + with pytest.raises(RuntimeError): + await client.connect() diff --git a/foreign/python/tests/utils.py b/foreign/python/tests/utils.py index 5d7654cbf4..b005dcabea 100644 --- a/foreign/python/tests/utils.py +++ b/foreign/python/tests/utils.py @@ -35,6 +35,7 @@ DEFAULT_TCP_PORT = 8090 DEFAULT_QUIC_PORT = 8080 DEFAULT_HTTP_PORT = 3000 +DEFAULT_WEBSOCKET_PORT = 8092 def get_transport_config(port_env_var: str, default_port: int) -> tuple[str, int]: @@ -96,6 +97,16 @@ def get_http_server_config() -> tuple[str, int]: return get_transport_config("IGGY_SERVER_HTTP_PORT", DEFAULT_HTTP_PORT) +def get_websocket_server_config() -> tuple[str, int]: + """ + Get WebSocket server configuration from environment variables or defaults. + + Returns: + tuple: (host, port) for the Iggy server + """ + return get_transport_config("IGGY_SERVER_WS_PORT", DEFAULT_WEBSOCKET_PORT) + + def wait_for_server(host: str, port: int, timeout: int = 60, interval: int = 2) -> None: """ Wait for the server to become available. From 6c07954a054ac2638cd4e7390e0549c91ae006fe Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Mon, 7 Sep 2026 22:53:55 +0200 Subject: [PATCH 086/182] chore(deps): Bump the csharp group with 1 update (#4087) --- foreign/csharp/Directory.Packages.props | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/foreign/csharp/Directory.Packages.props b/foreign/csharp/Directory.Packages.props index fdc9ef880e..2388831f38 100644 --- a/foreign/csharp/Directory.Packages.props +++ b/foreign/csharp/Directory.Packages.props @@ -36,7 +36,7 @@ under the License. - + all runtime; build; native; contentfiles; analyzers; buildtransitive From a71ed7420e1d00df0a6046e9f23eaaa4203490e0 Mon Sep 17 00:00:00 2001 From: Grzegorz Koszyk <112548209+numinnex@users.noreply.github.com> Date: Tue, 8 Sep 2026 11:26:13 +0200 Subject: [PATCH 087/182] fix(metadata): roll deleted partitions and topics out of parent stats (#4046) --- Cargo.toml | 6 +- core/common/src/types/streaming_stats.rs | 367 +++++-- .../verify_after_server_restart.rs | 52 + core/integration/tests/server/general.rs | 14 +- .../delete_stats_rollback_scenario.rs | 225 +++++ .../integration/tests/server/scenarios/mod.rs | 137 ++- .../server/scenarios/purge_delete_scenario.rs | 26 +- .../stream_size_validation_scenario.rs | 140 +-- core/metadata/src/stm/stream.rs | 899 ++++++++++++++++-- core/partitions/src/iggy_partition.rs | 159 +++- core/server/src/boot/mod.rs | 16 - core/server/src/boot/recovery.rs | 2 +- core/server/src/http/metrics.rs | 71 +- core/server/src/partition_reconciler.rs | 331 ++++++- core/server/src/responses.rs | 28 +- core/shard/src/lib.rs | 14 + 16 files changed, 2109 insertions(+), 378 deletions(-) create mode 100644 core/integration/tests/server/scenarios/delete_stats_rollback_scenario.rs diff --git a/Cargo.toml b/Cargo.toml index beb92e80f4..43747aae09 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -214,7 +214,11 @@ jsonwebtoken = { version = "11.0.0", features = ["rust_crypto"] } kafka-protocol = { version = "0.18.0", default-features = false, features = ["broker"] } keyring-core = "1.0.0" lazy_static = "1.5.0" -left-right = "0.11" +# Pinned exactly: `metadata::stm::stream`'s keyed stats evictions lean on +# left-right draining the previous batch's `absorb_second` before the current +# batch's `absorb_first`, which is an implementation detail of 0.11.8's +# `write.rs`, not a documented contract. +left-right = "=0.11.8" libc = "0.2.189" log = "0.4.34" lz4_flex = "0.14.0" diff --git a/core/common/src/types/streaming_stats.rs b/core/common/src/types/streaming_stats.rs index 2c5829d13e..5d64c632c4 100644 --- a/core/common/src/types/streaming_stats.rs +++ b/core/common/src/types/streaming_stats.rs @@ -19,6 +19,123 @@ use std::sync::{ Arc, atomic::{AtomicU32, AtomicU64, Ordering}, }; +use tracing::warn; + +/// Process-wide number of rollup decrements that could not be covered by the +/// counter they were subtracted from. One static for the whole node, summed +/// across every stream, topic and partition it holds. +/// +/// A decrement bigger than the total it targets means the tree lost +/// `parent >= sum(children)` somewhere upstream. The counters are unsigned, so +/// without a clamp that subtraction wraps to ~1.8e19 and every reader +/// (`get_stream`, `get_topic`, `/stats`, `/metrics`) serves the wrapped value +/// until the process restarts. +/// +/// Clamping trades that for a total that is merely low, and low is where it +/// stays: increments are `fetch_add`, so every later write stacks on the +/// clamped base and carries the shortfall with it. Only a rebuild +/// (`rebuild_parent_totals`) or a restart puts the total back. This counter +/// plus the `warn!` below is what tells an operator the divergence happened, +/// and that the aggregate needs a rebuild rather than time. +static ROLLUP_UNDERFLOWS: AtomicU64 = AtomicU64::new(0); + +/// Monotonic process-wide count of clamped rollup decrements. Surfaced on +/// `/metrics` so the clamp is alertable rather than log-grep-able. +/// +/// Named for the root it is reached from, not for this module: `iggy_common` +/// globs the module into its crate root, so a bare `rollup_underflows` would +/// sit in the published SDK surface next to the stream and topic types saying +/// nothing about which rollup it counts. +#[must_use] +pub fn stats_rollup_underflows() -> u64 { + ROLLUP_UNDERFLOWS.load(Ordering::Relaxed) +} + +/// One scope/counter pair's clamp state: the labels an operator reads, plus the +/// count the log throttle keys on. +struct UnderflowSite { + scope: &'static str, + counter: &'static str, + clamped: AtomicU64, +} + +impl UnderflowSite { + const fn new(scope: &'static str, counter: &'static str) -> Self { + Self { + scope, + counter, + clamped: AtomicU64::new(0), + } + } + + /// Count the clamp globally, then log this pair's first and every power of + /// two after it. + /// + /// The throttle is per pair, not global: a skewed tree emits up to three + /// lines per counter per rollback and a bulk delete turns that into a + /// flood, so a shared count would hold a quiet pair's first-ever + /// divergence back for up to 1023 events behind a noisy one. + /// + /// Both counts go on the line. `clamped` is this pair's, `total` is the + /// process-wide one [`stats_rollup_underflows`] exports, which carries no + /// scope or counter label -- without both an operator cannot tell which + /// pair moved the metric. + fn report(&self, shortfall: u64) { + let total = ROLLUP_UNDERFLOWS.fetch_add(1, Ordering::Relaxed) + 1; + let clamped = self.clamped.fetch_add(1, Ordering::Relaxed) + 1; + if clamped.is_power_of_two() { + warn!( + target: "iggy.stats.diag", + scope = self.scope, + counter = self.counter, + shortfall, + clamped, + total, + "rollup decrement exceeded the total it was subtracted from; clamped at zero" + ); + } + } +} + +static STREAM_SIZE_BYTES: UnderflowSite = UnderflowSite::new("stream", "size_bytes"); +static STREAM_MESSAGES_COUNT: UnderflowSite = UnderflowSite::new("stream", "messages_count"); +static STREAM_SEGMENTS_COUNT: UnderflowSite = UnderflowSite::new("stream", "segments_count"); +static TOPIC_SIZE_BYTES: UnderflowSite = UnderflowSite::new("topic", "size_bytes"); +static TOPIC_MESSAGES_COUNT: UnderflowSite = UnderflowSite::new("topic", "messages_count"); +static TOPIC_SEGMENTS_COUNT: UnderflowSite = UnderflowSite::new("topic", "segments_count"); +static PARTITION_SIZE_BYTES: UnderflowSite = UnderflowSite::new("partition", "size_bytes"); +static PARTITION_MESSAGES_COUNT: UnderflowSite = UnderflowSite::new("partition", "messages_count"); +static PARTITION_SEGMENTS_COUNT: UnderflowSite = UnderflowSite::new("partition", "segments_count"); + +/// Subtract `amount`, clamping at zero instead of wrapping. Returns what was +/// actually taken, which is what the caller passes on to its parent. +/// +/// `fetch_update` rather than a load followed by a subtract: the counters are +/// written from the metadata shard and from whichever shard owns the partition, +/// so a separate load leaves a window where the clamp reads one value and +/// subtracts from another. +macro_rules! clamped_sub { + ($name:ident, $counter:ty, $amount:ty) => { + fn $name(counter: &$counter, amount: $amount, site: &'static UnderflowSite) -> $amount { + let previous = counter + .fetch_update(Ordering::AcqRel, Ordering::Acquire, |current| { + // `None` on an already-zero counter: no store, and the + // `Err` it returns carries that same zero, so the clamp + // reports identically either way. That is the steady shape + // for a scope whose rollback has already run. + (current != 0).then(|| current.saturating_sub(amount)) + }) + .unwrap_or_else(|previous| previous); + if previous < amount { + site.report(u64::from(amount - previous)); + } + previous.min(amount) + } + }; +} + +clamped_sub!(clamped_sub_u64, AtomicU64, u64); +clamped_sub!(clamped_sub_u32, AtomicU32, u32); #[derive(Default, Debug)] pub struct StreamStats { @@ -42,18 +159,20 @@ impl StreamStats { .fetch_add(segments_count, Ordering::AcqRel); } + // A stream is the root, so a clamp here has nowhere left to be forwarded + // and no caller that could name a scope the `warn!` in `UnderflowSite` + // does not already carry. Only `PartitionStats` returns its shortfall, + // where the caller holds the namespace. pub fn decrement_size_bytes(&self, size_bytes: u64) { - self.size_bytes.fetch_sub(size_bytes, Ordering::AcqRel); + clamped_sub_u64(&self.size_bytes, size_bytes, &STREAM_SIZE_BYTES); } pub fn decrement_messages_count(&self, messages_count: u64) { - self.messages_count - .fetch_sub(messages_count, Ordering::AcqRel); + clamped_sub_u64(&self.messages_count, messages_count, &STREAM_MESSAGES_COUNT); } pub fn decrement_segments_count(&self, segments_count: u32) { - self.segments_count - .fetch_sub(segments_count, Ordering::AcqRel); + clamped_sub_u32(&self.segments_count, segments_count, &STREAM_SEGMENTS_COUNT); } pub fn size_bytes_inconsistent(&self) -> u64 { @@ -123,62 +242,41 @@ impl TopicStats { self.parent.clone() } - pub fn increment_parent_size_bytes(&self, size_bytes: u64) { - self.parent.increment_size_bytes(size_bytes); - } - - pub fn increment_parent_messages_count(&self, messages_count: u64) { - self.parent.increment_messages_count(messages_count); - } - - pub fn increment_parent_segments_count(&self, segments_count: u32) { - self.parent.increment_segments_count(segments_count); - } - pub fn increment_size_bytes(&self, size_bytes: u64) { self.size_bytes.fetch_add(size_bytes, Ordering::AcqRel); - self.increment_parent_size_bytes(size_bytes); + self.parent.increment_size_bytes(size_bytes); } pub fn increment_messages_count(&self, messages_count: u64) { self.messages_count .fetch_add(messages_count, Ordering::AcqRel); - self.increment_parent_messages_count(messages_count); + self.parent.increment_messages_count(messages_count); } pub fn increment_segments_count(&self, segments_count: u32) { self.segments_count .fetch_add(segments_count, Ordering::AcqRel); - self.increment_parent_segments_count(segments_count); - } - - pub fn decrement_parent_size_bytes(&self, size_bytes: u64) { - self.parent.decrement_size_bytes(size_bytes); - } - - pub fn decrement_parent_messages_count(&self, messages_count: u64) { - self.parent.decrement_messages_count(messages_count); - } - - pub fn decrement_parent_segments_count(&self, segments_count: u32) { - self.parent.decrement_segments_count(segments_count); + self.parent.increment_segments_count(segments_count); } + // Forward what this level actually gave up, not what was asked for. A + // decrement bigger than the counter holds means the bytes were never in + // this subtree, so the ancestors do not hold them either -- passing the + // full amount up would take them out of a sibling's live data instead. + // Under `parent == sum(children)` the two are equal and nothing changes. pub fn decrement_size_bytes(&self, size_bytes: u64) { - self.size_bytes.fetch_sub(size_bytes, Ordering::AcqRel); - self.decrement_parent_size_bytes(size_bytes); + let taken = clamped_sub_u64(&self.size_bytes, size_bytes, &TOPIC_SIZE_BYTES); + self.parent.decrement_size_bytes(taken); } pub fn decrement_messages_count(&self, messages_count: u64) { - self.messages_count - .fetch_sub(messages_count, Ordering::AcqRel); - self.decrement_parent_messages_count(messages_count); + let taken = clamped_sub_u64(&self.messages_count, messages_count, &TOPIC_MESSAGES_COUNT); + self.parent.decrement_messages_count(taken); } pub fn decrement_segments_count(&self, segments_count: u32) { - self.segments_count - .fetch_sub(segments_count, Ordering::AcqRel); - self.decrement_parent_segments_count(segments_count); + let taken = clamped_sub_u32(&self.segments_count, segments_count, &TOPIC_SEGMENTS_COUNT); + self.parent.decrement_segments_count(taken); } pub fn size_bytes_inconsistent(&self) -> u64 { @@ -255,60 +353,54 @@ impl PartitionStats { pub fn increment_size_bytes(&self, size_bytes: u64) { self.size_bytes.fetch_add(size_bytes, Ordering::AcqRel); - self.increment_parent_size_bytes(size_bytes); + self.parent.increment_size_bytes(size_bytes); } pub fn increment_messages_count(&self, messages_count: u64) { self.messages_count .fetch_add(messages_count, Ordering::AcqRel); - self.increment_parent_messages_count(messages_count); + self.parent.increment_messages_count(messages_count); } pub fn increment_segments_count(&self, segments_count: u32) { self.segments_count .fetch_add(segments_count, Ordering::AcqRel); - self.increment_parent_segments_count(segments_count); - } - - pub fn increment_parent_size_bytes(&self, size_bytes: u64) { - self.parent.increment_size_bytes(size_bytes); - } - - pub fn increment_parent_messages_count(&self, messages_count: u64) { - self.parent.increment_messages_count(messages_count); - } - - pub fn increment_parent_segments_count(&self, segments_count: u32) { self.parent.increment_segments_count(segments_count); } - pub fn decrement_size_bytes(&self, size_bytes: u64) { - self.size_bytes.fetch_sub(size_bytes, Ordering::AcqRel); - self.decrement_parent_size_bytes(size_bytes); - } - - pub fn decrement_messages_count(&self, messages_count: u64) { - self.messages_count - .fetch_sub(messages_count, Ordering::AcqRel); - self.decrement_parent_messages_count(messages_count); - } - - pub fn decrement_segments_count(&self, segments_count: u32) { - self.segments_count - .fetch_sub(segments_count, Ordering::AcqRel); - self.decrement_parent_segments_count(segments_count); - } - - pub fn decrement_parent_size_bytes(&self, size_bytes: u64) { - self.parent.decrement_size_bytes(size_bytes); - } - - pub fn decrement_parent_messages_count(&self, messages_count: u64) { - self.parent.decrement_messages_count(messages_count); - } - - pub fn decrement_parent_segments_count(&self, segments_count: u32) { - self.parent.decrement_segments_count(segments_count); + // Forward what this level actually gave up, not what was asked for. A + // decrement bigger than the counter holds means the bytes were never in + // this subtree, so the ancestors do not hold them either -- passing the + // full amount up would take them out of a sibling's live data instead. + // Under `parent == sum(children)` the two are equal and nothing changes. + // + // Each returns this level's shortfall -- how much of the amount it did not + // hold, zero when the decrement was covered -- so a caller that knows the + // ids can name where the divergence is. + pub fn decrement_size_bytes(&self, size_bytes: u64) -> u64 { + let taken = clamped_sub_u64(&self.size_bytes, size_bytes, &PARTITION_SIZE_BYTES); + self.parent.decrement_size_bytes(taken); + size_bytes - taken + } + + pub fn decrement_messages_count(&self, messages_count: u64) -> u64 { + let taken = clamped_sub_u64( + &self.messages_count, + messages_count, + &PARTITION_MESSAGES_COUNT, + ); + self.parent.decrement_messages_count(taken); + messages_count - taken + } + + pub fn decrement_segments_count(&self, segments_count: u32) -> u32 { + let taken = clamped_sub_u32( + &self.segments_count, + segments_count, + &PARTITION_SEGMENTS_COUNT, + ); + self.parent.decrement_segments_count(taken); + segments_count - taken } pub fn size_bytes_inconsistent(&self) -> u64 { @@ -357,3 +449,116 @@ impl PartitionStats { self.zero_out_current_offset(); } } + +#[cfg(test)] +mod tests { + use super::*; + + fn tree() -> (Arc, Arc, Arc) { + let stream = Arc::new(StreamStats::default()); + let topic = Arc::new(TopicStats::new(stream.clone())); + let partition = Arc::new(PartitionStats::new(topic.clone())); + (stream, topic, partition) + } + + #[test] + fn given_a_parent_short_of_its_child_when_rolling_back_should_clamp_instead_of_wrapping() { + let (stream, topic, partition) = tree(); + partition.increment_size_bytes(512); + // A snapshot restore stores absolute parent totals, so a parent can end + // up holding less than its children do. + topic.store_from_snapshot(0, 0, 0); + stream.store_from_snapshot(0, 0, 0); + + partition.zero_out_all(); + + assert_eq!(topic.size_bytes_inconsistent(), 0); + assert_eq!(stream.size_bytes_inconsistent(), 0); + } + + /// A deleted partition keeps its handle until the reconciler tears it down, + /// so a retention sweep can decrement counters the delete already rolled + /// back. Those bytes are not in the parents any more, and taking them again + /// would take a live sibling's instead. + #[test] + fn given_a_rolled_back_partition_when_a_late_decrement_arrives_should_leave_siblings_alone() { + let stream = Arc::new(StreamStats::default()); + let topic = Arc::new(TopicStats::new(stream.clone())); + let survivor = Arc::new(PartitionStats::new(topic.clone())); + let deleted = Arc::new(PartitionStats::new(topic.clone())); + survivor.increment_size_bytes(488); + deleted.increment_size_bytes(512); + + deleted.zero_out_all(); + assert_eq!(topic.size_bytes_inconsistent(), 488); + + // Retention retiring a segment of the partition that is on its way out. + deleted.decrement_size_bytes(100); + + assert_eq!( + topic.size_bytes_inconsistent(), + 488, + "the survivor's bytes are the only ones left; they are not the deleted partition's" + ); + assert_eq!(stream.size_bytes_inconsistent(), 488); + assert_eq!(survivor.size_bytes_inconsistent(), 488); + } + + /// `delete_topic` settles the topic's residue on the stream, and the + /// reconciler later settles the same partition's own handle. The second + /// pass must find nothing left to take, or it takes it from another topic. + #[test] + fn given_a_topic_already_settled_when_its_partition_settles_should_leave_other_topics_alone() { + let stream = Arc::new(StreamStats::default()); + let doomed_topic = Arc::new(TopicStats::new(stream.clone())); + let live_topic = Arc::new(TopicStats::new(stream.clone())); + let doomed_partition = Arc::new(PartitionStats::new(doomed_topic.clone())); + let live_partition = Arc::new(PartitionStats::new(live_topic.clone())); + live_partition.increment_size_bytes(900); + // Landed through the cached handle after the delete evicted its entry. + doomed_partition.increment_size_bytes(100); + assert_eq!(stream.size_bytes_inconsistent(), 1000); + + doomed_topic.zero_out_all(); + assert_eq!(stream.size_bytes_inconsistent(), 900); + + doomed_partition.zero_out_all(); + + assert_eq!( + stream.size_bytes_inconsistent(), + 900, + "the surviving topic's bytes must not pay for the deleted one's residue" + ); + assert_eq!(live_topic.size_bytes_inconsistent(), 900); + } + + #[test] + fn given_a_short_u32_segments_count_when_rolling_back_should_clamp_that_counter_too() { + let (stream, topic, partition) = tree(); + partition.increment_segments_count(3); + topic.store_from_snapshot(0, 0, 0); + stream.store_from_snapshot(0, 0, 0); + + partition.zero_out_all(); + + // `clamped_sub!` expands separately per width, so the u32 counter is + // its own wiring: same `saturating_sub` body, its own `UnderflowSite` + // and its own call sites at all three levels. + assert_eq!(topic.segments_count_inconsistent(), 0); + assert_eq!(stream.segments_count_inconsistent(), 0); + } + + #[test] + fn given_a_covered_decrement_when_rolling_back_should_subtract_exactly() { + let (stream, topic, partition) = tree(); + partition.increment_size_bytes(512); + partition.increment_messages_count(7); + + partition.decrement_size_bytes(112); + + assert_eq!(partition.size_bytes_inconsistent(), 400); + assert_eq!(topic.size_bytes_inconsistent(), 400); + assert_eq!(stream.size_bytes_inconsistent(), 400); + assert_eq!(stream.messages_count_inconsistent(), 7); + } +} diff --git a/core/integration/tests/data_integrity/verify_after_server_restart.rs b/core/integration/tests/data_integrity/verify_after_server_restart.rs index e831eae1fd..a9dece2003 100644 --- a/core/integration/tests/data_integrity/verify_after_server_restart.rs +++ b/core/integration/tests/data_integrity/verify_after_server_restart.rs @@ -15,6 +15,7 @@ // specific language governing permissions and limitations // under the License. +use bytes::Bytes; use iggy::prelude::*; use integration::bench_utils::run_bench_and_wait_for_finish; use integration::harness::{TestHarness, TestServerConfig}; @@ -307,6 +308,23 @@ async fn should_handle_resource_deletion_and_restart() { let topic_ident = Identifier::numeric(topic_idx).unwrap(); + // One pass per partition. Without messages every delete below + // removes an empty scope, and the stats assertions across the + // restart would hold whether or not a delete rolls its bytes out + // of the parent totals. + for partition_id in 0..3u32 { + let mut messages = deletion_test_messages(); + client + .send_messages( + &stream_ident, + &topic_ident, + &Partitioning::partition_id(partition_id), + &mut messages, + ) + .await + .unwrap(); + } + // Create 3 consumer groups per topic for cg_idx in 0..3 { client @@ -400,6 +418,11 @@ async fn should_handle_resource_deletion_and_restart() { stream_ids ); + // Deletes roll their bytes out of the parent totals at commit time; a + // restart rebuilds those totals from disk instead. The two have to agree, + // or one of the paths still counts data the other already removed. + let stats_before_restart = client.get_stats().await.unwrap(); + drop(client); // Restart server @@ -407,6 +430,21 @@ async fn should_handle_resource_deletion_and_restart() { let client = harness.tcp_root_client().await.unwrap(); + let stats_after_restart = client.get_stats().await.unwrap(); + assert!( + stats_before_restart.messages_count > 0, + "the deletes above must leave messages behind, or the comparison is vacuous" + ); + assert_eq!( + stats_before_restart.messages_count, stats_after_restart.messages_count, + "messages_count must survive the restart unchanged after the deletes" + ); + assert_eq!( + stats_before_restart.messages_size_bytes.as_bytes_u64(), + stats_after_restart.messages_size_bytes.as_bytes_u64(), + "messages_size_bytes must survive the restart unchanged after the deletes" + ); + // Verify streams after restart - should have 3 streams: 0, 1 (reused), 2 let streams = client.get_streams().await.unwrap(); assert_eq!(streams.len(), 3, "Expected 3 streams after restart"); @@ -520,3 +558,17 @@ async fn should_handle_resource_deletion_and_restart() { ); } } + +/// Payload for `should_handle_resource_deletion_and_restart`: small, fixed and +/// non-empty, so a scope that gets deleted has bytes worth rolling back. +fn deletion_test_messages() -> Vec { + (0..8u128) + .map(|offset| { + IggyMessage::builder() + .id(offset + 1) + .payload(Bytes::from_static(b"deletion-stats-payload")) + .build() + .expect("Failed to build message") + }) + .collect() +} diff --git a/core/integration/tests/server/general.rs b/core/integration/tests/server/general.rs index 861d24895a..f1dd8382bf 100644 --- a/core/integration/tests/server/general.rs +++ b/core/integration/tests/server/general.rs @@ -16,9 +16,9 @@ // under the License. use crate::server::scenarios::{ - authentication_scenario, consumer_timestamp_polling_scenario, invalid_consumer_offset_scenario, - message_headers_scenario, permissions_scenario, snapshot_scenario, - stream_size_validation_scenario, system_scenario, user_scenario, + authentication_scenario, consumer_timestamp_polling_scenario, delete_stats_rollback_scenario, + invalid_consumer_offset_scenario, message_headers_scenario, permissions_scenario, + snapshot_scenario, stream_size_validation_scenario, system_scenario, user_scenario, }; use integration::iggy_harness; @@ -88,6 +88,14 @@ async fn stream_size_validation(harness: &TestHarness) { stream_size_validation_scenario::run(harness).await; } +// One transport: the scenario asserts server-side rollback arithmetic, and +// every call it makes (`get_stream`, `get_topic`, `get_stats`) is already +// exercised on all four transports by the scenarios above. +#[iggy_harness(test_client_transport = [Tcp])] +async fn delete_stats_rollback(harness: &TestHarness) { + delete_stats_rollback_scenario::run(harness).await; +} + #[iggy_harness( test_client_transport = [Tcp, Http, Quic, WebSocket], server( diff --git a/core/integration/tests/server/scenarios/delete_stats_rollback_scenario.rs b/core/integration/tests/server/scenarios/delete_stats_rollback_scenario.rs new file mode 100644 index 0000000000..653806445e --- /dev/null +++ b/core/integration/tests/server/scenarios/delete_stats_rollback_scenario.rs @@ -0,0 +1,225 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Deleting data that is still counted must leave the parent totals correct. +//! +//! `stream_size_validation_scenario` covers delete too, but it purges each +//! topic first, so every delete it performs removes an already-empty scope and +//! cannot observe a rollback that never happened. These scenarios delete scopes +//! that still hold messages, which is what the parent totals are wrong about. +//! +//! What this guards, and what it does not. Two mechanisms roll a deleted scope +//! out of its parents: the metadata STM evicts the registry entries at commit +//! and zeroes them into their parents, and the reconciler's partition teardown +//! settles whatever landed after that. The STM half runs inside the apply that +//! produces the ack, on counters shared across every shard and both left-right +//! buffers, so it alone satisfies every assertion here. The reconciler runs as +//! a separate task, but a delete commit wakes it (`signal_reconcile_wake`), so +//! it too finishes before a client can complete another round trip. +//! +//! Measured, not reasoned: with only the STM rollback disabled this scenario +//! PASSES (while the `metadata::stm::stream` unit tests fail), and with only +//! the reconciler settle disabled it also passes. It fails only when BOTH are +//! disabled. Either mechanism alone keeps a client's view correct, so this is a +//! contract test for what a client observes and not a regression guard for +//! either half. +//! +//! Each half is guarded by unit tests instead: the `metadata::stm::stream` +//! module for the rollback and eviction, `server::partition_reconciler` for the +//! teardown settle and the membership gate. The integration layer cannot hold +//! the reconciler off by configuration -- the tick is capped at 30s, cannot be +//! set to zero, and the commit wake bypasses it regardless. +//! +//! Verifying any of this needs `cargo build --bin iggy-server` first. The +//! harness spawns the built binary, so a source-only mutation runs against the +//! previous build and reports a green that means nothing. +//! +//! The reads after each delete are deliberately single-shot. Retrying until the +//! numbers converge, the way the pre-delete assertions do for the async send +//! folding, would also wait out a broken rollback and pass regardless. + +use crate::server::scenarios::{ + PARTITIONS_COUNT, batch_size, create_client, create_messages, validate_stream, + validate_system_stats, validate_topic, +}; +use iggy::prelude::*; +use integration::harness::{TestHarness, assert_clean_system, login_root}; + +const STREAM_NAME: &str = "delete-stats-stream"; +const KEPT_TOPIC: &str = "kept-topic"; +const DELETED_TOPIC: &str = "deleted-topic"; +const MSGS_COUNT: u64 = 17; +const MSGS_SIZE: u64 = batch_size(MSGS_COUNT); + +pub async fn run(harness: &TestHarness) { + let client = create_client(harness).await; + client.ping().await.unwrap(); + login_root(&client).await.expect("login failed"); + + // Partition half first: its assertions are the tighter pair. + delete_partitions_rolls_back_topic_and_stream(&client).await; + delete_topic_rolls_back_the_stream(&client).await; + + // The plainest statement of what this guards: every scope the scenario + // created is gone, so the server-wide totals owe nothing. `assert_clean_system` + // reads the entity lists, never the stats. + validate_system_stats(&client, 0, 0).await; + assert_clean_system(&client).await; +} + +/// Deleting partitions that still hold messages must take their bytes out of +/// both the topic and the stream. The retained partition's messages must stay +/// counted, so a blanket zeroing fails here as loudly as no rollback at all. +async fn delete_partitions_rolls_back_topic_and_stream(client: &IggyClient) { + client.create_stream(STREAM_NAME).await.unwrap(); + create_topic(client, STREAM_NAME, KEPT_TOPIC).await; + + // One pass per partition, so the delete below removes a known share. + for partition_id in 0..PARTITIONS_COUNT { + send_one_pass(client, STREAM_NAME, KEPT_TOPIC, partition_id).await; + } + let all_partitions = MSGS_SIZE * u64::from(PARTITIONS_COUNT); + let all_messages = MSGS_COUNT * u64::from(PARTITIONS_COUNT); + validate_topic( + client, + STREAM_NAME, + KEPT_TOPIC, + all_partitions, + all_messages, + ) + .await; + validate_stream(client, STREAM_NAME, all_partitions, all_messages).await; + + client + .delete_partitions( + &Identifier::named(STREAM_NAME).unwrap(), + &Identifier::named(KEPT_TOPIC).unwrap(), + 1, + ) + .await + .unwrap(); + + // Single-shot again: the retained partitions' messages must still be + // counted, so this fails both on a missing rollback and on a blanket + // zeroing of the parents. + let retained = u64::from(PARTITIONS_COUNT - 1); + let topic = client + .get_topic( + &Identifier::named(STREAM_NAME).unwrap(), + &Identifier::named(KEPT_TOPIC).unwrap(), + ) + .await + .unwrap() + .expect("Failed to get topic"); + assert_eq!( + topic.size, + MSGS_SIZE * retained, + "the deleted partitions' bytes must leave the topic total on the delete's commit" + ); + assert_eq!(topic.messages_count, MSGS_COUNT * retained); + let stream = client + .get_stream(&Identifier::named(STREAM_NAME).unwrap()) + .await + .unwrap() + .expect("Failed to get stream"); + assert_eq!( + stream.size, + MSGS_SIZE * retained, + "the deleted partitions' bytes must leave the stream total too" + ); + assert_eq!(stream.messages_count, MSGS_COUNT * retained); + + client + .delete_stream(&Identifier::named(STREAM_NAME).unwrap()) + .await + .unwrap(); +} + +/// Deleting a topic that still holds messages must take its bytes out of the +/// stream total, not just remove the topic. +async fn delete_topic_rolls_back_the_stream(client: &IggyClient) { + client.create_stream(STREAM_NAME).await.unwrap(); + create_topic(client, STREAM_NAME, KEPT_TOPIC).await; + create_topic(client, STREAM_NAME, DELETED_TOPIC).await; + + send_one_pass(client, STREAM_NAME, KEPT_TOPIC, 0).await; + send_one_pass(client, STREAM_NAME, DELETED_TOPIC, 0).await; + validate_stream(client, STREAM_NAME, MSGS_SIZE * 2, MSGS_COUNT * 2).await; + + client + .delete_topic( + &Identifier::named(STREAM_NAME).unwrap(), + &Identifier::named(DELETED_TOPIC).unwrap(), + ) + .await + .unwrap(); + + // Read once, with no convergence retry: the delete acks on the metadata + // commit, so the totals must already be right in the reply the client is + // holding. A retry here would wait for the reconciler's on-disk wipe and + // hide the window entirely. + let stream = client + .get_stream(&Identifier::named(STREAM_NAME).unwrap()) + .await + .unwrap() + .expect("Failed to get stream"); + assert_eq!( + stream.size, MSGS_SIZE, + "the deleted topic's bytes must leave the stream total on the commit that acked the delete" + ); + assert_eq!(stream.messages_count, MSGS_COUNT); + validate_topic(client, STREAM_NAME, KEPT_TOPIC, MSGS_SIZE, MSGS_COUNT).await; + validate_system_stats(client, MSGS_SIZE, MSGS_COUNT).await; + + client + .delete_stream(&Identifier::named(STREAM_NAME).unwrap()) + .await + .unwrap(); +} + +async fn create_topic(client: &IggyClient, stream_name: &str, topic_name: &str) { + client + .create_topic( + &Identifier::named(stream_name).unwrap(), + topic_name, + &TopicCreateOptions { + partitions_count: Some(PARTITIONS_COUNT), + message_expiry: Some(IggyExpiry::NeverExpire), + ..TopicCreateOptions::default() + }, + ) + .await + .unwrap(); +} + +async fn send_one_pass( + client: &IggyClient, + stream_name: &str, + topic_name: &str, + partition_id: u32, +) { + let mut messages = create_messages(MSGS_COUNT); + client + .send_messages( + &Identifier::named(stream_name).unwrap(), + &Identifier::named(topic_name).unwrap(), + &Partitioning::partition_id(partition_id), + &mut messages, + ) + .await + .unwrap(); +} diff --git a/core/integration/tests/server/scenarios/mod.rs b/core/integration/tests/server/scenarios/mod.rs index 0143621e9c..c6451d7826 100644 --- a/core/integration/tests/server/scenarios/mod.rs +++ b/core/integration/tests/server/scenarios/mod.rs @@ -31,6 +31,7 @@ pub mod consumer_timestamp_polling_scenario; // shard-0 HTTP listener and the create/delete commit through the metadata STM, // so the token replicates to every shard a TCP client may land on. pub mod cross_protocol_pat_scenario; +pub mod delete_stats_rollback_scenario; pub mod encryption_scenario; pub mod invalid_consumer_offset_scenario; pub mod log_rotation_scenario; @@ -54,14 +55,21 @@ pub mod timestamp_scenario; pub mod user_scenario; pub mod websocket_tls_scenario; +use bytes::Bytes; use iggy::prelude::*; use integration::harness::{TestHarness, delete_user}; use std::time::{Duration, Instant}; use tokio::time::sleep; const PARTITION_ID: u32 = 0; -const POLL_CONVERGENCE_TIMEOUT: Duration = Duration::from_secs(10); -const POLL_RETRY_INTERVAL: Duration = Duration::from_millis(100); +// One pair for every wait in these scenarios, because they all wait out the +// same thing: the partition plane applies committed ops asynchronously on the +// owning shard (a send folds into the shared stats and into the servable log at +// commit-apply; purge and delete zero them when the reconciler drives the wipe), +// so a read racing that window sees a pre-apply value. Retry until the +// expectation holds, then make the terminal assertion for a real mismatch. +const CONVERGENCE_TIMEOUT: Duration = Duration::from_secs(10); +const RETRY_INTERVAL: Duration = Duration::from_millis(100); const STREAM_NAME: &str = "test-stream"; const TOPIC_NAME: &str = "test-topic"; const PARTITIONS_COUNT: u32 = 3; @@ -72,8 +80,127 @@ const USERNAME_3: &str = "user3"; const CONSUMER_KIND: ConsumerKind = ConsumerKind::Consumer; const MESSAGES_COUNT: u32 = 1337; +const MESSAGE_PAYLOAD_SIZE_BYTES: u64 = 57; +// The server accounts the actual on-disk batch framing: one 256-byte +// `SendMessages` command header per append pass plus a 48-byte per-message +// header (`server_common::send_messages::COMMAND_HEADER_SIZE` and +// `iggy_binary_protocol::batch::BATCH_MESSAGE_HEADER_SIZE`). +const NG_BATCH_HEADER_SIZE: u64 = 256; +const NG_MESSAGE_HEADER_SIZE: u64 = 48; + +/// What the server counts for one append pass of `messages_count` messages +/// built by [`create_messages`]. +const fn batch_size(messages_count: u64) -> u64 { + NG_BATCH_HEADER_SIZE + (NG_MESSAGE_HEADER_SIZE + MESSAGE_PAYLOAD_SIZE_BYTES) * messages_count +} + +/// One append pass worth of messages, sized so [`batch_size`] predicts what the +/// server will report for them. +fn create_messages(messages_count: u64) -> Vec { + (0..messages_count) + .map(|offset| { + let payload = Bytes::from(vec![0xD; MESSAGE_PAYLOAD_SIZE_BYTES as usize]); + IggyMessage::builder() + .id(u128::from(offset) + 1) + .payload(payload) + .build() + .expect("Failed to create message") + }) + .collect() +} + +/// Fetch the stream until its totals match or [`CONVERGENCE_TIMEOUT`] +/// expires, then assert on the last read. +async fn validate_stream( + client: &IggyClient, + stream_name: &str, + expected_size: u64, + expected_messages_count: u64, +) { + let deadline = Instant::now() + CONVERGENCE_TIMEOUT; + let stream = loop { + let stream = client + .get_stream(&Identifier::named(stream_name).unwrap()) + .await + .unwrap() + .expect("Failed to get stream"); + if (stream.size == expected_size && stream.messages_count == expected_messages_count) + || Instant::now() >= deadline + { + break stream; + } + sleep(RETRY_INTERVAL).await; + }; + assert_eq!(stream.size, expected_size, "stream size mismatch"); + assert_eq!( + stream.messages_count, expected_messages_count, + "stream messages_count mismatch" + ); +} + +/// Topic-level counterpart of [`validate_stream`]. +async fn validate_topic( + client: &IggyClient, + stream_name: &str, + topic_name: &str, + expected_size: u64, + expected_messages_count: u64, +) { + let deadline = Instant::now() + CONVERGENCE_TIMEOUT; + let topic = loop { + let topic = client + .get_topic( + &Identifier::named(stream_name).unwrap(), + &Identifier::named(topic_name).unwrap(), + ) + .await + .unwrap() + .expect("Failed to get topic"); + if (topic.size == expected_size && topic.messages_count == expected_messages_count) + || Instant::now() >= deadline + { + break topic; + } + sleep(RETRY_INTERVAL).await; + }; + assert_eq!(topic.size, expected_size, "topic size mismatch"); + assert_eq!( + topic.messages_count, expected_messages_count, + "topic messages_count mismatch" + ); +} + +/// Server-wide counterpart of [`validate_stream`], reading `[stats]` rather +/// than one entity. +async fn validate_system_stats( + client: &IggyClient, + expected_size: u64, + expected_messages_count: u64, +) { + let deadline = Instant::now() + CONVERGENCE_TIMEOUT; + let stats = loop { + let stats = client.get_stats().await.unwrap(); + if (stats.messages_count == expected_messages_count + && stats.messages_size_bytes.as_bytes_u64() == expected_size) + || Instant::now() >= deadline + { + break stats; + } + sleep(RETRY_INTERVAL).await; + }; + assert_eq!( + stats.messages_count, expected_messages_count, + "system stats messages_count mismatch" + ); + assert_eq!( + stats.messages_size_bytes.as_bytes_u64(), + expected_size, + "system stats messages_size_bytes mismatch" + ); +} + /// Poll until the partition serves `expected_count` messages or -/// [`POLL_CONVERGENCE_TIMEOUT`] expires, returning the last poll result. +/// [`CONVERGENCE_TIMEOUT`] expires, returning the last poll result. /// /// `send_messages` acks at consensus commit while the owning shard applies /// the batch asynchronously (see the materialisation race note at the top @@ -88,7 +215,7 @@ async fn poll_until_expected_count( strategy: &PollingStrategy, expected_count: u32, ) -> PolledMessages { - let deadline = Instant::now() + POLL_CONVERGENCE_TIMEOUT; + let deadline = Instant::now() + CONVERGENCE_TIMEOUT; loop { let polled = client .poll_messages( @@ -105,7 +232,7 @@ async fn poll_until_expected_count( if polled.messages.len() as u32 == expected_count || Instant::now() >= deadline { return polled; } - sleep(POLL_RETRY_INTERVAL).await; + sleep(RETRY_INTERVAL).await; } } diff --git a/core/integration/tests/server/scenarios/purge_delete_scenario.rs b/core/integration/tests/server/scenarios/purge_delete_scenario.rs index ce222ec623..b29130e2a3 100644 --- a/core/integration/tests/server/scenarios/purge_delete_scenario.rs +++ b/core/integration/tests/server/scenarios/purge_delete_scenario.rs @@ -15,7 +15,7 @@ // specific language governing permissions and limitations // under the License. -use super::{POLL_CONVERGENCE_TIMEOUT, POLL_RETRY_INTERVAL}; +use super::{CONVERGENCE_TIMEOUT, RETRY_INTERVAL}; use bytes::Bytes; use iggy::prelude::*; use iggy_common::Credentials; @@ -1203,14 +1203,14 @@ pub async fn run_resident_purge_no_resurface(harness: &mut TestHarness) { /// Poll from offset 0 with headroom (count 100) until exactly `expected` /// messages are served, so an extra resurfaced message fails the count /// instead of being cropped by the poll size. Panics after -/// [`POLL_CONVERGENCE_TIMEOUT`] with the last observed count. +/// [`CONVERGENCE_TIMEOUT`] with the last observed count. async fn poll_exactly( client: &IggyClient, stream_ident: &Identifier, topic_ident: &Identifier, expected: usize, ) -> PolledMessages { - let deadline = std::time::Instant::now() + POLL_CONVERGENCE_TIMEOUT; + let deadline = std::time::Instant::now() + CONVERGENCE_TIMEOUT; loop { let polled = client .poll_messages( @@ -1232,7 +1232,7 @@ async fn poll_exactly( "poll did not converge to {expected} messages, last saw {}", polled.messages.len() ); - tokio::time::sleep(POLL_RETRY_INTERVAL).await; + tokio::time::sleep(RETRY_INTERVAL).await; } } @@ -1318,7 +1318,7 @@ async fn maybe_restart(harness: &mut TestHarness, client: &IggyClient, restart_s let _ = client.disconnect().await; harness.restart_server().await.unwrap(); - let deadline = tokio::time::Instant::now() + POLL_CONVERGENCE_TIMEOUT; + let deadline = tokio::time::Instant::now() + CONVERGENCE_TIMEOUT; loop { // `connect` re-authenticates from the embedded credentials, so a // successful ping means the shards are serving, not merely listening. @@ -1327,12 +1327,12 @@ async fn maybe_restart(harness: &mut TestHarness, client: &IggyClient, restart_s } assert!( tokio::time::Instant::now() < deadline, - "server did not serve again within {POLL_CONVERGENCE_TIMEOUT:?} of restart" + "server did not serve again within {CONVERGENCE_TIMEOUT:?} of restart" ); // Back to Disconnected, else the next `connect` no-ops on a half-open // connection and the ping keeps failing until the deadline. let _ = client.disconnect().await; - tokio::time::sleep(POLL_RETRY_INTERVAL).await; + tokio::time::sleep(RETRY_INTERVAL).await; } } @@ -1501,17 +1501,17 @@ fn get_sorted_segment_offsets(partition_path: &str) -> Vec { } /// Wait until `dir` contains at least one entry, panicking with `context` -/// once [`POLL_CONVERGENCE_TIMEOUT`] expires. +/// once [`CONVERGENCE_TIMEOUT`] expires. /// /// A stored consumer offset is served from memory as soon as the store is /// acked, while the offset file is created asynchronously, so a single-shot /// existence check can run ahead of the flush. An offset that is never /// flushed still fails once the deadline expires. async fn await_dir_not_empty(dir: &str, context: &str) { - let deadline = std::time::Instant::now() + POLL_CONVERGENCE_TIMEOUT; + let deadline = std::time::Instant::now() + CONVERGENCE_TIMEOUT; while is_dir_empty(dir) { assert!(std::time::Instant::now() < deadline, "{context}"); - tokio::time::sleep(POLL_RETRY_INTERVAL).await; + tokio::time::sleep(RETRY_INTERVAL).await; } } @@ -1550,7 +1550,7 @@ async fn assert_fresh_empty_partition(partition_path: &str) { } /// Asserts no orphaned segment files remain after deletion, polling until the -/// counts converge or [`POLL_CONVERGENCE_TIMEOUT`] expires. +/// counts converge or [`CONVERGENCE_TIMEOUT`] expires. /// /// `get_sorted_segment_offsets` only checks .log files -- this additionally /// verifies that the .index file count matches, catching stale .index files @@ -1558,7 +1558,7 @@ async fn assert_fresh_empty_partition(partition_path: &str) { /// separate awaits, so a layout that already converged on .log files can /// transiently show one extra .index file. async fn assert_no_orphaned_segment_files(partition_path: &str, expected_count: usize) { - let deadline = std::time::Instant::now() + POLL_CONVERGENCE_TIMEOUT; + let deadline = std::time::Instant::now() + CONVERGENCE_TIMEOUT; loop { let log_count = count_files_with_ext(partition_path, LOG_EXTENSION); let index_count = count_files_with_ext(partition_path, INDEX_EXTENSION); @@ -1569,7 +1569,7 @@ async fn assert_no_orphaned_segment_files(partition_path: &str, expected_count: std::time::Instant::now() < deadline, "Expected {expected_count} .log and .index files, found {log_count} .log and {index_count} .index" ); - tokio::time::sleep(POLL_RETRY_INTERVAL).await; + tokio::time::sleep(RETRY_INTERVAL).await; } } diff --git a/core/integration/tests/server/scenarios/stream_size_validation_scenario.rs b/core/integration/tests/server/scenarios/stream_size_validation_scenario.rs index 8d6953afbb..a6dc82f83d 100644 --- a/core/integration/tests/server/scenarios/stream_size_validation_scenario.rs +++ b/core/integration/tests/server/scenarios/stream_size_validation_scenario.rs @@ -15,43 +15,24 @@ // specific language governing permissions and limitations // under the License. -use crate::server::scenarios::{PARTITION_ID, PARTITIONS_COUNT}; -use bytes::Bytes; +use crate::server::scenarios::{ + PARTITION_ID, PARTITIONS_COUNT, batch_size, create_client, create_messages, validate_stream, + validate_system_stats, validate_topic, +}; use iggy::prelude::*; use integration::harness::{TestHarness, assert_clean_system, login_root}; use std::str::FromStr; -use std::time::{Duration, Instant}; -use tokio::time::sleep; - -// The partition plane applies committed ops asynchronously on the owning -// shard (sends fold into the shared stats at commit-apply; purge/delete -// zero them when the reconciler drives the wipe), so a read racing that -// window can see a pre-apply value. Retry until the expectation holds, -// then make the terminal assertion for a real mismatch. -const STATS_CONVERGENCE_TIMEOUT: Duration = Duration::from_secs(10); -const STATS_RETRY_INTERVAL: Duration = Duration::from_millis(100); const S1_NAME: &str = "test-stream-1"; const T1_NAME: &str = "test-topic-1"; const S2_NAME: &str = "test-stream-2"; const T2_NAME: &str = "test-topic-2"; -const MESSAGE_PAYLOAD_SIZE_BYTES: u64 = 57; const MSGS_COUNT: u64 = 117; // number of messages in a single topic after one pass of appending -// The server accounts the actual on-disk batch framing: one 256-byte -// `SendMessages` command header per append pass plus a 48-byte per-message -// header (`server_common::send_messages::COMMAND_HEADER_SIZE` and -// `iggy_binary_protocol::batch::BATCH_MESSAGE_HEADER_SIZE`). Each pass below -// sends all `MSGS_COUNT` messages in one batch. -const NG_BATCH_HEADER_SIZE: u64 = 256; -const NG_MESSAGE_HEADER_SIZE: u64 = 48; -const MSGS_SIZE: u64 = - NG_BATCH_HEADER_SIZE + (NG_MESSAGE_HEADER_SIZE + MESSAGE_PAYLOAD_SIZE_BYTES) * MSGS_COUNT; +// Each pass below sends all `MSGS_COUNT` messages in one batch. +const MSGS_SIZE: u64 = batch_size(MSGS_COUNT); pub async fn run(harness: &TestHarness) { - let client = harness - .new_client() - .await - .expect("Failed to create new client"); + let client = create_client(harness).await; // 0. Ping server, login as root user and ensure that streams do not exist ping_login_and_validate(&client).await; @@ -198,7 +179,7 @@ async fn validate_operations_on_topic_twice( partition_id: u32, ) { // 1. Append messages to the topic - let mut messages = create_messages(); + let mut messages = create_messages(MSGS_COUNT); client .send_messages( &Identifier::from_str(stream_name).unwrap(), @@ -213,7 +194,7 @@ async fn validate_operations_on_topic_twice( validate_topic(client, stream_name, topic_name, MSGS_SIZE, MSGS_COUNT).await; // 3. Again append same number of messages to the topic - let mut messages = create_messages(); + let mut messages = create_messages(MSGS_COUNT); client .send_messages( &Identifier::from_str(stream_name).unwrap(), @@ -235,93 +216,6 @@ async fn validate_operations_on_topic_twice( .await; } -async fn validate_system_stats( - client: &IggyClient, - expected_size: u64, - expected_messages_count: u64, -) { - let deadline = Instant::now() + STATS_CONVERGENCE_TIMEOUT; - let stats = loop { - let stats = client.get_stats().await.unwrap(); - if (stats.messages_count == expected_messages_count - && stats.messages_size_bytes.as_bytes_u64() == expected_size) - || Instant::now() >= deadline - { - break stats; - } - sleep(STATS_RETRY_INTERVAL).await; - }; - assert_eq!( - stats.messages_count, expected_messages_count, - "system stats messages_count mismatch" - ); - assert_eq!( - stats.messages_size_bytes.as_bytes_u64(), - expected_size, - "system stats messages_size_bytes mismatch" - ); -} - -async fn validate_stream( - client: &IggyClient, - stream_name: &str, - expected_size: u64, - expected_messages_count: u64, -) { - // 1. Fetch until the async commit-apply converges (see the retry note at - // the top of the file). - let deadline = Instant::now() + STATS_CONVERGENCE_TIMEOUT; - let stream = loop { - let stream = client - .get_stream(&Identifier::from_str(stream_name).unwrap()) - .await - .unwrap() - .expect("Failed to get stream"); - if (stream.size == expected_size && stream.messages_count == expected_messages_count) - || Instant::now() >= deadline - { - break stream; - } - sleep(STATS_RETRY_INTERVAL).await; - }; - - // 2. Validate stream size and number of messages - assert_eq!(stream.size, expected_size); - assert_eq!(stream.messages_count, expected_messages_count); -} - -async fn validate_topic( - client: &IggyClient, - stream_name: &str, - topic_name: &str, - expected_size: u64, - expected_messages_count: u64, -) { - // 1. Fetch until the async commit-apply converges (see the retry note at - // the top of the file). - let deadline = Instant::now() + STATS_CONVERGENCE_TIMEOUT; - let topic = loop { - let topic = client - .get_topic( - &Identifier::from_str(stream_name).unwrap(), - &Identifier::from_str(topic_name).unwrap(), - ) - .await - .unwrap() - .expect("Failed to get topic"); - if (topic.size == expected_size && topic.messages_count == expected_messages_count) - || Instant::now() >= deadline - { - break topic; - } - sleep(STATS_RETRY_INTERVAL).await; - }; - - // 2. Validate topic size and number of messages - assert_eq!(topic.size, expected_size); - assert_eq!(topic.messages_count, expected_messages_count); -} - async fn delete_topic(client: &IggyClient, stream_name: &str, topic_name: &str) { client .delete_topic( @@ -355,19 +249,3 @@ async fn purge_stream(client: &IggyClient, stream_name: &str) { .await .unwrap(); } - -fn create_messages() -> Vec { - let mut messages = Vec::new(); - for offset in 0..MSGS_COUNT { - let id = (offset + 1) as u128; - let payload = Bytes::from(vec![0xD; MESSAGE_PAYLOAD_SIZE_BYTES as usize]); - - let message = IggyMessage::builder() - .id(id) - .payload(payload) - .build() - .expect("Failed to create message"); - messages.push(message); - } - messages -} diff --git a/core/metadata/src/stm/stream.rs b/core/metadata/src/stm/stream.rs index 119a9f4762..f2707ea0f7 100644 --- a/core/metadata/src/stm/stream.rs +++ b/core/metadata/src/stm/stream.rs @@ -151,6 +151,11 @@ impl Partition { } /// Stats snapshot representation for serialization. +/// +/// Carried for format compatibility only. Every restore path recomputes the +/// totals it describes: `restore_in_place` rebuilds them from the partition +/// entries this node holds, and boot zeroes them and re-folds the on-disk +/// deltas. Dropping the fields would break decode of existing checkpoints. #[derive(Debug, Clone, Serialize, Deserialize)] pub struct StatsSnapshot { pub size_bytes: u64, @@ -384,10 +389,33 @@ impl Stream { /// read. /// /// Shared across buffers and reader shards via `Arc` (a `StreamsInner` clone -/// shares it). Only shard 0's writer mutates the maps, under the -/// single-threaded Absorb; the `Mutex` is for `Sync` (uncontended), not -/// concurrency. Ids are deterministic across replicas (same op order), so both -/// buffers resolve the same key. +/// shares it). Shard 0's writer owns the map mutations that ride the op-log, +/// but it is NOT the only writer: [`Self::partition`] is a get-or-create the +/// owning shard's reconciler calls when it materializes a namespace, and boot +/// recovery calls it from every shard at once. The `Mutex` is load-bearing for +/// that concurrency, not merely for `Sync`. Ids are deterministic across +/// replicas (same op order), so both buffers resolve the same key. +/// +/// Every eviction here therefore runs twice, once per buffer, and the second +/// run is deferred. Two properties keep that from doing damage, and neither is +/// guaranteed by the `Absorb` trait: +/// +/// * `WriteCell::apply` is `append(cmd).publish()`, so a batch is one op. Batch +/// two and a delete could be absorbed on the second buffer AFTER a create +/// that re-minted the same id, and the delete would then evict the new +/// partition's entry. +/// * `left_right` drains the previous batch's `absorb_second` before the +/// current batch's `absorb_first` (0.11.8 `write.rs`), which is an +/// implementation detail, not a contract -- which is why the workspace +/// manifest pins `left-right` to `=0.11.8` rather than floating 0.11.x. +/// +/// Narrowing both, `fetch_partition_build_inputs` refuses to register a +/// partition the committed topic does not list. It does not close the window: +/// the check reads the left-right READ side, and the eviction lands in +/// `absorb_first` on the write side, so a get-or-create between that apply and +/// the next `publish()` still sees the partition listed and can mint the entry +/// back. What it does buy is that the resurrected entry carries an id the +/// evicting op named, so the deferred `absorb_second` run evicts it again. #[derive(Debug, Default)] pub struct StatsRegistry { streams: std::sync::Mutex>>, @@ -408,6 +436,14 @@ struct PartitionEntry { /// purge acked. Counters are shared side state (one `Arc` across buffers), /// so an ungated second reset would wipe messages sent since the purge. purged_generation: u64, + /// The committed [`Partition::created_revision`] these counters belong to. + /// + /// Slab keys are recycled, so the key alone does not say WHICH partition an + /// entry counts. [`StatsRegistry::retain_from_snapshot`] is where that + /// matters: a donor that recycled a key would otherwise leave the receiver + /// holding the dead occupant's counters, and `rebuild_parent_totals` then + /// makes them the topic and stream totals. + created_revision: u64, } impl StatsRegistry { @@ -439,22 +475,40 @@ impl StatsRegistry { /// reads the same `Arc`, so partition-plane counters are visible /// cross-shard without a gather. /// + /// Takes the committed record rather than a bare id, because a fresh entry + /// has to inherit two things from it. `created_revision` is the identity + /// `retain_from_snapshot` compares against. `purge_generation` is + /// the reset gate: mint it at 0 and a purge that committed while this + /// partition was torn down and rebuilt still counts as pending, so its + /// deferred second-buffer apply wipes everything appended since. + /// + /// Caller contract, unchecked either way: `partition` must be the committed + /// record listed under `(stream_id, topic_id)`, and `parent` the committed + /// topic's own `Arc`. A fresh entry inherits both without comparing them to + /// anything, so a record from another topic seeds the identity gate with a + /// revision that never matches, and a parent from another topic sends this + /// partition's increments into a stranger's totals for the entry's whole + /// life. Read all three out of one `streams().read` (see + /// `fetch_partition_build_inputs` in the reconciler) and both hold by + /// construction. + /// /// # Panics /// If the registry mutex is poisoned. pub fn partition( &self, stream_id: usize, topic_id: usize, - partition_id: usize, + partition: &Partition, parent: Arc, ) -> Arc { self.partitions .lock() .expect("stats registry mutex poisoned") - .entry((stream_id, topic_id, partition_id)) + .entry((stream_id, topic_id, partition.id)) .or_insert_with(|| PartitionEntry { stats: Arc::new(PartitionStats::new(parent)), - purged_generation: 0, + purged_generation: partition.purge_generation, + created_revision: partition.created_revision, }) .stats .clone() @@ -519,6 +573,7 @@ impl StatsRegistry { .or_insert_with(|| PartitionEntry { stats: Arc::new(PartitionStats::new(Arc::clone(parent))), purged_generation: 0, + created_revision: partition.created_revision, }); if entry.purged_generation >= partition.purge_generation { continue; @@ -534,39 +589,100 @@ impl StatsRegistry { } } - fn remove_stream(&self, id: usize) { + /// Evict only, no rollback: `StreamStats` is the root of the rollup, so + /// there is no parent total that owes the dropped counters back, and both + /// readers (`get_stats`, the `/metrics` gauges) reach stream totals by + /// walking the live streams, which this one has already left. + /// + /// Keyed off the committed tree the caller still holds, not a predicate + /// sweep: every entry in the three maps was inserted under ids taken from + /// that tree, so walking it names all of them, and a `retain` would instead + /// walk every entry on the node per deleted stream. + fn remove_stream(&self, id: usize, stream: &Stream) { self.streams .lock() .expect("stats registry mutex poisoned") .remove(&id); - self.topics - .lock() - .expect("stats registry mutex poisoned") - .retain(|(stream_id, _), _| *stream_id != id); - self.partitions - .lock() - .expect("stats registry mutex poisoned") - .retain(|(stream_id, _, _), _| *stream_id != id); + { + let mut partitions = self + .partitions + .lock() + .expect("stats registry mutex poisoned"); + for (topic_id, topic) in &stream.topics { + for partition in &topic.partitions { + partitions.remove(&(id, topic_id, partition.id)); + } + } + } + { + let mut topics = self.topics.lock().expect("stats registry mutex poisoned"); + for (topic_id, _) in &stream.topics { + topics.remove(&(id, topic_id)); + } + } } - fn remove_topic(&self, stream_id: usize, topic_id: usize) { - self.topics + /// Roll a deleted topic out of its stream and drop every entry under it. + /// + /// `partition_ids` comes off the committed topic the caller is about to + /// drop. Keyed, not swept: see [`Self::remove_partitions`]. + fn remove_topic(&self, stream_id: usize, topic_id: usize, partition_ids: &[usize]) { + // Partitions first: each one's rollback cascades through its parent + // topic into the stream, so zeroing the topic ahead of them would + // subtract the same bytes from the stream twice. + self.remove_partitions(stream_id, topic_id, partition_ids); + let topic = self + .topics .lock() .expect("stats registry mutex poisoned") .remove(&(stream_id, topic_id)); - self.partitions - .lock() - .expect("stats registry mutex poisoned") - .retain(|(sid, tid, _), _| !(*sid == stream_id && *tid == topic_id)); + // Whatever the topic still counts after the loop above is what its + // partitions did not account for: a snapshot restore that stored a + // total this node's partition entries never contributed, or bytes a + // partition kept adding after its own entry was evicted. Swapping it + // out settles that residue on the stream instead of stranding it. + if let Some(topic) = topic { + topic.zero_out_all(); + } } - fn remove_partitions_from(&self, stream_id: usize, topic_id: usize, first_removed: usize) { - self.partitions - .lock() - .expect("stats registry mutex poisoned") - .retain(|(sid, tid, pid), _| { - !(*sid == stream_id && *tid == topic_id && *pid >= first_removed) - }); + /// Roll the named partitions out of their parents and drop their entries. + /// + /// Both halves are load-bearing. A partition reports by incrementing its + /// parents through the `Arc`, so evicting alone strands what it contributed + /// in the topic and stream totals. And zeroing alone leaves an entry that + /// outlives its ids: a topic's slab key is recycled by the next + /// `create_topic`, and `DeletePartitions` truncates the tail so the next + /// `CreatePartitions` mints the freed ids again. + /// + /// Ids, never positions. `DeletePartitions` truncates the tail of a `Vec`, + /// and a count-based predicate (`id >= retained`) only picks the same set + /// while ids happen to be dense; when they are not it evicts and zeroes a + /// SURVIVING partition, which strips its bytes from the parents and resets + /// the offset its clients store against. + /// + /// # Panics + /// If the registry mutex is poisoned. + pub fn remove_partitions(&self, stream_id: usize, topic_id: usize, removed_ids: &[usize]) { + // Keyed removes, not a predicate: the ids are known, and a predicate + // makes the map walk every entry on the node to drop a handful of them. + // Bulk deletes call this per topic, so that walk squares. + let dropped: Vec> = { + let mut entries = self + .partitions + .lock() + .expect("stats registry mutex poisoned"); + removed_ids + .iter() + .filter_map(|partition_id| entries.remove(&(stream_id, topic_id, *partition_id))) + .map(|entry| entry.stats) + .collect() + }; + // Guard released first: the rollback cascades into parent totals, which + // the partition map has no part in. + for stats in dropped { + stats.zero_out_all(); + } } /// Drop every entry the snapshot does not describe, keeping the rest. @@ -577,18 +693,29 @@ impl StatsRegistry { /// or every already-materialized partition reads (0,0,0,0) forever. Slab /// keys are recycled, so anything the snapshot dropped has to go with it. /// + /// A partition key is kept only when the snapshot's `created_revision` + /// matches the entry's. The key alone says nothing about identity: behind a + /// donor that deleted and re-created a partition into the same slot, a + /// key-only retain hands the new occupant the dead one's counters, and + /// [`Self::rebuild_parent_totals`] then makes them the topic and stream + /// totals. Dropping the mismatch instead costs one re-registration by the + /// reconciler, which folds the on-disk delta back in. + /// /// # Panics /// If the registry mutex is poisoned. fn retain_from_snapshot(&self, snapshot: &StreamsSnapshot) { let mut live_streams: AHashSet = AHashSet::new(); let mut live_topics: AHashSet<(usize, usize)> = AHashSet::new(); - let mut live_partitions: AHashSet<(usize, usize, usize)> = AHashSet::new(); + let mut live_partitions: AHashMap<(usize, usize, usize), u64> = AHashMap::new(); for (stream_key, stream) in &snapshot.items { live_streams.insert(*stream_key); for (topic_key, topic) in &stream.topics { live_topics.insert((*stream_key, *topic_key)); for partition in &topic.partitions { - live_partitions.insert((*stream_key, *topic_key, partition.id)); + live_partitions.insert( + (*stream_key, *topic_key, partition.id), + partition.created_revision, + ); } } } @@ -600,10 +727,117 @@ impl StatsRegistry { .lock() .expect("stats registry mutex poisoned") .retain(|key, _| live_topics.contains(key)); - self.partitions - .lock() - .expect("stats registry mutex poisoned") - .retain(|key, _| live_partitions.contains(key)); + // Zeroing is the other half of the eviction, exactly as in + // [`Self::remove_partitions`]: a stale incarnation still mounted holds + // the same `Arc`, and its `ConfirmRemove` rolls those counters back + // through it. `rebuild_parent_totals` runs right after this and counts + // only survivors, so an entry dropped full leaves that later rollback + // to come out of a live sibling's totals. + let dropped: Vec> = { + let mut entries = self + .partitions + .lock() + .expect("stats registry mutex poisoned"); + entries + .extract_if(|key, entry| { + live_partitions + .get(key) + .is_none_or(|created_revision| *created_revision != entry.created_revision) + }) + .map(|(_, entry)| entry.stats) + .collect() + }; + // Guard released first: the rollback cascades into parent totals, which + // the partition map has no part in. + for stats in dropped { + stats.zero_out_all(); + } + } + + /// Recompute every topic and stream total as the sum of the partition + /// entries this node holds. + /// + /// A snapshot carries the DONOR's topic and stream totals, but partition + /// counters are node-local and never snapshotted: they keep moving on the + /// receiver while the snapshot is captured, shipped and installed. Storing + /// the donor's totals over the receiver's own children leaves + /// `topic < sum(partitions)` as the ordinary post-transfer shape, and the + /// next delete then subtracts a child from a parent that never counted it. + /// Every rollback in this file assumes `parent == sum(children)`; this is + /// where that gets re-established. + /// + /// Partitions this node has not materialized yet contribute nothing, the + /// same convergence boot relies on: the reconciler registers each one and + /// folds its on-disk delta in as it goes. + /// + /// The counters are sampled under the guard and summed without it: the walk + /// visits every partition in the tree, and the same mutex is on the + /// get-or-create path the reconciler and every shard's boot recovery take. + /// Holding it across the walk would serialize them behind it. + /// + /// The guard fences the MAP, never the counters, which the data plane + /// reaches through the `Arc` regardless. An append landing between the + /// sample and the parent store is folded into the parent by `fetch_add` and + /// then overwritten, so the invariant is restored modulo whatever arrives + /// in between. That residue is bounded by the walk and by one append, and + /// the saturating rollback absorbs it; quiescing the data plane for a + /// metadata install would cost far more than it buys. + /// + /// # Panics + /// If the registry mutex is poisoned. + fn rebuild_parent_totals(&self, streams: &IdSlab) { + let entries: AHashMap<(usize, usize, usize), (u64, u64, u32)> = { + let partitions = self + .partitions + .lock() + .expect("stats registry mutex poisoned"); + partitions + .iter() + .map(|(key, entry)| { + ( + *key, + ( + entry.stats.size_bytes_inconsistent(), + entry.stats.messages_count_inconsistent(), + entry.stats.segments_count_inconsistent(), + ), + ) + }) + .collect() + }; + for (stream_key, stream) in streams { + let mut stream_size_bytes = 0u64; + let mut stream_messages_count = 0u64; + let mut stream_segments_count = 0u32; + for (topic_key, topic) in &stream.topics { + let mut topic_size_bytes = 0u64; + let mut topic_messages_count = 0u64; + let mut topic_segments_count = 0u32; + for partition in &topic.partitions { + let Some((size_bytes, messages_count, segments_count)) = + entries.get(&(stream_key, topic_key, partition.id)) + else { + continue; + }; + topic_size_bytes = topic_size_bytes.saturating_add(*size_bytes); + topic_messages_count = topic_messages_count.saturating_add(*messages_count); + topic_segments_count = topic_segments_count.saturating_add(*segments_count); + } + topic.stats.store_from_snapshot( + topic_size_bytes, + topic_messages_count, + topic_segments_count, + ); + stream_size_bytes = stream_size_bytes.saturating_add(topic_size_bytes); + stream_messages_count = stream_messages_count.saturating_add(topic_messages_count); + stream_segments_count = stream_segments_count.saturating_add(topic_segments_count); + } + stream.stats.store_from_snapshot( + stream_size_bytes, + stream_messages_count, + stream_segments_count, + ); + } } } @@ -1729,10 +1963,11 @@ impl StateHandler for DeleteStreamRequest { return ApplyReply::err(DeleteStreamResult::StreamNotFound); }; let name = stream.name.clone(); + // Evict registry entries so a reused slab id starts with fresh stats. + // Before the removal: the committed tree is what names the entries. + state.stats_registry.remove_stream(stream_id, stream); state.items.remove(stream_id); state.index.remove(&name); - // Evict registry entries so a reused slab id starts with fresh stats. - state.stats_registry.remove_stream(stream_id); state.revision = state.revision.wrapping_add(1); // The dropped stream may have held groups with pending revocations. state.recompute_pending_revocations_count(); @@ -2046,10 +2281,20 @@ impl StateHandler for DeleteTopicRequest { return ApplyReply::err(DeleteTopicResult::TopicNotFound); }; let name = topic.name.clone(); + // Read the ids off the committed topic before it goes: they are what + // names its registry entries. + let partition_ids: Vec = topic + .partitions + .iter() + .map(|partition| partition.id) + .collect(); stream.topics.remove(topic_id); stream.topic_index.remove(&name); - // Evict registry entry so a reused slab id starts with fresh stats. - state.stats_registry.remove_topic(stream_id, topic_id); + // Roll the topic and its partitions out of the stream total and evict + // their entries, so a reused slab id starts from zero. + state + .stats_registry + .remove_topic(stream_id, topic_id, &partition_ids); state.revision = state.revision.wrapping_add(1); // The dropped topic may have held groups with pending revocations. state.recompute_pending_revocations_count(); @@ -2223,14 +2468,21 @@ impl StateHandler for DeletePartitionsRequest { // applies as the historical ok no-op. if count_to_delete > 0 { let retained = topic.partitions.len() - count_to_delete; + // Read the ids off the tail before it goes: `retained` is a count, + // and only a primary that minted dense ids makes it double as an id + // threshold. + let removed_ids: Vec = topic.partitions[retained..] + .iter() + .map(|partition| partition.id) + .collect(); topic.partitions.truncate(retained); // Members assigned the removed partitions must give them up. topic.rebalance_consumer_groups(); - // Evict registry entries so re-created partition ids start with - // fresh stats. + // Roll the removed partitions out of the topic and stream totals + // and evict their entries, so re-minted ids start from zero. state .stats_registry - .remove_partitions_from(stream_id, topic_id, retained); + .remove_partitions(stream_id, topic_id, &removed_ids); state.revision = state.revision.wrapping_add(1); } ApplyReply::ok(Bytes::new()) @@ -2338,7 +2590,12 @@ impl Snapshotable for Streams { // Boot: no live registry exists yet, so mint one. Safe because // `new_from_empty` clones this single inner onto the other left-right // buffer rather than building a second one. - Ok(StreamsInner::inner_from_snapshot(snapshot, Arc::new(StatsRegistry::default())).into()) + // + // The checkpoint's topic and stream aggregates are not adopted (see + // `inner_from_snapshot`), so this starts every total at zero and boot + // folds the real ones back in from disk. + let inner = StreamsInner::inner_from_snapshot(snapshot, Arc::new(StatsRegistry::default())); + Ok(inner.into()) } } @@ -2360,12 +2617,27 @@ impl StreamsInner { // that slot next. registry.retain_from_snapshot(&snapshot); *self = Self::inner_from_snapshot(snapshot, registry); + // The donor's aggregates describe the donor's partitions; this node kept + // its own the whole time, so the totals come from the surviving entries. + self.stats_registry.rebuild_parent_totals(&self.items); } /// Build a complete `StreamsInner` from a snapshot section against /// `stats_registry`. Shared by wrapper construction /// ([`Snapshotable::from_snapshot`]) and the in-place restore command /// (state transfer), which absorbs it on both left-right buffers. + /// + /// [`StatsSnapshot`] is decoded but never stored. Both callers derive the + /// topic and stream totals from this node's own partition entries instead: + /// state transfer through [`StatsRegistry::rebuild_parent_totals`], boot by + /// folding each shard's `load_partition` delta in as it materializes. + /// Adopting them would be actively wrong at both ends. A checkpoint reads a + /// stream's total and its topics' as separate loads while the partition + /// plane keeps counting, so they can disagree in either direction, and a + /// replayed `DeleteTopic` over that torn shape rolls a topic back against a + /// stream that never held it, clamps, and raises the rollup-underflow alarm + /// on every boot with no real divergence behind it. A snapshot's totals are + /// the DONOR's, over children the receiver kept the whole time. pub(crate) fn inner_from_snapshot( snapshot: StreamsSnapshot, stats_registry: Arc, @@ -2375,11 +2647,6 @@ impl StreamsInner { for (slab_key, stream_snap) in snapshot.items { let stream_stats = stats_registry.stream(slab_key); - stream_stats.store_from_snapshot( - stream_snap.stats.size_bytes, - stream_snap.stats.messages_count, - stream_snap.stats.segments_count, - ); let mut topic_index: AHashMap, usize> = AHashMap::new(); let mut topic_entries: Vec<(usize, Topic)> = Vec::new(); @@ -2387,11 +2654,6 @@ impl StreamsInner { for (topic_slab_key, topic_snap) in stream_snap.topics { let topic_stats = stats_registry.topic(slab_key, topic_slab_key, stream_stats.clone()); - topic_stats.store_from_snapshot( - topic_snap.stats.size_bytes, - topic_snap.stats.messages_count, - topic_snap.stats.segments_count, - ); let topic_name: Arc = Arc::from(topic_snap.name.as_str()); let topic = Topic { id: topic_snap.id, @@ -3187,6 +3449,422 @@ mod tests { assert_eq!(stream_stats.size_bytes_inconsistent(), 0); } + /// Deleting a partition has to roll its bytes out of the topic and stream + /// totals, not merely drop its registry entry: the aggregates are counters + /// the partition increments through its parent `Arc`, so an evicted entry + /// leaves what it contributed behind and `get_topic` / `get_stream` / + /// `/metrics` keep reporting deleted data until a restart rebuilds them. + #[test] + fn given_counted_partition_when_apply_delete_partitions_should_roll_it_out_of_the_parents() { + let mut inner = inner_with_registered_partition(); + let stats = inner.stats_registry.partition_get(0, 0, 0).expect("stats"); + stats.increment_segments_count(1); + stats.increment_messages_count(7); + stats.increment_size_bytes(512); + assert_eq!( + inner.items[0].stats.size_bytes_inconsistent(), + 512, + "partition counters must roll up before the delete, or the test proves nothing" + ); + + let delete = DeletePartitionsRequest { + stream_id: WireIdentifier::numeric(0), + topic_id: WireIdentifier::numeric(0), + partitions_count: 1, + }; + let apply = StateHandler::apply(&delete, &mut inner, IggyTimestamp::now()); + assert_eq!(apply.code, 0); + + let topic_stats = &inner.items[0].topics[0].stats; + assert_eq!(topic_stats.size_bytes_inconsistent(), 0); + assert_eq!(topic_stats.messages_count_inconsistent(), 0); + assert_eq!(topic_stats.segments_count_inconsistent(), 0); + let stream_stats = &inner.items[0].stats; + assert_eq!(stream_stats.size_bytes_inconsistent(), 0); + assert_eq!(stream_stats.messages_count_inconsistent(), 0); + assert_eq!(stream_stats.segments_count_inconsistent(), 0); + assert!( + inner.stats_registry.partition_get(0, 0, 0).is_none(), + "the entry has to go with the rollback: partition ids are re-minted \ + from max + 1, so the next create lands on this key" + ); + } + + /// The apply runs on BOTH left-right buffers off one shared registry, so the + /// rollback must land exactly once. A second pass that decremented again + /// would wrap the unsigned totals rather than settle at zero. + #[test] + fn given_replayed_delete_partitions_when_applied_twice_should_not_double_roll_back() { + let mut first = inner_with_registered_partition(); + let stats = first.stats_registry.partition_get(0, 0, 0).expect("stats"); + stats.increment_messages_count(7); + stats.increment_size_bytes(512); + let mut second = first.clone(); + + let delete = DeletePartitionsRequest { + stream_id: WireIdentifier::numeric(0), + topic_id: WireIdentifier::numeric(0), + partitions_count: 1, + }; + let _ = StateHandler::apply(&delete, &mut first, IggyTimestamp::now()); + let _ = StateHandler::apply(&delete, &mut second, IggyTimestamp::now()); + + assert!( + first.stats_registry.partition_get(0, 0, 0).is_none(), + "the first apply evicts; the second has to find nothing to roll back" + ); + let topic_stats = &second.items[0].topics[0].stats; + assert_eq!(topic_stats.size_bytes_inconsistent(), 0); + assert_eq!(topic_stats.messages_count_inconsistent(), 0); + let stream_stats = &second.items[0].stats; + assert_eq!( + stream_stats.size_bytes_inconsistent(), + 0, + "the second buffer's apply must not subtract the bytes again" + ); + assert_eq!(stream_stats.messages_count_inconsistent(), 0); + } + + /// `delete_topic` has the same duty one level up: dropping the topic entry + /// and its partitions leaves the stream total carrying the deleted topic. + #[test] + fn given_counted_topic_when_apply_delete_topic_should_roll_it_out_of_the_stream() { + let mut inner = inner_with_registered_partition(); + let stats = inner.stats_registry.partition_get(0, 0, 0).expect("stats"); + stats.increment_segments_count(1); + stats.increment_messages_count(7); + stats.increment_size_bytes(512); + assert_eq!( + inner.items[0].stats.size_bytes_inconsistent(), + 512, + "partition counters must roll up before the delete, or the test proves nothing" + ); + + let delete = DeleteTopicRequest { + stream_id: WireIdentifier::numeric(0), + topic_id: WireIdentifier::numeric(0), + }; + let apply = StateHandler::apply(&delete, &mut inner, IggyTimestamp::now()); + assert_eq!(apply.code, 0); + + let stream_stats = &inner.items[0].stats; + assert_eq!(stream_stats.size_bytes_inconsistent(), 0); + assert_eq!(stream_stats.messages_count_inconsistent(), 0); + assert_eq!(stream_stats.segments_count_inconsistent(), 0); + assert!( + inner.stats_registry.partition_get(0, 0, 0).is_none(), + "a recycled topic slab key would otherwise inherit these counters" + ); + } + + /// A partition keeps its `Arc` until the reconciler tears it down, which + /// happens after the commit that acked the delete, so writes in that window + /// land in a topic whose registry entry for them is already gone. Deleting + /// the topic has to settle that residue on the stream: the partition sweep + /// finds nothing to roll back, and only the topic's own swap can reach it. + #[test] + fn given_topic_residue_no_partition_entry_covers_when_apply_delete_topic_should_settle_it() { + let mut inner = inner_with_registered_partition(); + let stats = inner.stats_registry.partition_get(0, 0, 0).expect("stats"); + + let delete_partitions = DeletePartitionsRequest { + stream_id: WireIdentifier::numeric(0), + topic_id: WireIdentifier::numeric(0), + partitions_count: 1, + }; + let _ = StateHandler::apply(&delete_partitions, &mut inner, IggyTimestamp::now()); + assert!(inner.stats_registry.partition_get(0, 0, 0).is_none()); + + // The still-mounted partition, writing through the handle it cached + // before the delete. + stats.increment_size_bytes(100); + stats.increment_messages_count(2); + assert_eq!(inner.items[0].stats.size_bytes_inconsistent(), 100); + + let delete_topic = DeleteTopicRequest { + stream_id: WireIdentifier::numeric(0), + topic_id: WireIdentifier::numeric(0), + }; + let apply = StateHandler::apply(&delete_topic, &mut inner, IggyTimestamp::now()); + assert_eq!(apply.code, 0); + + let stream_stats = &inner.items[0].stats; + assert_eq!( + stream_stats.size_bytes_inconsistent(), + 0, + "bytes no surviving partition entry accounts for still owe the stream a rollback" + ); + assert_eq!(stream_stats.messages_count_inconsistent(), 0); + } + + /// A restore replaces the whole tree while the receiver's partition + /// counters keep running, and it adopts no aggregate of its own: without + /// the rebuild every topic and stream would come back at zero over live + /// children. Rolling a partition out of a parent that never counted it + /// subtracts past zero, and the totals are unsigned: `get_stream` / + /// `get_topic` / `/stats` would serve ~1.8e19 until the process restarted. + #[test] + fn given_a_restore_over_live_partition_counters_when_deleting_should_not_wrap() { + let mut inner = inner_with_registered_partition(); + let stats = inner.stats_registry.partition_get(0, 0, 0).expect("stats"); + stats.increment_segments_count(3); + stats.increment_messages_count(7); + stats.increment_size_bytes(512); + + // The donor captured this tree before any of that landed. + let snapshot = Streams::from(inner.clone()).to_snapshot(); + inner.restore_in_place(snapshot); + + // The rebuild has to put the receiver's own children back into the + // parents, or the delete below has nothing to subtract from. + assert_eq!( + inner.items[0].topics[0].stats.size_bytes_inconsistent(), + 512 + ); + assert_eq!(inner.items[0].stats.size_bytes_inconsistent(), 512); + + let delete = DeletePartitionsRequest { + stream_id: WireIdentifier::numeric(0), + topic_id: WireIdentifier::numeric(0), + partitions_count: 1, + }; + let apply = StateHandler::apply(&delete, &mut inner, IggyTimestamp::now()); + assert_eq!(apply.code, 0); + + let topic_stats = &inner.items[0].topics[0].stats; + assert_eq!(topic_stats.size_bytes_inconsistent(), 0); + assert_eq!(topic_stats.messages_count_inconsistent(), 0); + assert_eq!(topic_stats.segments_count_inconsistent(), 0); + let stream_stats = &inner.items[0].stats; + assert_eq!(stream_stats.size_bytes_inconsistent(), 0); + assert_eq!(stream_stats.messages_count_inconsistent(), 0); + // Its own `clamped_sub!` expansion and its own `UnderflowSite`, so the + // u32 wiring needs an assertion of its own. + assert_eq!(stream_stats.segments_count_inconsistent(), 0); + } + + /// `DeletePartitions` truncates the tail while the registry selects by id, + /// and only a primary that minted dense ids makes those the same set. With + /// ids `[3, 4]` a count-based predicate selects both, which strips the + /// SURVIVOR's bytes from its parents and resets the offset its clients + /// store against -- `IggyPartition` reads an empty offset space and denies + /// every `store_consumer_offset` after that. + #[test] + fn given_non_zero_base_partition_ids_when_apply_delete_partitions_should_keep_the_survivor() { + let mut inner = StreamsInner::new(); + create_stream(&mut inner, "alpha"); + let create_topic = CreateTopicWithAssignmentsRequest { + created_view: 0, + request: make_topic_request(0, 2, "logs"), + derived_options: WireOptions::empty(), + partitions: vec![ + CreatedPartitionAssignment { + partition_id: 3, + consensus_group_id: 1, + }, + CreatedPartitionAssignment { + partition_id: 4, + consensus_group_id: 2, + }, + ], + }; + let _ = StateHandler::apply(&create_topic, &mut inner, IggyTimestamp::now()); + let topic_stats = inner.items[0].topics[0].stats.clone(); + let survivor = inner.stats_registry.partition( + 0, + 0, + &committed_partition(&inner, 0, 0, 3), + topic_stats.clone(), + ); + let removed = inner.stats_registry.partition( + 0, + 0, + &committed_partition(&inner, 0, 0, 4), + topic_stats, + ); + survivor.increment_size_bytes(512); + survivor.set_current_offset(42); + removed.increment_size_bytes(64); + + let delete = DeletePartitionsRequest { + stream_id: WireIdentifier::numeric(0), + topic_id: WireIdentifier::numeric(0), + partitions_count: 1, + }; + let apply = StateHandler::apply(&delete, &mut inner, IggyTimestamp::now()); + assert_eq!(apply.code, 0); + + assert!( + inner.stats_registry.partition_get(0, 0, 4).is_none(), + "the truncated partition's entry must go" + ); + let survivor = inner + .stats_registry + .partition_get(0, 0, 3) + .expect("the retained partition keeps its entry"); + assert_eq!(survivor.size_bytes_inconsistent(), 512); + assert_eq!( + survivor.current_offset(), + 42, + "zeroing a live partition's offset denies every store_consumer_offset above it" + ); + assert_eq!( + inner.items[0].topics[0].stats.size_bytes_inconsistent(), + 512 + ); + assert_eq!(inner.items[0].stats.size_bytes_inconsistent(), 512); + } + + /// The entry has to be evicted, not just zeroed. Ids come back: the delete + /// truncates the tail and the next create mints `max + 1`, landing on the + /// same key. A surviving entry would hand the new partition its + /// predecessor's `purged_generation`, and the gate in + /// `reset_purged_partitions` would then skip the reset a purge just acked. + #[test] + fn given_recreated_partition_ids_when_purging_again_should_reset_the_new_counters() { + let mut inner = inner_with_registered_partition(); + let purge = PurgeTopicRequest { + stream_id: WireIdentifier::numeric(0), + topic_id: WireIdentifier::numeric(0), + }; + let _ = StateHandler::apply(&purge, &mut inner, IggyTimestamp::now()); + + let delete = DeletePartitionsRequest { + stream_id: WireIdentifier::numeric(0), + topic_id: WireIdentifier::numeric(0), + partitions_count: 1, + }; + let _ = StateHandler::apply(&delete, &mut inner, IggyTimestamp::now()); + + let create = CreatePartitionsWithAssignmentsRequest { + created_view: 0, + request: CreatePartitionsRequest { + stream_id: WireIdentifier::numeric(0), + topic_id: WireIdentifier::numeric(0), + partitions_count: 1, + }, + partitions: vec![CreatedPartitionAssignment { + partition_id: 0, + consensus_group_id: 2, + }], + }; + let apply = StateHandler::apply(&create, &mut inner, IggyTimestamp::now()); + assert_eq!(apply.code, 0); + assert_eq!( + inner.items[0].topics[0].partitions[0].id, 0, + "the freed id is re-minted, which is what makes the eviction load-bearing" + ); + + let topic_stats = inner.items[0].topics[0].stats.clone(); + let stats = inner.stats_registry.partition( + 0, + 0, + &committed_partition(&inner, 0, 0, 0), + topic_stats, + ); + stats.increment_size_bytes(512); + + let _ = StateHandler::apply(&purge, &mut inner, IggyTimestamp::now()); + assert_eq!( + stats.size_bytes_inconsistent(), + 0, + "a stale purged_generation would make this purge's gate skip the reset" + ); + assert_eq!(inner.items[0].topics[0].stats.size_bytes_inconsistent(), 0); + assert_eq!(inner.items[0].stats.size_bytes_inconsistent(), 0); + } + + /// The sweep has to reach every partition of the topic, not the first one. + #[test] + fn given_many_counted_partitions_when_apply_delete_topic_should_roll_all_of_them_out() { + let mut inner = StreamsInner::new(); + create_stream(&mut inner, "alpha"); + let create_topic = CreateTopicWithAssignmentsRequest { + created_view: 0, + request: make_topic_request(0, 4, "logs"), + derived_options: WireOptions::empty(), + partitions: (0..4) + .map(|partition_id| CreatedPartitionAssignment { + partition_id, + consensus_group_id: u64::from(partition_id) + 1, + }) + .collect(), + }; + let _ = StateHandler::apply(&create_topic, &mut inner, IggyTimestamp::now()); + let topic_stats = inner.items[0].topics[0].stats.clone(); + for partition_id in 0..4 { + let stats = inner.stats_registry.partition( + 0, + 0, + &committed_partition(&inner, 0, 0, partition_id), + topic_stats.clone(), + ); + stats.increment_size_bytes(512); + stats.increment_messages_count(7); + stats.increment_segments_count(1); + } + assert_eq!(inner.items[0].stats.size_bytes_inconsistent(), 512 * 4); + + let delete = DeleteTopicRequest { + stream_id: WireIdentifier::numeric(0), + topic_id: WireIdentifier::numeric(0), + }; + let apply = StateHandler::apply(&delete, &mut inner, IggyTimestamp::now()); + assert_eq!(apply.code, 0); + + let stream_stats = &inner.items[0].stats; + assert_eq!(stream_stats.size_bytes_inconsistent(), 0); + assert_eq!(stream_stats.messages_count_inconsistent(), 0); + assert_eq!(stream_stats.segments_count_inconsistent(), 0); + for partition_id in 0..4 { + assert!( + inner + .stats_registry + .partition_get(0, 0, partition_id) + .is_none(), + "partition {partition_id} kept its entry" + ); + } + } + + /// `remove_stream` is evict-only on purpose: `StreamStats` is the rollup + /// root, so nothing above it is owed the counters back. What it still has + /// to do is take the whole subtree's entries with it, or a recycled stream + /// slab key hands its topics' partitions the predecessor's counters. + #[test] + fn given_a_counted_stream_when_apply_delete_stream_should_evict_its_whole_subtree() { + let mut inner = inner_with_registered_partition(); + let stats = inner.stats_registry.partition_get(0, 0, 0).expect("stats"); + stats.increment_size_bytes(512); + stats.increment_messages_count(7); + + let delete = DeleteStreamRequest { + stream_id: WireIdentifier::numeric(0), + }; + let apply = StateHandler::apply(&delete, &mut inner, IggyTimestamp::now()); + assert_eq!(apply.code, 0); + + assert!( + inner.stats_registry.partition_get(0, 0, 0).is_none(), + "the partition entry must not outlive its stream" + ); + // Re-creating over the freed slab key must start from zero rather than + // inherit what the probe above counted. + create_stream(&mut inner, "beta"); + let create_topic = CreateTopicWithAssignmentsRequest { + created_view: 0, + request: make_topic_request(0, 1, "logs"), + derived_options: WireOptions::empty(), + partitions: vec![CreatedPartitionAssignment { + partition_id: 0, + consensus_group_id: 1, + }], + }; + let _ = StateHandler::apply(&create_topic, &mut inner, IggyTimestamp::now()); + assert_eq!(inner.items[0].stats.size_bytes_inconsistent(), 0); + assert_eq!(inner.items[0].topics[0].stats.size_bytes_inconsistent(), 0); + } + /// A stream purge walks every topic, so every topic's partitions must reset, /// not just the first one. #[test] @@ -3203,7 +3881,12 @@ mod tests { }; let _ = StateHandler::apply(&create_topic, &mut inner, IggyTimestamp::now()); let second_topic_stats = inner.items[0].topics[1].stats.clone(); - inner.stats_registry.partition(0, 1, 0, second_topic_stats); + inner.stats_registry.partition( + 0, + 1, + &committed_partition(&inner, 0, 1, 0), + second_topic_stats, + ); let counters: Vec> = (0..2) .map(|topic_id| { @@ -3312,7 +3995,12 @@ mod tests { // The data plane materializes the partition afterwards and counts what // it plants; the purge must not have invented a segment for it. let topic_stats = inner.items[0].topics[0].stats.clone(); - let stats = inner.stats_registry.partition(0, 0, 0, topic_stats); + let stats = inner.stats_registry.partition( + 0, + 0, + &committed_partition(&inner, 0, 0, 0), + topic_stats, + ); assert_eq!(stats.segments_count_inconsistent(), 0); stats.increment_segments_count(1); stats.increment_messages_count(5); @@ -3417,8 +4105,69 @@ mod tests { }; let _ = StateHandler::apply(&create_topic, &mut inner, IggyTimestamp::now()); let topic_stats = inner.items[0].topics[0].stats.clone(); - inner.stats_registry.partition(0, 0, 0, topic_stats); inner + .stats_registry + .partition(0, 0, &committed_partition(&inner, 0, 0, 0), topic_stats); + inner + } + + /// The committed record for one partition, which is what the registry keys + /// its entry's identity and purge gate off. + fn committed_partition( + inner: &StreamsInner, + stream_id: usize, + topic_id: usize, + partition_id: usize, + ) -> Partition { + inner.items[stream_id].topics[topic_id] + .partitions + .iter() + .find(|partition| partition.id == partition_id) + .expect("committed partition") + .clone() + } + + /// A checkpoint reads a stream's total and each of its topics' as separate + /// loads while the partition plane keeps counting, so the two can disagree + /// in either direction. The boot restore adopts neither level, and journal + /// replay runs over what it produces: a replayed `DeleteTopic` over an + /// adopted torn shape would roll a topic back against a stream that never + /// held it, clamp, and raise the underflow alarm on every boot after. + #[test] + fn given_a_torn_checkpoint_when_restoring_at_boot_should_adopt_neither_level() { + let mut snapshot = Streams::from(inner_with_registered_partition()).to_snapshot(); + // Topics summing above their stream is the torn shape: the stream was + // read first, and the topic kept counting before its own read. + snapshot.items[0].1.stats = StatsSnapshot { + size_bytes: 100, + messages_count: 1, + segments_count: 1, + }; + snapshot.items[0].1.topics[0].1.stats = StatsSnapshot { + size_bytes: 200, + messages_count: 2, + segments_count: 2, + }; + + let restored = Streams::from_snapshot(snapshot).expect("snapshot restore"); + + let (stream_size, stream_messages, stream_segments, topic_size) = restored.read(|inner| { + let stream = inner.items.get(0).expect("restored stream"); + let topic = stream.topics.get(0).expect("restored topic"); + ( + stream.stats.size_bytes_inconsistent(), + stream.stats.messages_count_inconsistent(), + stream.stats.segments_count_inconsistent(), + topic.stats.size_bytes_inconsistent(), + ) + }); + assert_eq!(topic_size, 0); + assert_eq!( + stream_size, 0, + "the stream total is snapshotted independently of its topics, so it has to go too" + ); + assert_eq!(stream_messages, 0); + assert_eq!(stream_segments, 0); } // The restore command is absorbed on BOTH left-right buffers. Minting a @@ -3487,6 +4236,44 @@ mod tests { ); } + /// The eviction has to empty what it drops. A partition the snapshot does + /// not carry stays mounted until the reconciler reaches it, holding the + /// same `Arc`, and its `ConfirmRemove` rolls those counters back through it. + /// The rebuild counts survivors only, so an entry dropped full leaves that + /// rollback to come out of a live sibling's totals. + #[test] + fn given_a_pruned_entry_when_its_partition_is_torn_down_later_should_leave_survivors_alone() { + let mut inner = inner_with_registered_partition(); + let doomed = inner.stats_registry.partition_get(0, 0, 0).expect("stats"); + doomed.increment_size_bytes(512); + + // A donor tree with the same shape but a partition this node's entry + // cannot be, so the retain drops it and the restore registers nothing + // in its place. + let mut snapshot = Streams::from(inner.clone()).to_snapshot(); + snapshot.items[0].1.topics[0].1.partitions[0].created_revision += 1; + inner.restore_in_place(snapshot); + assert!(inner.stats_registry.partition_get(0, 0, 0).is_none()); + + let survivor = inner.stats_registry.partition( + 0, + 0, + &committed_partition(&inner, 0, 0, 0), + inner.items[0].topics[0].stats.clone(), + ); + survivor.increment_size_bytes(900); + + // The mounted stale incarnation, reaching its drop point. + doomed.zero_out_all(); + + assert_eq!( + inner.items[0].topics[0].stats.size_bytes_inconsistent(), + 900, + "the survivor's bytes must not pay for the pruned entry's rollback" + ); + assert_eq!(inner.items[0].stats.size_bytes_inconsistent(), 900); + } + /// Admission is all that stands between a client and a namespace collision. /// `IggyNamespace::new` used to mask, so slab key `MAX_TOPICS` packed /// byte-identically to key 0: same shard, consensus group, directory and diff --git a/core/partitions/src/iggy_partition.rs b/core/partitions/src/iggy_partition.rs index f25616ad31..5d363ef24c 100644 --- a/core/partitions/src/iggy_partition.rs +++ b/core/partitions/src/iggy_partition.rs @@ -5947,6 +5947,8 @@ where budget_spent, ..SegmentRemoval::default() }; + let mut shortfall = CleanupShortfall::default(); + let mut removed_offsets: Option<(u64, u64)> = None; for _ in 0..removable { // The removable run is always a prefix (oldest first), so the next // victim is the front once the previous one is gone. @@ -5979,16 +5981,13 @@ where } } - let segment_size = segment.size.as_bytes_u64(); - // The removal loop above only reaches sealed segments, which always - // hold at least one message, so the count is inclusive end..=start. - // A one-message sealed segment has `start_offset == end_offset`, so - // the `+ 1` is required (a `start == end -> 0` special case would - // undercount it). - let messages_in_segment = segment.end_offset - segment.start_offset + 1; - self.stats.decrement_size_bytes(segment_size); - self.stats.decrement_segments_count(1); - self.stats.decrement_messages_count(messages_in_segment); + let (messages_in_segment, segment_shortfall) = + settle_cleaned_segment(&self.stats, &segment); + shortfall.absorb(segment_shortfall); + removed_offsets = Some(match removed_offsets { + Some((from, _)) => (from, segment.end_offset), + None => (segment.start_offset, segment.end_offset), + }); removal.segments += 1; removal.messages += messages_in_segment; @@ -6003,6 +6002,8 @@ where ); } + shortfall.report(namespace, removed_offsets); + removal } @@ -7263,6 +7264,71 @@ pub struct SegmentRemoval { pub budget_spent: bool, } +/// What a cleanup rollback could not take out of the partition counters, +/// summed over one [`IggyPartition::remove_sealed_segments_up_to`] call. +/// +/// Accumulated rather than reported per segment: `UnderflowSite::report` in +/// `iggy_common` already counts every clamp and power-of-two throttles its own +/// line, so a per-segment `warn!` here is an unthrottled second copy of it. One +/// line per call, carrying the offset range the call removed, says which +/// partition and which segments without that. +#[derive(Debug, Default, Clone, Copy)] +struct CleanupShortfall { + size_bytes: u64, + segments: u32, + messages: u64, +} + +impl CleanupShortfall { + const fn absorb(&mut self, other: Self) { + self.size_bytes += other.size_bytes; + self.segments += other.segments; + self.messages += other.messages; + } + + /// One line for the whole call, naming the partition and the offset range + /// it removed. Silent when the counters covered every rollback. + fn report(self, namespace: IggyNamespace, removed_offsets: Option<(u64, u64)>) { + if self.size_bytes == 0 && self.segments == 0 && self.messages == 0 { + return; + } + let (removed_from, removed_to) = removed_offsets.unwrap_or_default(); + warn!( + target: "iggy.partitions.diag", + plane = "partitions", + namespace_raw = namespace.inner(), + removed_from, + removed_to, + size_shortfall = self.size_bytes, + segments_shortfall = self.segments, + messages_shortfall = self.messages, + "segment cleanup gave back more than the partition counters held; the parent \ + totals are now low by the shortfall until a rebuild or a restart" + ); + } +} + +/// Roll one cleaned-up segment out of the partition counters. +/// +/// Returns the messages the segment held, which is what the caller reports as +/// removed, plus whatever the rollback could not cover. Retention is the +/// likeliest source of a clamped rollback: a partition on its way out keeps +/// serving cleanup passes after the delete already settled its counters into +/// the parents. +fn settle_cleaned_segment(stats: &PartitionStats, segment: &Segment) -> (u64, CleanupShortfall) { + // The removal loop only reaches sealed segments, which always hold at least + // one message, so the count is inclusive start..=end. A one-message sealed + // segment has `start_offset == end_offset`, so the `+ 1` is required (a + // `start == end -> 0` special case would undercount it). + let messages = segment.end_offset - segment.start_offset + 1; + let shortfall = CleanupShortfall { + size_bytes: stats.decrement_size_bytes(segment.size.as_bytes_u64()), + segments: stats.decrement_segments_count(1), + messages: stats.decrement_messages_count(messages), + }; + (messages, shortfall) +} + /// Highest `end_offset` among the leading run of expired sealed segments, or /// `None` when none are expired. The last element is the active segment and is /// never considered. `expiry` must be resolved; a `ServerDefault` expires @@ -10958,7 +11024,11 @@ mod tests { } } - fn armed_session(to_op: u64, floor: u64, first_batch_offset: Option) -> RepairSession { + pub(super) fn armed_session( + to_op: u64, + floor: u64, + first_batch_offset: Option, + ) -> RepairSession { armed_fetch_session(to_op, to_op, floor, first_batch_offset) } @@ -11005,7 +11075,11 @@ mod tests { /// A repaired `SendMessages` prepare with an explicit chain identity, as a /// serving peer ships it. - fn repaired_send_prepare(op: u64, parent: u128, checksum: u128) -> Message { + pub(super) fn repaired_send_prepare( + op: u64, + parent: u128, + checksum: u128, + ) -> Message { let namespace = IggyNamespace::new(1, 1, 0); let record = build_segment_record(namespace, op); let header_size = std::mem::size_of::(); @@ -12407,7 +12481,10 @@ mod retention_tests { #[cfg(test)] mod purge_floor_tests { - use super::tests::{build_segment_record, journal_send_batch, repair_config, test_partition}; + use super::tests::{ + armed_session, build_segment_record, journal_send_batch, repair_config, + repaired_send_prepare, test_partition, + }; use super::*; use iggy_binary_protocol::{Command, WireConsumer, WireEncode}; @@ -12863,6 +12940,62 @@ mod purge_floor_tests { let _ = std::fs::remove_dir_all(&dir); } + /// Eviction is what makes a second copy reachable at all: + /// `apply_repaired_prepare` refuses an op the journal still holds, so a + /// re-delivered repair frame is only re-journaled once the flush that + /// persisted it has evicted it. `commit_min` is what lets the frame past + /// the other guard: the flush persists the whole committed prefix while + /// the walk advances `commit_min` by at most [`COMMIT_WALK_OPS_MAX`], so + /// every op above the walk's reach is persisted, evicted, and still + /// re-deliverable. The flush is the last gate, and a durable line frozen at + /// boot cannot see its own writes -- a rejoining backup then diverged from + /// the group by exactly the ops it repaired. + /// + /// Journaled through the repair path, the one that keeps a batch's original + /// offsets; `apply_replicated_operation` re-stamps from the local counter + /// and so cannot collide with itself. + #[compio::test] + async fn given_offsets_already_persisted_when_flushed_again_should_not_append_a_second_copy() { + const CHECKSUM: u128 = 0x5a; + // One op past the walk budget, so the last one stays above `commit_min` + // after the flush that persists it. + const OPS: u64 = COMMIT_WALK_OPS_MAX as u64 + 1; + let (mut partition, dir) = purge_test_partition("persisted-twice"); + let record_len = build_segment_record(IggyNamespace::new(1, 1, 0), 1).len() as u64; + partition.repair = Some(armed_session(OPS, 0, None)); + + for op in 1..=OPS { + partition + .apply_repaired_prepare(repaired_send_prepare(op, 0, CHECKSUM)) + .await; + } + partition.consensus().advance_commit_max(OPS); + partition.commit_journal(&repair_config()).await; + assert_eq!( + partition.log.active_segment().size.as_bytes_u64(), + record_len * OPS, + "the first flush persists the whole committed prefix" + ); + assert!( + partition.consensus().commit_min() < OPS, + "the premise: the walk leaves the last op above `commit_min`, which \ + is what lets the repair ingest re-deliver it" + ); + + partition.repair = Some(armed_session(OPS, 0, None)); + partition + .apply_repaired_prepare(repaired_send_prepare(OPS, 0, CHECKSUM)) + .await; + partition.commit_journal(&repair_config()).await; + assert_eq!( + partition.log.active_segment().size.as_bytes_u64(), + record_len * OPS, + "offsets the segment already holds must not be appended twice" + ); + + let _ = std::fs::remove_dir_all(&dir); + } + /// A delete whose on-disk cleanup failed leaves the partition directory /// (and `purge.gen`) behind. The recreated topic's rows restart their purge /// generations at 0, so hydrating the DEAD incarnation's generation would diff --git a/core/server/src/boot/mod.rs b/core/server/src/boot/mod.rs index fc2de43979..04d2db3ede 100644 --- a/core/server/src/boot/mod.rs +++ b/core/server/src/boot/mod.rs @@ -541,22 +541,6 @@ async fn shard_main( .await .map_err(ServerError::MetadataRecovery)?; ensure_default_root_user(&recovered.mux_stm); - // The factory bundle hands every peer a read handle over the - // same `Inner`, so `Arc` (and the parent - // `Arc`) is shared across all shards. Zero the - // snapshot totals here, once, before any peer can observe the - // bundle. Per-shard `load_partition` deltas in - // `build_shard_for_thread` then race only against other - // atomic adds, never against a concurrent `swap(0)` that - // would mistake an in-flight delta for the snapshot total - // and decrement the parent `StreamStats` by it. - let () = recovered.mux_stm.streams().read(|inner| { - for (_, stream) in &inner.items { - for (_, topic) in &stream.topics { - topic.stats.zero_out_all(); - } - } - }); ( Rc::new(recovered.mux_stm), Some(bundle_tx), diff --git a/core/server/src/boot/recovery.rs b/core/server/src/boot/recovery.rs index 812e4ee932..25a7b67573 100644 --- a/core/server/src/boot/recovery.rs +++ b/core/server/src/boot/recovery.rs @@ -156,7 +156,7 @@ pub(in crate::boot) async fn build_shard_for_thread( let stats = inner.stats_registry.partition( stream.id, topic_id, - partition.id, + partition, topic.stats.clone(), ); owned.push(( diff --git a/core/server/src/http/metrics.rs b/core/server/src/http/metrics.rs index 6546c00c15..c409392ffe 100644 --- a/core/server/src/http/metrics.rs +++ b/core/server/src/http/metrics.rs @@ -21,13 +21,58 @@ //! other route handlers so this leaf never imports the state hub. use configs::http::HttpMetricsConfig; -use iggy_common::IggyError; +use iggy_common::{IggyError, stats_rollup_underflows}; +use prometheus_client::collector::Collector; use prometheus_client::encoding::text::encode; -use prometheus_client::metrics::counter::Counter; +use prometheus_client::encoding::{DescriptorEncoder, EncodeMetric}; +use prometheus_client::metrics::counter::{ConstCounter, Counter}; use prometheus_client::metrics::gauge::Gauge; use prometheus_client::registry::Registry; use tracing::error; +/// Exports the process-wide clamped-rollup count as a counter series, read at +/// encode time rather than mirrored into one. +/// +/// Non-zero means a partition, topic or stream total was asked to give back +/// more than it held, so what that scope now reports is low, and stays low +/// until a rebuild or a restart. All three levels clamp and all three feed this +/// one counter, which carries no scope label -- the `warn!` in `iggy_common` +/// names the scope and counter that moved it. +/// +/// Not "alert on any increase". A delete, a purge, a partition teardown and a +/// snapshot restore each open a window where a retention pass hands back bytes +/// the parents have already given up, and the clamp is the intended outcome +/// there -- `given_a_rolled_back_partition_when_a_late_decrement_arrives_should_leave_siblings_alone` +/// in `core/common/src/types/streaming_stats.rs` drives exactly that. What is +/// worth paging on is a bounded rate OUTSIDE those windows: that is the shape +/// that says the tree is diverging rather than settling. +/// +/// A counter, not a gauge: the source only ever climbs within a process, so +/// `rate()` and `increase()` are the queries an operator wants, and both are +/// counter-only. A restart resets the source and the exposition together, +/// which is the counter reset Prometheus expects. +/// +/// Read only by the `/metrics` scrape, so it needs `http.enabled` and +/// `http.metrics.enabled` -- as does every other series in this registry, +/// which is the only Prometheus surface the server has. A TCP-only or +/// QUIC-only deployment exports nothing, and the per-scope `warn!` in +/// `iggy_common` is the whole signal there. +#[derive(Debug)] +struct StatsRollupUnderflows; + +impl Collector for StatsRollupUnderflows { + fn encode(&self, mut encoder: DescriptorEncoder) -> Result<(), std::fmt::Error> { + let counter = ConstCounter::new(stats_rollup_underflows()); + let metric_encoder = encoder.encode_descriptor( + "stats_rollup_underflows", + "total count of aggregate stats decrements clamped at zero", + None, + counter.metric_type(), + )?; + counter.encode(metric_encoder) + } +} + /// The legacy server's metric set, registered under the same names and help /// texts so existing dashboards and alerts keep working unchanged. /// @@ -74,6 +119,9 @@ impl HttpMetrics { registry.register("messages", "total count of messages", messages.clone()); registry.register("users", "total count of users", users.clone()); registry.register("clients", "total count of clients", clients.clone()); + // Not a legacy-parity metric, and not a mirrored one: the source is a + // process-wide static the scrape reads directly. + registry.register_collector(Box::new(StatsRollupUnderflows)); // Every shard's drop / reconcile / partition counters, one // `shard`-labelled sub-registry per shard so series stay per-shard // without a `shard_id` label in the counter label sets (see @@ -203,6 +251,25 @@ mod tests { ); } + /// Outside `PARITY_METRIC_NAMES`: the legacy server had no such series, so + /// it gets its own assertion rather than a row in the parity list. + /// + /// The value is a process-wide static shared with every other test in this + /// binary, so this asserts the series and its type, not a number. + #[test] + fn rollup_underflow_collector_lands_in_the_exposition() { + let metrics = HttpMetrics::init(&[]); + let output = metrics.formatted_output(); + assert!( + output.contains("# TYPE stats_rollup_underflows counter\n"), + "expected the collector's series in the exposition:\n{output}" + ); + assert!( + output.contains("\nstats_rollup_underflows_total "), + "expected the collector's sample in the exposition:\n{output}" + ); + } + #[test] fn scraped_values_land_in_the_exposition() { let metrics = HttpMetrics::init(&[]); diff --git a/core/server/src/partition_reconciler.rs b/core/server/src/partition_reconciler.rs index 6ce661ee72..10be2b704d 100644 --- a/core/server/src/partition_reconciler.rs +++ b/core/server/src/partition_reconciler.rs @@ -179,7 +179,7 @@ use iggy_common::{ConsumerGroupId, IggyTimestamp}; use message_bus::AUTO_COMMIT_CLIENT_ID; use message_bus::MessageBus; use metadata::impls::metadata::StreamsFrontend; -use metadata::stm::stream::Partition; +use metadata::stm::stream::{Partition, StatsRegistry}; use server_common::Message; use server_common::sharding::{IggyNamespace, ShardId}; use shard::MetadataSubmit; @@ -228,6 +228,12 @@ pub struct ReconcilerCtx { pub cluster_id: u128, pub self_replica_id: u8, pub replica_count: u8, + /// The shared per-entity stats registry, cloned once at construction. It is + /// minted during metadata bootstrap and never re-minted, and a topic delete + /// settles one namespace per partition, so reading it out of the STM per + /// teardown would open a reader epoch per partition for a clone that cannot + /// change. + stats_registry: Arc, failure_state: RefCell>, /// `Streams::revision` observed at the end of the last pass that fully /// converged. Paired with `last_pass_noop` for the fast-skip in @@ -251,6 +257,12 @@ impl ReconcilerCtx { self_replica_id: u8, replica_count: u8, ) -> Self { + let stats_registry = shard + .plane + .metadata() + .mux_stm + .streams() + .read(|inner| Arc::clone(&inner.stats_registry)); Self { shard, total_shards, @@ -258,6 +270,7 @@ impl ReconcilerCtx { cluster_id, self_replica_id, replica_count, + stats_registry, failure_state: RefCell::new(AHashMap::new()), last_revision: Cell::new(None), last_pass_noop: Cell::new(false), @@ -567,12 +580,7 @@ async fn reconcile_additions( let partitions = ctx.shard.plane.partitions(); let total_shards = u32::from(ctx.total_shards); - for TargetPartition { - ns, - epoch, - created_view, - } in target - { + for TargetPartition { ns, epoch } in target { if partitions.contains(&ns) { // Tombstoned but still in the map. Two cases, told apart by // whether teardown's disk delete succeeded: @@ -647,6 +655,20 @@ async fn reconcile_additions( // bumps `Streams::revision`, which forces the next pass past the // fast-skip. if partitions.is_tombstoned(&ns) { + // Unless a teardown put it there. The mid-pass-recreate arm below + // tears down a prior life this shard never mounted, and a disk + // delete that failed leaves the tombstone standing with no + // `ConfirmRemove` behind it -- the same permanent fence the in-map + // branch above escapes, told apart by the same signal. + if ctx.has_pending_delete_failure(ns) { + trace!( + shard = shard_id, + ns_raw = ns.inner(), + "additions: ns tombstoned before materialisation with a failed disk delete; re-driving teardown" + ); + tear_down_owned_partition(ctx, ns, counters).await; + continue; + } trace!( shard = shard_id, ns_raw = ns.inner(), @@ -694,11 +716,12 @@ async fn reconcile_additions( continue; } - // Resolve the shared stats `Arc` only for namespaces actually - // built, not once per committed partition every pass. A topic that - // vanished between the target snapshot and this read defers to the - // next pass. - let Some((partition_stats, topic_runtime)) = fetch_partition_stats(ctx, ns) else { + // Resolved only for namespaces actually built, not once per committed + // partition every pass. A stream, topic or partition that vanished + // between the target snapshot and this read defers to the next pass. + let Some((partition_stats, topic_runtime, partition_metadata)) = + fetch_partition_build_inputs(ctx, ns) + else { continue; }; @@ -713,10 +736,36 @@ async fn reconcile_additions( ctx.config .system .get_partition_path(ns.stream_id(), ns.topic_id(), ns.partition_id()); - let built = if std::fs::metadata(&partition_dir).is_ok() { - let Some(partition_metadata) = fetch_partition_metadata(ctx, ns) else { - continue; - }; + let prior_life_on_disk = std::fs::metadata(&partition_dir).is_ok(); + + // The target was snapshotted before this read, so a delete plus a + // recreate of the same slab keys can commit in between. Everything + // below has to describe the incarnation the row will carry, so the + // committed record wins over the snapshot on both counts. + let created_revision = partition_metadata.created_revision; + let created_view = partition_metadata.created_view; + if created_revision != epoch { + trace!( + shard = shard_id, + ns_raw = ns.inner(), + target_epoch = epoch, + created_revision, + prior_life_on_disk, + "additions: recreate committed mid-pass; deferring the build to the next pass" + ); + counters.deferred += 1; + // Skipping alone only settles it when nothing is on disk. With a + // directory there, the next pass takes the loader arm below and + // hydrates the NEW incarnation out of the OLD segments, so the + // prior life has to go first. Teardown tombstones until its + // `ConfirmRemove` lands, which is what holds that pass off. + if prior_life_on_disk { + tear_down_owned_partition(ctx, ns, counters).await; + } + continue; + } + + let built = if prior_life_on_disk { load_partition_or_fence( ctx.config.as_ref(), ns, @@ -735,7 +784,7 @@ async fn reconcile_additions( ctx.config.as_ref(), ns, partition_stats, - epoch, + created_revision, topic_runtime, ctx.cluster_id, ctx.self_replica_id, @@ -751,7 +800,7 @@ async fn reconcile_additions( ctx.shard.enqueue_reconcile_op(ReconcileOp::InsertOwned { namespace: ns, partition: Box::new(partition), - epoch, + epoch: created_revision, }); ctx.record_success(ns, FailureCause::Add); counters.materialised += 1; @@ -988,6 +1037,17 @@ async fn tear_down_owned_partition( return; } + // Registry entry only. The mounted partition's OWN counters are settled by + // the `ConfirmRemove` arm, which is the point where the partition value is + // dropped: a handler suspended mid-append resumes and increments through + // its cached handle, and anything settled before that drop leaves the + // increment in the parent totals with nothing left to roll it back. + // + // Paired with the enqueue, not with the fence above it: a failed disk + // delete returns without a `ConfirmRemove` behind it, and an entry already + // evicted would leave the still-mounted partition serving the registry-miss + // shape while its files are still there. + settle_partition_stats(ctx, ns); ctx.shard .enqueue_reconcile_op(ReconcileOp::ConfirmRemove { namespace: ns }); ctx.record_success(ns, FailureCause::Delete); @@ -1202,12 +1262,9 @@ struct TargetPartition { ns: IggyNamespace, /// The committed `created_revision`. Lets the pass detect a stale local /// incarnation after slab-key reuse without an `Arc` clone per - /// partition; stats are fetched lazily in [`fetch_partition_stats`] only - /// for namespaces actually built. + /// partition: [`fetch_partition_build_inputs`] re-reads committed metadata + /// for the namespaces actually built, and takes the stats there. epoch: u64, - /// The view a fresh materialisation seeds its consensus group with; see - /// `build_partition_fresh`. - created_view: u32, } /// Every committed partition, as the additions pass needs it. @@ -1224,7 +1281,6 @@ fn snapshot_target_namespaces(ctx: &ReconcilerCtx) -> Vec { entries.push(TargetPartition { ns: IggyNamespace::new(stream.id, topic_id, partition.id), epoch: partition.created_revision, - created_view: partition.created_view, }); } } @@ -1244,47 +1300,79 @@ fn current_revision(ctx: &ReconcilerCtx) -> u64 { .read(|inner| inner.revision) } -/// Clone the parent topic's `Arc` for a single namespace. -/// `None` if the topic vanished between the target snapshot and this read. -fn fetch_partition_stats( +/// Roll a torn-down partition out of its parent topic and stream by dropping +/// its registry entry. +/// +/// Usually a no-op: the metadata apply evicts the entry on the commit that +/// acked the delete. What is left for this call is a stale incarnation being +/// torn down after a slab-key reuse, whose entry the apply never named. It must +/// not survive into the rebuild -- `StatsRegistry::partition` is a +/// get-or-create, so the rebuild would inherit the dead incarnation's counters. +/// (Its `purged_generation` no longer rides along either way: a fresh entry +/// seeds that gate from the committed partition.) +/// +/// The mounted partition's own handle is settled later, on `ConfirmRemove`. +/// +/// The window this opens, on the teardown-for-rebuild path only. Between this +/// eviction and the rebuild's get-or-create a pass later, `partition_get` +/// answers `None`, so every reply builder serves the registry-miss shape for +/// the namespace (no segments, offset 0) and its parents read low by whatever +/// the dead incarnation held. Deliberate: the alternative is carrying a +/// materialisation signal through the registry so readers could tell "not +/// mounted here" from "empty", and the numbers are wrong either way while the +/// rebuild is pending. The rebuild folds the on-disk delta back in and the +/// window closes on its own; a delete has no rebuild and so no window. +fn settle_partition_stats(ctx: &ReconcilerCtx, ns: IggyNamespace) { + ctx.stats_registry + .remove_partitions(ns.stream_id(), ns.topic_id(), &[ns.partition_id()]); +} + +/// Everything one namespace's build needs out of committed metadata: the shared +/// `Arc`, the topic's runtime options, and the committed +/// [`Partition`] record the disk loader reads `created_at` / `created_revision` +/// / `created_view` from. +/// +/// One read for all three. The `Partition` lookup doubles as the membership +/// check: the committed topic has to still list this partition, because a pass +/// captures its targets once and then awaits disk work per namespace, so +/// without it a target captured before a `DeletePartitions` re-creates the +/// entry that delete just evicted -- and the apply on the second left-right +/// buffer, deferred to a later publish, then zeroes a live partition's counters +/// and drops its entry again. +/// +/// `None` if the stream, the topic, or the partition vanished between the +/// target snapshot and this read. +fn fetch_partition_build_inputs( ctx: &ReconcilerCtx, ns: IggyNamespace, ) -> Option<( Arc, iggy_common::TopicRuntimeOptions, + Partition, )> { ctx.shard.plane.metadata().mux_stm.streams().read(|inner| { let stream = inner.items.get(ns.stream_id())?; let topic = stream.topics.get(ns.topic_id())?; + let partition = topic + .partitions + .iter() + .find(|partition| partition.id == ns.partition_id())? + .clone(); // Get-or-create in the shared registry so the owning shard's counters // are the same `Arc` every shard's `get_topic` reply reads. Some(( inner.stats_registry.partition( ns.stream_id(), ns.topic_id(), - ns.partition_id(), + &partition, topic.stats.clone(), ), iggy_common::TopicRuntimeOptions::from_resource_options(&topic.options), + partition, )) }) } -/// The committed [`Partition`] record for `ns`, which the disk loader needs -/// (`created_at`, `created_revision`, `created_view`). `None` if the topic or -/// partition vanished between the target snapshot and this read. -fn fetch_partition_metadata(ctx: &ReconcilerCtx, ns: IggyNamespace) -> Option { - ctx.shard.plane.metadata().mux_stm.streams().read(|inner| { - let stream = inner.items.get(ns.stream_id())?; - let topic = stream.topics.get(ns.topic_id())?; - topic - .partitions - .iter() - .find(|partition| partition.id == ns.partition_id()) - .cloned() - }) -} - /// `true` when this shard's routing row for `ns` already records `epoch`. A row /// carrying any other epoch (or none) is stale and must be rewritten, since the /// namespace is byte-identical across incarnations. @@ -1381,7 +1469,7 @@ pub fn install_tick_handler(shard: &Rc, wake_tx: WakeTx) { mod tests { use super::{ FailureCause, FailureRecord, PassCounters, ReconcilerCtx, build_partition_fresh, - current_revision, delete_partitions_from_disk, fetch_partition_stats, + current_revision, delete_partitions_from_disk, fetch_partition_build_inputs, reconcile_consumer_group_offsets, reconcile_once, }; use configs::server::{ServerConfig, ServerSystemConfig}; @@ -1926,6 +2014,155 @@ mod tests { ); } + /// A pass captures its targets once and then awaits disk work per + /// namespace, so a delete committed mid-pass leaves a stale target behind. + /// Resolving stats for it would get-or-CREATE the registry entry the delete + /// just evicted, and the apply on the second left-right buffer would then + /// zero and drop a live partition's counters. + #[compio::test] + async fn stats_are_refused_for_a_partition_the_committed_topic_no_longer_lists() { + let tmp = TempDir::new().expect("tempdir for system path"); + let config = test_config(&tmp); + let mux = TestMux::default(); + seed_stream(&mux, 1, "stream-a"); + seed_topic(&mux, 2, 0, "topic-a", vec![assignment(0, 1)]); + + let shard = build_test_shard(0, &config, mux); + let ctx = make_ctx(Rc::clone(&shard), 1, Rc::new(config)); + + let listed = IggyNamespace::new(0, 0, 0); + assert!( + fetch_partition_build_inputs(&ctx, listed).is_some(), + "a partition the topic lists must still resolve" + ); + + let unlisted = IggyNamespace::new(0, 0, 7); + assert!( + fetch_partition_build_inputs(&ctx, unlisted).is_none(), + "a partition the committed topic does not list must not resolve" + ); + assert!( + registry(&ctx).partition_get(0, 0, 7).is_none(), + "and the refusal must not have created its entry on the way" + ); + } + + /// The metadata apply rolls a deleted partition out of its parents at + /// commit, but the partition stays mounted until the reconciler tears it + /// down, and appends in that window land in parents with no entry left to + /// account for them. The `ConfirmRemove` arm is where that residue gets + /// settled: it drops the partition value, so it is the last moment a + /// suspended handler can still increment through its cached handle. + /// + /// Named for the drop point because that is what it holds: the apply had + /// already evicted the entry here, so [`settle_partition_stats`] finds + /// nothing and the rollback comes from the mounted partition's own handle + /// in `shard`'s pump. The eviction is guarded by the sibling test below. + #[compio::test] + async fn confirm_remove_rolls_back_what_the_metadata_delete_could_not_reach() { + let tmp = TempDir::new().expect("tempdir for system path"); + let config = test_config(&tmp); + let mux = TestMux::default(); + seed_stream(&mux, 1, "stream-a"); + seed_topic(&mux, 2, 0, "topic-a", vec![assignment(0, 1)]); + + let shard = build_test_shard(0, &config, mux); + let ctx = make_ctx(Rc::clone(&shard), 1, Rc::new(config)); + let ns = IggyNamespace::new(0, 0, 0); + + reconcile_once(&ctx).await; + ctx.shard.apply_reconcile_ops(); + let (stats, ..) = + fetch_partition_build_inputs(&ctx, ns).expect("materialised namespace has stats"); + stats.increment_size_bytes(512); + + seed_delete_topic(&ctx.shard.plane.metadata().mux_stm, 3, 0, 0); + // The apply evicted the entry and rolled back what it could see. This + // is the append that beat the teardown, through the handle the mounted + // partition still holds. + stats.increment_size_bytes(100); + assert_eq!(stream_size(&ctx), 100); + + reconcile_once(&ctx).await; + ctx.shard.apply_reconcile_ops(); + + assert_eq!( + stream_size(&ctx), + 0, + "the drop point must roll back what landed after the commit that acked the delete" + ); + } + + /// [`settle_partition_stats`], the half the test above cannot reach: there + /// the metadata apply had already dropped the entry, so the rollback came + /// from the mounted partition's own handle at the drop point. A stale + /// incarnation left by a slab-key reuse still holds an entry the apply + /// never named, and it has to go, or the rebuild's get-or-create inherits + /// the dead incarnation's counters. + #[compio::test] + async fn teardown_evicts_a_registry_entry_the_metadata_delete_never_named() { + let tmp = TempDir::new().expect("tempdir for system path"); + let config = test_config(&tmp); + let mux = TestMux::default(); + seed_stream(&mux, 1, "stream-a"); + seed_topic(&mux, 2, 0, "topic-a", vec![assignment(0, 1)]); + + let shard = build_test_shard(0, &config, mux); + let ctx = make_ctx(Rc::clone(&shard), 1, Rc::new(config)); + let ns = IggyNamespace::new(0, 0, 0); + + reconcile_once(&ctx).await; + ctx.shard.apply_reconcile_ops(); + let (stats, _, partition) = + fetch_partition_build_inputs(&ctx, ns).expect("materialised namespace has stats"); + let topic_stats = stats.parent(); + + seed_delete_topic(&ctx.shard.plane.metadata().mux_stm, 3, 0, 0); + assert!( + registry(&ctx).partition_get(0, 0, 0).is_none(), + "the apply evicts the entry it named, which is what leaves this test something else \ + to prove" + ); + // The entry a stale incarnation carries into the teardown, counting + // into parents that outlive it. Minted here rather than left behind by + // a slab-key reuse, which no in-crate fixture can stage. + registry(&ctx) + .partition(0, 0, &partition, topic_stats) + .increment_size_bytes(100); + assert_eq!(stream_size(&ctx), 100); + + reconcile_once(&ctx).await; + ctx.shard.apply_reconcile_ops(); + + assert!( + registry(&ctx).partition_get(0, 0, 0).is_none(), + "an entry surviving teardown hands the rebuild the dead incarnation's counters" + ); + assert_eq!( + stream_size(&ctx), + 0, + "and evicting without the rollback strands its bytes in the stream total" + ); + } + + fn registry(ctx: &ReconcilerCtx) -> Arc { + ctx.shard + .plane + .metadata() + .mux_stm + .streams() + .read(|inner| Arc::clone(&inner.stats_registry)) + } + + fn stream_size(ctx: &ReconcilerCtx) -> u64 { + ctx.shard.plane.metadata().mux_stm.streams().read(|inner| { + inner + .items + .get(0) + .map_or(0, |stream| stream.stats.size_bytes_inconsistent()) + }) + } + /// The cross-pass guard: a pass must not rebuild a namespace an earlier pass /// already built and left queued. Rebuilding is not merely wasted work -- /// the second build shares the namespace's `PartitionStats` with the queued @@ -1965,7 +2202,8 @@ mod tests { // `ensure_initial_segment` plants exactly one segment per build and // folds it into the namespace's shared stats, so this counter is the // observable that separates them. - let (stats, _) = fetch_partition_stats(&ctx, ns).expect("materialised namespace has stats"); + let (stats, ..) = + fetch_partition_build_inputs(&ctx, ns).expect("materialised namespace has stats"); assert_eq!( stats.segments_count_inconsistent(), 1, @@ -2002,7 +2240,8 @@ mod tests { let ctx = make_ctx(Rc::clone(&shard), 1, Rc::new(config.clone())); let ns = IggyNamespace::new(0, 0, 0); - let (stats, _) = fetch_partition_stats(&ctx, ns).expect("committed namespace has stats"); + let (stats, ..) = + fetch_partition_build_inputs(&ctx, ns).expect("committed namespace has stats"); let live = build_partition_fresh( &config, ns, diff --git a/core/server/src/responses.rs b/core/server/src/responses.rs index cd3460a331..1c6c581230 100644 --- a/core/server/src/responses.rs +++ b/core/server/src/responses.rs @@ -1415,13 +1415,14 @@ fn partition_response( // across all shards and both left-right buffers). // // Registration is NOT materialization: the owning shard's reconciler mints - // the entry (get-or-create in `fetch_partition_stats`) before it builds the - // partition, and `ensure_initial_segment` only bumps `segments_count` once - // the segment file is open. So a registry MISS and a registered entry still - // reading zero segments are the same thing to a caller -- committed, not yet - // holding storage -- and both report the deterministic shape every - // materialization lands on: one empty segment at offset 0. A bare zero - // would read as "no storage" to a client polling right after `create_topic`. + // the entry (get-or-create in `fetch_partition_build_inputs`) before it + // builds the partition, and `ensure_initial_segment` only bumps + // `segments_count` once the segment file is open. So a registry MISS and a + // registered entry still reading zero segments are the same thing to a + // caller -- committed, not yet holding storage -- and both report the + // deterministic shape every materialization lands on: one empty segment at + // offset 0. A bare zero would read as "no storage" to a client polling + // right after `create_topic`. // // Cost of the clamp: a partition fenced for rebuild (tombstoned after a // refused chain) also reads as one empty segment rather than zero. Telling @@ -2136,7 +2137,9 @@ mod tests { // before `ensure_initial_segment` runs, so this is the SAME state to a // caller and must not read as "no storage". let topic_stats = Arc::new(TopicStats::new(Arc::new(StreamStats::default()))); - let stats = streams.stats_registry.partition(0, 0, 0, topic_stats); + let stats = streams + .stats_registry + .partition(0, 0, &partition, topic_stats); let mid_build = partition_response(&streams, 0, 0, &partition).expect("response builds"); assert_eq!(mid_build.segments_count, 1); @@ -2184,7 +2187,12 @@ mod tests { // counter reported 1 here, so a caller polling `[stats]` twice saw the // total climb to 2 with no write in between (and `get_topic` already // reported 2 for the same partitions). - let stats = streams.stats_registry.partition(0, 0, 0, topic_stats); + let stats = streams.stats_registry.partition( + 0, + 0, + &Partition::new(0, 1, created_at, 0, 0), + topic_stats, + ); stats.increment_segments_count(1); let (_, _, partitions, segments, _, _) = @@ -2200,7 +2208,7 @@ mod tests { let late = streams.stats_registry.partition( 0, 0, - 1, + &Partition::new(1, 1, created_at, 0, 0), Arc::new(TopicStats::new(Arc::new(StreamStats::default()))), ); late.increment_segments_count(1); diff --git a/core/shard/src/lib.rs b/core/shard/src/lib.rs index ac4c2b0185..6f6f9d546c 100644 --- a/core/shard/src/lib.rs +++ b/core/shard/src/lib.rs @@ -2537,6 +2537,17 @@ where // the floor with the partition value. self.metrics .record_partition_prepare_gap_drops(partition.take_prepare_gap_drops()); + // Roll whatever this partition still counts out of its + // parent topic and stream. Here and not in the + // reconciler's teardown: a handler suspended mid-append + // holds its own `Arc` past the tombstone, and an earlier + // settle leaves its increment in the parents with the + // partition already gone. This is the drop point, so + // nothing can add through that handle afterwards. The + // rollback clamps, so the usual case -- the metadata + // apply already zeroed these counters at commit -- takes + // nothing. + partition.stats.zero_out_all(); } else { tracing::trace!( shard = self_shard_id, @@ -11297,6 +11308,9 @@ mod repair_scope_tests { assert_eq!(adopted_suffix_head(&pending, 98, 101), Some(100)); assert_eq!(adopted_suffix_head(&pending, 99, 101), Some(100)); assert_eq!(adopted_suffix_head(&pending, 100, 101), None); + // A parked head ABOVE the local head is a different shape -- ops this + // replica has not sequenced at all -- and stays out of scope. + assert_eq!(adopted_suffix_head(&pending, 98, 99), None); let suffix = adopted_suffix_head(&pending, 98, 101); assert_eq!( super::partition_repair_fetch_to_op(0, 98, suffix), From 2e38231960f4d3805ca72d8279d3e95557936ede Mon Sep 17 00:00:00 2001 From: Amrit <252372762+amr8t@users.noreply.github.com> Date: Tue, 8 Sep 2026 09:42:35 -0700 Subject: [PATCH 088/182] feat(connectors): add RabbitMQ sink (#3973) Replaces #3811 (could not be reopened after rebasing onto master). Rebased to current master and addressed all review comments. Relates to #3747 Adds the RabbitMQ sink connector via the lapin client. Source will be a separate PR. --- .github/workflows/_build_rust_artifacts.yml | 2 +- .github/workflows/edge-release.yml | 1 + Cargo.lock | 1495 ++++++++++++----- Cargo.toml | 2 + core/connectors/README.md | 1 + .../connectors/rabbitmq_sink.toml | 46 + core/connectors/sinks/README.md | 1 + .../connectors/sinks/rabbitmq_sink/Cargo.toml | 51 + core/connectors/sinks/rabbitmq_sink/README.md | 50 + .../sinks/rabbitmq_sink/config.toml | 46 + .../connectors/sinks/rabbitmq_sink/src/lib.rs | 783 +++++++++ core/integration/Cargo.toml | 1 + .../tests/connectors/fixtures/mod.rs | 6 + .../connectors/fixtures/rabbitmq/container.rs | 327 ++++ .../tests/connectors/fixtures/rabbitmq/mod.rs | 26 + .../connectors/fixtures/rabbitmq/sink.rs | 273 +++ core/integration/tests/connectors/mod.rs | 1 + .../tests/connectors/rabbitmq/mod.rs | 18 + .../connectors/rabbitmq/rabbitmq_sink.rs | 429 +++++ .../tests/connectors/rabbitmq/sink.toml | 20 + scripts/bump-version.sh | 2 +- 21 files changed, 3123 insertions(+), 458 deletions(-) create mode 100644 core/connectors/runtime/example_config/connectors/rabbitmq_sink.toml create mode 100644 core/connectors/sinks/rabbitmq_sink/Cargo.toml create mode 100644 core/connectors/sinks/rabbitmq_sink/README.md create mode 100644 core/connectors/sinks/rabbitmq_sink/config.toml create mode 100644 core/connectors/sinks/rabbitmq_sink/src/lib.rs create mode 100644 core/integration/tests/connectors/fixtures/rabbitmq/container.rs create mode 100644 core/integration/tests/connectors/fixtures/rabbitmq/mod.rs create mode 100644 core/integration/tests/connectors/fixtures/rabbitmq/sink.rs create mode 100644 core/integration/tests/connectors/rabbitmq/mod.rs create mode 100644 core/integration/tests/connectors/rabbitmq/rabbitmq_sink.rs create mode 100644 core/integration/tests/connectors/rabbitmq/sink.toml diff --git a/.github/workflows/_build_rust_artifacts.yml b/.github/workflows/_build_rust_artifacts.yml index bd56b10716..a1592ad78f 100644 --- a/.github/workflows/_build_rust_artifacts.yml +++ b/.github/workflows/_build_rust_artifacts.yml @@ -46,7 +46,7 @@ on: connector_plugins: type: string required: false - default: "iggy_connector_elasticsearch_sink,iggy_connector_elasticsearch_source,iggy_connector_iceberg_sink,iggy_connector_postgres_sink,iggy_connector_postgres_source,iggy_connector_quickwit_sink,iggy_connector_random_source,iggy_connector_s3_sink,iggy_connector_stdout_sink,iggy_connector_surrealdb_sink" + default: "iggy_connector_elasticsearch_sink,iggy_connector_elasticsearch_source,iggy_connector_iceberg_sink,iggy_connector_postgres_sink,iggy_connector_postgres_source,iggy_connector_quickwit_sink,iggy_connector_random_source,iggy_connector_s3_sink,iggy_connector_stdout_sink,iggy_connector_surrealdb_sink, iggy_connector_rabbitmq_sink" description: "Comma-separated list of connector plugin crates to build as shared libraries" outputs: artifact_name: diff --git a/.github/workflows/edge-release.yml b/.github/workflows/edge-release.yml index 2514c3227a..7dde3b7192 100644 --- a/.github/workflows/edge-release.yml +++ b/.github/workflows/edge-release.yml @@ -105,6 +105,7 @@ jobs: - `iggy_connector_s3_sink` - `iggy_connector_stdout_sink` - `iggy_connector_surrealdb_sink` + - `iggy_connector_rabbitmq_sink` ## Downloads diff --git a/Cargo.lock b/Cargo.lock index 7856dd4133..98b0f147d5 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -4,9 +4,9 @@ version = 4 [[package]] name = "actix-codec" -version = "0.5.3" +version = "0.5.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "31404e1443b7b7bcaa311c1af456775cc3f668e34573f1503d7ce4844327629e" +checksum = "5f7b0a21988c1bf877cf4759ef5ddaac04c1c9fe808c9142ecb78ba97d97a28a" dependencies = [ "bitflags 2.13.1", "bytes", @@ -59,12 +59,11 @@ dependencies = [ [[package]] name = "actix-http" -version = "3.13.3" +version = "3.13.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "11004b0e9b44b4eb3d15e0c3132b96fb178c7e50a74758b2f17bb9cc9a7fb4f6" +checksum = "86d62d1a48894ec9450bcde7ef1e3681205771ff6ea9c61cc5ae031145192787" dependencies = [ "actix-codec", - "actix-rt", "actix-service", "actix-utils", "base64 0.22.1", @@ -123,9 +122,9 @@ dependencies = [ [[package]] name = "actix-rt" -version = "2.13.0" +version = "2.11.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6a16bf2f19c2ad84842bdfe6f3665620e93197d5c607889bfa3aaac45e762fd1" +checksum = "92589714878ca59a7626ea19734f0e07a6a875197eec751bb5d3f99e64998c63" dependencies = [ "futures-core", "tokio", @@ -143,7 +142,7 @@ dependencies = [ "futures-core", "futures-util", "mio", - "socket2", + "socket2 0.6.5", "tokio", "tracing", ] @@ -205,7 +204,7 @@ dependencies = [ "serde_json", "serde_urlencoded", "smallvec", - "socket2", + "socket2 0.6.5", "time", "tracing", "url", @@ -235,7 +234,7 @@ version = "0.5.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d122413f284cf2d62fb1b7db97e02edb8cda96d769b16e443a4f6195e35662b0" dependencies = [ - "crypto-common 0.1.6", + "crypto-common 0.1.7", "generic-array", ] @@ -262,13 +261,13 @@ dependencies = [ [[package]] name = "aes" -version = "0.9.3" +version = "0.9.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "35f0f96ce78e38c3dc6d8948aa8163d06385be74000f3c7a95bf1eef35d3ea32" +checksum = "f1fc76eaeac4c9164506c466d4ffdd8ec9d0c5bf57ee97177c4d8eceb3a0e138" dependencies = [ "cipher 0.5.2", "cpubits", - "cpufeatures 0.3.1", + "cpufeatures 0.3.0", ] [[package]] @@ -292,7 +291,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7f2b8006a0c83f52b62ba44a97b58bf76fe2f70a329e588f67f89691d93d498f" dependencies = [ "aead 0.6.1", - "aes 0.9.3", + "aes 0.9.1", "cipher 0.5.2", "ctr 0.10.1", "ctutils", @@ -327,9 +326,9 @@ dependencies = [ [[package]] name = "aho-corasick" -version = "1.1.5" +version = "1.1.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c982642fa9e8606056828ee9a8505737230110bb1099153c79efe865c59d12ba" +checksum = "ddd31a130427c27518df266943a5308ed92d4b226cc639f5a8f1002816174301" dependencies = [ "memchr", ] @@ -379,11 +378,59 @@ version = "0.2.21" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "683d7910e743518b0e34f1186f92494becacb047c7b6bf616c96772180fef923" +[[package]] +name = "amq-protocol" +version = "7.2.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "587d313f3a8b4a40f866cc84b6059fe83133bf172165ac3b583129dd211d8e1c" +dependencies = [ + "amq-protocol-tcp", + "amq-protocol-types", + "amq-protocol-uri", + "cookie-factory", + "nom 7.1.3", + "serde", +] + +[[package]] +name = "amq-protocol-tcp" +version = "7.2.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "dc707ab9aa964a85d9fc25908a3fdc486d2e619406883b3105b48bf304a8d606" +dependencies = [ + "amq-protocol-uri", + "tcp-stream", + "tracing", +] + +[[package]] +name = "amq-protocol-types" +version = "7.2.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bf99351d92a161c61ec6ecb213bc7057f5b837dd4e64ba6cb6491358efd770c4" +dependencies = [ + "cookie-factory", + "nom 7.1.3", + "serde", + "serde_json", +] + +[[package]] +name = "amq-protocol-uri" +version = "7.2.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f89f8273826a676282208e5af38461a07fe939def57396af6ad5997fcf56577d" +dependencies = [ + "amq-protocol-types", + "percent-encoding", + "url", +] + [[package]] name = "android_system_properties" -version = "0.1.6" +version = "0.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ae221649c9976a6f6c56ae1facf410f3ddb33cc661c4b7b61020a912d4237fbc" +checksum = "819e7219dbd41043ac279b19830f2efc897156490d7fd6ea916720117ee66311" dependencies = [ "libc", ] @@ -503,14 +550,14 @@ checksum = "2b350bfd03649e07aa05c0a81b3e15934374e585c98204a57e20b9d49f49bb9a" dependencies = [ "keyring-core", "log", - "security-framework", + "security-framework 3.7.0", ] [[package]] name = "ar_archive_writer" -version = "0.5.3" +version = "0.5.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "73cd58deff2140a0a8eae87e417bd01db68a33e148aa93d1e8cd837e55e312b6" +checksum = "4087686b4b0a3427190bae57a1d9a478dbb2d40c5dc1bd6e2b6d797913bdd348" dependencies = [ "object", ] @@ -549,7 +596,7 @@ checksum = "134c52ddac6d63c576bef8168db10c83c49c26444ecbc68060fef078925a901c" dependencies = [ "base64ct", "blake2", - "cpufeatures 0.3.1", + "cpufeatures 0.3.0", "password-hash", ] @@ -714,7 +761,7 @@ dependencies = [ "arrow-select", "chrono", "half", - "indexmap 2.14.1", + "indexmap 2.14.2", "itoa", "lexical-core", "memchr", @@ -894,7 +941,7 @@ version = "0.7.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "435a87a52755b8f27fcf321ac4f04b2802e337c8c4872923137471ec39c37532" dependencies = [ - "event-listener", + "event-listener 5.4.1", "event-listener-strategy", "futures-core", "pin-project-lite", @@ -914,9 +961,9 @@ dependencies = [ [[package]] name = "async-compression" -version = "0.4.43" +version = "0.4.42" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3976abdc8fe7d1133d43d304afd42abdf5bc3e1319d263d223bde07b5efc4be8" +checksum = "e79b3f8a79cccc2898f31920fc69f304859b3bd567490f75ebf51ae1c792a9ac" dependencies = [ "compression-codecs", "compression-core", @@ -972,12 +1019,57 @@ checksum = "c96bf972d85afc50bf5ab8fe2d54d1586b4e0b46c97c50a0c9e71e2f7bcd812a" dependencies = [ "async-task", "concurrent-queue", - "fastrand", - "futures-lite", + "fastrand 2.5.0", + "futures-lite 2.6.1", "pin-project-lite", "slab", ] +[[package]] +name = "async-global-executor" +version = "3.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "13f937e26114b93193065fd44f507aa2e9169ad0cdabbb996920b1fe1ddea7ba" +dependencies = [ + "async-channel", + "async-executor", + "async-io 2.6.0", + "async-lock 3.4.2", + "blocking", + "futures-lite 2.6.1", +] + +[[package]] +name = "async-global-executor-trait" +version = "2.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9af57045d58eeb1f7060e7025a1631cbc6399e0a1d10ad6735b3d0ea7f8346ce" +dependencies = [ + "async-global-executor", + "async-trait", + "executor-trait", +] + +[[package]] +name = "async-io" +version = "1.13.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0fc5b45d93ef0529756f812ca52e44c221b35341892d3dcc34132ac02f3dd2af" +dependencies = [ + "async-lock 2.8.0", + "autocfg", + "cfg-if", + "concurrent-queue", + "futures-lite 1.13.0", + "log", + "parking", + "polling 2.8.0", + "rustix 0.37.28", + "slab", + "socket2 0.4.10", + "waker-fn", +] + [[package]] name = "async-io" version = "2.6.0" @@ -988,21 +1080,30 @@ dependencies = [ "cfg-if", "concurrent-queue", "futures-io", - "futures-lite", + "futures-lite 2.6.1", "parking", - "polling", + "polling 3.11.0", "rustix 1.1.4", "slab", "windows-sys 0.61.2", ] +[[package]] +name = "async-lock" +version = "2.8.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "287272293e9d8c41773cec55e365490fe034813a2f172f502d6ddcf75b2f582b" +dependencies = [ + "event-listener 2.5.3", +] + [[package]] name = "async-lock" version = "3.4.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "290f7f2596bd5b78a9fec8088ccd89180d7f9f55b94b0576823bbbdc72ee8311" dependencies = [ - "event-listener", + "event-listener 5.4.1", "event-listener-strategy", "pin-project-lite", ] @@ -1014,17 +1115,29 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "fc50921ec0055cdd8a16de48773bfeec5c972598674347252c0399676be7da75" dependencies = [ "async-channel", - "async-io", - "async-lock", + "async-io 2.6.0", + "async-lock 3.4.2", "async-signal", "async-task", "blocking", "cfg-if", - "event-listener", - "futures-lite", + "event-listener 5.4.1", + "futures-lite 2.6.1", "rustix 1.1.4", ] +[[package]] +name = "async-reactor-trait" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7a6012d170ad00de56c9ee354aef2e358359deb1ec504254e0e5a3774771de0e" +dependencies = [ + "async-io 1.13.0", + "async-trait", + "futures-core", + "reactor-trait", +] + [[package]] name = "async-recursion" version = "1.1.1" @@ -1053,8 +1166,8 @@ version = "0.2.14" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "52b5aaafa020cf5053a01f2a60e8ff5dccf550f0f77ec54a4e47285ac2bab485" dependencies = [ - "async-io", - "async-lock", + "async-io 2.6.0", + "async-lock 3.4.2", "atomic-waker", "cfg-if", "futures-core", @@ -1101,7 +1214,7 @@ checksum = "82f6aeea286b8eb4dd3431a1be1b59d290ace00f5bfd8e2a159bc2a05e2c1667" dependencies = [ "proc-macro2", "quote", - "syn 3.0.4", + "syn 3.0.5", ] [[package]] @@ -1129,7 +1242,7 @@ dependencies = [ "async-compression", "binrw", "crc32fast", - "futures-lite", + "futures-lite 2.6.1", "pin-project", "thiserror 2.0.20", "tokio", @@ -1245,9 +1358,9 @@ dependencies = [ [[package]] name = "aws-config" -version = "1.11.0" +version = "1.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a767267da9e2c2e189b2f9df8b5657e850ecf5352644734ba130d4a57095cf1b" +checksum = "47712fde1909402600ccfbb26e47d482d2e58bb9e9e603d9f17e67cc435a6319" dependencies = [ "aws-credential-types", "aws-runtime", @@ -1263,7 +1376,7 @@ dependencies = [ "aws-smithy-types", "aws-types", "bytes", - "fastrand", + "fastrand 2.5.0", "hex", "http 1.5.0", "sha1 0.10.7", @@ -1338,9 +1451,9 @@ dependencies = [ [[package]] name = "aws-runtime" -version = "1.9.1" +version = "1.8.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c9007227e10b5fed2f3e0a2beff489211e2b5604c400b7a9d5d81ca9d64c24bb" +checksum = "7816e98ee912159f45d307e5ee6bfea4a335a55aee15f7f3e32f81a6f3000f1d" dependencies = [ "aws-credential-types", "aws-sigv4", @@ -1352,7 +1465,7 @@ dependencies = [ "aws-types", "bytes", "bytes-utils", - "fastrand", + "fastrand 2.5.0", "http 1.5.0", "http-body 1.1.0", "percent-encoding", @@ -1363,9 +1476,9 @@ dependencies = [ [[package]] name = "aws-sdk-dynamodb" -version = "1.123.0" +version = "1.117.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dcd10d92058d5da989a7838fd09d70e9457390a81ba672abf5685099f3562bce" +checksum = "0f2b4adcf6592e06ebb2c78235ae09cee178263a5b7fc20ae5ea07ca67067bb6" dependencies = [ "arc-swap", "aws-credential-types", @@ -1380,19 +1493,18 @@ dependencies = [ "aws-smithy-types", "aws-types", "bytes", - "fastrand", + "fastrand 2.5.0", "http 0.2.12", "http 1.5.0", "regex-lite", "tracing", - "url", ] [[package]] name = "aws-sdk-sso" -version = "1.108.0" +version = "1.103.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c15301b04372832947916607983b114b3374b9db0be058a00fb7513800de1f05" +checksum = "0469f435f645ad2162cfb463b15bde37115966ee3acf2d87fb4871ee309b8401" dependencies = [ "arc-swap", "aws-credential-types", @@ -1407,7 +1519,7 @@ dependencies = [ "aws-smithy-types", "aws-types", "bytes", - "fastrand", + "fastrand 2.5.0", "http 0.2.12", "http 1.5.0", "regex-lite", @@ -1416,9 +1528,9 @@ dependencies = [ [[package]] name = "aws-sdk-ssooidc" -version = "1.110.0" +version = "1.105.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "72cc2c205cb27108183cf1856333f7d584c2ba0f505421b4209ca5828f9ea899" +checksum = "085faefb253f770655e162b9304321e62a1e71adf7f019ee1f4454228a377b3a" dependencies = [ "arc-swap", "aws-credential-types", @@ -1433,7 +1545,7 @@ dependencies = [ "aws-smithy-types", "aws-types", "bytes", - "fastrand", + "fastrand 2.5.0", "http 0.2.12", "http 1.5.0", "regex-lite", @@ -1442,9 +1554,9 @@ dependencies = [ [[package]] name = "aws-sdk-sts" -version = "1.113.0" +version = "1.108.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "68182ecb449f7537db0f4d5d25917789cf41e32074a9fe47b6a0b847fe1d2032" +checksum = "3c72b08911d8128dd360fe1b22a9fec0fa8b552dde8ec828dcf20ef5ec974e9f" dependencies = [ "arc-swap", "aws-credential-types", @@ -1460,7 +1572,7 @@ dependencies = [ "aws-smithy-types", "aws-smithy-xml", "aws-types", - "fastrand", + "fastrand 2.5.0", "http 0.2.12", "http 1.5.0", "regex-lite", @@ -1523,21 +1635,21 @@ dependencies = [ [[package]] name = "aws-smithy-http-client" -version = "1.4.0" +version = "1.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ebfd138fac0337cee7516c352757ea73b9f2266e57d0bcb5bc70e9547e45aef1" +checksum = "635d23afda0a6ab48d666c4d447c4873e8d1e83518a2be2093122397e50b838e" dependencies = [ "aws-smithy-async", "aws-smithy-runtime-api", "aws-smithy-types", - "h2 0.4.19", + "h2 0.4.15", "http 1.5.0", "hyper", "hyper-rustls", "hyper-util", "pin-project-lite", "rustls", - "rustls-native-certs", + "rustls-native-certs 0.8.4", "rustls-pki-types", "tokio", "tokio-rustls", @@ -1567,22 +1679,19 @@ dependencies = [ [[package]] name = "aws-smithy-query" -version = "0.62.0" +version = "0.61.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "512346c7212ab7436df2d77a16d976a468ae44a418835511d2a69269810aaf62" +checksum = "dd22a6ba36e3f113cb8d5b3d1fe0ed31c76ee608ef63322d753bb8d2c9479e77" dependencies = [ - "aws-smithy-runtime-api", - "aws-smithy-schema", "aws-smithy-types", - "aws-smithy-xml", "urlencoding", ] [[package]] name = "aws-smithy-runtime" -version = "1.14.0" +version = "1.12.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b82e438d30e02a825d363bd639a9efaed68a8089d86101054b0081e7e0d3e606" +checksum = "bea94a9ff8464016338c851e24b472d7131c388c88898a502e781815b2ee6045" dependencies = [ "aws-smithy-async", "aws-smithy-http", @@ -1592,7 +1701,7 @@ dependencies = [ "aws-smithy-schema", "aws-smithy-types", "bytes", - "fastrand", + "fastrand 2.5.0", "http 0.2.12", "http 1.5.0", "http-body 0.4.6", @@ -1606,9 +1715,9 @@ dependencies = [ [[package]] name = "aws-smithy-runtime-api" -version = "1.15.0" +version = "1.13.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "954c563ce84507722d2679f07a35d21b9c6466b3872d513020d0281fc8112ac9" +checksum = "22ed1ebe6e0a95ea84570225f5a8208dec4b8f77e61a9b0d6f51773fcb4612f0" dependencies = [ "aws-smithy-async", "aws-smithy-runtime-api-macros", @@ -1646,9 +1755,9 @@ dependencies = [ [[package]] name = "aws-smithy-types" -version = "1.6.2" +version = "1.6.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fce83ce9abbb198d25bc7131e468d0f9fe1257125e58c39f3f9fc9f5098c9647" +checksum = "d6dc683efb34b9e755675b37fedbe0103141e5b6df7bdc9eb6967756a8c167d8" dependencies = [ "base64-simd", "bytes", @@ -1672,9 +1781,9 @@ dependencies = [ [[package]] name = "aws-smithy-xml" -version = "0.62.0" +version = "0.61.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ce84f71c72fee2cbbadde6e7d082f5fb466e3a84733855295fa7aafd1b31b7d8" +checksum = "ea3f68eec3607f02acd24067969ce2abc6ba16aa7d5ce59ca450ed2fb5f78957" dependencies = [ "aws-smithy-runtime-api", "aws-smithy-schema", @@ -1684,9 +1793,9 @@ dependencies = [ [[package]] name = "aws-types" -version = "1.5.0" +version = "1.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "eec1cd5469f328c782dc3e33d4153cf118a54e33cbb3356d60d16f89883e1f94" +checksum = "e957a6c6dbce82b7a91f44231c09273159703769f447cbe85e854dfe9cf67f86" dependencies = [ "aws-credential-types", "aws-smithy-async", @@ -1795,7 +1904,7 @@ version = "1.6.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "cffb0e931875b666fc4fcb20fee52e9bbd1ef836fd9e9e04ec21555f9f85f7ef" dependencies = [ - "fastrand", + "fastrand 2.5.0", "gloo-timers 0.3.0", "tokio", ] @@ -2100,7 +2209,7 @@ dependencies = [ "cc", "cfg-if", "constant_time_eq", - "cpufeatures 0.3.1", + "cpufeatures 0.3.0", ] [[package]] @@ -2123,11 +2232,11 @@ dependencies = [ [[package]] name = "block-padding" -version = "0.4.2" +version = "0.3.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "710f1dd022ef4e93f8a438b4ba958de7f64308434fa6a87104481645cc30068b" +checksum = "a8894febbff9f758034a5b8e12d87918f56dfc64a8e1fe757d65e29041538d93" dependencies = [ - "hybrid-array", + "generic-array", ] [[package]] @@ -2141,14 +2250,14 @@ dependencies = [ [[package]] name = "blocking" -version = "1.7.0" +version = "1.6.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a70e4329df6cb94385eed412ec92375c3cdd8a6e502493d1229b6414e4036dfa" +checksum = "e83f8d02be6967315521be875afa792a316e28d57b5a2d401897e2a7921b7f21" dependencies = [ "async-channel", "async-task", "futures-io", - "futures-lite", + "futures-lite 2.6.1", "piper", ] @@ -2190,7 +2299,7 @@ dependencies = [ "pin-project-lite", "rand 0.9.5", "rustls", - "rustls-native-certs", + "rustls-native-certs 0.8.4", "rustls-pki-types", "serde", "serde_derive", @@ -2238,32 +2347,32 @@ dependencies = [ [[package]] name = "bon" -version = "3.10.0" +version = "3.10.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "9e3fac94a66da67200398458a25412bcc3f9b6443b5119a6cad9cf3ccfcd8cc6" +checksum = "60eafe0d77c3a2fc292c1d1346c3041b33c0a108085a2afabf672b70f69dbbc9" dependencies = [ "bon-macros", ] [[package]] name = "bon-macros" -version = "3.10.0" +version = "3.10.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d4654961ad0494e4774c5c60b4cb4cd0ae9b9d92d039d901638b1dba97ebebf5" +checksum = "bd0f9631d8aaaee112c41985d675ef269e02acbd4f33122836af4f0c5f699ff6" dependencies = [ "darling 0.24.1", "ident_case", "prettyplease 0.3.0", "proc-macro2", "quote", - "syn 3.0.4", + "syn 3.0.5", ] [[package]] name = "borsh" -version = "1.8.1" +version = "1.8.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "553c5d846a6ba5150c65e3b1b8ec073bcf1abc20f9b7220de384a4443ea4e20a" +checksum = "a88b7ea17d208c4193f2c1e6de3c35fe71f98c96982d5ced308bdcc749ff6e1f" dependencies = [ "borsh-derive", "bytes", @@ -2272,15 +2381,15 @@ dependencies = [ [[package]] name = "borsh-derive" -version = "1.8.1" +version = "1.8.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "12cdfe656708a01f89b451a7d36466e6fe6c414de0aa18fc54f864f6f9ca9f56" +checksum = "d8f347189c62a579b8cd5f80714efa178f52e461dc2e6d701d264f5ff22e566c" dependencies = [ "once_cell", "proc-macro-crate 3.3.0", "proc-macro2", "quote", - "syn 3.0.4", + "syn 2.0.119", ] [[package]] @@ -2325,7 +2434,7 @@ dependencies = [ "getrandom 0.2.17", "getrandom 0.3.4", "hex", - "indexmap 2.14.1", + "indexmap 2.14.2", "js-sys", "once_cell", "rand 0.9.5", @@ -2338,9 +2447,9 @@ dependencies = [ [[package]] name = "bstr" -version = "1.13.1" +version = "1.13.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6bb31b46c14244e20ee9984b11bf5c992b91fb6939fea616e3512c8baecdbe5f" +checksum = "1f7dc094d718f2e1c1559ad110e27eeaae14a5465d3d56dd6dbd793079fbd530" dependencies = [ "memchr", "regex-automata", @@ -2371,7 +2480,7 @@ dependencies = [ "chrono", "crc", "futures", - "indexmap 2.14.1", + "indexmap 2.14.2", "itertools 0.14.0", "object_store", "parquet", @@ -2410,7 +2519,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "4a813de7f2bbedb7dce265b64f1cf5908ebe4d56281ece8d847e98113788b9b0" dependencies = [ "rust_decimal", - "schemars 1.2.2", + "schemars 1.2.1", "serde", "utf8-width", ] @@ -2454,13 +2563,13 @@ dependencies = [ [[package]] name = "bytemuck_derive" -version = "1.12.0" +version = "1.11.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fc0e56a716f1e132ff6bf4bdac1c944a3fcdc1cae65f70a4a2a1ac3b401d2d1f" +checksum = "f65693059b6b9c588b9f62fed1cedbf0a8b805631457ea162d68f0de186f3de5" dependencies = [ "proc-macro2", "quote", - "syn 3.0.4", + "syn 2.0.119", ] [[package]] @@ -2511,9 +2620,9 @@ dependencies = [ [[package]] name = "camino" -version = "1.2.5" +version = "1.2.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bb1307f12aa967b5a58416e87b3653360e0fd614a016b6e970db08fecbb1b80d" +checksum = "5f2d30e4173c4026932d51d31d6b0613b1fd3014bf3f9f8943d4ba139c437ba0" dependencies = [ "serde_core", ] @@ -2573,18 +2682,18 @@ dependencies = [ [[package]] name = "cbc" -version = "0.2.1" +version = "0.1.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ce2dc9ee5f88d11e0beb842c88b33c8a5cf0d1329c4b19494af42b07dbfe8896" +checksum = "26b52a9543ae338f279b96b0b9fed9c8093744685043739079ce85cd58f289a6" dependencies = [ - "cipher 0.5.2", + "cipher 0.4.4", ] [[package]] name = "cc" -version = "1.4.4" +version = "1.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0ad534f4357a5264cce5019c989cf66a4f0dc4e0d1b1d15f8aacec0ff7360273" +checksum = "c89588d05638b5b4594a3348a2d6c20277e43a7f5c5202b05cc56888475a47b8" dependencies = [ "find-msvc-tools", "jobserver", @@ -2615,12 +2724,12 @@ checksum = "f079e83a288787bcd14a6aea84cee5c87a67c5a3e660c30f557a3d24761b3527" [[package]] name = "chacha20" -version = "0.10.2" +version = "0.10.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "65c35e4b699c7e15ccbe7ee35c005e4fc0a278d22238a2857e6ce2dadeda1b06" +checksum = "d524456ba66e72eb8b115ff89e01e497f8e6d11d78b70b1aa13c0fbd97540a81" dependencies = [ "cfg-if", - "cpufeatures 0.3.1", + "cpufeatures 0.3.0", "rand_core 0.10.1", ] @@ -2686,7 +2795,7 @@ version = "0.4.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "773f3b9af64447d2ce9850330c473515014aa235e6a783b02db81ff39e4a3dad" dependencies = [ - "crypto-common 0.1.6", + "crypto-common 0.1.7", "inout 0.1.4", ] @@ -2703,9 +2812,9 @@ dependencies = [ [[package]] name = "clang-sys" -version = "1.9.1" +version = "1.8.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "157a8ba7b480713b56f4c09fd13fc3e0a22a5dfab8097ba61cbc5feef950788a" +checksum = "0b023947811758c97c59bf9d1c188fd619ad4718dcaa767947df1cadb14f39f4" dependencies = [ "glob", "libc", @@ -2753,7 +2862,7 @@ dependencies = [ "heck 0.5.0", "proc-macro2", "quote", - "syn 3.0.4", + "syn 3.0.5", ] [[package]] @@ -2784,6 +2893,18 @@ version = "0.5.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0c9ea0ac24bc397ab3c98583a3c9ba74fa56b09a4449bbe172b9b1ddb016027a" +[[package]] +name = "cms" +version = "0.2.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7b77c319abfd5219629c45c34c89ba945ed3c5e49fcde9d16b6c3885f118a730" +dependencies = [ + "const-oid 0.9.6", + "der", + "spki", + "x509-cert", +] + [[package]] name = "cobs" version = "0.3.0" @@ -2816,9 +2937,9 @@ dependencies = [ [[package]] name = "combine" -version = "4.6.8" +version = "4.6.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cfc320937d09e6de266b31b9afb480f197d7a861be86be7cb2ea7e5d1bfffc5e" +checksum = "ba5a308b75df32fe02788e748662718f03fde005016435c444eea572398219fd" dependencies = [ "bytes", "memchr", @@ -2889,7 +3010,7 @@ dependencies = [ "compio-log", "compio-send-wrapper", "crossbeam-queue", - "flume", + "flume 0.12.0", "futures-util", "io-uring", "libc", @@ -2897,10 +3018,10 @@ dependencies = [ "mod_use", "once_cell", "pastey 0.2.3", - "polling", + "polling 3.11.0", "rustix 1.1.4", "smallvec", - "socket2", + "socket2 0.6.5", "synchrony", "thin-cell", "windows-sys 0.61.2", @@ -2974,7 +3095,7 @@ dependencies = [ "proc-macro-crate 3.3.0", "proc-macro2", "quote", - "syn 3.0.4", + "syn 3.0.5", ] [[package]] @@ -2992,7 +3113,7 @@ dependencies = [ "libc", "once_cell", "pin-project-lite", - "socket2", + "socket2 0.6.5", "synchrony", "widestring", "windows-sys 0.61.2", @@ -3010,7 +3131,7 @@ dependencies = [ "compio-log", "compio-net", "compio-runtime", - "flume", + "flume 0.12.0", "futures-util", "libc", "quinn-proto", @@ -3139,7 +3260,7 @@ dependencies = [ "darling 0.24.1", "proc-macro2", "quote", - "syn 3.0.4", + "syn 3.0.5", ] [[package]] @@ -3277,6 +3398,12 @@ dependencies = [ "version_check", ] +[[package]] +name = "cookie-factory" +version = "0.3.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9885fa71e26b8ab7855e2ec7cae6e9b380edff76cd052e07c683a0319d51b3a2" + [[package]] name = "core-foundation" version = "0.9.4" @@ -3338,9 +3465,9 @@ dependencies = [ [[package]] name = "cpufeatures" -version = "0.3.1" +version = "0.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5ca28b0ae3115b884660db4118d803791fd6756b6e88f39c0f3f7859060d7566" +checksum = "8b2a41393f66f16b0823bb79094d54ac5fbd34ab292ddafb9a0456ac9f87d201" dependencies = [ "libc", ] @@ -3371,9 +3498,9 @@ dependencies = [ [[package]] name = "crc32fast" -version = "1.5.1" +version = "1.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8498c871161e1742aaa9d52551b2d6ebdd4c3d45a3be423e3728f33b955be550" +checksum = "9481c1c90cbf2ac953f07c8d4a58aa3945c425b7185c9154d67a65e4230da511" dependencies = [ "cfg-if", ] @@ -3473,9 +3600,9 @@ dependencies = [ [[package]] name = "crypto-common" -version = "0.1.6" +version = "0.1.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1bfb12502f3fc46cca1bb51ac28df9d618d813cdc3d2f25b9fe775a34af26bb3" +checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a" dependencies = [ "generic-array", "rand_core 0.6.4", @@ -3707,7 +3834,7 @@ dependencies = [ "hyper", "hyper-util", "send_wrapper", - "socket2", + "socket2 0.6.5", "tokio", "tower", "tower-service", @@ -3815,7 +3942,7 @@ dependencies = [ "proc-macro2", "quote", "strsim", - "syn 3.0.4", + "syn 3.0.5", ] [[package]] @@ -3859,7 +3986,7 @@ checksum = "2ac7135c3ef02b2f7833bbeb1be5ba7f966dcde8a87c6b87f65a778d71a02785" dependencies = [ "darling_core 0.24.1", "quote", - "syn 3.0.4", + "syn 3.0.5", ] [[package]] @@ -3878,9 +4005,9 @@ dependencies = [ [[package]] name = "data-encoding" -version = "2.11.1" +version = "2.11.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4583a4551df46e2792f82ceeac45e850d2e2d5debba0b91f102385cda5b11f06" +checksum = "a4ae5f15dda3c708c0ade84bfee31ccab44a3da4f88015ed22f63732abe300c8" [[package]] name = "data-url" @@ -4030,7 +4157,7 @@ dependencies = [ "either", "futures", "humantime", - "indexmap 2.14.1", + "indexmap 2.14.2", "itertools 0.14.0", "num_cpus", "object_store", @@ -4103,7 +4230,7 @@ dependencies = [ "deno_path_util", "deno_unsync", "futures", - "indexmap 2.14.1", + "indexmap 2.14.2", "libc", "parking_lot", "percent-encoding", @@ -4158,7 +4285,7 @@ version = "0.227.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1bab1eaf578a8cc0ae6fb933e91dc3388b41df22e5974d5891c17ba66b3a0bbb" dependencies = [ - "indexmap 2.14.1", + "indexmap 2.14.2", "proc-macro-rules", "proc-macro2", "quote", @@ -4200,6 +4327,8 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e7c1832837b905bbfb5101e07cc24c8deddf52f93225eee6ead5f4d63d53ddcb" dependencies = [ "const-oid 0.9.6", + "der_derive", + "flagset", "pem-rfc7468", "zeroize", ] @@ -4218,6 +4347,17 @@ dependencies = [ "rusticata-macros", ] +[[package]] +name = "der_derive" +version = "0.7.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8034092389675178f570469e6c3b0465d3d30b4505c294a6550db47f3c17ad18" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + [[package]] name = "deranged" version = "0.5.8" @@ -4314,6 +4454,15 @@ dependencies = [ "unicode-xid", ] +[[package]] +name = "des" +version = "0.8.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ffdd80ce8ce993de27e9f063a444a4d53ce8e8db4c1f00cc03af5ad5a9867a1e" +dependencies = [ + "cipher 0.4.4", +] + [[package]] name = "difflib" version = "0.4.0" @@ -4328,7 +4477,7 @@ checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" dependencies = [ "block-buffer 0.10.4", "const-oid 0.9.6", - "crypto-common 0.1.6", + "crypto-common 0.1.7", "subtle", ] @@ -4390,13 +4539,13 @@ dependencies = [ [[package]] name = "displaydoc" -version = "0.2.7" +version = "0.2.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c6232dd377dcc64799954cbd3a9bb882e9cdc1308ccd87b1c098f1fb2eaf82a8" +checksum = "1ac70aa55017e108007fbaf5aa0f54b021c98f92ff8af59d42eda9da96e3dd4f" dependencies = [ "proc-macro2", "quote", - "syn 3.0.4", + "syn 2.0.119", ] [[package]] @@ -4424,7 +4573,7 @@ checksum = "1aacdd87ba97c2f83ff479e881e547f776b0a7c42921be22d4391f32b6284050" dependencies = [ "proc-macro2", "quote", - "syn 3.0.4", + "syn 3.0.5", ] [[package]] @@ -4436,6 +4585,12 @@ dependencies = [ "const-random", ] +[[package]] +name = "doc-comment" +version = "0.3.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "780955b8b195a21ab8e4ac6b60dd1dbdcec1dc6c51c0617964b08c81785e12c9" + [[package]] name = "docker_credential" version = "1.4.0" @@ -4541,9 +4696,9 @@ dependencies = [ [[package]] name = "either" -version = "1.18.0" +version = "1.16.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "252afb9ae5eaa683babdc6a068b3f5726eb19e05070c731f9b2a23a7c3e8ed34" +checksum = "91622ff5e7162018101f2fea40d6ebf4a78bbe5a49736a2020649edf9693679e" dependencies = [ "serde", ] @@ -4777,10 +4932,17 @@ dependencies = [ [[package]] name = "event-listener" -version = "5.4.2" +version = "2.5.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0206175f82b8d6bf6652ff7d71a1e27fd2e4efde587fd368662814d6ec1d9ce0" + +[[package]] +name = "event-listener" +version = "5.4.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5a23add41df1562121a9393cb065eab5146a1242410f23a644851e90cfd669d2" +checksum = "e13b66accf52311f30a0db42147dadea9850cb48cd070028831ae5f5d4b856ab" dependencies = [ + "concurrent-queue", "parking", "pin-project-lite", ] @@ -4791,10 +4953,19 @@ version = "0.5.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8be9f3dfaaffdae2972880079a491a1a8bb7cbed0b8dd7a347f668b4150a3b93" dependencies = [ - "event-listener", + "event-listener 5.4.1", "pin-project-lite", ] +[[package]] +name = "executor-trait" +version = "2.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "13c39dff9342e4e0e16ce96be751eb21a94e94a87bb2f6e63ad1961c2ce109bf" +dependencies = [ + "async-trait", +] + [[package]] name = "expect-test" version = "1.5.1" @@ -4852,6 +5023,15 @@ dependencies = [ "serde", ] +[[package]] +name = "fastrand" +version = "1.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e51093e27b0797c359783294ca4f0a911c270184cb10f85783b118614a1501be" +dependencies = [ + "instant", +] + [[package]] name = "fastrand" version = "2.5.0" @@ -4947,9 +5127,15 @@ dependencies = [ [[package]] name = "find-msvc-tools" -version = "0.1.11" +version = "0.1.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5baebc0774151f905a1a2cc41989300b1e6fbb29aff0ceffa1064fdd3088d582" + +[[package]] +name = "flagset" +version = "0.4.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d45db016d36b838f563236e9193d0ee6ce38f3f68b6c94e914b4929c96bbb890" +checksum = "b7ac824320a75a52197e8f2d787f6a38b6718bb6897a35142d749af3c0e8f4fe" [[package]] name = "flatbuffers" @@ -4963,12 +5149,12 @@ dependencies = [ [[package]] name = "flate2" -version = "1.1.10" +version = "1.1.9" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb" +checksum = "843fba2746e448b37e26a819579957415c8cef339bf08564fe8b7ddbd959573c" dependencies = [ "crc32fast", - "miniz_oxide 0.9.1", + "miniz_oxide 0.8.9", "zlib-rs", ] @@ -4987,13 +5173,24 @@ dependencies = [ "num-traits", ] +[[package]] +name = "flume" +version = "0.11.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "da0e4dd2a88388a1f4ccc7c9ce104604dab68d9f408dc34cd45823d5a9069095" +dependencies = [ + "futures-core", + "futures-sink", + "spin", +] + [[package]] name = "flume" version = "0.12.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "5e139bc46ca777eb5efaf62df0ab8cc5fd400866427e56c68b22e414e53bd3be" dependencies = [ - "fastrand", + "fastrand 2.5.0", "futures-core", "futures-sink", "spin", @@ -5162,13 +5359,28 @@ version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "53c0fa8157de1303bfffdaa1cc2a673bfffb60102f76b0ef4441659124373fed" +[[package]] +name = "futures-lite" +version = "1.13.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "49a9d51ce47660b1e808d3c990b4709f2f415d928835a17dfd16991515c46bce" +dependencies = [ + "fastrand 1.9.0", + "futures-core", + "futures-io", + "memchr", + "parking", + "pin-project-lite", + "waker-fn", +] + [[package]] name = "futures-lite" version = "2.6.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f78e10609fe0e0b3f4157ffab1876319b5b0db102a2c60dc4626306dc46b44ad" dependencies = [ - "fastrand", + "fastrand 2.5.0", "futures-core", "futures-io", "parking", @@ -5183,7 +5395,7 @@ checksum = "9fb9654ba8355388abeb8dcb4fc62f511300867002afc858860463bdd9fe0c44" dependencies = [ "proc-macro2", "quote", - "syn 3.0.4", + "syn 3.0.5", ] [[package]] @@ -5249,9 +5461,9 @@ dependencies = [ [[package]] name = "generic-array" -version = "0.14.9" +version = "0.14.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4bb6743198531e02858aeaea5398fcc883e71851fcbcb5a2f773e2fb6cb1edf2" +checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a" dependencies = [ "typenum", "version_check", @@ -5369,15 +5581,15 @@ dependencies = [ [[package]] name = "glob" -version = "0.3.4" +version = "0.3.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e4eba85ea1d0a966a983acd07deee566e67395d2d96b6fb39e62b5a833f1eb0b" +checksum = "0cc23270f6e1808e30a928bdc84dea0b9b4136a8bc82338574f23baf47bbd280" [[package]] name = "globset" -version = "0.4.20" +version = "0.4.19" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "07c34a9410465b45bd9787443bc7370f37735bad04b0f0cd57ff1a3186c98988" +checksum = "e47d37d2ae4464254884b60ab7071be2b876a9c35b696bd018ddcc76847309cd" dependencies = [ "aho-corasick", "bstr", @@ -5817,7 +6029,7 @@ dependencies = [ "futures-sink", "futures-util", "http 0.2.12", - "indexmap 2.14.1", + "indexmap 2.14.2", "slab", "tokio", "tokio-util", @@ -5826,9 +6038,9 @@ dependencies = [ [[package]] name = "h2" -version = "0.4.19" +version = "0.4.15" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ef8e5e5a340588f4452631496976cf8636d4a7ecf600239fdc27615d2530bc16" +checksum = "6cb093c84e8bd9b188d4c4a8cb6579fc016968d14c99882163cd3ff402a4f155" dependencies = [ "atomic-waker", "bytes", @@ -5836,7 +6048,7 @@ dependencies = [ "futures-core", "futures-sink", "http 1.5.0", - "indexmap 2.14.1", + "indexmap 2.14.2", "slab", "tokio", "tokio-util", @@ -5867,9 +6079,9 @@ dependencies = [ [[package]] name = "handlebars" -version = "6.4.4" +version = "6.4.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "75c54236f9045c8004a77942bebc52145b4844639db934a5c70fe08617fbe61a" +checksum = "4633d16a2350341713c379d6d06a4b9e1845329386026a49ce4fd09c2f3b16f6" dependencies = [ "derive_builder", "log", @@ -5887,7 +6099,7 @@ version = "0.1.0" dependencies = [ "proc-macro2", "quote", - "syn 3.0.4", + "syn 3.0.5", ] [[package]] @@ -5974,9 +6186,15 @@ checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea" [[package]] name = "hermit-abi" -version = "0.5.3" +version = "0.3.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d231dfb89cfffdbc30e7fc41579ed6066ad03abda9e567ccafae602b97ec5024" + +[[package]] +name = "hermit-abi" +version = "0.5.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e17592d60ebacc7d5e169f4663c5f84f9161cc90328abcfe8456f41e4dfcb284" +checksum = "fc0fef456e4baa96da950455cd02c081ca953b141298e41db3fc7e36b1da849c" [[package]] name = "hex" @@ -6154,9 +6372,9 @@ dependencies = [ [[package]] name = "http-body-util" -version = "0.1.5" +version = "0.1.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "23169fe34a5fbcdd3f3862e78fb9b6fccd5f02a6dc6f732547005d45631ce71c" +checksum = "e9f41fd6a08e4d4ec69df65976da761afd5ad5e58a9d4acb46bd1c953a9e3ff2" dependencies = [ "bytes", "futures-core", @@ -6241,9 +6459,9 @@ dependencies = [ [[package]] name = "hybrid-array" -version = "0.4.14" +version = "0.4.13" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "707114b52a152fa7bdb290cd7cd5912d9467273b6d74e21b8d81aca1f8533f6b" +checksum = "818356c5132c1fede50f837ca96afbe78ff42413047f4abb886217845e1b6c8c" dependencies = [ "typenum", ] @@ -6258,7 +6476,7 @@ dependencies = [ "bytes", "futures-channel", "futures-core", - "h2 0.4.19", + "h2 0.4.15", "http 1.5.0", "http-body 1.1.0", "httparse", @@ -6295,7 +6513,7 @@ dependencies = [ "hyper-util", "log", "rustls", - "rustls-native-certs", + "rustls-native-certs 0.8.4", "tokio", "tokio-rustls", "tower-service", @@ -6332,7 +6550,7 @@ dependencies = [ "libc", "percent-encoding", "pin-project-lite", - "socket2", + "socket2 0.6.5", "system-configuration", "tokio", "tower-service", @@ -6478,9 +6696,9 @@ dependencies = [ [[package]] name = "icu_collections" -version = "2.3.0" +version = "2.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "fa68d21081c4a05d5a901a1c62add574c77048b6a1c67be3b50ce0b60d4ca513" +checksum = "2984d1cd16c883d7935b9e07e44071dca8d917fd52ecc02c04d5fa0b5a3f191c" dependencies = [ "displaydoc", "potential_utf", @@ -6492,9 +6710,9 @@ dependencies = [ [[package]] name = "icu_locale_core" -version = "2.3.0" +version = "2.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d56e28588da92eee5c3201a6eff33fabdd49b62269c8938d4ff050ce4d900deb" +checksum = "92219b62b3e2b4d88ac5119f8904c10f8f61bf7e95b640d25ba3075e6cac2c29" dependencies = [ "displaydoc", "litemap", @@ -6505,9 +6723,9 @@ dependencies = [ [[package]] name = "icu_normalizer" -version = "2.3.0" +version = "2.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "12f9cf5f235641ed274641dd81c3f28d870e276763d0797aeeab72317b1c646f" +checksum = "c56e5ee99d6e3d33bd91c5d85458b6005a22140021cc324cea84dd0e72cff3b4" dependencies = [ "icu_collections", "icu_normalizer_data", @@ -6519,17 +6737,16 @@ dependencies = [ [[package]] name = "icu_normalizer_data" -version = "2.3.0" +version = "2.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1563da1ed3e0b3bf3d74c9b85917ac9c56464d2f57242270c09c9e752f8021a0" +checksum = "da3be0ae77ea334f4da67c12f149704f19f81d1adf7c51cf482943e84a2bad38" [[package]] name = "icu_properties" -version = "2.3.0" +version = "2.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e7ca276ad3145661a65914e6daf131ca5120cd3dcee8f8f3214b8875184a148" +checksum = "bee3b67d0ea5c2cca5003417989af8996f8604e34fb9ddf96208a033901e70de" dependencies = [ - "displaydoc", "icu_collections", "icu_locale_core", "icu_properties_data", @@ -6540,15 +6757,15 @@ dependencies = [ [[package]] name = "icu_properties_data" -version = "2.3.0" +version = "2.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e590f038c1464a96894fd6d10127e90a8be4509f56ff7ecef851b15cee0b7caa" +checksum = "8e2bbb201e0c04f7b4b3e14382af113e17ba4f63e2c9d2ee626b720cbce54a14" [[package]] name = "icu_provider" -version = "2.3.1" +version = "2.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d27bbb9d3abbefac45d55f647c9de1d44aafcd1186eb91879afef17c396c3e73" +checksum = "139c4cf31c8b5f33d7e199446eff9c1e02decfc2f0eec2c8d71f65befa45b421" dependencies = [ "displaydoc", "icu_locale_core", @@ -6603,7 +6820,7 @@ dependencies = [ "bytemuck", "bytes", "dashmap", - "flume", + "flume 0.12.0", "futures", "futures-util", "iggy_binary_protocol", @@ -6709,7 +6926,7 @@ dependencies = [ "tempfile", "thiserror 2.0.20", "tokio", - "toml 1.1.4+spec-1.1.0", + "toml 1.1.5+spec-1.1.0", "tracing", "tracing-appender", "tracing-subscriber", @@ -6735,7 +6952,7 @@ dependencies = [ "dotenvy", "figlet-rs", "figment", - "flume", + "flume 0.12.0", "futures", "iggy", "iggy_common", @@ -6764,7 +6981,7 @@ dependencies = [ "tempfile", "thiserror 2.0.20", "tokio", - "toml 1.1.4+spec-1.1.0", + "toml 1.1.5+spec-1.1.0", "tower-http 0.7.1", "tracing", "tracing-opentelemetry", @@ -6780,7 +6997,7 @@ dependencies = [ "bytes", "kafka-protocol", "libc", - "socket2", + "socket2 0.6.5", "thiserror 2.0.20", "tokio", "tokio-util", @@ -6811,7 +7028,7 @@ dependencies = [ "sd-notify", "serde", "serde_json", - "socket2", + "socket2 0.6.5", "strum 0.28.0", "tempfile", "thiserror 2.0.20", @@ -6885,7 +7102,7 @@ dependencies = [ "serde_json", "simd-json", "tokio", - "toml 1.1.4+spec-1.1.0", + "toml 1.1.5+spec-1.1.0", "tracing", ] @@ -6980,7 +7197,7 @@ dependencies = [ "simd-json", "strum_macros 0.28.0", "tokio", - "toml 1.1.4+spec-1.1.0", + "toml 1.1.5+spec-1.1.0", "tracing", ] @@ -7022,7 +7239,7 @@ dependencies = [ "serde_json", "simd-json", "tokio", - "toml 1.1.4+spec-1.1.0", + "toml 1.1.5+spec-1.1.0", "tracing", ] @@ -7047,7 +7264,7 @@ dependencies = [ "serde_json", "simd-json", "tokio", - "toml 1.1.4+spec-1.1.0", + "toml 1.1.5+spec-1.1.0", "tracing", "uuid", ] @@ -7141,6 +7358,23 @@ dependencies = [ "tracing", ] +[[package]] +name = "iggy_connector_rabbitmq_sink" +version = "0.4.1-edge.1" +dependencies = [ + "async-trait", + "dashmap", + "iggy", + "iggy_common", + "iggy_connector_sdk", + "lapin", + "secrecy", + "serde", + "serde_json", + "tokio", + "tracing", +] + [[package]] name = "iggy_connector_random_source" version = "0.5.0-edge.4" @@ -7290,9 +7524,9 @@ dependencies = [ [[package]] name = "ignore" -version = "0.4.33" +version = "0.4.31" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "00b69833ed729dc5aa7d19541d96d6cf8e9137194207a04916d658e43168402f" +checksum = "7f8a7b8211e695a1d0cd91cace480d4d0bd57667ab10277cc412c5f7f4884f83" dependencies = [ "crossbeam-deque", "globset", @@ -7324,7 +7558,7 @@ dependencies = [ "rayon", "rgb", "tiff", - "zune-core 0.5.3", + "zune-core 0.5.1", "zune-jpeg 0.5.15", ] @@ -7346,15 +7580,15 @@ checksum = "edcd27d72f2f071c64249075f42e205ff93c9a4c5f6c6da53e79ed9f9832c285" [[package]] name = "imgref" -version = "1.12.3" +version = "1.12.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6e44b0a4eaa4c82f441d50a963f2d5f05a787240aeee097597033e72accfd22f" +checksum = "89194689a993ab15268672e99e7b0e19da2da3268ac682e8f02d29d4d1434cd7" [[package]] name = "impl-more" -version = "0.3.5" +version = "0.3.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "277ff51754a3f68f12f58446c5d006aa8baa4914ea273cce24a599cfaff33d4f" +checksum = "35a84fd5aa25fae5c0f4a33d9cac2ca017fc622cbd089be2229993514990f870" [[package]] name = "implicit-clone" @@ -7363,7 +7597,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1689b939ee35e3a075b0834b5672efd43aec8a6e81a1c6002b76a5ca2f211ae0" dependencies = [ "implicit-clone-derive", - "indexmap 2.14.1", + "indexmap 2.14.2", ] [[package]] @@ -7389,9 +7623,9 @@ dependencies = [ [[package]] name = "indexmap" -version = "2.14.1" +version = "2.14.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "07aa2048142242915a31d35844fb311e0e53fcca590c3a0a40dcf1b841fa09eb" +checksum = "cc4e190f5d26ca7051642629da2c52fc03bde85a03197c99408dcd291734c855" dependencies = [ "equivalent", "hashbrown 0.17.1", @@ -7413,9 +7647,9 @@ checksum = "c8fae54786f62fb2918dcfae3d568594e50eb9b5c25bf04371af6fe7516452fb" [[package]] name = "inotify" -version = "0.11.5" +version = "0.11.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4cc00ea907cab49550b7da656f80ebb97be1b997d931fbcd28d39734e17ce592" +checksum = "153be1941a183ec9ccd095ddbe17a8b8d435ef6c76e9e02451b933c3999af2c8" dependencies = [ "bitflags 2.13.1", "inotify-sys", @@ -7437,6 +7671,7 @@ version = "0.1.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "879f10e63c20629ecabbb64a8010319738c66a5cd0c29b02d63d272b03751d01" dependencies = [ + "block-padding", "generic-array", ] @@ -7446,10 +7681,18 @@ version = "0.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "4250ce6452e92010fdf7268ccc5d14faa80bb12fc741938534c58f16804e03c7" dependencies = [ - "block-padding", "hybrid-array", ] +[[package]] +name = "instant" +version = "0.1.13" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e0242819d153cba4b4b05a5a8f2a7e9bbf97b6055b2a002b395c96b5ff3c0222" +dependencies = [ + "cfg-if", +] + [[package]] name = "integer-encoding" version = "3.0.4" @@ -7489,6 +7732,7 @@ dependencies = [ "journal", "jsonwebtoken 11.0.0", "keyring-core", + "lapin", "lazy_static", "libc", "mongodb", @@ -7515,7 +7759,7 @@ dependencies = [ "testcontainers-modules", "tokio", "tokio-postgres", - "toml 1.1.4+spec-1.1.0", + "toml 1.1.5+spec-1.1.0", "tracing", "tracing-subscriber", "twox-hash", @@ -7546,11 +7790,22 @@ dependencies = [ "rustversion", ] +[[package]] +name = "io-lifetimes" +version = "1.0.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "eae7b9aee968036d54dce06cebaefd919e4472e753296daccd6d344e3e2df0c2" +dependencies = [ + "hermit-abi 0.3.9", + "libc", + "windows-sys 0.48.0", +] + [[package]] name = "io-uring" -version = "0.7.14" +version = "0.7.13" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d64d8ca234d152948ceaede1f419b6a83983a5ecccaac05fb337a809c96d3aa6" +checksum = "9080b15e63775b9a2ac7dca720f7050a8b955e092ea0f6020a4a80f69998cdc0" dependencies = [ "bitflags 2.13.1", "cfg-if", @@ -7563,7 +7818,7 @@ version = "0.3.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "4d40460c0ce33d6ce4b0630ad68ff63d6661961c48b6dba35e5a4d81cfb48222" dependencies = [ - "socket2", + "socket2 0.6.5", "widestring", "windows-registry", "windows-result 0.4.1", @@ -7572,9 +7827,9 @@ dependencies = [ [[package]] name = "ipnet" -version = "2.12.1" +version = "2.12.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6a756c3fac73139e83f14c2d742155dd2b78d3ee56597b419a0579b7bdd6dd78" +checksum = "791930b43c0d5973160d90a8f3894509f2b273430f5c5c73b668636d0287c5c0" dependencies = [ "serde", ] @@ -7620,9 +7875,9 @@ checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" [[package]] name = "jiff" -version = "0.2.35" +version = "0.2.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "668b7183bd07af9a4885f5c35b0cc5c83c4607a913c16b7e17291832910d2dcc" +checksum = "e184d09547b80eb7e20d141ba2fb1fbac843ca53f4cf1b31210adc4c1adc6e16" dependencies = [ "defmt", "jiff-core", @@ -7648,9 +7903,9 @@ dependencies = [ [[package]] name = "jiff-static" -version = "0.2.35" +version = "0.2.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3a69dcb3a21cfb32ce1cd056169337ca284af0766dd766e7878819b251a49204" +checksum = "323da076b7a6faf914dc677cb05a4b907742ff7375c8322c9e7f5061e5e0e9de" dependencies = [ "jiff-core", "proc-macro2", @@ -7748,9 +8003,9 @@ dependencies = [ [[package]] name = "js-sys" -version = "0.3.104" +version = "0.3.103" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0e0c1080212aad755ea003d18543e8768dd432c48819efd73a7bf1e39b7a5a3a" +checksum = "53b44bfcdb3f8d5837a46dae1ca9660a837176eee74a28b229bc626816589102" dependencies = [ "cfg-if", "futures-util", @@ -7771,7 +8026,7 @@ dependencies = [ "p256", "p384", "pem 3.0.6", - "rand 0.8.8", + "rand 0.8.7", "rsa", "serde", "serde_json", @@ -7795,7 +8050,7 @@ dependencies = [ "p256", "p384", "pem 3.0.6", - "rand 0.8.8", + "rand 0.8.7", "rsa", "serde", "serde_json", @@ -7824,7 +8079,7 @@ dependencies = [ "clap", "hex", "iggy-gateway-kafka", - "indexmap 2.14.1", + "indexmap 2.14.2", "kafka-protocol", "tokio", "tracing", @@ -7841,18 +8096,18 @@ dependencies = [ "bytes", "crc", "crc32c", - "indexmap 2.14.1", + "indexmap 2.14.2", "uuid", ] [[package]] name = "keccak" -version = "0.2.2" +version = "0.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d8f198d1db720e4940b5a493201d199d9f24f568f8f746bd13706243a2f71598" +checksum = "9e24a010dd405bd7ed803e5253182815b41bf2e6a80cc3bfc066658e03a198aa" dependencies = [ "cfg-if", - "cpufeatures 0.3.1", + "cpufeatures 0.3.0", ] [[package]] @@ -7866,9 +8121,9 @@ dependencies = [ [[package]] name = "kqueue" -version = "1.2.1" +version = "1.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8d763e5b24120b4ddf50de6c92308156765aabfbbccebf401da7cff2d70a41ea" +checksum = "273c0752728918e0ac4976f2b275b6fefb9ecd400585dec929419f3844cd87b5" dependencies = [ "kqueue-sys", "libc", @@ -7901,6 +8156,28 @@ version = "0.3.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d4345964bb142484797b161f473a503a434de77149dd8c7427788c6e13379388" +[[package]] +name = "lapin" +version = "2.5.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "02d2aa4725b9607915fa1a73e940710a3be6af508ce700e56897cbe8847fbb07" +dependencies = [ + "amq-protocol", + "async-global-executor-trait", + "async-reactor-trait", + "async-trait", + "executor-trait", + "flume 0.11.1", + "futures-core", + "futures-io", + "parking_lot", + "pinky-swear", + "reactor-trait", + "serde", + "tracing", + "waker-fn", +] + [[package]] name = "lazy-regex" version = "3.6.1" @@ -8031,9 +8308,9 @@ dependencies = [ [[package]] name = "libgit2-sys" -version = "0.18.8+1.9.7" +version = "0.18.5+1.9.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7f7c568b25d7489bc3fb2988ed69ab111d2944d2f5fec3d5c987fe545ea97b50" +checksum = "005d6ae6eac1912906073e069f7db60b1fa98e052a68227824afe3e3a1c59ca2" dependencies = [ "cc", "libc", @@ -8053,18 +8330,18 @@ dependencies = [ [[package]] name = "liblzma" -version = "0.4.8" +version = "0.4.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2fe0a34ca854fd4f20c07f696fc8675aec78f87d88d29f5e10257a7490a1b2e1" +checksum = "45aec2360b3933207e27908049d8e4df4e476b58180afb1e56b2a4fb72efe4ba" dependencies = [ "liblzma-sys", ] [[package]] name = "liblzma-sys" -version = "0.4.8" +version = "0.4.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a0dad045e4b1b7b170be4b60b54b780cafb4490165461bac7d1cf7b703f61d5f" +checksum = "a046c7f353ba30f810545151e04f63545833803f5b86ee3ddf1517247fe560a5" dependencies = [ "cc", "libc", @@ -8088,9 +8365,9 @@ dependencies = [ [[package]] name = "libredox" -version = "0.1.23" +version = "0.1.18" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "8d8f1ea3f21fd3405dcaf6c9b5c1630af9afc422d9073ea39c5f6d6c772e08ed" +checksum = "c943259e342f1e06ff2da7a83eabdfe7f92ce10262688dbf1895ff0b3e6e4652" dependencies = [ "libc", ] @@ -8141,6 +8418,12 @@ version = "0.2.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7e57c38c1e860fd37c604281cdfb1dd2216977fd76a50f85ba2f388ef3219616" +[[package]] +name = "linux-raw-sys" +version = "0.3.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ef53942eb7bf7ff43a617b3e2c1c4a5ecf5944a7c1bc12d7ee39bbb15e5c1519" + [[package]] name = "linux-raw-sys" version = "0.4.15" @@ -8155,9 +8438,9 @@ checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53" [[package]] name = "litemap" -version = "0.8.3" +version = "0.8.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "47d9d19d1d6efa0109d2f65ff4c85cddd50bd572e5a00127ab10987290bcefae" +checksum = "92daf443525c4cce67b150400bc2316076100ce0b3686209eb8cf3c31612e6f0" [[package]] name = "local-channel" @@ -8172,9 +8455,9 @@ dependencies = [ [[package]] name = "local-event" -version = "0.1.3" +version = "0.1.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "23ab4b951e96ffb2da6e25cec3d038d0e2930f4c406d1ec194b5120ae41ec6d3" +checksum = "76dda8459b10a8960dfae91c1c77316fc15e7caf94f9f18794126512a8dc77e5" [[package]] name = "local-waker" @@ -8193,9 +8476,9 @@ dependencies = [ [[package]] name = "log" -version = "0.4.34" +version = "0.4.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" +checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad" [[package]] name = "logos" @@ -8524,7 +8807,7 @@ dependencies = [ "rustls-pemfile", "scopeguard", "server_common", - "socket2", + "socket2 0.6.5", "tempfile", "thiserror 2.0.20", "tracing", @@ -8634,7 +8917,6 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b63fbc4a50860e98e7b2aa7804ded1db5cbc3aff9193adaff57a6931bf7c4b4c" dependencies = [ "adler2", - "simd-adler32", ] [[package]] @@ -8683,16 +8965,16 @@ checksum = "21909324aa58f5c284d91cac514c6d081210901dc372d7c8ea6a9d7e0406097a" [[package]] name = "moka" -version = "0.12.16" +version = "0.12.15" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4293f18e7567a1caf3c584855554377025c65e0aa445344d04171f5ad63d19b9" +checksum = "957228ad12042ee839f93c8f257b62b4c0ab5eaae1d4fa60de53b27c9d7c5046" dependencies = [ - "async-lock", + "async-lock 3.4.2", "crossbeam-channel", "crossbeam-epoch", "crossbeam-utils", "equivalent", - "event-listener", + "event-listener 5.4.1", "futures-util", "parking_lot", "portable-atomic", @@ -8721,9 +9003,9 @@ checksum = "851fac73f7fe22f6a3ab87f720ce509cae7c9fd08e7dd27866cc232dee07ccf4" [[package]] name = "mongodb" -version = "3.8.2" +version = "3.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d220eb9ba80bad420e1f9efc4e895fede7418f8779a8e38a3b750e57ff397720" +checksum = "fd64784a1fdbcf2a334717446edc583061159067a2e7cc326ee567ecf035f07b" dependencies = [ "base64 0.22.1", "bitflags 2.13.1", @@ -8742,7 +9024,7 @@ dependencies = [ "md-5 0.11.0", "mongocrypt", "mongodb-internal-macros", - "pbkdf2", + "pbkdf2 0.13.0", "percent-encoding", "rand 0.9.5", "rustc_version_runtime", @@ -8752,7 +9034,7 @@ dependencies = [ "serde_with", "sha1 0.11.0", "sha2 0.11.0", - "socket2", + "socket2 0.6.5", "stringprep", "strsim", "take_mut", @@ -8767,9 +9049,9 @@ dependencies = [ [[package]] name = "mongodb-internal-macros" -version = "3.8.2" +version = "3.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e38ff3c46c59c2f4d9b26a86e6980dc23583e9d4b45aa579cef6584642093128" +checksum = "f243f9039382cd06378f13980b79700236468b6fb22f24abfa99783e9569f1a3" dependencies = [ "macro_magic", "proc-macro2", @@ -8947,7 +9229,7 @@ checksum = "c89e69e7e0f03bea5ef08013795c25018e101932225a656383bd384495ecc367" dependencies = [ "num-integer", "num-traits", - "rand 0.8.8", + "rand 0.8.7", "serde", ] @@ -8962,7 +9244,7 @@ dependencies = [ "num-integer", "num-iter", "num-traits", - "rand 0.8.8", + "rand 0.8.7", "smallvec", "zeroize", ] @@ -8996,9 +9278,9 @@ dependencies = [ [[package]] name = "num-integer" -version = "0.1.47" +version = "0.1.46" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7ce2d95d4b3734dc35aa2f45e1aa22cd416814592a4f9d9205e11affd5b8e10b" +checksum = "7969661fd2958a5cb096e56c8e1ad0444ac2bbcd0061bd28660485a44879858f" dependencies = [ "num-traits", ] @@ -9015,9 +9297,9 @@ dependencies = [ [[package]] name = "num-modular" -version = "0.6.5" +version = "0.6.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bd8e500409e6cd603b03e477c26a6caecdc27ac58979a53e881c75eafc079f44" +checksum = "fc41a1374056e9672221567958a66c16be12d0e2c1b408761e14d901c237d5e0" [[package]] name = "num-order" @@ -9055,7 +9337,7 @@ version = "1.17.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "91df4bbde75afed763b708b7eee1e8e7651e02d97f6d5dd763e89367e957b23b" dependencies = [ - "hermit-abi", + "hermit-abi 0.5.2", "libc", ] @@ -9136,9 +9418,9 @@ dependencies = [ [[package]] name = "object" -version = "0.39.1" +version = "0.37.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2e5a6c098c7a3b6547378093f5cc30bc54fd361ce711e05293a5cc589562739b" +checksum = "ff76201f031d8863c38aa7f905eca4f53abbfa15f609db4277d44cd8938f33fe" dependencies = [ "memchr", ] @@ -9378,6 +9660,12 @@ dependencies = [ "url", ] +[[package]] +name = "openssl-probe" +version = "0.1.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d05e27ee213611ffe7d6348b942e8f942b37114c00cc03cec254295a4a17852e" + [[package]] name = "openssl-probe" version = "0.2.1" @@ -9560,6 +9848,28 @@ version = "4.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "13c45bb4a6ae1280ec0803b1ef9d3455eb50f01efbbe1447ab020f1d54fba9d8" +[[package]] +name = "p12-keystore" +version = "0.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3cae83056e7cb770211494a0ecf66d9fa7eba7d00977e5bb91f0e925b40b937f" +dependencies = [ + "cbc", + "cms", + "der", + "des", + "hex", + "hmac 0.12.1", + "pkcs12", + "pkcs5", + "rand 0.9.5", + "rc2", + "sha1 0.10.7", + "sha2 0.10.9", + "thiserror 2.0.20", + "x509-parser 0.17.0", +] + [[package]] name = "p256" version = "0.13.2" @@ -9747,6 +10057,16 @@ version = "0.2.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "2ee67f1008b1ba2321834326597b8e186293b049a023cdef258527550b9935b4" +[[package]] +name = "pbkdf2" +version = "0.12.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f8ed6a7761f76e3b9f92dfb0a60a6a6477c61024b775147ff0973a02653abaf2" +dependencies = [ + "digest 0.10.7", + "hmac 0.12.1", +] + [[package]] name = "pbkdf2" version = "0.13.0" @@ -9849,9 +10169,9 @@ checksum = "3637c05577168127568a64e9dc5a6887da720efef07b3d9472d45f63ab191166" [[package]] name = "pest" -version = "2.9.0" +version = "2.8.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5a07a60cc7a4d00c91f95c685609d1d2f79050e6804b70ebedd7650f0b839bcf" +checksum = "47627dd7305c6a2d6c8c6bcd24c5a4c17dbbf425f4f9c5313e724b38fc9782e9" dependencies = [ "memchr", "ucd-trie", @@ -9859,9 +10179,9 @@ dependencies = [ [[package]] name = "pest_derive" -version = "2.9.0" +version = "2.8.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b3a83744a5c8455b8b3e0dc5031362780a347c878bdd11584d1a8984228cc88d" +checksum = "4b4254325ecad416ab689e27ba51da03ba01a9632bc6e108f5fe7c3c4ad29d58" dependencies = [ "pest", "pest_generator", @@ -9869,9 +10189,9 @@ dependencies = [ [[package]] name = "pest_generator" -version = "2.9.0" +version = "2.8.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e0cd3451aa3de60d4b9a1e736885e4dea6b31617598026f12256ad566d63304a" +checksum = "6c4c0e91ead7a8f7acecbca6f003fc2e8282b1dbe2dd9c9d2f16aba42995e0a7" dependencies = [ "pest", "pest_meta", @@ -9882,9 +10202,9 @@ dependencies = [ [[package]] name = "pest_meta" -version = "2.9.0" +version = "2.8.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e04d3a0849e241d7dfce834c83b1c5edc8622009e8dd51a12ba1927c32f05496" +checksum = "f9744bc48116fee06334924bb5f2bad41eed5e89bd26e29b0b799f9a3f82c210" dependencies = [ "pest", ] @@ -10019,6 +10339,18 @@ version = "0.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8b870d8c151b6f2fb93e84a13146138f05d02ed11c7e7c54f8826aaaf7c9f184" +[[package]] +name = "pinky-swear" +version = "6.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b1ea6e230dd3a64d61bcb8b79e597d3ab6b4c94ec7a234ce687dd718b4f2e657" +dependencies = [ + "doc-comment", + "flume 0.11.1", + "parking_lot", + "tracing", +] + [[package]] name = "pinned" version = "0.1.0" @@ -10037,7 +10369,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c835479a4443ded371d6c535cbfd8d31ad92c5d23ae9770a61bc155e4992a3c1" dependencies = [ "atomic-waker", - "fastrand", + "fastrand 2.5.0", "futures-io", ] @@ -10052,6 +10384,36 @@ dependencies = [ "spki", ] +[[package]] +name = "pkcs12" +version = "0.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "695b3df3d3cc1015f12d70235e35b6b79befc5fa7a9b95b951eab1dd07c9efc2" +dependencies = [ + "cms", + "const-oid 0.9.6", + "der", + "digest 0.10.7", + "spki", + "x509-cert", + "zeroize", +] + +[[package]] +name = "pkcs5" +version = "0.7.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e847e2c91a18bfa887dd028ec33f2fe6f25db77db3619024764914affe8b69a6" +dependencies = [ + "aes 0.8.4", + "cbc", + "der", + "pbkdf2 0.12.2", + "scrypt", + "sha2 0.10.9", + "spki", +] + [[package]] name = "pkcs8" version = "0.10.2" @@ -10064,9 +10426,9 @@ dependencies = [ [[package]] name = "pkg-config" -version = "0.3.34" +version = "0.3.33" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548" +checksum = "19f132c84eca552bf34cab8ec81f1c1dcc229b811638f9d283dceabe58c5569e" [[package]] name = "png" @@ -10100,6 +10462,22 @@ version = "0.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "5dcdc93847ad24990939cce6e1804361e903efcb5f99daa5abd87943a9d6d7ba" +[[package]] +name = "polling" +version = "2.8.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4b2d323e8ca7996b3e23126511a523f7e62924d93ecd5ae73b333815b0eb3dce" +dependencies = [ + "autocfg", + "bitflags 1.3.2", + "cfg-if", + "concurrent-queue", + "libc", + "log", + "pin-project-lite", + "windows-sys 0.48.0", +] + [[package]] name = "polling" version = "3.11.0" @@ -10108,7 +10486,7 @@ checksum = "5d0e4f59085d47d8241c88ead0f274e8a0cb551f3625263c05eb8dd897c34218" dependencies = [ "cfg-if", "concurrent-queue", - "hermit-abi", + "hermit-abi 0.5.2", "pin-project-lite", "rustix 1.1.4", "windows-sys 0.61.2", @@ -10133,15 +10511,15 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f0fa31d631f2b2cb2a544d0aa321ce847a94764d701ca2becc411138b93d49cd" dependencies = [ "cpubits", - "cpufeatures 0.3.1", + "cpufeatures 0.3.0", "universal-hash 0.6.1", ] [[package]] name = "portable-atomic" -version = "1.15.0" +version = "1.14.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85" +checksum = "3d20d5497ef88037a52ff98267d066e7f11fcc5e99bbfbd58a42336193aacec3" [[package]] name = "portable-atomic-util" @@ -10200,9 +10578,9 @@ dependencies = [ [[package]] name = "potential_utf" -version = "0.1.6" +version = "0.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d83eb9bc6d8e5cf568e7a1101d60ee05e81ed50ea106026f3d18deeb046d7661" +checksum = "0103b1cef7ec0cf76490e969665504990193874ea05c85ff9bab8b911d0a0564" dependencies = [ "zerovec", ] @@ -10280,7 +10658,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "2bfe0f4c752e450fc2faf62654f1c134747922825d5b04ca717b8874f41a40c0" dependencies = [ "proc-macro2", - "syn 3.0.4", + "syn 3.0.5", ] [[package]] @@ -10441,7 +10819,7 @@ checksum = "01e34894696ff94f64a20c2c373a6440903e9c2789a303d68ec6e6f953f890e4" dependencies = [ "proc-macro2", "quote", - "syn 3.0.4", + "syn 3.0.5", ] [[package]] @@ -10517,9 +10895,9 @@ dependencies = [ [[package]] name = "psm" -version = "0.1.32" +version = "0.1.31" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4dcd034599e63b970727f70d79e02d62390a4a84f7c6b827c27c46d5ac3fa622" +checksum = "645dbe486e346d9b5de3ef16ede18c26e6c70ad97418f4874b8b1889d6e761ea" dependencies = [ "ar_archive_writer", "cc", @@ -10653,7 +11031,7 @@ dependencies = [ "quinn-udp", "rustc-hash", "rustls", - "socket2", + "socket2 0.6.5", "thiserror 2.0.20", "tokio", "tracing", @@ -10662,9 +11040,9 @@ dependencies = [ [[package]] name = "quinn-proto" -version = "0.11.17" +version = "0.11.16" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "04759210543be93709136e28212294a659ef5001836ff4eab4d663e4529bba83" +checksum = "2f4bfc015262b9df63c8845072ce59068853ff5872180c2ce2f13038b970e560" dependencies = [ "aws-lc-rs", "bytes", @@ -10694,7 +11072,7 @@ dependencies = [ "cfg_aliases", "libc", "once_cell", - "socket2", + "socket2 0.6.5", "tracing", "windows-sys 0.61.2", ] @@ -10728,9 +11106,9 @@ checksum = "dc33ff2d4973d518d823d61aa239014831e521c75da58e3df4840d3f47749d09" [[package]] name = "rand" -version = "0.8.8" +version = "0.8.7" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e058c7de0b26af77780c769414d6257830bb240f3c38477dbc2c16e5f54d6d4c" +checksum = "22f6172bdec972074665ed81ed53b71da00bfc44b65a753cfde883ec4c702a1a" dependencies = [ "libc", "rand_chacha 0.3.1", @@ -10899,6 +11277,15 @@ dependencies = [ "crossbeam-utils", ] +[[package]] +name = "rc2" +version = "0.8.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "62c64daa8e9438b84aaae55010a93f396f8e60e3911590fcba770d04643fc1dd" +dependencies = [ + "cipher 0.4.4", +] + [[package]] name = "rcgen" version = "0.14.10" @@ -10909,10 +11296,21 @@ dependencies = [ "ring", "rustls-pki-types", "time", - "x509-parser", + "x509-parser 0.18.1", "yasna", ] +[[package]] +name = "reactor-trait" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "438a4293e4d097556730f4711998189416232f009c137389e0f961d2bc0ddc58" +dependencies = [ + "async-trait", + "futures-core", + "futures-io", +] + [[package]] name = "reborrow" version = "0.5.5" @@ -10961,22 +11359,22 @@ dependencies = [ [[package]] name = "ref-cast" -version = "1.0.27" +version = "1.0.26" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7e440fb4e4b4147295338efb76001ab9e4efc0e5839df2c47fc5ac2381d365c3" +checksum = "216e8f773d7923bcba9ceb86a86c93cabb3903a11872fc3f138c49630e50b96d" dependencies = [ "ref-cast-impl", ] [[package]] name = "ref-cast-impl" -version = "1.0.27" +version = "1.0.26" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "92ecd8964f8453721699a1ed72037b0db49ce2f5a5138486ee89bed6f67cdf3a" +checksum = "2c9283685feec7d69af75fb0e858d5e7378f33fe4fc699383b2916ab9273e03c" dependencies = [ "proc-macro2", "quote", - "syn 3.0.4", + "syn 3.0.5", ] [[package]] @@ -10993,9 +11391,9 @@ dependencies = [ [[package]] name = "regex-automata" -version = "0.4.18" +version = "0.4.16" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ad8553b9b26413251cbf30e620595c7a41b3887f03da04579c0e6b0d6a06b4b2" +checksum = "8fcfdb36bda0c880c5931cdc7a2bcdc8ba4556847b9d912bca70bc94708711ad" dependencies = [ "aho-corasick", "memchr", @@ -11102,7 +11500,7 @@ dependencies = [ "bytes", "futures-core", "futures-util", - "h2 0.4.19", + "h2 0.4.15", "http 1.5.0", "http-body 1.1.0", "http-body-util", @@ -11115,7 +11513,7 @@ dependencies = [ "pin-project-lite", "quinn", "rustls", - "rustls-native-certs", + "rustls-native-certs 0.8.4", "rustls-pki-types", "serde", "serde_json", @@ -11147,7 +11545,7 @@ dependencies = [ "futures-channel", "futures-core", "futures-util", - "h2 0.4.19", + "h2 0.4.15", "http 1.5.0", "http-body 1.1.0", "http-body-util", @@ -11346,13 +11744,13 @@ dependencies = [ "http 1.5.0", "http-body 1.1.0", "http-body-util", - "indexmap 2.14.1", + "indexmap 2.14.2", "pastey 0.2.3", "pin-project-lite", "rand 0.10.2", "reqwest 0.13.4", "rmcp-macros", - "schemars 1.2.2", + "schemars 1.2.1", "serde", "serde_json", "sse-stream", @@ -11375,7 +11773,7 @@ dependencies = [ "proc-macro2", "quote", "serde_json", - "syn 3.0.4", + "syn 3.0.5", ] [[package]] @@ -11399,9 +11797,9 @@ dependencies = [ [[package]] name = "roaring" -version = "0.11.5" +version = "0.11.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "18bd8a37d17a58532776dcdf6041ce64929adca78e8489d5cacbafe99229d3e1" +checksum = "1dedc5658c6ecb3bdb5ef5f3295bb9253f42dcf3fd1402c03f6b1f7659c3c4a9" dependencies = [ "bytemuck", "byteorder", @@ -11539,7 +11937,7 @@ dependencies = [ "bytes", "num-traits", "postgres-types", - "rand 0.8.8", + "rand 0.8.7", "rkyv", "serde", "serde_json", @@ -11580,6 +11978,20 @@ dependencies = [ "nom 7.1.3", ] +[[package]] +name = "rustix" +version = "0.37.28" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "519165d378b97752ca44bbe15047d5d3409e875f39327546b42ac81d7e18c1b6" +dependencies = [ + "bitflags 1.3.2", + "errno", + "io-lifetimes", + "libc", + "linux-raw-sys 0.3.8", + "windows-sys 0.48.0", +] + [[package]] name = "rustix" version = "0.38.44" @@ -11608,9 +12020,9 @@ dependencies = [ [[package]] name = "rustls" -version = "0.23.43" +version = "0.23.44" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0283386ce02abc0151e1761d08802dfe86c173b0b494af5cbc086574e453da06" +checksum = "6725596c3f2c3a0aef021139e145d4eafe314a6623e4680ca83852b2c67ab2ba" dependencies = [ "aws-lc-rs", "log", @@ -11622,16 +12034,42 @@ dependencies = [ "zeroize", ] +[[package]] +name = "rustls-connector" +version = "0.20.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "70cc376c6ba1823ae229bacf8ad93c136d93524eab0e4e5e0e4f96b9c4e5b212" +dependencies = [ + "log", + "rustls", + "rustls-native-certs 0.7.3", + "rustls-pki-types", + "rustls-webpki", +] + +[[package]] +name = "rustls-native-certs" +version = "0.7.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e5bfb394eeed242e909609f56089eecfe5fda225042e8b171791b9c95f5931e5" +dependencies = [ + "openssl-probe 0.1.6", + "rustls-pemfile", + "rustls-pki-types", + "schannel", + "security-framework 2.11.1", +] + [[package]] name = "rustls-native-certs" version = "0.8.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "dab5152771c58876a2146916e53e35057e1a4dfa2b9df0f0305b07f611fdea4d" dependencies = [ - "openssl-probe", + "openssl-probe 0.2.1", "rustls-pki-types", "schannel", - "security-framework", + "security-framework 3.7.0", ] [[package]] @@ -11645,9 +12083,9 @@ dependencies = [ [[package]] name = "rustls-pki-types" -version = "1.15.1" +version = "1.15.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2f4925028c7eb5d1fcdaf196971378ed9d2c1c4efc7dc5d011256f76c99c0a96" +checksum = "764899a24af3980067ee14bc143654f297b22eaebfe3c7b6b211920a5a59b046" dependencies = [ "web-time", "zeroize", @@ -11665,10 +12103,10 @@ dependencies = [ "log", "once_cell", "rustls", - "rustls-native-certs", + "rustls-native-certs 0.8.4", "rustls-platform-verifier-android", "rustls-webpki", - "security-framework", + "security-framework 3.7.0", "security-framework-sys", "webpki-root-certs", "windows-sys 0.61.2", @@ -11741,6 +12179,15 @@ version = "1.0.23" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9774ba4a74de5f7b1c1451ed6cd5285a32eddb5cccb8cc655a4e50009e06477f" +[[package]] +name = "salsa20" +version = "0.10.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "97a22f5af31f73a954c10289c93e8a50cc23d971e80ee446f1f6f7137a088213" +dependencies = [ + "cipher 0.4.4", +] + [[package]] name = "same-file" version = "1.0.6" @@ -11773,9 +12220,9 @@ dependencies = [ [[package]] name = "schemars" -version = "1.2.2" +version = "1.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "687274d293b6cdc6e73e0fee520bf2049650090d7164f87672d212a3c530cf4a" +checksum = "a2b42f36aa1cd011945615b92222f6bf73c599a102a300334cd7f8dbeec726cc" dependencies = [ "chrono", "dyn-clone", @@ -11787,14 +12234,14 @@ dependencies = [ [[package]] name = "schemars_derive" -version = "1.2.2" +version = "1.2.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d98c67716b46af2f0b8cf752abc930f6f9aecfbf671ecfb531db8a31dbe4e2ba" +checksum = "7d115b50f4aaeea07e79c1912f645c7513d81715d0420f8bc77a18c6260b307f" dependencies = [ "proc-macro2", "quote", "serde_derive_internals", - "syn 3.0.4", + "syn 2.0.119", ] [[package]] @@ -11809,6 +12256,17 @@ version = "1.2.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49" +[[package]] +name = "scrypt" +version = "0.11.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0516a385866c09368f0b5bcd1caff3366aace790fcd46e2bb032697bb172fd1f" +dependencies = [ + "pbkdf2 0.12.2", + "salsa20", + "sha2 0.10.9", +] + [[package]] name = "sd-notify" version = "0.5.0" @@ -11861,23 +12319,36 @@ dependencies = [ [[package]] name = "secret-service" -version = "5.2.0" +version = "5.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5107b24b91445dd2aa449a258a1807b63240942157292354dc5bfdbeb8bc6db8" +checksum = "9a62d7f86047af0077255a29494136b9aaaf697c76ff70b8e49cded4e2623c14" dependencies = [ - "aes 0.9.3", + "aes 0.8.4", "cbc", "futures-util", - "getrandom 0.4.3", - "hkdf 0.13.0", - "hybrid-array", + "generic-array", + "getrandom 0.2.17", + "hkdf 0.12.4", "num", "once_cell", "serde", - "sha2 0.11.0", + "sha2 0.10.9", "zbus", ] +[[package]] +name = "security-framework" +version = "2.11.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "897b2245f0b511c87893af39b033e5ca9cce68824c4d7e7630b5a1d339658d02" +dependencies = [ + "bitflags 2.13.1", + "core-foundation 0.9.4", + "core-foundation-sys", + "libc", + "security-framework-sys", +] + [[package]] name = "security-framework" version = "3.7.0" @@ -11993,18 +12464,18 @@ checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" dependencies = [ "proc-macro2", "quote", - "syn 3.0.4", + "syn 3.0.5", ] [[package]] name = "serde_derive_internals" -version = "0.30.0" +version = "0.29.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f852137cce035d6a4df67ccce505ff6b3e9fd3a10e3e52b24dc71e650bb1a9bd" +checksum = "18d26a20a969b9e3fdf2fc2d9f21eda6c40e2de84c9408bb5d3b05d499aae711" dependencies = [ "proc-macro2", "quote", - "syn 3.0.4", + "syn 2.0.119", ] [[package]] @@ -12013,7 +12484,7 @@ version = "1.0.151" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" dependencies = [ - "indexmap 2.14.1", + "indexmap 2.14.2", "itoa", "memchr", "serde", @@ -12040,7 +12511,7 @@ checksum = "8d3b1629de253c70a0508c3899572da79ca359fdab27c7920ff00406df418906" dependencies = [ "proc-macro2", "quote", - "syn 3.0.4", + "syn 3.0.5", ] [[package]] @@ -12098,10 +12569,10 @@ dependencies = [ "chrono", "hex", "indexmap 1.9.3", - "indexmap 2.14.1", + "indexmap 2.14.2", "jiff", "schemars 0.9.0", - "schemars 1.2.2", + "schemars 1.2.1", "serde_core", "serde_json", "serde_with_macros", @@ -12126,7 +12597,7 @@ version = "0.10.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7b4db627b98b36d4203a7b458cf3573730f2bb591b28871d916dfa9efabfd41f" dependencies = [ - "indexmap 2.14.1", + "indexmap 2.14.2", "itoa", "ryu", "serde", @@ -12155,7 +12626,7 @@ checksum = "a22144e767da4ddd8416dbf383700542ffd8a5dc493dfecedfe1fe3ad03c98ae" dependencies = [ "proc-macro2", "quote", - "syn 3.0.4", + "syn 3.0.5", ] [[package]] @@ -12229,15 +12700,15 @@ dependencies = [ "serde_json", "server_common", "shard", - "socket2", + "socket2 0.6.5", "strum 0.28.0", - "syn 3.0.4", + "syn 3.0.5", "sysinfo 0.39.6", "system_stats", "tempfile", "thiserror 2.0.20", "tokio", - "toml 1.1.4+spec-1.1.0", + "toml 1.1.5+spec-1.1.0", "tower-http 0.7.1", "tracing", "tracing-appender", @@ -12306,7 +12777,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "aacc4cc499359472b4abe1bf11d0b12e688af9a805fa5e3016f9a386dc2d0214" dependencies = [ "cfg-if", - "cpufeatures 0.3.1", + "cpufeatures 0.3.0", "digest 0.11.3", ] @@ -12328,7 +12799,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "446ba717509524cb3f22f17ecc096f10f4822d76ab5c0b9822c5f9c284e825f4" dependencies = [ "cfg-if", - "cpufeatures 0.3.1", + "cpufeatures 0.3.0", "digest 0.11.3", ] @@ -12486,7 +12957,7 @@ dependencies = [ "futures", "iggy_binary_protocol", "iggy_common", - "indexmap 2.14.1", + "indexmap 2.14.2", "journal", "message_bus", "metadata", @@ -12587,6 +13058,16 @@ version = "1.1.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "199905e6153d6405f9728fe44daace35f8f837bbf830bb6e85fbd5828709a886" +[[package]] +name = "socket2" +version = "0.4.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9f7916fc008ca5542385b89a3d3ce689953c143e9304a9bf8beec1de48994c0d" +dependencies = [ + "libc", + "winapi", +] + [[package]] name = "socket2" version = "0.6.5" @@ -12707,14 +13188,14 @@ dependencies = [ "crc", "crossbeam-queue", "either", - "event-listener", + "event-listener 5.4.1", "futures-core", "futures-intrusive", "futures-io", "futures-util", "hashbrown 0.16.1", "hashlink", - "indexmap 2.14.1", + "indexmap 2.14.2", "log", "memchr", "percent-encoding", @@ -12844,7 +13325,7 @@ checksum = "488e99c397a62007e4229aec669a179816339afc6d2620ca6fa420dbee2e982c" dependencies = [ "atoi", "chrono", - "flume", + "flume 0.12.0", "form_urlencoded", "futures-channel", "futures-core", @@ -12864,9 +13345,9 @@ dependencies = [ [[package]] name = "sse-stream" -version = "0.2.5" +version = "0.2.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c123f296ade4ec4b8b0f6162116e6629f5146922ca5ab40ca9d3c2e73ab4761e" +checksum = "39f24a9b78c40b90817bbcd1821c74ddfd74916aadd29403d001532a9195532d" dependencies = [ "bytes", "futures-util", @@ -12883,9 +13364,9 @@ checksum = "6ce2be8dc25455e1f91df71bfa12ad37d7af1092ae736f3a6cd0e37bc7810596" [[package]] name = "stacker" -version = "0.1.25" +version = "0.1.24" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "707f49d46706bacf8a2b00d51dace3f9de527c13eec3778f570c411f89e69967" +checksum = "640c8cdd92b6b12f5bcb1803ca3bbf5ab96e5e6b6b96b9ab77dabe9e880b3190" dependencies = [ "cc", "cfg-if", @@ -13057,9 +13538,9 @@ dependencies = [ [[package]] name = "syn" -version = "3.0.4" +version = "3.0.5" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e6275cddf4610d1775e6d1fe9469b2e77d0f39fd98fb7450901b821e0c53649f" +checksum = "12df2e0110f65b775f769bb17ef989067a1d931b2eb822bd4346631eeada89f9" dependencies = [ "proc-macro2", "quote", @@ -13081,7 +13562,7 @@ version = "0.1.9" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f9d6d5fbc4583cf3e5eee953506f13853a42ee6b4a21a983dffffb10c765cb80" dependencies = [ - "event-listener", + "event-listener 5.4.1", "futures-util", "local-event", "loom", @@ -13239,13 +13720,25 @@ dependencies = [ "xattr", ] +[[package]] +name = "tcp-stream" +version = "0.28.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "495b0abdce3dc1f8fd27240651c9e68890c14e9d9c61527b1ce44d8a5a7bd3d5" +dependencies = [ + "cfg-if", + "p12-keystore", + "rustls-connector", + "rustls-pemfile", +] + [[package]] name = "tempfile" version = "3.27.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd" dependencies = [ - "fastrand", + "fastrand 2.5.0", "getrandom 0.4.3", "once_cell", "rustix 1.1.4", @@ -13399,7 +13892,7 @@ checksum = "bc04cd3e1236dd4a98afca4569f2deb3f120e5422a4023be2cb683f8486292af" dependencies = [ "proc-macro2", "quote", - "syn 3.0.4", + "syn 3.0.5", ] [[package]] @@ -13438,9 +13931,9 @@ dependencies = [ [[package]] name = "time" -version = "0.3.55" +version = "0.3.54" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cdb87b95ec50ddfa440816d227a17b2ccbdda963a316a727fda0fc4334f7d134" +checksum = "3e1d5e639ff6bab73cb6885cc7e7b1de96c3f32c68ec55f3952614bec1092244" dependencies = [ "deranged", "libc", @@ -13505,9 +13998,9 @@ dependencies = [ [[package]] name = "tinystr" -version = "0.8.4" +version = "0.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b1e27c91459209c2986af3dcf603a5a74a4368754ce37414f59acc971167f643" +checksum = "c8323304221c2a851516f22236c5722a72eaa19749016521d6dff0824447d96d" dependencies = [ "displaydoc", "zerovec", @@ -13540,20 +14033,20 @@ dependencies = [ "parking_lot", "pin-project-lite", "signal-hook-registry", - "socket2", + "socket2 0.6.5", "tokio-macros", "windows-sys 0.61.2", ] [[package]] name = "tokio-macros" -version = "2.7.2" +version = "2.7.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "78773a2a397f451582ce068015985c33193cf6dea8b74d2a639fe457b2f07b0e" +checksum = "6328af13490e73a9b4694030fafd93f8c8c6a9dede33e821c3fc63eddf8042ba" dependencies = [ "proc-macro2", "quote", - "syn 3.0.4", + "syn 2.0.119", ] [[package]] @@ -13576,7 +14069,7 @@ dependencies = [ "postgres-protocol", "postgres-types", "rand 0.10.2", - "socket2", + "socket2 0.6.5", "tokio", "tokio-util", "whoami", @@ -13594,9 +14087,9 @@ dependencies = [ [[package]] name = "tokio-stream" -version = "0.1.19" +version = "0.1.18" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a3d06f0b082ba57c26b79407372e57cf2a1e28124f78e9479fe80322cf53420b" +checksum = "32da49809aab5c3bc678af03902d4ccddea2a87d028d86392a4b1560c6906c70" dependencies = [ "futures-core", "pin-project-lite", @@ -13666,11 +14159,11 @@ dependencies = [ [[package]] name = "toml" -version = "1.1.4+spec-1.1.0" +version = "1.1.5+spec-1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3aace63f4bbcdfc2c965b059de67119c89c4017a70d633be6c104910f67056f5" +checksum = "12c0ba9680044b4ce98d391a62094047eada0d64860b80166c39f4a6b5640785" dependencies = [ - "indexmap 2.14.1", + "indexmap 2.14.2", "serde_core", "serde_spanned 1.1.1", "toml_datetime 1.1.1+spec-1.1.0", @@ -13703,7 +14196,7 @@ version = "0.19.15" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1b5bb770da30e5cbfde35a2d7b9b8a2c4b8ef89548a7a6aeab5c9a576e3e7421" dependencies = [ - "indexmap 2.14.1", + "indexmap 2.14.2", "toml_datetime 0.6.11", "winnow 0.5.40", ] @@ -13714,7 +14207,7 @@ version = "0.22.27" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "41fe8c660ae4257887cf66394862d21dbca4a6ddd26f04a3560410406a2f819a" dependencies = [ - "indexmap 2.14.1", + "indexmap 2.14.2", "serde", "serde_spanned 0.6.9", "toml_datetime 0.6.11", @@ -13753,7 +14246,7 @@ dependencies = [ "axum", "base64 0.22.1", "bytes", - "h2 0.4.19", + "h2 0.4.15", "http 1.5.0", "http-body 1.1.0", "http-body-util", @@ -13762,7 +14255,7 @@ dependencies = [ "hyper-util", "percent-encoding", "pin-project", - "socket2", + "socket2 0.6.5", "sync_wrapper", "tokio", "tokio-stream", @@ -13815,7 +14308,7 @@ checksum = "ebe5ef63511595f1344e2d5cfa636d973292adc0eec1f0ad45fae9f0851ab1d4" dependencies = [ "futures-core", "futures-util", - "indexmap 2.14.1", + "indexmap 2.14.2", "pin-project-lite", "slab", "sync_wrapper", @@ -14153,7 +14646,7 @@ checksum = "f153acc4e99a5f2a5aefa09fb078be54e26271b2813f6041200b224c098d8328" dependencies = [ "proc-macro2", "quote", - "syn 3.0.4", + "syn 3.0.5", ] [[package]] @@ -14291,7 +14784,7 @@ version = "0.5.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "fc1de2c688dc15305988b563c3854064043356019f97a4b46276fe734c4f07ea" dependencies = [ - "crypto-common 0.1.6", + "crypto-common 0.1.7", "subtle", ] @@ -14325,11 +14818,11 @@ checksum = "8ecb6da28b8a351d773b68d5825ac39017e680750f980f3a1a85cd8dd28a47c1" [[package]] name = "ureq" -version = "3.4.0" +version = "3.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "972d7902c8735f2695410b8aed7df6ed12a47394aa1c8d7af49f0497b731a94d" +checksum = "dea7109cdcd5864d4eeb1b58a1648dc9bf520360d7af16ec26d0a9354bafcfc0" dependencies = [ - "base64 0.23.1", + "base64 0.22.1", "flate2", "log", "percent-encoding", @@ -14342,11 +14835,11 @@ dependencies = [ [[package]] name = "ureq-proto" -version = "0.6.1" +version = "0.6.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "da5f78b09e6941e1a0f2e30e695e4b120377b54d5e0aec11b594bb57b3971613" +checksum = "e994ba84b0bd1b1b0cf92878b7ef898a5c1760108fe7b6010327e274917a808c" dependencies = [ - "base64 0.23.1", + "base64 0.22.1", "http 1.5.0", "httparse", "log", @@ -14600,6 +15093,12 @@ dependencies = [ "libc", ] +[[package]] +name = "waker-fn" +version = "1.2.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "317211a0dc0ceedd78fb2ca9a44aed3d7b9b26f81870d485c07122b4350673b7" + [[package]] name = "walkdir" version = "2.5.0" @@ -14654,9 +15153,9 @@ dependencies = [ [[package]] name = "wasm-bindgen" -version = "0.2.127" +version = "0.2.126" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1b70935747edd64d89de3efa29d73789b806c15798f8e7dca4d8ac356b50ce70" +checksum = "4b067c0c11094aef6b7a801c1e34a26affafdf3d051dba08456b868789aaf9a4" dependencies = [ "cfg-if", "once_cell", @@ -14668,9 +15167,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-futures" -version = "0.4.77" +version = "0.4.76" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6b7777d5cc23d0e91404e53ce2d5e8ec7acae3026b16233dba62cd3246457950" +checksum = "c62df1340f32221cb9c54d6a27b030e3dba64361d4a95bed55f9aacb44da291d" dependencies = [ "js-sys", "wasm-bindgen", @@ -14678,9 +15177,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-macro" -version = "0.2.127" +version = "0.2.126" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "77775f8f3f7217702089053b94958f8f54061a3f663417df76e19cbdcca29bc1" +checksum = "167ce5e579f6bcf889c4f7175a8a5a585de84e8ff93976ce393efa5f2837aab1" dependencies = [ "quote", "wasm-bindgen-macro-support", @@ -14688,9 +15187,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-macro-support" -version = "0.2.127" +version = "0.2.126" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e11d33f857dc2fb11b8bc75aee111aa9cbeb12cd9f25efd3d4c2a3dd4e235284" +checksum = "f3997c7839262f4ef12cf90b818d6340c18e80f263f1a94bf157d0ec4420380e" dependencies = [ "bumpalo", "proc-macro2", @@ -14701,9 +15200,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-shared" -version = "0.2.127" +version = "0.2.126" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "7ef64dbcc55df09c7e5a46182d181c2cfa3e925f3da937ea764728b4bbb9dcbf" +checksum = "dc1b4cb0cc549fcf58d7dfc081778139b3d283a081644e833e84682ad71cea24" dependencies = [ "unicode-ident", ] @@ -14760,9 +15259,9 @@ dependencies = [ [[package]] name = "web-sys" -version = "0.3.104" +version = "0.3.103" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c435338968042f4f59a557f690a253676d47ce13ceb55d70100e7facf6620a30" +checksum = "8622dcb61c0bcc9fffa6938bed81210af2da9a7e4a1a834b2e37a59b6dfb6141" dependencies = [ "js-sys", "wasm-bindgen", @@ -14826,9 +15325,9 @@ dependencies = [ [[package]] name = "whoami" -version = "2.1.3" +version = "2.1.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "626c4bac6755d76ffc12cb01b2eac751db1996b9e0041de9aa02c8c211ddc82c" +checksum = "998767ef88740d1f5b0682a9c53c24431453923962269c2db68ee43788c5a40d" dependencies = [ "libc", "libredox", @@ -15079,6 +15578,15 @@ dependencies = [ "windows-link 0.2.1", ] +[[package]] +name = "windows-sys" +version = "0.48.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "677d2418bec65e3338edb076e806bc1ec15693c5d0104683f2efe857f61056a9" +dependencies = [ + "windows-targets 0.48.5", +] + [[package]] name = "windows-sys" version = "0.52.0" @@ -15115,6 +15623,21 @@ dependencies = [ "windows-link 0.2.1", ] +[[package]] +name = "windows-targets" +version = "0.48.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9a2fa6e2155d7247be68c096456083145c183cbbbc2764150dda45a87197940c" +dependencies = [ + "windows_aarch64_gnullvm 0.48.5", + "windows_aarch64_msvc 0.48.5", + "windows_i686_gnu 0.48.5", + "windows_i686_msvc 0.48.5", + "windows_x86_64_gnu 0.48.5", + "windows_x86_64_gnullvm 0.48.5", + "windows_x86_64_msvc 0.48.5", +] + [[package]] name = "windows-targets" version = "0.52.6" @@ -15166,6 +15689,12 @@ dependencies = [ "windows-link 0.2.1", ] +[[package]] +name = "windows_aarch64_gnullvm" +version = "0.48.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2b38e32f0abccf9987a4e3079dfb67dcd799fb61361e53e2882c3cbaf0d905d8" + [[package]] name = "windows_aarch64_gnullvm" version = "0.52.6" @@ -15178,6 +15707,12 @@ version = "0.53.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a9d8416fa8b42f5c947f8482c43e7d89e73a173cead56d044f6a56104a6d1b53" +[[package]] +name = "windows_aarch64_msvc" +version = "0.48.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "dc35310971f3b2dbbf3f0690a219f40e2d9afcf64f9ab7cc1be722937c26b4bc" + [[package]] name = "windows_aarch64_msvc" version = "0.52.6" @@ -15190,6 +15725,12 @@ version = "0.53.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b9d782e804c2f632e395708e99a94275910eb9100b2114651e04744e9b125006" +[[package]] +name = "windows_i686_gnu" +version = "0.48.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a75915e7def60c94dcef72200b9a8e58e5091744960da64ec734a6c6e9b3743e" + [[package]] name = "windows_i686_gnu" version = "0.52.6" @@ -15214,6 +15755,12 @@ version = "0.53.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "fa7359d10048f68ab8b09fa71c3daccfb0e9b559aed648a8f95469c27057180c" +[[package]] +name = "windows_i686_msvc" +version = "0.48.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f55c233f70c4b27f66c523580f78f1004e8b5a8b659e05a4eb49d4166cca406" + [[package]] name = "windows_i686_msvc" version = "0.52.6" @@ -15226,6 +15773,12 @@ version = "0.53.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1e7ac75179f18232fe9c285163565a57ef8d3c89254a30685b57d83a38d326c2" +[[package]] +name = "windows_x86_64_gnu" +version = "0.48.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "53d40abd2583d23e4718fddf1ebec84dbff8381c07cae67ff7768bbf19c6718e" + [[package]] name = "windows_x86_64_gnu" version = "0.52.6" @@ -15238,6 +15791,12 @@ version = "0.53.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9c3842cdd74a865a8066ab39c8a7a473c0778a3f29370b5fd6b4b9aa7df4a499" +[[package]] +name = "windows_x86_64_gnullvm" +version = "0.48.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0b7b52767868a23d5bab768e390dc5f5c55825b6d30b86c844ff2dc7414044cc" + [[package]] name = "windows_x86_64_gnullvm" version = "0.52.6" @@ -15250,6 +15809,12 @@ version = "0.53.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0ffa179e2d07eee8ad8f57493436566c7cc30ac536a3379fdf008f47f6bb7ae1" +[[package]] +name = "windows_x86_64_msvc" +version = "0.48.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ed94fce61571a4006852b7389a063ab983c02eb1bb37b47f8272ce92d06d9538" + [[package]] name = "windows_x86_64_msvc" version = "0.52.6" @@ -15326,9 +15891,9 @@ checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" [[package]] name = "writeable" -version = "0.6.4" +version = "0.6.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3ad82d2a33cdc9674dc7465672f271e096168fcdbe0f799d9e6db8c5892679dc" +checksum = "1ffae5123b2d3fc086436f8834ae3ab053a283cfac8fe0a0b8eaae044768a4c4" [[package]] name = "wyz" @@ -15339,6 +15904,17 @@ dependencies = [ "tap", ] +[[package]] +name = "x509-cert" +version = "0.2.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1301e935010a701ae5f8655edc0ad17c44bad3ac5ce8c39185f75453b720ae94" +dependencies = [ + "const-oid 0.9.6", + "der", + "spki", +] + [[package]] name = "x509-certificate" version = "0.25.0" @@ -15358,6 +15934,23 @@ dependencies = [ "zeroize", ] +[[package]] +name = "x509-parser" +version = "0.17.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4569f339c0c402346d4a75a9e39cf8dad310e287eef1ff56d4c68e5067f53460" +dependencies = [ + "asn1-rs", + "data-encoding", + "der-parser", + "lazy_static", + "nom 7.1.3", + "oid-registry", + "rusticata-macros", + "thiserror 2.0.20", + "time", +] + [[package]] name = "x509-parser" version = "0.18.1" @@ -15441,7 +16034,7 @@ dependencies = [ "futures", "gloo 0.11.0", "implicit-clone", - "indexmap 2.14.1", + "indexmap 2.14.2", "js-sys", "rustversion", "serde", @@ -15531,23 +16124,23 @@ checksum = "c6e61e59a957b7ccee15d2049f86e8bfd6f66968fcd88f018950662d9b86e675" [[package]] name = "zbus" -version = "5.19.0" +version = "5.18.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5db4be7c075cb421e4b7ee645541604239bd243ba7c357511f4ff3a74b555907" +checksum = "fe18fb60dc696039e738717b76eaea21e7a4489bbb1885020b43c94236d7e98a" dependencies = [ "async-broadcast", "async-executor", - "async-io", - "async-lock", + "async-io 2.6.0", + "async-lock 3.4.2", "async-process", "async-recursion", "async-task", "async-trait", "blocking", "enumflags2", - "event-listener", + "event-listener 5.4.1", "futures-core", - "futures-lite", + "futures-lite 2.6.1", "hex", "libc", "ordered-stream", @@ -15577,14 +16170,14 @@ dependencies = [ [[package]] name = "zbus_macros" -version = "5.19.0" +version = "5.18.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2990635d09ade6df1868f72f8cac69a876a90981e8bd3c40b1be413f8dc88f40" +checksum = "fe96480bed92df2b442a1a30df364e12d08eed03aeb061f2b8dc6afb2be91119" dependencies = [ "proc-macro-crate 3.3.0", "proc-macro2", "quote", - "syn 3.0.4", + "syn 2.0.119", "zbus_names", "zvariant", "zvariant_utils", @@ -15601,29 +16194,20 @@ dependencies = [ "zvariant", ] -[[package]] -name = "zcheapstr" -version = "1.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d1afec51604565183aeb5c54c20aeab286120d4e4460f7f76e3e8bb8c0d99473" -dependencies = [ - "serde", -] - [[package]] name = "zerocopy" -version = "0.8.56" +version = "0.8.55" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "556764e583adb45a9f8d413c2a147fa7e8d821e48e12b14fd560b607998b75eb" +checksum = "b5a105cd7b140f6eeec8acff2ea38135d3cab283ada58540f629fe51e46696eb" dependencies = [ "zerocopy-derive", ] [[package]] name = "zerocopy-derive" -version = "0.8.56" +version = "0.8.55" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f2ab42fc20575779bd240faa45f94a74256f755c0fa9e89f0ede20d91d0cdfc1" +checksum = "0fe976fb70c78cd64cccfe3a6fc142244e8a77b70959b30faf9d0ac37ee228eb" dependencies = [ "proc-macro2", "quote", @@ -15673,9 +16257,9 @@ dependencies = [ [[package]] name = "zerotrie" -version = "0.2.5" +version = "0.2.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4ea269c3bd32f0a32c321907a2ae912ba6f4649bb0fc764a15627e99a7095a3f" +checksum = "0f9152d31db0792fa83f70fb2f83148effb5c1f5b8c7686c3459e361d9bc20bf" dependencies = [ "displaydoc", "yoke", @@ -15684,9 +16268,9 @@ dependencies = [ [[package]] name = "zerovec" -version = "0.11.8" +version = "0.11.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bb0464e17806c1d976d5cba29399c7f08e516e279e2ba493f63123b5fca67dd8" +checksum = "90f911cbc359ab6af17377d242225f4d75119aec87ea711a880987b18cd7b239" dependencies = [ "yoke", "zerofrom", @@ -15695,13 +16279,13 @@ dependencies = [ [[package]] name = "zerovec-derive" -version = "0.11.6" +version = "0.11.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "34df6fc39dbd26ddc9c10e6a2984476e13acce22e64e4487636ef494369225da" +checksum = "625dc425cab0dca6dc3c3319506e6593dcb08a9f387ea3b284dbd52a92c40555" dependencies = [ "proc-macro2", "quote", - "syn 3.0.4", + "syn 2.0.119", ] [[package]] @@ -15724,7 +16308,7 @@ checksum = "2d04a6b5381502aa6087c94c669499eb1602eb9c5e8198e534de571f7154809b" dependencies = [ "crc32fast", "flate2", - "indexmap 2.14.1", + "indexmap 2.14.2", "memchr", "typed-path", "zopfli", @@ -15732,9 +16316,9 @@ dependencies = [ [[package]] name = "zlib-rs" -version = "0.6.7" +version = "0.6.6" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "34b31d188d9d685a4f9c7b46d6e36631b07058d2cfe190267adce54dc230bf12" +checksum = "b142a20ec14a91d5bc708c1dc21b080c550113d8aa77afa29635673a65dd02c5" [[package]] name = "zmij" @@ -15790,9 +16374,9 @@ checksum = "3f423a2c17029964870cfaabb1f13dfab7d092a62a29a89264f4d36990ca414a" [[package]] name = "zune-core" -version = "0.5.3" +version = "0.5.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d56377fd46368984a170bc5aac5567e52ca5da874caa60bea39fcbca78fb658b" +checksum = "cb8a0807f7c01457d0379ba880ba6322660448ddebc890ce29bb64da71fb40f9" [[package]] name = "zune-inflate" @@ -15818,46 +16402,45 @@ version = "0.5.15" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "27bc9d5b815bc103f142aa054f561d9187d191692ec7c2d1e2b4737f8dbd7296" dependencies = [ - "zune-core 0.5.3", + "zune-core 0.5.1", ] [[package]] name = "zvariant" -version = "5.15.0" +version = "5.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c1d34c27cc6cdd1f458427519dd6b8612f7b7e3f7b9a0b2355d041dda9869147" +checksum = "bee2a0bcd2a907786a456fff45aaaaf54c9ba5f50b71ae9ec1a4edd200c94911" dependencies = [ "endi", "enumflags2", "serde", "winnow 1.0.4", - "zcheapstr", "zvariant_derive", "zvariant_utils", ] [[package]] name = "zvariant_derive" -version = "5.15.0" +version = "5.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "864155e69b4352db0c7f374917bf45d1e0c8d17659c8b3dbf9795f3673f8c497" +checksum = "38a708216a18780796770bfe3f4739c7c83a3e8f789b755534bbbc06e4e23e12" dependencies = [ "proc-macro-crate 3.3.0", "proc-macro2", "quote", - "syn 3.0.4", + "syn 2.0.119", "zvariant_utils", ] [[package]] name = "zvariant_utils" -version = "4.2.0" +version = "3.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "bad0294361a320b694a328460dc73add56c306150f5cb6bfafc44446120008a3" +checksum = "90cb9383f9b45290407a1258b202d3f8f01db719eb60b4e4055c6375af4fc7c7" dependencies = [ "proc-macro2", "quote", "serde", - "syn 3.0.4", + "syn 2.0.119", "winnow 1.0.4", ] diff --git a/Cargo.toml b/Cargo.toml index 43747aae09..b84d880748 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -44,6 +44,7 @@ members = [ "core/connectors/sinks/mongodb_sink", "core/connectors/sinks/postgres_sink", "core/connectors/sinks/quickwit_sink", + "core/connectors/sinks/rabbitmq_sink", "core/connectors/sinks/redshift_sink", "core/connectors/sinks/s3_sink", "core/connectors/sinks/stdout_sink", @@ -213,6 +214,7 @@ js-sys = "0.3" jsonwebtoken = { version = "11.0.0", features = ["rust_crypto"] } kafka-protocol = { version = "0.18.0", default-features = false, features = ["broker"] } keyring-core = "1.0.0" +lapin = "2.5.1" lazy_static = "1.5.0" # Pinned exactly: `metadata::stm::stream`'s keyed stats evictions lean on # left-right draining the previous batch's `absorb_second` before the current diff --git a/core/connectors/README.md b/core/connectors/README.md index 890a263b5f..f88b3d2554 100644 --- a/core/connectors/README.md +++ b/core/connectors/README.md @@ -86,6 +86,7 @@ Each sink should have its own, custom configuration, which is passed along with - **Meilisearch Sink** - indexes messages in Meilisearch - **PostgreSQL Sink** - stores messages in PostgreSQL database tables - **Quickwit Sink** - indexes messages in Quickwit search engine +- **RabbitMQ Sink** - publishes messages to RabbitMQ exchanges via AMQP 0.9.1 - **Reshift Sink** - stores messages in Redshift warehouse tables via S3 as staging - **S3 Sink** - writes messages to Amazon S3 and S3-compatible stores (MinIO, R2, B2, DO Spaces) - **Stdout Sink** - prints messages to standard output (useful for debugging/development) diff --git a/core/connectors/runtime/example_config/connectors/rabbitmq_sink.toml b/core/connectors/runtime/example_config/connectors/rabbitmq_sink.toml new file mode 100644 index 0000000000..8d09f65ee7 --- /dev/null +++ b/core/connectors/runtime/example_config/connectors/rabbitmq_sink.toml @@ -0,0 +1,46 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +type = "sink" +key = "rabbitmq" +enabled = true +version = 0 +name = "RabbitMQ sink" +path = "target/release/libiggy_connector_rabbitmq_sink" +verbose = true + +[[streams]] +stream = "example_stream" +topics = ["example_topic"] +schema = "json" +batch_length = 100 +poll_interval = "5ms" +consumer_group = "rabbitmq_sink_connector" + +[plugin_config] +amqp_url = "amqp://guest:guest@localhost:5672" +exchange = "iggy_events" +exchange_type = "topic" +routing_key = "iggy.messages" +durable_exchange = true +delivery_mode = "persistent" +include_metadata = true +verbose_logging = false +max_retries = 3 +retry_delay_secs = 1 +max_retry_delay_secs = 5 +timeout_secs = 30 diff --git a/core/connectors/sinks/README.md b/core/connectors/sinks/README.md index 91de1fd084..1046edb4c3 100644 --- a/core/connectors/sinks/README.md +++ b/core/connectors/sinks/README.md @@ -19,6 +19,7 @@ Sink connectors are responsible for writing data from Iggy streams to external s | **s3_sink** | Writes messages to Amazon S3 and S3-compatible stores (MinIO, R2, B2, DO Spaces) | | **stdout_sink** | Prints messages to standard output (useful for debugging and development) | | **surrealdb_sink** | Writes messages into SurrealDB with deterministic record IDs for idempotent replay | +| **rabbitmq_sink** | Publishes messages to RabbitMQ exchanges via AMQP | The sink is represented by the single `Sink` trait, which defines the basic interface for all sink connectors. It provides methods for initializing the sink, writing data to external destination, and closing the sink. diff --git a/core/connectors/sinks/rabbitmq_sink/Cargo.toml b/core/connectors/sinks/rabbitmq_sink/Cargo.toml new file mode 100644 index 0000000000..2fd4dd715e --- /dev/null +++ b/core/connectors/sinks/rabbitmq_sink/Cargo.toml @@ -0,0 +1,51 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +[package] +name = "iggy_connector_rabbitmq_sink" +version = "0.4.1-edge.1" +description = "Iggy RabbitMQ sink connector for publishing stream messages to RabbitMQ exchanges via AMQP 0.9.1" +edition = "2024" +license = "Apache-2.0" +keywords = ["iggy", "messaging", "streaming", "rabbitmq", "amqp", "sink"] +categories = ["command-line-utilities", "database", "network-programming"] +homepage = "https://iggy.apache.org" +documentation = "https://iggy.apache.org/docs" +repository = "https://github.com/apache/iggy" +readme = "../../README.md" +publish = false + +[package.metadata.cargo-machete] +ignored = ["dashmap"] + +[lib] +crate-type = ["cdylib", "lib"] + +[dependencies] +async-trait = { workspace = true } +dashmap = { workspace = true } +iggy = { workspace = true } +iggy_common = { workspace = true } +iggy_connector_sdk = { workspace = true } +lapin = { workspace = true } +secrecy = { workspace = true } +serde = { workspace = true } +tokio = { workspace = true } +tracing = { workspace = true } + +[dev-dependencies] +serde_json = { workspace = true } diff --git a/core/connectors/sinks/rabbitmq_sink/README.md b/core/connectors/sinks/rabbitmq_sink/README.md new file mode 100644 index 0000000000..28a8ffd052 --- /dev/null +++ b/core/connectors/sinks/rabbitmq_sink/README.md @@ -0,0 +1,50 @@ +# RabbitMQ Sink + +The RabbitMQ sink connector publishes messages from Iggy streams to RabbitMQ exchanges via AMQP 0.9.1. + +## Configuration + +| Field | Type | Default | Description | +| --- | --- | --- | --- | +| `amqp_url` | string | `amqp://guest:guest@localhost:5672` | RabbitMQ connection URL, including credentials. Treated as a secret: never logged or serialized verbatim. | +| `exchange` | string | `iggy_events` | Exchange name to publish to. | +| `exchange_type` | string | `topic` | Exchange type: `direct`, `topic`, `fanout`, `headers`. | +| `routing_key` | string | `iggy.messages` | Routing key for published messages. | +| `durable_exchange` | bool | `true` | Declare the exchange as durable. Must match the durability of an operator-pre-created exchange, otherwise RabbitMQ closes the channel with `PRECONDITION_FAILED`. | +| `delivery_mode` | string | `persistent` | AMQP delivery mode: `persistent` (2) or `non_persistent` (1). Persistent messages survive broker restarts when the exchange and queue are durable; it forces an fsync on publish, so non-durable topologies may prefer `non_persistent`. | +| `include_metadata` | bool | `true` | Add `iggy_stream`, `iggy_topic`, `iggy_partition_id`, `iggy_offset` message headers. User-supplied headers are always preserved, regardless of this flag. | +| `verbose_logging` | bool | `false` | Log each published batch at `info` level instead of `debug`. | +| `max_retries` | u32 | `3` | Maximum transient publish retries before failing the batch. | +| `retry_delay_secs` | u64 | `1` | Base retry delay in seconds. | +| `max_retry_delay_secs` | u64 | `5` | Upper bound for exponential backoff. | +| `timeout_secs` | u64 | `30` | Timeout for connection, publish, and publisher-confirmation operations. | + +User headers on consumed Iggy messages are forwarded as AMQP headers: string values become AMQP `LongString`, raw binary values become `ByteArray`. This allows routing through a `headers` exchange on original user headers. + +Publishes are confirmed via `ConfirmSelect`. With `mandatory = true`, a message with a routing key that matches no binding is returned by RabbitMQ and the batch fails with a permanent error. + +**Delivery guarantee:** in-request retry only. Transient failures (connection loss, `Nack`, channel exceptions, +timeouts) are retried within a single `consume()` call up to `max_retries`, resuming at the first unconfirmed +message. Publishes are pipelined (all messages sent, then confirmed in order), so a failure mid-batch can already +have delivered the in-flight tail; resuming re-publishes those, so delivery is **at-least-once within a batch**. +Because lapin does not attribute a `Basic.Return` to a specific in-flight publish, any returned message fails the +whole batch and the connection is re-established before the next poll, so a channel with unresolved confirms is +never reused. The connectors runtime commits the consumer offset at poll time and discards `consume()`'s return +value, so there is no cross-poll redrive or DLQ: a failure that outlives the retry budget, or a crash mid-batch, is +**at-most-once** across polls. + +```toml +[plugin_config] +amqp_url = "amqp://guest:guest@localhost:5672" +exchange = "iggy_events" +exchange_type = "topic" +routing_key = "iggy.messages" +durable_exchange = true +delivery_mode = "persistent" +include_metadata = true +verbose_logging = false +max_retries = 3 +retry_delay_secs = 1 +max_retry_delay_secs = 5 +timeout_secs = 30 +``` diff --git a/core/connectors/sinks/rabbitmq_sink/config.toml b/core/connectors/sinks/rabbitmq_sink/config.toml new file mode 100644 index 0000000000..4c6bbfc00e --- /dev/null +++ b/core/connectors/sinks/rabbitmq_sink/config.toml @@ -0,0 +1,46 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +type = "sink" +key = "rabbitmq" +enabled = true +version = 0 +name = "RabbitMQ sink" +path = "../../target/release/libiggy_connector_rabbitmq_sink" +verbose = true + +[[streams]] +stream = "test_stream" +topics = ["test_topic"] +schema = "json" +batch_length = 100 +poll_interval = "5ms" +consumer_group = "rabbitmq_sink_connector" + +[plugin_config] +amqp_url = "amqp://guest:guest@localhost:5672" +exchange = "iggy_events" +exchange_type = "topic" +routing_key = "iggy.messages" +durable_exchange = true +delivery_mode = "persistent" +include_metadata = true +verbose_logging = false +max_retries = 3 +retry_delay_secs = 1 +max_retry_delay_secs = 5 +timeout_secs = 30 diff --git a/core/connectors/sinks/rabbitmq_sink/src/lib.rs b/core/connectors/sinks/rabbitmq_sink/src/lib.rs new file mode 100644 index 0000000000..69504096e9 --- /dev/null +++ b/core/connectors/sinks/rabbitmq_sink/src/lib.rs @@ -0,0 +1,783 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use async_trait::async_trait; +use iggy::prelude::HeaderKind; +use iggy_connector_sdk::retry::{exponential_backoff, jitter}; +use iggy_connector_sdk::{ + ConsumedMessage, Error, MessagesMetadata, Sink, TopicMetadata, sink_connector, +}; +use lapin::{ + BasicProperties, Channel, Connection, ConnectionProperties, ExchangeKind, + options::{ConfirmSelectOptions, ExchangeDeclareOptions}, + publisher_confirm::{Confirmation, PublisherConfirm}, + types::{AMQPValue, ByteArray, FieldTable, ShortString}, +}; +use secrecy::{ExposeSecret, SecretString}; +use serde::{Deserialize, Serialize}; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::time::Duration; +use tokio::sync::Mutex; +use tokio::time::timeout; +use tracing::{debug, info, warn}; + +sink_connector!(RabbitMQSink); + +#[derive(Debug)] +struct RabbitMqState { + connection: Connection, + channel: Channel, +} + +#[derive(Debug)] +pub struct RabbitMQSink { + id: u32, + amqp_url: SecretString, + exchange: String, + exchange_type: String, + routing_key: String, + include_metadata: bool, + verbose: bool, + durable_exchange: bool, + delivery_mode: u8, + timeout: Duration, + state: Mutex>, + reconnect_lock: Mutex<()>, + publish_lock: Mutex<()>, + max_retries: u32, + retry_delay: Duration, + max_retry_delay: Duration, + messages_published: AtomicU64, + publish_errors: AtomicU64, +} + +#[derive(Debug, Serialize, Deserialize)] +pub struct RabbitMQSinkConfig { + #[serde( + default = "default_amqp_url", + serialize_with = "iggy_common::serde_secret::serialize_secret" + )] + amqp_url: SecretString, + #[serde(default)] + exchange: Option, + #[serde(default = "default_exchange_type")] + exchange_type: Option, + #[serde(default)] + routing_key: Option, + #[serde(default = "default_true")] + include_metadata: Option, + #[serde(default)] + verbose_logging: Option, + #[serde(default = "default_max_retries")] + max_retries: Option, + #[serde(default = "default_retry_delay_secs")] + retry_delay_secs: Option, + #[serde(default = "default_max_retry_delay_secs")] + max_retry_delay_secs: Option, + #[serde(default = "default_true")] + durable_exchange: Option, + #[serde(default = "default_delivery_mode")] + delivery_mode: Option, + #[serde(default = "default_timeout_secs")] + timeout_secs: Option, +} + +fn default_exchange_type() -> Option { + Some("topic".into()) +} + +fn default_amqp_url() -> SecretString { + SecretString::from("amqp://guest:guest@localhost:5672") +} + +fn default_delivery_mode() -> Option { + Some("persistent".into()) +} + +fn default_timeout_secs() -> Option { + Some(30) +} + +fn append_connection_timeout(amqp_url: &SecretString, timeout_secs: u64) -> SecretString { + let timeout_ms = timeout_secs * 1000; + let url = amqp_url.expose_secret(); + let separator = if url.contains('?') { '&' } else { '?' }; + SecretString::from(format!("{url}{separator}connection_timeout={timeout_ms}")) +} + +fn default_true() -> Option { + Some(true) +} + +fn default_max_retries() -> Option { + Some(3) +} +fn default_retry_delay_secs() -> Option { + Some(1) +} +fn default_max_retry_delay_secs() -> Option { + Some(5) +} + +impl RabbitMQSink { + pub fn new(id: u32, config: RabbitMQSinkConfig) -> Self { + let delivery_mode = match config.delivery_mode.as_deref() { + Some("non_persistent") => 1, + Some("persistent") => 2, + Some(other) => { + warn!( + "Unknown delivery_mode: {other}, defaulting to persistent for connector ID: {id}" + ); + 2 + } + None => 2, + }; + let timeout_secs = match config.timeout_secs { + Some(0) | None => { + if config.timeout_secs == Some(0) { + warn!("timeout_secs must be >= 1, defaulting to 30 for connector ID: {id}"); + } + 30 + } + Some(secs) => secs, + }; + let amqp_url = append_connection_timeout(&config.amqp_url, timeout_secs); + RabbitMQSink { + id, + amqp_url, + exchange: config.exchange.unwrap_or_else(|| "iggy_events".into()), + exchange_type: config.exchange_type.unwrap_or_else(|| "topic".into()), + routing_key: config.routing_key.unwrap_or_else(|| "iggy.messages".into()), + include_metadata: config.include_metadata.unwrap_or(true), + verbose: config.verbose_logging.unwrap_or(false), + durable_exchange: config.durable_exchange.unwrap_or(true), + delivery_mode, + timeout: Duration::from_secs(timeout_secs), + state: Mutex::new(None), + reconnect_lock: Mutex::new(()), + publish_lock: Mutex::new(()), + max_retries: config.max_retries.unwrap_or(3), + retry_delay: Duration::from_secs(config.retry_delay_secs.unwrap_or(1)), + max_retry_delay: Duration::from_secs(config.max_retry_delay_secs.unwrap_or(5)), + messages_published: AtomicU64::new(0), + publish_errors: AtomicU64::new(0), + } + } + + fn exchange_kind(&self) -> Result { + match self.exchange_type.as_str() { + "direct" => Ok(ExchangeKind::Direct), + "topic" => Ok(ExchangeKind::Topic), + "fanout" => Ok(ExchangeKind::Fanout), + "headers" => Ok(ExchangeKind::Headers), + other => Err(Error::InvalidConfigValue(format!( + "unknown exchange_type: {other}. Valid: direct, topic, fanout, headers" + ))), + } + } + + async fn publish_batch_with_retry( + &self, + topic_metadata: &TopicMetadata, + messages_metadata: &MessagesMetadata, + messages: &[ConsumedMessage], + ) -> Result { + // The connectors runtime runs one consume task per (stream, topic) pair against a + // single sink instance, so batches would otherwise interleave on the shared channel. + // Serialize them: a Basic.Return staples to whichever confirm resolves next, and with + // two publishers in flight it can land on the other task's confirm, letting this batch + // count a returned message as delivered. + let _publish_guard = self.publish_lock.lock().await; + let mut attempts = 0u32; + let mut confirmed: usize = 0; + + loop { + let channel = match self + .state + .lock() + .await + .as_ref() + .map(|state| state.channel.clone()) + { + Some(channel) => channel, + None => { + let error = Error::Connection("RabbitMQ not connected".into()); + attempts += 1; + if attempts >= self.max_retries { + self.publish_errors + .fetch_add((messages.len() - confirmed) as u64, Ordering::Relaxed); + return Err(error); + } + if let Err(reconnect_error) = self.reconnect().await { + self.publish_errors + .fetch_add((messages.len() - confirmed) as u64, Ordering::Relaxed); + return Err(Error::Connection(format!( + "failed to reconnect: {reconnect_error}" + ))); + } + let delay = jitter(exponential_backoff( + self.retry_delay, + attempts.saturating_sub(1), + self.max_retry_delay, + )); + warn!( + "RabbitMQ not connected for connector ID: {} (attempt {attempts}/{}). Retrying in {:?}.", + self.id, self.max_retries, delay + ); + tokio::time::sleep(delay).await; + continue; + } + }; + + let mut last_error: Option = None; + let mut last_retryable = false; + let mut in_flight: Vec = + Vec::with_capacity(messages.len() - confirmed); + for message in &messages[confirmed..] { + let body = match message.payload.try_to_bytes() { + Ok(body) => body, + Err(error) => { + // Earlier publishes in this attempt are still in flight; drop the + // channel so their orphaned confirms never leak onto a reused one. + self.clear_state().await; + self.publish_errors + .fetch_add((messages.len() - confirmed) as u64, Ordering::Relaxed); + return Err(error); + } + }; + let mut props = BasicProperties::default().with_delivery_mode(self.delivery_mode); + let headers = self.build_headers(topic_metadata, messages_metadata, message); + if !headers.inner().is_empty() { + props = props.with_headers(headers); + } + + let confirm = match timeout( + self.timeout, + channel.basic_publish( + &self.exchange, + &self.routing_key, + lapin::options::BasicPublishOptions { + mandatory: true, + ..Default::default() + }, + &body, + props, + ), + ) + .await + { + Ok(Ok(confirm)) => confirm, + Ok(Err(e)) => { + last_error = Some(Error::CannotStoreData(e.to_string())); + last_retryable = is_lapin_error_retryable(&e); + self.clear_state().await; + break; + } + Err(_) => { + last_error = Some(Error::CannotStoreData("publish timed out".into())); + last_retryable = true; + self.clear_state().await; + break; + } + }; + in_flight.push(confirm); + } + + // Confirms resolve in publish order over already-overlapped RTTs. + // lapin staples a Basic.Return to whichever confirm resolves next, so under + // multiple-ack the returned message is not attributable to a specific publish: + // on any return the whole batch fails and the channel is cleared so a channel + // with unresolved confirms is never reused (a later batch could otherwise + // swallow the return and count the unroutable message as delivered). + for confirm in in_flight { + match timeout(self.timeout, confirm).await { + Ok(Ok(Confirmation::Ack(None))) => confirmed += 1, + Ok(Ok(Confirmation::Ack(Some(_)))) => { + last_error = Some(Error::InvalidRecordValue( + "message returned as unroutable by RabbitMQ".into(), + )); + last_retryable = false; + self.clear_state().await; + break; + } + Ok(Ok(Confirmation::Nack(_))) => { + last_error = + Some(Error::CannotStoreData("message nack'd by RabbitMQ".into())); + last_retryable = true; + self.clear_state().await; + break; + } + Ok(Ok(Confirmation::NotRequested)) => { + last_error = Some(Error::CannotStoreData( + "publisher confirms not enabled".into(), + )); + last_retryable = false; + self.clear_state().await; + break; + } + Ok(Err(e)) => { + last_error = Some(Error::CannotStoreData(format!("publish rejected: {e}"))); + last_retryable = is_lapin_error_retryable(&e); + self.clear_state().await; + break; + } + Err(_) => { + last_error = Some(Error::CannotStoreData( + "publisher confirmation timed out".into(), + )); + last_retryable = true; + self.clear_state().await; + break; + } + } + } + + if last_error.is_none() { + return Ok(confirmed as u64); + } + + let error = last_error.unwrap(); + attempts += 1; + + if !last_retryable || attempts >= self.max_retries { + self.publish_errors + .fetch_add((messages.len() - confirmed) as u64, Ordering::Relaxed); + return Err(Error::CannotStoreData(format!( + "batch publish failed after {attempts} attempts: {error}" + ))); + } + + match self.reconnect().await { + Ok(_) => {} + Err(reconnect_error) => { + self.publish_errors + .fetch_add((messages.len() - confirmed) as u64, Ordering::Relaxed); + return Err(Error::Connection(format!( + "failed to reconnect: {reconnect_error}" + ))); + } + } + + let delay = jitter(exponential_backoff( + self.retry_delay, + attempts.saturating_sub(1), + self.max_retry_delay, + )); + warn!( + "Transient RabbitMQ publish error for connector ID: {} (attempt {attempts}/{}): {error}. Retrying in {:?}.", + self.id, self.max_retries, delay + ); + tokio::time::sleep(delay).await; + } + } + + async fn clear_state(&self) { + *self.state.lock().await = None; + } + + fn build_headers( + &self, + topic_metadata: &TopicMetadata, + messages_metadata: &MessagesMetadata, + message: &ConsumedMessage, + ) -> FieldTable { + let mut headers = FieldTable::default(); + if let Some(user_headers) = &message.headers + && !user_headers.is_empty() + { + for (key, value) in user_headers { + let name = ShortString::from(key.to_string_value()); + let amqp_value = match value.kind() { + HeaderKind::String => AMQPValue::LongString(value.to_string_value().into()), + _ => AMQPValue::ByteArray(ByteArray::from(value.as_bytes())), + }; + headers.insert(name, amqp_value); + } + } + if self.include_metadata { + headers.insert( + "iggy_stream".into(), + AMQPValue::LongString(topic_metadata.stream.clone().into()), + ); + headers.insert( + "iggy_topic".into(), + AMQPValue::LongString(topic_metadata.topic.clone().into()), + ); + headers.insert( + "iggy_partition_id".into(), + AMQPValue::LongUInt(messages_metadata.partition_id), + ); + headers.insert( + "iggy_offset".into(), + AMQPValue::LongLongInt(message.offset as i64), + ); + } + headers + } + + async fn reconnect(&self) -> Result<(), Error> { + let _guard = self.reconnect_lock.lock().await; + if self.state.lock().await.is_some() { + return Ok(()); + } + warn!("Reconnecting RabbitMQ sink ID: {}", self.id); + let state = self.connect_with_timeout().await?; + *self.state.lock().await = Some(state); + Ok(()) + } + + async fn connect_with_timeout(&self) -> Result { + timeout(self.timeout, self.establish_connection()) + .await + .map_err(|_| Error::Connection("connection setup timed out".into()))? + } + + async fn establish_connection(&self) -> Result { + let connection = Connection::connect( + self.amqp_url.expose_secret(), + ConnectionProperties::default(), + ) + .await + .map_err(|e| Error::Connection(e.to_string()))?; + let channel = connection + .create_channel() + .await + .map_err(|e| Error::Connection(e.to_string()))?; + channel + .confirm_select(ConfirmSelectOptions::default()) + .await + .map_err(|e| Error::Connection(e.to_string()))?; + let exchange_kind = self.exchange_kind()?; + channel + .exchange_declare( + &self.exchange, + exchange_kind, + ExchangeDeclareOptions { + durable: self.durable_exchange, + ..Default::default() + }, + FieldTable::default(), + ) + .await + .map_err(|e| Error::Connection(e.to_string()))?; + Ok(RabbitMqState { + connection, + channel, + }) + } +} + +#[async_trait] +impl Sink for RabbitMQSink { + async fn open(&mut self) -> Result<(), Error> { + self.exchange_kind()?; + let state = self.connect_with_timeout().await?; + *self.state.get_mut() = Some(state); + info!( + "Opened RabbitMQ sink ID: {}, connected to exchange: {}", + self.id, self.exchange + ); + + Ok(()) + } + + async fn consume( + &self, + topic_metadata: &TopicMetadata, + messages_metadata: MessagesMetadata, + messages: Vec, + ) -> Result<(), Error> { + let published = self + .publish_batch_with_retry(topic_metadata, &messages_metadata, &messages) + .await?; + self.messages_published + .fetch_add(published, Ordering::Relaxed); + if self.verbose { + info!( + "Published {published} messages to exchange: {}", + self.exchange + ); + } else { + debug!( + "Published {published} messages to exchange: {}", + self.exchange + ); + } + Ok(()) + } + + async fn close(&mut self) -> Result<(), Error> { + let published = self.messages_published.load(Ordering::Relaxed); + let errors = self.publish_errors.load(Ordering::Relaxed); + info!( + "RabbitMQ sink ID: {} processed {} messages with {} errors", + self.id, published, errors + ); + + if let Some(state) = self.state.get_mut().take() { + state + .channel + .close(200, "OK") + .await + .map_err(|e| Error::Connection(e.to_string()))?; + state + .connection + .close(200, "OK") + .await + .map_err(|e| Error::Connection(e.to_string()))?; + } + Ok(()) + } +} + +fn is_lapin_error_retryable(error: &lapin::Error) -> bool { + match error { + lapin::Error::InvalidChannelState(_) + | lapin::Error::InvalidConnectionState(_) + | lapin::Error::IOError(_) + | lapin::Error::MissingHeartbeatError => true, + // AMQP soft channel errors: 405 RESOURCE_LOCKED, 320 CONNECTION_FORCED. + lapin::Error::ProtocolError(amqp_error) => matches!(amqp_error.get_id(), 320 | 405), + _ => false, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use iggy::prelude::{HeaderKey, HeaderValue}; + use iggy_connector_sdk::{Payload, Schema}; + use std::str::FromStr; + + fn test_config() -> RabbitMQSinkConfig { + RabbitMQSinkConfig { + amqp_url: SecretString::from("amqp://guest:guest@localhost:5672"), + exchange: None, + exchange_type: None, + routing_key: None, + include_metadata: Some(true), + verbose_logging: Some(false), + max_retries: Some(3), + retry_delay_secs: Some(1), + max_retry_delay_secs: Some(5), + durable_exchange: None, + delivery_mode: None, + timeout_secs: None, + } + } + + fn test_sink(config: RabbitMQSinkConfig) -> RabbitMQSink { + RabbitMQSink::new(1, config) + } + + fn test_message(offset: u64) -> ConsumedMessage { + ConsumedMessage { + id: 1, + offset, + checksum: 0, + timestamp: 0, + origin_timestamp: 0, + headers: None, + payload: Payload::Text("payload".into()), + } + } + + #[test] + fn given_offset_above_u32_max_when_build_headers_should_encode_full_value() { + let sink = test_sink(test_config()); + let topic = TopicMetadata { + stream: "test_stream".into(), + topic: "test_topic".into(), + }; + let metadata = MessagesMetadata { + partition_id: 0, + current_offset: 0, + schema: Schema::Json, + }; + let headers = sink.build_headers(&topic, &metadata, &test_message(u32::MAX as u64 + 1)); + match headers.inner().get(&ShortString::from("iggy_offset")) { + Some(AMQPValue::LongLongInt(value)) => assert_eq!(*value, u32::MAX as i64 + 1), + other => panic!("expected LongLongInt, got {other:?}"), + } + } + + #[test] + fn given_user_headers_when_build_headers_should_preserve_string_and_binary() { + let mut user_headers = std::collections::BTreeMap::new(); + user_headers.insert( + HeaderKey::from_str("content-type").unwrap(), + HeaderValue::from_str("application/json").unwrap(), + ); + user_headers.insert( + HeaderKey::from_str("trace-id").unwrap(), + HeaderValue::try_from(vec![0xDE, 0xAD, 0xBE, 0xEF]).unwrap(), + ); + let mut message = test_message(0); + message.headers = Some(user_headers); + + let sink = test_sink(test_config()); + let topic = TopicMetadata { + stream: "test_stream".into(), + topic: "test_topic".into(), + }; + let metadata = MessagesMetadata { + partition_id: 0, + current_offset: 0, + schema: Schema::Json, + }; + let headers = sink.build_headers(&topic, &metadata, &message); + let inner = headers.inner(); + assert!(matches!( + inner.get(&ShortString::from("content-type")), + Some(AMQPValue::LongString(s)) if s.as_bytes() == b"application/json" + )); + assert!(matches!( + inner.get(&ShortString::from("trace-id")), + Some(AMQPValue::ByteArray(b)) if b.as_slice() == [0xDE, 0xAD, 0xBE, 0xEF] + )); + } + + #[test] + fn given_persistent_delivery_mode_when_new_should_set_delivery_mode_2() { + let sink = test_sink(test_config()); + assert_eq!(sink.delivery_mode, 2); + } + + #[test] + fn given_non_persistent_delivery_mode_when_new_should_set_delivery_mode_1() { + let mut config = test_config(); + config.delivery_mode = Some("non_persistent".into()); + let sink = test_sink(config); + assert_eq!(sink.delivery_mode, 1); + } + + #[test] + fn given_durable_exchange_when_new_should_default_to_true() { + let sink = test_sink(test_config()); + assert!(sink.durable_exchange); + } + + #[test] + fn given_known_exchange_type_when_exchange_kind_should_map_to_lapin_kind() { + let sink = test_sink(test_config()); + assert!(matches!(sink.exchange_kind(), Ok(ExchangeKind::Topic))); + } + + #[test] + fn given_unknown_exchange_type_when_exchange_kind_should_error() { + let mut config = test_config(); + config.exchange_type = Some("mystery".into()); + let sink = test_sink(config); + assert!(matches!( + sink.exchange_kind(), + Err(Error::InvalidConfigValue(_)) + )); + } + + #[test] + fn given_transient_lapin_error_when_is_lapin_error_retryable_should_return_true() { + let errors = [ + lapin::Error::InvalidChannelState(lapin::ChannelState::Closed), + lapin::Error::InvalidConnectionState(lapin::ConnectionState::Error), + lapin::Error::IOError(std::sync::Arc::new(std::io::Error::new( + std::io::ErrorKind::ConnectionReset, + "reset by peer", + ))), + lapin::Error::MissingHeartbeatError, + lapin::Error::ProtocolError(lapin::protocol::AMQPError::new( + lapin::protocol::AMQPErrorKind::Soft( + lapin::protocol::AMQPSoftError::RESOURCELOCKED, + ), + "RESOURCE_LOCKED".into(), + )), + ]; + for error in errors { + assert!( + is_lapin_error_retryable(&error), + "expected {error:?} to be retryable" + ); + } + } + + #[test] + fn given_permanent_lapin_error_when_is_lapin_error_retryable_should_return_false() { + let errors = [ + lapin::Error::ProtocolError(lapin::protocol::AMQPError::new( + lapin::protocol::AMQPErrorKind::Soft( + lapin::protocol::AMQPSoftError::PRECONDITIONFAILED, + ), + "PRECONDITION_FAILED".into(), + )), + lapin::Error::InvalidChannel(0), + ]; + for error in errors { + assert!( + !is_lapin_error_retryable(&error), + "expected {error:?} to be permanent" + ); + } + } + + #[test] + fn given_timeout_secs_when_new_should_set_timeout() { + let mut config = test_config(); + config.timeout_secs = Some(7); + let sink = test_sink(config); + assert_eq!(sink.timeout, Duration::from_secs(7)); + } + + #[test] + fn given_zero_timeout_secs_when_new_should_default_to_30() { + let mut config = test_config(); + config.timeout_secs = Some(0); + let sink = test_sink(config); + assert_eq!(sink.timeout, Duration::from_secs(30)); + assert!( + sink.amqp_url + .expose_secret() + .ends_with("connection_timeout=30000") + ); + } + + #[test] + fn given_minimal_config_when_deserialized_should_apply_defaults() { + let config: RabbitMQSinkConfig = serde_json::from_str("{}").unwrap(); + let sink = RabbitMQSink::new(1, config); + assert_eq!( + sink.amqp_url.expose_secret(), + "amqp://guest:guest@localhost:5672?connection_timeout=30000" + ); + assert_eq!(sink.exchange, "iggy_events"); + assert_eq!(sink.exchange_type, "topic"); + assert_eq!(sink.routing_key, "iggy.messages"); + assert!(sink.include_metadata); + assert!(sink.durable_exchange); + assert_eq!(sink.delivery_mode, 2); + assert_eq!(sink.max_retries, 3); + assert_eq!(sink.retry_delay, Duration::from_secs(1)); + assert_eq!(sink.max_retry_delay, Duration::from_secs(5)); + assert_eq!(sink.timeout, Duration::from_secs(30)); + } + + #[test] + fn given_unknown_delivery_mode_when_new_should_default_to_persistent() { + let mut config = test_config(); + config.delivery_mode = Some("fancy".into()); + let sink = test_sink(config); + assert_eq!(sink.delivery_mode, 2); + } +} diff --git a/core/integration/Cargo.toml b/core/integration/Cargo.toml index 0450fa306d..c4d5008746 100644 --- a/core/integration/Cargo.toml +++ b/core/integration/Cargo.toml @@ -71,6 +71,7 @@ iggy_connector_sdk = { workspace = true, features = ["api"] } journal = { workspace = true } jsonwebtoken = { workspace = true } keyring-core = { workspace = true } +lapin = { workspace = true } lazy_static = { workspace = true } libc = { workspace = true } mongodb = { workspace = true } diff --git a/core/integration/tests/connectors/fixtures/mod.rs b/core/integration/tests/connectors/fixtures/mod.rs index e9d7a1332e..af9eb52990 100644 --- a/core/integration/tests/connectors/fixtures/mod.rs +++ b/core/integration/tests/connectors/fixtures/mod.rs @@ -28,6 +28,7 @@ mod meilisearch; mod mongodb; mod postgres; mod quickwit; +mod rabbitmq; mod redshift; mod s3; mod surrealdb; @@ -83,6 +84,11 @@ pub use postgres::{ PostgresSourceOps, }; pub use quickwit::{QuickwitFixture, QuickwitOps, QuickwitPreCreatedFixture}; +pub use rabbitmq::{ + RabbitMqOps, RabbitMqSinkDirectFixture, RabbitMqSinkFanoutFixture, RabbitMqSinkFixture, + RabbitMqSinkHeadersFixture, RabbitMqSinkRawSchemaFixture, RabbitMqSinkUnroutableFixture, + RabbitMqSinkWithoutMetadataFixture, +}; pub use redshift::{ RedshiftSinkFixture, RedshiftSinkJsonFixture, RedshiftSinkNoArchiveFixture, RedshiftSinkVarbyteFixture, diff --git a/core/integration/tests/connectors/fixtures/rabbitmq/container.rs b/core/integration/tests/connectors/fixtures/rabbitmq/container.rs new file mode 100644 index 0000000000..3864604490 --- /dev/null +++ b/core/integration/tests/connectors/fixtures/rabbitmq/container.rs @@ -0,0 +1,327 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use crate::connectors::fixtures; +use futures::StreamExt; +use integration::harness::TestBinaryError; +use lapin::{ + Connection, ConnectionProperties, ExchangeKind, + options::{BasicConsumeOptions, ExchangeDeclareOptions, QueueBindOptions, QueueDeclareOptions}, + types::{AMQPValue, FieldTable}, +}; +use std::time::Duration; +use testcontainers_modules::testcontainers::core::{IntoContainerPort, WaitFor}; +use testcontainers_modules::testcontainers::runners::AsyncRunner; +use testcontainers_modules::testcontainers::{ContainerAsync, GenericImage, ImageExt}; +use tokio::time::sleep; +use tracing::info; +use uuid::Uuid; + +const RABBITMQ_IMAGE: &str = "docker.io/rabbitmq"; +const RABBITMQ_TAG: &str = "4.0-management"; +const RABBITMQ_PORT: u16 = 5672; +const RABBITMQ_READY_MSG: &str = "Time to start RabbitMQ:"; +const RABBITMQ_BOOT_ATTEMPTS: usize = 60; +const RABBITMQ_BOOT_INTERVAL_MS: u64 = 1000; + +pub(super) const DEFAULT_TEST_STREAM: &str = "test_stream"; +pub(super) const DEFAULT_TEST_TOPIC: &str = "test_topic"; +pub(super) const DEFAULT_EXCHANGE: &str = "iggy_events"; +pub(super) const DEFAULT_EXCHANGE_TYPE: &str = "topic"; +pub(super) const DEFAULT_ROUTING_KEY: &str = "iggy.messages"; +pub(super) const DEFAULT_CONSUMER_GROUP: &str = "rabbitmq_sink_test_cg"; + +pub(super) const ENV_SINK_AMQP_URL: &str = "IGGY_CONNECTORS_SINK_RABBITMQ_PLUGIN_CONFIG_AMQP_URL"; +pub(super) const ENV_SINK_EXCHANGE: &str = "IGGY_CONNECTORS_SINK_RABBITMQ_PLUGIN_CONFIG_EXCHANGE"; +pub(super) const ENV_SINK_EXCHANGE_TYPE: &str = + "IGGY_CONNECTORS_SINK_RABBITMQ_PLUGIN_CONFIG_EXCHANGE_TYPE"; +pub(super) const ENV_SINK_ROUTING_KEY: &str = + "IGGY_CONNECTORS_SINK_RABBITMQ_PLUGIN_CONFIG_ROUTING_KEY"; +pub(super) const ENV_SINK_STREAMS_0_STREAM: &str = "IGGY_CONNECTORS_SINK_RABBITMQ_STREAMS_0_STREAM"; +pub(super) const ENV_SINK_STREAMS_0_TOPICS: &str = "IGGY_CONNECTORS_SINK_RABBITMQ_STREAMS_0_TOPICS"; +pub(super) const ENV_SINK_STREAMS_0_SCHEMA: &str = "IGGY_CONNECTORS_SINK_RABBITMQ_STREAMS_0_SCHEMA"; +pub(super) const ENV_SINK_STREAMS_0_CONSUMER_GROUP: &str = + "IGGY_CONNECTORS_SINK_RABBITMQ_STREAMS_0_CONSUMER_GROUP"; +pub(super) const ENV_SINK_PATH: &str = "IGGY_CONNECTORS_SINK_RABBITMQ_PATH"; +pub(super) const ENV_SINK_INCLUDE_METADATA: &str = + "IGGY_CONNECTORS_SINK_RABBITMQ_PLUGIN_CONFIG_INCLUDE_METADATA"; + +#[derive(PartialEq)] +pub(super) enum RabbitMqExchangeSetup { + Topic, + Fanout, + Direct, + Headers, +} +pub struct RabbitMqContainer { + #[allow(dead_code)] + container: ContainerAsync, + pub(super) amqp_url: String, + pub(super) queue_names: Vec, + pub(super) exchange_setup: RabbitMqExchangeSetup, +} + +impl RabbitMqContainer { + async fn start_container() -> Result<(ContainerAsync, String), TestBinaryError> { + let container = GenericImage::new(RABBITMQ_IMAGE, RABBITMQ_TAG) + .with_exposed_port(RABBITMQ_PORT.tcp()) + .with_wait_for(WaitFor::message_on_stdout(RABBITMQ_READY_MSG)) + .with_mapped_port(0, RABBITMQ_PORT.tcp()) + .with_container_name(fixtures::unique_container_name("rabbitmq")) + .start() + .await + .map_err(|e| TestBinaryError::FixtureSetup { + fixture_type: "RabbitMqContainer".to_string(), + message: format!("Failed to start container: {e}"), + })?; + + let mapped_port = container + .ports() + .await + .map_err(|e| TestBinaryError::FixtureSetup { + fixture_type: "RabbitMqContainer".to_string(), + message: format!("Failed to get ports: {e}"), + })? + .map_to_host_port_ipv4(RABBITMQ_PORT) + .ok_or_else(|| TestBinaryError::FixtureSetup { + fixture_type: "RabbitMqContainer".to_string(), + message: "No mapping for RabbitMQ port".to_string(), + })?; + let amqp_url = format!("amqp://guest:guest@127.0.0.1:{mapped_port}"); + Ok((container, amqp_url)) + } + + async fn start_with( + exchange_setup: RabbitMqExchangeSetup, + queue_count: usize, + ) -> Result { + let (container, amqp_url) = Self::start_container().await?; + let queue_names = (0..queue_count) + .map(|index| format!("test_queue_{}_{}", Uuid::new_v4().simple(), index)) + .collect(); + + let instance = Self { + container, + amqp_url, + exchange_setup, + queue_names, + }; + instance.wait_until_ready().await?; + + info!("RabbitMQ container available at {}", instance.amqp_url); + Ok(instance) + } + + pub(super) async fn start() -> Result { + Self::start_with(RabbitMqExchangeSetup::Topic, 1).await + } + + pub(super) async fn start_fanout() -> Result { + Self::start_with(RabbitMqExchangeSetup::Fanout, 2).await + } + + pub(super) async fn start_direct() -> Result { + Self::start_with(RabbitMqExchangeSetup::Direct, 1).await + } + + pub(super) async fn start_headers() -> Result { + Self::start_with(RabbitMqExchangeSetup::Headers, 1).await + } + + async fn wait_until_ready(&self) -> Result<(), TestBinaryError> { + let mut last_error = None; + + for _ in 0..RABBITMQ_BOOT_ATTEMPTS { + match Connection::connect(&self.amqp_url, ConnectionProperties::default()).await { + Ok(_) => { + return self.setup_exchange_and_queue().await; + } + Err(error) => last_error = Some(error.to_string()), + } + sleep(Duration::from_millis(RABBITMQ_BOOT_INTERVAL_MS)).await; + } + + let detail = last_error + .map(|error| format!(" Last error: {error}")) + .unwrap_or_default(); + Err(TestBinaryError::FixtureSetup { + fixture_type: "RabbitMqContainer".to_string(), + message: format!("RabbitMQ did not become ready.{detail}"), + }) + } + + async fn setup_exchange_and_queue(&self) -> Result<(), TestBinaryError> { + let conn = Connection::connect(&self.amqp_url, ConnectionProperties::default()) + .await + .map_err(|e| TestBinaryError::FixtureSetup { + fixture_type: "RabbitMqContainer".to_string(), + message: format!("Failed to create connection for consume: {e}"), + })?; + let channel = conn + .create_channel() + .await + .map_err(|e| TestBinaryError::FixtureSetup { + fixture_type: "RabbitMqContainer".to_string(), + message: format!("Failed to create channel for consume: {e}"), + })?; + + let exchange_kind = match self.exchange_setup { + RabbitMqExchangeSetup::Topic => ExchangeKind::Topic, + RabbitMqExchangeSetup::Fanout => ExchangeKind::Fanout, + RabbitMqExchangeSetup::Direct => ExchangeKind::Direct, + RabbitMqExchangeSetup::Headers => ExchangeKind::Headers, + }; + channel + .exchange_declare( + DEFAULT_EXCHANGE, + exchange_kind, + ExchangeDeclareOptions { + durable: true, + ..Default::default() + }, + FieldTable::default(), + ) + .await + .map_err(|e| TestBinaryError::FixtureSetup { + fixture_type: "RabbitMqContainer".to_string(), + message: format!("Failed to declare exchange for consume: {e}"), + })?; + + for queue_name in &self.queue_names { + channel + .queue_declare( + queue_name, + QueueDeclareOptions { + auto_delete: true, + ..Default::default() + }, + FieldTable::default(), + ) + .await + .map_err(|e| TestBinaryError::FixtureSetup { + fixture_type: "RabbitMqContainer".to_string(), + message: format!("Failed to create queue for consume: {e}"), + })?; + + let mut bind_arguments = FieldTable::default(); + if self.exchange_setup == RabbitMqExchangeSetup::Headers { + bind_arguments.insert("x-match".into(), AMQPValue::LongString("all".into())); + bind_arguments.insert("x-user".into(), AMQPValue::LongString("alice".into())); + } + channel + .queue_bind( + queue_name, + DEFAULT_EXCHANGE, + DEFAULT_ROUTING_KEY, + QueueBindOptions::default(), + bind_arguments, + ) + .await + .map_err(|e| TestBinaryError::FixtureSetup { + fixture_type: "RabbitMqContainer".to_string(), + message: format!("Failed to bind queue to exchange for consume: {e}"), + })?; + } + + Ok(()) + } +} + +pub struct ConsumedDelivery { + pub data: Vec, + pub headers: lapin::types::FieldTable, +} + +pub trait RabbitMqOps: Sync { + fn container(&self) -> &RabbitMqContainer; + + fn queue_names(&self) -> &[String] { + &self.container().queue_names + } + + async fn consume_messages( + &self, + count: usize, + ) -> Result, TestBinaryError> { + self.consume_messages_from(&self.container().queue_names[0], count) + .await + } + + async fn consume_messages_from( + &self, + queue_name: &str, + count: usize, + ) -> Result, TestBinaryError> { + self.consume_messages_from_with_timeout(queue_name, count, Duration::from_secs(60)) + .await + } + + async fn consume_messages_from_with_timeout( + &self, + queue_name: &str, + count: usize, + timeout: Duration, + ) -> Result, TestBinaryError> { + let conn = Connection::connect(&self.container().amqp_url, ConnectionProperties::default()) + .await + .map_err(|e| TestBinaryError::InvalidState { + message: format!("Failed to connect to RabbitMQ for consume: {e}"), + })?; + let channel = conn + .create_channel() + .await + .map_err(|e| TestBinaryError::InvalidState { + message: format!("Failed to create channel for consume: {e}"), + })?; + + let mut consumer = channel + .basic_consume( + queue_name, + "", + BasicConsumeOptions::default(), + FieldTable::default(), + ) + .await + .map_err(|e| TestBinaryError::InvalidState { + message: format!("Failed to start consumer: {e}"), + })?; + + let mut messages = Vec::with_capacity(count); + let deadline = tokio::time::Instant::now() + timeout; + while messages.len() < count && tokio::time::Instant::now() < deadline { + match tokio::time::timeout(Duration::from_secs(1), consumer.next()).await { + Ok(Some(delivery)) => { + let delivery = delivery.map_err(|e| TestBinaryError::InvalidState { + message: format!("Consumer error: {e}"), + })?; + let data = delivery.data.clone(); + let headers = delivery.properties.headers().clone().unwrap_or_default(); + delivery.ack(Default::default()).await.map_err(|e| { + TestBinaryError::InvalidState { + message: format!("Failed to ack message: {e}"), + } + })?; + messages.push(ConsumedDelivery { data, headers }); + } + Ok(None) => break, + Err(_) => continue, + } + } + + Ok(messages) + } +} diff --git a/core/integration/tests/connectors/fixtures/rabbitmq/mod.rs b/core/integration/tests/connectors/fixtures/rabbitmq/mod.rs new file mode 100644 index 0000000000..a8154b4e15 --- /dev/null +++ b/core/integration/tests/connectors/fixtures/rabbitmq/mod.rs @@ -0,0 +1,26 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +mod container; +mod sink; + +pub use container::RabbitMqOps; +pub use sink::{ + RabbitMqSinkDirectFixture, RabbitMqSinkFanoutFixture, RabbitMqSinkFixture, + RabbitMqSinkHeadersFixture, RabbitMqSinkRawSchemaFixture, RabbitMqSinkUnroutableFixture, + RabbitMqSinkWithoutMetadataFixture, +}; diff --git a/core/integration/tests/connectors/fixtures/rabbitmq/sink.rs b/core/integration/tests/connectors/fixtures/rabbitmq/sink.rs new file mode 100644 index 0000000000..8301dacdc1 --- /dev/null +++ b/core/integration/tests/connectors/fixtures/rabbitmq/sink.rs @@ -0,0 +1,273 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use super::container::{ + DEFAULT_CONSUMER_GROUP, DEFAULT_EXCHANGE, DEFAULT_EXCHANGE_TYPE, DEFAULT_ROUTING_KEY, + DEFAULT_TEST_STREAM, DEFAULT_TEST_TOPIC, ENV_SINK_AMQP_URL, ENV_SINK_EXCHANGE, + ENV_SINK_EXCHANGE_TYPE, ENV_SINK_INCLUDE_METADATA, ENV_SINK_PATH, ENV_SINK_ROUTING_KEY, + ENV_SINK_STREAMS_0_CONSUMER_GROUP, ENV_SINK_STREAMS_0_SCHEMA, ENV_SINK_STREAMS_0_STREAM, + ENV_SINK_STREAMS_0_TOPICS, RabbitMqContainer, RabbitMqOps, +}; +use async_trait::async_trait; +use iggy_connector_sdk::Schema; +use integration::harness::{TestBinaryError, TestFixture}; +use std::collections::HashMap; + +pub struct RabbitMqSinkFixture { + container: RabbitMqContainer, + include_metadata: bool, + schema: Schema, +} + +impl RabbitMqOps for RabbitMqSinkFixture { + fn container(&self) -> &RabbitMqContainer { + &self.container + } +} + +#[async_trait] +impl TestFixture for RabbitMqSinkFixture { + async fn setup() -> Result { + let container = RabbitMqContainer::start().await?; + Ok(Self { + container, + include_metadata: true, + schema: Schema::Json, + }) + } + + fn connectors_runtime_envs(&self) -> HashMap { + let mut envs = HashMap::new(); + envs.insert( + ENV_SINK_AMQP_URL.to_string(), + self.container.amqp_url.clone(), + ); + envs.insert(ENV_SINK_EXCHANGE.to_string(), DEFAULT_EXCHANGE.into()); + envs.insert( + ENV_SINK_EXCHANGE_TYPE.to_string(), + DEFAULT_EXCHANGE_TYPE.into(), + ); + envs.insert(ENV_SINK_ROUTING_KEY.to_string(), DEFAULT_ROUTING_KEY.into()); + envs.insert( + ENV_SINK_STREAMS_0_STREAM.to_string(), + DEFAULT_TEST_STREAM.into(), + ); + envs.insert( + ENV_SINK_STREAMS_0_TOPICS.to_string(), + format!("[{}]", DEFAULT_TEST_TOPIC), + ); + envs.insert( + ENV_SINK_STREAMS_0_SCHEMA.to_string(), + self.schema.to_string(), + ); + envs.insert( + ENV_SINK_STREAMS_0_CONSUMER_GROUP.to_string(), + DEFAULT_CONSUMER_GROUP.into(), + ); + envs.insert( + ENV_SINK_PATH.to_string(), + "../../target/debug/libiggy_connector_rabbitmq_sink".into(), + ); + envs.insert( + ENV_SINK_INCLUDE_METADATA.to_string(), + self.include_metadata.to_string(), + ); + envs + } +} + +pub struct RabbitMqSinkWithoutMetadataFixture { + inner: RabbitMqSinkFixture, +} + +impl std::ops::Deref for RabbitMqSinkWithoutMetadataFixture { + type Target = RabbitMqSinkFixture; + fn deref(&self) -> &Self::Target { + &self.inner + } +} + +#[async_trait] +impl TestFixture for RabbitMqSinkWithoutMetadataFixture { + async fn setup() -> Result { + let container = RabbitMqContainer::start().await?; + Ok(Self { + inner: RabbitMqSinkFixture { + container, + include_metadata: false, + schema: Schema::Json, + }, + }) + } + + fn connectors_runtime_envs(&self) -> HashMap { + self.inner.connectors_runtime_envs() + } +} + +pub struct RabbitMqSinkRawSchemaFixture { + inner: RabbitMqSinkFixture, +} + +impl std::ops::Deref for RabbitMqSinkRawSchemaFixture { + type Target = RabbitMqSinkFixture; + fn deref(&self) -> &Self::Target { + &self.inner + } +} + +#[async_trait] +impl TestFixture for RabbitMqSinkRawSchemaFixture { + async fn setup() -> Result { + let container = RabbitMqContainer::start().await?; + Ok(Self { + inner: RabbitMqSinkFixture { + container, + include_metadata: true, + schema: Schema::Raw, + }, + }) + } + + fn connectors_runtime_envs(&self) -> HashMap { + self.inner.connectors_runtime_envs() + } +} + +pub struct RabbitMqSinkFanoutFixture { + inner: RabbitMqSinkFixture, +} + +impl std::ops::Deref for RabbitMqSinkFanoutFixture { + type Target = RabbitMqSinkFixture; + fn deref(&self) -> &Self::Target { + &self.inner + } +} + +#[async_trait] +impl TestFixture for RabbitMqSinkFanoutFixture { + async fn setup() -> Result { + let container = RabbitMqContainer::start_fanout().await?; + Ok(Self { + inner: RabbitMqSinkFixture { + container, + include_metadata: true, + schema: Schema::Json, + }, + }) + } + + fn connectors_runtime_envs(&self) -> HashMap { + let mut envs = self.inner.connectors_runtime_envs(); + envs.insert(ENV_SINK_EXCHANGE_TYPE.to_string(), "fanout".into()); + envs + } +} + +pub struct RabbitMqSinkDirectFixture { + inner: RabbitMqSinkFixture, +} + +impl std::ops::Deref for RabbitMqSinkDirectFixture { + type Target = RabbitMqSinkFixture; + fn deref(&self) -> &Self::Target { + &self.inner + } +} + +#[async_trait] +impl TestFixture for RabbitMqSinkDirectFixture { + async fn setup() -> Result { + let container = RabbitMqContainer::start_direct().await?; + Ok(Self { + inner: RabbitMqSinkFixture { + container, + include_metadata: true, + schema: Schema::Json, + }, + }) + } + + fn connectors_runtime_envs(&self) -> HashMap { + let mut envs = self.inner.connectors_runtime_envs(); + envs.insert(ENV_SINK_EXCHANGE_TYPE.to_string(), "direct".into()); + envs + } +} + +pub struct RabbitMqSinkHeadersFixture { + inner: RabbitMqSinkFixture, +} + +impl std::ops::Deref for RabbitMqSinkHeadersFixture { + type Target = RabbitMqSinkFixture; + fn deref(&self) -> &Self::Target { + &self.inner + } +} + +#[async_trait] +impl TestFixture for RabbitMqSinkHeadersFixture { + async fn setup() -> Result { + let container = RabbitMqContainer::start_headers().await?; + Ok(Self { + inner: RabbitMqSinkFixture { + container, + include_metadata: true, + schema: Schema::Json, + }, + }) + } + + fn connectors_runtime_envs(&self) -> HashMap { + let mut envs = self.inner.connectors_runtime_envs(); + envs.insert(ENV_SINK_EXCHANGE_TYPE.to_string(), "headers".into()); + envs + } +} + +pub struct RabbitMqSinkUnroutableFixture { + inner: RabbitMqSinkFixture, +} + +impl std::ops::Deref for RabbitMqSinkUnroutableFixture { + type Target = RabbitMqSinkFixture; + fn deref(&self) -> &Self::Target { + &self.inner + } +} + +#[async_trait] +impl TestFixture for RabbitMqSinkUnroutableFixture { + async fn setup() -> Result { + let container = RabbitMqContainer::start().await?; + Ok(Self { + inner: RabbitMqSinkFixture { + container, + include_metadata: true, + schema: Schema::Json, + }, + }) + } + + fn connectors_runtime_envs(&self) -> HashMap { + let mut envs = self.inner.connectors_runtime_envs(); + envs.insert(ENV_SINK_ROUTING_KEY.to_string(), "unroutable.key".into()); + envs + } +} diff --git a/core/integration/tests/connectors/mod.rs b/core/integration/tests/connectors/mod.rs index f6346bcc20..54dd1437b9 100644 --- a/core/integration/tests/connectors/mod.rs +++ b/core/integration/tests/connectors/mod.rs @@ -29,6 +29,7 @@ mod meilisearch; mod mongodb; mod postgres; mod quickwit; +mod rabbitmq; mod random; mod random_source_liveness; mod redshift; diff --git a/core/integration/tests/connectors/rabbitmq/mod.rs b/core/integration/tests/connectors/rabbitmq/mod.rs new file mode 100644 index 0000000000..c4a30f2755 --- /dev/null +++ b/core/integration/tests/connectors/rabbitmq/mod.rs @@ -0,0 +1,18 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +mod rabbitmq_sink; diff --git a/core/integration/tests/connectors/rabbitmq/rabbitmq_sink.rs b/core/integration/tests/connectors/rabbitmq/rabbitmq_sink.rs new file mode 100644 index 0000000000..f9c2376fb1 --- /dev/null +++ b/core/integration/tests/connectors/rabbitmq/rabbitmq_sink.rs @@ -0,0 +1,429 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use crate::connectors::fixtures::{ + RabbitMqOps, RabbitMqSinkDirectFixture, RabbitMqSinkFanoutFixture, RabbitMqSinkFixture, + RabbitMqSinkHeadersFixture, RabbitMqSinkRawSchemaFixture, RabbitMqSinkUnroutableFixture, + RabbitMqSinkWithoutMetadataFixture, +}; +use bytes::Bytes; +use iggy::prelude::{HeaderKey, HeaderValue, IggyMessage, Partitioning}; +use iggy_common::Identifier; +use iggy_common::MessageClient; +use integration::harness::seeds; +use integration::iggy_harness; +use lapin::types::{AMQPValue, ShortString}; +use std::collections::BTreeMap; +use std::str::FromStr; +use std::time::Duration; + +#[iggy_harness( + server(connectors_runtime(config_path = "tests/connectors/rabbitmq/sink.toml")), + seed = seeds::connector_stream +)] + +async fn json_messages_are_published_to_rabbitmq_exchange( + harness: &TestHarness, + fixture: RabbitMqSinkFixture, +) { + let client = harness.root_client().await.unwrap(); + let stream_id: Identifier = seeds::names::STREAM.try_into().unwrap(); + let topic_id: Identifier = seeds::names::TOPIC.try_into().unwrap(); + + let payloads = [ + serde_json::json!({"name": "Alice"}), + serde_json::json!({"name": "Bob"}), + serde_json::json!({"name": "Carol"}), + ]; + let mut messages: Vec = payloads + .iter() + .enumerate() + .map(|(idx, payload)| { + IggyMessage::builder() + .id((idx + 1) as u128) + .payload(Bytes::from(serde_json::to_vec(payload).unwrap())) + .build() + .unwrap() + }) + .collect(); + + client + .send_messages( + &stream_id, + &topic_id, + &Partitioning::partition_id(0), + &mut messages, + ) + .await + .unwrap(); + + let delivered = fixture.consume_messages(3).await.unwrap(); + assert_eq!(delivered.len(), 3); + for (idx, delivery) in delivered.iter().enumerate() { + let value: serde_json::Value = serde_json::from_slice(&delivery.data).unwrap(); + assert_eq!(value, payloads[idx]); + assert_eq!( + header_str(&delivery.headers, "iggy_stream").as_deref(), + Some("test_stream") + ); + assert_eq!( + header_str(&delivery.headers, "iggy_topic").as_deref(), + Some("test_topic") + ); + } +} + +fn header_str(headers: &lapin::types::FieldTable, key: &str) -> Option { + headers + .inner() + .get(&ShortString::from(key)) + .and_then(|v| match v { + AMQPValue::LongString(s) => Some(s.to_string()), + _ => None, + }) +} + +#[iggy_harness( + server(connectors_runtime(config_path = "tests/connectors/rabbitmq/sink.toml")), + seed = seeds::connector_stream +)] + +async fn given_direct_exchange_when_published_should_deliver_to_bound_queue( + harness: &TestHarness, + fixture: RabbitMqSinkDirectFixture, +) { + let client = harness.root_client().await.unwrap(); + let stream_id: Identifier = seeds::names::STREAM.try_into().unwrap(); + let topic_id: Identifier = seeds::names::TOPIC.try_into().unwrap(); + + let payloads = [ + serde_json::json!({"name": "Alice"}), + serde_json::json!({"name": "Bob"}), + serde_json::json!({"name": "Carol"}), + ]; + let mut messages: Vec = payloads + .iter() + .enumerate() + .map(|(idx, payload)| { + IggyMessage::builder() + .id((idx + 1) as u128) + .payload(Bytes::from(serde_json::to_vec(payload).unwrap())) + .build() + .unwrap() + }) + .collect(); + + client + .send_messages( + &stream_id, + &topic_id, + &Partitioning::partition_id(0), + &mut messages, + ) + .await + .unwrap(); + + let delivered = fixture.consume_messages(3).await.unwrap(); + assert_eq!(delivered.len(), 3); + for (idx, delivery) in delivered.iter().enumerate() { + let value: serde_json::Value = serde_json::from_slice(&delivery.data).unwrap(); + assert_eq!(value, payloads[idx]); + assert_eq!( + header_str(&delivery.headers, "iggy_stream").as_deref(), + Some("test_stream") + ); + assert_eq!( + header_str(&delivery.headers, "iggy_topic").as_deref(), + Some("test_topic") + ); + } +} + +#[iggy_harness( + server(connectors_runtime(config_path = "tests/connectors/rabbitmq/sink.toml")), + seed = seeds::connector_stream +)] +async fn given_include_metadata_false_when_published_should_not_include_iggy_headers( + harness: &TestHarness, + fixture: RabbitMqSinkWithoutMetadataFixture, +) { + let client = harness.root_client().await.unwrap(); + let stream_id: Identifier = seeds::names::STREAM.try_into().unwrap(); + let topic_id: Identifier = seeds::names::TOPIC.try_into().unwrap(); + + let payload = serde_json::json!({"name": "Alice"}); + let mut messages = vec![ + IggyMessage::builder() + .id(1) + .payload(Bytes::from(serde_json::to_vec(&payload).unwrap())) + .build() + .unwrap(), + ]; + + client + .send_messages( + &stream_id, + &topic_id, + &Partitioning::partition_id(0), + &mut messages, + ) + .await + .unwrap(); + + let delivered = fixture.consume_messages(1).await.unwrap(); + assert_eq!(delivered.len(), 1); + let value: serde_json::Value = serde_json::from_slice(&delivered[0].data).unwrap(); + assert_eq!(value, payload); + assert!(delivered[0].headers.inner().is_empty()); +} + +#[iggy_harness( + server(connectors_runtime(config_path = "tests/connectors/rabbitmq/sink.toml")), + seed = seeds::connector_stream +)] +async fn given_raw_schema_when_published_should_preserve_raw_payload_bytes( + harness: &TestHarness, + fixture: RabbitMqSinkRawSchemaFixture, +) { + let client = harness.root_client().await.unwrap(); + let stream_id: Identifier = seeds::names::STREAM.try_into().unwrap(); + let topic_id: Identifier = seeds::names::TOPIC.try_into().unwrap(); + + let raw_payloads: Vec> = vec![ + b"plain text message".to_vec(), + vec![0x00, 0x01, 0x02, 0xFF, 0xFE, 0xFD], + vec![0xDE, 0xAD, 0xBE, 0xEF], + ]; + + let mut messages: Vec = raw_payloads + .iter() + .enumerate() + .map(|(idx, payload)| { + IggyMessage::builder() + .id((idx + 1) as u128) + .payload(Bytes::from(payload.clone())) + .build() + .unwrap() + }) + .collect(); + + client + .send_messages( + &stream_id, + &topic_id, + &Partitioning::partition_id(0), + &mut messages, + ) + .await + .unwrap(); + + let delivered = fixture.consume_messages(3).await.unwrap(); + assert_eq!(delivered.len(), 3); + for (idx, delivery) in delivered.iter().enumerate() { + assert_eq!(delivery.data, raw_payloads[idx]); + assert_eq!( + header_str(&delivery.headers, "iggy_stream").as_deref(), + Some("test_stream") + ); + assert_eq!( + header_str(&delivery.headers, "iggy_topic").as_deref(), + Some("test_topic") + ); + } +} + +#[iggy_harness( + server(connectors_runtime(config_path = "tests/connectors/rabbitmq/sink.toml")), + seed = seeds::connector_stream +)] +async fn given_fanout_exchange_when_published_should_deliver_to_all_bound_queues( + harness: &TestHarness, + fixture: RabbitMqSinkFanoutFixture, +) { + let client = harness.root_client().await.unwrap(); + let stream_id: Identifier = seeds::names::STREAM.try_into().unwrap(); + let topic_id: Identifier = seeds::names::TOPIC.try_into().unwrap(); + + let payload = serde_json::json!({"name": "Alice"}); + let mut messages = vec![ + IggyMessage::builder() + .id(1) + .payload(Bytes::from(serde_json::to_vec(&payload).unwrap())) + .build() + .unwrap(), + ]; + + client + .send_messages( + &stream_id, + &topic_id, + &Partitioning::partition_id(0), + &mut messages, + ) + .await + .unwrap(); + + assert_eq!( + fixture.queue_names().len(), + 2, + "fanout fixture should bind two queues" + ); + for queue_name in fixture.queue_names() { + let delivered = fixture.consume_messages_from(queue_name, 1).await.unwrap(); + assert_eq!( + delivered.len(), + 1, + "fanout exchange must deliver to every bound queue" + ); + let value: serde_json::Value = serde_json::from_slice(&delivered[0].data).unwrap(); + assert_eq!(value, payload); + assert_eq!( + header_str(&delivered[0].headers, "iggy_stream").as_deref(), + Some("test_stream") + ); + assert_eq!( + header_str(&delivered[0].headers, "iggy_topic").as_deref(), + Some("test_topic") + ); + } +} + +#[iggy_harness( + server(connectors_runtime(config_path = "tests/connectors/rabbitmq/sink.toml")), + seed = seeds::connector_stream +)] +async fn given_predeclared_durable_exchange_when_published_should_deliver( + harness: &TestHarness, + fixture: RabbitMqSinkFixture, +) { + let client = harness.root_client().await.unwrap(); + let stream_id: Identifier = seeds::names::STREAM.try_into().unwrap(); + let topic_id: Identifier = seeds::names::TOPIC.try_into().unwrap(); + + let payload = serde_json::json!({"name": "Alice"}); + let mut messages = vec![ + IggyMessage::builder() + .id(1) + .payload(Bytes::from(serde_json::to_vec(&payload).unwrap())) + .build() + .unwrap(), + ]; + + client + .send_messages( + &stream_id, + &topic_id, + &Partitioning::partition_id(0), + &mut messages, + ) + .await + .unwrap(); + + let delivered = fixture.consume_messages(1).await.unwrap(); + assert_eq!(delivered.len(), 1); + let value: serde_json::Value = serde_json::from_slice(&delivered[0].data).unwrap(); + assert_eq!(value, payload); +} + +#[iggy_harness( + server(connectors_runtime(config_path = "tests/connectors/rabbitmq/sink.toml")), + seed = seeds::connector_stream +)] +async fn given_unroutable_routing_key_when_published_should_not_deliver( + harness: &TestHarness, + fixture: RabbitMqSinkUnroutableFixture, +) { + let client = harness.root_client().await.unwrap(); + let stream_id: Identifier = seeds::names::STREAM.try_into().unwrap(); + let topic_id: Identifier = seeds::names::TOPIC.try_into().unwrap(); + + let payload = serde_json::json!({"name": "Alice"}); + let mut messages = vec![ + IggyMessage::builder() + .id(1) + .payload(Bytes::from(serde_json::to_vec(&payload).unwrap())) + .build() + .unwrap(), + ]; + + client + .send_messages( + &stream_id, + &topic_id, + &Partitioning::partition_id(0), + &mut messages, + ) + .await + .unwrap(); + + tokio::time::sleep(Duration::from_secs(5)).await; + let delivered = fixture + .consume_messages_from_with_timeout(&fixture.queue_names()[0], 1, Duration::from_secs(5)) + .await + .unwrap(); + assert!( + delivered.is_empty(), + "unroutable mandatory publish must not reach any queue" + ); +} + +#[iggy_harness( + server(connectors_runtime(config_path = "tests/connectors/rabbitmq/sink.toml")), + seed = seeds::connector_stream +)] +async fn given_user_header_when_published_through_headers_exchange_should_route( + harness: &TestHarness, + fixture: RabbitMqSinkHeadersFixture, +) { + let client = harness.root_client().await.unwrap(); + let stream_id: Identifier = seeds::names::STREAM.try_into().unwrap(); + let topic_id: Identifier = seeds::names::TOPIC.try_into().unwrap(); + + let user_headers = BTreeMap::from([( + HeaderKey::from_str("x-user").unwrap(), + HeaderValue::from_str("alice").unwrap(), + )]); + let payload = serde_json::json!({"name": "Alice"}); + let mut messages = vec![ + IggyMessage::builder() + .id(1) + .payload(Bytes::from(serde_json::to_vec(&payload).unwrap())) + .user_headers(user_headers) + .build() + .unwrap(), + ]; + + client + .send_messages( + &stream_id, + &topic_id, + &Partitioning::partition_id(0), + &mut messages, + ) + .await + .unwrap(); + + let delivered = fixture.consume_messages(1).await.unwrap(); + assert_eq!(delivered.len(), 1); + let value: serde_json::Value = serde_json::from_slice(&delivered[0].data).unwrap(); + assert_eq!(value, payload); + assert_eq!( + header_str(&delivered[0].headers, "x-user").as_deref(), + Some("alice"), + "user header must survive to AMQP headers for headers-exchange routing" + ); +} diff --git a/core/integration/tests/connectors/rabbitmq/sink.toml b/core/integration/tests/connectors/rabbitmq/sink.toml new file mode 100644 index 0000000000..e425ed503b --- /dev/null +++ b/core/integration/tests/connectors/rabbitmq/sink.toml @@ -0,0 +1,20 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +[connectors] +config_type = "local" +config_dir = "../connectors/sinks/rabbitmq_sink" diff --git a/scripts/bump-version.sh b/scripts/bump-version.sh index 88413af9d2..82d278eb62 100755 --- a/scripts/bump-version.sh +++ b/scripts/bump-version.sh @@ -87,7 +87,7 @@ EOF } RUST_COMPONENTS="rust-sdk rust-common rust-binary-protocol rust-server rust-cli rust-connector-sdk rust-mcp rust-bench rust-bench-dashboard-frontend rust-bench-dashboard-server rust-bench-report" -CONNECTOR_SINK_COMPONENTS="rust-connector-delta-sink rust-connector-elasticsearch-sink rust-connector-http-sink rust-connector-iceberg-sink rust-connector-influxdb-sink rust-connector-mongodb-sink rust-connector-postgres-sink rust-connector-quickwit-sink rust-connector-stdout-sink rust-connector-surrealdb-sink" +CONNECTOR_SINK_COMPONENTS="rust-connector-delta-sink rust-connector-elasticsearch-sink rust-connector-http-sink rust-connector-iceberg-sink rust-connector-influxdb-sink rust-connector-mongodb-sink rust-connector-postgres-sink rust-connector-quickwit-sink rust-connector-stdout-sink rust-connector-surrealdb-sink rust-connector-rabbitmq-sink" CONNECTOR_SOURCE_COMPONENTS="rust-connector-elasticsearch-source rust-connector-influxdb-source rust-connector-postgres-source rust-connector-random-source" CONNECTOR_COMPONENTS="rust-connector-runtime ${CONNECTOR_SINK_COMPONENTS} ${CONNECTOR_SOURCE_COMPONENTS}" SDK_COMPONENTS="sdk-python sdk-node sdk-go sdk-csharp sdk-java" From a2a4a2cb987b7c92d232c073c2138ce1b2237251 Mon Sep 17 00:00:00 2001 From: Krishna Vishal Date: Wed, 9 Sep 2026 03:16:18 +0530 Subject: [PATCH 089/182] fix(consensus): keep a replica off a hole in its committed prefix (#4073) A replica could be promoted to primary while missing operations the cluster had already committed. The promotion coverage scan opened at the merged commit point and never looked below it, so a hole under that point went unseen. The replica then served reads from incomplete state and walked commit_min over it, failing a swallowed assert. The scan, its repair requests and its retries now open at the lower of the merged commit point and commit_min + 1, and the scan reads the evicted repair ring, so eviction no longer looks like a gap. Repair sessions clear when satisfied, reopen where they were armed, and rotate their sources as a ring. Commit advances only over a contiguous run from commit_min + 1, and both journal walks stop below the pipeline head, the only entry that can answer the caller. A view change re-decides the recovery barrier it can truncate, so no primary waits out operations nobody will re-prepare. The simulator compares partition journal contents, not only commit positions, and fails a run on a wedged barrier, a commit hold that never clears, or a fenced partition task. --- core/consensus/src/impls.rs | 440 +++++++-- core/consensus/src/plane_helpers.rs | 169 +++- core/metadata/src/impls/metadata.rs | 18 + core/partitions/src/iggy_partition.rs | 150 ++- core/server/src/http/reads.rs | 10 +- core/shard/src/lib.rs | 983 ++++++++++++++++--- core/shard/src/router.rs | 9 + core/simulator/src/bin/workload-fuzz.rs | 44 +- core/simulator/src/lib.rs | 138 ++- core/simulator/src/workload/invariants.rs | 164 +++- core/simulator/src/workload/oracle.rs | 14 + core/simulator/src/workload/state_checker.rs | 170 +++- 12 files changed, 2034 insertions(+), 275 deletions(-) diff --git a/core/consensus/src/impls.rs b/core/consensus/src/impls.rs index 159e0ca5e6..0d1ebce944 100644 --- a/core/consensus/src/impls.rs +++ b/core/consensus/src/impls.rs @@ -1729,6 +1729,41 @@ impl> VsrConsensus { self.recovery_barrier.set(required_commit); } + /// Re-decide the barrier against a log head the cluster just settled. + /// + /// Boot arms it at the recovered journal head: those ops were acked before the + /// restart, so admitting writes before they re-commit rolls back committed + /// history. It otherwise clears only by `commit_max` passing it, which never + /// happens when a view change discards the suffix instead of re-committing it. + /// The boot re-pipeline already ran, so nothing re-prepares those ops, + /// `is_caught_up_primary` stays shut, and the primary drops the very requests + /// that would raise `commit_max`. + /// + /// Call this wherever the head is authoritatively re-decided: a merged log at + /// view start, an adopted `StartView`. `head` lowers the barrier when the view + /// truncated the suffix, keeps it when the suffix survived. + /// + /// Lowered, never cleared. `is_caught_up_primary` reads it against `commit_max` + /// and a met barrier costs it nothing, but `await_recovery_barrier` reads it + /// against `commit_min`, and adoption raises `commit_max` before walking the + /// suffix into the state machine. Zeroing a met barrier would open that read + /// gate over the unapplied window it exists to hold. + /// + /// `head == 0` returns instead of lowering: zero is the DISARMED value, not a + /// met barrier. `barrier_state` (`server/src/http/reads.rs`) reads zero as + /// "nothing gated" and skips the `commit_min` comparison, so writing it would + /// open the gate rather than lower it. Reachable on the wire: + /// `adopt_start_view_suffix` returns `StartViewHeader.op` verbatim when the + /// suffix fails to decode, `validate` does not require `op >= 1`, and the + /// low-op guard binds only while `msg_view == log_view`. + fn redecide_recovery_barrier(&self, head: u64) { + let barrier = self.recovery_barrier.get(); + if barrier == 0 || head == 0 { + return; + } + self.recovery_barrier.set(barrier.min(head)); + } + /// Deadline paired with [`Self::recovery_barrier`]; only meaningful while the /// barrier is armed (non-zero). #[must_use] @@ -3363,6 +3398,9 @@ impl> VsrConsensus { // frame, and either value leaves this replica chasing an unservable head. let announced = self.adopt_start_view_suffix(header, suffix_body); self.sequencer.set_sequence(announced); + // Settle a gated suffix's fate as a backup too, so a later election inherits + // a decided barrier rather than a latched one. + self.redecide_recovery_barrier(announced); // Update timeouts for normal backup operation { @@ -3654,20 +3692,46 @@ impl> VsrConsensus { /// and the view change is blocked on the round-trip. #[must_use] pub fn pending_view_body_sources(&self, op: u64) -> Vec { + self.pending_view_sources(|dvc| { + dvc.suffix + .index_of(dvc.op, op) + .is_some_and(|index| dvc.suffix.offers_body(index)) + }) + } + + /// Recorded `DoViewChange` senders that `serves` accepts, most-recent-`log_view` + /// first, never this replica. + /// + /// Ordering and self-exclusion are the same for every source list; only the + /// "can this sender serve the op" predicate differs. + fn pending_view_sources(&self, serves: impl Fn(&StoredDvc) -> bool) -> Vec { let quorum = self.do_view_change_from_all_replicas.borrow(); let mut sources: Vec<(u32, u8)> = dvc_iter(&quorum) .filter(|dvc| dvc.replica != self.replica) - .filter_map(|dvc| { - let index = dvc.suffix.index_of(dvc.op, op)?; - dvc.suffix - .offers_body(index) - .then_some((dvc.log_view, dvc.replica)) - }) + .filter(|dvc| serves(dvc)) + .map(|dvc| (dvc.log_view, dvc.replica)) .collect(); sources.sort_unstable_by_key(|(log_view, _)| std::cmp::Reverse(*log_view)); sources.into_iter().map(|(_, replica)| replica).collect() } + /// Replicas that committed `op`, most-recent-`log_view` first. + /// + /// The fallback for an op below the DVC suffixes. A suffix spans `commit..=op`, + /// so [`Self::pending_view_body_sources`] answers nothing about the committed + /// prefix and a merged log whose coverage gap sits there would have no source + /// at all. A sender that committed the op either still journals it or has + /// compacted it under a checkpoint, and both answers move the requester + /// forward: the prepare, or the `RangeEvicted` that says repair cannot close + /// this gap. + /// + /// Presence is not proven the way an offered body is, so prefer + /// [`Self::pending_view_body_sources`] wherever it returns anything. + #[must_use] + pub fn pending_view_commit_sources(&self, op: u64) -> Vec { + self.pending_view_sources(|dvc| dvc.commit >= op) + } + /// Finish the parked view change: this replica's journal now covers the merged /// log, so it can serve any op it is about to announce. /// @@ -3703,6 +3767,8 @@ impl> VsrConsensus { self.status.set(Status::Normal); self.ceded_primaryship.set(false); self.sequencer.set_sequence(new_op); + // Only place a promotion learns what the merge did to a gated suffix. + self.redecide_recovery_barrier(new_op); if let Some(head) = merged.headers.first() { // Keep the hash chain continuous: the next prepare must chain onto the // head this view adopted, not onto whatever was appended last. @@ -4418,31 +4484,48 @@ mod pipeline_entry_tests { } } +/// Fixtures every consensus test module needs. #[cfg(test)] -mod timestamp_clamp_tests { - //! Pin the monotonic-floor contract: a new primary must never stamp a - //! prepare below timestamps already in the replicated log, even when its - //! wall clock lags the predecessor's. - - use super::*; - use crate::LocalPipeline; - use message_bus::BusMessage; +pub mod test_bus { + use super::{Command, METADATA_GROUP, Message, StartViewHeader}; + use message_bus::{BusMessage, MessageBus}; use server_common::MESSAGE_ALIGN; use server_common::iobuf::Frozen; - /// Clock frozen at a fixed instant, standing in for a lagging wall - /// clock on a freshly elected primary. - struct FixedClock(u64); - - impl clock::Clock for FixedClock { - type Realtime = IggyTimestamp; - - fn realtime(&self) -> Self::Realtime { - IggyTimestamp::from(self.0) - } + /// A `StartView` at `view` announcing head `op`. `commit == op` is the steady + /// case (no suffix); a lower `commit` keeps an uncommitted suffix. + /// + /// # Panics + /// Never: a zeroed buffer of the right size is a valid `StartViewHeader`. + #[must_use] + #[allow(clippy::cast_possible_truncation)] + pub fn make_start_view( + view: u32, + op: u64, + commit: u64, + replica: u8, + incarnation: u128, + ) -> Message { + let size = std::mem::size_of::(); + let mut msg = Message::::new(size); + let header = bytemuck::checked::try_from_bytes_mut::( + &mut msg.as_mut_slice()[..size], + ) + .expect("zeroed bytes are a valid StartViewHeader"); + header.command = Command::StartView; + header.cluster = 1; + header.view = view; + header.op = op; + header.commit = commit; + header.replica = replica; + header.incarnation = incarnation; + header.group = METADATA_GROUP; + header.size = size as u32; + msg } - struct NoopBus; + /// A [`MessageBus`] that accepts everything and remembers nothing. + pub struct NoopBus; impl MessageBus for NoopBus { async fn send_to_client( @@ -4466,6 +4549,30 @@ mod timestamp_clamp_tests { fn set_client_forward_fn(&self, _f: message_bus::ClientForwardFn) {} fn track_background(&self, _handle: message_bus::JoinHandle<()>) {} } +} + +#[cfg(test)] +mod timestamp_clamp_tests { + //! Pin the monotonic-floor contract: a new primary must never stamp a + //! prepare below timestamps already in the replicated log, even when its + //! wall clock lags the predecessor's. + + use super::*; + use crate::LocalPipeline; + + /// Clock frozen at a fixed instant, standing in for a lagging wall + /// clock on a freshly elected primary. + struct FixedClock(u64); + + impl clock::Clock for FixedClock { + type Realtime = IggyTimestamp; + + fn realtime(&self) -> Self::Realtime { + IggyTimestamp::from(self.0) + } + } + + use crate::test_bus::{NoopBus, make_start_view}; #[test] fn observed_log_timestamp_floors_new_primary_stamps() { @@ -4519,31 +4626,6 @@ mod timestamp_clamp_tests { ); } - #[allow(clippy::cast_possible_truncation)] - fn make_start_view( - view: u32, - op: u64, - replica: u8, - incarnation: u128, - ) -> Message { - let size = std::mem::size_of::(); - let mut msg = Message::::new(size); - let header = bytemuck::checked::try_from_bytes_mut::( - &mut msg.as_mut_slice()[..size], - ) - .expect("zeroed bytes are a valid StartViewHeader"); - header.command = Command::StartView; - header.cluster = 1; - header.view = view; - header.op = op; - header.commit = op; - header.replica = replica; - header.incarnation = incarnation; - header.group = METADATA_GROUP; - header.size = size as u32; - msg - } - #[test] fn given_recovering_replica_when_start_view_incarnation_foreign_should_ignore() { // A StartView addressed to a PREVIOUS incarnation, still in flight when the @@ -4571,7 +4653,7 @@ mod timestamp_clamp_tests { assert_eq!(consensus.status(), Status::Recovering); // Same view, head behind ours, foreign incarnation: ignored. - let stale = make_start_view(1, 4, 1, STALE); + let stale = make_start_view(1, 4, 4, 1, STALE); assert!( consensus .handle_start_view(PlaneKind::Metadata, stale.header(), &[]) @@ -4591,7 +4673,7 @@ mod timestamp_clamp_tests { // Same view and head but echoing our current incarnation: adopted, since // the match proves the reply post-dates our restart. - let fresh = make_start_view(1, 4, 1, CURRENT); + let fresh = make_start_view(1, 4, 4, 1, CURRENT); assert!( !consensus .handle_start_view(PlaneKind::Metadata, fresh.header(), &[]) @@ -4688,7 +4770,7 @@ mod timestamp_clamp_tests { consensus .handle_start_view( PlaneKind::Metadata, - make_start_view(7, 104, 1, 0).header(), + make_start_view(7, 104, 104, 1, 0).header(), &[] ) .is_empty(), @@ -4706,7 +4788,7 @@ mod timestamp_clamp_tests { !consensus .handle_start_view( PlaneKind::Metadata, - make_start_view(7, 105, 1, 0).header(), + make_start_view(7, 105, 105, 1, 0).header(), &[] ) .is_empty(), @@ -5075,6 +5157,8 @@ mod vsr_consensus_tests { // now measuring op 2 rather than carrying op 1's elapsed ticks. consensus.advance_commit_max(1); assert_eq!(drain_committable_prefix(&consensus).len(), 1); + // As real callers do, per entry: the next drain starts at the op now owed. + consensus.advance_commit_min(1); assert!( prepare_ticking(&consensus), "a remaining prepare keeps the timer armed" @@ -5082,6 +5166,7 @@ mod vsr_consensus_tests { consensus.advance_commit_max(2); assert_eq!(drain_committable_prefix(&consensus).len(), 1); + consensus.advance_commit_min(2); assert!( !prepare_ticking(&consensus), "draining the last prepare disarms the timer without waiting for it to fire" @@ -5335,34 +5420,8 @@ mod quorum_tests { use super::*; use crate::LocalPipeline; - use message_bus::BusMessage; - use server_common::MESSAGE_ALIGN; - use server_common::iobuf::Frozen; - struct NoopBus; - - impl MessageBus for NoopBus { - async fn send_to_client( - &self, - _client_id: u128, - _data: impl Into, - ) -> Result<(), message_bus::SendError> { - Ok(()) - } - - async fn send_to_replica( - &self, - _replica: u8, - _data: Frozen, - ) -> Result<(), message_bus::SendError> { - Ok(()) - } - - fn set_connection_lost_fn(&self, _f: message_bus::ConnectionLostFn) {} - fn set_replica_forward_fn(&self, _f: message_bus::ReplicaForwardFn) {} - fn set_client_forward_fn(&self, _f: message_bus::ClientForwardFn) {} - fn track_background(&self, _handle: message_bus::JoinHandle<()>) {} - } + use crate::test_bus::NoopBus; fn consensus_with_replica_count(replica_count: u8) -> VsrConsensus { VsrConsensus::new( @@ -5404,3 +5463,222 @@ mod quorum_tests { } }; } + +#[cfg(test)] +mod view_source_tests { + //! Who a primary-elect may ask for an op its merged log names. A `DoViewChange` + //! suffix spans `commit..=op`, so the two selectors cover disjoint halves of + //! the merged log and the split is what keeps a coverage gap under the merged + //! commit point askable at all. + + use super::*; + use crate::LocalPipeline; + use crate::view_change_quorum::{DvcSuffix, StoredDvc, dvc_record}; + + use crate::test_bus::NoopBus; + + fn consensus() -> VsrConsensus { + VsrConsensus::new(1, 0, 3, METADATA_GROUP, NoopBus, LocalPipeline::new()) + } + + /// A sender whose suffix runs `commit..=op` with every body offered, which is + /// the widest window a real `DoViewChange` can carry. + fn sender(replica: u8, log_view: u32, op: u64, commit: u64) -> StoredDvc { + let headers: Vec = (commit..=op) + .rev() + .map(|op| PrepareHeader { + command: Command::Prepare, + op, + view: log_view, + ..Default::default() + }) + .collect(); + let present = (1u128 << headers.len()) - 1; + StoredDvc { + replica, + log_view, + op, + commit, + suffix: DvcSuffix::new(headers, 0, present), + } + } + + fn record(consensus: &VsrConsensus, senders: [StoredDvc; 2]) { + let mut quorum = consensus.do_view_change_from_all_replicas.borrow_mut(); + for dvc in senders { + assert!(dvc_record(&mut quorum, dvc)); + } + } + + #[test] + fn given_an_op_inside_the_suffixes_when_selecting_should_return_the_body_offers() { + let consensus = consensus(); + record(&consensus, [sender(1, 5, 12, 10), sender(2, 4, 12, 10)]); + + assert_eq!( + consensus.pending_view_body_sources(11), + vec![1, 2], + "both senders offer op 11, freshest log_view first" + ); + } + + #[test] + fn given_an_op_below_every_commit_point_when_selecting_should_need_the_committers() { + let consensus = consensus(); + record(&consensus, [sender(1, 5, 12, 10), sender(2, 4, 12, 10)]); + + assert!( + consensus.pending_view_body_sources(7).is_empty(), + "a suffix spans commit..=op, so it says nothing about op 7" + ); + assert_eq!( + consensus.pending_view_commit_sources(7), + vec![1, 2], + "a sender that committed op 7 holds it or compacted it, and either \ + answer moves the requester forward" + ); + } + + #[test] + fn given_a_sender_behind_the_op_when_selecting_committers_should_skip_it() { + let consensus = consensus(); + record(&consensus, [sender(1, 5, 12, 10), sender(2, 6, 6, 5)]); + + assert_eq!( + consensus.pending_view_commit_sources(7), + vec![1], + "replica 2 never committed op 7, so asking it wastes a retry interval \ + on a RangeEvicted it has no standing to send" + ); + } + + #[test] + fn given_this_replica_in_the_quorum_when_selecting_should_never_return_self() { + let consensus = consensus(); + record(&consensus, [sender(0, 5, 12, 10), sender(2, 4, 12, 10)]); + + assert_eq!( + consensus.pending_view_commit_sources(7), + vec![2], + "a replica cannot repair from itself" + ); + } +} + +#[cfg(test)] +mod recovery_barrier_tests { + //! The gate holding a restarted replica's reads and writes until the recovered + //! WAL suffix re-commits. Boot arms it at the recovered head; only a view that + //! settles that suffix's fate may move it, and only downward. + + use super::*; + use crate::LocalPipeline; + use crate::test_bus::{NoopBus, make_start_view}; + + /// Recovered at head 120, proven committed only through 100, in view 7. + fn recovered_with_gated_suffix() -> VsrConsensus { + let mut consensus = + VsrConsensus::new(1, 0, 3, METADATA_GROUP, NoopBus, LocalPipeline::new()); + consensus.set_view(7); + consensus.set_log_view(7); + consensus.sequencer().set_sequence(120); + consensus.restore_commit_state(100, 100); + consensus.set_recovery_barrier(120); + consensus + } + + /// The `commit_max >= recovery_barrier` clause of `is_caught_up_primary`, read + /// directly so these tests need not satisfy the primary/status clauses. + fn write_gate_open(consensus: &VsrConsensus) -> bool + where + P: Pipeline, + { + consensus.commit_max() >= consensus.recovery_barrier() + } + + /// The wedge `redecide_recovery_barrier` exists for: a view change discards the + /// recovered suffix, the replica later wins an election, and a barrier pinned + /// to a head that no longer exists shuts admission. + #[test] + fn given_a_discarded_suffix_when_adopting_a_view_should_lower_the_barrier() { + let consensus = recovered_with_gated_suffix(); + + // The view's head is 105: 106..=120 committed nowhere, so the view drops + // them and nothing re-prepares them. + assert!( + !consensus + .handle_start_view( + PlaneKind::Metadata, + make_start_view(7, 105, 105, 1, 0).header(), + &[] + ) + .is_empty(), + "the StartView at the commit floor must be adopted" + ); + assert_eq!( + consensus.recovery_barrier(), + 105, + "the adopted head settled the suffix's fate, so the barrier must fall to \ + it rather than latch at a head the view discarded" + ); + assert!( + write_gate_open(&consensus), + "a barrier the adopted commit point covers must stop gating admission" + ); + assert!( + consensus.commit_min() < consensus.recovery_barrier(), + "adoption raises commit_max before applying, so the local read gate \ + (which reads commit_min) must still hold" + ); + } + + /// The other half: lowering the barrier to the adopted head must not read as + /// clearing it while the suffix survives. + #[test] + fn given_a_surviving_suffix_when_adopting_a_view_should_keep_the_barrier() { + let consensus = recovered_with_gated_suffix(); + + // Head 120, commit still 100: 101..=120 re-replicate under the new view. + assert!( + !consensus + .handle_start_view( + PlaneKind::Metadata, + make_start_view(7, 120, 100, 1, 0).header(), + &[] + ) + .is_empty(), + "the StartView carrying the surviving suffix must be adopted" + ); + assert_eq!( + consensus.recovery_barrier(), + 120, + "a suffix the view kept is still unproven, so its gate must stand" + ); + assert!( + !write_gate_open(&consensus), + "an unproven suffix must keep admission shut" + ); + } + + #[test] + fn given_no_recovered_suffix_when_redeciding_should_stay_disarmed() { + // Nothing gated, so no view change may invent a gate. + let consensus = VsrConsensus::new(1, 0, 3, METADATA_GROUP, NoopBus, LocalPipeline::new()); + consensus.redecide_recovery_barrier(9); + assert_eq!(consensus.recovery_barrier(), 0); + } + + /// Zero is disarmed, not met: `barrier_state` skips the `commit_min` comparison + /// on it, so writing it opens the HTTP read gate over the window the barrier + /// holds. + #[test] + fn given_a_zero_head_when_redeciding_should_leave_the_barrier_armed() { + let consensus = recovered_with_gated_suffix(); + consensus.redecide_recovery_barrier(0); + assert_eq!( + consensus.recovery_barrier(), + 120, + "a zero head must not disarm the gate" + ); + } +} diff --git a/core/consensus/src/plane_helpers.rs b/core/consensus/src/plane_helpers.rs index 195120af88..a72995cf08 100644 --- a/core/consensus/src/plane_helpers.rs +++ b/core/consensus/src/plane_helpers.rs @@ -433,10 +433,48 @@ where false } +/// Whether a repair session may still run. +/// +/// `Normal` is the ordinary case. The exception is a primary-elect that parked a +/// merged log: its repair runs in `ViewChange` by design, because the log it is +/// repairing toward is one nobody has started yet. Reading `is_normal` alone at an +/// ingest site drops the frame AND the session, so the coverage scan re-arms every +/// tick and nothing it fetches can ever land. +/// +/// Every site that fences a repair session on status must use this, so the arming +/// side and the ingest side cannot drift apart. +pub fn repair_session_live(consensus: &VsrConsensus) -> bool +where + B: MessageBus, + P: Pipeline, +{ + consensus.is_normal() + || (consensus.view_log_is_pending() && consensus.is_primary_for_view(consensus.view())) +} + /// Drain and return committable prepares from the pipeline head. /// -/// Entries are drained only from the head and only while their op is covered -/// by the current commit frontier. +/// Entries are drained from the head, while covered by the commit frontier, and +/// only as a contiguous run starting at the next op owed to the state machine. +/// +/// Callers `advance_commit_min` per entry, so a run starting above +/// `commit_min + 1` or breaking partway hits that counter's sequential-advance +/// assert. Pipeline-side twin of `commit_journal`'s gap-stop. +/// +/// Reported, not asserted: a promoted partition primary reaches it legitimately +/// while its journal walk clears the apply backlog. Repair arms on the level, and +/// a hold that never clears fails the simulator's contiguity invariant. +/// +/// Do NOT read that as "the journal walk always finishes". It can stall on the +/// partition plane: `committed_headers_from` reads resident headers only, while +/// `commit_messages` evicts up to `commit_max` (the cluster frontier), so ops past +/// the per-call cap can be flushed out from under the walk. Pre-existing; see +/// `IggyPartition::collect_committable_from_journal`. +/// +/// A head at or below `commit_min` is the other shape: applied already, no repair +/// owed, only a pop that can no longer happen. The journal walks stop below the +/// head so they cannot create it, but a `set_commit_floor` jump past a live +/// pipeline still can, so it is logged apart rather than treated as impossible. /// /// # Panics /// If `head()` returns `Some` but `pop()` returns `None` (unreachable). @@ -446,18 +484,31 @@ where P: Pipeline, { let commit = consensus.commit_max(); + let commit_min = consensus.commit_min(); let mut drained = Vec::new(); consensus.with_pipeline_mut(|pipeline| { + let mut next = commit_min + 1; while let Some(head_op) = pipeline.head().map(|entry| entry.header.op) { if head_op > commit { break; } + if head_op != next { + report_uncommittable_head( + consensus.replica(), + head_op, + commit_min, + commit, + drained.len(), + ); + break; + } let entry = pipeline .pop() .expect("drain_committable_prefix: head exists"); drained.push(entry); + next += 1; } }); @@ -473,7 +524,8 @@ where drained } -/// Header of the pipeline head, iff its op is covered by the commit frontier. +/// Header of the pipeline head, iff its op is the next one this replica owes its +/// state machine and is covered by the commit frontier. /// /// Peek-only counterpart of [`drain_committable_prefix`] for commit paths that /// must survive their driving future being canceled between "committable" and @@ -482,15 +534,68 @@ where /// revalidates that the head is still this exact entry before popping and /// applying it. A driver dropped at an await strands nothing; a sibling driver /// that committed the op first fails the caller's revalidation and re-peeks. +/// +/// Bounded below for the reason [`drain_committable_prefix`] is, and reported +/// rather than asserted for the same one. Holding is safe: a shard pump's panic is +/// swallowed by `compio::runtime::spawn`, while `tick_metadata` re-arms repair on +/// the level. pub fn peek_committable_head(consensus: &VsrConsensus) -> Option where B: MessageBus, P: Pipeline, { let commit = consensus.commit_max(); - consensus + let commit_min = consensus.commit_min(); + let head = consensus .pipeline_head_header() - .filter(|header| header.op <= commit) + .filter(|header| header.op <= commit)?; + if head.op != commit_min + 1 { + report_uncommittable_head(consensus.replica(), head.op, commit_min, commit, 0); + return None; + } + Some(head) +} + +/// Log a held pipeline head, with the remedy for the arm it is in. +/// +/// `debug`, not `warn`, matching `tick_partitions` / `tick_metadata`: a hold is the +/// steady state for a whole rejoin, so `warn` is one line per group per tick. A +/// hold that never clears is caught by a simulator invariant, not by this line. +fn report_uncommittable_head( + replica: u8, + head_op: u64, + commit_min: u64, + commit_max: u64, + drained: usize, +) { + if head_op > commit_min { + tracing::debug!( + replica, + head_op, + expected_op = commit_min + 1, + commit_min, + commit_max, + drained, + "committable head sits above a hole in the committed prefix; holding the commit \ + walk until the ops below it are journaled" + ); + } else { + // Reported, not called a defect: a commit-floor jump reaches this state + // legitimately. `IggyMetadata`'s state-transfer install and + // `IggyPartition::complete_repair` both raise `commit_min` past a live + // pipeline via `set_commit_floor` without the `clear_pipeline` the + // partition transfer path pairs with it. Nothing pops a head below the + // floor, so it holds until a view change clears the pipeline. + tracing::warn!( + replica, + head_op, + commit_min, + commit_max, + drained, + "committable head sits at or below the applied commit point; nothing can pop \ + or answer this entry until a view change clears the pipeline" + ); + } } /// Build reply for a committed prepare. @@ -1991,10 +2096,64 @@ mod tests { assert_eq!(sent[0].0, 1); } + /// `advance_commit_min` is strictly sequential, so draining op 7 with 6 never + /// applied panics the shard pump. Both gates must hold instead. + #[test] + fn given_a_hole_below_the_head_when_committing_should_hold_both_gates() { + let consensus = VsrConsensus::new(1, 0, 3, 0, NoopBus, LocalPipeline::new()); + consensus.init(); + consensus.restore_commit_state(5, 5); + + // Op 6 never arrived; 7 and 8 did, and the cluster committed through 8. + consensus.pipeline_message(PlaneKind::Metadata, &prepare_message(7, 0, 70)); + consensus.pipeline_message(PlaneKind::Metadata, &prepare_message(8, 70, 80)); + consensus.advance_commit_max(8); + + assert!( + peek_committable_head(&consensus).is_none(), + "the head is covered by the frontier but op 6 is not applied" + ); + assert!( + drain_committable_prefix(&consensus).is_empty(), + "the drain must hold on the same hole the peek does" + ); + assert_eq!( + consensus.pipeline_head_header().map(|header| header.op), + Some(7), + "holding must not consume the entry" + ); + } + + /// The shape the journal-walk caps prevent. Both gates must still refuse it: + /// re-applying an applied op panics `advance_commit_min`. + #[test] + fn given_an_applied_head_when_committing_should_refuse_it() { + let consensus = VsrConsensus::new(1, 0, 3, 0, NoopBus, LocalPipeline::new()); + consensus.init(); + consensus.restore_commit_state(4, 4); + + consensus.pipeline_message(PlaneKind::Metadata, &prepare_message(5, 0, 50)); + consensus.advance_commit_max(6); + // The journal walk got there first and applied 5 and 6 out of the WAL. + consensus.advance_commit_min(5); + consensus.advance_commit_min(6); + + assert!( + peek_committable_head(&consensus).is_none(), + "op 5 is already applied; re-applying it panics advance_commit_min" + ); + assert!( + drain_committable_prefix(&consensus).is_empty(), + "the drain must refuse an applied head too" + ); + } + #[test] fn drains_only_up_to_commit_frontier_even_without_quorum_flags() { let consensus = VsrConsensus::new(1, 0, 3, 0, NoopBus, LocalPipeline::new()); consensus.init(); + // Pipeline opens at op 5, so the state machine must already be through 4. + consensus.restore_commit_state(4, 4); consensus.pipeline_message(PlaneKind::Metadata, &prepare_message(5, 0, 50)); consensus.pipeline_message(PlaneKind::Metadata, &prepare_message(6, 50, 60)); diff --git a/core/metadata/src/impls/metadata.rs b/core/metadata/src/impls/metadata.rs index 3c4d19f47d..1522cfbb50 100644 --- a/core/metadata/src/impls/metadata.rs +++ b/core/metadata/src/impls/metadata.rs @@ -3653,6 +3653,24 @@ where applied += 1; let op = consensus.commit_min() + 1; + // Never apply the op the pipeline holds: `on_ack` pops that entry + // and advances past it, so applying it here strands it below + // `peek_committable_head`'s floor. Nothing pops it after that (the + // only `pop_committed_prepare` sits inside that loop), so its wire + // reply is never built and its `reply_sender` neither fires nor drops. + // + // Compared per op, not hoisted: the body read below awaits and a + // sibling driver can move the head. Comparing against `op` also gets + // the two edge cases right for free -- an absent head is a backup's + // empty pipeline and caps nothing, and a head at or below `commit_min` + // is already stranded and must not freeze the walk on top of that. + if consensus + .pipeline_head_header() + .is_some_and(|head| head.op == op) + { + break; + } + let Some(header) = journal.handle().header(op as usize) else { // Gap-stop: the walk halts at the first missing prepare and // resumes once it is refilled -- by the primary's retransmit diff --git a/core/partitions/src/iggy_partition.rs b/core/partitions/src/iggy_partition.rs index 5d363ef24c..e6a2219710 100644 --- a/core/partitions/src/iggy_partition.rs +++ b/core/partitions/src/iggy_partition.rs @@ -43,9 +43,9 @@ use consensus::{ PlaneKind, Project, ReplicaLogContext, RequestLogEvent, Sequencer, SimEventKind, VsrConsensus, ack_preflight, ack_quorum_reached, build_deny_reply_from_request, build_reply_from_request, build_reply_message, drain_committable_prefix, emit_namespace_progress_event, - emit_partition_diag, emit_sim_event, fence_old_prepare_by_commit, repaired_frontier_update, - replicate_frozen_to_next_in_chain, replicate_preflight, restamp_prepare_view, - send_prepare_ok as send_prepare_ok_common, verify_prepare_integrity, + emit_partition_diag, emit_sim_event, fence_old_prepare_by_commit, repair_session_live, + repaired_frontier_update, replicate_frozen_to_next_in_chain, replicate_preflight, + restamp_prepare_view, send_prepare_ok as send_prepare_ok_common, verify_prepare_integrity, }; use iggy_binary_protocol::requests::consumer_offsets::{ DeleteConsumerOffsetRequest, StoreConsumerOffsetRequest, @@ -4304,13 +4304,34 @@ where /// Committable entries (ops `commit_min+1 ..= commit_max`) read from the /// journal, for a backup whose pipeline is empty. Stops at the first missing /// op: a replication gap must not be skipped, or `advance_commit_min`'s - /// sequential contract breaks. Like the metadata plane's `commit_journal`, - /// the journal keeps its committed entries until they are flushed - /// (`commit_messages` drains only the committed prefix), so this read finds - /// every committed op while the uncommitted tail stays resident. + /// sequential contract breaks. + /// + /// KNOWN GAP: resident headers only. `commit_messages` evicts up to + /// `commit_max` (the cluster frontier, not this replica's commit point) while + /// `committed_headers_from` never reads the evicted ring, so a backlog past + /// [`COMMIT_WALK_OPS_MAX`] can have its un-reached ops flushed out from under + /// it and stop. Repair refetches them and the simulator's contiguity invariant + /// catches a walk that never recovers. Reading the ring here would close it + /// directly, but not as a one-line swap: the apply path needs batch bytes and + /// the ring is capacity-bounded, so headers it cannot back with bytes would + /// fence the partition instead of stalling it. fn collect_committable_from_journal(&self, max_ops: usize) -> Vec { let from_op = self.consensus.commit_min() + 1; + // Stop below the pipeline head. The drain above holds rather than pops + // when the head is not the op owed next; walking that op out of the + // journal instead advances `commit_min` past a still-resident entry, and + // `on_ack` then finds the drain empty on every later ack -- no reply is + // ever shipped and each stranded entry leaves its awaiter parked. + // + // Only a head at or above `from_op` lowers the ceiling: an absent head is + // a backup's empty pipeline, and a lower head is already stranded and must + // not freeze the walk on top of that. let commit_max = self.consensus.commit_max(); + let commit_max = self + .consensus + .pipeline_head_header() + .filter(|head| head.op >= from_op) + .map_or(commit_max, |head| commit_max.min(head.op - 1)); self.log .journal() .inner @@ -6708,7 +6729,11 @@ where return; }; let consensus = self.consensus(); - if !consensus.is_normal() || consensus.view() != session.view { + // NOT `is_normal` alone: a primary-elect repairing toward its parked + // merged log runs this in `ViewChange`, and dropping the session on its + // first inbound frame leaves the coverage scan re-arming every tick over + // a stream it can never keep. + if !repair_session_live(consensus) || consensus.view() != session.view { self.repair = None; return; } @@ -8806,28 +8831,40 @@ mod tests { assert!(!dir.path().join("2").exists()); } + /// `AckLevel::NoAck` stores apply on the primary only and never replicate, so + /// which replicas hold an offset is not agreed and a committed delete can + /// legitimately find nothing. Erroring on that fails the committed apply, + /// fences the partition, then crash-loops on every replay of the op. + /// + /// Both kinds, because they are separate maps with separate directories. #[compio::test] async fn given_absent_offset_file_when_delete_commits_should_skip_directory_sync() { - let dir = tempfile::tempdir().unwrap(); - let (mut partition, sent) = recording_partition_at(0, 3); - partition.consumer_offsets_path = - Some(dir.path().join("missing").to_string_lossy().into_owned()); - partition.stage_consumer_offset_delete(1, ConsumerKind::Consumer, 7); - partition.consensus.restore_commit_state(0, 1); - let header = PrepareHeader { - op: 1, - operation: Operation::DeleteConsumerOffset, - client: 42, - request: 1, - ..Default::default() - }; - partition - .handle_committed_entries(vec![PipelineEntry::new(header)], &repair_config(), true) - .await; - assert!(partition.fatal.is_none()); - assert_eq!(partition.consensus.commit_min(), 1); - assert_eq!(partition.offset_dir_sync_count.get(), 0); - assert_eq!(sent.borrow().len(), 1); + for (op, kind) in [ + (1, ConsumerKind::Consumer), + (2, ConsumerKind::ConsumerGroup), + ] { + let dir = tempfile::tempdir().unwrap(); + let (mut partition, sent) = recording_partition_at(0, 3); + let missing = Some(dir.path().join("missing").to_string_lossy().into_owned()); + partition.consumer_offsets_path.clone_from(&missing); + partition.consumer_group_offsets_path = missing; + partition.stage_consumer_offset_delete(op, kind, 7); + partition.consensus.restore_commit_state(op - 1, op); + let header = PrepareHeader { + op, + operation: Operation::DeleteConsumerOffset, + client: 42, + request: 1, + ..Default::default() + }; + partition + .handle_committed_entries(vec![PipelineEntry::new(header)], &repair_config(), true) + .await; + assert!(partition.fatal.is_none(), "{kind:?} delete must not fence"); + assert_eq!(partition.consensus.commit_min(), op); + assert_eq!(partition.offset_dir_sync_count.get(), 0); + assert_eq!(sent.borrow().len(), 1); + } } #[compio::test] @@ -11073,6 +11110,63 @@ mod tests { .expect("journal append"); } + /// Walking through the head advances `commit_min` past a resident entry only + /// `on_ack` can pop and answer, after which every later ack finds the drain + /// empty and no reply is ever shipped. + #[compio::test] + async fn given_a_pipeline_head_when_walking_the_journal_should_stop_below_it() { + let partition = test_partition(); + for op in 1..=4 { + journal_prepare(&partition, op, Operation::CreateStream).await; + } + partition.consensus.restore_commit_state(0, 4); + partition.consensus.pipeline_message( + PlaneKind::Partitions, + &pipeline_prepare(3, Operation::CreateStream), + ); + + let ops: Vec = partition + .collect_committable_from_journal(COMMIT_WALK_OPS_MAX) + .into_iter() + .map(|entry| entry.header.op) + .collect(); + assert_eq!( + ops, + vec![1, 2], + "the walk stops at the op the pipeline holds, leaving 3 to on_ack" + ); + } + + /// A backup journals replicated prepares and never populates a pipeline, so an + /// absent head must mean NO ceiling. Read as a ceiling of zero it would stop + /// every backup's commit walk. + #[compio::test] + async fn given_an_empty_pipeline_when_walking_the_journal_should_not_cap() { + let partition = test_partition(); + for op in 1..=3 { + journal_prepare(&partition, op, Operation::CreateStream).await; + } + partition.consensus.restore_commit_state(0, 3); + assert!(partition.consensus.pipeline_head_header().is_none()); + + let ops: Vec = partition + .collect_committable_from_journal(COMMIT_WALK_OPS_MAX) + .into_iter() + .map(|entry| entry.header.op) + .collect(); + assert_eq!(ops, vec![1, 2, 3], "a backup walks its whole committed run"); + } + + fn pipeline_prepare(op: u64, operation: Operation) -> Message { + let size = std::mem::size_of::(); + Message::::new(size).transmute_header(|_, header: &mut PrepareHeader| { + header.command = Command::Prepare; + header.op = op; + header.operation = operation; + header.size = u32::try_from(size).expect("prepare header size fits in u32"); + }) + } + /// A repaired `SendMessages` prepare with an explicit chain identity, as a /// serving peer ships it. pub(super) fn repaired_send_prepare( diff --git a/core/server/src/http/reads.rs b/core/server/src/http/reads.rs index dc3262ff37..bcb387aa1a 100644 --- a/core/server/src/http/reads.rs +++ b/core/server/src/http/reads.rs @@ -276,17 +276,23 @@ pub(in crate::http) async fn await_recovery_barrier( let Some(consensus) = shard.plane.metadata().consensus.as_ref() else { return Ok(()); }; - let barrier = consensus.recovery_barrier(); // Gate on commit_MIN (locally applied), not commit_max (known committed): // a StartView adoption advances commit_max first and only then walks the // journal applying ops, and this task interleaves with that walk at its // await points -- a commit_max gate would serve state from before the // suffix applied (e.g. a pre-restart password change not yet visible). - if barrier_state(barrier, consensus.commit_min(), false) == BarrierWait::Ready { + if barrier_state(consensus.recovery_barrier(), consensus.commit_min(), false) + == BarrierWait::Ready + { return Ok(()); } let deadline = std::time::Instant::now() + consensus.recovery_deadline(); loop { + // Re-read per poll, like `commit_min`. `redecide_recovery_barrier` lowers + // the barrier when a view change settles the recovered suffix, so a reader + // holding the value it captured on entry waits out a barrier that no longer + // exists and then 503s on a replica that is serving. + let barrier = consensus.recovery_barrier(); let expired = std::time::Instant::now() >= deadline; match barrier_state(barrier, consensus.commit_min(), expired) { BarrierWait::Ready => return Ok(()), diff --git a/core/shard/src/lib.rs b/core/shard/src/lib.rs index 6f6f9d546c..ae32c2261e 100644 --- a/core/shard/src/lib.rs +++ b/core/shard/src/lib.rs @@ -906,6 +906,14 @@ pub const REPAIR_CHUNK_MAX: u64 = 128; #[derive(Debug, Clone, Copy)] struct MetadataRepairSession { nonce: u128, + /// Lowest op this session must fetch, and the floor its stall retry reopens at. + /// + /// Not re-derivable from `commit_min + 1`: the merged-log scan opens at + /// [`merged_log_scan_floor`], above the snapshot floor, while its + /// `committed_elsewhere` fallback reports ops below even that. A recomputed + /// retry asks for a different window than the one reported missing, and never + /// re-asks for the op that was. + from_op: u64, to_op: u64, /// Consensus view this session was armed in. A later view decides the log /// again, so the window this names may no longer be the one to fetch; @@ -1585,11 +1593,28 @@ where /// [`Self::metadata_transfer_decode_failures`]. metadata_transfer_attempts: Cell, - /// Consecutive stalled re-requests on the live metadata repair session, - /// against [`partitions::REPAIR_MAX_STALL_RETRIES`]. Survives the session, - /// so rotating the peer cannot reset it; cleared by an accepted repaired - /// prepare. - metadata_repair_attempts: Cell, + /// Consecutive stall rounds burned by the metadata repair session, against + /// [`partitions::REPAIR_MAX_STALL_RETRIES`]. Bounds how long one quiet peer + /// pins the commit walk. + /// + /// On the shard, not the session: rotation mints a fresh session, so a + /// per-session counter would reset itself. Nothing on the rotation path may + /// clear it either. + /// + /// Cleared only by [`Self::note_metadata_repair_walked`], which takes evidence + /// attributable to the targeted peer. A repair prepare carries no sender and + /// no nonce, so it restarts the stall clock only. Net effect: this bounds peers + /// that go silent, serve unusable bytes, or terminate without closing the gap. + /// It does not bound merely slow peers, whose chunks keep the clock from + /// firing. + /// + /// Paired with the view the rounds were charged in, and spent per view. The + /// merged-log arm refuses to re-arm once the budget is out, and the exhausting + /// path leaves no session behind, so nothing would ever be superseded or walk: + /// unfenced, one spent budget would refuse merged-log repair for every later + /// view for the life of the process, and a replica that keeps winning + /// elections would never repair again. + metadata_repair_attempts: Cell<(u32, u32)>, /// Decode failures charged against one snapshot generation, as /// `(snapshot_seq, failures)`. `None` until a pulled artifact set first @@ -1744,7 +1769,7 @@ where superblock_wedged_fatal_failures: Cell::new(0), bus_max_message_size: Cell::new(DEFAULT_BUS_MAX_MESSAGE_SIZE), metadata_transfer_attempts: Cell::new(0), - metadata_repair_attempts: Cell::new(0), + metadata_repair_attempts: Cell::new((0, 0)), metadata_transfer_decode_failures: Cell::new(None), }) } @@ -2231,7 +2256,7 @@ where superblock_wedged_fatal_failures: Cell::new(0), bus_max_message_size: Cell::new(DEFAULT_BUS_MAX_MESSAGE_SIZE), metadata_transfer_attempts: Cell::new(0), - metadata_repair_attempts: Cell::new(0), + metadata_repair_attempts: Cell::new((0, 0)), metadata_transfer_decode_failures: Cell::new(None), } } @@ -5037,6 +5062,12 @@ where // Applies to both planes, and is why a backup parks a log at all. The // view already decided which prepare belongs at this op; a different // one forks the log. An op the parked log omits is unconstrained. + // + // Which covers most of the range under `pending.commit_max`: the merged + // log names headers only from the DVC suffixes, and those span + // `commit..=op` per sender. Identity below that rests on crash-stop -- + // a committed op is the quorum's op. `verify_prepare_integrity` below + // guards corruption; neither guards Byzantine faults. let disagrees = consensus .with_pending_view_log(|pending| { pending @@ -5069,15 +5100,19 @@ where let Some(journal) = planes.0.journal.as_ref() else { return; }; - // Above the two returns below, not after them: only SILENCE should - // age the stream, and an in-scope frame proves the peer is serving - // whether or not this replica still needs the op it carries. The - // ops a re-request re-serves are exactly the ones already held, so - // counting accepted frames alone rotates away from a live peer. - if let Some(session) = self.metadata_repair.borrow_mut().as_mut() { - session.idle_ticks = 0; - } - self.note_metadata_repair_progress(); + // Below the divergence and integrity returns, above the two under it. + // + // Only silence should age the stream, and only a frame this replica + // would have accepted proves anything is being served. A forked or + // corrupted frame is neither: the peer re-serves the same stored bytes + // every re-request, so crediting those holds off the retry forever. + // + // Still above the dedup return: a re-request re-serves ops already + // held, and a stream re-covering ground is still a stream. + // + // Clock only -- the frame has no sender and no nonce, so it cannot be + // attributed. The budget is cleared from the terminator and the walk. + self.note_metadata_repair_clock(); let journal = journal.handle(); #[allow(clippy::cast_possible_truncation)] if journal.header(header.op as usize).is_some() { @@ -5190,6 +5225,14 @@ where done, "metadata journal repair walked" ); + // The one attributable credit: fenced on `session.nonce` + // above, so it came from the targeted peer, and the walk moved. + // Gating on the walk and not the frame keeps a peer that + // terminates every round while serving nothing useful from + // clearing its own budget and becoming un-rotatable. + if done || commit_min > before { + self.note_metadata_repair_walked(); + } if done { *self.metadata_repair.borrow_mut() = None; } else if repair_chunk_walked(before, commit_min, header.op) { @@ -5225,6 +5268,95 @@ where // `RangeEvicted` again if the primary checkpointed mid // transfer -- that reraises through the same path, and each // round lifts the local floor, so it converges. + // + // Never as primary-elect. A transfer replaces snapshot-shaped + // state wholesale, and this replica has a merged log parked + // against that state naming ops it was just told it cannot + // serve; installing under it starts the view over a log the new + // state no longer matches. Re-target instead, and let the + // view-change timeout escalate if nobody can serve it. + // + // DROP the session before returning. The serving peer follows + // `RangeEvicted` with `RepairDone(from_op - 1)` on the same + // nonce, and a `RepairDone` at or below `commit_min` walks + // nothing, so `repair_chunk_walked` is trivially true and that + // arm re-requests at once: no tick gate, no debounce, no attempt + // burned -- an unthrottled request/reply loop across two pumps. + // Dropping first lands the trailing frame on the `is_none` + // guard, as it did before this arm existed. + if consensus.view_log_is_pending() + && consensus.is_primary_for_view(consensus.view()) + { + tracing::warn!( + shard = self.id, + peer = header.replica, + retained_from = header.op, + local_commit = consensus.commit_min(), + "merged-log repair peer evicted the requested range; \ + re-targeting rather than transferring state mid view change" + ); + // Definitive, not a stall: this sender has said it cannot + // serve the window, so rotate now rather than spend a retry + // interval on a stream that will not come. Still charge a + // round, so a quorum that all answer this way stops asking + // instead of cycling the ring until the timeout. + if self.burn_metadata_repair_attempt(consensus.view()) { + *self.metadata_repair.borrow_mut() = None; + tracing::warn!( + shard = self.id, + from_op = session.from_op, + to_op = session.to_op, + "merged-log repair exhausted its senders; leaving the view \ + change to its timeout" + ); + return; + } + self.rotate_stalled_metadata_repair( + consensus, + header.replica, + session.from_op, + session.to_op, + ) + .await; + return; + } + + // The floor must also be ABOVE the op this replica needs. A + // peer behind the requested window walks its serve range off + // the end and answers `RangeEvicted` at the requested floor + // itself, having retained nothing and evicted nothing; + // converting on that arms a transfer against a replica with + // less state than this one and fences repair for a full + // transfer backoff. Drop the session and let the level trigger + // re-request from the primary instead. + if header.op <= consensus.commit_min() + 1 { + tracing::warn!( + shard = self.id, + peer = header.replica, + retained_from = header.op, + local_commit = consensus.commit_min(), + "metadata repair peer retained nothing in the requested range; \ + re-requesting rather than converting to state transfer" + ); + // Charge the round, and ACT on exhaustion. `gap_repair_peer` + // re-picks the primary deterministically, so dropping the + // session on its own re-arms the same peer at the debounce + // interval forever: the stall path never runs, so rotation + // is never reached and the state-transfer escalation this + // guard replaced stays out of reach. + if self.burn_metadata_repair_attempt(consensus.view()) { + self.rotate_stalled_metadata_repair( + consensus, + header.replica, + session.from_op, + session.to_op, + ) + .await; + return; + } + *self.metadata_repair.borrow_mut() = None; + return; + } if consensus.state_transfer_stage() == consensus::StateTransferStage::Idle { *self.metadata_repair.borrow_mut() = None; consensus.begin_state_transfer_await(); @@ -5269,7 +5401,12 @@ where if header.nonce != session.nonce { return; } - if !partition.consensus().is_normal() || partition.consensus().view() != session.view { + // Twin of the `apply_repaired_prepare` gate: a primary-elect's merged-log + // session legitimately runs outside `Normal`, and dropping it here would + // discard the terminator that closes the window it is repairing. + if !consensus::repair_session_live(partition.consensus()) + || partition.consensus().view() != session.view + { partition.repair = None; return; } @@ -5559,11 +5696,23 @@ where // cannot serve the state transfer it itself needs. A committed op // cannot diverge from the merged log, and its bytes stay serveable // from the evicted ring or the flushed segments. + let floor = ScanFloor { + repair_floor: consensus.commit_min(), + commit_min: consensus.commit_min(), + }; + // Ring AND resident, one pass. `header_by_op` reads the resident vec + // only, while `commit_messages` evicts up to `commit_max` (the cluster + // frontier), so a primary-elect with an apply backlog reads `None` for + // ops it holds in the repair ring and parks on a hole that is not one. + // One pass also because `header_by_op` is a linear scan and this window + // is the apply backlog, not the `prepare_queue_max` span the merge + // bounds -- probing per op is quadratic. let missing = { let journal = partition.log.journal(); - first_op_not_covered(&pending, consensus.commit_min(), |op| { - journal.inner.header_by_op(op) - }) + let window = journal + .inner + .repair_headers_in(floor.opens_at(&pending)..=pending.op_head); + first_op_not_covered(&pending, floor, |op| window.get(&op).copied()) }; if let Some(missing_op) = missing { tracing::debug!( @@ -5573,6 +5722,12 @@ where op_head = pending.op_head, "partition view change waiting on op {missing_op} before starting the view" ); + // And a way out of the wait: nothing else fetches this op. + // `maybe_request_partition_repair` refuses outside `Normal` and + // the sweep's gap detector needs `probe.normal`. Partition twin of + // the metadata plane's view repair. + self.request_partition_view_repair(partition, missing_op, pending.op_head, None) + .await; return; } @@ -5606,6 +5761,11 @@ where /// /// Repair frames are fire-and-forget, so a lost one leaves the session armed /// forever with the commit walk pinned below the frontier. + /// + /// Re-requests from the SAME peer while its stall budget holds. A peer that + /// never answers is a different problem, and + /// [`Self::rotate_stalled_metadata_repair`] owns it: the target lives on the + /// session, so it must be replaced there rather than shadowed for one send. #[allow(clippy::future_not_send)] async fn retry_stalled_metadata_repair

(&self, consensus: &VsrConsensus) where @@ -5640,7 +5800,7 @@ where "metadata repair session walked or superseded; closing it" ); *self.metadata_repair.borrow_mut() = None; - self.note_metadata_repair_progress(); + self.note_metadata_repair_walked(); return; } @@ -5654,55 +5814,32 @@ where return None; } session.idle_ticks = 0; - Some((session.peer, session.nonce, session.to_op)) + Some((session.peer, session.nonce, session.from_op, session.to_op)) }) }; - if let Some((peer, nonce, to_op)) = stalled { - // A session pins its peer and fences every arming site while it - // stands, so a peer that cannot answer wedges the plane harder than - // having no session at all -- and the gap-stopped-primary rotation - // can pick a peer that is simply down. Past the budget the session - // is dropped and re-armed one step around the ring; an ordinary lost - // frame is re-requested long before that. - if self.burn_metadata_repair_attempt() { - let next_peer = next_transfer_peer( - consensus.replica(), - peer, - consensus.replica_count(), - consensus.primary_index(consensus.view()), - ); - tracing::warn!( + if let Some((peer, nonce, session_from_op, to_op)) = stalled { + let from_op = + stalled_repair_from_op(session_from_op, consensus.commit_min(), repairing_view); + if from_op > to_op { + // `from_op` past `to_op` without `commit_min` reaching it: the + // primary-elect window above starts at the merged log's commit + // point, which can sit above what this replica has walked. The + // top-of-tick check closes the ordinary case; this closes the + // one it cannot see. Leaving it armed wedges the replica: no + // `RepairDone` clears a window the walk is already past, and the + // `is_none` gate then blocks the session the ops above it need. + tracing::info!( shard = self.id, - peer, - next_peer, to_op, - "metadata repair stalled past its retry budget; re-arming from another \ - replica" + peer, + "metadata repair window fully requested; closing the stalled session" ); *self.metadata_repair.borrow_mut() = None; - self.note_metadata_repair_progress(); - if next_peer != peer { - self.maybe_request_metadata_repair(consensus, next_peer) - .await; - } - // Nobody else to name (a solo group, or a two-replica group - // whose only peer went quiet): dropping the session is still - // right, since it unfences the detector, which re-arms after its - // debounce and logs the state each interval. - return; - } - // Primary-elect only. Its window starts at the merged log's commit - // point, which can sit below local `commit_min` (the headers inherited - // from senders behind the canonical log_view live there), so - // `commit_min + 1` would skip them. A backup's parked `StartView` - // suffix is only a verification reference; resuming from its commit - // point would restart at the view's opening head, not at the gap. - let from_op = consensus - .is_primary_for_view(consensus.view()) - .then(|| consensus.with_pending_view_log(|pending| pending.commit_max.max(1))) - .flatten() - .unwrap_or_else(|| consensus.commit_min() + 1); - if from_op <= to_op { + self.note_metadata_repair_walked(); + } else if self.burn_metadata_repair_attempt(consensus.view()) { + self.rotate_stalled_metadata_repair(consensus, peer, from_op, to_op) + .await; + } else { tracing::info!( shard = self.id, from_op, @@ -5720,22 +5857,95 @@ where consensus.group(), ) .await; - } else { - // `from_op` past `to_op` without `commit_min` reaching it: the - // primary-elect window above starts at the merged log's commit - // point, which can sit above what this replica has walked. The - // top-of-tick check closes the ordinary case; this closes the - // one it cannot see. - tracing::info!( + } + } + } + + /// Re-arm a repair session that spent its stall budget against another replica. + /// + /// A session pins its peer and fences every arming site while it stands, so a + /// peer that cannot answer wedges the walk harder than having no session at + /// all. Past the budget the session is dropped and re-armed one step on; an + /// ordinary lost frame is re-requested long before that. Mirrors the partition + /// rotation in [`Self::tick_partitions`]. + /// + /// Two rings, because two things decide who can serve. A `Normal` backup is + /// repairing its committed tail and any replica ahead of it will do, so it + /// walks the cluster preferring the primary. A primary-elect is repairing + /// toward a merged log, and only the `DoViewChange` senders that named the op + /// can serve it: walking the whole ring lands on a replica that answers + /// `RangeEvicted` for a range it never held. + /// + /// Does NOT spend the budget. It lives on the shard so rotation cannot reset it + /// (see [`Self::metadata_repair_attempts`]); clearing it here would bound one + /// round and re-target forever. + #[allow(clippy::future_not_send)] + async fn rotate_stalled_metadata_repair

( + &self, + consensus: &VsrConsensus, + peer: u8, + from_op: u64, + to_op: u64, + ) where + B: MessageBus, + P: Pipeline, + { + *self.metadata_repair.borrow_mut() = None; + + if consensus.view_log_is_pending() && consensus.is_primary_for_view(consensus.view()) { + let sources = view_repair_sources(consensus, from_op); + let Some(next_peer) = next_view_repair_peer(&sources, Some(peer)) else { + // Only the quiet peer named this op. The session is dropped either + // way: `advance_pending_metadata_view` re-scans on the next tick and + // re-requests it, and a peer that never comes back leaves the + // view-change timeout to escalate. + tracing::warn!( shard = self.id, - to_op, peer, - "metadata repair window fully requested; closing the stalled session" + from_op, + "no other replica offers op {from_op} for the merged log; \ + view change is stalled" ); - *self.metadata_repair.borrow_mut() = None; - self.note_metadata_repair_progress(); - } + return; + }; + tracing::warn!( + shard = self.id, + peer, + next_peer, + from_op, + to_op, + "merged-log repair stalled past its retry budget; re-arming from \ + another do_view_change sender" + ); + self.arm_metadata_repair_session(consensus, next_peer, from_op, to_op) + .await; + return; + } + + let primary = consensus.primary_index(consensus.view()); + let next_peer = next_transfer_peer( + consensus.replica(), + peer, + consensus.replica_count(), + primary, + ); + if next_peer == peer { + // The ring had nobody else to offer (a solo group, or a two-replica + // cluster whose only peer went quiet). Dropping the session is still + // right: it unfences the level trigger below, which re-requests the + // window on the next tick. + return; } + tracing::warn!( + shard = self.id, + peer, + next_peer, + from_op, + to_op, + "metadata repair stalled past its retry budget; re-arming from another replica" + ); + self.maybe_request_metadata_repair(consensus, next_peer) + .await; } /// Compare this replica's log against the headers the view decided, and drop or @@ -5926,7 +6136,11 @@ where // one back: demanding one parks the view change forever on an op already // applied and durable in the snapshot. let repair_floor = journal.handle().snapshot_op(); - let missing = first_op_not_covered(&pending, repair_floor, |op| { + let floor = ScanFloor { + repair_floor, + commit_min: consensus.commit_min(), + }; + let missing = first_op_not_covered(&pending, floor, |op| { usize::try_from(op) .ok() .and_then(|slot| journal.handle().header(slot)) @@ -5967,7 +6181,23 @@ where // Stream already running; the stall retry covers it drying up. return; } - let sources = consensus.pending_view_body_sources(missing_op); + // Level-triggered, so it would re-arm against the head of the same source + // list next tick -- including the sender that just answered `RangeEvicted`, + // which is how the budget got spent. Once every sender has been asked and + // charged, asking again is not progress; the view-change timeout is. + // + // Per view. A later view is a new merged log from a new quorum, so the + // senders that refused this one say nothing about it. + if self.metadata_repair_exhausted(consensus.view()) { + tracing::debug!( + shard = self.id, + missing_op, + view = consensus.view(), + "merged-log repair is out of retries for this view; not re-arming" + ); + return; + } + let sources = view_repair_sources(consensus, missing_op); let Some(peer) = sources.first().copied() else { // The merge only returns a startable log when some replica offered each // body, so an empty source list means that offer was withdrawn (peer @@ -5980,14 +6210,6 @@ where return; }; - let nonce = iggy_common::random_id::get_uuid(); - *self.metadata_repair.borrow_mut() = Some(MetadataRepairSession { - nonce, - to_op: pending.op_head, - view: consensus.view(), - peer, - idle_ticks: 0, - }); tracing::info!( shard = self.id, missing_op, @@ -5995,13 +6217,45 @@ where to_op = pending.op_head, "repairing toward the merged log before starting the view" ); + self.arm_metadata_repair_session(consensus, peer, missing_op, pending.op_head) + .await; + } + + /// Mint a metadata repair session and send its first request. + /// + /// Three sites arm one (the view-change scan, the tail-repair funnel, the + /// stall rotation). Keeping the invariant here -- fresh nonce, arming view, + /// clock from zero, a `from_op` the retry can reopen at -- stops a re-arm from + /// shipping a window its own retry cannot reproduce. + /// + /// Callers decide whether to arm; this decides what an armed session is. + #[allow(clippy::future_not_send)] + async fn arm_metadata_repair_session

( + &self, + consensus: &VsrConsensus, + peer: u8, + from_op: u64, + to_op: u64, + ) where + B: MessageBus, + P: Pipeline, + { + let nonce = iggy_common::random_id::get_uuid(); + *self.metadata_repair.borrow_mut() = Some(MetadataRepairSession { + nonce, + from_op, + to_op, + view: consensus.view(), + peer, + idle_ticks: 0, + }); self.send_request_prepares( consensus.cluster(), consensus.replica(), peer, nonce, - missing_op, - pending.op_head, + from_op, + to_op, consensus.group(), ) .await; @@ -6013,11 +6267,14 @@ where /// Every TAIL arming site funnels through here -- `StartView` adoption, the /// commit-heartbeat backstop, the state-transfer fallbacks, and /// `tick_metadata`'s gap detector -- so the guards below are what make the - /// level-triggered one idempotent. The one session this does not mint is - /// the view-change repair `advance_pending_metadata_view` builds inline: it - /// repairs toward a merged log rather than the commit frontier, from a peer - /// that offered the body rather than from the primary, so none of the - /// guards below describe it. + /// level-triggered one idempotent. + /// + /// It does not decide the merged-log sessions: + /// [`Self::advance_pending_metadata_view`] arms them and + /// [`Self::rotate_stalled_metadata_repair`] re-targets them. Those repair + /// toward a parked merged log rather than the commit frontier, from a sender + /// that named the op rather than the primary, so none of the guards below fit. + /// All three share [`Self::arm_metadata_repair_session`]. #[allow(clippy::future_not_send)] async fn maybe_request_metadata_repair

(&self, consensus: &VsrConsensus, peer: u8) where @@ -6038,7 +6295,6 @@ where && consensus.commit_min() < consensus.commit_max() && self.metadata_repair.borrow().is_none() { - let nonce = iggy_common::random_id::get_uuid(); let to_op = consensus.commit_max(); let from_op = consensus.commit_min() + 1; // Spent here rather than at the detector, so the edge-triggered @@ -6046,13 +6302,6 @@ where // the next tick would otherwise leave the count saturated and hand // the next real gap an arm on its first tick. self.metadata_gap_ticks.set(0); - *self.metadata_repair.borrow_mut() = Some(MetadataRepairSession { - nonce, - to_op, - view: consensus.view(), - peer, - idle_ticks: 0, - }); tracing::info!( shard = self.id, from_op, @@ -6060,16 +6309,8 @@ where peer, "metadata behind the group frontier; requesting repair" ); - self.send_request_prepares( - consensus.cluster(), - consensus.replica(), - peer, - nonce, - from_op, - to_op, - consensus.group(), - ) - .await; + self.arm_metadata_repair_session(consensus, peer, from_op, to_op) + .await; } } @@ -6600,21 +6841,23 @@ where /// on the shard here for the same reason `metadata_transfer_attempts` does: /// one metadata group per node. It has to outlive the SESSION either way, /// or the rotation that mints a new one would reset the count and re-target - /// forever without ever giving up on a peer. - fn burn_metadata_repair_attempt(&self) -> bool { - let attempts = self.metadata_repair_attempts.get() + 1; - self.metadata_repair_attempts.set(attempts); + /// forever without giving up on a peer. Only + /// [`Self::note_metadata_repair_walked`] clears it. + fn burn_metadata_repair_attempt(&self, view: u32) -> bool { + let (charged_view, attempts) = self.metadata_repair_attempts.get(); + let attempts = if charged_view == view { + attempts + 1 + } else { + 1 + }; + self.metadata_repair_attempts.set((view, attempts)); attempts > partitions::REPAIR_MAX_STALL_RETRIES } - /// The serving peer answered: reset the budget, so it bounds CONSECUTIVE - /// silence rather than the stalls a long healthy stream accumulates. - /// - /// Any in-scope repair frame, not only an accepted one. A re-request - /// re-serves ops this replica already holds, so charging those as silence - /// rotates away from a peer that is answering. - fn note_metadata_repair_progress(&self) { - self.metadata_repair_attempts.set(0); + /// Whether this view has already spent its merged-log repair budget. + const fn metadata_repair_exhausted(&self, view: u32) -> bool { + let (charged_view, attempts) = self.metadata_repair_attempts.get(); + charged_view == view && attempts > partitions::REPAIR_MAX_STALL_RETRIES } /// Burn one retry round; `true` once the budget is exhausted. @@ -6633,6 +6876,41 @@ where self.metadata_transfer_attempts.set(0); } + /// A usable repair frame landed in the window: restart the stall clock. + /// + /// A window is served in `REPAIR_CHUNK_MAX` slices and nothing else resets + /// `idle_ticks`, so a healthy multi-chunk stream would cross the retry interval + /// on its own and rotate off a peer that is answering. + /// + /// Clock only. A repair prepare carries no sender and no session nonce (the + /// frame IS the stored prepare, and its identity checksum covers every byte + /// that could hold one), so a peer this session already rotated away from can + /// land in-flight frames here and be credited to its successor. On the clock + /// that costs one retry interval and is bounded, since nothing re-requests from + /// that peer. On the budget it would cost rotation itself. See + /// [`Self::note_metadata_repair_walked`]. + fn note_metadata_repair_clock(&self) { + if let Some(session) = self.metadata_repair.borrow_mut().as_mut() { + session.idle_ticks = 0; + } + } + + /// The repair this session asked for is landing: restart the clock and the + /// budget. + /// + /// Takes only attributable signals, which a bare frame is not. Terminators are + /// fenced on `session.nonce` before reaching here, so they came from the + /// targeted peer; an advanced `commit_min` is the gap actually closing, + /// whoever supplied the bytes. + /// + /// Stricter than the "any frame" rule it replaced: a re-request re-serves ops + /// already held, so a peer answering with nothing new used to clear its own + /// budget and could never be rotated away from. + fn note_metadata_repair_walked(&self) { + self.metadata_repair_attempts.set((0, 0)); + self.note_metadata_repair_clock(); + } + /// Charge one decode failure against `snapshot_seq`'s generation; `true` /// once that generation's budget is spent. A different generation restarts /// the count: the peer checkpointed since, so the artifacts are new bytes @@ -7275,13 +7553,17 @@ where walk_cursor.get_or_insert(namespace); } } - let consensus_normal = partition.consensus().is_normal(); let consensus_view = partition.consensus().view(); let commit_min = partition.consensus().commit_min(); let cluster = partition.consensus().cluster(); let self_id = partition.consensus().replica(); + // A primary-elect's merged-log session legitimately runs outside + // `Normal` (`request_partition_view_repair`). Same predicate as the + // two ingest sites, so the arming side and the ingest side cannot + // drift apart. + let session_live = consensus::repair_session_live(partition.consensus()); let repair_finished = partition.repair.is_some_and(|session| { - if !consensus_normal || consensus_view != session.view { + if !session_live || consensus_view != session.view { return true; } // Floored at the LIVE commit point, like `complete_repair`: @@ -7313,7 +7595,7 @@ where continue; } let due = partition.repair.as_mut().and_then(|session| { - if !consensus_normal { + if !session_live { return None; } session.idle_ticks += 1; @@ -7362,6 +7644,20 @@ where ); partition.repair = None; partition.note_repair_progress(); + // A parked view change re-arms from the merged log's senders, + // not around the cluster ring: + // `maybe_request_partition_repair` refuses outside `Normal`, + // and a replica that never named the op answers `RangeEvicted` + // for a range it never held. + if partition.consensus().view_log_is_pending() + && partition + .consensus() + .is_primary_for_view(partition.consensus().view()) + { + self.request_partition_view_repair(partition, from_op, to_op, Some(peer)) + .await; + continue; + } if next_peer == peer { // The ring had nobody else to offer (a solo group, or a // two-replica group whose only peer is the one that @@ -8582,6 +8878,82 @@ where true } + /// Repair a primary-elect's merged log before it starts the view. + /// + /// Sibling of [`Self::maybe_request_partition_repair`], which refuses outside + /// `Normal` because its window comes from the live commit frontier. This window + /// comes from the parked merged log, so it runs in `ViewChange` for the replica + /// that parked it. Without it the coverage scan in + /// [`Self::start_pending_partition_view`] reports an op nothing ever fetches -- + /// the sweep's gap detector needs `probe.normal` too -- and only the + /// view-change timeout moves the replica. + /// + /// `avoid` is the peer a stall just gave up on, so rotation lands on a + /// different sender instead of the head of the same list. + #[allow(clippy::future_not_send)] + async fn request_partition_view_repair( + &self, + partition: &mut IggyPartition, + from_op: u64, + to_op: u64, + avoid: Option, + ) where + B: MessageBus, + { + if partition.repair.is_some() || from_op > to_op { + return; + } + if self.partition_repairs_inflight.get() >= PARTITION_REPAIRS_INFLIGHT_MAX { + return; + } + let consensus = partition.consensus(); + let sources = view_repair_sources(consensus, from_op); + let Some(peer) = next_view_repair_peer(&sources, avoid) else { + // Nobody else named this op. The view-change timeout escalates; + // re-arming the same silent sender would pin the scan behind a + // session for nothing. + tracing::warn!( + shard = self.id, + namespace_raw = consensus.group(), + from_op, + to_op, + "no replica offers op {from_op} for the merged partition log; view change is \ + stalled" + ); + return; + }; + let nonce = iggy_common::random_id::get_uuid(); + let cluster = consensus.cluster(); + let self_id = consensus.replica(); + let namespace = consensus.group(); + let view = consensus.view(); + self.partition_repairs_inflight + .set(self.partition_repairs_inflight.get() + 1); + partition.repair = Some(partitions::RepairSession { + nonce, + view, + // Merged-log numbers, not the live frontier: `commit_to_op` is what + // the walk must reach to finish the session, `fetch_to_op` the head + // the view will announce. + commit_to_op: pending_commit_max(consensus), + fetch_to_op: to_op, + floor: None, + peer, + first_batch_offset: None, + idle_ticks: 0, + }); + tracing::info!( + shard = self.id, + namespace_raw = namespace, + from_op, + to_op, + peer, + "repairing toward the merged partition log before starting the view" + ); + self.send_request_prepares(cluster, self_id, peer, nonce, from_op, to_op, namespace) + .await; + } + /// Receiver side of a partition descriptor: accept the manifest, adopt /// any reusable staged segments from an earlier attempt, and start /// pulling, or fall back to journal repair when the peer cannot serve. @@ -10004,6 +10376,111 @@ where } } +/// Lowest op of a primary-elect's merged log this replica can be held to. +/// +/// The merged commit point is what the cluster committed, `commit_min` what this +/// replica applied. They diverge whenever this replica has not applied the merged +/// commit point: a hole in the local prefix, or plain apply lag. Taking the lower +/// keeps coverage, repair scope and the stall retry asking about the same ops. +fn merged_log_scan_floor(pending: &MergedLog, commit_min: u64) -> u64 { + pending.commit_max.min(commit_min + 1).max(1) +} + +/// The two floors on a merged-log coverage scan. +/// +/// Named, not positional: both are `u64`, they sit next to each other, swapping +/// them compiles, and the partition site passes the same value for both. +#[derive(Debug, Clone, Copy)] +struct ScanFloor { + /// Ops at or below this are gone AND already settled: compacted under a + /// snapshot (metadata) or at or below the local commit point (partitions). + /// Neither can diverge from the merged log and no repair puts the entry back, + /// so demanding one parks the view change forever. + repair_floor: u64, + /// Highest op this replica has applied. See [`merged_log_scan_floor`]. + commit_min: u64, +} + +impl ScanFloor { + /// Lowest op the scan probes. + fn opens_at(self, pending: &MergedLog) -> u64 { + merged_log_scan_floor(pending, self.commit_min).max(self.repair_floor + 1) + } +} + +/// Replicas a primary-elect can ask for `op` while its merged log is parked. +/// +/// Offered bodies first: those senders proved they hold the entry. A gap below +/// every sender's commit point has none, since a DVC suffix spans `commit..=op` +/// and says nothing underneath, so fall back to senders that committed the op. +/// They hold it or compacted it, and `RangeEvicted` says which. +/// +/// Both planes: a partition primary-elect parks a merged log the same way. +fn view_repair_sources(consensus: &VsrConsensus, op: u64) -> Vec +where + B: MessageBus, + P: Pipeline, +{ + let offered = consensus.pending_view_body_sources(op); + if offered.is_empty() { + consensus.pending_view_commit_sources(op) + } else { + offered + } +} + +/// Where a stalled repair session reopens its window. +/// +/// A merged-log session reopens exactly where it was armed. Its floor is COVERAGE, +/// not the walk: `first_op_not_covered` reports ops whose journal entry is absent +/// or diverging, and an op can be applied (`commit_min` past it) while its entry is +/// gone. Raising the floor to `commit_min + 1` there skips the very op the scan +/// reported, and the view change parks on it forever. +/// +/// A tail-repair session is the other way round. Its window IS the commit gap, so +/// ops the walk has since consumed must not be asked for again. `session.from_op` +/// still floors it, carrying the initial arm's snapshot clamp so no retry asks for +/// compacted ops. +fn stalled_repair_from_op(session_from_op: u64, commit_min: u64, repairing_view: bool) -> u64 { + if repairing_view { + session_from_op + } else { + session_from_op.max(commit_min + 1) + } +} + +/// Walk a merged-log source list one step past `avoid`, wrapping. +/// +/// A ring, not a filter. The list is `log_view`-ordered and identical on every +/// call, so `find(|c| *c != avoid)` yields the head for every peer but the head +/// itself and a third sender is never reached. +/// +/// `None` when the list is empty or `avoid` is its only entry. +fn next_view_repair_peer(sources: &[u8], avoid: Option) -> Option { + let Some(avoid) = avoid else { + return sources.first().copied(); + }; + let Some(index) = sources.iter().position(|candidate| *candidate == avoid) else { + // The peer that stalled is not in this list at all (the DVC quorum moved + // under it), so nothing has been tried yet from where we now stand. + return sources.first().copied(); + }; + let next = sources[(index + 1) % sources.len()]; + if next == avoid { None } else { Some(next) } +} + +/// The merged log's commit point while a view change is parked; the live frontier +/// otherwise. +fn pending_commit_max(consensus: &VsrConsensus) -> u64 +where + B: MessageBus, + P: Pipeline, +{ + consensus + .with_pending_view_log(|pending| pending.commit_max) + .unwrap_or_else(|| consensus.commit_max()) +} + /// Whether a repaired prepare at `op` falls inside the range this replica is /// currently repairing. /// @@ -10024,7 +10501,7 @@ fn repair_op_in_scope( pending .filter(|_| is_primary_elect) .map_or(op > commit_min, |pending| { - (op >= pending.commit_max.max(1) && op <= pending.op_head) + (op >= merged_log_scan_floor(pending, commit_min) && op <= pending.op_head) || pending .committed_elsewhere .iter() @@ -10723,13 +11200,29 @@ const fn header_is_view_entry(local: &PrepareHeader, canonical: &PrepareHeader) /// repair cannot walk back to them. `repair_floor` drops the ops whose journal entry /// is legitimately gone AND whose identity is already settled: on the metadata plane /// ops compacted under a snapshot, on the partition plane ops at or below the local -/// commit point, whose flushed entries `evict_prefix` moves out of the header vec. -/// Neither can diverge from the merged log (a committed or compacted op is the -/// quorum's op), and no repair puts the journal entry back, so demanding one parks -/// the view change forever. +/// commit point. Neither can diverge from the merged log (a committed or compacted +/// op is the quorum's op), and no repair puts the journal entry back, so demanding +/// one parks the view change forever. +/// +/// Not a residency bound. `evict_prefix` clears the header vec up to `commit_max` +/// (the cluster frontier), so ops above `repair_floor` can be non-resident and +/// still serveable, from the evicted ring or the flushed segments. Callers pass a +/// `header_at` that reads both. +/// +/// Opens at [`merged_log_scan_floor`]: the merged commit point alone would declare +/// the log serveable over a local gap, promoting a replica whose `CommitJournal` +/// gap-stops below where `RebuildPipeline` seeds. +/// +/// Below the merged commit point, identity rests on the fault model rather than on +/// this scan. The merged log names headers only from the DVC suffixes, which span +/// `commit..=op` per sender, so an op the widened floor admits under +/// `pending.commit_max` usually has no canonical header and `held` degrades to +/// bare residency. Sound under crash-stop, where a committed op is the quorum's +/// op. Not a Byzantine or bit-rot guard: corruption is `verify_prepare_integrity`'s +/// job on the ingest side. fn first_op_not_covered( pending: &MergedLog, - repair_floor: u64, + floor: ScanFloor, header_at: impl Fn(u64) -> Option, ) -> Option { let held = |op: u64| { @@ -10743,14 +11236,19 @@ fn first_op_not_covered( .find(|header| header.op == op) .is_none_or(|canonical| header_is_view_entry(&local, canonical)) }; - (pending.commit_max.max(1).max(repair_floor + 1)..=pending.op_head) + (floor.opens_at(pending)..=pending.op_head) .find(|op| !held(*op)) .or_else(|| { + // NOT raised to `opens_at`: these ops sit outside the merged window + // by construction, and dropping the ones below it would start the + // view over a committed op this replica cannot serve. The repair + // window is floored to match instead, via + // `MetadataRepairSession::from_op`. pending .committed_elsewhere .iter() .map(|header| header.op) - .filter(|op| *op > repair_floor) + .filter(|op| *op > floor.repair_floor) .find(|op| !held(*op)) }) } @@ -11346,9 +11844,17 @@ mod repair_scope_tests { mod view_coverage_tests { //! Holding an op is not holding the view's op. - use super::{MergedLog, first_op_not_covered}; + use super::{MergedLog, ScanFloor, first_op_not_covered}; use iggy_binary_protocol::{Command, Operation, PrepareHeader}; + /// What a caught-up replica passes: nothing compacted, nothing lagging. + fn caught_up(pending: &MergedLog) -> ScanFloor { + ScanFloor { + repair_floor: 0, + commit_min: pending.commit_max, + } + } + fn sealed(op: u64, request: u64) -> PrepareHeader { let mut header = PrepareHeader { command: Command::Prepare, @@ -11373,7 +11879,7 @@ mod view_coverage_tests { committed_elsewhere: Vec::new(), }; let held = [sealed(100, 1), sealed(99, 7), sealed(98, 1)]; - let missing = first_op_not_covered(&pending, 0, |op| { + let missing = first_op_not_covered(&pending, caught_up(&pending), |op| { held.iter().find(|header| header.op == op).copied() }); assert_eq!(missing, Some(99)); @@ -11397,16 +11903,187 @@ mod view_coverage_tests { }; let nothing_resident = |_: u64| None; assert_eq!( - first_op_not_covered(&pending, 0, nothing_resident), + first_op_not_covered(&pending, caught_up(&pending), nothing_resident), Some(256), "unfloored, the evicted committed op reads as an unfillable hole" ); assert_eq!( - first_op_not_covered(&pending, 256, nothing_resident), + first_op_not_covered( + &pending, + ScanFloor { + repair_floor: 256, + commit_min: 256, + }, + nothing_resident + ), None, "floored at the local commit point, the view starts" ); } + + #[test] + fn given_a_hole_below_the_merged_commit_point_when_scanning_should_report_it() { + // Missed op 7 and kept taking prepares above it: the cluster committed + // through 10 while this state machine stopped at 6. From the merged commit + // point the view would start over the gap and the first quorum ack would + // apply an op with 7..=10 never executed locally. + let pending = MergedLog { + op_head: 12, + commit_max: 10, + headers: (7..=12).rev().map(|op| sealed(op, 1)).collect(), + committed_elsewhere: Vec::new(), + }; + let held: Vec<_> = (8..=12).map(|op| sealed(op, 1)).collect(); + let missing = first_op_not_covered( + &pending, + ScanFloor { + repair_floor: 0, + commit_min: 6, + }, + |op| held.iter().find(|header| header.op == op).copied(), + ); + assert_eq!( + missing, + Some(7), + "a hole below the merged commit point must park the view change" + ); + } + + #[test] + fn given_a_contiguous_prefix_when_scanning_should_open_at_the_merged_commit_point() { + // Nothing missing below, so both bounds coincide. Op 9 is held but is not + // the view's op 9, so the commit point itself is still identity-checked. + let pending = MergedLog { + op_head: 12, + commit_max: 9, + headers: (9..=12).rev().map(|op| sealed(op, 1)).collect(), + committed_elsewhere: Vec::new(), + }; + let held: Vec<_> = (9..=12) + .map(|op| sealed(op, if op == 9 { 7 } else { 1 })) + .collect(); + let missing = first_op_not_covered(&pending, caught_up(&pending), |op| { + held.iter().find(|header| header.op == op).copied() + }); + assert_eq!( + missing, + Some(9), + "the merged commit point stays in scope when the prefix is contiguous" + ); + } + + #[test] + fn given_a_held_run_below_the_hole_when_scanning_should_walk_to_the_hole() { + // The span the widened floor buys, and the one that costs: the open sits + // well below the merged commit point, the ops between are all held, and the + // scan must walk them to reach 15. A scan that stopped on its first probe + // would find op 4 covered and never look further. + let pending = MergedLog { + op_head: 22, + commit_max: 20, + headers: (4..=22).rev().map(|op| sealed(op, 1)).collect(), + committed_elsewhere: Vec::new(), + }; + let held: Vec<_> = (4..=22) + .filter(|op| *op != 15) + .map(|op| sealed(op, 1)) + .collect(); + let missing = first_op_not_covered( + &pending, + ScanFloor { + repair_floor: 0, + commit_min: 3, + }, + |op| held.iter().find(|header| header.op == op).copied(), + ); + assert_eq!( + missing, + Some(15), + "the scan must walk the held run below the merged commit point, not stop at its first covered probe" + ); + } + + #[test] + fn given_a_committed_elsewhere_op_below_the_open_when_scanning_should_still_report_it() { + // `committed_elsewhere` sits outside the merged window, so the fallback is + // floored at `repair_floor` and not at the scan's open. Which is why the + // repair window floors at the op reported here: a retry reopening at + // `opens_at` would skip op 5 forever. + let pending = MergedLog { + op_head: 12, + commit_max: 10, + headers: (7..=12).rev().map(|op| sealed(op, 1)).collect(), + committed_elsewhere: vec![sealed(5, 1)], + }; + let held: Vec<_> = (7..=12).map(|op| sealed(op, 1)).collect(); + let floor = ScanFloor { + repair_floor: 0, + commit_min: 6, + }; + assert_eq!(floor.opens_at(&pending), 7); + let missing = first_op_not_covered(&pending, floor, |op| { + held.iter().find(|header| header.op == op).copied() + }); + assert_eq!( + missing, + Some(5), + "an op committed elsewhere and below the open is still uncovered" + ); + } + + #[test] + fn given_a_hole_when_scoping_repair_should_admit_the_missing_op() { + // Coverage and scope must agree: the scan parks on op 7, so op 7's repaired + // prepare must be ingested. From the merged commit point it would be + // requested and then refused. + let pending = MergedLog { + op_head: 12, + commit_max: 10, + headers: (7..=12).rev().map(|op| sealed(op, 1)).collect(), + committed_elsewhere: Vec::new(), + }; + assert!( + super::repair_op_in_scope(Some(&pending), true, 6, 7), + "the op the coverage scan parked on must be in repair scope" + ); + } + + #[test] + fn given_a_source_list_when_rotating_should_walk_it_as_a_ring() { + use super::next_view_repair_peer; + + let sources = [1u8, 2, 3]; + assert_eq!(next_view_repair_peer(&sources, None), Some(1)); + assert_eq!(next_view_repair_peer(&sources, Some(1)), Some(2)); + assert_eq!( + next_view_repair_peer(&sources, Some(2)), + Some(3), + "a filter answers 1 here and never reaches the third sender" + ); + assert_eq!( + next_view_repair_peer(&sources, Some(3)), + Some(1), + "the walk wraps" + ); + } + + #[test] + fn given_a_sole_or_absent_source_when_rotating_should_report_nobody_left() { + use super::next_view_repair_peer; + + assert_eq!(next_view_repair_peer(&[], None), None); + assert_eq!(next_view_repair_peer(&[], Some(1)), None); + assert_eq!( + next_view_repair_peer(&[1], Some(1)), + None, + "the only sender is the one that went quiet" + ); + assert_eq!( + next_view_repair_peer(&[2, 3], Some(9)), + Some(2), + "a peer no longer in the list means nothing here has been tried yet" + ); + } } #[cfg(test)] @@ -12286,13 +12963,14 @@ mod metadata_repair_session_tests { use super::{ MetadataRepairSession, gap_repair_peer, metadata_repair_superseded, next_transfer_peer, - repair_chunk_walked, + repair_chunk_walked, stalled_repair_from_op, }; - /// Armed at view 3, against a window ending at op 20. + /// Armed at view 3, against the window `11..=20`. const fn session() -> MetadataRepairSession { MetadataRepairSession { nonce: 7, + from_op: 11, to_op: 20, view: 3, peer: 0, @@ -12300,6 +12978,31 @@ mod metadata_repair_session_tests { } } + /// A merged-log session must re-ask for the op the coverage scan reported, even + /// once the walk has passed it. Coverage is about the journal ENTRY; an op can + /// be applied and still have no entry to serve, which is exactly what + /// `committed_elsewhere` reports. + #[test] + fn given_a_dropped_response_below_commit_min_when_retrying_should_still_ask_for_it() { + assert_eq!( + stalled_repair_from_op(5, 6, true), + 5, + "clamping to commit_min + 1 would retry from 7 and skip the reported hole" + ); + } + + /// The tail-repair session is the other way round: its window is the commit gap, + /// so ops the walk consumed must not be re-requested. + #[test] + fn given_a_walked_window_when_retrying_a_tail_session_should_open_above_it() { + assert_eq!(stalled_repair_from_op(5, 6, false), 7); + assert_eq!( + stalled_repair_from_op(11, 3, false), + 11, + "the arm floor still holds, so no retry asks for compacted ops" + ); + } + #[test] fn given_a_gap_stopped_backup_when_picking_a_peer_should_ask_the_primary() { assert_eq!(gap_repair_peer(2, 3, 0), Some(0)); diff --git a/core/shard/src/router.rs b/core/shard/src/router.rs index b7465d2b7c..1e46ec9c9b 100644 --- a/core/shard/src/router.rs +++ b/core/shard/src/router.rs @@ -532,6 +532,15 @@ where }) } + /// `first_partition_commit_fault` for the simulator's lost-wakeup + /// tripwire: a fenced pump and a missed wake both leave frames undrained, and + /// only the second is a channel bug. Test/simulator only, like `inbox_len`. + #[cfg(any(test, feature = "simulator"))] + #[must_use] + pub fn fenced_partition_fault(&self) -> Option { + self.first_partition_commit_fault() + } + /// Sanity check at pump entry: every Consensus frame routed through /// [`Self::dispatch`] must land on the shard whose `id` matches the /// `target_shard` the sender stamped on the frame. The ctor diff --git a/core/simulator/src/bin/workload-fuzz.rs b/core/simulator/src/bin/workload-fuzz.rs index 8143016a7a..fb6ab966aa 100644 --- a/core/simulator/src/bin/workload-fuzz.rs +++ b/core/simulator/src/bin/workload-fuzz.rs @@ -132,11 +132,21 @@ struct Args { /// empty shadow against empty committed state and agrees. `0` opts out. #[arg(long, default_value_t = 1)] min_commits: u64, - /// Committed metadata ops that must have been witnessed by more than one live - /// replica, i.e. that exercised cross-replica agreement. Ignored below two live - /// replicas, where the property is untestable rather than untested. `0` opts out. + /// Committed ops, on EITHER plane, that must have been witnessed by more than + /// one live replica, i.e. that exercised cross-replica agreement. Ignored below + /// two live replicas, where the property is untestable rather than untested. + /// `0` opts out. #[arg(long, default_value_t = 1)] min_ops_compared: usize, + /// As `--min-ops-compared`, but METADATA ops only. + /// + /// Separate because a partition-only run satisfies the combined floor while the + /// metadata oracle compares an empty chain against an empty chain and agrees. + /// Defaults to `1`, which is the floor `--min-ops-compared` carried before it + /// counted both planes; a campaign that wants no metadata coverage opts out + /// with `0`. + #[arg(long, default_value_t = 1)] + min_metadata_ops_compared: usize, /// Fail the run if crash or restart injection was requested but never happened. /// Off by default, since a short run at low probability may legitimately draw /// none; on for a campaign where such a seed is silently wasted. @@ -562,9 +572,11 @@ fn run_quiesce_phase( }; println!( "quiesced and converged (leader-relative; entity oracle: {entity_oracle}; \ - evictions={}; ops_compared={} replicas_compared={} namespaces_checked={})", + evictions={}; ops_compared={} partition_ops_compared={} replicas_compared={} \ + namespaces_checked={})", workload.evictions(), convergence.ops_compared, + convergence.partition_ops_compared, convergence.replicas_compared, convergence.namespaces_checked, ); @@ -574,13 +586,29 @@ fn run_quiesce_phase( proved nothing about entity state (seed={seed:#x})" ); let live = usize::from(replicas) - sim.crashed.len(); + // Either plane satisfies it: a partition-plane run commits almost no metadata, + // so the metadata count alone called every such run vacuous. + let compared = convergence.ops_compared + convergence.partition_ops_compared; assert!( - args.min_ops_compared == 0 || live < 2 || convergence.ops_compared >= args.min_ops_compared, - "--min-ops-compared {}: {live} replicas live but only {} op(s) witnessed \ - by more than one, so cross-replica agreement went untested \ - (seed={seed:#x})", + args.min_ops_compared == 0 || live < 2 || compared >= args.min_ops_compared, + "--min-ops-compared {}: {live} replicas live but only {compared} op(s) witnessed \ + by more than one ({} metadata, {} partition), so cross-replica agreement went \ + untested (seed={seed:#x})", args.min_ops_compared, convergence.ops_compared, + convergence.partition_ops_compared, + ); + // The metadata half on its own: summing the planes above lets a partition-only + // run clear that floor while the metadata oracle compares nothing. + assert!( + args.min_metadata_ops_compared == 0 + || live < 2 + || convergence.ops_compared >= args.min_metadata_ops_compared, + "--min-metadata-ops-compared {}: {live} replicas live but only {} committed metadata \ + op(s) witnessed by more than one, so the metadata oracle compared an empty chain \ + (seed={seed:#x})", + args.min_metadata_ops_compared, + convergence.ops_compared, ); // Again after the drain: the drain both answers outstanding requests and // issues its own resends, so the pre-drain numbers are not the final ones. diff --git a/core/simulator/src/lib.rs b/core/simulator/src/lib.rs index d5c554023e..51a2d35598 100644 --- a/core/simulator/src/lib.rs +++ b/core/simulator/src/lib.rs @@ -33,7 +33,7 @@ use deps::SimClock; use deps::SimSuperblock; use deps::{MemStorage, SimJournal}; use executor::{DetExecutor, RunOutcome, TaskId}; -use iggy_binary_protocol::{Command, GenericHeader, ReplyHeader}; +use iggy_binary_protocol::{Command, GenericHeader, PrepareHeader, ReplyHeader}; use iggy_common::IggyError; use message_bus::installer::conn_info::{ClientConnMeta, ClientTransportKind}; use metadata::impls::metadata::StreamsFrontend; @@ -53,7 +53,7 @@ use server_common::sharding::{IggyNamespace, PartitionLocation, ShardId}; use shard::shards_table::{ShardsTable, calculate_shard_assignment}; use shard::{CONSENSUS_TICK_INTERVAL, PartitionMaterialisation}; use std::cell::RefCell; -use std::collections::{HashMap, HashSet}; +use std::collections::{BTreeMap, HashMap, HashSet}; use std::net::{IpAddr, Ipv4Addr, SocketAddr}; use std::rc::Rc; use std::sync::Arc; @@ -135,6 +135,54 @@ pub(crate) struct PartitionConsensusState { pub commit_min: u64, } +/// A pipeline head the commit walk is holding on: covered by the commit frontier, +/// but not the op the state machine is next owed. +/// +/// What `drain_committable_prefix` / `peek_committable_head` refuse to drain. Two +/// different faults share that refusal and need different thresholds, so +/// [`CommitPrefixHole::kind`] keeps them apart. +#[derive(Debug, Clone, Copy)] +pub(crate) struct CommitPrefixHole { + pub head_op: u64, + pub commit_min: u64, + pub commit_max: u64, + pub kind: CommitHoldKind, +} + +/// Why a commit walk is holding below its pipeline head. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub(crate) enum CommitHoldKind { + /// `head_op > commit_min + 1`: the ops between never arrived. Legitimate and + /// transient right after a promotion, while the bounded journal walk clears the + /// apply backlog `RebuildPipeline` seeded above. Repair clears it, and + /// `commit_min` climbing is the proof repair is working. + MissingOps, + /// `head_op <= commit_min`: the walk applied this op and advanced past a + /// still-resident entry. Nothing missing, no repair owed, only a pop that can + /// no longer happen -- so it never self-clears, and `commit_min` keeps climbing + /// while the head stays frozen. + AppliedHead, +} + +impl CommitPrefixHole { + fn read(head: Option, commit_min: u64, commit_max: u64) -> Option { + let head = head?; + if head.op > commit_max || head.op == commit_min + 1 { + return None; + } + Some(Self { + head_op: head.op, + commit_min, + commit_max, + kind: if head.op > commit_min { + CommitHoldKind::MissingOps + } else { + CommitHoldKind::AppliedHead + }, + }) + } +} + pub struct Simulator { /// All replicas, indexed by replica id. Always fully populated; crashed replicas /// stay alive but are skipped during dispatch. @@ -966,6 +1014,21 @@ impl Simulator { continue; } for shard in &replica.shards { + // First and unconditional. A fenced pump has exited, so its frames + // pile up exactly as a missed wake does and every lane assert below + // would misreport the cause -- and the pump's `FatalCommit` return + // is dropped at the spawn, so nothing else sees it. Gated on a + // non-empty inbox, a fenced pump that happened to drain would pass + // quiescence outright. + assert!( + shard.fenced_partition_fault().is_none(), + "fenced pump: replica {replica_id} shard {} exited on a fatal commit ({:?}). \ + Not a lost wakeup; fix the commit failure (seed {:#x}, schedule hash {:#x})", + shard.id, + shard.fenced_partition_fault(), + self.seed, + self.executor.schedule_hash(), + ); let pending = shard.inbox_len(); assert_eq!( pending, @@ -1364,6 +1427,30 @@ impl Simulator { Some(partition.offsets()) } + /// A replica's journaled partition-plane prepare headers over `ops`, or `None` + /// when it does not host the namespace. + /// + /// Repair headers, not resident ones. `evict_prefix` clears the resident vec as + /// the committed prefix flushes to segments and moves those entries to the + /// repair ring, so a resident-only read compares nothing at all once a run has + /// flushed. The ring is capacity-bounded, which makes this the recently + /// committed tail rather than the whole prefix -- and that tail is where a bad + /// repair or a mis-decided view change lands. + /// + /// One pass per replica per namespace, not one per op: both lookups behind + /// `repair_headers_in` are linear. + #[must_use] + pub(crate) fn partition_journaled_headers( + &self, + replica_idx: usize, + namespace: IggyNamespace, + ops: std::ops::RangeInclusive, + ) -> Option> { + let shard = self.replicas[replica_idx].partition_shard(namespace); + let partition = shard.plane.partitions().get_by_ns(&namespace)?; + Some(partition.log.journal().inner.repair_headers_in(ops)) + } + /// Consensus view for a replica's partition-plane group, or `None` if that /// replica does not host the namespace. #[must_use] @@ -1398,6 +1485,53 @@ impl Simulator { }) } + /// A replica's metadata consensus handle, or `None` when it hosts no metadata + /// plane. The one way to reach it; do not hand-walk the shard-0 / plane / + /// metadata chain. + #[must_use] + pub(crate) fn metadata_consensus( + &self, + replica_idx: usize, + ) -> Option<&consensus::VsrConsensus> { + self.replicas[replica_idx].shards[0] + .plane + .metadata() + .consensus + .as_ref() + } + + /// The metadata pipeline head the commit walk is holding on, if any. See + /// [`CommitPrefixHole`]. + #[must_use] + pub(crate) fn metadata_commit_prefix_hole( + &self, + replica_idx: usize, + ) -> Option { + let consensus = self.metadata_consensus(replica_idx)?; + CommitPrefixHole::read( + consensus.pipeline_head_header(), + consensus.commit_min(), + consensus.commit_max(), + ) + } + + /// Partition-plane twin of [`Self::metadata_commit_prefix_hole`]. + #[must_use] + pub(crate) fn partition_commit_prefix_hole( + &self, + replica_idx: usize, + namespace: IggyNamespace, + ) -> Option { + let shard = self.replicas[replica_idx].partition_shard(namespace); + let partition = shard.plane.partitions().get_by_ns(&namespace)?; + let consensus = partition.consensus(); + CommitPrefixHole::read( + consensus.pipeline_head_header(), + consensus.commit_min(), + consensus.commit_max(), + ) + } + /// Index of the current primary for `namespace`, as seen by the first live /// replica hosting it, or `None` if no live replica hosts it. /// diff --git a/core/simulator/src/workload/invariants.rs b/core/simulator/src/workload/invariants.rs index 4f6dd3a1d6..d2011bcb50 100644 --- a/core/simulator/src/workload/invariants.rs +++ b/core/simulator/src/workload/invariants.rs @@ -23,18 +23,49 @@ //! trace and the determinism baseline (`workload_replay_is_deterministic`) //! unchanged. -use crate::Simulator; use crate::workload::state_checker::StateChecker; use crate::workload::{CLIENT_REQUEST_QUEUE_MAX, Workload}; +use crate::{CommitHoldKind, CommitPrefixHole, Simulator}; +use consensus::Consensus; use server_common::sharding::IggyNamespace; use std::collections::HashMap; +/// Ticks a `Normal` metadata primary may sit behind its own recovery barrier +/// before the run is called wedged. +/// +/// A primary below the barrier admits nothing. Transient while a resumed primary +/// re-pipelines its suffix, a handful of round trips, so two orders of magnitude +/// of slack: only a barrier nothing will ever lower trips it. +const RECOVERY_BARRIER_WEDGE_TICKS: u32 = 2_000; + +/// Ticks a replica may hold its commit walk below a MISSING op. +/// +/// A promotion holds here legitimately while the journal walk clears its apply +/// backlog 64 ops per sweep, so this sits past any backlog a run generates. The +/// count also resets when `commit_min` advances, which is repair working. +const COMMIT_PREFIX_HOLE_WEDGE_TICKS: u32 = 2_000; + +/// Ticks a replica may hold its commit walk below an ALREADY APPLIED head. +/// +/// Reachable without a bug: a `set_commit_floor` jump can raise `commit_min` past +/// a live pipeline. Nothing pops a head below the floor, so only a view change +/// clears it, and this has to outlast the view-change escalation to avoid failing +/// a run on the window in between. +const COMMIT_PREFIX_APPLIED_HEAD_TICKS: u32 = 2_000; + /// Per-(replica, namespace) high-water marks carried across ticks so each new /// reading can be compared against the last. #[derive(Debug, Default)] pub struct Invariants { commit_offset: HashMap<(u8, IggyNamespace), u64>, view: HashMap<(u8, IggyNamespace), u64>, + /// Consecutive ticks a replica has been a `Normal` metadata primary still + /// gated by its recovery barrier. Reset as soon as any of that stops holding. + barrier_gated_ticks: HashMap, + /// Consecutive ticks a plane's commit walk has held below a committable + /// pipeline head, keyed by replica and namespace (`None` = metadata), paired + /// with the `commit_min` the count started from. + commit_hole_ticks: HashMap<(u8, Option), (u32, u64)>, /// Cross-replica committed-log agreement. Runs every tick like the rest, so a /// divergence is reported where it appears rather than at the next quiesce. state_checker: StateChecker, @@ -71,9 +102,25 @@ impl Invariants { for replica_idx in 0..sim.replica_count { if sim.is_crashed(replica_idx) { + self.barrier_gated_ticks.remove(&replica_idx); + self.commit_hole_ticks + .retain(|(replica, _), _| *replica != replica_idx); continue; } + self.check_recovery_barrier(sim, seed, replica_idx); + self.check_commit_prefix_contiguity( + seed, + replica_idx, + None, + sim.metadata_commit_prefix_hole(usize::from(replica_idx)), + ); for &ns in &workload.options.namespaces { + self.check_commit_prefix_contiguity( + seed, + replica_idx, + Some(ns), + sim.partition_commit_prefix_hole(usize::from(replica_idx), ns), + ); if let Some(offsets) = sim.offsets(usize::from(replica_idx), ns) { let cur = offsets.commit_offset; if let Some(&prev) = self.commit_offset.get(&(replica_idx, ns)) { @@ -102,6 +149,121 @@ impl Invariants { self.state_checker.check(sim, seed); } + /// Catch a metadata primary permanently shut behind its own recovery barrier. + /// + /// The shape `VsrConsensus::redecide_recovery_barrier` fixes, caught from the + /// outside: a primary that can never clear its barrier drops every request as + /// `NotReady`, which otherwise surfaces only as an unexplained stall. + /// + /// # Panics + /// When a `Normal` metadata primary sits below its barrier for + /// [`RECOVERY_BARRIER_WEDGE_TICKS`] consecutive ticks. + fn check_recovery_barrier(&mut self, sim: &Simulator, seed: u64, replica_idx: u8) { + let Some(consensus) = sim.metadata_consensus(usize::from(replica_idx)) else { + return; + }; + // Mirrors `consensus::is_caught_up_primary`, `!is_transferring` included: + // a primary in state transfer is shut by the transfer, not the barrier, so + // blaming the barrier is a false panic. + // + // Both gates, because they read different counters: admission compares + // `commit_max`, the HTTP read gate (`server/src/http/reads.rs`) compares + // `commit_min`. A barrier met by one and never the other 503s every read + // while admission looks open, which is the half + // `redecide_recovery_barrier` exists for. + let gated = consensus.is_primary() + && !consensus.has_ceded_primaryship() + && consensus.is_normal() + && !consensus.is_transferring() + && consensus.recovery_barrier() > 0 + && (consensus.commit_max() < consensus.recovery_barrier() + || consensus.commit_min() < consensus.recovery_barrier()); + if !gated { + self.barrier_gated_ticks.remove(&replica_idx); + return; + } + let ticks = self + .barrier_gated_ticks + .entry(replica_idx) + .and_modify(|ticks| *ticks += 1) + .or_insert(1); + assert!( + *ticks < RECOVERY_BARRIER_WEDGE_TICKS, + "replica {replica_idx} has been a Normal metadata primary gated by its recovery \ + barrier for {ticks} ticks: barrier={} commit={}..{} view={}. Nothing lowers the \ + barrier, so this primary drops every client request (commit_max gate) or 503s \ + every local read (commit_min gate) from here on (seed={seed:#x})", + consensus.recovery_barrier(), + consensus.commit_min(), + consensus.commit_max(), + consensus.view(), + ); + } + + /// Catch a commit walk permanently held below a committable pipeline head. + /// + /// The state `drain_committable_prefix` and `peek_committable_head` refuse to + /// drain. Two faults share that refusal and they are not the same wait, so they + /// are counted apart (see [`CommitHoldKind`]): + /// + /// - [`CommitHoldKind::MissingOps`] is legitimate while repair runs, so the + /// count restarts when `commit_min` advances and only a hole nothing refills + /// trips it. + /// - [`CommitHoldKind::AppliedHead`] never self-clears; only a view change + /// does. Resetting it on `commit_min` would be exactly wrong, since the head + /// is frozen while `commit_min` climbs. + /// + /// # Panics + /// When one plane holds past the threshold for its hold kind. + fn check_commit_prefix_contiguity( + &mut self, + seed: u64, + replica_idx: u8, + namespace: Option, + hole: Option, + ) { + let key = (replica_idx, namespace); + let Some(hole) = hole else { + self.commit_hole_ticks.remove(&key); + return; + }; + let (limit, diagnosis) = match hole.kind { + CommitHoldKind::MissingOps => ( + COMMIT_PREFIX_HOLE_WEDGE_TICKS, + "the ops between are missing and nothing is refilling them, so every reply \ + above the hole is owed forever", + ), + CommitHoldKind::AppliedHead => ( + COMMIT_PREFIX_APPLIED_HEAD_TICKS, + "the commit point moved past this entry, so nothing is missing and no repair \ + is owed: only a view change can clear it, and none has", + ), + }; + let entry = self + .commit_hole_ticks + .entry(key) + .or_insert((0, hole.commit_min)); + // A missing-op hold whose `commit_min` moved is repair landing, so the run + // starts over. An applied-head hold is measured while `commit_min` climbs, + // so it must not. + if hole.kind == CommitHoldKind::MissingOps && entry.1 != hole.commit_min { + *entry = (0, hole.commit_min); + } + entry.0 += 1; + let ticks = entry.0; + assert!( + ticks < limit, + "replica {replica_idx} has held its commit walk below a committable pipeline head \ + for {ticks} ticks on {} ({:?}): head_op={} commit={}..{}. {diagnosis} \ + (seed={seed:#x})", + namespace.map_or_else(|| "the metadata plane".to_owned(), |ns| format!("{ns:?}")), + hole.kind, + hole.head_op, + hole.commit_min, + hole.commit_max, + ); + } + /// The canonical committed chain built so far. Tests read it to prove the /// equality check compared replicas against each other rather than passing /// over an empty chain. diff --git a/core/simulator/src/workload/oracle.rs b/core/simulator/src/workload/oracle.rs index 4c84bdfe68..18b4b714dc 100644 --- a/core/simulator/src/workload/oracle.rs +++ b/core/simulator/src/workload/oracle.rs @@ -435,10 +435,20 @@ pub fn assert_converged(sim: &Simulator, workload: &mut Workload) -> Convergence } } + // Partition-plane twin of the walk above: that one bounds how much each + // replica committed, this compares what they committed. + let partition_ops_compared = + state_checker::assert_partition_prefixes_agree(sim, &workload.options.namespaces, seed); + tracing::info!( + partition_ops_compared, + "committed partition entries agree across every live replica" + ); + let replicas_compared = assert_committed_metadata_agrees(sim, &live, seed); let report = ConvergenceReport { ops_compared, + partition_ops_compared, replicas_compared, namespaces_checked, }; @@ -491,6 +501,10 @@ pub fn assert_converged(sim: &Simulator, workload: &mut Workload) -> Convergence pub struct ConvergenceReport { /// Committed metadata ops witnessed by more than one live replica. pub ops_compared: usize, + /// Committed partition ops witnessed by more than one live replica, summed over + /// every namespace. Separate from `ops_compared`: a partition-plane run commits + /// almost no metadata and would otherwise read as having compared nothing. + pub partition_ops_compared: usize, /// Live replicas whose committed metadata CONTENT was compared against a peer /// sharing its commit point. Zero on a solo cluster. pub replicas_compared: usize, diff --git a/core/simulator/src/workload/state_checker.rs b/core/simulator/src/workload/state_checker.rs index dd7e917291..53995a6ec6 100644 --- a/core/simulator/src/workload/state_checker.rs +++ b/core/simulator/src/workload/state_checker.rs @@ -32,9 +32,9 @@ //! non-vacuous: a chain nothing was compared against passes silently. use crate::Simulator; -use consensus::MetadataHandle; use iggy_binary_protocol::PrepareHeader; use journal::Journal; +use server_common::sharding::IggyNamespace; use std::collections::{BTreeMap, BTreeSet}; /// One op of the canonical committed chain. @@ -104,7 +104,7 @@ impl StateChecker { continue; } let replica = &sim.replicas[usize::from(replica_idx)]; - let Some(consensus) = replica.shards[0].plane.metadata().consensus.as_ref() else { + let Some(consensus) = sim.metadata_consensus(usize::from(replica_idx)) else { continue; }; let committed = consensus.commit_min(); @@ -254,7 +254,7 @@ pub fn assert_committed_prefixes_agree(sim: &Simulator, seed: u64) -> usize { continue; } let replica = &sim.replicas[usize::from(replica_idx)]; - let Some(consensus) = replica.shards[0].plane.metadata().consensus.as_ref() else { + let Some(consensus) = sim.metadata_consensus(usize::from(replica_idx)) else { continue; }; let committed = consensus.commit_min(); @@ -291,6 +291,71 @@ pub fn assert_committed_prefixes_agree(sim: &Simulator, seed: u64) -> usize { witnesses.values().filter(|&&count| count > 1).count() } +/// Assert every live replica's committed partition prefix agrees, op for op. +/// +/// Partition-plane counterpart of [`assert_committed_prefixes_agree`], and the only +/// check comparing what replicas committed on this plane rather than how much. The +/// quiesce oracle otherwise bounds only that no backup committed past its leader. +/// +/// Two differences from the metadata twin, both from the partition journal evicting +/// its committed prefix as it flushes to segments: +/// +/// * A missing header is ordinary, not a hole, so this compares only the ops two +/// replicas both still hold. `evict_prefix` moves a flushed entry to the repair +/// ring rather than dropping it, so that set is the ring plus whatever is still +/// resident: the recently committed tail, where a bad repair or a mis-decided +/// view change lands. +/// * Identity, not the sealed checksum. `restamp_prepare_view` rewrites the view on +/// retransmit, so `identity_checksum` compares the entry, not the delivery. +/// +/// Returns ops witnessed by more than one replica, summed over every namespace. +/// +/// # Panics +/// If two live replicas hold different entries at the same committed partition op. +#[must_use] +pub fn assert_partition_prefixes_agree( + sim: &Simulator, + namespaces: &[IggyNamespace], + seed: u64, +) -> usize { + let mut compared = 0; + for &namespace in namespaces { + let mut canonical: BTreeMap = BTreeMap::new(); + let mut witnesses: BTreeMap = BTreeMap::new(); + for replica_idx in 0..sim.replica_count { + if sim.is_crashed(replica_idx) { + continue; + } + let idx = usize::from(replica_idx); + let Some(state) = sim.partition_consensus_state(idx, namespace) else { + continue; + }; + let Some(headers) = + sim.partition_journaled_headers(idx, namespace, 1..=state.commit_min) + else { + continue; + }; + for (op, header) in headers { + let identity = header.identity_checksum(); + if let Some(&(expected, owner)) = canonical.get(&op) { + assert_eq!( + identity, expected, + "at quiesce replica {replica_idx} and replica {owner} disagree on \ + committed partition op {op} of ns {namespace:?}: {identity:#x} vs \ + {expected:#x} (seed={seed:#x})", + ); + *witnesses.entry(op).or_insert(1) += 1; + } else { + canonical.insert(op, (identity, replica_idx)); + witnesses.insert(op, 1); + } + } + } + compared += witnesses.values().filter(|&&count| count > 1).count(); + } + compared +} + /// Whether a committed op having no journal header is legitimate rather than a /// hole. /// @@ -320,6 +385,7 @@ mod tests { use super::*; use crate::client::SimClient; use crate::packet::PacketSimulatorOptions; + use consensus::PartitionsHandle; const SEED: u64 = 0x5C11; @@ -359,11 +425,8 @@ mod tests { #[should_panic(expected = "a hole in the committed log")] fn a_missing_committed_head_above_the_snapshot_floor_is_a_hole() { let sim = cluster_with_committed_ops(); - let committed = sim.replicas[1].shards[0] - .plane - .metadata() - .consensus - .as_ref() + let committed = sim + .metadata_consensus(1) .expect("shard 0 owns metadata consensus") .commit_min(); assert!( @@ -406,6 +469,97 @@ mod tests { checker.check(&sim, SEED); } + /// A three-replica cluster with committed partition ops flushed to segments. + /// + /// The state the partition comparison has to work in: `evict_prefix` clears the + /// resident header vec on flush, so a resident-only read sees nothing. + fn cluster_with_flushed_partition_ops() -> (Simulator, IggyNamespace) { + server_common::MemoryPool::init_pool(&server_common::MemoryPoolSettings { + enabled: false, + size: iggy_common::IggyByteSize::from(0u64), + bucket_capacity: 1, + }); + let client_id: u128 = 1; + let mut sim = Simulator::new( + 3, + std::iter::once(client_id), + PacketSimulatorOptions { + node_count: 3, + client_count: 1, + seed: SEED, + ..PacketSimulatorOptions::default() + }, + ); + let client = SimClient::new(client_id); + let namespace = IggyNamespace::new(1, 1, 0); + sim.init_partition(namespace); + sim.register_client_with_primary(&client); + for sequence in 0..6u32 { + let msg = client.send_messages( + namespace, + &[bytes::Bytes::from(format!("wl-part-{sequence}"))], + ); + sim.submit_request(client_id, 0, msg.into_generic()); + for _ in 0..60 { + sim.step(); + } + } + + for replica_idx in 0..3usize { + let shard = sim.replicas[replica_idx].partition_shard(namespace); + let partitions = shard.plane.partitions(); + let config = partitions.config(); + let Some(partition) = partitions.get_mut_by_ns(&namespace) else { + continue; + }; + futures::executor::block_on(partition.flush_committed_messages(config)) + .expect("the in-memory flush must succeed"); + } + (sim, namespace) + } + + /// The partition comparison must survive a flush. + /// + /// `header_by_op` reads the resident header vec, which `evict_prefix` clears as + /// the committed prefix flushes, so a resident-only read compares zero ops at + /// quiescence and the check passes over nothing. + #[test] + fn a_flushed_partition_prefix_is_still_compared() { + let (sim, namespace) = cluster_with_flushed_partition_ops(); + + let committed = sim + .partition_consensus_state(1, namespace) + .expect("replica 1 hosts the namespace") + .commit_min; + assert!( + committed > 1, + "the cluster committed {committed} partition op(s), so there is nothing \ + for two replicas to agree on" + ); + + let resident = sim.replicas[1] + .partition_shard(namespace) + .plane + .partitions() + .get_by_ns(&namespace) + .expect("replica 1 hosts the namespace") + .log + .journal() + .inner + .header_by_op(1); + assert!( + resident.is_none(), + "op 1 is still resident, so this test does not exercise the flushed path" + ); + + let compared = assert_partition_prefixes_agree(&sim, &[namespace], SEED); + assert!( + compared > 0, + "no committed partition op was witnessed by more than one replica after \ + the flush, so the check passed over an empty set" + ); + } + /// An undamaged prefix passes the recheck: the reset must turn a restart into a /// comparison, not into a failure. #[test] From 8dde7e7f67ec09242207d6137255d17ad812136f Mon Sep 17 00:00:00 2001 From: Maciej Modzelewski Date: Wed, 9 Sep 2026 07:32:41 +0200 Subject: [PATCH 090/182] feat(java): let TCP clients run one I/O thread or share a group (#4096) Every AsyncIggyTcpClient built its own event loop group with Netty's default of 2 x CPU threads. Its pool holds one channel, and a channel lives on one loop, so the other threads sat idle. Nothing let clients share a group. An application that opened one client per producer and consumer paid a thread count that grew with the client count. The group that a client creates for itself now has one thread by default, sized through ioThreads on the builder. eventLoopGroup registers the channel on a caller-owned group instead, and the client never shuts that group down. The builder rejects a group that cannot drive NIO channels or that already started to shut down, and ignores ioThreads when a group is supplied. The blocking builder forwards both options. The connection takes one loop from the group at construction and pins both the channel and the heartbeat to it. A next() call per heartbeat tick rotates the timer across loops the channel never uses. A second next() call, such as a pool handed the whole group, takes a slot of the group's round-robin counter and lands later connections off the sequence. Shutting the owned group down swept every channel. A shared group stays up, so close() now closes the tracked channels itself before it closes the pool. The pool closes only idle channels, and a login holds its lease until the reply, so the pool alone left that channel open. If a caller shuts the group down under a live connection, the heartbeat stops with a warning and close() completes instead of failing on a terminated loop. Refs #4021 --- foreign/java/README.md | 30 ++ .../client/async/tcp/AsyncIggyTcpClient.java | 32 +- .../async/tcp/AsyncIggyTcpClientBuilder.java | 79 ++- .../client/async/tcp/AsyncTcpConnection.java | 88 +++- .../blocking/tcp/IggyTcpClientBuilder.java | 31 ++ .../tcp/AsyncIggyTcpClientBuilderTest.java | 94 ++++ .../AsyncIggyTcpClientEventLoopGroupTest.java | 464 ++++++++++++++++++ .../AsyncTcpConnectionConcurrencyTest.java | 4 + 8 files changed, 799 insertions(+), 23 deletions(-) create mode 100644 foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/AsyncIggyTcpClientEventLoopGroupTest.java diff --git a/foreign/java/README.md b/foreign/java/README.md index ec18b00c90..949d8c7cda 100644 --- a/foreign/java/README.md +++ b/foreign/java/README.md @@ -182,6 +182,36 @@ var client = Iggy.tcpClientBuilder() .buildAndLogin(); ``` +### Event Loop Threads + +Each TCP client drives a single connection, so by default it creates an event loop group +with one thread. An application that opens many clients can instead register them all on +one caller-owned group. The clients never shut that group down. Close the clients first, +then shut the group down: + +```java +var group = new MultiThreadIoEventLoopGroup(2, NioIoHandler.newFactory()); + +var producer = Iggy.tcpClientBuilder() + .blocking() + .eventLoopGroup(group) + .credentials("iggy", "iggy") + .buildAndLogin(); +var consumer = Iggy.tcpClientBuilder() + .blocking() + .eventLoopGroup(group) + .credentials("iggy", "iggy") + .buildAndLogin(); + +// ... later +producer.close(); +consumer.close(); +group.shutdownGracefully(); +``` + +Do not block in a completion callback. Callbacks run on the group's loops, so a blocked +callback stalls every client that shares the group. + ### Version Information ```java diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/AsyncIggyTcpClient.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/AsyncIggyTcpClient.java index d65622f532..e0a88922b2 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/AsyncIggyTcpClient.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/AsyncIggyTcpClient.java @@ -22,6 +22,7 @@ import io.netty.buffer.ByteBuf; import io.netty.buffer.Unpooled; import io.netty.channel.ConnectTimeoutException; +import io.netty.channel.IoEventLoopGroup; import org.apache.iggy.IggyVersion; import org.apache.iggy.client.ConnectionInfo; import org.apache.iggy.client.async.ConsumerGroupsClient; @@ -108,8 +109,12 @@ * response handling is performed asynchronously. * *

Resource Management

- *

Always call {@link #close()} when the client is no longer needed. This shuts down - * the Netty event loop group and releases all associated resources. + *

When the client is no longer needed, call {@link #close()}. This closes the + * connection and shuts down the event loop group that the client created for itself. A + * group supplied through {@link AsyncIggyTcpClientBuilder#eventLoopGroup(IoEventLoopGroup)} + * stays up. The caller owns that group. After every client on the group is closed, the + * caller shuts the group down. Do not block in a completion callback. Callbacks run on + * that group's loops, so a blocked callback stalls every client that shares the group. * * @see AsyncIggyTcpClientBuilder * @see org.apache.iggy.Iggy#tcpClientBuilder() @@ -141,6 +146,8 @@ public class AsyncIggyTcpClient { private final Optional retryPolicy; private final boolean enableTls; private final Optional tlsCertificate; + private final Optional sharedEventLoopGroup; + private final int ioThreads; private final TcpConnectionPoolConfig poolConfig; private final ClientRoutingState routingState = new ClientRoutingState(); private final LoginRoutingHook loginRoutingHook = new LoginRoutingHook() { @@ -207,7 +214,9 @@ public AsyncIggyTcpClient(String host, int port) { VsrFrameDecoder.DEFAULT_MAX_FRAME_SIZE, null, false, - Optional.empty()); + Optional.empty(), + Optional.empty(), + AsyncTcpConnection.DEFAULT_IO_THREADS); } @SuppressWarnings("checkstyle:ParameterNumber") @@ -223,7 +232,9 @@ public AsyncIggyTcpClient(String host, int port) { int maxVsrFrameSize, RetryPolicy retryPolicy, boolean enableTls, - Optional tlsCertificate) { + Optional tlsCertificate, + Optional sharedEventLoopGroup, + int ioThreads) { this.connectionInfo = new ConnectionInfo(host, port); this.seedConnectionInfo = this.connectionInfo; this.username = Optional.ofNullable(username); @@ -236,6 +247,8 @@ public AsyncIggyTcpClient(String host, int port) { this.retryPolicy = Optional.ofNullable(retryPolicy); this.enableTls = enableTls; this.tlsCertificate = tlsCertificate; + this.sharedEventLoopGroup = sharedEventLoopGroup; + this.ioThreads = ioThreads; var poolConfigBuilder = TcpConnectionPoolConfig.builder(); this.acquireTimeout.ifPresent(timeout -> poolConfigBuilder.setAcquireTimeoutMillis(timeout.toMillis())); @@ -460,10 +473,13 @@ public ConsumerOffsetsClient consumerOffsets() { } /** - * Closes the TCP connection and releases all Netty resources. + * Closes the TCP connection and releases the Netty resources that this client owns. * - *

This shuts down the event loop group gracefully. After calling this method, - * the client cannot be reused — create a new instance if needed. + *

If the client created its own event loop group, it shuts that group down gracefully. + * The client does not shut down a group supplied through + * {@link AsyncIggyTcpClientBuilder#eventLoopGroup(IoEventLoopGroup)}. The caller shuts + * that group down. After you call this method, you cannot reuse the client. If you need + * a client again, create a new instance. * * @return a {@link CompletableFuture} that completes when all resources are released */ @@ -498,6 +514,8 @@ private AsyncTcpConnection openConnection(ConnectionInfo target) { enableTls, tlsCertificate, poolConfig, + sharedEventLoopGroup, + ioThreads, dialTimeout(), requestTimeout, heartbeatInterval, diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/AsyncIggyTcpClientBuilder.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/AsyncIggyTcpClientBuilder.java index e6a6ed510b..95d9833095 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/AsyncIggyTcpClientBuilder.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/AsyncIggyTcpClientBuilder.java @@ -19,6 +19,10 @@ package org.apache.iggy.client.async.tcp; +import io.netty.channel.IoEventLoop; +import io.netty.channel.IoEventLoopGroup; +import io.netty.channel.nio.NioIoHandle; +import io.netty.util.concurrent.EventExecutor; import org.apache.commons.lang3.StringUtils; import org.apache.iggy.client.async.tcp.vsr.VsrFrameDecoder; import org.apache.iggy.client.async.tcp.vsr.VsrHeaders; @@ -61,6 +65,13 @@ * .credentials("admin", "secret") * .buildAndLogin() * .join(); + * + * // Many clients on one caller-owned event loop group + * var group = new MultiThreadIoEventLoopGroup(2, NioIoHandler.newFactory()); + * var producer = AsyncIggyTcpClient.builder().eventLoopGroup(group).build(); + * var consumer = AsyncIggyTcpClient.builder().eventLoopGroup(group).build(); + * // ... close both clients, then: + * group.shutdownGracefully(); * } * * @see AsyncIggyTcpClient#builder() @@ -78,6 +89,8 @@ public final class AsyncIggyTcpClientBuilder { private Duration acquireTimeout; private Duration heartbeatInterval = Duration.ofSeconds(5); private long maxVsrFrameSize = VsrFrameDecoder.DEFAULT_MAX_FRAME_SIZE; + private int ioThreads = AsyncTcpConnection.DEFAULT_IO_THREADS; + private IoEventLoopGroup eventLoopGroup; public AsyncIggyTcpClientBuilder() {} @@ -227,6 +240,42 @@ public AsyncIggyTcpClientBuilder retryPolicy(RetryPolicy retryPolicy) { return this; } + /** + * Sets the number of event loop threads in the group that the client creates for itself. + * + *

The client drives a single channel, so the default of 1 is enough. If + * {@link #eventLoopGroup(IoEventLoopGroup)} is set, the client ignores this value. + * + * @param ioThreads the event loop thread count, at least 1 + * @return this builder + */ + public AsyncIggyTcpClientBuilder ioThreads(int ioThreads) { + this.ioThreads = ioThreads; + return this; + } + + /** + * Sets a caller-owned event loop group shared across clients. + * + *

The client pins its channel to one loop of the group and never shuts the group down. + * After every client on the group is closed, the caller shuts the group down. The group + * must drive NIO channels, for example + * {@code new MultiThreadIoEventLoopGroup(threads, NioIoHandler.newFactory())}. + * Do not block in a completion callback. Callbacks run on the group's loops, so a + * blocked callback stalls every client that shares the group. + * + * @param eventLoopGroup the group to register the client's channel on + * @return this builder + * @throws IggyInvalidArgumentException if the group is null + */ + public AsyncIggyTcpClientBuilder eventLoopGroup(IoEventLoopGroup eventLoopGroup) { + if (eventLoopGroup == null) { + throw new IggyInvalidArgumentException("EventLoopGroup cannot be null"); + } + this.eventLoopGroup = eventLoopGroup; + return this; + } + /** * Builds and returns a configured AsyncIggyTcpClient instance. * Note: You still need to call {@link AsyncIggyTcpClient#connect()} on the returned client. @@ -242,6 +291,8 @@ public AsyncIggyTcpClient build() { validateRequestTimeout(); validateHeartbeatInterval(); validateMaxVsrFrameSize(); + validateIoThreads(); + validateEventLoopGroup(); return new AsyncIggyTcpClient( host, @@ -255,7 +306,9 @@ public AsyncIggyTcpClient build() { (int) maxVsrFrameSize, retryPolicy, enableTls, - Optional.ofNullable(tlsCertificate)); + Optional.ofNullable(tlsCertificate), + Optional.ofNullable(eventLoopGroup), + ioThreads); } private void validateHost() { @@ -308,6 +361,30 @@ private void validateMaxVsrFrameSize() { } } + private void validateIoThreads() { + if (eventLoopGroup != null) { + return; + } + if (ioThreads < 1) { + throw new IggyInvalidArgumentException("IoThreads must be at least 1"); + } + } + + private void validateEventLoopGroup() { + if (eventLoopGroup == null) { + return; + } + for (EventExecutor executor : eventLoopGroup) { + if (!(executor instanceof IoEventLoop loop) || !loop.isCompatible(NioIoHandle.class)) { + throw new IggyInvalidArgumentException( + "EventLoopGroup must drive NIO channels, for example MultiThreadIoEventLoopGroup with NioIoHandler"); + } + } + if (eventLoopGroup.isShuttingDown()) { + throw new IggyInvalidArgumentException("EventLoopGroup shutdown already started"); + } + } + /** * Builds, connects, and logs in using the provided credentials. * This is a convenience method equivalent to calling {@code build()}, {@code connect()}, diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/AsyncTcpConnection.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/AsyncTcpConnection.java index bdf97f90ad..656103416e 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/AsyncTcpConnection.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/async/tcp/AsyncTcpConnection.java @@ -27,8 +27,11 @@ import io.netty.channel.ChannelOption; import io.netty.channel.ChannelPipeline; import io.netty.channel.ConnectTimeoutException; +import io.netty.channel.EventLoop; import io.netty.channel.IoEventLoopGroup; import io.netty.channel.MultiThreadIoEventLoopGroup; +import io.netty.channel.group.ChannelGroup; +import io.netty.channel.group.DefaultChannelGroup; import io.netty.channel.nio.NioIoHandler; import io.netty.channel.pool.AbstractChannelPoolHandler; import io.netty.channel.pool.ChannelHealthChecker; @@ -37,7 +40,10 @@ import io.netty.handler.ssl.SslContext; import io.netty.handler.ssl.SslContextBuilder; import io.netty.handler.ssl.SslHandler; +import io.netty.util.concurrent.DefaultThreadFactory; +import io.netty.util.concurrent.Future; import io.netty.util.concurrent.FutureListener; +import io.netty.util.concurrent.GlobalEventExecutor; import io.netty.util.concurrent.ScheduledFuture; import org.apache.iggy.client.ConnectionInfo; import org.apache.iggy.client.async.tcp.vsr.ConsensusSession; @@ -91,6 +97,8 @@ public class AsyncTcpConnection { // and a transient one is not a rejected credential. static final int TRANSIENT_NOT_COMMITTED = 57; static final int TRANSIENT_NOT_ACCEPTED = 58; + // The pool holds one channel, and one channel lives on one loop. + static final int DEFAULT_IO_THREADS = 1; private static final Logger log = LoggerFactory.getLogger(AsyncTcpConnection.class); private static final Duration DEFAULT_CONNECTION_TIMEOUT = Duration.ofMillis(3000); // A missing reply must not hold the single VSR-pinned channel forever. @@ -98,9 +106,13 @@ public class AsyncTcpConnection { private static final long TRANSIENT_RETRY_INTERVAL_MS = 50; private static final Duration TRANSIENT_RETRY_BUDGET = Duration.ofSeconds(30); private static final Duration NOT_ACCEPTED_RETRY_BUDGET = Duration.ofSeconds(2); + private static final String EVENT_LOOP_THREAD_PREFIX = "iggy-tcp-io"; private final IoEventLoopGroup eventLoopGroup; + private final boolean ownsEventLoopGroup; + private final EventLoop eventLoop; private final FixedChannelPool channelPool; + private final ChannelGroup channels = new DefaultChannelGroup(GlobalEventExecutor.INSTANCE, true); private final AtomicBoolean isClosed = new AtomicBoolean(false); private final AtomicLong authGeneration = new AtomicLong(0); private final VsrRequestEncoder vsrEncoder; @@ -130,6 +142,8 @@ public AsyncTcpConnection( enableTls, tlsCertificate, poolConfig, + Optional.empty(), + DEFAULT_IO_THREADS, connectionTimeout, Optional.empty(), Duration.ofSeconds(5), @@ -146,6 +160,8 @@ public AsyncTcpConnection( boolean enableTls, Optional tlsCertificate, TcpConnectionPoolConfig poolConfig, + Optional sharedEventLoopGroup, + int ioThreads, Optional connectionTimeout, Optional requestTimeout, Duration heartbeatInterval, @@ -171,12 +187,17 @@ public AsyncTcpConnection( ConsensusSession consensusSession = new ConsensusSession(); this.vsrEncoder = new VsrRequestEncoder(consensusSession); - this.eventLoopGroup = new MultiThreadIoEventLoopGroup(NioIoHandler.newFactory()); + this.ownsEventLoopGroup = sharedEventLoopGroup.isEmpty(); + this.eventLoopGroup = sharedEventLoopGroup.orElseGet(() -> new MultiThreadIoEventLoopGroup( + ioThreads, + new DefaultThreadFactory(EVENT_LOOP_THREAD_PREFIX, false, Thread.MAX_PRIORITY), + NioIoHandler.newFactory())); + this.eventLoop = eventLoopGroup.next(); long dialTimeoutMillis = connectionTimeout.orElse(DEFAULT_CONNECTION_TIMEOUT).toMillis(); var bootstrap = new Bootstrap() - .group(eventLoopGroup) + .group(eventLoop) .channel(NioSocketChannel.class) .option(ChannelOption.TCP_NODELAY, true) .option(ChannelOption.CONNECT_TIMEOUT_MILLIS, (int) dialTimeoutMillis) @@ -196,7 +217,8 @@ public AsyncTcpConnection( dialTimeoutMillis, consensusSession, maxVsrFrameSize, - this::onSessionEvicted), + this::onSessionEvicted, + channels::add), ChannelHealthChecker.ACTIVE, FixedChannelPool.AcquireTimeoutAction.FAIL, poolConfig.getAcquireTimeoutMillis(), @@ -248,8 +270,13 @@ private void scheduleNextHeartbeat() { if (!heartbeatRunning || isClosed.get()) { return; } - heartbeatTask = - eventLoopGroup.next().schedule(this::sendHeartbeat, heartbeatIntervalNanos, TimeUnit.NANOSECONDS); + try { + heartbeatTask = eventLoop.schedule(this::sendHeartbeat, heartbeatIntervalNanos, TimeUnit.NANOSECONDS); + } catch (RejectedExecutionException loopGone) { + // Only a caller-owned group shuts down under a live connection. + heartbeatRunning = false; + log.warn("Event loop rejected the heartbeat, stopping it: {}", loopGone.getMessage()); + } } } @@ -289,6 +316,16 @@ private void stopHeartbeat() { } } + boolean heartbeatScheduled() { + synchronized (heartbeatLock) { + return heartbeatTask != null; + } + } + + EventLoop eventLoop() { + return eventLoop; + } + public CompletableFuture exchangeForEntity( CommandCode commandCode, ByteBuf payload, Function func) { return send(commandCode, payload).thenApply(response -> { @@ -920,18 +957,35 @@ public CompletableFuture close() { stopHeartbeat(); releaseLoginPayload(); CompletableFuture shutdownFuture = new CompletableFuture<>(); - channelPool - .closeAsync() - .addListener(f -> eventLoopGroup.shutdownGracefully().addListener(sf -> { - if (sf.isSuccess()) { - shutdownFuture.complete(null); - } else { - shutdownFuture.completeExceptionally(sf.cause()); - } - })); + channels.close().addListener(channelsClosed -> closePool(shutdownFuture)); return shutdownFuture; } + private void closePool(CompletableFuture shutdownFuture) { + try { + channelPool.closeAsync().addListener(poolClosed -> { + if (!ownsEventLoopGroup) { + completeShutdown(shutdownFuture, poolClosed); + return; + } + eventLoopGroup + .shutdownGracefully() + .addListener(groupClosed -> completeShutdown(shutdownFuture, groupClosed)); + }); + } catch (RejectedExecutionException loopGone) { + log.warn("Event loop rejected the pool close, channel already gone: {}", loopGone.getMessage()); + shutdownFuture.complete(null); + } + } + + private static void completeShutdown(CompletableFuture shutdownFuture, Future step) { + if (step.isSuccess()) { + shutdownFuture.complete(null); + } else { + shutdownFuture.completeExceptionally(step.cause()); + } + } + private static final class PoolChannelHandler extends AbstractChannelPoolHandler { private final String host; private final int port; @@ -941,6 +995,7 @@ private static final class PoolChannelHandler extends AbstractChannelPoolHandler private final ConsensusSession consensusSession; private final int maxVsrFrameSize; private final IntConsumer onEviction; + private final Consumer onChannelCreated; @SuppressWarnings("checkstyle:ParameterNumber") PoolChannelHandler( @@ -951,7 +1006,8 @@ private static final class PoolChannelHandler extends AbstractChannelPoolHandler long dialTimeoutMillis, ConsensusSession consensusSession, int maxVsrFrameSize, - IntConsumer onEviction) { + IntConsumer onEviction, + Consumer onChannelCreated) { this.host = host; this.port = port; this.enableTls = enableTls; @@ -960,10 +1016,12 @@ private static final class PoolChannelHandler extends AbstractChannelPoolHandler this.consensusSession = consensusSession; this.maxVsrFrameSize = maxVsrFrameSize; this.onEviction = onEviction; + this.onChannelCreated = onChannelCreated; } @Override public void channelCreated(Channel ch) { + onChannelCreated.accept(ch); ChannelPipeline pipeline = ch.pipeline(); if (enableTls) { SslHandler ssl = sslContext.newHandler(ch.alloc(), host, port); diff --git a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/blocking/tcp/IggyTcpClientBuilder.java b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/blocking/tcp/IggyTcpClientBuilder.java index 65a09d215a..e79ecfa3ac 100644 --- a/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/blocking/tcp/IggyTcpClientBuilder.java +++ b/foreign/java/java-sdk/src/main/java/org/apache/iggy/client/blocking/tcp/IggyTcpClientBuilder.java @@ -19,6 +19,7 @@ package org.apache.iggy.client.blocking.tcp; +import io.netty.channel.IoEventLoopGroup; import org.apache.iggy.client.async.tcp.AsyncIggyTcpClient; import org.apache.iggy.client.async.tcp.AsyncIggyTcpClientBuilder; import org.apache.iggy.config.RetryPolicy; @@ -136,6 +137,36 @@ public IggyTcpClientBuilder retryPolicy(RetryPolicy retryPolicy) { return this; } + /** + * Sets the number of event loop threads in the group that the client creates for itself. + * + *

The client drives a single channel, so the default of 1 is enough. If + * {@link #eventLoopGroup(IoEventLoopGroup)} is set, the client ignores this value. + * + * @param ioThreads the event loop thread count, at least 1 + * @return this builder + */ + public IggyTcpClientBuilder ioThreads(int ioThreads) { + asyncBuilder.ioThreads(ioThreads); + return this; + } + + /** + * Sets a caller-owned event loop group shared across clients. + * + *

The client registers its channel on the group and never shuts the group down. + * After every client on the group is closed, the caller shuts the group down. The group + * must drive NIO channels. + * + * @param eventLoopGroup the group to register the client's channel on + * @return this builder + * @throws org.apache.iggy.exception.IggyInvalidArgumentException if the group is null + */ + public IggyTcpClientBuilder eventLoopGroup(IoEventLoopGroup eventLoopGroup) { + asyncBuilder.eventLoopGroup(eventLoopGroup); + return this; + } + /** * Enables or disables TLS for the TCP connection. * diff --git a/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/AsyncIggyTcpClientBuilderTest.java b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/AsyncIggyTcpClientBuilderTest.java index ce7234a35f..248486b2d0 100644 --- a/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/AsyncIggyTcpClientBuilderTest.java +++ b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/AsyncIggyTcpClientBuilderTest.java @@ -19,6 +19,10 @@ package org.apache.iggy.client.async.tcp; +import io.netty.channel.IoEventLoopGroup; +import io.netty.channel.MultiThreadIoEventLoopGroup; +import io.netty.channel.local.LocalIoHandler; +import io.netty.channel.nio.NioIoHandler; import org.apache.iggy.client.BaseIntegrationTest; import org.apache.iggy.config.RetryPolicy; import org.apache.iggy.exception.IggyAuthenticationException; @@ -250,6 +254,96 @@ void shouldRejectNonPositiveHeartbeatInterval() { .isInstanceOf(IggyInvalidArgumentException.class); } + @Test + void shouldRejectNonPositiveIoThreads() { + assertThatThrownBy(() -> AsyncIggyTcpClient.builder().ioThreads(0).build()) + .isInstanceOf(IggyInvalidArgumentException.class); + assertThatThrownBy(() -> AsyncIggyTcpClient.builder().ioThreads(-1).build()) + .isInstanceOf(IggyInvalidArgumentException.class); + } + + @Test + void shouldIgnoreIoThreadsWhenAnEventLoopGroupIsSupplied() throws Exception { + IoEventLoopGroup group = new MultiThreadIoEventLoopGroup(1, NioIoHandler.newFactory()); + try { + AsyncIggyTcpClient.builder().ioThreads(0).eventLoopGroup(group).build(); + } finally { + group.shutdownGracefully(0, 1, TimeUnit.SECONDS).get(5, TimeUnit.SECONDS); + } + } + + @Test + void shouldRejectNullEventLoopGroup() { + assertThatThrownBy(() -> AsyncIggyTcpClient.builder().eventLoopGroup(null)) + .isInstanceOf(IggyInvalidArgumentException.class); + } + + @Test + void shouldRejectAnEventLoopGroupThatCannotDriveNioChannels() throws Exception { + IoEventLoopGroup localGroup = new MultiThreadIoEventLoopGroup(1, LocalIoHandler.newFactory()); + try { + assertThatThrownBy(() -> AsyncIggyTcpClient.builder() + .eventLoopGroup(localGroup) + .build()) + .isInstanceOf(IggyInvalidArgumentException.class) + .hasMessageContaining("NIO"); + } finally { + localGroup.shutdownGracefully(0, 1, TimeUnit.SECONDS).get(5, TimeUnit.SECONDS); + } + } + + @Test + void shouldRejectAShutDownEventLoopGroup() throws Exception { + IoEventLoopGroup group = new MultiThreadIoEventLoopGroup(1, NioIoHandler.newFactory()); + group.shutdownGracefully(0, 1, TimeUnit.SECONDS).get(5, TimeUnit.SECONDS); + + assertThatThrownBy( + () -> AsyncIggyTcpClient.builder().eventLoopGroup(group).build()) + .isInstanceOf(IggyInvalidArgumentException.class); + } + + @Test + void shouldShareOneEventLoopGroupAcrossClients() throws Exception { + IoEventLoopGroup group = new MultiThreadIoEventLoopGroup(1, NioIoHandler.newFactory()); + AsyncIggyTcpClient other = null; + try { + client = AsyncIggyTcpClient.builder() + .host(serverHost()) + .port(serverTcpPort()) + .eventLoopGroup(group) + .build(); + other = AsyncIggyTcpClient.builder() + .host(serverHost()) + .port(serverTcpPort()) + .eventLoopGroup(group) + .build(); + client.connect().get(TEST_TIMEOUT_SECONDS, TimeUnit.SECONDS); + other.connect().get(TEST_TIMEOUT_SECONDS, TimeUnit.SECONDS); + client.users().login(TEST_USERNAME, TEST_PASSWORD).get(TEST_TIMEOUT_SECONDS, TimeUnit.SECONDS); + other.users().login(TEST_USERNAME, TEST_PASSWORD).get(TEST_TIMEOUT_SECONDS, TimeUnit.SECONDS); + assertThat(client.streams().getStreams().get(TEST_TIMEOUT_SECONDS, TimeUnit.SECONDS)) + .isNotNull(); + assertThat(other.streams().getStreams().get(TEST_TIMEOUT_SECONDS, TimeUnit.SECONDS)) + .isNotNull(); + + client.close().get(TEST_TIMEOUT_SECONDS, TimeUnit.SECONDS); + client = null; + + assertThat(group.isShuttingDown()) + .as("closing a client must not take the shared group down") + .isFalse(); + assertThat(other.streams().getStreams().get(TEST_TIMEOUT_SECONDS, TimeUnit.SECONDS)) + .as("the other client keeps working on the shared group") + .isNotNull(); + } finally { + if (other != null) { + other.close().get(TEST_TIMEOUT_SECONDS, TimeUnit.SECONDS); + } + assertThat(group.isShuttingDown()).isFalse(); + group.shutdownGracefully(0, 1, TimeUnit.SECONDS).get(TEST_TIMEOUT_SECONDS, TimeUnit.SECONDS); + } + } + @Test void shouldMaintainBackwardCompatibilityWithOldConstructor() throws Exception { // Given: Old constructor approach diff --git a/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/AsyncIggyTcpClientEventLoopGroupTest.java b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/AsyncIggyTcpClientEventLoopGroupTest.java new file mode 100644 index 0000000000..24b0146dd7 --- /dev/null +++ b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/AsyncIggyTcpClientEventLoopGroupTest.java @@ -0,0 +1,464 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, + * software distributed under the License is distributed on an + * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY + * KIND, either express or implied. See the License for the + * specific language governing permissions and limitations + * under the License. + */ + +package org.apache.iggy.client.async.tcp; + +import io.netty.buffer.ByteBuf; +import io.netty.buffer.Unpooled; +import io.netty.channel.EventLoop; +import io.netty.channel.IoEventLoopGroup; +import io.netty.channel.MultiThreadIoEventLoopGroup; +import io.netty.channel.nio.NioIoHandler; +import org.apache.iggy.client.ConnectionInfo; +import org.junit.jupiter.api.Test; + +import java.io.EOFException; +import java.io.IOException; +import java.io.InputStream; +import java.io.OutputStream; +import java.net.InetAddress; +import java.net.ServerSocket; +import java.net.Socket; +import java.nio.ByteBuffer; +import java.nio.ByteOrder; +import java.nio.charset.StandardCharsets; +import java.time.Duration; +import java.util.ArrayList; +import java.util.List; +import java.util.Optional; +import java.util.Set; +import java.util.concurrent.CompletableFuture; +import java.util.concurrent.CopyOnWriteArrayList; +import java.util.concurrent.TimeUnit; +import java.util.concurrent.TimeoutException; +import java.util.concurrent.atomic.AtomicInteger; +import java.util.function.BooleanSupplier; +import java.util.stream.Collectors; + +import static org.assertj.core.api.Assertions.assertThat; +import static org.assertj.core.api.Assertions.assertThatCode; + +/** + * One client drives one channel, so the group that it creates for itself runs one loop + * and shuts down with the client. If the caller supplies a group, the client pins its + * channel to one loop of that group and never shuts the group down. + */ +class AsyncIggyTcpClientEventLoopGroupTest { + + private static final String OWNED_THREAD_PREFIX = "iggy-tcp-io-"; + private static final Duration HEARTBEAT_INTERVAL = Duration.ofMillis(25); + // Graceful shutdown waits a two-second quiet period before the loop thread exits. + private static final Duration SHUTDOWN_TIMEOUT = Duration.ofSeconds(15); + + private static final int HEADER_SIZE = 256; + private static final int SIZE_OFFSET = 48; + private static final int COMMAND_OFFSET = 60; + private static final int REQUEST_ID_OFFSET = 168; + private static final int REQUEST_OPERATION_OFFSET = 176; + private static final int REQUEST_CODE_OFFSET = 196; + private static final int REPLY_REQUEST_ID_OFFSET = 200; + private static final int REPLY_OPERATION_OFFSET = 208; + private static final int REPLY_STATUS_OFFSET = 216; + private static final int COMMAND_REPLY = 8; + private static final int OPERATION_REGISTER = 1; + private static final int OPERATION_NON_REPLICATED = 2; + private static final int PING_CODE = 1; + private static final int LOGIN_CODE = 38; + + @Test + void shouldRunOneEventLoopThreadPerClientByDefault() throws Exception { + try (MockVsrServer server = MockVsrServer.start()) { + Set before = ownedEventLoopThreads(); + AsyncIggyTcpClient client = builder(server).build(); + client.connect().get(5, TimeUnit.SECONDS); + + // Enough ticks for a round-robin heartbeat to wake more than one loop. + server.awaitPings(8, Duration.ofSeconds(5)); + assertThat(ownedEventLoopThreadsSince(before)) + .as("one channel needs one loop") + .isEqualTo(1); + + client.close().get(SHUTDOWN_TIMEOUT.toSeconds(), TimeUnit.SECONDS); + awaitOwnedEventLoopThreadsSince(before, 0, SHUTDOWN_TIMEOUT); + } + } + + @Test + void shouldLeaveASharedGroupRunningWhenClientsClose() throws Exception { + IoEventLoopGroup group = new MultiThreadIoEventLoopGroup(1, NioIoHandler.newFactory()); + try (MockVsrServer server = MockVsrServer.start()) { + Set before = ownedEventLoopThreads(); + AsyncIggyTcpClient first = builder(server).eventLoopGroup(group).build(); + AsyncIggyTcpClient second = builder(server).eventLoopGroup(group).build(); + first.connect().get(5, TimeUnit.SECONDS); + second.connect().get(5, TimeUnit.SECONDS); + assertThat(ownedEventLoopThreadsSince(before)) + .as("clients on a shared group create no group of their own") + .isZero(); + + first.close().get(5, TimeUnit.SECONDS); + assertThat(group.isShuttingDown()).isFalse(); + second.sendBinaryRequest(PING_CODE, new byte[0]).get(5, TimeUnit.SECONDS); + + second.close().get(5, TimeUnit.SECONDS); + assertThat(group.isShuttingDown()).isFalse(); + } finally { + group.shutdownGracefully(0, 1, TimeUnit.SECONDS).get(5, TimeUnit.SECONDS); + } + } + + @Test + void shouldPinEachConnectionOnASharedGroupToTheNextLoop() throws Exception { + IoEventLoopGroup group = new MultiThreadIoEventLoopGroup(4, NioIoHandler.newFactory()); + List loops = new ArrayList<>(); + group.forEach(executor -> loops.add((EventLoop) executor)); + List connections = new ArrayList<>(); + try (MockVsrServer server = MockVsrServer.start()) { + for (int i = 0; i < loops.size(); i++) { + AsyncTcpConnection connection = connection(server, group); + connection.connect().get(5, TimeUnit.SECONDS); + connections.add(connection); + } + server.awaitPings(loops.size() * 2, Duration.ofSeconds(5)); + + // A fresh group hands its loops out round-robin from the first. Every + // extra next() call per connection, such as a pool handed the whole + // group, skips loops and lands later connections off this sequence. + assertThat(connections.stream().map(AsyncTcpConnection::eventLoop).toList()) + .as("one connection takes one slot of the group's round-robin chooser") + .containsExactlyElementsOf(loops); + } finally { + for (AsyncTcpConnection connection : connections) { + connection.close().get(5, TimeUnit.SECONDS); + } + group.shutdownGracefully(0, 1, TimeUnit.SECONDS).get(5, TimeUnit.SECONDS); + } + } + + @Test + void shouldCloseQuietlyAfterTheCallerShutTheSharedGroupDown() throws Exception { + IoEventLoopGroup group = new MultiThreadIoEventLoopGroup(1, NioIoHandler.newFactory()); + try (MockVsrServer server = MockVsrServer.start()) { + AsyncIggyTcpClient client = builder(server).eventLoopGroup(group).build(); + client.connect().get(5, TimeUnit.SECONDS); + group.shutdownGracefully(0, 1, TimeUnit.SECONDS).get(5, TimeUnit.SECONDS); + + // The pool closes on its loop, and a terminated loop rejects that task. + assertThatCode(() -> client.close().get(5, TimeUnit.SECONDS)) + .as("nothing is left to release once the group took the channel down") + .doesNotThrowAnyException(); + } + } + + @Test + void shouldCloseTheChannelHeldByAPendingLoginOnASharedGroup() throws Exception { + IoEventLoopGroup group = new MultiThreadIoEventLoopGroup(1, NioIoHandler.newFactory()); + try (MockVsrServer server = MockVsrServer.start()) { + server.withholdRegisterReplies(); + // Far past the test bound, so only close() can settle the login. + AsyncTcpConnection connection = connection(server, group, Duration.ofMinutes(5)); + connection.connect().get(5, TimeUnit.SECONDS); + CompletableFuture login = connection.send(LOGIN_CODE, loginPayload()); + await(() -> server.registers() == 1, "the server holds the login", Duration.ofSeconds(5)); + + connection.close().get(5, TimeUnit.SECONDS); + + // The pool only closes idle channels, and a login holds its lease + // until the reply, so close() must reach that channel itself. + await(() -> server.closedSockets() == 1, "the login's socket is closed", Duration.ofSeconds(5)); + await(login::isDone, "the pending login settles", Duration.ofSeconds(5)); + assertThat(login).isCompletedExceptionally(); + assertThat(group.isShuttingDown()).isFalse(); + } finally { + group.shutdownGracefully(0, 1, TimeUnit.SECONDS).get(5, TimeUnit.SECONDS); + } + } + + @Test + void shouldCancelThePendingHeartbeatOnASharedGroupWhenTheConnectionCloses() throws Exception { + IoEventLoopGroup group = new MultiThreadIoEventLoopGroup(1, NioIoHandler.newFactory()); + try (MockVsrServer server = MockVsrServer.start()) { + AsyncTcpConnection connection = connection(server, group); + connection.connect().get(5, TimeUnit.SECONDS); + server.awaitPings(2, Duration.ofSeconds(5)); + await(connection::heartbeatScheduled, "a live connection keeps a tick armed", Duration.ofSeconds(5)); + + connection.close().get(5, TimeUnit.SECONDS); + + // A stray timer on a shared loop cannot be seen through the server: the + // channel is gone, so a tick that still fired would fail before sending. + assertThat(connection.heartbeatScheduled()) + .as("close cancels the armed tick instead of leaving it to the shared loop") + .isFalse(); + Thread.sleep(HEARTBEAT_INTERVAL.multipliedBy(4).toMillis()); + assertThat(connection.heartbeatScheduled()) + .as("nothing re-arms the heartbeat after close") + .isFalse(); + assertThat(group.isShuttingDown()).isFalse(); + } finally { + group.shutdownGracefully(0, 1, TimeUnit.SECONDS).get(5, TimeUnit.SECONDS); + } + } + + @Test + void shouldReleaseTheGroupOfAConnectionReplacedByRetarget() throws Exception { + try (MockVsrServer primary = MockVsrServer.start(); + MockVsrServer survivor = MockVsrServer.start()) { + Set before = ownedEventLoopThreads(); + AsyncIggyTcpClient client = builder(primary).build(); + client.connect().get(5, TimeUnit.SECONDS); + + client.retarget(new ConnectionInfo(loopback(), survivor.port())).get(5, TimeUnit.SECONDS); + assertThat(client.getConnectionInfo().port()).isEqualTo(survivor.port()); + + awaitOwnedEventLoopThreadsSince(before, 1, SHUTDOWN_TIMEOUT); + client.close().get(SHUTDOWN_TIMEOUT.toSeconds(), TimeUnit.SECONDS); + awaitOwnedEventLoopThreadsSince(before, 0, SHUTDOWN_TIMEOUT); + } + } + + private static AsyncIggyTcpClientBuilder builder(MockVsrServer server) { + return AsyncIggyTcpClient.builder() + .host(loopback()) + .port(server.port()) + .heartbeatInterval(HEARTBEAT_INTERVAL) + .requestTimeout(Duration.ofSeconds(1)); + } + + private static AsyncTcpConnection connection(MockVsrServer server, IoEventLoopGroup group) { + return connection(server, group, Duration.ofSeconds(1)); + } + + private static AsyncTcpConnection connection( + MockVsrServer server, IoEventLoopGroup group, Duration requestTimeout) { + return new AsyncTcpConnection( + loopback(), + server.port(), + false, + Optional.empty(), + new AsyncTcpConnection.TcpConnectionPoolConfig(1000, 3000), + Optional.of(group), + 1, + Optional.of(Duration.ofSeconds(1)), + Optional.of(requestTimeout), + HEARTBEAT_INTERVAL, + 1024 * 1024, + null, + errorCode -> {}, + ignored -> {}); + } + + private static String loopback() { + return InetAddress.getLoopbackAddress().getHostAddress(); + } + + private static ByteBuf loginPayload() { + ByteBuf payload = Unpooled.buffer(); + payload.writeByte(4); + payload.writeBytes("iggy".getBytes(StandardCharsets.UTF_8)); + payload.writeByte(4); + payload.writeBytes("iggy".getBytes(StandardCharsets.UTF_8)); + payload.writeIntLE(0); + payload.writeIntLE(0); + return payload; + } + + private static Set ownedEventLoopThreads() { + return Thread.getAllStackTraces().keySet().stream() + .filter(Thread::isAlive) + .filter(thread -> thread.getName().startsWith(OWNED_THREAD_PREFIX)) + .collect(Collectors.toSet()); + } + + // Loops that earlier tests left draining die on their own schedule, so + // only threads born after the snapshot count. + private static int ownedEventLoopThreadsSince(Set before) { + return (int) ownedEventLoopThreads().stream() + .filter(thread -> !before.contains(thread)) + .count(); + } + + private static void awaitOwnedEventLoopThreadsSince(Set before, int expected, Duration timeout) + throws Exception { + await( + () -> ownedEventLoopThreadsSince(before) == expected, + "Expected " + expected + " owned event loop threads started by this test", + timeout); + } + + private static void await(BooleanSupplier condition, String description, Duration timeout) throws Exception { + long deadline = System.nanoTime() + timeout.toNanos(); + while (!condition.getAsBoolean()) { + if (System.nanoTime() > deadline) { + throw new TimeoutException(description + " within " + timeout); + } + Thread.sleep(20); + } + } + + /** + * Replies success to every request and counts the pings that it saw across all sockets. + * Register replies can be withheld to keep a login pending on the client. + */ + private static final class MockVsrServer implements AutoCloseable { + private final ServerSocket serverSocket; + private final List accepted = new CopyOnWriteArrayList<>(); + private final AtomicInteger pings = new AtomicInteger(); + private final AtomicInteger registers = new AtomicInteger(); + private final AtomicInteger closedSockets = new AtomicInteger(); + private volatile boolean withholdRegisterReplies; + private volatile boolean closed; + + private MockVsrServer(ServerSocket serverSocket) { + this.serverSocket = serverSocket; + } + + static MockVsrServer start() throws IOException { + MockVsrServer server = new MockVsrServer(new ServerSocket(0, 4, InetAddress.getLoopbackAddress())); + Thread acceptor = new Thread(server::acceptLoop, "mock-vsr-acceptor-" + server.port()); + acceptor.setDaemon(true); + acceptor.start(); + return server; + } + + int port() { + return serverSocket.getLocalPort(); + } + + int pings() { + return pings.get(); + } + + int registers() { + return registers.get(); + } + + int closedSockets() { + return closedSockets.get(); + } + + void withholdRegisterReplies() { + withholdRegisterReplies = true; + } + + void awaitPings(int expected, Duration timeout) throws Exception { + long deadline = System.nanoTime() + timeout.toNanos(); + while (pings.get() < expected) { + if (System.nanoTime() > deadline) { + throw new TimeoutException("Expected " + expected + " pings, saw " + pings.get()); + } + Thread.sleep(5); + } + } + + private void acceptLoop() { + while (!closed) { + try { + Socket socket = serverSocket.accept(); + accepted.add(socket); + Thread exchange = new Thread(() -> exchange(socket), "mock-vsr-exchange-" + port()); + exchange.setDaemon(true); + exchange.start(); + } catch (IOException stopped) { + return; + } + } + } + + private void exchange(Socket socket) { + try (socket) { + InputStream input = socket.getInputStream(); + OutputStream output = socket.getOutputStream(); + Request request; + while (!closed && (request = readRequest(input)) != null) { + if (request.operation() == OPERATION_NON_REPLICATED && request.commandCode() == PING_CODE) { + pings.incrementAndGet(); + } + if (request.operation() == OPERATION_REGISTER) { + registers.incrementAndGet(); + if (withholdRegisterReplies) { + continue; + } + } + byte[] body = request.operation() == OPERATION_REGISTER ? registerBody() : new byte[0]; + writeResponse(output, request, body); + } + } catch (IOException clientWentAway) { + // A closed client and a closed server look the same here. + } finally { + closedSockets.incrementAndGet(); + } + } + + @Override + public void close() throws IOException { + closed = true; + for (Socket socket : accepted) { + socket.close(); + } + serverSocket.close(); + } + } + + private static Request readRequest(InputStream input) throws IOException { + byte[] header = input.readNBytes(HEADER_SIZE); + if (header.length == 0) { + return null; + } + if (header.length != HEADER_SIZE) { + throw new EOFException("Truncated VSR request header"); + } + ByteBuffer fields = ByteBuffer.wrap(header).order(ByteOrder.LITTLE_ENDIAN); + int size = fields.getInt(SIZE_OFFSET); + byte[] body = input.readNBytes(size - HEADER_SIZE); + if (body.length != size - HEADER_SIZE) { + throw new EOFException("Truncated VSR request body"); + } + return new Request( + Byte.toUnsignedInt(header[REQUEST_OPERATION_OFFSET]), + fields.getInt(REQUEST_CODE_OFFSET), + fields.getLong(REQUEST_ID_OFFSET)); + } + + private static void writeResponse(OutputStream output, Request request, byte[] body) throws IOException { + byte[] header = new byte[HEADER_SIZE]; + ByteBuffer fields = ByteBuffer.wrap(header).order(ByteOrder.LITTLE_ENDIAN); + fields.putInt(SIZE_OFFSET, HEADER_SIZE + body.length); + header[COMMAND_OFFSET] = COMMAND_REPLY; + fields.putLong(REPLY_REQUEST_ID_OFFSET, request.requestId()); + header[REPLY_OPERATION_OFFSET] = (byte) request.operation(); + fields.putInt(REPLY_STATUS_OFFSET, 0); + output.write(header); + output.write(body); + output.flush(); + } + + private static byte[] registerBody() { + return ByteBuffer.allocate(21) + .order(ByteOrder.LITTLE_ENDIAN) + .putInt(0) + .putInt(1) + .putLong(42) + .putInt(11 << 10) + .put((byte) 0) + .array(); + } + + private record Request(int operation, int commandCode, long requestId) {} +} diff --git a/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/AsyncTcpConnectionConcurrencyTest.java b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/AsyncTcpConnectionConcurrencyTest.java index 354f09d95b..07812d8539 100644 --- a/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/AsyncTcpConnectionConcurrencyTest.java +++ b/foreign/java/java-sdk/src/test/java/org/apache/iggy/client/async/tcp/AsyncTcpConnectionConcurrencyTest.java @@ -106,6 +106,8 @@ void shouldCorrelateConcurrentPartitionResponsesInReverseOrder() throws Exceptio false, Optional.empty(), new AsyncTcpConnection.TcpConnectionPoolConfig(1000, 50), + Optional.empty(), + 1, Optional.of(Duration.ofSeconds(1)), Optional.of(Duration.ofSeconds(2)), Duration.ofHours(1), @@ -303,6 +305,8 @@ private static AsyncTcpConnection newConnection( false, Optional.empty(), new AsyncTcpConnection.TcpConnectionPoolConfig(1000, acquireTimeoutMillis), + Optional.empty(), + 1, Optional.of(Duration.ofSeconds(1)), Optional.of(Duration.ofSeconds(2)), heartbeatInterval, From 422810f82a711ac837fbac6c9e60d44d73230951 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Wed, 9 Sep 2026 08:11:28 +0200 Subject: [PATCH 091/182] chore(deps): Bump the java group across 3 directories with 1 update (#4098) --- bdd/java/build.gradle.kts | 2 +- examples/java/build.gradle.kts | 2 +- foreign/java/gradle/libs.versions.toml | 2 +- 3 files changed, 3 insertions(+), 3 deletions(-) diff --git a/bdd/java/build.gradle.kts b/bdd/java/build.gradle.kts index 7f673b1f42..0a0633799d 100644 --- a/bdd/java/build.gradle.kts +++ b/bdd/java/build.gradle.kts @@ -20,7 +20,7 @@ plugins { java jacoco - id("com.diffplug.spotless") version "8.10.0" + id("com.diffplug.spotless") version "8.10.1" } repositories { diff --git a/examples/java/build.gradle.kts b/examples/java/build.gradle.kts index 91c8e6fec1..7dc1318af5 100644 --- a/examples/java/build.gradle.kts +++ b/examples/java/build.gradle.kts @@ -19,7 +19,7 @@ plugins { java - id("com.diffplug.spotless") version "8.10.0" + id("com.diffplug.spotless") version "8.10.1" } repositories { diff --git a/foreign/java/gradle/libs.versions.toml b/foreign/java/gradle/libs.versions.toml index 700eec36e6..621d4bdfed 100644 --- a/foreign/java/gradle/libs.versions.toml +++ b/foreign/java/gradle/libs.versions.toml @@ -55,7 +55,7 @@ typesafe-config = "1.4.9" picocli = "4.7.7" # Build plugins -spotless = "8.10.0" +spotless = "8.10.1" shadow = "9.6.1" checkstyle = "12.3.1" jacoco = "0.8.15" From 6d894501babbd88c026c5227d45d71171919d0b2 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Wed, 9 Sep 2026 12:17:17 +0200 Subject: [PATCH 092/182] chore(deps): Bump the github-actions group with 3 updates (#4099) --- .github/workflows/_common.yml | 4 ++-- .github/workflows/coverage-baseline.yml | 4 ++-- .github/workflows/edge-release.yml | 2 +- .github/workflows/post-merge.yml | 2 +- 4 files changed, 6 insertions(+), 6 deletions(-) diff --git a/.github/workflows/_common.yml b/.github/workflows/_common.yml index b6681b6bfb..a9735cf56a 100644 --- a/.github/workflows/_common.yml +++ b/.github/workflows/_common.yml @@ -93,7 +93,7 @@ jobs: run: echo "version=$(cat .github/config/hawkeye.version)" >> "$GITHUB_OUTPUT" - name: Install HawkEye - uses: taiki-e/install-action@v2.86.7 + uses: taiki-e/install-action@v2.87.3 with: tool: hawkeye@${{ steps.hawkeye-version.outputs.version }} @@ -206,7 +206,7 @@ jobs: with: fetch-depth: 0 - name: Check typos - uses: crate-ci/typos@v1.49.0 + uses: crate-ci/typos@v1.50.1 summary: name: Common checks summary diff --git a/.github/workflows/coverage-baseline.yml b/.github/workflows/coverage-baseline.yml index b4bd7c3156..42a90fccf3 100644 --- a/.github/workflows/coverage-baseline.yml +++ b/.github/workflows/coverage-baseline.yml @@ -104,7 +104,7 @@ jobs: free-disk-space-aggressive: "true" - name: Install cargo-llvm-cov - uses: taiki-e/install-action@v2.86.7 + uses: taiki-e/install-action@v2.87.3 with: tool: cargo-llvm-cov @@ -299,7 +299,7 @@ jobs: save-cache: "false" - name: Install cargo-llvm-cov - uses: taiki-e/install-action@v2.86.7 + uses: taiki-e/install-action@v2.87.3 with: tool: cargo-llvm-cov diff --git a/.github/workflows/edge-release.yml b/.github/workflows/edge-release.yml index 7dde3b7192..d68b36fcda 100644 --- a/.github/workflows/edge-release.yml +++ b/.github/workflows/edge-release.yml @@ -75,7 +75,7 @@ jobs: fi - name: Create edge pre-release - uses: softprops/action-gh-release@v3.0.2 + uses: softprops/action-gh-release@v3.0.3 with: tag_name: edge name: edge diff --git a/.github/workflows/post-merge.yml b/.github/workflows/post-merge.yml index 6b0d3fdb4f..0ddf30d4dc 100644 --- a/.github/workflows/post-merge.yml +++ b/.github/workflows/post-merge.yml @@ -67,7 +67,7 @@ jobs: # Pinned for the same reason as .github/actions/rust/pre-merge/action.yml: # 0.24 dropped the `-f json` edge-affected-images.sh relies on. - name: Install cargo-rail - uses: taiki-e/install-action@v2.86.7 + uses: taiki-e/install-action@v2.87.3 with: tool: cargo-rail@0.23.0 From 1290e4fffcfca85abdad6a213567d61e1f2f21f4 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=C5=81ukasz=20Zborek?= Date: Thu, 10 Sep 2026 07:31:52 +0200 Subject: [PATCH 093/182] chore(deps): remove Microsoft.SourceLink.GitHub package reference (#4113) Remove the explicit `Microsoft.SourceLink.GitHub` package reference. Since .NET 8 SourceLink is built into the SDK (https://github.com/dotnet/sourcelink), so the package is not needed. The pinned 10.0.108 also pulled in `Microsoft.Build.Tasks.Git` 10.0.108, flagged by CVE-2026-62900 (GHSA-23fw-v26w-5fgq), which failed CI restore for `Iggy_SDK.csproj` under `TreatWarningsAsErrors`. --- foreign/csharp/Directory.Build.props | 4 ---- foreign/csharp/Directory.Packages.props | 1 - 2 files changed, 5 deletions(-) diff --git a/foreign/csharp/Directory.Build.props b/foreign/csharp/Directory.Build.props index 684da2c884..085aff8257 100644 --- a/foreign/csharp/Directory.Build.props +++ b/foreign/csharp/Directory.Build.props @@ -26,8 +26,4 @@ under the License. true - - - - diff --git a/foreign/csharp/Directory.Packages.props b/foreign/csharp/Directory.Packages.props index 2388831f38..525f09c3c9 100644 --- a/foreign/csharp/Directory.Packages.props +++ b/foreign/csharp/Directory.Packages.props @@ -27,7 +27,6 @@ under the License. - From 7a77f45929912ce502d5d0ce6ef7ed177ea01799 Mon Sep 17 00:00:00 2001 From: Richard Cocks <50965970+richardcocks@users.noreply.github.com> Date: Thu, 10 Sep 2026 10:46:36 +0100 Subject: [PATCH 094/182] fix(connectors): use DLL_EXTENSION for plugin path suffix (#4114) Fixes #4111 --- core/connectors/runtime/src/main.rs | 12 ++---------- 1 file changed, 2 insertions(+), 10 deletions(-) diff --git a/core/connectors/runtime/src/main.rs b/core/connectors/runtime/src/main.rs index fcf8cafa91..12d627c84a 100644 --- a/core/connectors/runtime/src/main.rs +++ b/core/connectors/runtime/src/main.rs @@ -354,12 +354,7 @@ pub(crate) fn resolve_plugin_path(path: &str) -> Result { let with_extension = if ALLOWED_PLUGIN_EXTENSIONS.contains(&extension) { path.to_string() } else { - let os_extension = match std::env::consts::OS { - "macos" => "dylib", - "windows" => "dll", - _ => "so", - }; - format!("{path}.{os_extension}") + format!("{path}.{}", std::env::consts::DLL_EXTENSION) }; let candidate = std::path::Path::new(&with_extension); @@ -548,10 +543,7 @@ mod tests { fn path_without_extension_gets_os_suffix() { let result = resolve_plugin_path("/tmp/nonexistent_test_plugin"); let err = result.unwrap_err().to_string(); - let expected_ext = match std::env::consts::OS { - "macos" => "dylib", - _ => "so", - }; + let expected_ext = std::env::consts::DLL_EXTENSION; assert!( err.contains(&format!("nonexistent_test_plugin.{expected_ext}")), "Error should mention OS-specific extension, got: {err}" From 62f945a3451d690a02243c4a99f0cdb4012c4f50 Mon Sep 17 00:00:00 2001 From: Justin Mclean Date: Thu, 10 Sep 2026 20:25:05 +1000 Subject: [PATCH 095/182] chore: add justinmclean to .asf.yaml collaborators (#4115) --- .asf.yaml | 3 +++ 1 file changed, 3 insertions(+) diff --git a/.asf.yaml b/.asf.yaml index 79102664da..e3ae0fa846 100644 --- a/.asf.yaml +++ b/.asf.yaml @@ -59,6 +59,9 @@ github: edit_comment_discussion: "Re: {title}" delete_comment_discussion: "Re: {title}" + collaborators: + - justinmclean + notifications: commits: commits@iggy.apache.org issues: commits@iggy.apache.org From 46a9075963bbc6c9f6d834d9bb140bb78b7fffdf Mon Sep 17 00:00:00 2001 From: Mehmet YUCE Date: Thu, 10 Sep 2026 13:53:49 +0300 Subject: [PATCH 096/182] fix(connectors): bring quickwit_sink up to convention (#3523) Closes #3814 --- Cargo.lock | 5 +- .../connectors/quickwit_sink.toml | 22 +- .../src/configs/connectors/local_provider.rs | 136 ++- .../connectors/sinks/quickwit_sink/Cargo.toml | 10 +- core/connectors/sinks/quickwit_sink/README.md | 64 +- .../sinks/quickwit_sink/config.toml | 9 + .../connectors/sinks/quickwit_sink/src/lib.rs | 957 ++++++++++++++++-- .../tests/connectors/fixtures/mod.rs | 5 +- .../connectors/fixtures/quickwit/container.rs | 86 +- .../tests/connectors/fixtures/quickwit/mod.rs | 5 +- .../connectors/quickwit/quickwit_sink.rs | 128 ++- 11 files changed, 1291 insertions(+), 136 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 98b0f147d5..3afc2f782f 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -7349,12 +7349,15 @@ name = "iggy_connector_quickwit_sink" version = "0.5.0-edge.4" dependencies = [ "async-trait", - "dashmap", + "base64 0.23.1", + "humantime", "iggy_connector_sdk", "reqwest 0.13.4", + "reqwest-middleware", "serde", "serde_yaml_ng", "simd-json", + "tokio", "tracing", ] diff --git a/core/connectors/runtime/example_config/connectors/quickwit_sink.toml b/core/connectors/runtime/example_config/connectors/quickwit_sink.toml index e4622ffcf8..27a3c75607 100644 --- a/core/connectors/runtime/example_config/connectors/quickwit_sink.toml +++ b/core/connectors/runtime/example_config/connectors/quickwit_sink.toml @@ -53,13 +53,22 @@ fields = ["email", "created_at"] [plugin_config] url = "http://localhost:7280" +verbose_logging = false +# Total attempts including the first; 1 disables retries. +# max_retries = 3 +# retry_delay = "1s" +# retry_max_delay = "5s" +# Total readiness probes including the first; 1 disables retries. +# max_open_retries = 10 +# open_retry_max_delay = "30s" +# timeout = "30s" index = """ version: 0.9 index_id: events doc_mapping: - mode: strict + mode: dynamic field_mappings: - name: timestamp type: datetime @@ -93,12 +102,15 @@ doc_mapping: type: text tokenizer: default - timestamp_field: timestamp + # Enable timestamp sharding only when every document contains timestamp. + # Raw/text wrappers do not acquire fields from the add_fields transform. + # timestamp_field: timestamp indexing_settings: commit_timeout_secs: 10 -retention: - period: 7 days - schedule: daily +# Retention requires timestamp_field above. +# retention: +# period: 7 days +# schedule: daily """ diff --git a/core/connectors/runtime/src/configs/connectors/local_provider.rs b/core/connectors/runtime/src/configs/connectors/local_provider.rs index 5df18e2386..fc012bad25 100644 --- a/core/connectors/runtime/src/configs/connectors/local_provider.rs +++ b/core/connectors/runtime/src/configs/connectors/local_provider.rs @@ -32,6 +32,8 @@ use std::collections::HashMap; use std::path::Path; use tracing::{debug, info, warn}; +const PLUGIN_CONFIG_FORMAT_ENV_SUFFIX: &str = "FORMAT"; + #[derive(Eq, PartialEq, Hash, Clone, Debug)] struct ConnectorId { key: String, @@ -355,6 +357,10 @@ impl LocalConnectorsConfigProvider { } let field_path = &env_key_upper[prefix.len()..]; + // ConfigEnv handles plugin_config_format on the connector itself. + if field_path == PLUGIN_CONFIG_FORMAT_ENV_SUFFIX { + continue; + } let field_name = field_path.to_lowercase(); let parsed_value = ::configs::parse_env_value_to_json(&env_value); @@ -855,9 +861,13 @@ impl Provider for ConnectorEnvProvider { #[cfg(test)] mod tests { - use super::*; + use std::process::Command; + use tempfile::TempDir; + use super::*; + use crate::configs::connectors::ConfigFormat; + #[tokio::test] async fn given_valid_key_when_creating_source_config_should_write_prefixed_file() { let dir = TempDir::new().unwrap(); @@ -880,4 +890,128 @@ mod tests { .collect(); assert_eq!(entries, vec!["source_random_0.toml"]); } + + #[test] + fn given_runtime_format_env_when_loading_configs_should_preserve_plugin_overrides() { + const CHILD_PROCESS_ENV: &str = "IGGY_CONNECTORS_FORMAT_OVERRIDE_TEST"; + const CONNECTOR_KEY: &str = "format_override_test"; + const FORMAT_ONLY_KEY: &str = "format_only_test"; + + if std::env::var_os(CHILD_PROCESS_ENV).is_none() { + // Environment overrides stay in a child process to avoid racing other tests. + let mut command = Command::new( + std::env::current_exe().expect("Test binary path should be available"), + ); + command + .arg("--exact") + .arg( + std::thread::current() + .name() + .expect("Test thread should have a name"), + ) + .env_clear() + .env(CHILD_PROCESS_ENV, "1"); + for connector_type in ["SINK", "SOURCE"] { + let prefix = format!( + "IGGY_CONNECTORS_{connector_type}_{}_PLUGIN_CONFIG_", + CONNECTOR_KEY.to_uppercase() + ); + command + .env(format!("{prefix}FORMAT"), "yaml") + .env(format!("{prefix}URL"), "http://overridden.test") + .env(format!("{prefix}FORMAT_OPTIONS"), r#"["compact","json"]"#) + .env( + format!( + "IGGY_CONNECTORS_{connector_type}_{}_PLUGIN_CONFIG_FORMAT", + FORMAT_ONLY_KEY.to_uppercase() + ), + "yaml", + ); + } + let output = command.output().expect("Config override test should run"); + assert!( + output.status.success(), + "Config override test failed:\n{}\n{}", + String::from_utf8_lossy(&output.stdout), + String::from_utf8_lossy(&output.stderr) + ); + return; + } + + let dir = TempDir::new().expect("Config directory should be created"); + for connector_type in ["sink", "source"] { + for key in [CONNECTOR_KEY, FORMAT_ONLY_KEY] { + let mut config = format!( + r#" +type = "{connector_type}" +key = "{key}" +enabled = true +version = 0 +name = "format override test" +path = "unused" +streams = [] +plugin_config_format = "json" +"# + ); + if key == CONNECTOR_KEY { + config.push_str( + r#" +[plugin_config] +url = "http://original.test" +format_options = ["original"] +[plugin_config.headers] +content_type = "application/json" +"#, + ); + } + std::fs::write( + dir.path().join(format!("{connector_type}_{key}.toml")), + config, + ) + .expect("Connector config should be written"); + } + } + + let runtime = tokio::runtime::Runtime::new().expect("Test runtime should be created"); + runtime.block_on(async { + let provider = LocalConnectorsConfigProvider::new( + dir.path() + .to_str() + .expect("Config directory should be UTF-8"), + ) + .init() + .await + .expect("Connector configs should load"); + + for key in [CONNECTOR_KEY, FORMAT_ONLY_KEY] { + let sink = provider + .get_sink_config(key, None) + .await + .expect("Sink config should be available") + .expect("Sink config should exist"); + let source = provider + .get_source_config(key, None) + .await + .expect("Source config should be available") + .expect("Source config should exist"); + let expected_plugin_config = (key == CONNECTOR_KEY).then(|| { + serde_json::json!({ + "url": "http://overridden.test", + "headers": {"content_type": "application/json"}, + "format_options": ["compact", "json"] + }) + }); + for (connector_type, format, plugin_config) in [ + ("sink", sink.plugin_config_format, sink.plugin_config), + ("source", source.plugin_config_format, source.plugin_config), + ] { + assert_eq!(format, Some(ConfigFormat::Yaml), "{connector_type} {key}"); + assert_eq!( + plugin_config, expected_plugin_config, + "{connector_type} {key}" + ); + } + } + }); + } } diff --git a/core/connectors/sinks/quickwit_sink/Cargo.toml b/core/connectors/sinks/quickwit_sink/Cargo.toml index b9b9fe8f4b..75141dc8a7 100644 --- a/core/connectors/sinks/quickwit_sink/Cargo.toml +++ b/core/connectors/sinks/quickwit_sink/Cargo.toml @@ -29,21 +29,23 @@ repository = "https://github.com/apache/iggy" readme = "../../README.md" publish = false -[package.metadata.cargo-machete] -ignored = ["dashmap"] - [lib] crate-type = ["cdylib", "lib"] [dependencies] async-trait = { workspace = true } -dashmap = { workspace = true } +base64 = { workspace = true } +humantime = { workspace = true } iggy_connector_sdk = { workspace = true } reqwest = { workspace = true } +reqwest-middleware = { workspace = true } serde = { workspace = true } serde_yaml_ng = { workspace = true } simd-json = { workspace = true } tracing = { workspace = true } +[dev-dependencies] +tokio = { workspace = true } + [lints] workspace = true diff --git a/core/connectors/sinks/quickwit_sink/README.md b/core/connectors/sinks/quickwit_sink/README.md index c8fc338ce2..f3067418ac 100644 --- a/core/connectors/sinks/quickwit_sink/README.md +++ b/core/connectors/sinks/quickwit_sink/README.md @@ -1,22 +1,44 @@ # Quickwit Sink -The Quickwit connector allows you to send data to the Quickwit API using HTTP. This sink will ensure that the index exists (create it if it doesn't) and will append the data to the index using the same batch size as specified in the Iggy configuration. +The Quickwit connector sends data to the Quickwit API using HTTP. It checks readiness when opening, creates the index if needed, and appends messages as NDJSON. Requests are split at 8 MiB, including the newline after each document. A larger individual document is logged and rejected while other documents continue. ## Configuration -- `url`: The URL of the Quickwit server. -- `index`: The index configuration using YAML, as described in the [Quickwit index configuration docs](https://quickwit.io/docs/configuration/index-config) +| Key | Default | Description | +| --- | --- | --- | +| `url` | Required | Quickwit base URL with an `http` or `https` scheme and host. Path prefixes and a trailing slash are supported; query strings and fragments are rejected. | +| `index` | Required | Index configuration as YAML, with a nonempty `index_id`. See the [Quickwit index configuration docs](https://quickwit.io/docs/configuration/index-config). | +| `verbose_logging` | `false` | Log received and ingested message counts at `info` instead of `debug`. | +| `max_retries` | `3` | Total HTTP attempts including the first; `1` disables retries. | +| `retry_delay` | `"1s"` | Base exponential delay for HTTP retries and readiness probes. | +| `retry_max_delay` | `"5s"` | Maximum delay between HTTP retry attempts. | +| `max_open_retries` | `10` | Total readiness probes including the first; `1` disables retries. | +| `open_retry_max_delay` | `"30s"` | Maximum delay between readiness probes. | +| `timeout` | `"30s"` | Timeout for each HTTP request. | + +Duration values require units, such as `250ms` or `30s`. Invalid or zero durations prevent initialization. Unknown plugin configuration keys are rejected. + +Set `plugin_config_format` in the connector TOML or with the `IGGY_CONNECTORS_SINK_QUICKWIT_PLUGIN_CONFIG_FORMAT` environment variable. ```toml [plugin_config] url = "http://localhost:7280" +verbose_logging = false +# Total attempts including the first; 1 disables retries. +max_retries = 3 +retry_delay = "1s" +retry_max_delay = "5s" +# Total readiness probes including the first; 1 disables retries. +max_open_retries = 10 +open_retry_max_delay = "30s" +timeout = "30s" index = """ version: 0.9 index_id: events doc_mapping: - mode: strict + mode: dynamic field_mappings: - name: timestamp type: datetime @@ -50,13 +72,39 @@ doc_mapping: type: text tokenizer: default - timestamp_field: timestamp + # Enable only when every document contains timestamp. + # timestamp_field: timestamp indexing_settings: commit_timeout_secs: 10 -retention: - period: 7 days - schedule: daily +# Retention requires timestamp_field above. +# retention: +# period: 7 days +# schedule: daily """ ``` + +## Document shapes + +The stream's `schema` selects the runtime decoder. The sink sends JSON objects using these shapes: + +| Payload | Example document | +| --- | --- | +| JSON object, including an object parsed from raw bytes | `{"message":"ready"}` | +| JSON array or scalar | `{"data":[1,2],"data_type":"json"}` or `{"data":42,"data_type":"json"}` | +| Raw UTF-8 that is not a JSON object | `{"data":"ready","data_type":"raw","data_encoding":"utf8"}` | +| Raw non-UTF-8 bytes | `{"data":"/wCA","data_type":"raw","data_encoding":"base64"}` | +| Text | `{"text":"ready","data_type":"text"}` | + +Raw JSON arrays and scalars remain UTF-8 strings in the raw wrapper. Malformed JSON also preserves the original bytes. `data_encoding` distinguishes literal text from base64. The sink handles `Payload::Avro` and `Payload::FlatBuffer` with the raw path, and `Payload::Proto` with the text wrapper. Runtime decoder settings determine which payload variant reaches the sink. + +The examples use `mode: dynamic` to retain wrapper fields. With `mode: strict`, map every field emitted by the selected payload shape or Quickwit rejects the document during indexing, even after a successful HTTP response. Timestamp sharding and retention are optional here: raw/text wrappers have no `timestamp`, and the `add_fields` transform only enriches JSON payloads. Configure them only when every document supplies the required timestamp. + +## Delivery semantics + +Transient HTTP failures, including 429, can retry a request that Quickwit already accepted. Quickwit ingest has no deduplication key, so these retries can produce duplicate documents. Set `max_retries = 1` to disable HTTP retries and use at-most-once request submission. + +The sink cannot guarantee at-least-once delivery. The runtime commits offsets when polling and ignores the plugin's consume return code. A permanent error or exhausted retry budget is logged, but affected messages are not redelivered. The runtime's processed-message count does not prove successful indexing. These runtime limitations are tracked in [#2927](https://github.com/apache/iggy/issues/2927) and [#2928](https://github.com/apache/iggy/issues/2928). + +Chunks are independent: successful writes remain committed if another chunk fails. The sink continues later chunks and returns the last error. A successful ingest response acknowledges submission for indexing, not that every document passed the index mapping. This sink has no circuit breaker. diff --git a/core/connectors/sinks/quickwit_sink/config.toml b/core/connectors/sinks/quickwit_sink/config.toml index 205c37268a..c0f1a68b06 100644 --- a/core/connectors/sinks/quickwit_sink/config.toml +++ b/core/connectors/sinks/quickwit_sink/config.toml @@ -34,3 +34,12 @@ consumer_group = "quickwit_sink" [plugin_config] url = "" index = "" +verbose_logging = true +# Total attempts including the first; 1 disables retries. +max_retries = 2 +retry_delay = "250ms" +retry_max_delay = "2s" +# Total readiness probes including the first; 1 disables retries. +max_open_retries = 8 +open_retry_max_delay = "10s" +timeout = "10s" diff --git a/core/connectors/sinks/quickwit_sink/src/lib.rs b/core/connectors/sinks/quickwit_sink/src/lib.rs index 0ee0d8f2b4..cd5392d585 100644 --- a/core/connectors/sinks/quickwit_sink/src/lib.rs +++ b/core/connectors/sinks/quickwit_sink/src/lib.rs @@ -15,159 +15,406 @@ // specific language governing permissions and limitations // under the License. +use std::time::Duration; + use async_trait::async_trait; +use base64::{Engine as _, engine::general_purpose}; +use iggy_connector_sdk::retry::{ + ConnectivityConfig, build_retry_client, check_connectivity_with_retry, is_transient_status, +}; use iggy_connector_sdk::{ ConsumedMessage, Error, MessagesMetadata, Payload, Sink, TopicMetadata, sink_connector, }; -use serde::{Deserialize, Serialize}; -use tracing::{error, info, warn}; +use reqwest::StatusCode; +use reqwest::Url; +use reqwest_middleware::ClientWithMiddleware; +use serde::Deserialize; +use simd_json::OwnedValue; +use tracing::{debug, error, info}; sink_connector!(QuickwitSink); +const DEFAULT_MAX_RETRIES: u32 = 3; +const DEFAULT_RETRY_DELAY: &str = "1s"; +const DEFAULT_RETRY_MAX_DELAY: &str = "5s"; +const DEFAULT_MAX_OPEN_RETRIES: u32 = 10; +const DEFAULT_OPEN_RETRY_MAX_DELAY: &str = "30s"; +const DEFAULT_TIMEOUT: &str = "30s"; +const MAX_INGEST_BODY_BYTES: usize = 8 * 1024 * 1024; +const ESTIMATED_RECORD_BYTES: usize = 512; + #[derive(Debug)] pub struct QuickwitSink { id: u32, config: QuickwitSinkConfig, - client: reqwest::Client, + client: Option, + verbose: bool, index_id: String, + indexes_url: String, + index_url: String, + ingest_url: String, } -#[derive(Debug, Serialize, Deserialize)] +/// Configuration for the Quickwit sink connector, deserialized from `[plugin_config]` in config.toml. +#[derive(Debug, Deserialize)] +#[serde(deny_unknown_fields)] pub struct QuickwitSinkConfig { - url: String, - index: String, + /// Target URL for the Quickwit service. + pub url: String, + /// Full Quickwit index config YAML, passed to `POST /api/v1/indexes` on first open. + /// `index_id` is extracted from this YAML to build ingest URLs. + pub index: String, + /// Enable verbose logging for ingested messages (default: false). + pub verbose_logging: Option, + /// Total HTTP attempts including the first (default: 3). 1 disables retries. + pub max_retries: Option, + /// Initial retry delay as a human-readable duration string, e.g. "1s" (default: 1s). + pub retry_delay: Option, + /// Maximum retry delay cap as a human-readable duration string, e.g. "5s" (default: 5s). + pub retry_max_delay: Option, + /// Total readiness attempts including the first (default: 10). 1 disables retries. + pub max_open_retries: Option, + /// Maximum retry delay cap when opening the sink, e.g. "30s" (default: 30s). + pub open_retry_max_delay: Option, + /// HTTP request timeout as a human-readable duration string, e.g. "30s" (default: 30s). + pub timeout: Option, } -#[derive(Debug, Serialize, Deserialize)] +#[derive(Debug, Deserialize)] struct IndexConfig { index_id: String, } impl QuickwitSink { pub fn new(id: u32, config: QuickwitSinkConfig) -> Self { - let index_config = - serde_yaml_ng::from_str::(&config.index).expect("Invalid index config."); - QuickwitSink { + let verbose = config.verbose_logging.unwrap_or(false); + Self { id, config, - index_id: index_config.index_id, - client: reqwest::Client::new(), + client: None, + verbose, + index_id: String::new(), + indexes_url: String::new(), + index_url: String::new(), + ingest_url: String::new(), } } + fn client(&self) -> Result<&ClientWithMiddleware, Error> { + self.client + .as_ref() + .ok_or_else(|| Error::InitError("Quickwit sink client not initialized".into())) + } + async fn has_index(&self) -> Result { - let url = format!("{}/api/v1/indexes/{}", self.config.url, self.index_id); - let response = self.client.get(&url).send().await.map_err(|error| { - error!( - "Failed to send HTTP request to check if index with ID: {} exists. {error}", - self.index_id - ); - Error::HttpRequestFailed(error.to_string()) - })?; + let client = self.client()?; + let response = client + .get(&self.index_url) + .send() + .await + .map_err(|e| Error::HttpRequestFailed(e.to_string()))?; let status = response.status(); if status.is_success() { Ok(true) - } else if status == reqwest::StatusCode::NOT_FOUND { + } else if status == StatusCode::NOT_FOUND { Ok(false) } else { - Err(Error::HttpRequestFailed(format!( - "Unexpected status code: {status}", + let reason = response + .text() + .await + .unwrap_or_else(|error| format!("failed to read response: {error}")); + Err(Error::InitError(format!( + "Checking Quickwit index '{}': {status}, {reason}", + self.index_id ))) } } async fn create_index(&self) -> Result<(), Error> { - info!("Creating index: {}", self.index_id); - let url = format!("{}/api/v1/indexes", self.config.url); - let response = self - .client - .post(&url) - .header("content-type", "application/yaml") - .body(self.config.index.to_owned()) + info!( + "Creating Quickwit index: {} for connector ID: {}", + self.index_id, self.id + ); + let client = self.client()?; + let response = client + .post(&self.indexes_url) + .header("Content-Type", "application/yaml") + .body(self.config.index.clone()) .send() .await - .map_err(|error| { - error!( - "Failed to send HTTP request to create index: {}. {error}", - self.index_id - ); - Error::HttpRequestFailed(error.to_string()) - })?; + .map_err(|error| Error::HttpRequestFailed(error.to_string()))?; - if !response.status().is_success() { - let status = response.status(); - let reason = response.text().await.unwrap_or_default(); - error!( - "Received an invalid HTTP response when creating index: {}. Status code: {status}, reason: {reason}", - self.index_id + let status = response.status(); + if status.is_success() { + info!( + "Created Quickwit index: {} for connector ID: {}", + self.index_id, self.id ); - return Err(Error::InitError(format!( - "Failed to create index: {}. {reason}", - self.index_id - ))); + Ok(()) + } else { + let reason = response + .text() + .await + .unwrap_or_else(|error| format!("failed to read response: {error}")); + // A competing creator or a retried POST may already have created the index. + if self.has_index().await? { + info!( + "Quickwit index already exists ({status}): {} for connector ID: {}", + self.index_id, self.id + ); + Ok(()) + } else { + Err(Error::InitError(format!( + "Failed to create index '{0}': {status} {reason}", + self.index_id + ))) + } } + } - info!("Created index: {}", self.index_id); - Ok(()) + async fn ingest(&self, messages: Vec) -> Result<(), Error> { + let capacity = messages + .len() + .saturating_mul(ESTIMATED_RECORD_BYTES) + .min(MAX_INGEST_BODY_BYTES); + let mut body = Vec::with_capacity(capacity); + let mut messages_count = 0; + let mut last_error = None; + for (position, record) in messages.into_iter().enumerate() { + let previous_len = body.len(); + if let Err(error) = simd_json::to_writer(&mut body, &record) { + body.truncate(previous_len); + error!( + "Quickwit sink connector ID: {} failed to serialize record {position}: {error}", + self.id + ); + last_error = Some(Error::Serialization(error.to_string())); + continue; + } + body.push(b'\n'); + let record_len = body.len() - previous_len; + if record_len > MAX_INGEST_BODY_BYTES { + body.truncate(previous_len); + let reason = format!( + "record {position} is {record_len} bytes, exceeding the {MAX_INGEST_BODY_BYTES}-byte ingest limit" + ); + error!("Quickwit sink connector ID: {}: {reason}", self.id); + last_error = Some(Error::InvalidRecordValue(reason)); + continue; + } + if body.len() > MAX_INGEST_BODY_BYTES { + let next_body = body.split_off(previous_len); + let completed_body = std::mem::replace(&mut body, next_body); + if let Err(error) = self.ingest_batch(completed_body, messages_count).await { + last_error = Some(error); + } + messages_count = 0; + } + messages_count += 1; + } + if messages_count > 0 + && let Err(error) = self.ingest_batch(body, messages_count).await + { + last_error = Some(error); + } + last_error.map_or(Ok(()), Err) } - pub async fn ingest(&self, messages: Vec) -> Result<(), Error> { - let url = format!( - "{}/api/v1/{}/ingest?commit=auto", - self.config.url, self.index_id - ); - info!("Ingesting messages for index: {}...", self.index_id); - let messages_count = messages.len(); - let messages = messages - .into_iter() - .filter_map(|record| simd_json::to_string(&record).ok()) - .collect::>() - .join("\n"); - - let response = self - .client - .post(&url) - .body(messages) + async fn ingest_batch(&self, body: Vec, messages_count: usize) -> Result<(), Error> { + let client = self.client()?; + // Retries can duplicate accepted documents. Final failures are logged but + // not redelivered because the runtime commits offsets when polling. + // Error classification remains diagnostic across the sink FFI boundary. + let response = client + .post(&self.ingest_url) + .header("Content-Type", "application/x-ndjson") + .body(body) .send() .await - .map_err(|error| { + .map_err(|e| { error!( - "Failed to send HTTP request to ingest messages for index: {}. {error}", - self.index_id + "Failed to ingest {messages_count} messages into Quickwit index: {} for connector ID: {}. {e}", + self.index_id, self.id ); - Error::HttpRequestFailed(error.to_string()) + Error::HttpRequestFailed(e.to_string()) })?; - if !response.status().is_success() { - let status = response.status(); - let text = response.text().await.unwrap_or_default(); - error!( - "Received an invalid HTTP response when ingesting messages for index: {}. Status code: {status}, reason: {text}", - self.index_id - ); - return Err(Error::HttpRequestFailed(format!( - "Status code: {status}, reason: {text}" - ))); + let status = response.status(); + if status.is_success() { + if self.verbose { + info!( + "Ingested {messages_count} messages into Quickwit index: {} for connector ID: {}", + self.index_id, self.id + ); + } else { + debug!( + "Ingested {messages_count} messages into Quickwit index: {} for connector ID: {}", + self.index_id, self.id + ); + } + return Ok(()); } - info!( - "Ingested {messages_count} messages for index: {}", - self.index_id + let reason = response + .text() + .await + .unwrap_or_else(|error| format!("failed to read response: {error}")); + let transient = is_transient_status(status); + let category = if transient { "Transient" } else { "Permanent" }; + error!( + "{category} error ingesting into Quickwit index: {} for connector ID: {}. status: {status}, reason: {reason}", + self.index_id, self.id ); - Ok(()) + let reason = format!("status: {status}, reason: {reason}"); + if transient { + Err(Error::HttpRequestFailed(reason)) + } else { + Err(Error::PermanentHttpError(reason)) + } + } + + fn extract_json_payloads(&self, messages: Vec) -> Vec { + let mut json_payloads = Vec::with_capacity(messages.len()); + for message in messages { + let val = match message.payload { + Payload::Json(value @ OwnedValue::Object(_)) => value, + Payload::Json(value) => simd_json::json!({ + "data": value, + "data_type": "json" + }), + Payload::Raw(bytes) | Payload::Avro(bytes) | Payload::FlatBuffer(bytes) => { + if bytes.iter().find(|byte| !byte.is_ascii_whitespace()) == Some(&b'{') { + // SIMD parsing mutates its input, even on failure. Preserve the fallback. + let mut json_bytes = bytes.clone(); + if let Ok(value @ OwnedValue::Object(_)) = + simd_json::from_slice::(&mut json_bytes) + { + json_payloads.push(value); + continue; + } + } + let (data, encoding) = match String::from_utf8(bytes) { + Ok(text) => (text, "utf8"), + Err(error) => ( + general_purpose::STANDARD.encode(error.into_bytes()), + "base64", + ), + }; + simd_json::json!({ + "data": data, + "data_type": "raw", + "data_encoding": encoding + }) + } + Payload::Text(text) | Payload::Proto(text) => simd_json::json!({ + "text": text, + "data_type": "text" + }), + }; + json_payloads.push(val); + } + json_payloads } } #[async_trait] impl Sink for QuickwitSink { async fn open(&mut self) -> Result<(), Error> { + let result = async { + let index_config = serde_yaml_ng::from_str::(&self.config.index) + .map_err(|error| Error::InvalidConfigValue(format!("index: {error}")))?; + if index_config.index_id.trim().is_empty() { + return Err(Error::InvalidConfigValue( + "index_id must not be empty".into(), + )); + } + let base_url = Url::parse(self.config.url.trim_end_matches('/')) + .map_err(|error| Error::InvalidConfigValue(format!("url: {error}")))?; + if !matches!(base_url.scheme(), "http" | "https") + || !base_url.has_host() + || base_url.query().is_some() + || base_url.fragment().is_some() + { + return Err(Error::InvalidConfigValue( + "url must be an HTTP(S) base URL with a host and no query or fragment".into(), + )); + } + let retry_delay = parse_duration( + self.config.retry_delay.as_deref(), + DEFAULT_RETRY_DELAY, + "retry_delay", + )?; + let retry_max_delay = parse_duration( + self.config.retry_max_delay.as_deref(), + DEFAULT_RETRY_MAX_DELAY, + "retry_max_delay", + )?; + let open_retry_max_delay = parse_duration( + self.config.open_retry_max_delay.as_deref(), + DEFAULT_OPEN_RETRY_MAX_DELAY, + "open_retry_max_delay", + )?; + let timeout = + parse_duration(self.config.timeout.as_deref(), DEFAULT_TIMEOUT, "timeout")?; + + self.index_id = index_config.index_id; + self.indexes_url = endpoint_url(&base_url, &["api", "v1", "indexes"])?.into(); + self.index_url = + endpoint_url(&base_url, &["api", "v1", "indexes", &self.index_id])?.into(); + let mut ingest_url = endpoint_url(&base_url, &["api", "v1", &self.index_id, "ingest"])?; + ingest_url.set_query(Some("commit=auto")); + self.ingest_url = ingest_url.into(); + + let raw_client = reqwest::Client::builder() + .timeout(timeout) + .build() + .map_err(|error| Error::InitError(format!("reqwest client: {error}")))?; + check_connectivity_with_retry( + &raw_client, + endpoint_url(&base_url, &["health", "readyz"])?, + "Quickwit sink", + self.id, + &ConnectivityConfig { + max_open_retries: self + .config + .max_open_retries + .unwrap_or(DEFAULT_MAX_OPEN_RETRIES) + .max(1), + open_retry_max_delay, + retry_delay, + }, + ) + .await?; + + self.client = Some(build_retry_client( + raw_client, + self.config + .max_retries + .unwrap_or(DEFAULT_MAX_RETRIES) + .max(1), + retry_delay, + retry_max_delay, + "Quickwit", + )); + if !self.has_index().await? { + self.create_index().await?; + } + Ok(()) + } + .await; + if let Err(error) = result { + self.client = None; + error!( + "Failed to open Quickwit sink connector ID: {}: {error}", + self.id + ); + return Err(error); + } + info!( - "Opened Quickwit sink connector with ID: {} for URL: {}", - self.id, self.config.url + "Opened Quickwit sink connector ID: {}, index: {}", + self.id, self.index_id ); - if !self.has_index().await? { - self.create_index().await?; - } Ok(()) } @@ -177,33 +424,529 @@ impl Sink for QuickwitSink { messages_metadata: MessagesMetadata, messages: Vec, ) -> Result<(), Error> { - info!( - "Quickwit sink with ID: {} received: {} messages, format: {}", - self.id, - messages.len(), - messages_metadata.schema - ); - - let mut json_payloads = Vec::with_capacity(messages.len()); - for message in messages { - match message.payload { - Payload::Json(value) => json_payloads.push(value), - _ => { - warn!("Unsupported payload format: {}", messages_metadata.schema); - } - } + let total = messages.len(); + if self.verbose { + info!( + "Quickwit sink connector ID: {} received {total} messages, schema: {}", + self.id, messages_metadata.schema + ); + } else { + debug!( + "Quickwit sink connector ID: {} received {total} messages, schema: {}", + self.id, messages_metadata.schema + ); } + let json_payloads = self.extract_json_payloads(messages); if json_payloads.is_empty() { return Ok(()); } - self.ingest(json_payloads).await?; - Ok(()) + self.ingest(json_payloads).await } async fn close(&mut self) -> Result<(), Error> { - info!("Quickwit sink connector with ID: {} is closed.", self.id); + let _ = self.client.take(); + info!("Closed Quickwit sink connector ID: {}", self.id); Ok(()) } } + +fn parse_duration(value: Option<&str>, default: &str, field: &str) -> Result { + let duration = humantime::parse_duration(value.unwrap_or(default)) + .map_err(|error| Error::InvalidConfigValue(format!("{field}: {error}")))?; + if duration.is_zero() { + return Err(Error::InvalidConfigValue(format!( + "{field} must be greater than zero" + ))); + } + Ok(duration) +} + +fn endpoint_url(base_url: &Url, segments: &[&str]) -> Result { + let mut url = base_url.clone(); + url.path_segments_mut() + .map_err(|()| Error::InvalidConfigValue("url cannot be used as a base URL".into()))? + .pop_if_empty() + .extend(segments); + Ok(url) +} + +#[cfg(test)] +mod tests { + use std::time::Duration; + + use iggy_connector_sdk::Schema; + use tokio::io::{AsyncReadExt, AsyncWriteExt}; + use tokio::net::{TcpListener, TcpStream}; + use tokio::task::JoinHandle; + use tokio::time::timeout; + + use super::*; + + const TEST_TIMEOUT: Duration = Duration::from_secs(5); + const HTTP_READ_BUFFER_BYTES: usize = 8192; + + fn test_config() -> QuickwitSinkConfig { + QuickwitSinkConfig { + url: "http://localhost:7280".to_string(), + index: "index_id: test\nversion: 0.8\n".to_string(), + verbose_logging: None, + max_retries: Some(1), + retry_delay: Some("1ms".to_string()), + retry_max_delay: None, + max_open_retries: Some(1), + open_retry_max_delay: None, + timeout: Some("1s".to_string()), + } + } + + fn test_message(payload: Payload) -> ConsumedMessage { + ConsumedMessage { + id: 1, + offset: 0, + checksum: 0, + timestamp: 0, + origin_timestamp: 0, + headers: None, + payload, + } + } + + #[test] + fn given_invalid_index_config_when_opened_should_return_config_error() { + let runtime = test_runtime(); + runtime.block_on(async { + for index in [ + "", + "index_id: [", + "version: 0.8", + "index_id: ''", + "index_id: ' '", + ] { + let mut config = test_config(); + config.index = index.to_string(); + let mut sink = QuickwitSink::new(1, config); + assert!( + matches!(sink.open().await, Err(Error::InvalidConfigValue(_))), + "invalid index config should fail open: {index:?}" + ); + assert!(sink.client.is_none()); + } + }); + } + + #[test] + fn given_invalid_url_when_opened_should_return_config_error() { + let runtime = test_runtime(); + runtime.block_on(async { + for url in [ + "localhost:7280", + "ftp://localhost:7280", + "http://", + "http://localhost:7280/?token=value", + "http://localhost:7280/#fragment", + ] { + let mut config = test_config(); + config.url = url.to_string(); + let mut sink = QuickwitSink::new(1, config); + assert!( + matches!(sink.open().await, Err(Error::InvalidConfigValue(_))), + "invalid URL should fail before probing: {url}" + ); + } + }); + } + + #[test] + fn given_invalid_duration_when_opened_should_return_config_error() { + let runtime = test_runtime(); + runtime.block_on(async { + for field in ["retry_delay", "retry_max_delay", "open_retry_max_delay", "timeout"] { + for value in ["30", "0s"] { + let mut config = test_config(); + let setting = match field { + "retry_delay" => &mut config.retry_delay, + "retry_max_delay" => &mut config.retry_max_delay, + "open_retry_max_delay" => &mut config.open_retry_max_delay, + "timeout" => &mut config.timeout, + _ => unreachable!(), + }; + *setting = Some(value.to_string()); + let mut sink = QuickwitSink::new(1, config); + assert!( + matches!(sink.open().await, Err(Error::InvalidConfigValue(reason)) if reason.contains(field)), + "invalid {field} should fail before probing: {value}" + ); + } + } + }); + } + + #[test] + fn given_unknown_config_key_when_deserialized_should_reject_it() { + let config = "url: http://localhost:7280\nindex: 'index_id: test'\nmax_retires: 1"; + assert!(serde_yaml_ng::from_str::(config).is_err()); + } + + #[test] + fn given_payload_variants_when_extracted_should_preserve_object_and_text_documents() { + let sink = QuickwitSink::new(1, test_config()); + let object = simd_json::json!({"key": "value"}); + let raw_object = b" \n{\"key\": \"value\"}".to_vec(); + let messages = vec![ + test_message(Payload::Json(object.clone())), + test_message(Payload::Raw(raw_object.clone())), + test_message(Payload::Avro(raw_object.clone())), + test_message(Payload::FlatBuffer(raw_object)), + test_message(Payload::Text("hello quickwit".to_string())), + test_message(Payload::Proto("hello quickwit".to_string())), + ]; + let extracted = sink.extract_json_payloads(messages); + assert_eq!( + &extracted[..4], + &[object.clone(), object.clone(), object.clone(), object] + ); + let text = simd_json::json!({"text": "hello quickwit", "data_type": "text"}); + assert_eq!(&extracted[4..], &[text.clone(), text]); + } + + #[test] + fn given_raw_nonobjects_when_extracted_should_preserve_original_bytes() { + let sink = QuickwitSink::new(1, test_config()); + for raw_text in [ + "42", + "\"text\"", + "[1,2]", + "null", + "true", + "", + "plain text", + r#"{"message":"escaped\ntext","broken":}"#, + ] { + let extracted = sink.extract_json_payloads(vec![test_message(Payload::Raw( + raw_text.as_bytes().to_vec(), + ))]); + assert_eq!( + extracted, + vec![simd_json::json!({ + "data": raw_text, + "data_type": "raw", + "data_encoding": "utf8" + })], + "raw input should remain unchanged: {raw_text:?}" + ); + } + let binary = vec![0, 15, 255]; + let extracted = + sink.extract_json_payloads(vec![test_message(Payload::Raw(binary.clone()))]); + assert_eq!( + extracted, + vec![simd_json::json!({ + "data": general_purpose::STANDARD.encode(binary), + "data_type": "raw", + "data_encoding": "base64" + })] + ); + } + + #[test] + fn given_json_nonobjects_when_extracted_should_wrap_original_values() { + let sink = QuickwitSink::new(1, test_config()); + for value in [ + simd_json::json!(42), + simd_json::json!("text"), + simd_json::json!([1, 2]), + simd_json::json!(null), + simd_json::json!(true), + ] { + let extracted = + sink.extract_json_payloads(vec![test_message(Payload::Json(value.clone()))]); + assert_eq!( + extracted, + vec![simd_json::json!({"data": value, "data_type": "json"})] + ); + } + } + + #[test] + fn given_index_creation_race_when_opened_should_verify_the_existing_index() { + let runtime = test_runtime(); + runtime.block_on(async { + for status in [400, 409, 503] { + let (url, server) = start_test_server(vec![ + ("GET /prefix/health/readyz HTTP/1.1", 200, ""), + ("GET /prefix/api/v1/indexes/test HTTP/1.1", 404, ""), + ( + "POST /prefix/api/v1/indexes HTTP/1.1", + status, + "index `test` already exist(s)", + ), + ("GET /prefix/api/v1/indexes/test HTTP/1.1", 200, "{}"), + ]) + .await; + let mut config = test_config(); + config.url = format!("{url}/prefix///"); + let mut sink = QuickwitSink::new(1, config); + sink.open() + .await + .expect("index creation race should recover"); + assert_eq!(sink.index_id, "test"); + assert_eq!( + sink.ingest_url, + format!("{url}/prefix/api/v1/test/ingest?commit=auto") + ); + let requests = server.await.expect("test server should finish"); + assert_eq!(requests[2], sink.config.index.as_bytes()); + } + }); + } + + #[test] + fn given_failed_index_creation_when_index_still_absent_should_fail_open() { + let runtime = test_runtime(); + runtime.block_on(async { + let (url, server) = start_test_server(vec![ + ("GET /health/readyz HTTP/1.1", 200, ""), + ("GET /api/v1/indexes/test HTTP/1.1", 404, ""), + ( + "POST /api/v1/indexes HTTP/1.1", + 409, + "index creation failed", + ), + ("GET /api/v1/indexes/test HTTP/1.1", 404, ""), + ]) + .await; + let mut config = test_config(); + config.url = url; + let mut sink = QuickwitSink::new(1, config); + assert!(matches!(sink.open().await, Err(Error::InitError(_)))); + assert!(sink.client.is_none()); + server.await.expect("test server should finish"); + }); + } + + #[test] + fn given_ingest_failure_when_consumed_should_classify_http_errors() { + let runtime = test_runtime(); + runtime.block_on(async { + for status in [400, 429, 503] { + let (url, server) = start_test_server(vec![ + ("GET /health/readyz HTTP/1.1", 200, ""), + ("GET /api/v1/indexes/test HTTP/1.1", 200, "{}"), + ( + "POST /api/v1/test/ingest?commit=auto HTTP/1.1", + status, + "rejected", + ), + ]) + .await; + let mut config = test_config(); + config.url = url; + let mut sink = QuickwitSink::new(1, config); + sink.open().await.expect("sink should open"); + let result = + consume_payloads(&sink, vec![Payload::Text("message".to_string())]).await; + if status == 400 { + assert!(matches!(result, Err(Error::PermanentHttpError(_)))); + } else { + assert!(matches!(result, Err(Error::HttpRequestFailed(_)))); + } + let requests = server.await.expect("test server should finish"); + let mut body = requests + .into_iter() + .last() + .expect("ingest request should exist"); + assert_eq!( + simd_json::from_slice::(&mut body) + .expect("ingest body should be JSON"), + simd_json::json!({"text": "message", "data_type": "text"}) + ); + } + }); + } + + #[test] + fn given_full_chunk_when_consumed_should_split_and_continue_after_http_failure() { + let runtime = test_runtime(); + runtime.block_on(async { + let (url, server) = start_test_server(vec![ + ("GET /health/readyz HTTP/1.1", 200, ""), + ("GET /api/v1/indexes/test HTTP/1.1", 200, "{}"), + ( + "POST /api/v1/test/ingest?commit=auto HTTP/1.1", + 503, + "unavailable", + ), + ("POST /api/v1/test/ingest?commit=auto HTTP/1.1", 200, "{}"), + ]) + .await; + let mut config = test_config(); + config.url = url; + let mut sink = QuickwitSink::new(1, config); + sink.open().await.expect("sink should open"); + let overhead = simd_json::to_vec(&simd_json::json!({"data": ""})) + .expect("empty document should serialize") + .len() + + 1; + let full_record = + simd_json::json!({"data": "x".repeat(MAX_INGEST_BODY_BYTES - overhead)}); + let last_record = simd_json::json!({"sequence": 2}); + let result = consume_payloads( + &sink, + vec![ + Payload::Json(full_record), + Payload::Json(last_record.clone()), + ], + ) + .await; + assert!(matches!(result, Err(Error::HttpRequestFailed(_)))); + let requests = server.await.expect("test server should finish"); + assert_eq!(requests[2].len(), MAX_INGEST_BODY_BYTES); + assert_eq!(requests[2].last(), Some(&b'\n')); + let mut last_body = requests + .into_iter() + .last() + .expect("last chunk should exist"); + assert_eq!( + simd_json::from_slice::(&mut last_body) + .expect("last chunk should be JSON"), + last_record + ); + }); + } + + #[test] + fn given_oversized_record_when_consumed_should_report_it_and_send_other_records() { + let runtime = test_runtime(); + runtime.block_on(async { + let (url, server) = start_test_server(vec![ + ("GET /health/readyz HTTP/1.1", 200, ""), + ("GET /api/v1/indexes/test HTTP/1.1", 200, "{}"), + ("POST /api/v1/test/ingest?commit=auto HTTP/1.1", 200, "{}"), + ]) + .await; + let mut config = test_config(); + config.url = url; + let mut sink = QuickwitSink::new(1, config); + sink.open().await.expect("sink should open"); + let result = consume_payloads( + &sink, + vec![ + Payload::Json(simd_json::json!({"sequence": 1})), + Payload::Json(simd_json::json!({"data": "x".repeat(MAX_INGEST_BODY_BYTES)})), + Payload::Json(simd_json::json!({"sequence": 3})), + ], + ) + .await; + assert!(matches!(result, Err(Error::InvalidRecordValue(_)))); + let requests = server.await.expect("test server should finish"); + assert_eq!(requests[2], b"{\"sequence\":1}\n{\"sequence\":3}\n"); + }); + } + + fn test_runtime() -> tokio::runtime::Runtime { + tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .expect("test runtime should start") + } + + async fn consume_payloads(sink: &QuickwitSink, payloads: Vec) -> Result<(), Error> { + sink.consume( + &TopicMetadata { + stream: "test".to_string(), + topic: "test".to_string(), + }, + MessagesMetadata { + partition_id: 1, + current_offset: 0, + schema: Schema::Raw, + }, + payloads.into_iter().map(test_message).collect(), + ) + .await + } + + async fn start_test_server( + responses: Vec<(&'static str, u16, &'static str)>, + ) -> (String, JoinHandle>>) { + let listener = TcpListener::bind("127.0.0.1:0") + .await + .expect("test server should bind"); + let url = format!( + "http://{}", + listener + .local_addr() + .expect("test server should have address") + ); + let server = tokio::spawn(async move { + let mut bodies = Vec::with_capacity(responses.len()); + for (expected_request, status, response_body) in responses { + let (mut stream, _) = timeout(TEST_TIMEOUT, listener.accept()) + .await + .expect("request should arrive before timeout") + .expect("request should connect"); + let (request_line, body) = timeout(TEST_TIMEOUT, read_http_request(&mut stream)) + .await + .expect("request should finish before timeout"); + assert_eq!(request_line, expected_request); + bodies.push(body); + let response = format!( + "HTTP/1.1 {status} Test\r\ncontent-length: {}\r\nconnection: close\r\n\r\n{response_body}", + response_body.len() + ); + stream + .write_all(response.as_bytes()) + .await + .expect("response should be written"); + } + bodies + }); + (url, server) + } + + async fn read_http_request(stream: &mut TcpStream) -> (String, Vec) { + let mut buffer = Vec::new(); + let mut chunk = [0u8; HTTP_READ_BUFFER_BYTES]; + let headers_end = loop { + let read = stream.read(&mut chunk).await.expect("request should read"); + assert_ne!(read, 0, "request should include headers"); + buffer.extend_from_slice(&chunk[..read]); + if let Some(position) = buffer + .windows(b"\r\n\r\n".len()) + .position(|window| window == b"\r\n\r\n") + { + break position; + } + }; + let headers = std::str::from_utf8(&buffer[..headers_end]).expect("headers should be UTF-8"); + let request_line = headers + .lines() + .next() + .expect("request line should exist") + .to_string(); + let content_length = headers + .lines() + .find_map(|line| { + let (name, value) = line.split_once(':')?; + name.eq_ignore_ascii_case("content-length").then(|| { + value + .trim() + .parse::() + .expect("content length should be valid") + }) + }) + .unwrap_or(0); + let body_start = headers_end + b"\r\n\r\n".len(); + while buffer.len() < body_start + content_length { + let read = stream.read(&mut chunk).await.expect("body should read"); + assert_ne!(read, 0, "request should include declared body"); + buffer.extend_from_slice(&chunk[..read]); + } + ( + request_line, + buffer[body_start..body_start + content_length].to_vec(), + ) + } +} diff --git a/core/integration/tests/connectors/fixtures/mod.rs b/core/integration/tests/connectors/fixtures/mod.rs index af9eb52990..f7af061440 100644 --- a/core/integration/tests/connectors/fixtures/mod.rs +++ b/core/integration/tests/connectors/fixtures/mod.rs @@ -83,7 +83,10 @@ pub use postgres::{ PostgresSourceJsonFixture, PostgresSourceJsonbFixture, PostgresSourceMarkFixture, PostgresSourceOps, }; -pub use quickwit::{QuickwitFixture, QuickwitOps, QuickwitPreCreatedFixture}; +pub use quickwit::{ + QuickwitFixture, QuickwitOps, QuickwitPreCreatedFixture, QuickwitRawFixture, + QuickwitTextFixture, +}; pub use rabbitmq::{ RabbitMqOps, RabbitMqSinkDirectFixture, RabbitMqSinkFanoutFixture, RabbitMqSinkFixture, RabbitMqSinkHeadersFixture, RabbitMqSinkRawSchemaFixture, RabbitMqSinkUnroutableFixture, diff --git a/core/integration/tests/connectors/fixtures/quickwit/container.rs b/core/integration/tests/connectors/fixtures/quickwit/container.rs index 2628290638..b6d671839f 100644 --- a/core/integration/tests/connectors/fixtures/quickwit/container.rs +++ b/core/integration/tests/connectors/fixtures/quickwit/container.rs @@ -42,6 +42,7 @@ const QUICKWIT_LISTEN_ADDRESS: &str = "0.0.0.0"; const ENV_PLUGIN_URL: &str = "IGGY_CONNECTORS_SINK_QUICKWIT_PLUGIN_CONFIG_URL"; const ENV_PLUGIN_INDEX: &str = "IGGY_CONNECTORS_SINK_QUICKWIT_PLUGIN_CONFIG_INDEX"; +const ENV_PLUGIN_CONFIG_FORMAT: &str = "IGGY_CONNECTORS_SINK_QUICKWIT_PLUGIN_CONFIG_FORMAT"; const ENV_STREAMS_0_STREAM: &str = "IGGY_CONNECTORS_SINK_QUICKWIT_STREAMS_0_STREAM"; const ENV_STREAMS_0_TOPICS: &str = "IGGY_CONNECTORS_SINK_QUICKWIT_STREAMS_0_TOPICS"; const ENV_STREAMS_0_SCHEMA: &str = "IGGY_CONNECTORS_SINK_QUICKWIT_STREAMS_0_SCHEMA"; @@ -129,6 +130,10 @@ pub trait QuickwitOps: Sync { fn container(&self) -> &QuickwitContainer; fn http_client(&self) -> &HttpClient; + fn timestamp_field(&self) -> Option<&str> { + Some(INDEX_TIMESTAMP_FIELD) + } + fn create_index( &self, index_config: &str, @@ -176,6 +181,7 @@ pub trait QuickwitOps: Sync { .http_client() .post(&ingest_url) .query(&[("commit", "force")]) + // A scalar forces a commit without adding a searchable document. .json("{}") .send() .await @@ -204,13 +210,12 @@ pub trait QuickwitOps: Sync { { async move { let search_url = format!("{}/api/v1/{}/search", self.container().base_url(), index_id); - let descending = "-"; + let mut request = self.http_client().get(&search_url).query(&[("query", "")]); + if let Some(timestamp_field) = self.timestamp_field() { + request = request.query(&[("sort_by", format!("-{timestamp_field}"))]); + } - let response = self - .http_client() - .get(&search_url) - .query(&[("query", "")]) - .query(&[("sort_by", format!("{descending}{INDEX_TIMESTAMP_FIELD}"))]) + let response = request .send() .await .map_err(|e| TestBinaryError::InvalidState { @@ -302,6 +307,7 @@ retention: fn build_connector_envs(base_url: &str) -> HashMap { HashMap::from([ (ENV_PLUGIN_URL.to_string(), base_url.to_string()), + (ENV_PLUGIN_CONFIG_FORMAT.to_string(), "yaml".to_string()), ( ENV_PLUGIN_INDEX.to_string(), get_index_config(seeds::names::TOPIC), @@ -326,6 +332,7 @@ fn build_connector_envs(base_url: &str) -> HashMap { pub struct QuickwitFixture { container: QuickwitContainer, http_client: HttpClient, + timestamp_field: Option<&'static str>, } impl QuickwitOps for QuickwitFixture { @@ -336,6 +343,10 @@ impl QuickwitOps for QuickwitFixture { fn http_client(&self) -> &HttpClient { &self.http_client } + + fn timestamp_field(&self) -> Option<&str> { + self.timestamp_field + } } #[async_trait] @@ -347,6 +358,7 @@ impl TestFixture for QuickwitFixture { Ok(Self { container, http_client, + timestamp_field: Some(INDEX_TIMESTAMP_FIELD), }) } @@ -386,6 +398,7 @@ impl TestFixture for QuickwitPreCreatedFixture { let inner = QuickwitFixture { container, http_client, + timestamp_field: Some(INDEX_TIMESTAMP_FIELD), }; let index_config = get_index_config(seeds::names::TOPIC); @@ -398,3 +411,64 @@ impl TestFixture for QuickwitPreCreatedFixture { self.inner.connectors_runtime_envs() } } + +pub struct QuickwitRawFixture { + inner: QuickwitFixture, +} + +impl std::ops::Deref for QuickwitRawFixture { + type Target = QuickwitFixture; + + fn deref(&self) -> &Self::Target { + &self.inner + } +} + +#[async_trait] +impl TestFixture for QuickwitRawFixture { + async fn setup() -> Result { + let mut inner = QuickwitFixture::setup().await?; + inner.timestamp_field = None; + Ok(Self { inner }) + } + + fn connectors_runtime_envs(&self) -> HashMap { + let mut envs = self.inner.connectors_runtime_envs(); + envs.insert(ENV_STREAMS_0_SCHEMA.to_string(), "raw".to_string()); + envs.insert( + ENV_PLUGIN_INDEX.to_string(), + format!( + "version: 0.8\nindex_id: {}\ndoc_mapping:\n mode: dynamic\n", + seeds::names::TOPIC + ), + ); + envs + } +} + +pub struct QuickwitTextFixture { + inner: QuickwitRawFixture, +} + +impl std::ops::Deref for QuickwitTextFixture { + type Target = QuickwitRawFixture; + + fn deref(&self) -> &Self::Target { + &self.inner + } +} + +#[async_trait] +impl TestFixture for QuickwitTextFixture { + async fn setup() -> Result { + Ok(Self { + inner: QuickwitRawFixture::setup().await?, + }) + } + + fn connectors_runtime_envs(&self) -> HashMap { + let mut envs = self.inner.connectors_runtime_envs(); + envs.insert(ENV_STREAMS_0_SCHEMA.to_string(), "text".to_string()); + envs + } +} diff --git a/core/integration/tests/connectors/fixtures/quickwit/mod.rs b/core/integration/tests/connectors/fixtures/quickwit/mod.rs index c8d3f41556..66f11243a1 100644 --- a/core/integration/tests/connectors/fixtures/quickwit/mod.rs +++ b/core/integration/tests/connectors/fixtures/quickwit/mod.rs @@ -17,4 +17,7 @@ mod container; -pub use container::{QuickwitFixture, QuickwitOps, QuickwitPreCreatedFixture}; +pub use container::{ + QuickwitFixture, QuickwitOps, QuickwitPreCreatedFixture, QuickwitRawFixture, + QuickwitTextFixture, +}; diff --git a/core/integration/tests/connectors/quickwit/quickwit_sink.rs b/core/integration/tests/connectors/quickwit/quickwit_sink.rs index 240e2a62f3..bdc1eda0b8 100644 --- a/core/integration/tests/connectors/quickwit/quickwit_sink.rs +++ b/core/integration/tests/connectors/quickwit/quickwit_sink.rs @@ -16,12 +16,15 @@ // under the License. use crate::connectors::create_test_messages; -use crate::connectors::fixtures::{QuickwitFixture, QuickwitOps, QuickwitPreCreatedFixture}; +use crate::connectors::fixtures::{ + QuickwitFixture, QuickwitOps, QuickwitPreCreatedFixture, QuickwitRawFixture, + QuickwitTextFixture, +}; use bytes::Bytes; use iggy::prelude::{IggyMessage, Partitioning}; use iggy_common::Identifier; use iggy_common::MessageClient; -use integration::harness::seeds; +use integration::harness::{TestHarness, seeds}; use integration::iggy_harness; use serde::{Deserialize, Serialize}; @@ -253,3 +256,124 @@ async fn given_invalid_messages_should_not_store(harness: &TestHarness, fixture: ); } } + +#[iggy_harness( + server(connectors_runtime(config_path = "tests/connectors/quickwit/sink.toml")), + seed = seeds::connector_stream +)] +async fn given_raw_messages_should_store_objects_and_encoded_wrappers( + harness: &TestHarness, + fixture: QuickwitRawFixture, +) { + let cases = [ + ( + Bytes::from_static(br#"{"message":"raw JSON object"}"#), + serde_json::json!({"message": "raw JSON object"}), + ), + ( + Bytes::from_static(b"YWJj"), + serde_json::json!({"data": "YWJj", "data_type": "raw", "data_encoding": "utf8"}), + ), + ( + Bytes::from_static(b"\xff\x00\x80"), + serde_json::json!({"data": "/wCA", "data_type": "raw", "data_encoding": "base64"}), + ), + ( + Bytes::from_static(br#"{"message":"escaped\ntext",broken}"#), + serde_json::json!({ + "data": r#"{"message":"escaped\ntext",broken}"#, + "data_type": "raw", + "data_encoding": "utf8" + }), + ), + ( + Bytes::from_static(b"42"), + serde_json::json!({"data": "42", "data_type": "raw", "data_encoding": "utf8"}), + ), + ( + Bytes::from_static(br#""text""#), + serde_json::json!({"data": "\"text\"", "data_type": "raw", "data_encoding": "utf8"}), + ), + ( + Bytes::from_static(b"[1,2]"), + serde_json::json!({"data": "[1,2]", "data_type": "raw", "data_encoding": "utf8"}), + ), + ( + Bytes::from_static(b"null"), + serde_json::json!({"data": "null", "data_type": "raw", "data_encoding": "utf8"}), + ), + ]; + + assert_documents_stored(harness, &fixture, &cases).await; +} + +#[iggy_harness( + server(connectors_runtime(config_path = "tests/connectors/quickwit/sink.toml")), + seed = seeds::connector_stream +)] +async fn given_text_messages_should_store_text_wrappers( + harness: &TestHarness, + fixture: QuickwitTextFixture, +) { + let cases = [ + ( + Bytes::from_static(b"log entry\n\"ready\""), + serde_json::json!({"text": "log entry\n\"ready\"", "data_type": "text"}), + ), + ( + Bytes::from_static(br#"{"message":"text containing JSON"}"#), + serde_json::json!({ + "text": r#"{"message":"text containing JSON"}"#, + "data_type": "text" + }), + ), + ]; + + assert_documents_stored(harness, &fixture, &cases).await; +} + +async fn assert_documents_stored( + harness: &TestHarness, + fixture: &QuickwitFixture, + cases: &[(Bytes, serde_json::Value)], +) { + let client = harness.root_client().await.expect("create Iggy client"); + let stream_id: Identifier = seeds::names::STREAM.try_into().expect("stream identifier"); + let topic_id: Identifier = seeds::names::TOPIC.try_into().expect("topic identifier"); + let mut messages: Vec = cases + .iter() + .enumerate() + .map(|(message_index, (payload, _))| { + IggyMessage::builder() + .id(message_index as u128 + 1) + .payload(payload.clone()) + .build() + .expect("build message") + }) + .collect(); + + client + .send_messages( + &stream_id, + &topic_id, + &Partitioning::partition_id(0), + &mut messages, + ) + .await + .expect("send messages"); + + let search = fixture + .wait_for_documents(seeds::names::TOPIC, cases.len()) + .await + .expect("wait for indexed documents"); + + assert_eq!(search.num_hits, cases.len()); + assert_eq!(search.hits.len(), cases.len()); + for (_, expected) in cases { + assert!( + search.hits.contains(expected), + "Missing document {expected}; search hits: {:?}", + search.hits + ); + } +} From df4c75d34614865ea3bc1ffb2b117c8247146091 Mon Sep 17 00:00:00 2001 From: Elioooon <102770919+Elioooon@users.noreply.github.com> Date: Thu, 10 Sep 2026 19:17:08 +0800 Subject: [PATCH 097/182] feat(python): expose partition management (#4017) Expose `create_partitions` and `delete_partitions` on the Python `IggyClient`, delegating to the existing Rust SDK operations and error conversion. Document ID reuse, validation limits, asynchronous materialization, consumer-group rebalancing, and the permission hierarchy. Creating a topic accepts zero partitions; adding or deleting partitions requires a positive count. Closes #4014 --- core/simulator/src/lib.rs | 20 +++ foreign/python/apache_iggy.pyi | 62 ++++++- foreign/python/src/client.rs | 203 ++++++++++++++-------- foreign/python/tests/test_partition.py | 228 +++++++++++++++++++++++++ 4 files changed, 439 insertions(+), 74 deletions(-) create mode 100644 foreign/python/tests/test_partition.py diff --git a/core/simulator/src/lib.rs b/core/simulator/src/lib.rs index 51a2d35598..3fd111c555 100644 --- a/core/simulator/src/lib.rs +++ b/core/simulator/src/lib.rs @@ -6305,6 +6305,17 @@ mod partition_repair_driver_tests { below would have nothing to lose" ); let counted = gap_drops(&sim, LAGGING, namespace); + let partition_stats = sim.replicas[LAGGING as usize] + .partition_shard(namespace) + .plane + .partitions() + .get_by_ns(&namespace) + .expect("partition exists before ConfirmRemove") + .stats + .clone(); + let topic_stats = partition_stats.parent(); + let stream_stats = topic_stats.parent(); + assert!(partition_stats.messages_count_inconsistent() > 0); // The reconciler's teardown order: the tombstone lands first and the // disk delete runs before `ConfirmRemove`, so from here `get_mut_by_ns` @@ -6327,6 +6338,15 @@ mod partition_repair_driver_tests { "the {buffered} prepare(s) buffered on the partition went to the floor \ with it; the drops are the only record those frames existed" ); + assert_eq!(partition_stats.messages_count_inconsistent(), 0); + assert_eq!(partition_stats.size_bytes_inconsistent(), 0); + assert_eq!(partition_stats.segments_count_inconsistent(), 0); + assert_eq!(topic_stats.messages_count_inconsistent(), 0); + assert_eq!(topic_stats.size_bytes_inconsistent(), 0); + assert_eq!(topic_stats.segments_count_inconsistent(), 0); + assert_eq!(stream_stats.messages_count_inconsistent(), 0); + assert_eq!(stream_stats.size_bytes_inconsistent(), 0); + assert_eq!(stream_stats.segments_count_inconsistent(), 0); } } diff --git a/foreign/python/apache_iggy.pyi b/foreign/python/apache_iggy.pyi index 901af64aa4..adebabee94 100644 --- a/foreign/python/apache_iggy.pyi +++ b/foreign/python/apache_iggy.pyi @@ -1325,7 +1325,7 @@ class IggyClient: Args: stream: Stream identifier as `str | int`. name: Topic name as `str`. - partitions_count: Number of partitions as `int`. + partitions_count: Number of partitions as `int`, at most 1000. compression_algorithm: Compression algorithm as `str | None`. message_expiry: Message expiry as `IggyExpiry | None`. max_topic_size: Maximum topic size as `MaxTopicSize | None`. @@ -1443,6 +1443,66 @@ class IggyClient: Raises: RuntimeError: If an identifier is invalid or the request fails. """ + def create_partitions( + self, + stream_id: builtins.str | builtins.int, + topic_id: builtins.str | builtins.int, + partitions_count: builtins.int, + ) -> collections.abc.Awaitable[None]: + r""" + Create partitions for a topic. New partition IDs continue from one past the + current highest ID; IDs removed by deletion can be reused. Existing consumer + groups are immediately rebalanced across all partitions, advancing their + generation and dropping pending revocations. + + Args: + stream_id: Stream identifier as `str | int`. + topic_id: Topic identifier as `str | int`. + partitions_count: Number of partitions to create as `int`, between 1 and + 1000 inclusive. + + Returns: + An awaitable that resolves to `None` when the partitions are committed; + storage materialization completes asynchronously. + + Raises: + ValueError: If an identifier is invalid. + OverflowError: If `partitions_count` is outside the unsigned 32-bit range. + RuntimeError: If the client is not authenticated, lacks global + `manage_streams` or `manage_topics`, per-stream `manage_stream` or + `manage_topics`, or per-topic `manage_topic` permission, or the + request fails. + """ + def delete_partitions( + self, + stream_id: builtins.str | builtins.int, + topic_id: builtins.str | builtins.int, + partitions_count: builtins.int, + ) -> collections.abc.Awaitable[None]: + r""" + Delete the last partitions from a topic, including all messages stored in them. + Existing consumer groups are immediately rebalanced across the remaining + partitions, advancing their generation and dropping pending revocations. + + Args: + stream_id: Stream identifier as `str | int`. + topic_id: Topic identifier as `str | int`. + partitions_count: Number of partitions to delete as `int` from the end of + the topic; must be between 1 and 1000 inclusive and no greater than + its current count. + + Returns: + An awaitable that resolves to `None` when deletion is accepted; storage + teardown completes asynchronously. + + Raises: + ValueError: If an identifier is invalid. + OverflowError: If `partitions_count` is outside the unsigned 32-bit range. + RuntimeError: If the client is not authenticated, lacks global + `manage_streams` or `manage_topics`, per-stream `manage_stream` or + `manage_topics`, or per-topic `manage_topic` permission, or the + request fails. + """ def create_consumer_group( self, stream_id: builtins.str | builtins.int, diff --git a/foreign/python/src/client.rs b/foreign/python/src/client.rs index 0fdef39d9c..0e3a7b9376 100644 --- a/foreign/python/src/client.rs +++ b/foreign/python/src/client.rs @@ -60,7 +60,7 @@ pub struct IggyClient { inner: Arc, } -/// Keeps the SDK's own message on the `RuntimeError` the Python surface raises. +/// Converts SDK errors to the RuntimeError exposed by the Python API. fn to_runtime_error(error: E) -> PyErr { PyErr::new::(error.to_string()) } @@ -174,8 +174,8 @@ impl IggyClient { // is a no-op for the other transports since the protocol isn't known until the // connection string is parsed. let _guard = pyo3_async_runtimes::tokio::get_runtime().enter(); - let client = RustIggyClient::from_connection_string(&connection_string) - .map_err(|e| PyErr::new::(e.to_string()))?; + let client = + RustIggyClient::from_connection_string(&connection_string).map_err(to_runtime_error)?; Ok(Self { inner: Arc::new(client), }) @@ -186,12 +186,10 @@ impl IggyClient { #[gen_stub(override_return_type(type_repr="collections.abc.Awaitable[None]", imports=("collections.abc")))] fn ping<'a>(&self, py: Python<'a>) -> PyResult> { let inner = self.inner.clone(); - future_into_py(py, async move { - inner - .ping() - .await - .map_err(|e| PyErr::new::(e.to_string())) - }) + future_into_py( + py, + async move { inner.ping().await.map_err(to_runtime_error) }, + ) } /// Get the statistics and details of the server and its running process. @@ -210,10 +208,7 @@ impl IggyClient { fn get_stats<'a>(&self, py: Python<'a>) -> PyResult> { let inner = self.inner.clone(); future_into_py(py, async move { - let stats = inner - .get_stats() - .await - .map_err(|e| PyErr::new::(e.to_string()))?; + let stats = inner.get_stats().await.map_err(to_runtime_error)?; Ok(PyStats::from(stats)) }) } @@ -243,7 +238,7 @@ impl IggyClient { let specs = inner .describe_options(scope) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(specs .into_iter() .map(PyOptionSpec::from) @@ -265,7 +260,7 @@ impl IggyClient { inner .login_user(&username, &password) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(()) }) } @@ -288,10 +283,7 @@ impl IggyClient { let inner = self.inner.clone(); future_into_py(py, async move { - let user = inner - .get_user(&user_id) - .await - .map_err(|e| PyErr::new::(e.to_string()))?; + let user = inner.get_user(&user_id).await.map_err(to_runtime_error)?; Ok(user.map(PyUserInfoDetails::from)) }) } @@ -308,10 +300,7 @@ impl IggyClient { let inner = self.inner.clone(); future_into_py(py, async move { - let users = inner - .get_users() - .await - .map_err(|e| PyErr::new::(e.to_string()))?; + let users = inner.get_users().await.map_err(to_runtime_error)?; Ok(users.into_iter().map(PyUserInfo::from).collect::>()) }) } @@ -349,7 +338,7 @@ impl IggyClient { let user = inner .create_user(&username, &password, status, permissions) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(PyUserInfoDetails::from(user)) }) } @@ -390,7 +379,7 @@ impl IggyClient { &UserUpdateOptions::default(), ) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(()) }) } @@ -415,7 +404,7 @@ impl IggyClient { inner .delete_user(&user_id) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(()) }) } @@ -453,7 +442,7 @@ impl IggyClient { inner .update_permissions(&user_id, permissions) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(()) }) } @@ -486,7 +475,7 @@ impl IggyClient { inner .change_password(&user_id, ¤t_password, &new_password) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(()) }) } @@ -503,10 +492,7 @@ impl IggyClient { let inner = self.inner.clone(); future_into_py(py, async move { - inner - .logout_user() - .await - .map_err(|e| PyErr::new::(e.to_string()))?; + inner.logout_user().await.map_err(to_runtime_error)?; Ok(()) }) } @@ -519,10 +505,7 @@ impl IggyClient { fn connect<'a>(&self, py: Python<'a>) -> PyResult> { let inner = self.inner.clone(); future_into_py(py, async move { - inner - .connect() - .await - .map_err(|e| PyErr::new::(e.to_string()))?; + inner.connect().await.map_err(to_runtime_error)?; Ok(()) }) } @@ -534,10 +517,7 @@ impl IggyClient { fn create_stream<'a>(&self, py: Python<'a>, name: String) -> PyResult> { let inner = self.inner.clone(); future_into_py(py, async move { - inner - .create_stream(&name) - .await - .map_err(|e| PyErr::new::(e.to_string()))?; + inner.create_stream(&name).await.map_err(to_runtime_error)?; Ok(()) }) } @@ -558,7 +538,7 @@ impl IggyClient { let stream = inner .get_stream(&stream_id) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(stream.map(StreamDetails::from)) }) } @@ -575,10 +555,7 @@ impl IggyClient { fn get_streams<'a>(&self, py: Python<'a>) -> PyResult> { let inner = self.inner.clone(); future_into_py(py, async move { - let streams = inner - .get_streams() - .await - .map_err(|e| PyErr::new::(e.to_string()))?; + let streams = inner.get_streams().await.map_err(to_runtime_error)?; Ok(streams.into_iter().map(Stream::from).collect::>()) }) } @@ -627,7 +604,7 @@ impl IggyClient { inner .update_stream(&stream_id, &name, &update_options) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(()) }) } @@ -660,7 +637,7 @@ impl IggyClient { inner .delete_stream(&stream_id) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(()) }) } @@ -694,7 +671,7 @@ impl IggyClient { inner .purge_stream(&stream_id) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(()) }) } @@ -704,7 +681,7 @@ impl IggyClient { /// Args: /// stream: Stream identifier as `str | int`. /// name: Topic name as `str`. - /// partitions_count: Number of partitions as `int`. + /// partitions_count: Number of partitions as `int`, at most 1000. /// compression_algorithm: Compression algorithm as `str | None`. /// message_expiry: Message expiry as `IggyExpiry | None`. /// max_topic_size: Maximum topic size as `MaxTopicSize | None`. @@ -784,7 +761,7 @@ impl IggyClient { inner .create_topic(&stream, &name, &topic_options) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(()) }) } @@ -807,7 +784,7 @@ impl IggyClient { let topic = inner .get_topic(&stream_id, &topic_id) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(topic.map(TopicDetails::from)) }) } @@ -835,7 +812,7 @@ impl IggyClient { let topics = inner .get_topics(&stream_id) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(topics.into_iter().map(Topic::from).collect::>()) }) } @@ -889,10 +866,7 @@ impl IggyClient { // Absent stays absent: a key the caller did not pass is left alone // server-side rather than reset to a default. let compression_algorithm = compression_algorithm - .map(|algo| { - CompressionAlgorithm::from_str(&algo) - .map_err(|e| PyErr::new::(e.to_string())) - }) + .map(|algo| CompressionAlgorithm::from_str(&algo).map_err(to_runtime_error)) .transpose()?; let update_options = TopicUpdateOptions { compression_algorithm, @@ -909,7 +883,7 @@ impl IggyClient { inner .update_topic(&stream_id, &topic_id, &name, &update_options) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(()) }) } @@ -940,7 +914,7 @@ impl IggyClient { inner .delete_topic(&stream_id, &topic_id) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(()) }) } @@ -971,7 +945,93 @@ impl IggyClient { inner .purge_topic(&stream_id, &topic_id) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; + Ok(()) + }) + } + + /// Create partitions for a topic. New partition IDs continue from one past the + /// current highest ID; IDs removed by deletion can be reused. Existing consumer + /// groups are immediately rebalanced across all partitions, advancing their + /// generation and dropping pending revocations. + /// + /// Args: + /// stream_id: Stream identifier as `str | int`. + /// topic_id: Topic identifier as `str | int`. + /// partitions_count: Number of partitions to create as `int`, between 1 and + /// 1000 inclusive. + /// + /// Returns: + /// An awaitable that resolves to `None` when the partitions are committed; + /// storage materialization completes asynchronously. + /// + /// Raises: + /// ValueError: If an identifier is invalid. + /// OverflowError: If `partitions_count` is outside the unsigned 32-bit range. + /// RuntimeError: If the client is not authenticated, lacks global + /// `manage_streams` or `manage_topics`, per-stream `manage_stream` or + /// `manage_topics`, or per-topic `manage_topic` permission, or the + /// request fails. + #[gen_stub(override_return_type(type_repr="collections.abc.Awaitable[None]", imports=("collections.abc")))] + fn create_partitions<'a>( + &self, + py: Python<'a>, + stream_id: PyIdentifier, + topic_id: PyIdentifier, + partitions_count: u32, + ) -> PyResult> { + let stream_id = Identifier::try_from(stream_id)?; + let topic_id = Identifier::try_from(topic_id)?; + let inner = self.inner.clone(); + + future_into_py(py, async move { + inner + .create_partitions(&stream_id, &topic_id, partitions_count) + .await + .map_err(to_runtime_error)?; + Ok(()) + }) + } + + /// Delete the last partitions from a topic, including all messages stored in them. + /// Existing consumer groups are immediately rebalanced across the remaining + /// partitions, advancing their generation and dropping pending revocations. + /// + /// Args: + /// stream_id: Stream identifier as `str | int`. + /// topic_id: Topic identifier as `str | int`. + /// partitions_count: Number of partitions to delete as `int` from the end of + /// the topic; must be between 1 and 1000 inclusive and no greater than + /// its current count. + /// + /// Returns: + /// An awaitable that resolves to `None` when deletion is accepted; storage + /// teardown completes asynchronously. + /// + /// Raises: + /// ValueError: If an identifier is invalid. + /// OverflowError: If `partitions_count` is outside the unsigned 32-bit range. + /// RuntimeError: If the client is not authenticated, lacks global + /// `manage_streams` or `manage_topics`, per-stream `manage_stream` or + /// `manage_topics`, or per-topic `manage_topic` permission, or the + /// request fails. + #[gen_stub(override_return_type(type_repr="collections.abc.Awaitable[None]", imports=("collections.abc")))] + fn delete_partitions<'a>( + &self, + py: Python<'a>, + stream_id: PyIdentifier, + topic_id: PyIdentifier, + partitions_count: u32, + ) -> PyResult> { + let stream_id = Identifier::try_from(stream_id)?; + let topic_id = Identifier::try_from(topic_id)?; + let inner = self.inner.clone(); + + future_into_py(py, async move { + inner + .delete_partitions(&stream_id, &topic_id, partitions_count) + .await + .map_err(to_runtime_error)?; Ok(()) }) } @@ -1005,7 +1065,7 @@ impl IggyClient { inner .create_consumer_group(&stream_id, &topic_id, &name) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(()) }) } @@ -1041,7 +1101,7 @@ impl IggyClient { let group = inner .get_consumer_group(&stream_id, &topic_id, &group_id) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(group.map(PyConsumerGroupDetails::from)) }) } @@ -1073,7 +1133,7 @@ impl IggyClient { let groups = inner .get_consumer_groups(&stream_id, &topic_id) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(groups .into_iter() .map(PyConsumerGroup::from) @@ -1111,7 +1171,7 @@ impl IggyClient { inner .delete_consumer_group(&stream_id, &topic_id, &group_id) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(()) }) } @@ -1149,7 +1209,7 @@ impl IggyClient { inner .join_consumer_group(&stream_id, &topic_id, &group_id) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(()) }) } @@ -1189,7 +1249,7 @@ impl IggyClient { inner .leave_consumer_group(&stream_id, &topic_id, &group_id) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(()) }) } @@ -1247,7 +1307,7 @@ impl IggyClient { let response = inner .send_messages(&stream, &topic, &partitioning, messages.as_mut()) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(PySendMessagesResponse::from(response)) }) } @@ -1289,7 +1349,7 @@ impl IggyClient { auto_commit, ) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; let partition_id = polled_messages.partition_id; let messages = polled_messages .messages @@ -1365,7 +1425,7 @@ impl IggyClient { let mut builder = self .inner .consumer_group(name, stream, topic) - .map_err(|e| PyErr::new::(e.to_string()))? + .map_err(to_runtime_error)? .without_encryptor() .partition(partition_id); @@ -1425,10 +1485,7 @@ impl IggyClient { let mut consumer = builder.build(); future_into_py(py, async move { - consumer - .init() - .await - .map_err(|e| PyErr::new::(e.to_string()))?; + consumer.init().await.map_err(to_runtime_error)?; let state = consumer.state(); let name = consumer.name().to_string(); let stream = PyIdentifier::try_from(consumer.stream())?; @@ -1469,7 +1526,7 @@ impl IggyClient { let response = inner .send_binary_request(code, Bytes::from(payload)) .await - .map_err(|e| PyErr::new::(e.to_string()))?; + .map_err(to_runtime_error)?; Ok(Python::attach(|py| PyBytes::new(py, &response).unbind())) }) } diff --git a/foreign/python/tests/test_partition.py b/foreign/python/tests/test_partition.py new file mode 100644 index 0000000000..193b3cb42a --- /dev/null +++ b/foreign/python/tests/test_partition.py @@ -0,0 +1,228 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +import asyncio + +import pytest + +from apache_iggy import ( + IggyClient, + Permissions, + SendMessage, + StreamPermissions, + TopicPermissions, +) + +from .utils import login_fresh_client, unique_credentials + + +async def _create_topic(iggy_client: IggyClient, unique_name): + stream_name = unique_name() + topic_name = unique_name() + + await iggy_client.create_stream(stream_name) + await iggy_client.create_topic( + stream=stream_name, name=topic_name, partitions_count=2 + ) + return stream_name, topic_name + + +async def _wait_for_messages(iggy_client, stream_id, topic_id, expected): + for _ in range(100): + topic = await iggy_client.get_topic(stream_id, topic_id) + if topic is not None and topic.messages_count == expected: + return topic + await asyncio.sleep(0.01) + raise AssertionError(f"topic messages_count did not reach {expected}") + + +class TestPartitionManagement: + @pytest.mark.asyncio + @pytest.mark.parametrize("numeric_ids", [False, True]) + async def test_create_and_delete_partitions( + self, iggy_client: IggyClient, unique_name, numeric_ids: bool + ): + stream_name, topic_name = await _create_topic(iggy_client, unique_name) + stream = await iggy_client.get_stream(stream_name) + assert stream is not None + topic = await iggy_client.get_topic(stream.id, topic_name) + assert topic is not None + stream_id = stream.id if numeric_ids else stream_name + topic_id = topic.id if numeric_ids else topic_name + + await iggy_client.create_partitions(stream_id, topic_id, 2) + created = await iggy_client.get_topic(stream_id, topic_id) + assert created is not None + assert created.partitions_count == 4 + assert [partition.id for partition in created.partitions] == [0, 1, 2, 3] + + await iggy_client.delete_partitions(stream_id, topic_id, 2) + deleted = await iggy_client.get_topic(stream_id, topic_id) + assert deleted is not None + assert deleted.partitions_count == 2 + assert [partition.id for partition in deleted.partitions] == [0, 1] + + @pytest.mark.asyncio + async def test_delete_partitions_rolls_back_stats( + self, iggy_client: IggyClient, unique_name + ): + stream_name, topic_name = await _create_topic(iggy_client, unique_name) + await iggy_client.create_partitions(stream_name, topic_name, 2) + await iggy_client.send_messages( + stream_name, topic_name, 0, [SendMessage("retained")] + ) + await iggy_client.send_messages( + stream_name, topic_name, 3, [SendMessage("deleted")] + ) + await _wait_for_messages(iggy_client, stream_name, topic_name, 2) + + await iggy_client.delete_partitions(stream_name, topic_name, 2) + deleted = await _wait_for_messages(iggy_client, stream_name, topic_name, 1) + assert [partition.id for partition in deleted.partitions] == [0, 1] + assert deleted.partitions[0].messages_count == 1 + + @pytest.mark.asyncio + @pytest.mark.parametrize("method", ["create_partitions", "delete_partitions"]) + @pytest.mark.parametrize("partitions_count", [0, 1001]) + async def test_partition_management_rejects_invalid_count( + self, + iggy_client: IggyClient, + unique_name, + method: str, + partitions_count: int, + ): + stream_name, topic_name = await _create_topic(iggy_client, unique_name) + + # Zero shares the legacy TooManyPartitions code with an over-limit count. + with pytest.raises(RuntimeError, match="Too many partitions"): + await getattr(iggy_client, method)( + stream_name, topic_name, partitions_count + ) + + @pytest.mark.asyncio + async def test_delete_partitions_rejects_count_larger_than_topic( + self, iggy_client: IggyClient, unique_name + ): + stream_name, topic_name = await _create_topic(iggy_client, unique_name) + + with pytest.raises(RuntimeError, match="Invalid partitions count"): + await iggy_client.delete_partitions(stream_name, topic_name, 3) + + @pytest.mark.asyncio + @pytest.mark.parametrize("method", ["create_partitions", "delete_partitions"]) + @pytest.mark.parametrize("missing", ["stream", "topic"]) + async def test_partition_management_rejects_missing_stream_or_topic( + self, iggy_client: IggyClient, unique_name, method: str, missing: str + ): + stream_name, topic_name = await _create_topic(iggy_client, unique_name) + missing_name = unique_name() + stream_id = missing_name if missing == "stream" else stream_name + topic_id = missing_name if missing == "topic" else topic_name + + with pytest.raises(RuntimeError, match=r"was not found\."): + await getattr(iggy_client, method)(stream_id, topic_id, 1) + + @pytest.mark.asyncio + @pytest.mark.parametrize("method", ["create_partitions", "delete_partitions"]) + async def test_partition_management_rejects_invalid_identifier( + self, iggy_client: IggyClient, unique_name, method: str + ): + _, topic_name = await _create_topic(iggy_client, unique_name) + + with pytest.raises(ValueError): + await getattr(iggy_client, method)("", topic_name, 1) + + @pytest.mark.asyncio + @pytest.mark.parametrize("method", ["create_partitions", "delete_partitions"]) + @pytest.mark.parametrize("partitions_count", [-1, 2**32]) + async def test_partition_management_rejects_out_of_range_python_integer( + self, + iggy_client: IggyClient, + unique_name, + method: str, + partitions_count: int, + ): + stream_name, topic_name = await _create_topic(iggy_client, unique_name) + + with pytest.raises(OverflowError): + await getattr(iggy_client, method)( + stream_name, topic_name, partitions_count + ) + + @pytest.mark.asyncio + async def test_delete_partitions_accepts_deleting_all_partitions( + self, iggy_client: IggyClient, unique_name + ): + stream_name, topic_name = await _create_topic(iggy_client, unique_name) + + await iggy_client.delete_partitions(stream_name, topic_name, 2) + topic = await iggy_client.get_topic(stream_name, topic_name) + assert topic is not None + assert topic.partitions_count == 0 + assert topic.partitions == [] + + @pytest.mark.asyncio + async def test_partition_management_requires_scoped_manage_topic( + self, iggy_client: IggyClient, unique_name + ): + stream_name, topic_name = await _create_topic(iggy_client, unique_name) + other_topic_name = unique_name() + await iggy_client.create_topic( + stream_name, other_topic_name, partitions_count=2 + ) + stream = await iggy_client.get_stream(stream_name) + assert stream is not None + topic = await iggy_client.get_topic(stream.id, topic_name) + other_topic = await iggy_client.get_topic(stream.id, other_topic_name) + assert topic is not None + assert other_topic is not None + + denied_username, denied_password = unique_credentials(unique_name) + denied_user = await iggy_client.create_user(denied_username, denied_password) + denied = await login_fresh_client(denied_username, denied_password) + for method in ("create_partitions", "delete_partitions"): + with pytest.raises(RuntimeError, match="Unauthorized"): + await getattr(denied, method)(stream.id, topic.id, 1) + unchanged = await iggy_client.get_topic(stream.id, topic.id) + assert unchanged is not None + assert unchanged.partitions_count == 2 + + allowed_username, allowed_password = unique_credentials(unique_name) + allowed_user = await iggy_client.create_user( + allowed_username, + allowed_password, + permissions=Permissions( + streams={ + stream.id: StreamPermissions( + topics={topic.id: TopicPermissions(manage_topic=True)} + ) + } + ), + ) + allowed = await login_fresh_client(allowed_username, allowed_password) + await allowed.create_partitions(stream.id, topic.id, 1) + await allowed.delete_partitions(stream.id, topic.id, 1) + for method in ("create_partitions", "delete_partitions"): + with pytest.raises(RuntimeError, match="Unauthorized"): + await getattr(allowed, method)(stream.id, other_topic.id, 1) + scoped = await iggy_client.get_topic(stream.id, topic.id) + untouched = await iggy_client.get_topic(stream.id, other_topic.id) + assert scoped is not None and scoped.partitions_count == 2 + assert untouched is not None and untouched.partitions_count == 2 + + await iggy_client.delete_user(denied_user.id) + await iggy_client.delete_user(allowed_user.id) From 3f0646fcede10781d3614e142e3ac8ceaa1394e7 Mon Sep 17 00:00:00 2001 From: haubur Date: Thu, 10 Sep 2026 13:30:35 +0200 Subject: [PATCH 098/182] chore(ci): add team-review-slim, cheaper team-review and plain english (#4095) Closes #4094 --- .claude/skills/team-review-slim/SKILL.md | 230 +++++++++++++++++++++++ .claude/skills/team-review/SKILL.md | 184 ++++++++++++++---- 2 files changed, 380 insertions(+), 34 deletions(-) create mode 100644 .claude/skills/team-review-slim/SKILL.md diff --git a/.claude/skills/team-review-slim/SKILL.md b/.claude/skills/team-review-slim/SKILL.md new file mode 100644 index 0000000000..d2a6ce0bc0 --- /dev/null +++ b/.claude/skills/team-review-slim/SKILL.md @@ -0,0 +1,230 @@ +--- +name: team-review-slim +description: | + Team-review without the verification chain. Reviews a PR, the current branch against master, or of a ref range. +argument-hint: "[PR number | branch | ref range]" +disable-model-invocation: true +--- + +# Apache Iggy Team Review (slim version) + +`` = `$ARGUMENTS`: a PR number, a branch, or a ref range. Empty means `origin/master..HEAD`. Mission critical +code. + +You = **moderator**. You never open the diff or a source file: you route paths, merge claims, synthesize. Every token +you load rides along every later turn. Reviewers are one-shot agents that deliver by writing a file, they do not chat. + +## Charter (paste VERBATIM into every expert prompt) + +> You think deep. You write plain. Keep the two apart. +> +> **Thinking, unchanged.** Read the diff, then every changed file in full from the local checkout, then whatever call +> sites you need. Trace call chains. Verify invariants. Prove findings, don't guess. Cite exact `file:line`. Running +> tests or builds needs a stated justification: reading and tracing settles most claims, and parallel cargo runs block +> on one target-dir lock. +> +> **Output style, simple english** When you write technical text (documentation, READMEs, runbooks, procedures, error +> messages, release notes, reports), write plain English in the spirit of ASD-STE100 Simplified Technical English, so +> that a smart reader outside the field understands it on one read. Obey these rules: +> +> CLASSIFY FIRST. Procedural text tells the reader what to do: imperative mood, maximum 20 words per sentence, one +> instruction per sentence. Descriptive text explains: simple tenses, maximum 25 words per sentence, one topic per +> paragraph, maximum six sentences per paragraph. Never mix the two in one passage. +> +> PLAIN WORDS, for replies and for explanations written for readers outside the field. Use the common word when one +> exists ("use", not "utilize"). Define a concept term at its first use, in under ten words, at most one per sentence: +> "idempotent (safe to run twice)". Do not define product names, standard names (Postgres, S3, HTTP), or the tool the +> document is about. Address the reader as "you". Lead with the point. Procedures and reference documents follow the +> rules above alone. +> +> VERBS. Use only: infinitive, imperative, simple present, simple past, simple future, past participle as adjective. No +> present perfect ("has completed" -> "completed"). No "-ing" verb forms ("making it easy" -> new sentence). Active +> voice; passive only in descriptions when the agent is unknown. Approved modals: can, will, must. Banned: should, +> would, may, might, could. For "should": write "must" if required, delete if optional. +> +> SENTENCES. Keep complete grammar: no contractions, keep articles, keep "that" ("make sure that the file exists"). Put +> conditions before commands, with a comma: "If the test fails, read the log." No semicolons: write two sentences. No +> em-dashes: an em-dash hides the logic between two statements. Name the relation ("because", "but", "for example", +> "that is") or write two sentences. Use a vertical list for more than two items or steps. +> +> WORDS. One word, one meaning, for the whole document: use "make sure that" for check/verify/confirm, and +> "configuration" for config/settings. Noun chains of maximum three words. Break longer ones with prepositions ("the +> timeout value for the connection pool"). Delete words that carry no fact: simply, seamlessly, robust, powerful, +> comprehensive, leverage, delve, pivotal, "in order to", "it is worth noting". Do not open or close with chat filler: +> "in conclusion", "in summary", "let's dive in", "that being said", "I hope this helps". +> +> AVOID THE AI DRIFTS. Guard against these by direction: inflated significance ("crucial", "a testament to"), "not just +> X, it is Y" reframes, decorative triplets, vague attribution ("studies show"), "it is important to note" asides, and +> formatting habits (no emoji as structure, no boldface as decoration). State the fact. The fact carries itself. +> Replace: utilize -> use, prior to -> before, in the event that -> if, e.g. -> for example. American spelling. +> +> WARNINGS. Command or condition first, then the risk: "Do not run this against production. The command deletes rows." +> +> NEVER TOUCH. Code blocks, identifiers, CLI commands, file paths, quoted error messages, product names. Each counts as +> one word toward sentence limits. Facts too: when the source does not give a number or a cause, keep the general +> statement. Do not invent specifics. +> +> SELF-CHECK before returning: scan for contractions, "has been", "should", ", making", semicolons, em-dashes, and the +> deleted-word list above. Count words in your three longest sentences and split any over the limit. Collapse synonym +> rotation. +> +> REPLIES TO THE USER. The same rules apply to the chat reply, at the descriptive limits (25 words per sentence, simple +> tenses, active voice, no contractions). Start with the answer or the result. If a concept term is necessary, define it +> in a few words. Do not restate the request. Keep the whole reply to 5 sentences or fewer, code and lists excluded. Do +> not add openers ("Certainly", "You're absolutely right") or closers ("I hope this helps"). Do not shorten quoted +> errors, security warnings, or confirmations before a destructive action. +> +> Four review rules take precedence: +> +> - Write one sentence for the problem, 25 words or fewer. Write one sentence for the fix, imperative, 20 words or +> fewer. +> - Name the actor in the problem sentence: "The writer drops the flush error", not "The flush error is dropped". +> - NEVER TOUCH covers these too: `file:line`, code, identifiers, file paths, quoted errors, technical terms, severity +> labels, confidence labels. Keep them EXACT. +> - These writing rules govern your own output only. Never review the code, the comments, or the commit messages against +> them. +> +> **Finding format.** One entry per finding: +> +> ```text +> [sev] file:line - problem. Fix: action. (origin, conf:H|M|L) +> Evidence: the traced path or the line that proves the finding. +> ``` +> +> The `Evidence:` line is required for every `critical` only. Not for `warning`, `nit` and `simplify`. +> +> - `sev`: `critical` = correctness, safety, data loss, or security, and it blocks the merge. `warning` = a real defect, +> a performance hit, or an API problem. `nit` = style or naming. `simplify` = less complexity or dead code, in the +> format +> `[simplify] file:line - the current shape. Simpler: alternative. Saves: about N lines, or removes indirection. (origin, conf)`. +> - `origin`: `intro` (the change introduced it), `pre-surfaced` (it existed, and the change exposed it), +> `pre-untouched` (it existed, and the change did not touch it). +> - `conf`: `H` = a traced call path proves it. `M` = a strong reading, but one gap remains. `L` = a suspicion, and the +> reader must check it. +> - Never flag em dashes or other punctuation style in the reviewed code as a finding. +> - Simplification mandate: less code beats more code. For each changed file, ask whether a 30% smaller file keeps the +> same behavior. Look for dead fields, dead parameters, dead branches, dead imports, duplication of an existing helper +> (cite the helper), single-implementation traits, premature generics, and checks for impossible states. Do not +> propose a simplification that changes the semantics or that breaks the public API. If nothing qualifies, write +> `Simplifications: none`. +> +> **Self-verification, before you write the file.** Run this pass: +> +> 1. Re-open every cited `file:line`. Make sure that the anchor still names the code that you describe. +> 2. Trace one reachable call path for each `critical` and `warning`. Record that path for `critical` in the `Evidence:` +> line. +> 3. Delete every finding that you cannot prove from the code that you read. An unreachable concern is not a finding. +> 4. Set the confidence label from the evidence that you hold, not from how much the defect worries you. +> 5. Ask once whether the severity is calibrated. A cold-path clone is never `critical`. +> +> Deep analysis, plain words. Dig deep. Write short and clear. + +## Step 1: Identify the target (no reading) + +- Classify ``: matches `^#?(pr)?[0-9]+$` case-insensitively -> PR, the digits are ``. Anything else -> ref + range or bare branch. Empty -> review `origin/master..HEAD`. +- ``: `` lowercased, chars outside `[a-z0-9-]` replaced by `-`, repeats collapsed, trimmed, max 40 chars + (`PR3123` -> `pr3123`, `origin/master..HEAD` -> `origin-master-head`). Empty -> `date +%s`. +- `

` = `/review-`. `mkdir -p` it. +- PR: `gh pr view --json title,body,headRefOid > /pr.json`, `gh pr diff > /diff.patch`, + `gh pr diff --name-only > /files.txt`. `` = first 8 of `headRefOid`. +- Ref range or bare branch: `git diff $(git merge-base origin/master HEAD)..HEAD > /diff.patch`, same with + `--name-only`, `` = `git rev-parse --short=8 HEAD`. No `pr.json` on this path. +- Guard: `git rev-parse HEAD` must equal the reviewed head. Experts read the local checkout; if it differs, stop and ask + the user to check out the reviewed head. +- ``: 1-3 word `snake_case` summary, `[a-z0-9_]`, \<= 24 chars. From the PR title; no PR -> from + `git log -1 --format=%s`. +- Report path: `/report.md`. + +Do not `cat` any of the files you just wrote. `wc -l /diff.patch` is the only look you take. + +## Step 2: Four one-shot experts (one message, parallel) + +Spawn 4 `Agent` calls in a single message: `subagent_type: general-purpose`, `name: -` (bare role names +collide with concurrent sessions: one shared agent namespace), no `model` (inherits). Prompt = role block + Charter + +this brief, with ``, ``, `` filled in: + +> Target: `` at ``. Diff: `/diff.patch`. Changed files: `/files.txt`. PR title and body: +> `/pr.json` (drop this sentence when there is no PR). Classify each finding's origin; check existing codebase +> conventions before calling a deviation `intro`. Deliverable = the file `/.md`, written with the Write tool +> BEFORE you end your turn: findings in Charter format, then `Simplifications: ...`, then +> `Verdict: APPROVE | REQUEST CHANGES - reason`. A previous worker finished reading and then idled without delivering; +> the Write call IS the delivery, your final message is just the path. Budget 3/4 reading, 1/4 writing; partial beats +> unshipped. Run the self-verification pass of the Charter before you write. No validator follows you, and an unproven +> `critical` finding costs the user. You work alone: no teammates, no SendMessage, no questions back. + +Role blocks: + +- **storage**: Senior storage/DB engineer, 15 years of WAL, B-trees, LSM, crash recovery, fsync semantics. Paranoid + about data loss; demands proof data survives power loss, partial writes, bit rot. Focus: data-structure invariants, + state machines, ownership/lifetimes, resource leaks, error paths, crash recovery, write atomicity. Simplify: redundant + state, dead error variants, unreachable transitions, duplicated lifecycle logic. +- **perf**: Performance engineer / kernel dev. Flamegraphs, cache lines, io_uring, allocators. Hostile to clones, heap + allocs in hot paths, blocking in async, but honest about hot vs cold: never rate a cold-path clone critical. Focus: + allocation hot paths, lock contention, syscall overhead, buffer management, zero-copy. Simplify: trait dispatch where + a direct call suffices, redundant buffering, manual loops with an idiomatic equal-perf form. +- **distsys**: Distributed-systems architect, formal methods. TLA+, linearizability, "message arrives twice / out of + order / never". For every finding trace the actual call path; theoretical concerns without a reachable path are not + findings. Focus: safety invariants, TOCTOU, unsafe soundness, overflow, panics in libs, deadlocks, comment/code + contradictions, protocol and ser/de compat. Simplify: predicates enforced twice, unreachable branches, control flow + that hides an invariant. +- **ecosystem**: SDK and API ecosystem lead across the client languages. Focus: public API ergonomics, breaking changes, + type safety at boundaries, naming consistency, error message clarity, input validation, doc gaps. Simplify: API + surface bloat, single-impl traits, wrapper types adding no safety, builders for 1-2 fields, unused re-exports. + +Collect: wait for the completion notifications, then `ls /*.md`. A role with no file gets one `SendMessage` nudge +to `-` ("Write `/.md` now, then stop."); still missing after that, respawn the role once with +the same prompt. Never open a subagent transcript via `TaskOutput` (it is the whole JSONL). + +## Step 3: Merge the role files (moderator, no agents) + +Read the 4 role files. Merge them straight into the report sections of Step 4, and write no intermediate file. + +- The same anchor and the same defect from several roles = one entry, at the highest severity of the group. Record the + roles that raised it, and keep the strongest `Evidence:` line of the group. +- When two roles disagree on the severity of one anchor, keep the higher severity and add + `(disputed: rates it )`. You do not adjudicate, and the user decides at the cited line. +- Keep the confidence label of every entry. It is the reader's map for the manual verification. +- Simplify items are entries too, and they go into their own section. +- Rewrite nothing. Keep the plain English of the experts, and fix a sentence only when it breaks a Charter rule. + +No findings at all: go to Step 4 with empty sections and `Verdict: APPROVE`. The report file still gets written. + +## Step 4: Write the report, then stop + +Output: + +```text +## Review: + +### Findings +- [sev] file:line - problem. Fix: action. (raised: role[, role]; conf:H|M|L) + Evidence: the traced path or the line that proves the finding. + +### Unconfirmed (conf:L, check these first) +- [sev] file:line - problem. Open question: what the reader must check. + +### Pre-existing (origin pre-*, does not block the merge) +- file:line - problem. The code follows the pattern in . + +### Simplification opportunities (does not block the merge) +- file:line - the current shape. Simpler: alternative. Saves: about N lines, or removes indirection. + +### Verdict: APPROVE | REQUEST CHANGES +Reason: one sentence. Only `critical` and `warning` entries with conf:H or conf:M decide the verdict. +Simplifications are informational. + +Counts: critical N, warning N, nit N, simplify N +Verification: none ran. Open each cited line and confirm the finding before you change the code. +``` + +Then write `/report.md` with: + +1. H1 `# Iggy Team Review (small) - ()`. +2. Metadata, one line each: target ``, reviewed commit, ISO timestamp, roles, expert count, + `validation: none (manual)`. +3. The report above, verbatim. +4. Appendix `## Raw findings per expert`: each role file verbatim, in a fenced block. + +Last user-facing line: `Findings written: /report.md`. Then name the highest-severity entry in one sentence, and +stop. There is no cleanup: the one-shot agents end themselves, and `` stays in the scratchpad. diff --git a/.claude/skills/team-review/SKILL.md b/.claude/skills/team-review/SKILL.md index 5d43f6117a..4d4605ace7 100644 --- a/.claude/skills/team-review/SKILL.md +++ b/.claude/skills/team-review/SKILL.md @@ -1,85 +1,200 @@ --- name: team-review -description: Adversarial 4-expert review (storage, perf, distsys, ecosystem) of a PR, branch, or ref range, with clean-room validation of every finding. Experts work alone, no peer debate. Expensive, one run spawns ~10 subagents. +description: | + Adversarial 4-expert review (storage, perf, distsys, ecosystem) of a PR, branch, or ref range, with clean-room + validation of every finding. Experts work alone, no peer debate. Expensive, one run spawns ~10 subagents. argument-hint: "[PR number | branch | ref range]" disable-model-invocation: true --- # Apache Iggy Team Review -`` = `$ARGUMENTS`: a PR number, a branch, or a ref range. Empty means `origin/master..HEAD`. Mission critical code. +`` = `$ARGUMENTS`: a PR number, a branch, or a ref range. Empty means `origin/master..HEAD`. Mission critical +code. -You = **moderator**. You never open the diff or a source file: you route paths, merge claims, synthesize. Every token you load rides along every later turn. Reviewers and validators are one-shot agents that deliver by writing a file; nobody chats. +You = **moderator**. You never open the diff or a source file: you route paths, merge claims, synthesize. Every token +you load rides along every later turn. Reviewers and validators are one-shot agents that deliver by writing a file; +nobody chats. ## Charter (paste VERBATIM into every expert, validator, and tiebreak prompt) -> You think big brain. You speak caveman. Separate things. +> You think deep. You write plain. Keep the two apart. > -> **Thinking, unchanged.** Read the diff, then every changed file in full from the local checkout, then whatever call sites you need. Trace call chains. Verify invariants. Prove findings, don't guess. Cite exact `file:line`. Running tests or builds needs a stated justification: reading and tracing settles most claims, and parallel cargo runs block on one target-dir lock. +> **Thinking, unchanged.** Read the diff, then every changed file in full from the local checkout, then whatever call +> sites you need. Trace call chains. Verify invariants. Prove findings, do not guess. Cite exact `file:line`. Running +> tests or builds needs a stated justification: reading and tracing settles most claims, and parallel cargo runs block +> on one target-dir lock. > -> **Output style.** Drop articles, filler, pleasantries, hedging. Fragments OK. Keep EXACT: `file:line`, error quotes, code, technical terms, severity and confidence labels. +> **Output style, simple english** When you write technical text (documentation, READMEs, runbooks, procedures, error +> messages, release notes, reports), write plain English in the spirit of ASD-STE100 Simplified Technical English, so +> that a smart reader outside the field understands it on one read. Obey these rules: +> +> CLASSIFY FIRST. Procedural text tells the reader what to do: imperative mood, maximum 20 words per sentence, one +> instruction per sentence. Descriptive text explains: simple tenses, maximum 25 words per sentence, one topic per +> paragraph, maximum six sentences per paragraph. Never mix the two in one passage. +> +> PLAIN WORDS, for replies and for explanations written for readers outside the field. Use the common word when one +> exists ("use", not "utilize"). Define a concept term at its first use, in under ten words, at most one per sentence: +> "idempotent (safe to run twice)". Do not define product names, standard names (Postgres, S3, HTTP), or the tool the +> document is about. Address the reader as "you". Lead with the point. Procedures and reference documents follow the +> rules above alone. +> +> VERBS. Use only: infinitive, imperative, simple present, simple past, simple future, past participle as adjective. No +> present perfect ("has completed" -> "completed"). No "-ing" verb forms ("making it easy" -> new sentence). Active +> voice; passive only in descriptions when the agent is unknown. Approved modals: can, will, must. Banned: should, +> would, may, might, could. For "should": write "must" if required, delete if optional. +> +> SENTENCES. Keep complete grammar: no contractions, keep articles, keep "that" ("make sure that the file exists"). Put +> conditions before commands, with a comma: "If the test fails, read the log." No semicolons: write two sentences. No +> em-dashes: an em-dash hides the logic between two statements. Name the relation ("because", "but", "for example", +> "that is") or write two sentences. Use a vertical list for more than two items or steps. +> +> WORDS. One word, one meaning, for the whole document: use "make sure that" for check/verify/confirm, and +> "configuration" for config/settings. Noun chains of maximum three words. Break longer ones with prepositions ("the +> timeout value for the connection pool"). Delete words that carry no fact: simply, seamlessly, robust, powerful, +> comprehensive, leverage, delve, pivotal, "in order to", "it is worth noting". Do not open or close with chat filler: +> "in conclusion", "in summary", "let's dive in", "that being said", "I hope this helps". +> +> AVOID THE AI DRIFTS. Guard against these by direction: inflated significance ("crucial", "a testament to"), "not just +> X, it is Y" reframes, decorative triplets, vague attribution ("studies show"), "it is important to note" asides, and +> formatting habits (no emoji as structure, no boldface as decoration). State the fact. The fact carries itself. +> Replace: utilize -> use, prior to -> before, in the event that -> if, e.g. -> for example. American spelling. +> +> WARNINGS. Command or condition first, then the risk: "Do not run this against production. The command deletes rows." +> +> NEVER TOUCH. Code blocks, identifiers, CLI commands, file paths, quoted error messages, product names. Each counts as +> one word toward sentence limits. Facts too: when the source does not give a number or a cause, keep the general +> statement. Do not invent specifics. +> +> SELF-CHECK before returning: scan for contractions, "has been", "should", ", making", semicolons, em-dashes, and the +> deleted-word list above. Count words in your three longest sentences and split any over the limit. Collapse synonym +> rotation. +> +> REPLIES TO THE USER. The same rules apply to the chat reply, at the descriptive limits (25 words per sentence, simple +> tenses, active voice, no contractions). Start with the answer or the result. If a concept term is necessary, define it +> in a few words. Do not restate the request. Keep the whole reply to 5 sentences or fewer, code and lists excluded. Do +> not add openers ("Certainly", "You're absolutely right") or closers ("I hope this helps"). Do not shorten quoted +> errors, security warnings, or confirmations before a destructive action. +> +> Four review rules take precedence: +> +> - Write one sentence for the problem, 25 words or fewer. Write one sentence for the fix, imperative, 20 words or +> fewer. +> - Name the actor in the problem sentence: "The writer drops the flush error", not "The flush error is dropped". +> - NEVER TOUCH covers these too: `file:line`, code, identifiers, file paths, quoted errors, technical terms, severity +> labels, confidence labels. Keep them EXACT. +> - These writing rules govern your own output only. Never review the code, the comments, or the commit messages against +> them. +> +> **Finding format.** > > - Finding, one line each: `[sev] file:line - problem. Fix: action. (origin, conf:H|M|L)` -> - `sev`: `critical` = correctness/safety/data-loss/security, blocks merge; `warning` = real defect, perf hit, API issue; `nit` = style/naming; `simplify` = complexity/dead-code reduction, format `[simplify] file:line - what's complex. Simpler: alternative. Saves: ~N lines / removes indirection. (origin, conf)`. +> - `sev`: `critical` = correctness/safety/data-loss/security, blocks merge; `warning` = real defect, perf hit, API +> issue; `nit` = style/naming; `simplify` = complexity/dead-code reduction, format +> `[simplify] file:line - what's complex. Simpler: alternative. Saves: ~N lines / removes indirection. (origin, conf)`. > - `origin`: `intro` (PR introduced), `pre-surfaced` (existed, exposed by PR), `pre-untouched` (existed, not touched). > - Never flag em dashes or other punctuation style as a finding. -> - Simplification mandate: less code > more code. Per changed file ask whether ~30% smaller keeps correctness: dead fields/params/branches/imports, duplication of an existing helper (cite it), single-impl traits, premature generics, checks for impossible states. Do not propose simplifications that change semantics or break public API. If nothing qualifies, write `Simplifications: none`. +> - Simplification mandate: less code > more code. Per changed file ask whether ~30% smaller keeps correctness: dead +> fields/params/branches/imports, duplication of an existing helper (cite it), single-impl traits, premature generics, +> checks for impossible states. Do not propose simplifications that change semantics or break public API. If nothing +> qualifies, write `Simplifications: none`. > -> Caveman = output compression, not analysis compression. Dig deep. Write short. +> Deep analysis, plain words. Dig deep. Write short and clear. ## Step 1: Identify the target (no reading) -- Classify ``: matches `^#?(pr)?[0-9]+$` case-insensitively -> PR, the digits are ``. Anything else -> ref range or bare branch. Empty -> review `origin/master..HEAD`. -- ``: `` lowercased, chars outside `[a-z0-9-]` replaced by `-`, repeats collapsed, trimmed, max 40 chars (`PR3123` -> `pr3123`, `origin/master..HEAD` -> `origin-master-head`). Empty -> `date +%s`. +- Classify ``: matches `^#?(pr)?[0-9]+$` case-insensitively -> PR, the digits are ``. Anything else -> ref + range or bare branch. Empty -> review `origin/master..HEAD`. +- ``: `` lowercased, chars outside `[a-z0-9-]` replaced by `-`, repeats collapsed, trimmed, max 40 chars + (`PR3123` -> `pr3123`, `origin/master..HEAD` -> `origin-master-head`). Empty -> `date +%s`. - `` = `/review-`. `mkdir -p` it. -- PR: `gh pr view --json title,body,headRefOid > /pr.json`, `gh pr diff > /diff.patch`, `gh pr diff --name-only > /files.txt`. `` = first 8 of `headRefOid`. -- Ref range or bare branch: `git diff $(git merge-base origin/master HEAD)..HEAD > /diff.patch`, same with `--name-only`, `` = `git rev-parse --short=8 HEAD`. No `pr.json` on this path. -- Guard: `git rev-parse HEAD` must equal the reviewed head. Experts read the local checkout; if it differs, stop and ask the user to check out the reviewed head. -- ``: 1-3 word `snake_case` summary, `[a-z0-9_]`, <= 24 chars. From the PR title; no PR -> from `git log -1 --format=%s`. +- PR: `gh pr view --json title,body,headRefOid > /pr.json`, `gh pr diff > /diff.patch`, + `gh pr diff --name-only > /files.txt`. `` = first 8 of `headRefOid`. +- Ref range or bare branch: `git diff $(git merge-base origin/master HEAD)..HEAD > /diff.patch`, same with + `--name-only`, `` = `git rev-parse --short=8 HEAD`. No `pr.json` on this path. +- Guard: `git rev-parse HEAD` must equal the reviewed head. Experts read the local checkout; if it differs, stop and ask + the user to check out the reviewed head. +- ``: 1-3 word `snake_case` summary, `[a-z0-9_]`, \<= 24 chars. From the PR title; no PR -> from + `git log -1 --format=%s`. - Report path: `/report.md`. Do not `cat` any of the files you just wrote. `wc -l /diff.patch` is the only look you take. ## Step 2: Round 1, four one-shot experts (one message, parallel) -Spawn 4 `Agent` calls in a single message: `subagent_type: general-purpose`, `name: -` (bare role names collide with concurrent sessions: one shared agent namespace), no `model` (inherits). Prompt = role block + Charter + this brief, with ``, ``, `` filled in: +Spawn 4 `Agent` calls in a single message: `subagent_type: general-purpose`, `name: -` (bare role names +collide with concurrent sessions: one shared agent namespace), no `model` (inherits). Prompt = role block + Charter + +this brief, with ``, ``, `` filled in: -> Target: `` at ``. Diff: `/diff.patch`. Changed files: `/files.txt`. PR title and body: `/pr.json` (drop this sentence when there is no PR). Classify each finding's origin; check existing codebase conventions before calling a deviation `intro`. -> Deliverable = the file `/.md`, written with the Write tool BEFORE you end your turn: findings in Charter format, then `Simplifications: ...`, then `Verdict: APPROVE | REQUEST CHANGES - reason`. A previous worker finished reading and then idled without delivering; the Write call IS the delivery, your final message is just the path. Budget 3/4 reading, 1/4 writing; partial beats unshipped. -> You work alone: no teammates, no SendMessage, no questions back. +> Target: `` at ``. Diff: `/diff.patch`. Changed files: `/files.txt`. PR title and body: +> `/pr.json` (drop this sentence when there is no PR). Classify each finding's origin; check existing codebase +> conventions before calling a deviation `intro`. Deliverable = the file `/.md`, written with the Write tool +> BEFORE you end your turn: findings in Charter format, then `Simplifications: ...`, then +> `Verdict: APPROVE | REQUEST CHANGES - reason`. A previous worker finished reading and then idled without delivering; +> the Write call IS the delivery, your final message is just the path. Budget 3/4 reading, 1/4 writing; partial beats +> unshipped. You work alone: no teammates, no SendMessage, no questions back. Role blocks: -- **storage**: Senior storage/DB engineer, 15 years of WAL, B-trees, LSM, crash recovery, fsync semantics. Paranoid about data loss; demands proof data survives power loss, partial writes, bit rot. Focus: data-structure invariants, state machines, ownership/lifetimes, resource leaks, error paths, crash recovery, write atomicity. Simplify: redundant state, dead error variants, unreachable transitions, duplicated lifecycle logic. -- **perf**: Performance engineer / kernel dev. Flamegraphs, cache lines, io_uring, allocators. Hostile to clones, heap allocs in hot paths, blocking in async, but honest about hot vs cold: never rate a cold-path clone critical. Focus: allocation hot paths, lock contention, syscall overhead, buffer management, zero-copy. Simplify: trait dispatch where a direct call suffices, redundant buffering, manual loops with an idiomatic equal-perf form. -- **distsys**: Distributed-systems architect, formal methods. TLA+, linearizability, "message arrives twice / out of order / never". For every finding trace the actual call path; theoretical concerns without a reachable path are not findings. Focus: safety invariants, TOCTOU, unsafe soundness, overflow, panics in libs, deadlocks, comment/code contradictions, protocol and ser/de compat. Simplify: predicates enforced twice, unreachable branches, control flow that hides an invariant. -- **ecosystem**: SDK and API ecosystem lead across the client languages. Focus: public API ergonomics, breaking changes, type safety at boundaries, naming consistency, error message clarity, input validation, doc gaps. Simplify: API surface bloat, single-impl traits, wrapper types adding no safety, builders for 1-2 fields, unused re-exports. - -Collect: wait for the completion notifications, then `ls /*.md`. A role with no file gets one `SendMessage` nudge to `-` ("Write `/.md` now, then stop."); still missing after that, respawn the role once with the same prompt. Never open a subagent transcript via `TaskOutput` (it is the whole JSONL). +- **storage**: Senior storage/DB engineer, 15 years of WAL, B-trees, LSM, crash recovery, fsync semantics. Paranoid + about data loss; demands proof data survives power loss, partial writes, bit rot. Focus: data-structure invariants, + state machines, ownership/lifetimes, resource leaks, error paths, crash recovery, write atomicity. Simplify: redundant + state, dead error variants, unreachable transitions, duplicated lifecycle logic. +- **perf**: Performance engineer / kernel dev. Flamegraphs, cache lines, io_uring, allocators. Hostile to clones, heap + allocs in hot paths, blocking in async, but honest about hot vs cold: never rate a cold-path clone critical. Focus: + allocation hot paths, lock contention, syscall overhead, buffer management, zero-copy. Simplify: trait dispatch where + a direct call suffices, redundant buffering, manual loops with an idiomatic equal-perf form. +- **distsys**: Distributed-systems architect, formal methods. TLA+, linearizability, "message arrives twice / out of + order / never". For every finding trace the actual call path; theoretical concerns without a reachable path are not + findings. Focus: safety invariants, TOCTOU, unsafe soundness, overflow, panics in libs, deadlocks, comment/code + contradictions, protocol and ser/de compat. Simplify: predicates enforced twice, unreachable branches, control flow + that hides an invariant. +- **ecosystem**: SDK and API ecosystem lead across the client languages. Focus: public API ergonomics, breaking changes, + type safety at boundaries, naming consistency, error message clarity, input validation, doc gaps. Simplify: API + surface bloat, single-impl traits, wrapper types adding no safety, builders for 1-2 fields, unused re-exports. + +Collect: wait for the completion notifications, then `ls /*.md`. A role with no file gets one `SendMessage` nudge +to `-` ("Write `/.md` now, then stop."); still missing after that, respawn the role once with +the same prompt. Never open a subagent transcript via `TaskOutput` (it is the whole JSONL). ## Step 3: Merge into neutral claims (moderator) -Read the 4 role files. Write `/claims.md`, one line per claim: `C [sev] file:line - claim. Fix: action. (origin)`. Strip role names, confidence, and argument. Same anchor + same defect from several roles = one claim at the highest severity; keep a private raised-by map for the report. Simplify items are claims too. +Read the 4 role files. Write `/claims.md`, one line per claim: +`C [sev] file:line - claim. Fix: action. (origin)`. Strip role names, confidence, and argument. Same anchor + same +defect from several roles = one claim at the highest severity; keep a private raised-by map for the report. Simplify +items are claims too. -No claims at all: skip Steps 4 and 5, go to Step 6 with empty sections and `Verdict: APPROVE`. The report file still gets written. +No claims at all: skip Steps 4 and 5, go to Step 6 with empty sections and `Verdict: APPROVE`. The report file still +gets written. ## Step 4: Clean-room validation (one message, parallel) -Shard claims ~5 per validator. Spawn one `Agent` per shard plus one sweep validator, all in one message: `subagent_type: general-purpose`, `model: opus`, `name: validator--` / `sweep-`. Each gets ONLY: its claims verbatim, `/files.txt`, `/diff.patch`, the target identity, the Charter. Not the role files, not raised-by, not your reasoning; the missing context is what removes the anchoring bias. +Shard claims ~5 per validator. Spawn one `Agent` per shard plus one sweep validator, all in one message: +`subagent_type: general-purpose`, `model: opus`, `name: validator--` / `sweep-`. Each gets ONLY: its +claims verbatim, `/files.txt`, `/diff.patch`, the target identity, the Charter. Not the role files, not +raised-by, not your reasoning; the missing context is what removes the anchoring bias. -Validator mandate (adversarial): for each claim open the cited `file:line`, trace call sites, then rate `C: PASS | FIX: | REMOVE: `; judge whether the severity is calibrated; re-check the anchor. Deliverable `/validate-.md` via Write, same idle rule as Step 2. +Validator mandate (adversarial): for each claim open the cited `file:line`, trace call sites, then rate +`C: PASS | FIX: | REMOVE: `; judge whether +the severity is calibrated; re-check the anchor. Deliverable `/validate-.md` via Write, same idle rule as Step +2\. -Sweep mandate: all claims + the diff. Two questions only: which real defects in the diff are missing from the list, and which listed items wrongly clear a bug. Deliverable `/sweep.md`, additions in Charter format tagged `(sweep)`. +Sweep mandate: all claims + the diff. Two questions only: which real defects in the diff are missing from the list, and +which listed items wrongly clear a bug. Deliverable `/sweep.md`, additions in Charter format tagged `(sweep)`. -Apply: drop REMOVE, apply FIX (wording, line, severity), fold sweep additions in as `(sweep, unvalidated)`. A `critical` sweep addition gets one extra validator before it may block the verdict. +Apply: drop REMOVE, apply FIX (wording, line, severity), fold sweep additions in as `(sweep, unvalidated)`. A `critical` +sweep addition gets one extra validator before it may block the verdict. ## Step 5: Contested items (only when triggered) -Contested = a validator REMOVEs or downgrades a `critical` or `warning`, or a sweep addition contradicts a PASS. Per item spawn one `Agent` (`model: opus`) with the claim, the validator's verdict text, the expert's original line, and the paths; it writes `UPHELD | OVERTURNED - reason (cite path)` to `/contested-.md`. Cap 5 per run; past the cap you adjudicate and mark `(moderator call)`. +Contested = a validator REMOVEs or downgrades a `critical` or `warning`, or a sweep addition contradicts a PASS. Per +item spawn one `Agent` (`model: opus`) with the claim, the validator's verdict text, the expert's original line, and the +paths; it writes `UPHELD | OVERTURNED - reason (cite path)` to `/contested-.md`. Cap 5 per run; past the cap you +adjudicate and mark `(moderator call)`. ## Step 6: Synthesize, write, done -Output in caveman style: +Output in the simple English of the Charter. The Charter binds you too: ```text ## Review: [change desc] @@ -114,4 +229,5 @@ Then write `/report.md` with: 4. Appendix `## Raw findings per expert`: each role file verbatim in a fenced block. 5. `## Validation record`: counts of PASS / FIX / REMOVE, sweep additions, contested outcomes. -Last user-facing line: `Findings written: /report.md`. No cleanup: one-shot agents end themselves, `` stays in the scratchpad. +Last user-facing line: `Findings written: /report.md`. No cleanup: one-shot agents end themselves, `` stays in +the scratchpad. From e9f3362e3d79f1becc0cc4232de7098c270d167d Mon Sep 17 00:00:00 2001 From: Ryan Huang Date: Fri, 11 Sep 2026 00:36:41 +0800 Subject: [PATCH 099/182] feat(connectors)!: add shared retry_async helper to the connector SDK (#4104) Relates to #3702 #4084 --- .claude/skills/connector-sdk/SKILL.md | 6 +- .claude/skills/connector-sink/SKILL.md | 3 +- Cargo.lock | 1 + core/connectors/sdk/Cargo.toml | 5 + core/connectors/sdk/README.md | 13 + core/connectors/sdk/src/retry.rs | 589 +++++++++++++++--- core/connectors/sinks/doris_sink/src/lib.rs | 59 +- core/connectors/sinks/influxdb_sink/README.md | 6 +- .../connectors/sinks/influxdb_sink/src/lib.rs | 11 +- .../sinks/meilisearch_sink/src/lib.rs | 42 +- .../connectors/sinks/quickwit_sink/src/lib.rs | 10 +- .../connectors/sinks/rabbitmq_sink/src/lib.rs | 14 +- core/connectors/sinks/s3_sink/src/sink.rs | 6 +- .../sinks/surrealdb_sink/src/lib.rs | 8 +- .../sources/influxdb_source/README.md | 4 +- .../sources/influxdb_source/src/lib.rs | 11 +- 16 files changed, 604 insertions(+), 184 deletions(-) diff --git a/.claude/skills/connector-sdk/SKILL.md b/.claude/skills/connector-sdk/SKILL.md index d155b4b541..7d00fd484a 100644 --- a/.claude/skills/connector-sdk/SKILL.md +++ b/.claude/skills/connector-sdk/SKILL.md @@ -47,7 +47,7 @@ sdk/src/ ├── api.rs ConnectorStatus, ConnectorStats (feature = "api"). ├── convert.rs owned_value_to_serde_json (simd_json ⇄ serde_json bridge). ├── log.rs CallbackLayer for tracing across FFI. -├── retry.rs CircuitBreaker, HttpRetryMiddleware, exponential_backoff, jitter. +├── retry.rs retry_async + RetryPolicy, CircuitBreaker, HttpRetryMiddleware. ├── decoders/ One per schema: json, raw, text, proto, flatbuffer, avro. ├── encoders/ Mirror of decoders. └── transforms/ add_fields, delete_fields, update_fields, filter_fields, @@ -149,8 +149,10 @@ Plugin authors call this on every consumed message. The implementation in `lib.r ## Retry helpers (`retry.rs`) - `CircuitBreaker`: threshold + cooldown, `try_lock()` on the success path to avoid hot-path contention. +- `retry_async(policy, context, should_retry, op)`: the retry loop for anything failing as `Err`. Owns attempt counting, backoff and the per-retry log; returns `RetryFailure { error, attempts, exhausted }` so callers write their own terminal log. +- `retry_backoff(base, retry, max)`: backoff only, for a loop that computes its own delay. `retry` is 1-based. `exponential_backoff` is the 0-based primitive underneath and applies no jitter, so call it directly only when the delay must be exact (`source.rs::nack_retry_delay`, whose tests assert exact values). - `HttpRetryMiddleware`: integrates with `reqwest-middleware`. Retries 429 + 5xx + network errors. Honors `Retry-After`. -- `max_retries` = **total attempts** including the first try, not extra retries. Document if you change this convention. +- `max_retries` = **total attempts** including the first try, not extra retries. Document if you change this convention. `meilisearch_sink` is the standing exception: its `max_retries` / `max_open_retries` count retries *after* the first, as its README states. - New helpers must take `Duration` (not `u64 millis`) on the public API. Internal computation uses `humantime` parsing of `String`. ## `ConnectorState` diff --git a/.claude/skills/connector-sink/SKILL.md b/.claude/skills/connector-sink/SKILL.md index 80533036b6..207e553547 100644 --- a/.claude/skills/connector-sink/SKILL.md +++ b/.claude/skills/connector-sink/SKILL.md @@ -82,7 +82,8 @@ for getting them to the external system reliably and efficiently. - SDK helpers cover the simple case: `iggy_connector_sdk::retry::check_connectivity_with_retry(...)` for `open()`, `HttpRetryMiddleware` for default 429/5xx/network policy on reqwest clients. - Custom strategies: `reqwest-middleware` + `reqwest_retry::RetryTransientMiddleware::new_with_policy_and_strategy`. `http_sink` defines its own `HttpSinkRetryStrategy` (honors `success_status_codes`, per-status decisions). - Non-HTTP clients: write `is_transient_error(&e)` mapping driver-specific codes. `postgres_sink::is_transient_error` maps SQLSTATEs `40001`, `40P01`, `57P01-03`, `08000/03/06`. -- Backoff: `iggy_connector_sdk::retry::exponential_backoff(base, attempt, max)` + `jitter()`. +- Retry loop: new connectors use `iggy_connector_sdk::retry::retry_async(policy, context, should_retry, op)` for anything that fails as `Err`. It owns attempt counting, backoff and the per-retry log, and returns `RetryFailure { error, attempts, exhausted }`; the caller logs the terminal failure. +- Backoff only, for a loop that computes its own delay (it retries on an `Ok` response, carries a deadline, or reconnects between attempts): `retry_backoff(base, retry, max)`, where `retry` is 1-based. `exponential_backoff` is the 0-based primitive underneath, and it applies no jitter; call it directly only when the delay must be exact, as `sdk/src/source.rs::nack_retry_delay` needs for its tests. - Cap retries at 3 total attempts. ### Idempotency diff --git a/Cargo.lock b/Cargo.lock index 3afc2f782f..4f18e57409 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -7472,6 +7472,7 @@ dependencies = [ "tracing", "tracing-subscriber", "uuid", + "wiremock", ] [[package]] diff --git a/core/connectors/sdk/Cargo.toml b/core/connectors/sdk/Cargo.toml index 5d6b98daa9..3d511d3f5a 100644 --- a/core/connectors/sdk/Cargo.toml +++ b/core/connectors/sdk/Cargo.toml @@ -68,5 +68,10 @@ tracing = { workspace = true } tracing-subscriber = { workspace = true } uuid = { workspace = true } +[dev-dependencies] +# `test-util` is not part of `full`; the retry tests need its paused time. +tokio = { workspace = true, features = ["full", "test-util"] } +wiremock = { workspace = true } + [lints] workspace = true diff --git a/core/connectors/sdk/README.md b/core/connectors/sdk/README.md index 1c72e143d4..96f8037ad2 100644 --- a/core/connectors/sdk/README.md +++ b/core/connectors/sdk/README.md @@ -48,6 +48,19 @@ key = "message" value.static = "hello" ``` +## Retry helpers + +`retry_async` runs an operation that fails with `Err` and retries it while `should_retry` accepts the error. It owns attempt counting, backoff and the per-retry log, and returns `RetryFailure { error, attempts, exhausted }` so the caller logs the terminal failure. `retry_backoff` computes a single delay for a loop that cannot use `retry_async`, such as `HttpRetryMiddleware`, which retries on an `Ok` response rather than an `Err`. Its `retry` argument is 1-based. + +Two items changed in a way that breaks out-of-tree plugins, so those plugins must be rebuilt against the current source: + +| Removed | Replacement | +| --- | --- | +| `ConnectivityConfig` | `RetryPolicy`. `max_open_retries` becomes `max_attempts`, `retry_delay` becomes `base_delay`, and `open_retry_max_delay` becomes `max_delay`. | +| `jitter` (was public) | `retry_backoff`, which applies the jitter itself. | + +Both types carry `(u32, Duration, Duration)` and the two delay roles cross over, so a field-by-field rename compiles and swaps the base delay for the cap. Map the fields by name. + ## Protocol Buffers Support The SDK includes support for Protocol Buffers (protobuf) format with both encoding and decoding capabilities. Protocol Buffers provide efficient serialization and are particularly useful for high-performance data streaming scenarios. diff --git a/core/connectors/sdk/src/retry.rs b/core/connectors/sdk/src/retry.rs index 06866e8da0..b1d63294d3 100644 --- a/core/connectors/sdk/src/retry.rs +++ b/core/connectors/sdk/src/retry.rs @@ -24,11 +24,13 @@ //! - [`build_retry_client`] — wraps a `reqwest::Client` with the middleware //! - [`check_connectivity`] — single health-check probe (GET /health) //! - [`check_connectivity_with_retry`] — startup probe with exponential backoff -//! - [`ConnectivityConfig`] — parameters for the startup retry loop +//! - [`retry_async`] — generic retry loop for fallible async operations +//! - [`RetryFailure`] — terminal outcome of [`retry_async`], with attempt count +//! - [`RetryPolicy`] — attempt budget and backoff bounds for [`retry_async`] //! - [`is_transient_status`] — transient HTTP status predicate //! - [`parse_duration`] — humantime duration parsing with fallback -//! - [`jitter`] — ±20 % random jitter for retry delays //! - [`exponential_backoff`] — capped exponential backoff +//! - [`retry_backoff`] — jittered backoff for a 1-based retry number //! - [`parse_retry_after`] — HTTP `Retry-After` header parsing use anyhow::anyhow; @@ -36,10 +38,12 @@ use http::Extensions; use humantime::Duration as HumanDuration; use rand::RngExt as _; use reqwest_middleware::{ClientBuilder, ClientWithMiddleware, Middleware, Next}; +use std::fmt; +use std::future::Future; use std::str::FromStr; use std::time::Duration; use tokio::sync::Mutex; -use tracing::{info, warn}; +use tracing::{error, info, warn}; // --------------------------------------------------------------------------- // Circuit breaker @@ -150,7 +154,7 @@ pub fn parse_duration(value: Option<&str>, default_value: &str) -> Duration { } /// Apply ±20 % random jitter to `base` to spread retry storms. -pub fn jitter(base: Duration) -> Duration { +pub(crate) fn jitter(base: Duration) -> Duration { let millis = base.as_millis() as u64; let jitter_range = millis / 5; // 20% of base if jitter_range == 0 { @@ -161,14 +165,17 @@ pub fn jitter(base: Duration) -> Duration { } /// True exponential backoff: `base × 2^attempt`, capped at `max_delay`. +/// +/// `attempt` is 0-based. Retry loops count from 1, so passing their counter +/// here makes the first retry wait twice `base`; use [`retry_backoff`], which +/// takes a 1-based retry number and applies jitter and the cap. pub fn exponential_backoff(base: Duration, attempt: u32, max_delay: Duration) -> Duration { let factor = 2u64.saturating_pow(attempt); let millis = base .as_millis() .saturating_mul(factor as u128) .min(max_delay.as_millis()); - let millis_u64 = u64::try_from(millis).unwrap_or(u64::MAX); - Duration::from_millis(millis_u64) + Duration::from_millis(u64::try_from(millis).unwrap_or(u64::MAX)) } /// Parse a `Retry-After` header value (integer seconds). @@ -181,6 +188,160 @@ pub fn parse_retry_after(value: &str) -> Option { None } +// --------------------------------------------------------------------------- +// Generic retry loop +// --------------------------------------------------------------------------- + +/// Parameters for [`retry_async`]. +/// +/// `max_attempts` is a *total attempt count*, not a count of extra retries: +/// `1` runs the operation once and never retries, `3` allows two retries. `0` +/// behaves as `1`, so a misconfigured value degrades to a single attempt +/// rather than skipping the operation entirely. +#[derive(Debug, Clone, Copy)] +pub struct RetryPolicy { + pub max_attempts: u32, + pub base_delay: Duration, + pub max_delay: Duration, +} + +impl RetryPolicy { + /// Jittered backoff before retry number `retry` (1-based). See + /// [`retry_backoff`]. + pub fn backoff(&self, retry: u32) -> Duration { + retry_backoff(self.base_delay, retry, self.max_delay) + } +} + +/// Jittered, capped backoff before retry number `retry` (1-based): the first +/// retry waits `base_delay`, the second `2 × base_delay`, and so on, which is +/// the convention the `retry_delay` config fields document. +/// +/// The cap is re-applied after jittering, because ±20 % jitter on an +/// already-capped delay can otherwise land above `max_delay`, which the config +/// fields document as a strict upper bound. +/// +/// Prefer [`retry_async`], which calls this for you. Reach for it directly +/// only in a loop that cannot be expressed as a retried `Result`. +pub fn retry_backoff(base_delay: Duration, retry: u32, max_delay: Duration) -> Duration { + jitter(exponential_backoff( + base_delay, + retry.saturating_sub(1), + max_delay, + )) + .min(max_delay) +} + +/// Why [`retry_async`] stopped. +/// +/// Carries the attempt count so a caller can write its own terminal log: the +/// helper owns the per-retry line, giving up is the caller's to report. +#[derive(Debug)] +pub struct RetryFailure { + pub error: E, + /// Attempts actually made, including the first. + pub attempts: u32, + /// `true` when the attempt budget ran out, `false` when `should_retry` + /// rejected the error and no further attempt was made. + pub exhausted: bool, +} + +impl RetryFailure { + /// Discard the attempt bookkeeping and keep the underlying error. + pub fn into_error(self) -> E { + self.error + } +} + +// No `source()`: `Display` already prints the inner error, so returning it +// here repeats the same text in an error chain. +impl std::error::Error for RetryFailure where E: std::error::Error + 'static {} + +impl fmt::Display for RetryFailure { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + let reason = if self.exhausted { + "ran out of attempts" + } else { + "hit a non-retryable error" + }; + let plural = if self.attempts == 1 { + "attempt" + } else { + "attempts" + }; + write!( + f, + "{reason} after {} {plural}: {}", + self.attempts, self.error + ) + } +} + +/// Run `operation`, retrying while it fails with an error `should_retry` +/// accepts and the attempt budget in `policy` is not exhausted. +/// +/// This is the retry skeleton for connectors whose failures surface as `Err`, +/// including backends that report failure in-band (a 200 response carrying a +/// failure status in its body) and so cannot use [`HttpRetryMiddleware`], +/// which classifies on the status code alone. +/// +/// `context` identifies the connector and operation in every log line this +/// emits, e.g. `"Doris sink ID 3 Stream Load (label=abc)"`. Callers build it +/// before the first attempt, so keep it off per-message paths. +/// +/// The per-retry line is logged here. Giving up is not: the returned +/// [`RetryFailure`] carries the attempt count and which condition ended the +/// loop, so a caller logs the terminal failure at the level and wording that +/// suit it. The error itself is returned unchanged. +pub async fn retry_async( + policy: RetryPolicy, + context: &str, + should_retry: S, + mut operation: Op, +) -> Result> +where + S: Fn(&E) -> bool, + Op: FnMut() -> Fut, + Fut: Future>, + E: fmt::Display, +{ + let max_attempts = policy.max_attempts.max(1); + let mut attempt = 0u32; + + loop { + let error = match operation().await { + Ok(value) => { + if attempt > 0 { + let plural = if attempt == 1 { "retry" } else { "retries" }; + info!("{context} succeeded after {attempt} {plural}."); + } + return Ok(value); + } + Err(error) => error, + }; + + attempt += 1; + let retryable = should_retry(&error); + let budget_spent = attempt >= max_attempts; + if !retryable || budget_spent { + return Err(RetryFailure { + error, + attempts: attempt, + // A non-retryable error is the reason we stopped even when the + // budget happened to run out on the same attempt. + exhausted: retryable && budget_spent, + }); + } + + let delay = policy.backoff(attempt); + warn!( + "{context} failed on attempt {attempt}/{max_attempts}: {error}. \ + Retrying in {delay:?}..." + ); + tokio::time::sleep(delay).await; + } +} + // --------------------------------------------------------------------------- // reqwest-middleware retry implementation // --------------------------------------------------------------------------- @@ -201,18 +362,19 @@ pub fn is_transient_status(status: reqwest::StatusCode) -> bool { /// (e.g. `"InfluxDB"`, `"Elasticsearch"`), allowing this middleware to be /// reused across connectors without misleading log output. /// -/// The `max_retries` parameter is the *total attempt count* (not the number -/// of extra attempts), consistent with the rest of the connector retry config: -/// - `max_retries = 1` → one attempt, no retries on failure -/// - `max_retries = 3` → up to three attempts (two retries after a failure) +/// `max_retries` is a total attempt count, as described on [`RetryPolicy`]. /// /// Non-transient error responses (4xx except 429) are returned as-is so /// callers can inspect the status and body to build a meaningful error. +/// +/// The loop is hand-written rather than delegating to [`retry_async`] because +/// a retry here is driven by an `Ok(Response)` carrying a transient status, +/// and the final response is handed back to the caller instead of being turned +/// into an `Err`. Backoff still comes from [`RetryPolicy`], so the timing +/// matches every other connector retry: the first retry waits `retry_delay`. #[derive(Debug, Clone)] pub struct HttpRetryMiddleware { - max_retries: u32, - retry_delay: Duration, - max_delay: Duration, + policy: RetryPolicy, log_prefix: &'static str, } @@ -224,9 +386,11 @@ impl HttpRetryMiddleware { log_prefix: &'static str, ) -> Self { Self { - max_retries, - retry_delay, - max_delay, + policy: RetryPolicy { + max_attempts: max_retries, + base_delay: retry_delay, + max_delay, + }, log_prefix, } } @@ -256,34 +420,28 @@ impl Middleware for HttpRetryMiddleware { return Ok(response); } - // Parse Retry-After header on 429 before falling back to - // our own calculated backoff. + // Parse Retry-After on 429 before falling back to our own + // calculated backoff. let retry_after = if status == reqwest::StatusCode::TOO_MANY_REQUESTS { response .headers() .get("Retry-After") - .and_then(|v| v.to_str().ok()) + .and_then(|value| value.to_str().ok()) .and_then(parse_retry_after) } else { None }; attempts += 1; - if is_transient_status(status) && attempts < self.max_retries { + if is_transient_status(status) && attempts < self.policy.max_attempts { // Consume the error body for logging, then retry. let body_text = response.text().await.unwrap_or_default(); - let delay = retry_after.unwrap_or_else(|| { - jitter(exponential_backoff( - self.retry_delay, - attempts, - self.max_delay, - )) - }); + let delay = retry_after.unwrap_or_else(|| self.policy.backoff(attempts)); warn!( "{} transient error {status} \ (attempt {attempts}/{}): {body_text}. \ Retrying in {delay:?}...", - self.log_prefix, self.max_retries + self.log_prefix, self.policy.max_attempts ); tokio::time::sleep(delay).await; current_req = match next_req { @@ -304,16 +462,12 @@ impl Middleware for HttpRetryMiddleware { } Err(e) => { attempts += 1; - if attempts < self.max_retries { - let delay = jitter(exponential_backoff( - self.retry_delay, - attempts, - self.max_delay, - )); + if attempts < self.policy.max_attempts { + let delay = self.policy.backoff(attempts); warn!( "{} network error (attempt {attempts}/{}): {e}. \ Retrying in {delay:?}...", - self.log_prefix, self.max_retries + self.log_prefix, self.policy.max_attempts ); tokio::time::sleep(delay).await; current_req = match next_req { @@ -358,18 +512,6 @@ pub fn build_retry_client( // Shared connectivity helper // --------------------------------------------------------------------------- -/// Configuration for the startup connectivity retry loop. -/// -/// This is intentionally separate from per-request retry config so that -/// startup can wait patiently for a service (e.g. 10 retries over 60 s) -/// without affecting the shorter per-request retry window used during -/// normal operation. -pub struct ConnectivityConfig { - pub max_open_retries: u32, - pub open_retry_max_delay: Duration, - pub retry_delay: Duration, -} - /// Probe `url` with a plain GET and return `Ok(())` if the response is 2xx. /// /// This is a single, non-retried attempt. The caller is responsible for the @@ -398,49 +540,334 @@ pub async fn check_connectivity( /// Retry [`check_connectivity`] with exponential backoff + jitter. /// -/// `connector_label` is used in log messages (e.g. `"InfluxDB sink connector ID: 1"`). +/// `connector_label` names the connector in log messages (e.g. `"InfluxDB sink"`). /// `connector_id` is included in log messages for multi-instance deployments. +/// +/// Startup usually wants a more patient policy than the per-request one (ten +/// attempts over a minute, say), so callers pass their own [`RetryPolicy`] +/// rather than reusing the one that governs live traffic. pub async fn check_connectivity_with_retry( client: &reqwest::Client, url: reqwest::Url, connector_label: &str, connector_id: u32, - cfg: &ConnectivityConfig, + policy: RetryPolicy, ) -> Result<(), crate::Error> { - let max_open_retries = cfg.max_open_retries.max(1); - let mut attempt = 0u32; + let context = + format!("{connector_label} startup connectivity for connector ID: {connector_id}"); + + retry_async( + policy, + &context, + |_| true, + || check_connectivity(client, url.clone(), connector_label), + ) + .await + .map_err(|failure| { + // `open()`'s Err reaches the FFI boundary and is dropped there, so the + // runtime logs only "Plugin initialization failed". Some callers log + // the error again themselves. + error!("{context} {failure}"); + failure.into_error() + }) +} - loop { - match check_connectivity(client, url.clone(), connector_label).await { - Ok(()) => { - if attempt > 0 { - tracing::info!( - "{connector_label} connectivity established after {attempt} retries \ - for connector ID: {connector_id}" - ); - } - return Ok(()); +#[cfg(test)] +mod tests { + use super::*; + use std::cell::Cell; + use std::time::Instant; + use wiremock::matchers::method; + use wiremock::{Mock, MockServer, ResponseTemplate}; + + const BASE: Duration = Duration::from_millis(100); + const MAX: Duration = Duration::from_secs(10); + + // `jitter` is ±20 %, so every timing assertion below is a band around the + // nominal delay rather than an equality. + const JITTER_LOW: f64 = 0.8; + const JITTER_HIGH: f64 = 1.2; + + #[derive(Debug, Clone, Copy, PartialEq, Eq)] + enum TestError { + Transient, + Permanent, + } + + impl fmt::Display for TestError { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::Transient => write!(f, "transient"), + Self::Permanent => write!(f, "permanent"), } - Err(e) => { - attempt += 1; - if attempt >= max_open_retries { - tracing::error!( - "{connector_label} connectivity check failed after {attempt} attempts \ - for connector ID: {connector_id}. Giving up: {e}" - ); - return Err(e); - } - let backoff = jitter(exponential_backoff( - cfg.retry_delay, - attempt, - cfg.open_retry_max_delay, - )); - tracing::warn!( - "{connector_label} health check failed \ - (attempt {attempt}/{max_open_retries}) \ - for connector ID: {connector_id}. Retrying in {backoff:?}: {e}" - ); - tokio::time::sleep(backoff).await; + } + } + + fn should_retry(error: &TestError) -> bool { + matches!(error, TestError::Transient) + } + + fn policy(max_attempts: u32) -> RetryPolicy { + RetryPolicy { + max_attempts, + base_delay: BASE, + max_delay: MAX, + } + } + + /// Operation that fails with `error` for the first `failures` calls and then + /// succeeds, returning the 1-based number of the call that succeeded. + fn failing_times( + calls: &Cell, + failures: u32, + error: TestError, + ) -> impl FnMut() -> std::future::Ready> + '_ { + move || { + let call = calls.get() + 1; + calls.set(call); + std::future::ready(if call <= failures { + Err(error) + } else { + Ok(call) + }) + } + } + + #[tokio::test(start_paused = true)] + async fn given_a_transient_failure_should_retry_until_it_succeeds() { + let calls = Cell::new(0); + let result = retry_async( + policy(4), + "test", + should_retry, + failing_times(&calls, 2, TestError::Transient), + ) + .await; + + assert_eq!(result.map_err(RetryFailure::into_error), Ok(3)); + assert_eq!(calls.get(), 3); + } + + #[tokio::test(start_paused = true)] + async fn given_a_permanent_failure_should_not_retry() { + let calls = Cell::new(0); + let result = retry_async( + policy(4), + "test", + should_retry, + failing_times(&calls, 2, TestError::Permanent), + ) + .await; + + let failure = result.expect_err("permanent error should not retry"); + assert_eq!(failure.error, TestError::Permanent); + assert_eq!(failure.attempts, 1); + assert!( + !failure.exhausted, + "should_retry rejected it, budget untouched" + ); + assert_eq!(calls.get(), 1); + } + + #[tokio::test(start_paused = true)] + async fn given_a_permanent_failure_on_the_last_attempt_should_not_report_exhaustion() { + // Budget of 1 makes the last attempt also the first, so both stop + // conditions fire at once; the non-retryable one is the real reason. + let calls = Cell::new(0); + let result = retry_async( + policy(1), + "test", + should_retry, + failing_times(&calls, u32::MAX, TestError::Permanent), + ) + .await; + + let failure = result.expect_err("permanent error should fail"); + assert!( + !failure.exhausted, + "reported exhaustion for an error that was never retryable" + ); + } + + #[tokio::test(start_paused = true)] + async fn given_an_exhausted_budget_should_return_the_last_error() { + let calls = Cell::new(0); + // Distinct error per attempt, so "last" is actually discriminated. + let result: Result> = retry_async( + policy(3), + "test", + |_| true, + || { + let call = calls.get() + 1; + calls.set(call); + std::future::ready(Err(call)) + }, + ) + .await; + + let failure = result.expect_err("budget should be exhausted"); + assert_eq!(failure.error, 3, "should surface the final attempt's error"); + assert_eq!(failure.attempts, 3); + assert!(failure.exhausted); + assert_eq!(calls.get(), 3, "max_attempts is a total attempt count"); + } + + #[tokio::test(start_paused = true)] + async fn given_a_single_attempt_budget_should_run_the_operation_once() { + for max_attempts in [0, 1] { + let calls = Cell::new(0); + let result = retry_async( + policy(max_attempts), + "test", + should_retry, + failing_times(&calls, u32::MAX, TestError::Transient), + ) + .await; + + let failure = result.expect_err("single attempt should fail"); + assert_eq!(failure.error, TestError::Transient); + assert!(failure.exhausted, "max_attempts = {max_attempts}"); + assert_eq!(calls.get(), 1, "max_attempts = {max_attempts}"); + } + } + + #[tokio::test(start_paused = true)] + async fn given_successive_retries_should_double_the_delay() { + let calls = Cell::new(0); + let started = tokio::time::Instant::now(); + let result = retry_async( + policy(3), + "test", + should_retry, + failing_times(&calls, 2, TestError::Transient), + ) + .await; + let elapsed = started.elapsed(); + + assert_eq!(result.map_err(RetryFailure::into_error), Ok(3)); + // base + 2 × base, each independently jittered. Passing the retry + // number straight to `exponential_backoff` would give 2 + 4 × base. + let nominal = BASE * 3; + assert!( + elapsed >= nominal.mul_f64(JITTER_LOW) && elapsed <= nominal.mul_f64(JITTER_HIGH), + "two retries waited {elapsed:?}, expected roughly {nominal:?}" + ); + } + + // `HttpRetryMiddleware` is the path every HTTP connector rides, and nothing + // covered it before: these cases pin the first delay and that a server's + // `Retry-After` outranks the computed backoff. + fn retry_client(max_attempts: u32, base: Duration, max: Duration) -> ClientWithMiddleware { + build_retry_client(reqwest::Client::new(), max_attempts, base, max, "test") + } + + async fn mock_then_ok(status: u16, headers: &[(&str, &str)]) -> MockServer { + let server = MockServer::start().await; + let mut first = ResponseTemplate::new(status); + for (name, value) in headers { + first = first.insert_header(*name, *value); + } + Mock::given(method("GET")) + .respond_with(first) + .up_to_n_times(1) + .mount(&server) + .await; + Mock::given(method("GET")) + .respond_with(ResponseTemplate::new(200)) + .mount(&server) + .await; + server + } + + #[tokio::test] + async fn given_no_retry_after_should_wait_the_base_delay_on_the_first_retry() { + let server = mock_then_ok(503, &[]).await; + // 1s rather than 200ms so the pass band and the bug's band do not + // overlap: the bug produces jitter(2s) in [1.6s, 2.4s], a correct run + // produces jitter(1s) in [0.8s, 1.2s], and 1.4s separates them with + // room for two loopback round trips. + let base = Duration::from_secs(1); + let client = retry_client(3, base, Duration::from_secs(30)); + + let started = Instant::now(); + let response = client.get(server.uri()).send().await.unwrap(); + let elapsed = started.elapsed(); + + assert_eq!(response.status(), 200); + assert_eq!( + server.received_requests().await.unwrap().len(), + 2, + "expected exactly one retry" + ); + assert!( + elapsed >= base.mul_f64(JITTER_LOW), + "first retry waited {elapsed:?}, expected roughly {base:?}" + ); + // Guards the middleware's own `policy.backoff(attempts)` wiring, which + // the pure `retry_backoff` tests do not reach: feeding a 1-based + // counter to the 0-based `exponential_backoff` doubles this delay. + // A correct run tops out at 1.2 x base and the bug starts at 1.6 x, so + // 1.5 x leaves the widest margin for loopback round trips on a loaded + // runner while still failing on the bug. + assert!( + elapsed < base.mul_f64(1.5), + "first retry waited {elapsed:?}, past the {base:?} the config asks for" + ); + } + + #[tokio::test] + async fn given_a_retry_after_should_take_precedence_over_the_computed_backoff() { + let server = mock_then_ok(429, &[("Retry-After", "1")]).await; + // Backoff bounded far below the header, so honoring it is visible. + let client = retry_client(3, Duration::from_millis(10), Duration::from_millis(50)); + + let started = Instant::now(); + let response = client.get(server.uri()).send().await.unwrap(); + let elapsed = started.elapsed(); + + assert_eq!(response.status(), 200); + assert_eq!( + server.received_requests().await.unwrap().len(), + 2, + "expected exactly one retry" + ); + assert!( + elapsed >= Duration::from_millis(900), + "used the computed backoff instead of Retry-After: waited {elapsed:?}" + ); + // The header asks for 1s. Anything far past it means the middleware + // added its own backoff on top instead of honoring the header. + assert!( + elapsed < Duration::from_millis(1500), + "waited {elapsed:?}, past the 1s the header asked for" + ); + } + + #[test] + fn given_a_retry_number_should_back_off_from_the_base_delay() { + let policy = policy(8); + for (retry, factor) in [(1u32, 1.0), (2, 2.0), (3, 4.0), (4, 8.0)] { + let delay = policy.backoff(retry); + let nominal = BASE.mul_f64(factor); + assert!( + delay >= nominal.mul_f64(JITTER_LOW) && delay <= nominal.mul_f64(JITTER_HIGH), + "retry {retry} backed off {delay:?}, expected roughly {nominal:?}" + ); + } + } + + #[test] + fn given_jitter_on_a_capped_delay_should_never_exceed_max_delay() { + // Base far above the cap, so every draw starts clamped and only jitter + // could push it back over. + let policy = RetryPolicy { + max_attempts: 8, + base_delay: Duration::from_secs(30), + max_delay: Duration::from_secs(1), + }; + for retry in 1..=8 { + for _ in 0..64 { + assert!(policy.backoff(retry) <= policy.max_delay); } } } diff --git a/core/connectors/sinks/doris_sink/src/lib.rs b/core/connectors/sinks/doris_sink/src/lib.rs index b58b112a8b..78e2d70d37 100644 --- a/core/connectors/sinks/doris_sink/src/lib.rs +++ b/core/connectors/sinks/doris_sink/src/lib.rs @@ -19,7 +19,7 @@ use async_trait::async_trait; use base64::{Engine as _, engine::general_purpose}; use bytes::Bytes; use humantime::Duration as HumanDuration; -use iggy_connector_sdk::retry::{exponential_backoff, jitter}; +use iggy_connector_sdk::retry::{RetryPolicy, retry_async}; use iggy_connector_sdk::{ ConsumedMessage, Error, MessagesMetadata, Payload, Sink, TopicMetadata, sink_connector, }; @@ -300,7 +300,7 @@ impl DorisSink { } let response = request.send().await.map_err(|e| { - error!("Doris sink ID {} HTTP request failed: {e}", self.id); + warn!("Doris sink ID {} HTTP request failed: {e}", self.id); Error::HttpRequestFailed(e.to_string()) })?; @@ -385,7 +385,9 @@ impl DorisSink { "Doris sink ID {} stream load returned HTTP {status}: {response_for_log}", self.id ); - error!("{msg}"); + // Per-attempt detail only: `retry_async` logs every retried + // attempt, and `consume` logs the terminal error carrying `msg`. + warn!("{msg}"); // 408/429 are 4xx but transient, so include them in the bounded // in-request retry path. return Err(match status { @@ -417,36 +419,31 @@ impl DorisSink { )) })?; - let mut attempt = 0u32; - loop { - let error = match self - .send_stream_load(connected, label, body.clone()) - .await - .and_then(|response| classify_status(self.id, &response).map(|()| response)) - { - Ok(response) => return Ok(response), - Err(error) => error, - }; - - attempt += 1; - if attempt >= connected.max_retries || !is_transient_error(&error) { - return Err(error); + let policy = RetryPolicy { + max_attempts: connected.max_retries, + base_delay: connected.retry_delay, + max_delay: connected.max_retry_delay, + }; + let context = format!("Doris sink ID {} Stream Load (label={label})", self.id); + + retry_async(policy, &context, is_transient_error, || { + let body = body.clone(); + async move { + let response = self.send_stream_load(connected, label, body).await?; + classify_status(self.id, &response)?; + Ok(response) } - - // `attempt` counts completed attempts. Subtract one so the first - // retry waits exactly the configured base delay (base * 2^0). - let delay = jitter(exponential_backoff( - connected.retry_delay, - attempt - 1, - connected.max_retry_delay, - )) - .min(connected.max_retry_delay); + }) + .await + .map_err(|failure| { + // The only place the attempt count and the reason survive: + // `consume` logs the error itself, which carries neither. warn!( - "Doris sink ID {} transient Stream Load failure on attempt {attempt}/{} (label={label}): {error}; retrying in {delay:?}", - self.id, connected.max_retries + "Doris sink ID {} Stream Load (label={label}) {failure}", + self.id ); - tokio::time::sleep(delay).await; - } + failure.into_error() + }) } } @@ -1011,7 +1008,7 @@ impl Sink for DorisSink { self.id ); } - // `exponential_backoff` already caps at the max, but a base above the cap + // `retry_backoff` already caps at the max, but a base above the cap // is a config mistake worth surfacing rather than silently flattening. let (retry_delay, max_retry_delay) = if retry_delay > max_retry_delay { warn!( diff --git a/core/connectors/sinks/influxdb_sink/README.md b/core/connectors/sinks/influxdb_sink/README.md index 37a8e01252..64813f8d89 100644 --- a/core/connectors/sinks/influxdb_sink/README.md +++ b/core/connectors/sinks/influxdb_sink/README.md @@ -69,10 +69,10 @@ verbose_logging = false ```toml timeout = "30s" # per-request timeout -max_retries = 3 # retries per write on transient errors (429/5xx) +max_retries = 3 # total write attempts, including the first (429/5xx) retry_delay = "1s" # initial backoff between retries retry_max_delay = "5s" # backoff cap -max_open_retries = 10 # retries during open() health check +max_open_retries = 10 # total open() health-check attempts, including the first open_retry_max_delay = "60s" # backoff cap for open() retries circuit_breaker_threshold = 5 # consecutive failures before circuit trips circuit_breaker_cool_down = "30s" # how long circuit stays open before half-open probe @@ -119,6 +119,6 @@ circuit_breaker_cool_down = "15s" The sink uses a layered design: - **Batch accumulator**: messages are serialised to line protocol and buffered until `batch_size` is reached, then flushed in a single HTTP POST. -- **Retry middleware**: `reqwest-retry` with exponential backoff handles 429 and 5xx responses automatically before the connector-level retry logic runs. +- **Retry middleware**: `iggy_connector_sdk::retry::HttpRetryMiddleware` retries 429, 5xx and network errors with exponential backoff and jitter. - **Circuit breaker**: after `circuit_breaker_threshold` consecutive failures the connector stops issuing writes and waits for the cool-down window before probing again. - **Precision mapping**: V3's `/api/v3/write_lp` endpoint requires full English words (`nanosecond`, `microsecond`, `millisecond`, `second`); the connector maps the short forms automatically. diff --git a/core/connectors/sinks/influxdb_sink/src/lib.rs b/core/connectors/sinks/influxdb_sink/src/lib.rs index 51a6828d17..2018bd90f8 100644 --- a/core/connectors/sinks/influxdb_sink/src/lib.rs +++ b/core/connectors/sinks/influxdb_sink/src/lib.rs @@ -23,8 +23,7 @@ use base64::{Engine as _, engine::general_purpose}; use bytes::Bytes; use iggy_common::serde_secret::serialize_secret; use iggy_connector_sdk::retry::{ - CircuitBreaker, ConnectivityConfig, build_retry_client, check_connectivity_with_retry, - parse_duration, + CircuitBreaker, RetryPolicy, build_retry_client, check_connectivity_with_retry, parse_duration, }; use iggy_connector_sdk::{ ConsumedMessage, Error, MessagesMetadata, Sink, TopicMetadata, sink_connector, @@ -749,13 +748,13 @@ impl Sink for InfluxDbSink { self.config.build_health_url()?, "InfluxDB sink", self.id, - &ConnectivityConfig { - max_open_retries: self.config.max_open_retries(), - open_retry_max_delay: parse_duration( + RetryPolicy { + max_attempts: self.config.max_open_retries(), + base_delay: self.retry_delay, + max_delay: parse_duration( self.config.open_retry_max_delay(), DEFAULT_OPEN_RETRY_MAX_DELAY, ), - retry_delay: self.retry_delay, }, ) .await?; diff --git a/core/connectors/sinks/meilisearch_sink/src/lib.rs b/core/connectors/sinks/meilisearch_sink/src/lib.rs index 64b42ab8c2..27f0a04ae7 100644 --- a/core/connectors/sinks/meilisearch_sink/src/lib.rs +++ b/core/connectors/sinks/meilisearch_sink/src/lib.rs @@ -20,7 +20,7 @@ use base64::{Engine as _, engine::general_purpose}; use iggy_common::IggyTimestamp; use iggy_connector_sdk::{ ConsumedMessage, Error, MessagesMetadata, Payload, Sink, TopicMetadata, - retry::{exponential_backoff, jitter, parse_duration}, + retry::{parse_duration, retry_backoff}, sink_connector, }; use meilisearch_sdk::{ @@ -228,11 +228,7 @@ impl MeilisearchSink { ))); } retries += 1; - let delay = jitter(exponential_backoff( - self.config.retry_delay, - retries, - self.config.max_retry_delay, - )); + let delay = self.backoff(retries); warn!( "Meilisearch health check returned status '{}' (retry {}/{}). Retrying in {:?}...", health.status, retries, self.config.max_open_retries, delay @@ -246,11 +242,7 @@ impl MeilisearchSink { return Err(map_sdk_error(error)); } retries += 1; - let delay = jitter(exponential_backoff( - self.config.retry_delay, - retries, - self.config.max_retry_delay, - )); + let delay = self.backoff(retries); warn!( "Meilisearch health check failed (retry {}/{}): {}. Retrying in {:?}...", retries, self.config.max_open_retries, error, delay @@ -265,11 +257,7 @@ impl MeilisearchSink { ))); } retries += 1; - let delay = jitter(exponential_backoff( - self.config.retry_delay, - retries, - self.config.max_retry_delay, - )); + let delay = self.backoff(retries); warn!( "Meilisearch health check timed out after {:?} (retry {}/{}). Retrying in {:?}...", self.config.timeout, retries, self.config.max_open_retries, delay @@ -642,6 +630,16 @@ impl MeilisearchSink { ))) } + /// Backoff before retry number `retry` (1-based), from the configured bounds. + /// + /// Only the delay is shared with the SDK. This connector's `max_retries` / + /// `max_open_retries` count retries *after* the first request, as the + /// README documents, unlike `RetryPolicy::max_attempts` which is a total. + /// Aligning them would silently cut every deployed budget by one. + fn backoff(&self, retry: u32) -> Duration { + retry_backoff(self.config.retry_delay, retry, self.config.max_retry_delay) + } + async fn retry_sdk_operation( &self, operation: &str, @@ -699,11 +697,7 @@ impl MeilisearchSink { return Err(map_sdk_error(error)); } retries += 1; - let delay = jitter(exponential_backoff( - self.config.retry_delay, - retries, - self.config.max_retry_delay, - )); + let delay = self.backoff(retries); warn!( "Meilisearch {operation} failed (retry {retries}/{max_retries}): {error}. Retrying in {delay:?}..." ); @@ -721,11 +715,7 @@ impl MeilisearchSink { ))); } retries += 1; - let delay = jitter(exponential_backoff( - self.config.retry_delay, - retries, - self.config.max_retry_delay, - )); + let delay = self.backoff(retries); warn!( "Meilisearch {operation} timed out after {:?} (retry {retries}/{max_retries}). Retrying in {delay:?}...", self.config.timeout diff --git a/core/connectors/sinks/quickwit_sink/src/lib.rs b/core/connectors/sinks/quickwit_sink/src/lib.rs index cd5392d585..fadcb8b551 100644 --- a/core/connectors/sinks/quickwit_sink/src/lib.rs +++ b/core/connectors/sinks/quickwit_sink/src/lib.rs @@ -20,7 +20,7 @@ use std::time::Duration; use async_trait::async_trait; use base64::{Engine as _, engine::general_purpose}; use iggy_connector_sdk::retry::{ - ConnectivityConfig, build_retry_client, check_connectivity_with_retry, is_transient_status, + RetryPolicy, build_retry_client, check_connectivity_with_retry, is_transient_status, }; use iggy_connector_sdk::{ ConsumedMessage, Error, MessagesMetadata, Payload, Sink, TopicMetadata, sink_connector, @@ -374,14 +374,14 @@ impl Sink for QuickwitSink { endpoint_url(&base_url, &["health", "readyz"])?, "Quickwit sink", self.id, - &ConnectivityConfig { - max_open_retries: self + RetryPolicy { + max_attempts: self .config .max_open_retries .unwrap_or(DEFAULT_MAX_OPEN_RETRIES) .max(1), - open_retry_max_delay, - retry_delay, + base_delay: retry_delay, + max_delay: open_retry_max_delay, }, ) .await?; diff --git a/core/connectors/sinks/rabbitmq_sink/src/lib.rs b/core/connectors/sinks/rabbitmq_sink/src/lib.rs index 69504096e9..a428d12507 100644 --- a/core/connectors/sinks/rabbitmq_sink/src/lib.rs +++ b/core/connectors/sinks/rabbitmq_sink/src/lib.rs @@ -17,7 +17,7 @@ use async_trait::async_trait; use iggy::prelude::HeaderKind; -use iggy_connector_sdk::retry::{exponential_backoff, jitter}; +use iggy_connector_sdk::retry::retry_backoff; use iggy_connector_sdk::{ ConsumedMessage, Error, MessagesMetadata, Sink, TopicMetadata, sink_connector, }; @@ -229,11 +229,7 @@ impl RabbitMQSink { "failed to reconnect: {reconnect_error}" ))); } - let delay = jitter(exponential_backoff( - self.retry_delay, - attempts.saturating_sub(1), - self.max_retry_delay, - )); + let delay = retry_backoff(self.retry_delay, attempts, self.max_retry_delay); warn!( "RabbitMQ not connected for connector ID: {} (attempt {attempts}/{}). Retrying in {:?}.", self.id, self.max_retries, delay @@ -372,11 +368,7 @@ impl RabbitMQSink { } } - let delay = jitter(exponential_backoff( - self.retry_delay, - attempts.saturating_sub(1), - self.max_retry_delay, - )); + let delay = retry_backoff(self.retry_delay, attempts, self.max_retry_delay); warn!( "Transient RabbitMQ publish error for connector ID: {} (attempt {attempts}/{}): {error}. Retrying in {:?}.", self.id, self.max_retries, delay diff --git a/core/connectors/sinks/s3_sink/src/sink.rs b/core/connectors/sinks/s3_sink/src/sink.rs index 59b1a59f80..7ca3e1296c 100644 --- a/core/connectors/sinks/s3_sink/src/sink.rs +++ b/core/connectors/sinks/s3_sink/src/sink.rs @@ -20,7 +20,7 @@ use crate::formatter; use crate::path::{PathContext, render_s3_key}; use crate::{BufferKey, S3Sink}; use async_trait::async_trait; -use iggy_connector_sdk::retry::{exponential_backoff, jitter}; +use iggy_connector_sdk::retry::retry_backoff; use iggy_connector_sdk::{ConsumedMessage, Error, MessagesMetadata, Sink, TopicMetadata}; use std::sync::Arc; use std::time::Duration; @@ -392,9 +392,7 @@ impl S3Sink { ); } } - // exponential_backoff expects a 0-based retry index - let retry_index = attempt - 1; - let delay = jitter(exponential_backoff(base_delay, retry_index, MAX_BACKOFF)); + let delay = retry_backoff(base_delay, attempt, MAX_BACKOFF); tokio::time::sleep(delay).await; } } diff --git a/core/connectors/sinks/surrealdb_sink/src/lib.rs b/core/connectors/sinks/surrealdb_sink/src/lib.rs index 6e354e9ae0..0d774083fc 100644 --- a/core/connectors/sinks/surrealdb_sink/src/lib.rs +++ b/core/connectors/sinks/surrealdb_sink/src/lib.rs @@ -20,7 +20,7 @@ use base64::Engine; use base64::engine::general_purpose; use bytes::Bytes; use iggy_connector_sdk::convert::owned_value_to_serde_json; -use iggy_connector_sdk::retry::{exponential_backoff, jitter, parse_duration}; +use iggy_connector_sdk::retry::{parse_duration, retry_backoff}; use iggy_connector_sdk::{ ConsumedMessage, Error, MessagesMetadata, Payload, Sink, TopicMetadata, sink_connector, }; @@ -773,11 +773,7 @@ impl SurrealDbSink { } } - let delay = jitter(exponential_backoff( - self.retry_delay, - attempts.saturating_sub(1), - self.max_retry_delay, - )); + let delay = retry_backoff(self.retry_delay, attempts, self.max_retry_delay); warn!( "Transient SurrealDB write error for connector ID: {} (attempt {attempts}/{}): {error}. Retrying in {:?}.", self.id, self.max_retries, delay diff --git a/core/connectors/sources/influxdb_source/README.md b/core/connectors/sources/influxdb_source/README.md index f7110fc3ad..333e4f3d76 100644 --- a/core/connectors/sources/influxdb_source/README.md +++ b/core/connectors/sources/influxdb_source/README.md @@ -108,10 +108,10 @@ verbose_logging = false ```toml timeout = "10s" # per-request timeout -max_retries = 3 # retries per query on transient errors (429/5xx) +max_retries = 3 # total query attempts, including the first (429/5xx) retry_delay = "1s" # initial backoff retry_max_delay = "5s" # backoff cap -max_open_retries = 10 # retries during open() health check +max_open_retries = 10 # total open() health-check attempts, including the first open_retry_max_delay = "60s" # backoff cap for open() retries circuit_breaker_threshold = 5 # consecutive failures before circuit trips circuit_breaker_cool_down = "30s" # cool-down before half-open probe diff --git a/core/connectors/sources/influxdb_source/src/lib.rs b/core/connectors/sources/influxdb_source/src/lib.rs index ac0ed9fda8..59a1b5e124 100644 --- a/core/connectors/sources/influxdb_source/src/lib.rs +++ b/core/connectors/sources/influxdb_source/src/lib.rs @@ -31,8 +31,7 @@ use common::{ validate_cursor_field, }; use iggy_connector_sdk::retry::{ - CircuitBreaker, ConnectivityConfig, build_retry_client, check_connectivity_with_retry, - parse_duration, + CircuitBreaker, RetryPolicy, build_retry_client, check_connectivity_with_retry, parse_duration, }; use iggy_connector_sdk::{ ConnectorState, Error, ProducedMessages, Schema, Source, source_connector, @@ -420,13 +419,13 @@ impl Source for InfluxDbSource { health_url, CONNECTOR_NAME, self.id, - &ConnectivityConfig { - max_open_retries: self.config.max_open_retries(), - open_retry_max_delay: parse_duration( + RetryPolicy { + max_attempts: self.config.max_open_retries(), + base_delay: self.retry_delay, + max_delay: parse_duration( self.config.open_retry_max_delay(), DEFAULT_OPEN_RETRY_MAX_DELAY, ), - retry_delay: self.retry_delay, }, ) .await?; From d693eb93d878a7759221fba05dcea20e52dd1249 Mon Sep 17 00:00:00 2001 From: Ryan Huang Date: Fri, 11 Sep 2026 01:38:58 +0800 Subject: [PATCH 100/182] test(integration): cleanup & drop the doris nextest test-group (#4116) Closes #3371 --- .config/nextest.toml | 15 --------------- .github/config/components.yml | 1 + 2 files changed, 1 insertion(+), 15 deletions(-) diff --git a/.config/nextest.toml b/.config/nextest.toml index 1d98894330..ef58df2622 100644 --- a/.config/nextest.toml +++ b/.config/nextest.toml @@ -27,21 +27,6 @@ nextest-version = { required = "0.9.77" } filter = 'package(integration) and test(cli::system::test_cli_session_scenario::should_be_successful)' threads-required = "num-cpus" -# Doris tests are serialized among themselves, but no longer monopolize the -# whole runner. The all-in-one image's BE advertises 127.0.0.1:8040 for the -# FE→BE 307 redirect, so only one Doris container per process can bind -# host:8040. Within a nextest binary the fixture caches one shared container in -# `SHARED_DORIS` so the first doris test pays the ~40s boot and the rest reuse -# it; `max-threads = 1` keeps a second nextest binary (or a re-run) from -# racing the same host port, while still letting unrelated light tests fill -# the rest of the runner. -[test-groups.doris] -max-threads = 1 - -[[profile.default.overrides]] -filter = 'package(integration) and test(/connectors::doris::/)' -test-group = "doris" - # Elasticsearch tests share one reusable container (fixed name # `iggy-test-elasticsearch`, ReuseDirective::Always). Serializing the group # lets the first test create it and the rest attach by name, instead of racing diff --git a/.github/config/components.yml b/.github/config/components.yml index d63cdd54a7..a42745dacd 100644 --- a/.github/config/components.yml +++ b/.github/config/components.yml @@ -23,6 +23,7 @@ components: - "Cargo.lock" - "rust-toolchain.toml" - ".cargo/**" + - ".config/**" # nextest + cargo-rail config gate every Rust test job # crates.io publish chain verification. Runs scripts/verify-crates-publish.sh, # which spins up cargo-http-registry as a local alt-registry and chain-publishes From d75890f67f4c8d71d7828646b421c23754dbf000 Mon Sep 17 00:00:00 2001 From: Hubert Gruszecki Date: Fri, 11 Sep 2026 09:26:18 +0200 Subject: [PATCH 101/182] fix(repo): include gateways so the source release tarball builds (#4126) --- scripts/prepare-release.sh | 1 + 1 file changed, 1 insertion(+) diff --git a/scripts/prepare-release.sh b/scripts/prepare-release.sh index 0d270e285b..aee27cfc5e 100755 --- a/scripts/prepare-release.sh +++ b/scripts/prepare-release.sh @@ -60,6 +60,7 @@ RELEASE_PATHS=( "core" "examples" "foreign" + "gateways" "helm" "scripts" "web" From e0027506f2a0379ee3bb5508427ff570e3a9a435 Mon Sep 17 00:00:00 2001 From: Piotr Gankiewicz Date: Fri, 11 Sep 2026 10:05:54 +0200 Subject: [PATCH 102/182] feat(cluster)!: let topics require durable acks and flatten config (#4092) Producer acknowledgments did not wait for recoverable storage. The `enforce_fsync` option synchronized segment flushes, but replicas sent `PrepareOk` before prepares reached stable storage. Consumer-offset synchronization was one server-wide key. This PR adds two create-only topic options, `durability` and `consumer_offset_durability`. Each is `replicated` (the default) or `persisted`, and neither inherits the other. Both policies write to disk and complete after VSR quorum commit. Persisted also requires recoverable stable-storage copies on the quorum, or a local sync in a single-replica group. In replicated groups, a persisted policy adds a bounded per-partition prepare WAL, sized by `partition.wal_bytes_max`. WAL records reference message bodies in segment files, and hard links keep them until reclamation. Replicas forward each prepare before their own WAL write completes. Reclamation waits for durable materialized state, and missing history or storage errors fail closed. AWS i4i benchmarks found two costs outside the WAL, which this PR removes. Produce admission zero-filled each request buffer before the copy, and repair ran for operations that were already resident. The configuration drops `[system]`. Its `path` key and remaining tables move to the root, and `IGGY_SYSTEM_*` variables lose `SYSTEM_`. This PR removes the stream, topic and partition path keys, `archive_expired`, `recreate_missing_state` and `consumer_offset_enforce_fsync`. The server refuses to boot with a relocated key or a stored topic with `enforce_fsync=true`. The Rust, Java, C#, Go, Node, Python, PHP and C++ SDKs, the CLI and the benchmark expose both options. The HTTP `Iggy-Durability` header reports `replicated` or `persisted` instead of `replicated-memory`. Poll auto-commit and `ack=none` produce stay asynchronous. A deterministic simulator storage model injects crashes, power loss and torn writes into WAL tests. Cluster tests cover crashes and corruption, and a compatibility test boots a baseline data directory. The in-memory partition simulator does not run persisted topics. --- .github/actions/go/pre-merge/action.yml | 8 +- .../actions/java-gradle/pre-merge/action.yml | 4 +- Cargo.lock | 19 +- Cargo.toml | 10 +- README.md | 34 +- bdd/docker-compose.cluster.yml | 4 +- bdd/docker-compose.server.yml | 2 +- bdd/python/uv.lock | 2 +- core/ai/mcp/Cargo.toml | 2 +- core/bench/Cargo.toml | 2 +- core/bench/README.md | 60 + core/bench/dashboard/frontend/index.html | 2 +- core/bench/src/args/common.rs | 31 +- core/bench/src/args/defaults.rs | 6 +- core/bench/src/args/examples.rs | 355 +- .../balanced/producer_and_consumer_group.rs | 4 +- .../end_to_end/producing_consumer_group.rs | 2 +- core/bench/src/args/kinds/pinned/consumer.rs | 6 +- core/bench/src/args/transport.rs | 21 +- core/bench/src/benchmarks/benchmark.rs | 9 +- core/bench/src/utils/mod.rs | 116 +- core/binary_protocol/Cargo.toml | 2 +- .../binary_protocol/src/primitives/options.rs | 27 +- core/cli/Cargo.toml | 2 +- core/cli/src/args/topic.rs | 69 + .../commands/binary_topics/create_topic.rs | 10 +- core/cli/src/main.rs | 2 + core/common/Cargo.toml | 2 +- core/common/src/traits/message_client.rs | 6 +- core/common/src/types/options/durability.rs | 53 + core/common/src/types/options/mod.rs | 297 +- core/configs/src/common/defaults.rs | 90 +- core/configs/src/common/displays.rs | 50 +- core/configs/src/common/mod.rs | 2 +- core/configs/src/common/server.rs | 2 +- core/configs/src/common/system.rs | 219 +- core/configs/src/common/validators.rs | 40 +- .../configs/src/configs_impl/file_provider.rs | 108 +- core/configs/src/configs_impl/mod.rs | 2 +- .../src/configs_impl/typed_env_provider.rs | 2 +- core/configs/src/lib.rs | 2 +- core/configs/src/server_config/defaults.rs | 16 +- core/configs/src/server_config/displays.rs | 18 +- core/configs/src/server_config/partition.rs | 71 +- core/configs/src/server_config/server.rs | 259 +- core/configs/src/server_config/sharding.rs | 16 +- core/configs/src/server_config/validators.rs | 57 +- core/connectors/sdk/Cargo.toml | 2 +- core/connectors/sdk/README.md | 6 +- core/consensus/src/impls.rs | 175 +- core/consensus/src/plane_helpers.rs | 229 +- core/harness_derive/src/attrs.rs | 2 +- .../integration/src/harness/config/resolve.rs | 40 +- core/integration/src/harness/disk.rs | 13 +- core/integration/src/harness/handle/server.rs | 24 +- .../src/harness/orchestrator/builder.rs | 4 +- .../cli/topic/test_topic_create_command.rs | 30 +- .../tests/cluster/crash_durability.rs | 339 +- .../cluster/crash_recovery_corruption.rs | 227 +- .../cluster/failover_client_continuity.rs | 2 +- .../tests/cluster/fast_primary_rejoin.rs | 2 +- .../cluster/metadata_checkpoint_restart.rs | 2 +- .../multi_shard_partition_convergence.rs | 4 +- .../tests/cluster/parked_frame_redispatch.rs | 8 +- .../tests/cluster/partition_dedup.rs | 12 +- .../tests/cluster/partition_state_transfer.rs | 6 +- .../tests/cluster/register_forwarding.rs | 14 +- .../tests/cluster/staggered_bootstrap.rs | 49 +- core/integration/tests/config_provider/mod.rs | 6 +- .../tests/data_integrity/storage_compat.rs | 228 +- core/integration/tests/sdk/options.rs | 15 +- .../tests/server/a2a_jwt/config.toml | 33 +- .../tests/server/cluster_metadata_vsr.rs | 2 +- .../server/cluster_view_durability_vsr.rs | 2 +- .../tests/server/http_read_your_writes.rs | 4 +- .../tests/server/http_view_header.rs | 4 +- core/integration/tests/server/http_vsr.rs | 158 +- .../tests/server/message_retrieval.rs | 2 +- .../server/partition_view_durability_vsr.rs | 2 +- .../concurrent_produce_consume_scenario.rs | 2 +- ...tiple_clients_polling_messages_scenario.rs | 6 +- .../server/scenarios/encryption_scenario.rs | 9 +- .../server/scenarios/log_rotation_scenario.rs | 8 +- .../scenarios/message_cleanup_scenario.rs | 156 +- .../server/scenarios/purge_delete_scenario.rs | 2 +- .../read_during_persistence_scenario.rs | 2 +- .../reconnect_after_restart_scenario.rs | 4 +- .../scenarios/restart_offset_skip_scenario.rs | 2 +- core/journal/Cargo.toml | 6 +- core/journal/src/durable_storage.rs | 452 +++ core/journal/src/lib.rs | 4 + core/journal/src/partition_journal.rs | 2993 +++++++++++++++++ .../journal/src/partition_journal/segments.rs | 848 +++++ core/metadata/src/impls/metadata.rs | 13 +- core/metadata/src/stm/stream.rs | 24 +- core/partitions/Cargo.toml | 4 +- core/partitions/src/iggy_partition.rs | 2949 +++++++++++++++- core/partitions/src/iggy_partitions.rs | 17 +- core/partitions/src/install_backup.rs | 213 ++ core/partitions/src/journal.rs | 26 +- core/partitions/src/lib.rs | 5 + core/partitions/src/messages_writer.rs | 58 +- core/partitions/src/offset_storage.rs | 69 +- core/partitions/src/persistence.rs | 1504 +++++++++ core/partitions/src/state_transfer.rs | 526 ++- core/partitions/src/types.rs | 24 +- core/sdk/Cargo.toml | 2 +- core/sdk/src/http/messages.rs | 99 +- core/sdk/src/http/topics.rs | 2 +- core/sdk/src/prelude.rs | 8 +- core/server/Cargo.toml | 2 +- core/server/config.toml | 277 +- core/server/src/args.rs | 6 +- core/server/src/boot/credentials.rs | 2 +- core/server/src/boot/listeners.rs | 2 +- core/server/src/boot/mod.rs | 49 +- core/server/src/boot/recovery.rs | 84 +- core/server/src/boot/threads.rs | 11 +- core/server/src/config_writer.rs | 2 +- core/server/src/consumer_group.rs | 643 +++- core/server/src/dispatch/failure.rs | 6 +- core/server/src/dispatch/mod.rs | 57 +- core/server/src/dispatch/partition.rs | 133 +- core/server/src/dispatch/reads.rs | 10 +- core/server/src/dispatch/session_ops.rs | 3 +- core/server/src/dispatch/test_support.rs | 3 +- core/server/src/http.rs | 6 +- core/server/src/http/handlers.rs | 53 +- core/server/src/http/reads.rs | 221 ++ core/server/src/http/state.rs | 4 +- core/server/src/main.rs | 3 +- core/server/src/partition_helpers.rs | 557 ++- core/server/src/partition_reconciler.rs | 39 +- core/server/src/responses.rs | 15 +- core/server/src/segment_recovery.rs | 353 +- core/server/src/server_error.rs | 66 +- core/server/src/snapshot.rs | 28 +- core/server/tests/sdk_e2e.rs | 4 +- core/server_common/src/fs_utils.rs | 46 + core/server_common/src/iobuf.rs | 4 + .../src/segment_storage/messages_reader.rs | 13 +- core/server_common/src/segment_storage/mod.rs | 55 + core/server_common/src/send_messages.rs | 6 +- core/shard/src/lib.rs | 485 ++- core/shard/src/metrics.rs | 115 +- core/shard/src/router.rs | 21 + core/simulator/Cargo.toml | 1 + core/simulator/src/lib.rs | 207 +- core/simulator/src/replica.rs | 9 +- core/simulator/src/storage.rs | 585 ++++ core/simulator/src/storage/tests.rs | 2519 ++++++++++++++ examples/node/package-lock.json | 2 +- examples/python/uv.lock | 2 +- foreign/cpp/Cargo.toml | 2 +- foreign/cpp/MODULE.bazel | 2 +- foreign/cpp/include/iggy.hpp | 32 +- foreign/cpp/src/client.rs | 31 +- foreign/cpp/src/lib.rs | 5 +- foreign/cpp/tests/e2e/topic.cpp | 25 +- foreign/cpp/tests/unit/unit_tests.cpp | 23 +- .../Fixtures/VsrCluster.cs | 4 +- .../OptionsTests.cs | 21 +- .../Contracts/SendMessagesResponse.cs | 4 +- .../csharp/Iggy_SDK/Contracts/TopicOptions.cs | 42 +- foreign/csharp/Iggy_SDK/Enums/Durability.cs | 27 + .../Implementations/HttpMessageStream.cs | 2 + .../Implementations/TcpMessageStream.cs | 2 + foreign/csharp/Iggy_SDK/Iggy_SDK.csproj | 2 +- .../ClientTests/HttpTopicOptionsTests.cs | 17 +- .../MapperTests/BinaryMapper.cs | 4 +- .../OptionsBlockGoldenVectorTests.cs | 26 +- .../ResourceOptionsConverterTests.cs | 8 +- .../UtilityTests/TopicOptionsTests.cs | 20 +- foreign/go/README.md | 2 +- .../contracts/compression_algorithm_test.go | 14 +- foreign/go/contracts/topic_options.go | 19 +- foreign/go/contracts/topic_options_test.go | 24 +- foreign/go/contracts/version.go | 2 +- foreign/go/internal/command/topic.go | 20 +- foreign/go/internal/command/topic_test.go | 20 +- foreign/go/tests/e2e_helpers_test.go | 2 +- foreign/go/tests/e2e_test.go | 18 +- .../iggy-connector-flink/docker-compose.yml | 2 +- .../iggy-connector-pinot/docker-compose.yml | 2 +- .../pinot/IggyPinotIntegrationTest.java | 2 +- foreign/java/gradle.properties | 2 +- .../client/async/tcp/TopicsTcpClient.java | 8 +- .../blocking/http/TopicsHttpClient.java | 2 +- .../org/apache/iggy/topic/Durability.java | 35 + .../org/apache/iggy/topic/TopicOptions.java | 37 +- .../iggy/client/BaseIntegrationTest.java | 2 +- .../client/blocking/SystemClientBaseTest.java | 2 +- .../client/blocking/TopicsClientBaseTest.java | 12 +- .../blocking/tcp/TopicsTcpClientTest.java | 4 +- .../serde/OptionsBlockGoldenVectorTest.java | 16 +- .../apache/iggy/topic/TopicOptionsTest.java | 20 +- foreign/node/package-lock.json | 4 +- foreign/node/package.json | 2 +- .../src/wire/message/send-messages.command.ts | 5 +- foreign/node/src/wire/options.utils.test.ts | 15 +- .../wire/topic/create-topic.command.test.ts | 42 +- .../src/wire/topic/create-topic.command.ts | 27 +- foreign/node/src/wire/topic/index.ts | 2 +- foreign/node/src/wire/topic/topic.utils.ts | 7 + foreign/php/Cargo.toml | 2 +- foreign/php/iggy-php.stubs.php | 15 +- foreign/php/src/client.rs | 7 +- foreign/php/src/durability.rs | 38 + foreign/php/src/lib.rs | 2 + foreign/php/src/send_message.rs | 5 +- foreign/python/Cargo.toml | 4 +- foreign/python/apache_iggy.pyi | 21 +- foreign/python/pyproject.toml | 2 +- foreign/python/src/bin/stub_gen.rs | 8 +- foreign/python/src/client.rs | 17 +- foreign/python/src/durability.rs | 58 + foreign/python/src/lib.rs | 4 +- foreign/python/src/send_message.rs | 10 +- foreign/python/tests/test_topic.py | 21 +- foreign/python/uv.lock | 2 +- helm/charts/iggy/README.md | 13 +- helm/charts/iggy/README.md.gotmpl | 8 +- helm/charts/iggy/templates/_helpers.tpl | 19 +- helm/charts/iggy/values.yaml | 4 + scripts/ci/license-headers.sh | 6 +- scripts/ci/test-helm.sh | 47 +- .../run-standard-performance-suite.sh | 53 +- scripts/performance/utils.sh | 9 +- 228 files changed, 18802 insertions(+), 2758 deletions(-) create mode 100644 core/common/src/types/options/durability.rs create mode 100644 core/journal/src/durable_storage.rs create mode 100644 core/journal/src/partition_journal.rs create mode 100644 core/journal/src/partition_journal/segments.rs create mode 100644 core/partitions/src/install_backup.rs create mode 100644 core/partitions/src/persistence.rs create mode 100644 core/simulator/src/storage.rs create mode 100644 core/simulator/src/storage/tests.rs create mode 100644 foreign/csharp/Iggy_SDK/Enums/Durability.cs create mode 100644 foreign/java/java-sdk/src/main/java/org/apache/iggy/topic/Durability.java create mode 100644 foreign/php/src/durability.rs create mode 100644 foreign/python/src/durability.rs diff --git a/.github/actions/go/pre-merge/action.yml b/.github/actions/go/pre-merge/action.yml index 53a381edcd..1d06e1da92 100644 --- a/.github/actions/go/pre-merge/action.yml +++ b/.github/actions/go/pre-merge/action.yml @@ -134,7 +134,7 @@ runs: log-file: ${{ runner.temp }}/iggy-go-e2e.log wait-timeout-seconds: "90" env: - IGGY_SYSTEM_PATH: ${{ runner.temp }}/iggy-go-e2e-data + IGGY_PATH: ${{ runner.temp }}/iggy-go-e2e-data - name: Run e2e tests shell: bash @@ -190,7 +190,7 @@ runs: log-file: ${{ runner.temp }}/iggy-go-e2e-tls.log wait-timeout-seconds: "90" env: - IGGY_SYSTEM_PATH: ${{ runner.temp }}/iggy-go-e2e-tls-data + IGGY_PATH: ${{ runner.temp }}/iggy-go-e2e-tls-data IGGY_TCP_TLS_ENABLED: "true" IGGY_TCP_TLS_CERT_FILE: core/certs/iggy_cert.pem IGGY_TCP_TLS_KEY_FILE: core/certs/iggy_key.pem @@ -229,7 +229,7 @@ runs: wait-timeout-seconds: "90" env: IGGY_CLUSTER_ENABLED: "true" - IGGY_SYSTEM_PATH: ${{ runner.temp }}/iggy-go-cluster-0-data + IGGY_PATH: ${{ runner.temp }}/iggy-go-cluster-0-data - name: Start Iggy VSR cluster node 1 id: iggy-cluster-1 @@ -245,7 +245,7 @@ runs: wait-timeout-seconds: "90" env: IGGY_CLUSTER_ENABLED: "true" - IGGY_SYSTEM_PATH: ${{ runner.temp }}/iggy-go-cluster-1-data + IGGY_PATH: ${{ runner.temp }}/iggy-go-cluster-1-data - name: Run cluster e2e tests shell: bash diff --git a/.github/actions/java-gradle/pre-merge/action.yml b/.github/actions/java-gradle/pre-merge/action.yml index d5568883b5..db377bd43d 100644 --- a/.github/actions/java-gradle/pre-merge/action.yml +++ b/.github/actions/java-gradle/pre-merge/action.yml @@ -88,7 +88,7 @@ runs: cargo-bin: iggy-server wait-timeout-seconds: "90" env: - IGGY_SYSTEM_PATH: ${{ runner.temp }}/iggy-java-data + IGGY_PATH: ${{ runner.temp }}/iggy-java-data - name: Test if: inputs.task == 'test' @@ -166,7 +166,7 @@ runs: pid-file: ${{ runner.temp }}/iggy-server-tls.pid log-file: ${{ runner.temp }}/iggy-server-tls.log env: - IGGY_SYSTEM_PATH: ${{ runner.temp }}/iggy-java-tls-data + IGGY_PATH: ${{ runner.temp }}/iggy-java-tls-data IGGY_TCP_TLS_ENABLED: "true" IGGY_TCP_TLS_CERT_FILE: core/certs/iggy_cert.pem IGGY_TCP_TLS_KEY_FILE: core/certs/iggy_key.pem diff --git a/Cargo.lock b/Cargo.lock index 4f18e57409..007cf389c1 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -6811,7 +6811,7 @@ checksum = "cd62e6b5e86ea8eeeb8db1de02880a6abc01a397b2ebb64b5d74ac255318f5cb" [[package]] name = "iggy" -version = "0.11.0-edge.7" +version = "0.11.0-edge.8" dependencies = [ "async-broadcast", "async-dropper", @@ -6845,7 +6845,7 @@ dependencies = [ [[package]] name = "iggy-bench" -version = "0.6.0-edge.7" +version = "0.6.0-edge.8" dependencies = [ "async-trait", "bench-report", @@ -6902,7 +6902,7 @@ dependencies = [ [[package]] name = "iggy-cli" -version = "0.14.0-edge.7" +version = "0.14.0-edge.8" dependencies = [ "anyhow", "apple-native-keyring-store", @@ -7008,7 +7008,7 @@ dependencies = [ [[package]] name = "iggy-mcp" -version = "0.5.0-edge.6" +version = "0.5.0-edge.7" dependencies = [ "axum", "axum-server", @@ -7042,7 +7042,7 @@ dependencies = [ [[package]] name = "iggy_binary_protocol" -version = "0.11.0-edge.7" +version = "0.11.0-edge.8" dependencies = [ "aligned-vec", "bytemuck", @@ -7055,7 +7055,7 @@ dependencies = [ [[package]] name = "iggy_common" -version = "0.11.0-edge.7" +version = "0.11.0-edge.8" dependencies = [ "aes-gcm 0.11.1", "async-broadcast", @@ -7440,7 +7440,7 @@ dependencies = [ [[package]] name = "iggy_connector_sdk" -version = "0.4.0-edge.4" +version = "0.4.0-edge.5" dependencies = [ "anyhow", "apache-avro 0.22.0", @@ -7999,6 +7999,8 @@ dependencies = [ "compio", "futures", "iggy_binary_protocol", + "iggy_common", + "nix", "server_common", "tempfile", "tracing", @@ -12635,7 +12637,7 @@ dependencies = [ [[package]] name = "server" -version = "0.9.0-edge.7" +version = "0.9.0-edge.8" dependencies = [ "ahash 0.8.12", "argon2", @@ -12976,6 +12978,7 @@ dependencies = [ "tempfile", "tracing", "tracing-subscriber", + "twox-hash", ] [[package]] diff --git a/Cargo.toml b/Cargo.toml index b84d880748..cb03ebd191 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -199,13 +199,13 @@ hyper-util = { version = "0.1.20", features = ["server-auto", "service"] } iceberg = "0.10.1" iceberg-catalog-rest = "0.10.1" iceberg-storage-opendal = "0.10.1" -iggy = { path = "core/sdk", version = "0.11.0-edge.7" } -iggy-cli = { path = "core/cli", version = "0.14.0-edge.7" } +iggy = { path = "core/sdk", version = "0.11.0-edge.8" } +iggy-cli = { path = "core/cli", version = "0.14.0-edge.8" } iggy-gateway-kafka = { path = "gateways/kafka" } -iggy_binary_protocol = { path = "core/binary_protocol", version = "0.11.0-edge.7" } -iggy_common = { path = "core/common", version = "0.11.0-edge.7" } +iggy_binary_protocol = { path = "core/binary_protocol", version = "0.11.0-edge.8" } +iggy_common = { path = "core/common", version = "0.11.0-edge.8" } iggy_connector_doris_sink = { path = "core/connectors/sinks/doris_sink" } -iggy_connector_sdk = { path = "core/connectors/sdk", version = "0.4.0-edge.4" } +iggy_connector_sdk = { path = "core/connectors/sdk", version = "0.4.0-edge.5" } indexmap = "2.14.1" integration = { path = "core/integration" } ipnet = "2.12.1" diff --git a/README.md b/README.md index 3b00daf6c1..e78873f139 100644 --- a/README.md +++ b/README.md @@ -231,6 +231,18 @@ The configuration file is loaded from the current working directory, but you can When config file is not found, the default values from embedded `config.toml` file are used. +Topic creation accepts two independent policies: `durability` for message acknowledgments and `consumer_offset_durability` for explicit offset stores and deletes. Both default to `replicated`. This means VSR quorum commit without waiting for stable storage. `persisted` also requires recoverable stable-storage copies on the replication quorum. Both policies normally store data on disk. Poll auto-commit remains asynchronous and is not covered by the poll response's completion. + +The data directory is configured with `path` or `IGGY_PATH`. The layout beneath it is `streams//topics//partitions/`, with fixed directory names. + +The HTTP `Iggy-Durability` header reports `replicated` or `persisted` for awaited writes, and `none` for early dispatch acceptance. + +Segment flush thresholds control scheduling, independently of acknowledgment durability. + +Rust HTTP callers can use `HttpClient::send_messages_with_durability` to read the advertised guarantee alongside confirmations. + +The CLI exposes `--durability persisted` and `--consumer-offset-durability persisted` on `topic create`. Select either independently. The policy names describe completion guarantees and do not prescribe an I/O syscall. + For the detailed documentation of the configuration file, please refer to the [configuration](https://iggy.apache.org/docs/server/configuration) section. --- @@ -271,7 +283,7 @@ Start the server: `cargo run --bin iggy-server` -All the data used by the server will be persisted under the `local_data` directory by default, unless specified differently in the configuration (see `system.path` in `config.toml`). +All the data used by the server will be persisted under the `local_data` directory by default, unless specified differently in the configuration (see `path` in `config.toml`). One can use default root credentials with optional `--with-default-root-credentials`. This flag is equivalent to setting `IGGY_ROOT_USERNAME=iggy` and `IGGY_ROOT_PASSWORD=iggy`, plus @@ -291,7 +303,7 @@ You can also use environment variables to override any configuration setting: `IGGY_TCP_ADDRESS=127.0.0.1:8090 cargo run --bin iggy-server` - Set custom data path - `IGGY_SYSTEM_PATH=/data/iggy cargo run --bin iggy-server` + `IGGY_PATH=/data/iggy cargo run --bin iggy-server` - Enable HTTP transport `IGGY_HTTP_ENABLED=true cargo run --bin iggy-server` @@ -430,7 +442,7 @@ To benchmark the project, first build the project in release mode: cargo build --release ``` -Then, run the benchmarking app with the desired options: +Start `iggy-server` separately, then run the benchmarking app with the desired options: 1. Sending (writing) benchmark @@ -474,14 +486,26 @@ Then, run the benchmarking app with the desired options: cargo run --bin iggy-bench -r -- end-to-end-producing-consumer tcp ``` -These benchmarks would start the server with the default configuration, create a stream, topic and partition, and then send or poll the messages. The default configuration is optimized for the best performance, so you might want to tweak it for your needs. If you need more options, please refer to `iggy-bench` subcommands `help` and `examples`. +8. End to end producing and consuming through a consumer group: -For example, to run the benchmark for the already started server, provide the additional argument `--server-address 0.0.0.0:8090`. + ```bash + cargo run --bin iggy-bench -r -- end-to-end-producing-consumer-group tcp + ``` + +The benchmark connects to a running server and creates the streams, topics, and partitions needed by the selected workload. Use `iggy-bench --help` and `iggy-bench examples` for all benchmark variants, transports, and topic-option examples. Both message and consumer-offset durability independently default to `replicated`. + +For example, to run the benchmark for the already started server, provide the additional argument `--server-address 127.0.0.1:8090`. **Iggy is already capable of processing millions of messages per second at the microseconds range for p99+ latency** Depending on the hardware, transport protocol (`quic`, `websocket`, `tcp` or `http`) and payload size (`messages-per-batch * message-size`) you might expect **over 5000 MB/s (e.g. 5M of 1 KB msg/sec) throughput for writes and reads**. Please refer to the mentioned [benchmarking platform](https://benchmarks.iggy.apache.org) where you can browse the results achieved on the different hardware configurations, using the different Iggy server versions. +### Host preparation + +Check `io_uring` access, process limits, memory headroom, CPU/NUMA placement, and sustained disk/network capacity before comparing runs. Measure host-tuning changes with the same workload and durability policies. + +Use the [benchmark host checklist](core/bench/README.md#host-preparation) for practical setup and repeatable measurements. The [Linux tuning guide](https://iggy.apache.org/docs/server/linux-tuning) explains swappiness, huge pages, writeback, CPU placement, and networking, with commands and upstream references. + --- ## Contributing diff --git a/bdd/docker-compose.cluster.yml b/bdd/docker-compose.cluster.yml index df773b8a5e..d84da3475d 100644 --- a/bdd/docker-compose.cluster.yml +++ b/bdd/docker-compose.cluster.yml @@ -105,7 +105,7 @@ services: environment: <<: *cluster-topology RUST_LOG: info - IGGY_SYSTEM_PATH: local_data_leader + IGGY_PATH: local_data_leader IGGY_TCP_ADDRESS: 0.0.0.0:8091 IGGY_HTTP_ADDRESS: 0.0.0.0:3001 IGGY_QUIC_ADDRESS: 0.0.0.0:8081 @@ -131,7 +131,7 @@ services: environment: <<: *cluster-topology RUST_LOG: info - IGGY_SYSTEM_PATH: local_data_follower + IGGY_PATH: local_data_follower IGGY_TCP_ADDRESS: 0.0.0.0:8092 IGGY_HTTP_ADDRESS: 0.0.0.0:3002 IGGY_QUIC_ADDRESS: 0.0.0.0:8082 diff --git a/bdd/docker-compose.server.yml b/bdd/docker-compose.server.yml index c57b3cdd82..88868ace30 100644 --- a/bdd/docker-compose.server.yml +++ b/bdd/docker-compose.server.yml @@ -63,7 +63,7 @@ services: - RUST_LOG=info - IGGY_ROOT_USERNAME=iggy - IGGY_ROOT_PASSWORD=iggy - - IGGY_SYSTEM_PATH=local_data + - IGGY_PATH=local_data - IGGY_TCP_ADDRESS=0.0.0.0:8090 - IGGY_NODE_ADVERTISED_ADDRESS=iggy-server - IGGY_HTTP_ADDRESS=0.0.0.0:3000 diff --git a/bdd/python/uv.lock b/bdd/python/uv.lock index fad1a3fac5..2f3e7d0783 100644 --- a/bdd/python/uv.lock +++ b/bdd/python/uv.lock @@ -8,7 +8,7 @@ exclude-newer-span = "P7D" [[package]] name = "apache-iggy" -version = "0.9.0.dev7" +version = "0.9.0.dev8" source = { directory = "../../foreign/python" } [package.metadata] diff --git a/core/ai/mcp/Cargo.toml b/core/ai/mcp/Cargo.toml index 1d1953f9f2..688c650988 100644 --- a/core/ai/mcp/Cargo.toml +++ b/core/ai/mcp/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "iggy-mcp" -version = "0.5.0-edge.6" +version = "0.5.0-edge.7" description = "MCP Server for Iggy message streaming platform" edition = "2024" license = "Apache-2.0" diff --git a/core/bench/Cargo.toml b/core/bench/Cargo.toml index 7a3f1ce421..8dc2f65c1f 100644 --- a/core/bench/Cargo.toml +++ b/core/bench/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "iggy-bench" -version = "0.6.0-edge.7" +version = "0.6.0-edge.8" edition = "2024" license = "Apache-2.0" repository = "https://github.com/apache/iggy" diff --git a/core/bench/README.md b/core/bench/README.md index b3ae1ac4ef..66d1ab5ab0 100644 --- a/core/bench/README.md +++ b/core/bench/README.md @@ -5,3 +5,63 @@ The interactive Bench CLI allows you to perform various benchmarking on the Apac Iggy Bench CLI can be installed with `cargo install iggy-bench` and then simply accessed by typing `iggy-bench` in your terminal. ![CLI](../../assets/bench.png) + +The WebSocket transport command is `websocket`, with `ws` as a shorthand. + +Producer and consumer counts default to six. Pinned workloads also default to six streams. Override the counts for the workload and available CPUs. + +## Examples and topic options + +Start Iggy before running benchmarks. The tool connects to an existing server. Run `iggy-bench examples` for all eight benchmark kinds, their aliases, all four transports, and topic-option combinations. Global options precede the benchmark kind, kind-specific options follow it, and the server address follows the transport. + +Both `--durability` and `--consumer-offset-durability` independently default to `replicated`. The following runs use the same workload with different topic policies: + +```bash +# Both policies replicated, the default. +iggy-bench balanced-producer-and-consumer-group tcp + +# Persisted message acknowledgments with replicated offsets. +iggy-bench --durability persisted balanced-producer-and-consumer-group tcp + +# Replicated message acknowledgments with persisted offsets. +iggy-bench --consumer-offset-durability persisted balanced-producer-and-consumer-group tcp + +# Both policies persisted. +iggy-bench --durability persisted --consumer-offset-durability persisted balanced-producer-and-consumer-group tcp +``` + +These options apply when the benchmark creates topics. They do not change existing topics with `--reuse-streams`. Both policies normally write data to disk. `persisted` adds a stable-storage completion requirement. Consumer workloads using poll auto-commit still receive asynchronous poll responses, so this flag does not turn their poll latency into a measurement of acknowledged offset-store latency. + +`--messages-required-to-save` controls a topic's segment-flush cadence independently of durability. `--max-topic-size` and `--message-expiry` are kind-specific topic options. Use the selected kind's `--help` to inspect supported flags. OS writeback settings do not replace these policies. + +## Host preparation + +Use a dedicated server and a separate load generator where possible. If they share a host, assign disjoint CPU sets and leave CPU capacity for kernel and network work. Match the deployment's CPU topology, storage, replication-group size, and durability policies. + +| Area | Starting point | What to check | +| --- | --- | --- | +| Runtime access | Permit the service to create `io_uring` instances. | Startup diagnostics, container syscall policy, and effective process limits. | +| File descriptors and locked memory | Size limits for connections, partitions, and runtime allocations. | `/proc/PID/limits`, not just the shell's `ulimit`. Set systemd service limits explicitly. | +| Swap | With enough RAM and disk-backed swap, compare `vm.swappiness=10` against the existing value. | Swap activity, memory pressure, and p99 latency. This does not disable swap. | +| Huge pages | Start without an explicit reservation, then test a budgeted pool with a compatible allocator. | Actual page size, usage, NUMA placement, and memory left for page cache. | +| CPU placement | Use the server's `[sharding]` options within its allowed CPU set. | Per-core load, CPU steal time, cgroup throttling, and IRQ distribution. | +| Writeback | Start with the OS defaults. | Sustained disk latency and dirty-page buildup. Large percentage limits can mask short-run bottlenecks. | +| Networking | Start with the OS defaults. | Receive drops, retransmissions, connection backlogs, and bandwidth-delay requirements before increasing limits. | + +Read the [Linux tuning guide](https://iggy.apache.org/docs/server/linux-tuning) before reserving huge pages. `vm.nr_hugepages=2048` reserves 4 GiB only with 2 MiB pages. **`MIMALLOC_RESERVE_HUGE_OS_PAGES=2048` requests 2048 one-GiB pages, or 2 TiB.** They are different controls. The guide explains allocator configuration, THP, service limits, permissions, and reboot persistence. + +### Repeatable measurements + +1. Record the server and benchmark revisions, build profile, kernel, VM type, CPU allocation, allocator, memory limits, filesystem, and provisioned disk/network capacity. +2. Keep workload shape, both durability policies, replication-group size, and topic options identical between compared runs. Use an explicit warmup and repeat runs. +3. Observe `vmstat 1`, `iostat -xz 1`, `mpstat -P ALL 1`, memory/I/O pressure, and network errors. Confirm the client has spare CPU and network capacity. +4. Run long enough to include segment flushes, checkpoints, and sustained device limits. Distinguish warm-cache tests, cold-cache tests, and storage throughput. Do not drop caches during a production workload. +5. Change one host setting at a time, retain its previous value, and compare throughput, p50/p99/p99.9 latency, memory, and errors. Record successful changes in provisioning and recheck after reboot. + +The `output` subcommand records results and accepts context for the run: + +```bash +iggy-bench --warmup-time 10s --total-data 10GiB balanced-producer tcp output --identifier baseline --remark replicated-defaults +``` + +Choose `--total-data` for the duration and storage behavior being tested. This example is not a universal sufficient data volume. diff --git a/core/bench/dashboard/frontend/index.html b/core/bench/dashboard/frontend/index.html index 92fd763a23..5d3b066927 100644 --- a/core/bench/dashboard/frontend/index.html +++ b/core/bench/dashboard/frontend/index.html @@ -1,4 +1,4 @@ - + +Sends encode `Partitioning.PartitionId`, `Partitioning.Balanced` or +`Partitioning.MessageKey` in the payload. The server resolves the target +partition at admission. VSR works over TCP and TLS. It restricts `Client` to one pooled connection because authentication, request sequencing, and consumer-group assignments belong to one consensus session. Configurations requesting more than one pooled connection fail before a socket is opened. @@ -69,8 +68,9 @@ The client pings every `heartbeatInterval` milliseconds, 5000 by default, which keeps an idle session alive when the server's `[heartbeat]` eviction is enabled. `heartbeatInterval` also accepts a duration expression such as `"10s"` or `"1h 30m"`, like the Rust SDK. -The server evicts a connection silent for 36 s, which is 1.2 x its 30 s -heartbeat interval. Raising the client interval past that window, or setting it +With the default heartbeat settings, a group member becomes eligible for +eviction after 36 s of silence (1.2 x the 30 s interval); the verifier checks +once per interval. Raising the client interval past that window, or setting it to 0 to disable client heartbeats, exposes an idle consumer-group member to eviction; a connection holding no group membership is left alone. Any other unusable value is rejected instead of silently disabling the heartbeat. diff --git a/foreign/php/README.md b/foreign/php/README.md index 15007bb901..dc2ef121c3 100644 --- a/foreign/php/README.md +++ b/foreign/php/README.md @@ -11,7 +11,7 @@ future resolves; it does not provide fiber-aware or non-blocking I/O. ## Requirements - Rust and Cargo -- PHP 8.3 or newer with `php-config` +- PHP 8.3 or newer with `php-config` (non-thread-safe builds only, no ZTS) - `cargo-php` - Composer, for installing PHPUnit - Docker, for running the integration test server @@ -33,9 +33,12 @@ cargo build --release Generate IDE stubs after changing the exported PHP API: ```sh -cargo php stubs --manifest Cargo.toml -o iggy-php.stubs.php +cargo build +cargo php stubs target/debug/libiggy_php.so -o /tmp/iggy-php.stubs.php ``` +Stub generation requires a debug build. On macOS, use `libiggy_php.dylib`. +Preserve the existing Apache license header when updating `iggy-php.stubs.php`. The CI lint job regenerates this file and fails if the checked-in stubs drift from the Rust signatures. @@ -60,12 +63,18 @@ php -r 'var_dump(extension_loaded("iggy-php"));' ## Run Iggy +Use Iggy 0.9.0. When testing unreleased SDK changes, build the server from the +same source checkout. + ```sh docker run --rm --name iggy-php-test \ + --cap-add=SYS_NICE --security-opt seccomp=unconfined --ulimit memlock=-1:-1 \ -p 8090:8090 \ -p 3000:3000 \ + -e IGGY_TCP_ADDRESS=0.0.0.0:8090 -e IGGY_HTTP_ADDRESS=0.0.0.0:3000 \ -e IGGY_NODE_ADVERTISED_ADDRESS=localhost \ - apache/iggy:latest + -e IGGY_ROOT_USERNAME=iggy -e IGGY_ROOT_PASSWORD=iggy \ + apache/iggy:0.9.0 ``` You can also run a local server from the repository root: @@ -74,6 +83,10 @@ You can also run a local server from the repository root: cargo run --bin iggy-server -- --fresh --with-default-root-credentials ``` +The root variables bootstrap a new data directory; they do not replace stored +credentials. Environment credentials override `--with-default-root-credentials`. +Use `--fresh` only with disposable development data: it deletes local replica state. + The tests assume: - host: `127.0.0.1` @@ -100,7 +113,8 @@ $client->createStream($stream); $client->createTopic($stream, $topic, 1, null, null, null, null); $client->sendMessages($stream, $topic, $partitionId, [ - new \Iggy\SendMessage('hello from PHP'), + new \Iggy\SendMessage('hello from PHP 1'), + new \Iggy\SendMessage('hello from PHP 2'), ]); $messages = $client->pollMessages( @@ -119,7 +133,8 @@ foreach ($messages as $message) { Consumer group callbacks require a finite message limit. The partition id argument is ignored for a consumer group, since the member reads the partitions -the server assigns to it: +the server assigns to it. After the example above, each loop below consumes one +of its two messages: ```php consumerGroup( $consumer->consumeMessages( function (\Iggy\ReceiveMessage $message) use ($consumer): void { - process($message->payload()); + echo $message->payload(), PHP_EOL; $consumer->storeOffset($message->offset(), $message->partitionId()); }, - 100, + 1, ); foreach ($consumer->iterMessages() as $message) { - process($message->payload()); + echo $message->payload(), PHP_EOL; $consumer->storeOffset($message->offset(), $message->partitionId()); - if (shouldStop()) { - break; - } + break; } ``` @@ -174,10 +187,11 @@ composer install composer test ``` -Run Rust verification: +Run Rust verification with a matching PHP embedding library (`libphp`) available +to the linker and dynamic loader: ```sh -cargo test +cargo test --features ext-php-rs/embed ``` TLS tests are opt-in because they require a TLS-enabled Iggy server and certificate @@ -215,8 +229,8 @@ iggy+tcp://iggy:iggy@127.0.0.1:8090?tls=true&tls_domain=localhost&tls_ca_file=/p - `Iggy\SendMessage::payload` and `Iggy\ReceiveMessage::payload()` copy the payload bytes into a PHP string on each read. Cache large payloads in PHP if they will be read repeatedly. -- Large unsigned values that can overflow PHP integers, such as message checksums, - are returned as decimal strings. +- Message IDs and checksums are returned as decimal strings. Offset and timestamp + getters return PHP integers and are limited to `PHP_INT_MAX`. - `Iggy\Client::sendBinaryRequest(int $code, string $payload): string` sends a command code and payload and returns the raw response body. - `Iggy\Client` is synchronous and blocks the current PHP thread. diff --git a/foreign/php/src/client.rs b/foreign/php/src/client.rs index fbe8d0bbd5..6cef2060e3 100644 --- a/foreign/php/src/client.rs +++ b/foreign/php/src/client.rs @@ -64,6 +64,8 @@ impl IggyClient { /// Constructs a new IggyClient from a connection string. pub fn from_connection_string(connection_string: String) -> PhpResult { + // QUIC creates its endpoint before a blocking API call enters the runtime. + let _guard = runtime().enter(); let client = RustIggyClient::from_connection_string(&connection_string).map_err(to_php_exception)?; diff --git a/foreign/php/tests/IggySdkTest.php b/foreign/php/tests/IggySdkTest.php index a5e682de89..336fb206b8 100644 --- a/foreign/php/tests/IggySdkTest.php +++ b/foreign/php/tests/IggySdkTest.php @@ -30,6 +30,12 @@ final class IggySdkTest extends TestCase { + private const CREATE_USER_CODE = 33; + private const DELETE_USER_CODE = 34; + private const GET_ME_CODE = 20; + private const ACTIVE_USER_STATUS = 1; + private const STRING_IDENTIFIER_KIND = 2; + #[TestDox('A connected client can ping the server')] public function testPing(): void { @@ -67,6 +73,62 @@ public function testClientFromConnectionString(): void assert_true($client instanceof IggyClient); } + #[TestDox('Examples authenticate using literal environment credentials')] + public function testExamplesAuthenticateWithLiteralEnvironmentCredentials(): void + { + require_once __DIR__ . '/../../../examples/php/src/common.php'; + + $admin = new_client(); + $username = unique_name('php-example-user'); + $password = 'example % @:+= password'; + $request = chr(strlen($username)) . $username + . chr(strlen($password)) . $password + . chr(self::ACTIVE_USER_STATUS) . chr(0); + $created = $admin->sendBinaryRequest(self::CREATE_USER_CODE, $request); + $userId = unpack('VuserId', $created)['userId']; + $keys = ['IGGY_CONNECTION_STRING', 'IGGY_HOST', 'IGGY_PORT', 'IGGY_USERNAME', 'IGGY_PASSWORD']; + $previous = array_combine($keys, array_map('getenv', $keys)); + + try { + putenv('IGGY_CONNECTION_STRING'); + putenv('IGGY_HOST=' . server_host()); + putenv('IGGY_PORT=' . server_port()); + putenv('IGGY_USERNAME=' . $username); + putenv('IGGY_PASSWORD=' . $password); + + $client = iggy_client(); + $identity = $client->sendBinaryRequest(self::GET_ME_CODE, ''); + assert_same($userId, unpack('VclientId/VuserId', $identity)['userId']); + + putenv('IGGY_PASSWORD=incorrect-password'); + assert_instance_of( + \Iggy\Exception\AuthenticationException::class, + assert_throws(static fn () => iggy_client()), + ); + } finally { + foreach ($previous as $key => $value) { + putenv($value === false ? $key : $key . '=' . $value); + } + $admin->sendBinaryRequest( + self::DELETE_USER_CODE, + chr(self::STRING_IDENTIFIER_KIND) . chr(strlen($username)) . $username, + ); + } + } + + #[TestDox('Connection strings construct clients for every transport before connecting')] + public function testConnectionStringsConstructEveryTransport(): void + { + foreach ([ + 'iggy+tcp://iggy:iggy@127.0.0.1:8090', + 'iggy+quic://iggy:iggy@127.0.0.1:8080', + 'iggy+http://iggy:iggy@127.0.0.1:3000', + 'iggy+ws://iggy:iggy@127.0.0.1:8092', + ] as $connectionString) { + assert_instance_of(IggyClient::class, IggyClient::fromConnectionString($connectionString)); + } + } + #[TestDox('A stream can be created and fetched by name')] public function testCreateAndGetStream(): void { diff --git a/foreign/python/README.md b/foreign/python/README.md index b4a33a1167..1718d2d89b 100644 --- a/foreign/python/README.md +++ b/foreign/python/README.md @@ -28,14 +28,19 @@ pip install apache-iggy ### Prerequisites -Every installation below compiles the Rust extension, so you'll need: - - Python 3.10+ + +Published wheels include the Rust extension; installing a wheel does not require +Rust. Building from source and running the development checks below also requires: + - Rust toolchain: `curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh` - `uv`: `curl -LsSf https://astral.sh/uv/install.sh | sh` - All checks tooling from [CONTRIBUTING.md](https://github.com/apache/iggy/blob/master/CONTRIBUTING.md). - Docker +Use an SDK release compatible with your server. For unreleased changes, build +the SDK and server from the same source checkout. + ### Local Development **IMPORTANT: All commands are supposed to be ran from `foreign/python` unless it's specified to run in repository's root folder.** diff --git a/foreign/python/apache_iggy.pyi b/foreign/python/apache_iggy.pyi index 532c0d0802..1150fa018d 100644 --- a/foreign/python/apache_iggy.pyi +++ b/foreign/python/apache_iggy.pyi @@ -989,9 +989,9 @@ class IggyClient: ) -> IggyClient: r""" Constructs a new IggyClient from a TCP server address, a `TcpConfig`, a - `QuicConfig`, an `HttpConfig`, or a `WebSocketConfig`. This initializes a - new runtime for asynchronous operations. - Future versions might utilize asyncio for more Pythonic async. + `QuicConfig`, an `HttpConfig`, or a `WebSocketConfig`. Construction is + synchronous; async methods return asyncio awaitables backed by the shared + Tokio runtime. Args: conn: A `host:port` address, a `TcpConfig`, a `QuicConfig`, an @@ -1661,7 +1661,7 @@ class IggyClient: Returns: An awaitable that resolves to `SendMessagesResponse`. Its confirmations report the committed partition and batch base offset. The list is empty - when the server reports no offsets, including on the legacy server. + when the server reports no offsets. Raises: ValueError: If a string stream or topic identifier is invalid. @@ -2332,9 +2332,6 @@ class SendMessagesConfirmation: Confirmation follows VSR quorum commit. A topic with persisted message durability also waits for recoverable stable-storage copies on the quorum. - - The legacy server confirms nothing, so its confirmation list is empty - and this value is never reached. """ @typing.final @@ -2347,9 +2344,8 @@ class SendMessagesResponse: r""" Gets the commit confirmations, one per partition the batch was written to. - The list is empty when the server reports no offsets, and the legacy - server never reports any, so branch on it being empty rather than - indexing into it. + The list is empty when the server reports no offsets, so check whether + it is empty before indexing into it. A reported `base_offset` never implies uniqueness, because delivery is at-least-once and an earlier retry may already have committed the same diff --git a/foreign/python/src/client.rs b/foreign/python/src/client.rs index 6ce2e8c2a8..40e53bb0cd 100644 --- a/foreign/python/src/client.rs +++ b/foreign/python/src/client.rs @@ -94,9 +94,9 @@ fn resolve_topic_params( #[pymethods] impl IggyClient { /// Constructs a new IggyClient from a TCP server address, a `TcpConfig`, a - /// `QuicConfig`, an `HttpConfig`, or a `WebSocketConfig`. This initializes a - /// new runtime for asynchronous operations. - /// Future versions might utilize asyncio for more Pythonic async. + /// `QuicConfig`, an `HttpConfig`, or a `WebSocketConfig`. Construction is + /// synchronous; async methods return asyncio awaitables backed by the shared + /// Tokio runtime. /// /// Args: /// conn: A `host:port` address, a `TcpConfig`, a `QuicConfig`, an @@ -1277,7 +1277,7 @@ impl IggyClient { /// Returns: /// An awaitable that resolves to `SendMessagesResponse`. Its confirmations /// report the committed partition and batch base offset. The list is empty - /// when the server reports no offsets, including on the legacy server. + /// when the server reports no offsets. /// /// Raises: /// ValueError: If a string stream or topic identifier is invalid. diff --git a/foreign/python/src/send_message.rs b/foreign/python/src/send_message.rs index f0eba1a8d8..8cb940fc40 100644 --- a/foreign/python/src/send_message.rs +++ b/foreign/python/src/send_message.rs @@ -148,9 +148,6 @@ impl SendMessagesConfirmation { /// /// Confirmation follows VSR quorum commit. A topic with persisted message /// durability also waits for recoverable stable-storage copies on the quorum. - /// - /// The legacy server confirms nothing, so its confirmation list is empty - /// and this value is never reached. #[getter] pub fn base_offset(&self) -> u64 { self.inner.base_offset @@ -175,9 +172,8 @@ impl From for SendMessagesResponse { impl SendMessagesResponse { /// Gets the commit confirmations, one per partition the batch was written to. /// - /// The list is empty when the server reports no offsets, and the legacy - /// server never reports any, so branch on it being empty rather than - /// indexing into it. + /// The list is empty when the server reports no offsets, so check whether + /// it is empty before indexing into it. /// /// A reported `base_offset` never implies uniqueness, because delivery is /// at-least-once and an earlier retry may already have committed the same diff --git a/web/README.md b/web/README.md index e3dcc4224c..ef213b7d7a 100644 --- a/web/README.md +++ b/web/README.md @@ -8,11 +8,11 @@ This project hosts the web user interface for Apache Iggy. The web UI is built u The Iggy Web UI provides a user-friendly panel for managing various aspects of the Iggy platform, including streams, topics, partitions, and more. -The [docker image](https://hub.docker.com/r/apache/iggy-web-ui) is available, and can be fetched via `docker pull apache/iggy-web-ui`. +The [docker image](https://hub.docker.com/r/apache/iggy-web-ui) is available, and can be fetched via `docker pull apache/iggy-web-ui:edge`. ## Tooling -- Node.js: use a version supported by the current frontend toolchain, `^20.19.0 || ^22.13.0 || >=24`. `22.13+` LTS is the safest default. +- Node.js: use a version supported by the current frontend toolchain, `^20.19.0 || ^22.13.0 || >=24`. - Package manager: `npm` - `pnpm` and `yarn` are not part of the supported workflow for this package. CI, Docker builds, and the committed lockfile use `npm`. @@ -21,12 +21,14 @@ The [docker image](https://hub.docker.com/r/apache/iggy-web-ui) is available, an 1. **Run Iggy server first** ```sh - docker pull apache/iggy:latest + docker pull apache/iggy:edge ``` ```sh - docker run -p 3000:3000 -p 8090:8090 \ - -e IGGY_NODE_ADVERTISED_ADDRESS=localhost apache/iggy:latest + docker run --security-opt seccomp=unconfined -p 3000:3000 -p 8090:8090 \ + -e IGGY_ROOT_USERNAME=iggy -e IGGY_ROOT_PASSWORD=iggy \ + -e IGGY_HTTP_ADDRESS=0.0.0.0:3000 -e IGGY_TCP_ADDRESS=0.0.0.0:8090 \ + -e IGGY_NODE_ADVERTISED_ADDRESS=localhost apache/iggy:edge ``` 2. **Clone the repository:** @@ -38,7 +40,7 @@ The [docker image](https://hub.docker.com/r/apache/iggy-web-ui) is available, an 3. **Build the project:** ```sh - cd web + cd iggy/web npm ci ``` @@ -54,17 +56,13 @@ The [docker image](https://hub.docker.com/r/apache/iggy-web-ui) is available, an npm run dev -- --host --port 3333 ``` - **If Iggy server was run using cargo directly we need to change PUBLIC_IGGY_API_URL env in web ui root folder to:** + Set `PUBLIC_IGGY_API_URL` in `web/.env` to the HTTP address reachable by your browser, for example: ```sh PUBLIC_IGGY_API_URL=http://127.0.0.1:3000 ``` - **instead of** - - ```sh - PUBLIC_IGGY_API_URL=http://localhost:3000 - ``` + This applies to both container and source-built servers. Use the host-published HTTP port for a container. ## Roadmap diff --git a/web/package.json b/web/package.json index efeea0ac6b..10f0324e4a 100644 --- a/web/package.json +++ b/web/package.json @@ -10,6 +10,7 @@ "server": "vite server", "check": "svelte-kit sync && svelte-check --tsconfig ./tsconfig.json", "check:watch": "svelte-kit sync && svelte-check --tsconfig ./tsconfig.json --watch", + "test": "node --experimental-strip-types --test", "lint": "eslint . && prettier --check .", "format": "prettier --write ." }, diff --git a/web/src/lib/api/ApiSchema.ts b/web/src/lib/api/ApiSchema.ts index cc6c3cdd1e..8af3153403 100644 --- a/web/src/lib/api/ApiSchema.ts +++ b/web/src/lib/api/ApiSchema.ts @@ -123,7 +123,6 @@ type Topics = message_expiry: number; name: string; partitions_count: number; - stream_id: number; }; } | { @@ -131,9 +130,9 @@ type Topics = path: `/streams/${number}/topics/${number}`; body: { name: string; - message_expiry: number; - compression_algorithm: number; - max_topic_size: number; + message_expiry?: number; + compression_algorithm?: 'none' | 'gzip'; + max_topic_size?: number; }; } | { diff --git a/web/src/lib/components/Breadcrumbs.svelte b/web/src/lib/components/Breadcrumbs.svelte index 09150949f1..e697376183 100644 --- a/web/src/lib/components/Breadcrumbs.svelte +++ b/web/src/lib/components/Breadcrumbs.svelte @@ -24,9 +24,10 @@ under the License. import { isNumber } from '$lib/utils/parsers'; import { twMerge } from 'tailwind-merge'; import { resolve } from '$app/paths'; + import type { Pathname } from '$app/types'; type Crumb = { - path: string; + path: Pathname; label: string; }; @@ -50,7 +51,7 @@ under the License. } function formatPathSegment(segment: string, index: number, parts: string[]): Crumb { - const path = `/dashboard/${parts.slice(0, index + 1).join('/')}`; + const path = `/dashboard/${parts.slice(0, index + 1).join('/')}` as Pathname; if (isNumber(segment)) { const prevSegment = parts[index - 1]; diff --git a/web/src/lib/components/Modals/DeletePartitionsModal.svelte b/web/src/lib/components/Modals/DeletePartitionsModal.svelte index 31bcd76660..53b0f65f09 100644 --- a/web/src/lib/components/Modals/DeletePartitionsModal.svelte +++ b/web/src/lib/components/Modals/DeletePartitionsModal.svelte @@ -130,7 +130,7 @@ under the License. }); let messagesToDelete = $derived( - arraySum(topic.partitions.slice($form.partitions_count).map((p) => p.messagesCount)) + arraySum(topic.partitions.slice(-$form.partitions_count).map((p) => p.messagesCount)) ); diff --git a/web/src/lib/components/Modals/StreamSettingsModal.svelte b/web/src/lib/components/Modals/StreamSettingsModal.svelte index 1b63b1fe36..8d8c6fcc6c 100644 --- a/web/src/lib/components/Modals/StreamSettingsModal.svelte +++ b/web/src/lib/components/Modals/StreamSettingsModal.svelte @@ -127,7 +127,7 @@ under the License. confirmationOpen = false; if (result) { - const { ok } = await fetchRouteApi({ + const { data, ok } = await fetchRouteApi({ method: 'DELETE', path: `/streams/${stream.id}` }); diff --git a/web/src/lib/components/Modals/TopicSettingsModal.svelte b/web/src/lib/components/Modals/TopicSettingsModal.svelte index 8fcd0b5fbf..fc962d265c 100644 --- a/web/src/lib/components/Modals/TopicSettingsModal.svelte +++ b/web/src/lib/components/Modals/TopicSettingsModal.svelte @@ -31,6 +31,7 @@ under the License. import { fetchRouteApi } from '$lib/api/fetchRouteApi'; import { goto } from '$app/navigation'; import { resolve } from '$app/paths'; + import type { Pathname } from '$app/types'; import { showToast } from '../AppToasts.svelte'; import ModalConfirmation from '../ModalConfirmation.svelte'; import { browser } from '$app/environment'; @@ -41,7 +42,7 @@ under the License. interface Props { topic: TopicDetails; closeModal: CloseModalFn; - onDeleteRedirectPath: string; + onDeleteRedirectPath: Pathname; } let { topic, closeModal, onDeleteRedirectPath }: Props = $props(); @@ -71,9 +72,7 @@ under the License. path: `/streams/${+page.params.streamId}/topics/${topic.id}`, body: { name: form.data.name, - message_expiry: form.data.message_expiry, - compression_algorithm: topic.compressionAlgorithm, - max_topic_size: 0 + message_expiry: form.data.message_expiry } }); diff --git a/web/src/lib/utils/base64Utils.test.mjs b/web/src/lib/utils/base64Utils.test.mjs new file mode 100644 index 0000000000..316a94a276 --- /dev/null +++ b/web/src/lib/utils/base64Utils.test.mjs @@ -0,0 +1,32 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +import assert from 'node:assert/strict'; +import { test } from 'node:test'; +import { decodeBase64 } from './base64Utils.ts'; + +test('decode message payloads as UTF-8 text', () => { + for (const payload of ['plain text', 'Kraków 🦀', '{"city":"Kraków"}', 'Kraków']) { + const encoded = Buffer.from(payload, 'utf8').toString('base64'); + assert.equal(decodeBase64(encoded), payload); + } +}); + +test('reject invalid Base64 and preserve empty text', () => { + assert.equal(decodeBase64('%%%'), null); + assert.equal(decodeBase64(''), ''); +}); diff --git a/web/src/lib/utils/base64Utils.ts b/web/src/lib/utils/base64Utils.ts index 325d650ae8..9b5988900f 100644 --- a/web/src/lib/utils/base64Utils.ts +++ b/web/src/lib/utils/base64Utils.ts @@ -17,7 +17,8 @@ export function decodeBase64(str: string): string | null { try { - return atob(str); + const bytes = Uint8Array.from(atob(str), (character) => character.charCodeAt(0)); + return new TextDecoder().decode(bytes); } catch { return null; } diff --git a/web/src/routes/dashboard/streams/[streamId=i32]/topics/[topicId=i32]/+page.svelte b/web/src/routes/dashboard/streams/[streamId=i32]/topics/[topicId=i32]/+page.svelte index 05fbe8469a..c151d11574 100644 --- a/web/src/routes/dashboard/streams/[streamId=i32]/topics/[topicId=i32]/+page.svelte +++ b/web/src/routes/dashboard/streams/[streamId=i32]/topics/[topicId=i32]/+page.svelte @@ -39,7 +39,7 @@ under the License. let { data }: Props = $props(); let topic = $derived(data.topic); - let prevPage = $derived(`/dashboard/streams/${page.params.streamId}/`); + let prevPage = $derived(typedRoute(`/dashboard/streams/${+(page.params.streamId || '')}`));
From bab0e331ed4443112f0bb779c0675d586bd5b300 Mon Sep 17 00:00:00 2001 From: Hubert Gruszecki Date: Sun, 13 Sep 2026 12:01:40 +0200 Subject: [PATCH 123/182] fix(connectors): report failures and correct format conversion (#4152) The runtime can count failed sink deliveries as processed messages, misclassify transformation errors and leave failed instances in running metrics. Protobuf and Avro conversion can corrupt encoded data or apply mappings incorrectly. Inspect FFI outcomes, correct error and lifecycle accounting, redact state URL secrets and run HTTP telemetry on the existing runtime. Repair codec and transform behavior with regression coverage and matching configuration guidance. Failed sink batches are logged and counted as errors. Runtime batch replay is not introduced by these fixes. Validation: formatting, manifest sorting, Clippy, build, Markdown and TOML checks passed. Selected crate suites passed 470 tests, and runtime/HTTP integration passed 32 tests. CI owns the remaining validation. --- core/connectors/README.md | 49 +- core/connectors/runtime/README.md | 49 +- core/connectors/runtime/src/log.rs | 117 ++- core/connectors/runtime/src/manager/sink.rs | 55 +- core/connectors/runtime/src/manager/source.rs | 40 + core/connectors/runtime/src/sink.rs | 24 +- core/connectors/runtime/src/state.rs | 2 +- core/connectors/runtime/src/state/http.rs | 70 +- core/connectors/sdk/README.md | 272 ++++--- core/connectors/sdk/src/decoders/proto.rs | 136 ++-- core/connectors/sdk/src/encoders/proto.rs | 206 ++++-- core/connectors/sdk/src/lib.rs | 2 +- .../sdk/src/transforms/avro_convert.rs | 48 +- .../sdk/src/transforms/filter_fields.rs | 6 +- .../sdk/src/transforms/json/filter_fields.rs | 53 ++ .../sdk/src/transforms/proto_convert.rs | 199 +++-- .../sdk/tests/protobuf_integration.rs | 683 +++++++++++++++++- core/connectors/sinks/http_sink/README.md | 115 +-- core/connectors/sinks/http_sink/config.toml | 2 +- core/connectors/sinks/http_sink/src/lib.rs | 9 +- .../sources/random_source/README.md | 10 +- .../sources/random_source/src/lib.rs | 31 +- .../connectors/fixtures/http/container.rs | 23 + .../tests/connectors/http/http_sink.rs | 124 +++- .../connectors/runtime/error_isolation.rs | 82 ++- .../tests/connectors/runtime/http_state.rs | 32 +- .../runtime/sink_transform_error.toml | 20 + .../sink_transform_error_config/stdout.toml | 47 ++ 28 files changed, 2111 insertions(+), 395 deletions(-) create mode 100644 core/integration/tests/connectors/runtime/sink_transform_error.toml create mode 100644 core/integration/tests/connectors/runtime/sink_transform_error_config/stdout.toml diff --git a/core/connectors/README.md b/core/connectors/README.md index f88b3d2554..821a3cd909 100644 --- a/core/connectors/README.md +++ b/core/connectors/README.md @@ -2,7 +2,7 @@ The highly performant and modular runtime for statically typed, yet dynamically loaded connectors. Ingest the data from the external sources and push it further to the Iggy streams, or fetch the data from the Iggy streams and push it further to the external sources. Create your own Rust plugins by simply implementing either the `Source` or `Sink` trait and build custom pipelines for the data processing. -The [docker image](https://hub.docker.com/r/apache/iggy-connect) is available, and can be fetched via `docker pull apache/iggy-connect`. +The [docker image](https://hub.docker.com/r/apache/iggy-connect) is available, and can be fetched via `docker pull apache/iggy-connect:edge`. ## Features @@ -20,24 +20,45 @@ The [docker image](https://hub.docker.com/r/apache/iggy-connect) is available, a ## Quick Start -1. Build the project in release mode (or debug, and update the connectors paths in the config accordingly), and make sure that the plugins specified in `core/connectors/runtime/example_config/connectors/` directory under `path` are available. The configuration must be provided in `toml` format, with files following the `{connector_name}_{type}[_v{N}].toml` naming convention. +Run these commands from the root of the same Iggy source checkout used for the server and plugins. This guide targets server 0.9.0, including its edge builds. -2. Run `docker compose up -d` from `/examples/rust/src/sink-data-producer` which will start the Quickwit server to be used by an example sink connector. At this point, you can access the Quickwit UI at [http://localhost:7280](http://localhost:7280) - check this dashboard again later on, after the `events` index will be created. +1. Build the server, CLI, runtime and quick-start plugins: -3. Set environment variable `IGGY_CONNECTORS_CONFIG_PATH=core/connectors/runtime/example_config/config.toml` (adjust the path as needed) pointing to the runtime configuration file. + ```bash + cargo build --release -p server -p iggy-cli -p iggy-connectors \ + -p iggy_connector_random_source -p iggy_connector_stdout_sink \ + -p iggy_connector_quickwit_sink + ``` + + For a debug build, omit `--release` and replace `target/release` with `target/debug` in both the commands and plugin paths. Make sure that the plugins specified in `core/connectors/runtime/example_config/connectors/` directory under `path` are available. The configuration must be provided in `toml` format. + The example directory also enables connectors for ClickHouse, Delta Lake, Apache Doris, Apache Iceberg, and InfluxDB. Without their backing services (or their compiled plugins) these are reported with the `Error` status, but they don't block the remaining connectors. Set `enabled = false` in their files to skip them entirely. + +2. Run `docker compose -f examples/rust/src/sink-data-producer/docker-compose.yml up -d`, which will start the Quickwit server to be used by an example sink connector. At this point, you can access the Quickwit UI at [http://localhost:7280](http://localhost:7280) - check this dashboard again later on, after the `events` index will be created. + +3. In the terminal that will run the connectors, set the runtime configuration path: + + ```bash + export IGGY_CONNECTORS_CONFIG_PATH=core/connectors/runtime/example_config/config.toml + ``` + +4. Start the Iggy server in a separate terminal with credentials matching the sample connector configuration: + + ```bash + IGGY_ROOT_USERNAME=iggy IGGY_ROOT_PASSWORD=iggy cargo run --bin iggy-server --release + ``` -4. Start the Iggy server and invoke the following commands via Iggy CLI to create the example streams and topics used by the sample connectors. + With the server running, create the example streams and topics using the CLI from this checkout. An existing server must have these credentials, or you must adjust the commands and connector configuration to match it. ```bash - iggy --username iggy --password iggy stream create example_stream - iggy --username iggy --password iggy topic create example_stream example_topic 1 none 1d - iggy --username iggy --password iggy stream create qw - iggy --username iggy --password iggy topic create qw records 1 none 1d + target/release/iggy --username iggy --password iggy stream create example_stream + target/release/iggy --username iggy --password iggy topic create example_stream example_topic 1 none 1d + target/release/iggy --username iggy --password iggy stream create qw + target/release/iggy --username iggy --password iggy topic create qw records 1 none 1d ``` -5. Execute `cargo run --example sink-data-producer -r` which will start the example data producer application, sending the messages to previously created `qw` stream and `records` topic (this will be used by the Quickwit sink connector). +5. Execute `cargo run --example sink-data-producer --release`, which sends 100 batches of messages to previously created `qw` stream and `records` topic (this will be used by the Quickwit sink connector). -6. Start the connector runtime `cargo run --bin iggy-connectors -r` - you should be able to browse Quickwit UI with records being constantly added to the `events` index. At the same time, you should see the new messages being added to the `example` stream and `topic1` topic by the test source connector - you can use Iggy Web UI to browse the data. The messages will have applied the basic fields transformations. +6. Start the connector runtime `cargo run --bin iggy-connectors --release` in the terminal configured in step 3. The Quickwit sink indexes the produced records in the `events` index. At the same time, you should see the new messages being added to the `example_stream` stream and `example_topic` topic by the Random source connector - you can [start the Iggy Web UI](https://iggy.apache.org/docs/web_ui/start) to browse the data. The messages will have applied the basic fields transformations. ## Runtime @@ -94,7 +115,7 @@ Each sink should have its own, custom configuration, which is passed along with ## Source -Sources are responsible for producing the messages to the configured stream(s) and topic(s). For example, the Test source connector will generate the random messages that will be then sent to the configured stream and topic. +Sources produce messages to an Iggy stream and topic. Configure one `[[streams]]` entry per source instance: the runtime retains only the last configured producer. For example, the Random source generates messages for that stream and topic. Please refer to the **[Source documentation](https://github.com/apache/iggy/tree/master/core/connectors/sources)** for the details about the configuration and the sample implementation. @@ -106,10 +127,10 @@ Please refer to the **[Source documentation](https://github.com/apache/iggy/tree ## Building the connectors -New connector can be built simply by implementing either `Sink` or `Source` trait. Please check the **[sink](https://github.com/apache/iggy/tree/master/core/connectors/sinks)** or **[source](https://github.com/apache/iggy/tree/master/core/connectors/sources)** documentation, as well as the existing examples under `/sinks` and `/sources` directories. +New connector can be built simply by implementing either `Sink` or `Source` trait. Please check the **[sink](https://github.com/apache/iggy/tree/master/core/connectors/sinks)** or **[source](https://github.com/apache/iggy/tree/master/core/connectors/sources)** documentation, as well as the existing examples under `core/connectors/sinks` and `core/connectors/sources`. ## Transformations -Field transformations (depending on the supported payload formats) can be applied to the messages either before they are sent to the specified topic (e.g. when produced by the source connectors), or before consumed by the sink connectors. To add the new transformation, simply implement the `Transform` trait and extend the existing `load` function. Each transform may have its own, custom configuration. +Field transformations (depending on the supported payload formats) can be applied to the messages either before they are sent to the specified topic (e.g. when produced by the source connectors), or before consumed by the sink connectors. To add a transformation, implement the `Transform` trait in the SDK, add its `TransformType` variant and extend `transforms::from_config`. Each transform may have its own, custom configuration. To find out more about the transforms, stream decoders or encoders, please refer to the **[SDK documentation](https://github.com/apache/iggy/tree/master/core/connectors/sdk)**. diff --git a/core/connectors/runtime/README.md b/core/connectors/runtime/README.md index ec824abda6..37f8e54342 100644 --- a/core/connectors/runtime/README.md +++ b/core/connectors/runtime/README.md @@ -2,19 +2,19 @@ Runtime is responsible for managing the lifecycle of the connectors and providing the necessary infrastructure for the connectors to run. -The runtime uses a shared [Tokio runtime](https://tokio.rs) to manage the asynchronous tasks and events across all connectors. Additionally, it has built-in support for logging via [tracing](https://docs.rs/tracing/latest/tracing/) crate. +The runtime uses a shared [Tokio runtime](https://tokio.rs) for its connector-management and forwarding tasks. Each loaded plugin library also has an SDK Tokio runtime shared by its instances. Additionally, it has built-in support for logging via [tracing](https://docs.rs/tracing/latest/tracing/) crate. The connector are implemented as Rust libraries, and these are loaded dynamically during the runtime initialization process. -Internally, [dlopen2](https://github.com/OpenByteDev/dlopen2) provides a safe and efficient way of loading the plugins via C FFI. +Internally, [dlopen2](https://github.com/OpenByteDev/dlopen2) loads plugin libraries and resolves their C FFI symbols. Plugins execute inside the runtime process. By default, runtime will look for the configuration file, to decide which connectors to load and how to configure them. -To start the connector runtime, simply run `cargo run --bin iggy-connectors`. +Set the broker credentials and connector configuration directory before starting the runtime. The embedded default has an empty connector directory and cannot start unchanged. Follow the [connector quick start](../README.md#quick-start) for a complete setup. -The [docker image](https://hub.docker.com/r/apache/iggy-connect) is available, and can be fetched via `docker pull apache/iggy-connect`. +The [docker image](https://hub.docker.com/r/apache/iggy-connect) is available, and can be fetched via `docker pull apache/iggy-connect:edge`. -The minimal viable configuration requires at least the Iggy credentials to create 2 separate instances of producer & consumer connections, the state directory path where source connectors can store their optional state, and the connectors configuration provider settings. +The runtime opens two Iggy TCP clients, one for producers and one for consumers. Set credentials matching the broker and a connector configuration provider. Omitted settings use the embedded defaults, including file-based source state storage. Save this example as `connectors.toml` in the repository root and replace `path/to/connectors` with your connector configuration directory. ```toml [iggy] @@ -39,14 +39,20 @@ config_dir = "path/to/connectors" format = "text" # Options: "text" (default), "json" ``` -The path to the configuration can be overridden by `IGGY_CONNECTORS_CONFIG_PATH` environment variable. Each configuration section can be also additionally updated by using the following convention `IGGY_CONNECTORS_SECTION_NAME.KEY_NAME` e.g. `IGGY_CONNECTORS_IGGY_USERNAME` and so on. +Start it from the repository root: + +```bash +IGGY_CONNECTORS_CONFIG_PATH=connectors.toml cargo run --bin iggy-connectors +``` + +Supported scalar fields and indexed list entries use environment variables with nested keys joined by underscores, for example `IGGY_CONNECTORS_IGGY_USERNAME`. Header and URL-template maps are configured in TOML. The runtime loads the first `.env` file found in the working directory or its parents, or the file specified by `IGGY_CONNECTORS_ENV_PATH`. ## State storage -Source connectors checkpoint their progress (an opaque byte blob) through the runtime's state storage. The backend is selected via `state.storage`: +Source plugins can supply optional checkpoint bytes. The runtime stores these opaque bytes using the backend selected by `state.storage`: -- `file` (default): one file per source at `{state.path}/source_{key}.state`, written crash-atomically. Ties the cursor to the local disk. -- `http`: one resource per source at `{state.http.url}/source_{key}` on any HTTP-speaking store (a sidecar in front of a database, an object-store gateway, a coordination service). Cursors survive node replacement and failover to another runtime instance. +- `file` (default): one file at `{state.path}/source_{key}.state` for each source that supplies checkpoints. Writes use a temporary file, file synchronization and atomic rename. On Unix, the parent directory is synchronized too. Ties the cursor to the local disk. +- `http`: one resource per source at `{state.http.url}/source_{key}` on any HTTP-speaking store (a sidecar in front of a database, an object-store gateway, a coordination service). A replacement runtime can read the same checkpoint from that state server; its durability and availability depend on the server. ```toml [state] @@ -54,7 +60,7 @@ path = "local_state" # used by storage = "file" storage = "http" # "file" | "http" [state.http] -url = "http://127.0.0.1:8080/connectors/state" # base URL, no trailing slash +url = "http://127.0.0.1:8080/connectors/state" load_method = "get" # "get" (default) | "post" save_method = "put" # "put" (default) | "post" | "patch" timeout = "5s" @@ -83,7 +89,7 @@ The configured base URL may contain a query string, which is preserved when `sou - Every read uses the configured `load_method` and remembers the returned `ETag`. Every write uses the configured `save_method` and is conditional: `If-Match: ` when a version is tracked, `If-None-Match: *` for the first-ever write. There is no unconditional overwrite path. - Every write carries an `Idempotency-Key` header, minted once per logical save and reused byte-identically across that save's retries, so a server that committed a write but lost the response can replay the original outcome instead of failing the retry with a spurious `412`. -- State is sent and returned as opaque MessagePack bytes with `Content-Type: application/octet-stream`. The runtime never converts connector state to JSON, so each source retains its own compact state schema. +- State is sent and returned as opaque bytes. The SDK provides MessagePack helpers. Writes use the bytes unchanged with `Content-Type: application/octet-stream`. The runtime never converts connector state to JSON, so each source retains its own compact state schema. - `425`/`429`/`503`/`5xx`, timeouts and connect failures are retried with exponential backoff (honoring `Retry-After`, capped at `max_backoff`) and classified transient when exhausted: the batch is Nacked and the plugin re-polls. - `412`/`409` (version conflict), `401`/`403` (authorization lost) and protocol violations are permanent: the provider latches and every later save fails fast without touching the network, until the connector is restarted. A permanent error means another writer took over or this writer's authority was revoked - retrying cannot help and would mask the original error. - Durability is the server's durability. The runtime guarantees only that the checkpoint is not advanced (the batch is not Acked) unless the server confirmed the write. @@ -119,7 +125,7 @@ The runtime supports two types of configuration providers for managing connector ### Local File Provider -The default configuration provider reads connector configurations from local files. Each connector (source or sink) is configured in its own separate file within the directory specified by `connectors.config_dir`. If `config_dir` is empty or the directory doesn't exist, no connectors will be loaded. +The default configuration provider reads connector configurations from local files. Each connector (source or sink) is configured in its own separate file within the directory specified by `connectors.config_dir`. An empty `config_dir` is a fatal startup error. A missing directory is created automatically with a warning, and no connectors are loaded from it. Only nonhidden `*.toml` files directly inside the directory are read; `Cargo.toml` is skipped. ```toml [connectors] @@ -129,7 +135,7 @@ config_dir = "path/to/connectors" ### HTTP Configuration Provider -The HTTP configuration provider allows the runtime to fetch connector configurations from a remote HTTP/REST API. This enables centralized configuration management and dynamic configuration updates. +The HTTP configuration provider allows the runtime to fetch connector configurations from a remote HTTP/REST API. The provider fetches active configurations at startup and handles configuration operations requested through the runtime API. It does not periodically poll for remote changes. ```toml [connectors] @@ -142,7 +148,7 @@ api-key = "your-api-key" [connectors.retry] enabled = true -max_attempts = 3 +max_attempts = 3 # Retries after the first request, up to four requests total initial_backoff = "1 s" max_backoff = "30 s" backoff_multiplier = 2 @@ -166,8 +172,8 @@ error_path = "error" # Path to error in response (e.g., {"error": "..."}) - **timeout** (optional): HTTP request timeout (default: 10s) - **request_headers** (optional): Custom headers to include in all HTTP requests (e.g., authentication headers) - **url_templates** (optional): Custom URL templates for API endpoints. Supports variable substitution with `{key}` and `{version}` placeholders. -- **response.data_path** (optional): JSON path to extract response data from nested structures (e.g., "data.config") -- **response.error_path** (optional): JSON path to check for errors in responses +- **response.data_path** (optional): Dot-separated object keys or numeric array indexes used to extract data (e.g., `data.config` or `data.0`). +- **response.error_path** (optional): A path with the same syntax. Any non-null value at this path is treated as an error, including `false` or an empty string. #### Default URL Templates @@ -192,7 +198,7 @@ The HTTP provider expects the remote API to implement these endpoints and return ## HTTP API -Connector runtime has an optional HTTP API that can be enabled by setting the `enabled` flag to `true` in the `[http]` section. +The HTTP API is enabled by default at `127.0.0.1:8081`. Set `[http].enabled = false` to disable it. ```toml [http] # Optional HTTP API configuration @@ -300,7 +306,7 @@ key_file = "core/certs/iggy_key.pem" Currently, it does expose the following endpoints: - `GET /`: welcome message. -- `GET /health`: health status of the runtime. +- `GET /health`: process liveness response. It does not check connector health; inspect `/stats`, `/sources` or `/sinks` for connector status. - `GET /stats`: runtime statistics including process info, memory/CPU usage, and connector status. - `GET {http.metrics.endpoint}` (default `/metrics`): Prometheus-formatted metrics, when `http.metrics.enabled` is `true`. - `GET /sinks`: list of sinks. @@ -359,14 +365,15 @@ transport = "grpc" # Options: "grpc", "http" endpoint = "http://localhost:4317" ``` +For `transport = "http"`, use complete signal endpoints such as `http://localhost:4318/v1/logs` and `http://localhost:4318/v1/traces`. The runtime does not append those paths. + ## Benchmark Mode Each connector configuration accepts an optional `benchmark` flag. When set to `true`, the runtime emits a per-batch `info!` event on the `iggy_connectors::benchmark` tracing target with stage timings in microseconds. This is opt-in and adds a single tracing call per processed batch. +Set the flag before any section headers in an existing connector configuration: + ```toml -type = "sink" -key = "stdout" -# ... other fields ... benchmark = true ``` diff --git a/core/connectors/runtime/src/log.rs b/core/connectors/runtime/src/log.rs index 82273a56b6..05df5635eb 100644 --- a/core/connectors/runtime/src/log.rs +++ b/core/connectors/runtime/src/log.rs @@ -22,7 +22,10 @@ use opentelemetry::{KeyValue, global}; use opentelemetry_appender_tracing::layer::OpenTelemetryTracingBridge; use opentelemetry_otlp::WithExportConfig; use opentelemetry_sdk::Resource; +use opentelemetry_sdk::logs::log_processor_with_async_runtime; use opentelemetry_sdk::propagation::TraceContextPropagator; +use opentelemetry_sdk::runtime::Tokio; +use opentelemetry_sdk::trace::span_processor_with_async_runtime; use tracing::info; use tracing_opentelemetry::OpenTelemetryLayer; use tracing_subscriber::layer::SubscriberExt; @@ -116,7 +119,13 @@ fn init_logs_exporter( .expect("Failed to initialize HTTP logger."); opentelemetry_sdk::logs::SdkLoggerProvider::builder() .with_resource(resource) - .with_batch_exporter(log_exporter) + .with_log_processor( + log_processor_with_async_runtime::BatchLogProcessor::builder( + log_exporter, + Tokio, + ) + .build(), + ) .build() } } @@ -146,7 +155,13 @@ fn init_traces_exporter( .expect("Failed to initialize HTTP tracer."); opentelemetry_sdk::trace::SdkTracerProvider::builder() .with_resource(resource) - .with_batch_exporter(trace_exporter) + .with_span_processor( + span_processor_with_async_runtime::BatchSpanProcessor::builder( + trace_exporter, + Tokio, + ) + .build(), + ) .build() } } @@ -181,3 +196,101 @@ pub extern "C" fn runtime_log_callback( } pub const LOG_CALLBACK: LogCallback = runtime_log_callback; + +#[cfg(test)] +mod tests { + use opentelemetry::logs::{LogRecord, Logger, LoggerProvider}; + use opentelemetry::trace::{Span, Tracer, TracerProvider}; + use wiremock::matchers::method; + use wiremock::{Mock, MockServer, ResponseTemplate}; + + use super::init_telemetry; + use crate::configs::runtime::{ + TelemetryConfig, TelemetryLogsConfig, TelemetryTracesConfig, TelemetryTransport, + }; + + const TEST_SCOPE: &str = "connectors-telemetry-test"; + const LOG_BODY: &str = "connector-telemetry-log"; + const TRACE_NAME: &str = "connector-telemetry-span"; + + #[test] + fn given_http_telemetry_when_flushed_should_export_logs_and_traces() { + let runtime = tokio::runtime::Runtime::new().expect("Tokio runtime should start"); + runtime.block_on(async { + let collector = MockServer::start().await; + Mock::given(method("POST")) + .respond_with( + ResponseTemplate::new(200) + .insert_header("content-type", "application/x-protobuf"), + ) + .mount(&collector) + .await; + let config = TelemetryConfig { + enabled: true, + logs: TelemetryLogsConfig { + transport: TelemetryTransport::Http, + endpoint: format!("{}/v1/logs", collector.uri()), + }, + traces: TelemetryTracesConfig { + transport: TelemetryTransport::Http, + endpoint: format!("{}/v1/traces", collector.uri()), + }, + ..TelemetryConfig::default() + }; + let (logger_provider, tracer_provider) = + init_telemetry(&config, env!("CARGO_PKG_VERSION")); + let logger = logger_provider.logger(TEST_SCOPE); + let mut record = logger.create_log_record(); + record.set_body(LOG_BODY.into()); + logger.emit(record); + let tracer = tracer_provider.tracer(TEST_SCOPE); + let mut span = tracer.start(TRACE_NAME); + span.end(); + + let results = tokio::task::spawn_blocking(move || { + ( + logger_provider.force_flush(), + tracer_provider.force_flush(), + logger_provider.shutdown(), + tracer_provider.shutdown(), + ) + }) + .await + .expect("telemetry flush task should complete"); + assert!(results.0.is_ok(), "log export failed: {:?}", results.0); + assert!(results.1.is_ok(), "trace export failed: {:?}", results.1); + assert!(results.2.is_ok(), "logger shutdown failed: {:?}", results.2); + assert!(results.3.is_ok(), "tracer shutdown failed: {:?}", results.3); + + let requests = collector + .received_requests() + .await + .expect("collector should record requests"); + assert_eq!( + requests.len(), + 2, + "both telemetry signals should be exported" + ); + for (path, marker) in [("/v1/logs", LOG_BODY), ("/v1/traces", TRACE_NAME)] { + let request = requests + .iter() + .find(|request| request.url.path() == path) + .expect("each signal should reach its configured endpoint"); + assert_eq!( + request + .headers + .get("content-type") + .expect("OTLP content type"), + "application/x-protobuf" + ); + assert!( + request + .body + .windows(marker.len()) + .any(|bytes| bytes == marker.as_bytes()), + "the exported signal should contain its emitted record: {path}" + ); + } + }); + } +} diff --git a/core/connectors/runtime/src/manager/sink.rs b/core/connectors/runtime/src/manager/sink.rs index 270d442d9b..07e94527e0 100644 --- a/core/connectors/runtime/src/manager/sink.rs +++ b/core/connectors/runtime/src/manager/sink.rs @@ -100,9 +100,14 @@ impl SinkManager { } } - pub async fn set_error(&self, key: &str, error_message: &str) { + pub async fn set_error(&self, key: &str, error_message: &str, metrics: Option<&Arc>) { if let Some(sink) = self.sinks.get(key) { let mut sink = sink.lock().await; + if sink.info.status == ConnectorStatus::Running + && let Some(metrics) = metrics + { + metrics.decrement_sinks_running(); + } sink.info.status = ConnectorStatus::Error; sink.info.last_error = Some(ConnectorError::new(error_message)); } @@ -444,7 +449,7 @@ mod tests { #[tokio::test] async fn should_clear_error_when_status_becomes_running() { let manager = SinkManager::new(vec![create_test_sink_details("es", 1)]); - manager.set_error("es", "some error").await; + manager.set_error("es", "some error", None).await; manager .update_status("es", ConnectorStatus::Running, None) @@ -459,7 +464,7 @@ mod tests { async fn should_set_error_status_and_message() { let manager = SinkManager::new(vec![create_test_sink_details("es", 1)]); - manager.set_error("es", "connection failed").await; + manager.set_error("es", "connection failed", None).await; let sink = manager.get("es").await.unwrap(); let details = sink.lock().await; @@ -527,7 +532,7 @@ mod tests { #[tokio::test] async fn should_clear_error_when_status_becomes_stopped() { let manager = SinkManager::new(vec![create_test_sink_details("es", 1)]); - manager.set_error("es", "some error").await; + manager.set_error("es", "some error", None).await; manager .update_status("es", ConnectorStatus::Stopped, None) @@ -579,6 +584,46 @@ mod tests { async fn set_error_should_be_noop_for_unknown_key() { let manager = SinkManager::new(vec![]); - manager.set_error("nonexistent", "some error").await; + manager.set_error("nonexistent", "some error", None).await; + } + + #[tokio::test] + async fn given_running_connector_when_error_repeats_should_decrement_once() { + const FAILED_KEY: &str = "failed"; + let metrics = Arc::new(Metrics::init()); + let manager = SinkManager::new(vec![ + create_test_sink_details(FAILED_KEY, 1), + create_test_sink_details("healthy", 2), + ]); + metrics.increment_sinks_running(); + metrics.increment_sinks_running(); + + manager + .set_error(FAILED_KEY, "first error", Some(&metrics)) + .await; + assert_eq!( + metrics.get_sinks_running(), + 1, + "the healthy connector remains running" + ); + + manager + .set_error(FAILED_KEY, "repeated error", Some(&metrics)) + .await; + assert_eq!( + metrics.get_sinks_running(), + 1, + "repeated errors must not decrement twice" + ); + + manager + .stop_connector(FAILED_KEY, &metrics) + .await + .expect("failed connector should stop"); + assert_eq!( + metrics.get_sinks_running(), + 1, + "stopping an errored connector must not decrement again" + ); } } diff --git a/core/connectors/runtime/src/manager/source.rs b/core/connectors/runtime/src/manager/source.rs index dcdd85b937..3cff2305ac 100644 --- a/core/connectors/runtime/src/manager/source.rs +++ b/core/connectors/runtime/src/manager/source.rs @@ -759,4 +759,44 @@ mod tests { manager.set_error("nonexistent", "some error", None).await; } + + #[tokio::test] + async fn given_running_connector_when_error_repeats_should_decrement_once() { + const FAILED_KEY: &str = "failed"; + let metrics = Arc::new(Metrics::init()); + let manager = SourceManager::new(vec![ + create_test_source_details(FAILED_KEY, 1), + create_test_source_details("healthy", 2), + ]); + metrics.increment_sources_running(); + metrics.increment_sources_running(); + + manager + .set_error(FAILED_KEY, "first error", Some(&metrics)) + .await; + assert_eq!( + metrics.get_sources_running(), + 1, + "the healthy connector remains running" + ); + + manager + .set_error(FAILED_KEY, "repeated error", Some(&metrics)) + .await; + assert_eq!( + metrics.get_sources_running(), + 1, + "repeated errors must not decrement twice" + ); + + manager + .stop_connector(FAILED_KEY, &metrics) + .await + .expect("failed connector should stop"); + assert_eq!( + metrics.get_sources_running(), + 1, + "stopping an errored connector must not decrement again" + ); + } } diff --git a/core/connectors/runtime/src/sink.rs b/core/connectors/runtime/src/sink.rs index 88bc43a659..994cfc392e 100644 --- a/core/connectors/runtime/src/sink.rs +++ b/core/connectors/runtime/src/sink.rs @@ -289,7 +289,7 @@ pub(crate) fn spawn_consume_tasks( metrics.inc_errors_with_labels(&labels.counter); context .sinks - .set_error(&plugin_key, &error.to_string()) + .set_error(&plugin_key, &error.to_string(), Some(&metrics)) .await; } }); @@ -602,12 +602,13 @@ async fn process_messages( let mut messages = Vec::with_capacity(decoded.len()); for message in decoded { let mut current_message = Some(message); + let mut transform_failed = false; for transform in transforms.iter() { let Some(message) = current_message.take() else { break; }; - // Drop-and-continue on a single bad message, mirroring the source - // side - one malformed payload must not kill the whole batch. + // Sink batches can deliver valid siblings after a transform failure. + // Source batches instead reject the checkpoint for the entire batch. match transform.transform(topic_metadata, message) { Ok(next) => current_message = next, Err(error) => { @@ -618,11 +619,14 @@ async fn process_messages( topic_metadata.topic ); error_count += 1; - current_message = None; + transform_failed = true; break; } } } + if transform_failed { + continue; + } // Filter contract: transform returning Ok(None) is an intentional drop. let Some(message) = current_message else { @@ -734,7 +738,7 @@ async fn process_messages( })?; let ffi_start = Instant::now(); - (consume)( + let result = (consume)( plugin_id, topic_meta.as_ptr(), topic_meta.len(), @@ -744,6 +748,16 @@ async fn process_messages( messages.len(), ); let ffi_elapsed = ffi_start.elapsed(); + let processed_count = if result == 0 { + processed_count + } else { + error!( + "Failed to consume {processed_count} messages for sink connector with ID: {plugin_id}, stream: {}, topic: {}, status: {result}", + topic_metadata.stream, topic_metadata.topic + ); + metrics.inc_errors_with_labels(&labels.counter); + 0 + }; Ok(SinkBatchTiming { processed_count, diff --git a/core/connectors/runtime/src/state.rs b/core/connectors/runtime/src/state.rs index 34693f50cb..08c0b2d33e 100644 --- a/core/connectors/runtime/src/state.rs +++ b/core/connectors/runtime/src/state.rs @@ -93,7 +93,7 @@ pub fn factory_from_config( } StateStorageKind::Http => { let factory = HttpStateFactory::new(&config.http)?; - info!("State will be stored via HTTP at: {}", config.http.url); + info!("State will be stored via HTTP: {}", config.http); Ok(Arc::new(factory)) } } diff --git a/core/connectors/runtime/src/state/http.rs b/core/connectors/runtime/src/state/http.rs index 9d216656f3..d9875dc8c6 100644 --- a/core/connectors/runtime/src/state/http.rs +++ b/core/connectors/runtime/src/state/http.rs @@ -291,14 +291,16 @@ impl HttpStateProvider { let etag = required_etag(&response, "load", &self.resource_label)?; match response.bytes().await { Ok(bytes) => return Ok(LoadResponse::Found { etag, bytes }), - Err(read_error) => TransientFailure::Read(read_error.to_string()), + Err(read_error) => { + TransientFailure::Read(read_error.without_url().to_string()) + } } } Ok(response) if response.status() == StatusCode::NOT_FOUND => { return Ok(LoadResponse::NotFound); } Ok(response) => return Ok(LoadResponse::Failure(response)), - Err(send_error) => TransientFailure::Send(send_error), + Err(send_error) => TransientFailure::Send(send_error.without_url()), }; if attempt >= max_attempts { return Err(Error::TransientState(failure.describe( @@ -342,7 +344,7 @@ impl HttpStateProvider { let failure = match request.send().await { Ok(response) if !is_transient_status(response.status()) => return Ok(response), Ok(response) => TransientFailure::Status(response), - Err(send_error) => TransientFailure::Send(send_error), + Err(send_error) => TransientFailure::Send(send_error.without_url()), }; if attempt >= max_attempts { return Err(Error::TransientState(failure.describe( @@ -725,10 +727,13 @@ mod tests { use std::sync::atomic::AtomicU64; use std::sync::{Arc, Mutex as StdMutex}; use std::time::Instant; + use tokio::io::{AsyncBufReadExt, AsyncWriteExt, BufReader}; + use tokio::net::TcpListener; use wiremock::matchers::{header, method, path, query_param}; use wiremock::{Mock, MockServer, Request, Respond, ResponseTemplate}; const RESOURCE_PATH: &str = "/source_test"; + const STATE_URL_QUERY_SECRET: &str = "state-query-secret-not-for-logs"; fn test_config(url: &str) -> HttpStateConfig { HttpStateConfig { @@ -970,6 +975,52 @@ mod tests { ); } + #[tokio::test] + async fn given_truncated_body_when_loaded_should_redact_url_secrets() { + let listener = TcpListener::bind("127.0.0.1:0") + .await + .expect("bind state server"); + let address = listener.local_addr().expect("state server address"); + let mut config = test_config(&format!("http://{address}?token={STATE_URL_QUERY_SECRET}")); + config.retry.enabled = false; + let storage = storage_for(&config); + let respond = async { + let (socket, _) = listener.accept().await.expect("accept state request"); + let mut connection = BufReader::new(socket); + let mut line = String::new(); + loop { + line.clear(); + let read = connection + .read_line(&mut line) + .await + .expect("read request headers"); + assert_ne!(read, 0, "request must include complete headers"); + if line == "\r\n" { + break; + } + } + connection + .write_all(b"HTTP/1.1 200 OK\r\nETag: \"v1\"\r\nContent-Length: 2\r\nConnection: close\r\n\r\nx") + .await + .expect("send truncated state body"); + connection.shutdown().await.expect("close state response"); + }; + let (_, result) = tokio::time::timeout(config.timeout.get_duration(), async { + tokio::join!(respond, storage.load()) + }) + .await + .expect("state request must finish"); + let Err(Error::TransientState(message)) = result else { + panic!("body-read failure must be transient, got {result:?}"); + }; + assert!( + message.contains("while reading the response body"), + "{message}" + ); + assert!(message.contains(RESOURCE_PATH), "{message}"); + assert!(!message.contains(STATE_URL_QUERY_SECRET), "{message}"); + } + #[tokio::test] async fn given_version_conflict_when_saved_should_latch() { let server = MockServer::start().await; @@ -1140,15 +1191,18 @@ mod tests { .mount(&server) .await; - let mut config = test_config(&server.uri()); + let mut config = test_config(&format!("{}?token={STATE_URL_QUERY_SECRET}", server.uri())); config.timeout = IggyDuration::new(Duration::from_millis(50)); config.retry.enabled = false; let storage = storage_for(&config); storage.load().await.unwrap(); - assert!(matches!( - storage.save(ConnectorState(vec![1, 2, 3])).await, - Err(Error::TransientState(_)) - )); + let result = storage.save(ConnectorState(vec![1, 2, 3])).await; + assert!( + matches!(result, Err(Error::TransientState(_))), + "{result:?}" + ); + let error_log = format!("{result:?}"); + assert!(!error_log.contains(STATE_URL_QUERY_SECRET), "{error_log}"); storage .resolve_pending() .await diff --git a/core/connectors/sdk/README.md b/core/connectors/sdk/README.md index dbb284baca..d91628058e 100644 --- a/core/connectors/sdk/README.md +++ b/core/connectors/sdk/README.md @@ -10,7 +10,7 @@ Source connectors use a one-in-flight-batch contract between the plugin and the 1. `Source::poll()` returns messages and candidate state without committing cursor changes or destructive operations. 2. The runtime sends the batch to Iggy and waits for the producer result. -3. After a successful send, the runtime persists the candidate state. +3. After a successful send, the runtime persists the candidate state if the batch provides it. 4. The runtime reports `SourceBatchResult::Ack` to the plugin. A send or state-save failure reports `SourceBatchResult::Nack` instead. 5. `Source::on_batch_result()` commits or discards the plugin's staged work before the next poll starts. @@ -37,7 +37,7 @@ This contract is a breaking FFI change. Source plugins must be rebuilt with the Moreover, it contains both, the `decoders` and `encoders` modules, implementing either `StreamDecoder` or `StreamEncoder` traits, which are used when consuming or producing data from/to Iggy streams. -SDK is WiP, and it'd certainly benefit from having the support of multiple format schemas, such as Protobuf, Avro, Flatbuffers etc. including decoding/encoding the data between the different formats (when applicable) and supporting the data transformations whenever possible (easy for JSON, but complex for Bincode for example). +The SDK provides decoders and encoders for JSON, raw bytes, text, Protocol Buffers, FlatBuffers, and Avro. Their supported conversions have format-specific limits; see below and the [transforms guide](https://iggy.apache.org/docs/connectors/transforms). Last but not least, the different `transforms` are available, to transform (add, update, delete etc.) the particular fields of the data being processed via external configuration. It's as simple as adding a new transform to the `transforms` section of the particular connector configuration file: @@ -69,9 +69,16 @@ The SDK includes support for Protocol Buffers (protobuf) format with both encodi ### Configuration Example -Here's a complete example configuration for using Protocol Buffers with Iggy connectors. +This example uses the Random source and Stdout sink from the matching checkout. Start a server using the **[getting-started guide](https://iggy.apache.org/docs/introduction/getting-started)**, then build the plugins, runtime, and CLI from the repository root: -**Main runtime config (config.toml):** +```bash +cargo build --release -p iggy_connector_random_source -p iggy_connector_stdout_sink -p iggy-connectors -p iggy-cli +mkdir -p connectors +``` + +The source's `schema = "proto"` selects the default protobuf encoder, which wraps each JSON record in a `google.protobuf.StringValue` inside `google.protobuf.Any`. The sink reads raw bytes and applies `proto_convert` to expose the Any envelope as JSON. This does not unpack a custom protobuf message schema. + +**Main runtime config (connectors.toml):** ```toml [iggy] @@ -81,7 +88,7 @@ password = "iggy" [connectors] config_type = "local" -config_dir = "path/to/connectors" +config_dir = "connectors" ``` **Source connector config (connectors/protobuf_source.toml):** @@ -92,19 +99,20 @@ key = "protobuf" enabled = true version = 0 name = "Protobuf Source" -path = "target/release/libiggy_connector_protobuf_source" +path = "target/release/libiggy_connector_random_source" [[streams]] stream = "protobuf_stream" topic = "protobuf_topic" schema = "proto" -batch_size = 1000 -send_interval = "5ms" +batch_length = 1000 +linger_time = "5ms" [plugin_config] -schema_path = "schemas/message.proto" -message_type = "com.example.Message" -use_any_wrapper = true +interval = "100ms" +messages_range = [1, 10] +payload_size = 32 +max_count = 100 ``` **Sink connector config (connectors/protobuf_sink.toml):** @@ -115,40 +123,66 @@ key = "protobuf" enabled = true version = 0 name = "Protobuf Sink" -path = "target/release/libiggy_connector_protobuf_sink" +path = "target/release/libiggy_connector_stdout_sink" [[streams]] stream = "protobuf_stream" -topic = "protobuf_topic" -schema = "proto" +topics = ["protobuf_topic"] +schema = "raw" -[[transforms]] -type = "proto_convert" +[plugin_config] +print_payload = true + +[transforms.proto_convert] +enabled = true +source_format = "proto" target_format = "json" -preserve_structure = true +include_paths = ["."] +preserve_unknown_fields = false + +[transforms.proto_convert.conversion_options] +validate_messages = true +pretty_json = false +include_metadata = false +type_url_prefix = "type.googleapis.com" +strict_mode = false +``` -field_mappings = { "old_field" = "new_field", "legacy_id" = "id" } +Create the stream and topic, then start the runtime from the repository root: -[[transforms]] -type = "proto_convert" -target_format = "proto" -preserve_structure = false +```bash +./target/release/iggy --username iggy --password iggy stream create protobuf_stream +./target/release/iggy --username iggy --password iggy topic create protobuf_stream protobuf_topic 1 none 1d +IGGY_CONNECTORS_CONFIG_PATH=connectors.toml ./target/release/iggy-connectors ``` +The source sends 100 records, then continues polling without new messages. Stdout logs message offsets and the serialized JSON envelope bytes, containing `type_url` and base64 `value`. The sink's `raw` schema also determines how the plugin receives those transformed bytes. + +The format-conversion transforms define no per-key defaults. Every non-optional key shown above must be present, or the configuration fails to deserialize (`schema_path`, `message_type`, `field_mappings`, and `descriptor_set` are optional). + +The two `[[streams]]` shapes differ: a source produces to a single `topic` and can tune batching via `batch_length` and `linger_time`, while a sink consumes from a list of `topics` and can additionally set `batch_length`, `poll_interval`, and `consumer_group`. + ### Key Configuration Options -#### Source Configuration +#### Programmatic Encoder and Decoder Configuration + +These are SDK configuration fields, not Random or Stdout `plugin_config` keys. The runtime's `schema = "proto"` uses the default encoder or decoder. - **`schema_path`**: Path to the `.proto` file containing message definitions - **`message_type`**: Fully qualified name of the protobuf message type to use -- **`use_any_wrapper`**: Whether to wrap messages in `google.protobuf.Any` for type safety +- **`use_any_wrapper`**: Selects the Any fallback when no message descriptor is loaded; a loaded descriptor takes precedence #### Transform Options - **`proto_convert`**: Transform for converting between protobuf and other formats - - **`target_format`**: Target format for conversion (`json`, `proto`, `text`) - - **`preserve_structure`**: Whether to preserve the original message structure during conversion - - **`field_mappings`**: Mapping of field names for transformation (e.g., `"old_field" = "new_field"`) +- **`source_format`** / **`target_format`**: Formats to convert between - any schema value (`json`, `raw`, `text`, `proto`, `flat_buffer`, `avro`) +- **`preserve_unknown_fields`**: Accepted by `proto_convert`, but currently has no effect +- **`include_paths`**: Additional directories searched for imported `.proto` files +- **`field_mappings`**: Renames fields in a JSON input object before conversion (e.g., `"old_field" = "new_field"`) +- **`conversion_options`**: `pretty_json` controls JSON text output and `include_metadata` enriches supported protobuf-to-JSON paths. `validate_messages`, `type_url_prefix`, and `strict_mode` are accepted but currently have no effect + +The `schema_registry_url` field is reserved and currently not implemented. The SDK never contacts a schema registry, and schemas are loaded only from `schema_path` or `descriptor_set`. + - **`unwrap_envelope`**: Extracts a nested JSON field and promotes it as the top-level payload. Required when a source emits envelope-wrapped records (with metadata fields alongside a nested data object) and the downstream sink expects flat JSON matching the target table schema. @@ -162,90 +196,162 @@ field = "data" ### Supported Features -- **Encoding**: Convert JSON, Text, and Raw data to protobuf format -- **Decoding**: Parse protobuf messages into JSON format with type information -- **Transforms**: Convert between protobuf and other formats (JSON, Text) -- **Field Mapping**: Transform field names during format conversion -- **Any Wrapper**: Support for `google.protobuf.Any` message wrapper +- **Encoding**: A loaded message descriptor encodes matching JSON fields. The encoder supports booleans, strings, all protobuf integer types, and base64 strings for bytes or already-encoded nested messages. Float, double, and enum fields are unsupported by the encoder; nested JSON objects, repeated fields, maps, and proto2 groups are not a general-purpose schema conversion path. +- **Decoding**: A loaded descriptor extracts present fields. Integer, floating-point, boolean, and string fields become JSON values; bytes become base64 and nested messages become metadata with base64 content. Missing fields are not filled with protobuf defaults. Without a descriptor, the default decoder returns an Any envelope's `type_url` and base64 `value`. +- **Transforms**: `proto_convert` supports JSON-to-protobuf schema encoding for scalar fields, including floating-point numbers and numeric enum values; bytes and already-encoded nested messages use base64 strings. It logs and omits fields it cannot encode. Its protobuf-to-JSON path exposes Any metadata or raw-data metadata rather than decoding a custom message descriptor. Converting protobuf to `flat_buffer` or `avro` rewraps bytes without transcoding them. +- **Field Mapping**: Encoder/decoder mappings use protobuf field names as keys and JSON field names as values. The encoder applies that mapping in reverse. Transform mappings rename JSON input keys directly. +- **Any Wrapper**: The default encoder puts JSON/text in a `google.protobuf.StringValue`, or binary data in a `google.protobuf.BytesValue`, inside `google.protobuf.Any`. The default decoder exposes the envelope without unpacking its inner message. ### Programmatic Usage +From the matching repository root, create an example crate and schema directory: + +```bash +mkdir -p connector-sdk-example/src schemas +``` + +Save this as `connector-sdk-example/Cargo.toml`. The path dependency uses the SDK from the same checkout as the runtime: + +```toml +[package] +name = "connector-sdk-example" +version = "0.1.0" +edition = "2024" + +[dependencies] +iggy_connector_sdk = { path = "../core/connectors/sdk" } +simd-json = { version = "0.18.1", features = ["serde_impl"] } + +[workspace] +``` + +Save this as `schemas/user.proto`: + +```protobuf +syntax = "proto3"; +package com.example; + +message User { + uint64 id = 1; + string name = 2; +} +``` + +Each Rust example below is a complete `connector-sdk-example/src/main.rs`. Run it from the repository root so the relative schema path resolves: + +```bash +cargo run --manifest-path connector-sdk-example/Cargo.toml +``` + #### Dynamic Schema Loading You can load or reload schemas programmatically: ```rust -use iggy_connector_sdk::decoders::proto::{ProtoStreamDecoder, ProtoConfig}; +use iggy_connector_sdk::decoders::proto::{ProtoConfig, ProtoStreamDecoder}; +use iggy_connector_sdk::encoders::proto::{ProtoEncoderConfig, ProtoStreamEncoder}; +use iggy_connector_sdk::{Error, Payload, StreamDecoder, StreamEncoder}; use std::path::PathBuf; -let mut decoder = ProtoStreamDecoder::new(ProtoConfig { - schema_path: None, - use_any_wrapper: true, - ..Default::default() -}); - -let config_with_schema = ProtoConfig { - schema_path: Some(PathBuf::from("schemas/user.proto")), - message_type: Some("com.example.User".to_string()), - ..Default::default() -}; - -match decoder.update_config(config_with_schema, true) { - Ok(()) => println!("Schema loaded successfully"), - Err(e) => eprintln!("Failed to load schema: {}", e), +fn main() -> Result<(), Error> { + let mut decoder = ProtoStreamDecoder::new_default(); + decoder.update_config( + ProtoConfig { + schema_path: Some(PathBuf::from("schemas/user.proto")), + message_type: Some("com.example.User".to_string()), + ..ProtoConfig::default() + }, + true, + )?; + let encoder = ProtoStreamEncoder::new_with_config(ProtoEncoderConfig { + schema_path: Some(PathBuf::from("schemas/user.proto")), + message_type: Some("com.example.User".to_string()), + ..ProtoEncoderConfig::default() + }); + let encoded = encoder.encode(Payload::Json(simd_json::json!({ + "id": 1, + "name": "Alice" + })))?; + println!("{}", decoder.decode(encoded)?); + Ok(()) } ``` -#### Schema Registry Integration +The encoder follows the same pattern: ```rust -use iggy_connector_sdk::encoders::proto::{ProtoStreamEncoder, ProtoEncoderConfig}; - -let mut encoder = ProtoStreamEncoder::new_with_config(ProtoEncoderConfig { - schema_registry_url: Some("http://schema-registry:8081".to_string()), - message_type: Some("com.example.Event".to_string()), - use_any_wrapper: false, - ..Default::default() -}); +use iggy_connector_sdk::encoders::proto::{ProtoEncoderConfig, ProtoStreamEncoder}; +use iggy_connector_sdk::{Error, Payload, StreamEncoder}; +use std::path::PathBuf; -if let Err(e) = encoder.load_schema() { - eprintln!("Schema reload failed: {}", e); +fn main() -> Result<(), Error> { + let mut encoder = ProtoStreamEncoder::new_with_config(ProtoEncoderConfig { + schema_path: Some(PathBuf::from("schemas/user.proto")), + message_type: Some("com.example.User".to_string()), + use_any_wrapper: false, + ..ProtoEncoderConfig::default() + }); + encoder.load_schema()?; + let encoded = encoder.encode(Payload::Json(simd_json::json!({ + "id": 1, + "name": "Alice" + })))?; + println!("{encoded:?}"); + Ok(()) } ``` #### Creating Converters with Schema +The loaded schema is used for JSON-to-protobuf conversion. This example maps `user_id` and `full_name` to the schema's field names: + ```rust -use iggy_connector_sdk::transforms::proto_convert::{ProtoConvert, ProtoConvertConfig}; -use iggy_connector_sdk::Schema; +use iggy_connector_sdk::transforms::{ProtoConvert, ProtoConvertConfig, Transform}; +use iggy_connector_sdk::{DecodedMessage, Error, Payload, Schema, TopicMetadata}; use std::collections::HashMap; use std::path::PathBuf; -let converter = ProtoConvert::new(ProtoConvertConfig { - source_format: Schema::Proto, - target_format: Schema::Json, - schema_path: Some(PathBuf::from("schemas/user.proto")), - message_type: Some("com.example.User".to_string()), - field_mappings: Some(HashMap::from([ - ("user_id".to_string(), "id".to_string()), - ("full_name".to_string(), "name".to_string()), - ])), - ..ProtoConvertConfig::default() -}); - -let mut converter_with_manual_loading = ProtoConvert::new(ProtoConvertConfig::default()); -if let Err(e) = converter_with_manual_loading.load_schema() { - eprintln!("Manual schema loading failed: {}", e); +fn main() -> Result<(), Error> { + let converter = ProtoConvert::new(ProtoConvertConfig { + source_format: Schema::Json, + target_format: Schema::Proto, + schema_path: Some(PathBuf::from("schemas/user.proto")), + message_type: Some("com.example.User".to_string()), + field_mappings: Some(HashMap::from([ + ("user_id".to_string(), "id".to_string()), + ("full_name".to_string(), "name".to_string()), + ])), + ..ProtoConvertConfig::default() + }); + let metadata = TopicMetadata { + stream: "users".to_string(), + topic: "users".to_string(), + }; + let message = DecodedMessage { + id: None, + offset: None, + checksum: None, + timestamp: None, + origin_timestamp: None, + headers: None, + payload: Payload::Json(simd_json::json!({ + "user_id": 1, + "full_name": "Alice" + })), + }; + if let Some(converted) = converter.transform(&metadata, message)? { + println!("{:?}", converted.payload); + } + Ok(()) } ``` ### Usage Notes -- **Automatic Loading**: Schemas are loaded automatically when `schema_path` or `descriptor_set` is provided in config -- **Manual Loading**: Use `load_schema()` method for dynamic schema loading or reloading -- **Error Handling**: Schema loading errors are handled gracefully with fallback to Any wrapper mode -- **Immutable Design**: Converters are created with fixed configuration - create new instances for different schemas -- When `use_any_wrapper` is enabled, messages are wrapped in `google.protobuf.Any` for better type safety -- The `proto_convert` transform can be used to convert protobuf messages to JSON for easier processing -- Field mappings allow you to rename fields during format conversion +- **Automatic Loading**: Constructors attempt to load `schema_path` or `descriptor_set`; `schema_path` takes precedence when both are set. Constructors log loading errors and return an instance without a loaded schema. +- **Manual Loading**: `load_schema()` reloads the configured source. Missing or unreadable files, invalid protobuf syntax, compilation failures, and malformed descriptor bytes return errors and preserve an already-loaded schema. `update_config(config, true)` also restores the previous configuration on error; `false` changes the configuration while retaining the cached schema. +- **Fallbacks**: Absent schema sources or an unmatched `message_type` can return `Ok(())` without an active message descriptor. Successful reloads into fallback mode clear the previous descriptor. Check the actual encoded/decoded result when validating a schema setup. +- **Encoding Errors**: Errors encoding a loaded message descriptor are returned to the caller. The encoder does not retry that message as Any. The decoder attempts Any after a schema decoding error. +- **Transform Configuration**: Create a new converter to change its configuration. `load_schema()` can reload its existing source. Without a descriptor, JSON-to-protobuf conversion produces JSON text in `Payload::Proto`, not a schema-encoded binary message. +- **Format Options**: Encoder `preserve_unknown_fields`, `compact_encoding`, `validate_message`, and `deterministic_encoding` are accepted but have no effect. Decoder `preserve_unknown_fields` retains unknown varints as numbers and length-delimited data as base64; fixed-width unknown fields become placeholders. It does not retain the original wire encoding. See the [Transforms page](https://iggy.apache.org/docs/connectors/transforms) for conversion-specific limits. - Protocol Buffers provide efficient binary serialization compared to JSON diff --git a/core/connectors/sdk/src/decoders/proto.rs b/core/connectors/sdk/src/decoders/proto.rs index 9e97ba52bc..a4d774a6e2 100644 --- a/core/connectors/sdk/src/decoders/proto.rs +++ b/core/connectors/sdk/src/decoders/proto.rs @@ -83,39 +83,39 @@ impl ProtoStreamDecoder { } pub fn update_config(&mut self, config: ProtoConfig, reload_schema: bool) -> Result<(), Error> { - self.config = config; - if reload_schema - && (self.config.schema_path.is_some() || self.config.descriptor_set.is_some()) - { - self.load_schema() - } else { - Ok(()) + let old_config = std::mem::replace(&mut self.config, config); + if reload_schema && let Err(error) = self.load_schema() { + self.config = old_config; + return Err(error); } + Ok(()) } pub fn load_schema(&mut self) -> Result<(), Error> { let schema_path = self.config.schema_path.clone(); let descriptor_set = self.config.descriptor_set.clone(); - if let Some(path) = schema_path { - self.compile_schema_internal(&path)?; + let old_message_descriptor = self.message_descriptor.take(); + let old_file_descriptor_set = self.file_descriptor_set.take(); + let result = if let Some(path) = schema_path { + self.compile_schema_internal(&path) } else if let Some(descriptor_bytes) = descriptor_set { - self.load_descriptor_set_internal(&descriptor_bytes)?; + self.load_descriptor_set_internal(&descriptor_bytes) + } else { + Ok(()) + }; + if result.is_err() { + self.message_descriptor = old_message_descriptor; + self.file_descriptor_set = old_file_descriptor_set; } - Ok(()) + result } fn compile_schema_internal(&mut self, schema_path: &PathBuf) -> Result<(), Error> { info!("Compiling protobuf schema from: {:?}", schema_path); - let proto_content = match fs::read_to_string(schema_path) { - Ok(content) => content, - Err(e) => { - error!("Failed to read proto file: {}", e); - error!("Falling back to Any wrapper mode"); - return Ok(()); - } - }; + let proto_content = fs::read_to_string(schema_path) + .map_err(|error| Error::InitError(format!("Failed to read proto file: {error}")))?; let parsed_file = parse(&schema_path.to_string_lossy(), &proto_content) .map_err(|e| Error::InitError(format!("Failed to parse proto file: {e}")))?; @@ -149,12 +149,9 @@ impl ProtoStreamDecoder { self.file_descriptor_set = Some(file_descriptor_set); Ok(()) } - Err(e) => { - error!("Failed to compile proto schema: {}", e); - error!("Falling back to Any wrapper mode"); - - Ok(()) - } + Err(error) => Err(Error::InitError(format!( + "Failed to compile proto schema: {error}" + ))), } } @@ -197,23 +194,24 @@ impl ProtoStreamDecoder { package: &str, ) -> Option { let parent_name = parent_message.name.as_deref().unwrap_or(""); - - let package_prefix = if package.is_empty() { - String::new() + let parent_prefix = if package.is_empty() { + parent_name.to_string() } else { - format!("{package}.") + format!("{package}.{parent_name}") }; for nested_message in &parent_message.nested_type { let nested_name = nested_message.name.as_deref().unwrap_or(""); - let full_name = format!("{package_prefix}{parent_name}.{nested_name}"); + let full_name = format!("{parent_prefix}.{nested_name}"); if full_name == target_type { info!("Found nested message descriptor: {}", full_name); return Some(nested_message.clone()); } - if let Some(deeper) = self.find_nested_message(nested_message, target_type, package) { + if let Some(deeper) = + self.find_nested_message(nested_message, target_type, &parent_prefix) + { return Some(deeper); } } @@ -369,6 +367,22 @@ impl ProtoStreamDecoder { Err(Error::InvalidProtobufPayload) } + fn parse_fixed_integer( + data: &[u8], + cursor: usize, + wire_type: u8, + ) -> Result<(u64, usize), Error> { + let width = match wire_type { + 1 => size_of::(), + 5 => size_of::(), + _ => return Err(Error::InvalidProtobufPayload), + }; + let end_cursor = Self::length_delimited_end(cursor, width as u64, data.len())?; + let mut bytes = [0; size_of::()]; + bytes[..width].copy_from_slice(&data[cursor..end_cursor]); + Ok((u64::from_le_bytes(bytes), end_cursor)) + } + fn decode_field_value( &self, data: &[u8], @@ -384,11 +398,14 @@ impl ProtoStreamDecoder { Type::Bool => { simd_json::OwnedValue::Static(simd_json::StaticNode::Bool(value != 0)) } - Type::Int32 | Type::Sint32 | Type::Sfixed32 => { - simd_json::OwnedValue::from(value as i32) + Type::Int32 | Type::Sfixed32 => simd_json::OwnedValue::from(value as i32), + Type::Int64 | Type::Sfixed64 => simd_json::OwnedValue::from(value as i64), + Type::Sint32 => { + let value = value as u32; + simd_json::OwnedValue::from(((value >> 1) as i32) ^ -((value & 1) as i32)) } - Type::Int64 | Type::Sint64 | Type::Sfixed64 => { - simd_json::OwnedValue::from(value as i64) + Type::Sint64 => { + simd_json::OwnedValue::from(((value >> 1) as i64) ^ -((value & 1) as i64)) } Type::Uint32 | Type::Fixed32 => simd_json::OwnedValue::from(value as u32), Type::Uint64 | Type::Fixed64 => simd_json::OwnedValue::from(value), @@ -396,6 +413,19 @@ impl ProtoStreamDecoder { }; Ok((json_value, new_cursor)) } + 1 | 5 => { + let (value, new_cursor) = Self::parse_fixed_integer(data, cursor, wire_type)?; + let json_value = match (wire_type, field_desc.r#type()) { + (1, Type::Double) => simd_json::OwnedValue::from(f64::from_bits(value)), + (1, Type::Fixed64) => simd_json::OwnedValue::from(value), + (1, Type::Sfixed64) => simd_json::OwnedValue::from(value as i64), + (5, Type::Float) => simd_json::OwnedValue::from(f32::from_bits(value as u32)), + (5, Type::Fixed32) => simd_json::OwnedValue::from(value as u32), + (5, Type::Sfixed32) => simd_json::OwnedValue::from(value as u32 as i32), + _ => simd_json::OwnedValue::String("unsupported_wire_type".into()), + }; + Ok((json_value, new_cursor)) + } 2 => { let (length, mut new_cursor) = self.parse_simple_varint(data, cursor)?; let end_cursor = Self::length_delimited_end(new_cursor, length, data.len())?; @@ -445,6 +475,10 @@ impl ProtoStreamDecoder { let (value, new_cursor) = self.parse_simple_varint(data, cursor)?; Ok((simd_json::OwnedValue::from(value), new_cursor)) } + 1 | 5 => { + let (_, new_cursor) = Self::parse_fixed_integer(data, cursor, wire_type)?; + Ok((simd_json::OwnedValue::String("unknown".into()), new_cursor)) + } 2 => { let (length, mut new_cursor) = self.parse_simple_varint(data, cursor)?; let end_cursor = Self::length_delimited_end(new_cursor, length, data.len())?; @@ -465,6 +499,10 @@ impl ProtoStreamDecoder { let (_, new_cursor) = self.parse_simple_varint(data, cursor)?; Ok(new_cursor) } + 1 | 5 => { + let (_, new_cursor) = Self::parse_fixed_integer(data, cursor, wire_type)?; + Ok(new_cursor) + } 2 => { let (length, new_cursor) = self.parse_simple_varint(data, cursor)?; let end_cursor = Self::length_delimited_end(new_cursor, length, data.len())?; @@ -672,7 +710,7 @@ mod tests { } #[test] - fn load_schema_should_handle_missing_proto_file_gracefully() { + fn given_missing_proto_file_when_loading_schema_should_return_error() { let mut decoder = ProtoStreamDecoder::new(ProtoConfig { schema_path: Some(PathBuf::from("nonexistent.proto")), message_type: Some("com.example.Test".to_string()), @@ -681,10 +719,7 @@ mod tests { let result = decoder.load_schema(); - assert!( - result.is_ok(), - "Should handle missing proto file gracefully" - ); + assert!(matches!(result, Err(Error::InitError(_))), "{result:?}"); } #[test] @@ -735,21 +770,26 @@ mod tests { } #[test] - fn update_config_should_reload_schema_when_requested() { + fn given_valid_schema_when_updating_config_should_decode_with_reloaded_schema() { let mut decoder = ProtoStreamDecoder::new(ProtoConfig::default()); let new_config = ProtoConfig { - schema_path: Some(PathBuf::from("schemas/test.proto")), - message_type: Some("com.example.Test".to_string()), + schema_path: Some(PathBuf::from("examples/user.proto")), + message_type: Some("com.example.User".to_string()), use_any_wrapper: false, ..ProtoConfig::default() }; - let result = decoder.update_config(new_config.clone(), true); - assert!(result.is_ok()); - assert_eq!(decoder.config.schema_path, new_config.schema_path); - assert_eq!(decoder.config.message_type, new_config.message_type); - assert_eq!(decoder.config.use_any_wrapper, new_config.use_any_wrapper); + decoder + .update_config(new_config, true) + .expect("reload user schema"); + let Payload::Json(decoded) = decoder + .decode(42i32.encode_to_vec()) + .expect("decode user id") + else { + panic!("expected reloaded schema"); + }; + assert_eq!(decoded, simd_json::json!({"id": 42})); } #[test] diff --git a/core/connectors/sdk/src/encoders/proto.rs b/core/connectors/sdk/src/encoders/proto.rs index 8a2590b9aa..44dca94f05 100644 --- a/core/connectors/sdk/src/encoders/proto.rs +++ b/core/connectors/sdk/src/encoders/proto.rs @@ -105,26 +105,32 @@ impl ProtoStreamEncoder { config: ProtoEncoderConfig, reload_schema: bool, ) -> Result<(), Error> { - self.config = config; - if reload_schema - && (self.config.schema_path.is_some() || self.config.descriptor_set.is_some()) - { - self.load_schema() - } else { - Ok(()) + let old_config = std::mem::replace(&mut self.config, config); + if reload_schema && let Err(error) = self.load_schema() { + self.config = old_config; + return Err(error); } + Ok(()) } pub fn load_schema(&mut self) -> Result<(), Error> { let schema_path = self.config.schema_path.clone(); let descriptor_set = self.config.descriptor_set.clone(); - if let Some(path) = schema_path { - self.compile_schema_internal(&path)?; + let old_message_descriptor = self.message_descriptor.take(); + let old_file_descriptor_set = self.file_descriptor_set.take(); + let result = if let Some(path) = schema_path { + self.compile_schema_internal(&path) } else if let Some(descriptor_bytes) = descriptor_set { - self.load_descriptor_set_internal(&descriptor_bytes)?; + self.load_descriptor_set_internal(&descriptor_bytes) + } else { + Ok(()) + }; + if result.is_err() { + self.message_descriptor = old_message_descriptor; + self.file_descriptor_set = old_file_descriptor_set; } - Ok(()) + result } fn compile_schema_internal(&mut self, schema_path: &PathBuf) -> Result<(), Error> { @@ -137,14 +143,8 @@ impl ProtoStreamEncoder { schema_path ); - let proto_content = match fs::read_to_string(schema_path) { - Ok(content) => content, - Err(e) => { - error!("Failed to read proto file: {}", e); - error!("Falling back to Any wrapper mode"); - return Ok(()); - } - }; + let proto_content = fs::read_to_string(schema_path) + .map_err(|error| Error::InitError(format!("Failed to read proto file: {error}")))?; let parsed_file = parse(&schema_path.to_string_lossy(), &proto_content) .map_err(|e| Error::InitError(format!("Failed to parse proto file: {e}")))?; @@ -181,11 +181,9 @@ impl ProtoStreamEncoder { self.file_descriptor_set = Some(file_descriptor_set); Ok(()) } - Err(e) => { - error!("Failed to compile proto schema: {}", e); - error!("Falling back to Any wrapper mode"); - Ok(()) - } + Err(error) => Err(Error::InitError(format!( + "Failed to compile proto schema: {error}" + ))), } } @@ -231,14 +229,15 @@ impl ProtoStreamEncoder { package: &str, ) -> Option { let parent_name = parent_message.name.as_deref().unwrap_or(""); + let parent_prefix = if package.is_empty() { + parent_name.to_string() + } else { + format!("{package}.{parent_name}") + }; for nested_message in &parent_message.nested_type { let nested_name = nested_message.name.as_deref().unwrap_or(""); - let full_name = if package.is_empty() { - format!("{parent_name}.{nested_name}") - } else { - format!("{package}.{parent_name}.{nested_name}") - }; + let full_name = format!("{parent_prefix}.{nested_name}"); if full_name == target_type { info!( @@ -248,7 +247,9 @@ impl ProtoStreamEncoder { return Some(nested_message.clone()); } - if let Some(deeper) = self.find_nested_message(nested_message, target_type, package) { + if let Some(deeper) = + self.find_nested_message(nested_message, target_type, &parent_prefix) + { return Some(deeper); } } @@ -432,30 +433,58 @@ impl ProtoStreamEncoder { self.encode_varint(&mut bytes, value); Ok(bytes) } - Type::Int32 | Type::Sint32 | Type::Sfixed32 => { + Type::Int32 => { let value = self.extract_i32_from_json(json_value)? as i64 as u64; let mut bytes = Vec::new(); self.encode_varint(&mut bytes, value); Ok(bytes) } - Type::Int64 | Type::Sint64 | Type::Sfixed64 => { + Type::Int64 => { let value = self.extract_i64_from_json(json_value)? as u64; let mut bytes = Vec::new(); self.encode_varint(&mut bytes, value); Ok(bytes) } - Type::Uint32 | Type::Fixed32 => { - let value = self.extract_u32_from_json(json_value)? as u64; + Type::Uint32 => { + let value = Self::extract_u32_from_json(json_value)? as u64; let mut bytes = Vec::new(); self.encode_varint(&mut bytes, value); Ok(bytes) } - Type::Uint64 | Type::Fixed64 => { - let value = self.extract_u64_from_json(json_value)?; + Type::Uint64 => { + let value = Self::extract_u64_from_json(json_value)?; let mut bytes = Vec::new(); self.encode_varint(&mut bytes, value); Ok(bytes) } + Type::Sint32 => { + let value = self.extract_i32_from_json(json_value)?; + let zigzag = ((value << 1) ^ (value >> 31)) as u32; + let mut bytes = Vec::new(); + self.encode_varint(&mut bytes, u64::from(zigzag)); + Ok(bytes) + } + Type::Sint64 => { + let value = self.extract_i64_from_json(json_value)?; + let zigzag = ((value << 1) ^ (value >> 63)) as u64; + let mut bytes = Vec::new(); + self.encode_varint(&mut bytes, zigzag); + Ok(bytes) + } + Type::Fixed32 => Ok(Self::extract_u32_from_json(json_value)? + .to_le_bytes() + .to_vec()), + Type::Fixed64 => Ok(Self::extract_u64_from_json(json_value)? + .to_le_bytes() + .to_vec()), + Type::Sfixed32 => Ok(self + .extract_i32_from_json(json_value)? + .to_le_bytes() + .to_vec()), + Type::Sfixed64 => Ok(self + .extract_i64_from_json(json_value)? + .to_le_bytes() + .to_vec()), Type::String => { let text = match json_value { simd_json::OwnedValue::String(s) => s.as_str(), @@ -468,35 +497,13 @@ impl ProtoStreamEncoder { result.extend_from_slice(text_bytes); Ok(result) } - Type::Bytes => { - let bytes = match json_value { - simd_json::OwnedValue::String(s) => general_purpose::STANDARD - .decode(s.as_str()) - .map_err(|_| Error::InvalidJsonPayload)?, - _ => return Err(Error::InvalidJsonPayload), - }; + Type::Bytes | Type::Message => { + let bytes = Self::extract_bytes_from_json(json_value)?; let mut result = Vec::new(); - self.encode_varint(&mut result, bytes.len() as u64); result.extend_from_slice(&bytes); Ok(result) } - Type::Message => { - let message_bytes = match json_value { - simd_json::OwnedValue::String(s) => general_purpose::STANDARD - .decode(s.as_str()) - .map_err(|_| Error::InvalidJsonPayload)?, - simd_json::OwnedValue::Object(_) => { - return Err(Error::InvalidJsonPayload); - } - _ => return Err(Error::InvalidJsonPayload), - }; - let mut result = Vec::new(); - - self.encode_varint(&mut result, message_bytes.len() as u64); - result.extend_from_slice(&message_bytes); - Ok(result) - } _ => { error!("Unsupported field type: {:?}", field_desc.r#type()); Err(Error::InvalidJsonPayload) @@ -554,7 +561,18 @@ impl ProtoStreamEncoder { } } - fn extract_u32_from_json(&self, json_value: &simd_json::OwnedValue) -> Result { + pub(crate) fn extract_bytes_from_json( + json_value: &simd_json::OwnedValue, + ) -> Result, Error> { + let simd_json::OwnedValue::String(value) = json_value else { + return Err(Error::InvalidJsonPayload); + }; + general_purpose::STANDARD + .decode(value) + .map_err(|_| Error::InvalidJsonPayload) + } + + pub(crate) fn extract_u32_from_json(json_value: &simd_json::OwnedValue) -> Result { match json_value { simd_json::OwnedValue::String(s) => { s.parse::().map_err(|_| Error::InvalidJsonPayload) @@ -566,7 +584,7 @@ impl ProtoStreamEncoder { } } - fn extract_u64_from_json(&self, json_value: &simd_json::OwnedValue) -> Result { + pub(crate) fn extract_u64_from_json(json_value: &simd_json::OwnedValue) -> Result { match json_value { simd_json::OwnedValue::String(s) => { s.parse::().map_err(|_| Error::InvalidJsonPayload) @@ -588,7 +606,7 @@ impl ProtoStreamEncoder { "{}/google.protobuf.StringValue", self.config.format_options.type_url_prefix ), - json_string.into_bytes(), + json_string.encode_to_vec(), ) } Payload::Text(text) => { @@ -604,7 +622,7 @@ impl ProtoStreamEncoder { "{}/google.protobuf.StringValue", self.config.format_options.type_url_prefix ), - json_string.into_bytes(), + json_string.encode_to_vec(), ) } Payload::Raw(data) => ( @@ -612,28 +630,28 @@ impl ProtoStreamEncoder { "{}/google.protobuf.BytesValue", self.config.format_options.type_url_prefix ), - data, + data.encode_to_vec(), ), Payload::Proto(text) => ( format!( "{}/google.protobuf.StringValue", self.config.format_options.type_url_prefix ), - text.into_bytes(), + text.encode_to_vec(), ), Payload::FlatBuffer(data) => ( format!( "{}/google.protobuf.BytesValue", self.config.format_options.type_url_prefix ), - data, + data.encode_to_vec(), ), Payload::Avro(data) => ( format!( "{}/google.protobuf.BytesValue", self.config.format_options.type_url_prefix ), - data, + data.encode_to_vec(), ), }; @@ -751,7 +769,19 @@ mod tests { let any = decoded_any.unwrap(); assert!(any.type_url.contains("google.protobuf.StringValue")); - assert!(!any.value.is_empty()); + let mut json_bytes = any + .to_msg::() + .expect("unpack StringValue") + .into_bytes(); + let decoded_json = simd_json::to_owned_value(&mut json_bytes).expect("decode JSON"); + assert_eq!( + decoded_json, + simd_json::json!({ + "user_id": 123, + "name": "John Doe", + "email": "john@example.com" + }) + ); } #[test] @@ -765,7 +795,7 @@ mod tests { let encoded_bytes = result.unwrap(); let decoded_any = Any::decode(encoded_bytes.as_slice()).unwrap(); - let json_string = String::from_utf8(decoded_any.value).unwrap(); + let json_string = decoded_any.to_msg::().expect("unpack StringValue"); let mut json_bytes = json_string.into_bytes(); let json_value = simd_json::to_owned_value(&mut json_bytes).unwrap(); @@ -790,7 +820,10 @@ mod tests { let decoded_any = Any::decode(encoded_bytes.as_slice()).unwrap(); assert!(decoded_any.type_url.contains("google.protobuf.BytesValue")); - assert_eq!(decoded_any.value, raw_data); + assert_eq!( + decoded_any.to_msg::>().expect("unpack BytesValue"), + raw_data + ); } #[test] @@ -805,7 +838,29 @@ mod tests { let decoded_any = Any::decode(encoded_bytes.as_slice()).unwrap(); assert!(decoded_any.type_url.contains("google.protobuf.StringValue")); - assert_eq!(decoded_any.value, proto_text.as_bytes()); + assert_eq!( + decoded_any.to_msg::().expect("unpack StringValue"), + proto_text + ); + } + + #[test] + fn given_binary_payload_when_wrapped_should_unpack_bytes_value() { + let encoder = ProtoStreamEncoder::default(); + for data in [Vec::new(), vec![0, 255, 128, 10]] { + for payload in [ + Payload::Raw(data.clone()), + Payload::FlatBuffer(data.clone()), + Payload::Avro(data.clone()), + ] { + let encoded = encoder.encode(payload).expect("encode binary payload"); + let wrapped = Any::decode(encoded.as_slice()).expect("decode Any"); + assert_eq!( + wrapped.to_msg::>().expect("unpack BytesValue"), + data + ); + } + } } #[test] @@ -901,7 +956,7 @@ mod tests { } #[test] - fn load_schema_should_handle_missing_proto_file_gracefully() { + fn given_missing_proto_file_when_loading_schema_should_return_error() { let mut encoder = ProtoStreamEncoder::new_with_config(ProtoEncoderConfig { schema_path: Some(PathBuf::from("nonexistent.proto")), message_type: Some("com.example.Test".to_string()), @@ -909,10 +964,7 @@ mod tests { }); let result = encoder.load_schema(); - assert!( - result.is_ok(), - "Should handle missing proto file gracefully" - ); + assert!(matches!(result, Err(Error::InitError(_))), "{result:?}"); } #[test] diff --git a/core/connectors/sdk/src/lib.rs b/core/connectors/sdk/src/lib.rs index 7aeb02bc1a..0d6efc51b5 100644 --- a/core/connectors/sdk/src/lib.rs +++ b/core/connectors/sdk/src/lib.rs @@ -108,7 +108,7 @@ pub trait Source: Send + Sync { /// Invoked when the source is initialized, allowing it to perform any necessary setup. async fn open(&mut self) -> Result<(), Error>; - /// Invoked every time a batch of messages is produced to the configured stream and topic. + /// Retrieves the next batch for the runtime to process and deliver. async fn poll(&self) -> Result; /// Invoked after the runtime has finished processing the most recently polled batch. diff --git a/core/connectors/sdk/src/transforms/avro_convert.rs b/core/connectors/sdk/src/transforms/avro_convert.rs index 6ac1819ca1..5a14dd0007 100644 --- a/core/connectors/sdk/src/transforms/avro_convert.rs +++ b/core/connectors/sdk/src/transforms/avro_convert.rs @@ -120,7 +120,7 @@ impl Transform for AvroConvert { let encoder_config = AvroEncoderConfig { schema_path: self.config.schema_path.clone(), schema_json: self.config.schema_json.clone(), - field_mappings: self.config.field_mappings.clone(), + field_mappings: None, }; let encoder = AvroStreamEncoder::new(encoder_config); let encoded_bytes = encoder.encode(message.payload)?; @@ -184,6 +184,10 @@ impl Default for AvroConvert { #[cfg(test)] mod tests { + use apache_avro::{ + Schema as AvroSchema, reader::datum::GenericDatumReader, types::Value as AvroValue, + }; + use super::*; use crate::TopicMetadata; @@ -233,6 +237,48 @@ mod tests { .unwrap() } + #[test] + fn given_chained_field_names_when_encoding_should_apply_each_mapping_once() { + let schema_json = create_test_schema_json(); + let schema = AvroSchema::parse_str(&schema_json).expect("parse User schema"); + let converter = AvroConvert::new(AvroConvertConfig { + source_format: Schema::Json, + target_format: Schema::Avro, + schema_json: Some(schema_json), + field_mappings: Some(HashMap::from([ + ("old_name".to_string(), "name".to_string()), + ("name".to_string(), "archived_name".to_string()), + ])), + ..AvroConvertConfig::default() + }); + let message = create_test_message(Payload::Json(simd_json::json!({ + "old_name": "Alice", + "age": 30, + }))); + let transformed = converter + .transform(&create_test_metadata(), message) + .expect("encode with one field rename") + .expect("preserve message"); + let Payload::Avro(data) = transformed.payload else { + panic!("expected Avro datum"); + }; + let reader = GenericDatumReader::builder(&schema) + .build() + .expect("create Avro reader"); + let mut bytes = data.as_slice(); + let decoded = reader + .read_value(&mut bytes) + .expect("decode transformed datum"); + assert_eq!( + decoded, + AvroValue::Record(vec![ + ("name".to_string(), AvroValue::String("Alice".to_string())), + ("age".to_string(), AvroValue::Int(30)), + ]) + ); + assert!(bytes.is_empty()); + } + #[test] fn transform_should_convert_avro_to_json_successfully() { let schema_json = create_test_schema_json(); diff --git a/core/connectors/sdk/src/transforms/filter_fields.rs b/core/connectors/sdk/src/transforms/filter_fields.rs index a678882489..bb6460b680 100644 --- a/core/connectors/sdk/src/transforms/filter_fields.rs +++ b/core/connectors/sdk/src/transforms/filter_fields.rs @@ -129,9 +129,9 @@ impl ValuePattern { Equals(x) => v == x, Contains(s) => v.as_str().is_some_and(|x| x.contains(s)), Regex(re) => v.as_str().is_some_and(|x| re.is_match(x)), - GreaterThan(t) => v.as_f64().is_some_and(|n| n > *t), - LessThan(t) => v.as_f64().is_some_and(|n| n < *t), - Between(a, b) => v.as_f64().is_some_and(|n| n >= *a && n <= *b), + GreaterThan(t) => v.cast_f64().is_some_and(|n| n > *t), + LessThan(t) => v.cast_f64().is_some_and(|n| n < *t), + Between(a, b) => v.cast_f64().is_some_and(|n| n >= *a && n <= *b), IsNull => v.is_null(), IsNotNull => !v.is_null(), IsString => v.is_str(), diff --git a/core/connectors/sdk/src/transforms/json/filter_fields.rs b/core/connectors/sdk/src/transforms/json/filter_fields.rs index ec6ce5e086..340dce2008 100644 --- a/core/connectors/sdk/src/transforms/json/filter_fields.rs +++ b/core/connectors/sdk/src/transforms/json/filter_fields.rs @@ -63,6 +63,59 @@ mod tests { }, }; + #[test] + fn given_integer_and_float_fields_when_comparing_should_filter_all_numbers() { + let input = r#"{ + "negative": -2, + "zero": 0, + "positive": 2, + "fraction": 2.5, + "unsigned": 18446744073709551615, + "text": "2", + "boolean": true, + "null": null + }"#; + for (pattern, included, excluded) in [ + ( + ValuePattern::GreaterThan(1.0), + r#"{"positive":2,"fraction":2.5,"unsigned":18446744073709551615}"#, + r#"{"negative":-2,"zero":0,"text":"2","boolean":true,"null":null}"#, + ), + ( + ValuePattern::LessThan(1.0), + r#"{"negative":-2,"zero":0}"#, + r#"{"positive":2,"fraction":2.5,"unsigned":18446744073709551615,"text":"2","boolean":true,"null":null}"#, + ), + ( + ValuePattern::Between(-2.0, 2.0), + r#"{"negative":-2,"zero":0,"positive":2}"#, + r#"{"fraction":2.5,"unsigned":18446744073709551615,"text":"2","boolean":true,"null":null}"#, + ), + ] { + for (include_matching, expected) in [(true, included), (false, excluded)] { + let transform = FilterFields::new(FilterFieldsConfig { + keep_fields: vec![], + patterns: vec![FilterPattern { + key_pattern: None, + value_pattern: Some(pattern.clone()), + }], + include_matching, + }) + .expect("construct numeric filter"); + let result = transform + .transform(&create_test_topic_metadata(), create_test_message(input)) + .expect("filter numeric fields") + .expect("preserve message"); + let expected = create_test_message(expected); + assert_eq!( + extract_json_object(&result), + extract_json_object(&expected), + "pattern={pattern:?}, include_matching={include_matching}", + ); + } + } + } + #[test] fn should_keep_only_specified_fields_when_keep_list_provided() { let transform = FilterFields::new(FilterFieldsConfig { diff --git a/core/connectors/sdk/src/transforms/proto_convert.rs b/core/connectors/sdk/src/transforms/proto_convert.rs index a8874ea1d7..510efa8c03 100644 --- a/core/connectors/sdk/src/transforms/proto_convert.rs +++ b/core/connectors/sdk/src/transforms/proto_convert.rs @@ -19,11 +19,13 @@ use base64::Engine; use iggy_common::IggyTimestamp; use prost::Message; use serde::{Deserialize, Serialize}; +use simd_json::prelude::ValueAsScalar; use std::collections::HashMap; use std::path::PathBuf; use tracing::{error, info}; use super::{Transform, TransformType}; +use crate::encoders::proto::ProtoStreamEncoder; use crate::{DecodedMessage, Error, Payload, Schema, TopicMetadata}; #[derive(Debug, Clone, Serialize, Deserialize)] @@ -109,12 +111,20 @@ impl ProtoConvert { let schema_path = self.config.schema_path.clone(); let descriptor_set = self.config.descriptor_set.clone(); - if let Some(path) = schema_path { - self.compile_schema_internal(&path)?; + let old_message_descriptor = self.message_descriptor.take(); + let old_file_descriptor_set = self.file_descriptor_set.take(); + let result = if let Some(path) = schema_path { + self.compile_schema_internal(&path) } else if let Some(descriptor_bytes) = descriptor_set { - self.load_descriptor_set_internal(&descriptor_bytes)?; + self.load_descriptor_set_internal(&descriptor_bytes) + } else { + Ok(()) + }; + if result.is_err() { + self.message_descriptor = old_message_descriptor; + self.file_descriptor_set = old_file_descriptor_set; } - Ok(()) + result } fn compile_schema_internal(&mut self, schema_path: &PathBuf) -> Result<(), Error> { @@ -127,14 +137,8 @@ impl ProtoConvert { schema_path ); - let proto_content = match fs::read_to_string(schema_path) { - Ok(content) => content, - Err(e) => { - error!("Failed to read proto file: {}", e); - error!("Falling back to basic conversion methods"); - return Ok(()); - } - }; + let proto_content = fs::read_to_string(schema_path) + .map_err(|error| Error::InitError(format!("Failed to read proto file: {error}")))?; let parsed_file = parse(&schema_path.to_string_lossy(), &proto_content) .map_err(|e| Error::InitError(format!("Failed to parse proto file: {e}")))?; @@ -171,12 +175,9 @@ impl ProtoConvert { self.file_descriptor_set = Some(file_descriptor_set); Ok(()) } - Err(e) => { - error!("Failed to compile proto schema: {}", e); - error!("Falling back to basic conversion methods"); - - Ok(()) - } + Err(error) => Err(Error::InitError(format!( + "Failed to compile proto schema: {error}" + ))), } } @@ -219,6 +220,11 @@ impl ProtoConvert { package: &str, ) -> Option { let parent_name = parent_message.name.as_deref().unwrap_or(""); + let parent_prefix = if package.is_empty() { + parent_name.to_string() + } else { + format!("{package}.{parent_name}") + }; let package_prefix = if package.is_empty() { String::new() @@ -238,7 +244,9 @@ impl ProtoConvert { return Some(nested_message.clone()); } - if let Some(deeper) = self.find_nested_message(nested_message, target_type, package) { + if let Some(deeper) = + self.find_nested_message(nested_message, target_type, &parent_prefix) + { return Some(deeper); } } @@ -281,19 +289,21 @@ impl ProtoConvert { let field_name = field_desc.name.as_deref().unwrap_or(""); if let Some(json_field_value) = json_map.get(field_name) { + let field_data = match self + .encode_field_value_for_conversion(json_field_value, field_desc) + { + Ok(field_data) => field_data, + Err(error) => { + error!("Failed to encode field {}: {}", field_name, error); + continue; + } + }; let field_number = field_desc.number() as u64; let wire_type = self.get_wire_type_for_conversion_field(field_desc); let tag = (field_number << 3) | (wire_type as u64); self.encode_varint_for_conversion(&mut buffer, tag); - - match self.encode_field_value_for_conversion(json_field_value, field_desc) { - Ok(field_data) => buffer.extend_from_slice(&field_data), - Err(e) => { - error!("Failed to encode field {}: {}", field_name, e); - continue; - } - } + buffer.extend_from_slice(&field_data); } } @@ -343,18 +353,75 @@ impl ProtoConvert { Err(Error::InvalidJsonPayload) } } - Type::Int32 | Type::Sint32 => { + Type::Float => Ok( + (json_value.cast_f64().ok_or(Error::InvalidJsonPayload)? as f32) + .to_le_bytes() + .to_vec(), + ), + Type::Double => Ok(json_value + .cast_f64() + .ok_or(Error::InvalidJsonPayload)? + .to_le_bytes() + .to_vec()), + Type::Bytes | Type::Message => { + let bytes = ProtoStreamEncoder::extract_bytes_from_json(json_value)?; + let mut result = Vec::new(); + self.encode_varint_for_conversion(&mut result, bytes.len() as u64); + result.extend_from_slice(&bytes); + Ok(result) + } + Type::Int32 => { let value = self.extract_i32_from_json_for_conversion(json_value)?; let mut result = Vec::new(); self.encode_varint_for_conversion(&mut result, value as u64); Ok(result) } - Type::Int64 | Type::Sint64 => { + Type::Int64 => { let value = self.extract_i64_from_json_for_conversion(json_value)?; let mut result = Vec::new(); self.encode_varint_for_conversion(&mut result, value as u64); Ok(result) } + Type::Sint32 => { + let value = self.extract_i32_from_json_for_conversion(json_value)?; + let zigzag = ((value << 1) ^ (value >> 31)) as u32; + let mut result = Vec::new(); + self.encode_varint_for_conversion(&mut result, u64::from(zigzag)); + Ok(result) + } + Type::Sint64 => { + let value = self.extract_i64_from_json_for_conversion(json_value)?; + let zigzag = ((value << 1) ^ (value >> 63)) as u64; + let mut result = Vec::new(); + self.encode_varint_for_conversion(&mut result, zigzag); + Ok(result) + } + Type::Uint32 => { + let value = ProtoStreamEncoder::extract_u32_from_json(json_value)?; + let mut result = Vec::new(); + self.encode_varint_for_conversion(&mut result, u64::from(value)); + Ok(result) + } + Type::Uint64 => { + let value = ProtoStreamEncoder::extract_u64_from_json(json_value)?; + let mut result = Vec::new(); + self.encode_varint_for_conversion(&mut result, value); + Ok(result) + } + Type::Fixed32 => Ok(ProtoStreamEncoder::extract_u32_from_json(json_value)? + .to_le_bytes() + .to_vec()), + Type::Fixed64 => Ok(ProtoStreamEncoder::extract_u64_from_json(json_value)? + .to_le_bytes() + .to_vec()), + Type::Sfixed32 => Ok(self + .extract_i32_from_json_for_conversion(json_value)? + .to_le_bytes() + .to_vec()), + Type::Sfixed64 => Ok(self + .extract_i64_from_json_for_conversion(json_value)? + .to_le_bytes() + .to_vec()), Type::Bool => { let value = if let simd_json::OwnedValue::Static(simd_json::StaticNode::Bool(b)) = json_value @@ -382,15 +449,6 @@ impl ProtoConvert { result.extend_from_slice(bytes); Ok(result) } - _ => { - let json_string = - simd_json::to_string(json_value).map_err(|_| Error::InvalidJsonPayload)?; - let bytes = json_string.as_bytes(); - let mut result = Vec::new(); - self.encode_varint_for_conversion(&mut result, bytes.len() as u64); - result.extend_from_slice(bytes); - Ok(result) - } } } @@ -746,6 +804,64 @@ mod tests { use std::collections::HashMap; use std::path::PathBuf; + #[derive(Clone, PartialEq, prost::Message)] + struct StringRecord { + #[prost(string, tag = "1")] + first: String, + #[prost(string, tag = "2")] + second: String, + } + + #[test] + fn given_invalid_field_when_converting_with_schema_should_keep_valid_wire_data() { + let descriptor = protox_parse::parse( + "record.proto", + r#"syntax = "proto3"; + message StringRecord { + string first = 1; + string second = 2; + }"#, + ) + .expect("test schema must parse"); + let descriptor_set = prost_types::FileDescriptorSet { + file: vec![descriptor], + }; + let converter = ProtoConvert::new(ProtoConvertConfig { + source_format: Schema::Json, + target_format: Schema::Proto, + message_type: Some("StringRecord".to_string()), + descriptor_set: Some(descriptor_set.encode_to_vec()), + ..ProtoConvertConfig::default() + }); + for (payload, first, second) in [ + ( + simd_json::json!({"first": 123, "second": "kept"}), + "", + "kept", + ), + ( + simd_json::json!({"first": "kept", "second": 123}), + "kept", + "", + ), + ] { + let converted = converter + .transform( + &create_test_metadata(), + create_test_message(Payload::Json(payload)), + ) + .expect("invalid fields are skipped") + .expect("the valid sibling keeps the message"); + let Payload::Raw(bytes) = converted.payload else { + panic!("a loaded schema must produce binary protobuf"); + }; + let decoded = StringRecord::decode(bytes.as_slice()) + .expect("skipping a field must not leave an incomplete protobuf tag"); + assert_eq!(decoded.first, first); + assert_eq!(decoded.second, second); + } + } + fn create_test_message(payload: Payload) -> DecodedMessage { DecodedMessage { id: Some(123), @@ -1098,7 +1214,7 @@ mod tests { } #[test] - fn load_schema_should_log_warning_for_unimplemented_schema_compilation() { + fn given_missing_proto_file_when_loading_schema_should_return_error() { let mut converter = ProtoConvert::new(ProtoConvertConfig { schema_path: Some(PathBuf::from("test.proto")), message_type: Some("com.example.Test".to_string()), @@ -1107,10 +1223,7 @@ mod tests { let result = converter.load_schema(); - assert!( - result.is_ok(), - "Should handle unimplemented schema compilation gracefully" - ); + assert!(matches!(result, Err(Error::InitError(_))), "{result:?}"); } #[test] diff --git a/core/connectors/sdk/tests/protobuf_integration.rs b/core/connectors/sdk/tests/protobuf_integration.rs index aaf80c8408..287f0b7772 100644 --- a/core/connectors/sdk/tests/protobuf_integration.rs +++ b/core/connectors/sdk/tests/protobuf_integration.rs @@ -15,15 +15,619 @@ // specific language governing permissions and limitations // under the License. +use base64::Engine; use iggy_connector_sdk::decoders::proto::{ProtoConfig, ProtoStreamDecoder}; use iggy_connector_sdk::encoders::proto::{ProtoEncoderConfig, ProtoStreamEncoder}; use iggy_connector_sdk::transforms::{ProtoConvert, ProtoConvertConfig, Transform}; -use iggy_connector_sdk::{Payload, Schema, StreamDecoder, StreamEncoder}; +use iggy_connector_sdk::{Error, Payload, Schema, StreamDecoder, StreamEncoder}; use prost::Message; use prost_types::Any; +use simd_json::prelude::ValueAsScalar; use std::collections::HashMap; use std::path::PathBuf; +const INTEGER_SCHEMA: &str = r#" + syntax = "proto3"; + message IntegerRecord { + int32 int32_value = 1; + int64 int64_value = 2; + uint32 uint32_value = 3; + uint64 uint64_value = 4; + sint32 sint32_value = 5; + sint64 sint64_value = 6; + fixed32 fixed32_value = 7; + fixed64 fixed64_value = 8; + sfixed32 sfixed32_value = 9; + sfixed64 sfixed64_value = 10; + } + "#; + +#[derive(Clone, PartialEq, Message)] +struct ValueRecord { + #[prost(float, tag = "1")] + single: f32, + #[prost(double, tag = "2")] + double: f64, + #[prost(bytes, tag = "3")] + bytes: Vec, + #[prost(message, optional, tag = "4")] + nested: Option, +} + +#[test] +fn given_float_and_binary_fields_when_converting_should_roundtrip_values() { + let mut schema = protox_parse::parse( + "values.proto", + r#" + syntax = "proto3"; + message Nested { string value = 1; } + message ValueRecord { + float single = 1; + double double = 2; + bytes bytes = 3; + Nested nested = 4; + } + "#, + ) + .expect("parse scalar and binary schema"); + let record_schema = schema + .message_type + .iter_mut() + .find(|message| message.name() == "ValueRecord") + .expect("find record schema"); + let nested_field = record_schema + .field + .iter_mut() + .find(|field| field.name() == "nested") + .expect("find nested message field"); + nested_field.r#type = Some(prost_types::field_descriptor_proto::Type::Message as i32); + nested_field.type_name = Some(".Nested".to_string()); + let descriptor_set = prost_types::FileDescriptorSet { file: vec![schema] }.encode_to_vec(); + let decoder = ProtoStreamDecoder::new(ProtoConfig { + descriptor_set: Some(descriptor_set.clone()), + message_type: Some("ValueRecord".to_string()), + ..ProtoConfig::default() + }); + let converter = ProtoConvert::new(ProtoConvertConfig { + source_format: Schema::Json, + target_format: Schema::Proto, + descriptor_set: Some(descriptor_set), + message_type: Some("ValueRecord".to_string()), + ..ProtoConvertConfig::default() + }); + for (single, double) in [ + (1.25, -2.5), + (f32::MIN, f64::MIN), + (f32::MAX, f64::MAX), + (f32::MIN_POSITIVE, f64::MIN_POSITIVE), + (f32::from_bits(1), f64::from_bits(1)), + (0.0, -0.0), + (-0.0, 0.0), + ] { + let expected = ValueRecord { + single, + double, + bytes: vec![0, 1, 255], + nested: Some("nested".to_string()), + }; + let converted = converter.transform( + &iggy_connector_sdk::TopicMetadata { stream: "values".to_string(), topic: "values".to_string() }, + iggy_connector_sdk::DecodedMessage { + id: None, offset: None, checksum: None, timestamp: None, + origin_timestamp: None, headers: None, + payload: Payload::Json(simd_json::json!({ + "single": single, "double": double, + "bytes": base64::engine::general_purpose::STANDARD.encode(&expected.bytes), + "nested": base64::engine::general_purpose::STANDARD.encode("nested".to_string().encode_to_vec()), + })), + }, + ).expect("convert scalar and binary record").expect("preserve record"); + let Payload::Raw(encoded) = converted.payload else { + panic!("expected schema-encoded protobuf bytes"); + }; + let decoded = + ValueRecord::decode(encoded.as_slice()).expect("decode converted values with prost"); + assert_eq!(decoded, expected); + let Payload::Json(decoded) = decoder.decode(encoded).expect("decode converted values") + else { + panic!("expected schema-decoded JSON"); + }; + assert_eq!( + decoded["single"].as_f64().expect("decode float").to_bits(), + f64::from(single).to_bits() + ); + assert_eq!( + decoded["double"].as_f64().expect("decode double").to_bits(), + double.to_bits() + ); + } +} + +#[derive(Clone, PartialEq, Message)] +struct IntegerRecord { + #[prost(int32, tag = "1")] + int32_value: i32, + #[prost(int64, tag = "2")] + int64_value: i64, + #[prost(uint32, tag = "3")] + uint32_value: u32, + #[prost(uint64, tag = "4")] + uint64_value: u64, + #[prost(sint32, tag = "5")] + sint32_value: i32, + #[prost(sint64, tag = "6")] + sint64_value: i64, + #[prost(fixed32, tag = "7")] + fixed32_value: u32, + #[prost(fixed64, tag = "8")] + fixed64_value: u64, + #[prost(sfixed32, tag = "9")] + sfixed32_value: i32, + #[prost(sfixed64, tag = "10")] + sfixed64_value: i64, +} + +#[test] +fn given_nested_message_when_encoding_should_resolve_qualified_name() { + let encoder = ProtoStreamEncoder::new_with_config(ProtoEncoderConfig { + descriptor_set: Some(nested_descriptor_set()), + message_type: Some("example.Outer.Middle.Inner".to_string()), + ..ProtoEncoderConfig::default() + }); + let encoded = encoder + .encode(Payload::Json(simd_json::json!({"label": "nested"}))) + .expect("encode nested message type"); + let record = + FixedFieldRecord::decode(encoded.as_slice()).expect("decode nested record with prost"); + assert_eq!(record.label, "nested"); +} + +#[test] +fn given_nested_message_when_decoding_should_resolve_qualified_name() { + let decoder = ProtoStreamDecoder::new(ProtoConfig { + descriptor_set: Some(nested_descriptor_set()), + message_type: Some("example.Outer.Middle.Inner".to_string()), + ..ProtoConfig::default() + }); + let record = FixedFieldRecord { + label: "nested".to_string(), + ..Default::default() + }; + let Payload::Json(decoded) = decoder + .decode(record.encode_to_vec()) + .expect("decode nested message type") + else { + panic!("expected nested message fields"); + }; + assert_eq!(decoded, simd_json::json!({"label": "nested"})); +} + +#[test] +fn given_nested_message_when_converting_should_resolve_qualified_name() { + let converter = ProtoConvert::new(ProtoConvertConfig { + source_format: Schema::Json, + target_format: Schema::Proto, + descriptor_set: Some(nested_descriptor_set()), + message_type: Some("example.Outer.Middle.Inner".to_string()), + ..ProtoConvertConfig::default() + }); + let converted = converter + .transform( + &iggy_connector_sdk::TopicMetadata { + stream: "nested".to_string(), + topic: "nested".to_string(), + }, + iggy_connector_sdk::DecodedMessage { + id: None, + offset: None, + checksum: None, + timestamp: None, + origin_timestamp: None, + headers: None, + payload: Payload::Json(simd_json::json!({"label": "nested"})), + }, + ) + .expect("convert nested message type") + .expect("preserve message"); + let Payload::Raw(encoded) = converted.payload else { + panic!("expected schema-encoded nested message"); + }; + let record = + FixedFieldRecord::decode(encoded.as_slice()).expect("decode nested record with prost"); + assert_eq!(record.label, "nested"); +} + +#[test] +fn given_loaded_encoder_when_config_changes_should_match_reload_setting() { + for reload_schema in [false, true] { + for schema_path in [ + None, + Some(std::env::temp_dir().join(format!("{}.proto", uuid::Uuid::new_v4()))), + ] { + let missing_schema = schema_path.is_some(); + let mut encoder = ProtoStreamEncoder::new_with_config(ProtoEncoderConfig { + descriptor_set: Some(integer_descriptor_set()), + message_type: Some("IntegerRecord".to_string()), + ..ProtoEncoderConfig::default() + }); + let result = encoder.update_config( + ProtoEncoderConfig { + schema_path, + ..ProtoEncoderConfig::default() + }, + reload_schema, + ); + if reload_schema && missing_schema { + assert!(matches!(result, Err(Error::InitError(_))), "{result:?}"); + encoder + .load_schema() + .expect("reload restored configuration"); + } else { + result.expect("update encoder configuration"); + } + let (record, json) = integer_cases()[2].clone(); + let encoded = encoder + .encode(Payload::Json(json.clone())) + .expect("encode record"); + if reload_schema && !missing_schema { + let wrapped = Any::decode(encoded.as_slice()).expect("decode fallback Any"); + let text = wrapped + .to_msg::() + .expect("unpack fallback StringValue"); + let decoded: simd_json::OwnedValue = + simd_json::from_slice(&mut text.into_bytes()).expect("parse fallback JSON"); + assert_eq!(decoded, json); + } else { + assert_eq!(IntegerRecord::decode(encoded.as_slice()).unwrap(), record); + } + } + } +} + +#[test] +fn given_loaded_decoder_when_config_changes_should_match_reload_setting() { + for reload_schema in [false, true] { + for schema_path in [ + None, + Some(std::env::temp_dir().join(format!("{}.proto", uuid::Uuid::new_v4()))), + ] { + let missing_schema = schema_path.is_some(); + let mut decoder = ProtoStreamDecoder::new(ProtoConfig { + descriptor_set: Some(integer_descriptor_set()), + message_type: Some("IntegerRecord".to_string()), + ..ProtoConfig::default() + }); + let result = decoder.update_config( + ProtoConfig { + schema_path, + use_any_wrapper: false, + ..ProtoConfig::default() + }, + reload_schema, + ); + if reload_schema && missing_schema { + assert!(matches!(result, Err(Error::InitError(_))), "{result:?}"); + decoder + .load_schema() + .expect("reload restored configuration"); + } else { + result.expect("update decoder configuration"); + } + let (record, json) = integer_cases()[2].clone(); + let encoded = record.encode_to_vec(); + let decoded = decoder.decode(encoded.clone()).expect("decode record"); + if reload_schema && !missing_schema { + let Payload::Raw(bytes) = decoded else { + panic!("expected configured raw fallback"); + }; + assert_eq!(bytes, encoded); + } else { + let Payload::Json(fields) = decoded else { + panic!("expected retained schema"); + }; + assert_eq!(fields, json); + } + } + } +} + +#[test] +fn given_loaded_encoder_when_update_fails_should_preserve_configuration() { + let mut encoder = ProtoStreamEncoder::new_with_config(ProtoEncoderConfig { + descriptor_set: Some(integer_descriptor_set()), + message_type: Some("IntegerRecord".to_string()), + ..ProtoEncoderConfig::default() + }); + assert!( + encoder + .update_config( + ProtoEncoderConfig { + descriptor_set: Some(vec![0xff]), + field_mappings: Some(HashMap::from([( + "int32_value".to_string(), + "renamed".to_string() + )])), + ..ProtoEncoderConfig::default() + }, + true + ) + .is_err() + ); + let (record, json) = integer_cases()[2].clone(); + let encoded = encoder + .encode(Payload::Json(json)) + .expect("encode after failed update"); + assert_eq!(IntegerRecord::decode(encoded.as_slice()).unwrap(), record); + encoder + .load_schema() + .expect("reload restored configuration"); +} + +#[test] +fn given_loaded_decoder_when_update_fails_should_preserve_configuration() { + let mut decoder = ProtoStreamDecoder::new(ProtoConfig { + descriptor_set: Some(integer_descriptor_set()), + message_type: Some("IntegerRecord".to_string()), + ..ProtoConfig::default() + }); + assert!( + decoder + .update_config( + ProtoConfig { + descriptor_set: Some(vec![0xff]), + field_mappings: Some(HashMap::from([( + "int32_value".to_string(), + "renamed".to_string() + )])), + ..ProtoConfig::default() + }, + true + ) + .is_err() + ); + let (record, json) = integer_cases()[2].clone(); + let Payload::Json(decoded) = decoder + .decode(record.encode_to_vec()) + .expect("decode after failed update") + else { + panic!("expected retained schema"); + }; + assert_eq!(decoded, json); + decoder + .load_schema() + .expect("reload restored configuration"); +} + +#[test] +fn given_loaded_schemas_when_reload_fails_should_preserve_last_good_schema() { + let directory = std::env::temp_dir(); + let path = directory.join(format!("{}.proto", uuid::Uuid::new_v4())); + std::fs::write(&path, INTEGER_SCHEMA).expect("write owned schema fixture"); + let mut encoder = ProtoStreamEncoder::new_with_config(ProtoEncoderConfig { + schema_path: Some(path.clone()), + message_type: Some("IntegerRecord".to_string()), + include_paths: vec![directory.clone()], + ..ProtoEncoderConfig::default() + }); + let mut decoder = ProtoStreamDecoder::new(ProtoConfig { + schema_path: Some(path.clone()), + message_type: Some("IntegerRecord".to_string()), + include_paths: vec![directory.clone()], + use_any_wrapper: false, + ..ProtoConfig::default() + }); + let mut converter = ProtoConvert::new(ProtoConvertConfig { + source_format: Schema::Json, + target_format: Schema::Proto, + schema_path: Some(path.clone()), + message_type: Some("IntegerRecord".to_string()), + include_paths: vec![directory], + ..ProtoConvertConfig::default() + }); + let (record, json) = integer_cases()[2].clone(); + let convert = |converter: &ProtoConvert| { + converter + .transform( + &iggy_connector_sdk::TopicMetadata { + stream: "integers".to_string(), + topic: "integers".to_string(), + }, + iggy_connector_sdk::DecodedMessage { + id: None, + offset: None, + checksum: None, + timestamp: None, + origin_timestamp: None, + headers: None, + payload: Payload::Json(json.clone()), + }, + ) + .expect("convert record") + .expect("preserve record") + .payload + }; + for schema_content in [ + Some("syntax = broken"), + Some(r#"syntax = "proto3"; message IntegerRecord { Missing value = 1; }"#), + None, + ] { + if let Some(content) = schema_content { + std::fs::write(&path, content).expect("replace owned schema fixture"); + } + let encoder_error = encoder.load_schema(); + let decoder_error = decoder.load_schema(); + let converter_error = converter.load_schema(); + if schema_content.is_some() { + std::fs::remove_file(&path).expect("remove owned schema fixture"); + } + assert!( + matches!(encoder_error, Err(Error::InitError(_))), + "encoder reload for {schema_content:?}: {encoder_error:?}" + ); + assert!( + matches!(decoder_error, Err(Error::InitError(_))), + "decoder reload for {schema_content:?}: {decoder_error:?}" + ); + assert!( + matches!(converter_error, Err(Error::InitError(_))), + "converter reload for {schema_content:?}: {converter_error:?}" + ); + + let encoded = encoder + .encode(Payload::Json(json.clone())) + .expect("encode retained schema"); + assert_eq!(IntegerRecord::decode(encoded.as_slice()).unwrap(), record); + let Payload::Json(decoded) = decoder + .decode(record.encode_to_vec()) + .expect("decode retained schema") + else { + panic!("expected retained decoder schema"); + }; + assert_eq!(decoded, json); + let Payload::Raw(converted) = convert(&converter) else { + panic!("expected retained converter schema"); + }; + assert_eq!(IntegerRecord::decode(converted.as_slice()).unwrap(), record); + } +} + +#[test] +fn given_integer_schema_when_encoding_should_match_prost_values() { + let encoder = ProtoStreamEncoder::new_with_config(ProtoEncoderConfig { + descriptor_set: Some(integer_descriptor_set()), + message_type: Some("IntegerRecord".to_string()), + ..ProtoEncoderConfig::default() + }); + for (expected, json) in integer_cases() { + let encoded = encoder + .encode(Payload::Json(json)) + .expect("encode integer record"); + let decoded = + IntegerRecord::decode(encoded.as_slice()).expect("decode integer record with prost"); + assert_eq!(decoded, expected); + } +} + +#[test] +fn given_integer_schema_when_converting_should_match_prost_values() { + let converter = ProtoConvert::new(ProtoConvertConfig { + source_format: Schema::Json, + target_format: Schema::Proto, + descriptor_set: Some(integer_descriptor_set()), + message_type: Some("IntegerRecord".to_string()), + ..ProtoConvertConfig::default() + }); + let metadata = iggy_connector_sdk::TopicMetadata { + stream: "integers".to_string(), + topic: "integers".to_string(), + }; + for (expected, json) in integer_cases() { + let message = iggy_connector_sdk::DecodedMessage { + id: None, + offset: None, + checksum: None, + timestamp: None, + origin_timestamp: None, + headers: None, + payload: Payload::Json(json), + }; + let converted = converter + .transform(&metadata, message) + .expect("convert integer record") + .expect("preserve message"); + let Payload::Raw(encoded) = converted.payload else { + panic!("expected schema-encoded protobuf bytes"); + }; + let decoded = IntegerRecord::decode(encoded.as_slice()) + .expect("decode converted integer record with prost"); + assert_eq!(decoded, expected); + } +} + +#[test] +fn given_prost_integers_when_decoding_should_preserve_values() { + let decoder = ProtoStreamDecoder::new(ProtoConfig { + descriptor_set: Some(integer_descriptor_set()), + message_type: Some("IntegerRecord".to_string()), + ..ProtoConfig::default() + }); + for (record, expected) in integer_cases() { + let encoded = record.encode_to_vec(); + if encoded.is_empty() { + assert!(matches!( + decoder.decode(encoded), + Err(iggy_connector_sdk::Error::InvalidPayloadType) + )); + continue; + } + let Payload::Json(decoded) = decoder + .decode(encoded) + .expect("decode prost integer record") + else { + panic!("expected JSON integer fields"); + }; + assert_eq!(decoded, expected); + } +} + +#[derive(Clone, PartialEq, Message)] +struct FixedFieldRecord { + #[prost(fixed32, tag = "1")] + narrow: u32, + #[prost(fixed64, tag = "2")] + wide: u64, + #[prost(string, tag = "3")] + label: String, +} + +#[test] +fn given_unknown_fixed_fields_when_decoding_should_preserve_following_fields() { + let record = FixedFieldRecord { + narrow: 1, + wide: u64::MAX, + label: "after fixed fields".to_string(), + }; + for preserve_unknown_fields in [false, true] { + let decoder = fixed_field_decoder(preserve_unknown_fields); + let Payload::Json(simd_json::OwnedValue::Object(decoded)) = decoder + .decode(record.encode_to_vec()) + .expect("decode after unknown fixed fields") + else { + panic!("expected JSON object"); + }; + assert_eq!(decoded["label"], record.label); + assert_eq!(decoded.len(), if preserve_unknown_fields { 3 } else { 1 }); + assert_eq!( + decoded.contains_key("unknown_field_1"), + preserve_unknown_fields + ); + assert_eq!( + decoded.contains_key("unknown_field_2"), + preserve_unknown_fields + ); + } +} + +#[test] +fn given_truncated_fixed_fields_when_decoding_should_reject_payload() { + for preserve_unknown_fields in [false, true] { + let decoder = fixed_field_decoder(preserve_unknown_fields); + for (tag, width) in [ + (1 << 3 | 5, size_of::()), + (2 << 3 | 1, size_of::()), + ] { + for length in 0..width { + let mut truncated = vec![tag]; + truncated.resize(1 + length, 0); + assert!( + decoder.decode(truncated).is_err(), + "tag={tag}, length={length}, preserve={preserve_unknown_fields}" + ); + } + } + } +} + #[tokio::test] async fn should_transform_with_real_schema_and_field_mapping() { let mut field_mappings = HashMap::new(); @@ -425,3 +1029,80 @@ async fn should_encode_complex_nested_data_with_any_wrapper() { encoded_bytes.len() ); } + +fn integer_descriptor_set() -> Vec { + let schema = + protox_parse::parse("integers.proto", INTEGER_SCHEMA).expect("parse integer schema"); + prost_types::FileDescriptorSet { file: vec![schema] }.encode_to_vec() +} + +fn integer_cases() -> [(IntegerRecord, simd_json::OwnedValue); 5] { + [ + (0, 0, 0, 0), + (-1, -1, 1, 1), + (150, 9_000_000_000, 150, 9_000_000_000), + (i32::MIN, i64::MIN, u32::MAX, u64::MAX), + (i32::MAX, i64::MAX, u32::MAX, u64::MAX), + ] + .map(|(signed32, signed64, unsigned32, unsigned64)| { + let expected = IntegerRecord { + int32_value: signed32, + int64_value: signed64, + uint32_value: unsigned32, + uint64_value: unsigned64, + sint32_value: signed32, + sint64_value: signed64, + fixed32_value: unsigned32, + fixed64_value: unsigned64, + sfixed32_value: signed32, + sfixed64_value: signed64, + }; + let json = simd_json::json!({ + "int32_value": signed32, + "int64_value": signed64, + "uint32_value": unsigned32, + "uint64_value": unsigned64, + "sint32_value": signed32, + "sint64_value": signed64, + "fixed32_value": unsigned32, + "fixed64_value": unsigned64, + "sfixed32_value": signed32, + "sfixed64_value": signed64, + }); + (expected, json) + }) +} + +fn fixed_field_decoder(preserve_unknown_fields: bool) -> ProtoStreamDecoder { + let schema = protox_parse::parse( + "fixed.proto", + r#" + syntax = "proto3"; + message FixedFieldRecord { string label = 3; } + "#, + ) + .expect("parse schema without fixed fields"); + ProtoStreamDecoder::new(ProtoConfig { + descriptor_set: Some(prost_types::FileDescriptorSet { file: vec![schema] }.encode_to_vec()), + message_type: Some("FixedFieldRecord".to_string()), + preserve_unknown_fields, + ..ProtoConfig::default() + }) +} + +fn nested_descriptor_set() -> Vec { + let schema = protox_parse::parse( + "nested.proto", + r#" + syntax = "proto3"; + package example; + message Outer { + message Middle { + message Inner { string label = 3; } + } + } + "#, + ) + .expect("parse nested message schema"); + prost_types::FileDescriptorSet { file: vec![schema] }.encode_to_vec() +} diff --git a/core/connectors/sinks/http_sink/README.md b/core/connectors/sinks/http_sink/README.md index cccd5329cc..f8d9dd8b8d 100644 --- a/core/connectors/sinks/http_sink/README.md +++ b/core/connectors/sinks/http_sink/README.md @@ -68,13 +68,15 @@ IGGY_CONNECTORS_CONFIG_PATH=/tmp/http-sink-test/config.toml ./target/debug/iggy- ./target/debug/iggy -u iggy -p iggy message send demo_stream demo_topic '{"hello":"http"}' ``` -Expected output on the Python receiver: +Example receiver output (message IDs and timestamps vary): ```json { "metadata": { "iggy_id": "00000000000000000000000000000001", "iggy_offset": 0, + "iggy_timestamp": 1710064800000000, + "iggy_partition_id": 0, "iggy_stream": "demo_stream", "iggy_topic": "demo_topic" }, @@ -88,7 +90,16 @@ Cleanup: `rm -rf /tmp/http-sink-test` ## Quick Start +Use the runtime setup from [Try It](#try-it), replace the URL with your receiver, and save this connector entry in its connector directory: + ```toml +type = "sink" +key = "http" +enabled = true +version = 0 +name = "HTTP sink" +path = "target/debug/libiggy_connector_http_sink" + [[streams]] stream = "events" topics = ["notifications"] @@ -132,7 +143,8 @@ batch_mode = "ndjson" One HTTP request per message. Best for webhooks and endpoints that accept single events. -> With `batch_length = 50`, this produces 50 sequential HTTP round trips per poll cycle. +> A full poll with `batch_length = 50` can produce 50 sequential requests before retries. +> Shorter polls, serialization or size failures, and the consecutive-failure abort can produce fewer. > For production throughput, use `ndjson` or `json_array`. ```text @@ -170,7 +182,7 @@ POST /ingest Content-Type: application/octet-stream ## Message Flow: What Goes In vs. What Comes Out -The connector does **not** require or expect any particular message structure. It receives raw bytes from the Iggy runtime — whatever you published to the topic is what arrives in `consume()`. The `{metadata: {}, payload: {}}` envelope is something the **sink adds on the way out**, not something it expects on the way in. +The connector does **not** expect a metadata envelope from producers. The runtime decodes the configured stream schema and applies transforms before passing messages to the plugin. The `{metadata: {}, payload: {}}` envelope is something the **sink adds on the way out**. Producers must supply payloads accepted by the configured decoder and destination. ```text Your app publishes: {"order_id": 123, "amount": 9.99} @@ -179,7 +191,7 @@ Your app publishes: {"order_id": 123, "amount": 9.99} Iggy stores: raw bytes of that JSON | v -Runtime delivers: those same raw bytes to consume() +Runtime delivers: decoded/transformed payload to consume() | v HTTP sink wraps: {"metadata": {"iggy_offset": 0, ...}, @@ -189,21 +201,21 @@ HTTP sink wraps: {"metadata": {"iggy_offset": 0, ...}, HTTP endpoint gets: the wrapped envelope ``` -With `include_metadata = false`, the sink skips wrapping — your original message goes through as-is: +With `include_metadata = false`, the sink skips wrapping. JSON values are still reserialized, so original bytes and whitespace are not preserved: ```text HTTP endpoint gets: {"order_id": 123, "amount": 9.99} ``` -The `schema` field in `[[streams]]` controls how the sink **interprets** the incoming bytes for output formatting: +The runtime first applies `schema` decoding and any configured transforms. The sink formats the resulting payload variants in JSON batch modes as follows: -| Schema | Interpretation | Payload in envelope | +| Payload variant | Interpretation | Payload in envelope | | ------ | -------------- | ------------------- | -| `json` | Parses bytes as JSON | Embedded as JSON value | -| `text` | Treats bytes as UTF-8 string | Embedded as string | -| `raw` / `flatbuffer` / `proto` | Opaque binary | Base64-encoded with `"iggy_payload_encoding": "base64"` | +| `json` | JSON value | Embedded as JSON value | +| `text` | UTF-8 string | Embedded as string | +| `raw` / `flatbuffer` / `proto` / `avro` | Payload bytes | Base64-encoded with `"iggy_payload_encoding": "base64"` | -You can publish any struct serialized in any format (JSON, protobuf, raw bytes). Set the matching `schema` in `[[streams]]`, and choose whether you want the metadata envelope (`include_metadata`) or not. +For opaque byte forwarding, use `schema = "raw"`, `batch_mode = "raw"`, and no transforms. Other schemas have decoder-specific requirements; see the connector SDK format documentation. ## Metadata Envelope @@ -219,15 +231,17 @@ When `include_metadata = true` (default), payloads are wrapped: "iggy_topic": "my_topic", "iggy_partition_id": 0 }, - "payload": { ... } + "payload": {"key": "value"} } ``` - **`iggy_id`**: Message ID formatted as 32-character lowercase hex string (no dashes) -- **Non-JSON payloads** (Raw, FlatBuffer, Proto): base64-encoded with `"iggy_payload_encoding": "base64"` in payload -- **JSON/Text payloads**: Embedded as-is +- **Non-JSON payload variants** (Raw, FlatBuffer, Proto, Avro): base64-encoded with `"iggy_payload_encoding": "base64"` in payload +- **JSON/Text payloads**: JSON values are reserialized; text becomes a JSON string. +- **Timestamps**: `iggy_timestamp` and optional `iggy_origin_timestamp` use Unix epoch microseconds. +- **Optional metadata**: `include_checksum` and `include_origin_timestamp` default to false. Nonempty message headers are included as `iggy_headers`. -Set `include_metadata = false` to send the raw payload without wrapping. +Set `include_metadata = false` to omit the envelope; payload formatting still follows the selected batch mode. ## Retry Strategy @@ -235,18 +249,20 @@ Uses `reqwest-middleware` with `RetryTransientMiddleware` for automatic exponent ```text Initial request: no delay -Retry 1: retry_delay = 1s -Retry 2: retry_delay * backoff = 2s -Retry 3: retry_delay * backoff^2 = min(4s, 30s) = 4s +Retry 1: full jitter from 0 to retry_delay = 1s +Retry 2: full jitter from 0 to retry_delay * backoff = 2s +Retry 3: full jitter from 0 to min(retry_delay * backoff^2, 30s) = 4s ``` +`max_retries` counts additional attempts after the initial request, so the default allows at most four attempts. + A custom `HttpSinkRetryStrategy` respects user-configured `success_status_codes` — codes in the success set are never retried, even if normally transient (e.g., 429 configured as "queued"). **Transient errors** (retry): Network errors, HTTP 429, 500, 502, 503, 504. **Non-transient errors** (fail immediately): HTTP 400, 401, 403, 404, 405, etc. -**HTTP 429 `Retry-After`**: The middleware does not natively support `Retry-After` headers. When a response carries `Retry-After`, a warning is logged with the header value. The middleware uses computed exponential backoff instead. +**HTTP 429 `Retry-After`**: The middleware does not natively support `Retry-After` headers. When an unsuccessful response carries `Retry-After`, a warning is logged with the header value. The middleware uses computed exponential backoff instead. **Partial delivery** (`individual`/`raw` modes): If a message fails after exhausting retries, subsequent messages continue processing. After 3 consecutive HTTP failures, the remaining batch is aborted to avoid hammering a dead endpoint. @@ -342,7 +358,7 @@ Authorization = "Bearer observability-token" ## Authentication -The HTTP sink supports authentication via custom headers in `[plugin_config.headers]`. All headers are sent with every request, including health checks. +The HTTP sink supports authentication via custom headers in `[plugin_config.headers]`. Custom headers are sent with data requests and health checks, except `Content-Type`: configured values are ignored, and the data request content type comes from the batch mode. ### Bearer Token @@ -388,18 +404,18 @@ X-Client-Version = "iggy-http-sink/0.1" ### Connector Runtime Model -A **connector instance** is a single OS process — the `iggy-connectors` binary loading one shared library (`libiggy_connector_http_sink.so`/`.dylib`) with one config file. Each process reads exactly one `config.toml` (set via `IGGY_CONNECTORS_CONFIG_PATH`), which defines one `[plugin_config]` block — including the target `url`, authentication headers, batch mode, and retry settings. +A **connector instance** is one configured plugin loaded by an `iggy-connectors` process. The runtime reads its main configuration from `IGGY_CONNECTORS_CONFIG_PATH`; the local provider loads connector TOML files from `connectors.config_dir`. One process can host multiple connector instances, each with its own key, `[plugin_config]`, destination URL and HTTP client. -Within that single process, the runtime spawns one async task per topic listed in `[[streams]]`. All tasks share the same plugin instance (and therefore the same HTTP client and `[plugin_config]`). There is no built-in orchestrator, no multi-connector-in-one-process mode, and no routing table that maps different topics to different URLs. +For each instance, the runtime spawns one async consumer task per configured stream/topic. Those tasks share that plugin instance and can consume concurrently. One HTTP sink instance has one destination URL; use multiple entries with distinct keys and URLs for multiple destinations. How this works in the runtime source code: -- **One consumer per topic**: `setup_sink_consumers()` in [`runtime/src/sink.rs`](../../../runtime/src/sink.rs) iterates `for topic in stream.topics.iter()` and creates a separate `IggyConsumer` for each topic. -- **One async task per consumer**: `spawn_consume_tasks()` in [`runtime/src/sink.rs`](../../../runtime/src/sink.rs) wraps each consumer in `tokio::spawn`, so topics are consumed concurrently within the same process. +- **One consumer per topic**: `setup_sink_consumers()` in [`runtime/src/sink.rs`](../../runtime/src/sink.rs) iterates `for topic in stream.topics.iter()` and creates a separate `IggyConsumer` for each topic. +- **One async task per consumer**: `spawn_consume_tasks()` in [`runtime/src/sink.rs`](../../runtime/src/sink.rs) wraps each consumer in `tokio::spawn`, so topics are consumed concurrently within the same process. - **One plugin instance per ID**: The `sink_connector!` macro in [`sdk/src/sink.rs`](../../sdk/src/sink.rs) creates a `static INSTANCES: DashMap` — each `plugin_id` passed to `iggy_sink_open` gets its own entry, and all topic tasks call `consume()` on the same instance. -- **Sequential consume within each topic**: `consume_messages()` in [`runtime/src/sink.rs`](../../../runtime/src/sink.rs) awaits `consume()` before polling the next batch — there is no pipelining within a single topic task. +- **Sequential consume within each topic**: `consume_messages()` in [`runtime/src/sink.rs`](../../runtime/src/sink.rs) awaits `consume()` before polling the next batch — there is no pipelining within a single topic task. -**"Deploying multiple instances"** means running N separate `iggy-connectors` processes — each with its own config directory, its own `[plugin_config]` (and therefore its own destination URL, headers, batch mode, etc.). In Docker or Kubernetes, this means N containers from the same image with different config mounts or environment variables. In systemd, N service units. In ECS, N task definitions. +**"Deploying multiple instances"** can mean multiple connector entries in one runtime, or separate processes with their own configuration directories. Separate containers, service units or task definitions are optional when process isolation is needed. ### What's Achievable Today vs. Not @@ -407,12 +423,12 @@ How this works in the runtime source code: | ------- | :-: | --- | | Single destination, single topic | Yes | One connector instance, one `[[streams]]` entry | | Single destination, multiple topics | Yes | One connector instance, multiple topics in `[[streams]]` | -| Multiple destinations (topic-per-destination) | Yes | N connector instances, one per destination, each a separate OS process | +| Multiple destinations (topic-per-destination) | Yes | N connector entries with distinct keys and URLs, in one or more processes | | Fan-out (same topic to multiple destinations) | Yes | N connector instances consuming same topic with different `consumer_group` names | | Per-topic URL routing within one instance | **No** | Not supported — each instance has exactly one `url`. Requires N instances. See [Known Limitations](#known-limitations) item 6 | | OAuth2 / OIDC token refresh | **No** | Static headers only. Use an auth proxy | | mTLS client certificates | **No** | Use a sidecar proxy for mTLS termination | -| Environment variable expansion in config values | **No** | Use env var overrides at the process level (see [Environment Variable Overrides](#environment-variable-overrides)) | +| Environment variable expansion in config values | **No** | Flat fields support env overrides; header tables require a configuration file (see [Environment Variable Overrides](#environment-variable-overrides)) | ### Single Destination, Multiple Topics @@ -469,9 +485,9 @@ Authorization = "Bearer shared-token" ### Multiple Destinations (One Connector Per Destination) -*Achievable today — requires N separate OS processes.* +*Achievable with multiple connector entries, in one or more processes.* -When different topics need to go to different services, deploy separate connector instances. Each gets its own config directory and runs as a **separate `iggy-connectors` process** (not a config option within one process — see [Connector Runtime Model](#connector-runtime-model)). +When different topics need to go to different services, configure separate connector instances. They can share a runtime using distinct keys. The example below chooses separate configuration directories and processes for isolation. ```text ┌───────────────────┐ @@ -593,9 +609,9 @@ IGGY_CONNECTORS_CONFIG_PATH=/opt/connectors/slack/config.toml iggy-connectors ### Fan-Out: One Topic to Multiple Destinations -*Achievable today — requires N separate OS processes with different consumer groups.* +*Achievable with separate connector entries and different consumer groups.* -When a single topic needs to be delivered to multiple HTTP endpoints (e.g., send order events to both the billing service AND an analytics pipeline), deploy multiple connector instances that consume from the **same topic with different consumer groups**. Each instance is a separate `iggy-connectors` process (see [Connector Runtime Model](#connector-runtime-model)). +When a single topic needs to be delivered to multiple HTTP endpoints (e.g., send order events to both the billing service AND an analytics pipeline), deploy multiple connector instances that consume from the **same topic with different consumer groups**. The instances can share a runtime using distinct keys, or run in separate processes (see [Connector Runtime Model](#connector-runtime-model)). ```text connector-billing ──▶ billing-api.example.com @@ -607,7 +623,7 @@ When a single topic needs to be delivered to multiple HTTP endpoints (e.g., send (consumer_group: analytics_sink) ``` -Each consumer group maintains its own offset, so both connectors independently receive every message. This is the standard Iggy fan-out pattern — not an antipattern. +Each consumer group maintains its own offset, so both connectors can independently consume the retained messages. The delivery limits below still apply. **Key requirement**: Each connector instance MUST use a **different `consumer_group`**. If they share a consumer group, messages are load-balanced (split) across instances rather than duplicated. @@ -643,13 +659,13 @@ batch_mode = "ndjson" *Achievable today.* -Each connector instance maps naturally to one container (one process = one container). Share the compiled `.so`/`.dylib` via a volume mount or bake it into the image: +A container can run one runtime process with one or more connector entries. Share the compiled `.so`/`.dylib` via a volume mount or bake it into the image: ```dockerfile FROM rust:latest AS builder WORKDIR /app COPY . . -RUN cargo build -p iggy_connector_http_sink --release +RUN cargo build -p iggy_connector_http_sink -p iggy-connectors --release FROM debian:bookworm-slim RUN apt-get update && apt-get install -y ca-certificates && rm -rf /var/lib/apt/lists/* @@ -682,12 +698,11 @@ services: ### Environment Variable Overrides -The connector runtime supports overriding any config field via environment variables using the convention `IGGY_CONNECTORS_SINK_{KEY}_
_`. This is useful for keeping secrets out of config files: +The local provider accepts flat plugin overrides with `IGGY_CONNECTORS_SINK_{KEY}_PLUGIN_CONFIG_`, such as the destination URL. It does not support nested headers or JSON objects: `HEADERS_AUTHORIZATION` creates an unused field, and a JSON object passed through `HEADERS` becomes a string and fails plugin initialization. Supply header secrets through a protected connector TOML file. ```bash -# Override the URL and auth token at runtime +# Override a flat plugin field at runtime export IGGY_CONNECTORS_SINK_HTTP_PLUGIN_CONFIG_URL="https://prod-api.example.com/ingest" -export IGGY_CONNECTORS_SINK_HTTP_PLUGIN_CONFIG_HEADERS_AUTHORIZATION="Bearer prod-token" iggy-connectors ``` @@ -695,7 +710,7 @@ iggy-connectors ### Batch Mode Selection -The connector runtime calls `consume()` **sequentially** — the next poll cycle does not start until the current batch completes. Batch mode choice directly impacts throughput: +The connector runtime calls `consume()` **sequentially within each topic task**. The task does not process its next batch until the current call completes. Request counts below assume a nonempty batch with no retries or skipped messages. Batch mode choice directly impacts throughput: | Mode | HTTP Requests per Poll | Latency per Poll | Best For | | ---- | ---------------------- | ----------------- | -------- | @@ -704,7 +719,7 @@ The connector runtime calls `consume()` **sequentially** — the next poll cycle | `json_array` | 1 | 1 × round-trip | APIs expecting array payloads | | `raw` | N (one per message) | N × round-trip | Binary payloads (protobuf, avro) | -With `batch_length=50` in `individual` mode, each poll cycle performs 50 sequential HTTP round trips. If each takes 100ms, the poll cycle takes 5 seconds — during which no new messages are consumed from that topic. Use `ndjson` or `json_array` to collapse this to a single round trip. +For a full 50-message batch in `individual` mode, without retries or skipped messages, 50 sequential 100ms HTTP round trips take 5 seconds before other processing costs. The topic task does not process the next batch during those requests. Use `ndjson` or `json_array` to collapse this to a single round trip. ### Memory @@ -720,19 +735,19 @@ reqwest uses HTTP/1.1 persistent connections (keep-alive) by default. The connec - **TCP keep-alive** (30s) — Sends TCP keep-alive probes on idle connections to detect silent drops by cloud load balancers. Without this, a connection silently closed by an intermediate LB (AWS ALB drops idle connections after ~60s, GCP after ~600s) would only be discovered on the next HTTP request, causing a failed attempt and retry delay. - **Pool idle timeout** (90s) — Closes connections unused for 90 seconds to prevent stale connection accumulation in the pool. -Because `reqwest::Client` clones are cheap (they share the same connection pool via `Arc`), all topic tasks within a single connector process share one pool. This means multi-topic connectors benefit from connection reuse when all topics target the same host — a connection returned to the pool by topic A's task can be reused by topic B's task. +Because `reqwest::Client` clones are cheap (they share the same connection pool via `Arc`), all topic tasks within a single connector instance share one pool. This means multi-topic connectors benefit from connection reuse when all topics target the same host — a connection returned to the pool by topic A's task can be reused by topic B's task. -For multiple connector instances (separate processes), each process has its own independent `reqwest::Client` and its own connection pool. There is no cross-process connection sharing. +Each connector instance has its own `reqwest::Client` and connection pool, even when several instances share a process. Separate processes do not share pools. ### Retry Impact on Throughput -Each failed message in `individual`/`raw` mode burns through the retry budget (default: 3 retries with exponential backoff up to 30s) before moving to the next message. The backoff delays are 1s + 2s + 4s = 7 seconds per message, but each attempt also incurs the request timeout (default 30s) for a dead endpoint. Worst case per message: 4 attempts × 30s timeout + 7s backoff = 127 seconds. +Each transiently failing message in `individual`/`raw` mode can use the retry budget (default: 3 retries with exponential backoff up to 30s) before moving to the next message. The jittered backoff delays total at most 1s + 2s + 4s = 7 seconds per message, but each attempt also incurs the request timeout (default 30s) for a dead endpoint. Worst case per message: 4 attempts × 30s timeout + 7s backoff = 127 seconds. -The consecutive failure abort (`MAX_CONSECUTIVE_FAILURES = 3`) mitigates this: after 3 consecutive HTTP failures, remaining messages in the batch are skipped. This limits worst-case blocking to: 3 × (4 × 30s + 1s + 2s + 4s) = 381 seconds with default timeout, or 3 × 7s = 21 seconds of backoff delay alone. +The consecutive failure abort (`MAX_CONSECUTIVE_FAILURES = 3`) mitigates this: after 3 consecutive HTTP failures, remaining messages in the batch are skipped. This limits worst-case blocking to: 3 × (4 × 30s + 1s + 2s + 4s) = 381 seconds with default timeout, or at most 3 × 7s = 21 seconds of backoff delay alone. ### Multiple Instances vs. Single Instance -Multiple connector instances (one per destination) provide: +Separate runtime processes (one per destination) can provide: - **Performance isolation**: A slow destination doesn't block other topics - **Failure isolation**: One dead endpoint doesn't affect unrelated connectors @@ -781,21 +796,21 @@ cargo test -p iggy_connector_http_sink Integration tests (requires Docker for WireMock container): ```bash -cargo test -p integration --test connectors -- http_sink +cargo test -p integration -- connectors::http::http_sink:: ``` ## Delivery Semantics -All retry logic lives inside `consume()`. The connector runtime invokes `consume()` via an FFI callback that returns an `i32` status code. The runtime does not inspect this return value (see `process_messages()` in `runtime/src/sink.rs`), so errors logged by the sink are not propagated to the runtime's retry or alerting mechanisms. Additionally, consumer group offsets are committed before processing ([runtime issue #1](#known-limitations)). This means: +All retry logic lives inside `consume()`. The connector runtime invokes `consume()` via an FFI callback that returns an `i32` status code. A nonzero result is logged and increments the runtime error counter; that batch contributes no processed messages. The runtime continues polling without retrying the failed batch. Consumer group offsets are committed before processing ([runtime issue #2](#known-limitations)). This means: - Failed messages are **not retried by the runtime** — only by the sink's internal retry loop - Messages are committed **before delivery** — a crash after commit but before delivery loses messages -The effective delivery guarantee is **at-most-once** at the runtime level. The sink's internal retries provide best-effort delivery within each `consume()` call. +A failed batch may already be partially delivered, and a retry after an ambiguous response can duplicate delivery. There is no end-to-end at-least-once or exactly-once guarantee. A successful HTTP status is accepted without inspecting the response body for per-item failures. ## Known Limitations -1. **Runtime ignores `consume()` status**: The connector runtime invokes `consume()` via an FFI callback returning `i32`. The `process_messages()` function in `runtime/src/sink.rs` does not inspect the return value. Errors are logged internally by the sink but do not trigger runtime-level retry or alerting. ([#2927](https://github.com/apache/iggy/issues/2927)) +1. **No runtime batch retry**: Nonzero `consume()` results are logged and counted as errors, but the runtime does not retry the batch. A failed batch may have been partially delivered; the FFI result does not report a per-message outcome. 2. **Offsets committed before processing**: The `PollingMessages` auto-commit strategy commits consumer group offsets before `consume()` is called. Combined with limitation 1, at-least-once delivery is not achievable. ([#2928](https://github.com/apache/iggy/issues/2928)) @@ -809,4 +824,4 @@ The effective delivery guarantee is **at-most-once** at the runtime level. The s 7. **No OAuth2 token refresh**: Bearer tokens are static. Use an auth proxy for services requiring automatic token rotation. -8. **No environment variable expansion in config values**: Secrets in `[plugin_config.headers]` are stored as plaintext. Use environment variable overrides (see [Environment Variable Overrides](#environment-variable-overrides)) or mount secrets from a secrets manager. +8. **No environment variable expansion in config values**: Secrets in `[plugin_config.headers]` are stored as plaintext. Mount or generate a protected connector TOML file containing the header values; nested header environment overrides are unsupported. diff --git a/core/connectors/sinks/http_sink/config.toml b/core/connectors/sinks/http_sink/config.toml index 98b3360e74..83af481414 100644 --- a/core/connectors/sinks/http_sink/config.toml +++ b/core/connectors/sinks/http_sink/config.toml @@ -84,7 +84,7 @@ max_connections = 10 verbose_logging = false # Custom HTTP headers. Replace placeholder values with real credentials. -# Do not commit actual secrets — use environment variable overrides for production. +# Do not commit actual secrets. Supply headers through a protected config file. [plugin_config.headers] Authorization = "Bearer " X-Custom-Header = "custom-value" diff --git a/core/connectors/sinks/http_sink/src/lib.rs b/core/connectors/sinks/http_sink/src/lib.rs index ea97cedf1e..1b4f8d306e 100644 --- a/core/connectors/sinks/http_sink/src/lib.rs +++ b/core/connectors/sinks/http_sink/src/lib.rs @@ -1194,8 +1194,8 @@ impl Sink for HttpSink { /// /// **Runtime note**: The FFI boundary in `sdk/src/sink.rs` maps `consume()`'s `Result` to /// `i32` (0=ok, 1=err), but the runtime's `process_messages()` in `runtime/src/sink.rs` - /// discards that return code. All retry logic lives inside this method — returning `Err` - /// does not trigger a runtime-level retry. + /// logs and counts a nonzero code, records no processed messages for that batch, + /// and continues polling. Returning `Err` does not trigger a runtime retry. async fn consume( &self, topic_metadata: &TopicMetadata, @@ -1235,10 +1235,7 @@ impl Sink for HttpSink { }; if let Err(ref e) = result { - error!( - "HTTP sink ID: {} — consume() returning error (runtime ignores FFI status code): {}", - self.id, e - ); + error!("HTTP sink ID: {} - consume() failed: {}", self.id, e); } result diff --git a/core/connectors/sources/random_source/README.md b/core/connectors/sources/random_source/README.md index 730304e341..e4154741ba 100644 --- a/core/connectors/sources/random_source/README.md +++ b/core/connectors/sources/random_source/README.md @@ -1,12 +1,12 @@ # Random Source -The Random Source connector generates random data and sends it to the specified stream(s). +The Random Source connector generates random data and sends it to the configured stream and topic. ## Configuration -- `interval`: A string representing the interval at which the connector generates random data. Defaults to `"1s"`. -- `max_count`: An integer representing the maximum number of messages to generate. Defaults to none. -- `messages_range`: An array of two integers representing the range of amount of messages to generate. Defaults to `[10, 50]`. +- `interval`: A string representing the interval at which the connector generates random data. Defaults to `"1s"`; invalid duration strings also fall back to `"1s"`. +- `max_count`: An integer representing the maximum number of messages to generate. Omit for no limit; `0` produces no messages. The last batch is capped by the remaining count. +- `messages_range`: An array of two integers specifying the half-open message-count range (lower included, upper excluded). Defaults to `[10, 50]`. Non-increasing ranges return a configuration error from polling. - `payload_size`: An integer representing the size in bytes of the `text` field payload to generate. Defaults to `100`. ```toml @@ -16,3 +16,5 @@ max_count = 1000 messages_range = [10, 50] payload_size = 200 ``` + +The count is staged while a batch is in flight and committed on acknowledgement, after delivery and checkpoint storage. A restored checkpoint preserves the count across restarts; missing or undecodable state starts from zero. After reaching the limit, polling continues with empty batches and no new checkpoint. diff --git a/core/connectors/sources/random_source/src/lib.rs b/core/connectors/sources/random_source/src/lib.rs index 908640d3f7..ca5de470db 100644 --- a/core/connectors/sources/random_source/src/lib.rs +++ b/core/connectors/sources/random_source/src/lib.rs @@ -91,12 +91,12 @@ impl RandomSource { ConnectorState::serialize(state, CONNECTOR_NAME, self.id) } - fn generate_messages(&self, remaining: Option) -> Vec { + fn generate_messages(&self, remaining: Option) -> Result, Error> { let mut messages = Vec::new(); let mut rng = rand::rng(); - let messages_count = rng - .sample(Uniform::new(self.messages_range.0, self.messages_range.1).unwrap()) - as usize; + let distribution = Uniform::new(self.messages_range.0, self.messages_range.1) + .map_err(|error| Error::InvalidConfigValue(format!("messages_range: {error}")))?; + let messages_count = rng.sample(distribution) as usize; let messages_count = remaining.map_or(messages_count, |remaining| messages_count.min(remaining)); for _ in 0..messages_count { @@ -124,7 +124,7 @@ impl RandomSource { }; messages.push(message); } - messages + Ok(messages) } fn generate_random_text(&self) -> String { @@ -171,7 +171,7 @@ impl Source for RandomSource { let remaining = self .max_count .map(|max_count| max_count.saturating_sub(messages_produced)); - let messages = self.generate_messages(remaining); + let messages = self.generate_messages(remaining)?; let candidate_state = State { messages_produced: messages_produced + messages.len(), }; @@ -234,6 +234,25 @@ mod tests { } } + #[tokio::test] + async fn given_non_increasing_message_ranges_when_polling_should_return_config_error() { + for messages_range in [(0, 0), (5, 5), (10, 5)] { + let mut config = test_config(); + config.messages_range = Some(messages_range); + let source = RandomSource::new(1, config, None); + let error = source + .poll() + .await + .expect_err("an invalid range must return an error instead of panicking"); + assert!( + matches!(error, Error::InvalidConfigValue(ref reason) if reason.contains("messages_range")), + "range {messages_range:?} returned the wrong error: {error:?}" + ); + assert_eq!(source.state.lock().await.messages_produced, 0); + assert!(source.pending_state.lock().await.is_none()); + } + } + #[test] fn given_persisted_state_should_restore_messages_produced() { let state = State { diff --git a/core/integration/tests/connectors/fixtures/http/container.rs b/core/integration/tests/connectors/fixtures/http/container.rs index c31d88fd8f..81dcca2c98 100644 --- a/core/integration/tests/connectors/fixtures/http/container.rs +++ b/core/integration/tests/connectors/fixtures/http/container.rs @@ -118,6 +118,29 @@ impl HttpSinkWireMockContainer { }) } + pub async fn set_ingest_status( + &self, + status: reqwest::StatusCode, + ) -> Result<(), TestBinaryError> { + let url = format!("{}/__admin/mappings", self.base_url); + let client = reqwest::Client::new(); + let mapping = serde_json::json!({ + "request": { "method": "POST", "urlPath": "/ingest" }, + "response": { "status": status.as_u16() } + }); + // Deleting mappings also removes the bind-mounted fixture files. + client + .post(&url) + .json(&mapping) + .send() + .await + .and_then(reqwest::Response::error_for_status) + .map_err(|error| TestBinaryError::InvalidState { + message: format!("Failed to set WireMock ingest response: {error}"), + })?; + Ok(()) + } + /// Query WireMock's admin API and return all received requests. pub async fn get_received_requests(&self) -> Result, TestBinaryError> { let url = format!("{}/__admin/requests", self.base_url); diff --git a/core/integration/tests/connectors/http/http_sink.rs b/core/integration/tests/connectors/http/http_sink.rs index 0466e71f46..a44a09d59b 100644 --- a/core/integration/tests/connectors/http/http_sink.rs +++ b/core/integration/tests/connectors/http/http_sink.rs @@ -47,10 +47,9 @@ //! 4. **WireMock**: Docker container accepting all POSTs to `/ingest`, recording //! requests for later verification via `/__admin/requests` //! -//! **Runtime model**: 1 process = 1 config = 1 plugin. The runtime reads `config.toml`, -//! loads the plugin binary, iterates `for topic in stream.topics`, and spawns one -//! `tokio::spawn` task per topic. Each task creates an `IggyConsumer` and polls -//! sequentially — `consume()` is awaited before the next poll. +//! **Runtime model**: One process can load multiple connector configurations. +//! Each plugin instance has a consumer task per configured stream/topic. Topic tasks +//! can run concurrently; each awaits `consume()` before processing its next batch. //! //! See `setup_sink_consumers()` and `spawn_consume_tasks()` in `runtime/src/sink.rs`. //! @@ -127,13 +126,13 @@ //! cargo build -p iggy_connector_http_sink //! //! # Run all HTTP sink integration tests -//! cargo test -p integration --test connectors -- http_sink --nocapture +//! cargo test -p integration -- http_sink --nocapture //! //! # Run a specific test -//! cargo test -p integration --test connectors -- individual_json_messages --nocapture +//! cargo test -p integration -- individual_json_messages --nocapture //! //! # Run with test isolation (sequential) -//! cargo test -p integration --test connectors -- http_sink --test-threads=1 --nocapture +//! cargo test -p integration -- http_sink --test-threads=1 --nocapture //! ``` //! //! ## Success Criteria @@ -159,11 +158,10 @@ //! //! ## Known Limitations //! -//! 1. **FFI return value ignored**: The runtime's `process_messages()` discards `consume()`'s -//! `i32` return code. Errors are logged by the sink but invisible to the runtime. -//! See [#2927](https://github.com/apache/iggy/issues/2927). +//! 1. **No runtime batch retry**: Nonzero FFI results are logged and counted as errors, +//! but the runtime continues polling without retrying the failed batch. //! 2. **Offsets committed before processing**: `PollingMessages` auto-commit strategy commits -//! offsets before `consume()`. Combined with (1), effective guarantee is at-most-once. +//! offsets before `consume()`. Failed delivery can lose messages; sink retries can duplicate them. //! See [#2928](https://github.com/apache/iggy/issues/2928). //! //! ## Test History @@ -187,8 +185,112 @@ use crate::connectors::fixtures::{ use bytes::Bytes; use iggy::prelude::IggyClient; use iggy_common::{Identifier, IggyMessage, MessageClient, Partitioning}; +use iggy_connector_sdk::api::{ConnectorRuntimeStats, ConnectorStatus}; use integration::harness::seeds; use integration::iggy_harness; +use reqwest::StatusCode; +use std::time::Duration; +use tokio::time::{sleep, timeout}; + +const HTTP_SINK_KEY: &str = "http"; +const SINK_STATS_TIMEOUT: Duration = Duration::from_secs(10); +const SINK_STATS_POLL_INTERVAL: Duration = Duration::from_millis(100); + +#[iggy_harness( + server(connectors_runtime(config_path = "tests/connectors/http/sink.toml")), + seed = seeds::connector_stream +)] +async fn given_sink_rejection_when_delivery_recovers_should_report_batch_outcomes( + harness: &TestHarness, + fixture: HttpSinkJsonArrayFixture, +) { + let client = harness + .root_client() + .await + .expect("root client should connect"); + let stream_id: Identifier = seeds::names::STREAM.try_into().expect("valid stream name"); + let topic_id: Identifier = seeds::names::TOPIC.try_into().expect("valid topic name"); + let stats_url = format!( + "{}/stats", + harness + .connectors_runtime() + .expect("connector runtime should be available") + .http_url() + ); + let http_client = reqwest::Client::new(); + + for (status, expected_processed) in [(StatusCode::BAD_REQUEST, 0), (StatusCode::OK, 1)] { + fixture + .container() + .set_ingest_status(status) + .await + .expect("ingest response should be configured"); + let mut messages = vec![ + IggyMessage::builder() + .payload(Bytes::from_static(br#"{"message":"sink status"}"#)) + .build() + .expect("valid JSON message"), + ]; + client + .send_messages( + &stream_id, + &topic_id, + &Partitioning::partition_id(0), + &mut messages, + ) + .await + .expect("message should reach Iggy"); + + let sink = timeout(SINK_STATS_TIMEOUT, async { + loop { + let snapshot: ConnectorRuntimeStats = http_client + .get(&stats_url) + .send() + .await + .expect("stats request should complete") + .error_for_status() + .expect("stats endpoint should succeed") + .json() + .await + .expect("stats response should decode"); + let sink = snapshot + .connectors + .into_iter() + .find(|connector| connector.key == HTTP_SINK_KEY) + .expect("HTTP sink should be reported"); + assert!( + sink.messages_processed.unwrap_or_default() <= expected_processed, + "rejected messages must not count as processed: {sink:?}" + ); + if sink.errors > 0 && sink.messages_processed == Some(expected_processed) { + break sink; + } + sleep(SINK_STATS_POLL_INTERVAL).await; + } + }) + .await + .expect("sink should report the completed batch outcome"); + + assert_eq!(sink.messages_consumed, Some(expected_processed + 1)); + assert_eq!( + sink.errors, 1, + "one rejected batch should count as one error" + ); + assert_eq!(sink.messages_filtered, Some(0)); + assert_eq!(sink.status, ConnectorStatus::Running); + } + + let requests = fixture + .container() + .get_received_requests() + .await + .expect("received requests should be available"); + assert_eq!( + requests.len(), + 2, + "both batches must reach the HTTP endpoint" + ); +} // ============================================================================ // Test 1: Individual Batch Mode diff --git a/core/integration/tests/connectors/runtime/error_isolation.rs b/core/integration/tests/connectors/runtime/error_isolation.rs index d23429609d..a3a899b00f 100644 --- a/core/integration/tests/connectors/runtime/error_isolation.rs +++ b/core/integration/tests/connectors/runtime/error_isolation.rs @@ -30,14 +30,15 @@ //! * source state-load failure (unreachable state file), //! * post-container setup failure (invalid duration in stream config). +use iggy_common::{Identifier, IggyMessage, MessageClient, Partitioning}; use iggy_connector_sdk::api::{ - ConnectorStatus, HealthResponse, SinkInfoResponse, SourceInfoResponse, + ConnectorRuntimeStats, ConnectorStatus, HealthResponse, SinkInfoResponse, SourceInfoResponse, }; use integration::harness::seeds; use integration::iggy_harness; use reqwest::Client; use std::time::Duration; -use tokio::time::sleep; +use tokio::time::{sleep, timeout}; async fn assert_runtime_healthy(http_client: &Client, api_address: &str) { let response = http_client @@ -320,3 +321,80 @@ async fn source_with_invalid_config_does_not_abort_runtime(harness: &TestHarness "Healthy sibling source should have no last_error" ); } + +#[iggy_harness( + server(connectors_runtime( + config_path = "tests/connectors/runtime/sink_transform_error.toml" + )), + seed = seeds::connector_stream +)] +async fn given_sink_transform_error_when_batch_processed_should_not_count_filter( + harness: &TestHarness, +) { + const SINK_KEY: &str = "stdout_transform_error"; + const STATS_TIMEOUT: Duration = Duration::from_secs(10); + const STATS_POLL_INTERVAL: Duration = Duration::from_millis(100); + + let client = harness + .root_client() + .await + .expect("root client should connect"); + let stream_id: Identifier = seeds::names::STREAM.try_into().expect("valid stream name"); + let topic_id: Identifier = seeds::names::TOPIC.try_into().expect("valid topic name"); + let mut messages = vec![ + IggyMessage::from("not JSON"), + IggyMessage::from(r#"{"message":"valid"}"#), + ]; + client + .send_messages( + &stream_id, + &topic_id, + &Partitioning::partition_id(0), + &mut messages, + ) + .await + .expect("messages should reach Iggy"); + + let stats_url = format!( + "{}/stats", + harness + .connectors_runtime() + .expect("connector runtime should be available") + .http_url() + ); + let http_client = Client::new(); + let sink = timeout(STATS_TIMEOUT, async { + loop { + let snapshot: ConnectorRuntimeStats = http_client + .get(&stats_url) + .send() + .await + .expect("stats request should complete") + .error_for_status() + .expect("stats endpoint should succeed") + .json() + .await + .expect("stats response should decode"); + let sink = snapshot + .connectors + .into_iter() + .find(|connector| connector.key == SINK_KEY) + .expect("sink should be reported"); + if sink.messages_processed == Some(1) && sink.errors > 0 { + break sink; + } + sleep(STATS_POLL_INTERVAL).await; + } + }) + .await + .expect("the valid message should survive the other message's transform failure"); + + assert_eq!(sink.messages_consumed, Some(2)); + assert_eq!(sink.errors, 1); + assert_eq!( + sink.messages_filtered, + Some(0), + "transform errors are not intentional filters" + ); + assert_eq!(sink.status, ConnectorStatus::Running); +} diff --git a/core/integration/tests/connectors/runtime/http_state.rs b/core/integration/tests/connectors/runtime/http_state.rs index d36bac5fde..ed038362f0 100644 --- a/core/integration/tests/connectors/runtime/http_state.rs +++ b/core/integration/tests/connectors/runtime/http_state.rs @@ -34,7 +34,7 @@ use assert_cmd::prelude::CommandCargoExt; use async_trait::async_trait; -use iggy_connector_sdk::api::{ConnectorStatus, SourceInfoResponse}; +use iggy_connector_sdk::api::{ConnectorRuntimeStats, ConnectorStatus, SourceInfoResponse}; use integration::harness::config::TestServerConfig; use integration::harness::{TestBinaryError, TestFixture, TestHarness, seeds}; use integration::iggy_harness; @@ -56,6 +56,7 @@ const RUNTIME_CONFIG_PATH: &str = "tests/connectors/runtime/http_state.toml"; const WAIT_DEADLINE: Duration = Duration::from_secs(15); const POLL_INTERVAL: Duration = Duration::from_millis(50); const IDEMPOTENCY_KEY_HEADER: HeaderName = HeaderName::from_static("idempotency-key"); +const STATE_URL_QUERY_SECRET: &str = "state-query-secret-not-for-logs"; /// In-memory state server backing the wiremock responder. Enforces the /// conditional-write contract and exposes counters plus injectable failure @@ -255,7 +256,10 @@ async fn given_unavailable_state_store_when_booting_should_fail_startup() { // Bound but never accepted, so every state request times out instead of // racing other tests for a recycled port. let listener = std::net::TcpListener::bind("127.0.0.1:0").expect("reserve a port"); - let state_url = format!("http://127.0.0.1:{}", listener.local_addr().unwrap().port()); + let state_url = format!( + "http://127.0.0.1:{}?token={STATE_URL_QUERY_SECRET}", + listener.local_addr().unwrap().port() + ); let mut command = Command::cargo_bin("iggy-connectors").expect("iggy-connectors binary"); command @@ -288,6 +292,10 @@ async fn given_unavailable_state_store_when_booting_should_fail_startup() { logs.contains("failed to load state") || logs.contains("StateLoadFailed"), "startup failure should point at the state load, got:\n{logs}" ); + assert!( + !logs.contains(STATE_URL_QUERY_SECRET), + "runtime startup and state-load failures must not log URL query secrets:\n{logs}" + ); } #[iggy_harness( @@ -326,6 +334,26 @@ async fn given_conflict_mid_stream_should_nack_and_latch( fixture.store.conflict_mode.store(true, Ordering::SeqCst); wait_for_status(harness, ConnectorStatus::Error).await; + let api_url = harness + .connectors_runtime() + .expect("connector runtime should be available") + .http_url(); + let stats: ConnectorRuntimeStats = Client::new() + .get(format!("{api_url}/stats")) + .send() + .await + .expect("stats request should complete") + .error_for_status() + .expect("stats endpoint should succeed") + .json() + .await + .expect("stats response should decode"); + assert_eq!(stats.sources_total, 1); + assert_eq!( + stats.sources_running, 0, + "a latched source has Error status and must not count as running" + ); + let version_after_conflict = fixture.store.version.load(Ordering::SeqCst); let puts_after_conflict = fixture.store.put_count.load(Ordering::SeqCst); sleep(Duration::from_millis(500)).await; diff --git a/core/integration/tests/connectors/runtime/sink_transform_error.toml b/core/integration/tests/connectors/runtime/sink_transform_error.toml new file mode 100644 index 0000000000..648cbf96b0 --- /dev/null +++ b/core/integration/tests/connectors/runtime/sink_transform_error.toml @@ -0,0 +1,20 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +[connectors] +config_type = "local" +config_dir = "tests/connectors/runtime/sink_transform_error_config" diff --git a/core/integration/tests/connectors/runtime/sink_transform_error_config/stdout.toml b/core/integration/tests/connectors/runtime/sink_transform_error_config/stdout.toml new file mode 100644 index 0000000000..b6b1bcef6e --- /dev/null +++ b/core/integration/tests/connectors/runtime/sink_transform_error_config/stdout.toml @@ -0,0 +1,47 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +type = "sink" +key = "stdout_transform_error" +enabled = true +version = 0 +name = "Stdout transform error sink" +path = "../../target/debug/libiggy_connector_stdout_sink" + +[[streams]] +stream = "test_stream" +topics = ["test_topic"] +schema = "raw" +batch_length = 100 +poll_interval = "5ms" + +[plugin_config] +print_payload = true + +[transforms.proto_convert] +enabled = true +source_format = "raw" +target_format = "json" +include_paths = [] +preserve_unknown_fields = false + +[transforms.proto_convert.conversion_options] +validate_messages = true +pretty_json = false +include_metadata = false +type_url_prefix = "type.googleapis.com" +strict_mode = false From af237a124c5f63881ebb63d804e2e492b7d8a6aa Mon Sep 17 00:00:00 2001 From: diegomrsantos Date: Sun, 13 Sep 2026 13:19:42 +0200 Subject: [PATCH 124/182] fix(partitions): complete polls on the owning shard (#4119) A disk poll can remain pending while purge replaces a partition's message history and resets its offsets to zero. Previously, the detached read worker retained shared handles that could advance consumer offsets and group progress after purge. If fresh messages reused those numeric offsets, the stale completion could make a later `Next` poll skip messages from the new history. Poll workers now return owned read results to the partition owner. The owner validates the captured history before accepting progress or authorizing a successful reply. Fixes #4117. --- .../common/src/http/messages/poll_messages.rs | 4 + core/common/src/traits/message_client.rs | 30 +- .../src/types/message/polling_strategy.rs | 7 +- core/configs/src/server_config/sharding.rs | 81 +- core/consensus/src/impls.rs | 100 + .../src/consumer_offset_capacity.rs | 141 +- core/partitions/src/iggy_partition.rs | 1866 ++++++++++++++--- core/partitions/src/iggy_partitions.rs | 152 +- core/partitions/src/lib.rs | 3 +- core/partitions/src/poll_plan.rs | 696 +----- core/partitions/src/state_transfer.rs | 6 +- core/sdk/src/clients/consumer.rs | 6 + core/server/Cargo.toml | 1 + core/server/config.toml | 7 + core/server/src/boot/mod.rs | 2 - core/server/src/boot/recovery.rs | 3 +- core/server/src/boot/threads.rs | 39 + core/server/src/consumer_group.rs | 281 ++- core/server/src/dispatch/failure.rs | 2 +- core/server/src/dispatch/mod.rs | 5 +- core/server/src/dispatch/partition.rs | 677 +++--- core/server/src/http/error.rs | 52 +- core/server/src/server_error.rs | 2 + core/server/src/shell.rs | 4 +- core/server_common/src/lib.rs | 1 + core/server_common/src/poll.rs | 279 +++ core/shard/Cargo.toml | 1 + core/shard/src/builder.rs | 15 +- core/shard/src/lib.rs | 158 +- core/shard/src/metrics.rs | 97 +- core/shard/src/poll.rs | 257 +++ core/shard/src/poll/completion.rs | 507 +++++ core/shard/src/poll/completion_tests.rs | 506 +++++ core/shard/src/poll/test_support.rs | 110 + core/shard/src/poll/timeout_tests.rs | 532 +++++ core/shard/src/router.rs | 93 +- core/simulator/src/bin/simulator-ui.rs | 21 +- core/simulator/src/bus.rs | 22 + core/simulator/src/lib.rs | 216 +- core/simulator/src/replica.rs | 3 +- 40 files changed, 5366 insertions(+), 1619 deletions(-) create mode 100644 core/server_common/src/poll.rs create mode 100644 core/shard/src/poll.rs create mode 100644 core/shard/src/poll/completion.rs create mode 100644 core/shard/src/poll/completion_tests.rs create mode 100644 core/shard/src/poll/test_support.rs create mode 100644 core/shard/src/poll/timeout_tests.rs diff --git a/core/common/src/http/messages/poll_messages.rs b/core/common/src/http/messages/poll_messages.rs index 5f3e48668b..98995c8c39 100644 --- a/core/common/src/http/messages/poll_messages.rs +++ b/core/common/src/http/messages/poll_messages.rs @@ -53,6 +53,10 @@ pub struct PollMessages { #[serde(default = "PollMessages::default_number_of_messages_to_poll")] pub count: u32, /// Whether to commit offset on the server automatically after polling the messages. + /// The cursor can advance before the response arrives, and a successful + /// response does not acknowledge a durable offset commit. See the + /// [poll recovery contract](crate::MessageClient::poll_messages) for retrying + /// a missing response with explicit offsets for each partition. #[serde(default)] pub auto_commit: bool, } diff --git a/core/common/src/traits/message_client.rs b/core/common/src/traits/message_client.rs index bf850aeef6..1bd3ab13e8 100644 --- a/core/common/src/traits/message_client.rs +++ b/core/common/src/traits/message_client.rs @@ -31,15 +31,39 @@ pub trait MessageClient { /// Polling a consumer group the client is not (or no longer) a member of fails with `ConsumerGroupMemberNotFound` rather than returning an empty batch, so the caller can rejoin. /// A member that holds no partitions gets an empty batch whose `partition_id` is [`NO_ASSIGNED_PARTITION`](crate::NO_ASSIGNED_PARTITION). /// - /// With server-side auto-commit enabled, a new consumer offset key can be + /// With automatic commits enabled, a new consumer offset key can be /// rejected with `TooManyConsumerOffsets` at the partition's configured /// limit. That poll returns no messages. Existing keys remain writable, - /// and polling without auto-commit does not allocate a stored offset. - /// A refused auto-commit submission returns `TransientNotAccepted` with no + /// and polling without automatic commits does not allocate a stored offset. + /// A refused automatic commit submission returns `TransientNotAccepted` with no /// messages and may be retried. A capacity error requires capacity to be freed. /// Local cursors without a committed offset can be evicted at the limit. /// Their next `Next` poll resumes from the earliest retained messages, /// which can redeliver messages from earlier polls. + /// + /// # Recovering from a missing response + /// + /// A timeout or communication failure does not establish that the server + /// rejected the poll. With `auto_commit` enabled, the server can advance its + /// consumer cursor before the caller receives the messages. Retrying with + /// [`PollingStrategy::next()`] can then skip messages from the missing response. + /// Even a successful poll response does not acknowledge a durable offset commit. + /// + /// Keep a checkpoint for each partition: one past the last message successfully + /// processed in order, or the intended starting offset if none was processed. + /// Recover with [`PollingStrategy::offset(checkpoint)`](PollingStrategy::offset) + /// and advance the checkpoint only after processing the returned messages. + /// Use each message's header offset, not [`PolledMessages::current_offset`], + /// which reports the partition's current offset rather than the last message + /// returned. Persist checkpoints if recovery must survive a client restart. + /// + /// Continue using explicit offsets until all messages through the server's + /// cursor have been processed: replay does not rewind an advanced cursor. + /// For a consumer group, recover each partition with its own checkpoint and + /// respect the current assignment. [`Self::poll_messages_with_strategy_for`] + /// selects the strategy after the partition is known. + /// Recovery assumes the same partition message history and requires the + /// messages to remain retained. It can repeat messages already processed. #[allow(clippy::too_many_arguments)] async fn poll_messages( &self, diff --git a/core/common/src/types/message/polling_strategy.rs b/core/common/src/types/message/polling_strategy.rs index d0bfde0926..2939dfeafb 100644 --- a/core/common/src/types/message/polling_strategy.rs +++ b/core/common/src/types/message/polling_strategy.rs @@ -92,7 +92,12 @@ impl PollingStrategy { } } - /// Poll messages from the next message after the last polled message based on the stored consumer offset. Should be used with `auto_commit` set to `true`. + /// Poll messages after the consumer offset stored on the server. + /// + /// Advance that offset with automatic commits or explicit offset stores. + /// A missing response to a poll with automatic commits can leave it ahead + /// of messages the caller received. See the + /// [poll recovery contract](crate::MessageClient::poll_messages) before retrying. pub fn next() -> Self { Self { kind: PollingKind::Next, diff --git a/core/configs/src/server_config/sharding.rs b/core/configs/src/server_config/sharding.rs index c7f6335781..67f163869f 100644 --- a/core/configs/src/server_config/sharding.rs +++ b/core/configs/src/server_config/sharding.rs @@ -33,8 +33,8 @@ use configs::ConfigEnv; // alongside the rest of the section. pub use cpu_allocation::{CpuAllocation, NumaConfig}; -/// Maximum permitted per-shard inbox depth. The channel is allocated -/// up-front per shard, so a runaway value here OOMs the process at boot. +/// Maximum permitted capacity of an inbox or completion lane on each shard. +/// Channels are allocated at boot, so a runaway value can exhaust memory. /// `1 << 20` (~1M frames) is several orders of magnitude above any /// realistic backpressure target and still fits comfortably in process /// address space. @@ -95,6 +95,12 @@ pub struct ShardingConfig { /// reply forwards and each lane is sized for its own worst case: this /// one against peak client-reply fan-out per shard. pub reply_inbox_capacity: usize, + /// Maximum number of running disk polls plus queued completions per shard. + /// A read reserves a slot before I/O and retains it until the owner dequeues + /// or discards its result, including after a requester timeout. Exhaustion + /// rejects new disk polls before I/O. Main and reply inbox traffic uses + /// separate capacity. This counts operations, not retained message bytes. + pub poll_completion_capacity: usize, /// Wall-clock budget for a single shard's bus drain on shutdown. /// Drives `IggyMessageBus::shutdown(..)` from the per-shard watchdog /// and the parallel-join survivor path. Sized larger than typical @@ -142,6 +148,7 @@ impl Default for ShardingConfig { pin_cores: SERVER_CONFIG.sharding.pin_cores, inbox_capacity: SERVER_CONFIG.sharding.inbox_capacity as usize, reply_inbox_capacity: SERVER_CONFIG.sharding.reply_inbox_capacity as usize, + poll_completion_capacity: SERVER_CONFIG.sharding.poll_completion_capacity as usize, shutdown_drain_timeout: SERVER_CONFIG .sharding .shutdown_drain_timeout @@ -200,6 +207,21 @@ impl Validatable for ShardingConfig { ); return Err(ConfigurationError::InvalidConfigurationValue); } + if self.poll_completion_capacity == 0 { + eprintln!( + "Invalid sharding configuration: poll_completion_capacity must be > 0 \ + (each disk poll must reserve a completion slot before I/O)" + ); + return Err(ConfigurationError::InvalidConfigurationValue); + } + if self.poll_completion_capacity > INBOX_CAPACITY_MAX { + eprintln!( + "Invalid sharding configuration: poll_completion_capacity {} exceeds the {} \ + cap (each shard preallocates a completion lane of this size)", + self.poll_completion_capacity, INBOX_CAPACITY_MAX + ); + return Err(ConfigurationError::InvalidConfigurationValue); + } let drain = self.shutdown_drain_timeout.get_duration(); if drain.is_zero() { @@ -298,6 +320,58 @@ mod tests { assert!(ShardingConfig::default().validate().is_ok()); } + #[test] + fn given_invalid_poll_completion_capacity_when_validated_should_reject() { + for capacity in [0, INBOX_CAPACITY_MAX + 1] { + let config = ShardingConfig { + poll_completion_capacity: capacity, + ..ShardingConfig::default() + }; + assert!(config.validate().is_err(), "accepted capacity {capacity}"); + } + } + + #[test] + fn given_poll_completion_capacity_at_boundaries_when_validated_should_accept() { + for capacity in [1, INBOX_CAPACITY_MAX] { + let config = ShardingConfig { + poll_completion_capacity: capacity, + ..ShardingConfig::default() + }; + assert!(config.validate().is_ok(), "rejected capacity {capacity}"); + } + } + + #[test] + fn given_legacy_inbox_settings_when_deserialized_should_default_poll_completion_capacity() { + let sharding: ShardingConfig = Figment::new() + .merge(Toml::string( + "inbox_capacity = 7\nreply_inbox_capacity = 11", + )) + .extract() + .expect("existing inbox settings remain valid without the new field"); + + assert_eq!(sharding.inbox_capacity, 7); + assert_eq!(sharding.reply_inbox_capacity, 11); + assert_eq!(sharding.poll_completion_capacity, 1024); + assert!(sharding.validate().is_ok()); + } + + #[test] + fn given_explicit_poll_completion_capacity_when_deserialized_should_use_independent_limit() { + let sharding: ShardingConfig = Figment::new() + .merge(Toml::string( + "inbox_capacity = 7\nreply_inbox_capacity = 11\npoll_completion_capacity = 17", + )) + .extract() + .expect("completion capacity can be configured independently"); + + assert_eq!(sharding.inbox_capacity, 7); + assert_eq!(sharding.reply_inbox_capacity, 11); + assert_eq!(sharding.poll_completion_capacity, 17); + assert!(sharding.validate().is_ok()); + } + #[test] fn zero_drain_is_rejected() { let cfg = ShardingConfig { @@ -395,6 +469,7 @@ mod tests { assert!(sharding.pin_cores); assert_eq!(sharding.inbox_capacity, 1024); assert_eq!(sharding.reply_inbox_capacity, 1024); + assert_eq!(sharding.poll_completion_capacity, 1024); assert_eq!(sharding.shutdown_drain_timeout, "10 s".parse().unwrap()); assert_eq!(sharding.shutdown_poll_interval, "50 ms".parse().unwrap()); assert_eq!(sharding.shutdown_join_timeout, "30 s".parse().unwrap()); @@ -414,6 +489,7 @@ mod tests { assert!(!sharding.pin_cores); assert_eq!(sharding.inbox_capacity, 1024); assert_eq!(sharding.reply_inbox_capacity, 1024); + assert_eq!(sharding.poll_completion_capacity, 1024); assert_eq!(sharding.shutdown_drain_timeout, "10 s".parse().unwrap()); assert_eq!(sharding.shutdown_poll_interval, "50 ms".parse().unwrap()); assert_eq!(sharding.shutdown_join_timeout, "30 s".parse().unwrap()); @@ -430,6 +506,7 @@ mod tests { assert!(sharding.pin_cores); assert_eq!(sharding.inbox_capacity, 1024); assert_eq!(sharding.reply_inbox_capacity, 1024); + assert_eq!(sharding.poll_completion_capacity, 1024); assert_eq!(sharding.shutdown_drain_timeout, "10 s".parse().unwrap()); assert_eq!(sharding.shutdown_poll_interval, "50 ms".parse().unwrap()); assert_eq!(sharding.shutdown_join_timeout, "30 s".parse().unwrap()); diff --git a/core/consensus/src/impls.rs b/core/consensus/src/impls.rs index f445d25dde..64e28825a7 100644 --- a/core/consensus/src/impls.rs +++ b/core/consensus/src/impls.rs @@ -36,6 +36,7 @@ use iggy_common::calculate_checksum; use message_bus::IggyMessageBus; use message_bus::MessageBus; use server_common::Message; +use server_common::poll::{AutoCommitReservation, PollHistoryId}; use server_common::sharding::{IggyNamespace, METADATA_GROUP}; use std::cell::{Cell, RefCell}; use std::collections::VecDeque; @@ -268,9 +269,20 @@ impl PipelineEntry { } } +/// Identity and capacity held by a pending automatic commit. +#[derive(Debug)] +pub struct AutoCommitRequestContext { + /// History accepted with the poll, which must still match at promotion. + pub history: PollHistoryId, + /// Keeps this request's consumer key occupied through promotion and staging. + pub reservation: AutoCommitReservation, +} + /// Accepted request waiting in `request_queue` for a prepare slot. #[derive(Debug)] pub struct RequestEntry { + /// Automatic commit context owned by this entry until promotion or removal. + auto_commit: Option, pub message: Message, /// When the request was parked, in microseconds from the consensus-injected /// clock ([`VsrConsensus::clock_realtime_micros`]). `0` until @@ -294,6 +306,30 @@ pub struct RequestEntry { } impl RequestEntry { + /// Build an automatic commit entry that owns its capacity guard. + /// The context must match the consumer key encoded in `message`. + #[must_use] + pub fn with_auto_commit( + message: Message, + context: AutoCommitRequestContext, + ) -> Self { + let mut entry = Self::new(message); + entry.auto_commit = Some(context); + entry + } + + /// Transfer the context without dropping its reservation. + /// Promotion must retain the returned context through replication staging. + pub const fn take_auto_commit(&mut self) -> Option { + self.auto_commit.take() + } + + /// Inspect the context without acquiring another reservation. + #[must_use] + pub const fn auto_commit(&self) -> Option<&AutoCommitRequestContext> { + self.auto_commit.as_ref() + } + /// Queued request on the network reply path: no in-process subscriber. #[must_use] pub const fn new(message: Message) -> Self { @@ -322,6 +358,7 @@ impl RequestEntry { reply_sender: Option>>, ) -> Self { Self { + auto_commit: None, message, received_at: 0, reply_sender, @@ -730,6 +767,13 @@ impl LocalPipeline { } } + /// Remove pending requests while preserving the order of those retained. + /// Removed entries drop their reply senders and automatic commit reservations. + /// Prepare entries are unaffected. + pub fn retain_requests(&mut self, mut keep: impl FnMut(&RequestEntry) -> bool) { + self.request_queue.retain(|request| keep(request)); + } + /// Drop `request_queue` only; preserve `prepare_queue`. View-change reset. /// /// # Safety @@ -4225,6 +4269,10 @@ mod fresh_group_start_tests { mod request_queue_tests { use super::*; use iggy_binary_protocol::{Command, Operation}; + use iggy_common::ConsumerKind; + use server_common::poll::AutoCommitReservationToken; + use std::cell::Cell; + use std::rc::Rc; fn make_request(client: u128, request_num: u64) -> Message { let header_size = std::mem::size_of::(); @@ -4244,6 +4292,58 @@ mod request_queue_tests { msg } + #[test] + fn queued_contexts_keep_each_reservation_until_their_own_removal() { + let consumer_id = 7; + let client_id = 1; + let reclaim_epoch = Rc::new(Cell::new(0)); + let active_keys = Rc::new(Cell::new(0)); + let token = Rc::new(AutoCommitReservationToken::new( + ConsumerKind::Consumer, + consumer_id, + reclaim_epoch, + Rc::clone(&active_keys), + )); + let history = PollHistoryId::default(); + let mut pipeline = LocalPipeline::new(); + + // Each queued request holds its own guard for the same consumer key. + // The queue stores these messages without interpreting their payloads. + for request_number in 1..=2 { + let context = AutoCommitRequestContext { + history, + reservation: token.acquire(), + }; + pipeline + .push_request(RequestEntry::with_auto_commit( + make_request(client_id, request_number), + context, + )) + .expect("both requests fit in the queue"); + } + + // Removing the first entry transfers its context to the caller, as + // promotion would. Releasing it must preserve the second reservation. + let mut first_request = pipeline.pop_request().expect("first queued request"); + let first_context = first_request + .take_auto_commit() + .expect("context follows first request"); + assert_eq!(first_request.message.header().request, 1); + assert_eq!(first_context.history, history); + drop(first_context); + assert_eq!(token.active_count(), 1); + assert_eq!( + active_keys.get(), + 1, + "the queued request still holds the key" + ); + + // Clearing the queue drops the remaining context and frees the key. + pipeline.clear_request_queue(); + assert_eq!(token.active_count(), 0); + assert_eq!(active_keys.get(), 0); + } + #[test] fn push_request_buffers_when_prepare_queue_full() { let mut pipeline = LocalPipeline::new(); diff --git a/core/partitions/src/consumer_offset_capacity.rs b/core/partitions/src/consumer_offset_capacity.rs index aaaa932087..b6618175c3 100644 --- a/core/partitions/src/consumer_offset_capacity.rs +++ b/core/partitions/src/consumer_offset_capacity.rs @@ -15,13 +15,14 @@ // specific language governing permissions and limitations // under the License. -use iggy_common::ConsumerKind; use std::cell::{Cell, RefCell}; use std::collections::hash_map::Entry; use std::collections::{HashMap, HashSet}; use std::rc::Rc; -use std::sync::Arc; -use std::sync::atomic::{AtomicU64, AtomicUsize, Ordering}; + +use iggy_common::ConsumerKind; +pub use server_common::poll::AutoCommitReservation; +use server_common::poll::AutoCommitReservationToken; #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub struct DurableOffsetState { @@ -172,13 +173,15 @@ pub struct ConsumerOffsetCapacity { kind: ConsumerKind, limit: Cell, pending: RefCell>, - provisional: RefCell>>, - active_provisional_keys: Arc, + /// Tokens cached by consumer key, which may outlive their last request guard. + provisional: RefCell>>, + /// Tokens that still have guards, excluding inactive entries in the cache. + active_provisional_keys: Rc>, stranded: RefCell>, uncertain: Cell, durable_warned: Cell, map_warned: Cell, - reclaim_epoch: Arc, + reclaim_epoch: Rc>, last_reclaim: Cell>, } @@ -189,12 +192,12 @@ impl ConsumerOffsetCapacity { limit: Cell::new(limit), pending: RefCell::new(HashMap::new()), provisional: RefCell::new(HashMap::new()), - active_provisional_keys: Arc::new(AtomicUsize::new(0)), + active_provisional_keys: Rc::new(Cell::new(0)), stranded: RefCell::new(HashSet::new()), uncertain: Cell::new(false), durable_warned: Cell::new(false), map_warned: Cell::new(false), - reclaim_epoch: Arc::new(AtomicU64::new(0)), + reclaim_epoch: Rc::new(Cell::new(0)), last_reclaim: Cell::new(None), } } @@ -230,7 +233,7 @@ impl ConsumerOffsetCapacity { let fixed = durable_count .saturating_add(self.pending.borrow().len()) .saturating_add(self.stranded.borrow().len()); - let provisional_len = self.active_provisional_keys.load(Ordering::Relaxed); + let provisional_len = self.active_provisional_keys.get(); let upper_bound = fixed.saturating_add(provisional_len); if !self.uncertain.get() && upper_bound < limit { self.durable_warned.set(false); @@ -255,40 +258,42 @@ impl ConsumerOffsetCapacity { Ok(()) } + /// Check capacity and hold a provisional claim for one automatic commit. + /// Requests sharing a key share occupancy but retain independent guards. + /// + /// Keep the capacity check and token acquisition in one owner operation. + /// Splitting them across owner turns could let two new keys claim the last + /// available slot, even with atomic counters. pub(crate) fn reserve_provisional( - self: &Rc, + &self, id: u32, - durable: &Rc, + durable: &DurableConsumerOffsets, ) -> Result { self.check(id, durable)?; let mut provisional = self.provisional.borrow_mut(); if provisional.len() >= self.limit.get() && !provisional.contains_key(&id) { - provisional.retain(|_, token| token.active.load(Ordering::Relaxed) > 0); + provisional.retain(|_, token| token.active_count() > 0); } - let token = Arc::clone(provisional.entry(id).or_insert_with(|| { - Arc::new(ProvisionalToken { - reclaim_epoch: Arc::clone(&self.reclaim_epoch), - active_keys: Arc::clone(&self.active_provisional_keys), - active: AtomicUsize::new(0), - }) - })); - if token.active.fetch_add(1, Ordering::Relaxed) == 0 { - token.active_keys.fetch_add(1, Ordering::Relaxed); - } - Ok(AutoCommitReservation { - token, - kind: self.kind, - consumer_id: id, - }) + let token = provisional.entry(id).or_insert_with(|| { + Rc::new(AutoCommitReservationToken::new( + self.kind, + id, + Rc::clone(&self.reclaim_epoch), + Rc::clone(&self.active_provisional_keys), + )) + }); + Ok(token.acquire()) } + /// Check that the guard belongs to this tracker's current token for its key. + /// Matching consumer identifiers cannot validate a guard from another tracker. pub(crate) fn owns(&self, reservation: &AutoCommitReservation) -> bool { - reservation.kind == self.kind + reservation.kind() == self.kind && self .provisional .borrow() - .get(&reservation.consumer_id) - .is_some_and(|token| Arc::ptr_eq(token, &reservation.token)) + .get(&reservation.consumer_id()) + .is_some_and(|token| token.owns(reservation)) } pub(crate) fn holds(&self, id: u32, durable: &DurableConsumerOffsets) -> bool { @@ -298,7 +303,7 @@ impl ConsumerOffsetCapacity { .provisional .borrow() .get(&id) - .is_some_and(|token| token.active.load(Ordering::Relaxed) > 0) + .is_some_and(|token| token.active_count() > 0) } /// Assigns the pending count outright while [`Self::release_reservation`] @@ -407,14 +412,15 @@ impl ConsumerOffsetCapacity { } pub(crate) fn note_local_key_change(&self) { - self.reclaim_epoch.fetch_add(1, Ordering::Relaxed); + self.reclaim_epoch + .set(self.reclaim_epoch.get().wrapping_add(1)); } pub(crate) fn forget_inactive_provisional(&self, id: u32) { let mut provisional = self.provisional.borrow_mut(); if provisional .get(&id) - .is_some_and(|token| token.active.load(Ordering::Relaxed) == 0) + .is_some_and(|token| token.active_count() == 0) { provisional.remove(&id); } @@ -424,10 +430,7 @@ impl ConsumerOffsetCapacity { if self.uncertain.get() { return false; } - let epoch = ( - self.reclaim_epoch.load(Ordering::Relaxed), - durable.membership_epoch.get(), - ); + let epoch = (self.reclaim_epoch.get(), durable.membership_epoch.get()); self.last_reclaim.replace(Some(epoch)) != Some(epoch) } @@ -441,7 +444,7 @@ impl ConsumerOffsetCapacity { local.extend( provisional .iter() - .filter(|(_, token)| token.active.load(Ordering::Relaxed) > 0) + .filter(|(_, token)| token.active_count() > 0) .map(|(id, _)| *id), ); local.extend(stranded.iter().copied()); @@ -453,52 +456,28 @@ impl ConsumerOffsetCapacity { } } -/// Keeps a provisional key occupied until the pump admits or drops its request. -#[derive(Debug)] -pub struct AutoCommitReservation { - token: Arc, - pub(crate) kind: ConsumerKind, - pub(crate) consumer_id: u32, -} - -#[derive(Debug)] -struct ProvisionalToken { - reclaim_epoch: Arc, - active_keys: Arc, - active: AtomicUsize, -} - -impl Drop for AutoCommitReservation { - fn drop(&mut self) { - if self.token.active.fetch_sub(1, Ordering::Relaxed) == 1 { - self.token.active_keys.fetch_sub(1, Ordering::Relaxed); - self.token.reclaim_epoch.fetch_add(1, Ordering::Relaxed); - } - } -} - #[cfg(test)] mod tests { use super::*; #[test] fn given_cached_inactive_tokens_when_admitting_should_count_only_active_keys() { - let durable = Rc::new(DurableConsumerOffsets::default()); - let capacity = Rc::new(ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 2)); + let durable = DurableConsumerOffsets::default(); + let capacity = ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 2); let first = capacity.reserve_provisional(1, &durable).unwrap(); let repeated = capacity.reserve_provisional(1, &durable).unwrap(); - assert_eq!(capacity.active_provisional_keys.load(Ordering::Relaxed), 1); + assert_eq!(capacity.active_provisional_keys.get(), 1); drop(first); - assert_eq!(capacity.active_provisional_keys.load(Ordering::Relaxed), 1); + assert_eq!(capacity.active_provisional_keys.get(), 1); drop(repeated); - assert_eq!(capacity.active_provisional_keys.load(Ordering::Relaxed), 0); + assert_eq!(capacity.active_provisional_keys.get(), 0); assert_eq!(capacity.provisional.borrow().len(), 1); capacity.check(2, &durable).unwrap(); let second = capacity.reserve_provisional(2, &durable).unwrap(); - assert_eq!(capacity.active_provisional_keys.load(Ordering::Relaxed), 1); + assert_eq!(capacity.active_provisional_keys.get(), 1); drop(second); capacity.check(3, &durable).unwrap(); - assert_eq!(capacity.active_provisional_keys.load(Ordering::Relaxed), 0); + assert_eq!(capacity.active_provisional_keys.get(), 0); } #[test] @@ -524,8 +503,8 @@ mod tests { #[test] fn given_unchanged_protection_when_reclaim_repeats_should_skip_until_guard_drops() { - let durable = Rc::new(DurableConsumerOffsets::default()); - let capacity = Rc::new(ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 2)); + let durable = DurableConsumerOffsets::default(); + let capacity = ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 2); let held = capacity.reserve_provisional(7, &durable).unwrap(); assert!(capacity.should_reclaim(&durable)); for _ in 0..100 { @@ -540,21 +519,21 @@ mod tests { #[test] fn given_repeated_reservation_for_same_key_should_reuse_token_allocation() { - let durable = Rc::new(DurableConsumerOffsets::default()); - let capacity = Rc::new(ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 2)); + let durable = DurableConsumerOffsets::default(); + let capacity = ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 2); let first = capacity.reserve_provisional(7, &durable).unwrap(); - let token = Arc::clone(&first.token); + let token = Rc::clone(capacity.provisional.borrow().get(&7).unwrap()); drop(first); drop(capacity.reserve_provisional(8, &durable).unwrap()); assert_eq!(capacity.provisional.borrow().len(), 2); let second = capacity.reserve_provisional(7, &durable).unwrap(); - assert!(Arc::ptr_eq(&token, &second.token)); + assert!(token.owns(&second)); } #[test] fn given_inactive_token_cache_at_limit_when_new_key_arrives_should_prune_it() { - let durable = Rc::new(DurableConsumerOffsets::default()); - let capacity = Rc::new(ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 2)); + let durable = DurableConsumerOffsets::default(); + let capacity = ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 2); drop(capacity.reserve_provisional(7, &durable).unwrap()); drop(capacity.reserve_provisional(8, &durable).unwrap()); assert_eq!(capacity.provisional.borrow().len(), 2); @@ -579,8 +558,8 @@ mod tests { #[test] fn given_provisional_and_journal_reservations_when_rebuilt_and_canceled_should_preserve_journal_slot() { - let durable = Rc::new(DurableConsumerOffsets::default()); - let capacity = Rc::new(ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 1)); + let durable = DurableConsumerOffsets::default(); + let capacity = ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 1); let provisional = capacity .reserve_provisional(7, &durable) .expect("reserve poll"); @@ -593,8 +572,8 @@ mod tests { #[test] fn given_dropped_submit_when_guard_leaves_scope_should_release_only_its_key() { - let durable = Rc::new(DurableConsumerOffsets::default()); - let capacity = Rc::new(ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 2)); + let durable = DurableConsumerOffsets::default(); + let capacity = ConsumerOffsetCapacity::new(ConsumerKind::Consumer, 2); let first = capacity .reserve_provisional(7, &durable) .expect("reserve first poll"); diff --git a/core/partitions/src/iggy_partition.rs b/core/partitions/src/iggy_partition.rs index 4f399f0749..0f5fac0af5 100644 --- a/core/partitions/src/iggy_partition.rs +++ b/core/partitions/src/iggy_partition.rs @@ -29,27 +29,28 @@ use crate::offset_storage::{ }; use crate::persistence::{PartitionPersistence, PersistenceCompletion, PersistenceNotifier}; use crate::poll_plan::{ - AutoCommitCtx, AutoCommitTarget, DiskReadPlan, DiskSegment, LastPolledCtx, - PartitionDirResolution, PollPlan, PollTier, ResidentTailSnapshot, + DiskReadPlan, DiskSegment, PartitionDirResolution, PollContext, PollPlan, PollReadResult, + PollTier, ResidentTailSnapshot, }; use crate::segment::Segment; use crate::state_transfer::{PartitionTransferSession, PendingTransferRearm}; use crate::types::{COMMIT_WALK_OPS_MAX, FatalCommit, RepairConclusion, RepairSession}; use crate::{ - AppendResult, Partition, PartitionOffsets, PartitionsConfig, PollQueryResult, PollingArgs, - PollingConsumer, + AppendResult, Partition, PartitionOffsets, PartitionsConfig, PollFragments, PollQueryResult, + PollingArgs, PollingConsumer, }; use consensus::Pipeline; use consensus::{ - ClientTable, ClientTableMode, CommitLogEvent, Consensus, PartitionDiagEvent, PipelineEntry, - PlaneKind, Project, ReplicaLogContext, RequestLogEvent, Sequencer, SimEventKind, VsrConsensus, - ack_preflight, ack_quorum_reached, build_deny_reply_from_request, build_reply_from_request, - build_reply_message, drain_committable_prefix, emit_namespace_progress_event, - emit_partition_diag, emit_sim_event, fence_old_prepare_by_commit, repair_session_live, - repaired_frontier_update, replicate_frozen_to_next_in_chain, replicate_preflight, - report_uncommittable_head, restamp_prepare_view, send_prepare_ok as send_prepare_ok_common, - verify_prepare_integrity, + AutoCommitRequestContext, ClientTable, ClientTableMode, CommitLogEvent, Consensus, + PartitionDiagEvent, PipelineEntry, PlaneKind, Project, ReplicaLogContext, RequestLogEvent, + Sequencer, SimEventKind, VsrConsensus, ack_preflight, ack_quorum_reached, + build_deny_reply_from_request, build_reply_from_request, build_reply_message, + drain_committable_prefix, emit_namespace_progress_event, emit_partition_diag, emit_sim_event, + fence_old_prepare_by_commit, repair_session_live, repaired_frontier_update, + replicate_frozen_to_next_in_chain, replicate_preflight, report_uncommittable_head, + restamp_prepare_view, send_prepare_ok as send_prepare_ok_common, verify_prepare_integrity, }; +use iggy_binary_protocol::primitives::consumer::WireConsumer; use iggy_binary_protocol::requests::consumer_offsets::{ DeleteConsumerOffsetRequest, StoreConsumerOffsetRequest, }; @@ -57,7 +58,7 @@ use iggy_binary_protocol::responses::messages::{ SendMessagesConfirmationResponse, SendMessagesResponse, }; use iggy_binary_protocol::{ - AckLevel, Operation, PrepareHeader, WireDecode, WireEncode, WireIdentifier, + AckLevel, Command, Operation, PrepareHeader, WireDecode, WireEncode, WireIdentifier, }; use iggy_binary_protocol::{PrepareOkHeader, ReplyHeader, RoutedRequestHeader}; use iggy_common::{ @@ -71,7 +72,8 @@ use journal::superblock::{ PingPongSuperblock, SUPERBLOCK_RETRY_BACKOFF_BASE_MICROS, SUPERBLOCK_RETRY_BACKOFF_MAX_MICROS, SUPERBLOCK_RETRY_BACKOFF_MAX_SHIFT, SuperblockStore, }; -use message_bus::{IggyMessageBus, MessageBus, is_auto_commit_client}; +use message_bus::{AUTO_COMMIT_CLIENT_ID, IggyMessageBus, MessageBus, is_auto_commit_client}; +use server_common::poll::{AutoCommitReservation, PollHistoryId}; use server_common::{ Message, SegmentStorage, iobuf::Frozen, @@ -215,15 +217,15 @@ where /// server down without the partition moving again in the meantime. fatal: Option, pub(crate) pending_consumer_offset_commits: HashMap, - pub(crate) queued_auto_commit_reservations: - RefCell>>, + /// Identity shared with pending polls and replaced when their history retires. + poll_history: PollHistoryId, /// Committed consumer-offset membership and values. This is deliberately /// separate from the eager poll maps because follower-local and uncommitted /// auto-commit progress must never consume a durable slot or enter a state /// transfer artifact. - pub(crate) durable_consumer_offsets: Rc, - pub(crate) consumer_offset_capacity: Rc, - pub(crate) consumer_group_offset_capacity: Rc, + pub(crate) durable_consumer_offsets: DurableConsumerOffsets, + pub(crate) consumer_offset_capacity: ConsumerOffsetCapacity, + pub(crate) consumer_group_offset_capacity: ConsumerOffsetCapacity, pub(crate) observed_view: u32, offset_reservations_need_resync: Cell, offset_reservations_scan_state: Option<(u64, u64, u64, Option)>, @@ -386,6 +388,28 @@ where } } +/// Read accepted by the owner, with any local progress updates already applied. +/// Acceptance does not imply that its automatic offset commit is durable. +#[derive(Debug)] +pub struct PollCompletion { + /// Selected message bytes that the owner has authorized for the reply. + pub fragments: PollFragments, + /// Partition message frontier from planning, which may now lag new commits. + pub current_offset: u64, + /// Assigned prepare to replicate after releasing the reply. `None` can also + /// mean the automatic commit is queued and has no prepare slot yet. + pub replication: Option, +} + +/// A prepare assigned by completion, with capacity held through journal staging. +#[derive(Debug)] +pub struct PollReplication { + /// Automatic offset commit with an operation number already assigned. + prepare: Message, + /// Keeps the consumer key reserved until replication stages the prepare. + reservation: AutoCommitReservation, +} + /// Post-preflight dispatch in `on_request`: replicate via VSR or take the /// `NoAck` leader-local fast path. `RoutedRequestHeader` is boxed to avoid the /// 277-byte inline variant tripping clippy's `large_enum_variant`. @@ -585,16 +609,16 @@ where installed_frontier: None, fatal: None, pending_consumer_offset_commits: HashMap::new(), - queued_auto_commit_reservations: RefCell::new(HashMap::new()), - durable_consumer_offsets: Rc::new(DurableConsumerOffsets::default()), - consumer_offset_capacity: Rc::new(ConsumerOffsetCapacity::new( + poll_history: PollHistoryId::default(), + durable_consumer_offsets: DurableConsumerOffsets::default(), + consumer_offset_capacity: ConsumerOffsetCapacity::new( ConsumerKind::Consumer, crate::DEFAULT_CONSUMER_OFFSETS_MAX, - )), - consumer_group_offset_capacity: Rc::new(ConsumerOffsetCapacity::new( + ), + consumer_group_offset_capacity: ConsumerOffsetCapacity::new( ConsumerKind::ConsumerGroup, crate::DEFAULT_CONSUMER_OFFSETS_MAX, - )), + ), observed_view, offset_reservations_need_resync: Cell::new(false), offset_reservations_scan_state: None, @@ -1676,6 +1700,7 @@ where /// frontier must not lower it. #[cfg(any(test, feature = "simulator"))] pub fn adopt_retained_log(&mut self, state: crate::RetainedPartitionState) { + self.invalidate_poll_history(); let crate::RetainedPartitionState { log, durable_offset, @@ -2558,6 +2583,7 @@ where consumer_offsets: ConsumerOffsets, consumer_group_offsets: ConsumerGroupOffsets, ) { + self.invalidate_poll_history(); self.consumer_offsets = Arc::new(consumer_offsets); self.consumer_group_offsets = Arc::new(consumer_group_offsets); self.consumer_offsets_path = Some(consumer_offsets_path); @@ -2614,7 +2640,7 @@ where /// must already have been appended to `self.log.journal` by the caller /// so `VsrAction::RetransmitPrepares` can recover it during a view /// change. The on-disk offset table is NOT touched here: persist runs - /// from [`apply_staged_consumer_offset_commit`] at commit-time so a + /// from [`Self::apply_staged_consumer_offset_commit`] at commit time so a /// view-change rollback of the in-memory pending entry also rolls /// back the disk write (by never having performed it). pub(crate) fn stage_consumer_offset_upsert( @@ -2638,7 +2664,7 @@ where } /// Stage a consumer offset delete for the replicated op. See - /// [`stage_consumer_offset_upsert`] for the ordering contract. + /// [`Self::stage_consumer_offset_upsert`] for the ordering contract. /// /// Deliberately infallible: this runs on the replicated-apply path (every /// replica), where the offset may legitimately be absent (e.g. a backup @@ -2861,19 +2887,282 @@ where } } - /// Reject completion from a removed partition or a view whose retained - /// prepares have not yet been accounted for by the pump. - #[must_use] - pub fn auto_commit_admission_ready(&self, applied: &crate::AutoCommitApplied) -> bool { - applied.belongs_to(&self.durable_consumer_offsets) - && self.observed_view == self.consensus.view() + /// Accept a read only while its message history still belongs to this owner. + /// Validation, admission, and progress updates must stay in one synchronous + /// owner turn so purge or recovery cannot interleave between them. + /// Nonempty group reads update `last_polled` even without automatic commits. + /// Rejection leaves this read's progress unapplied. + pub(crate) fn complete_poll( + &mut self, + result: PollReadResult, + ) -> Result { + // Recovery can become necessary while disk I/O is pending, even if + // the history identity has not changed. + if result.context.history != self.poll_history + || self.fatal.is_some() + || self.materialization_missing + { + return Err(IggyError::TransientNotAccepted); + } + self.resynchronize_consumer_offset_reservations(); + let mut replication = None; + if let Some(offset) = result + .last_matching_offset + .filter(|_| !result.fragments.is_empty()) + { + if result.context.auto_commit { + let pending = PendingConsumerOffsetCommit::try_from_polling_consumer( + result.context.consumer, + offset, + )?; + let kind = pending.kind; + let consumer_id = pending.consumer_id; + // Admit before changing either progress map, or a rejected + // read could make the next poll skip its messages. + replication = self.admit_poll_auto_commit(kind, consumer_id, offset)?; + self.apply_local_poll_offset(kind, consumer_id, offset); + } + if let PollingConsumer::ConsumerGroup(group_id, _) = result.context.consumer { + upsert_offset_max( + &self.last_polled_offsets, + ConsumerGroupId(group_id), + offset, + || { + ConsumerOffset::new( + ConsumerKind::ConsumerGroup, + u32::try_from(group_id).unwrap_or(u32::MAX), + offset, + String::new(), + ) + }, + ); + } + } + Ok(PollCompletion { + fragments: result.fragments, + current_offset: result.commit_offset, + replication, + }) + } + + /// Stage an assigned prepare and release its provisional capacity guard. + /// The owner releases the poll reply first because replica sends may wait. + pub(crate) async fn replicate_poll_completion(&mut self, replication: PollReplication) { + let PollReplication { + prepare, + reservation, + } = replication; + self.on_replicate(prepare).await; + drop(reservation); + } + + /// Invalidate after preflight, before replacing data or progress. + /// Preserve the identity when preflight leaves served state unchanged. + /// Queued automatic commits also belong to the old history, even when + /// their reads have already completed. + pub(crate) fn invalidate_poll_history(&mut self) { + self.poll_history = PollHistoryId::default(); + self.discard_queued_auto_commits(); + } + + fn discard_queued_auto_commits(&self) { + self.consensus.with_pipeline_mut(|pipeline| { + pipeline.retain_requests(|request| request.auto_commit().is_none()); + }); + } + + /// Admit an automatic commit without advancing this read's progress. + /// `Some` carries an assigned prepare. `None` means the request is queued, + /// the durable offset already covers it, or the current consensus role or + /// state cannot originate a prepare. Errors release any provisional guard. + fn admit_poll_auto_commit( + &self, + kind: ConsumerKind, + consumer_id: u32, + offset: u64, + ) -> Result, IggyError> { + if !self.auto_commit_admission_ready(kind, consumer_id) { + return Err(IggyError::TransientNotAccepted); + } + self.check_local_poll_key(kind, consumer_id) + .map_err(|error| self.poll_capacity_error(error))?; + let consensus = self.consensus(); + if !consensus.is_primary() + || !consensus.is_normal() + || consensus.is_transferring() + || self + .durable_consumer_offsets + .covers(kind, consumer_id, offset) + { + return Ok(None); + } + + let reservation = self + .consumer_offset_capacity_for(kind) + .reserve_provisional(consumer_id, &self.durable_consumer_offsets) + .map_err(|error| self.poll_capacity_error(error))?; + let request = self.build_poll_auto_commit_request(kind, consumer_id, offset)?; + // Automatic commits bypass the request handler, so check journal + // capacity here before queueing or assigning an operation. + if self + .persistence + .as_ref() + .is_some_and(|persistence| !persistence.has_capacity(request.as_slice().len())) + { + return Err(IggyError::TransientNotAccepted); + } + if self.consensus.pipeline_is_full() { + let context = AutoCommitRequestContext { + history: self.poll_history, + reservation, + }; + self.consensus + .push_queued_request(consensus::RequestEntry::with_auto_commit(request, context)) + .map_err(|_| IggyError::TransientNotAccepted)?; + Ok(None) + } else { + self.reserve_consumer_offset(kind, consumer_id) + .map_err(|error| self.poll_capacity_error(error))?; + let prepare = request.project(self.consensus()); + self.consensus + .pipeline_message(PlaneKind::Partitions, &prepare); + Ok(Some(PollReplication { + prepare, + reservation, + })) + } + } + + fn auto_commit_admission_ready(&self, kind: ConsumerKind, consumer_id: u32) -> bool { + self.observed_view == self.consensus.view() && !self.offset_reservations_need_resync.get() - && (!self - .consumer_offset_capacity_for(applied.kind) - .is_uncertain() - || self - .durable_consumer_offsets - .contains(applied.kind, applied.consumer_id)) + && (!self.consumer_offset_capacity_for(kind).is_uncertain() + || self.durable_consumer_offsets.contains(kind, consumer_id)) + } + + fn check_local_poll_key( + &self, + kind: ConsumerKind, + consumer_id: u32, + ) -> Result<(), ConsumerOffsetCapacityError> { + let exists = match kind { + ConsumerKind::Consumer => self + .consumer_offsets + .pin() + .contains_key(&(consumer_id as usize)), + ConsumerKind::ConsumerGroup => self + .consumer_group_offsets + .pin() + .contains_key(&ConsumerGroupId(consumer_id as usize)), + }; + if exists { + return Ok(()); + } + let capacity = self.consumer_offset_capacity_for(kind); + let count = self.consumer_offset_map_count(kind); + if count >= capacity.limit() { + self.reclaim_phantom_offsets(kind, count); + } + capacity.admit_local_map_key( + self.consumer_offset_map_count(kind), + self.durable_consumer_offsets.count(kind) >= capacity.limit(), + ) + } + + /// Advance automatic progress without letting a slower read move it back. + /// Explicit offset stores retain their separate semantics and may rewind. + fn apply_local_poll_offset(&self, kind: ConsumerKind, consumer_id: u32, offset: u64) { + let existed = match kind { + ConsumerKind::Consumer => self + .consumer_offsets + .pin() + .contains_key(&(consumer_id as usize)), + ConsumerKind::ConsumerGroup => self + .consumer_group_offsets + .pin() + .contains_key(&ConsumerGroupId(consumer_id as usize)), + }; + let create = || { + ConsumerOffset::new( + kind, + consumer_id, + offset, + self.persisted_offset_path(kind, consumer_id) + .unwrap_or_default(), + ) + }; + match kind { + ConsumerKind::Consumer => { + upsert_offset_max(&self.consumer_offsets, consumer_id as usize, offset, create); + } + ConsumerKind::ConsumerGroup => upsert_offset_max( + &self.consumer_group_offsets, + ConsumerGroupId(consumer_id as usize), + offset, + create, + ), + } + if !existed { + self.consumer_offset_capacity_for(kind) + .note_local_key_change(); + } + } + + fn poll_capacity_error(&self, error: ConsumerOffsetCapacityError) -> IggyError { + if error.first_in_episode { + warn!(namespace_raw = self.namespace().inner(), kind = ?error.kind, + occupied = error.occupied, limit = error.limit, uncertain = error.uncertain, + "consumer offset admission refused during poll completion"); + } + error.into() + } + + fn build_poll_auto_commit_request( + &self, + kind: ConsumerKind, + consumer_id: u32, + offset: u64, + ) -> Result, IggyError> { + let namespace = self.namespace(); + let request = StoreConsumerOffsetRequest { + consumer: WireConsumer { + kind: kind.as_code(), + id: WireIdentifier::Numeric(consumer_id), + }, + stream_id: WireIdentifier::Numeric( + u32::try_from(namespace.stream_id()) + .map_err(|_| IggyError::InvalidConfiguration)?, + ), + topic_id: WireIdentifier::Numeric( + u32::try_from(namespace.topic_id()).map_err(|_| IggyError::InvalidConfiguration)?, + ), + partition_id: Some( + u32::try_from(namespace.partition_id()) + .map_err(|_| IggyError::InvalidConfiguration)?, + ), + offset, + ack: AckLevel::Quorum, + }; + let body = request.to_bytes(); + let header_size = std::mem::size_of::(); + let total_size = header_size + body.len(); + let size = u32::try_from(total_size).map_err(|_| IggyError::InvalidConfiguration)?; + let mut message = Message::::new(total_size); + message.as_mut_slice()[header_size..].copy_from_slice(&body); + Ok( + message.transmute_header(|_, header: &mut RoutedRequestHeader| { + *header = RoutedRequestHeader { + command: Command::Request, + operation: Operation::StoreConsumerOffset, + size, + client: AUTO_COMMIT_CLIENT_ID, + session: 1, + request: 1, + group: namespace.inner(), + ..Default::default() + }; + }), + ) } fn apply_consumer_offset_commit(&self, pending: PendingConsumerOffsetCommit) { @@ -3157,7 +3446,7 @@ where } } - pub(crate) fn consumer_offset_capacity_for( + pub(crate) const fn consumer_offset_capacity_for( &self, kind: ConsumerKind, ) -> &ConsumerOffsetCapacity { @@ -3328,7 +3617,7 @@ where } if current_view != self.observed_view { - self.queued_auto_commit_reservations.borrow_mut().clear(); + self.discard_queued_auto_commits(); self.mark_consumer_group_offsets_need_reconcile(); } @@ -3483,10 +3772,8 @@ where } } - /// Build an owned [`PollPlan`] synchronously (no `.await`), so the caller - /// can run the disk read + offset persist off the partition borrow. The - /// in-memory journal tier is read here directly (mem reads never yield); - /// the disk tier is captured as owned descriptors in [`DiskReadPlan`]. + /// Snapshot read resources synchronously on the partition owner. + /// Only the owned snapshot crosses a suspension during disk I/O. #[allow(clippy::too_many_lines)] pub(crate) fn build_poll_plan( &mut self, @@ -3498,11 +3785,15 @@ where // commit). Also used below as the poll's high-water bound: this function // is fully synchronous, so the single load cannot drift mid-plan. let commit_offset = self.offsets().commit_offset; + let context = PollContext { + history: self.poll_history, + consumer, + auto_commit: args.auto_commit, + }; if !self.offset_space.committed_seeded || args.count == 0 { return PollPlan { commit_offset, - auto_commit: None, - last_polled: None, + context, tier: PollTier::Empty, }; } @@ -3526,8 +3817,7 @@ where if start_offset > commit_offset { return PollPlan { commit_offset, - auto_commit: None, - last_polled: None, + context, tier: PollTier::Empty, }; } @@ -3563,21 +3853,6 @@ where } } - // Past the empty-return guards: only now build the auto-commit context, - // whose offset-path `format!()` is wasted on the early returns above. - let auto_commit = self.auto_commit_ctx(consumer, args.auto_commit); - // Cooperative-rebalance: record the highest offset served to a group so - // the drain reconciler can tell committed >= last-polled. Captured here - // as an owned `Arc` and applied off the borrow in `PollPlan::execute`, - // since the served offset is unknown until the poll completes. - let last_polled = match consumer { - PollingConsumer::ConsumerGroup(group_id, _) => Some(LastPolledCtx { - offsets: self.last_polled_offsets.clone(), - group_id, - }), - PollingConsumer::Consumer(..) => None, - }; - let serve_journal_first = match query { MessageLookup::Offset { offset, .. } => self .log @@ -3598,8 +3873,7 @@ where }; return PollPlan { commit_offset, - auto_commit, - last_polled, + context, tier, }; } @@ -3639,8 +3913,7 @@ where let resident_tail = self.resident_tail_snapshot(); PollPlan { commit_offset, - auto_commit, - last_polled, + context, tier: PollTier::Disk { disk, query, @@ -3649,43 +3922,6 @@ where } } - /// Capture the owned inputs for an auto-commit, if requested: the lock-free - /// offset-map `Arc` and the target consumer/group id, so the in-memory apply - /// runs off the partition borrow once the poll's served offset is known. - /// Durability is not captured here: the poll no longer writes the offset - /// file, the serving shard replicates the offset through consensus instead. - fn auto_commit_ctx( - &self, - consumer: PollingConsumer, - auto_commit: bool, - ) -> Option { - if !auto_commit { - return None; - } - let pending = PendingConsumerOffsetCommit::try_from_polling_consumer(consumer, 0).ok()?; - let target = match pending.kind { - ConsumerKind::Consumer => AutoCommitTarget::Consumer { - offsets: self.consumer_offsets.clone(), - consumer_id: pending.consumer_id, - create_path: self.consumer_offsets_path.clone(), - }, - ConsumerKind::ConsumerGroup => AutoCommitTarget::ConsumerGroup { - offsets: self.consumer_group_offsets.clone(), - group_id: pending.consumer_id, - create_path: self.consumer_group_offsets_path.clone(), - }, - }; - let capacity = match pending.kind { - ConsumerKind::Consumer => Rc::clone(&self.consumer_offset_capacity), - ConsumerKind::ConsumerGroup => Rc::clone(&self.consumer_group_offset_capacity), - }; - Some(AutoCommitCtx { - target, - capacity, - durable: Rc::clone(&self.durable_consumer_offsets), - }) - } - /// Synchronous in-memory journal poll, for the resident tier. Never awaits /// (see [`PartitionJournal::get_sync`]), so it is safe under a partition /// borrow. @@ -3942,34 +4178,12 @@ where /// Panics if called when this partition's consensus instance is not the /// primary, is not in normal status, or is currently syncing. #[allow(clippy::future_not_send, clippy::too_many_lines)] - pub async fn on_request( - &mut self, - message: Message, - reply: Option>>, - ) { - self.on_request_with_reservation(message, reply, None).await; - } - #[allow(clippy::too_many_lines)] - pub(crate) async fn on_request_with_reservation( + pub async fn on_request( &mut self, message: Message, reply: Option>>, - mut reservation: Option, ) { - if reservation.as_ref().is_some_and(|reservation| { - !self - .consumer_offset_capacity_for(reservation.kind) - .owns(reservation) - }) { - debug!( - namespace_raw = self.namespace().inner(), - kind = ?reservation.as_ref().map(|reservation| reservation.kind), - consumer_id = ?reservation.as_ref().map(|reservation| reservation.consumer_id), - "dropping auto-commit reservation owned by another partition incarnation" - ); - return; - } // Taken by whichever arm answers: the deny paths, the NoAck fast path, // or the pipeline entry that fires it at commit. Exactly one runs. let mut reply = reply; @@ -4290,12 +4504,6 @@ where waiter, ) .await; - } else if let Some(reservation) = reservation.take() { - self.queued_auto_commit_reservations - .borrow_mut() - .entry((reservation.kind, reservation.consumer_id)) - .or_default() - .push(reservation); } return; } @@ -4387,21 +4595,21 @@ where &req.message, ) }); - let _reservation = parsed_store - .as_ref() - .and_then(|parsed| parsed.as_ref().ok()) - .and_then(|(kind, id, _, _)| { - if !is_auto_commit_client(req.message.header().client) { - return None; - } - let mut queued = self.queued_auto_commit_reservations.borrow_mut(); - let reservations = queued.get_mut(&(*kind, *id))?; - let reservation = reservations.pop(); - if reservations.is_empty() { - queued.remove(&(*kind, *id)); - } - reservation - }); + let context = req.take_auto_commit(); + // History and capacity ownership can change while the request + // waits for a prepare slot, so admission alone is not sufficient. + if context.as_ref().is_some_and(|context| { + context.history != self.poll_history + || !self + .consumer_offset_capacity_for(context.reservation.kind()) + .owns(&context.reservation) + }) { + consecutive_denials += 1; + if consecutive_denials >= PROMOTION_DENIALS_MAX { + break; + } + continue; + } // Taken before the preflight so a refusal answers the parked waiter // instead of waking it with `Canceled`. @@ -4492,6 +4700,7 @@ where }; promoted += 1; self.on_replicate(prepare).await; + drop(context); } } @@ -7378,6 +7587,7 @@ where } } self.record_purge_frontier_reset(generation).await?; + self.invalidate_poll_history(); // The purge recreates segment files at the paths it unlinks below, so // an in-flight poll's cached read fd would keep serving the unlinked @@ -7609,7 +7819,7 @@ where } self.durable_consumer_offsets.clear(); self.pending_consumer_offset_commits.clear(); - self.queued_auto_commit_reservations.borrow_mut().clear(); + self.consumer_offset_capacity .rebuild(&self.durable_consumer_offsets, std::iter::empty()); self.consumer_group_offset_capacity @@ -8106,12 +8316,8 @@ where } } -/// Commit-apply an upserted offset into a lock-free offset map. A server -/// auto-commit already advanced this offset in memory on the serving poll and -/// this replicated commit can land behind a newer poll, so it must be -/// monotone (`fetch_max`) or it rewinds the map and re-serves consumed -/// messages. An explicit client store keeps the rewinding `store` (an offset -/// reset is a valid action). +/// Automatic commits remain monotone because an earlier poll can commit after +/// a later one. Explicit stores may intentionally rewind the cursor. fn upsert_committed_offset( map: &papaya::HashMap, key: K, @@ -8122,9 +8328,45 @@ fn upsert_committed_offset( K: Hash + Eq + Clone + Send + Sync, { if auto_commit { - crate::poll_plan::upsert_offset_max(map, key, offset, create_on_miss); + upsert_offset_max(map, key, offset, create_on_miss); + } else { + upsert_offset(map, key, offset, create_on_miss); + } +} + +fn upsert_offset( + map: &papaya::HashMap, + key: K, + offset: u64, + create_on_miss: impl FnOnce() -> ConsumerOffset, +) where + K: Hash + Eq + Clone + Send + Sync, +{ + let guard = map.pin(); + if let Some(existing) = guard.get(&key) { + existing.offset.store(offset, Ordering::Relaxed); + } else { + let created = create_on_miss(); + created.offset.store(offset, Ordering::Relaxed); + guard.insert(key, created); + } +} + +fn upsert_offset_max( + map: &papaya::HashMap, + key: K, + offset: u64, + create_on_miss: impl FnOnce() -> ConsumerOffset, +) where + K: Hash + Eq + Clone + Send + Sync, +{ + let guard = map.pin(); + if let Some(existing) = guard.get(&key) { + existing.offset.fetch_max(offset, Ordering::Relaxed); } else { - crate::poll_plan::upsert_offset(map, key, offset, create_on_miss); + let created = create_on_miss(); + created.offset.store(offset, Ordering::Relaxed); + guard.insert(key, created); } } @@ -8962,12 +9204,13 @@ mod tests { ); assert_eq!(partition.log.journal().inner.resident_count(), 0); let args = PollingArgs::new(iggy_common::PollingStrategy::offset(1), 1, false); - let (fragments, _, _) = partition + let result = partition .build_poll_plan(PollingConsumer::Consumer(1, 0), &args, true) .execute() - .await - .unwrap(); - let polled: Vec<_> = fragments + .await; + let completion = partition.complete_poll(result).unwrap(); + let polled: Vec<_> = completion + .fragments .iter() .flat_map(|fragment| fragment.as_slice().iter().copied()) .collect(); @@ -10755,6 +10998,17 @@ mod tests { pub(super) fn recording_partition_at( replica: u8, replica_count: u8, + ) -> (IggyPartition, SentFrames) { + recording_partition_with_pipeline(replica, replica_count, LocalPipeline::new()) + } + + /// Creates an empty partition with the supplied consensus queue capacities. + /// Its bus records sends without contacting clients or replicas; the returned + /// frame list contains client replies only. + fn recording_partition_with_pipeline( + replica: u8, + replica_count: u8, + pipeline: LocalPipeline, ) -> (IggyPartition, SentFrames) { let namespace = IggyNamespace::new(1, 1, 0); let bus = RecordingBus::default(); @@ -10765,7 +11019,7 @@ mod tests { replica_count, namespace.inner(), bus, - LocalPipeline::new(), + pipeline, ); consensus.init(); let partition = IggyPartition::with_in_memory_storage( @@ -11681,37 +11935,23 @@ mod tests { None, ) .await; - let reservation = partition - .consumer_offset_capacity - .reserve_provisional(8, &partition.durable_consumer_offsets) - .unwrap(); - partition - .on_request_with_reservation( - store_offset_request( - message_bus::AUTO_COMMIT_CLIENT_ID, - 1, - ConsumerKind::Consumer, - 8, - 0, - AckLevel::Quorum, - ), - None, - Some(reservation), - ) - .await; + let result = poll_read_result(&partition, PollingConsumer::Consumer(8, 0), true, Some(0)); + let completion = partition + .complete_poll(result) + .expect("queue automatic commit"); + assert!(completion.replication.is_none()); + assert_eq!( + partition.get_consumer_offset(PollingConsumer::Consumer(8, 0)), + Some(0) + ); assert_eq!( partition.occupied_consumer_offset_count(ConsumerKind::Consumer), 2 ); - assert_eq!(partition.queued_auto_commit_reservations.borrow().len(), 1); + assert_eq!(partition.consensus.request_queue_len(), 1); partition.consensus.set_view(3); partition.resynchronize_consumer_offset_reservations(); - assert!( - partition - .queued_auto_commit_reservations - .borrow() - .is_empty() - ); + assert_eq!(partition.consensus.request_queue_len(), 0); assert_eq!( partition.occupied_consumer_offset_count(ConsumerKind::Consumer), 1 @@ -11723,23 +11963,13 @@ mod tests { let (mut partition, sent) = recording_partition_at(0, 3); partition.stats.increment_messages_count(1); partition.set_consumer_offsets_max(1); - let reservation = partition - .consumer_offset_capacity - .reserve_provisional(7, &partition.durable_consumer_offsets) - .unwrap(); + let result = poll_read_result(&partition, PollingConsumer::Consumer(7, 0), true, Some(0)); + let completion = partition + .complete_poll(result) + .expect("accept automatic commit"); + assert!(partition.pending_consumer_offset_commits.is_empty()); partition - .on_request_with_reservation( - store_offset_request( - message_bus::AUTO_COMMIT_CLIENT_ID, - 1, - ConsumerKind::Consumer, - 7, - 0, - AckLevel::Quorum, - ), - None, - Some(reservation), - ) + .replicate_poll_completion(completion.replication.expect("assigned prepare")) .await; assert_eq!(partition.pending_consumer_offset_commits.len(), 1); assert_eq!( @@ -11832,27 +12062,762 @@ mod tests { drop(held); } - #[test] - fn given_full_live_map_when_polling_existing_key_should_reclaim_only_for_new_keys() { - let (mut partition, _) = recording_partition(); - partition.set_consumer_offsets_max(2); - partition.offset_space.committed_seeded = true; - partition.seed_recovered_consumer_offset(ConsumerKind::Consumer, 7, 0, 0); - for id in 7..=8 { - partition.consumer_offsets.pin().insert( - id as usize, - ConsumerOffset::new(ConsumerKind::Consumer, id, 0, String::new()), - ); + /// Constructs a result awaiting owner acceptance, without reading messages or + /// updating progress. `Some` adds a placeholder fragment so completion takes + /// the nonempty path; these bytes are never decoded. `None` models an empty read. + fn poll_read_result( + partition: &IggyPartition, + consumer: PollingConsumer, + auto_commit: bool, + last_matching_offset: Option, + ) -> PollReadResult { + let mut fragments = PollFragments::new(); + if last_matching_offset.is_some() { + fragments.push(crate::Fragment::whole(Owned::<4096>::zeroed(8).into())); + } + PollReadResult { + context: PollContext { + history: partition.poll_history, + consumer, + auto_commit, + }, + fragments, + commit_offset: partition.offsets().commit_offset, + last_matching_offset, } - let args = PollingArgs::new(iggy_common::PollingStrategy::first(), 1, true); - let _ = partition.build_poll_plan(PollingConsumer::Consumer(7, 0), &args, false); - assert_eq!(partition.consumer_offsets.len(), 2); - let _ = partition.build_poll_plan(PollingConsumer::Consumer(9, 0), &args, false); - assert_eq!(partition.consumer_offsets.len(), 1); - assert!(partition.consumer_offsets.pin().contains_key(&7)); } - #[compio::test] + #[test] + fn given_read_result_when_owner_accepts_should_advance_before_replication() { + let (mut partition, _) = recording_partition(); + // Let Next compare its starting offset with the committed frontier (0) + // instead of taking the shortcut for a partition that has never had data. + partition.offset_space.committed_seeded = true; + let consumer_id = 7; + let partition_id = 0; + let auto_commit = true; + let consumer = PollingConsumer::Consumer(consumer_id, partition_id); + let read_result = poll_read_result(&partition, consumer, auto_commit, Some(0)); + assert_eq!(partition.get_consumer_offset(consumer), None); + + // Acceptance advances local progress and assigns replication work, but + // the automatic commit has not yet been staged in the journal. + let completion = partition.complete_poll(read_result).expect("accept poll"); + assert_eq!(partition.get_consumer_offset(consumer), Some(0)); + assert!(completion.replication.is_some()); + assert!(partition.pending_consumer_offset_commits.is_empty()); + + // Next already starts after offset 0, before replication runs. + let validate_checksum = false; + let next_result = partition + .build_poll_plan( + consumer, + &PollingArgs::new(iggy_common::PollingStrategy::next(), 1, auto_commit), + validate_checksum, + ) + .execute_resident(); + assert!(next_result.fragments.is_empty()); + } + + #[test] + fn given_capacity_refusal_when_poll_completes_should_preserve_existing_progress() { + let (mut partition, _) = recording_partition(); + let durable_consumer_id = 8; + let polling_consumer_id = 7; + let partition_id = 0; + let auto_commit = true; + + // Another consumer owns the only durable slot. The polling consumer has + // local progress at 4, but cannot admit an automatic commit through 9. + partition.set_consumer_offsets_max(1); + partition.seed_recovered_consumer_offset(ConsumerKind::Consumer, durable_consumer_id, 0, 0); + partition.apply_local_poll_offset(ConsumerKind::Consumer, polling_consumer_id, 4); + let consumer = PollingConsumer::Consumer(polling_consumer_id as usize, partition_id); + let read_result = poll_read_result(&partition, consumer, auto_commit, Some(9)); + + assert!(matches!( + partition.complete_poll(read_result), + Err(IggyError::TooManyConsumerOffsets) + )); + assert_eq!(partition.get_consumer_offset(consumer), Some(4)); + assert_eq!(partition.consensus.pipeline_len(), 0); + } + + #[test] + fn given_group_read_without_auto_commit_when_history_changes_should_not_record_last_polled() { + let (mut partition, _) = recording_partition(); + let group_id = 7; + let member_id = 1; + let auto_commit = false; + let consumer = PollingConsumer::ConsumerGroup(group_id, member_id); + let old_result = poll_read_result(&partition, consumer, auto_commit, Some(0)); + + // Group reads still record last_polled with automatic commits disabled. + // Retiring the history must prevent even that local progress update. + partition.invalidate_poll_history(); + assert!(matches!( + partition.complete_poll(old_result), + Err(IggyError::TransientNotAccepted) + )); + let (last_polled, committed) = partition.group_offset_state(group_id as u64); + assert_eq!( + last_polled, None, + "the old read must not record group progress" + ); + assert_eq!(committed, None, "automatic commits are disabled"); + } + + #[compio::test] + async fn given_stale_poll_when_reservations_need_resync_should_reject_before_reconciliation() { + let mut partition = test_partition(); + partition.set_consumer_offsets_max(1); + let discarded_consumer_id = 8; + let discarded_operation = 1; + journal_prepare( + &partition, + discarded_operation, + Operation::StoreConsumerOffset, + ) + .await; + partition + .consensus + .sequencer() + .set_sequence(discarded_operation); + partition.stage_consumer_offset_upsert( + discarded_operation, + ConsumerKind::Consumer, + discarded_consumer_id, + 0, + false, + ); + let consumer = PollingConsumer::Consumer(7, 0); + let auto_commit = true; + let stale_result = poll_read_result(&partition, consumer, auto_commit, Some(0)); + + // Truncation removes the pending operation but leaves its capacity + // reservation for reconciliation. The old read cannot belong to the + // replacement history or trigger that maintenance. + partition.invalidate_poll_history(); + partition + .truncate_uncommitted_from(discarded_operation) + .await + .unwrap(); + assert!(partition.pending_consumer_offset_commits.is_empty()); + assert!(partition.offset_reservations_need_resync.get()); + assert_eq!( + partition.occupied_consumer_offset_count(ConsumerKind::Consumer), + 1 + ); + + assert!(matches!( + partition.complete_poll(stale_result), + Err(IggyError::TransientNotAccepted) + )); + assert_eq!( + partition.occupied_consumer_offset_count(ConsumerKind::Consumer), + 1, + "rejecting a stale read must not reconcile the discarded reservation" + ); + assert!(partition.offset_reservations_need_resync.get()); + assert_eq!(partition.get_consumer_offset(consumer), None); + assert_eq!(partition.consensus.pipeline_len(), 0); + + // A valid completion still reconciles before admission, freeing the + // only slot so the owner can admit an automatic commit for this consumer. + let fresh_result = poll_read_result(&partition, consumer, auto_commit, Some(0)); + let completion = partition + .complete_poll(fresh_result) + .expect("accept fresh poll"); + assert!(completion.replication.is_some()); + assert!(!partition.offset_reservations_need_resync.get()); + assert_eq!(partition.get_consumer_offset(consumer), Some(0)); + assert_eq!( + partition.occupied_consumer_offset_count(ConsumerKind::Consumer), + 1 + ); + } + + #[compio::test] + async fn given_full_wal_when_poll_completes_should_reject_without_progress() { + // A primary with persisted offsets needs write-ahead log capacity before + // accepting an automatic commit. Use real persistence, then exhaust it. + let directory = tempfile::tempdir().unwrap(); + let primary_replica = 0; + let replica_count = 3; + let (mut partition, _) = recording_partition_at(primary_replica, replica_count); + partition.runtime_options.consumer_offset_durability = iggy_common::Durability::Persisted; + partition.set_partition_dir(directory.path().to_string_lossy().into_owned()); + partition.open_persistence().await.unwrap(); + let persistence = Rc::clone(partition.persistence.as_ref().unwrap()); + let group_id = 7; + let member_id = 1; + let auto_commit = true; + let consumer = PollingConsumer::ConsumerGroup(group_id, member_id); + let read_result = poll_read_result(&partition, consumer, auto_commit, Some(0)); + let operation_before_poll = partition.consensus.sequencer().current_sequence(); + persistence.exhaust_capacity_for_test(); + + // Refusal must leave progress, queues, operation assignment, and + // occupied capacity unchanged. + assert!(matches!( + partition.complete_poll(read_result), + Err(IggyError::TransientNotAccepted) + )); + let (last_polled, committed) = partition.group_offset_state(group_id as u64); + assert_eq!(last_polled, None); + assert_eq!(committed, None); + assert_eq!(partition.consensus.pipeline_len(), 0); + assert_eq!(partition.consensus.request_queue_len(), 0); + assert_eq!( + partition.consensus.sequencer().current_sequence(), + operation_before_poll + ); + assert_eq!( + partition + .consumer_group_offset_capacity + .occupied(&partition.durable_consumer_offsets), + 0 + ); + + // The same offset becomes acceptable when capacity returns. These are + // local progress updates; the assigned replication has not run yet. + persistence.release_capacity_for_test(); + let retry_result = poll_read_result(&partition, consumer, auto_commit, Some(0)); + let retry_completion = partition.complete_poll(retry_result).expect("accept retry"); + assert!(retry_completion.replication.is_some()); + let (last_polled, committed) = partition.group_offset_state(group_id as u64); + assert_eq!(last_polled, Some(0)); + assert_eq!(committed, Some(0)); + assert_eq!(partition.consensus.pipeline_len(), 1); + assert_eq!( + partition.consensus.sequencer().current_sequence(), + operation_before_poll + 1 + ); + } + + #[test] + fn given_pending_poll_when_materialization_is_missing_should_reject_without_progress() { + let group_id = 7; + let member_id = 1; + let consumer = PollingConsumer::ConsumerGroup(group_id, member_id); + for auto_commit in [false, true] { + let (mut partition, _) = recording_partition(); + let read_result = poll_read_result(&partition, consumer, auto_commit, Some(9)); + + // The partition can lose its materialized data after planning even + // when its history identity has not changed. + partition.materialization_missing = true; + assert!(matches!( + partition.complete_poll(read_result), + Err(IggyError::TransientNotAccepted) + )); + let (last_polled, committed) = partition.group_offset_state(group_id as u64); + assert_eq!( + last_polled, None, + "a rejected read must not record group progress" + ); + assert_eq!(committed, None, "a rejected read must not commit an offset"); + assert_eq!(partition.consensus.pipeline_len(), 0); + assert_eq!(partition.consensus.request_queue_len(), 0); + } + } + + #[test] + fn given_empty_read_when_partition_is_replaced_should_reject_old_result() { + let (partition, _) = recording_partition(); + let consumer_id = 7; + let partition_id = 0; + let auto_commit = true; + let consumer = PollingConsumer::Consumer(consumer_id, partition_id); + let old_empty_result = poll_read_result(&partition, consumer, auto_commit, None); + + // The same namespace does not give a replacement ownership of + // the old instance's results, even when they contain no messages. + let (mut replacement, _) = recording_partition(); + assert!(matches!( + replacement.complete_poll(old_empty_result), + Err(IggyError::TransientNotAccepted) + )); + } + + #[test] + fn given_group_reads_completing_in_reverse_order_should_keep_progress_monotone() { + // A backup accepts local poll progress without assigning replication. + let backup_replica = 1; + let replica_count = 3; + let (mut partition, _) = recording_partition_at(backup_replica, replica_count); + let group_id = 7; + let member_id = 1; + let auto_commit = true; + let consumer = PollingConsumer::ConsumerGroup(group_id, member_id); + let earlier_result = poll_read_result(&partition, consumer, auto_commit, Some(4)); + let later_result = poll_read_result(&partition, consumer, auto_commit, Some(9)); + + // Accept the result through 9 before the slower result through 4. + assert!( + partition + .complete_poll(later_result) + .expect("accept later poll") + .replication + .is_none() + ); + partition + .complete_poll(earlier_result) + .expect("accept earlier poll"); + let (last_polled, committed) = partition.group_offset_state(group_id as u64); + assert_eq!( + last_polled, + Some(9), + "the slower read must not rewind group progress" + ); + assert_eq!( + committed, + Some(9), + "the slower read must not rewind its offset" + ); + assert_eq!(partition.consensus.pipeline_len(), 0); + } + + #[test] + fn given_pending_auto_commit_when_history_changes_should_release_only_its_queue_entry() { + let (mut partition, _) = recording_partition(); + partition.set_consumer_offsets_max(1); + let automatic_consumer_id = 7; + let explicit_consumer_id = 8; + let explicit_client_id = 42; + + // Only the automatic request carries a history identity and holds the + // provisional slot. Queue an explicit store beside it to prove that + // retiring the history does not discard unrelated client requests. + let automatic_reservation = partition + .consumer_offset_capacity + .reserve_provisional(automatic_consumer_id, &partition.durable_consumer_offsets) + .unwrap(); + let automatic_request = partition + .build_poll_auto_commit_request(ConsumerKind::Consumer, automatic_consumer_id, 0) + .unwrap(); + partition + .consensus + .push_queued_request(consensus::RequestEntry::with_auto_commit( + automatic_request, + AutoCommitRequestContext { + history: partition.poll_history, + reservation: automatic_reservation, + }, + )) + .unwrap(); + let explicit_request = store_offset_request( + explicit_client_id, + 1, + ConsumerKind::Consumer, + explicit_consumer_id, + 0, + AckLevel::Quorum, + ); + partition + .consensus + .push_queued_request(consensus::RequestEntry::new(explicit_request)) + .unwrap(); + assert_eq!( + partition.occupied_consumer_offset_count(ConsumerKind::Consumer), + 1 + ); + + partition.invalidate_poll_history(); + assert_eq!( + partition.occupied_consumer_offset_count(ConsumerKind::Consumer), + 0 + ); + let retained_request = partition + .consensus + .pop_queued_request() + .expect("explicit request survives"); + assert_eq!(retained_request.message.header().client, explicit_client_id); + assert!(partition.consensus.pop_queued_request().is_none()); + } + + #[compio::test] + async fn given_old_auto_commit_context_when_promoted_should_not_assign_an_operation() { + let (mut partition, _) = recording_partition(); + let consumer_id = 7; + let old_reservation = partition + .consumer_offset_capacity + .reserve_provisional(consumer_id, &partition.durable_consumer_offsets) + .unwrap(); + let old_context = AutoCommitRequestContext { + history: partition.poll_history, + reservation: old_reservation, + }; + let old_request = partition + .build_poll_auto_commit_request(ConsumerKind::Consumer, consumer_id, 0) + .unwrap(); + + // Inject the obsolete context after invalidation has swept the queue. + // Promotion must independently check its history before assigning an op. + partition.invalidate_poll_history(); + partition + .consensus + .push_queued_request(consensus::RequestEntry::with_auto_commit( + old_request, + old_context, + )) + .unwrap(); + partition.drain_request_queue_into_prepares(1).await; + + assert_eq!(partition.consensus.pipeline_len(), 0); + assert_eq!(partition.consensus.sequencer().current_sequence(), 0); + assert_eq!( + partition.occupied_consumer_offset_count(ConsumerKind::Consumer), + 0 + ); + } + + #[compio::test] + async fn given_read_when_state_is_installed_should_reject_the_previous_history() { + // State installation writes replacement offset tables, so the fixture + // needs filesystem paths even though its message log starts in memory. + let directory = tempfile::tempdir().unwrap(); + let (mut partition, _) = recording_partition(); + partition.set_partition_dir(directory.path().to_string_lossy().into_owned()); + partition.consumer_offsets_path = Some( + directory + .path() + .join("consumers") + .to_string_lossy() + .into_owned(), + ); + partition.consumer_group_offsets_path = Some( + directory + .path() + .join("groups") + .to_string_lossy() + .into_owned(), + ); + let group_id = 7; + let member_id = 1; + let auto_commit_enabled = true; + let auto_commit_disabled = false; + let consumer = PollingConsumer::ConsumerGroup(group_id, member_id); + let old_automatic_result = + poll_read_result(&partition, consumer, auto_commit_enabled, Some(4)); + let old_manual_result = + poll_read_result(&partition, consumer, auto_commit_disabled, Some(4)); + + // Replace the history after both reads have captured its old identity. + // Offset 4 is below the new frontier, so its value alone cannot establish + // that either result still belongs to the installed history. + let installed_state = crate::state_transfer::ConsumerOffsetsWire { + purge_generation: 0, + prepare_checksum: None, + checkpoint_prepare: Vec::new(), + next_offset: 10, + consumers: Vec::new(), + groups: Vec::new(), + dedup: Vec::new(), + }; + let installed_commit_operation = 12; + let committed_purge_generation = 0; + partition + .install_state_transfer( + &repair_config(), + installed_commit_operation, + Vec::new(), + &installed_state.encode(), + committed_purge_generation, + ) + .await + .unwrap(); + + for old_result in [old_automatic_result, old_manual_result] { + assert!(matches!( + partition.complete_poll(old_result), + Err(IggyError::TransientNotAccepted) + )); + } + let (last_polled, committed) = partition.group_offset_state(group_id as u64); + assert_eq!( + last_polled, None, + "old reads must not restore group progress" + ); + assert_eq!(committed, None, "old reads must not commit an offset"); + assert_eq!(partition.consensus.pipeline_len(), 0); + } + + #[compio::test] + async fn given_read_when_failed_install_converges_should_reject_the_previous_history() { + // Valid offset table paths let installation reach the segment swap. + let directory = tempfile::tempdir().unwrap(); + let (mut partition, _) = recording_partition(); + partition.set_partition_dir(directory.path().to_string_lossy().into_owned()); + partition.consumer_offsets_path = Some( + directory + .path() + .join("consumers") + .to_string_lossy() + .into_owned(), + ); + partition.consumer_group_offsets_path = Some( + directory + .path() + .join("groups") + .to_string_lossy() + .into_owned(), + ); + let group_id = 7; + let member_id = 1; + let auto_commit = true; + let consumer = PollingConsumer::ConsumerGroup(group_id, member_id); + let old_result = poll_read_result(&partition, consumer, auto_commit, Some(4)); + + // Describe staged files without creating them. The swap fails after + // preflight, forcing recovery to replace the served log with an empty one. + let missing_staged_segment = crate::state_transfer::StagedSegmentMeta { + start_offset: 0, + end_offset: 0, + index_size: 0, + size: 8, + start_timestamp: 0, + end_timestamp: 0, + max_timestamp: 0, + log_staging: directory.path().join("00000000000000000000.log.staging"), + index_staging: directory.path().join("00000000000000000000.index.staging"), + }; + let offered_state = crate::state_transfer::ConsumerOffsetsWire { + purge_generation: 0, + prepare_checksum: None, + checkpoint_prepare: Vec::new(), + next_offset: 1, + consumers: Vec::new(), + groups: Vec::new(), + dedup: Vec::new(), + }; + let offered_commit_operation = 12; + let committed_purge_generation = 0; + let install_error = partition + .install_state_transfer( + &repair_config(), + offered_commit_operation, + vec![missing_staged_segment], + &offered_state.encode(), + committed_purge_generation, + ) + .await + .unwrap_err(); + assert!( + matches!( + install_error, + crate::state_transfer::PartitionInstallError::SwapIo { .. } + ), + "unexpected failure: {install_error:?}" + ); + + // Failure did not preserve the old history: recovery left an empty log, + // and its owner must reject the result captured before installation. + assert_eq!(partition.log.segments().len(), 1); + assert_eq!(partition.log.active_segment().size.as_bytes_u64(), 0); + assert!(matches!( + partition.complete_poll(old_result), + Err(IggyError::TransientNotAccepted) + )); + let (last_polled, committed) = partition.group_offset_state(group_id as u64); + assert_eq!(last_polled, None); + assert_eq!(committed, None); + } + + #[compio::test] + async fn given_read_when_install_preflight_fails_should_keep_the_existing_history() { + let directory = tempfile::tempdir().unwrap(); + let (mut partition, _) = recording_partition(); + partition.set_partition_dir(directory.path().to_string_lossy().into_owned()); + let group_id = 7; + let member_id = 1; + let auto_commit = false; + let consumer = PollingConsumer::ConsumerGroup(group_id, member_id); + let pending_result = poll_read_result(&partition, consumer, auto_commit, Some(4)); + + // Empty offset bytes fail decoding before installation mutates served + // state. The result's original history identity must remain acceptable. + let offered_commit_operation = 12; + let committed_purge_generation = 0; + assert!( + partition + .install_state_transfer( + &repair_config(), + offered_commit_operation, + Vec::new(), + &[], + committed_purge_generation, + ) + .await + .is_err() + ); + partition + .complete_poll(pending_result) + .expect("unmodified history remains valid"); + let (last_polled, committed) = partition.group_offset_state(group_id as u64); + assert_eq!( + last_polled, + Some(4), + "the read still belongs to the served history" + ); + assert_eq!(committed, None, "automatic commits are disabled"); + } + + #[compio::test] + async fn given_full_prepare_queue_when_poll_completes_should_queue_its_context_and_advance() { + let primary_replica = 0; + let replica_count = 3; + let prepare_capacity = 1; + let request_capacity = 1; + let pipeline = LocalPipeline::with_capacities(prepare_capacity, request_capacity); + let (mut partition, _) = + recording_partition_with_pipeline(primary_replica, replica_count, pipeline); + let client_id = 42; + let explicit_consumer_id = 7; + + // One visible message makes offset 0 valid. Its explicit store occupies + // the only prepare slot while the request queue remains available. + partition.stats.increment_messages_count(1); + partition + .on_request( + store_offset_request( + client_id, + 1, + ConsumerKind::Consumer, + explicit_consumer_id, + 0, + AckLevel::Quorum, + ), + None, + ) + .await; + let group_id = 8; + let member_id = 1; + let auto_commit = true; + let consumer = PollingConsumer::ConsumerGroup(group_id, member_id); + let read_result = poll_read_result(&partition, consumer, auto_commit, Some(0)); + + // Acceptance can advance local progress after queueing the automatic + // commit, even though no operation or replication work is assigned yet. + let completion = partition + .complete_poll(read_result) + .expect("request queue has room"); + assert!(completion.replication.is_none()); + let (last_polled, committed) = partition.group_offset_state(group_id as u64); + assert_eq!(last_polled, Some(0)); + assert_eq!(committed, Some(0)); + assert_eq!(partition.consensus.sequencer().current_sequence(), 1); + + // The queued request owns both the history identity and the capacity + // reservation. Taking it out of the queue must keep that slot occupied. + let queued_request = partition + .consensus + .pop_queued_request() + .expect("automatic commit queued"); + let queued_context = queued_request + .auto_commit() + .expect("request carries its context"); + assert_eq!(queued_context.history, partition.poll_history); + assert_eq!( + queued_context.reservation.kind(), + ConsumerKind::ConsumerGroup + ); + assert_eq!(queued_context.reservation.consumer_id() as usize, group_id); + assert_eq!( + partition.occupied_consumer_offset_count(ConsumerKind::ConsumerGroup), + 1 + ); + drop(queued_request); + assert_eq!( + partition.occupied_consumer_offset_count(ConsumerKind::ConsumerGroup), + 0 + ); + } + + #[compio::test] + async fn given_full_prepare_and_request_queues_when_poll_completes_should_preserve_progress() { + let primary_replica = 0; + let replica_count = 3; + let prepare_capacity = 1; + let request_capacity = 1; + let pipeline = LocalPipeline::with_capacities(prepare_capacity, request_capacity); + let (mut partition, _) = + recording_partition_with_pipeline(primary_replica, replica_count, pipeline); + let client_id = 42; + let prepared_consumer_id = 7; + let queued_consumer_id = 9; + + // Fill both admission paths: the first explicit store owns the prepare + // slot, and the second waits in the only request slot. + partition.stats.increment_messages_count(1); + partition + .on_request( + store_offset_request( + client_id, + 1, + ConsumerKind::Consumer, + prepared_consumer_id, + 0, + AckLevel::Quorum, + ), + None, + ) + .await; + partition + .consensus + .push_queued_request(consensus::RequestEntry::new(store_offset_request( + client_id, + 2, + ConsumerKind::Consumer, + queued_consumer_id, + 0, + AckLevel::Quorum, + ))) + .unwrap(); + let group_id = 8; + let member_id = 1; + let auto_commit = true; + let consumer = PollingConsumer::ConsumerGroup(group_id, member_id); + let read_result = poll_read_result(&partition, consumer, auto_commit, Some(0)); + + // The group cannot queue its automatic commit. Rejection must leave its + // progress empty, release its reservation, and retain the existing work. + assert!(matches!( + partition.complete_poll(read_result), + Err(IggyError::TransientNotAccepted) + )); + let (last_polled, committed) = partition.group_offset_state(group_id as u64); + assert_eq!(last_polled, None); + assert_eq!(committed, None); + assert_eq!( + partition.occupied_consumer_offset_count(ConsumerKind::ConsumerGroup), + 0 + ); + assert_eq!(partition.consensus.request_queue_len(), 1); + assert_eq!(partition.consensus.sequencer().current_sequence(), 1); + } + + #[test] + fn given_full_live_map_when_polling_existing_key_should_reclaim_only_for_new_keys() { + let (mut partition, _) = recording_partition(); + partition.set_consumer_offsets_max(2); + partition.offset_space.committed_seeded = true; + partition.seed_recovered_consumer_offset(ConsumerKind::Consumer, 7, 0, 0); + for id in 7..=8 { + partition.consumer_offsets.pin().insert( + id as usize, + ConsumerOffset::new(ConsumerKind::Consumer, id, 0, String::new()), + ); + } + let args = PollingArgs::new(iggy_common::PollingStrategy::first(), 1, true); + let _ = partition.build_poll_plan(PollingConsumer::Consumer(7, 0), &args, false); + assert_eq!(partition.consumer_offsets.len(), 2); + let _ = partition.build_poll_plan(PollingConsumer::Consumer(9, 0), &args, false); + assert_eq!(partition.consumer_offsets.len(), 1); + assert!(partition.consumer_offsets.pin().contains_key(&7)); + } + + #[compio::test] async fn given_missing_retained_header_when_journal_progresses_should_retry_without_view_change() { let mut partition = test_partition(); @@ -11874,18 +12839,8 @@ mod tests { assert_ne!(partition.offset_reservations_scan_state, failed_scan); assert!(partition.consumer_offset_capacity.is_uncertain()); partition.seed_recovered_consumer_offset(ConsumerKind::Consumer, 7, 0, 0); - let existing = partition - .auto_commit_ctx(PollingConsumer::Consumer(7, 0), true) - .unwrap() - .apply(0) - .unwrap(); - assert!(partition.auto_commit_admission_ready(&existing)); - let new = partition - .auto_commit_ctx(PollingConsumer::Consumer(8, 0), true) - .unwrap() - .apply(0) - .unwrap(); - assert!(!partition.auto_commit_admission_ready(&new)); + assert!(partition.auto_commit_admission_ready(ConsumerKind::Consumer, 7)); + assert!(!partition.auto_commit_admission_ready(ConsumerKind::Consumer, 8)); partition.log.journal().inner.clear_all(); for op in 1..=3 { journal_prepare(&partition, op, Operation::SendMessages).await; @@ -11939,11 +12894,7 @@ mod tests { partition.set_consumer_offsets_max(2); partition.offset_space.committed_seeded = true; for id in 1..=2 { - partition - .auto_commit_ctx(PollingConsumer::Consumer(id, 0), true) - .unwrap() - .apply(0) - .unwrap(); + partition.apply_local_poll_offset(ConsumerKind::Consumer, id, 0); } let args = PollingArgs::new(iggy_common::PollingStrategy::first(), 1, true); let _ = partition.build_poll_plan(PollingConsumer::Consumer(3, 0), &args, false); @@ -11952,11 +12903,10 @@ mod tests { assert!(retained_id == 1 || retained_id == 2); assert!( partition - .auto_commit_ctx(PollingConsumer::Consumer(3, 0), true) - .unwrap() - .apply(0) + .check_local_poll_key(ConsumerKind::Consumer, 3) .is_ok() ); + partition.apply_local_poll_offset(ConsumerKind::Consumer, 3, 0); assert!(partition.consumer_offsets.pin().contains_key(&retained_id)); assert_eq!(partition.consumer_offsets.len(), 2); assert!(!partition.consensus.is_primary()); @@ -15202,6 +16152,438 @@ mod retention_tests { } } +#[cfg(test)] +mod purge_poll_tests { + //! A disk poll starts reading messages at offsets 0-2. + //! Before it completes, purge removes those messages and clears consumer + //! progress. Fresh messages are then appended starting at offset 0. + //! Completing the old poll must not advance progress over the fresh messages + //! or restore a group's last polled mark when automatic commits are disabled. + + use super::tests::{journal_send_batch, repair_config, test_partition}; + use super::*; + use crate::PollFragments; + use iggy_common::PollingStrategy; + use server_common::send_messages::decode_batch_slice; + + #[compio::test] + async fn given_pending_disk_poll_when_purged_should_preserve_fresh_consumer_progress() { + let auto_commit = true; + let fresh_message_count = 2; + // The old offset 2 lies beyond the replacement history's offsets 0-1. + Box::pin(assert_delayed_poll_preserves_fresh_progress( + auto_commit, + fresh_message_count, + )) + .await; + } + + #[compio::test] + async fn given_pending_disk_poll_when_fresh_history_covers_old_offset_should_preserve_progress() + { + // The delayed poll targets old offsets 0-2. After purge, five fresh + // messages occupy offsets 0-4. Recording progress 2 from the old poll + // would make `Next` skip fresh offsets 0-2 even though 2 is in range. + let auto_commit = true; + let fresh_message_count = 5; + Box::pin(assert_delayed_poll_preserves_fresh_progress( + auto_commit, + fresh_message_count, + )) + .await; + } + + #[compio::test] + async fn given_disk_poll_without_auto_commit_when_purged_should_preserve_fresh_progress() { + // An individual consumer does not record progress without automatic + // commits. Here the regression assertion is rejection of the old result; + // the group case below also detects an unwanted last_polled update. + let auto_commit = false; + for fresh_message_count in [2, 5] { + Box::pin(assert_delayed_poll_preserves_fresh_progress( + auto_commit, + fresh_message_count, + )) + .await; + } + } + + /// Accepted group polls with messages update `last_polled` even with + /// automatic commits disabled. Purge clears that progress, so a read started + /// before purge must not restore it when its result arrives afterward. + #[compio::test] + async fn given_pending_group_poll_without_auto_commit_when_purged_should_not_restore_progress() + { + Box::pin(assert_delayed_group_poll_preserves_progress()).await; + } + + async fn assert_delayed_group_poll_preserves_progress() { + let group_id = 7; + let member_id = 1; + let auto_commit = false; + let validate_checksum = true; + let consumer = PollingConsumer::ConsumerGroup(group_id, member_id); + + // 1. Put three messages on disk and establish the group's old progress. + // The save threshold of one makes commit_journal write these messages + // to disk. + let config = repair_config(); + let (_directory, mut partition) = Box::pin(disk_poll_partition(&config)).await; + for operation_number in 1..=3 { + journal_send_batch(&mut partition, operation_number).await; + } + partition.consensus().advance_commit_max(3); + partition.commit_journal(&config).await; + + // These accepted reads cache the file descriptor and record progress + // through offset 2, without committing a consumer offset. + accept_initial_disk_polls(&mut partition, consumer).await; + let (last_polled, committed) = partition.group_offset_state(group_id as u64); + assert_eq!(last_polled, Some(2)); + assert_eq!(committed, None, "automatic commits are disabled"); + + // 2. Start another read against the old file, leaving its future pending. + // Using an unsealed segment with a cached descriptor makes the first + // suspension the message read. The read keeps the old file open even + // after purge removes it from the partition's directory. + assert_eq!(partition.log.segments().len(), 1); + assert!(!partition.log.active_segment().sealed); + let old_segment_read_state = Rc::clone(&partition.log.sealed_read_state()[0]); + assert!(old_segment_read_state.fd.borrow().is_some()); + + let old_poll_plan = partition.build_poll_plan( + consumer, + &PollingArgs::new(PollingStrategy::offset(0), 3, auto_commit), + validate_checksum, + ); + assert!(old_poll_plan.needs_off_pump_io()); + let mut old_disk_read = std::pin::pin!(old_poll_plan.execute()); + assert!( + futures::poll!(old_disk_read.as_mut()).is_pending(), + "the group disk poll must suspend before purge", + ); + + // 3. Purge removes the messages and clears both kinds of group progress. + let purge_generation = 1; + partition + .purge(&config, purge_generation) + .await + .expect("purge partition"); + assert_eq!(partition.applied_purge_generation(), purge_generation); + let (last_polled, committed) = partition.group_offset_state(group_id as u64); + assert_eq!(last_polled, None, "purge must clear group progress"); + assert_eq!(committed, None); + assert!(old_segment_read_state.fd.borrow().is_none()); + + // 4. Write five replacement messages while leaving the old read unpolled. + // Consensus operation numbers continue at 4, but message offsets restart + // at 0. The old offset 2 is now in range again, so a bounds check alone + // cannot tell that the old read belongs to the deleted history. + for operation_number in 4..=8 { + journal_send_batch(&mut partition, operation_number).await; + } + partition.consensus().advance_commit_max(8); + partition.commit_journal(&config).await; + assert_eq!(partition.offsets().commit_offset, 4); + + // 5. The old read returns real messages, but accepting its result must + // fail without restoring the group's progress over the new history. + let old_result = old_disk_read.await; + assert_eq!(polled_offsets(&old_result.fragments), [0, 1, 2]); + assert_eq!(old_result.last_matching_offset, Some(2)); + let old_completion = partition.complete_poll(old_result); + let (last_polled, committed) = partition.group_offset_state(group_id as u64); + assert_eq!( + last_polled, None, + "the old read must not restore group progress after purge", + ); + assert_eq!(committed, None, "the old read must not commit an offset"); + assert!(matches!( + old_completion, + Err(IggyError::TransientNotAccepted) + )); + + // 6. A new poll reads all replacement messages and records last_polled. + // The committed offset stays unset because automatic commits are disabled. + let fresh_poll_plan = partition.build_poll_plan( + consumer, + &PollingArgs::new(PollingStrategy::next(), 5, auto_commit), + validate_checksum, + ); + assert!(fresh_poll_plan.needs_off_pump_io()); + let fresh_result = fresh_poll_plan.execute().await; + let fresh_completion = partition + .complete_poll(fresh_result) + .expect("accept fresh group read"); + assert_eq!(polled_offsets(&fresh_completion.fragments), [0, 1, 2, 3, 4]); + assert!(fresh_completion.replication.is_none()); + let (last_polled, committed) = partition.group_offset_state(group_id as u64); + assert_eq!(last_polled, Some(4), "the fresh poll must record progress"); + assert_eq!(committed, None, "automatic commits are disabled"); + } + + // Keep the pause, purge, and late acceptance in one visible sequence. + #[allow(clippy::too_many_lines)] + async fn assert_delayed_poll_preserves_fresh_progress( + auto_commit: bool, + fresh_message_count: u32, + ) { + let consumer_id = 7; + let partition_id = 0; + let validate_checksum = true; + let consumer = PollingConsumer::Consumer(consumer_id, partition_id); + + // 1. Write three messages and evict their journal data so polls must + // read the segment file. The save threshold is one message. + let config = repair_config(); + let (_directory, mut partition) = Box::pin(disk_poll_partition(&config)).await; + for operation_number in 1..=3 { + journal_send_batch(&mut partition, operation_number).await; + } + partition.consensus().advance_commit_max(3); + partition.commit_journal(&config).await; + assert_eq!(partition.offsets().commit_offset, 2); + assert_eq!(partition.log.segments().len(), 1); + assert!(!partition.log.active_segment().sealed); + assert!(partition.log.active_segment().size.as_bytes_u64() > 0); + assert!( + partition + .log + .journal() + .inner + .oldest_resident_offset() + .is_none() + ); + + // 2. Cache the old file descriptor, then start a read against that file. + // The active segment needs no index I/O, so the first suspension holds + // a file clone in the message read even if purge later unlinks the file. + accept_initial_disk_polls(&mut partition, consumer).await; + let old_segment_read_state = Rc::clone(&partition.log.sealed_read_state()[0]); + assert!(old_segment_read_state.fd.borrow().is_some()); + + let old_poll_plan = partition.build_poll_plan( + consumer, + &PollingArgs::new(PollingStrategy::offset(0), 3, auto_commit), + validate_checksum, + ); + assert!(old_poll_plan.needs_off_pump_io()); + let mut old_disk_read = std::pin::pin!(old_poll_plan.execute()); + assert!( + futures::poll!(old_disk_read.as_mut()).is_pending(), + "the disk poll must suspend before purge", + ); + + if auto_commit { + // Model another poll's queued commit from the old history. Purge + // must remove that request and release its reserved capacity too. + queue_old_auto_commit(&partition); + } + + // 3. Purge clears progress, queued commits, and the cached descriptor. + // Leave the read future unpolled until purge and the fresh append finish. + // The underlying I/O may finish, but its result has not been accepted. + let purge_generation = 1; + partition + .purge(&config, purge_generation) + .await + .expect("purge partition"); + assert_eq!(partition.applied_purge_generation(), purge_generation); + assert_eq!(partition.consensus.request_queue_len(), 0); + assert_eq!( + partition.occupied_consumer_offset_count(ConsumerKind::Consumer), + 0 + ); + assert_eq!(partition.get_consumer_offset(consumer), None); + assert!(old_segment_read_state.fd.borrow().is_none()); + assert_eq!(partition.log.active_segment().size.as_bytes_u64(), 0); + + // 4. Replace the deleted messages. Consensus operation numbers continue + // through purge, while message offsets restart at zero in the new file. + let last_fresh_operation = 3 + u64::from(fresh_message_count); + for operation_number in 4..=last_fresh_operation { + journal_send_batch(&mut partition, operation_number).await; + } + partition + .consensus() + .advance_commit_max(last_fresh_operation); + partition.commit_journal(&config).await; + assert_eq!(partition.consensus().commit_min(), last_fresh_operation); + assert_eq!( + partition.offsets().commit_offset, + u64::from(fresh_message_count) - 1 + ); + + // 5. Reject the result from the deleted history before it can advance + // this consumer's cursor over replacement messages. + let old_result = old_disk_read.await; + let old_offsets = polled_offsets(&old_result.fragments); + assert!(matches!( + partition.complete_poll(old_result), + Err(IggyError::TransientNotAccepted) + )); + assert_eq!( + partition.get_consumer_offset(consumer), + None, + "the rejected old result must not restore the consumer cursor" + ); + + // 6. Both an explicit offset and Next must read the full new history. + // Disable commits for these checks so the first read does not advance + // the cursor and change the starting point of the second read. + let fresh_auto_commit = false; + let fresh_poll_plan = partition.build_poll_plan( + consumer, + &PollingArgs::new( + PollingStrategy::offset(0), + fresh_message_count, + fresh_auto_commit, + ), + validate_checksum, + ); + assert!(fresh_poll_plan.needs_off_pump_io()); + let fresh_result = fresh_poll_plan.execute().await; + let fresh_completion = partition + .complete_poll(fresh_result) + .expect("accept fresh read"); + let next_result = partition + .build_poll_plan( + consumer, + &PollingArgs::new( + PollingStrategy::next(), + fresh_message_count, + fresh_auto_commit, + ), + validate_checksum, + ) + .execute() + .await; + let next_completion = partition + .complete_poll(next_result) + .expect("accept fresh Next read"); + let expected_fresh_offsets = (0..u64::from(fresh_message_count)).collect::>(); + assert_eq!( + polled_offsets(&fresh_completion.fragments), + expected_fresh_offsets + ); + assert_eq!( + polled_offsets(&next_completion.fragments), + expected_fresh_offsets, + "old offsets={old_offsets:?}, auto_commit={auto_commit}" + ); + } + + /// Queue a commit for consumer 7 at offset 2 with the current history and a + /// capacity reservation. Purge must discard both the request and its guard. + fn queue_old_auto_commit(partition: &IggyPartition) { + let consumer_id = 7; + let last_polled_offset = 2; + let reservation = partition + .consumer_offset_capacity + .reserve_provisional(consumer_id, &partition.durable_consumer_offsets) + .unwrap(); + let request = partition + .build_poll_auto_commit_request(ConsumerKind::Consumer, consumer_id, last_polled_offset) + .unwrap(); + partition + .consensus + .push_queued_request(consensus::RequestEntry::with_auto_commit( + request, + AutoCommitRequestContext { + history: partition.poll_history, + reservation, + }, + )) + .unwrap(); + assert_eq!( + partition.occupied_consumer_offset_count(ConsumerKind::Consumer), + 1 + ); + } + + /// Read offsets 0 and then 0-2 without automatic commits, caching the segment + /// descriptor. For a group, accepting these reads also records `last_polled`. + async fn accept_initial_disk_polls( + partition: &mut IggyPartition, + consumer: PollingConsumer, + ) { + let auto_commit = false; + let validate_checksum = true; + let warm_poll_plan = partition.build_poll_plan( + consumer, + &PollingArgs::new(PollingStrategy::offset(0), 1, auto_commit), + validate_checksum, + ); + assert!(warm_poll_plan.needs_off_pump_io()); + let warm_result = warm_poll_plan.execute().await; + let warm_completion = partition + .complete_poll(warm_result) + .expect("accept warm read"); + assert_eq!(polled_offsets(&warm_completion.fragments), [0]); + assert!(warm_completion.replication.is_none()); + assert_eq!(partition.get_consumer_offset(consumer), None); + let initial_result = partition + .build_poll_plan( + consumer, + &PollingArgs::new(PollingStrategy::offset(0), 3, auto_commit), + validate_checksum, + ) + .execute() + .await; + let initial_completion = partition + .complete_poll(initial_result) + .expect("accept old read"); + assert_eq!(polled_offsets(&initial_completion.fragments), [0, 1, 2]); + assert!(initial_completion.replication.is_none()); + } + + /// Replace the fixture's initial segment with real files and offset stores. + /// Keep the returned directory alive until every read using those files ends. + async fn disk_poll_partition( + config: &PartitionsConfig, + ) -> (tempfile::TempDir, IggyPartition) { + let directory = tempfile::tempdir().expect("create partition directory"); + let mut partition = test_partition(); + partition.set_partition_dir(directory.path().to_string_lossy().into_owned()); + partition.log.retire_front().expect("retire empty segment"); + partition + .install_empty_segment(config, 0) + .await + .expect("install segment with real writers"); + + let consumer_path = directory.path().join("consumer_offsets"); + let group_path = directory.path().join("consumer_group_offsets"); + compio::fs::create_dir_all(&consumer_path) + .await + .expect("create consumer offsets directory"); + compio::fs::create_dir_all(&group_path) + .await + .expect("create group offsets directory"); + partition.configure_consumer_offset_storage( + consumer_path.to_string_lossy().into_owned(), + group_path.to_string_lossy().into_owned(), + ConsumerOffsets::with_capacity(1), + ConsumerGroupOffsets::with_capacity(1), + ); + (directory, partition) + } + + fn polled_offsets(fragments: &PollFragments) -> Vec { + fragments + .iter() + .map(|fragment| { + let batch = decode_batch_slice(fragment.as_slice()).expect("decode polled batch"); + assert_eq!( + batch.message_count(), + 1, + "fixture sends one message per batch" + ); + batch.header.base_offset + }) + .collect() + } +} + #[cfg(test)] mod review_4092_tests { use super::tests::{checksummed_segment_prepare, recording_partition_at}; diff --git a/core/partitions/src/iggy_partitions.rs b/core/partitions/src/iggy_partitions.rs index 96dffd4424..3a68203a86 100644 --- a/core/partitions/src/iggy_partitions.rs +++ b/core/partitions/src/iggy_partitions.rs @@ -17,9 +17,10 @@ #![allow(dead_code)] -use crate::poll_plan::PollPlan; +use crate::poll_plan::{PollPlan, PollReadResult}; use crate::types::PartitionsConfig; use crate::{IggyPartition, Partition, PollingArgs, PollingConsumer}; +use crate::{PollCompletion, PollReplication}; use ahash::AHashSet; use consensus::{ Consensus, Plane, PlaneIdentity, VsrConsensus, build_deny_reply_from_request_header, @@ -469,17 +470,9 @@ where self.tombstoned.borrow_mut().remove(namespace); } - /// Build an owned [`PollPlan`] for a partition poll synchronously, under a - /// single pump-only `&mut` borrow (the in-memory journal tier + the - /// resident-tail straddle snapshot are read here, and the sealed-read-handle - /// LRU is touched; mem reads never yield). Returns `None` for a missing or - /// tombstoned namespace. - /// - /// Pairs with [`PollPlan::execute`], which runs the disk read + - /// offset persist/apply off the borrow on the owned plan. Splitting the - /// borrow-bound plan build from the borrow-free execution is what keeps - /// poll-read sound: the only partition reference is taken here, on the pump, - /// sequential with the pump's own `&mut` mutations, never on a sibling task. + /// Snapshot read resources under a synchronous borrow on the owning pump. + /// Returns `None` for a missing or tombstoned namespace. Execution carries + /// no partition borrow; its result returns to [`Self::complete_poll`]. pub fn build_poll_snapshot( &self, namespace: &IggyNamespace, @@ -495,6 +488,55 @@ where Some(partition.build_poll_plan(consumer, args, validate_checksum)) } + /// Validate and accept a poll synchronously on the owning pump. + /// Attempt the reply synchronously, then immediately await any returned + /// continuation through [`Self::replicate_poll_completion`] on the same pump. + /// Do not suspend between acceptance and that call or discard the continuation. + /// Success does not acknowledge a durable offset commit. + /// + /// Acceptance is independent of reply delivery. A caller that stopped + /// waiting does not cancel completion: an accepted read can still advance + /// progress and admit an automatic commit even if its reply cannot be sent. + /// A history rejection accepts no progress from this read, but cannot + /// replace a response the transport already sent. + /// + /// # Errors + /// Rejects stale history, unavailable partition or admission state, invalid + /// consumer identifiers, and exhausted consumer, queue, or journal capacity. + /// Rejection does not accept progress from this read. + pub fn complete_poll( + &self, + namespace: &IggyNamespace, + result: PollReadResult, + ) -> Result { + let partition = self + .get_mut_by_ns(namespace) + .ok_or(IggyError::TransientNotAccepted)?; + partition.complete_poll(result) + } + + /// Drive an accepted poll's replication on the owning pump after its reply. + /// Use the continuation returned by [`Self::complete_poll`] for this namespace. + /// Begin this call immediately after the synchronous reply attempt, with no + /// intervening suspension. Dropping the assigned prepare would leave an + /// unjournaled operation in the pipeline and prevent later commits. + /// This call may suspend with a partition borrow, so it must remain on the + /// owning pump. + /// + /// # Panics + /// Panics if the namespace is missing or tombstoned. Its availability cannot + /// change between acceptance and this call in the same synchronous owner turn. + pub async fn replicate_poll_completion( + &self, + namespace: &IggyNamespace, + replication: PollReplication, + ) { + let partition = self + .get_mut_by_ns(namespace) + .expect("IggyPartitions invariant: accepted poll namespace missing or tombstoned"); + partition.replicate_poll_completion(replication).await; + } + /// Read a consumer's stored offset + the partition commit offset. Fully /// synchronous (atomics + lock-free maps), so it runs under a single /// [`Self::with_partition`] borrow. `None` for a missing/tombstoned @@ -665,31 +707,6 @@ where let _ = waiter.send(build_deny_reply_from_request_header(header, status)); } } - - pub async fn on_auto_commit_request( - &self, - request: Message, - reservation: crate::AutoCommitReservation, - ) { - let namespace = IggyNamespace::from_raw(request.header().group); - if self.is_tombstoned(&namespace) { - tracing::debug!( - namespace_raw = namespace.inner(), - "dropping auto-commit for tombstoned partition" - ); - return; - } - if let Some(partition) = self.get_mut_by_ns(&namespace) { - partition - .on_request_with_reservation(request, None, Some(reservation)) - .await; - } else { - tracing::debug!( - namespace_raw = namespace.inner(), - "dropping auto-commit for missing partition" - ); - } - } } impl Plane> for IggyPartitions @@ -778,14 +795,17 @@ mod tests { use bytes::Bytes; use consensus::LocalPipeline; use iggy_binary_protocol::Operation; - use iggy_common::{IggyByteSize, PartitionStats}; + use iggy_common::{IggyByteSize, PartitionStats, PollingStrategy}; use journal::Journal as _; use message_bus::IggyMessageBus; use server_common::send_messages::{ IggyMessage, IggyMessageHeader, IggyMessages, PREPARE_SPLIT_POINT, SendMessagesOwned, stamp_prepare_for_persistence, }; - use server_common::{Message, iobuf::Frozen}; + use server_common::{ + Message, + iobuf::{Frozen, Owned}, + }; use std::sync::Arc; const TEST_CLUSTER: u128 = 1; @@ -1082,4 +1102,58 @@ mod tests { "no committed message at or after offset 3, so the poll is empty", ); } + + #[compio::test] + #[should_panic(expected = "accepted poll namespace missing or tombstoned")] + async fn given_accepted_poll_when_namespace_is_tombstoned_should_trip_invariant() { + let namespace = IggyNamespace::new(1, 1, 0); + let config = PartitionsConfig { + messages_required_to_save: 1, + size_of_messages_required_to_save: IggyByteSize::from(1024 * 1024), + validate_checksum: false, + segment_size: IggyByteSize::from(1024 * 1024), + preallocate_segments: false, + encryptor: None, + path_layout: crate::PartitionPathLayout::default(), + }; + let partitions = IggyPartitions::new(ShardId::new(0), config); + partitions.insert(namespace, build_partition()); + let consumer = PollingConsumer::Consumer(7, 0); + let auto_commit = true; + let mut read_result = partitions + .build_poll_snapshot( + &namespace, + consumer, + &PollingArgs::new(PollingStrategy::next(), 1, auto_commit), + ) + .expect("snapshot the current history") + .execute_resident(); + + // Supply a nonempty result with the captured history so acceptance + // assigns an automatic commit, without needing message I/O for this test. + read_result + .fragments + .push(crate::Fragment::whole(Owned::<4096>::zeroed(8).into())); + read_result.last_matching_offset = Some(0); + let replication = partitions + .complete_poll(&namespace, read_result) + .expect("accept automatic commit") + .replication + .expect("assign a prepare"); + let partition = partitions.get_by_ns(&namespace).expect("partition exists"); + let prepare_header = partition + .consensus() + .pipeline_head_header() + .expect("acceptance queued the prepare"); + assert_eq!(prepare_header.operation, Operation::StoreConsumerOffset); + assert!(!partition.log.journal().inner.holds_op(prepare_header.op)); + + // A future caller that yields after acceptance could permit this + // tombstone. The partition still exists and must not lose its prepare. + partitions.tombstone(namespace); + assert!(partitions.local_idx(&namespace).is_some()); + partitions + .replicate_poll_completion(&namespace, replication) + .await; + } } diff --git a/core/partitions/src/lib.rs b/core/partitions/src/lib.rs index 71cc995e98..200892f8ba 100644 --- a/core/partitions/src/lib.rs +++ b/core/partitions/src/lib.rs @@ -64,9 +64,10 @@ pub const DEFAULT_OFFSET_RESERVATION_LEASE: u32 = 64 * 1024; /// Shipped per-kind durable consumer-offset limit for one partition. pub const DEFAULT_CONSUMER_OFFSETS_MAX: usize = 4096; +pub use iggy_partition::{PollCompletion, PollReplication}; pub use messages_writer::MessagesWriter; pub use offset_storage::delete_persisted_offset; -pub use poll_plan::{AutoCommitApplied, PollPlan}; +pub use poll_plan::{PollPlan, PollReadResult}; pub use segment::Segment; use server_common::Message; pub use server_common::send_messages::{IggyMessage, IggyMessageHeader, IggyMessages}; diff --git a/core/partitions/src/poll_plan.rs b/core/partitions/src/poll_plan.rs index 6b8a01e009..74943432eb 100644 --- a/core/partitions/src/poll_plan.rs +++ b/core/partitions/src/poll_plan.rs @@ -15,42 +15,25 @@ // specific language governing permissions and limitations // under the License. -//! Owned, borrow-free poll execution. +//! Owned poll reads without authority to change consumer progress. //! -//! A poll must not hold a partition reference across an `.await`: the shard pump -//! can reallocate the partitions `Vec` (`ReconcileOp::InsertOwned`) or take a -//! `&mut` to the same namespace while a poll is parked, dangling the reference. -//! So `IggyPartition::build_poll_plan` captures everything a poll needs -//! synchronously under the borrow into the owned types here, drops the borrow, -//! then [`PollPlan::execute`] runs the disk read + the in-memory auto-commit -//! apply on owned data alone: consumer offsets are already `Arc`, the journal -//! tail is a point-in-time `Frozen` snapshot, and each sealed segment carries a -//! shared [`SealedSegmentReadState`] handle (a plain `Rc`, not a partition -//! reference) whose read fd + sparse index the read reuses or fills on a miss. -//! No value in this module holds a partition reference, so executing a plan is -//! sound on a detached task concurrently with the pump's own writes. +//! The pump snapshots file handles and resident fragments before a read can +//! yield. Execution returns facts about that snapshot. Only the partition +//! owner validates its history and accepts progress. -use crate::PollFragments; -use crate::consumer_offset_capacity::{ - AutoCommitReservation, ConsumerOffsetCapacity, ConsumerOffsetCapacityError, - DurableConsumerOffsets, -}; use crate::iggy_index::{IGGY_INDEX_SIZE, IggyIndexCache}; use crate::iggy_index_reader::IggyIndexReader; use crate::journal::{ MessageLookup, push_selected_batch_fragments, select_batch_slice, unpin_sparse_source, }; +use crate::{PollFragments, PollingConsumer}; use compio::io::AsyncReadAtExt; -use iggy_common::{ - ConsumerGroupId, ConsumerGroupOffsets, ConsumerKind, ConsumerOffset, ConsumerOffsets, IggyError, -}; +use iggy_common::{ConsumerKind, IggyError}; use server_common::iobuf::{Frozen, Owned}; +use server_common::poll::PollHistoryId; use server_common::send_messages::{BatchIntegrity, COMMAND_HEADER_SIZE, decode_batch_slice_with}; use std::cell::{Cell, RefCell}; -use std::hash::Hash; use std::rc::Rc; -use std::sync::Arc; -use std::sync::atomic::Ordering; use tracing::{error, warn}; /// Byte cap for materializing a sealed segment's sparse index into its shared @@ -170,76 +153,37 @@ pub struct DiskSegment { pub(crate) sealed: bool, } -/// Owned auto-commit input, applied off the partition borrow after a poll (see -/// module docs). Only the in-memory apply happens here; durability is the -/// replicated [`crate::iggy_partition::IggyPartition::apply_staged_consumer_offset_commit`] -/// path's job on every node, driven by the `StoreConsumerOffset` op the serving -/// shard submits from [`AutoCommitApplied`]. A poll-local disk write would be -/// node-local only and diverge on failover. -pub struct AutoCommitCtx { - pub(crate) target: AutoCommitTarget, - pub(crate) capacity: Rc, - pub(crate) durable: Rc, -} - -/// The offset an `auto_commit` poll applied in memory, surfaced for replication. -/// -/// The serving shard replicates it through the partition consensus (the only -/// cross-node durable path); `kind` + `consumer_id` are the offset key the -/// submitted `StoreConsumerOffset` op must carry. -pub struct AutoCommitApplied { - pub kind: ConsumerKind, - pub consumer_id: u32, - pub offset: u64, - previous_offset: Option, - target: AutoCommitTarget, - capacity: Rc, - durable: Rc, - last_polled: Option, +/// Admission inputs carried by a read without access to consumer progress maps. +#[derive(Debug)] +pub struct PollContext { + /// History captured at planning, which must still match at completion. + pub(crate) history: PollHistoryId, + /// Consumer or group whose progress the owner may update after admission. + pub(crate) consumer: PollingConsumer, + /// For nonempty results, whether to advance the stored offset locally. + /// Group `last_polled` progress also advances when this is false. + pub(crate) auto_commit: bool, } -/// The lock-free offset map this auto-commit updates, captured as an owned -/// `Arc` so the apply needs no partition borrow. `create_path` builds the -/// `ConsumerOffset` entry on first commit for a consumer that has none yet. -pub enum AutoCommitTarget { - Consumer { - offsets: Arc, - consumer_id: u32, - create_path: Option, - }, - ConsumerGroup { - offsets: Arc, - group_id: u32, - create_path: Option, - }, -} - -/// Owned cooperative-rebalance input: a group's lock-free `last_polled` map -/// (captured as an `Arc`) plus its id, so the highest offset served to the group -/// is recorded off the partition borrow after the poll completes (the served -/// offset is unknown until then). See [`PollPlan::execute`]. -pub struct LastPolledCtx { - pub(crate) offsets: Arc, - pub(crate) group_id: usize, +/// An owned read result awaiting validation by the partition owner. +/// Finishing I/O does not authorize a successful reply or a progress update. +#[derive(Debug)] +pub struct PollReadResult { + pub(crate) context: PollContext, + /// Selected message bytes, which remain unaccepted until owner validation. + pub(crate) fragments: PollFragments, + /// Partition message frontier captured at planning, not consumer progress. + pub(crate) commit_offset: u64, + /// Inclusive offset of the last selected message, or `None` for no match. + pub(crate) last_matching_offset: Option, } -impl LastPolledCtx { - /// Bump the group's recorded high-water served offset (monotone via - /// `fetch_max`). Lock-free `papaya` on an owned `Arc`, so sound off the pump. - #[allow(clippy::cast_possible_truncation)] - fn record(&self, last_offset: u64) { - let guard = self.offsets.pin(); - let key = ConsumerGroupId(self.group_id); - if let Some(existing) = guard.get(&key) { - existing.offset.fetch_max(last_offset, Ordering::Relaxed); - } else { - let created = ConsumerOffset::new( - ConsumerKind::ConsumerGroup, - u32::try_from(self.group_id).unwrap_or(u32::MAX), - last_offset, - String::new(), - ); - guard.insert(key, created); +impl PollReadResult { + #[must_use] + pub const fn consumer_kind(&self) -> ConsumerKind { + match self.context.consumer { + PollingConsumer::Consumer(..) => ConsumerKind::Consumer, + PollingConsumer::ConsumerGroup(..) => ConsumerKind::ConsumerGroup, } } } @@ -275,49 +219,29 @@ impl ResidentTailSnapshot { } } -/// Everything a poll needs, captured by `IggyPartition::build_poll_plan` (see -/// module docs for the borrow contract). +/// Owned read snapshot that may outlive the partition history it captured. +/// +/// Execution yields a [`PollReadResult`] that the owner must accept through +/// [`crate::IggyPartitions::complete_poll`] before replying or updating progress. pub struct PollPlan { /// Monotone high-water snapshot taken before the disk read, so it may lag a /// concurrent producer by the poll duration and self-corrects next poll. pub(crate) commit_offset: u64, - pub(crate) auto_commit: Option, - pub(crate) last_polled: Option, + pub(crate) context: PollContext, pub(crate) tier: PollTier, } impl PollPlan { - /// Whether executing this plan needs off-pump IO: only a `Disk` tier read. - /// When `false` the result is fully resident and the caller runs - /// [`Self::execute_resident`] + replies on the pump; when `true` it must - /// spawn [`Self::execute`] so the pump is not blocked on file IO. - /// - /// Auto-commit no longer forces a detached task: its in-memory apply is - /// synchronous and its durability rides consensus off the serving shard - /// (no poll-local disk write), so a fully-resident `auto_commit` poll still - /// replies inline. + /// Whether this snapshot needs disk I/O on a detached read task. + /// Resident reads execute and complete synchronously on the owner. #[must_use] pub const fn needs_off_pump_io(&self) -> bool { matches!(self.tier, PollTier::Disk { .. }) } - /// Execute this plan off the partition borrow: disk read (if any), straddle - /// splice into the owned resident-tail snapshot, then apply the auto-commit - /// to the owned `Arc` offset map. Holds no partition reference (see module - /// docs), so it is safe on a detached task. Returns the served fragments, - /// the poll's high-water offset, and the auto-committed offset (if any) for - /// the serving shard to replicate through consensus. - /// - /// # Errors - /// Returns a capacity error when auto-commit would create a new live-map - /// entry after the configured per-kind bound has been reached. - /// The serving shard also rejects completion with `TransientNotAccepted` - /// if the partition disappeared or changed incarnation during disk I/O. - /// No fragments are returned for that rejected completion. - pub async fn execute( - self, - ) -> Result<(PollFragments<4096>, u64, Option), ConsumerOffsetCapacityError> - { + /// Read the captured snapshot without changing consumer progress. + /// The result still requires owner validation, including when it is empty. + pub async fn execute(self) -> PollReadResult { let commit_offset = self.commit_offset; let (fragments, last_matching_offset) = match self.tier { PollTier::Empty => (PollFragments::new(), None), @@ -382,27 +306,21 @@ impl PollPlan { }, }; - finish( - self.last_polled, - self.auto_commit, + PollReadResult { + context: self.context, commit_offset, fragments, last_matching_offset, - ) + } } - /// Synchronous fast path for a fully-resident poll - /// ([`Self::needs_off_pump_io`] is `false`): no disk read, so the pump - /// applies the auto-commit in memory and replies inline without spawning. - /// The auto-committed offset is returned for the serving shard to replicate. + /// Read a resident snapshot synchronously on the pump. + /// The returned result requires the same owner validation as a disk read. /// - /// # Errors - /// Returns a capacity error when auto-commit would create a new live-map - /// entry after the configured per-kind bound has been reached. - pub fn execute_resident( - self, - ) -> Result<(PollFragments<4096>, u64, Option), ConsumerOffsetCapacityError> - { + /// # Panics + /// Panics if [`Self::needs_off_pump_io`] is true. + #[must_use] + pub fn execute_resident(self) -> PollReadResult { let commit_offset = self.commit_offset; let (fragments, last_matching_offset) = match self.tier { PollTier::Empty => (PollFragments::new(), None), @@ -416,59 +334,13 @@ impl PollPlan { unreachable!("execute_resident on Disk tier; needs_off_pump_io guards this") } }; - finish( - self.last_polled, - self.auto_commit, + PollReadResult { + context: self.context, commit_offset, fragments, last_matching_offset, - ) - } -} - -/// Common tail of [`PollPlan::execute`] and [`PollPlan::execute_resident`], -/// factored out so the high-water record, auto-commit, and returned triple -/// stay identical across both. -fn finish( - last_polled: Option, - auto_commit: Option, - commit_offset: u64, - fragments: PollFragments<4096>, - last_matching_offset: Option, -) -> Result<(PollFragments<4096>, u64, Option), ConsumerOffsetCapacityError> { - let mut auto_commit_applied = apply_auto_commit(auto_commit, &fragments, last_matching_offset)?; - if let Some(applied) = &mut auto_commit_applied { - applied.last_polled = last_polled; - } else if let (Some(last_polled), Some(last_offset)) = (last_polled, last_matching_offset) { - last_polled.record(last_offset); - } - Ok((fragments, commit_offset, auto_commit_applied)) -} - -/// Apply an `auto_commit` to the in-memory offset map (monotone) and surface -/// the committed offset so the serving shard can replicate it through -/// consensus. `None` when the poll served nothing (empty fragments) or no -/// auto-commit was requested. Shared by [`PollPlan::execute`] and -/// [`PollPlan::execute_resident`] so both apply identically. -/// -/// The eager in-memory apply preserves read-your-own-poll for a tight -/// `Consumer::Next` loop that reads before the replicated commit lands; the -/// commit's apply is an idempotent monotone set, so the double-apply converges. -fn apply_auto_commit( - auto_commit: Option, - fragments: &PollFragments<4096>, - last_matching_offset: Option, -) -> Result, ConsumerOffsetCapacityError> { - let Some(auto_commit) = auto_commit else { - return Ok(None); - }; - if fragments.is_empty() { - return Ok(None); + } } - let Some(last_offset) = last_matching_offset else { - return Ok(None); - }; - auto_commit.apply(last_offset).map(Some) } pub enum PollTier { @@ -868,241 +740,6 @@ fn resolve_index_position(index: &IggyIndexCache, query: MessageLookup) -> Optio .map(|entry| entry.position) } -impl AutoCommitCtx { - /// The offset key (kind + numeric id) this auto-commit targets, for the - /// replicated `StoreConsumerOffset` op the serving shard submits. - pub(crate) const fn kind_and_id(&self) -> (ConsumerKind, u32) { - match &self.target { - AutoCommitTarget::Consumer { consumer_id, .. } => { - (ConsumerKind::Consumer, *consumer_id) - } - AutoCommitTarget::ConsumerGroup { group_id, .. } => { - (ConsumerKind::ConsumerGroup, *group_id) - } - } - } - - /// Apply the committed offset to the in-memory map on the owned `Arc` - /// handle, with NO partition reference. Uses the monotone - /// [`upsert_offset_max`] so a stale off-pump auto-commit cannot rewind a - /// newer explicit store; the maps are lock-free (`papaya`), so this is - /// sound off the pump task. - #[allow(clippy::cast_possible_truncation)] - pub(crate) fn apply( - self, - offset: u64, - ) -> Result { - let (kind, consumer_id) = self.kind_and_id(); - let create = |path: Option<&str>| { - ConsumerOffset::new( - kind, - consumer_id, - offset, - path.map_or_else(String::new, |path| format!("{path}/{consumer_id}")), - ) - }; - let previous_offset = match &self.target { - AutoCommitTarget::Consumer { - offsets, - consumer_id, - create_path, - } => apply_local_offset( - offsets, - *consumer_id as usize, - offset, - &self.capacity, - self.durable.count(kind) >= self.capacity.limit(), - || create(create_path.as_deref()), - )?, - AutoCommitTarget::ConsumerGroup { - offsets, - group_id, - create_path, - } => apply_local_offset( - offsets, - ConsumerGroupId(*group_id as usize), - offset, - &self.capacity, - self.durable.count(kind) >= self.capacity.limit(), - || create(create_path.as_deref()), - )?, - }; - Ok(AutoCommitApplied { - kind, - consumer_id, - offset, - previous_offset, - target: self.target, - capacity: self.capacity, - durable: self.durable, - last_polled: None, - }) - } -} - -impl AutoCommitApplied { - /// Record the group handoff frontier only after poll admission succeeds. - fn mark_served(&self) { - if let Some(last_polled) = &self.last_polled { - last_polled.record(self.offset); - } - } - /// Reserve a durable key before the synthetic store is submitted. - /// Returns `None` when committed state already covers this offset. - /// - /// # Errors - /// Returns a capacity error when this is a new durable key and the - /// partition's per-kind limit has been reached. - pub fn reserve_durable( - &self, - ) -> Result, ConsumerOffsetCapacityError> { - if self - .durable - .covers(self.kind, self.consumer_id, self.offset) - { - return Ok(None); - } - self.capacity - .reserve_provisional(self.consumer_id, &self.durable) - .map(Some) - } - - pub(crate) fn belongs_to(&self, durable: &Rc) -> bool { - Rc::ptr_eq(&self.durable, durable) - } - - /// Run the serving shard's synchronous admission and settle this apply in - /// the same call: `Ok` marks the cursor served, `Err` rolls the eager - /// local update back and returns the error. The rollback is a bare store of - /// the previous offset, so nothing may yield between the decision and it. - /// Keeping both inside one synchronous method is what makes that hold for - /// every caller. - /// - /// # Errors - /// Whatever `decide` returned, after the rollback. - pub fn admit(self, decide: impl FnOnce(&Self) -> Result<(), E>) -> Result<(), E> { - match decide(&self) { - Ok(()) => { - self.mark_served(); - Ok(()) - } - Err(error) => { - self.rollback_created(); - Err(error) - } - } - } - - /// Undo this poll's eager update after synchronous admission fails. - fn rollback_created(&self) { - match &self.target { - AutoCommitTarget::Consumer { - offsets, - consumer_id, - .. - } => rollback_local_offset(offsets, *consumer_id as usize, self.previous_offset), - AutoCommitTarget::ConsumerGroup { - offsets, group_id, .. - } => rollback_local_offset( - offsets, - ConsumerGroupId(*group_id as usize), - self.previous_offset, - ), - } - if self.previous_offset.is_none() { - self.capacity.note_local_key_change(); - self.capacity.forget_inactive_provisional(self.consumer_id); - } - } -} - -fn rollback_local_offset( - map: &papaya::HashMap, - key: K, - previous: Option, -) { - let guard = map.pin(); - if let Some(previous) = previous { - if let Some(entry) = guard.get(&key) { - entry.offset.store(previous, Ordering::Relaxed); - } - } else { - guard.remove(&key); - } -} - -fn apply_local_offset( - map: &papaya::HashMap, - key: K, - offset: u64, - capacity: &ConsumerOffsetCapacity, - durable_full: bool, - create: impl FnOnce() -> ConsumerOffset, -) -> Result, ConsumerOffsetCapacityError> { - let guard = map.pin(); - if let Some(existing) = guard.get(&key) { - return Ok(Some(existing.offset.fetch_max(offset, Ordering::Relaxed))); - } - // The `len()` read and the insert are not atomic on this lock-free map. - // The bound holds because every poll of one partition runs on that - // partition's own shard thread, so no second inserter exists. - capacity.admit_local_map_key(guard.len(), durable_full)?; - guard.insert(key, create()); - capacity.note_local_key_change(); - Ok(None) -} - -/// Upsert a committed offset into a lock-free `papaya` offset map: bump an -/// existing entry in place, or build one via `create_on_miss` on first commit -/// for a consumer/group that has none yet. Shared by the pump's -/// [`IggyPartition::apply_consumer_offset_commit`] and the off-pump -/// [`AutoCommitCtx::apply`] so both store offsets identically. -pub fn upsert_offset( - map: &papaya::HashMap, - key: K, - offset: u64, - create_on_miss: impl FnOnce() -> ConsumerOffset, -) where - K: Hash + Eq + Clone + Send + Sync, -{ - let guard = map.pin(); - if let Some(existing) = guard.get(&key) { - existing.offset.store(offset, Ordering::Relaxed); - } else { - let created = create_on_miss(); - created.offset.store(offset, Ordering::Relaxed); - guard.insert(key, created); - } -} - -/// Monotone variant of [`upsert_offset`] for the off-pump auto-commit: an -/// existing entry is bumped via `fetch_max` so a stale auto-commit racing a -/// newer explicit `StoreConsumerOffset` cannot rewind it backward. The -/// on-miss create branch is identical. The explicit pump path keeps -/// [`upsert_offset`] (`store`), since an explicit store may legitimately rewind. -/// -/// Also used by the replicated commit-apply for a server auto-commit op -/// ([`crate::iggy_partition::IggyPartition::apply_consumer_offset_commit`]): its -/// offset was already advanced in memory by the eager poll-path apply, and this -/// commit can land behind a newer poll, so it must not `store` (rewind) it. -pub fn upsert_offset_max( - map: &papaya::HashMap, - key: K, - offset: u64, - create_on_miss: impl FnOnce() -> ConsumerOffset, -) where - K: Hash + Eq + Clone + Send + Sync, -{ - let guard = map.pin(); - if let Some(existing) = guard.get(&key) { - existing.offset.fetch_max(offset, Ordering::Relaxed); - } else { - let created = create_on_miss(); - created.offset.store(offset, Ordering::Relaxed); - guard.insert(key, created); - } -} - /// Walk stamped `[256B BatchHeader][blob]` batches in one disk /// chunk, pushing matching fragments. Returns bytes consumed: the start /// of the first batch that did not fully fit in the chunk (the caller @@ -1274,7 +911,9 @@ mod tests { assert_eq!(past_interval, None); } - fn non_empty_fragments() -> PollFragments<4096> { + // Resident execution passes already selected bytes through unchanged. + // The snapshot test needs a nonempty payload but does not decode messages. + fn placeholder_fragments() -> PollFragments<4096> { let mut fragments = PollFragments::new(); fragments.push(crate::types::Fragment::whole( Owned::<4096>::zeroed(8).into(), @@ -1282,199 +921,54 @@ mod tests { fragments } - fn consumer_auto_commit(offsets: Arc, consumer_id: u32) -> AutoCommitCtx { - consumer_auto_commit_with_limit(offsets, consumer_id, crate::DEFAULT_CONSUMER_OFFSETS_MAX) - } - - fn consumer_auto_commit_with_limit( - offsets: Arc, - consumer_id: u32, - limit: usize, - ) -> AutoCommitCtx { - AutoCommitCtx { - target: AutoCommitTarget::Consumer { - offsets, - consumer_id, - create_path: None, - }, - capacity: Rc::new(ConsumerOffsetCapacity::new(ConsumerKind::Consumer, limit)), - durable: Rc::new(DurableConsumerOffsets::default()), - } - } - #[test] - fn given_existing_phantom_when_primary_reservation_is_denied_should_restore_previous_offset() { - let offsets = Arc::new(ConsumerOffsets::with_capacity(2)); - offsets.pin().insert( - 7, - ConsumerOffset::new(ConsumerKind::Consumer, 7, 4, String::new()), - ); - let context = consumer_auto_commit_with_limit(Arc::clone(&offsets), 7, 1); - context - .durable - .record_explicit(ConsumerKind::Consumer, 8, 0, 0); - let applied = context - .apply(9) - .expect("existing phantom is locally writable"); - assert!(applied.reserve_durable().is_err()); - applied.rollback_created(); - assert_eq!( - offsets - .pin() - .get(&7) - .expect("phantom retained") - .offset - .load(Ordering::Relaxed), - 4 - ); - } - - #[test] - fn given_new_auto_commit_when_provisional_guard_is_dropped_should_reopen_capacity() { - let offsets = Arc::new(ConsumerOffsets::with_capacity(1)); - let context = consumer_auto_commit_with_limit(Arc::clone(&offsets), 7, 1); - let capacity = Rc::clone(&context.capacity); - let durable = Rc::clone(&context.durable); - let applied = context.apply(9).expect("new poll fits"); - let reservation = applied - .reserve_durable() - .expect("primary admits") - .expect("new slot"); - assert!(capacity.check(8, &durable).is_err()); - drop(reservation); - applied.rollback_created(); - assert!(capacity.check(8, &durable).is_ok()); - assert!(offsets.pin().is_empty()); - } - - #[test] - fn given_full_auto_commit_map_when_creating_key_should_reject_without_insertion() { - let offsets = Arc::new(ConsumerOffsets::with_capacity(1)); - offsets.pin().insert( - 1, - ConsumerOffset::new(ConsumerKind::Consumer, 1, 0, String::new()), - ); - let plan = PollPlan { - commit_offset: 42, - auto_commit: Some(consumer_auto_commit_with_limit(offsets.clone(), 2, 1)), - last_polled: None, - tier: PollTier::Resident { - fragments: non_empty_fragments(), - last_matching_offset: Some(5), + fn resident_read_returns_snapshot_facts() { + let snapshot_history = PollHistoryId::default(); + let partition_commit_offset = 42; + let last_selected_offset = 5; + let consumer_id = 7; + let partition_id = 0; + let resident_plan = PollPlan { + commit_offset: partition_commit_offset, + context: PollContext { + history: snapshot_history, + consumer: PollingConsumer::Consumer(consumer_id, partition_id), + auto_commit: true, }, - }; - - let Err(error) = plan.execute_resident() else { - panic!("a missing key at the map limit must be rejected"); - }; - assert_eq!(error.kind, ConsumerKind::Consumer); - assert_eq!(error.occupied, 1); - assert!(offsets.pin().get(&2).is_none()); - } - - #[test] - fn resident_auto_commit_applies_in_memory_and_surfaces_offset() { - // A resident auto_commit poll stays on the inline fast path (no detached - // task since the poll no longer persists), applies the committed offset - // to the in-memory map for read-your-own-poll, AND surfaces it so the - // serving shard replicates it through consensus. - let offsets = Arc::new(ConsumerOffsets::with_capacity(1)); - let plan = PollPlan { - commit_offset: 42, - auto_commit: Some(consumer_auto_commit(offsets.clone(), 7)), - last_polled: None, tier: PollTier::Resident { - fragments: non_empty_fragments(), - last_matching_offset: Some(5), + fragments: placeholder_fragments(), + last_matching_offset: Some(last_selected_offset), }, }; + assert!(!resident_plan.needs_off_pump_io()); - assert!( - !plan.needs_off_pump_io(), - "a resident auto_commit no longer persists on the poll path; the pump must not spawn", - ); - - let (fragments, commit_offset, applied) = - plan.execute_resident().expect("auto-commit is admitted"); - assert!(!fragments.is_empty(), "resident fragments must be returned"); - assert_eq!(commit_offset, 42, "commit offset is forwarded verbatim"); - - let applied = applied.expect("auto_commit must surface the applied offset for replication"); - assert!(matches!(applied.kind, ConsumerKind::Consumer)); - assert_eq!(applied.consumer_id, 7); - assert_eq!(applied.offset, 5); - - let stored = offsets - .pin() - .get(&7usize) - .map(|entry| entry.offset.load(Ordering::Relaxed)); - assert_eq!( - stored, - Some(5), - "the in-memory auto-commit must be applied on the resident path", - ); + // Reading preserves both the partition frontier and the last selected + // message offset. These are snapshot facts awaiting owner acceptance. + let read_result = resident_plan.execute_resident(); + assert_eq!(read_result.context.history, snapshot_history); + assert_eq!(read_result.commit_offset, partition_commit_offset); + assert_eq!(read_result.last_matching_offset, Some(last_selected_offset)); + assert!(!read_result.fragments.is_empty()); } #[test] - fn empty_resident_poll_surfaces_no_auto_commit() { - // Nothing served -> nothing to commit: no offset is surfaced and the - // in-memory map stays untouched. - let offsets = Arc::new(ConsumerOffsets::with_capacity(1)); - let plan = PollPlan { + fn empty_read_returns_no_progress() { + let consumer_id = 7; + let partition_id = 0; + let empty_plan = PollPlan { commit_offset: 9, - auto_commit: Some(consumer_auto_commit(offsets.clone(), 7)), - last_polled: None, + context: PollContext { + history: PollHistoryId::default(), + consumer: PollingConsumer::Consumer(consumer_id, partition_id), + auto_commit: true, + }, tier: PollTier::Empty, }; - let (fragments, _commit_offset, applied) = - plan.execute_resident().expect("empty poll needs no slot"); - assert!(fragments.is_empty()); - assert!( - applied.is_none(), - "empty poll must not surface an auto-commit" - ); - assert!( - offsets.pin().get(&7usize).is_none(), - "an empty poll must not touch the offset map", - ); - } - - #[test] - fn auto_commit_apply_is_monotone_but_explicit_store_rewinds() { - // Auto-commit must never rewind a newer offset (anti-rewind via - // fetch_max); an explicit StoreConsumerOffset may legitimately rewind. - let offsets = Arc::new(ConsumerOffsets::with_capacity(1)); - consumer_auto_commit(offsets.clone(), 7) - .apply(10) - .expect("first auto-commit is admitted"); - let after_high = offsets - .pin() - .get(&7usize) - .map(|entry| entry.offset.load(Ordering::Relaxed)); - assert_eq!(after_high, Some(10)); - // A stale auto-commit with a smaller offset must not rewind. - consumer_auto_commit(offsets.clone(), 7) - .apply(4) - .expect("existing key remains admitted"); - let after_stale = offsets - .pin() - .get(&7usize) - .map(|entry| entry.offset.load(Ordering::Relaxed)); - assert_eq!(after_stale, Some(10), "auto-commit fetch_max must hold"); - - // The explicit pump path (store-semantics) still rewinds to 4. - upsert_offset(&offsets, 7usize, 4, || { - ConsumerOffset::new(ConsumerKind::Consumer, 7, 0, String::new()) - }); - let after_explicit = offsets - .pin() - .get(&7usize) - .map(|entry| entry.offset.load(Ordering::Relaxed)); - assert_eq!( - after_explicit, - Some(4), - "explicit store may rewind below the auto-committed offset", - ); + // Automatic commits are enabled, but no matching message means there + // is no selected offset for the owner to apply as consumer progress. + let read_result = empty_plan.execute_resident(); + assert!(read_result.fragments.is_empty()); + assert_eq!(read_result.last_matching_offset, None); } } diff --git a/core/partitions/src/state_transfer.rs b/core/partitions/src/state_transfer.rs index 1a81ff54f8..e4ffae9e27 100644 --- a/core/partitions/src/state_transfer.rs +++ b/core/partitions/src/state_transfer.rs @@ -2891,6 +2891,7 @@ where // on a receiver that missed the purge. Same hazard and same fix as // `purge`: wipe the shared read-state slots first, so suspended walks // re-resolve by path and see the fresh files. + self.invalidate_poll_history(); self.log.invalidate_sealed_read_state(); self.segment_checksum_cache.borrow_mut().clear(); // Every staging file this install does not rename away is gone by the @@ -3255,7 +3256,7 @@ where } self.durable_consumer_offsets.clear(); self.pending_consumer_offset_commits.clear(); - self.queued_auto_commit_reservations.borrow_mut().clear(); + self.consumer_offset_capacity .rebuild(&self.durable_consumer_offsets, std::iter::empty()); self.consumer_group_offset_capacity @@ -3507,6 +3508,7 @@ where // The empty plant below can land on a base offset this sweep unlinks, // so an in-flight poll's cached read fd would keep serving the retired // inodes as live data. Same hazard and same fix as `purge`. + self.invalidate_poll_history(); self.log.invalidate_sealed_read_state(); while let Some((_, mut storage)) = self.log.retire_front() { let _ = storage.shutdown(); @@ -3518,7 +3520,7 @@ where self.last_polled_offsets.pin().clear(); self.durable_consumer_offsets.clear(); self.pending_consumer_offset_commits.clear(); - self.queued_auto_commit_reservations.borrow_mut().clear(); + for kind in [ConsumerKind::Consumer, ConsumerKind::ConsumerGroup] { self.consumer_offset_capacity_for(kind) .rebuild(&self.durable_consumer_offsets, std::iter::empty()); diff --git a/core/sdk/src/clients/consumer.rs b/core/sdk/src/clients/consumer.rs index 961f5b89e6..52a953c4df 100644 --- a/core/sdk/src/clients/consumer.rs +++ b/core/sdk/src/clients/consumer.rs @@ -473,6 +473,12 @@ unsafe impl Sync for IggyConsumer {} /// automatically; the next call parks for [`polling_retry_interval()`] while polling is paused. Hence, /// deciding when to give up on repeated errors is up to you. /// +/// With [`AutoCommitWhen::PollingMessages`], a missing response can leave the server's +/// cursor ahead of messages this consumer received. Continuing with +/// [`PollingStrategy::next()`] can skip those messages. See the +/// [poll recovery contract](MessageClient::poll_messages) for recovery from explicit +/// checkpoints for each partition. +/// /// For a boilerplate implementation of such a loop Iggy provides [`IggyConsumerMessageExt::consume_messages`]. /// /// # Tracking what has been read diff --git a/core/server/Cargo.toml b/core/server/Cargo.toml index d112462088..fba3b0c91f 100644 --- a/core/server/Cargo.toml +++ b/core/server/Cargo.toml @@ -80,6 +80,7 @@ name = "iggy-server" path = "src/main.rs" [features] +poll-diagnostics = ["shard/poll-diagnostics"] default = ["mimalloc", "iggy-web"] disable-mimalloc = [] mimalloc = ["dep:mimalloc"] diff --git a/core/server/config.toml b/core/server/config.toml index 618ca6cad7..e5009ce264 100644 --- a/core/server/config.toml +++ b/core/server/config.toml @@ -735,6 +735,13 @@ inbox_capacity = 1024 # peak client-reply fan-out per shard. reply_inbox_capacity = 1024 +# Maximum running disk polls plus queued completions per shard. Each disk read +# reserves a slot before I/O and retains it until the owner dequeues or discards +# the result, including after a requester timeout. When all slots are occupied, +# new disk polls are rejected before I/O. Main and reply inbox traffic uses +# separate capacity. This limits operation count, not retained message bytes. +poll_completion_capacity = 1024 + # Wall-clock budget for a single shard's bus drain on shutdown. Drives # the per-shard watchdog and the parallel-join survivor path; sized # larger than typical TCP RTT times in-flight write-batch so writers diff --git a/core/server/src/boot/mod.rs b/core/server/src/boot/mod.rs index ca55df8c33..e7a218695e 100644 --- a/core/server/src/boot/mod.rs +++ b/core/server/src/boot/mod.rs @@ -55,7 +55,6 @@ use crate::boot::threads::{ spawn_shutdown_watchdog, validate_sharding_runtime_knobs, }; use crate::boot::topology::{RosterCells, resolve_tcp_topology}; -use crate::dispatch::partition::make_partition_read_handler; use crate::dispatch::reads::read_frontier_budget; use crate::dispatch::session_ops::warm_dummy_password_hash; use crate::dispatch::submit::make_metadata_submit_handler; @@ -127,7 +126,6 @@ where ), on_metadata_submit: make_metadata_submit_handler(shard_handle), on_list_clients: make_list_clients_handler(&sessions), - on_partition_read: make_partition_read_handler(shard_handle), sessions, } } diff --git a/core/server/src/boot/recovery.rs b/core/server/src/boot/recovery.rs index 03f41a9de4..9f3dd9628e 100644 --- a/core/server/src/boot/recovery.rs +++ b/core/server/src/boot/recovery.rs @@ -258,7 +258,6 @@ pub(in crate::boot) async fn build_shard_for_thread( on_client_request, on_metadata_submit, on_list_clients, - on_partition_read, sessions, } = wire_shell_handlers( &bus, @@ -282,12 +281,12 @@ pub(in crate::boot) async fn build_shard_for_thread( Rc::clone(&on_client_request), on_metadata_submit, on_list_clients, - on_partition_read, metadata, partitions, senders, inbox, reply_inbox, + config.sharding.poll_completion_capacity, shards_table, PartitionConsensusConfig::new( topology.cluster_id, diff --git a/core/server/src/boot/threads.rs b/core/server/src/boot/threads.rs index 4aaf0cc968..79760ede09 100644 --- a/core/server/src/boot/threads.rs +++ b/core/server/src/boot/threads.rs @@ -580,6 +580,13 @@ pub(in crate::boot) fn validate_sharding_runtime_knobs( max: INBOX_CAPACITY_MAX, }); } + let poll_completion_capacity = sharding.poll_completion_capacity; + if poll_completion_capacity == 0 || poll_completion_capacity > INBOX_CAPACITY_MAX { + return Err(ServerError::InvalidPollCompletionCapacity { + value: poll_completion_capacity, + max: INBOX_CAPACITY_MAX, + }); + } let drain_timeout = sharding.shutdown_drain_timeout.get_duration(); if drain_timeout.is_zero() || drain_timeout > SHUTDOWN_DRAIN_TIMEOUT_MAX { return Err(ServerError::InvalidShutdownDrainTimeout { @@ -858,6 +865,38 @@ impl StopSignals { mod tests { use super::*; + #[test] + fn given_invalid_poll_completion_capacity_when_boot_validates_should_report_setting() { + for capacity in [0, INBOX_CAPACITY_MAX + 1] { + let sharding = configs::sharding::ShardingConfig { + poll_completion_capacity: capacity, + ..configs::sharding::ShardingConfig::default() + }; + let error = validate_sharding_runtime_knobs(&sharding) + .expect_err("boot must reject invalid completion capacity before allocation"); + + assert!(matches!( + error, + ServerError::InvalidPollCompletionCapacity { value, max } + if value == capacity && max == INBOX_CAPACITY_MAX + )); + } + } + + #[test] + fn given_poll_completion_capacity_at_boundaries_when_boot_validates_should_accept() { + for capacity in [1, INBOX_CAPACITY_MAX] { + let sharding = configs::sharding::ShardingConfig { + poll_completion_capacity: capacity, + ..configs::sharding::ShardingConfig::default() + }; + assert!( + validate_sharding_runtime_knobs(&sharding).is_ok(), + "boot rejected completion capacity {capacity}" + ); + } + } + #[test] fn shutdown_on_drop_armed_flips_flag() { let flag = Arc::new(AtomicBool::new(false)); diff --git a/core/server/src/consumer_group.rs b/core/server/src/consumer_group.rs index 0150be4069..29cf1d3f32 100644 --- a/core/server/src/consumer_group.rs +++ b/core/server/src/consumer_group.rs @@ -264,10 +264,10 @@ where #[cfg(test)] mod tests { - use std::cell::RefCell; use std::future::Future; use std::path::Path; use std::sync::Arc; + use std::sync::atomic::AtomicBool; use bytes::Bytes; use consensus::{Consensus, LocalPipeline, PartitionsHandle, Sequencer, VsrConsensus}; @@ -284,23 +284,30 @@ mod tests { use iggy_binary_protocol::{WireName, WireOptions}; use iggy_common::{ ConsumerGroupId, ConsumerGroupOffsets, ConsumerKind, ConsumerOffset, ConsumerOffsets, - PartitionStats, + IggyByteSize, PartitionStats, }; + use metadata::IggyMetadata; use metadata::stm::StateMachine; use partitions::state_transfer::mark_materialization_missing; - use partitions::{IggyIndexWriter, IggyPartition, MessagesWriter, PartitionsConfig}; + use partitions::{ + IggyIndexWriter, IggyPartition, IggyPartitions, MessagesWriter, PartitionPathLayout, + PartitionsConfig, + }; use server_common::SegmentStorage; use server_common::send_messages::{ IggyMessage, IggyMessageHeader, IggyMessages, SendMessagesOwned, }; use server_common::sharding::{IggyNamespace, PartitionLocation, ShardId}; - use shard::shards_table::ShardsTable; - use shard::{LifecycleFrame, Receiver, ShardFrame, shard_channel}; + use shard::metrics::ShardMetrics; + use shard::shards_table::{PapayaShardsTable, ShardsTable}; + use shard::{ + LifecycleFrame, PartitionConsensusConfig, ReplicaTopology, ShardFrame, ShardIdentity, + channel, shard_channel, + }; use super::*; - use crate::dispatch::partition::make_partition_read_handler; use crate::dispatch::test_support::{ - SpyBus, TestShard, prepare_message, request_message, test_shard, + SpyBus, TestMux, TestShard, prepare_message, request_message, }; const STREAM_ID: WireIdentifier = WireIdentifier::Numeric(0); @@ -317,7 +324,7 @@ mod tests { #[compio::test] async fn given_rejected_join_when_recovered_should_retry_without_stale_revocations() { - let (shard, inbox) = group_shard(); + let shard = group_shard(); let directory = tempfile::tempdir().unwrap(); let stale_namespace = namespace(&shard, STALE_PARTITION); let recovering_namespace = namespace(&shard, RECOVERING_PARTITION); @@ -340,9 +347,8 @@ mod tests { let before = streams .consumer_group_details(&STREAM_ID, &TOPIC_ID, &GROUP_ID) .unwrap(); - let rejected = with_partition_reads( + let rejected = run_with_partition_message_pump( &shard, - &inbox, maybe_rewrite_consumer_group_request(&shard, join_request(FIRST_CLIENT)), ) .await; @@ -361,9 +367,8 @@ mod tests { ); recover_partition(&shard, &directory.path().join("donor")).await; - let accepted = with_partition_reads( + let accepted = run_with_partition_message_pump( &shard, - &inbox, maybe_rewrite_consumer_group_request(&shard, join_request(FIRST_CLIENT)), ) .await @@ -391,9 +396,8 @@ mod tests { vec![STALE_PARTITION, RECOVERING_PARTITION] ); - let next_join = with_partition_reads( + let next_join = run_with_partition_message_pump( &shard, - &inbox, maybe_rewrite_consumer_group_request(&shard, join_request(SECOND_CLIENT)), ) .await @@ -419,7 +423,7 @@ mod tests { #[compio::test] async fn given_transferring_partition_when_joining_should_preserve_commit_ownership() { for missing in [true, false] { - let (shard, inbox) = group_shard(); + let shard = group_shard(); apply_initial_join(&shard); let directory = tempfile::tempdir().unwrap(); let mut recovering = partition(&shard, RECOVERING_PARTITION, directory.path()); @@ -455,9 +459,8 @@ mod tests { .consumer_group_member_assignment(&STREAM_ID, &TOPIC_ID, &GROUP_ID, FIRST_CLIENT) .unwrap(); - let rewritten = with_partition_reads( + let rewritten = run_with_partition_message_pump( &shard, - &inbox, maybe_rewrite_consumer_group_request(&shard, join_request(SECOND_CLIENT)), ) .await; @@ -538,59 +541,156 @@ mod tests { } #[compio::test] - async fn given_unanswered_or_missing_partitions_when_joining_should_allow_eager_handoff() { - let (shard, inbox) = group_shard(); + async fn given_unroutable_partition_when_joining_should_preserve_ownership() { + let monotonic_group_id = 0; + let shard = group_shard(); apply_initial_join(&shard); - let unanswered = namespace(&shard, STALE_PARTITION); - shard.shards_table().remove(&unanswered); - assert!( + let streams = shard.plane.metadata().mux_stm.streams(); + let group_before_join = streams + .consumer_group_details(&STREAM_ID, &TOPIC_ID, &GROUP_ID) + .unwrap(); + + // The first member owns both partitions. Remove one route so gathering + // its progress is explicitly refused before any owner receives that read. + let unroutable_namespace = namespace(&shard, RECOVERING_PARTITION); + shard.shards_table().remove(&unroutable_namespace); + assert!(matches!( shard - .partition_read(unanswered, PartitionRead::GroupOffsetState { group_id: 0 }) - .await - .is_none() + .partition_read( + unroutable_namespace, + PartitionRead::GroupOffsetState { + group_id: monotonic_group_id + }, + ) + .await, + Some(PartitionReadReply::Rejected( + IggyError::TransientNotAccepted + )) + )); + + // A refusal aborts the second member's join before it can be replicated. + let join_result = run_with_partition_message_pump( + &shard, + maybe_rewrite_consumer_group_request(&shard, join_request(SECOND_CLIENT)), + ); + assert!(matches!( + join_result.await, + Err(IggyError::TransientNotAccepted) + )); + assert_eq!( + streams.consumer_group_details(&STREAM_ID, &TOPIC_ID, &GROUP_ID), + Some(group_before_join), + "a refused progress read must leave membership and assignments unchanged" + ); + + // Committing previously polled work remains allowed for the existing owner. + let require_pollable = false; + assert_eq!( + streams.consumer_group_fence( + &STREAM_ID, + &TOPIC_ID, + &GROUP_ID, + FIRST_CLIENT, + RECOVERING_PARTITION, + require_pollable + ), + Some(monotonic_group_id), + "the existing owner must retain permission to commit" ); - let not_found = with_partition_reads( + assert!( + !streams.has_pending_revocations(), + "the rejected join must not start a handoff" + ); + } + + #[compio::test] + async fn given_dropped_or_not_found_group_replies_when_joining_should_allow_eager_handoff() { + let mut shard = group_shard(); + apply_initial_join(&shard); + + // Routes exist but no partitions have been installed. The real owner + // pump answers NotFound, establishing the missing partition case. + let missing_partition_reply = run_with_partition_message_pump( &shard, - &inbox, shard.partition_read( namespace(&shard, RECOVERING_PARTITION), PartitionRead::GroupOffsetState { group_id: 0 }, ), ) .await; - assert!(matches!(not_found, Some(PartitionReadReply::NotFound))); + assert!(matches!( + missing_partition_reply, + Some(PartitionReadReply::NotFound) + )); - let rewritten = with_partition_reads( - &shard, - &inbox, - maybe_rewrite_consumer_group_request(&shard, join_request(SECOND_CLIENT)), - ) - .await - .unwrap(); + // Receive both join reads through a controlled owner inbox. Lose one + // reply after submission and answer NotFound for the other partition. + let (sender, owner_inbox, _owner_replies) = + shard_channel(0, INBOX_CAPACITY, INBOX_CAPACITY); + Rc::get_mut(&mut shard) + .unwrap() + .attach_senders(vec![sender]); + let unanswered_namespace = namespace(&shard, STALE_PARTITION); + let missing_namespace = namespace(&shard, RECOVERING_PARTITION); + let owner_task = compio::runtime::spawn(async move { + let mut reply_was_dropped = false; + let mut not_found_was_sent = false; + for _ in 0..PARTITION_COUNT { + let ShardFrame::Lifecycle(LifecycleFrame::PartitionRead { + namespace, + read: PartitionRead::GroupOffsetState { group_id: 0 }, + reply, + }) = owner_inbox.recv().await.unwrap() + else { + panic!("the group read must have reached the owner"); + }; + if namespace == unanswered_namespace { + drop(reply); + reply_was_dropped = true; + } else { + assert_eq!(namespace, missing_namespace); + reply.try_send(PartitionReadReply::NotFound).unwrap(); + not_found_was_sent = true; + } + } + assert!(reply_was_dropped, "one submitted read must lose its reply"); + assert!( + not_found_was_sent, + "the other read must report a missing partition" + ); + }); + + // Neither outcome establishes outstanding work to drain. The join + // therefore permits immediate reassignment under the existing policy. + let join_for_replication = + maybe_rewrite_consumer_group_request(&shard, join_request(SECOND_CLIENT)) + .await + .unwrap(); + owner_task.await.expect("the owner task must finish"); + let replicated_join = + ReplicatedJoinConsumerGroupRequest::decode_from(request_body(&join_for_replication)) + .unwrap(); assert!( - ReplicatedJoinConsumerGroupRequest::decode_from(request_body(&rewritten)) - .unwrap() - .in_flight - .is_empty() + replicated_join.in_flight.is_empty(), + "neither partition should wait for its previous owner to drain" ); - apply_join(&shard, &rewritten); + apply_join(&shard, &join_for_replication); let streams = shard.plane.metadata().mux_stm.streams(); assert!(!streams.has_pending_revocations()); + let (_generation, assigned_partitions) = streams + .consumer_group_member_assignment(&STREAM_ID, &TOPIC_ID, &GROUP_ID, SECOND_CLIENT) + .unwrap(); assert_eq!( - streams - .consumer_group_member_assignment(&STREAM_ID, &TOPIC_ID, &GROUP_ID, SECOND_CLIENT) - .unwrap() - .1, - vec![RECOVERING_PARTITION] + assigned_partitions, + vec![RECOVERING_PARTITION], + "the new member can poll its partition immediately after the join is applied" ); } - fn group_shard() -> (Rc, Receiver) { - let bus = SpyBus::default(); - let mut shard = test_shard(&bus, 0, 3, 1); - let (sender, inbox, _replies) = shard_channel(0, INBOX_CAPACITY, INBOX_CAPACITY); - shard.attach_senders(vec![sender]); - let shard = Rc::new(shard); + /// Create a group and routes for two partitions, with no members or local + /// partitions. Tests install partition state and apply joins explicitly. + fn group_shard() -> Rc { + let shard = partition_read_shard(); let mux = &shard.plane.metadata().mux_stm; mux.update(prepare_message( Operation::CreateStream, @@ -644,7 +744,62 @@ mod tests { PartitionLocation::new(ShardId::new(0), 0), ); } - (shard, inbox) + shard + } + + /// Build a shard with its own inbox so tests can serve group progress reads + /// and clears through the production message pump. + fn partition_read_shard() -> Rc { + // These tests do not dispatch disk polls, so keep that lane minimal. + const POLL_COMPLETION_CAPACITY: usize = 1; + + let bus = SpyBus::default(); + let consensus = VsrConsensus::new( + 1, + 0, + 3, + server_common::sharding::METADATA_GROUP, + bus.clone(), + LocalPipeline::new(), + ); + consensus.set_incarnation(1); + consensus.init(); + let metadata = + IggyMetadata::new(Some(consensus), None, None, None, TestMux::default(), None); + let partitions = IggyPartitions::new( + ShardId::new(0), + PartitionsConfig { + messages_required_to_save: 1, + size_of_messages_required_to_save: IggyByteSize::from(1024_u64), + validate_checksum: true, + segment_size: IggyByteSize::from(1_048_576_u64), + preallocate_segments: false, + encryptor: None, + path_layout: PartitionPathLayout::default(), + }, + ); + let (sender, inbox, replies) = shard_channel(0, INBOX_CAPACITY, INBOX_CAPACITY); + Rc::new( + TestShard::new( + ShardIdentity::new(0, "consumer-group-test".to_string()), + bus.clone(), + Rc::new(|_, _| {}), + Rc::new(|_, _| {}), + Rc::new(|_| {}), + Rc::new(|_| {}), + metadata, + partitions, + vec![sender], + inbox, + replies, + POLL_COMPLETION_CAPACITY, + PapayaShardsTable::new(), + PartitionConsensusConfig::new(1, ReplicaTopology::new(0, 3), bus), + None, + ShardMetrics::for_shard(), + ) + .unwrap(), + ) } fn namespace(shard: &Rc, partition_id: u32) -> IggyNamespace { @@ -746,26 +901,14 @@ mod tests { .unwrap(); } - async fn with_partition_reads( + /// Poll the shard's message pump alongside the operation until it finishes. + /// This serves progress reads and also executes any requested stale clears. + async fn run_with_partition_message_pump( shard: &Rc, - inbox: &Receiver, operation: impl Future, ) -> T { - let handle = Rc::new(RefCell::new(Some(Rc::downgrade(shard)))); - let handler = make_partition_read_handler(&handle); - let serve = async { - loop { - let ShardFrame::Lifecycle(LifecycleFrame::PartitionRead { - namespace, - read, - reply, - }) = inbox.recv().await.unwrap() - else { - panic!("unexpected frame while serving the join's partition reads"); - }; - handler(namespace, read, reply); - } - }; + let (_stop, stop) = channel(1); + let serve = shard.run_message_pump(stop, Arc::new(AtomicBool::new(false))); match select(Box::pin(operation), Box::pin(serve)).await { Either::Left((result, _)) => result, Either::Right(_) => { diff --git a/core/server/src/dispatch/failure.rs b/core/server/src/dispatch/failure.rs index 3501221fda..19025351b5 100644 --- a/core/server/src/dispatch/failure.rs +++ b/core/server/src/dispatch/failure.rs @@ -26,7 +26,7 @@ //! | [`FrameChannel::TypedDeny`] | Reply, nonzero status + empty body, or a result-framed rejection body | rejections that must unblock the SDK's lockstep request slot: checksum, authz, pre-consensus rewrite, unknown or unsupported non-replicated code, unbound non-PING read, transient replay hints | //! | [`FrameChannel::Eviction`] | session-terminal Eviction frame with a typed reason | the client must register again: `NoSession`, `MalformedLogin`, heartbeat and login evictions. The reason rides the channel label, since one `context` covers four of them | //! | [`FrameChannel::ResyncSentinel`] | status-0 poll reply, body carries `RESYNC_REQUIRED_PARTITION_SENTINEL` | a fenced consumer-group poll: the consumer must re-sync its assignment; HTTP mirrors it as `resync_required_polled_messages` in `crate::http::wire` | -//! | [`FrameChannel::EmptyFrame`] | status-0 fail-fast body, the 16-byte empty poll | the partition cannot answer yet; the SDK fails fast (empty poll) and retries. A permanent client error never rides this channel: an undecodable body, an unresolved target, and a request the resolve rejects all deny typed, because there is nothing to retry | +//! | [`FrameChannel::EmptyFrame`] | status 0 with the 16-byte empty poll | fallback for an unexpected owner reply or a poll encoding failure; this does not prove the partition is empty. Missing owner replies and explicit owner rejections use `TypedDeny` | //! | [`FrameChannel::Reply`] | status-0 success frame | host-built success replies: login/register, ping, logout, non-replicated read bodies, committed metadata replies | //! | silent drop | no frame | one deliberate case, a transient consensus submit failure: the SDK read-timeout replays the same request id, and a synthesized failure could contradict a write that commits moments later. A header `RequestHeader::validate` rejected also drops, but that one is a GAP, not a contract - the fields decode, so a deny could be echoed under the transport id, and the client instead waits out its read timeout | //! | HTTP status | HTTP status code | the HTTP spine maps the same rejections in `crate::http::error`; it never rides these frames | diff --git a/core/server/src/dispatch/mod.rs b/core/server/src/dispatch/mod.rs index 5a5e9a2d33..9767cfc479 100644 --- a/core/server/src/dispatch/mod.rs +++ b/core/server/src/dispatch/mod.rs @@ -897,6 +897,9 @@ mod tests { /// A test shard wired to its own lanes (the held sender feeds them), /// for the reply-lane pump tests below. fn reply_lane_test_shard(name: &str) -> (SpyBus, shard::TaggedSender, Rc) { + // These tests do not dispatch disk polls, so keep that lane minimal. + const POLL_COMPLETION_CAPACITY: usize = 1; + let bus = SpyBus::default(); let metadata = IggyMetadata::new(None, None, None, None, TestMux::default(), None); let partitions = IggyPartitions::new( @@ -921,12 +924,12 @@ mod tests { Rc::new(|_, _| {}), Rc::new(|_| {}), Rc::new(|_| {}), - Rc::new(|_, _, _| {}), metadata, partitions, vec![sender], inbox_rx, reply_inbox_rx, + POLL_COMPLETION_CAPACITY, PapayaShardsTable::new(), PartitionConsensusConfig::new(1, ReplicaTopology::new(0, 1), bus.clone()), None, diff --git a/core/server/src/dispatch/partition.rs b/core/server/src/dispatch/partition.rs index 11255dcb0a..3c87ded5e3 100644 --- a/core/server/src/dispatch/partition.rs +++ b/core/server/src/dispatch/partition.rs @@ -15,21 +15,13 @@ // specific language governing permissions and limitations // under the License. -//! The partition data plane, both sides of the mesh on one page. +//! Partition requests and replies through the shard mesh. //! -//! Server side: [`make_partition_read_handler`] answers `PartitionRead` -//! frames on the owning shard (poll snapshots, consumer offsets, segment -//! deletes). Client side: the funnel routes partition writes through -//! [`dispatch_partition_request`], and the non-replicated read arms call -//! [`handle_poll_messages`] / [`handle_get_consumer_offset`], which read via -//! the shard mesh. -//! -//! Deliberate asymmetry (the plane-split reply trio): the partitions engine -//! replies to committed writes itself, straight from the owning shard, while -//! everything the host builds here -- read bodies, denies, empty-poll shapes -//! -- goes out on the connection's home shard. The reply path is therefore -//! split by plane, not unified, and the deny helpers in `failure` are the -//! third leg (typed status replies for requests that never reach a plane). +//! Writes enter through [`dispatch_partition_request`]. Reads enter through +//! [`handle_poll_messages`] and [`handle_get_consumer_offset`], then execute +//! on the owning shard. The partition engine sends replies for committed writes; +//! this module sends read replies and admission errors through the connection's +//! home shard. use crate::consumer_group::maybe_rewrite_consumer_offset_request; use crate::dispatch::authz::{authorize_partition_op, authorize_partition_read}; @@ -38,372 +30,37 @@ use crate::dispatch::failure::{ send_non_replicated_deny, send_result_rejection, }; use crate::dispatch::submit::submit_client_request_on_owner; -use crate::dispatch::upgrade_shard_handle; use crate::responses::{ build_consumer_offset_body, build_polled_messages_reply, current_metadata_commit, resolve_partition_namespace, resolve_partition_request_namespace, }; -use crate::shell::{ShellBus, ShellShard, ShellShardHandle}; -use crate::wire::{request_body, usize_to_u32}; +use crate::shell::{ShellBus, ShellShard}; +use crate::wire::request_body; use bytes::Bytes; -use consensus::{Consensus, MetadataHandle, PartitionsHandle}; +use consensus::{MetadataHandle, PartitionsHandle}; use iggy_binary_protocol::PrepareHeader; use iggy_binary_protocol::primitives::consumer::WireConsumer; use iggy_binary_protocol::primitives::polling_strategy::WirePollingStrategy; -use iggy_binary_protocol::requests::consumer_offsets::{ - GetConsumerOffsetRequest, StoreConsumerOffsetRequest, -}; +use iggy_binary_protocol::requests::consumer_offsets::GetConsumerOffsetRequest; use iggy_binary_protocol::requests::messages::PollMessagesRequest; use iggy_binary_protocol::requests::segments::DeleteSegmentsRequest; -use iggy_binary_protocol::{ - AckLevel, Command, KIND_CONSUMER_GROUP, Operation, RoutedRequestHeader, WireDecode, WireEncode, - WireIdentifier, -}; +use iggy_binary_protocol::{KIND_CONSUMER_GROUP, Operation, RoutedRequestHeader, WireDecode}; use iggy_common::{ConsumerKind, IggyError, PollingStrategy, RESYNC_REQUIRED_PARTITION_SENTINEL}; use journal::superblock::SuperblockStore; use journal::{Journal, JournalHandle}; -use message_bus::{AUTO_COMMIT_CLIENT_ID, BusMessage}; +use message_bus::BusMessage; use metadata::impls::metadata::{ StreamsFrontend, build_truncate_partition_client_message, build_truncate_partition_client_message_with_identifiers, }; -use partitions::{ - AutoCommitApplied, ConsumerOffsetCapacityError, PollFragments, PollPlan, PollingArgs, - PollingConsumer, -}; +use partitions::{PollingArgs, PollingConsumer}; use server_common::Message; use server_common::sharding::IggyNamespace; use shard::shards_table::ShardsTable; -use shard::{PartitionRead, PartitionReadHandler, PartitionReadReply}; +use shard::{PartitionRead, PartitionReadReply}; use std::rc::Rc; use tracing::{debug, warn}; -/// Build the per-shard [`PartitionReadHandler`]: on a `PartitionRead` frame -/// (this shard owns the namespace), run the poll / consumer-offset lookup -/// against the local partitions plane and push the result back over the -/// carried reply sender. The requesting shard bounds the wait with a -/// timeout, so a dropped reply degrades to a client-visible read failure. -pub fn make_partition_read_handler( - shard_handle: &ShellShardHandle, -) -> PartitionReadHandler -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let shard_handle = Rc::clone(shard_handle); - // Runs synchronously on the shard pump (see `process_lifecycle` -> - // `on_partition_read`). `build_poll_snapshot` takes a pump-only `&mut` - // partition borrow (synchronous, so no sibling task can realloc under it) and - // returns an owned `PollPlan`; only owned data crosses into `spawn_poll_io`. A - // fully-resident poll replies here without spawning. See the `poll_plan` module docs. - Rc::new(move |namespace, read, reply| { - let Some(shard) = upgrade_shard_handle(&shard_handle) else { - return; - }; - let partitions = shard.plane.partitions(); - if partitions.with_partition( - &namespace, - partitions::IggyPartition::requires_state_transfer, - ) == Some(true) - { - let _ = reply.try_send(PartitionReadReply::Rejected( - IggyError::TransientNotAccepted, - )); - return; - } - match read { - PartitionRead::Poll { consumer, args } => { - match partitions.build_poll_snapshot(&namespace, consumer, &args) { - None => { - let _ = reply.try_send(PartitionReadReply::NotFound); - } - Some(plan) if plan.needs_off_pump_io() => { - spawn_poll_io(Rc::clone(&shard), namespace, plan, reply); - } - Some(plan) => { - let result = poll_reply(&shard, namespace, plan.execute_resident()); - let _ = reply.try_send(result); - } - } - } - PartitionRead::ConsumerOffset { consumer } => { - let result = match partitions.consumer_offset_read(&namespace, consumer) { - Some((stored, current_offset)) => PartitionReadReply::ConsumerOffset { - stored, - current_offset, - }, - None => PartitionReadReply::NotFound, - }; - let _ = reply.try_send(result); - } - PartitionRead::GroupOffsetState { group_id } => { - let result = match partitions.group_offset_state(&namespace, group_id) { - Some((last_polled, committed)) => PartitionReadReply::GroupOffsetState { - last_polled, - committed, - }, - None => PartitionReadReply::NotFound, - }; - let _ = reply.try_send(result); - } - PartitionRead::ClearGroupLastPolled { group_id } => { - let result = match partitions.clear_group_last_polled(&namespace, group_id) { - Some(()) => PartitionReadReply::Ack, - None => PartitionReadReply::NotFound, - }; - let _ = reply.try_send(result); - } - PartitionRead::ResolveSegmentDeleteOffset { count } => { - let result = partitions - .segment_delete_resolution(&namespace, count) - .map_or_else( - || PartitionReadReply::NotFound, - |(up_to_offset, lagging)| PartitionReadReply::SegmentDeleteOffset { - up_to_offset, - lagging, - }, - ); - let _ = reply.try_send(result); - } - } - }) -} - -/// Spawn the off-pump leg of a partition poll: disk read + auto-commit apply on -/// the OWNED plan (disk descriptors, resident-tail `Frozen` clones, `Arc` offset -/// map), then replicate the auto-committed offset and send the reply. Holds no -/// partition reference across the IO, so it is sound concurrently with the -/// pump's `&mut` writes; the auto-commit submit re-borrows synchronously after. -fn spawn_poll_io( - shard: Rc>, - namespace: IggyNamespace, - plan: PollPlan, - reply: shard::Sender, -) where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - let bus = shard.bus.clone(); - bus.spawn(async move { - // Diagnostic-only wall clock: `elapsed` gates the slow-poll `warn!` - // below and is never folded into a reply or the deterministic schedule, - // so it stays sound under the simulator's virtual clock (there it just - // measures near-zero real time and never fires). Do not derive any - // replicated or reply value from it, or replay determinism breaks. - let poll_started = std::time::Instant::now(); - let result = plan.execute().await; - let elapsed = poll_started.elapsed(); - if elapsed > std::time::Duration::from_secs(1) { - warn!( - namespace_raw = namespace.inner(), - elapsed_ms = u64::try_from(elapsed.as_millis()).unwrap_or(u64::MAX), - "slow partition poll; gather side may have timed out" - ); - } - let result = poll_reply(&shard, namespace, result); - let _ = reply.try_send(result); - }); -} - -fn poll_reply( - shard: &Rc>, - namespace: IggyNamespace, - result: Result<(PollFragments, u64, Option), ConsumerOffsetCapacityError>, -) -> PartitionReadReply -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - match result { - Ok((fragments, current_offset, auto_commit)) => { - if let Some(applied) = auto_commit - && let Err(error) = submit_auto_commit(shard, namespace, applied) - { - PartitionReadReply::Rejected(error) - } else { - PartitionReadReply::Poll { - fragments, - current_offset, - } - } - } - Err(error) => { - if !error.uncertain { - shard.metrics().record_consumer_offset_denied(error.kind); - } - warn_auto_commit_capacity(namespace, error); - PartitionReadReply::Rejected(error.into()) - } - } -} - -/// Replicate a poll's auto-committed offset through the partition consensus so -/// it survives failover, mirroring the explicit `StoreConsumerOffset` path: the -/// same op code, submitted onto the owning shard's own pipeline. The poll reply -/// does not wait for commit, but a primary reserves cardinality before this -/// submission and rejects the poll if the local inbox cannot accept it. -/// -/// The partition plane admits writes on the primary only (it asserts so), and a -/// poll is served on whichever node owns the namespace locally, which may be a -/// backup. So gate on primary status here. Auto-commit -/// is server-managed best-effort, so a follower-served -/// poll simply does not advance the durable offset. The same contract covers a -/// local cursor that never became durable: when the per-kind live map is over -/// its limit the partition evicts such a cursor, and that consumer's next -/// `Next` poll restarts from offset 0. -/// -/// A poll whose auto-commit cannot be submitted is answered -/// `TransientNotAccepted` and returns no messages, even though the fragments -/// were already read: the owning shard's inbox refused the frame, or the -/// partition changed primary or incarnation during the read. Like the -/// `TooManyConsumerOffsets` refusal at the key limit, it returns no batch. -/// Transient refusals permit retry. Capacity refusals need available capacity. -/// -/// Coalescing: an offset the partition's committed high-water already covers is -/// dropped without a consensus op (the steady state for a re-poll of committed -/// data, hence no log). The gate reads committed state only, so an offset that -/// merely sits in flight keeps resubmitting until its covering op commits -- a -/// dropped op self-heals on the next poll instead of being suppressed forever. -fn submit_auto_commit( - shard: &Rc>, - namespace: IggyNamespace, - applied: AutoCommitApplied, -) -> Result<(), IggyError> -where - B: ShellBus, - MJ: JournalHandle + 'static, - MJ::Target: Journal, Header = PrepareHeader>, - S: 'static, - SB: SuperblockStore + 'static, -{ - // Everything inside is synchronous: `admit` rolls the eager cursor update - // back on `Err` in the same call, and no await may sit between the update - // and that rollback. - applied.admit(|applied| { - let primary = shard - .plane - .partitions() - .with_partition(&namespace, |partition| { - let consensus = partition.consensus(); - if !partition.auto_commit_admission_ready(applied) { - return Err(IggyError::TransientNotAccepted); - } - Ok(consensus.is_primary() && consensus.is_normal() && !consensus.is_transferring()) - }); - if matches!(primary, None | Some(Err(_))) { - return Err(IggyError::TransientNotAccepted); - } - if primary == Some(Ok(false)) { - debug!( - namespace_raw = namespace.inner(), - "auto-commit offset not replicated: partition not primary on this node (best-effort)" - ); - return Ok(()); - } - let reservation = match applied.reserve_durable() { - Ok(Some(reservation)) => reservation, - Ok(None) => return Ok(()), - Err(error) => { - if !error.uncertain { - shard.metrics().record_consumer_offset_denied(applied.kind); - } - warn_auto_commit_capacity(namespace, error); - return Err(error.into()); - } - }; - let message = build_auto_commit_request(namespace, applied).inspect_err(|error| { - warn!( - namespace_raw = namespace.inner(), - error = %error, - "failed to build auto-commit store-offset request" - ); - })?; - // Routes by namespace to this same owning primary shard's inbox. The - // pump admits it next turn exactly like a client store. `dispatch` - // never blocks. - shard - .submit_auto_commit_offset(message, reservation) - .map_err(|_| IggyError::TransientNotAccepted) - }) -} - -fn warn_auto_commit_capacity( - namespace: IggyNamespace, - error: partitions::ConsumerOffsetCapacityError, -) { - if error.first_in_episode && error.uncertain { - warn!( - namespace_raw = namespace.inner(), - kind = ?error.kind, - "consumer offset accounting unavailable during auto-commit" - ); - } else if error.first_in_episode { - warn!( - namespace_raw = namespace.inner(), - kind = ?error.kind, - occupied = error.occupied, - limit = error.limit, - config = "[partition] consumer_offsets_max", - "consumer offset map limit reached during auto-commit" - ); - } -} - -/// Build the synthetic `StoreConsumerOffset` request for an auto-commit, keyed -/// to the resolved numeric consumer/group id and stamped with the reserved -/// [`AUTO_COMMIT_CLIENT_ID`] so the commit path skips the (unwaited) reply. The -/// wire stream/topic ids are cosmetic here -- admission and apply key off the -/// header namespace and the consumer id -- but are set from the namespace for a -/// well-formed body. `ack` is `Quorum` so the offset actually replicates. -fn build_auto_commit_request( - namespace: IggyNamespace, - applied: &AutoCommitApplied, -) -> Result, IggyError> { - let request = StoreConsumerOffsetRequest { - consumer: WireConsumer { - kind: applied.kind.as_code(), - id: WireIdentifier::Numeric(applied.consumer_id), - }, - stream_id: WireIdentifier::Numeric(usize_to_u32(namespace.stream_id())?), - topic_id: WireIdentifier::Numeric(usize_to_u32(namespace.topic_id())?), - partition_id: Some(usize_to_u32(namespace.partition_id())?), - offset: applied.offset, - ack: AckLevel::Quorum, - }; - let body = request.to_bytes(); - let header_size = std::mem::size_of::(); - let total_size = header_size + body.len(); - let size = u32::try_from(total_size).map_err(|_| IggyError::InvalidConfiguration)?; - let mut message = Message::::new(total_size); - message.as_mut_slice()[header_size..].copy_from_slice(&body); - Ok( - message.transmute_header(|_, header: &mut RoutedRequestHeader| { - *header = RoutedRequestHeader { - command: Command::Request, - operation: Operation::StoreConsumerOffset, - size, - client: AUTO_COMMIT_CLIENT_ID, - // The reserved sentinel client is never deduped and never - // replied to; a nonzero session + request just satisfy the wire - // header validation. - session: 1, - request: 1, - group: namespace.inner(), - ..Default::default() - }; - }), - ) -} - /// Route a partition data-plane op (`SendMessages` / consumer-offset writes) /// through the shard mesh by namespace: the op belongs to the partition's /// own consensus group, not the metadata group. The owning shard's @@ -680,12 +337,12 @@ fn consumer_offset_kind(request: &Message) -> Option( shard: &Rc>, @@ -790,8 +447,10 @@ pub(in crate::dispatch) async fn handle_poll_messages( } /// Run the resolved poll on the owning shard and re-encode the stored -/// batches into the wire `PolledMessages` reply. A failed read or re-encode -/// hands back the fail-fast empty poll for the partition instead. +/// batches into the wire `PolledMessages` reply. Owner rejections preserve +/// their error; a missing reply reports a communication error without claiming +/// that the owner rejected the read. Unexpected replies and encoding failures +/// retain the empty poll fallback. #[allow(clippy::future_not_send)] async fn read_polled_messages( shard: &Rc>, @@ -830,12 +489,25 @@ where ReadPolledMessagesError::Fallback(empty_poll_fallback(partition_id)) }), Some(PartitionReadReply::Rejected(error)) => Err(ReadPolledMessagesError::Rejected(error)), - other => { + None => { + // The owner rejects a disconnected receiver before poll admission, + // but timeout can race with that check. Missing replies therefore + // do not prove rejection and must not be reported as empty success. warn!( transport_client_id, namespace = namespace.inner(), - reply_was_none = other.is_none(), - "partition read failed; replying empty poll" + "partition read reply unavailable; acceptance unknown" + ); + Err(ReadPolledMessagesError::Rejected( + IggyError::ShardCommunicationError, + )) + } + Some(other) => { + warn!( + transport_client_id, + namespace = namespace.inner(), + reply = ?other, + "unexpected partition poll reply; replying empty poll" ); Err(ReadPolledMessagesError::Fallback(empty_poll_fallback( partition_id, @@ -1413,20 +1085,23 @@ mod tests { use consensus::Sequencer; use iggy_binary_protocol::primitives::partition_assignment::CreatedPartitionAssignment; use iggy_binary_protocol::requests::consumer_offsets::DeleteConsumerOffsetRequest; + use iggy_binary_protocol::requests::consumer_offsets::StoreConsumerOffsetRequest; use iggy_binary_protocol::requests::messages::SendMessagesHeader; use iggy_binary_protocol::requests::streams::CreateStreamRequest; use iggy_binary_protocol::requests::topics::{ CreateTopicRequest, CreateTopicWithAssignmentsRequest, }; - use iggy_binary_protocol::{PrepareOkHeader, ReplyHeader}; - use iggy_binary_protocol::{WireName, WireOptions, WirePartitioning}; + use iggy_binary_protocol::{ + Command, PrepareOkHeader, ReplyHeader, WireEncode, WireIdentifier, WireName, WireOptions, + WirePartitioning, + }; use iggy_common::Identifier; use iggy_common::defaults::DEFAULT_ROOT_USER_ID; use metadata::IggyMetadata; use metadata::stm::StateMachine as _; use partitions::{IggyPartitions, PartitionPathLayout, PartitionsConfig}; use server_common::MessageBag; - use server_common::sharding::ShardId; + use server_common::sharding::{PartitionLocation, ShardId}; use shard::metrics::{ShardMetrics, frame_drop_reason, frame_drop_variant}; use shard::shards_table::PapayaShardsTable; use shard::{ @@ -1823,6 +1498,234 @@ mod tests { } } + #[compio::test] + async fn given_unroutable_partition_when_polling_should_report_not_accepted() { + let bus = SpyBus::default(); + let mut shard = test_shard(&bus, 0, 1, 1); + let (sender, owner_inbox, _owner_replies) = shard_channel(0, 1, 1); + shard.attach_senders(vec![sender]); + let shard = Rc::new(shard); + let namespace = IggyNamespace::new(1, 1, 0); + + // The owner has a live inbox, but no route identifies it for this partition. + assert_eq!( + poll_and_expect_error(&shard, namespace).await, + IggyError::TransientNotAccepted + ); + assert!( + owner_inbox.try_recv().is_err(), + "a poll rejected during routing must never reach the owner" + ); + } + + #[compio::test] + async fn given_missing_owner_sender_when_polling_should_report_not_accepted() { + let bus = SpyBus::default(); + let shard = Rc::new(test_shard(&bus, 0, 1, 1)); + let namespace = IggyNamespace::new(1, 1, 0); + shard + .shards_table() + .insert(namespace, PartitionLocation::new(ShardId::new(0), 0)); + + // A route exists, but no sender was attached to submit work to its owner. + assert_eq!( + poll_and_expect_error(&shard, namespace).await, + IggyError::TransientNotAccepted + ); + } + + #[compio::test] + async fn given_full_owner_inbox_when_polling_should_report_not_accepted() { + let bus = SpyBus::default(); + let mut shard = test_shard(&bus, 0, 1, 1); + let inbox_capacity = 1; + let (sender, owner_inbox, _owner_replies) = shard_channel(0, inbox_capacity, 1); + // Occupy the only slot before polling; no task drains this inbox. + sender + .try_send(ShardFrame::lifecycle(LifecycleFrame::MetadataCommitTick)) + .unwrap(); + shard.attach_senders(vec![sender]); + let shard = Rc::new(shard); + let namespace = IggyNamespace::new(1, 1, 0); + shard + .shards_table() + .insert(namespace, PartitionLocation::new(ShardId::new(0), 0)); + + assert_eq!( + poll_and_expect_error(&shard, namespace).await, + IggyError::TransientNotAccepted + ); + assert!(matches!( + owner_inbox.try_recv().unwrap(), + ShardFrame::Lifecycle(LifecycleFrame::MetadataCommitTick) + )); + assert!( + owner_inbox.try_recv().is_err(), + "only the filler frame may be queued; the poll was never submitted" + ); + } + + #[compio::test] + async fn given_closed_owner_inbox_when_polling_should_report_not_accepted() { + let bus = SpyBus::default(); + let mut shard = test_shard(&bus, 0, 1, 1); + let (sender, owner_inbox, _owner_replies) = shard_channel(0, 1, 1); + shard.attach_senders(vec![sender]); + // Closing the destination leaves its sender attached, so submission fails. + drop(owner_inbox); + let shard = Rc::new(shard); + let namespace = IggyNamespace::new(1, 1, 0); + shard + .shards_table() + .insert(namespace, PartitionLocation::new(ShardId::new(0), 0)); + + assert_eq!( + poll_and_expect_error(&shard, namespace).await, + IggyError::TransientNotAccepted + ); + } + + #[compio::test] + async fn given_submitted_poll_when_owner_drops_reply_should_report_unknown_acceptance() { + let bus = SpyBus::default(); + let mut shard = test_shard(&bus, 0, 1, 1); + let (sender, owner_inbox, _owner_replies) = shard_channel(0, 1, 1); + shard.attach_senders(vec![sender]); + let shard = Rc::new(shard); + let namespace = IggyNamespace::new(1, 1, 0); + shard + .shards_table() + .insert(namespace, PartitionLocation::new(ShardId::new(0), 0)); + + // Receive the request before dropping its reply. Submission succeeded, + // so the caller cannot infer whether the owner accepted any progress. + let owner_task = compio::runtime::spawn(async move { + let ShardFrame::Lifecycle(LifecycleFrame::PartitionRead { reply, .. }) = + owner_inbox.recv().await.unwrap() + else { + panic!("the poll must have reached the owner"); + }; + drop(reply); + }); + + assert_eq!( + poll_and_expect_error(&shard, namespace).await, + IggyError::ShardCommunicationError + ); + owner_task.await.expect("the owner task must finish"); + } + + #[compio::test] + async fn given_pending_poll_when_owner_reply_times_out_should_report_unknown_acceptance() { + const TRANSPORT_CLIENT_ID: u128 = 91; + const VSR_CLIENT_ID: u128 = 1; + let bus = SpyBus::default(); + // Expire the reply timeout deterministically, without a wall clock wait. + bus.instant_timers.set(true); + let mut shard = test_shard(&bus, 0, 1, 1); + let (sender, owner_inbox, _owner_replies) = shard_channel(0, 1, 1); + shard.attach_senders(vec![sender]); + let shard = Rc::new(shard); + + // Create metadata and a route so authorization and resolution let the + // poll reach its owner inbox. + let metadata = shard.plane.metadata(); + metadata.mux_stm.users().ensure_root_user("iggy", "hash"); + metadata + .mux_stm + .update(prepare_message( + Operation::CreateStream, + VSR_CLIENT_ID, + 1, + &CreateStreamRequest { + name: WireName::new("stream").unwrap(), + options: WireOptions::empty(), + } + .to_bytes(), + )) + .unwrap(); + metadata + .mux_stm + .update(prepare_message( + Operation::CreateTopicWithAssignments, + VSR_CLIENT_ID, + 2, + &CreateTopicWithAssignmentsRequest { + request: CreateTopicRequest { + stream_id: WireIdentifier::numeric(0), + partitions_count: 1, + name: WireName::new("topic").unwrap(), + options: WireOptions::empty(), + }, + derived_options: WireOptions::empty(), + partitions: vec![CreatedPartitionAssignment { + partition_id: 0, + consensus_group_id: 1, + }], + created_view: 0, + } + .to_bytes(), + )) + .unwrap(); + let namespace = metadata + .mux_stm + .streams() + .namespace_from_partition(&WireIdentifier::numeric(0), &WireIdentifier::numeric(0), 0) + .unwrap(); + shard + .shards_table() + .insert(namespace, PartitionLocation::new(ShardId::new(0), 0)); + let poll_body = PollMessagesRequest { + consumer: WireConsumer::consumer(WireIdentifier::numeric(1)), + stream_id: WireIdentifier::numeric(0), + topic_id: WireIdentifier::numeric(0), + partition_id: Some(0), + strategy: WirePollingStrategy::offset(0), + count: 10, + auto_commit: true, + } + .to_bytes(); + let request = request_message(Operation::NonReplicated, VSR_CLIENT_ID, 1, 1, &poll_body); + + // Keep the owner request queued so its live reply sender forces the + // timeout path. The owner can still receive the request after that timeout. + handle_poll_messages( + &shard, + TRANSPORT_CLIENT_ID, + &request, + Some(DEFAULT_ROOT_USER_ID), + ) + .await; + + // The client receives one error header, with no successful empty poll body. + let replies = bus.client_replies.borrow(); + assert_eq!(replies.len(), 1); + let (client_id, frame) = &replies[0]; + assert_eq!(*client_id, TRANSPORT_CLIENT_ID); + assert_eq!(frame.len(), std::mem::size_of::()); + let status_start = std::mem::offset_of!(ReplyHeader, status); + let status = u32::from_le_bytes(frame[status_start..status_start + 4].try_into().unwrap()); + assert_eq!(status, IggyError::ShardCommunicationError.as_code()); + drop(replies); + + // Timing out abandons the reply receiver; a late owner reply cannot + // replace the communication error or produce a second client response. + let ShardFrame::Lifecycle(LifecycleFrame::PartitionRead { reply, .. }) = + owner_inbox.try_recv().unwrap() + else { + panic!("the timed out poll must remain queued for its owner"); + }; + assert!( + reply + .try_send(PartitionReadReply::Rejected( + IggyError::TransientNotAccepted + )) + .is_err(), + "the timeout must have dropped the caller's reply receiver" + ); + assert_eq!(bus.client_replies.borrow().len(), 1); + } + /// A partition write whose routable wait exhausts (namespace committed, /// but no reconciler ever seeds this shard's routing row -- the state a /// teardown/rematerialise churn leaves behind) must answer a nonzero @@ -2140,6 +2043,8 @@ mod tests { const TRANSPORT: u128 = 91; const SESSION: u64 = 1; const STATUS_OFFSET: usize = std::mem::offset_of!(ReplyHeader, status); + // This test does not dispatch disk polls, so keep that lane minimal. + const POLL_COMPLETION_CAPACITY: usize = 1; let bus = SpyBus::default(); let metadata = IggyMetadata::new(None, None, None, None, TestMux::default(), None); @@ -2168,12 +2073,12 @@ mod tests { Rc::new(|_, _| {}), Rc::new(|_| {}), Rc::new(|_| {}), - Rc::new(|_, _, _| {}), metadata, partitions, vec![sender], inbox_rx, reply_inbox_rx, + POLL_COMPLETION_CAPACITY, PapayaShardsTable::new(), PartitionConsensusConfig::new(1, ReplicaTopology::new(0, 1), bus.clone()), None, @@ -2218,4 +2123,32 @@ mod tests { status so the SDK replays it instead of timing out" ); } + + /// Attempt an already resolved poll and require an error, including when no + /// owner reply arrives. Bypassing metadata resolution isolates owner routing + /// and reply handling from authentication and request decoding. + async fn poll_and_expect_error(shard: &Rc, namespace: IggyNamespace) -> IggyError { + let vsr_client_id = 1; + let transport_client_id = 91; + let consumer_id = 1; + let partition_id = 0; + let request = request_message(Operation::NonReplicated, vsr_client_id, 1, 1, &[]); + let resolved_poll = ( + namespace, + partition_id, + PollingConsumer::Consumer(consumer_id, usize::try_from(partition_id).unwrap()), + PollingArgs { + strategy: PollingStrategy::offset(0), + count: 1, + auto_commit: true, + }, + ); + match read_polled_messages(shard, transport_client_id, &request, resolved_poll).await { + Err(ReadPolledMessagesError::Rejected(error)) => error, + Err(ReadPolledMessagesError::Fallback(_)) => { + panic!("poll must not return an empty reply") + } + Ok(_) => panic!("poll must not succeed without an owner reply"), + } + } } diff --git a/core/server/src/http/error.rs b/core/server/src/http/error.rs index 8a0d04a318..aef9821819 100644 --- a/core/server/src/http/error.rs +++ b/core/server/src/http/error.rs @@ -353,10 +353,9 @@ pub(in crate::http) fn error_response(status: StatusCode, code: &str, reason: &s .into_response() } -/// Shared 504 rendering for an in-band request the partition plane did not -/// answer in time, shaped like every other HTTP error (`ErrorResponse`) so -/// clients parse one error schema. Consumed by the partition-write reply wait, -/// the partition reads ([`ReadError::Timeout`]), and the forward attempt bound. +/// Shared 504 rendering for partition writes and forwarding attempts that did +/// not answer in time. Partition reads use [`ReadError::Timeout`] to preserve +/// the same Iggy error identity as binary transports. pub(in crate::http) fn gateway_timeout_response(code: &str, reason: &str) -> Response { error_response(StatusCode::GATEWAY_TIMEOUT, code, reason) } @@ -479,9 +478,9 @@ pub(in crate::http) enum ReadError { /// refusal (`TransientNotCommitted`) already renders as, so an SDK that /// speaks both sees one answer. Never a 2xx with stale state. MetadataFrontierUnreached, - /// A partition read (poll / consumer-offset) got no reply from the owning - /// shard within the mesh budget. 504 like a produce timeout: the outcome is - /// unknown (the abandoned read may still be running), so the caller retries. + /// A partition read got no reply after submission to the owning shard. + /// Render HTTP 504 with the same Iggy error identity as binary transports. + /// A missing reply does not prove that the owner rejected the read. Timeout, } @@ -495,10 +494,13 @@ impl IntoResponse for ReadError { Self::NotPrimary => not_primary_response(), Self::RedirectToPrimary(location) => primary_redirect_response(&location), Self::RecoveryIncomplete | Self::MetadataFrontierUnreached => service_unavailable(), - Self::Timeout => gateway_timeout_response( - "partition_read_timeout", - "the partition owner did not answer the read in time; retry", - ), + Self::Timeout => ( + StatusCode::GATEWAY_TIMEOUT, + Json(ErrorResponse::from_error( + &IggyError::ShardCommunicationError, + )), + ) + .into_response(), } } } @@ -596,7 +598,9 @@ fn primary_node(roster: &ClusterRoster, primary_index: u8) -> Option<(&ResolvedC mod tests { use super::*; + use axum::body::to_bytes; use configs::cluster::{ClusterNodeConfig, TransportPorts}; + use serde_json::Value; const READ_PATH: &str = "/streams?consistency=linearizable"; fn node(replica_id: u8, ip: &str, http: Option) -> ClusterNodeConfig { @@ -774,6 +778,32 @@ mod tests { assert!(response.headers().contains_key(RETRY_AFTER)); } + #[tokio::test] + async fn partition_read_timeout_renders_504_with_shard_communication_error() { + let response = ReadError::Timeout.into_response(); + + assert_eq!(response.status(), StatusCode::GATEWAY_TIMEOUT); + let body = to_bytes(response.into_body(), 1024).await.unwrap(); + let error: Value = serde_json::from_slice(&body).unwrap(); + assert_eq!(error["id"], 11001); + assert_eq!(error["code"], "shard_communication_error"); + } + + #[tokio::test] + async fn rejected_partition_read_renders_503_with_transient_not_accepted() { + let response = ReadError::Rejected(IggyError::TransientNotAccepted).into_response(); + + assert_eq!(response.status(), StatusCode::SERVICE_UNAVAILABLE); + assert_eq!( + response.headers().get(RETRY_AFTER), + Some(&HeaderValue::from(RETRY_AFTER_SECONDS)) + ); + let body = to_bytes(response.into_body(), 1024).await.unwrap(); + let error: Value = serde_json::from_slice(&body).unwrap(); + assert_eq!(error["id"], 58); + assert_eq!(error["code"], "transient_not_accepted"); + } + #[test] fn business_error_renders_without_retry_after() { let response = CustomError::from(IggyError::UserAlreadyExists).into_response(); diff --git a/core/server/src/server_error.rs b/core/server/src/server_error.rs index 5e9bc61885..52d4d17fac 100644 --- a/core/server/src/server_error.rs +++ b/core/server/src/server_error.rs @@ -104,6 +104,8 @@ pub enum ServerError { InvalidInboxCapacity { value: usize, max: usize }, #[error("sharding.reply_inbox_capacity must be in 1..={max}; got {value}")] InvalidReplyInboxCapacity { value: usize, max: usize }, + #[error("sharding.poll_completion_capacity must be in 1..={max}; got {value}")] + InvalidPollCompletionCapacity { value: usize, max: usize }, #[error("sharding.shutdown_drain_timeout must be in (0, {max:?}]; got {value:?}")] InvalidShutdownDrainTimeout { value: std::time::Duration, diff --git a/core/server/src/shell.rs b/core/server/src/shell.rs index 10da7aecf7..5c95fef006 100644 --- a/core/server/src/shell.rs +++ b/core/server/src/shell.rs @@ -39,7 +39,7 @@ use metadata::stm::mux::WithFactory; use metadata::stm::stream::Streams; use metadata::stm::user::Users; use shard::shards_table::PapayaShardsTable; -use shard::{IggyShard, ListClientsHandler, MetadataSubmitHandler, PartitionReadHandler}; +use shard::{IggyShard, ListClientsHandler, MetadataSubmitHandler}; use std::cell::RefCell; use std::rc::{Rc, Weak}; use std::time::Duration; @@ -90,7 +90,6 @@ pub struct ShellHandlers { pub on_client_request: RequestHandler, pub on_metadata_submit: MetadataSubmitHandler, pub on_list_clients: ListClientsHandler, - pub on_partition_read: PartitionReadHandler, /// Bound by the client-request handler, read by the get-clients /// handler; the caller keeps it to reach locally-homed sessions. pub sessions: Rc>, @@ -108,7 +107,6 @@ impl ShellHandlers { on_client_request: Rc::new(|_, _| {}), on_metadata_submit: Rc::new(|_| {}), on_list_clients: Rc::new(|_| {}), - on_partition_read: Rc::new(|_, _, _| {}), sessions: Rc::new(RefCell::new(SessionManager::new())), } } diff --git a/core/server_common/src/lib.rs b/core/server_common/src/lib.rs index 9c7c1a55bc..e7b8c81511 100644 --- a/core/server_common/src/lib.rs +++ b/core/server_common/src/lib.rs @@ -26,6 +26,7 @@ pub mod fs_utils; pub mod iobuf; pub mod log; mod memory_pool; +pub mod poll; mod reactor_yield; mod segment_storage; pub mod send_messages; diff --git a/core/server_common/src/poll.rs b/core/server_common/src/poll.rs new file mode 100644 index 0000000000..ec10069829 --- /dev/null +++ b/core/server_common/src/poll.rs @@ -0,0 +1,279 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use std::cell::Cell; +use std::rc::Rc; +use std::sync::atomic::{AtomicU64, Ordering}; + +use iggy_common::ConsumerKind; + +static NEXT_POLL_HISTORY_ID: AtomicU64 = AtomicU64::new(0); + +/// Identity of one serviceable message history. It is never serialized. +/// +/// A process counter gives each new history a unique value, even if a rebuilt +/// partition reuses its namespace and offsets. Polls copy the value without +/// accessing the counter. `Default` creates a fresh identity. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct PollHistoryId(u64); + +impl Default for PollHistoryId { + /// # Panics + /// Panics when the process counter is exhausted. It never wraps, so a + /// pending read cannot match a later history through identity reuse. + fn default() -> Self { + Self::allocate(&NEXT_POLL_HISTORY_ID) + } +} + +impl PollHistoryId { + fn allocate(counter: &AtomicU64) -> Self { + // The counter provides uniqueness, not publication of partition state. + let id = counter + .fetch_update(Ordering::Relaxed, Ordering::Relaxed, |next| { + next.checked_add(1) + }) + .expect("poll history ID counter exhausted"); + Self(id) + } +} + +/// Lifetime accounting for one provisional consumer key. +/// +/// Requests for the same key hold separate guards, so dropping one request +/// cannot release capacity still held by another. +/// +/// Shared only on the owning shard thread. Acquisition and release update +/// the current counters without yielding or reentering task execution, so +/// `Rc` and `Cell` suffice even when guards outlive an async suspension. +#[derive(Debug)] +pub struct AutoCommitReservationToken { + kind: ConsumerKind, + consumer_id: u32, + /// Shared change stamp advanced on the last guard drop to retry reclamation. + /// Wrapping is allowed. + reclaim_epoch: Rc>, + /// Number of tokens with outstanding guards in this capacity tracker. + active_keys: Rc>, + /// Outstanding guards for this key, excluding cached handles to the token. + active: Cell, +} + +impl AutoCommitReservationToken { + /// Create an inactive token using counters shared by one capacity tracker. + /// Reuse the token for concurrent reservations of the same key. Construction + /// neither checks the configured limit nor occupies capacity. + #[must_use] + pub fn new( + kind: ConsumerKind, + consumer_id: u32, + reclaim_epoch: Rc>, + active_keys: Rc>, + ) -> Self { + Self { + kind, + consumer_id, + reclaim_epoch, + active_keys, + active: Cell::new(0), + } + } + + /// Hold this key until the returned guard is dropped. + /// The first guard increments the shared key count. Capacity admission must + /// already have succeeded, without yielding between that check and this call. + #[must_use] + pub fn acquire(self: &Rc) -> AutoCommitReservation { + let active = self.active.get(); + self.active.set(active.wrapping_add(1)); + if active == 0 { + self.active_keys.set(self.active_keys.get().wrapping_add(1)); + } + AutoCommitReservation { + token: Rc::clone(self), + } + } + + /// Count outstanding guards, excluding cached handles to this token. + #[must_use] + pub fn active_count(&self) -> usize { + self.active.get() + } + + /// Test token identity, not just the consumer kind and ID. + #[must_use] + pub fn owns(self: &Rc, reservation: &AutoCommitReservation) -> bool { + Rc::ptr_eq(self, &reservation.token) + } +} + +/// Guard for provisional capacity while a request waits or enters replication. +/// Dropping the last guard releases the key's provisional occupancy and enables +/// reclamation retries. Durable membership and pending prepare reservations +/// for the same key remain unchanged. +/// +/// The owner creates this guard when accepting a poll result, after any disk +/// completion has crossed the inbox. Local request entries or replication +/// continuations retain it. It never enters a shard channel. +#[derive(Debug)] +pub struct AutoCommitReservation { + token: Rc, +} + +impl AutoCommitReservation { + #[must_use] + pub fn kind(&self) -> ConsumerKind { + self.token.kind + } + + #[must_use] + pub fn consumer_id(&self) -> u32 { + self.token.consumer_id + } +} + +impl Drop for AutoCommitReservation { + fn drop(&mut self) { + let active = self.token.active.get(); + self.token.active.set(active.wrapping_sub(1)); + if active == 1 { + self.token + .active_keys + .set(self.token.active_keys.get().wrapping_sub(1)); + self.token + .reclaim_epoch + .set(self.token.reclaim_epoch.get().wrapping_add(1)); + } + } +} + +#[cfg(test)] +mod tests { + use std::collections::HashSet; + use std::panic::catch_unwind; + use std::sync::Barrier; + use std::thread; + + use super::*; + + #[test] + fn histories_match_only_their_own_copies() { + let history = PollHistoryId::default(); + let copied_history = history; + assert_eq!(history, copied_history); + let other_history = PollHistoryId::default(); + assert_ne!(history, other_history); + } + + #[test] + fn histories_created_on_different_threads_are_unique() { + const THREAD_COUNT: usize = 4; + const HISTORIES_PER_THREAD: usize = 128; + let start = Barrier::new(THREAD_COUNT); + + let histories = thread::scope(|scope| { + let workers: Vec<_> = (0..THREAD_COUNT) + .map(|_| { + scope.spawn(|| { + start.wait(); + (0..HISTORIES_PER_THREAD) + .map(|_| PollHistoryId::default()) + .collect::>() + }) + }) + .collect(); + workers + .into_iter() + .flat_map(|worker| worker.join().unwrap()) + .map(|history| history.0) + .collect::>() + }); + + assert_eq!(histories.len(), THREAD_COUNT * HISTORIES_PER_THREAD); + } + + #[test] + fn exhausted_history_counter_never_reuses_an_identity() { + let counter = AtomicU64::new(u64::MAX - 1); + let last_history = PollHistoryId::allocate(&counter); + assert_eq!(last_history.0, u64::MAX - 1); + + // A failed allocation must leave the counter exhausted on later attempts. + for _ in 0..2 { + assert!(catch_unwind(|| PollHistoryId::allocate(&counter)).is_err()); + assert_eq!(counter.load(Ordering::Relaxed), u64::MAX); + } + } + + #[test] + fn last_reservation_releases_the_key() { + let consumer_id = 7; + let reclaim_epoch = Rc::new(Cell::new(0)); + let active_keys = Rc::new(Cell::new(0)); + let token = Rc::new(AutoCommitReservationToken::new( + ConsumerKind::Consumer, + consumer_id, + Rc::clone(&reclaim_epoch), + Rc::clone(&active_keys), + )); + + // Two requests for the same consumer occupy one capacity slot. + let first_reservation = token.acquire(); + let second_reservation = token.acquire(); + assert_eq!(first_reservation.kind(), ConsumerKind::Consumer); + assert_eq!(first_reservation.consumer_id(), consumer_id); + assert_eq!(active_keys.get(), 1); + + drop(first_reservation); + assert_eq!( + active_keys.get(), + 1, + "the second request still holds the key" + ); + + drop(second_reservation); + assert_eq!(active_keys.get(), 0); + assert_eq!( + reclaim_epoch.get(), + 1, + "releasing the key enables reclamation" + ); + } + + #[test] + fn last_reservation_wraps_reclaim_epoch() { + let group_id = 7; + let reclaim_epoch = Rc::new(Cell::new(u64::MAX)); + let active_keys = Rc::new(Cell::new(0)); + let token = Rc::new(AutoCommitReservationToken::new( + ConsumerKind::ConsumerGroup, + group_id, + Rc::clone(&reclaim_epoch), + Rc::clone(&active_keys), + )); + let reservation = token.acquire(); + assert_eq!(reservation.kind(), ConsumerKind::ConsumerGroup); + assert_eq!(reservation.consumer_id(), group_id); + + // The guard retains the token until release, even after its cached + // handle is gone. Advancing reclamation past u64::MAX must wrap. + drop(token); + drop(reservation); + assert_eq!(active_keys.get(), 0); + assert_eq!(reclaim_epoch.get(), 0); + } +} diff --git a/core/shard/Cargo.toml b/core/shard/Cargo.toml index 1ee8ae7135..00cc3d6427 100644 --- a/core/shard/Cargo.toml +++ b/core/shard/Cargo.toml @@ -23,6 +23,7 @@ license = "Apache-2.0" publish = false [features] +poll-diagnostics = [] # Simulator-only test hook (`IggyShard::init_partition`): bypasses the # reconciler's `ReconcileOp::InsertOwned` funnel, mutating `IggyPartitions` # off the pump task. A `-p iggy-server` build excludes it; `cargo build diff --git a/core/shard/src/builder.rs b/core/shard/src/builder.rs index 07363f451e..c2c18eb242 100644 --- a/core/shard/src/builder.rs +++ b/core/shard/src/builder.rs @@ -31,8 +31,7 @@ use crate::coordinator::{ShardZeroCoordinator, classify_try_send_err}; use crate::metrics::{ShardMetrics, frame_drop_variant}; use crate::{ CoordinatorConfig, IggyShard, LifecycleFrame, ListClientsHandler, MetadataSubmitHandler, - PartitionConsensusConfig, PartitionReadHandler, Receiver, ShardCtorError, ShardFrame, - ShardIdentity, TaggedSender, + PartitionConsensusConfig, Receiver, ShardCtorError, ShardFrame, ShardIdentity, TaggedSender, }; use consensus::VsrConsensus; use journal::JournalHandle; @@ -68,12 +67,12 @@ where on_client_request: RequestHandler, on_metadata_submit: MetadataSubmitHandler, on_list_clients: ListClientsHandler, - on_partition_read: PartitionReadHandler, metadata: IggyMetadata, MJ, S, M, SB>, partitions: IggyPartitions, senders: Vec, inbox: Receiver, reply_inbox: Receiver, + poll_completion_capacity: usize, shards_table: T, partition_consensus: PartitionConsensusConfig, coord_config: CoordinatorConfig, @@ -99,12 +98,12 @@ where on_client_request: RequestHandler, on_metadata_submit: MetadataSubmitHandler, on_list_clients: ListClientsHandler, - on_partition_read: PartitionReadHandler, metadata: IggyMetadata, MJ, S, M, SB>, partitions: IggyPartitions, senders: Vec, inbox: Receiver, reply_inbox: Receiver, + poll_completion_capacity: usize, shards_table: T, partition_consensus: PartitionConsensusConfig, coord_config: CoordinatorConfig, @@ -117,12 +116,12 @@ where on_client_request, on_metadata_submit, on_list_clients, - on_partition_read, metadata, partitions, senders, inbox, reply_inbox, + poll_completion_capacity, shards_table, partition_consensus, coord_config, @@ -142,6 +141,10 @@ where /// [`ShardCtorError::ShardCountOverflow`] if `senders.len()` does not /// fit in `u16`. Both are bootstrap programming errors and the /// `u16` overflow check fires on every shard, not only shard 0. + /// + /// # Panics + /// + /// Panics if `poll_completion_capacity` is zero. pub fn build(self) -> Result, ShardCtorError> { let is_shard_zero = self.identity.id == 0; @@ -228,12 +231,12 @@ where self.on_client_request, self.on_metadata_submit, self.on_list_clients, - self.on_partition_read, self.metadata, self.partitions, self.senders, self.inbox, self.reply_inbox, + self.poll_completion_capacity, self.shards_table, self.partition_consensus, coordinator, diff --git a/core/shard/src/lib.rs b/core/shard/src/lib.rs index 2a0be3f471..1fcb8536c1 100644 --- a/core/shard/src/lib.rs +++ b/core/shard/src/lib.rs @@ -19,10 +19,12 @@ pub mod builder; pub mod config; pub mod coordinator; pub mod metrics; +mod poll; mod router; pub mod shards_table; pub use config::CoordinatorConfig; +pub use poll::PollCompleted; pub use router::CONSENSUS_TICK_INTERVAL; #[cfg(feature = "simulator")] @@ -366,15 +368,13 @@ pub enum PartitionReadReply { stored: Option, current_offset: u64, }, - /// The read was refused and returns no messages, even where fragments were - /// already gathered. For a poll with `auto_commit`, `TooManyConsumerOffsets` - /// when the poll needed a new offset key past `[partition] - /// consumer_offsets_max`, and `TransientNotAccepted` when the auto-commit - /// could not be submitted: the owning shard's inbox was full, or the - /// partition changed primary or incarnation during the read. Transient - /// refusal also covers any read while the partition requires state transfer, - /// and permits retrying. A capacity refusal needs a slot reclaimed - /// or a higher configured limit before a new key can succeed. + /// The read was refused and returns no messages. A poll that needs a new + /// consumer offset key beyond the configured limit returns + /// `TooManyConsumerOffsets`. A completion whose history changed, whose owner + /// inbox is unavailable, or whose automatic commit cannot be admitted returns + /// `TransientNotAccepted`, allowing the client to retry. Reads also return + /// `TransientNotAccepted` when submission fails before reaching the owner + /// or while the partition requires state transfer. Rejected(IggyError), /// Reply to [`PartitionRead::GroupOffsetState`]: the group's last-polled and /// committed offsets on this partition (each `None` if absent). @@ -402,22 +402,15 @@ pub enum PartitionReadReply { NotFound, } -/// Handler the owning shard runs for an inbound -/// [`LifecycleFrame::PartitionRead`]. The server wires it to its partitions -/// plane; the handler pushes the result back over the carried reply sender. -pub type PartitionReadHandler = - Rc)>; - -/// Reply budget for a cross-shard [`IggyShard::partition_read`]. Bounds a -/// wedged owning shard; the caller maps expiry to a client-visible error. +/// Reply budget for [`IggyShard::partition_read`]. Bounds a wedged owning shard; +/// the caller maps expiry to an error with unknown acceptance. /// /// 10s, not lower: a disk poll over tiny segments opens one file per -/// segment, so a 1024-message read can legitimately take several seconds -/// on an oversubscribed host (8 parallel test clusters). Expiry is masked -/// as an empty poll downstream while the abandoned walk keeps running, so -/// a too-small budget turns slow reads into missing data plus duplicated -/// walks from client retries. Must stay below the SDK's 30s request -/// deadline. +/// segment, so a read of 1024 messages can legitimately take several seconds +/// on an oversubscribed host (8 parallel test clusters). Disk I/O continues +/// after expiry, so a short budget can waste completed reads. Acceptance racing +/// with expiry can still leave an unknown outcome. Must stay below the SDK's +/// 30s request deadline. const PARTITION_READ_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(10); /// Budget for a partition write's wait on its committed reply. Longer than a @@ -772,12 +765,6 @@ pub enum LifecycleFrame { request: Message, reply: Sender>>, }, - /// Local auto-commit submission. The guard travels with the frame so an - /// inbox drop or admission refusal releases its provisional key directly. - AutoCommitSubmit { - request: Message, - reservation: partitions::AutoCommitReservation, - }, /// Shard 0 broadcasts after a partition-shaped metadata commit; wakes /// the per-shard reconciler. No payload: reconciler re-reads target /// state. Drops covered by the periodic safety tick. @@ -1441,16 +1428,11 @@ where /// the simulator stub ctor. on_list_clients: ListClientsHandler, - /// Handler for inbound [`LifecycleFrame::PartitionRead`] queries. - /// The server wires it to this shard's partitions plane. Defaults to a - /// no-op for the simulator stub ctor. - on_partition_read: PartitionReadHandler, - /// Channel senders to every shard, indexed by shard id. /// Includes a sender to self so that local routing goes through the /// same channel path as remote routing. /// - /// [`assert_sender_ordering`] is invoked in the ctor so `senders[i]` + /// [`validate_sender_ordering`] runs during construction so `senders[i]` /// is guaranteed to feed the shard whose `id == i`. Call sites can /// therefore index by `target_shard` without re-checking. senders: Vec, @@ -1470,6 +1452,10 @@ where /// fill the main lane. Fed via [`TaggedSender::reply_sender`]. reply_inbox: Receiver, + /// Disk reads reserve this lane before I/O so ordinary frames cannot + /// displace their results. Only the owner pump validates completions. + poll_completions: poll::completion::PollCompletionLane, + /// Partition namespace -> owning shard lookup. shards_table: T, @@ -1484,8 +1470,7 @@ where /// and in single-shard tests that bypass the coordinator. coordinator: Option>, - /// Per-shard observability counters. Cloned at metric increment sites, - /// so cheap (`Arc` clone) regardless of label cardinality. + /// Observability counters shared with the metrics registry. metrics: crate::metrics::ShardMetrics, /// Late-bound `MetadataCommitTick` handler. `None` until reconciler @@ -1675,6 +1660,13 @@ where self.reply_inbox.len() } + /// Queued disk completions covered by the simulator's lost wakeup check. + #[cfg(any(test, feature = "simulator"))] + #[must_use] + pub fn poll_completion_inbox_len(&self) -> usize { + self.poll_completions.len() + } + /// The armed metadata repair window as `(to_op, peer)`, `None` when no session /// is running. /// @@ -1702,6 +1694,8 @@ where /// * `inbox` - the receiver that this shard drains in its message pump. /// * `reply_inbox` - the reply lane's receiver, drained by the same /// pump (client `Reply` forwards only; see [`TaggedSender::reply_sender`]). + /// * `poll_completion_capacity` - separate limit on running disk polls plus + /// results awaiting dequeue by this shard's owner pump. Must be nonzero. /// * `shards_table` - namespace -> shard routing table. /// * `coordinator` - `Some` on shard 0 (supplied by the builder when /// `is_shard_zero`), `None` everywhere else. Immutable post-ctor: @@ -1718,6 +1712,10 @@ where /// fit in `u16`. Both are bootstrap programming errors: the /// permutation would silently misroute every inter-shard frame, or /// addressing space (u16) would wrap. + /// + /// # Panics + /// + /// Panics if `poll_completion_capacity` is zero. #[allow(clippy::too_many_arguments)] pub fn new( identity: ShardIdentity, @@ -1726,12 +1724,12 @@ where on_client_request: RequestHandler, on_metadata_submit: MetadataSubmitHandler, on_list_clients: ListClientsHandler, - on_partition_read: PartitionReadHandler, metadata: IggyMetadata, MJ, S, M, SB>, partitions: IggyPartitions, senders: Vec, inbox: Receiver, reply_inbox: Receiver, + poll_completion_capacity: usize, shards_table: T, partition_consensus: PartitionConsensusConfig, coordinator: Option>, @@ -1745,6 +1743,8 @@ where let nonce_seed = forward_nonce_seed(metadata.consensus.as_ref()); let plane = MuxPlane::new(variadic!(metadata, partitions)); let ShardIdentity { id, name } = identity; + let poll_completions = + poll::completion::PollCompletionLane::new(poll_completion_capacity, &metrics); Ok(Self { id, name, @@ -1754,11 +1754,11 @@ where on_client_request, on_metadata_submit, on_list_clients, - on_partition_read, senders, shard_count, inbox, reply_inbox, + poll_completions, shards_table, partition_consensus, coordinator, @@ -1992,14 +1992,22 @@ where clients } - /// Run a partition read (message poll / consumer-offset lookup) on the - /// shard owning `namespace` and await the reply. + /// Run a partition read on the shard owning `namespace` and await the reply. /// /// Routes a [`LifecycleFrame::PartitionRead`] through the shards table - /// (self-sends included, so a locally-owned partition takes the same - /// path). `None` = unroutable namespace, full owning-shard inbox, - /// dropped reply sender, or `PARTITION_READ_TIMEOUT` expiry; the - /// caller maps it to a client-visible error. + /// (including sends to this shard). An unroutable namespace, missing sender, + /// or rejected inbox submission returns [`PartitionReadReply::Rejected`] + /// with [`IggyError::TransientNotAccepted`]: the owner never received the + /// request, so this read cannot advance progress and is safe to retry. + /// + /// `None` means submission succeeded but the reply sender was dropped or + /// `PARTITION_READ_TIMEOUT` expired. A timeout drops the reply receiver + /// without canceling a queued request or detached read. The owner discards + /// a poll completion if it observes disconnection before admission. If + /// timeout races with that check, completion can still advance progress and + /// admit an automatic commit. Callers must not treat a missing reply as an + /// accepted empty poll or as evidence that retrying cannot advance progress + /// again. #[allow(clippy::future_not_send)] pub async fn partition_read( &self, @@ -2012,7 +2020,9 @@ where namespace_raw = namespace.inner(), "partition_read: namespace not routable (not materialised yet or deleted)" ); - return None; + return Some(PartitionReadReply::Rejected( + IggyError::TransientNotAccepted, + )); }; let (reply_tx, reply_rx) = channel::(1); let frame = ShardFrame::lifecycle(LifecycleFrame::PartitionRead { @@ -2020,14 +2030,20 @@ where read, reply: reply_tx, }); - let sender = self.senders.get(target as usize)?; + let Some(sender) = self.senders.get(target as usize) else { + return Some(PartitionReadReply::Rejected( + IggyError::TransientNotAccepted, + )); + }; if let Err(error) = sender.try_send(frame) { tracing::warn!( shard = self.id, target, "partition_read: inbox rejected PartitionRead frame: {error:?}" ); - return None; + return Some(PartitionReadReply::Rejected( + IggyError::TransientNotAccepted, + )); } match bus_timeout(&self.bus, PARTITION_READ_TIMEOUT, reply_rx.recv()).await { Some(Ok(reply)) => Some(reply), @@ -2107,32 +2123,6 @@ where }) } - /// Submit an auto-commit back to the partition-owning shard's pump. - /// - /// # Errors - /// Returns a refusal if the local inbox cannot accept the frame. - pub fn submit_auto_commit_offset( - &self, - request: Message, - reservation: partitions::AutoCommitReservation, - ) -> Result<(), PartitionSubmitRefused> { - let frame = ShardFrame::lifecycle(LifecycleFrame::AutoCommitSubmit { - request, - reservation, - }); - let sender = self - .senders - .get(usize::from(self.id)) - .ok_or(PartitionSubmitRefused)?; - sender.try_send(frame).map_err(|error| { - self.metrics.record_frame_drop( - crate::metrics::frame_drop_variant::PARTITION_AUTO_COMMIT, - crate::coordinator::classify_try_send_err(&error), - ); - PartitionSubmitRefused - }) - } - /// Wait out a submitted write's committed reply. /// /// `None` = reply channel dropped before a reply (view-change reset, park @@ -2215,6 +2205,10 @@ where shards_table: T, partition_consensus: PartitionConsensusConfig, ) -> Self { + // Direct owner tests can keep one disk poll outstanding. Tests needing + // concurrent reads construct a lane with their chosen capacity. + const POLL_COMPLETION_CAPACITY: usize = 1; + // Placeholder lanes: the simulator delivers frames straight to // `on_message` (see the `shard_count` note below), so nothing ever // sends here and capacity 1 exists only to satisfy the fields. The @@ -2223,6 +2217,7 @@ where // unbounded variant is wanted here either. let (_tx, inbox) = channel(1); let (_reply_tx, reply_inbox) = channel(1); + let metrics = crate::metrics::ShardMetrics::for_shard(); let nonce_seed = forward_nonce_seed(metadata.consensus.as_ref()); let plane = MuxPlane::new(variadic!(metadata, partitions)); let ShardIdentity { id, name } = identity; @@ -2234,7 +2229,6 @@ where on_client_request: std::rc::Rc::new(|_, _| {}), on_metadata_submit: std::rc::Rc::new(|_| {}), on_list_clients: std::rc::Rc::new(|_| {}), - on_partition_read: std::rc::Rc::new(|_, _, _| {}), plane, coordinator: None, senders: Vec::new(), @@ -2247,9 +2241,13 @@ where shard_count: 1, inbox, reply_inbox, + poll_completions: poll::completion::PollCompletionLane::new( + POLL_COMPLETION_CAPACITY, + &metrics, + ), shards_table, partition_consensus, - metrics: crate::metrics::ShardMetrics::for_shard(), + metrics, metadata_tick_handler: RefCell::new(None), reconcile_queue: RefCell::new(VecDeque::new()), pending_partition_frames: RefCell::new(BTreeMap::new()), @@ -9062,9 +9060,9 @@ where /// `Normal` because its window comes from the live commit frontier. This window /// comes from the parked merged log, so it runs in `ViewChange` for the replica /// that parked it. Without it the coverage scan in - /// [`Self::start_pending_partition_view`] reports an op nothing ever fetches -- - /// the sweep's gap detector needs `probe.normal` too -- and only the - /// view-change timeout moves the replica. + /// [`Self::advance_pending_partition_view`] reports an op that nothing fetches. + /// The sweep's gap detector also requires `probe.normal`, so only the view + /// change timeout moves the replica. /// /// `avoid` is the peer a stall just gave up on, so rotation lands on a /// different sender instead of the head of the same list. @@ -10463,7 +10461,7 @@ where /// Snapshot a partition's uncommitted suffix into its consensus. /// -/// Same contract as [`Self::refresh_metadata_dvc_suffix`]. The partition journal +/// Same contract as [`refresh_metadata_dvc_suffix`]. The partition journal /// is in-memory only, so after a restart it reads empty and this replica votes /// all-nack: correct, since the ops really are lost and the merge needs a peer /// that still holds them. diff --git a/core/shard/src/metrics.rs b/core/shard/src/metrics.rs index 2ac87c6be7..c5ee4f29ad 100644 --- a/core/shard/src/metrics.rs +++ b/core/shard/src/metrics.rs @@ -107,13 +107,10 @@ pub mod frame_drop_variant { /// dropped; the shard-0 deadline expiry recovers the slot / pending /// entry, so this stays informational. pub const REPLICA_HANDSHAKE_ACK: &str = "replica_handshake_ack"; - /// A poll's auto-commit submit refused by the owning shard's own inbox. - /// - /// Its own series, not `PARTITION`: the poll is answered with a retriable - /// status and no frame of the client's was dropped, so counting it with - /// shed frames would read as a routing loss. - pub const PARTITION_AUTO_COMMIT: &str = "partition_auto_commit"; pub const PARTITION_PERSISTENCE_COMPLETED: &str = "partition_persistence_completed"; + /// A disk poll could not reserve completion capacity or return its result + /// to the owning shard. + pub const PARTITION_POLL_COMPLETION: &str = "partition_poll_completion"; } /// Reason labels used in `frame_drops_total`. @@ -158,10 +155,9 @@ pub mod frame_drop_reason { pub const SUBMIT_TIMEOUT: &str = "submit_timeout"; } -// The tables only index the lazy fast-path cache below; a `{variant, reason}` +// The tables only index the lazy cache below. A `{variant, reason}` // pair enters the `Family` (and therefore the scrape) the first time a drop -// site actually produces it, so the unreachable corners of the 7 x 9 cross -// product never appear as permanent zero-valued series. +// site produces it, so unproduced combinations never enter the scrape. const VARIANT_COUNT: usize = 9; const REASON_COUNT: usize = 11; @@ -173,8 +169,8 @@ const VARIANTS: [&str; VARIANT_COUNT] = [ frame_drop_variant::FORWARD_REPLICA_SEND, frame_drop_variant::METADATA_COMMIT_TICK, frame_drop_variant::REPLICA_HANDSHAKE_ACK, - frame_drop_variant::PARTITION_AUTO_COMMIT, frame_drop_variant::PARTITION_PERSISTENCE_COMPLETED, + frame_drop_variant::PARTITION_POLL_COMPLETION, ]; const REASONS: [&str; REASON_COUNT] = [ @@ -233,8 +229,7 @@ pub struct ShardMetrics { partition_wal_prepares: Counter, partition_wal_checkpoints: Counter, partition_wal_errors: Counter, - frame_drops_total: Family, - cached_counters: Arc<[[OnceLock; REASON_COUNT]; VARIANT_COUNT]>, + frame_drops: FrameDropMetrics, partitions_materialised_total: Counter, partitions_removed_total: Counter, partitions_reconcile_failures_total: Counter, @@ -302,8 +297,10 @@ impl ShardMetrics { partition_wal_prepares: Counter::default(), partition_wal_checkpoints: Counter::default(), partition_wal_errors: Counter::default(), - frame_drops_total, - cached_counters, + frame_drops: FrameDropMetrics { + total: frame_drops_total, + cached_counters, + }, partitions_materialised_total: Counter::default(), partitions_removed_total: Counter::default(), partitions_reconcile_failures_total: Counter::default(), @@ -390,9 +387,8 @@ impl ShardMetrics { ); } - /// Best effort: counts explicit client denials read off the reply status - /// and the poll-side reservation refusals. A denial the pump answers to an - /// auto-commit submit has no client reply to read and is not counted. + /// Count consumer offset capacity denials from explicit client requests + /// and automatic commit admission during poll completion. pub fn record_consumer_offset_denied(&self, kind: ConsumerKind) { self.consumer_offset_denied_counters[consumer_kind_index(kind)].inc(); } @@ -459,19 +455,12 @@ impl ShardMetrics { /// path so accounting is preserved even if a future caller forgets /// to extend the const tables above. pub fn record_frame_drop(&self, variant: &'static str, reason: &'static str) { - if let (Some(v_idx), Some(r_idx)) = (variant_index(variant), reason_index(reason)) { - self.cached_counters[v_idx][r_idx] - .get_or_init(|| { - self.frame_drops_total - .get_or_create(&FrameDropLabel { variant, reason }) - .clone() - }) - .inc(); - } else { - self.frame_drops_total - .get_or_create(&FrameDropLabel { variant, reason }) - .inc(); - } + self.frame_drops.record(variant, reason); + } + + /// The completion lane shares these handles once during shard setup. + pub(crate) const fn frame_drop_metrics(&self) -> &FrameDropMetrics { + &self.frame_drops } /// Bumped on the owning shard each time the partition reconciliation @@ -554,7 +543,8 @@ impl ShardMetrics { #[cfg(any(test, feature = "simulator"))] #[must_use] pub fn frame_drops_value(&self) -> u64 { - self.cached_counters + self.frame_drops + .cached_counters .iter() .flatten() .filter_map(OnceLock::get) @@ -574,7 +564,8 @@ impl ShardMetrics { let Some(reason_idx) = reason_index(reason) else { return 0; }; - self.cached_counters + self.frame_drops + .cached_counters .iter() .filter_map(|variant| variant[reason_idx].get()) .map(prometheus_client::metrics::counter::Counter::get) @@ -728,7 +719,7 @@ impl ShardMetrics { #[must_use] pub fn frame_drop_count(&self, variant: &'static str, reason: &'static str) -> u64 { match (variant_index(variant), reason_index(reason)) { - (Some(v_idx), Some(r_idx)) => self.cached_counters[v_idx][r_idx] + (Some(v_idx), Some(r_idx)) => self.frame_drops.cached_counters[v_idx][r_idx] .get() .map_or(0, prometheus_client::metrics::counter::Counter::get), _ => 0, @@ -745,7 +736,7 @@ impl ShardMetrics { registry.register( "frame_drops", "frames shed instead of delivered, by frame class and refusal reason", - self.frame_drops_total.clone(), + self.frame_drops.total.clone(), ); registry.register( "partitions_materialised", @@ -827,6 +818,35 @@ impl ShardMetrics { } } +/// Frame drop accounting shared with the registered shard metrics. +/// Clones retain the same family and lazy cache, so detached work records +/// visible drops without carrying unrelated counters or creating unused series. +#[derive(Clone)] +pub(crate) struct FrameDropMetrics { + total: Family, + cached_counters: Arc<[[OnceLock; REASON_COUNT]; VARIANT_COUNT]>, +} + +impl FrameDropMetrics { + pub(crate) fn record(&self, variant: &'static str, reason: &'static str) { + if let (Some(variant_index), Some(reason_index)) = + (variant_index(variant), reason_index(reason)) + { + self.cached_counters[variant_index][reason_index] + .get_or_init(|| { + self.total + .get_or_create(&FrameDropLabel { variant, reason }) + .clone() + }) + .inc(); + } else { + self.total + .get_or_create(&FrameDropLabel { variant, reason }) + .inc(); + } + } +} + #[cfg(test)] mod tests { use super::*; @@ -843,7 +863,8 @@ mod tests { let count = |variant, reason| { metrics - .frame_drops_total + .frame_drops + .total .get_or_create(&FrameDropLabel { variant, reason }) .get() }; @@ -944,7 +965,8 @@ mod tests { metrics.record_frame_drop(frame_drop_variant::PARTITION, frame_drop_reason::UNROUTABLE); } let from_family = metrics - .frame_drops_total + .frame_drops + .total .get_or_create(&FrameDropLabel { variant: frame_drop_variant::PARTITION, reason: frame_drop_reason::UNROUTABLE, @@ -990,7 +1012,8 @@ mod tests { let metrics = ShardMetrics::for_shard(); metrics.record_frame_drop("unexpected_variant", "unexpected_reason"); let from_family = metrics - .frame_drops_total + .frame_drops + .total .get_or_create(&FrameDropLabel { variant: "unexpected_variant", reason: "unexpected_reason", diff --git a/core/shard/src/poll.rs b/core/shard/src/poll.rs new file mode 100644 index 0000000000..ca9e86e495 --- /dev/null +++ b/core/shard/src/poll.rs @@ -0,0 +1,257 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Poll reads complete under the partition owner's control. +//! +//! 1. The owner snapshots the history identity and read resources. +//! 2. Resident reads complete inline. Disk reads reserve completion capacity +//! before detached I/O and return through the owner's completion lane. +//! 3. The owner checks the reply connection, history, and recovery state, admits +//! any automatic commit, and updates progress before releasing the reply. +//! +//! A disk read can yield while purge or state transfer replaces the history, +//! even on the same shard thread. The detached task therefore cannot advance +//! progress or authorize a successful reply. Completion validation and progress +//! updates run synchronously on the owner, before any replication wait. + +use crate::shards_table::ShardsTable; +use crate::{IggyShard, PartitionRead, PartitionReadReply, Sender}; +use consensus::PartitionsHandle; +use iggy_common::IggyError; +use journal::superblock::SuperblockStore; +use message_bus::MessageBus; +use partitions::{PollCompletion, PollPlan, PollReadResult}; +use server_common::sharding::IggyNamespace; + +pub mod completion; +#[cfg(test)] +mod completion_tests; +#[cfg(test)] +mod test_support; +#[cfg(test)] +mod timeout_tests; + +/// A read result awaiting acceptance by its partition owner. +/// Disk tasks send it through the reserved completion lane. Resident reads +/// pass it directly to the same completion handler. +/// +/// The channel requires `Send` even for messages addressed to this same shard. +/// Completion capacity is released on dequeue. Consumer offset capacity is +/// reserved separately during owner acceptance; those guards stay in the +/// owner's local request queue or replication state, outside this channel. +pub struct PollCompleted { + /// Namespace whose current partition must validate the captured history. + namespace: IggyNamespace, + /// Snapshot facts that have not yet been accepted as consumer progress. + result: PollReadResult, + /// Return path for the accepted read or a rejection. + reply: Sender, + /// Inbox enqueue time for disk diagnostics, or `None` for resident completion. + #[cfg(feature = "poll-diagnostics")] + queued_at: Option, +} + +impl IggyShard +where + B: MessageBus + 'static, + T: ShardsTable, + SB: SuperblockStore, +{ + /// Execute a routed read on the partition owner's pump. + /// Partitions missing materialized data reject the read. Resident polls finish + /// inline. Disk polls return to the pump for acceptance after detached I/O. + #[allow(clippy::future_not_send)] + pub(crate) async fn on_partition_read( + &self, + namespace: IggyNamespace, + read: PartitionRead, + reply: Sender, + ) { + let partitions = self.plane.partitions(); + if partitions.with_partition( + &namespace, + partitions::IggyPartition::requires_state_transfer, + ) == Some(true) + { + let _ = reply.try_send(PartitionReadReply::Rejected( + IggyError::TransientNotAccepted, + )); + return; + } + let result = match read { + PartitionRead::Poll { consumer, args } => { + match partitions.build_poll_snapshot(&namespace, consumer, &args) { + None => PartitionReadReply::NotFound, + Some(plan) if plan.needs_off_pump_io() => { + let route_failure = self.senders.get(usize::from(self.id)).map_or( + Some(crate::metrics::frame_drop_reason::UNROUTABLE), + |sender| { + sender + .is_disconnected() + .then_some(crate::metrics::frame_drop_reason::DISCONNECTED) + }, + ); + if let Some(reason) = route_failure { + completion::reject(&reply, self.metrics.frame_drop_metrics(), reason); + return; + } + let Some(completion) = self.poll_completions.try_reserve(namespace, reply) + else { + return; + }; + #[cfg(feature = "poll-diagnostics")] + tracing::debug!( + target: "iggy.shard.poll_diagnostics", + namespace_raw = namespace.inner(), + phase = "dispatch", + tier = "disk", + "partition poll dispatch" + ); + self.bus.spawn(read_poll(namespace, plan, completion)); + return; + } + Some(plan) => { + #[cfg(feature = "poll-diagnostics")] + tracing::debug!( + target: "iggy.shard.poll_diagnostics", + namespace_raw = namespace.inner(), + phase = "dispatch", + tier = "resident", + "partition poll dispatch" + ); + self.on_poll_completed(PollCompleted { + namespace, + result: plan.execute_resident(), + reply, + #[cfg(feature = "poll-diagnostics")] + queued_at: None, + }) + .await; + return; + } + } + } + PartitionRead::ConsumerOffset { consumer } => partitions + .consumer_offset_read(&namespace, consumer) + .map_or(PartitionReadReply::NotFound, |(stored, current_offset)| { + PartitionReadReply::ConsumerOffset { + stored, + current_offset, + } + }), + PartitionRead::GroupOffsetState { group_id } => partitions + .group_offset_state(&namespace, group_id) + .map_or(PartitionReadReply::NotFound, |(last_polled, committed)| { + PartitionReadReply::GroupOffsetState { + last_polled, + committed, + } + }), + PartitionRead::ClearGroupLastPolled { group_id } => partitions + .clear_group_last_polled(&namespace, group_id) + .map_or(PartitionReadReply::NotFound, |()| PartitionReadReply::Ack), + PartitionRead::ResolveSegmentDeleteOffset { count } => partitions + .segment_delete_resolution(&namespace, count) + .map_or(PartitionReadReply::NotFound, |(up_to_offset, lagging)| { + PartitionReadReply::SegmentDeleteOffset { + up_to_offset, + lagging, + } + }), + }; + let _ = reply.try_send(result); + } + + /// Discard a read if its reply receiver is already disconnected at the check. + /// Otherwise accept it on the owner's pump and attempt the reply before + /// replication, which may suspend. Another shard can disconnect after the + /// check, so delivery is not guaranteed and admitted progress is not rolled + /// back if the reply fails. + #[allow(clippy::future_not_send)] + pub(crate) async fn on_poll_completed(&self, completion: PollCompleted) { + let PollCompleted { + namespace, + result, + reply, + #[cfg(feature = "poll-diagnostics")] + queued_at, + } = completion; + #[cfg(feature = "poll-diagnostics")] + if let Some(queued_at) = queued_at { + tracing::debug!( + target: "iggy.shard.poll_diagnostics", + namespace_raw = namespace.inner(), + phase = "completion", + tier = "disk", + queue_wait_us = u64::try_from(queued_at.elapsed().as_micros()).unwrap_or(u64::MAX), + "partition poll completion" + ); + } + if reply.is_disconnected() { + return; + } + let partitions = self.plane.partitions(); + let consumer_kind = result.consumer_kind(); + match partitions.complete_poll(&namespace, result) { + Ok(PollCompletion { + fragments, + current_offset, + replication, + }) => { + // The owner has admitted progress, but the offset may still be + // queued. Release the reply before waiting for replication. + // A poll reply does not acknowledge a durable offset commit. + let _ = reply.try_send(PartitionReadReply::Poll { + fragments, + current_offset, + }); + if let Some(replication) = replication { + partitions + .replicate_poll_completion(&namespace, replication) + .await; + } + } + Err(error) => { + if matches!(error, IggyError::TooManyConsumerOffsets) { + self.metrics.record_consumer_offset_denied(consumer_kind); + } + let _ = reply.try_send(PartitionReadReply::Rejected(error)); + } + } + } +} + +/// Read an owned snapshot without borrowing the partition or changing progress. +/// The completion sender returns the result to the owner for acceptance. +#[allow(clippy::future_not_send)] +async fn read_poll( + namespace: IggyNamespace, + plan: PollPlan, + completion: completion::PollCompletionSender, +) { + let poll_started = std::time::Instant::now(); + let result = plan.execute().await; + let elapsed = poll_started.elapsed(); + if elapsed > std::time::Duration::from_secs(1) { + tracing::warn!( + namespace_raw = namespace.inner(), + elapsed_ms = u64::try_from(elapsed.as_millis()).unwrap_or(u64::MAX), + "slow partition poll; gather side may have timed out" + ); + } + completion.complete(result); +} diff --git a/core/shard/src/poll/completion.rs b/core/shard/src/poll/completion.rs new file mode 100644 index 0000000000..186e7d2719 --- /dev/null +++ b/core/shard/src/poll/completion.rs @@ -0,0 +1,507 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Capacity belongs to a disk read from dispatch until its result is dequeued. +//! Only a reservation can enqueue a completion, so ordinary work cannot occupy +//! its space and a finished read never waits for space. The pump still decides +//! whether the result may advance progress or produce a successful reply. +//! +//! For capacity N, running reads plus queued results never exceed N. A requester +//! timeout does not cancel detached I/O, so its reservation remains occupied. +//! Dropping the read, dequeuing its result, or discarding it releases the slot. +//! +//! Reservation, delivery, and closure run on the owner's shard. Delivery never +//! yields between checking closure and enqueueing, so the pump cannot close and +//! drain the lane between those steps. The atomic flag alone would not provide +//! that ordering if delivery moved to another thread. + +use std::sync::Arc; +use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; + +use crossfire::{RecvError, TryRecvError, TrySendError}; +use iggy_common::IggyError; +use partitions::PollReadResult; +use server_common::sharding::IggyNamespace; + +use super::PollCompleted; +use crate::coordinator::classify_try_send_err; +use crate::metrics::{FrameDropMetrics, ShardMetrics, frame_drop_reason, frame_drop_variant}; +use crate::{PartitionReadReply, Receiver, Sender, channel}; + +/// A private bounded lane, with capacity shared by pending reads and results. +/// It lives on the owner; its raw sender cannot bypass reservation accounting. +pub struct PollCompletionLane { + sender: Sender, + receiver: Receiver, + state: Arc, +} + +impl PollCompletionLane { + /// Create a lane with the same limit for reservations and queued results. + /// + /// # Panics + /// + /// Panics if `capacity` is zero. + pub(crate) fn new(capacity: usize, metrics: &ShardMetrics) -> Self { + assert!(capacity > 0, "poll completion capacity must be nonzero"); + let (sender, receiver) = channel(capacity); + Self { + sender, + receiver, + state: Arc::new(LaneState { + capacity, + reserved: AtomicUsize::new(0), + closed: AtomicBool::new(false), + metrics: metrics.frame_drop_metrics().clone(), + }), + } + } + + /// Reserve before starting I/O. Exhaustion or shutdown rejects the poll + /// immediately, without reading data or waiting on the owner's pump. + pub(crate) fn try_reserve( + &self, + namespace: IggyNamespace, + reply: Sender, + ) -> Option { + if self.state.closed.load(Ordering::Relaxed) || self.sender.is_disconnected() { + reject(&reply, &self.state.metrics, frame_drop_reason::DISCONNECTED); + return None; + } + if self + .state + .reserved + .fetch_update(Ordering::Relaxed, Ordering::Relaxed, |reserved| { + (reserved < self.state.capacity).then_some(reserved + 1) + }) + .is_err() + { + reject(&reply, &self.state.metrics, frame_drop_reason::FULL); + return None; + } + Some(PollCompletionSender { + inbox: self.sender.clone(), + slot: CompletionSlot { + state: self.state.clone(), + }, + namespace, + reply, + }) + } + + /// Dequeueing frees capacity before owner validation can await replication. + #[allow(clippy::future_not_send)] + pub(crate) async fn recv(&self) -> Result, RecvError> { + self.receiver + .recv() + .await + .map(QueuedCompletion::into_result) + } + + pub(crate) fn try_recv(&self) -> Result, TryRecvError> { + self.receiver.try_recv().map(QueuedCompletion::into_result) + } + + #[cfg(any(test, feature = "simulator"))] + pub(crate) fn len(&self) -> usize { + self.receiver.len() + } + + /// Stop admission and late delivery on the owning shard while leaving + /// queued results available for its graceful drain. Closing never accepts + /// consumer progress. + pub(crate) fn close(&self) { + self.state.closed.store(true, Ordering::Relaxed); + } + + /// Close on cancellation too: the shard can outlive its borrowed pump. + pub(crate) const fn pump_guard(&self) -> CompletionPumpGuard<'_> { + CompletionPumpGuard { lane: self } + } +} + +impl Drop for PollCompletionLane { + fn drop(&mut self) { + self.close(); + // Crossfire senders may outlive the receiver. Drain explicitly so + // queued permits and reply senders do not wait for those tasks to end. + while self.receiver.try_recv().is_ok() {} + } +} + +/// A detached read can reject delivery, but cannot authorize a successful poll. +/// Its slot follows the result into the queue. Caller cancellation retains +/// capacity until the owner dequeues and discards the result. Dropping this +/// sender instead abandons the read and releases its reservation. +pub struct PollCompletionSender { + inbox: Sender, + slot: CompletionSlot, + namespace: IggyNamespace, + reply: Sender, +} + +impl PollCompletionSender { + /// Transfer the reservation into the completion queue without waiting. + /// Must run on the owning shard, like admission and closure. A closed + /// owner rejects the result and releases the reservation. + pub(crate) fn complete(self, result: PollReadResult) { + if self.slot.state.closed.load(Ordering::Relaxed) || self.inbox.is_disconnected() { + reject( + &self.reply, + &self.slot.state.metrics, + frame_drop_reason::DISCONNECTED, + ); + return; + } + let completion = QueuedCompletion { + result: Box::new(PollCompleted { + namespace: self.namespace, + result, + reply: self.reply, + #[cfg(feature = "poll-diagnostics")] + queued_at: Some(std::time::Instant::now()), + }), + slot: self.slot, + }; + if let Err(error) = self.inbox.try_send(completion) { + let reason = classify_try_send_err(&error); + let completion = match error { + TrySendError::Full(completion) => { + debug_assert!(false, "reserved poll completion slot was unavailable"); + completion + } + TrySendError::Disconnected(completion) => completion, + }; + reject( + &completion.result.reply, + &completion.slot.state.metrics, + reason, + ); + } + } +} + +/// Closes admission and abandons any undrained results when the pump exits, +/// including when its future is canceled while the shard remains alive. +pub struct CompletionPumpGuard<'lane> { + lane: &'lane PollCompletionLane, +} + +impl Drop for CompletionPumpGuard<'_> { + fn drop(&mut self) { + self.lane.close(); + // Graceful shutdown already delivered its queue. Cancellation or a + // fatal commit instead abandons results without accepting progress. + while self.lane.receiver.try_recv().is_ok() {} + } +} + +/// Report a completion route or admission failure without changing progress. +pub(super) fn reject( + reply: &Sender, + metrics: &FrameDropMetrics, + reason: &'static str, +) { + metrics.record(frame_drop_variant::PARTITION_POLL_COMPLETION, reason); + let _ = reply.try_send(PartitionReadReply::Rejected( + IggyError::TransientNotAccepted, + )); +} + +struct LaneState { + capacity: usize, + /// Includes slots held by running reads and by results awaiting dequeue. + reserved: AtomicUsize, + /// Closing stops both new reservations and delivery by existing reads. + closed: AtomicBool, + /// Share the registered family and cache through the existing slot handle. + /// Disk reads need no additional metric handle clones at dispatch. + metrics: FrameDropMetrics, +} + +/// Unique ownership of one slot, transferred from read to queued completion. +struct CompletionSlot { + state: Arc, +} + +impl Drop for CompletionSlot { + fn drop(&mut self) { + let previous = self.state.reserved.fetch_sub(1, Ordering::Relaxed); + debug_assert!(previous > 0, "poll completion reservation underflow"); + } +} + +struct QueuedCompletion { + result: Box, + /// Kept until dequeue, even after the disk task has returned. + slot: CompletionSlot, +} + +impl QueuedCompletion { + fn into_result(self) -> Box { + self.result + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + use std::sync::atomic::Ordering; + + use consensus::{LocalPipeline, VsrConsensus}; + use iggy_common::{IggyByteSize, IggyError, PartitionStats, PollingStrategy}; + use message_bus::IggyMessageBus; + use partitions::{ + IggyPartition, IggyPartitions, PartitionPathLayout, PartitionsConfig, PollReadResult, + PollingArgs, PollingConsumer, + }; + use prometheus_client::encoding::text::encode; + use prometheus_client::registry::Registry; + use server_common::sharding::{IggyNamespace, ShardId}; + + use super::{PollCompletionLane, PollCompletionSender}; + use crate::metrics::{ShardMetrics, frame_drop_reason, frame_drop_variant}; + use crate::{PartitionReadReply, Receiver, channel}; + + #[test] + fn given_pending_read_when_capacity_is_reserved_should_reject_another_reservation() { + let metrics = ShardMetrics::for_shard(); + let lane = PollCompletionLane::new(1, &metrics); + let (pending_read, _pending_replies) = reserve_read(&lane); + let (reply, rejected_replies) = channel(1); + + assert!(lane.try_reserve(namespace(), reply).is_none()); + assert!(matches!( + rejected_replies.try_recv(), + Ok(PartitionReadReply::Rejected( + IggyError::TransientNotAccepted + )) + )); + assert_eq!(metrics.frame_drops_value(), 1); + assert_eq!(lane.len(), 0, "the first read still owns the empty slot"); + + // Canceling the detached read releases capacity without a result. + drop(pending_read); + let (_next_read, _next_replies) = reserve_read(&lane); + } + + #[test] + fn given_queued_result_when_read_finishes_should_keep_reservation_until_dequeue() { + let metrics = ShardMetrics::for_shard(); + let lane = PollCompletionLane::new(1, &metrics); + let (read, replies) = reserve_read(&lane); + read.complete(read_empty_partition()); + assert_eq!(lane.len(), 1); + assert!(matches!( + replies.try_recv(), + Err(crossfire::TryRecvError::Empty) + )); + + let (next_reply, _next_replies) = channel(1); + assert!(lane.try_reserve(namespace(), next_reply).is_none()); + + // Dequeue frees capacity even while owner validation still holds bytes. + let _result_for_owner = lane.try_recv().expect("result reaches owner"); + let (_next_read, _next_replies) = reserve_read(&lane); + assert!(matches!( + replies.try_recv(), + Err(crossfire::TryRecvError::Empty) + )); + } + + #[test] + fn given_closed_caller_when_read_is_pending_should_retain_reservation() { + let metrics = ShardMetrics::for_shard(); + let lane = PollCompletionLane::new(1, &metrics); + let (pending_read, replies) = reserve_read(&lane); + drop(replies); + let (next_reply, _next_replies) = channel(1); + assert!(lane.try_reserve(namespace(), next_reply).is_none()); + + pending_read.complete(read_empty_partition()); + assert_eq!( + lane.len(), + 1, + "caller cancellation retains capacity until the owner discards the result" + ); + let _result_for_owner = lane + .try_recv() + .expect("owner dequeues the late result before discarding it"); + let (_next_read, _next_replies) = reserve_read(&lane); + } + + #[test] + fn given_closed_lane_when_reserved_read_finishes_should_reject_without_enqueue() { + let metrics = ShardMetrics::for_shard(); + let lane = PollCompletionLane::new(1, &metrics); + let (pending_read, replies) = reserve_read(&lane); + lane.close(); + pending_read.complete(read_empty_partition()); + + assert!(matches!( + replies.try_recv(), + Ok(PartitionReadReply::Rejected( + IggyError::TransientNotAccepted + )) + )); + assert_eq!(lane.len(), 0); + assert_eq!(lane.state.reserved.load(Ordering::Relaxed), 0); + let (next_reply, next_replies) = channel(1); + assert!(lane.try_reserve(namespace(), next_reply).is_none()); + assert!(matches!( + next_replies.try_recv(), + Ok(PartitionReadReply::Rejected( + IggyError::TransientNotAccepted + )) + )); + } + + #[test] + fn given_dropped_lane_when_read_is_pending_should_release_queued_and_pending_slots() { + let metrics = ShardMetrics::for_shard(); + let mut registry = Registry::default(); + metrics.register(&mut registry); + let lane = PollCompletionLane::new(2, &metrics); + let state = lane.state.clone(); + let (queued_read, queued_replies) = reserve_read(&lane); + let (pending_read, pending_replies) = reserve_read(&lane); + queued_read.complete(read_empty_partition()); + + let mut scrape = String::new(); + encode(&mut scrape, ®istry).expect("metrics are encodable"); + assert!( + !scrape.contains(frame_drop_variant::PARTITION_POLL_COMPLETION), + "reserving and delivering reads must not create unused drop series", + ); + drop(lane); + + assert_eq!(state.reserved.load(Ordering::Relaxed), 1); + assert!(matches!( + queued_replies.try_recv(), + Err(crossfire::TryRecvError::Disconnected) + )); + pending_read.complete(read_empty_partition()); + assert!(matches!( + pending_replies.try_recv(), + Ok(PartitionReadReply::Rejected( + IggyError::TransientNotAccepted + )) + )); + assert_eq!(state.reserved.load(Ordering::Relaxed), 0); + assert_eq!( + metrics.frame_drop_count( + frame_drop_variant::PARTITION_POLL_COMPLETION, + frame_drop_reason::DISCONNECTED, + ), + 1, + "the late read must record its drop in the original shard metrics", + ); + scrape.clear(); + encode(&mut scrape, ®istry).expect("metrics are encodable"); + assert!( + scrape.lines().any(|line| { + line.starts_with("frame_drops_total{") + && line.contains("variant=\"partition_poll_completion\"") + && line.contains("reason=\"disconnected\"") + && line.ends_with(" 1") + }), + "the registered family must observe the late read's drop" + ); + } + + #[test] + fn given_canceled_pump_when_completions_exist_should_abandon_without_acceptance() { + let metrics = ShardMetrics::for_shard(); + let lane = PollCompletionLane::new(2, &metrics); + let pump = lane.pump_guard(); + let (queued_read, queued_replies) = reserve_read(&lane); + let (pending_read, pending_replies) = reserve_read(&lane); + queued_read.complete(read_empty_partition()); + drop(pump); + + assert_eq!(lane.len(), 0); + assert!(matches!( + queued_replies.try_recv(), + Err(crossfire::TryRecvError::Disconnected) + )); + pending_read.complete(read_empty_partition()); + assert!(matches!( + pending_replies.try_recv(), + Ok(PartitionReadReply::Rejected( + IggyError::TransientNotAccepted + )) + )); + assert_eq!(lane.state.reserved.load(Ordering::Relaxed), 0); + } + + fn reserve_read( + lane: &PollCompletionLane, + ) -> (PollCompletionSender, Receiver) { + let (reply, replies) = channel(1); + let reservation = lane + .try_reserve(namespace(), reply) + .expect("scenario has capacity for this read"); + (reservation, replies) + } + + fn namespace() -> IggyNamespace { + IggyNamespace::new(1, 1, 0) + } + + /// Read an empty partition without invoking owner acceptance. These tests + /// exercise completion delivery, so no messages or disk I/O are needed. + fn read_empty_partition() -> PollReadResult { + let partitions = IggyPartitions::::new( + ShardId::new(0), + PartitionsConfig { + messages_required_to_save: 1, + size_of_messages_required_to_save: IggyByteSize::from(1024_u64), + validate_checksum: true, + segment_size: IggyByteSize::from(1_048_576_u64), + preallocate_segments: false, + encryptor: None, + path_layout: PartitionPathLayout::default(), + }, + ); + let consensus = VsrConsensus::new( + 1, + 0, + 1, + namespace().inner(), + IggyMessageBus::new(0), + LocalPipeline::new(), + ); + partitions.insert( + namespace(), + IggyPartition::new(Arc::new(PartitionStats::default()), consensus), + ); + let consumer_id = 7; + let partition_id = 0; + partitions + .build_poll_snapshot( + &namespace(), + PollingConsumer::Consumer(consumer_id, partition_id), + &PollingArgs { + strategy: PollingStrategy::next(), + count: 0, + auto_commit: true, + }, + ) + .expect("partition has a read snapshot") + .execute_resident() + } +} diff --git a/core/shard/src/poll/completion_tests.rs b/core/shard/src/poll/completion_tests.rs new file mode 100644 index 0000000000..5d3cac3baa --- /dev/null +++ b/core/shard/src/poll/completion_tests.rs @@ -0,0 +1,506 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use std::future::Future; +use std::rc::Rc; +use std::sync::Arc; +use std::sync::atomic::{AtomicBool, Ordering}; +use std::task::{Context, Poll, Wake}; + +use consensus::PartitionsHandle; +use iggy_common::{IggyError, PollingStrategy}; +use journal::prepare_journal::PrepareJournal; +use message_bus::IggyMessageBus; +use metadata::IggyMetadata; +use metadata::impls::metadata::IggySnapshot; +use partitions::{IggyPartitions, PartitionsConfig, PollingArgs, PollingConsumer}; +use server_common::send_messages::decode_batch_slice; +use server_common::sharding::{IggyNamespace, PartitionLocation, ShardId}; + +use super::test_support::{PollTestMetadata, partition_with_messages}; +use crate::metrics::ShardMetrics; +use crate::shards_table::{PapayaShardsTable, ShardsTable}; +use crate::{ + IggyShard, LifecycleFrame, PartitionConsensusConfig, PartitionRead, PartitionReadReply, + Receiver, ReplicaTopology, ShardFrame, ShardIdentity, TaggedSender, channel, shard_channel, +}; + +/// Replacement can reuse every message offset from the old history. The pump +/// must reject the old completion by history, then accept a fresh completion +/// through the same completion lane without inheriting stale group progress. +#[compio::test] +#[allow(clippy::too_many_lines)] +async fn given_pending_group_read_when_partition_is_replaced_should_reject_stale_completion_through_owner_pump() + { + let namespace = IggyNamespace::new(1, 1, 0); + let group_id = 7; + let consumer = PollingConsumer::ConsumerGroup( + usize::try_from(group_id).expect("group id fits the consumer key"), + 0, + ); + let bus = Rc::new(IggyMessageBus::new(0)); + let old_payloads = ["old zero", "old one", "old two"]; + let (old_partition, config) = partition_with_messages(&bus, namespace, &old_payloads).await; + let (owner, _owner_sender) = owner_with_inbox(&bus, config, namespace); + let partitions = owner.plane.partitions(); + partitions.insert(namespace, old_partition); + let poll_args = PollingArgs { + strategy: PollingStrategy::offset(0), + count: 3, + auto_commit: true, + }; + + // Read the committed old batch but hold its result before owner acceptance. + // Resident bytes make the release ordering explicit without disk timing. + let (stale_reply_sender, stale_replies) = channel(1); + let old_completion = owner + .poll_completions + .try_reserve(namespace, stale_reply_sender) + .expect("reserve the old read before executing it"); + let old_plan = partitions + .build_poll_snapshot(&namespace, consumer, &poll_args) + .expect("old partition has a read snapshot"); + assert!(!old_plan.needs_off_pump_io()); + let delayed_result = old_plan.execute_resident(); + + // Reuse offsets 0 through 2 in a different history. Checking only that the + // old offset fits the current partition would wrongly accept this result. + // The pump has not started, so no partition borrow can span replacement. + let fresh_payloads = ["fresh zero", "fresh one", "fresh two"]; + let (replacement, _) = partition_with_messages(&bus, namespace, &fresh_payloads).await; + drop( + partitions + .remove(&namespace) + .expect("remove the old history"), + ); + partitions.insert(namespace, replacement); + let (last_polled, committed) = partitions.group_offset_state(&namespace, group_id).unwrap(); + assert_eq!(last_polled, None); + assert_eq!(committed, None); + + // Keep stop open and drive the real pump only after each completion is + // queued. Dropping the pump at the end avoids an unrelated shutdown flush. + let (_stop_sender, stop_receiver) = channel(1); + let pump = owner.run_message_pump(stop_receiver, Arc::new(AtomicBool::new(false))); + futures::pin_mut!(pump); + old_completion.complete(delayed_result); + assert_eq!( + owner.poll_completion_inbox_len(), + 1, + "stale result is queued for owner validation" + ); + assert!( + matches!( + stale_replies.try_recv(), + Err(crossfire::TryRecvError::Empty) + ), + "the sender must leave rejection to the owner" + ); + assert!(futures::poll!(pump.as_mut()).is_pending()); + let stale_reply = stale_replies + .try_recv() + .expect("owner processed the stale completion"); + + // Check both offsets before a fresh read can hide a stale update. These + // assertions also expose nonempty stale admission if the history check fails. + let (last_polled, committed) = partitions.group_offset_state(&namespace, group_id).unwrap(); + assert_eq!( + last_polled, None, + "stale completion must not restore the group's last polled offset" + ); + assert_eq!( + committed, None, + "stale completion must not admit an automatic commit" + ); + assert!( + matches!( + stale_reply, + PartitionReadReply::Rejected(IggyError::TransientNotAccepted) + ), + "old history must be rejected, got {stale_reply:?}" + ); + + // A fresh result takes the same completion lane and pump. Distinct + // payloads prove the reply belongs to the replacement at the reused offsets. + let (fresh_reply_sender, fresh_replies) = channel(1); + let fresh_completion = owner + .poll_completions + .try_reserve(namespace, fresh_reply_sender) + .expect("reserve the fresh read before executing it"); + let fresh_plan = partitions + .build_poll_snapshot(&namespace, consumer, &poll_args) + .expect("replacement has a read snapshot"); + assert!(!fresh_plan.needs_off_pump_io()); + let fresh_result = fresh_plan.execute_resident(); + fresh_completion.complete(fresh_result); + assert_eq!( + owner.poll_completion_inbox_len(), + 1, + "fresh result uses the same completion lane" + ); + assert!( + matches!( + fresh_replies.try_recv(), + Err(crossfire::TryRecvError::Empty) + ), + "the sender must leave success to the owner" + ); + assert!(futures::poll!(pump.as_mut()).is_pending()); + let PartitionReadReply::Poll { + fragments, + current_offset, + } = fresh_replies + .try_recv() + .expect("owner processed the fresh completion") + else { + panic!("fresh history should produce a successful poll reply"); + }; + assert_eq!(current_offset, 2); + let bytes: Vec = fragments + .iter() + .flat_map(|fragment| fragment.as_slice().iter().copied()) + .collect(); + let batch = decode_batch_slice(&bytes).expect("decode the fresh batch"); + let offsets: Vec = batch + .iter() + .map(|message| batch.header.base_offset + u64::from(message.header.offset_delta)) + .collect(); + let payloads: Vec<&[u8]> = batch.iter().map(|message| message.payload).collect(); + assert_eq!(offsets, vec![0, 1, 2]); + assert_eq!(payloads, fresh_payloads.map(str::as_bytes)); + let (last_polled, committed) = partitions.group_offset_state(&namespace, group_id).unwrap(); + assert_eq!(last_polled, Some(2)); + assert_eq!( + committed, + Some(2), + "fresh acceptance advances the stored offset locally" + ); +} + +/// A full ordinary inbox must not refuse a completed read. Interleaving offset +/// queries with two reads also proves neither lane drains its whole backlog +/// before giving the other lane a turn. +#[compio::test] +#[allow(clippy::too_many_lines)] +async fn given_full_owner_inbox_when_reserved_reads_complete_should_interleave_both_lanes() { + let namespace = IggyNamespace::new(1, 1, 0); + let group_id = 7; + let consumer = PollingConsumer::ConsumerGroup( + usize::try_from(group_id).expect("group id fits the consumer key"), + 0, + ); + let bus = Rc::new(IggyMessageBus::new(0)); + let payloads = ["first completion", "second completion"]; + let (partition, config) = partition_with_messages(&bus, namespace, &payloads).await; + let (owner, owner_sender) = owner_with_inbox(&bus, config, namespace); + let partitions = owner.plane.partitions(); + partitions.insert(namespace, partition); + let mut delayed_reads = Vec::new(); + let mut poll_replies = Vec::new(); + + // Reserve both reads before executing them. Each nonempty result advances + // the same group's progress by one offset, making acceptance order visible. + for offset in [0, 1] { + let (reply_sender, replies) = channel(1); + let completion = owner + .poll_completions + .try_reserve(namespace, reply_sender) + .expect("reserve completion capacity before reading"); + let plan = partitions + .build_poll_snapshot( + &namespace, + consumer, + &PollingArgs { + strategy: PollingStrategy::offset(offset), + count: 1, + auto_commit: false, + }, + ) + .expect("fixture partition has a read snapshot"); + assert!(!plan.needs_off_pump_io()); + delayed_reads.push((completion, plan.execute_resident())); + poll_replies.push(replies); + } + + // The fixture's ordinary inbox has two slots. These queries fill it before + // either completed read is returned, reproducing the former refusal path. + let mut progress_replies = Vec::new(); + for _ in 0..2 { + let (reply, replies) = channel(1); + assert!( + owner_sender + .try_send(ShardFrame::lifecycle(LifecycleFrame::PartitionRead { + namespace, + read: PartitionRead::GroupOffsetState { group_id }, + reply, + })) + .is_ok() + ); + progress_replies.push(replies); + } + assert!(matches!( + owner_sender.try_send(ShardFrame::lifecycle(LifecycleFrame::ReconcileApply)), + Err(crossfire::TrySendError::Full(_)) + )); + for (completion, result) in delayed_reads { + completion.complete(result); + } + assert_eq!(owner.inbox_len(), 2, "ordinary work remains queued"); + assert_eq!(owner.poll_completion_inbox_len(), 2); + assert_eq!(owner.metrics().frame_drops_value(), 0); + for replies in &poll_replies { + assert!(matches!( + replies.try_recv(), + Err(crossfire::TryRecvError::Empty) + )); + } + + let (_stop_sender, stop_receiver) = channel(1); + let pump = owner.run_message_pump(stop_receiver, Arc::new(AtomicBool::new(false))); + futures::pin_mut!(pump); + assert!(futures::poll!(pump.as_mut()).is_pending()); + + // The first query runs before either read is accepted. The second runs + // after exactly one acceptance: each lane yields while the other has work. + for (replies, expected_last_polled) in progress_replies.iter().zip([None, Some(0)]) { + let PartitionReadReply::GroupOffsetState { + last_polled, + committed, + } = replies.try_recv().expect("ordinary query was processed") + else { + panic!("expected the group's progress at this pump turn"); + }; + assert_eq!(last_polled, expected_last_polled); + assert_eq!(committed, None, "automatic commits were disabled"); + } + + // Both accepted results contain their requested message, so an empty read + // or an early rejection cannot make the progress observations pass. + for (expected_offset, replies) in poll_replies.iter().enumerate() { + let PartitionReadReply::Poll { fragments, .. } = replies + .try_recv() + .expect("completed read reached its caller") + else { + panic!("a full ordinary inbox must not reject a reserved completion"); + }; + let bytes: Vec = fragments + .iter() + .flat_map(|fragment| fragment.as_slice().iter().copied()) + .collect(); + let batch = decode_batch_slice(&bytes).expect("decode the completed read"); + let offsets: Vec = batch + .iter() + .map(|message| batch.header.base_offset + u64::from(message.header.offset_delta)) + .collect(); + let returned_payloads: Vec<&[u8]> = batch.iter().map(|message| message.payload).collect(); + assert_eq!(offsets, vec![expected_offset as u64]); + assert_eq!( + returned_payloads, + vec![payloads[expected_offset].as_bytes()] + ); + } + let (last_polled, committed) = partitions.group_offset_state(&namespace, group_id).unwrap(); + assert_eq!(last_polled, Some(1)); + assert_eq!(committed, None); + assert_eq!(owner.inbox_len(), 0); + assert_eq!(owner.poll_completion_inbox_len(), 0); +} + +#[compio::test] +async fn given_owner_processed_completion_when_shutdown_arrives_should_wake_and_drain_queued_completion() + { + let namespace = IggyNamespace::new(1, 1, 0); + let bus = Rc::new(IggyMessageBus::new(0)); + let (partition, config) = partition_with_messages(&bus, namespace, &["message"]).await; + let (owner, _owner_sender) = owner_with_inbox(&bus, config, namespace); + owner.plane.partitions().insert(namespace, partition); + let (stop_sender, stop_receiver) = channel(1); + let shutdown_flag = Arc::new(AtomicBool::new(false)); + let wake_observer = Arc::new(PumpWakeObserver::default()); + let waker = Arc::clone(&wake_observer).into(); + let mut context = Context::from_waker(&waker); + let pump = owner.run_message_pump(stop_receiver, Arc::clone(&shutdown_flag)); + futures::pin_mut!(pump); + + // Processing a completion advances the pump into another wait. Shutdown + // must still wake it after a completion branch has already won. + let first_reply = queue_resident_poll(&owner, namespace); + assert!(pump.as_mut().poll(&mut context).is_pending()); + assert_single_message_reply(&first_reply); + wake_observer.notified.store(false, Ordering::Relaxed); + stop_sender.try_send(()).expect("signal shutdown"); + assert!(wake_observer.notified.load(Ordering::Relaxed)); + + // Both shutdown and a completion are ready before the pump resumes. + // Graceful shutdown must drain the completion before the final flush. + let queued_reply = queue_resident_poll(&owner, namespace); + assert!(matches!( + pump.as_mut().poll(&mut context), + Poll::Ready(None) + )); + assert_single_message_reply(&queued_reply); + assert_eq!(owner.inbox_len(), 0); + assert_eq!(owner.poll_completion_inbox_len(), 0); + assert!(!shutdown_flag.load(Ordering::Relaxed)); +} + +#[compio::test] +async fn given_owner_processed_completion_when_shutdown_sender_drops_should_wake_and_stop() { + let namespace = IggyNamespace::new(1, 1, 0); + let bus = Rc::new(IggyMessageBus::new(0)); + let (partition, config) = partition_with_messages(&bus, namespace, &["message"]).await; + let (owner, _owner_sender) = owner_with_inbox(&bus, config, namespace); + owner.plane.partitions().insert(namespace, partition); + let (stop_sender, stop_receiver) = channel(1); + let shutdown_flag = Arc::new(AtomicBool::new(false)); + let wake_observer = Arc::new(PumpWakeObserver::default()); + let waker = Arc::clone(&wake_observer).into(); + let mut context = Context::from_waker(&waker); + let pump = owner.run_message_pump(stop_receiver, Arc::clone(&shutdown_flag)); + futures::pin_mut!(pump); + + let reply = queue_resident_poll(&owner, namespace); + assert!(pump.as_mut().poll(&mut context).is_pending()); + assert_single_message_reply(&reply); + + // Losing the final shutdown sender must wake an otherwise idle owner; + // manually polling to completion alone would miss a lost notification. + wake_observer.notified.store(false, Ordering::Relaxed); + drop(stop_sender); + assert!(wake_observer.notified.load(Ordering::Relaxed)); + assert!(matches!( + pump.as_mut().poll(&mut context), + Poll::Ready(None) + )); + assert_eq!(owner.inbox_len(), 0); + assert_eq!(owner.poll_completion_inbox_len(), 0); + assert!(!shutdown_flag.load(Ordering::Relaxed)); +} + +type CompletionTestShard = IggyShard< + Rc, + PrepareJournal, + IggySnapshot, + PollTestMetadata, + PapayaShardsTable, +>; + +fn owner_with_inbox( + bus: &Rc, + config: PartitionsConfig, + namespace: IggyNamespace, +) -> (CompletionTestShard, TaggedSender) { + let shard_id = ShardId::new(0); + let partitions = IggyPartitions::new(shard_id, config); + let metadata = IggyMetadata::new(None, None, None, None, PollTestMetadata::default(), None); + let (sender, inbox, replies) = shard_channel(0, 2, 1); + let routes = PapayaShardsTable::new(); + routes.insert(namespace, PartitionLocation::new(shard_id, 0)); + let owner = CompletionTestShard::new( + ShardIdentity::new(0, "poll-completion-test".to_string()), + bus.clone(), + Rc::new(|_, _| {}), + Rc::new(|_, _| {}), + Rc::new(|_| {}), + Rc::new(|_| {}), + metadata, + partitions, + vec![sender.clone()], + inbox, + replies, + 2, + routes, + PartitionConsensusConfig::new(1, ReplicaTopology::new(0, 3), bus.clone()), + None, + ShardMetrics::for_shard(), + ) + .expect("valid owner inbox wiring"); + (owner, sender) +} + +fn queue_resident_poll( + owner: &CompletionTestShard, + namespace: IggyNamespace, +) -> Receiver { + let plan = owner + .plane + .partitions() + .build_poll_snapshot( + &namespace, + PollingConsumer::Consumer(1, 0), + &PollingArgs { + strategy: PollingStrategy::offset(0), + count: 1, + auto_commit: false, + }, + ) + .expect("fixture has a read snapshot"); + assert!(!plan.needs_off_pump_io()); + let (reply_sender, replies) = channel(1); + owner + .poll_completions + .try_reserve(namespace, reply_sender) + .expect("reserve capacity before completing the read") + .complete(plan.execute_resident()); + assert_eq!( + owner.poll_completion_inbox_len(), + 1, + "completion awaits owner acceptance" + ); + assert!(matches!( + replies.try_recv(), + Err(crossfire::TryRecvError::Empty) + )); + replies +} + +fn assert_single_message_reply(replies: &Receiver) { + let PartitionReadReply::Poll { + fragments, + current_offset, + } = replies.try_recv().expect("owner replied to the completion") + else { + panic!("the fixture's read should succeed"); + }; + assert_eq!(current_offset, 0); + let bytes: Vec = fragments + .iter() + .flat_map(|fragment| fragment.as_slice().iter().copied()) + .collect(); + let batch = decode_batch_slice(&bytes).expect("decode the reply"); + let payloads: Vec<&[u8]> = batch.iter().map(|message| message.payload).collect(); + assert_eq!(payloads, vec![b"message".as_slice()]); + assert!(matches!( + replies.try_recv(), + Err(crossfire::TryRecvError::Disconnected) + )); +} + +#[derive(Default)] +struct PumpWakeObserver { + notified: AtomicBool, +} + +impl Wake for PumpWakeObserver { + fn wake(self: Arc) { + self.wake_by_ref(); + } + + fn wake_by_ref(self: &Arc) { + self.notified.store(true, Ordering::Relaxed); + } +} diff --git a/core/shard/src/poll/test_support.rs b/core/shard/src/poll/test_support.rs new file mode 100644 index 0000000000..6664288ed9 --- /dev/null +++ b/core/shard/src/poll/test_support.rs @@ -0,0 +1,110 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use std::sync::Arc; + +use consensus::{LocalPipeline, Sequencer, VsrConsensus, oneshot_channel}; +use futures::FutureExt; +use iggy_binary_protocol::{Command, Operation, RoutedRequestHeader}; +use iggy_common::{IggyByteSize, PartitionStats, variadic}; +use message_bus::MessageBus; +use metadata::MuxStateMachine; +use metadata::stm::stream::Streams; +use metadata::stm::user::Users; +use partitions::{IggyPartition, Partition, PartitionPathLayout, PartitionsConfig}; +use server_common::send_messages::{ + IggyMessage, IggyMessageHeader, IggyMessages, SendMessagesOwned, +}; +use server_common::sharding::IggyNamespace; + +pub(super) type PollTestMetadata = MuxStateMachine; + +/// Commit one batch starting at offset zero and keep it resident. Each call +/// creates an independent partition history, even for the same namespace. +#[allow(clippy::future_not_send)] +pub(super) async fn partition_with_messages( + bus: &B, + namespace: IggyNamespace, + payloads: &[&str], +) -> (IggyPartition, PartitionsConfig) { + let cluster_id = 1; + let replica_id = 0; + let replica_count = 3; + let segment_size = IggyByteSize::from(1_048_576_u64); + let config = PartitionsConfig { + messages_required_to_save: 100, + size_of_messages_required_to_save: segment_size, + validate_checksum: true, + segment_size, + preallocate_segments: false, + encryptor: None, + path_layout: PartitionPathLayout::default(), + }; + let consensus = VsrConsensus::new( + cluster_id, + replica_id, + replica_count, + namespace.inner(), + bus.clone(), + LocalPipeline::new(), + ); + consensus.init(); + let mut partition = Box::new(IggyPartition::with_in_memory_storage( + Arc::new(PartitionStats::default()), + consensus, + segment_size, + )); + + assert!(!payloads.is_empty(), "fixture requires messages"); + let mut messages = IggyMessages::with_capacity(payloads.len()); + for (index, payload) in payloads.iter().enumerate() { + messages.push(IggyMessage { + header: IggyMessageHeader { + id: index as u128 + 1, + payload_length: u32::try_from(payload.len()).expect("fixture payload fits u32"), + ..Default::default() + }, + payload: payload.as_bytes().to_vec().into(), + user_headers: None, + }); + } + let append = SendMessagesOwned::from_messages(namespace, &messages) + .expect("encode fixture messages") + .encode_request(RoutedRequestHeader { + command: Command::Request, + operation: Operation::SendMessages, + client: 1, + session: 1, + request: 1, + group: namespace.inner(), + ..Default::default() + }) + .expect("encode append request"); + let (append_reply, append_result) = oneshot_channel(); + partition.on_request(append, Some(append_reply)).await; + assert_eq!(partition.consensus().sequencer().current_sequence(), 1); + partition.consensus().advance_commit_max(1); + partition.commit_journal(&config).await; + assert_eq!(partition.consensus().commit_min(), 1); + assert!( + matches!(append_result.now_or_never(), Some(Ok(_))), + "fixture append was committed" + ); + assert_eq!(partition.offsets().commit_offset, payloads.len() as u64 - 1); + + (*partition, config) +} diff --git a/core/shard/src/poll/timeout_tests.rs b/core/shard/src/poll/timeout_tests.rs new file mode 100644 index 0000000000..faad4a6f34 --- /dev/null +++ b/core/shard/src/poll/timeout_tests.rs @@ -0,0 +1,532 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use std::cell::RefCell; +use std::pin::Pin; +use std::rc::Rc; +use std::time::Duration; + +use consensus::PartitionsHandle; +use futures::channel::oneshot; +use iggy_common::{IggyError, PollingStrategy}; +use message_bus::{ + BusMessage, ClientForwardFn, ConnectionLostFn, JoinHandle, MessageBus, ReplicaForwardFn, + SendError, +}; +use metadata::IggyMetadata; +use partitions::{IggyPartitions, PollFragments, PollingArgs, PollingConsumer}; +use server_common::MESSAGE_ALIGN; +use server_common::iobuf::Frozen; +use server_common::send_messages::decode_batch_slice; +use server_common::sharding::{IggyNamespace, PartitionLocation, ShardId}; + +use super::completion::PollCompletionLane; +use super::test_support::{PollTestMetadata, partition_with_messages}; +use crate::shards_table::{PapayaShardsTable, ShardsTable}; +use crate::{ + IggyShard, LifecycleFrame, PARTITION_READ_TIMEOUT, PartitionConsensusConfig, PartitionRead, + PartitionReadReply, ReplicaTopology, ShardFrame, ShardIdentity, channel, shard_channel, +}; + +/// Expire the requester before releasing a nonempty completion to the owner. +/// The closed reply channel must prevent admission so a later Next can return +/// the unseen messages. Timeout racing with admission is a separate case. +#[compio::test] +#[allow(clippy::too_many_lines)] +async fn given_auto_commit_poll_when_completion_arrives_after_timeout_should_return_unseen_messages_on_next() + { + let namespace = IggyNamespace::new(1, 1, 0); + let consumer = PollingConsumer::Consumer(7, 0); + let (expire_timeout, timeout_elapsed) = oneshot::channel(); + let bus = PollTestBus { + next_timeout: Rc::new(RefCell::new(Some(timeout_elapsed))), + ..Default::default() + }; + let mut owner = owner_with_messages(&bus, namespace).await; + let (owner_sender, owner_inbox, _owner_reply_lane) = shard_channel(0, 2, 1); + owner.attach_senders(vec![owner_sender.clone()]); + let partitions = owner.plane.partitions(); + let (stored_offset, partition_offset) = partitions + .consumer_offset_read(&namespace, consumer) + .expect("fixture partition exists"); + assert_eq!(stored_offset, None); + assert_eq!( + partition_offset, 3, + "four messages are available at offsets 0 through 3" + ); + + // Start a real routed request, but let the test control when its completed + // read reaches the owner. The fourth message will identify the next batch. + let first_poll = owner.partition_read( + namespace, + PartitionRead::Poll { + consumer, + args: PollingArgs { + strategy: PollingStrategy::next(), + count: 3, + auto_commit: true, + }, + }, + ); + futures::pin_mut!(first_poll); + assert!(futures::poll!(first_poll.as_mut()).is_pending()); + let ShardFrame::Lifecycle(LifecycleFrame::PartitionRead { + namespace: requested_namespace, + read: + PartitionRead::Poll { + consumer: requested_consumer, + args, + }, + reply, + }) = owner_inbox + .try_recv() + .expect("poll reached the owner inbox") + else { + panic!("expected the routed poll request"); + }; + let late_reply = reply.clone(); + let completion = owner + .poll_completions + .try_reserve(requested_namespace, reply) + .expect("reserve the read before the requester times out"); + let read_plan = partitions + .build_poll_snapshot(&requested_namespace, requested_consumer, &args) + .expect("poll has a read snapshot"); + // Resident bytes keep disk scheduling out of this test. Holding the result + // here models the delay before acceptance shared with detached disk reads. + assert!(!read_plan.needs_off_pump_io()); + let delayed_result = read_plan.execute_resident(); + + // The reply sender is still alive, so None must come from the timeout. + expire_timeout + .send(()) + .expect("requester is waiting on its timer"); + assert!( + first_poll.await.is_none(), + "the caller received no successful poll reply" + ); + assert!( + matches!( + late_reply.try_send(PartitionReadReply::Ack), + Err(crossfire::TrySendError::Disconnected(_)) + ), + "timeout closed the reply channel before completion" + ); + let (stored_offset, _) = partitions + .consumer_offset_read(&namespace, consumer) + .unwrap(); + assert_eq!( + stored_offset, None, + "reading alone has not accepted progress" + ); + + // The reservation survives requester timeout and follows the late result + // into the completion lane. The owner discards it before admitting progress. + completion.complete(delayed_result); + let completion = owner + .poll_completions + .try_recv() + .expect("late completion reached its reserved lane"); + owner.on_poll_completed(*completion).await; + let (stored_offset, _) = partitions + .consumer_offset_read(&namespace, consumer) + .unwrap(); + assert_eq!( + stored_offset, None, + "the late result must leave progress unset for the three unseen messages" + ); + partitions + .with_partition(&namespace, |partition| { + assert!( + partition.consensus().pipeline_is_empty(), + "the late completion must not assign an automatic commit" + ); + assert_eq!( + partition.consensus().request_queue_len(), + 0, + "the late completion must not queue an automatic commit" + ); + }) + .expect("fixture partition exists"); + + // A new Next must include the three unseen messages and the fourth message. + // Disabling its automatic commit keeps the final cursor attributable to + // the timed out poll alone. + let next_poll = owner.partition_read( + namespace, + PartitionRead::Poll { + consumer, + args: PollingArgs { + strategy: PollingStrategy::next(), + count: 4, + auto_commit: false, + }, + }, + ); + futures::pin_mut!(next_poll); + assert!(futures::poll!(next_poll.as_mut()).is_pending()); + let ShardFrame::Lifecycle(LifecycleFrame::PartitionRead { + namespace, + read, + reply, + }) = owner_inbox + .try_recv() + .expect("subsequent Next reached the owner inbox") + else { + panic!("expected the subsequent poll request"); + }; + owner.on_partition_read(namespace, read, reply).await; + let Some(PartitionReadReply::Poll { fragments, .. }) = next_poll.await else { + panic!("subsequent Next must return messages successfully"); + }; + assert_eq!( + message_offsets(&fragments), + vec![0, 1, 2, 3], + "Next must include offsets 0 through 2, which the caller never received" + ); + let (stored_offset, _) = partitions + .consumer_offset_read(&namespace, consumer) + .unwrap(); + assert_eq!(stored_offset, None); +} + +/// Admission is checked against a synthetic disk plan: the fixture's journal +/// bytes are evicted and spawned futures are captured without execution. This +/// proves dispatch ordering; the partition tests separately cover real disk I/O. +#[compio::test] +#[allow(clippy::too_many_lines)] +async fn given_reserved_completion_capacity_when_disk_polls_arrive_should_reject_until_owner_dequeues() + { + let namespace = IggyNamespace::new(1, 1, 0); + let group_id = 7; + let consumer = PollingConsumer::ConsumerGroup(7, 0); + let bus = PollTestBus::default(); + let mut owner = owner_with_messages(&bus, namespace).await; + owner.poll_completions = PollCompletionLane::new(1, owner.metrics()); + let (owner_sender, _owner_inbox, _owner_replies) = shard_channel(0, 2, 1); + owner.attach_senders(vec![owner_sender]); + let partitions = owner.plane.partitions(); + let args = PollingArgs::new(PollingStrategy::offset(0), 3, true); + let resident_plan = partitions + .build_poll_snapshot(&namespace, consumer, &args) + .expect("fixture has messages before eviction"); + assert!(!resident_plan.needs_off_pump_io()); + let delayed_result = resident_plan.execute_resident(); + evict_messages_for_disk_dispatch(&owner, namespace).await; + assert!( + partitions + .build_poll_snapshot(&namespace, consumer, &args) + .expect("evicted messages have a disk plan") + .needs_off_pump_io() + ); + assert!(bus.spawned_tasks.borrow().is_empty()); + + // A running read owns the sole slot even while its queue is empty. + let (held_reply, _held_replies) = channel(1); + let reservation = owner + .poll_completions + .try_reserve(namespace, held_reply) + .expect("reserve the only slot"); + assert_eq!(owner.poll_completion_inbox_len(), 0); + let (reply, replies) = channel(1); + owner + .on_partition_read( + namespace, + PartitionRead::Poll { + consumer, + args: args.clone(), + }, + reply, + ) + .await; + assert!(matches!( + replies.try_recv(), + Ok(PartitionReadReply::Rejected( + IggyError::TransientNotAccepted + )) + )); + assert!( + bus.spawned_tasks.borrow().is_empty(), + "full capacity must reject before spawning I/O" + ); + + // Returning the result transfers its reservation into queue occupancy. + // A finished read cannot admit a replacement until the owner takes it out. + reservation.complete(delayed_result); + assert_eq!(owner.poll_completion_inbox_len(), 1); + let (reply, replies) = channel(1); + owner + .on_partition_read( + namespace, + PartitionRead::Poll { + consumer, + args: args.clone(), + }, + reply, + ) + .await; + assert!(matches!( + replies.try_recv(), + Ok(PartitionReadReply::Rejected( + IggyError::TransientNotAccepted + )) + )); + assert!( + bus.spawned_tasks.borrow().is_empty(), + "queued results still consume capacity" + ); + let (last_polled, committed) = partitions.group_offset_state(&namespace, group_id).unwrap(); + assert_eq!( + last_polled, None, + "rejected dispatch must not advance group progress" + ); + assert_eq!( + committed, None, + "rejected dispatch must not commit an offset" + ); + partitions + .with_partition(&namespace, |partition| { + assert!(partition.consensus().pipeline_is_empty()); + assert_eq!(partition.consensus().request_queue_len(), 0); + }) + .expect("fixture partition exists"); + + // Discarding the dequeued result frees the slot without accepting its data. + drop( + owner + .poll_completions + .try_recv() + .expect("owner dequeues held result"), + ); + let (reply, replies) = channel(1); + owner + .on_partition_read(namespace, PartitionRead::Poll { consumer, args }, reply) + .await; + assert_eq!( + bus.spawned_tasks.borrow().len(), + 1, + "a newly available slot permits disk dispatch" + ); + assert!(matches!( + replies.try_recv(), + Err(crossfire::TryRecvError::Empty) + )); + assert_eq!( + owner.poll_completion_inbox_len(), + 0, + "captured disk task has not run" + ); + bus.spawned_tasks.borrow_mut().clear(); +} + +#[compio::test] +async fn given_missing_owner_route_when_disk_poll_arrives_should_reject_before_dispatch() { + let namespace = IggyNamespace::new(1, 1, 0); + let bus = PollTestBus::default(); + let owner = owner_with_messages(&bus, namespace).await; + evict_messages_for_disk_dispatch(&owner, namespace).await; + let consumer = PollingConsumer::ConsumerGroup(7, 0); + let args = PollingArgs::new(PollingStrategy::offset(0), 3, true); + assert!( + owner + .plane + .partitions() + .build_poll_snapshot(&namespace, consumer, &args) + .expect("fixture must enter the disk path") + .needs_off_pump_io() + ); + let (reply, replies) = channel(1); + + owner + .on_partition_read(namespace, PartitionRead::Poll { consumer, args }, reply) + .await; + + assert!(matches!( + replies.try_recv(), + Ok(PartitionReadReply::Rejected( + IggyError::TransientNotAccepted + )) + )); + assert!(bus.spawned_tasks.borrow().is_empty()); + assert_eq!(owner.poll_completion_inbox_len(), 0); + assert_eq!(owner.metrics().frame_drops_value(), 1); + let (last_polled, committed) = owner + .plane + .partitions() + .group_offset_state(&namespace, 7) + .unwrap(); + assert_eq!(last_polled, None); + assert_eq!(committed, None); +} + +#[compio::test] +async fn given_disconnected_owner_when_disk_poll_arrives_should_reject_before_dispatch() { + let namespace = IggyNamespace::new(1, 1, 0); + let bus = PollTestBus::default(); + let mut owner = owner_with_messages(&bus, namespace).await; + let (owner_sender, owner_inbox, owner_replies) = shard_channel(0, 2, 1); + owner.attach_senders(vec![owner_sender]); + drop(owner_inbox); + drop(owner_replies); + evict_messages_for_disk_dispatch(&owner, namespace).await; + let consumer = PollingConsumer::ConsumerGroup(7, 0); + let args = PollingArgs::new(PollingStrategy::offset(0), 3, true); + assert!( + owner + .plane + .partitions() + .build_poll_snapshot(&namespace, consumer, &args) + .expect("fixture must enter the disk path") + .needs_off_pump_io() + ); + let (reply, replies) = channel(1); + + owner + .on_partition_read(namespace, PartitionRead::Poll { consumer, args }, reply) + .await; + + assert!(matches!( + replies.try_recv(), + Ok(PartitionReadReply::Rejected( + IggyError::TransientNotAccepted + )) + )); + assert!(bus.spawned_tasks.borrow().is_empty()); + assert_eq!(owner.poll_completion_inbox_len(), 0); + assert_eq!(owner.metrics().frame_drops_value(), 1); + let (last_polled, committed) = owner + .plane + .partitions() + .group_offset_state(&namespace, 7) + .unwrap(); + assert_eq!(last_polled, None); + assert_eq!(committed, None); +} + +/// Remove the fixture's only resident batch to select disk dispatch without +/// creating files. Captured tasks must stay unpolled: these tests establish +/// admission behavior, while real disk reads are covered in partition tests. +#[allow(clippy::future_not_send)] +async fn evict_messages_for_disk_dispatch(owner: &PollTestShard, namespace: IggyNamespace) { + let partitions = owner.plane.partitions(); + let partition = partitions + .remove(&namespace) + .expect("fixture partition exists"); + let retained = partition.log.journal().inner.evict_prefix(1).await; + assert!(retained.is_empty(), "fixture has exactly one batch"); + assert!( + partition + .log + .journal() + .inner + .oldest_resident_offset() + .is_none() + ); + partitions.insert(namespace, partition); +} + +fn message_offsets(fragments: &PollFragments) -> Vec { + // A partial batch has separate header and payload fragments. Reassemble + // the fixture's single batch before decoding its actual message offsets. + let bytes: Vec = fragments + .iter() + .flat_map(|fragment| fragment.as_slice().iter().copied()) + .collect(); + let batch = decode_batch_slice(&bytes).expect("decode polled batch"); + batch + .iter() + .map(|message| batch.header.base_offset + u64::from(message.header.offset_delta)) + .collect() +} + +type PollTestShard = IggyShard; + +/// Commit four messages at offsets 0 through 3 while retaining their bytes in +/// the resident journal. The regression controls completion acceptance, so +/// setup needs neither disk files nor a running shard pump. +#[allow(clippy::future_not_send)] +async fn owner_with_messages(bus: &PollTestBus, namespace: IggyNamespace) -> PollTestShard { + let shard_id = ShardId::new(0); + let (partition, config) = partition_with_messages(bus, namespace, &["x"; 4]).await; + let partitions = IggyPartitions::new(shard_id, config); + partitions.insert(namespace, partition); + let metadata = IggyMetadata::new(None, None, None, None, PollTestMetadata::default(), None); + let routes = PapayaShardsTable::new(); + routes.insert(namespace, PartitionLocation::new(shard_id, 0)); + PollTestShard::without_inbox( + ShardIdentity::new(0, "poll-timeout-test".to_string()), + bus.clone(), + metadata, + partitions, + routes, + PartitionConsensusConfig::new(1, ReplicaTopology::new(0, 3), bus.clone()), + ) +} + +type CapturedTask = Pin>>; + +/// The first configured timer expires on demand; other timers stay pending. +/// Detached tasks are captured so dispatch tests can observe admission without +/// executing their synthetic disk plans. +#[derive(Clone, Default)] +struct PollTestBus { + next_timeout: Rc>>>, + spawned_tasks: Rc>>, +} + +#[allow(clippy::future_not_send)] +impl MessageBus for PollTestBus { + fn spawn(&self, future: impl Future + 'static) { + self.spawned_tasks.borrow_mut().push(Box::pin(future)); + } + + fn sleep(&self, duration: Duration) -> impl Future { + assert_eq!(duration, PARTITION_READ_TIMEOUT); + let timeout = self.next_timeout.borrow_mut().take(); + async move { + if let Some(timeout) = timeout { + timeout + .await + .expect("test controls when the timeout expires"); + } else { + std::future::pending::<()>().await; + } + } + } + + async fn send_to_client( + &self, + _client_id: u128, + _data: impl Into, + ) -> Result<(), SendError> { + panic!("partition reads reply through their channel"); + } + + fn send_to_replica( + &self, + _replica: u8, + _data: Frozen, + ) -> impl Future> { + // This test observes local progress, without acknowledging replication. + std::future::ready(Ok(())) + } + + fn set_connection_lost_fn(&self, _f: ConnectionLostFn) {} + fn set_replica_forward_fn(&self, _f: ReplicaForwardFn) {} + fn set_client_forward_fn(&self, _f: ClientForwardFn) {} + fn track_background(&self, _handle: JoinHandle<()>) {} +} diff --git a/core/shard/src/router.rs b/core/shard/src/router.rs index 0eea8a6785..a9c62d08fe 100644 --- a/core/shard/src/router.rs +++ b/core/shard/src/router.rs @@ -274,6 +274,10 @@ where /// the bounded pump drain that turn a stalled flush into a timed-out /// non-zero exit instead of a process that reports healthy forever. #[allow(clippy::future_not_send)] + #[allow( + clippy::too_many_lines, + reason = "Keep the owner select loop and its shutdown sequence together" + )] pub async fn run_message_pump( &self, stop: Receiver<()>, @@ -286,6 +290,7 @@ where Journal, Header = PrepareHeader>, M: RestorableMetadataStm, { + let _completion_pump = self.poll_completions.pump_guard(); if let Some(sender) = self.senders.get(self.id as usize).cloned() { let metrics = self.metrics.clone(); self.plane @@ -318,6 +323,9 @@ where // simulator (see `MessageBus::sleep`). let rearm_tick = || self.bus.sleep(CONSENSUS_TICK_INTERVAL).fuse(); let mut consensus_tick = std::pin::pin!(rearm_tick()); + // Keep the receive registered: recreating it repeats Crossfire's initial + // backoff on every pump turn while the shutdown channel is empty. + let mut stop_signal = std::pin::pin!(stop.recv().fuse()); let mut fatal: Option = None; loop { // `select_biased!`, not `select!`: the unbiased macro draws its @@ -326,13 +334,12 @@ where // intended priority anyway: stop, then tick, redispatch, then // newly received frames. futures::select_biased! { - _ = stop.recv().fuse() => break, + _ = stop_signal.as_mut() => break, () = consensus_tick.as_mut() => { - // Sharing the pump task is what keeps `tick_partitions` - // borrow-safe, but it bounds the tick's worst-case delay - // to one main frame body's longest `.await` (replication - // append + commit_journal fsync/rotate + reply) plus the - // one reply-lane bus send drained per main frame. + // Sharing the pump task keeps `tick_partitions` borrows + // safe, but a tick can wait for a main frame's processing + // (replication, journal fsync or rotation, and reply), plus + // one reply and one poll completion, including its loopback. // TODO(hubcio): if a load test shows tick starvation, // make `tick_partitions` borrow-free so the tick can be // decoupled from the pump again without reintroducing the @@ -386,6 +393,7 @@ where { self.process_frame(reply).await; } + self.process_one_poll_completion(&mut loopback_buf, &mut namespace_scratch).await; } frame = self.inbox.recv().fuse() => { match frame { @@ -411,6 +419,7 @@ where { self.process_frame(reply).await; } + self.process_one_poll_completion(&mut loopback_buf, &mut namespace_scratch).await; } Err(_) => break, } @@ -424,6 +433,17 @@ where if self.accept_frame_for_self(&frame) { self.process_frame(frame).await; } + self.process_one_poll_completion(&mut loopback_buf, &mut namespace_scratch).await; + } + Err(_) => break, + } + } + completion = self.poll_completions.recv().fuse() => { + match completion { + Ok(completion) => { + self.on_poll_completed(*completion).await; + self.process_loopback(&mut loopback_buf, &mut namespace_scratch).await; + self.apply_reconcile_ops(); } Err(_) => break, } @@ -431,6 +451,8 @@ where } } + self.poll_completions.close(); + // A stop can win the select immediately after a frame fenced a // partition, before the next tick observes it. Preserve that fault so // shutdown cannot turn a durability failure into a clean pump exit. @@ -480,6 +502,27 @@ where fatal } + /// A busy ordinary lane yields one completion per frame. Keeping this + /// service bounded lets ordinary work progress under a completion flood. + #[allow(clippy::future_not_send)] + async fn process_one_poll_completion( + &self, + loopback_buf: &mut Vec>, + namespace_scratch: &mut Vec, + ) where + B: MessageBus + 'static, + MJ: JournalHandle, + ::Target: + Journal, Header = PrepareHeader>, + M: RestorableMetadataStm, + { + if let Ok(completion) = self.poll_completions.try_recv() { + self.on_poll_completed(*completion).await; + self.process_loopback(loopback_buf, namespace_scratch).await; + self.apply_reconcile_ops(); + } + } + /// Process queued work after the select loop has stopped. Redispatch keeps /// its live-pump rank over the inbox, and every delivered frame gets its /// loopback before another frame can run. @@ -504,8 +547,24 @@ where if let Some(fault) = self.first_partition_commit_fault() { return Some(fault); } + self.process_one_poll_completion(loopback_buf, namespace_scratch) + .await; + if let Some(fault) = self.first_partition_commit_fault() { + return Some(fault); + } } let Ok(frame) = self.inbox.try_recv() else { + if let Ok(completion) = self.poll_completions.try_recv() { + self.on_poll_completed(*completion).await; + self.process_loopback(loopback_buf, namespace_scratch).await; + self.apply_reconcile_ops(); + if let Some(fault) = self.first_partition_commit_fault() { + return Some(fault); + } + // Completion processing can stage ordinary work. Return + // to the main drain before accepting another completion. + continue; + } break; }; if self.accept_frame_for_self(&frame) { @@ -516,6 +575,11 @@ where return Some(fault); } } + self.process_one_poll_completion(loopback_buf, namespace_scratch) + .await; + if let Some(fault) = self.first_partition_commit_fault() { + return Some(fault); + } } // Retired, not delivered: the pump is going away, so a staged frame has @@ -736,13 +800,7 @@ where read, reply, } => { - // Addressed to the shard owning `namespace` (the sender - // resolved it via the shards table). The handler (wired by - // the server) runs the read against this shard's partitions - // plane and pushes the result over `reply`; a dropped - // sender means the read is skipped and the gather side - // times out. - (self.on_partition_read)(namespace, read, reply); + self.on_partition_read(namespace, read, reply).await; } LifecycleFrame::PartitionSubmit { request, reply } => { // Addressed to the shard owning the request's namespace (the @@ -752,15 +810,6 @@ where // already made. self.on_partition_submit(request, reply).await; } - LifecycleFrame::AutoCommitSubmit { - request, - reservation, - } => { - self.plane - .partitions() - .on_auto_commit_request(request, reservation) - .await; - } LifecycleFrame::MetadataCommitTick => { // Reconciler may not yet be wired (e.g. mid-bootstrap, or // single-shard tests that never enable the reconciler loop). diff --git a/core/simulator/src/bin/simulator-ui.rs b/core/simulator/src/bin/simulator-ui.rs index 041a7c8860..86b022801f 100644 --- a/core/simulator/src/bin/simulator-ui.rs +++ b/core/simulator/src/bin/simulator-ui.rs @@ -16,6 +16,7 @@ // under the License. use bytes::Bytes; +use futures::FutureExt; use iggy_binary_protocol::ReplyHeader; use iggy_common::{IggyByteSize, PollingStrategy}; use partitions::{PollingArgs, PollingConsumer}; @@ -66,6 +67,7 @@ fn main() { // Initialize partition on all replicas println!("[sim] Initializing test partition: {test_namespace:?}"); sim.init_partition(test_namespace); + sim.register_client_with_primary(&client); // 1. Send messages to a partition println!("[sim] Sending messages to partition"); @@ -118,12 +120,21 @@ fn main() { // 4. Poll messages and check offsets on the leader let consumer = PollingConsumer::Consumer(1, 0); let args = PollingArgs::new(PollingStrategy::first(), 10, false); - match sim.poll_messages(leader as usize, test_namespace, consumer, &args) { + let poll = sim.poll_messages(leader as usize, test_namespace, consumer, &args); + futures::pin_mut!(poll); + // Drive the caller between steps so the owner can process a pending poll. + let poll_result = (0..100) + .find_map(|_| { + let result = poll.as_mut().now_or_never(); + if result.is_none() { + sim.step(); + } + result + }) + .expect("poll should reply within the simulation budget"); + match poll_result { Ok(fragments) => { - println!( - "[sim] Poll returned {} fragments (expected 4)", - fragments.len() - ); + println!("[sim] Poll returned {} fragments", fragments.len()); } Err(e) => { println!("[sim] Poll failed: {e}"); diff --git a/core/simulator/src/bus.rs b/core/simulator/src/bus.rs index daa65c8dc1..7002ea4a1d 100644 --- a/core/simulator/src/bus.rs +++ b/core/simulator/src/bus.rs @@ -18,6 +18,8 @@ use crate::deps::SimClock; use crate::executor::{PendingSpawns, TimerHandle}; use clock::Clock; +#[cfg(test)] +use futures::channel::oneshot; use iggy_binary_protocol::GenericHeader; use message_bus::client_listener::RequestHandler; use message_bus::fd_transfer::DupedFd; @@ -89,6 +91,8 @@ pub struct SimOutbox { /// fires it via `notify_client_connection_lost` to drive the real /// session-removal + logout path when it models a client disconnect. client_lost_fn: RefCell>, + #[cfg(test)] + next_replica_send: RefCell>>, } impl std::fmt::Debug for SimOutbox { @@ -119,9 +123,19 @@ impl SimOutbox { spawns, client_metas: RefCell::new(HashMap::new()), client_lost_fn: RefCell::new(None), + #[cfg(test)] + next_replica_send: RefCell::new(None), } } + /// Suspend the next replica send so tests can inspect owner reply ordering. + #[cfg(test)] + pub(crate) fn delay_next_replica_send(&self) -> oneshot::Sender<()> { + let (resume, wait) = oneshot::channel(); + assert!(self.next_replica_send.borrow_mut().replace(wait).is_none()); + resume + } + /// Drain all staged messages from this outbox. pub fn drain(&self) -> Vec { self.pending_messages.borrow_mut().drain(..).collect() @@ -215,6 +229,14 @@ impl MessageBus for SimOutbox { return Err(SendError::ReplicaNotConnected(replica)); } + #[cfg(test)] + { + let wait = self.next_replica_send.borrow_mut().take(); + if let Some(wait) = wait { + let _ = wait.await; + } + } + self.pending_messages.borrow_mut().push_back(Envelope { from_replica: Some(self.self_id), to_replica: Some(replica), diff --git a/core/simulator/src/lib.rs b/core/simulator/src/lib.rs index 7ad93eaa4a..31527f6ea9 100644 --- a/core/simulator/src/lib.rs +++ b/core/simulator/src/lib.rs @@ -1050,6 +1050,16 @@ impl Simulator { self.seed, self.executor.schedule_hash(), ); + let pending_completions = shard.poll_completion_inbox_len(); + assert_eq!( + pending_completions, + 0, + "lost wakeup: replica {replica_id} shard {} completion lane holds \ + {pending_completions} result(s) at quiescence (seed {:#x}, schedule hash {:#x})", + shard.id, + self.seed, + self.executor.schedule_hash(), + ); let pending_redispatch = shard.redispatched_frame_count(); assert_eq!( pending_redispatch, @@ -1375,45 +1385,46 @@ impl Simulator { self.run_pumps(); } - /// Poll messages directly from a replica's partition. + /// Poll messages through a replica's partition owner. + /// + /// The returned future owns its request and shard handle, so it can run on + /// the simulator executor while the caller advances the simulation or + /// releases paused work. The owner accepts the read and replies before + /// awaiting replication of an automatic commit. /// /// # Errors - /// `IggyError::ResourceNotFound` if the namespace is not on this replica. + /// Returns routing and owner rejections, `IggyError::ResourceNotFound` if + /// the owner reports a missing partition, or `IggyError::ShardCommunicationError` + /// if a submitted request times out or loses its reply. + /// + /// # Panics + /// If `replica_idx` is out of bounds or the owner replies for a different read type. + #[allow(clippy::future_not_send)] pub fn poll_messages( &self, replica_idx: usize, namespace: IggyNamespace, consumer: PollingConsumer, args: &PollingArgs, - ) -> Result, IggyError> { - let shard = self.replicas[replica_idx].partition_shard(namespace); - // Build the owned poll plan synchronously, then execute off the borrow. - // - // The one `block_on` allowed to stay, and only because `plan.execute()` - // cannot suspend here: the sim's partitions are in-memory (no - // `partition_dir`), so the plan serves the resident journal tier with no disk - // IO and no `bus.sleep`. A suspending await would fail two ways. On the - // virtual clock it would hang this thread forever, the clock advancing only - // through `advance_time`, which does not run during `block_on`. On the retry - // path it would panic on the compio timer outside a compio runtime. Safe only - // because it runs between `run_pumps` calls, with the executor quiescent and - // no pump holding the partition commit lock in a suspended frame. - let Some(plan) = shard - .plane - .partitions() - .build_poll_snapshot(&namespace, consumer, args) - else { - return Err(IggyError::ResourceNotFound(format!( - "partition not found for namespace {namespace:?} on replica {replica_idx}" - ))); - }; - // Partitions are driven directly, so a poll's auto-commit is never - // replicated (the serving shard's job in the real server). Offset discarded. - let (fragments, _commit_offset, auto_commit) = futures::executor::block_on(plan.execute())?; - if let Some(applied) = auto_commit { - applied.admit(|_| Ok::<(), IggyError>(()))?; + ) -> impl Future, IggyError>> + use<> { + let owner = Rc::clone(self.replicas[replica_idx].partition_shard(namespace)); + let args = args.clone(); + async move { + match owner + .partition_read(namespace, shard::PartitionRead::Poll { consumer, args }) + .await + { + Some(shard::PartitionReadReply::Poll { fragments, .. }) => Ok(fragments), + Some(shard::PartitionReadReply::Rejected(error)) => Err(error), + Some(shard::PartitionReadReply::NotFound) => { + Err(IggyError::ResourceNotFound(format!( + "partition not found for namespace {namespace:?} on replica {replica_idx}" + ))) + } + None => Err(IggyError::ShardCommunicationError), + Some(reply) => panic!("unexpected reply to a poll request: {reply:?}"), + } } - Ok(fragments) } /// Partition offsets from a replica. @@ -1677,6 +1688,7 @@ mod tests { use crate::workload::apply_sim_commands; use bytes::Bytes; use consensus::Status; + use futures::FutureExt; use iggy_binary_protocol::{AckLevel, RoutedRequestHeader}; use iggy_common::ConsumerKind; use server_common::sharding::IggyNamespace; @@ -1701,6 +1713,142 @@ mod tests { panic!("request {request_id} did not receive a reply"); } + /// The simulator helper must yield while its owner runs, then return the + /// accepted resident read even if the automatic commit's replica send stalls. + /// Releasing that send must let the same commit reach the other replicas. + #[test] + #[allow(clippy::too_many_lines)] + fn given_simulator_poll_when_replica_send_stalls_should_reply_and_resume_replication() { + // 1. Publish one message and let the cluster finish the initial work. + // The pause installed later must affect the poll's automatic commit. + server_common::MemoryPool::init_pool(&server_common::MemoryPoolSettings { + enabled: false, + size: iggy_common::IggyByteSize::from(0u64), + bucket_capacity: 1, + }); + let client_id = 1u128; + let replica_count = 3u8; + let primary_replica = 0u8; + let namespace = IggyNamespace::new(1, 1, 0); + let mut simulator = Simulator::new( + usize::from(replica_count), + std::iter::once(client_id), + packet::PacketSimulatorOptions { + node_count: replica_count, + client_count: 1, + seed: 0xC011_EC71, + ..packet::PacketSimulatorOptions::default() + }, + ); + simulator.init_partition(namespace); + let client = SimClient::new(client_id); + simulator.register_client_with_primary(&client); + let append_reply = submit_and_wait_for_reply( + &mut simulator, + client_id, + primary_replica, + client.send_messages( + namespace, + &[Bytes::from_static(b"reply-before-replication")], + ), + ); + assert_eq!(append_reply.header().status, 0); + simulator.run_pumps(); + + // 2. Confirm that the poll uses resident data. This scenario checks + // reply timing after inline acceptance, without a detached disk read. + let owner = + Rc::clone(simulator.replicas[usize::from(primary_replica)].partition_shard(namespace)); + let consumer_id = 7; + let partition_id = 0; + let consumer = PollingConsumer::Consumer(consumer_id, partition_id); + let auto_commit = true; + let poll_args = PollingArgs::new(iggy_common::PollingStrategy::next(), 1, auto_commit); + let resident_plan = owner + .plane + .partitions() + .build_poll_snapshot(&namespace, consumer, &poll_args) + .expect("poll snapshot"); + assert!( + !resident_plan.needs_off_pump_io(), + "fixture must exercise inline resident completion" + ); + drop(resident_plan); + + // 3. Run the poll task with the next replica send paused. + // The separate caller task lets the owner send a reply while + // its replication continuation is still waiting for release. + let resume_replication = owner.bus.delay_next_replica_send(); + let (poll_result_sender, poll_result_receiver) = shard::channel(1); + let poll = simulator.poll_messages( + usize::from(primary_replica), + namespace, + consumer, + &poll_args, + ); + simulator.executor.spawn(async move { + let _ = poll_result_sender.try_send(poll.await); + }); + simulator.run_pumps(); + + // 4. The nonempty reply must already be available before releasing + // replication. Waiting for it asynchronously here would hide a stall. + let fragments = poll_result_receiver + .recv() + .now_or_never() + .expect("admitted poll replies while replication remains suspended") + .expect("reply channel is open") + .expect("owning shard accepted the poll"); + assert!(!fragments.is_empty()); + + // 5. Before release, neither backup has received the automatic commit. + // Inspect backups because the stalled owner still borrows its partition. + for backup in &simulator.replicas[1..] { + let (consumer_offset, _) = backup + .partition_shard(namespace) + .plane + .partitions() + .consumer_offset_read(&namespace, consumer) + .expect("partition exists on the backup"); + assert_eq!(consumer_offset, None); + } + assert!( + resume_replication.send(()).is_ok(), + "replication must still be waiting after the poll reply" + ); + + // 6. Drive network delivery as well as the executor. A reply alone + // must not let the helper silently discard its replication continuation. + let replicated = (0..100).any(|_| { + simulator.step(); + simulator.replicas.iter().all(|replica| { + replica + .partition_shard(namespace) + .plane + .partitions() + .with_partition(&namespace, |partition| { + partition.durable_consumer_offset_count(iggy_common::ConsumerKind::Consumer) + == 1 + }) + == Some(true) + }) + }); + assert!( + replicated, + "automatic commit reaches every replica after release" + ); + let (consumer_offset, _partition_commit_offset) = owner + .plane + .partitions() + .consumer_offset_read(&namespace, consumer) + .expect("partition exists"); + assert_eq!( + consumer_offset, + Some(0), + "admission records the served cursor" + ); + } + #[test] #[allow(clippy::too_many_lines)] fn given_small_offset_limit_when_using_quorum_and_no_ack_should_bound_every_replica() { @@ -2740,7 +2888,7 @@ mod tests { } // Poll through the dispatch shell (`on_client_request`, drain, - // `handle_poll_messages`, `partition_read`, `on_partition_read`), running as a + // `handle_poll_messages`, `partition_read`, the owning shard), running as a // task the executor interleaves with the pump. let poll = client.poll_messages(ns, 10); sim.submit_request(client_id, 0, poll.into_generic()); @@ -2757,7 +2905,7 @@ mod tests { /// A `SimClient` poll returns the produced messages through the real dispatch /// read path (`on_client_request`, `handle_poll_messages`, `partition_read`, - /// `on_partition_read`), running as a task the executor interleaves with the pump, + /// the owning shard), running as a task the executor interleaves with the pump, /// and the whole login/produce/poll round-trip replays byte-for-byte on one seed. #[test] fn shell_poll_returns_produced_messages_deterministically() { @@ -3036,8 +3184,8 @@ mod tests { /// Injected through the synthetic `hold_borrow_across_await` rather than the real /// read, because the production read has no borrow-holding suspension to seed: the /// journal read is a synchronous memory copy, and `with_partition` returns an owned - /// `PollPlan` before the only awaits (disk read, offset persist) run off the borrow - /// in `spawn_poll_io`. + /// `PollPlan` before disk reads run off the borrow. Completion returns to the owner + /// before consumer progress changes. /// /// TODO: once storage faults are modelled, the disk-tier read /// (`PollPlan::execute`, `read_disk`) becomes a real seedable await in the read diff --git a/core/simulator/src/replica.rs b/core/simulator/src/replica.rs index a828bbe80f..3b963168b9 100644 --- a/core/simulator/src/replica.rs +++ b/core/simulator/src/replica.rs @@ -407,7 +407,6 @@ pub fn new_shard( on_client_request, on_metadata_submit, on_list_clients, - on_partition_read, // Step 6 keeps this to register client sessions; unused shell-off. sessions: _, } = if shell { @@ -431,12 +430,12 @@ pub fn new_shard( on_client_request, on_metadata_submit, on_list_clients, - on_partition_read, metadata, partitions, senders, inbox, reply_inbox, + ServerConfig::default().sharding.poll_completion_capacity, PapayaShardsTable::new(), shard::PartitionConsensusConfig::with_clock( CLUSTER_ID, From 30bdb469b9525c3d7e0faa4758326d8c8ab13ef7 Mon Sep 17 00:00:00 2001 From: Rudra Prasad Bhuyan Date: Sun, 13 Sep 2026 21:06:54 +0530 Subject: [PATCH 125/182] fix(configs): allow sibling IGGY variables (#4149) Closes #4143 --- Cargo.lock | 1 + core/configs/Cargo.toml | 3 ++ .../configs/src/configs_impl/file_provider.rs | 28 +++++++++++++++---- .../src/configs_impl/typed_env_provider.rs | 1 + core/configs/src/server_config/server.rs | 6 ++++ 5 files changed, 33 insertions(+), 6 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 5a191fc137..8e0573eca1 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3248,6 +3248,7 @@ dependencies = [ "serde", "serde_json", "serde_with", + "serial_test", "server_common", "static-toml", "tracing", diff --git a/core/configs/Cargo.toml b/core/configs/Cargo.toml index a1f5d51420..2649bf37db 100644 --- a/core/configs/Cargo.toml +++ b/core/configs/Cargo.toml @@ -37,5 +37,8 @@ server_common = { workspace = true } static-toml = { workspace = true } tracing = { workspace = true } +[dev-dependencies] +serial_test = { workspace = true } + [lints] workspace = true diff --git a/core/configs/src/configs_impl/file_provider.rs b/core/configs/src/configs_impl/file_provider.rs index f384352fd6..f182f2cd24 100644 --- a/core/configs/src/configs_impl/file_provider.rs +++ b/core/configs/src/configs_impl/file_provider.rs @@ -71,6 +71,7 @@ pub struct FileConfigProvider

{ env_prefix: &'static str, relocated_keys: &'static [RelocatedKey], known_env_names: Option>, + allowed_env_prefixes: &'static [&'static str], } impl FileConfigProvider

{ @@ -95,6 +96,7 @@ impl FileConfigProvider

{ env_prefix: "", relocated_keys: &[], known_env_names: None, + allowed_env_prefixes: &[], } } @@ -120,6 +122,11 @@ impl FileConfigProvider

{ self } + pub fn with_allowed_env_prefixes(mut self, prefixes: &'static [&'static str]) -> Self { + self.allowed_env_prefixes = prefixes; + self + } + fn reject_unknown_env_names(&self) -> Result<(), ConfigurationError> { let Some(known) = &self.known_env_names else { return Ok(()); @@ -128,9 +135,10 @@ impl FileConfigProvider

{ env::vars_os().filter_map(|(name, _)| name.into_string().ok()), self.env_prefix, known, + self.allowed_env_prefixes, ); for name in &unknown { - eprintln!("Unknown configuration environment variable '{name}'"); + eprintln!("Unknown configuration environment variable '{name}'. Unset it to boot."); } let rejected = !unknown.is_empty(); if rejected { @@ -279,9 +287,16 @@ fn unknown_env_names( names: impl Iterator, prefix: &str, known: &[&str], + allowed_prefixes: &[&str], ) -> Vec { names - .filter(|name| name.starts_with(prefix) && !known.contains(&name.as_str())) + .filter(|name| { + name.starts_with(prefix) + && !known.contains(&name.as_str()) + && !allowed_prefixes + .iter() + .any(|allowed| name.starts_with(allowed)) + }) .collect() } @@ -339,6 +354,7 @@ mod tests { .into_iter(), "IGGY_", &["IGGY_TCP_ADDRESS", "IGGY_ROOT_PASSWORD"], + &[], ); assert_eq!(unknown, vec!["IGGY_ENCRYPTION_UNKNOWN"]); } @@ -407,7 +423,6 @@ mod tests { /// environment, a shared `env_file`, or the `.env` that `main.rs` loads /// through `dotenvy` before `load_config` runs will refuse server boot. #[test] - #[ignore = "PR #4092 review: the `IGGY_` prefix fence refuses the repo's own `IGGY_CONNECTORS_*`, `IGGY_MCP_*` and CLI variables, with no opt-out"] fn given_a_sibling_binarys_env_vars_when_rejecting_then_the_server_should_still_boot() { let siblings = [ "IGGY_CONNECTORS_CONFIG_PATH", @@ -422,8 +437,8 @@ mod tests { names(&siblings).into_iter(), "IGGY_", crate::server_config::server::SERVER_PROCESS_ENV_VARS, + crate::server_config::server::SERVER_ALLOWED_ENV_PREFIXES, ); - assert!( unknown.is_empty(), "the server refuses to boot when its own sibling products' variables are present: {unknown:?}" @@ -449,7 +464,7 @@ mod tests { /// Mutates the process environment, so it must not run beside another test /// that reads it. #[test] - #[ignore = "PR #4092 review: a `.env` loaded by `dotenvy` before `load_config` refuses server boot; also mutates the process environment, so it must not run in parallel"] + #[serial_test::serial] fn given_a_dotenv_with_a_connectors_variable_when_loading_then_the_server_should_boot() { // SAFETY: single-threaded assertion over a variable no other test reads. unsafe { std::env::set_var("IGGY_CONNECTORS_CONFIG_PATH", "/etc/iggy/connectors.toml") }; @@ -461,7 +476,8 @@ mod tests { None, ) .with_relocated_keys("IGGY_", &[]) - .with_known_env_names(crate::server_config::server::SERVER_PROCESS_ENV_VARS.to_vec()); + .with_known_env_names(crate::server_config::server::SERVER_PROCESS_ENV_VARS.to_vec()) + .with_allowed_env_prefixes(crate::server_config::server::SERVER_ALLOWED_ENV_PREFIXES); let rejected = provider.reject_unknown_env_names(); // SAFETY: paired with the set above. diff --git a/core/configs/src/configs_impl/typed_env_provider.rs b/core/configs/src/configs_impl/typed_env_provider.rs index 72ac9fe018..5225f6a8a2 100644 --- a/core/configs/src/configs_impl/typed_env_provider.rs +++ b/core/configs/src/configs_impl/typed_env_provider.rs @@ -584,6 +584,7 @@ mod tests { } #[test] + #[serial_test::serial] fn typed_provider_deserializes_env_vars() { unsafe { env::set_var("TEST_ENABLED", "true"); diff --git a/core/configs/src/server_config/server.rs b/core/configs/src/server_config/server.rs index 2420fcc07b..c2a3455a0d 100644 --- a/core/configs/src/server_config/server.rs +++ b/core/configs/src/server_config/server.rs @@ -61,8 +61,13 @@ pub const SERVER_PROCESS_ENV_VARS: &[&str] = &[ "IGGY_SHARD_RUNTIME_CAPACITY", "IGGY_SHARD_EVENT_INTERVAL", "IGGY_CI_BUILD", + "IGGY_HOME", + "IGGY_USERNAME", + "IGGY_PASSWORD", ]; +pub(crate) const SERVER_ALLOWED_ENV_PREFIXES: &[&str] = &["IGGY_CONNECTORS_", "IGGY_MCP_"]; + const DEFAULT_CONFIG_PATH: &str = "core/server/config.toml"; /// Server config keys that became per-topic options, or went away with the @@ -261,6 +266,7 @@ impl ServerConfig { .chain(SERVER_PROCESS_ENV_VARS.iter().copied()) .collect(), ) + .with_allowed_env_prefixes(SERVER_ALLOWED_ENV_PREFIXES) } /// All recognised env var names for [`ServerConfig`]. From 2588e99268e03905d47a040acca9fb8126a6cf52 Mon Sep 17 00:00:00 2001 From: Gunther Xing Date: Sun, 13 Sep 2026 23:57:42 +0800 Subject: [PATCH 126/182] feat(python): add high-level producer API (#4156) Closes #4019 --- examples/python/README.md | 38 + .../python/high-level/background_producer.py | 108 ++ examples/python/high-level/consumer.py | 93 + examples/python/high-level/producer.py | 122 ++ foreign/python/README.md | 149 ++ foreign/python/apache_iggy.pyi | 246 +++ foreign/python/src/client.rs | 110 ++ foreign/python/src/lib.rs | 11 + foreign/python/src/partitioning.rs | 27 +- foreign/python/src/producer.rs | 1307 ++++++++++++++ foreign/python/src/send_message.rs | 26 +- foreign/python/tests/test_producer.py | 1608 +++++++++++++++++ 12 files changed, 3830 insertions(+), 15 deletions(-) create mode 100644 examples/python/high-level/background_producer.py create mode 100644 examples/python/high-level/consumer.py create mode 100644 examples/python/high-level/producer.py create mode 100644 foreign/python/src/producer.rs create mode 100644 foreign/python/tests/test_producer.py diff --git a/examples/python/README.md b/examples/python/README.md index 14ed2adb7e..8bbc4f03dc 100644 --- a/examples/python/README.md +++ b/examples/python/README.md @@ -56,6 +56,44 @@ pip install ../../foreign/python . ## Basic Examples +### High-Level Producer and Consumer + +The Python high-level producer API is a port of the Rust high-level producer +API. For detailed producer behavior and configuration, see the +[Rust high-level SDK documentation](https://iggy.apache.org/docs/sdk/rust/high-level-sdk/). + +The high-level producer binds the destination once, initializes missing +resources, applies producer-level batching and retry settings, and shuts down +deterministically through an async context manager. The high-level consumer +joins a consumer group, polls all assigned partitions, invokes an async handler, +and commits each message after it has been handled. + +Run either producer first. Both create the stream and topic and publish 12 +messages for the consumer, which exits after receiving all of them. `producer.py` +uses direct mode and waits for server confirmations; `background_producer.py` +uses bounded background workers and flushes them on context-manager exit: + +```bash +# Using uv +uv run high-level/producer.py +uv run high-level/consumer.py + +# Or use the background producer before starting the same consumer +uv run high-level/background_producer.py +uv run high-level/consumer.py + +# Without using uv +python high-level/producer.py +python high-level/consumer.py + +# Or use the background producer before starting the same consumer +python high-level/background_producer.py +python high-level/consumer.py +``` + +The existing examples below use the low-level `IggyClient.send_messages()` API +and remain useful when each call needs to specify its own destination. + ### Getting Started Perfect introduction for newcomers to Iggy: diff --git a/examples/python/high-level/background_producer.py b/examples/python/high-level/background_producer.py new file mode 100644 index 0000000000..41033ee324 --- /dev/null +++ b/examples/python/high-level/background_producer.py @@ -0,0 +1,108 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +import argparse +import asyncio +from datetime import timedelta +from typing import NamedTuple + +from apache_iggy import ( + BackgroundProducerConfig, + BackpressureMode, + IggyClient, + Partitioning, + ProducerSharding, + SendMessage, +) +from loguru import logger + +STREAM_NAME = "high-level-stream" +TOPIC_NAME = "high-level-topic" +MESSAGES_TO_SEND = 12 + + +class ArgNamespace(NamedTuple): + connection_string: str + + +def parse_args() -> ArgNamespace: + parser = argparse.ArgumentParser() + parser.add_argument( + "connection_string", + help=( + "Connection string for Iggy, for example " + "'iggy+tcp://iggy:iggy@127.0.0.1:8090'" + ), + default="iggy+tcp://iggy:iggy@127.0.0.1:8090", + nargs="?", + ) + return ArgNamespace(**vars(parser.parse_args())) + + +async def main() -> None: + args = parse_args() + client = IggyClient.from_connection_string(args.connection_string) + + logger.info("Connecting to Iggy") + await client.connect() + + producer = await client.producer( + STREAM_NAME, + TOPIC_NAME, + partitioning=Partitioning.balanced(), + mode=BackgroundProducerConfig( + num_shards=4, + linger_time=timedelta(milliseconds=10), + batch_size=1024 * 1024, + batch_length=100, + max_buffer_size=32 * 1024 * 1024, + failure_mode=BackpressureMode.block_with_timeout(timedelta(seconds=1)), + max_in_flight=4, + # Ordered keeps one destination on one sequential worker. Balanced + # uses all workers but does not guarantee order for one destination. + sharding=ProducerSharding.ORDERED, + ), + create_stream_if_not_exists=True, + create_topic_if_not_exists=True, + topic_partitions_count=3, + send_retries=3, + send_retry_interval=timedelta(seconds=1), + ) + + async with producer: + first = await producer.send_one(SendMessage("background message 0")) + logger.info( + "Dispatcher accepted the first message with {} confirmations", + len(first.confirmations), + ) + + messages = [ + SendMessage(f"background message {index}") + for index in range(1, MESSAGES_TO_SEND) + ] + accepted = await producer.send(messages) + logger.info( + "Dispatcher accepted {} more messages with {} confirmations", + len(messages), + len(accepted.confirmations), + ) + + logger.info("Shutdown flushed every accepted message") + + +if __name__ == "__main__": + asyncio.run(main()) diff --git a/examples/python/high-level/consumer.py b/examples/python/high-level/consumer.py new file mode 100644 index 0000000000..67253e2b38 --- /dev/null +++ b/examples/python/high-level/consumer.py @@ -0,0 +1,93 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +import argparse +import asyncio +from datetime import timedelta +from typing import NamedTuple + +from apache_iggy import ( + AutoCommit, + AutoCommitAfter, + IggyClient, + PollingStrategy, + ReceiveMessage, +) +from loguru import logger + +STREAM_NAME = "high-level-stream" +TOPIC_NAME = "high-level-topic" +CONSUMER_GROUP_NAME = "high-level-consumer" +MESSAGES_TO_CONSUME = 12 + + +class ArgNamespace(NamedTuple): + connection_string: str + + +def parse_args() -> ArgNamespace: + parser = argparse.ArgumentParser() + parser.add_argument( + "connection_string", + help=( + "Connection string for Iggy, for example " + "'iggy+tcp://iggy:iggy@127.0.0.1:8090'" + ), + default="iggy+tcp://iggy:iggy@127.0.0.1:8090", + nargs="?", + ) + return ArgNamespace(**vars(parser.parse_args())) + + +async def main() -> None: + args = parse_args() + client = IggyClient.from_connection_string(args.connection_string) + + logger.info("Connecting to Iggy") + await client.connect() + + consumer = await client.consumer_group( + CONSUMER_GROUP_NAME, + STREAM_NAME, + TOPIC_NAME, + polling_strategy=PollingStrategy.First(), + batch_length=100, + auto_commit=AutoCommit.After(AutoCommitAfter.ConsumingEachMessage()), + poll_interval=timedelta(milliseconds=100), + ) + + shutdown_event = asyncio.Event() + consumed_messages = 0 + + async def handle_message(message: ReceiveMessage) -> None: + nonlocal consumed_messages + consumed_messages += 1 + logger.info( + "Received message from partition {} at offset {}: {}", + message.partition_id(), + message.offset(), + message.payload().decode("utf-8"), + ) + if consumed_messages == MESSAGES_TO_CONSUME: + shutdown_event.set() + + await consumer.consume_messages(handle_message, shutdown_event) + logger.info("Consumed {} messages, exiting", consumed_messages) + + +if __name__ == "__main__": + asyncio.run(main()) diff --git a/examples/python/high-level/producer.py b/examples/python/high-level/producer.py new file mode 100644 index 0000000000..ceb01da0a7 --- /dev/null +++ b/examples/python/high-level/producer.py @@ -0,0 +1,122 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +import argparse +import asyncio +from datetime import timedelta +from typing import NamedTuple + +from apache_iggy import DirectProducerConfig, IggyClient, Partitioning, SendMessage +from loguru import logger + +STREAM_NAME = "high-level-stream" +TOPIC_NAME = "high-level-topic" +SEND_TO_STREAM_NAME = "high-level-send-to-stream" +SEND_TO_TOPIC_NAME = "high-level-send-to-topic" + + +class ArgNamespace(NamedTuple): + connection_string: str + + +def parse_args() -> ArgNamespace: + parser = argparse.ArgumentParser() + parser.add_argument( + "connection_string", + help=( + "Connection string for Iggy, for example " + "'iggy+tcp://iggy:iggy@127.0.0.1:8090'" + ), + default="iggy+tcp://iggy:iggy@127.0.0.1:8090", + nargs="?", + ) + return ArgNamespace(**vars(parser.parse_args())) + + +async def ensure_send_to_destination(client: IggyClient) -> None: + if await client.get_stream(SEND_TO_STREAM_NAME) is None: + await client.create_stream(SEND_TO_STREAM_NAME) + if await client.get_topic(SEND_TO_STREAM_NAME, SEND_TO_TOPIC_NAME) is None: + await client.create_topic( + stream=SEND_TO_STREAM_NAME, + name=SEND_TO_TOPIC_NAME, + partitions_count=2, + ) + + +async def main() -> None: + args = parse_args() + client = IggyClient.from_connection_string(args.connection_string) + + logger.info("Connecting to Iggy") + await client.connect() + await ensure_send_to_destination(client) + + producer = await client.producer( + STREAM_NAME, + TOPIC_NAME, + partitioning=Partitioning.balanced(), + mode=DirectProducerConfig( + batch_length=100, + linger_time=timedelta(milliseconds=5), + ), + create_stream_if_not_exists=True, + create_topic_if_not_exists=True, + topic_partitions_count=3, + topic_message_expiry=None, + topic_max_size=None, + send_retries=3, + send_retry_interval=timedelta(seconds=1), + ) + + async with producer: + confirmation = await producer.send_one(SendMessage("single message")) + logger.info( + "Sent one message with {} confirmation(s)", + len(confirmation.confirmations), + ) + + messages = [SendMessage(f"batch message {index}") for index in range(10)] + confirmation = await producer.send(messages) + logger.info( + "Sent a batch with {} confirmation(s)", + len(confirmation.confirmations), + ) + + confirmation = await producer.send_with_partitioning( + [SendMessage("partitioned message")], + Partitioning.partition_id(1), + ) + logger.info( + "Sent to a selected partition with {} confirmation(s)", + len(confirmation.confirmations), + ) + + confirmation = await producer.send_to( + SEND_TO_STREAM_NAME, + SEND_TO_TOPIC_NAME, + [SendMessage("message for another topic")], + Partitioning.partition_id(0), + ) + logger.info( + "Sent to another topic with {} confirmation(s)", + len(confirmation.confirmations), + ) + + +if __name__ == "__main__": + asyncio.run(main()) diff --git a/foreign/python/README.md b/foreign/python/README.md index 1718d2d89b..c727cb5aec 100644 --- a/foreign/python/README.md +++ b/foreign/python/README.md @@ -209,6 +209,155 @@ async def main(): asyncio.run(main()) ``` +## High-Level Producer + +The Python high-level producer API is a port of the Rust high-level producer +API. For detailed producer semantics and configuration guidance, see the +[Rust high-level SDK documentation](https://iggy.apache.org/docs/sdk/rust/high-level-sdk/). + +Use `IggyClient.producer()` when an application repeatedly publishes to one +stream and topic. Producer creation is asynchronous because it initializes the +destination before returning. By default, it creates a missing stream and topic, +uses balanced partitioning, sends directly in batches of up to 1,000 messages, +and retries failed sends up to three times with a one-second retry interval. + +The default mode is direct. Pass `BackgroundProducerConfig` to queue sends on +background workers instead. + +```python +import asyncio +from datetime import timedelta + +from apache_iggy import DirectProducerConfig, IggyClient, Partitioning, SendMessage + + +async def main(): + client = IggyClient.from_connection_string("iggy+tcp://iggy:iggy@127.0.0.1:8090") + await client.connect() + + producer = await client.producer( + "orders", + "created", + partitioning=Partitioning.balanced(), + mode=DirectProducerConfig( + batch_length=500, + linger_time=timedelta(milliseconds=5), + ), + create_stream_if_not_exists=True, + create_topic_if_not_exists=True, + topic_partitions_count=3, + topic_message_expiry=None, + topic_max_size=None, + send_retries=3, + send_retry_interval=timedelta(seconds=1), + ) + + async with producer: + await producer.send_one(SendMessage("order-1")) + response = await producer.send([SendMessage("order-2"), SendMessage("order-3")]) + print(f"Received {len(response.confirmations)} partition confirmations") + + +asyncio.run(main()) +``` + +The producer is bound to the stream and topic passed to `producer()`. Use +`send_with_partitioning(messages, partitioning)` to override its partitioning +strategy for one call, or `send_to(stream, topic, messages, partitioning)` to +send to another existing destination. `send_to()` does not create or initialize +that destination. + +Direct sends use at-least-once delivery. A request can commit even when its +response is lost, so any retry can write the same batch again. `send_retries` +counts retries after the initial attempt. The first retry runs immediately, and +`send_retry_interval` delays only later retries. Set `send_retries` to `None` or +`0` to disable producer retries. Set `send_retry_interval` to `None` to run all +enabled retries without a delay. A zero interval raises `ValueError`. + +Transport retries are separate from producer retries. For example, the default +`HttpConfig(retries=3)` gives each producer attempt up to four HTTP attempts. + +A failed direct send raises `ProducerSendError`, which is a `RuntimeError` +subclass. Its `cause` property contains the underlying error text. Its +`committed` property contains confirmations from completed chunks, and `failed` +contains the remaining unconfirmed messages. If encryption is enabled, the +failed messages contain encrypted payloads. Restore the original payloads +before submitting them to the same producer again. + +### Background Mode + +A successful background send means the dispatcher accepted the messages. The +worker writes them later, so all four send methods return a +`SendMessagesResponse` with an empty `confirmations` list. It does not mean the +server has committed the messages. + +```python +from datetime import timedelta + +from apache_iggy import ( + BackgroundProducerConfig, + BackpressureMode, + ProducerSharding, + SendMessage, +) + +producer = await client.producer( + "orders", + "created", + mode=BackgroundProducerConfig( + num_shards=4, + linger_time=timedelta(milliseconds=10), + batch_size=1024 * 1024, + batch_length=100, + max_buffer_size=32 * 1024 * 1024, + failure_mode=BackpressureMode.block_with_timeout(timedelta(seconds=1)), + max_in_flight=4, + sharding=ProducerSharding.ORDERED, + ), +) + +async with producer: + accepted = await producer.send_one(SendMessage("order-1")) + assert accepted.confirmations == [] +``` + +Each shard flushes when any configured condition is met: `batch_length` queued +send calls, `batch_size` reported bytes, or `linger_time` since the first send +entered an empty buffer. `batch_length` counts calls, not individual messages. +Zero disables either batching threshold, while a zero linger flushes as soon as +the worker receives a send. + +`ProducerSharding.ORDERED` hashes the stream/topic destination so its sends use +one sequential worker and retain dispatch order. `ProducerSharding.BALANCED` +assigns consecutive sends round-robin across the shards for throughput; ordering +for one destination is not guaranteed. + +`max_buffer_size` bounds bytes queued or in flight across the whole producer. +When it is full, `BackpressureMode.block()` waits indefinitely, +`block_with_timeout(duration)` waits up to that duration, and +`fail_immediately()` raises `RuntimeError` without accepting the batch. A single +batch larger than the whole budget always fails. A zero byte budget is +unlimited. `max_in_flight` separately bounds concurrent write requests across +all shards; zero uses the runtime maximum. + +Producer retries and transport reconnection happen inside background workers. +The Python API does not currently expose a background error callback, so a write +that still fails after its retries is logged by the Rust SDK and its unconfirmed +messages are dropped. + +See the complete runnable +[`background_producer.py`](../../examples/python/high-level/background_producer.py) +example for all background configuration fields and deterministic shutdown. + +Cleanup is asynchronous and must be explicit. Prefer `async with`, as above, so +shutdown runs on both successful and exceptional exits. Otherwise, call +`await producer.shutdown()` in a `finally` block. Shutdown waits for active +sends, is safe to call more than once, and rejects later sends with +`RuntimeError`. In background mode it also drains every queue and flushes all +accepted messages before returning. Object destruction does not perform +asynchronous cleanup; dropping a background producer without `shutdown()` can +lose buffered messages. + ## Examples Refer to the [examples/python/](https://github.com/apache/iggy/tree/master/examples/python) directory for usage examples. diff --git a/foreign/python/apache_iggy.pyi b/foreign/python/apache_iggy.pyi index 1150fa018d..e628a7c4d3 100644 --- a/foreign/python/apache_iggy.pyi +++ b/foreign/python/apache_iggy.pyi @@ -31,12 +31,15 @@ __all__ = [ "AutoCommitAfter", "AutoCommitWhen", "AutoLogin", + "BackgroundProducerConfig", + "BackpressureMode", "CacheMetrics", "CacheMetricsKey", "Consumer", "ConsumerGroup", "ConsumerGroupDetails", "ConsumerGroupMember", + "DirectProducerConfig", "GlobalPermissions", "HeaderKey", "HeaderValue", @@ -44,12 +47,15 @@ __all__ = [ "IggyClient", "IggyConsumer", "IggyExpiry", + "IggyProducer", "MaxTopicSize", "OptionSpec", "Partition", "Partitioning", "Permissions", "PollingStrategy", + "ProducerSendError", + "ProducerSharding", "QuicConfig", "QuicReconnectionConfig", "ReceiveMessage", @@ -302,6 +308,107 @@ class AutoLogin: """ def __repr__(self) -> builtins.str: ... +@typing.final +class BackgroundProducerConfig: + r""" + Immutable configuration for a producer that queues sends on background workers. + + For detailed background-producer semantics, see + https://iggy.apache.org/docs/sdk/rust/high-level-sdk/. + """ + @property + def num_shards(self) -> builtins.int: + r""" + Number of background worker shards, each with its own queue. + A value of zero is treated as one shard. + """ + @property + def linger_time(self) -> datetime.timedelta: + r""" + Maximum time a worker holds a non-empty buffer before flushing it. + A zero duration flushes as soon as the worker receives a send. + """ + @property + def batch_size(self) -> builtins.int: + r""" + Per-worker flush threshold in buffered bytes. + A value of zero disables this threshold. + """ + @property + def batch_length(self) -> builtins.int: + r""" + Per-worker flush threshold in queued sends, not individual messages. + A value of zero disables this threshold. + """ + @property + def max_buffer_size(self) -> builtins.int: + r""" + Maximum bytes buffered or in flight across all worker shards. + A value of zero makes the byte budget unlimited. + """ + @property + def failure_mode(self) -> BackpressureMode: + r""" + Behavior when `max_buffer_size` is exhausted. + """ + @property + def max_in_flight(self) -> builtins.int: + r""" + Maximum number of requests written concurrently across all workers. + A value of zero uses the runtime's maximum semaphore permit count. + """ + @property + def sharding(self) -> ProducerSharding: + r""" + Strategy used to assign each send to a worker shard. + Ordered sharding preserves per-destination dispatch order, while balanced + sharding distributes sends round-robin and may reorder them. + """ + def __new__( + cls, + *, + num_shards: builtins.int = 1, + linger_time: datetime.timedelta = ..., + batch_size: builtins.int = 1048576, + batch_length: builtins.int = 1000, + max_buffer_size: builtins.int = 33554432, + failure_mode: BackpressureMode = ..., + max_in_flight: builtins.int = 1, + sharding: ProducerSharding = ..., + ) -> BackgroundProducerConfig: + r""" + Constructs background batching, capacity, backpressure, and sharding configuration. + """ + def __repr__(self) -> builtins.str: ... + +@typing.final +class BackpressureMode: + r""" + What a background send does when the producer buffer is full. + """ + @property + def timeout(self) -> datetime.timedelta | None: + r""" + The configured timeout, or `None` for modes without one. + """ + def __eq__(self, other: builtins.object, /) -> builtins.bool: ... + @staticmethod + def block() -> BackpressureMode: + r""" + Wait indefinitely for buffer capacity. + """ + @staticmethod + def block_with_timeout(timeout: datetime.timedelta) -> BackpressureMode: + r""" + Wait up to `timeout` for buffer capacity. + """ + @staticmethod + def fail_immediately() -> BackpressureMode: + r""" + Fail immediately when the producer buffer is full. + """ + def __repr__(self) -> builtins.str: ... + @typing.final class CacheMetrics: r""" @@ -451,6 +558,30 @@ class ConsumerGroupMember: Gets the collection of partitions the consumer group member is consuming. """ +@typing.final +class DirectProducerConfig: + r""" + Configuration for a producer that sends from the calling task. + """ + @property + def batch_length(self) -> builtins.int: + r""" + Maximum number of messages sent in one request. + A value of zero uses the internal limit of 1,000,000 messages. + """ + @property + def linger_time(self) -> datetime.timedelta: + r""" + Minimum gap requested between sequential direct sends. + """ + def __new__( + cls, *, batch_length: builtins.int = 1000, linger_time: datetime.timedelta = ... + ) -> DirectProducerConfig: + r""" + Constructs direct-producer batching and pacing configuration. + """ + def __repr__(self) -> builtins.str: ... + @typing.final class GlobalPermissions: r""" @@ -1670,6 +1801,32 @@ class IggyClient: the supported unsigned 32-bit range. RuntimeError: If the request fails. """ + def producer( + self, + stream: builtins.str, + topic: builtins.str, + partitioning: Partitioning | None = None, + mode: DirectProducerConfig | BackgroundProducerConfig | None = None, + create_stream_if_not_exists: builtins.bool = True, + create_topic_if_not_exists: builtins.bool = True, + topic_partitions_count: builtins.int = 1, + topic_message_expiry: IggyExpiry | None = None, + topic_max_size: MaxTopicSize | None = None, + send_retries: builtins.int | None = 3, + send_retry_interval: datetime.timedelta | None = ..., + ) -> collections.abc.Awaitable[IggyProducer]: + r""" + Creates and initializes a high-level producer bound to a stream and topic. + + This is a Python port of the Rust high-level producer API. For detailed + producer semantics, see https://iggy.apache.org/docs/sdk/rust/high-level-sdk/. + `None` selects direct mode. `BackgroundProducerConfig` starts background + workers and makes successful sends mean queue acceptance rather than a + server commit. The returned producer is ready to send. + + Raises `ValueError` for invalid names or numeric ranges and `RuntimeError` + when stream/topic initialization fails. + """ def poll_messages( self, stream: builtins.str | builtins.int, @@ -1863,6 +2020,63 @@ class IggyExpiry: ... +@typing.final +class IggyProducer: + r""" + Python port of the Rust high-level producer API, bound to one stream and topic. + + For detailed producer semantics, see + https://iggy.apache.org/docs/sdk/rust/high-level-sdk/. + + Direct sends complete after the server responds and contain commit confirmations. + Background sends complete once accepted by a worker and contain no confirmations. + Always use the async context manager or call `shutdown()` explicitly; dropping a + background producer can lose accepted buffered messages. + """ + def send( + self, messages: list[SendMessage] + ) -> collections.abc.Awaitable[SendMessagesResponse]: + r""" + Sends a batch to the producer's bound stream and topic. + In background mode, success means accepted into the dispatcher and the + returned confirmation list is empty. + """ + def send_one( + self, message: SendMessage + ) -> collections.abc.Awaitable[SendMessagesResponse]: + r""" + Sends one message to the producer's bound stream and topic. + It has the same mode-dependent completion semantics as `send()`. + """ + def send_with_partitioning( + self, messages: list[SendMessage], partitioning: Partitioning | None = None + ) -> collections.abc.Awaitable[SendMessagesResponse]: + r""" + Sends a batch with an optional per-call partitioning override. + It has the same mode-dependent completion semantics as `send()`. + """ + def send_to( + self, + stream: builtins.str | builtins.int, + topic: builtins.str | builtins.int, + messages: list[SendMessage], + partitioning: Partitioning | None = None, + ) -> collections.abc.Awaitable[SendMessagesResponse]: + r""" + Sends a batch to another existing stream and topic. + It has the same mode-dependent completion semantics as `send()` and does + not create the alternate destination. + """ + def shutdown(self) -> collections.abc.Awaitable[None]: + r""" + Waits for active sends and closes the producer. Repeated calls are safe. + Background shutdown flushes every accepted buffered message before returning. + """ + def __aenter__(self) -> collections.abc.Awaitable[IggyProducer]: ... + def __aexit__( + self, _exc_type: typing.Any, _exc_value: typing.Any, _traceback: typing.Any + ) -> collections.abc.Awaitable[builtins.bool]: ... + class MaxTopicSize: r""" The maximum size of a topic. @@ -2080,6 +2294,29 @@ class PollingStrategy: ... +@typing.final +class ProducerSendError(builtins.RuntimeError): + r""" + A direct producer error that preserves partial-send recovery state. + """ + @property + def cause(self) -> builtins.str: + r""" + The underlying Iggy error message. + """ + @property + def failed(self) -> builtins.list[SendMessage]: + r""" + Messages without a usable confirmation after the failure. + An encryptor can leave these messages encrypted, so do not submit them + to the same producer without restoring their original payloads. + """ + @property + def committed(self) -> builtins.list[SendMessagesConfirmation]: + r""" + Confirmations returned for chunks committed before the failure. + """ + @typing.final class QuicConfig: r""" @@ -3248,6 +3485,15 @@ class WebSocketReconnectionConfig: """ def __repr__(self) -> builtins.str: ... +@typing.final +class ProducerSharding(enum.Enum): + r""" + How a background producer distributes sends among its workers. + """ + + ORDERED = ... + BALANCED = ... + @typing.final class UserStatus(enum.Enum): r""" diff --git a/foreign/python/src/client.rs b/foreign/python/src/client.rs index 40e53bb0cd..d75f7c350b 100644 --- a/foreign/python/src/client.rs +++ b/foreign/python/src/client.rs @@ -42,6 +42,7 @@ use crate::identifier::PyIdentifier; use crate::options::OptionSpec as PyOptionSpec; use crate::partitioning::PyPartitioning; use crate::permissions::Permissions as PyPermissions; +use crate::producer::{IggyProducer, ProducerMode, RetryInterval, u32_param as producer_u32_param}; use crate::receive_message::{PollingStrategy, ReceiveMessage}; use crate::send_message::{SendMessage, SendMessagesResponse as PySendMessagesResponse}; use crate::stats::Stats as PyStats; @@ -65,6 +66,10 @@ fn to_runtime_error(error: E) -> PyErr { PyErr::new::(error.to_string()) } +fn to_value_error(error: E) -> PyErr { + PyErr::new::(error.to_string()) +} + /// Resolves the shared `create_topic`/`update_topic` parameters, applying /// server defaults where the caller left them unset. fn resolve_topic_params( @@ -1321,6 +1326,111 @@ impl IggyClient { }) } + /// Creates and initializes a high-level producer bound to a stream and topic. + /// + /// This is a Python port of the Rust high-level producer API. For detailed + /// producer semantics, see https://iggy.apache.org/docs/sdk/rust/high-level-sdk/. + /// `None` selects direct mode. `BackgroundProducerConfig` starts background + /// workers and makes successful sends mean queue acceptance rather than a + /// server commit. The returned producer is ready to send. + /// + /// Raises `ValueError` for invalid names or numeric ranges and `RuntimeError` + /// when stream/topic initialization fails. + #[allow(clippy::too_many_arguments)] + #[pyo3(signature = ( + stream, + topic, + partitioning=None, + mode=None, + create_stream_if_not_exists=true, + create_topic_if_not_exists=true, + topic_partitions_count=1, + topic_message_expiry=None, + topic_max_size=None, + send_retries=Some(3), + send_retry_interval=RetryInterval::default(), + ))] + #[gen_stub(override_return_type(type_repr = "collections.abc.Awaitable[IggyProducer]", imports=("collections.abc")))] + fn producer<'a>( + &self, + py: Python<'a>, + stream: &str, + topic: &str, + #[gen_stub(override_type(type_repr = "Partitioning | None"))] partitioning: Option< + &crate::partitioning::Partitioning, + >, + #[gen_stub(override_type( + type_repr = "DirectProducerConfig | BackgroundProducerConfig | None" + ))] + mode: Option, + create_stream_if_not_exists: bool, + create_topic_if_not_exists: bool, + topic_partitions_count: i64, + #[gen_stub(override_type(type_repr = "IggyExpiry | None"))] topic_message_expiry: Option< + &IggyExpiry, + >, + #[gen_stub(override_type(type_repr = "MaxTopicSize | None"))] topic_max_size: Option< + &MaxTopicSize, + >, + send_retries: Option, + send_retry_interval: RetryInterval, + ) -> PyResult> { + let mode = mode.unwrap_or_default(); + + let topic_partitions_count = + producer_u32_param(topic_partitions_count, "topic_partitions_count")?; + let topic_message_expiry = topic_message_expiry + .map(RustIggyExpiry::try_from) + .transpose()? + .unwrap_or(RustIggyExpiry::ServerDefault); + let topic_max_size = topic_max_size + .map(RustMaxTopicSize::try_from) + .transpose()? + .unwrap_or(RustMaxTopicSize::ServerDefault); + let send_retries = send_retries + .map(|retries| producer_u32_param(retries, "send_retries")) + .transpose()? + .filter(|retries| *retries != 0); + let send_retry_interval = send_retry_interval.resolve()?; + + let mut builder = self + .inner + .producer(stream, topic) + .map_err(to_value_error)? + .send_retries(send_retries, send_retry_interval); + + builder = match mode { + ProducerMode::Direct(config) => builder.direct((&config).into()), + ProducerMode::Background(config) => builder.background((&config).try_into()?), + }; + + if let Some(partitioning) = partitioning { + builder = builder.partitioning(partitioning.inner.as_ref().clone()); + } + if create_stream_if_not_exists { + builder = builder.create_stream_if_not_exists(); + } else { + builder = builder.do_not_create_stream_if_not_exists(); + } + if create_topic_if_not_exists { + builder = builder.create_topic_if_not_exists( + topic_partitions_count, + topic_message_expiry, + topic_max_size, + ); + } else { + builder = builder.do_not_create_topic_if_not_exists(); + } + + future_into_py(py, async move { + // A background build starts Tokio worker tasks, so it must happen + // while this future is executing on the Rust runtime. + let producer = builder.build(); + producer.init().await.map_err(to_runtime_error)?; + Ok(IggyProducer::new(producer)) + }) + } + /// Polls for messages from the specified topic on behalf of the given consumer. /// Omitting `partition_id` reads partition 0 for a regular consumer, and /// polls the member's assigned partitions for a consumer group. diff --git a/foreign/python/src/lib.rs b/foreign/python/src/lib.rs index 4dbcd48c0b..6462b786d0 100644 --- a/foreign/python/src/lib.rs +++ b/foreign/python/src/lib.rs @@ -24,6 +24,7 @@ mod identifier; mod options; mod partitioning; mod permissions; +mod producer; mod receive_message; mod send_message; mod stats; @@ -44,6 +45,10 @@ use consumer::{ use options::OptionSpec; use partitioning::Partitioning; use permissions::{GlobalPermissions, Permissions, StreamPermissions, TopicPermissions}; +use producer::{ + BackgroundProducerConfig, BackpressureMode, DirectProducerConfig, IggyProducer, + ProducerSendError, ProducerSharding, +}; use pyo3::prelude::*; use receive_message::{PollingStrategy, ReceiveMessage}; use send_message::{SendMessage, SendMessagesConfirmation, SendMessagesResponse}; @@ -62,6 +67,12 @@ fn apache_iggy(py: Python, m: &Bound<'_, PyModule>) -> PyResult<()> { m.add_class::()?; m.add_class::()?; m.add_class::()?; + m.add_class::()?; + m.add_class::()?; + m.add_class::()?; + m.add_class::()?; + m.add_class::()?; + m.add_class::()?; m.add_class::()?; m.add_class::()?; m.add_class::()?; diff --git a/foreign/python/src/partitioning.rs b/foreign/python/src/partitioning.rs index 83d9715d8c..bb345cf511 100644 --- a/foreign/python/src/partitioning.rs +++ b/foreign/python/src/partitioning.rs @@ -15,6 +15,8 @@ // specific language governing permissions and limitations // under the License. +use std::sync::Arc; + use iggy::prelude::Partitioning as RustPartitioning; use pyo3::{exceptions::PyValueError, prelude::*, types::PyBytes}; use pyo3_stub_gen::{ @@ -27,7 +29,7 @@ use pyo3_stub_gen::{ #[pyclass(from_py_object)] #[gen_stub_pyclass] pub struct Partitioning { - pub(crate) inner: RustPartitioning, + pub(crate) inner: Arc, } #[gen_stub_pymethods] @@ -37,7 +39,7 @@ impl Partitioning { #[staticmethod] pub fn balanced() -> Self { Self { - inner: RustPartitioning::balanced(), + inner: Arc::new(RustPartitioning::balanced()), } } @@ -53,7 +55,7 @@ impl Partitioning { #[staticmethod] pub fn partition_id(partition_id: u32) -> Self { Self { - inner: RustPartitioning::partition_id(partition_id), + inner: Arc::new(RustPartitioning::partition_id(partition_id)), } } @@ -73,7 +75,9 @@ impl Partitioning { }; let inner = RustPartitioning::messages_key(&key) .map_err(|error| PyValueError::new_err(error.to_string()))?; - Ok(Self { inner }) + Ok(Self { + inner: Arc::new(inner), + }) } } @@ -98,8 +102,21 @@ impl_stub_type!(PyPartitioning = Partitioning | isize); impl From for RustPartitioning { fn from(partitioning: PyPartitioning) -> Self { match partitioning { - PyPartitioning::Strategy(partitioning) => partitioning.inner, + PyPartitioning::Strategy(partitioning) => partitioning.inner.as_ref().clone(), PyPartitioning::PartitionId(partition_id) => Self::partition_id(partition_id), } } } + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn clone_shares_the_rust_partitioning() { + let partitioning = Partitioning::balanced(); + let cloned = partitioning.clone(); + + assert!(Arc::ptr_eq(&partitioning.inner, &cloned.inner)); + } +} diff --git a/foreign/python/src/producer.rs b/foreign/python/src/producer.rs new file mode 100644 index 0000000000..c51e34d2fa --- /dev/null +++ b/foreign/python/src/producer.rs @@ -0,0 +1,1307 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use std::{str::FromStr, sync::Arc}; + +use iggy::clients::producer_config::BackpressureMode as RustBackpressureMode; +use iggy::prelude::{ + BackgroundConfig as RustBackgroundConfig, BalancedSharding, DirectConfig as RustDirectConfig, + Identifier, IggyByteSize, IggyDuration, IggyError, IggyMessage as RustIggyMessage, + IggyProducer as RustIggyProducer, OrderedSharding, + SendMessagesConfirmationResponse as RustSendMessagesConfirmationResponse, Sharding, +}; +use pyo3::IntoPyObjectExt; +use pyo3::conversion::FromPyObject; +use pyo3::exceptions::{PyRuntimeError, PyTypeError, PyValueError}; +use pyo3::prelude::*; +use pyo3::types::{PyAny, PyDelta, PyInt, PyList, PyString}; +use pyo3_async_runtimes::tokio::future_into_py; +use pyo3_stub_gen::derive::{gen_stub_pyclass, gen_stub_pyclass_enum, gen_stub_pymethods}; +use pyo3_stub_gen::{PyStubType, TypeInfo}; +use tokio::sync::{RwLock, Semaphore}; + +use crate::duration::{duration_repr, iggy_duration_to_py_delta, py_delta_to_iggy_duration}; +use crate::partitioning::Partitioning; +use crate::send_message::{SendMessage, SendMessagesConfirmation, SendMessagesResponse}; + +const DEFAULT_BACKGROUND_NUM_SHARDS: usize = 1; +const DEFAULT_BACKGROUND_BATCH_SIZE: usize = 1024 * 1024; +const DEFAULT_BACKGROUND_BATCH_LENGTH: usize = 1_000; +const DEFAULT_BACKGROUND_MAX_BUFFER_SIZE: u64 = 32 * 1024 * 1024; +const DEFAULT_BACKGROUND_MAX_IN_FLIGHT: usize = 1; + +/// Configuration for a producer that sends from the calling task. +#[derive(Clone)] +#[gen_stub_pyclass] +#[pyclass(frozen, from_py_object)] +pub struct DirectProducerConfig { + pub(crate) inner: RustDirectConfig, +} + +impl Default for DirectProducerConfig { + fn default() -> Self { + Self { + inner: RustDirectConfig::builder().build(), + } + } +} + +impl From<&DirectProducerConfig> for RustDirectConfig { + fn from(config: &DirectProducerConfig) -> Self { + config.inner.clone() + } +} + +#[gen_stub_pymethods] +#[pymethods] +impl DirectProducerConfig { + /// Constructs direct-producer batching and pacing configuration. + #[new] + #[pyo3(signature = (*, batch_length=1000, linger_time=DefaultDuration::default()))] + fn new(batch_length: i64, linger_time: DefaultDuration) -> PyResult { + let batch_length = u32_param(batch_length, "batch_length")?; + let linger_time = linger_time.resolve(IggyDuration::from(0))?; + if linger_time.get_duration().as_micros() > u128::from(u64::MAX) { + return Err(PyValueError::new_err(format!( + "'linger_time' must not exceed {} microseconds", + u64::MAX + ))); + } + Ok(Self { + inner: RustDirectConfig::builder() + .batch_length(batch_length) + .linger_time(linger_time) + .build(), + }) + } + + /// Maximum number of messages sent in one request. + /// A value of zero uses the internal limit of 1,000,000 messages. + #[getter] + fn batch_length(&self) -> u32 { + self.inner.batch_length + } + + /// Minimum gap requested between sequential direct sends. + #[gen_stub(override_return_type(type_repr = "datetime.timedelta", imports=("datetime")))] + #[getter] + fn linger_time<'py>(&self, py: Python<'py>) -> PyResult> { + iggy_duration_to_py_delta(py, self.inner.linger_time) + } + + fn __repr__(&self) -> String { + format!( + "DirectProducerConfig(batch_length={}, linger_time={})", + self.inner.batch_length, + duration_repr(self.inner.linger_time) + ) + } +} + +/// How a background producer distributes sends among its workers. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +#[gen_stub_pyclass_enum] +#[pyclass(eq, from_py_object, rename_all = "UPPERCASE")] +pub enum ProducerSharding { + Ordered, + Balanced, +} + +impl From for Box { + fn from(sharding: ProducerSharding) -> Self { + match sharding { + ProducerSharding::Ordered => Box::new(OrderedSharding), + ProducerSharding::Balanced => Box::new(BalancedSharding::default()), + } + } +} + +/// What a background send does when the producer buffer is full. +#[derive(Debug, Clone, PartialEq, Eq)] +#[gen_stub_pyclass] +#[pyclass(eq, frozen, from_py_object)] +pub struct BackpressureMode { + kind: BackpressureKind, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +enum BackpressureKind { + Block, + BlockWithTimeout(IggyDuration), + FailImmediately, +} + +impl From<&BackpressureMode> for RustBackpressureMode { + fn from(mode: &BackpressureMode) -> Self { + match mode.kind { + BackpressureKind::Block => Self::Block, + BackpressureKind::BlockWithTimeout(timeout) => Self::BlockWithTimeout(timeout), + BackpressureKind::FailImmediately => Self::FailImmediately, + } + } +} + +#[gen_stub_pymethods] +#[pymethods] +impl BackpressureMode { + /// Wait indefinitely for buffer capacity. + #[staticmethod] + fn block() -> Self { + Self { + kind: BackpressureKind::Block, + } + } + + /// Wait up to `timeout` for buffer capacity. + #[staticmethod] + fn block_with_timeout( + #[gen_stub(override_type(type_repr = "datetime.timedelta", imports=("datetime")))] + timeout: Py, + ) -> PyResult { + Ok(Self { + kind: BackpressureKind::BlockWithTimeout(py_delta_to_iggy_duration(&timeout)?), + }) + } + + /// Fail immediately when the producer buffer is full. + #[staticmethod] + fn fail_immediately() -> Self { + Self { + kind: BackpressureKind::FailImmediately, + } + } + + /// The configured timeout, or `None` for modes without one. + #[gen_stub(override_return_type(type_repr = "datetime.timedelta | None", imports=("datetime")))] + #[getter] + fn timeout<'py>(&self, py: Python<'py>) -> PyResult>> { + match self.kind { + BackpressureKind::BlockWithTimeout(timeout) => { + iggy_duration_to_py_delta(py, timeout).map(Some) + } + BackpressureKind::Block | BackpressureKind::FailImmediately => Ok(None), + } + } + + fn __repr__(&self) -> String { + match self.kind { + BackpressureKind::Block => "BackpressureMode.block()".to_owned(), + BackpressureKind::BlockWithTimeout(timeout) => format!( + "BackpressureMode.block_with_timeout({})", + duration_repr(timeout) + ), + BackpressureKind::FailImmediately => "BackpressureMode.fail_immediately()".to_owned(), + } + } +} + +/// Immutable configuration for a producer that queues sends on background workers. +/// +/// For detailed background-producer semantics, see +/// https://iggy.apache.org/docs/sdk/rust/high-level-sdk/. +#[derive(Clone)] +#[gen_stub_pyclass] +#[pyclass(frozen, from_py_object)] +pub struct BackgroundProducerConfig { + num_shards: usize, + linger_time: IggyDuration, + batch_size: usize, + batch_length: usize, + max_buffer_size: IggyByteSize, + failure_mode: BackpressureMode, + max_in_flight: usize, + sharding: ProducerSharding, +} + +impl Default for BackgroundProducerConfig { + fn default() -> Self { + Self { + num_shards: DEFAULT_BACKGROUND_NUM_SHARDS, + linger_time: IggyDuration::from(1_000), + batch_size: DEFAULT_BACKGROUND_BATCH_SIZE, + batch_length: DEFAULT_BACKGROUND_BATCH_LENGTH, + max_buffer_size: IggyByteSize::from(DEFAULT_BACKGROUND_MAX_BUFFER_SIZE), + failure_mode: BackpressureMode::block(), + max_in_flight: DEFAULT_BACKGROUND_MAX_IN_FLIGHT, + sharding: ProducerSharding::Ordered, + } + } +} + +impl TryFrom<&BackgroundProducerConfig> for RustBackgroundConfig { + type Error = PyErr; + + fn try_from(config: &BackgroundProducerConfig) -> PyResult { + let max_buffer_size = config.max_buffer_size.as_bytes_u64(); + if max_buffer_size != 0 && max_buffer_size > Semaphore::MAX_PERMITS as u64 { + return Err(PyValueError::new_err(format!( + "'max_buffer_size' must not exceed {}", + Semaphore::MAX_PERMITS + ))); + } + if config.max_in_flight != 0 && config.max_in_flight > Semaphore::MAX_PERMITS { + return Err(PyValueError::new_err(format!( + "'max_in_flight' must not exceed {}", + Semaphore::MAX_PERMITS + ))); + } + + Ok(RustBackgroundConfig::builder() + .num_shards(config.num_shards) + .linger_time(config.linger_time) + .batch_size(config.batch_size) + .batch_length(config.batch_length) + .max_buffer_size(config.max_buffer_size) + .failure_mode((&config.failure_mode).into()) + .max_in_flight(config.max_in_flight) + .sharding(config.sharding.into()) + .build()) + } +} + +#[gen_stub_pymethods] +#[pymethods] +impl BackgroundProducerConfig { + /// Constructs background batching, capacity, backpressure, and sharding configuration. + #[new] + #[allow(clippy::too_many_arguments)] + #[pyo3(signature = ( + *, + num_shards=1, + linger_time=DefaultDuration::one_millisecond(), + batch_size=1048576, + batch_length=1000, + max_buffer_size=33554432, + failure_mode=DefaultBackpressureMode::default(), + max_in_flight=1, + sharding=DefaultProducerSharding::default(), + ))] + fn new( + num_shards: i128, + linger_time: DefaultDuration, + batch_size: i128, + batch_length: i128, + max_buffer_size: i128, + failure_mode: DefaultBackpressureMode, + max_in_flight: i128, + sharding: DefaultProducerSharding, + ) -> PyResult { + let linger_time = linger_time.resolve(IggyDuration::from(1_000))?; + Ok(Self { + num_shards: usize_param(num_shards, "num_shards")?, + linger_time, + batch_size: usize_param(batch_size, "batch_size")?, + batch_length: usize_param(batch_length, "batch_length")?, + max_buffer_size: IggyByteSize::from(u64_param(max_buffer_size, "max_buffer_size")?), + failure_mode: failure_mode.resolve(), + max_in_flight: usize_param(max_in_flight, "max_in_flight")?, + sharding: sharding.resolve(), + }) + } + + /// Number of background worker shards, each with its own queue. + /// A value of zero is treated as one shard. + #[getter] + fn num_shards(&self) -> usize { + self.num_shards + } + + /// Maximum time a worker holds a non-empty buffer before flushing it. + /// A zero duration flushes as soon as the worker receives a send. + #[gen_stub(override_return_type(type_repr = "datetime.timedelta", imports=("datetime")))] + #[getter] + fn linger_time<'py>(&self, py: Python<'py>) -> PyResult> { + iggy_duration_to_py_delta(py, self.linger_time) + } + + /// Per-worker flush threshold in buffered bytes. + /// A value of zero disables this threshold. + #[getter] + fn batch_size(&self) -> usize { + self.batch_size + } + + /// Per-worker flush threshold in queued sends, not individual messages. + /// A value of zero disables this threshold. + #[getter] + fn batch_length(&self) -> usize { + self.batch_length + } + + /// Maximum bytes buffered or in flight across all worker shards. + /// A value of zero makes the byte budget unlimited. + #[getter] + fn max_buffer_size(&self) -> u64 { + self.max_buffer_size.as_bytes_u64() + } + + /// Behavior when `max_buffer_size` is exhausted. + #[getter] + fn failure_mode(&self) -> BackpressureMode { + self.failure_mode.clone() + } + + /// Maximum number of requests written concurrently across all workers. + /// A value of zero uses the runtime's maximum semaphore permit count. + #[getter] + fn max_in_flight(&self) -> usize { + self.max_in_flight + } + + /// Strategy used to assign each send to a worker shard. + /// Ordered sharding preserves per-destination dispatch order, while balanced + /// sharding distributes sends round-robin and may reorder them. + #[getter] + fn sharding(&self) -> ProducerSharding { + self.sharding + } + + fn __repr__(&self) -> String { + let sharding = match self.sharding { + ProducerSharding::Ordered => "ProducerSharding.ORDERED", + ProducerSharding::Balanced => "ProducerSharding.BALANCED", + }; + format!( + "BackgroundProducerConfig(num_shards={}, linger_time={}, batch_size={}, batch_length={}, max_buffer_size={}, failure_mode={}, max_in_flight={}, sharding={sharding})", + self.num_shards, + duration_repr(self.linger_time), + self.batch_size, + self.batch_length, + self.max_buffer_size.as_bytes_u64(), + self.failure_mode.__repr__(), + self.max_in_flight, + ) + } +} + +/// A direct producer error that preserves partial-send recovery state. +#[gen_stub_pyclass] +#[pyclass(frozen, extends=PyRuntimeError)] +pub struct ProducerSendError { + cause: String, + failed: Arc>, + committed: Arc>, +} + +#[gen_stub_pymethods] +#[pymethods] +impl ProducerSendError { + /// The underlying Iggy error message. + #[getter] + fn cause(&self) -> &str { + &self.cause + } + + /// Messages without a usable confirmation after the failure. + /// An encryptor can leave these messages encrypted, so do not submit them + /// to the same producer without restoring their original payloads. + #[getter] + fn failed(&self) -> Vec { + self.failed + .iter() + .map(SendMessage::clone_from_rust) + .collect() + } + + /// Confirmations returned for chunks committed before the failure. + #[getter] + fn committed(&self) -> Vec { + self.committed + .iter() + .map(SendMessagesConfirmation::from) + .collect() + } +} + +impl ProducerSendError { + fn new_err( + cause: IggyError, + failed: Arc>, + committed: Arc>, + ) -> PyErr { + Python::attach(|py| { + let cause = cause.to_string(); + let message = format!("Producer send failed: {cause}"); + let cause_error = PyRuntimeError::new_err(cause.clone()); + let instance = match Bound::new( + py, + Self { + cause, + failed, + committed, + }, + ) { + Ok(instance) => instance, + Err(error) => return error, + }; + if let Err(error) = instance.setattr("args", (message,)) { + return error; + } + let error = PyErr::from_value(instance.into_any()); + error.set_cause(py, Some(cause_error)); + error + }) + } +} + +/// Python port of the Rust high-level producer API, bound to one stream and topic. +/// +/// For detailed producer semantics, see +/// https://iggy.apache.org/docs/sdk/rust/high-level-sdk/. +/// +/// Direct sends complete after the server responds and contain commit confirmations. +/// Background sends complete once accepted by a worker and contain no confirmations. +/// Always use the async context manager or call `shutdown()` explicitly; dropping a +/// background producer can lose accepted buffered messages. +#[derive(Clone)] +#[gen_stub_pyclass] +#[pyclass(from_py_object)] +pub struct IggyProducer { + inner: Arc>>, +} + +impl IggyProducer { + pub(crate) fn new(producer: RustIggyProducer) -> Self { + Self { + inner: Arc::new(RwLock::new(Some(producer))), + } + } +} + +#[gen_stub_pymethods] +#[pymethods] +impl IggyProducer { + /// Sends a batch to the producer's bound stream and topic. + /// In background mode, success means accepted into the dispatcher and the + /// returned confirmation list is empty. + #[gen_stub(override_return_type(type_repr = "collections.abc.Awaitable[SendMessagesResponse]", imports=("collections.abc")))] + fn send<'py>( + &self, + py: Python<'py>, + #[gen_stub(override_type(type_repr = "list[SendMessage]"))] messages: &Bound<'_, PyList>, + ) -> PyResult> { + let messages = extract_messages(messages)?; + let inner = self.inner.clone(); + future_into_py(py, async move { + // Every send keeps a read guard for its full future. Concurrent sends + // remain possible while shutdown cannot consume an active producer. + let producer = inner.read().await; + let producer = producer + .as_ref() + .ok_or_else(|| PyRuntimeError::new_err("producer has been shut down"))?; + producer + .send(messages) + .await + .map(SendMessagesResponse::from) + .map_err(to_send_error) + }) + } + + /// Sends one message to the producer's bound stream and topic. + /// It has the same mode-dependent completion semantics as `send()`. + #[gen_stub(override_return_type(type_repr = "collections.abc.Awaitable[SendMessagesResponse]", imports=("collections.abc")))] + fn send_one<'py>( + &self, + py: Python<'py>, + message: PyRef<'_, SendMessage>, + ) -> PyResult> { + let message = clone_rust_message(&message); + let inner = self.inner.clone(); + future_into_py(py, async move { + let producer = inner.read().await; + let producer = producer + .as_ref() + .ok_or_else(|| PyRuntimeError::new_err("producer has been shut down"))?; + producer + .send_one(message) + .await + .map(SendMessagesResponse::from) + .map_err(to_send_error) + }) + } + + /// Sends a batch with an optional per-call partitioning override. + /// It has the same mode-dependent completion semantics as `send()`. + #[pyo3(signature = (messages, partitioning=None))] + #[gen_stub(override_return_type(type_repr = "collections.abc.Awaitable[SendMessagesResponse]", imports=("collections.abc")))] + fn send_with_partitioning<'py>( + &self, + py: Python<'py>, + #[gen_stub(override_type(type_repr = "list[SendMessage]"))] messages: &Bound<'_, PyList>, + #[gen_stub(override_type(type_repr = "Partitioning | None"))] partitioning: Option< + &Partitioning, + >, + ) -> PyResult> { + let messages = extract_messages(messages)?; + let partitioning = partitioning.map(|value| value.inner.clone()); + let inner = self.inner.clone(); + future_into_py(py, async move { + let producer = inner.read().await; + let producer = producer + .as_ref() + .ok_or_else(|| PyRuntimeError::new_err("producer has been shut down"))?; + producer + .send_with_partitioning(messages, partitioning) + .await + .map(SendMessagesResponse::from) + .map_err(to_send_error) + }) + } + + /// Sends a batch to another existing stream and topic. + /// It has the same mode-dependent completion semantics as `send()` and does + /// not create the alternate destination. + #[pyo3(signature = (stream, topic, messages, partitioning=None))] + #[gen_stub(override_return_type(type_repr = "collections.abc.Awaitable[SendMessagesResponse]", imports=("collections.abc")))] + fn send_to<'py>( + &self, + py: Python<'py>, + #[gen_stub(override_type(type_repr = "builtins.str | builtins.int"))] stream: &Bound< + '_, + PyAny, + >, + #[gen_stub(override_type(type_repr = "builtins.str | builtins.int"))] topic: &Bound< + '_, + PyAny, + >, + #[gen_stub(override_type(type_repr = "list[SendMessage]"))] messages: &Bound<'_, PyList>, + #[gen_stub(override_type(type_repr = "Partitioning | None"))] partitioning: Option< + &Partitioning, + >, + ) -> PyResult> { + let stream = Arc::new(extract_send_to_identifier(stream, "stream")?); + let topic = Arc::new(extract_send_to_identifier(topic, "topic")?); + let messages = extract_messages(messages)?; + let partitioning = partitioning.map(|value| value.inner.clone()); + let inner = self.inner.clone(); + future_into_py(py, async move { + let producer = inner.read().await; + let producer = producer + .as_ref() + .ok_or_else(|| PyRuntimeError::new_err("producer has been shut down"))?; + producer + .send_to(stream, topic, messages, partitioning) + .await + .map(SendMessagesResponse::from) + .map_err(to_send_error) + }) + } + + #[gen_stub(skip)] + fn _is_send_active(&self) -> bool { + self.inner.try_write().is_err() + } + + /// Waits for active sends and closes the producer. Repeated calls are safe. + /// Background shutdown flushes every accepted buffered message before returning. + #[gen_stub(override_return_type(type_repr = "collections.abc.Awaitable[None]", imports=("collections.abc")))] + fn shutdown<'py>(&self, py: Python<'py>) -> PyResult> { + let inner = self.inner.clone(); + future_into_py(py, async move { + shutdown(inner).await?; + Ok(Python::attach(|py| py.None())) + }) + } + + #[gen_stub(override_return_type(type_repr = "collections.abc.Awaitable[IggyProducer]", imports=("collections.abc")))] + fn __aenter__<'py>(this: Py, py: Python<'py>) -> PyResult> { + future_into_py(py, async move { Ok(this) }) + } + + #[gen_stub(override_return_type(type_repr = "collections.abc.Awaitable[builtins.bool]", imports=("collections.abc")))] + fn __aexit__<'py>( + &self, + py: Python<'py>, + _exc_type: &Bound<'_, PyAny>, + _exc_value: &Bound<'_, PyAny>, + _traceback: &Bound<'_, PyAny>, + ) -> PyResult> { + let inner = self.inner.clone(); + future_into_py(py, async move { + shutdown(inner).await?; + Ok(false) + }) + } +} + +#[derive(Clone, FromPyObject)] +pub(crate) enum ProducerMode { + #[pyo3(transparent)] + Direct(DirectProducerConfig), + #[pyo3(transparent)] + Background(BackgroundProducerConfig), +} + +impl Default for ProducerMode { + fn default() -> Self { + Self::Direct(DirectProducerConfig::default()) + } +} + +#[derive(FromPyObject)] +#[pyo3(transparent)] +struct DefaultBackpressureMode(BackpressureMode); + +impl Default for DefaultBackpressureMode { + fn default() -> Self { + Self(BackpressureMode::block()) + } +} + +impl DefaultBackpressureMode { + fn resolve(self) -> BackpressureMode { + self.0 + } +} + +impl PyStubType for DefaultBackpressureMode { + fn type_output() -> TypeInfo { + BackpressureMode::type_output() + } + + fn type_input() -> TypeInfo { + BackpressureMode::type_input() + } +} + +impl<'py> IntoPyObject<'py> for DefaultBackpressureMode { + type Target = PyAny; + type Output = Bound<'py, PyAny>; + type Error = PyErr; + + fn into_pyobject(self, py: Python<'py>) -> Result { + Ok(opaque_stub_default(py)) + } +} + +#[derive(FromPyObject)] +#[pyo3(transparent)] +struct DefaultProducerSharding(ProducerSharding); + +impl Default for DefaultProducerSharding { + fn default() -> Self { + Self(ProducerSharding::Ordered) + } +} + +impl DefaultProducerSharding { + fn resolve(self) -> ProducerSharding { + self.0 + } +} + +impl PyStubType for DefaultProducerSharding { + fn type_output() -> TypeInfo { + ProducerSharding::type_output() + } + + fn type_input() -> TypeInfo { + ProducerSharding::type_input() + } +} + +impl<'py> IntoPyObject<'py> for DefaultProducerSharding { + type Target = PyAny; + type Output = Bound<'py, PyAny>; + type Error = PyErr; + + fn into_pyobject(self, py: Python<'py>) -> Result { + Ok(opaque_stub_default(py)) + } +} + +#[derive(Default)] +pub(crate) enum RetryInterval { + #[default] + Omitted, + Disabled, + Duration(Py), +} + +impl<'a, 'py> FromPyObject<'a, 'py> for RetryInterval { + type Error = PyErr; + + fn extract(object: Borrowed<'a, 'py, PyAny>) -> PyResult { + if object.is_none() { + Ok(Self::Disabled) + } else { + Ok(Self::Duration(object.extract::>()?)) + } + } +} + +impl PyStubType for RetryInterval { + fn type_output() -> TypeInfo { + ::type_output() | TypeInfo::none() + } + + fn type_input() -> TypeInfo { + ::type_input() | TypeInfo::none() + } +} + +impl<'py> IntoPyObject<'py> for RetryInterval { + type Target = PyAny; + type Output = Bound<'py, PyAny>; + type Error = PyErr; + + fn into_pyobject(self, py: Python<'py>) -> Result { + match self { + Self::Omitted => std::time::Duration::from_secs(1).into_bound_py_any(py), + Self::Disabled => Ok(py.None().into_bound(py)), + Self::Duration(duration) => Ok(duration.into_bound(py).into_any()), + } + } +} + +impl RetryInterval { + pub(crate) fn resolve(self) -> PyResult> { + match self { + Self::Omitted => Ok(Some(iggy::prelude::NonZeroIggyDuration::ONE_SECOND)), + Self::Disabled => Ok(None), + Self::Duration(duration) => py_delta_to_iggy_duration(&duration).and_then(|duration| { + iggy::prelude::NonZeroIggyDuration::try_from(duration) + .map(Some) + .map_err(|_| PyValueError::new_err("'send_retry_interval' must not be zero")) + }), + } + } +} + +#[derive(Default)] +enum DefaultDuration { + #[default] + Zero, + OneMillisecond, + Value(Py), +} + +impl DefaultDuration { + fn one_millisecond() -> Self { + Self::OneMillisecond + } +} + +impl PyStubType for DefaultDuration { + fn type_output() -> TypeInfo { + timedelta_type_info() + } + + fn type_input() -> TypeInfo { + timedelta_type_info() + } +} + +impl<'py> IntoPyObject<'py> for DefaultDuration { + type Target = PyDelta; + type Output = Bound<'py, PyDelta>; + type Error = PyErr; + + fn into_pyobject(self, py: Python<'py>) -> Result { + match self { + Self::Zero => std::time::Duration::ZERO.into_pyobject(py), + Self::OneMillisecond => std::time::Duration::from_millis(1).into_pyobject(py), + Self::Value(duration) => Ok(duration.into_bound(py)), + } + } +} + +impl<'a, 'py> FromPyObject<'a, 'py> for DefaultDuration { + type Error = PyErr; + + fn extract(object: Borrowed<'a, 'py, PyAny>) -> PyResult { + Ok(Self::Value(object.extract::>()?)) + } +} + +impl DefaultDuration { + fn resolve(self, default: IggyDuration) -> PyResult { + match self { + Self::Value(duration) => py_delta_to_iggy_duration(&duration), + Self::Zero | Self::OneMillisecond => Ok(default), + } + } +} + +fn timedelta_type_info() -> TypeInfo { + let mut type_info = ::type_input(); + type_info.source_module = None; + type_info +} + +fn opaque_stub_default(py: Python<'_>) -> Bound<'_, PyAny> { + // The stub generator renders objects without a stable Python expression as + // `...`. An actual Ellipsis is rendered as `Ellipsis`, which type checkers + // reject as a default for these public configuration types. + py.None().into_bound(py).get_type().into_any() +} + +async fn shutdown(inner: Arc>>) -> PyResult<()> { + // Exclusive access must span consuming shutdown so another close waits for + // completion and no send can observe a producer being closed underneath it. + let mut producer = inner.write().await; + if let Some(producer) = producer.take() { + producer.shutdown().await; + } + Ok(()) +} + +fn extract_messages(messages: &Bound<'_, PyList>) -> PyResult> { + messages + .iter() + .map(|item| { + let message = item.extract::>()?; + Ok(clone_rust_message(&message)) + }) + .collect() +} + +fn extract_send_to_identifier(value: &Bound<'_, PyAny>, parameter: &str) -> PyResult { + if let Ok(value) = value.cast::() { + return Identifier::from_str(value.to_str()?) + .map_err(|error| PyValueError::new_err(error.to_string())); + } + if value.is_instance_of::() { + let value = value.extract::()?; + return Identifier::numeric(value) + .map_err(|error| PyValueError::new_err(error.to_string())); + } + Err(PyTypeError::new_err(format!( + "'{parameter}' must be a string or an integer" + ))) +} + +fn clone_rust_message(message: &SendMessage) -> RustIggyMessage { + message.clone().inner +} + +fn to_send_error(error: IggyError) -> PyErr { + match error { + IggyError::ProducerSendFailed { + cause, + failed, + committed, + .. + } => ProducerSendError::new_err(*cause, failed, committed), + error => PyRuntimeError::new_err(error.to_string()), + } +} + +pub(crate) fn u32_param(value: i64, parameter: &str) -> PyResult { + u32::try_from(value).map_err(|_| { + PyValueError::new_err(format!("'{parameter}' must be between 0 and {}", u32::MAX)) + }) +} + +fn usize_param(value: i128, parameter: &str) -> PyResult { + usize::try_from(value).map_err(|_| { + PyValueError::new_err(format!( + "'{parameter}' must be between 0 and {}", + usize::MAX + )) + }) +} + +fn u64_param(value: i128, parameter: &str) -> PyResult { + u64::try_from(value).map_err(|_| { + PyValueError::new_err(format!("'{parameter}' must be between 0 and {}", u64::MAX)) + }) +} + +#[cfg(test)] +mod tests { + use std::future::Future; + use std::sync::atomic::{AtomicUsize, Ordering}; + use std::time::Duration; + + use iggy::clients::producer::ProducerCoreBackend; + use iggy::clients::producer_dispatcher::ProducerDispatcher; + use pyo3::exceptions::{PyOverflowError, PyRuntimeError, PyTypeError, PyValueError}; + + use super::*; + + #[derive(Debug)] + struct ConcurrencyTrackingBackend { + active: Arc, + maximum: Arc, + } + + impl ProducerCoreBackend for ConcurrencyTrackingBackend { + fn send_internal( + &self, + _stream: &Identifier, + _topic: &Identifier, + _messages: Vec, + _partitioning: Option>, + ) -> impl Future> + Send + { + let active = self.active.clone(); + let maximum = self.maximum.clone(); + async move { + let active_now = active.fetch_add(1, Ordering::SeqCst) + 1; + maximum.fetch_max(active_now, Ordering::SeqCst); + tokio::time::sleep(Duration::from_millis(50)).await; + active.fetch_sub(1, Ordering::SeqCst); + Ok(iggy::prelude::SendMessagesResponse { + confirmations: Vec::new(), + }) + } + } + } + + #[test] + fn direct_defaults_match_rust() { + let python = DirectProducerConfig::default(); + let rust = RustDirectConfig::builder().build(); + + assert_eq!(python.inner.batch_length, rust.batch_length); + assert_eq!(python.inner.linger_time, rust.linger_time); + } + + #[test] + fn direct_linger_rejects_values_above_u64_microseconds() { + Python::initialize(); + Python::attach(|py| { + let duration = PyDelta::new(py, 999_999_999, 0, 0, false).unwrap().unbind(); + let error = DirectProducerConfig::new(1_000, DefaultDuration::Value(duration)) + .err() + .unwrap(); + + assert!(error.is_instance_of::(py)); + assert!(error.to_string().contains("linger_time")); + }); + } + + #[test] + fn producer_send_error_preserves_recovery_state() { + Python::initialize(); + Python::attach(|py| { + let cause = IggyError::CannotSendMessagesDueToClientDisconnection; + let cause_message = cause.to_string(); + let error = to_send_error(IggyError::ProducerSendFailed { + cause: Box::new(cause), + failed: Arc::new(vec![RustIggyMessage::from_str("failed").unwrap()]), + committed: Arc::new(vec![RustSendMessagesConfirmationResponse { + stream_id: 1, + topic_id: 2, + partition_id: 3, + base_offset: 4, + }]), + stream_name: "stream".to_owned(), + topic_name: "topic".to_owned(), + }); + + assert!(error.is_instance_of::(py)); + assert!(error.is_instance_of::(py)); + assert_eq!( + error + .value(py) + .getattr("cause") + .unwrap() + .extract::() + .unwrap(), + cause_message + ); + + let failed = error + .value(py) + .getattr("failed") + .unwrap() + .extract::>>() + .unwrap(); + assert_eq!(failed.len(), 1); + assert_eq!(failed[0].borrow(py).inner.payload.as_ref(), b"failed"); + + let committed = error + .value(py) + .getattr("committed") + .unwrap() + .extract::>>() + .unwrap(); + assert_eq!(committed.len(), 1); + assert_eq!(committed[0].borrow(py).inner.base_offset, 4); + + let source = error.cause(py).unwrap(); + assert!(source.is_instance_of::(py)); + assert_eq!( + source.value(py).str().unwrap().to_str().unwrap(), + cause_message + ); + }); + } + + #[test] + fn background_defaults_match_rust() { + let python = BackgroundProducerConfig::default(); + let rust = RustBackgroundConfig::builder().build(); + + assert_eq!(python.num_shards, rust.num_shards); + assert_eq!(python.linger_time, rust.linger_time); + assert_eq!(python.batch_size, rust.batch_size); + assert_eq!(python.batch_length, rust.batch_length); + assert_eq!(python.max_buffer_size, rust.max_buffer_size); + assert!(matches!(rust.failure_mode, RustBackpressureMode::Block)); + assert_eq!(python.max_in_flight, rust.max_in_flight); + assert!(matches!(python.sharding, ProducerSharding::Ordered)); + assert_eq!(format!("{:?}", rust.sharding), "OrderedSharding"); + } + + #[test] + fn background_configuration_converts_every_field_to_rust() { + let python = BackgroundProducerConfig { + num_shards: 4, + linger_time: IggyDuration::from(2_000), + batch_size: 2_048, + batch_length: 32, + max_buffer_size: IggyByteSize::from(8_192), + failure_mode: BackpressureMode::fail_immediately(), + max_in_flight: 3, + sharding: ProducerSharding::Balanced, + }; + let rust = RustBackgroundConfig::try_from(&python).unwrap(); + + assert_eq!(rust.num_shards, 4); + assert_eq!(rust.linger_time, IggyDuration::from(2_000)); + assert_eq!(rust.batch_size, 2_048); + assert_eq!(rust.batch_length, 32); + assert_eq!(rust.max_buffer_size.as_bytes_u64(), 8_192); + assert!(matches!( + rust.failure_mode, + RustBackpressureMode::FailImmediately + )); + assert_eq!(rust.max_in_flight, 3); + assert_eq!( + format!("{:?}", rust.sharding), + "BalancedSharding { counter: 0 }" + ); + } + + #[test] + fn backpressure_modes_convert_to_rust() { + let block = RustBackpressureMode::from(&BackpressureMode::block()); + let timeout = RustBackpressureMode::from(&BackpressureMode { + kind: BackpressureKind::BlockWithTimeout(IggyDuration::from(250_000)), + }); + let immediate = RustBackpressureMode::from(&BackpressureMode::fail_immediately()); + + assert!(matches!(block, RustBackpressureMode::Block)); + assert!(matches!( + timeout, + RustBackpressureMode::BlockWithTimeout(value) + if value == IggyDuration::from(250_000) + )); + assert!(matches!(immediate, RustBackpressureMode::FailImmediately)); + } + + #[test] + fn sharding_modes_convert_to_the_selected_rust_strategy() { + let messages = vec![RustIggyMessage::from_str("message").unwrap()]; + let stream = Identifier::from_str("stream").unwrap(); + let topic = Identifier::from_str("topic").unwrap(); + + let ordered = RustBackgroundConfig::try_from(&BackgroundProducerConfig { + num_shards: 4, + ..BackgroundProducerConfig::default() + }) + .unwrap(); + let first = ordered.sharding.pick_shard(4, &messages, &stream, &topic); + let second = ordered.sharding.pick_shard(4, &messages, &stream, &topic); + assert_eq!(first, second); + + let balanced = RustBackgroundConfig::try_from(&BackgroundProducerConfig { + num_shards: 3, + sharding: ProducerSharding::Balanced, + ..BackgroundProducerConfig::default() + }) + .unwrap(); + let picked = (0..4) + .map(|_| balanced.sharding.pick_shard(3, &messages, &stream, &topic)) + .collect::>(); + assert_eq!(picked, vec![0, 1, 2, 0]); + } + + #[test] + fn background_conversion_rejects_values_that_would_panic_semaphore() { + let invalid_max_buffer = BackgroundProducerConfig { + max_buffer_size: IggyByteSize::from(Semaphore::MAX_PERMITS as u64 + 1), + ..BackgroundProducerConfig::default() + }; + let invalid_max_in_flight = BackgroundProducerConfig { + max_in_flight: Semaphore::MAX_PERMITS + 1, + ..BackgroundProducerConfig::default() + }; + + assert!(RustBackgroundConfig::try_from(&invalid_max_buffer).is_err()); + assert!(RustBackgroundConfig::try_from(&invalid_max_in_flight).is_err()); + } + + #[tokio::test] + async fn converted_max_in_flight_limits_concurrent_background_writes() { + let active = Arc::new(AtomicUsize::new(0)); + let maximum = Arc::new(AtomicUsize::new(0)); + let backend = Arc::new(ConcurrencyTrackingBackend { + active, + maximum: maximum.clone(), + }); + let config = RustBackgroundConfig::try_from(&BackgroundProducerConfig { + num_shards: 4, + linger_time: IggyDuration::from(0), + batch_size: 0, + batch_length: 1, + max_buffer_size: IggyByteSize::from(0), + failure_mode: BackpressureMode::block(), + max_in_flight: 2, + sharding: ProducerSharding::Balanced, + }) + .unwrap(); + let dispatcher = ProducerDispatcher::new(backend, config); + let stream = Arc::new(Identifier::numeric(1).unwrap()); + let topic = Arc::new(Identifier::numeric(1).unwrap()); + + for index in 0..4 { + dispatcher + .dispatch( + vec![RustIggyMessage::from_str(&index.to_string()).unwrap()], + stream.clone(), + topic.clone(), + None, + ) + .await + .unwrap(); + } + dispatcher.shutdown().await; + + assert_eq!(maximum.load(Ordering::SeqCst), 2); + } + + #[test] + fn direct_batch_length_rejects_values_outside_u32() { + assert!(u32_param(-1, "batch_length").is_err()); + assert!(u32_param(i64::from(u32::MAX) + 1, "batch_length").is_err()); + assert_eq!( + u32_param(i64::from(u32::MAX), "batch_length").unwrap(), + u32::MAX + ); + } + + #[test] + fn send_to_identifier_preserves_python_error_categories() { + Python::initialize(); + Python::attach(|py| { + let string = PyString::new(py, "stream"); + let numeric = 7_u32.into_pyobject(py).unwrap(); + let negative = (-1_i64).into_pyobject(py).unwrap(); + let too_large = (u64::from(u32::MAX) + 1).into_pyobject(py).unwrap(); + let wrong_type = py.None().into_bound(py); + + assert_eq!( + extract_send_to_identifier(string.as_any(), "stream") + .unwrap() + .get_string_value() + .unwrap(), + "stream" + ); + assert_eq!( + extract_send_to_identifier(numeric.as_any(), "stream") + .unwrap() + .get_u32_value() + .unwrap(), + 7 + ); + + let negative = extract_send_to_identifier(negative.as_any(), "stream").unwrap_err(); + let too_large = extract_send_to_identifier(too_large.as_any(), "stream").unwrap_err(); + let wrong_type = extract_send_to_identifier(&wrong_type, "stream").unwrap_err(); + + assert!(negative.is_instance_of::(py)); + assert!(too_large.is_instance_of::(py)); + assert!(wrong_type.is_instance_of::(py)); + }); + } + + #[test] + fn representations_name_public_python_constructors() { + let direct = DirectProducerConfig::default(); + let background = BackgroundProducerConfig::default(); + + assert_eq!( + direct.__repr__(), + "DirectProducerConfig(batch_length=1000, linger_time=datetime.timedelta(seconds=0))" + ); + assert_eq!( + BackpressureMode::block().__repr__(), + "BackpressureMode.block()" + ); + assert_eq!( + BackpressureMode::fail_immediately().__repr__(), + "BackpressureMode.fail_immediately()" + ); + assert!( + background + .__repr__() + .ends_with("sharding=ProducerSharding.ORDERED)") + ); + } + + #[test] + fn numeric_getters_return_configured_values() { + let background = BackgroundProducerConfig::default(); + + assert_eq!(background.num_shards(), 1); + assert_eq!(background.batch_size(), 1_048_576); + assert_eq!(background.batch_length(), 1_000); + assert_eq!(background.max_buffer_size(), 33_554_432); + assert_eq!(background.max_in_flight(), 1); + assert_eq!(background.sharding(), ProducerSharding::Ordered); + assert_eq!(background.failure_mode(), BackpressureMode::block()); + } + + #[test] + fn numeric_validation_uses_python_semantic_ranges() { + assert!(usize_param(-1, "num_shards").is_err()); + assert!(u64_param(-1, "max_buffer_size").is_err()); + assert_eq!( + usize_param(usize::MAX as i128, "num_shards").unwrap(), + usize::MAX + ); + assert_eq!( + u64_param(u64::MAX as i128, "max_buffer_size").unwrap(), + u64::MAX + ); + } + + #[test] + fn direct_default_constants_match_builder() { + let rust = RustDirectConfig::builder().build(); + assert_eq!(1_000, rust.batch_length); + assert_eq!(IggyDuration::from(0), rust.linger_time); + } + + #[test] + fn background_default_constants_match_builder() { + let rust = RustBackgroundConfig::builder().build(); + assert_eq!(DEFAULT_BACKGROUND_NUM_SHARDS, rust.num_shards); + assert_eq!(IggyDuration::from(1_000), rust.linger_time); + assert_eq!(DEFAULT_BACKGROUND_BATCH_SIZE, rust.batch_size); + assert_eq!(DEFAULT_BACKGROUND_BATCH_LENGTH, rust.batch_length); + assert_eq!( + DEFAULT_BACKGROUND_MAX_BUFFER_SIZE, + rust.max_buffer_size.as_bytes_u64() + ); + assert_eq!(DEFAULT_BACKGROUND_MAX_IN_FLIGHT, rust.max_in_flight); + } +} diff --git a/foreign/python/src/send_message.rs b/foreign/python/src/send_message.rs index 8cb940fc40..454fbb2acf 100644 --- a/foreign/python/src/send_message.rs +++ b/foreign/python/src/send_message.rs @@ -38,20 +38,26 @@ pub struct SendMessage { impl Clone for SendMessage { fn clone(&self) -> Self { + Self::clone_from_rust(&self.inner) + } +} + +impl SendMessage { + pub(crate) fn clone_from_rust(message: &RustIggyMessage) -> Self { Self { inner: RustIggyMessage { header: IggyMessageHeader { - checksum: self.inner.header.checksum, - id: self.inner.header.id, - offset: self.inner.header.offset, - timestamp: self.inner.header.timestamp, - origin_timestamp: self.inner.header.origin_timestamp, - user_headers_length: self.inner.header.user_headers_length, - payload_length: self.inner.header.payload_length, - reserved: self.inner.header.reserved, + checksum: message.header.checksum, + id: message.header.id, + offset: message.header.offset, + timestamp: message.header.timestamp, + origin_timestamp: message.header.origin_timestamp, + user_headers_length: message.header.user_headers_length, + payload_length: message.header.payload_length, + reserved: message.header.reserved, }, - payload: self.inner.payload.clone(), - user_headers: self.inner.user_headers.clone(), + payload: message.payload.clone(), + user_headers: message.user_headers.clone(), }, } } diff --git a/foreign/python/tests/test_producer.py b/foreign/python/tests/test_producer.py new file mode 100644 index 0000000000..418b7631ea --- /dev/null +++ b/foreign/python/tests/test_producer.py @@ -0,0 +1,1608 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +"""Tests for the high-level producer API and both send modes.""" + +import ast +import asyncio +import contextlib +import os +import subprocess +import sys +import time +from datetime import timedelta +from pathlib import Path + +import pytest + +from apache_iggy import ( + BackgroundProducerConfig, + BackpressureMode, + Consumer, + DirectProducerConfig, + IggyClient, + IggyExpiry, + IggyProducer, + MaxTopicSize, + Partitioning, + PollingStrategy, + ProducerSendError, + ProducerSharding, + SendMessage, + SendMessagesResponse, +) + +from .utils import wait_for_server + +REPOSITORY_ROOT = Path(__file__).resolve().parents[3] +SERVER_BINARY = ( + REPOSITORY_ROOT + / "target" + / "debug" + / ("iggy-server.exe" if sys.platform == "win32" else "iggy-server") +) + + +def spawn_restartable_server( + data_path: Path, + tcp_address: str, +) -> subprocess.Popen[bytes]: + data_path.mkdir(parents=True, exist_ok=True) + environment = os.environ.copy() + # Python test connection settings use the IGGY_SERVER_ prefix, but the + # server rejects them as unknown configuration overrides. + for name in tuple(environment): + if name.startswith("IGGY_SERVER_"): + del environment[name] + environment.update( + { + "IGGY_PATH": str(data_path), + "IGGY_TCP_ADDRESS": tcp_address, + "IGGY_HTTP_ENABLED": "false", + "IGGY_QUIC_ENABLED": "false", + "IGGY_WEBSOCKET_ENABLED": "false", + "IGGY_LOGGING_LEVEL": "error", + } + ) + with (data_path / "server.log").open("ab") as server_log: + return subprocess.Popen( # noqa: S603 + [str(SERVER_BINARY), "--with-default-root-credentials"], + cwd=REPOSITORY_ROOT, + env=environment, + stdout=server_log, + stderr=subprocess.STDOUT, + ) + + +async def stop_restartable_server(process: subprocess.Popen[bytes]) -> None: + if process.poll() is not None: + return + + process.terminate() + try: + await asyncio.to_thread(process.wait, 10) + except subprocess.TimeoutExpired: + process.kill() + await asyncio.to_thread(process.wait) + + +async def discover_server_address( + data_path: Path, + process: subprocess.Popen[bytes], + timeout: float = 15, +) -> tuple[str, int]: + config_path = data_path / "runtime" / "current_config.toml" + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + if process.poll() is not None: + server_log = (data_path / "server.log").read_text(errors="replace") + raise RuntimeError( + f"Restartable Iggy server exited with code {process.returncode}:\n" + f"{server_log[-4_000:]}" + ) + if config_path.exists(): + in_tcp_section = False + for line in config_path.read_text().splitlines(): + stripped = line.strip() + if stripped.startswith("["): + in_tcp_section = stripped == "[tcp]" + elif in_tcp_section and stripped.startswith("address = "): + address = stripped.split('"', 2)[1] + host, port = address.rsplit(":", 1) + return host, int(port) + await asyncio.sleep(0.05) + + raise TimeoutError("Restartable Iggy server did not publish its TCP address") + + +async def poll_payloads( + client: IggyClient, + stream: str, + topic: str, + *, + partition_id: int = 0, +) -> list[str]: + messages = await client.poll_messages( + stream=stream, + topic=topic, + consumer=Consumer.Single(1), + partition_id=partition_id, + polling_strategy=PollingStrategy.First(), + count=100, + auto_commit=False, + ) + return [message.payload().decode("utf-8") for message in messages] + + +async def wait_for_payloads( + client: IggyClient, + stream: str, + topic: str, + expected: list[str], + *, + partition_id: int = 0, + timeout: float = 2, +) -> list[str]: + deadline = time.monotonic() + timeout + payloads: list[str] = [] + while time.monotonic() < deadline: + payloads = await poll_payloads( + client, + stream, + topic, + partition_id=partition_id, + ) + if payloads == expected: + return payloads + await asyncio.sleep(0.01) + raise AssertionError( + f"Timed out waiting for payloads {expected!r}; last payloads were {payloads!r}" + ) + + +@pytest.mark.unit +class TestDirectProducerConfig: + """Test direct-mode configuration independently of a server.""" + + def test_defaults_match_the_rust_sdk(self): + config = DirectProducerConfig() + + assert config.batch_length == 1_000 + assert config.linger_time == timedelta(0) + + def test_fields_are_keyword_only_and_round_trip(self): + config = DirectProducerConfig( + batch_length=25, + linger_time=timedelta(milliseconds=125), + ) + + assert config.batch_length == 25 + assert config.linger_time == timedelta(milliseconds=125) + + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-count + DirectProducerConfig(25, timedelta(milliseconds=125)) + + def test_zero_batch_length_retains_the_rust_sentinel(self): + assert DirectProducerConfig(batch_length=0).batch_length == 0 + + def test_largest_batch_length_round_trips(self): + assert DirectProducerConfig(batch_length=2**32 - 1).batch_length == 2**32 - 1 + + @pytest.mark.parametrize("batch_length", [-1, 2**32]) + def test_batch_length_outside_u32_is_rejected(self, batch_length: int): + with pytest.raises(ValueError, match="batch_length"): + DirectProducerConfig(batch_length=batch_length) + + def test_batch_length_binding_overflow_is_distinct(self): + with pytest.raises(OverflowError): + DirectProducerConfig(batch_length=2**63) + + @pytest.mark.parametrize( + "linger_time", + [timedelta(microseconds=-1), timedelta(seconds=-1)], + ) + def test_negative_linger_is_rejected(self, linger_time: timedelta): + with pytest.raises(ValueError, match="negative"): + DirectProducerConfig(linger_time=linger_time) + + def test_linger_above_u64_microseconds_is_rejected(self): + with pytest.raises(ValueError, match="linger_time"): + DirectProducerConfig(linger_time=timedelta(days=999_999_999)) + + def test_wrong_field_types_are_rejected(self): + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-type + DirectProducerConfig(batch_length="1000") + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-type + DirectProducerConfig(linger_time=1) + + def test_repr_contains_pasteable_python_values(self): + config = DirectProducerConfig( + batch_length=25, + linger_time=timedelta(milliseconds=125), + ) + + printed = repr(config) + + assert "batch_length=25" in printed + assert "linger_time=datetime.timedelta(microseconds=125000)" in printed + ast.parse(printed) + + +@pytest.mark.unit +class TestBackgroundProducerSurface: + """Test the background configuration model independently of a server.""" + + def test_sharding_values_are_public_and_distinct(self): + assert ProducerSharding.ORDERED != ProducerSharding.BALANCED + assert repr(ProducerSharding.ORDERED) == "ProducerSharding.ORDERED" + assert repr(ProducerSharding.BALANCED) == "ProducerSharding.BALANCED" + + def test_backpressure_constructors_and_getters(self): + blocking = BackpressureMode.block() + timed = BackpressureMode.block_with_timeout(timedelta(milliseconds=250)) + immediate = BackpressureMode.fail_immediately() + + assert blocking.timeout is None + assert timed.timeout == timedelta(milliseconds=250) + assert immediate.timeout is None + assert repr(blocking) == "BackpressureMode.block()" + assert repr(timed) == ( + "BackpressureMode.block_with_timeout(" + "datetime.timedelta(microseconds=250000))" + ) + assert repr(immediate) == "BackpressureMode.fail_immediately()" + + def test_backpressure_timeout_validation(self): + zero = BackpressureMode.block_with_timeout(timedelta(0)) + assert zero.timeout == timedelta(0) + + with pytest.raises(ValueError, match="negative"): + BackpressureMode.block_with_timeout(timedelta(microseconds=-1)) + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-type + BackpressureMode.block_with_timeout(1) + + def test_defaults_match_the_rust_sdk(self): + config = BackgroundProducerConfig() + + assert config.num_shards == 1 + assert config.linger_time == timedelta(milliseconds=1) + assert config.batch_size == 1024 * 1024 + assert config.batch_length == 1_000 + assert config.max_buffer_size == 32 * 1024 * 1024 + assert repr(config.failure_mode) == "BackpressureMode.block()" + assert config.max_in_flight == 1 + assert config.sharding == ProducerSharding.ORDERED + + def test_every_field_round_trips(self): + failure_mode = BackpressureMode.block_with_timeout(timedelta(milliseconds=500)) + config = BackgroundProducerConfig( + num_shards=4, + linger_time=timedelta(milliseconds=10), + batch_size=256 * 1024, + batch_length=50, + max_buffer_size=8 * 1024 * 1024, + failure_mode=failure_mode, + max_in_flight=3, + sharding=ProducerSharding.BALANCED, + ) + + assert config.num_shards == 4 + assert config.linger_time == timedelta(milliseconds=10) + assert config.batch_size == 256 * 1024 + assert config.batch_length == 50 + assert config.max_buffer_size == 8 * 1024 * 1024 + assert repr(config.failure_mode) == repr(failure_mode) + assert config.max_in_flight == 3 + assert config.sharding == ProducerSharding.BALANCED + + def test_fields_are_keyword_only_and_read_only(self): + config = BackgroundProducerConfig() + + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-count + BackgroundProducerConfig(2) + with pytest.raises(AttributeError): + # pyrefly: ignore # read-only + config.num_shards = 2 + + @pytest.mark.parametrize( + "field", + [ + "num_shards", + "batch_size", + "batch_length", + "max_buffer_size", + "max_in_flight", + ], + ) + def test_unsigned_fields_reject_negative_values(self, field: str): + with pytest.raises(ValueError, match=field): + # pyrefly: ignore # bad-argument-type + BackgroundProducerConfig(**{field: -1}) + + @pytest.mark.parametrize( + "field", + [ + "num_shards", + "batch_size", + "batch_length", + "max_buffer_size", + "max_in_flight", + ], + ) + def test_unsigned_fields_reject_wrong_types(self, field: str): + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-type + BackgroundProducerConfig(**{field: "1"}) + + def test_max_buffer_size_checks_semantic_range_before_binding_overflow(self): + maximum = 2**64 - 1 + + assert ( + BackgroundProducerConfig(max_buffer_size=maximum).max_buffer_size == maximum + ) + with pytest.raises(ValueError, match="max_buffer_size"): + BackgroundProducerConfig(max_buffer_size=2**64) + with pytest.raises(OverflowError): + BackgroundProducerConfig(max_buffer_size=2**127) + + @pytest.mark.parametrize("value", [0, 1]) + def test_zero_and_one_are_preserved_for_threshold_fields(self, value: int): + config = BackgroundProducerConfig( + num_shards=value, + batch_size=value, + batch_length=value, + max_buffer_size=value, + max_in_flight=value, + ) + + assert config.num_shards == value + assert config.batch_size == value + assert config.batch_length == value + assert config.max_buffer_size == value + assert config.max_in_flight == value + + def test_wrong_variant_and_duration_types_are_rejected(self): + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-type + BackgroundProducerConfig(linger_time=1) + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-type + BackgroundProducerConfig(failure_mode="block") + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-type + BackgroundProducerConfig(sharding="ordered") + + def test_negative_linger_is_rejected(self): + with pytest.raises(ValueError, match="negative"): + BackgroundProducerConfig(linger_time=timedelta(microseconds=-1)) + + def test_repr_contains_every_public_field(self): + config = BackgroundProducerConfig( + num_shards=2, + linger_time=timedelta(milliseconds=5), + batch_size=4_096, + batch_length=8, + max_buffer_size=65_536, + failure_mode=BackpressureMode.fail_immediately(), + max_in_flight=2, + sharding=ProducerSharding.BALANCED, + ) + + printed = repr(config) + + for expected in [ + "num_shards=2", + "linger_time=datetime.timedelta(microseconds=5000)", + "batch_size=4096", + "batch_length=8", + "max_buffer_size=65536", + "failure_mode=BackpressureMode.fail_immediately()", + "max_in_flight=2", + "sharding=ProducerSharding.BALANCED", + ]: + assert expected in printed + ast.parse(printed) + + +class TestProducerCreation: + """Test initialization and producer-owned resource creation.""" + + @pytest.mark.asyncio + async def test_default_producer_creates_resources_and_is_immediately_usable( + self, iggy_client: IggyClient, unique_name + ): + stream_name = unique_name() + topic_name = unique_name() + + producer = await iggy_client.producer(stream_name, topic_name) + try: + response = await producer.send_one(SendMessage("ready")) + + assert isinstance(producer, IggyProducer) + assert isinstance(response, SendMessagesResponse) + assert len(response.confirmations) == 1 + assert await iggy_client.get_stream(stream_name) is not None + assert await iggy_client.get_topic(stream_name, topic_name) is not None + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_none_mode_uses_the_default_direct_mode( + self, iggy_client: IggyClient, unique_name + ): + producer = await iggy_client.producer( + unique_name(), + unique_name(), + mode=None, + ) + try: + response = await producer.send_one(SendMessage("default mode")) + assert len(response.confirmations) == 1 + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_existing_resources_work_when_creation_is_disabled( + self, iggy_client: IggyClient, unique_name + ): + stream_name = unique_name() + topic_name = unique_name() + await iggy_client.create_stream(stream_name) + await iggy_client.create_topic(stream_name, topic_name, 1) + + producer = await iggy_client.producer( + stream_name, + topic_name, + create_stream_if_not_exists=False, + create_topic_if_not_exists=False, + ) + try: + response = await producer.send_one(SendMessage("existing")) + assert len(response.confirmations) == 1 + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_missing_stream_fails_without_returning_a_producer( + self, iggy_client: IggyClient, unique_name + ): + stream_name = unique_name() + topic_name = unique_name() + not_returned = object() + result = not_returned + + with pytest.raises(RuntimeError): + result = await iggy_client.producer( + stream_name, + topic_name, + create_stream_if_not_exists=False, + ) + + assert result is not_returned + assert await iggy_client.get_stream(stream_name) is None + + @pytest.mark.asyncio + async def test_missing_topic_fails_after_creating_the_bound_stream( + self, iggy_client: IggyClient, unique_name + ): + stream_name = unique_name() + topic_name = unique_name() + not_returned = object() + result = not_returned + + with pytest.raises(RuntimeError): + result = await iggy_client.producer( + stream_name, + topic_name, + create_topic_if_not_exists=False, + ) + + assert result is not_returned + assert await iggy_client.get_stream(stream_name) is not None + assert await iggy_client.get_topic(stream_name, topic_name) is None + + @pytest.mark.asyncio + async def test_created_topic_uses_requested_topology_and_retention( + self, iggy_client: IggyClient, unique_name + ): + stream_name = unique_name() + topic_name = unique_name() + expiry = timedelta(minutes=15) + maximum_size = 2_000_000_000 + + producer = await iggy_client.producer( + stream_name, + topic_name, + topic_partitions_count=3, + topic_message_expiry=IggyExpiry.ExpireDuration(expiry), + topic_max_size=MaxTopicSize.Custom(maximum_size), + ) + try: + topic = await iggy_client.get_topic(stream_name, topic_name) + + assert topic is not None + assert topic.partitions_count == 3 + assert isinstance(topic.message_expiry, IggyExpiry.ExpireDuration) + assert topic.message_expiry.duration == expiry + assert isinstance(topic.max_topic_size, MaxTopicSize.Custom) + assert topic.max_topic_size.bytes == maximum_size + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_background_producer_is_initialized_and_immediately_usable( + self, iggy_client: IggyClient, unique_name + ): + stream_name = unique_name("background-", min_bytes=20, max_bytes=20) + topic_name = unique_name("topic-", min_bytes=20, max_bytes=20) + producer = await iggy_client.producer( + stream_name, + topic_name, + mode=BackgroundProducerConfig(), + ) + try: + response = await producer.send_one(SendMessage("ready")) + + assert isinstance(producer, IggyProducer) + assert response.confirmations == [] + finally: + await producer.shutdown() + + assert await wait_for_payloads( + iggy_client, stream_name, topic_name, ["ready"] + ) == ["ready"] + + +class TestProducerSends: + """Test all direct send operations and partitioning fallbacks.""" + + @pytest.mark.asyncio + async def test_send_and_send_one_return_confirmations( + self, iggy_client: IggyClient, unique_name + ): + producer = await iggy_client.producer(unique_name(), unique_name()) + try: + batch = await producer.send([SendMessage("first"), SendMessage("second")]) + single = await producer.send_one(SendMessage("third")) + + assert isinstance(batch, SendMessagesResponse) + assert isinstance(single, SendMessagesResponse) + assert len(batch.confirmations) == 1 + assert len(single.confirmations) == 1 + assert single.confirmations[0].base_offset == 2 + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_default_producer_partitioning_is_balanced( + self, iggy_client: IggyClient, unique_name + ): + producer = await iggy_client.producer( + unique_name(), + unique_name(), + topic_partitions_count=3, + ) + try: + responses = [ + await producer.send_one(SendMessage(str(index))) for index in range(3) + ] + + assert { + response.confirmations[0].partition_id for response in responses + } == {0, 1, 2} + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_per_call_partitioning_overrides_and_falls_back( + self, iggy_client: IggyClient, unique_name + ): + producer = await iggy_client.producer( + unique_name(), + unique_name(), + partitioning=Partitioning.partition_id(2), + topic_partitions_count=3, + ) + try: + configured = await producer.send([SendMessage("configured")]) + overridden = await producer.send_with_partitioning( + [SendMessage("override")], Partitioning.partition_id(0) + ) + omitted = await producer.send_with_partitioning([SendMessage("omitted")]) + explicit_none = await producer.send_with_partitioning( + [SendMessage("none")], None + ) + + assert configured.confirmations[0].partition_id == 2 + assert overridden.confirmations[0].partition_id == 0 + assert omitted.confirmations[0].partition_id == 2 + assert explicit_none.confirmations[0].partition_id == 2 + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_send_to_accepts_string_and_numeric_identifiers( + self, iggy_client: IggyClient, unique_name + ): + destination_stream = unique_name() + destination_topic = unique_name() + await iggy_client.create_stream(destination_stream) + await iggy_client.create_topic(destination_stream, destination_topic, 2) + stream = await iggy_client.get_stream(destination_stream) + topic = await iggy_client.get_topic(destination_stream, destination_topic) + assert stream is not None + assert topic is not None + + producer = await iggy_client.producer( + unique_name(), + unique_name(), + partitioning=Partitioning.partition_id(1), + ) + try: + by_name = await producer.send_to( + destination_stream, + destination_topic, + [SendMessage("names")], + ) + by_id = await producer.send_to( + stream.id, + topic.id, + [SendMessage("ids")], + None, + ) + + assert by_name.confirmations[0].stream_id == stream.id + assert by_name.confirmations[0].topic_id == topic.id + assert by_name.confirmations[0].partition_id == 1 + assert by_id.confirmations[0].stream_id == stream.id + assert by_id.confirmations[0].topic_id == topic.id + assert by_id.confirmations[0].partition_id == 1 + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_empty_batches_are_successful_no_ops_before_shutdown( + self, iggy_client: IggyClient, unique_name + ): + stream_name = unique_name() + topic_name = unique_name() + producer = await iggy_client.producer(stream_name, topic_name) + try: + responses = [ + await producer.send([]), + await producer.send_with_partitioning([]), + await producer.send_to(stream_name, topic_name, []), + ] + + assert all( + isinstance(response, SendMessagesResponse) for response in responses + ) + assert all(response.confirmations == [] for response in responses) + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_batch_length_splits_direct_requests( + self, iggy_client: IggyClient, unique_name + ): + producer = await iggy_client.producer( + unique_name(), + unique_name(), + mode=DirectProducerConfig(batch_length=2), + ) + try: + response = await producer.send( + [SendMessage(str(index)) for index in range(5)] + ) + + assert len(response.confirmations) == 3 + assert [ + confirmation.base_offset for confirmation in response.confirmations + ] == [0, 2, 4] + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_linger_paces_sequential_direct_sends( + self, iggy_client: IggyClient, unique_name + ): + linger = timedelta(milliseconds=150) + producer = await iggy_client.producer( + unique_name(), + unique_name(), + mode=DirectProducerConfig(linger_time=linger), + ) + try: + await producer.send_one(SendMessage("first")) + started_at = time.monotonic() + await producer.send_one(SendMessage("second")) + elapsed = time.monotonic() - started_at + + assert elapsed >= 0.1 + finally: + await producer.shutdown() + + +class TestBackgroundProducerSends: + """Test background acceptance, batching, routing, and graceful flush.""" + + @pytest.mark.asyncio + async def test_all_send_methods_accept_messages_without_confirmations( + self, iggy_client: IggyClient, unique_name + ): + stream_name = unique_name("bg-stream-", min_bytes=20, max_bytes=20) + topic_name = unique_name("bg-topic-", min_bytes=20, max_bytes=20) + destination_stream = unique_name("to-stream-", min_bytes=20, max_bytes=20) + destination_topic = unique_name("to-topic-", min_bytes=20, max_bytes=20) + await iggy_client.create_stream(destination_stream) + await iggy_client.create_topic(destination_stream, destination_topic, 1) + + producer = await iggy_client.producer( + stream_name, + topic_name, + partitioning=Partitioning.partition_id(1), + mode=BackgroundProducerConfig( + num_shards=2, + linger_time=timedelta(seconds=60), + batch_size=0, + batch_length=0, + max_in_flight=2, + sharding=ProducerSharding.BALANCED, + ), + topic_partitions_count=3, + ) + try: + responses = [ + await producer.send( + [SendMessage("batch-one"), SendMessage("batch-two")] + ), + await producer.send_one(SendMessage("single")), + await producer.send_with_partitioning( + [SendMessage("override")], Partitioning.partition_id(2) + ), + await producer.send_to( + destination_stream, + destination_topic, + [SendMessage("send-to")], + Partitioning.partition_id(0), + ), + ] + + assert all( + isinstance(response, SendMessagesResponse) for response in responses + ) + assert all(response.confirmations == [] for response in responses) + finally: + await producer.shutdown() + + payloads = await poll_payloads( + iggy_client, stream_name, topic_name, partition_id=1 + ) + assert sorted(payloads) == ["batch-one", "batch-two", "single"] + assert await poll_payloads( + iggy_client, stream_name, topic_name, partition_id=2 + ) == ["override"] + assert await poll_payloads( + iggy_client, destination_stream, destination_topic + ) == ["send-to"] + + @pytest.mark.asyncio + async def test_batch_length_flushes_after_the_configured_number_of_sends( + self, iggy_client: IggyClient, unique_name + ): + stream_name = unique_name("length-", min_bytes=12, max_bytes=12) + topic_name = unique_name("topic-", min_bytes=12, max_bytes=12) + producer = await iggy_client.producer( + stream_name, + topic_name, + mode=BackgroundProducerConfig( + linger_time=timedelta(seconds=5), + batch_size=0, + batch_length=2, + ), + ) + try: + await producer.send([SendMessage("first-a"), SendMessage("first-b")]) + await asyncio.sleep(0.05) + assert await poll_payloads(iggy_client, stream_name, topic_name) == [] + + await producer.send_one(SendMessage("second-send")) + assert await wait_for_payloads( + iggy_client, + stream_name, + topic_name, + ["first-a", "first-b", "second-send"], + ) == ["first-a", "first-b", "second-send"] + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_batch_size_flushes_after_buffered_bytes_reach_the_threshold( + self, iggy_client: IggyClient, unique_name + ): + stream_name = unique_name("size-", min_bytes=12, max_bytes=12) + topic_name = unique_name("topic-", min_bytes=12, max_bytes=12) + first = "a" * 50 + second = "b" * 50 + producer = await iggy_client.producer( + stream_name, + topic_name, + mode=BackgroundProducerConfig( + linger_time=timedelta(seconds=5), + batch_size=200, + batch_length=0, + ), + ) + try: + # With two 12-byte identifiers and the 64-byte message header, + # either send is below 200 reported bytes while both exceed it. + await producer.send_one(SendMessage(first)) + await asyncio.sleep(0.05) + assert await poll_payloads(iggy_client, stream_name, topic_name) == [] + + await producer.send_one(SendMessage(second)) + assert await wait_for_payloads( + iggy_client, stream_name, topic_name, [first, second] + ) == [first, second] + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_linger_time_flushes_a_non_empty_buffer( + self, iggy_client: IggyClient, unique_name + ): + stream_name = unique_name("linger-", min_bytes=12, max_bytes=12) + topic_name = unique_name("topic-", min_bytes=12, max_bytes=12) + producer = await iggy_client.producer( + stream_name, + topic_name, + mode=BackgroundProducerConfig( + linger_time=timedelta(milliseconds=200), + batch_size=0, + batch_length=0, + ), + ) + try: + await producer.send_one(SendMessage("linger")) + await asyncio.sleep(0.05) + assert await poll_payloads(iggy_client, stream_name, topic_name) == [] + assert await wait_for_payloads( + iggy_client, stream_name, topic_name, ["linger"] + ) == ["linger"] + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_zero_sentinels_use_one_shard_and_unlimited_capacity( + self, iggy_client: IggyClient, unique_name + ): + stream_name = unique_name("zero-", min_bytes=12, max_bytes=12) + topic_name = unique_name("topic-", min_bytes=12, max_bytes=12) + producer = await iggy_client.producer( + stream_name, + topic_name, + mode=BackgroundProducerConfig( + num_shards=0, + linger_time=timedelta(0), + batch_size=0, + batch_length=0, + max_buffer_size=0, + max_in_flight=0, + ), + ) + try: + response = await producer.send_one(SendMessage("zero sentinels")) + assert response.confirmations == [] + finally: + await producer.shutdown() + + assert await poll_payloads(iggy_client, stream_name, topic_name) == [ + "zero sentinels" + ] + + @pytest.mark.asyncio + async def test_ordered_sharding_preserves_order_for_one_destination( + self, iggy_client: IggyClient, unique_name + ): + stream_name = unique_name("ordered-", min_bytes=16, max_bytes=16) + topic_name = unique_name("topic-", min_bytes=16, max_bytes=16) + expected = [str(index) for index in range(30)] + producer = await iggy_client.producer( + stream_name, + topic_name, + mode=BackgroundProducerConfig( + num_shards=4, + linger_time=timedelta(0), + batch_size=0, + batch_length=1, + max_in_flight=4, + sharding=ProducerSharding.ORDERED, + ), + ) + try: + for payload in expected: + await producer.send_one(SendMessage(payload)) + finally: + await producer.shutdown() + + assert await poll_payloads(iggy_client, stream_name, topic_name) == expected + + +class TestBackgroundProducerBackpressure: + """Test each byte-budget backpressure policy through the Python API.""" + + @staticmethod + def config( + failure_mode: BackpressureMode, + *, + linger_time: timedelta, + ) -> BackgroundProducerConfig: + return BackgroundProducerConfig( + linger_time=linger_time, + batch_size=0, + batch_length=0, + max_buffer_size=250, + failure_mode=failure_mode, + ) + + @pytest.mark.asyncio + async def test_fail_immediately_rejects_a_send_when_the_buffer_is_full( + self, iggy_client: IggyClient, unique_name + ): + producer = await iggy_client.producer( + unique_name("fail-", min_bytes=12, max_bytes=12), + unique_name("topic-", min_bytes=12, max_bytes=12), + mode=self.config( + BackpressureMode.fail_immediately(), + linger_time=timedelta(seconds=5), + ), + ) + try: + assert (await producer.send_one(SendMessage("a" * 128))).confirmations == [] + with pytest.raises(RuntimeError, match="(?i)buffer|overflow"): + await producer.send_one(SendMessage("b" * 128)) + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_block_with_timeout_waits_then_reports_a_timeout( + self, iggy_client: IggyClient, unique_name + ): + producer = await iggy_client.producer( + unique_name("timeout-", min_bytes=16, max_bytes=16), + unique_name("topic-", min_bytes=16, max_bytes=16), + mode=self.config( + BackpressureMode.block_with_timeout(timedelta(milliseconds=100)), + linger_time=timedelta(seconds=5), + ), + ) + try: + await producer.send_one(SendMessage("a" * 128)) + started_at = time.monotonic() + with pytest.raises(RuntimeError, match="(?i)timeout"): + await producer.send_one(SendMessage("b" * 128)) + elapsed = time.monotonic() - started_at + + assert 0.05 <= elapsed < 1 + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_block_waits_until_a_previous_batch_releases_capacity( + self, iggy_client: IggyClient, unique_name + ): + producer = await iggy_client.producer( + unique_name("block-", min_bytes=12, max_bytes=12), + unique_name("topic-", min_bytes=12, max_bytes=12), + mode=self.config( + BackpressureMode.block(), + linger_time=timedelta(milliseconds=200), + ), + ) + try: + await producer.send_one(SendMessage("a" * 128)) + started_at = time.monotonic() + response = await producer.send_one(SendMessage("b" * 128)) + elapsed = time.monotonic() - started_at + + assert response.confirmations == [] + assert elapsed >= 0.1 + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_batch_larger_than_the_total_budget_fails_without_blocking( + self, iggy_client: IggyClient, unique_name + ): + producer = await iggy_client.producer( + unique_name("oversize-", min_bytes=16, max_bytes=16), + unique_name("topic-", min_bytes=16, max_bytes=16), + mode=BackgroundProducerConfig( + linger_time=timedelta(seconds=5), + batch_size=0, + batch_length=0, + max_buffer_size=100, + failure_mode=BackpressureMode.block(), + ), + ) + try: + with pytest.raises(RuntimeError, match="(?i)buffer|overflow"): + await asyncio.wait_for( + producer.send_one(SendMessage("larger than the budget")), + timeout=1, + ) + finally: + await producer.shutdown() + + +class TestProducerLifecycle: + """Test producer ownership, shutdown, and asynchronous context management.""" + + @pytest.mark.asyncio + async def test_linger_sends_run_concurrently_under_shared_lifecycle_access( + self, iggy_client: IggyClient, unique_name + ): + linger_seconds = 2 + producer = await iggy_client.producer( + unique_name(), + unique_name(), + mode=DirectProducerConfig(linger_time=timedelta(seconds=linger_seconds)), + ) + try: + await producer.send_one(SendMessage("prime linger")) + started_at = time.monotonic() + responses = await asyncio.gather( + producer.send_one(SendMessage("concurrent one")), + producer.send_one(SendMessage("concurrent two")), + ) + elapsed = time.monotonic() - started_at + + assert len(responses) == 2 + assert all(len(response.confirmations) == 1 for response in responses) + # Both reads wait against the same timestamp. A mutex around the + # producer would make the second wait through another full linger. + assert 1.2 <= elapsed < 3.6 + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_shutdown_is_sequentially_and_concurrently_idempotent( + self, iggy_client: IggyClient, unique_name + ): + producer = await iggy_client.producer(unique_name(), unique_name()) + + assert await asyncio.gather(producer.shutdown(), producer.shutdown()) == [ + None, + None, + ] + assert await producer.shutdown() is None + + @pytest.mark.asyncio + async def test_background_shutdown_is_idempotent_and_flushes_accepted_messages( + self, iggy_client: IggyClient, unique_name + ): + stream_name = unique_name("shutdown-", min_bytes=20, max_bytes=20) + topic_name = unique_name("topic-", min_bytes=20, max_bytes=20) + producer = await iggy_client.producer( + stream_name, + topic_name, + mode=BackgroundProducerConfig( + linger_time=timedelta(seconds=60), + batch_size=0, + batch_length=0, + ), + ) + await producer.send([SendMessage("one"), SendMessage("two")]) + + assert await asyncio.gather(producer.shutdown(), producer.shutdown()) == [ + None, + None, + ] + assert await producer.shutdown() is None + assert await poll_payloads(iggy_client, stream_name, topic_name) == [ + "one", + "two", + ] + + @pytest.mark.asyncio + async def test_every_send_rejects_use_after_shutdown( + self, iggy_client: IggyClient, unique_name + ): + stream_name = unique_name() + topic_name = unique_name() + producer = await iggy_client.producer(stream_name, topic_name) + await producer.shutdown() + + calls = [ + producer.send([]), + producer.send_one(SendMessage("closed")), + producer.send_with_partitioning([], None), + producer.send_to(stream_name, topic_name, [], None), + ] + for call in calls: + with pytest.raises(RuntimeError, match="closed|shut down"): + await call + + @pytest.mark.asyncio + async def test_shutdown_waits_for_a_send_holding_lifecycle_access( + self, iggy_client: IggyClient, unique_name + ): + producer = await iggy_client.producer( + unique_name(), + unique_name(), + mode=DirectProducerConfig(linger_time=timedelta(milliseconds=150)), + ) + await producer.send_one(SendMessage("prime linger")) + + sending = asyncio.ensure_future(producer.send_one(SendMessage("in flight"))) + + async def wait_for_active_send(): + # pyrefly: ignore # missing-attribute + while not producer._is_send_active(): + await asyncio.sleep(0) + + await asyncio.wait_for(wait_for_active_send(), timeout=1) + shutting_down = asyncio.ensure_future(producer.shutdown()) + + response, shutdown_result = await asyncio.wait_for( + asyncio.gather(sending, shutting_down), timeout=2 + ) + assert len(response.confirmations) == 1 + assert shutdown_result is None + with pytest.raises(RuntimeError, match="closed|shut down"): + await producer.send_one(SendMessage("too late")) + + @pytest.mark.asyncio + async def test_background_send_racing_shutdown_completes_without_deadlock( + self, iggy_client: IggyClient, unique_name + ): + stream_name = unique_name("race-", min_bytes=12, max_bytes=12) + topic_name = unique_name("topic-", min_bytes=12, max_bytes=12) + producer = await iggy_client.producer( + stream_name, + topic_name, + mode=BackgroundProducerConfig( + linger_time=timedelta(milliseconds=200), + batch_size=0, + batch_length=0, + max_buffer_size=250, + failure_mode=BackpressureMode.block(), + ), + ) + await producer.send_one(SendMessage("a" * 128)) + sending = asyncio.ensure_future(producer.send_one(SendMessage("b" * 128))) + + async def wait_for_blocked_send(): + # pyrefly: ignore # missing-attribute + while not producer._is_send_active(): + await asyncio.sleep(0) + + await asyncio.wait_for(wait_for_blocked_send(), timeout=1) + shutting_down = asyncio.ensure_future(producer.shutdown()) + response, shutdown_result = await asyncio.wait_for( + asyncio.gather(sending, shutting_down), timeout=2 + ) + + assert response.confirmations == [] + assert shutdown_result is None + assert await poll_payloads(iggy_client, stream_name, topic_name) == [ + "a" * 128, + "b" * 128, + ] + + @pytest.mark.asyncio + async def test_async_context_manager_returns_self_and_closes_normally( + self, iggy_client: IggyClient, unique_name + ): + producer = await iggy_client.producer(unique_name(), unique_name()) + + async with producer as entered: + assert entered is producer + assert ( + len((await entered.send_one(SendMessage("inside"))).confirmations) == 1 + ) + + with pytest.raises(RuntimeError, match="closed|shut down"): + await producer.send_one(SendMessage("outside")) + + @pytest.mark.asyncio + async def test_async_context_manager_closes_on_exception_without_suppressing_it( + self, iggy_client: IggyClient, unique_name + ): + producer = await iggy_client.producer(unique_name(), unique_name()) + + with pytest.raises(LookupError, match="application failure"): + async with producer: + raise LookupError("application failure") + + with pytest.raises(RuntimeError, match="closed|shut down"): + await producer.send_one(SendMessage("outside")) + + @pytest.mark.asyncio + @pytest.mark.parametrize( + "raises_inside", [False, True], ids=["normal", "exception"] + ) + async def test_background_context_manager_flushes_on_every_exit( + self, iggy_client: IggyClient, unique_name, raises_inside: bool + ): + stream_name = unique_name("context-", min_bytes=20, max_bytes=20) + topic_name = unique_name("topic-", min_bytes=20, max_bytes=20) + producer = await iggy_client.producer( + stream_name, + topic_name, + mode=BackgroundProducerConfig( + linger_time=timedelta(seconds=60), + batch_size=0, + batch_length=0, + ), + ) + + async def use_context(): + async with producer: + await producer.send_one(SendMessage("buffered")) + if raises_inside: + raise LookupError("application failure") + + if raises_inside: + with pytest.raises(LookupError, match="application failure"): + await use_context() + else: + await use_context() + + assert await poll_payloads(iggy_client, stream_name, topic_name) == ["buffered"] + with pytest.raises(RuntimeError, match="closed|shut down"): + await producer.send_one(SendMessage("outside")) + + +class TestProducerValidationAndRetries: + """Test stable error categories and observable retry policy behavior.""" + + @pytest.mark.asyncio + @pytest.mark.parametrize( + ("kwargs", "expected_exception"), + [ + ({"partitioning": 0}, TypeError), + ({"mode": object()}, TypeError), + ({"topic_partitions_count": -1}, ValueError), + ({"topic_partitions_count": 2**32}, ValueError), + ({"topic_partitions_count": 2**63}, OverflowError), + ({"topic_message_expiry": timedelta(seconds=1)}, TypeError), + ({"topic_max_size": 1000}, TypeError), + ({"send_retries": -1}, ValueError), + ({"send_retries": 2**32}, ValueError), + ({"send_retries": 2**63}, OverflowError), + ({"send_retry_interval": timedelta(0)}, ValueError), + ({"send_retry_interval": timedelta(microseconds=-1)}, ValueError), + ({"send_retry_interval": 1}, TypeError), + ], + ) + async def test_producer_configuration_errors_have_stable_categories( + self, + iggy_client: IggyClient, + unique_name, + kwargs: dict, + expected_exception: type[Exception], + ): + with pytest.raises(expected_exception): + await iggy_client.producer(unique_name(), unique_name(), **kwargs) + + @pytest.mark.asyncio + @pytest.mark.parametrize( + "mode", + [ + BackgroundProducerConfig(max_buffer_size=2**64 - 1), + BackgroundProducerConfig(max_in_flight=sys.maxsize), + ], + ids=["max-buffer-size", "max-in-flight"], + ) + async def test_background_limits_that_would_panic_rust_are_value_errors( + self, + iggy_client: IggyClient, + unique_name, + mode: BackgroundProducerConfig, + ): + with pytest.raises(ValueError, match="must not exceed"): + await iggy_client.producer( + unique_name(), + unique_name(), + mode=mode, + ) + + @pytest.mark.asyncio + async def test_bound_destination_requires_names( + self, iggy_client: IggyClient, unique_name + ): + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-type + await iggy_client.producer(1, unique_name()) + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-type + await iggy_client.producer(unique_name(), 1) + + @pytest.mark.asyncio + @pytest.mark.parametrize( + ("stream", "topic"), + [ + ("", "topic"), + ("stream", ""), + ("x" * 256, "topic"), + ("stream", "x" * 256), + ], + ) + async def test_bound_destination_rejects_invalid_names_as_value_errors( + self, iggy_client: IggyClient, stream: str, topic: str + ): + with pytest.raises(ValueError): + await iggy_client.producer(stream, topic) + + @pytest.mark.asyncio + async def test_send_arguments_reject_low_level_shorthands_and_wrong_types( + self, iggy_client: IggyClient, unique_name + ): + producer = await iggy_client.producer(unique_name(), unique_name()) + try: + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-type + await producer.send((SendMessage("tuple"),)) + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-type + await producer.send(["payload"]) + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-type + await producer.send_one("payload") + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-type + await producer.send_with_partitioning([SendMessage("integer")], 0) + with pytest.raises(TypeError): + # pyrefly: ignore # bad-argument-type + await producer.send_to(object(), "topic", [SendMessage("bad stream")]) + finally: + await producer.shutdown() + + @pytest.mark.asyncio + @pytest.mark.parametrize("identifier", [-1, 2**32, 2**63]) + async def test_send_to_identifier_overflow_is_not_folded_into_type_error( + self, iggy_client: IggyClient, unique_name, identifier: int + ): + producer = await iggy_client.producer(unique_name(), unique_name()) + try: + for stream, topic in [(identifier, "topic"), ("stream", identifier)]: + with pytest.raises(OverflowError): + await producer.send_to( + stream, + topic, + [SendMessage("outside u32")], + ) + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_server_send_errors_preserve_recovery_state( + self, iggy_client: IggyClient, unique_name + ): + producer = await iggy_client.producer( + unique_name(), + unique_name(), + send_retries=0, + ) + try: + with pytest.raises(ProducerSendError) as raised: + await producer.send_with_partitioning( + [SendMessage("missing partition")], + Partitioning.partition_id(1), + ) + + error = raised.value + assert isinstance(error, RuntimeError) + assert error.cause + assert isinstance(error.__cause__, RuntimeError) + assert str(error.__cause__) == error.cause + assert len(error.failed) == 1 + assert isinstance(error.failed[0], SendMessage) + assert error.committed == [] + finally: + await producer.shutdown() + + @pytest.mark.asyncio + @pytest.mark.parametrize("send_retries", [None, 0]) + async def test_none_and_zero_disable_retries( + self, iggy_client: IggyClient, unique_name, send_retries: int | None + ): + producer = await iggy_client.producer( + unique_name(), + unique_name(), + send_retries=send_retries, + send_retry_interval=timedelta(seconds=5), + ) + try: + with pytest.raises(RuntimeError): + await asyncio.wait_for( + producer.send_to( + unique_name(), unique_name(), [SendMessage("single attempt")] + ), + timeout=1, + ) + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_explicit_none_interval_retries_without_delay( + self, iggy_client: IggyClient, unique_name + ): + producer = await iggy_client.producer( + unique_name(), + unique_name(), + send_retries=10, + send_retry_interval=None, + ) + try: + with pytest.raises(RuntimeError): + await asyncio.wait_for( + producer.send_to( + unique_name(), unique_name(), [SendMessage("immediate retries")] + ), + timeout=1, + ) + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_custom_retry_interval_paces_later_retries( + self, iggy_client: IggyClient, unique_name + ): + producer = await iggy_client.producer( + unique_name(), + unique_name(), + send_retries=2, + send_retry_interval=timedelta(milliseconds=150), + ) + try: + started_at = time.monotonic() + with pytest.raises(RuntimeError): + await producer.send_to( + unique_name(), unique_name(), [SendMessage("paced retries")] + ) + elapsed = time.monotonic() - started_at + + assert elapsed >= 0.1 + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_default_retry_policy_retains_the_rust_interval( + self, iggy_client: IggyClient, unique_name + ): + producer = await iggy_client.producer(unique_name(), unique_name()) + try: + started_at = time.monotonic() + with pytest.raises(RuntimeError): + await asyncio.wait_for( + producer.send_to( + unique_name(), unique_name(), [SendMessage("default retries")] + ), + timeout=6, + ) + elapsed = time.monotonic() - started_at + + # The first retry is immediate. Three retries with the one-second + # default wait for two later interval ticks. + assert elapsed >= 1.5 + finally: + await producer.shutdown() + + @pytest.mark.asyncio + async def test_background_write_retries_until_a_destination_appears( + self, iggy_client: IggyClient, unique_name + ): + destination_stream = unique_name("retry-stream-", min_bytes=24, max_bytes=24) + destination_topic = unique_name("retry-topic-", min_bytes=24, max_bytes=24) + producer = await iggy_client.producer( + unique_name("bound-stream-", min_bytes=24, max_bytes=24), + unique_name("bound-topic-", min_bytes=24, max_bytes=24), + mode=BackgroundProducerConfig( + linger_time=timedelta(0), + batch_length=1, + ), + send_retries=10, + send_retry_interval=timedelta(milliseconds=100), + ) + try: + response = await producer.send_to( + destination_stream, + destination_topic, + [SendMessage("eventual destination")], + ) + assert response.confirmations == [] + + # The first retry is immediate. Create the destination before a + # later interval tick so the worker can recover asynchronously. + await asyncio.sleep(0.03) + await iggy_client.create_stream(destination_stream) + await iggy_client.create_topic(destination_stream, destination_topic, 1) + finally: + await producer.shutdown() + + assert await wait_for_payloads( + iggy_client, + destination_stream, + destination_topic, + ["eventual destination"], + ) == ["eventual destination"] + + @pytest.mark.asyncio + async def test_background_write_reconnects_after_server_restart( + self, + tmp_path: Path, + unique_name, + ): + data_path = tmp_path / "restartable-server" + server = spawn_restartable_server(data_path, "127.0.0.1:0") + producer: IggyProducer | None = None + + try: + host, port = await discover_server_address(data_path, server) + await asyncio.to_thread(wait_for_server, host, port, 15, 1) + client = IggyClient.from_connection_string( + f"iggy+tcp://iggy:iggy@{host}:{port}" + "?reconnection_interval=100ms&reestablish_after=0" + ) + await client.connect() + + stream = unique_name("reconnect-stream-", min_bytes=28, max_bytes=28) + topic = unique_name("reconnect-topic-", min_bytes=28, max_bytes=28) + producer = await client.producer( + stream, + topic, + mode=BackgroundProducerConfig( + linger_time=timedelta(0), + batch_length=1, + ), + send_retries=20, + send_retry_interval=timedelta(milliseconds=100), + ) + + await stop_restartable_server(server) + server = None + accepted = await producer.send_one(SendMessage("after restart")) + assert accepted.confirmations == [] + + server = spawn_restartable_server(data_path, f"{host}:{port}") + await asyncio.to_thread(wait_for_server, host, port, 15, 1) + await asyncio.wait_for(producer.shutdown(), timeout=15) + producer = None + + assert await wait_for_payloads( + client, + stream, + topic, + ["after restart"], + timeout=10, + ) == ["after restart"] + finally: + if producer is not None: + if server is None: + server = spawn_restartable_server(data_path, f"{host}:{port}") + await asyncio.to_thread(wait_for_server, host, port, 15, 1) + with contextlib.suppress(RuntimeError, TimeoutError): + await asyncio.wait_for(producer.shutdown(), timeout=5) + if server is not None: + await stop_restartable_server(server) From 78151b95111a7bc4427e7defa2b90260aa0e6c05 Mon Sep 17 00:00:00 2001 From: Gunther Xing Date: Mon, 14 Sep 2026 00:13:25 +0800 Subject: [PATCH 127/182] fix(ci): support multiline README commands (#4162) Closes #4161 --- examples/rust/README.md | 6 ++++-- scripts/utils.sh | 40 ++++++++++++++++++++++++++++++++++++++-- 2 files changed, 42 insertions(+), 4 deletions(-) diff --git a/examples/rust/README.md b/examples/rust/README.md index bbb444f1cf..4cf14bbfce 100644 --- a/examples/rust/README.md +++ b/examples/rust/README.md @@ -96,8 +96,10 @@ Demonstrates fundamental client connection, authentication, batch message sendin To run the pair over HTTP: ```bash -cargo run --example basic-producer -- --transport http -cargo run --example basic-consumer -- --transport http +cargo run --example basic-producer -- \ + --transport http +cargo run --example basic-consumer -- \ + --transport http ``` ## Message Pattern Examples diff --git a/scripts/utils.sh b/scripts/utils.sh index 0c77704877..cc5d4399c3 100755 --- a/scripts/utils.sh +++ b/scripts/utils.sh @@ -343,9 +343,45 @@ function portable_timeout() { fi } +# Grep must see complete shell commands instead of individual README lines. +function read_readme_logical_lines() { + local readme_file="$1" + + awk ' + function has_line_continuation(line, position, count) { + count = 0 + for (position = length(line); position > 0; position--) { + if (substr(line, position, 1) != "\\") { + break + } + count++ + } + return count % 2 == 1 + } + + { + if (has_line_continuation($0)) { + pending = pending substr($0, 1, length($0) - 1) + continued = 1 + next + } + + print pending $0 + pending = "" + continued = 0 + } + + END { + if (continued) { + print pending "\\" + } + } + ' "${readme_file}" +} + # Run commands extracted from a README file. # Usage: run_readme_commands readme_file grep_pattern [cmd_timeout [grep_exclude]] -# Reads matching lines, strips backticks/comments, executes each. +# Reads matching logical lines, strips backticks/comments, executes each. # Calls TRANSFORM_COMMAND function on each command if defined. # Returns: sets global EXAMPLES_EXIT_CODE and README_COMMANDS_EXECUTED. # Zero matches is not an error here: callers iterate multiple README @@ -364,7 +400,7 @@ function run_readme_commands() { fi local commands - commands=$(grep -E "${grep_pattern}" "${readme_file}" || true) + commands=$(read_readme_logical_lines "${readme_file}" | grep -E "${grep_pattern}" || true) if [ -n "${grep_exclude}" ]; then commands=$(echo "${commands}" | grep -v -e "${grep_exclude}" || true) fi From c2bd0c0cd3b227299ca441f8a650ab5bd86ebaa0 Mon Sep 17 00:00:00 2001 From: Justin Mclean Date: Mon, 14 Sep 2026 16:14:20 +1000 Subject: [PATCH 128/182] fix(ci): don't mark PRs stale while they wait on review (#4172) --- .github/workflows/stale-prs.yml | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/.github/workflows/stale-prs.yml b/.github/workflows/stale-prs.yml index 0038d49a12..54e013dc0a 100644 --- a/.github/workflows/stale-prs.yml +++ b/.github/workflows/stale-prs.yml @@ -51,7 +51,9 @@ jobs: This pull request was automatically closed because it has been inactive for 14 days. Feel free to reopen it if you'd like to continue working on it. - exempt-pr-labels: "pinned" + # S-waiting-on-review means the next move is ours, not the author's, + # so the inactivity clock should not run against them. + exempt-pr-labels: "pinned,S-waiting-on-review" exempt-draft-pr: true days-before-issue-stale: -1 days-before-issue-close: -1 From 1b4ebeecfd546d7b5f1aec25e7ebc26337ed9119 Mon Sep 17 00:00:00 2001 From: Matthew Patton Date: Mon, 14 Sep 2026 03:37:26 -0400 Subject: [PATCH 129/182] fix(node): stop dropping and corrupting tokens in list responses (#4171) --- .../node/src/wire/token/token.utils.test.ts | 88 +++++++++++++++++++ foreign/node/src/wire/token/token.utils.ts | 10 +-- 2 files changed, 91 insertions(+), 7 deletions(-) create mode 100644 foreign/node/src/wire/token/token.utils.test.ts diff --git a/foreign/node/src/wire/token/token.utils.test.ts b/foreign/node/src/wire/token/token.utils.test.ts new file mode 100644 index 0000000000..571146bd24 --- /dev/null +++ b/foreign/node/src/wire/token/token.utils.test.ts @@ -0,0 +1,88 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +import { describe, it } from "node:test"; +import assert from "node:assert/strict"; +import { deserializeTokens } from "./token.utils.js"; + +// Real server wire shape: [nameLength: u8][name][expiry: 8-byte LE, always present, 0 = never-expiring]. +// See core/binary_protocol/.../get_personal_access_tokens.rs and core/server/src/responses.rs:1195-1209. +const tokenRecord = (name: string, expiry: bigint = 0n): Buffer => { + const nameBuf = Buffer.from(name, "utf-8"); + const head = Buffer.from([nameBuf.length]); + const expiryBuf = Buffer.alloc(8); + expiryBuf.writeBigUInt64LE(expiry); + return Buffer.concat([head, nameBuf, expiryBuf]); +}; + +describe("deserializeTokens", () => { + it("reads all tokens in a 3-token buffer, including the one after the 2nd", () => { + const t1 = tokenRecord("ci-a", 123n); + const t2 = tokenRecord("ci-b"); + const t3 = tokenRecord("x"); + const buffer = Buffer.concat([t1, t2, t3]); + + const tokens = deserializeTokens(buffer); + + assert.deepEqual( + tokens.map((t) => t.name), + ["ci-a", "ci-b", "x"], + ); + assert.notEqual(tokens[0].expiry, null); + assert.equal(tokens[1].expiry, null); + assert.equal(tokens[2].expiry, null); + }); + + it("reads all tokens cleanly regardless of record length ordering", () => { + const t1 = tokenRecord("short"); + const t2 = tokenRecord("second"); + const t3 = tokenRecord("a-much-longer-token-name-here"); + const t4 = tokenRecord("last"); + const buffer = Buffer.concat([t1, t2, t3, t4]); + + const tokens = deserializeTokens(buffer); + + assert.deepEqual( + tokens.map((t) => t.name), + ["short", "second", "a-much-longer-token-name-here", "last"], + ); + assert.ok(tokens.every((t) => t.expiry === null)); + }); + + it("decodes a single token with no trailing data", () => { + const buffer = tokenRecord("solo"); + + const tokens = deserializeTokens(buffer); + + assert.deepEqual(tokens, [{ name: "solo", expiry: null }]); + }); + + it("decodes a non-zero expiry into the correct point in time", () => { + const expiryMicros = 1700000000000000n; + const buffer = tokenRecord("with-expiry", expiryMicros); + + const tokens = deserializeTokens(buffer); + + assert.equal(tokens[0].expiry?.getTime(), Number(expiryMicros / 1000n)); + }); + + it("throws on a buffer truncated before the expiry field", () => { + const partial = Buffer.concat([Buffer.from([4]), Buffer.from("iggy")]); + + assert.throws(() => deserializeTokens(partial), RangeError); + }); +}); diff --git a/foreign/node/src/wire/token/token.utils.ts b/foreign/node/src/wire/token/token.utils.ts index e4b7a641c7..efe1b3de49 100644 --- a/foreign/node/src/wire/token/token.utils.ts +++ b/foreign/node/src/wire/token/token.utils.ts @@ -78,13 +78,9 @@ export const deserializeCreateToken = (p: Buffer, pos = 0): TokenDeserialized => export const deserializeToken = (p: Buffer, pos = 0): TokenSerialized => { const nameLength = p.readUInt8(pos); const name = p.subarray(pos + 1, pos + 1 + nameLength).toString(); - const rest = p.subarray(pos + 1 + nameLength); - let expiry = null; - let bytesRead = pos + 1 + nameLength; - if (rest.length >= 8) { - expiry = toDate(rest.readBigUInt64LE(0)); - bytesRead += 8; - } + const expiryRaw = p.readBigUInt64LE(pos + 1 + nameLength); + const expiry = expiryRaw === 0n ? null : toDate(expiryRaw); + const bytesRead = 1 + nameLength + 8; return { bytesRead, data: { From 153088b3ec29ab47a2fc91cc04bbc1306a0b187b Mon Sep 17 00:00:00 2001 From: Rohan Dubey Date: Mon, 14 Sep 2026 18:48:50 +0530 Subject: [PATCH 130/182] fix(connectors): defer postgres source progress until ack (#3957) Closes #3635 --- .claude/skills/connector-source/SKILL.md | 55 +- .claude/skills/connector-source/TEMPLATE.md | 49 +- Cargo.lock | 5 + Cargo.toml | 5 +- core/connectors/README.md | 2 +- core/connectors/runtime/README.md | 2 + core/connectors/runtime/src/manager/source.rs | 35 + core/connectors/runtime/src/source.rs | 122 +- core/connectors/runtime/src/stream.rs | 47 +- core/connectors/sdk/src/source.rs | 174 +- core/connectors/sources/README.md | 2 +- .../sources/postgres_source/Cargo.toml | 3 + .../sources/postgres_source/README.md | 101 +- .../sources/postgres_source/src/lib.rs | 1927 +++++++++++++++-- .../src/harness/handle/connectors_runtime.rs | 16 +- core/integration/src/harness/seeds.rs | 8 +- .../tests/connectors/fixtures/mod.rs | 6 +- .../tests/connectors/fixtures/postgres/cdc.rs | 41 + .../tests/connectors/fixtures/postgres/mod.rs | 8 +- .../connectors/fixtures/postgres/source.rs | 313 ++- .../tests/connectors/postgres/mod.rs | 58 +- .../connectors/postgres/postgres_source.rs | 410 +++- .../postgres/postgres_source_cdc.rs | 342 ++- .../tests/connectors/postgres/restart.rs | 3 +- 24 files changed, 3378 insertions(+), 356 deletions(-) diff --git a/.claude/skills/connector-source/SKILL.md b/.claude/skills/connector-source/SKILL.md index e6f747ea8b..23613c5b05 100644 --- a/.claude/skills/connector-source/SKILL.md +++ b/.claude/skills/connector-source/SKILL.md @@ -45,16 +45,16 @@ The macro shares the source as `Arc` across the FFI callback and forwarding l ### Lock discipline -Never hold the state `Mutex` across upstream I/O. Canonical pattern (matches `sources/postgres_source/src/lib.rs::poll_tables`): +Never hold the state `Mutex` across upstream I/O. Build a candidate from committed +state, then stage it until the runtime reports the batch result: ```rust -let cursor = { self.state.lock().await.cursor.clone() }; // brief read -let rows = client.query(&sql, &[&cursor]).await?; // no lock held -let persisted = { // brief write - let mut state = self.state.lock().await; - state.cursor = Some(new_cursor); - ConnectorState::serialize(&*state, CONNECTOR_NAME, self.id) -}; +let mut candidate = self.state.lock().await.clone(); +let rows = client.query(&sql, &[&candidate.cursor]).await?; +candidate.cursor = Some(new_cursor); +let persisted = ConnectorState::serialize(&candidate, CONNECTOR_NAME, self.id) + .ok_or_else(|| Error::Serialization("failed to serialize source state".into()))?; +*self.pending.lock().await = Some(candidate); ``` ### Delivery acknowledgment @@ -78,13 +78,36 @@ of backoff, without calling `close()`. ### State persistence -- `ConnectorState` is `Vec` via MessagePack (`rmp_serde`). Use `ConnectorState::serialize(&state, NAME, id)` + `ConnectorState::deserialize::(NAME, id)`. Both return `Option` and log on failure (non-fatal). -- Runtime saves to `{state_path}/source_{key}.state` only after a successful Iggy send. Between `poll()` returning and the runtime persisting the save, a crash leaves the same cursor for the next poll - downstream must tolerate at-least-once. -- **Return state on a batch whose send cannot fail.** The runtime saves state only on the success branch of the Iggy send, so state riding a batch of messages is skipped whenever that send fails, while the source has already cleared whatever dirty flag it tracks. -- Timestamp sources that stage their cursor and apply it in `on_batch_result` can attach state to any batch. A source whose state is a control-plane record rather than a cursor, such as `http_source`'s endpoint registry, should ride it on an empty batch instead, which cannot fail for want of a publish. -- The runtime can still NACK an empty batch: it short-circuits the send stage when its own state storage is latched or a pending checkpoint will not resolve. Never treat a hand-off as durable; `on_batch_result` is how you learn. +- `ConnectorState` is `Vec` via MessagePack (`rmp_serde`). Use + `ConnectorState::serialize(&state, NAME, id)` and + `ConnectorState::deserialize::(NAME, id)`. +- `poll()` must not commit cursors or destructive work. Return messages with candidate state and + keep the corresponding work staged. +- The runtime sends the batch, saves its candidate state to + `{state_path}/source_{key}.state`, then calls `on_batch_result(Ack)`. Commit staged in-memory + state and external delete or mark operations only on ACK. A NACK discards the candidate so the + same data can be polled again. +- A crash between `poll()` returning and state persistence leaves the prior cursor for the next + poll, so downstream must tolerate at-least-once delivery. +- Cursor sources that stage state until `on_batch_result` can attach state to the corresponding + message batch. Control-plane state that is independent of a publish can ride an empty batch + instead. +- Return `state: None` for an empty poll when no watermark changed. If an empty poll advances a + watermark, stage and return the new state through the same ACK handshake. +- The runtime can still NACK an empty batch when state storage is latched or a pending checkpoint + cannot resolve. Never treat hand-off as durable; `on_batch_result` reports the outcome. +- Treat candidate-state serialization failure as a poll error. Do not send messages without the + state needed to resume them safely. - Keep `State` small - rewritten every batch. No unbounded vecs. +The SDK allows one in-flight batch. Five consecutive NACKs stop the source and +require a manual restart. Returning `Err` from `on_batch_result` is fatal, so +retry transient backend failures inside the callback before returning an error. +The runtime must report ACK or NACK within the SDK's 30-second batch-result +window. Once the result is received, the SDK waits for `on_batch_result` to +finish, so the callback must bound its own connection acquisition and retry +budget rather than relying on the SDK deadline. + ### Sleep first `poll()` must `sleep(self.poll_interval).await` before any work. Without it, an empty source spins a CPU. @@ -119,7 +142,7 @@ Match `ProducedMessages.schema` to the bytes in `messages[i].payload`: | Transient fetch failure (retry-worthy) | `Error::Connection` or `Error::HttpRequestFailed` | | Permanent fetch failure (auth, schema gone) | `Error::PermanentHttpError` | | Row failed to serialize | `Error::Serialization(...)` | -| State serialization failed | log + skip (non-fatal) | +| State serialization failed | `Error::Serialization(...)` | Returning `Err` from `poll()` is only logged by the SDK's FFI bridge (`sdk/src/source.rs::handle_messages`) - the loop continues, the next @@ -149,7 +172,7 @@ Iggy consumer-loop labels use literal API names (`offset=`, `current_offset=`). 1. `async fn poll(&mut self)` - won't compile. Use `&self` + `Mutex`. 2. Holding `state.lock()` across the fetch I/O - blocks `close()`, causes shutdown timeouts. 3. Forgetting to sleep - 100% CPU on idle source. -4. Returning state only on success - state should advance on empty polls too. +4. Committing a cursor or deleting source data in `poll()` - stage it and wait for ACK. 5. Unbounded data in `State` - rewritten every batch. keep O(constant). 6. `std::sync::Mutex` - blocks the executor. Use `tokio::sync::Mutex`. 7. Not setting `ProducedMessage.id` when a stable ID exists - leaves a consumer nothing to dedupe a replayed duplicate on. It does not make the write idempotent server-side, because nothing there reads it. @@ -159,7 +182,7 @@ Iggy consumer-loop labels use literal API names (`offset=`, `current_offset=`). Mandatory four canonical source state tests (see [connector-testing](../connector-testing/SKILL.md) for the full pattern). Copy from `sources/random_source/src/lib.rs::tests`. Plus config defaults, payload building, schema selection. -Integration tests under `core/integration/tests/connectors//` for any source backed by external infra. Use `#[iggy_harness]` + a `TestFixture` backed by `testcontainers-modules`. Reference: `core/integration/tests/connectors/postgres/postgres_source.rs` (multi-mode tests) + `restart.rs` (state survives restart). +Integration tests under `core/integration/tests/connectors//` for any source backed by external infra. Use `#[iggy_harness]` + a `TestFixture` backed by `testcontainers-modules`. Reference: `core/integration/tests/connectors/postgres/postgres_source.rs` (multi-mode tests) + `restart.rs` (state survives restart). Exercise both ACK and NACK paths when the source stages cursors or destructive work. ## Before declaring done diff --git a/.claude/skills/connector-source/TEMPLATE.md b/.claude/skills/connector-source/TEMPLATE.md index 7a27953654..04b8c6062e 100644 --- a/.claude/skills/connector-source/TEMPLATE.md +++ b/.claude/skills/connector-source/TEMPLATE.md @@ -16,13 +16,15 @@ helpers below. ```rust /* Apache 2.0 header */ +use std::str::FromStr; +use std::time::Duration; + use async_trait::async_trait; use iggy_connector_sdk::{ - ConnectorState, Error, ProducedMessage, ProducedMessages, Schema, Source, source_connector, + ConnectorState, Error, ProducedMessage, ProducedMessages, Schema, Source, + source::SourceBatchResult, source_connector, }; use serde::{Deserialize, Serialize}; -use std::str::FromStr; -use std::time::Duration; use tokio::sync::Mutex; use tokio::time::sleep; use tracing::{debug, error, info, warn}; @@ -41,7 +43,7 @@ pub struct MySourceConfig { pub verbose_logging: Option, } -#[derive(Debug, Serialize, Deserialize)] +#[derive(Debug, Clone, Serialize, Deserialize)] struct State { cursor: Option, // WAL LSN, scroll id, timestamp, ... last_offset: u64, @@ -56,6 +58,7 @@ pub struct MySource { verbose: bool, client: Option, state: Mutex, + pending: Mutex>, } impl MySource { @@ -88,6 +91,7 @@ impl MySource { last_offset: 0, messages_produced: 0, })), + pending: Mutex::new(None), } } } @@ -109,9 +113,9 @@ impl Source for MySource { async fn poll(&self) -> Result { sleep(self.poll_interval).await; // sleep first - backpressure - let cursor = { self.state.lock().await.cursor.clone() }; // brief read + let mut candidate = self.state.lock().await.clone(); - let fetched = self.fetch_since(cursor.as_deref()).await?; // no lock held + let fetched = self.fetch_since(candidate.cursor.as_deref()).await?; let mut messages = Vec::with_capacity(fetched.len()); let mut next_cursor = None; @@ -137,22 +141,37 @@ impl Source for MySource { ); } - let persisted = { // brief write - let mut state = self.state.lock().await; - state.messages_produced += messages.len() as u64; - if let Some(c) = next_cursor { - state.cursor = Some(c); - } - ConnectorState::serialize(&*state, CONNECTOR_NAME, self.id) - }; + if messages.is_empty() { + return Ok(ProducedMessages { + schema: Schema::Json, + messages, + state: None, + }); + } + + candidate.messages_produced += messages.len() as u64; + candidate.cursor = next_cursor; + let persisted = ConnectorState::serialize(&candidate, CONNECTOR_NAME, self.id) + .ok_or_else(|| Error::Serialization("failed to serialize source state".into()))?; + *self.pending.lock().await = Some(candidate); Ok(ProducedMessages { schema: Schema::Json, messages, - state: persisted, + state: Some(persisted), }) } + async fn on_batch_result(&self, result: SourceBatchResult) -> Result<(), Error> { + let (SourceBatchResult::Ack, Some(candidate)) = + (result, self.pending.lock().await.take()) + else { + return Ok(()); + }; + *self.state.lock().await = candidate; + Ok(()) + } + async fn close(&mut self) -> Result<(), Error> { if let Some(client) = self.client.take() { let _ = client; // or `client.close().await;` for sqlx pools diff --git a/Cargo.lock b/Cargo.lock index 8e0573eca1..0a71fdfffc 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -7362,6 +7362,7 @@ dependencies = [ "humantime", "iggy_common", "iggy_connector_sdk", + "rmp-serde", "secrecy", "serde", "serde_json", @@ -13218,6 +13219,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "05b44e85bf579a8eeb4ceaa77a3a523baf2bf0e9bac7e40f405d537b5d2d5ccb" dependencies = [ "base64 0.22.1", + "bigdecimal", "bytes", "cfg-if", "chrono", @@ -13294,6 +13296,7 @@ version = "0.9.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "90b8020fe17c5f2c245bfa2505d7ef59c5604839527c740266ad2214acebea27" dependencies = [ + "bigdecimal", "bitflags 2.13.1", "byteorder", "bytes", @@ -13324,6 +13327,7 @@ checksum = "87a2bdd6e83f6b3ea525ca9fee568030508b58355a43d0b2c1674d5f79dcd65e" dependencies = [ "atoi", "base64 0.22.1", + "bigdecimal", "bitflags 2.13.1", "byteorder", "chrono", @@ -13340,6 +13344,7 @@ dependencies = [ "log", "md-5 0.11.0", "memchr", + "num-bigint", "rand 0.10.2", "serde", "serde_json", diff --git a/Cargo.toml b/Cargo.toml index e97a0a5911..13676a9cd2 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -314,7 +314,10 @@ sqlx = { version = "0.9.0", features = [ "runtime-tokio", "tls-rustls", "postgres", - # "mysql": the Doris integration fixture talks to Doris over its MySQL frontend. + "bigdecimal", + # "mysql": Doris exposes a MySQL-wire frontend; the Doris sink's + # integration-test fixture talks to it over this driver. Cargo unifies + # features across the workspace, so it is declared once here. "mysql", "chrono", "uuid", diff --git a/core/connectors/README.md b/core/connectors/README.md index 821a3cd909..b334009278 100644 --- a/core/connectors/README.md +++ b/core/connectors/README.md @@ -51,7 +51,7 @@ Run these commands from the root of the same Iggy source checkout used for the s ```bash target/release/iggy --username iggy --password iggy stream create example_stream - target/release/iggy --username iggy --password iggy topic create example_stream example_topic 1 none 1d + target/release/iggy --username iggy --password iggy topic create example_stream example_topic 1 none 1d --durability persisted target/release/iggy --username iggy --password iggy stream create qw target/release/iggy --username iggy --password iggy topic create qw records 1 none 1d ``` diff --git a/core/connectors/runtime/README.md b/core/connectors/runtime/README.md index 37f8e54342..eb907521ea 100644 --- a/core/connectors/runtime/README.md +++ b/core/connectors/runtime/README.md @@ -47,6 +47,8 @@ IGGY_CONNECTORS_CONFIG_PATH=connectors.toml cargo run --bin iggy-connectors Supported scalar fields and indexed list entries use environment variables with nested keys joined by underscores, for example `IGGY_CONNECTORS_IGGY_USERNAME`. Header and URL-template maps are configured in TOML. The runtime loads the first `.env` file found in the working directory or its parents, or the file specified by `IGGY_CONNECTORS_ENV_PATH`. +Source destination topics must persist every acknowledged batch before the runtime checkpoints the source or invokes its Ack hook. Missing topics are therefore created with `durability = "persisted"` and `messages_required_to_save = 1`. An existing topic must use `durability = "persisted"`; its save threshold may differ because persisted acknowledgments already wait for durable storage. + ## State storage Source plugins can supply optional checkpoint bytes. The runtime stores these opaque bytes using the backend selected by `state.storage`: diff --git a/core/connectors/runtime/src/manager/source.rs b/core/connectors/runtime/src/manager/source.rs index 3cff2305ac..fe962b1ebf 100644 --- a/core/connectors/runtime/src/manager/source.rs +++ b/core/connectors/runtime/src/manager/source.rs @@ -103,6 +103,15 @@ impl SourceManager { } } + pub async fn recover_from_error(&self, key: &str, metrics: Option<&Arc>) { + if let Some(source) = self.sources.get(key) { + let mut source = source.lock().await; + if source.info.status == ConnectorStatus::Error { + source.apply_status(ConnectorStatus::Running, metrics); + } + } + } + pub async fn is_stopping_or_stopped(&self, key: &str) -> bool { let Some(source) = self.sources.get(key).map(|entry| entry.value().clone()) else { return true; @@ -650,6 +659,32 @@ mod tests { assert!(details.info.last_error.is_some()); } + #[tokio::test] + async fn recover_from_error_should_not_overwrite_stopping_status() { + let mut details = create_test_source_details("pg", 1); + details.info.status = ConnectorStatus::Stopping; + let manager = SourceManager::new(vec![details]); + + manager.recover_from_error("pg", None).await; + + let source = manager.get("pg").await.unwrap(); + let details = source.lock().await; + assert_eq!(details.info.status, ConnectorStatus::Stopping); + } + + #[tokio::test] + async fn recover_from_error_should_restore_running_status_and_clear_error() { + let manager = SourceManager::new(vec![create_test_source_details("pg", 1)]); + manager.set_error("pg", "connection failed", None).await; + + manager.recover_from_error("pg", None).await; + + let source = manager.get("pg").await.unwrap(); + let details = source.lock().await; + assert_eq!(details.info.status, ConnectorStatus::Running); + assert!(details.info.last_error.is_none()); + } + #[tokio::test] async fn stop_should_return_not_found_for_unknown_key() { let metrics = Arc::new(Metrics::init()); diff --git a/core/connectors/runtime/src/source.rs b/core/connectors/runtime/src/source.rs index 7c54108d8f..a6a747b2c9 100644 --- a/core/connectors/runtime/src/source.rs +++ b/core/connectors/runtime/src/source.rs @@ -19,9 +19,10 @@ use dashmap::DashMap; use dlopen2::wrapper::Container; use flume::{Receiver, Sender}; use iggy::prelude::{ - DirectConfig, HeaderKey, HeaderValue, IggyClient, IggyDuration, IggyError, IggyMessage, - IggyProducer, + DirectConfig, HeaderKey, HeaderValue, Identifier, IggyClient, IggyDuration, IggyError, + IggyMessage, IggyProducer, StreamClient, TopicClient, TopicCreateOptions, }; +use iggy_common::{Durability, TopicRuntimeOptions}; use iggy_connector_sdk::encoders::avro::{AvroEncoderConfig, AvroStreamEncoder}; use iggy_connector_sdk::{ ConnectorState, DecodedMessage, Error as SdkError, ProducedMessages, Schema, StreamEncoder, @@ -56,6 +57,7 @@ use tokio::runtime::Handle; use tokio::task::JoinHandle; const MAX_FAILED_TAIL_RETRIES: u32 = 3; +const SOURCE_TOPIC_MESSAGES_REQUIRED_TO_SAVE: u32 = 1; pub(crate) struct SourceSenderEntry { pub(crate) sender: Sender, @@ -452,9 +454,14 @@ pub(crate) async fn setup_source_producer( .map_err(|error| { RuntimeError::InvalidConfiguration(format!("Invalid linger time: {error}")) })?; + ensure_durable_source_topic(iggy_client, &stream.stream, &stream.topic).await?; let batch_length = stream.batch_length.unwrap_or(1000); let producer = iggy_client .producer(&stream.stream, &stream.topic)? + // The topic was validated above. If it disappears before init, + // recreating it with producer defaults would drop the durability guarantee. + .do_not_create_stream_if_not_exists() + .do_not_create_topic_if_not_exists() .direct( DirectConfig::builder() .batch_length(batch_length) @@ -493,6 +500,57 @@ pub(crate) async fn setup_source_producer( Ok((producer, encoder, transforms)) } +async fn ensure_durable_source_topic( + client: &IggyClient, + stream_name: &str, + topic_name: &str, +) -> Result<(), RuntimeError> { + let stream_id = Identifier::try_from(stream_name)?; + if client.get_stream(&stream_id).await?.is_none() { + client.create_stream(stream_name).await?; + } + + let topic_id = Identifier::try_from(topic_name)?; + let topic = match client.get_topic(&stream_id, &topic_id).await? { + Some(topic) => topic, + None => { + client + .create_topic( + &stream_id, + topic_name, + &TopicCreateOptions { + partitions_count: Some(1), + durability: Durability::Persisted, + messages_required_to_save: Some(SOURCE_TOPIC_MESSAGES_REQUIRED_TO_SAVE), + ..TopicCreateOptions::default() + }, + ) + .await? + } + }; + + validate_source_topic_durability( + stream_name, + topic_name, + TopicRuntimeOptions::from_resource_options(&topic.options), + ) +} + +fn validate_source_topic_durability( + stream_name: &str, + topic_name: &str, + options: TopicRuntimeOptions, +) -> Result<(), RuntimeError> { + if options.durability == Durability::Persisted { + return Ok(()); + } + + Err(RuntimeError::InvalidConfiguration(format!( + "Source destination topic '{stream_name}/{topic_name}' must use durability=persisted; found durability={}", + options.durability + ))) +} + #[allow(clippy::too_many_arguments)] pub(crate) async fn source_forwarding_loop( plugin_id: u32, @@ -729,6 +787,11 @@ pub(crate) async fn source_forwarding_loop( .set_error(&plugin_key, &error_msg, Some(&context.metrics)) .await; } + } else if should_recover_source(batch_result, sent_count) { + context + .sources + .recover_from_error(&plugin_key, Some(&context.metrics)) + .await; } let total_elapsed = total_start.elapsed(); @@ -763,6 +826,10 @@ pub(crate) async fn source_forwarding_loop( .await; } +fn should_recover_source(batch_result: SourceBatchResult, sent_count: usize) -> bool { + batch_result == SourceBatchResult::Ack && sent_count > 0 +} + #[allow(clippy::too_many_arguments)] pub(crate) fn spawn_source_handler( plugin_id: u32, @@ -1329,6 +1396,57 @@ mod tests { ); } + #[test] + fn given_acknowledged_nonempty_batch_when_recovering_should_restore_source_status() { + assert!(should_recover_source(SourceBatchResult::Ack, 1)); + } + + #[test] + fn given_acknowledged_empty_batch_when_recovering_should_preserve_source_error() { + assert!(!should_recover_source(SourceBatchResult::Ack, 0)); + } + + #[test] + fn given_rejected_nonempty_batch_when_recovering_should_preserve_source_error() { + assert!(!should_recover_source(SourceBatchResult::Nack, 1)); + } + + #[test] + fn given_persisted_per_batch_topic_when_validating_source_destination_should_accept() { + let options = TopicRuntimeOptions { + durability: Durability::Persisted, + messages_required_to_save: Some(1), + ..TopicRuntimeOptions::default() + }; + + assert!(validate_source_topic_durability("stream", "topic", options).is_ok()); + } + + #[test] + fn given_replicated_topic_when_validating_source_destination_should_reject() { + let options = TopicRuntimeOptions { + durability: Durability::Replicated, + messages_required_to_save: Some(1), + ..TopicRuntimeOptions::default() + }; + + let error = validate_source_topic_durability("stream", "topic", options) + .expect_err("replicated topic must not receive checkpointed source data"); + + assert!(error.to_string().contains("durability=replicated")); + } + + #[test] + fn given_persisted_buffered_topic_when_validating_source_destination_should_accept() { + let options = TopicRuntimeOptions { + durability: Durability::Persisted, + messages_required_to_save: Some(10), + ..TopicRuntimeOptions::default() + }; + + assert!(validate_source_topic_durability("stream", "topic", options).is_ok()); + } + #[test] fn given_partially_committed_send_should_retry_only_failed_tail() { let runtime = tokio::runtime::Runtime::new().expect("failed to create test runtime"); diff --git a/core/connectors/runtime/src/stream.rs b/core/connectors/runtime/src/stream.rs index 4082ffc8ae..83b2dc6687 100644 --- a/core/connectors/runtime/src/stream.rs +++ b/core/connectors/runtime/src/stream.rs @@ -24,6 +24,14 @@ use crate::error::RuntimeError; const TOKEN_FILE_PREFIX: &str = "file:"; +/// `address` must match the suffix of `connection_string`. It is inspected +/// separately because credentials may contain `?`, which does not start the +/// address query string. +fn append_query_parameters(connection_string: &str, address: &str, parameters: &str) -> String { + let separator = if address.contains('?') { '&' } else { '?' }; + format!("{connection_string}{separator}{parameters}") +} + fn expand_home(path: &str) -> PathBuf { if let Some(rest) = path.strip_prefix("~/") { if let Some(home) = dirs::home_dir() { @@ -133,8 +141,10 @@ fn connection_string_with_token( .filter(|domain| !domain.is_empty()) .map(|domain| format!("&tls_domain={domain}")) .unwrap_or_default(); - Ok(format!( - "{connection_string}?tls=true&tls_ca_file={ca_file}{domain}" + Ok(append_query_parameters( + &connection_string, + &config.address, + &format!("tls=true&tls_ca_file={ca_file}{domain}"), )) } else { Ok(connection_string) @@ -192,6 +202,39 @@ mod tests { assert_eq!(result, PathBuf::from("relative/path")); } + #[test] + fn given_existing_query_when_appending_parameters_should_use_ampersand() { + let connection_string = "iggy://user:password@127.0.0.1:8090?reconnection_retries=0"; + let address = "127.0.0.1:8090?reconnection_retries=0"; + + let result = append_query_parameters(connection_string, address, "tls=true"); + + assert_eq!( + result, + "iggy://user:password@127.0.0.1:8090?reconnection_retries=0&tls=true" + ); + } + + #[test] + fn given_no_query_when_appending_parameters_should_use_question_mark() { + let connection_string = "iggy://user:password@127.0.0.1:8090"; + let address = "127.0.0.1:8090"; + + let result = append_query_parameters(connection_string, address, "tls=true"); + + assert_eq!(result, "iggy://user:password@127.0.0.1:8090?tls=true"); + } + + #[test] + fn given_question_mark_in_credentials_should_use_address_separator() { + let connection_string = "iggy://user:pass?word@127.0.0.1:8090"; + let address = "127.0.0.1:8090"; + + let result = append_query_parameters(connection_string, address, "tls=true"); + + assert_eq!(result, "iggy://user:pass?word@127.0.0.1:8090?tls=true"); + } + #[test] fn test_resolve_token_direct_value() { let token = "my-secret-token"; diff --git a/core/connectors/sdk/src/source.rs b/core/connectors/sdk/src/source.rs index a264a56df4..09d18bf673 100644 --- a/core/connectors/sdk/src/source.rs +++ b/core/connectors/sdk/src/source.rs @@ -51,10 +51,12 @@ pub type SendCallback = extern "C" fn( pub type BatchResultCallback = extern "C" fn(plugin_id: u32, batch_id: u64, result: u8) -> i32; -const BATCH_RESULT_TIMEOUT: Duration = Duration::from_secs(30); +/// Maximum time the runtime may take to report a source batch result. +pub const BATCH_RESULT_TIMEOUT: Duration = Duration::from_secs(30); const NACK_RETRY_DELAY: Duration = Duration::from_millis(100); const MAX_NACK_RETRY_DELAY: Duration = Duration::from_secs(5); -const MAX_CONSECUTIVE_NACKS: u32 = 5; +/// Number of consecutive rejected batches after which a source is stopped. +pub const MAX_CONSECUTIVE_NACKS: u32 = 5; /// Delivery result for the single batch currently in flight from a source plugin. #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -82,6 +84,14 @@ impl TryFrom for SourceBatchResult { struct PendingBatch { id: u64, result_sender: oneshot::Sender, + result_received: bool, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum PendingBatchClearResult { + Cleared, + ResultReceived, + Unavailable, } #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -331,6 +341,7 @@ async fn handle_messages( *pending = Some(PendingBatch { id: batch_id, result_sender, + result_received: false, }); } @@ -338,7 +349,9 @@ async fn handle_messages( drop(messages); if callback_result != 0 { - if !clear_pending_batch(&pending_batch, batch_id, plugin_id) { + if clear_pending_batch(&pending_batch, batch_id, plugin_id) + != PendingBatchClearResult::Cleared + { break; } let completion = apply_batch_result( @@ -357,22 +370,30 @@ async fn handle_messages( continue; } + let mut result_receiver = Box::pin(result_receiver); let (completion, shutting_down) = tokio::select! { biased; - result = result_receiver => { + result = &mut result_receiver => { (result.unwrap_or(BatchCompletion::Stop), false) }, _ = shutdown.changed() => { - let completion = if clear_pending_batch(&pending_batch, batch_id, plugin_id) { - apply_batch_result( + let completion = match clear_pending_batch( + &pending_batch, + batch_id, + plugin_id, + ) { + PendingBatchClearResult::Cleared => apply_batch_result( &source, &consecutive_nacks, SourceBatchResult::Nack, plugin_id, policy.max_consecutive_nacks, - ).await - } else { - BatchCompletion::Stop + ).await, + PendingBatchClearResult::ResultReceived + | PendingBatchClearResult::Unavailable => result_receiver + .as_mut() + .await + .unwrap_or(BatchCompletion::Stop), }; (completion, true) }, @@ -380,16 +401,23 @@ async fn handle_messages( warn!( "Timed out waiting for batch result for source connector with ID: {plugin_id}, batch ID: {batch_id}" ); - let completion = if clear_pending_batch(&pending_batch, batch_id, plugin_id) { - apply_batch_result( + let completion = match clear_pending_batch( + &pending_batch, + batch_id, + plugin_id, + ) { + PendingBatchClearResult::Cleared => apply_batch_result( &source, &consecutive_nacks, SourceBatchResult::Nack, plugin_id, policy.max_consecutive_nacks, - ).await - } else { - BatchCompletion::Stop + ).await, + PendingBatchClearResult::ResultReceived + | PendingBatchClearResult::Unavailable => result_receiver + .as_mut() + .await + .unwrap_or(BatchCompletion::Stop), }; (completion, false) } @@ -438,9 +466,9 @@ where } }; - let Some(current) = take_pending_batch(pending_batch, batch_id, plugin_id) else { + if !mark_batch_result_received(pending_batch, batch_id, plugin_id) { return -1; - }; + } let completion = get_runtime().block_on(apply_batch_result( source, @@ -449,6 +477,9 @@ where plugin_id, max_consecutive_nacks, )); + let Some(current) = take_pending_batch(pending_batch, batch_id, plugin_id) else { + return -1; + }; if current.result_sender.send(completion).is_err() { error!( "Failed to deliver batch result for source connector with ID: {plugin_id}, batch ID: {batch_id}" @@ -463,6 +494,34 @@ where } } +fn mark_batch_result_received( + pending_batch: &Mutex>, + batch_id: u64, + plugin_id: u32, +) -> bool { + let mut pending = lock_pending_batch(pending_batch); + let Some(current) = pending.as_mut() else { + error!("No batch is awaiting a result for source connector with ID: {plugin_id}"); + return false; + }; + if current.id != batch_id { + error!( + "Batch result ID mismatch for source connector with ID: {plugin_id}. Expected: {}, received: {batch_id}", + current.id + ); + return false; + } + if current.result_received { + error!( + "Batch result was already received for source connector with ID: {plugin_id}, batch ID: {batch_id}" + ); + return false; + } + + current.result_received = true; + true +} + fn take_pending_batch( pending_batch: &Mutex>, batch_id: u64, @@ -488,18 +547,24 @@ fn clear_pending_batch( pending_batch: &Mutex>, batch_id: u64, plugin_id: u32, -) -> bool { +) -> PendingBatchClearResult { let mut pending = lock_pending_batch(pending_batch); - if let Some(current) = pending.as_ref() - && current.id != batch_id - { + let Some(current) = pending.as_ref() else { + return PendingBatchClearResult::Unavailable; + }; + if current.id != batch_id { error!( "Batch result ID mismatch for source connector with ID: {plugin_id}. Expected: {}, received: {batch_id}", current.id ); - return false; + return PendingBatchClearResult::Unavailable; } - pending.take_if(|current| current.id == batch_id).is_some() + if current.result_received { + return PendingBatchClearResult::ResultReceived; + } + + pending.take(); + PendingBatchClearResult::Cleared } fn lock_pending_batch( @@ -678,6 +743,7 @@ mod tests { polls: AtomicUsize, results: Mutex>, fail_batch_result: AtomicBool, + batch_result_delay: Duration, } #[async_trait::async_trait] @@ -696,6 +762,9 @@ mod tests { } async fn on_batch_result(&self, result: SourceBatchResult) -> Result<(), crate::Error> { + if !self.batch_result_delay.is_zero() { + tokio::time::sleep(self.batch_result_delay).await; + } self.results .lock() .unwrap_or_else(PoisonError::into_inner) @@ -882,6 +951,7 @@ mod tests { *lock_pending_batch(&pending_batch) = Some(PendingBatch { id: 41, result_sender, + result_received: false, }); assert_eq!( @@ -927,6 +997,7 @@ mod tests { *lock_pending_batch(&pending_batch) = Some(PendingBatch { id: 51, result_sender, + result_received: false, }); assert_eq!( @@ -1006,6 +1077,65 @@ mod tests { }); } + #[test] + fn given_result_received_before_timeout_when_hook_finishes_late_should_continue_polling() { + let runtime = tokio::runtime::Runtime::new().expect("failed to create test runtime"); + runtime.block_on(async { + let source = Arc::new(TestSource { + batch_result_delay: Duration::from_millis(200), + ..TestSource::default() + }); + let pending_batch = Arc::new(Mutex::new(None)); + let consecutive_nacks = Arc::new(AtomicU32::new(0)); + let (_shutdown_sender, shutdown_receiver) = watch::channel(()); + let (batch_sender, mut batch_receiver) = mpsc::unbounded_channel(); + let policy = BatchPolicy { + result_timeout: Duration::from_millis(100), + ..test_policy() + }; + + let task = tokio::spawn(handle_messages( + 29, + Arc::clone(&source), + move |_, batch_id, _, _| { + batch_sender + .send(batch_id) + .expect("batch receiver should remain open"); + 0 + }, + shutdown_receiver, + Arc::clone(&pending_batch), + Arc::clone(&consecutive_nacks), + policy, + )); + + let batch_id = batch_receiver + .recv() + .await + .expect("first batch should be sent"); + assert_eq!( + complete_test_batch( + Arc::clone(&pending_batch), + Arc::clone(&source), + Arc::clone(&consecutive_nacks), + batch_id, + SourceBatchResult::Ack, + 29, + ) + .await, + 0 + ); + let next_batch_id = tokio::time::timeout(Duration::from_secs(1), batch_receiver.recv()) + .await + .expect("source did not poll after the received ACK was applied") + .expect("batch channel closed"); + assert_eq!(next_batch_id, 2); + + task.abort(); + let _ = task.await; + }); + } + #[test] fn given_batch_result_handler_failure_should_stop_polling() { let runtime = tokio::runtime::Runtime::new().expect("failed to create test runtime"); diff --git a/core/connectors/sources/README.md b/core/connectors/sources/README.md index eb2d3e99b4..e63f8cb17f 100644 --- a/core/connectors/sources/README.md +++ b/core/connectors/sources/README.md @@ -285,7 +285,7 @@ And before starting the runtime, do not forget to create the specified stream an ```bash iggy --username iggy --password iggy stream create example_stream -iggy --username iggy --password iggy topic create example_stream example_topic 1 none 1d +iggy --username iggy --password iggy topic create example_stream example_topic 1 none 1d --durability persisted ``` And that's all, enjoy using the source connector! diff --git a/core/connectors/sources/postgres_source/Cargo.toml b/core/connectors/sources/postgres_source/Cargo.toml index 0bfb77f65e..0802680fce 100644 --- a/core/connectors/sources/postgres_source/Cargo.toml +++ b/core/connectors/sources/postgres_source/Cargo.toml @@ -57,5 +57,8 @@ tokio = { workspace = true } tracing = { workspace = true } uuid = { workspace = true } +[dev-dependencies] +rmp-serde = { workspace = true } + [lints] workspace = true diff --git a/core/connectors/sources/postgres_source/README.md b/core/connectors/sources/postgres_source/README.md index d0619bdeab..d3f2feb3c4 100644 --- a/core/connectors/sources/postgres_source/README.md +++ b/core/connectors/sources/postgres_source/README.md @@ -12,7 +12,7 @@ The PostgreSQL source connector fetches data from PostgreSQL databases and strea - **Mark as Processed**: Mark rows as processed using a boolean column - **Multiple Tables**: Monitor multiple tables simultaneously - **Batch Processing**: Fetch data in configurable batch sizes -- **Offset Tracking**: Keep track of processed records to avoid duplicates +- **Offset Tracking**: Resume incremental polling from the last acknowledged offset ## Configuration @@ -54,26 +54,45 @@ cdc_backend = "builtin" | `connection_string` | string | required | PostgreSQL connection string | | `mode` | string | required | `polling` or `cdc` | | `tables` | array | required | List of tables to monitor | -| `poll_interval` | string | `1s` | How often to poll (e.g., `1s`, `5m`) | +| `poll_interval` | string | `10s` | How often to poll (e.g., `1s`, `5m`) | | `batch_size` | u32 | `1000` | Max rows per poll | -| `tracking_column` | string | `id` | Column for incremental updates | +| `tracking_column` | string | `id` | Unique, non-null column for incremental updates | | `initial_offset` | string | none | Starting value for tracking column | | `max_connections` | u32 | `10` | Max database connections | | `snake_case_columns` | bool | `false` | Convert column names to snake_case | | `include_metadata` | bool | `true` | Wrap results with metadata | | `payload_column` | string | none | Column to extract as payload | | `payload_format` | string | `bytea` | Format of payload_column: `bytea`, `text`, or `json_direct` | -| `delete_after_read` | bool | `false` | Delete rows after reading | -| `processed_column` | string | none | Boolean column to mark as processed | -| `primary_key_column` | string | tracking_column | PK for delete/mark operations | +| `delete_after_read` | bool | `false` | Delete rows after reading; takes precedence over `processed_column` | +| `processed_column` | string | none | Boolean column to mark as processed when `delete_after_read` is false | +| `primary_key_column` | string | tracking_column | Unique, non-null key for delete/mark operations | | `custom_query` | string | none | Custom SQL with parameter substitution | | `replication_slot` | string | `iggy_slot` | Replication slot name (only used when `mode = "cdc"`) | | `capture_operations` | array | `["INSERT","UPDATE","DELETE"]` | CDC operations to capture | | `cdc_backend` | string | `builtin` | `builtin` or `pg_replicate` | | `verbose_logging` | bool | `false` | Log at info level instead of debug | -| `max_retries` | u32 | `3` | Max retry attempts for transient errors | +| `max_retries` | u32 | `3` | Max attempts for transient errors; `0` and `1` both perform one attempt | | `retry_delay` | string | `1s` | Base delay between retries (e.g., `500ms`, `2s`) | +## Delivery Failures + +Each batch selected by the polling query is delivered at least once, so +consumers must tolerate duplicates. Complete polling capture additionally +requires transactions to become visible in tracking-column order. A transaction +that commits below an already acknowledged cursor is not selected by a later +poll, even when the tracking column is unique. A failed send NACKs the batch and +leaves its database progress uncommitted for redelivery. +After five consecutive NACKs, the source stops and requires a manual connector +restart. + +Cleanup work and replication-slot advances are stored with the acknowledged +checkpoint before they run. After a restart, the connector replays that work +before polling new rows and then saves a state-only checkpoint to retire it. +Checkpoints created by older connector versions remain compatible unless they +contain unfinished row cleanup without the row-version receipt required for +safe replay. In that case, verify the affected rows and clear the connector +state before restarting. + ## Output Modes ### JSON Mode (Default) @@ -199,17 +218,41 @@ Example: ```sql SELECT * FROM $table -WHERE created_at > '$offset' +WHERE created_at > $offset AND (scheduled_at IS NULL OR scheduled_at <= '$now') ORDER BY created_at LIMIT $limit ``` +Custom queries containing `$offset` advance the connector-managed offset after +the batch is acknowledged. The tracking column must be unique and non-null, and +the query must return rows ordered by that column in ascending order. Custom +queries without `$offset` do not advance the connector-managed offset. Cleanup +operations remain bounded by the selected primary keys rather than by the custom +query's cursor. When cleanup is enabled, the query result must include the +resolved cleanup key. The connector joins the result to the configured source +table in the same PostgreSQL snapshot to capture each selected row's version. + +The generated polling query also uses this scalar cursor. At startup, the +connector rejects any generated query or `$offset` custom query whose tracking +column lacks a valid single-column unique index or permits null values. + ## Delete After Read / Mark as Processed +Both options may be present for compatibility. When `delete_after_read` is +`true`, rows are deleted and `processed_column` is ignored. + +The resolved cleanup key is `primary_key_column`, or `tracking_column` when the +former is unset. For every configured table, it must be a non-null column with +a valid single-column unique index. The connector validates this requirement +at startup before enabling delete or mark operations. +The resolved key, tracking column, cleanup action, and selected row versions are +stored in the acknowledged checkpoint. A restart rejects incompatible cleanup +configuration instead of applying old work to a different column or action. + ### Delete After Read -Deletes rows from the source table after successful processing: +Deletes rows from the source table only after Iggy acknowledges the batch: ```toml [plugin_config] @@ -219,7 +262,7 @@ primary_key_column = "id" ### Mark as Processed -Updates a boolean column instead of deleting: +Updates a boolean column after Iggy acknowledges the batch instead of deleting: ```toml [plugin_config] @@ -235,6 +278,18 @@ ALTER TABLE users ADD COLUMN is_processed BOOLEAN DEFAULT false; When `processed_column` is set, the connector automatically adds a `WHERE is_processed = FALSE` filter to the polling query, so only unprocessed rows are fetched. This improves polling efficiency as the table grows. +With the generated polling query, a row whose tracking value moves past the +batch boundary between poll and acknowledgement is left unchanged and returns +in a later poll. Custom queries do not apply this boundary because their result +order is not guaranteed. Cleanup also matches the row version captured by the +poll, so replay cannot delete or mark a replacement row that reused the same +key. + +For generated polling queries, the connector persists the acknowledged offset +before deleting or marking rows. If it stops in between, the rows have been +delivered but may remain unchanged in PostgreSQL. The persisted offset prevents +those rows from being selected again. + ## Supported Column Types The connector handles these PostgreSQL types in JSON mode: @@ -244,7 +299,7 @@ The connector handles these PostgreSQL types in JSON mode: | `BOOL` | boolean | | `INT2`, `INT4`, `INT8` | number | | `FLOAT4`, `FLOAT8` | number | -| `NUMERIC` | number (parsed as f64) | +| `NUMERIC` | string (exact decimal representation) | | `VARCHAR`, `TEXT`, `CHAR` | string | | `TIMESTAMP`, `TIMESTAMPTZ` | string (RFC3339) | | `UUID` | string | @@ -254,7 +309,7 @@ The connector handles these PostgreSQL types in JSON mode: ## CDC Mode -CDC requires PostgreSQL logical replication setup: +CDC requires PostgreSQL 11 or newer and logical replication setup: 1. Set `wal_level = logical` in `postgresql.conf` 2. Restart PostgreSQL @@ -267,6 +322,15 @@ tables = ["users", "orders"] capture_operations = ["INSERT", "UPDATE", "DELETE"] ``` +The connector peeks at logical changes and advances the replication slot only +after Iggy acknowledges the batch. A failed delivery leaves the slot unchanged +so the next poll can read the same changes again. + +Advancing the slot fast-forwards through the WAL range that was just peeked, so +each acknowledged batch is decoded twice. Poll and decode errors do not change +the connector's runtime status. Monitor `confirmed_flush_lsn`, retained WAL, and +replication slot lag in PostgreSQL to detect a stuck CDC poller. + The `pg_replicate` backend requires the `cdc_pg_replicate` feature flag at build time. ### Slot Naming @@ -274,8 +338,9 @@ The `pg_replicate` backend requires the `cdc_pg_replicate` feature flag at build Each CDC connector must use a unique `replication_slot`. Setup accepts any pre-existing `test_decoding` slot, so two connectors pointed at the same database with the default `replication_slot = "iggy_slot"` will silently -share one slot. `pg_logical_slot_get_changes` consumes changes on read, so -each connector only sees a subset of the other's changes instead of erroring. +share one slot. Each connector peeks from and advances the same slot after +delivery, so one connector can move the shared position past changes that the +other has not processed. Set an explicit, distinct `replication_slot` per connector instance. ### Decommissioning @@ -368,7 +433,13 @@ capture_operations = ["INSERT", "UPDATE"] ### Automatic Retries -The connector automatically retries transient database errors (connection issues, deadlocks, serialization failures) with exponential backoff. Configure with `max_retries` (default: 3) and `retry_delay` (default: `1s`). The actual delay is `retry_delay * attempt_number`. Non-transient errors fail immediately. +The connector automatically retries transient database errors (connection issues, deadlocks, serialization failures) with linear backoff. Configure with `max_retries` (default: 3) and `retry_delay` (default: `1s`). The actual delay is `retry_delay * attempt_number`. Non-transient errors fail immediately. + +Polling operations use the complete configured retry schedule. Database work performed after an ACK, such as deleting or marking rows and advancing a replication slot, uses the same schedule but shares a 10-second deadline across the batch. When the configured schedule exceeds that window, unfinished ACK operations remain staged and are retried before the connector polls new rows. + +ACK statements use a transaction-local 9-second server-side timeout so a stalled backend is cancelled before the 10-second ACK callback backstop expires. Polling and CDC reads retain the PostgreSQL connection's configured timeout. + +The connector stops after three consecutive row-cleanup or replication-slot advance failures. This prevents permanent cleanup errors from stalling polling while the connector continues to report healthy progress, and prevents repeated CDC delivery while WAL continues to grow. ### SQL Injection Protection diff --git a/core/connectors/sources/postgres_source/src/lib.rs b/core/connectors/sources/postgres_source/src/lib.rs index 8deb355169..a337fb0d70 100644 --- a/core/connectors/sources/postgres_source/src/lib.rs +++ b/core/connectors/sources/postgres_source/src/lib.rs @@ -15,21 +15,25 @@ // specific language governing permissions and limitations // under the License. +use std::collections::HashMap; +use std::str::FromStr; +use std::sync::atomic::{AtomicU32, Ordering}; +use std::time::Duration; + use async_trait::async_trait; use chrono::{NaiveDate, NaiveDateTime, NaiveTime}; use humantime::Duration as HumanDuration; use iggy_common::{DateTime, Utc}; use iggy_connector_sdk::{ - ConnectorState, Error, ProducedMessage, ProducedMessages, Schema, Source, source_connector, + ConnectorState, Error, ProducedMessage, ProducedMessages, Schema, Source, + source::SourceBatchResult, source_connector, }; use secrecy::{ExposeSecret, SecretString}; use serde::{Deserialize, Serialize}; -use sqlx::postgres::PgPoolOptions; use sqlx::postgres::types::{Oid, PgInterval, PgTimeTz}; -use sqlx::{Column, Pool, Postgres, Row, TypeInfo, ValueRef}; -use std::collections::HashMap; -use std::str::FromStr; -use std::time::Duration; +use sqlx::postgres::{PgConnectOptions, PgPoolOptions, PgValueFormat, PgValueRef}; +use sqlx::types::BigDecimal; +use sqlx::{Column, Pool, Postgres, Row, Transaction, TypeInfo, ValueRef}; use tokio::sync::Mutex; use tracing::{debug, error, info, warn}; use uuid::Uuid; @@ -38,6 +42,16 @@ source_connector!(PostgresSource); const DEFAULT_MAX_RETRIES: u32 = 3; const DEFAULT_RETRY_DELAY: &str = "1s"; +const ACK_BATCH_TIMEOUT: Duration = Duration::from_secs(10); +const ACK_STATEMENT_TIMEOUT: Duration = Duration::from_secs(9); +const _: () = assert!(ACK_STATEMENT_TIMEOUT.as_nanos() < ACK_BATCH_TIMEOUT.as_nanos()); +const MAX_CONSECUTIVE_PROCESS_FAILURES: u32 = 3; +const MAX_CONSECUTIVE_ADVANCE_FAILURES: u32 = 3; +const MIN_CDC_POSTGRES_VERSION_NUM: i32 = 110_000; +const NUMERIC_NAN_SIGN: u16 = 0xC000; +const NUMERIC_POSITIVE_INFINITY_SIGN: u16 = 0xD000; +const NUMERIC_NEGATIVE_INFINITY_SIGN: u16 = 0xF000; +const ROW_VERSION_COLUMN: &str = "__iggy_row_version"; #[derive(Debug)] pub struct PostgresSource { @@ -45,6 +59,9 @@ pub struct PostgresSource { pool: Option>, config: PostgresSourceConfig, state: Mutex, + pending_batch: Mutex>, + consecutive_process_failures: AtomicU32, + consecutive_advance_failures: AtomicU32, verbose: bool, retry_delay: Duration, poll_interval: Duration, @@ -97,11 +114,56 @@ impl PayloadFormat { } } -#[derive(Debug, Serialize, Deserialize)] +#[derive(Debug, Clone, Serialize, Deserialize)] struct State { last_poll_time: DateTime, tracking_offsets: HashMap, processed_rows: u64, + #[serde(default)] + pending_operations: Vec, +} + +#[derive(Debug)] +struct PolledBatch { + messages: Vec, + pending: Option, +} + +#[derive(Debug, Clone)] +struct PendingBatch { + state: State, + acknowledged: bool, + retiring_operations: bool, + operations_checkpointed: bool, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +enum PendingOperation { + ProcessRows { + table: String, + ids: Vec, + tracking_boundary: Option, + #[serde(default)] + target: Option, + #[serde(default)] + row_versions: Vec, + }, + AdvanceReplicationSlot { + lsn: String, + }, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +struct CleanupTarget { + key_column: String, + tracking_column: String, + action: CleanupAction, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +enum CleanupAction { + Delete, + MarkProcessed { column: String }, } #[derive(Debug, Serialize, Deserialize)] @@ -128,6 +190,7 @@ struct ProcessedRow { message: ProducedMessage, max_offset: Option, row_pk: Option, + row_version: Option, } const CONNECTOR_NAME: &str = "PostgreSQL source"; @@ -144,6 +207,18 @@ impl PostgresSource { s.tracking_offsets, s.processed_rows ); }); + let state = restored_state.unwrap_or(State { + last_poll_time: Utc::now(), + tracking_offsets: HashMap::new(), + processed_rows: 0, + pending_operations: Vec::new(), + }); + let pending_batch = (!state.pending_operations.is_empty()).then(|| PendingBatch { + state: state.clone(), + acknowledged: true, + retiring_operations: false, + operations_checkpointed: true, + }); let delay_str = config.retry_delay.as_deref().unwrap_or(DEFAULT_RETRY_DELAY); let retry_delay = HumanDuration::from_str(delay_str) @@ -157,11 +232,10 @@ impl PostgresSource { id, pool: None, config, - state: Mutex::new(restored_state.unwrap_or(State { - last_poll_time: Utc::now(), - tracking_offsets: HashMap::new(), - processed_rows: 0, - })), + state: Mutex::new(state), + pending_batch: Mutex::new(pending_batch), + consecutive_process_failures: AtomicU32::new(0), + consecutive_advance_failures: AtomicU32::new(0), verbose, retry_delay, poll_interval, @@ -181,12 +255,21 @@ impl Source for PostgresSource { self.id, self.config.mode, self.config.tables ); + if self.config.delete_after_read.unwrap_or(false) && self.config.processed_column.is_some() + { + warn!( + "PostgreSQL source connector ID: {} has both delete_after_read and \ + processed_column configured; delete_after_read takes precedence", + self.id + ); + } self.connect().await?; validate_payload_format(self.config.payload_format.as_deref())?; match self.config.mode.as_str() { "cdc" => { + self.validate_cdc_server_version().await?; let backend = validate_cdc_backend(self.config.cdc_backend.as_deref())?; validate_capture_operations(self.config.capture_operations.as_deref())?; self.setup_cdc().await?; @@ -196,6 +279,8 @@ impl Source for PostgresSource { ); } "polling" => { + self.validate_pending_cleanup()?; + self.validate_polling_keys().await?; info!( "PostgreSQL polling mode enabled for connector ID: {}", self.id @@ -218,10 +303,58 @@ impl Source for PostgresSource { } async fn poll(&self) -> Result { - let poll_interval = self.poll_interval; - tokio::time::sleep(poll_interval).await; + let schema = match self.payload_format() { + PayloadFormat::Bytea => Schema::Raw, + PayloadFormat::Text => Schema::Text, + PayloadFormat::JsonDirect | PayloadFormat::Json => Schema::Json, + }; + + { + let pending = self.pending_batch.lock().await; + if let Some(pending) = pending + .as_ref() + .filter(|pending| pending.retiring_operations) + { + let state = self.serialize_state(&pending.state).ok_or_else(|| { + Error::Serialization("failed to serialize PostgreSQL source state".to_string()) + })?; + return Ok(ProducedMessages { + schema, + messages: Vec::new(), + state: Some(state), + }); + } + } - let messages = match self.config.mode.as_str() { + tokio::time::sleep(self.poll_interval).await; + + { + let pending = self.pending_batch.lock().await; + if let Some(pending) = pending.as_ref() { + let state = if pending.retiring_operations { + Some(self.serialize_state(&pending.state).ok_or_else(|| { + Error::Serialization( + "failed to serialize PostgreSQL source state".to_string(), + ) + })?) + } else if pending.acknowledged { + None + } else { + error!( + "PostgreSQL source connector ID: {} was polled while a batch was still in flight", + self.id + ); + return Err(Error::InvalidState); + }; + return Ok(ProducedMessages { + schema, + messages: Vec::new(), + state, + }); + } + } + + let polled = match self.config.mode.as_str() { "polling" => self.poll_tables().await?, "cdc" => self.poll_cdc().await?, _ => { @@ -230,38 +363,94 @@ impl Source for PostgresSource { } }; - let state = self.state.lock().await; + let processed_rows = self.state.lock().await.processed_rows; if self.verbose { info!( "PostgreSQL source connector ID: {} produced {} messages. Total processed: {}", self.id, - messages.len(), - state.processed_rows + polled.messages.len(), + processed_rows ); } else { debug!( "PostgreSQL source connector ID: {} produced {} messages. Total processed: {}", self.id, - messages.len(), - state.processed_rows + polled.messages.len(), + processed_rows ); } - let schema = match self.payload_format() { - PayloadFormat::Bytea => Schema::Raw, - PayloadFormat::Text => Schema::Text, - PayloadFormat::JsonDirect | PayloadFormat::Json => Schema::Json, - }; - - let persisted_state = self.serialize_state(&state); + let persisted_state = polled + .pending + .as_ref() + .filter(|pending| pending.operations_checkpointed) + .map(|pending| { + self.serialize_state(&pending.state).ok_or_else(|| { + Error::Serialization("failed to serialize PostgreSQL source state".to_string()) + }) + }) + .transpose()?; + *self.pending_batch.lock().await = polled.pending; Ok(ProducedMessages { schema, - messages, + messages: polled.messages, state: persisted_state, }) } + async fn on_batch_result(&self, result: SourceBatchResult) -> Result<(), Error> { + if result == SourceBatchResult::Nack { + let mut pending = self.pending_batch.lock().await; + if pending + .as_ref() + .is_some_and(|pending| !pending.acknowledged && !pending.retiring_operations) + { + pending.take(); + } + return Ok(()); + } + + let Some(pending) = self.pending_batch.lock().await.as_ref().cloned() else { + return Ok(()); + }; + + let PendingBatch { + mut state, + operations_checkpointed, + .. + } = pending; + let had_operations = !state.pending_operations.is_empty(); + let operations = std::mem::take(&mut state.pending_operations); + let deadline = tokio::time::Instant::now() + ACK_BATCH_TIMEOUT; + let failed_operations = self.apply_pending_operations(operations, deadline).await?; + + if failed_operations.is_empty() { + *self.state.lock().await = state.clone(); + let mut pending = self.pending_batch.lock().await; + if had_operations && operations_checkpointed { + *pending = Some(PendingBatch { + state, + acknowledged: false, + retiring_operations: true, + operations_checkpointed: false, + }); + } else { + pending.take(); + } + } else { + state.pending_operations = failed_operations; + *self.pending_batch.lock().await = Some(PendingBatch { + state, + acknowledged: true, + retiring_operations: false, + operations_checkpointed, + }); + } + + Ok(()) + } + async fn close(&mut self) -> Result<(), Error> { if let Some(pool) = self.pool.take() { pool.close().await; @@ -284,12 +473,16 @@ impl PostgresSource { async fn connect(&mut self) -> Result<(), Error> { let max_connections = self.config.max_connections.unwrap_or(10); let redacted = redact_connection_string(self.config.connection_string.expose_secret()); + let connect_options = PgConnectOptions::from_str( + self.config.connection_string.expose_secret(), + ) + .map_err(|e| Error::InitError(format!("Invalid PostgreSQL connection string: {e}")))?; info!("Connecting to PostgreSQL with max {max_connections} connections: {redacted}"); let pool = PgPoolOptions::new() .max_connections(max_connections) - .connect(self.config.connection_string.expose_secret()) + .connect_with(connect_options) .await .map_err(|e| Error::InitError(format!("Failed to connect to PostgreSQL: {e}")))?; @@ -303,6 +496,100 @@ impl PostgresSource { Ok(()) } + async fn validate_polling_keys(&self) -> Result<(), Error> { + let tracking_column = self.tracking_column(); + if self.uses_tracking_cursor() { + self.validate_unique_key("Tracking column", tracking_column) + .await?; + } + + if self.should_process_rows() { + let cleanup_key = self.cleanup_key_column(); + if cleanup_key != tracking_column || !self.uses_tracking_cursor() { + self.validate_unique_key("Cleanup key", cleanup_key).await?; + } + } + + Ok(()) + } + + fn validate_pending_cleanup(&mut self) -> Result<(), Error> { + let configured_target = self.cleanup_target(); + for operation in &self.state.get_mut().pending_operations { + let PendingOperation::ProcessRows { + ids, + target, + row_versions, + .. + } = operation + else { + continue; + }; + + if ids.is_empty() { + continue; + } + let Some(target) = target else { + return Err(Error::InitError( + "the restored PostgreSQL cleanup operation predates row-version safety; \ + clear the connector state only after verifying the rows manually" + .to_string(), + )); + }; + if row_versions.len() != ids.len() { + return Err(Error::InitError( + "the restored PostgreSQL cleanup operation has incomplete row-version data" + .to_string(), + )); + } + if configured_target.as_ref() != Some(target) { + return Err(Error::InitError(format!( + "the PostgreSQL cleanup configuration changed while acknowledged cleanup is \ + pending; restore key '{}', tracking column '{}', and action {:?}", + target.key_column, target.tracking_column, target.action + ))); + } + } + Ok(()) + } + + async fn validate_unique_key(&self, key_kind: &str, column: &str) -> Result<(), Error> { + let pool = self.get_pool()?; + for table in &self.config.tables { + let relation = quote_qualified_identifier(table)?; + let is_unique: bool = sqlx::query_scalar( + "SELECT EXISTS (\ + SELECT 1 FROM pg_catalog.pg_index AS idx \ + JOIN pg_catalog.pg_attribute AS attr \ + ON attr.attrelid = idx.indrelid AND attr.attnum = idx.indkey[0] \ + WHERE idx.indrelid = pg_catalog.to_regclass($1) \ + AND idx.indisunique AND idx.indisvalid \ + AND idx.indpred IS NULL AND idx.indexprs IS NULL \ + AND idx.indnkeyatts = 1 \ + AND attr.attname = $2 AND attr.attnotnull\ + )", + ) + .bind(&relation) + .bind(column) + .fetch_one(pool) + .await + .map_err(|e| { + Error::InitError(format!( + "Failed to validate {key_kind} '{column}' for table '{table}': {e}" + )) + })?; + + if !is_unique { + return Err(Error::InitError(format!( + "{key_kind} '{column}' for table '{table}' must be a non-null column \ + backed by a valid single-column unique index" + ))); + } + } + + Ok(()) + } + async fn setup_cdc(&self) -> Result<(), Error> { let pool = self.get_pool()?; @@ -347,11 +634,7 @@ impl PostgresSource { } } - let slot_name = self - .config - .replication_slot - .as_deref() - .unwrap_or("iggy_slot"); + let slot_name = self.replication_slot(); let existing_plugin: Option = sqlx::query_scalar("SELECT plugin FROM pg_replication_slots WHERE slot_name = $1") @@ -389,7 +672,19 @@ impl PostgresSource { Ok(()) } - async fn poll_cdc(&self) -> Result, Error> { + async fn validate_cdc_server_version(&self) -> Result<(), Error> { + let pool = self.get_pool()?; + let server_version_num: i32 = + sqlx::query_scalar("SELECT current_setting('server_version_num')::int") + .fetch_one(pool) + .await + .map_err(|e| { + Error::InitError(format!("Failed to read PostgreSQL server version: {e}")) + })?; + validate_cdc_server_version_num(server_version_num) + } + + async fn poll_cdc(&self) -> Result { match self.config.cdc_backend.as_deref().unwrap_or("builtin") { "builtin" => self.poll_cdc_builtin().await, "pg_replicate" => Err(Error::InitError( @@ -399,14 +694,10 @@ impl PostgresSource { } } - async fn poll_cdc_builtin(&self) -> Result, Error> { + async fn poll_cdc_builtin(&self) -> Result { let pool = self.get_pool()?; - let slot_name = self - .config - .replication_slot - .as_deref() - .unwrap_or("iggy_slot"); + let slot_name = self.replication_slot(); let capture_ops = self .config .capture_operations @@ -417,43 +708,65 @@ impl PostgresSource { (!self.config.tables.is_empty()).then_some(self.config.tables.as_slice()); let batch_size = self.config.batch_size.unwrap_or(1000) as i32; + let pre_peek_lsn = with_retry( + || { + sqlx::query_scalar( + "SELECT CASE WHEN pg_is_in_recovery() \ + THEN pg_last_wal_replay_lsn() \ + ELSE pg_current_wal_flush_lsn() END::text", + ) + .fetch_one(pool) + }, + self.get_max_retries(), + self.retry_delay.as_millis() as u64, + ) + .await + .map_err(|e| Error::Connection(format!("failed to read current WAL position: {e}")))?; + // Database I/O without holding the lock. upto_nchanges is only // checked at transaction-commit boundaries (a single huge transaction // can still exceed it), so this isn't a hard per-call cap - but it // stops the backlog from growing unbounded across many transactions // the way NULL (no limit at all) did. - let rows = - sqlx::query("SELECT lsn, xid, data FROM pg_logical_slot_get_changes($1, NULL, $2)") + let rows = with_retry( + || { + sqlx::query( + "SELECT lsn::text AS lsn, data FROM pg_logical_slot_peek_changes($1, NULL, $2)", + ) .bind(slot_name) .bind(batch_size) .fetch_all(pool) - .await - .map_err(|e| { - error!("Failed to fetch CDC changes: {e}"); + }, + self.get_max_retries(), + self.retry_delay.as_millis() as u64, + ) + .await + .map_err(|e| Error::Connection(format!("failed to fetch CDC changes: {e}")))?; + + let last_lsn: Option = rows + .last() + .map(|row| { + row.try_get("lsn").map_err(|e| { + error!("Failed to read CDC row LSN: {e}"); Error::InvalidRecord - })?; - + }) + }) + .transpose()?; let mut messages = Vec::new(); for row in rows { - let data: String = match row.try_get("data") { - Ok(data) => data, - Err(e) => { - error!("Skipping CDC row with unreadable data column: {e}"); - continue; - } - }; + let data: String = row.try_get("data").map_err(|e| { + error!("Failed to read CDC row data: {e}"); + Error::InvalidRecord + })?; if let Some(change_record) = self.parse_logical_replication_message(&data, &capture_ops, captured_tables) { - let payload = match simd_json::to_vec(&change_record) { - Ok(payload) => payload, - Err(e) => { - error!("Skipping CDC row that failed to serialize: {e}"); - continue; - } - }; + let payload = simd_json::to_vec(&change_record).map_err(|e| { + error!("Failed to serialize CDC row: {e}"); + Error::InvalidRecord + })?; let message = ProducedMessage { id: Some(Uuid::new_v4().as_u128()), @@ -468,31 +781,37 @@ impl PostgresSource { } } - // Update state with minimal lock time - if !messages.is_empty() { - let mut state = self.state.lock().await; - state.processed_rows += messages.len() as u64; - } - if self.verbose { info!("CDC: Fetched {} change records", messages.len()); } else { debug!("CDC: Fetched {} change records", messages.len()); } - Ok(messages) + let lsn = last_lsn.unwrap_or(pre_peek_lsn); + let mut state = self.state.lock().await.clone(); + if !messages.is_empty() { + state.processed_rows += messages.len() as u64; + state.last_poll_time = Utc::now(); + } + state.pending_operations = vec![PendingOperation::AdvanceReplicationSlot { lsn }]; + let pending = Some(PendingBatch { + state, + acknowledged: false, + retiring_operations: false, + operations_checkpointed: !messages.is_empty(), + }); + + Ok(PolledBatch { messages, pending }) } - async fn poll_tables(&self) -> Result, Error> { + async fn poll_tables(&self) -> Result { let pool = self.get_pool()?; let mut messages = Vec::new(); + let mut operations = Vec::new(); + let mut candidate_state = self.state.lock().await.clone(); let batch_size = self.config.batch_size.unwrap_or(1000); - let tracking_column = self.config.tracking_column.as_deref().unwrap_or("id"); - let pk_column = self - .config - .primary_key_column - .as_deref() - .unwrap_or(tracking_column); + let tracking_column = self.tracking_column(); + let pk_column = self.cleanup_key_column(); let row_config = RowProcessingConfig { table: "", @@ -504,8 +823,6 @@ impl PostgresSource { include_metadata: self.config.include_metadata.unwrap_or(true), }; - // Collect state updates to apply after processing - let mut state_updates: Vec<(String, String)> = Vec::new(); let mut total_processed: u64 = 0; for table in &self.config.tables { @@ -514,11 +831,7 @@ impl PostgresSource { ..row_config }; - // Get last offset with minimal lock time - let last_offset = { - let state = self.state.lock().await; - state.tracking_offsets.get(table).cloned() - }; + let last_offset = candidate_state.tracking_offsets.get(table).cloned(); let query = if let Some(custom_query) = &self.config.custom_query { self.validate_custom_query(custom_query)?; @@ -526,23 +839,38 @@ impl PostgresSource { } else { self.build_polling_query(table, tracking_column, &last_offset, batch_size)? }; + let query = if self.should_process_rows() { + self.attach_row_versions(&query, table, pk_column, tracking_column)? + } else { + query + }; // Database I/O without holding the lock let rows = with_retry( - || sqlx::query(sqlx::AssertSqlSafe(query.as_str())).fetch_all(pool), + || { + sqlx::query(sqlx::AssertSqlSafe(query.as_str())) + .persistent(false) + .fetch_all(pool) + }, self.get_max_retries(), self.retry_delay.as_millis() as u64, ) - .await?; + .await + .map_err(|e| { + Error::Connection(format!("failed to poll PostgreSQL table '{table}': {e}")) + })?; let mut max_offset: Option = None; - let mut processed_ids: Vec = Vec::new(); + let mut processed_ids = Vec::with_capacity(rows.len()); + let mut row_versions = Vec::with_capacity(rows.len()); + let mut table_processed = 0; for row in rows { let processed = self.process_row(&row, &table_config)?; - if let Some(pk) = processed.row_pk { + if let (Some(pk), Some(row_version)) = (processed.row_pk, processed.row_version) { processed_ids.push(pk); + row_versions.push(row_version); } if let Some(offset) = processed.max_offset { max_offset = Some(offset); @@ -550,68 +878,253 @@ impl PostgresSource { messages.push(processed.message); total_processed += 1; + table_processed += 1; } - // Database I/O without holding the lock - if !processed_ids.is_empty() { - self.mark_or_delete_processed_rows(pool, table, pk_column, &processed_ids) - .await?; + let cleanup_boundary = self.cleanup_boundary(max_offset.clone()); + let tracking_cursor = self.tracking_cursor(max_offset); + + if self.should_process_rows() && !processed_ids.is_empty() { + operations.push(PendingOperation::ProcessRows { + table: table.clone(), + ids: processed_ids, + tracking_boundary: cleanup_boundary, + target: self.cleanup_target(), + row_versions, + }); } - // Collect offset update for later - if let Some(offset) = max_offset { - state_updates.push((table.clone(), offset)); + if let Some(offset) = tracking_cursor { + candidate_state + .tracking_offsets + .insert(table.clone(), offset); } if self.verbose { - info!("Fetched {} rows from table '{table}'", messages.len()); + info!("Fetched {table_processed} rows from table '{table}'"); } else { - debug!("Fetched {} rows from table '{table}'", messages.len()); + debug!("Fetched {table_processed} rows from table '{table}'"); } } - // Apply all state updates with a single lock acquisition - { - let mut state = self.state.lock().await; - state.processed_rows += total_processed; - for (table, offset) in state_updates { - state.tracking_offsets.insert(table, offset); + let pending = if total_processed > 0 { + candidate_state.processed_rows += total_processed; + candidate_state.last_poll_time = Utc::now(); + candidate_state.pending_operations = operations; + Some(PendingBatch { + state: candidate_state, + acknowledged: false, + retiring_operations: false, + operations_checkpointed: true, + }) + } else { + None + }; + + Ok(PolledBatch { messages, pending }) + } + + async fn advance_replication_slot(&self, lsn: &str) -> Result<(), Error> { + let slot_name = self.replication_slot(); + let pool = self.get_pool()?; + with_retry( + || async { + let mut transaction = begin_ack_transaction(pool).await?; + let result = sqlx::query("SELECT pg_replication_slot_advance($1, $2::pg_lsn)") + .bind(slot_name) + .bind(lsn) + .execute(&mut *transaction) + .await + .map(|_| ()); + let result = finish_ack_transaction(transaction, result).await; + + if let Err(error) = result { + match replication_slot_reached_target(pool, slot_name, lsn).await { + Ok(true) => { + warn!( + "PostgreSQL source connector ID: {} confirmed replication slot \ + '{slot_name}' is already at or beyond {lsn} after advance failed: \ + {error}", + self.id + ); + return Ok(()); + } + Ok(false) => {} + Err(read_error) => warn!( + "PostgreSQL source connector ID: {} failed to read replication slot \ + '{slot_name}' after advance failed: {read_error}", + self.id + ), + } + return Err(error); + } + + Ok(()) + }, + self.get_max_retries(), + self.retry_delay.as_millis() as u64, + ) + .await + .map_err(|e| { + Error::Connection(format!( + "failed to advance replication slot '{slot_name}' to {lsn}: {e}" + )) + })?; + Ok(()) + } + + async fn apply_pending_operations( + &self, + operations: Vec, + deadline: tokio::time::Instant, + ) -> Result, Error> { + let mut failed_operations = Vec::new(); + let mut operations = operations.into_iter(); + while let Some(operation) = operations.next() { + if tokio::time::Instant::now() >= deadline { + failed_operations.push(operation); + failed_operations.extend(operations); + break; + } + if !self.apply_pending_operation(&operation, deadline).await? { + failed_operations.push(operation); } - state.last_poll_time = Utc::now(); } + Ok(failed_operations) + } - Ok(messages) + async fn apply_pending_operation( + &self, + operation: &PendingOperation, + deadline: tokio::time::Instant, + ) -> Result { + let result = tokio::time::timeout_at(deadline, async { + match operation { + PendingOperation::ProcessRows { + table, + ids, + tracking_boundary, + target, + row_versions, + } => { + if ids.is_empty() { + return Ok(()); + } + let target = target.as_ref().ok_or(Error::InvalidState)?; + let pool = self.get_pool()?; + self.mark_or_delete_processed_rows( + pool, + table, + ids, + row_versions, + target, + tracking_boundary.as_deref(), + ) + .await + } + PendingOperation::AdvanceReplicationSlot { lsn } => { + self.advance_replication_slot(lsn).await + } + } + }) + .await; + + match result { + Ok(Ok(())) => { + match operation { + PendingOperation::ProcessRows { .. } => { + self.consecutive_process_failures + .store(0, Ordering::Relaxed); + } + PendingOperation::AdvanceReplicationSlot { .. } => { + self.consecutive_advance_failures + .store(0, Ordering::Relaxed); + } + } + Ok(true) + } + Ok(Err(error)) => { + self.record_pending_operation_failure(operation, &error)?; + Ok(false) + } + Err(_) => { + self.record_pending_operation_failure( + operation, + "operation exceeded the shared 10s ACK batch budget", + )?; + Ok(false) + } + } + } + + fn record_pending_operation_failure( + &self, + operation: &PendingOperation, + error: impl std::fmt::Display, + ) -> Result<(), Error> { + match operation { + PendingOperation::ProcessRows { .. } => { + let consecutive_failures = self + .consecutive_process_failures + .fetch_add(1, Ordering::Relaxed) + .saturating_add(1); + error!( + "Failed to process rows for PostgreSQL source connector ID: {}. \ + Consecutive row processing failures: {consecutive_failures}. {error}", + self.id, + ); + if consecutive_failures >= MAX_CONSECUTIVE_PROCESS_FAILURES { + return Err(Error::Connection(format!( + "stopping PostgreSQL source connector ID {} after {consecutive_failures} \ + consecutive row processing failures", + self.id + ))); + } + } + PendingOperation::AdvanceReplicationSlot { .. } => { + let consecutive_failures = self + .consecutive_advance_failures + .fetch_add(1, Ordering::Relaxed) + .saturating_add(1); + error!( + "Failed to advance replication slot for PostgreSQL source connector ID: {}. \ + Consecutive replication slot advance failures: {consecutive_failures}. \ + {error}", + self.id + ); + if consecutive_failures >= MAX_CONSECUTIVE_ADVANCE_FAILURES { + return Err(Error::Connection(format!( + "stopping PostgreSQL source connector ID {} after {consecutive_failures} \ + consecutive replication slot advance failures", + self.id + ))); + } + } + } + Ok(()) } async fn mark_or_delete_processed_rows( &self, pool: &Pool, table: &str, - pk_column: &str, ids: &[String], + row_versions: &[String], + target: &CleanupTarget, + tracking_boundary: Option<&str>, ) -> Result<(), Error> { if ids.is_empty() { return Ok(()); } let quoted_table = quote_qualified_identifier(table)?; - let quoted_pk = quote_identifier(pk_column)?; + let row_condition = build_row_version_condition(&target.key_column, ids, row_versions)?; + let tracking_condition = + build_tracking_condition(&target.tracking_column, tracking_boundary)?; - let ids_list = ids - .iter() - .map(|id| { - if id.parse::().is_ok() { - id.clone() - } else { - format!("'{}'", id.replace('\'', "''")) - } - }) - .collect::>() - .join(", "); - - if self.config.delete_after_read.unwrap_or(false) { + if target.action == CleanupAction::Delete { let delete_query = - format!("DELETE FROM {quoted_table} WHERE {quoted_pk} IN ({ids_list})"); + format!("DELETE FROM {quoted_table} WHERE ({row_condition}){tracking_condition}"); if self.verbose { info!("Deleting {} processed rows from '{table}'", ids.len()); @@ -619,17 +1132,18 @@ impl PostgresSource { debug!("Deleting {} processed rows from '{table}'", ids.len()); } - sqlx::query(sqlx::AssertSqlSafe(delete_query)) - .execute(pool) - .await - .map_err(|e| { - error!("Failed to delete processed rows: {e}"); - Error::InvalidRecord - })?; - } else if let Some(processed_col) = &self.config.processed_column { - let quoted_processed = quote_identifier(processed_col)?; + with_retry( + || execute_ack_statement(pool, delete_query.as_str()), + self.get_max_retries(), + self.retry_delay.as_millis() as u64, + ) + .await + .map_err(|e| Error::Connection(format!("failed to delete processed rows: {e}")))?; + } else if let CleanupAction::MarkProcessed { column } = &target.action { + let quoted_processed = quote_identifier(column)?; let update_query = format!( - "UPDATE {quoted_table} SET {quoted_processed} = TRUE WHERE {quoted_pk} IN ({ids_list})" + "UPDATE {quoted_table} SET {quoted_processed} = TRUE \ + WHERE ({row_condition}){tracking_condition}" ); if self.verbose { @@ -638,13 +1152,13 @@ impl PostgresSource { debug!("Marking {} rows as processed in '{table}'", ids.len()); } - sqlx::query(sqlx::AssertSqlSafe(update_query)) - .execute(pool) - .await - .map_err(|e| { - error!("Failed to mark rows as processed: {e}"); - Error::InvalidRecord - })?; + with_retry( + || execute_ack_statement(pool, update_query.as_str()), + self.get_max_retries(), + self.retry_delay.as_millis() as u64, + ) + .await + .map_err(|e| Error::Connection(format!("failed to mark rows as processed: {e}")))?; } Ok(()) @@ -669,6 +1183,66 @@ impl PostgresSource { self.config.max_retries.unwrap_or(DEFAULT_MAX_RETRIES) } + fn should_process_rows(&self) -> bool { + self.config.delete_after_read.unwrap_or(false) || self.config.processed_column.is_some() + } + + fn cleanup_target(&self) -> Option { + let action = if self.config.delete_after_read.unwrap_or(false) { + CleanupAction::Delete + } else { + CleanupAction::MarkProcessed { + column: self.config.processed_column.clone()?, + } + }; + Some(CleanupTarget { + key_column: self.cleanup_key_column().to_string(), + tracking_column: self.tracking_column().to_string(), + action, + }) + } + + fn cleanup_boundary(&self, max_offset: Option) -> Option { + if self.config.custom_query.is_some() { + None + } else { + max_offset + } + } + + fn tracking_cursor(&self, max_offset: Option) -> Option { + if self.uses_tracking_cursor() { + max_offset + } else { + None + } + } + + fn uses_tracking_cursor(&self) -> bool { + self.config + .custom_query + .as_ref() + .is_none_or(|query| query.contains("$offset")) + } + + fn tracking_column(&self) -> &str { + self.config.tracking_column.as_deref().unwrap_or("id") + } + + fn cleanup_key_column(&self) -> &str { + self.config + .primary_key_column + .as_deref() + .unwrap_or_else(|| self.tracking_column()) + } + + fn replication_slot(&self) -> &str { + self.config + .replication_slot + .as_deref() + .unwrap_or("iggy_slot") + } + fn build_polling_query( &self, table: &str, @@ -714,6 +1288,31 @@ impl PostgresSource { )) } + fn attach_row_versions( + &self, + query: &str, + table: &str, + key_column: &str, + tracking_column: &str, + ) -> Result { + let quoted_table = quote_qualified_identifier(table)?; + let quoted_key = quote_identifier(key_column)?; + let quoted_tracking = quote_identifier(tracking_column)?; + let query = query.trim().trim_end_matches(';'); + let order_clause = if self.uses_tracking_cursor() { + format!(" ORDER BY __iggy_rows.{quoted_tracking} ASC") + } else { + String::new() + }; + + Ok(format!( + "SELECT __iggy_rows.*, __iggy_source.xmin::text AS \"{ROW_VERSION_COLUMN}\" \ + FROM ({query}) AS __iggy_rows \ + JOIN {quoted_table} AS __iggy_source \ + ON __iggy_source.{quoted_key} = __iggy_rows.{quoted_key}{order_clause}" + )) + } + fn validate_custom_query(&self, query: &str) -> Result<(), Error> { let query_upper = query.to_uppercase(); if !query_upper.contains("SELECT") { @@ -733,18 +1332,25 @@ impl PostgresSource { batch_size: u32, ) -> String { let offset_value = last_offset - .clone() - .or_else(|| self.config.initial_offset.clone()) + .as_deref() + .or(self.config.initial_offset.as_deref()) .unwrap_or_default(); + let formatted_offset = format_offset_value(offset_value); let now = Utc::now(); + let batch_size = batch_size.to_string(); + let now_unix = now.timestamp().to_string(); + let now = now.to_rfc3339(); + let replacements = [ + ("$table", table), + ("'$offset'", formatted_offset.as_str()), + ("$offset", formatted_offset.as_str()), + ("$limit", batch_size.as_str()), + ("$now_unix", now_unix.as_str()), + ("$now", now.as_str()), + ]; - query - .replace("$table", table) - .replace("$offset", &offset_value) - .replace("$limit", &batch_size.to_string()) - .replace("$now", &now.to_rfc3339()) - .replace("$now_unix", &now.timestamp().to_string()) + substitute_query_tokens(query, &replacements) } fn parse_logical_replication_message( @@ -816,11 +1422,21 @@ impl PostgresSource { config: &RowProcessingConfig, ) -> Result { let mut row_pk: Option = None; + let mut row_version: Option = None; let mut max_offset: Option = None; let mut extracted_payload: Option> = None; let mut data = serde_json::Map::new(); for (i, column) in row.columns().iter().enumerate() { + if column.name() == ROW_VERSION_COLUMN { + row_version = Some(row.try_get(i).map_err(|error| { + Error::InvalidRecordValue(format!( + "failed to decode PostgreSQL row version: {error}" + )) + })?); + continue; + } + let column_name = if config.snake_case_columns { to_snake_case(column.name()) } else { @@ -834,23 +1450,23 @@ impl PostgresSource { } let value = extract_column_value(row, i)?; - data.insert(column_name.clone(), value.clone()); - if column.name() == config.tracking_column { - if let serde_json::Value::String(ref s) = value { - max_offset = Some(s.clone()); - } else if let serde_json::Value::Number(ref n) = value { - max_offset = Some(n.to_string()); - } + max_offset = extract_tracking_value(&value); } if column.name() == config.pk_column { - if let serde_json::Value::String(ref s) = value { - row_pk = Some(s.clone()); - } else if let serde_json::Value::Number(ref n) = value { - row_pk = Some(n.to_string()); - } + row_pk = extract_tracking_value(&value); } + + data.insert(column_name, value); + } + + if self.should_process_rows() && (row_pk.is_none() || row_version.is_none()) { + return Err(Error::InvalidRecordValue(format!( + "polling cleanup requires query results to include key column '{}' and the \ + source row version", + config.pk_column + ))); } let payload = if let Some(bytes) = extracted_payload { @@ -891,6 +1507,7 @@ impl PostgresSource { message, max_offset, row_pk, + row_version, }) } @@ -994,15 +1611,7 @@ fn extract_column_value( .map(serde_json::Value::from) .unwrap_or(serde_json::Value::Null)) } - "NUMERIC" => { - let value: Option = row - .try_get(column_index) - .map_err(|_| Error::InvalidRecord)?; - Ok(value - .and_then(|s| s.parse::().ok()) - .map(serde_json::Value::from) - .unwrap_or(serde_json::Value::Null)) - } + "NUMERIC" => extract_numeric_value(row, column_index), "VARCHAR" | "TEXT" | "CHAR" | "NAME" | "BPCHAR" => { let value: Option = row .try_get(column_index) @@ -1375,6 +1984,49 @@ fn extract_column_value( } } +fn extract_numeric_value( + row: &sqlx::postgres::PgRow, + column_index: usize, +) -> Result { + let raw = row + .try_get_raw(column_index) + .map_err(|_| Error::InvalidRecord)?; + if raw.is_null() { + return Ok(serde_json::Value::Null); + } + if let Some(value) = extract_special_numeric_value(raw)? { + return Ok(serde_json::Value::String(value.to_string())); + } + + let value: BigDecimal = row + .try_get(column_index) + .map_err(|_| Error::InvalidRecord)?; + Ok(serde_json::Value::String(value.normalized().to_string())) +} + +fn extract_special_numeric_value(raw: PgValueRef<'_>) -> Result, Error> { + match raw.format() { + PgValueFormat::Text => match raw.as_str().map_err(|_| Error::InvalidRecord)? { + "NaN" => Ok(Some("NaN")), + "Infinity" => Ok(Some("Infinity")), + "-Infinity" => Ok(Some("-Infinity")), + _ => Ok(None), + }, + PgValueFormat::Binary => { + let bytes = raw.as_bytes().map_err(|_| Error::InvalidRecord)?; + if bytes.len() < 6 { + return Ok(None); + } + match u16::from_be_bytes([bytes[4], bytes[5]]) { + NUMERIC_NAN_SIGN => Ok(Some("NaN")), + NUMERIC_POSITIVE_INFINITY_SIGN => Ok(Some("Infinity")), + NUMERIC_NEGATIVE_INFINITY_SIGN => Ok(Some("-Infinity")), + _ => Ok(None), + } + } + } +} + fn format_pg_interval(interval: &PgInterval) -> String { let mut parts = Vec::new(); @@ -1453,10 +2105,138 @@ fn quote_qualified_identifier(name: &str) -> Result { } fn format_offset_value(value: &str) -> String { - if value.parse::().is_ok() || value.parse::().is_ok() { - value.to_string() - } else { - format!("'{}'", value.replace('\'', "''")) + // Let PostgreSQL infer the literal type from the compared column. + format!("'{}'", value.replace('\'', "''")) +} + +fn build_row_version_condition( + key_column: &str, + ids: &[String], + row_versions: &[String], +) -> Result { + if ids.len() != row_versions.len() { + return Err(Error::InvalidState); + } + let quoted_key = quote_identifier(key_column)?; + Ok(ids + .iter() + .zip(row_versions) + .map(|(id, row_version)| { + format!( + "({quoted_key} = {} AND xmin = {}::xid)", + format_offset_value(id), + format_offset_value(row_version) + ) + }) + .collect::>() + .join(" OR ")) +} + +fn substitute_query_tokens(query: &str, replacements: &[(&str, &str)]) -> String { + let mut result = String::with_capacity(query.len()); + let mut remaining = query; + + loop { + let mut next_replacement: Option<(usize, &str, &str)> = None; + for &(token, replacement) in replacements { + let Some(index) = remaining.find(token) else { + continue; + }; + let should_replace = match next_replacement { + Some((next_index, next_token, _)) => { + index < next_index || (index == next_index && token.len() > next_token.len()) + } + None => true, + }; + if should_replace { + next_replacement = Some((index, token, replacement)); + } + } + + let Some((index, token, replacement)) = next_replacement else { + result.push_str(remaining); + return result; + }; + result.push_str(&remaining[..index]); + result.push_str(replacement); + remaining = &remaining[index + token.len()..]; + } +} + +fn build_tracking_condition( + tracking_column: &str, + tracking_boundary: Option<&str>, +) -> Result { + let Some(boundary) = tracking_boundary else { + return Ok(String::new()); + }; + let quoted_tracking = quote_identifier(tracking_column)?; + Ok(format!( + " AND ({quoted_tracking} <= {} OR {quoted_tracking} IS NULL)", + format_offset_value(boundary) + )) +} + +async fn replication_slot_reached_target( + pool: &Pool, + slot_name: &str, + target_lsn: &str, +) -> Result { + let mut transaction = begin_ack_transaction(pool).await?; + let result = sqlx::query_scalar::<_, bool>( + "SELECT COALESCE(confirmed_flush_lsn >= $2::pg_lsn, FALSE) \ + FROM pg_replication_slots WHERE slot_name = $1", + ) + .bind(slot_name) + .bind(target_lsn) + .fetch_optional(&mut *transaction) + .await + .map(|reached| reached.unwrap_or(false)); + finish_ack_transaction(transaction, result).await +} + +async fn begin_ack_transaction( + pool: &Pool, +) -> Result, sqlx::Error> { + let mut transaction = pool.begin().await?; + sqlx::query("SELECT set_config('statement_timeout', $1, true)") + .bind(format!("{}ms", ACK_STATEMENT_TIMEOUT.as_millis())) + .execute(&mut *transaction) + .await?; + Ok(transaction) +} + +async fn finish_ack_transaction( + transaction: Transaction<'_, Postgres>, + result: Result, +) -> Result { + match result { + Ok(value) => { + transaction.commit().await?; + Ok(value) + } + Err(error) => { + transaction.rollback().await?; + Err(error) + } + } +} + +async fn execute_ack_statement(pool: &Pool, statement: &str) -> Result<(), sqlx::Error> { + let mut transaction = begin_ack_transaction(pool).await?; + let result = sqlx::query(sqlx::AssertSqlSafe(statement)) + .persistent(false) + .execute(&mut *transaction) + .await + .map(|_| ()); + finish_ack_transaction(transaction, result).await +} + +fn extract_tracking_value(value: &serde_json::Value) -> Option { + match value { + serde_json::Value::String(value) => Some(value.clone()), + serde_json::Value::Number(value) => Some(value.to_string()), + _ => None, } } @@ -1506,6 +2286,16 @@ fn validate_cdc_backend(cdc_backend: Option<&str>) -> Result<&str, Error> { } } +fn validate_cdc_server_version_num(server_version_num: i32) -> Result<(), Error> { + if server_version_num < MIN_CDC_POSTGRES_VERSION_NUM { + return Err(Error::InitError(format!( + "PostgreSQL CDC requires PostgreSQL 11 or newer; server_version_num is \ + {server_version_num}" + ))); + } + Ok(()) +} + fn validate_capture_operations(capture_operations: Option<&[String]>) -> Result<(), Error> { const VALID: [&str; 3] = ["INSERT", "UPDATE", "DELETE"]; let Some(ops) = capture_operations else { @@ -1647,7 +2437,11 @@ fn parse_bare_scalar(token: &str) -> serde_json::Value { } } -async fn with_retry(operation: F, max_retries: u32, delay_ms: u64) -> Result +async fn with_retry( + operation: F, + max_retries: u32, + delay_ms: u64, +) -> Result where F: Fn() -> Fut, Fut: std::future::Future>, @@ -1660,7 +2454,7 @@ where attempts += 1; if attempts >= max_retries || !is_transient_error(&e) { error!("Database operation failed after {attempts} attempts: {e}"); - return Err(Error::InvalidRecord); + return Err(e); } warn!( "Transient database error (attempt {attempts}/{max_retries}): {e}. Retrying in {delay_ms}ms..." @@ -1677,16 +2471,20 @@ fn is_transient_error(e: &sqlx::Error) -> bool { sqlx::Error::PoolTimedOut => true, sqlx::Error::PoolClosed => false, sqlx::Error::Protocol(_) => false, - sqlx::Error::Database(db_err) => db_err.code().is_some_and(|code| { - matches!( - code.as_ref(), - "40001" | "40P01" | "57P01" | "57P02" | "57P03" | "08000" | "08003" | "08006" - ) - }), + sqlx::Error::Database(db_err) => db_err + .code() + .is_some_and(|code| is_transient_sqlstate(code.as_ref())), _ => false, } } +fn is_transient_sqlstate(code: &str) -> bool { + matches!( + code, + "40001" | "40P01" | "55006" | "57P01" | "57P02" | "57P03" | "08000" | "08003" | "08006" + ) +} + fn redact_connection_string(conn_str: &str) -> String { if let Some(scheme_end) = conn_str.find("://") { let scheme = &conn_str[..scheme_end + 3]; @@ -1704,6 +2502,8 @@ mod cdc_fixtures; #[cfg(test)] mod tests { + use std::sync::atomic::{AtomicU32, Ordering}; + use super::*; fn test_config() -> PostgresSourceConfig { @@ -1734,51 +2534,167 @@ mod tests { } #[test] - fn given_last_offset_polling_query_should_be_built() { + fn given_last_offset_polling_query_should_be_built() { + let src = PostgresSource::new(1, test_config(), None); + let query = src + .build_polling_query("users", "updated_at", &Some("2024-01-01".to_string()), 500) + .expect("Failed to build query"); + assert_eq!( + query, + "SELECT * FROM \"users\" WHERE \"updated_at\" > '2024-01-01' ORDER BY \"updated_at\" ASC LIMIT 500" + ); + } + + #[test] + fn given_initial_offset_polling_query_should_be_built() { + let mut config = test_config(); + config.tracking_column = Some("id".to_string()); + config.initial_offset = Some("100".to_string()); + let src = PostgresSource::new(1, config, None); + let query = src + .build_polling_query("users", "id", &None, 1000) + .expect("Failed to build query"); + assert_eq!( + query, + "SELECT * FROM \"users\" WHERE \"id\" > '100' ORDER BY \"id\" ASC LIMIT 1000" + ); + } + + #[test] + fn given_processed_column_polling_query_should_include_filter() { + let mut config = test_config(); + config.processed_column = Some("is_processed".to_string()); + let src = PostgresSource::new(1, config, None); + let query = src + .build_polling_query("events", "id", &None, 100) + .expect("Failed to build query"); + assert!(query.contains("\"is_processed\" = FALSE")); + } + + #[test] + fn given_delete_and_processed_column_should_include_processed_filter() { + let mut config = test_config(); + config.delete_after_read = Some(true); + config.processed_column = Some("is_processed".to_string()); + let src = PostgresSource::new(1, config, None); + + let query = src + .build_polling_query("events", "id", &None, 100) + .expect("Failed to build query"); + + assert!(query.contains("\"is_processed\" = FALSE")); + } + + #[test] + fn given_numeric_offset_should_allow_database_type_inference() { let src = PostgresSource::new(1, test_config(), None); let query = src - .build_polling_query("users", "updated_at", &Some("2024-01-01".to_string()), 500) + .build_polling_query("users", "id", &Some("42".to_string()), 100) .expect("Failed to build query"); + assert!(query.contains("\"id\" > '42'")); + } + + #[test] + fn given_exact_numeric_boundary_should_preserve_text_and_include_null_rows() { + let condition = build_tracking_condition("offset", Some("9007199254740993.25")) + .expect("Failed to build tracking condition"); + assert_eq!( - query, - "SELECT * FROM \"users\" WHERE \"updated_at\" > '2024-01-01' ORDER BY \"updated_at\" ASC LIMIT 500" + condition, + " AND (\"offset\" <= '9007199254740993.25' OR \"offset\" IS NULL)" ); } #[test] - fn given_initial_offset_polling_query_should_be_built() { + fn given_non_finite_offsets_should_quote_values() { + for value in ["NaN", "inf", "-infinity"] { + assert_eq!(format_offset_value(value), format!("'{value}'")); + } + } + + #[test] + fn given_cleanup_receipts_should_match_keys_and_row_versions() { + let ids = vec!["100".to_string(), "O'Reilly".to_string()]; + let row_versions = vec!["42".to_string(), "43".to_string()]; + + assert_eq!( + build_row_version_condition("id", &ids, &row_versions) + .expect("Failed to build row-version condition"), + "(\"id\" = '100' AND xmin = '42'::xid) OR \ + (\"id\" = 'O''Reilly' AND xmin = '43'::xid)" + ); + } + + #[test] + fn given_cleanup_query_should_capture_row_versions_in_the_same_snapshot() { let mut config = test_config(); - config.tracking_column = Some("id".to_string()); - config.initial_offset = Some("100".to_string()); + config.delete_after_read = Some(true); + config.primary_key_column = Some("id".to_string()); let src = PostgresSource::new(1, config, None); + let query = src - .build_polling_query("users", "id", &None, 1000) - .expect("Failed to build query"); + .attach_row_versions( + "SELECT * FROM \"users\" ORDER BY \"updated_at\" ASC LIMIT 500", + "users", + "id", + "updated_at", + ) + .expect("Failed to attach row versions"); + + assert!(query.contains("__iggy_source.xmin::text AS \"__iggy_row_version\"")); + assert!(query.contains( + "ON __iggy_source.\"id\" = __iggy_rows.\"id\" ORDER BY \ + __iggy_rows.\"updated_at\" ASC" + )); + } + + #[test] + fn given_no_tracking_boundary_should_not_add_tracking_condition() { + let condition = + build_tracking_condition("offset", None).expect("Failed to build tracking condition"); + + assert!(condition.is_empty()); + } + + #[test] + fn given_custom_query_should_not_apply_last_row_as_cleanup_boundary() { + let mut config = test_config(); + config.custom_query = Some("SELECT id FROM users".to_string()); + let src = PostgresSource::new(1, config, None); + + assert_eq!(src.cleanup_boundary(Some("42".to_string())), None); + } + + #[test] + fn given_generated_query_should_use_last_row_as_cleanup_boundary() { + let src = PostgresSource::new(1, test_config(), None); + assert_eq!( - query, - "SELECT * FROM \"users\" WHERE \"id\" > 100 ORDER BY \"id\" ASC LIMIT 1000" + src.cleanup_boundary(Some("42".to_string())), + Some("42".to_string()) ); } #[test] - fn given_processed_column_polling_query_should_include_filter() { + fn given_custom_query_with_offset_should_advance_tracking_cursor() { let mut config = test_config(); - config.processed_column = Some("is_processed".to_string()); + config.custom_query = + Some("SELECT * FROM users WHERE id > $offset ORDER BY id ASC LIMIT $limit".to_string()); let src = PostgresSource::new(1, config, None); - let query = src - .build_polling_query("events", "id", &None, 100) - .expect("Failed to build query"); - assert!(query.contains("\"is_processed\" = FALSE")); + + assert_eq!( + src.tracking_cursor(Some("42".to_string())), + Some("42".to_string()) + ); } #[test] - fn given_numeric_offset_should_not_quote_value() { - let src = PostgresSource::new(1, test_config(), None); - let query = src - .build_polling_query("users", "id", &Some("42".to_string()), 100) - .expect("Failed to build query"); - assert!(query.contains("\"id\" > 42")); - assert!(!query.contains("'42'")); + fn given_custom_query_without_offset_should_not_advance_tracking_cursor() { + let mut config = test_config(); + config.custom_query = Some("SELECT * FROM users ORDER BY id ASC".to_string()); + let src = PostgresSource::new(1, config, None); + + assert_eq!(src.tracking_cursor(Some("42".to_string())), None); } #[test] @@ -1979,6 +2895,19 @@ mod tests { assert!(validate_capture_operations(None).is_ok()); } + #[test] + fn given_postgres_10_should_fail_cdc_version_validation() { + let error = validate_cdc_server_version_num(100_000).unwrap_err(); + + assert!(matches!(error, Error::InitError(_))); + } + + #[test] + fn given_postgres_11_or_newer_should_pass_cdc_version_validation() { + assert!(validate_cdc_server_version_num(110_000).is_ok()); + assert!(validate_cdc_server_version_num(170_000).is_ok()); + } + #[test] fn given_unsupported_payload_format_should_fail_validation() { let err = validate_payload_format(Some("btea")).unwrap_err(); @@ -2433,19 +3362,64 @@ mod tests { let result = src.substitute_query_params(query, "events", &Some("100".to_string()), 50); assert!(result.contains("FROM events")); - assert!(result.contains("id > 100")); + assert!(result.contains("id > '100'")); assert!(result.contains("LIMIT 50")); } + #[test] + fn given_text_offset_should_escape_it_as_a_sql_literal() { + let src = PostgresSource::new(1, test_config(), None); + + let query = "SELECT * FROM events WHERE name > $offset"; + let result = + src.substitute_query_params(query, "events", &Some("O'Reilly".to_string()), 50); + + assert_eq!(result, "SELECT * FROM events WHERE name > 'O''Reilly'"); + } + + #[test] + fn given_quoted_text_offset_should_not_double_quote_the_literal() { + let src = PostgresSource::new(1, test_config(), None); + + let query = "SELECT * FROM events WHERE name > '$offset'"; + let result = + src.substitute_query_params(query, "events", &Some("O'Reilly".to_string()), 50); + + assert_eq!(result, "SELECT * FROM events WHERE name > 'O''Reilly'"); + } + + #[test] + fn given_offset_containing_placeholder_should_substitute_original_token_once() { + let src = PostgresSource::new(1, test_config(), None); + + let query = "SELECT * FROM events WHERE name > '$offset'"; + let result = src.substitute_query_params( + query, + "events", + &Some("$offset' OR TRUE --".to_string()), + 50, + ); + + assert_eq!( + result, + "SELECT * FROM events WHERE name > '$offset'' OR TRUE --'" + ); + } + #[test] fn given_custom_query_with_time_params_should_substitute_correctly() { let src = PostgresSource::new(1, test_config(), None); - let query = "SELECT * FROM $table WHERE created_at < '$now'"; + let query = "SELECT * FROM $table WHERE created_at < '$now' AND epoch < $now_unix"; let result = src.substitute_query_params(query, "logs", &None, 100); assert!(result.contains("FROM logs")); assert!(!result.contains("$now")); + let unix_value = result + .rsplit_once("epoch < ") + .map(|(_, value)| value) + .unwrap(); + assert!(unix_value.parse::().is_ok()); } #[test] @@ -2457,7 +3431,7 @@ mod tests { let query = "SELECT * FROM $table WHERE id > $offset"; let result = src.substitute_query_params(query, "data", &None, 100); - assert!(result.contains("id > 500")); + assert!(result.contains("id > '500'")); } #[test] @@ -2481,8 +3455,8 @@ mod tests { assert_eq!(redacted, "postgresql://adm***"); } - #[test] - fn given_persisted_state_should_restore_tracking_offsets() { + #[tokio::test] + async fn given_persisted_state_should_restore_tracking_offsets() { let state = State { last_poll_time: Utc::now(), tracking_offsets: HashMap::from([ @@ -2490,6 +3464,7 @@ mod tests { ("orders".to_string(), "2024-01-15T10:30:00Z".to_string()), ]), processed_rows: 500, + pending_operations: Vec::new(), }; let connector_state = @@ -2497,44 +3472,538 @@ mod tests { let src = PostgresSource::new(1, test_config(), Some(connector_state)); - let runtime = tokio::runtime::Runtime::new().unwrap(); - runtime.block_on(async { - let restored = src.state.lock().await; - assert_eq!( - restored.tracking_offsets.get("users"), - Some(&"100".to_string()) - ); - assert_eq!( - restored.tracking_offsets.get("orders"), - Some(&"2024-01-15T10:30:00Z".to_string()) + let restored = src.state.lock().await; + assert_eq!( + restored.tracking_offsets.get("users"), + Some(&"100".to_string()) + ); + assert_eq!( + restored.tracking_offsets.get("orders"), + Some(&"2024-01-15T10:30:00Z".to_string()) + ); + assert_eq!(restored.processed_rows, 500); + } + + #[tokio::test] + async fn given_legacy_checkpoint_when_source_restarts_should_restore_without_pending_operations() + { + #[derive(Serialize)] + struct LegacyState { + last_poll_time: DateTime, + tracking_offsets: HashMap, + processed_rows: u64, + } + + let legacy_state = LegacyState { + last_poll_time: Utc::now(), + tracking_offsets: HashMap::from([("users".to_string(), "100".to_string())]), + processed_rows: 500, + }; + let connector_state = ConnectorState( + rmp_serde::to_vec(&legacy_state).expect("Failed to serialize legacy state"), + ); + + let src = PostgresSource::new(1, test_config(), Some(connector_state)); + + let restored = src.state.lock().await; + assert_eq!( + restored.tracking_offsets.get("users").map(String::as_str), + Some("100") + ); + assert_eq!(restored.processed_rows, 500); + assert!(restored.pending_operations.is_empty()); + drop(restored); + assert!(src.pending_batch.lock().await.is_none()); + } + + #[tokio::test] + async fn given_checkpoint_with_cleanup_when_source_restarts_should_replay_and_retire_operation() + { + let mut config = test_config(); + config.poll_interval = Some("0s".to_string()); + let state = State { + last_poll_time: Utc::now(), + tracking_offsets: HashMap::from([("users".to_string(), "3".to_string())]), + processed_rows: 3, + pending_operations: vec![PendingOperation::ProcessRows { + table: "users".to_string(), + ids: Vec::new(), + tracking_boundary: Some("3".to_string()), + target: None, + row_versions: Vec::new(), + }], + }; + let connector_state = + ConnectorState::serialize(&state, "test", 1).expect("Failed to serialize state"); + let mut src = PostgresSource::new(1, config, Some(connector_state)); + src.pool = Some( + PgPoolOptions::new() + .connect_lazy("postgres://postgres:postgres@localhost/postgres") + .expect("lazy PostgreSQL pool should be created"), + ); + + { + let pending = src.pending_batch.lock().await; + let pending = pending + .as_ref() + .expect("restored cleanup should be staged for replay"); + assert!(pending.acknowledged); + assert_eq!(pending.state.pending_operations, state.pending_operations); + } + + src.on_batch_result(SourceBatchResult::Ack) + .await + .expect("restored cleanup should replay successfully"); + { + let pending = src.pending_batch.lock().await; + assert!( + pending + .as_ref() + .is_some_and(|pending| pending.retiring_operations) ); - assert_eq!(restored.processed_rows, 500); - }); + } + + src.on_batch_result(SourceBatchResult::Nack) + .await + .expect("failed retirement checkpoint should remain retryable"); + assert!(src.pending_batch.lock().await.is_some()); + + let retirement = src + .poll() + .await + .expect("retirement checkpoint should be produced"); + assert!(retirement.messages.is_empty()); + let retired_state = retirement + .state + .expect("retirement should include cleared state") + .deserialize::("test", 1) + .expect("retirement state should deserialize"); + assert!(retired_state.pending_operations.is_empty()); + + src.on_batch_result(SourceBatchResult::Ack) + .await + .expect("retirement checkpoint should complete"); + assert!(src.pending_batch.lock().await.is_none()); + } + + #[test] + fn given_pending_cleanup_when_cleanup_target_changes_should_reject_restart() { + let mut config = test_config(); + config.delete_after_read = Some(true); + config.primary_key_column = Some("id".to_string()); + let state = State { + last_poll_time: Utc::now(), + tracking_offsets: HashMap::new(), + processed_rows: 1, + pending_operations: vec![PendingOperation::ProcessRows { + table: "users".to_string(), + ids: vec!["1".to_string()], + tracking_boundary: None, + target: Some(CleanupTarget { + key_column: "legacy_id".to_string(), + tracking_column: "updated_at".to_string(), + action: CleanupAction::Delete, + }), + row_versions: vec!["42".to_string()], + }], + }; + let connector_state = + ConnectorState::serialize(&state, "test", 1).expect("Failed to serialize state"); + let mut src = PostgresSource::new(1, config, Some(connector_state)); + + assert!(matches!( + src.validate_pending_cleanup(), + Err(Error::InitError(_)) + )); + } + + #[test] + fn given_legacy_pending_cleanup_without_row_versions_should_reject_restart() { + let mut config = test_config(); + config.delete_after_read = Some(true); + let state = State { + last_poll_time: Utc::now(), + tracking_offsets: HashMap::new(), + processed_rows: 1, + pending_operations: vec![PendingOperation::ProcessRows { + table: "users".to_string(), + ids: vec!["1".to_string()], + tracking_boundary: None, + target: None, + row_versions: Vec::new(), + }], + }; + let connector_state = + ConnectorState::serialize(&state, "test", 1).expect("Failed to serialize state"); + let mut src = PostgresSource::new(1, config, Some(connector_state)); + + assert!(matches!( + src.validate_pending_cleanup(), + Err(Error::InitError(_)) + )); + } + + #[tokio::test] + async fn given_no_state_should_start_fresh() { + let src = PostgresSource::new(1, test_config(), None); + + let state = src.state.lock().await; + assert!(state.tracking_offsets.is_empty()); + assert_eq!(state.processed_rows, 0); + } + + #[tokio::test] + async fn given_transient_database_errors_when_retrying_should_eventually_succeed() { + let attempts = AtomicU32::new(0); + + let result = with_retry( + || async { + if attempts.fetch_add(1, Ordering::Relaxed) < 2 { + Err(sqlx::Error::PoolTimedOut) + } else { + Ok(()) + } + }, + 3, + 0, + ) + .await; + + assert!(result.is_ok()); + assert_eq!(attempts.load(Ordering::Relaxed), 3); + } + + #[tokio::test] + async fn given_zero_max_retries_should_attempt_operation_once() { + let attempts = AtomicU32::new(0); + + let result: Result<(), sqlx::Error> = with_retry( + || async { + attempts.fetch_add(1, Ordering::Relaxed); + Err(sqlx::Error::PoolTimedOut) + }, + 0, + 0, + ) + .await; + + assert!(result.is_err()); + assert_eq!(attempts.load(Ordering::Relaxed), 1); } #[test] - fn given_no_state_should_start_fresh() { + fn given_active_replication_slot_sqlstate_should_be_transient() { + assert!(is_transient_sqlstate("55006")); + } + + #[tokio::test] + async fn given_nack_when_batch_is_staged_should_keep_committed_state() { let src = PostgresSource::new(1, test_config(), None); + let mut candidate_state = src.state.lock().await.clone(); + candidate_state + .tracking_offsets + .insert("users".to_string(), "3".to_string()); + candidate_state.processed_rows = 3; + candidate_state.pending_operations = vec![ + PendingOperation::ProcessRows { + table: "users".to_string(), + ids: vec!["3".to_string()], + tracking_boundary: Some("3".to_string()), + target: None, + row_versions: Vec::new(), + }, + PendingOperation::AdvanceReplicationSlot { + lsn: "0/16D32A0".to_string(), + }, + ]; + *src.pending_batch.lock().await = Some(PendingBatch { + state: candidate_state, + acknowledged: false, + retiring_operations: false, + operations_checkpointed: true, + }); - let runtime = tokio::runtime::Runtime::new().unwrap(); - runtime.block_on(async { + src.on_batch_result(SourceBatchResult::Nack) + .await + .expect("NACK should discard the candidate state"); + + { let state = src.state.lock().await; assert!(state.tracking_offsets.is_empty()); assert_eq!(state.processed_rows, 0); + } + assert!(src.pending_batch.lock().await.is_none()); + } + + #[tokio::test] + async fn given_nack_when_ack_retry_is_staged_should_keep_pending_operation() { + let src = PostgresSource::new(1, test_config(), None); + let mut candidate_state = src.state.lock().await.clone(); + candidate_state.pending_operations = vec![PendingOperation::AdvanceReplicationSlot { + lsn: "0/16D32A0".to_string(), + }]; + *src.pending_batch.lock().await = Some(PendingBatch { + state: candidate_state, + acknowledged: true, + retiring_operations: false, + operations_checkpointed: true, }); + + src.on_batch_result(SourceBatchResult::Nack) + .await + .expect("NACK should not discard an operation from an acknowledged batch"); + + assert!(src.pending_batch.lock().await.is_some()); } - #[test] - fn given_invalid_state_should_start_fresh() { - let invalid_state = ConnectorState(b"not valid json".to_vec()); - let src = PostgresSource::new(1, test_config(), Some(invalid_state)); + #[tokio::test] + async fn given_ack_when_staged_operation_fails_should_keep_candidate_state_staged() { + let mut src = PostgresSource::new(1, test_config(), None); + src.pool = Some( + PgPoolOptions::new() + .connect_lazy("postgres://postgres:postgres@localhost/postgres") + .expect("lazy PostgreSQL pool should be created"), + ); + let mut candidate_state = src.state.lock().await.clone(); + candidate_state + .tracking_offsets + .insert("users".to_string(), "3".to_string()); + candidate_state.processed_rows = 3; + candidate_state.pending_operations = vec![PendingOperation::ProcessRows { + table: "public.".to_string(), + ids: vec!["3".to_string()], + tracking_boundary: Some("3".to_string()), + target: Some(CleanupTarget { + key_column: "updated_at".to_string(), + tracking_column: "updated_at".to_string(), + action: CleanupAction::Delete, + }), + row_versions: vec!["1".to_string()], + }]; + *src.pending_batch.lock().await = Some(PendingBatch { + state: candidate_state, + acknowledged: false, + retiring_operations: false, + operations_checkpointed: true, + }); + + src.on_batch_result(SourceBatchResult::Ack) + .await + .expect("ACK callback should keep the failed operation staged"); - let runtime = tokio::runtime::Runtime::new().unwrap(); - runtime.block_on(async { + { let state = src.state.lock().await; assert!(state.tracking_offsets.is_empty()); assert_eq!(state.processed_rows, 0); + } + let pending = src.pending_batch.lock().await; + let pending = pending + .as_ref() + .expect("failed operation should remain staged"); + assert_eq!( + pending + .state + .tracking_offsets + .get("users") + .map(String::as_str), + Some("3") + ); + assert_eq!(pending.state.processed_rows, 3); + assert_eq!(pending.state.pending_operations.len(), 1); + assert!(pending.acknowledged); + assert_eq!(src.consecutive_process_failures.load(Ordering::Relaxed), 1); + } + + #[tokio::test] + async fn given_ack_operation_when_row_processing_repeatedly_fails_should_stop_and_keep_work() { + let mut src = PostgresSource::new(1, test_config(), None); + src.pool = Some( + PgPoolOptions::new() + .connect_lazy("postgres://postgres:postgres@localhost/postgres") + .expect("lazy PostgreSQL pool should be created"), + ); + let mut candidate_state = src.state.lock().await.clone(); + candidate_state.pending_operations = vec![PendingOperation::ProcessRows { + table: "public.".to_string(), + ids: vec!["3".to_string()], + tracking_boundary: Some("3".to_string()), + target: Some(CleanupTarget { + key_column: "updated_at".to_string(), + tracking_column: "updated_at".to_string(), + action: CleanupAction::Delete, + }), + row_versions: vec!["1".to_string()], + }]; + *src.pending_batch.lock().await = Some(PendingBatch { + state: candidate_state, + acknowledged: false, + retiring_operations: false, + operations_checkpointed: true, + }); + + for _ in 1..MAX_CONSECUTIVE_PROCESS_FAILURES { + src.on_batch_result(SourceBatchResult::Ack) + .await + .expect("row processing should remain retryable below the failure threshold"); + } + let error = src + .on_batch_result(SourceBatchResult::Ack) + .await + .expect_err("row processing should stop the source at the failure threshold"); + + assert!(matches!(error, Error::Connection(_))); + assert_eq!( + src.consecutive_process_failures.load(Ordering::Relaxed), + MAX_CONSECUTIVE_PROCESS_FAILURES + ); + assert!(src.pending_batch.lock().await.is_some()); + } + + #[tokio::test] + async fn given_prior_failure_when_row_processing_succeeds_should_reset_failure_count() { + let mut src = PostgresSource::new(1, test_config(), None); + src.pool = Some( + PgPoolOptions::new() + .connect_lazy("postgres://postgres:postgres@localhost/postgres") + .expect("lazy PostgreSQL pool should be created"), + ); + src.consecutive_process_failures.store(1, Ordering::Relaxed); + let operation = PendingOperation::ProcessRows { + table: "users".to_string(), + ids: Vec::new(), + tracking_boundary: None, + target: None, + row_versions: Vec::new(), + }; + + assert!( + src.apply_pending_operation( + &operation, + tokio::time::Instant::now() + ACK_BATCH_TIMEOUT, + ) + .await + .expect("empty row processing should succeed") + ); + assert_eq!(src.consecutive_process_failures.load(Ordering::Relaxed), 0); + } + + #[tokio::test] + async fn given_staged_operation_when_retry_succeeds_should_commit_candidate_state() { + let mut src = PostgresSource::new(1, test_config(), None); + src.pool = Some( + PgPoolOptions::new() + .connect_lazy("postgres://postgres:postgres@localhost/postgres") + .expect("lazy PostgreSQL pool should be created"), + ); + let mut candidate_state = src.state.lock().await.clone(); + candidate_state + .tracking_offsets + .insert("users".to_string(), "3".to_string()); + candidate_state.processed_rows = 3; + candidate_state.pending_operations = vec![PendingOperation::ProcessRows { + table: "users".to_string(), + ids: Vec::new(), + tracking_boundary: None, + target: None, + row_versions: Vec::new(), + }]; + *src.pending_batch.lock().await = Some(PendingBatch { + state: candidate_state, + acknowledged: true, + retiring_operations: false, + operations_checkpointed: true, + }); + + src.on_batch_result(SourceBatchResult::Ack) + .await + .expect("staged operation should be retried successfully"); + + { + let state = src.state.lock().await; + assert_eq!( + state.tracking_offsets.get("users").map(String::as_str), + Some("3") + ); + assert_eq!(state.processed_rows, 3); + } + { + let pending = src.pending_batch.lock().await; + let pending = pending + .as_ref() + .expect("successful cleanup should stage operation retirement"); + assert!(pending.retiring_operations); + assert!(pending.state.pending_operations.is_empty()); + } + + src.on_batch_result(SourceBatchResult::Ack) + .await + .expect("retirement checkpoint ACK should clear the pending batch"); + assert!(src.pending_batch.lock().await.is_none()); + } + + #[tokio::test] + async fn given_failed_slot_advance_should_increment_consecutive_failures() { + let src = PostgresSource::new(1, test_config(), None); + let mut candidate_state = src.state.lock().await.clone(); + candidate_state.pending_operations = vec![PendingOperation::AdvanceReplicationSlot { + lsn: "0/16D32A0".to_string(), + }]; + *src.pending_batch.lock().await = Some(PendingBatch { + state: candidate_state, + acknowledged: false, + retiring_operations: false, + operations_checkpointed: true, + }); + + src.on_batch_result(SourceBatchResult::Ack) + .await + .expect("ACK should record the failed slot advance"); + + assert_eq!(src.consecutive_advance_failures.load(Ordering::Relaxed), 1); + let pending = src.pending_batch.lock().await; + assert!(pending.as_ref().is_some_and(|pending| pending.acknowledged)); + } + + #[tokio::test] + async fn given_repeated_slot_advance_failures_should_stop_source() { + let src = PostgresSource::new(1, test_config(), None); + let mut candidate_state = src.state.lock().await.clone(); + candidate_state.pending_operations = vec![PendingOperation::AdvanceReplicationSlot { + lsn: "0/16D32A0".to_string(), + }]; + *src.pending_batch.lock().await = Some(PendingBatch { + state: candidate_state, + acknowledged: false, + retiring_operations: false, + operations_checkpointed: true, }); + + for _ in 1..MAX_CONSECUTIVE_ADVANCE_FAILURES { + src.on_batch_result(SourceBatchResult::Ack) + .await + .expect("slot advance should remain retryable below the failure threshold"); + } + let error = src + .on_batch_result(SourceBatchResult::Ack) + .await + .expect_err("slot advance should stop the source at the failure threshold"); + + assert!(matches!(error, Error::Connection(_))); + assert_eq!( + src.consecutive_advance_failures.load(Ordering::Relaxed), + MAX_CONSECUTIVE_ADVANCE_FAILURES + ); + assert!(src.pending_batch.lock().await.is_some()); + } + + #[tokio::test] + async fn given_invalid_state_should_start_fresh() { + let invalid_state = ConnectorState(b"not valid json".to_vec()); + let src = PostgresSource::new(1, test_config(), Some(invalid_state)); + + let state = src.state.lock().await; + assert!(state.tracking_offsets.is_empty()); + assert_eq!(state.processed_rows, 0); } #[test] @@ -2545,6 +4014,7 @@ mod tests { .with_timezone(&Utc), tracking_offsets: HashMap::from([("table1".to_string(), "42".to_string())]), processed_rows: 1000, + pending_operations: Vec::new(), }; let connector_state = @@ -2556,6 +4026,7 @@ mod tests { assert_eq!(original.last_poll_time, deserialized.last_poll_time); assert_eq!(original.tracking_offsets, deserialized.tracking_offsets); assert_eq!(original.processed_rows, deserialized.processed_rows); + assert_eq!(original.pending_operations, deserialized.pending_operations); } #[test] diff --git a/core/integration/src/harness/handle/connectors_runtime.rs b/core/integration/src/harness/handle/connectors_runtime.rs index 8a581c4813..66d2a90faf 100644 --- a/core/integration/src/harness/handle/connectors_runtime.rs +++ b/core/integration/src/harness/handle/connectors_runtime.rs @@ -44,6 +44,7 @@ pub struct ConnectorsRuntimeHandle { child_handle: Option, server_address: SocketAddr, iggy_address: Option, + iggy_connection_options: Option, stdout_path: Option, stderr_path: Option, _port_reserver: SinglePortReserver, @@ -80,6 +81,14 @@ impl ConnectorsRuntimeHandle { common::collect_logs(&self.stdout_path, &self.stderr_path) } + pub fn set_iggy_connection_options(&mut self, options: impl Into) { + self.iggy_connection_options = Some(options.into()); + } + + pub fn clear_iggy_connection_options(&mut self) { + self.iggy_connection_options = None; + } + fn build_envs(&mut self) { let state_path = self.context.connectors_runtime_state_path(self.server_id); self.envs.insert( @@ -92,8 +101,12 @@ impl ConnectorsRuntimeHandle { ); if let Some(addr) = self.iggy_address { + let address = self + .iggy_connection_options + .as_ref() + .map_or_else(|| addr.to_string(), |options| format!("{addr}?{options}")); self.envs - .insert("IGGY_CONNECTORS_IGGY_ADDRESS".to_string(), addr.to_string()); + .insert("IGGY_CONNECTORS_IGGY_ADDRESS".to_string(), address); } if let Some(ref config_path) = self.config.config_path { @@ -127,6 +140,7 @@ impl ConnectorsRuntimeHandle { child_handle: None, server_address, iggy_address: None, + iggy_connection_options: None, stdout_path: None, stderr_path: None, _port_reserver: reserver, diff --git a/core/integration/src/harness/seeds.rs b/core/integration/src/harness/seeds.rs index 5485dbc6d9..b0f54f6440 100644 --- a/core/integration/src/harness/seeds.rs +++ b/core/integration/src/harness/seeds.rs @@ -17,7 +17,7 @@ use iggy::prelude::{IggyClient, StreamClient, TopicClient, TopicCreateOptions}; use iggy_common::{ - Consumer, Identifier, IggyExpiry, IggyMessage, MaxTopicSize, Partitioning, + Consumer, Durability, Identifier, IggyExpiry, IggyMessage, MaxTopicSize, Partitioning, PersonalAccessTokenExpiry, UserStatus, }; use iggy_common::{ @@ -78,6 +78,8 @@ pub async fn connector_stream(client: &IggyClient) -> Result<(), SeedError> { names::TOPIC, &TopicCreateOptions { partitions_count: Some(1), + durability: Durability::Persisted, + messages_required_to_save: Some(1), ..TopicCreateOptions::default() }, ) @@ -100,6 +102,8 @@ pub async fn connector_multi_topic_stream(client: &IggyClient) -> Result<(), See names::TOPIC, &TopicCreateOptions { partitions_count: Some(1), + durability: Durability::Persisted, + messages_required_to_save: Some(1), ..TopicCreateOptions::default() }, ) @@ -111,6 +115,8 @@ pub async fn connector_multi_topic_stream(client: &IggyClient) -> Result<(), See names::TOPIC_2, &TopicCreateOptions { partitions_count: Some(1), + durability: Durability::Persisted, + messages_required_to_save: Some(1), ..TopicCreateOptions::default() }, ) diff --git a/core/integration/tests/connectors/fixtures/mod.rs b/core/integration/tests/connectors/fixtures/mod.rs index 5a80f86358..dcad25ce48 100644 --- a/core/integration/tests/connectors/fixtures/mod.rs +++ b/core/integration/tests/connectors/fixtures/mod.rs @@ -81,8 +81,10 @@ pub use mongodb::{ }; pub use postgres::{ PostgresOps, PostgresSinkByteaFixture, PostgresSinkFixture, PostgresSinkJsonFixture, - PostgresSourceByteaFixture, PostgresSourceCdcFixture, PostgresSourceDeleteFixture, - PostgresSourceJsonFixture, PostgresSourceJsonbFixture, PostgresSourceMarkFixture, + PostgresSourceByteaFixture, PostgresSourceCdcFixture, PostgresSourceCdcSlowPollFixture, + PostgresSourceDeleteFixture, PostgresSourceDeleteSlowPollFixture, PostgresSourceJsonFixture, + PostgresSourceJsonbFixture, PostgresSourceMarkFixture, PostgresSourceNonUniqueCleanupFixture, + PostgresSourceNonUniqueTrackingFixture, PostgresSourceNumericTrackingFixture, PostgresSourceOps, }; pub use quickwit::{ diff --git a/core/integration/tests/connectors/fixtures/postgres/cdc.rs b/core/integration/tests/connectors/fixtures/postgres/cdc.rs index 88d69a0544..9c346d3f97 100644 --- a/core/integration/tests/connectors/fixtures/postgres/cdc.rs +++ b/core/integration/tests/connectors/fixtures/postgres/cdc.rs @@ -204,3 +204,44 @@ impl TestFixture for PostgresSourceCdcFixture { envs } } + +/// The longer poll interval leaves time to restart Iggy before the SDK's +/// consecutive-NACK limit stops the source. +pub struct PostgresSourceCdcSlowPollFixture { + inner: PostgresSourceCdcFixture, +} + +impl std::ops::Deref for PostgresSourceCdcSlowPollFixture { + type Target = PostgresSourceCdcFixture; + + fn deref(&self) -> &Self::Target { + &self.inner + } +} + +impl PostgresOps for PostgresSourceCdcSlowPollFixture { + fn container(&self) -> &PostgresContainer { + self.inner.container() + } +} + +impl PostgresSourceOps for PostgresSourceCdcSlowPollFixture { + fn table_name(&self) -> &str { + self.inner.table_name() + } +} + +#[async_trait] +impl TestFixture for PostgresSourceCdcSlowPollFixture { + async fn setup() -> Result { + Ok(Self { + inner: PostgresSourceCdcFixture::setup().await?, + }) + } + + fn connectors_runtime_envs(&self) -> HashMap { + let mut envs = self.inner.connectors_runtime_envs(); + envs.insert(ENV_SOURCE_POLL_INTERVAL.to_string(), "5s".to_string()); + envs + } +} diff --git a/core/integration/tests/connectors/fixtures/postgres/mod.rs b/core/integration/tests/connectors/fixtures/postgres/mod.rs index cca22e6e33..87c0c22fbe 100644 --- a/core/integration/tests/connectors/fixtures/postgres/mod.rs +++ b/core/integration/tests/connectors/fixtures/postgres/mod.rs @@ -20,10 +20,12 @@ mod container; mod sink; mod source; -pub use cdc::PostgresSourceCdcFixture; +pub use cdc::{PostgresSourceCdcFixture, PostgresSourceCdcSlowPollFixture}; pub use container::{PostgresOps, PostgresSourceOps}; pub use sink::{PostgresSinkByteaFixture, PostgresSinkFixture, PostgresSinkJsonFixture}; pub use source::{ - PostgresSourceByteaFixture, PostgresSourceDeleteFixture, PostgresSourceJsonFixture, - PostgresSourceJsonbFixture, PostgresSourceMarkFixture, + PostgresSourceByteaFixture, PostgresSourceDeleteFixture, PostgresSourceDeleteSlowPollFixture, + PostgresSourceJsonFixture, PostgresSourceJsonbFixture, PostgresSourceMarkFixture, + PostgresSourceNonUniqueCleanupFixture, PostgresSourceNonUniqueTrackingFixture, + PostgresSourceNumericTrackingFixture, }; diff --git a/core/integration/tests/connectors/fixtures/postgres/source.rs b/core/integration/tests/connectors/fixtures/postgres/source.rs index 5ad2900c63..f37c59bd46 100644 --- a/core/integration/tests/connectors/fixtures/postgres/source.rs +++ b/core/integration/tests/connectors/fixtures/postgres/source.rs @@ -103,7 +103,11 @@ impl PostgresSourceJsonFixture { impl TestFixture for PostgresSourceJsonFixture { async fn setup() -> Result { let container = PostgresContainer::start().await?; - Ok(Self { container }) + let fixture = Self { container }; + let pool = fixture.create_pool().await?; + fixture.create_table(&pool).await; + pool.close().await; + Ok(fixture) } fn connectors_runtime_envs(&self) -> HashMap { @@ -182,7 +186,11 @@ impl PostgresSourceByteaFixture { impl TestFixture for PostgresSourceByteaFixture { async fn setup() -> Result { let container = PostgresContainer::start().await?; - Ok(Self { container }) + let fixture = Self { container }; + let pool = fixture.create_pool().await?; + fixture.create_table(&pool).await; + pool.close().await; + Ok(fixture) } fn connectors_runtime_envs(&self) -> HashMap { @@ -262,7 +270,11 @@ impl PostgresSourceJsonbFixture { impl TestFixture for PostgresSourceJsonbFixture { async fn setup() -> Result { let container = PostgresContainer::start().await?; - Ok(Self { container }) + let fixture = Self { container }; + let pool = fixture.create_pool().await?; + fixture.create_table(&pool).await; + pool.close().await; + Ok(fixture) } fn connectors_runtime_envs(&self) -> HashMap { @@ -350,7 +362,11 @@ impl PostgresSourceDeleteFixture { impl TestFixture for PostgresSourceDeleteFixture { async fn setup() -> Result { let container = PostgresContainer::start().await?; - Ok(Self { container }) + let fixture = Self { container }; + let pool = fixture.create_pool().await?; + fixture.create_table(&pool).await; + pool.close().await; + Ok(fixture) } fn connectors_runtime_envs(&self) -> HashMap { @@ -382,6 +398,139 @@ impl TestFixture for PostgresSourceDeleteFixture { } } +/// The longer poll interval leaves time to restart Iggy before the SDK's +/// consecutive-NACK limit stops the source. +pub struct PostgresSourceDeleteSlowPollFixture { + inner: PostgresSourceDeleteFixture, +} + +impl std::ops::Deref for PostgresSourceDeleteSlowPollFixture { + type Target = PostgresSourceDeleteFixture; + + fn deref(&self) -> &Self::Target { + &self.inner + } +} + +impl PostgresOps for PostgresSourceDeleteSlowPollFixture { + fn container(&self) -> &PostgresContainer { + self.inner.container() + } +} + +impl PostgresSourceOps for PostgresSourceDeleteSlowPollFixture { + fn table_name(&self) -> &str { + self.inner.table_name() + } +} + +#[async_trait] +impl TestFixture for PostgresSourceDeleteSlowPollFixture { + async fn setup() -> Result { + Ok(Self { + inner: PostgresSourceDeleteFixture::setup().await?, + }) + } + + fn connectors_runtime_envs(&self) -> HashMap { + let mut envs = self.inner.connectors_runtime_envs(); + envs.insert(ENV_SOURCE_POLL_INTERVAL.to_string(), "5s".to_string()); + envs + } +} + +/// PostgreSQL source fixture with an exact NUMERIC tracking column. +pub struct PostgresSourceNumericTrackingFixture { + container: PostgresContainer, +} + +impl PostgresOps for PostgresSourceNumericTrackingFixture { + fn container(&self) -> &PostgresContainer { + &self.container + } +} + +impl PostgresSourceOps for PostgresSourceNumericTrackingFixture { + fn table_name(&self) -> &str { + Self::TABLE + } +} + +impl PostgresSourceNumericTrackingFixture { + const TABLE: &'static str = "test_numeric_tracking"; + + pub async fn create_table(&self, pool: &Pool) { + let query = format!( + "CREATE TABLE IF NOT EXISTS {} ( + tracking_value NUMERIC PRIMARY KEY + )", + Self::TABLE + ); + sqlx::query(sqlx::AssertSqlSafe(query)) + .execute(pool) + .await + .unwrap_or_else(|e| panic!("Failed to create table: {e}")); + } + + pub async fn insert_row(&self, pool: &Pool, tracking_value: &str) { + let query = format!( + "INSERT INTO {} (tracking_value) VALUES ($1::numeric)", + Self::TABLE + ); + sqlx::query(sqlx::AssertSqlSafe(query)) + .bind(tracking_value) + .execute(pool) + .await + .unwrap_or_else(|e| panic!("Failed to insert row: {e}")); + } + + pub async fn count_rows(&self, pool: &Pool) -> i64 { + PostgresSourceOps::count_rows(self, pool).await + } +} + +#[async_trait] +impl TestFixture for PostgresSourceNumericTrackingFixture { + async fn setup() -> Result { + let container = PostgresContainer::start().await?; + let fixture = Self { container }; + let pool = fixture.create_pool().await?; + fixture.create_table(&pool).await; + pool.close().await; + Ok(fixture) + } + + fn connectors_runtime_envs(&self) -> HashMap { + let mut envs = HashMap::new(); + envs.insert( + ENV_SOURCE_CONNECTION_STRING.to_string(), + self.container.connection_string.clone(), + ); + envs.insert(ENV_SOURCE_TABLES.to_string(), format!("[{}]", Self::TABLE)); + envs.insert( + ENV_SOURCE_TRACKING_COLUMN.to_string(), + "tracking_value".to_string(), + ); + envs.insert(ENV_SOURCE_DELETE_AFTER_READ.to_string(), "true".to_string()); + envs.insert(ENV_SOURCE_INCLUDE_METADATA.to_string(), "true".to_string()); + envs.insert( + ENV_SOURCE_STREAMS_0_STREAM.to_string(), + DEFAULT_TEST_STREAM.to_string(), + ); + envs.insert( + ENV_SOURCE_STREAMS_0_TOPIC.to_string(), + DEFAULT_TEST_TOPIC.to_string(), + ); + envs.insert(ENV_SOURCE_STREAMS_0_SCHEMA.to_string(), "json".to_string()); + envs.insert(ENV_SOURCE_POLL_INTERVAL.to_string(), "10ms".to_string()); + envs.insert( + ENV_SOURCE_PATH.to_string(), + "../../target/debug/libiggy_connector_postgres_source".to_string(), + ); + envs + } +} + /// PostgreSQL source fixture with processed_column marking. pub struct PostgresSourceMarkFixture { container: PostgresContainer, @@ -465,7 +614,11 @@ impl PostgresSourceMarkFixture { impl TestFixture for PostgresSourceMarkFixture { async fn setup() -> Result { let container = PostgresContainer::start().await?; - Ok(Self { container }) + let fixture = Self { container }; + let pool = fixture.create_pool().await?; + fixture.create_table(&pool).await; + pool.close().await; + Ok(fixture) } fn connectors_runtime_envs(&self) -> HashMap { @@ -499,3 +652,153 @@ impl TestFixture for PostgresSourceMarkFixture { envs } } + +/// PostgreSQL source fixture with an explicitly configured non-unique cleanup key. +pub struct PostgresSourceNonUniqueCleanupFixture { + container: PostgresContainer, +} + +impl PostgresOps for PostgresSourceNonUniqueCleanupFixture { + fn container(&self) -> &PostgresContainer { + &self.container + } +} + +impl PostgresSourceOps for PostgresSourceNonUniqueCleanupFixture { + fn table_name(&self) -> &str { + Self::TABLE + } +} + +impl PostgresSourceNonUniqueCleanupFixture { + const TABLE: &'static str = "test_non_unique_cleanup"; + + async fn create_table(&self, pool: &Pool) { + let query = format!( + "CREATE TABLE IF NOT EXISTS {} ( + id SERIAL PRIMARY KEY, + group_id INTEGER NOT NULL + )", + Self::TABLE + ); + sqlx::query(sqlx::AssertSqlSafe(query)) + .execute(pool) + .await + .unwrap_or_else(|e| panic!("Failed to create table: {e}")); + } +} + +#[async_trait] +impl TestFixture for PostgresSourceNonUniqueCleanupFixture { + async fn setup() -> Result { + let container = PostgresContainer::start().await?; + let fixture = Self { container }; + let pool = fixture.create_pool().await?; + fixture.create_table(&pool).await; + pool.close().await; + Ok(fixture) + } + + fn connectors_runtime_envs(&self) -> HashMap { + let mut envs = HashMap::new(); + envs.insert( + ENV_SOURCE_CONNECTION_STRING.to_string(), + self.container.connection_string.clone(), + ); + envs.insert(ENV_SOURCE_TABLES.to_string(), format!("[{}]", Self::TABLE)); + envs.insert(ENV_SOURCE_TRACKING_COLUMN.to_string(), "id".to_string()); + envs.insert( + ENV_SOURCE_PRIMARY_KEY_COLUMN.to_string(), + "group_id".to_string(), + ); + envs.insert(ENV_SOURCE_DELETE_AFTER_READ.to_string(), "true".to_string()); + envs.insert( + ENV_SOURCE_STREAMS_0_STREAM.to_string(), + DEFAULT_TEST_STREAM.to_string(), + ); + envs.insert( + ENV_SOURCE_STREAMS_0_TOPIC.to_string(), + DEFAULT_TEST_TOPIC.to_string(), + ); + envs.insert(ENV_SOURCE_STREAMS_0_SCHEMA.to_string(), "json".to_string()); + envs.insert( + ENV_SOURCE_PATH.to_string(), + "../../target/debug/libiggy_connector_postgres_source".to_string(), + ); + envs + } +} + +/// PostgreSQL source fixture with a non-unique scalar tracking cursor. +pub struct PostgresSourceNonUniqueTrackingFixture { + container: PostgresContainer, +} + +impl PostgresOps for PostgresSourceNonUniqueTrackingFixture { + fn container(&self) -> &PostgresContainer { + &self.container + } +} + +impl PostgresSourceOps for PostgresSourceNonUniqueTrackingFixture { + fn table_name(&self) -> &str { + Self::TABLE + } +} + +impl PostgresSourceNonUniqueTrackingFixture { + const TABLE: &'static str = "test_non_unique_tracking"; + + async fn create_table(&self, pool: &Pool) { + let query = format!( + "CREATE TABLE IF NOT EXISTS {} ( + id SERIAL PRIMARY KEY, + cursor_value INTEGER NOT NULL + )", + Self::TABLE + ); + sqlx::query(sqlx::AssertSqlSafe(query)) + .execute(pool) + .await + .unwrap_or_else(|e| panic!("Failed to create table: {e}")); + } +} + +#[async_trait] +impl TestFixture for PostgresSourceNonUniqueTrackingFixture { + async fn setup() -> Result { + let container = PostgresContainer::start().await?; + let fixture = Self { container }; + let pool = fixture.create_pool().await?; + fixture.create_table(&pool).await; + pool.close().await; + Ok(fixture) + } + + fn connectors_runtime_envs(&self) -> HashMap { + let mut envs = HashMap::new(); + envs.insert( + ENV_SOURCE_CONNECTION_STRING.to_string(), + self.container.connection_string.clone(), + ); + envs.insert(ENV_SOURCE_TABLES.to_string(), format!("[{}]", Self::TABLE)); + envs.insert( + ENV_SOURCE_TRACKING_COLUMN.to_string(), + "cursor_value".to_string(), + ); + envs.insert( + ENV_SOURCE_STREAMS_0_STREAM.to_string(), + DEFAULT_TEST_STREAM.to_string(), + ); + envs.insert( + ENV_SOURCE_STREAMS_0_TOPIC.to_string(), + DEFAULT_TEST_TOPIC.to_string(), + ); + envs.insert(ENV_SOURCE_STREAMS_0_SCHEMA.to_string(), "json".to_string()); + envs.insert( + ENV_SOURCE_PATH.to_string(), + "../../target/debug/libiggy_connector_postgres_source".to_string(), + ); + envs + } +} diff --git a/core/integration/tests/connectors/postgres/mod.rs b/core/integration/tests/connectors/postgres/mod.rs index ee992b36b8..250e6b082c 100644 --- a/core/integration/tests/connectors/postgres/mod.rs +++ b/core/integration/tests/connectors/postgres/mod.rs @@ -20,12 +20,22 @@ mod postgres_source; mod postgres_source_cdc; mod restart; -use crate::connectors::TestMessage; +use std::time::Duration; + +use iggy_connector_sdk::api::{ConnectorRuntimeStats, ConnectorStats, ConnectorStatus}; +use reqwest::Client; use serde::Deserialize; +use tokio::time::{sleep, timeout}; +use crate::connectors::TestMessage; + +const API_KEY: &str = "test-api-key"; +const SOURCE_KEY: &str = "postgres"; +const DEFAULT_SLOT: &str = "iggy_slot"; const TEST_MESSAGE_COUNT: usize = 3; const POLL_ATTEMPTS: usize = 100; const POLL_INTERVAL_MS: u64 = 50; +const SEND_FAILURE_TIMEOUT: Duration = Duration::from_secs(25); #[derive(Debug, Deserialize)] struct DatabaseRecord { @@ -33,3 +43,49 @@ struct DatabaseRecord { operation_type: String, data: TestMessage, } + +async fn source_stats(http: &Client, api_url: &str) -> Option { + let response = http + .get(format!("{api_url}/stats")) + .header("api-key", API_KEY) + .send() + .await + .ok()?; + let stats = response.json::().await.ok()?; + stats + .connectors + .into_iter() + .find(|connector| connector.key == SOURCE_KEY) +} + +async fn wait_for_source_status(http: &Client, api_url: &str, expected: ConnectorStatus) { + for _ in 0..POLL_ATTEMPTS { + if let Some(source) = source_stats(http, api_url).await + && source.status == expected + { + return; + } + sleep(Duration::from_millis(POLL_INTERVAL_MS)).await; + } + panic!("Source connector did not reach {expected:?} status in time"); +} + +async fn wait_for_source_errors( + http: &Client, + api_url: &str, + minimum_errors: u64, +) -> ConnectorStats { + timeout(SEND_FAILURE_TIMEOUT, async { + loop { + if let Some(source) = source_stats(http, api_url).await + && source.status == ConnectorStatus::Error + && source.errors >= minimum_errors + { + return source; + } + sleep(Duration::from_millis(POLL_INTERVAL_MS)).await; + } + }) + .await + .expect("PostgreSQL source did not retry the NACKed batch") +} diff --git a/core/integration/tests/connectors/postgres/postgres_source.rs b/core/integration/tests/connectors/postgres/postgres_source.rs index a66a3c5e67..375cb785e0 100644 --- a/core/integration/tests/connectors/postgres/postgres_source.rs +++ b/core/integration/tests/connectors/postgres/postgres_source.rs @@ -15,20 +15,61 @@ // specific language governing permissions and limitations // under the License. -use super::{DatabaseRecord, POLL_ATTEMPTS, POLL_INTERVAL_MS, TEST_MESSAGE_COUNT}; -use crate::connectors::create_test_messages; -use crate::connectors::fixtures::{ - PostgresOps, PostgresSourceByteaFixture, PostgresSourceDeleteFixture, - PostgresSourceJsonFixture, PostgresSourceJsonbFixture, PostgresSourceMarkFixture, - PostgresSourceOps, -}; +use std::time::Duration; + use iggy_common::MessageClient; use iggy_common::{Consumer, Identifier, PollingStrategy}; +use iggy_connector_sdk::api::ConnectorStatus; use integration::harness::seeds; use integration::iggy_harness; -use std::time::Duration; +use reqwest::Client; use tokio::time::sleep; +use super::{ + DatabaseRecord, POLL_ATTEMPTS, POLL_INTERVAL_MS, TEST_MESSAGE_COUNT, source_stats, + wait_for_source_errors, wait_for_source_status, +}; +use crate::connectors::create_test_messages; +use crate::connectors::fixtures::{ + PostgresOps, PostgresSourceByteaFixture, PostgresSourceDeleteFixture, + PostgresSourceDeleteSlowPollFixture, PostgresSourceJsonFixture, PostgresSourceJsonbFixture, + PostgresSourceMarkFixture, PostgresSourceNonUniqueCleanupFixture, + PostgresSourceNonUniqueTrackingFixture, PostgresSourceNumericTrackingFixture, + PostgresSourceOps, +}; + +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/postgres/source.toml")), + seed = seeds::connector_stream +)] +async fn given_non_unique_cleanup_key_when_source_opens_should_reject_config( + harness: &TestHarness, + _fixture: PostgresSourceNonUniqueCleanupFixture, +) { + let api_url = harness + .connectors_runtime() + .expect("connectors runtime") + .http_url(); + wait_for_source_status(&Client::new(), &api_url, ConnectorStatus::Error).await; +} + +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/postgres/source.toml")), + seed = seeds::connector_stream +)] +async fn given_non_unique_tracking_column_when_source_opens_should_reject_config( + harness: &TestHarness, + _fixture: PostgresSourceNonUniqueTrackingFixture, +) { + let api_url = harness + .connectors_runtime() + .expect("connectors runtime") + .http_url(); + wait_for_source_status(&Client::new(), &api_url, ConnectorStatus::Error).await; +} + #[iggy_harness( server(connectors_runtime(config_path = "tests/connectors/postgres/source.toml")), seed = seeds::connector_stream @@ -127,6 +168,220 @@ async fn json_rows_source_produces_messages_to_iggy( } } +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/postgres/source.toml")), + seed = seeds::connector_stream +)] +async fn given_delete_after_read_when_iggy_crashes_should_delete_only_after_redelivery( + harness: &mut TestHarness, + fixture: PostgresSourceDeleteFixture, +) { + let pool = fixture.create_pool().await.expect("Failed to create pool"); + fixture.create_table(&pool).await; + + harness + .server_mut() + .stop_dependents() + .expect("Failed to stop connectors runtime"); + harness + .server_mut() + .connectors_runtime_mut() + .expect("connectors runtime") + // Keep a failed send bounded instead of waiting indefinitely for Iggy to return. + .set_iggy_connection_options("reconnection_retries=0"); + harness + .server_mut() + .start_dependents() + .await + .expect("Failed to restart connectors runtime"); + + let api_url = harness + .connectors_runtime() + .expect("connectors runtime") + .http_url(); + let http = Client::new(); + let errors_before_failure = source_stats(&http, &api_url) + .await + .expect("PostgreSQL source stats should be present") + .errors; + + harness.kill_node(0).expect("Failed to kill Iggy server"); + + for index in 0..TEST_MESSAGE_COUNT { + fixture + .insert_row(&pool, &format!("row_{index}"), index as i32) + .await; + } + + let failed_source = wait_for_source_errors(&http, &api_url, errors_before_failure + 2).await; + assert_eq!(failed_source.status, ConnectorStatus::Error); + assert_eq!( + fixture.count_rows(&pool).await, + TEST_MESSAGE_COUNT as i64, + "NACKed rows must not be deleted" + ); + + harness + .server_mut() + .stop_dependents() + .expect("Failed to stop connectors runtime"); + harness + .restart_node(0) + .expect("Failed to restart Iggy server"); + harness + .server_mut() + .connectors_runtime_mut() + .expect("connectors runtime") + .clear_iggy_connection_options(); + harness + .server_mut() + .start_dependents() + .await + .expect("Failed to restart connectors runtime"); + + let client = harness.root_client().await.unwrap(); + let stream_id: Identifier = seeds::names::STREAM.try_into().unwrap(); + let topic_id: Identifier = seeds::names::TOPIC.try_into().unwrap(); + let consumer_id: Identifier = "send_failure_consumer".try_into().unwrap(); + let mut received = 0; + + for _ in 0..POLL_ATTEMPTS { + if let Ok(polled) = client + .poll_messages( + &stream_id, + &topic_id, + None, + &Consumer::new(consumer_id.clone()), + &PollingStrategy::next(), + 10, + true, + ) + .await + { + received += polled.messages.len(); + if received >= TEST_MESSAGE_COUNT { + break; + } + } + sleep(Duration::from_millis(POLL_INTERVAL_MS)).await; + } + + assert_eq!( + received, TEST_MESSAGE_COUNT, + "Rows polled during the failed send should be delivered after restart" + ); + + let mut remaining_rows = fixture.count_rows(&pool).await; + for _ in 0..POLL_ATTEMPTS { + if remaining_rows == 0 { + break; + } + sleep(Duration::from_millis(POLL_INTERVAL_MS)).await; + remaining_rows = fixture.count_rows(&pool).await; + } + assert_eq!(remaining_rows, 0, "ACKed rows should be deleted"); + + pool.close().await; +} + +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/postgres/source.toml")), + seed = seeds::connector_stream +)] +async fn given_delivery_failure_when_iggy_restarts_should_redeliver_without_runtime_restart( + harness: &mut TestHarness, + fixture: PostgresSourceDeleteSlowPollFixture, +) { + const REDELIVERY_ATTEMPTS: usize = POLL_ATTEMPTS * 3; + + let pool = fixture.create_pool().await.expect("Failed to create pool"); + fixture.create_table(&pool).await; + + harness + .server_mut() + .stop_dependents() + .expect("Failed to stop connectors runtime"); + harness + .server_mut() + .connectors_runtime_mut() + .expect("connectors runtime") + .set_iggy_connection_options("reconnection_retries=0"); + harness + .server_mut() + .start_dependents() + .await + .expect("Failed to restart connectors runtime"); + + let api_url = harness + .connectors_runtime() + .expect("connectors runtime") + .http_url(); + let http = Client::new(); + let errors_before_failure = source_stats(&http, &api_url) + .await + .expect("PostgreSQL source stats should be present") + .errors; + + harness.kill_node(0).expect("Failed to kill Iggy server"); + fixture.insert_row(&pool, "single_nack", 1).await; + + let failed_source = wait_for_source_errors(&http, &api_url, errors_before_failure + 1).await; + assert_eq!(failed_source.status, ConnectorStatus::Error); + assert_eq!( + fixture.count_rows(&pool).await, + 1, + "NACKed row must not be deleted" + ); + + harness + .restart_node(0) + .expect("Failed to restart only the Iggy server"); + + let client = harness.root_client().await.unwrap(); + let stream_id: Identifier = seeds::names::STREAM.try_into().unwrap(); + let topic_id: Identifier = seeds::names::TOPIC.try_into().unwrap(); + let consumer_id: Identifier = "nack_survival_consumer".try_into().unwrap(); + let mut received = 0; + + for _ in 0..REDELIVERY_ATTEMPTS { + if let Ok(polled) = client + .poll_messages( + &stream_id, + &topic_id, + None, + &Consumer::new(consumer_id.clone()), + &PollingStrategy::next(), + 10, + true, + ) + .await + { + received += polled.messages.len(); + if received == 1 { + break; + } + } + sleep(Duration::from_millis(POLL_INTERVAL_MS)).await; + } + + assert_eq!(received, 1, "NACKed row should be redelivered"); + + let mut remaining_rows = fixture.count_rows(&pool).await; + for _ in 0..REDELIVERY_ATTEMPTS { + if remaining_rows == 0 { + break; + } + sleep(Duration::from_millis(POLL_INTERVAL_MS)).await; + remaining_rows = fixture.count_rows(&pool).await; + } + assert_eq!(remaining_rows, 0, "ACKed row should be deleted"); + wait_for_source_status(&http, &api_url, ConnectorStatus::Running).await; + + pool.close().await; +} + #[iggy_harness( server(connectors_runtime(config_path = "tests/connectors/postgres/source.toml")), seed = seeds::connector_stream @@ -327,6 +582,145 @@ async fn delete_after_read_source_removes_rows_after_producing( pool.close().await; } +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/postgres/source.toml")), + seed = seeds::connector_stream +)] +async fn numeric_tracking_source_preserves_exact_ack_boundary( + harness: &TestHarness, + fixture: PostgresSourceNumericTrackingFixture, +) { + const TRACKING_VALUE: &str = "9007199254740993.25"; + + let client = harness.root_client().await.unwrap(); + let pool = fixture.create_pool().await.expect("Failed to create pool"); + fixture.create_table(&pool).await; + fixture.insert_row(&pool, TRACKING_VALUE).await; + + let stream_id: Identifier = seeds::names::STREAM.try_into().unwrap(); + let topic_id: Identifier = seeds::names::TOPIC.try_into().unwrap(); + let consumer_id: Identifier = "numeric_tracking_consumer".try_into().unwrap(); + let mut received = None; + + for _ in 0..POLL_ATTEMPTS { + if let Ok(polled) = client + .poll_messages( + &stream_id, + &topic_id, + None, + &Consumer::new(consumer_id.clone()), + &PollingStrategy::next(), + 1, + true, + ) + .await + { + for message in polled.messages { + if let Ok(record) = serde_json::from_slice::(&message.payload) { + received = Some(record); + break; + } + } + if received.is_some() { + break; + } + } + sleep(Duration::from_millis(POLL_INTERVAL_MS)).await; + } + let received = received.expect("NUMERIC tracking row should be delivered"); + assert_eq!( + received["data"]["tracking_value"], + serde_json::json!(TRACKING_VALUE), + "NUMERIC payload should preserve its exact decimal representation" + ); + + let mut remaining_rows = fixture.count_rows(&pool).await; + for _ in 0..POLL_ATTEMPTS { + if remaining_rows == 0 { + break; + } + sleep(Duration::from_millis(POLL_INTERVAL_MS)).await; + remaining_rows = fixture.count_rows(&pool).await; + } + assert_eq!( + remaining_rows, 0, + "Exact NUMERIC tracking boundary should allow ACK cleanup" + ); + + pool.close().await; +} + +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/postgres/source.toml")), + seed = seeds::connector_stream +)] +async fn given_numeric_nan_when_source_polls_should_deliver_and_clean_up_row( + harness: &TestHarness, + fixture: PostgresSourceNumericTrackingFixture, +) { + const TRACKING_VALUE: &str = "NaN"; + + let client = harness.root_client().await.unwrap(); + let pool = fixture.create_pool().await.expect("Failed to create pool"); + fixture.create_table(&pool).await; + fixture.insert_row(&pool, TRACKING_VALUE).await; + + let stream_id: Identifier = seeds::names::STREAM.try_into().unwrap(); + let topic_id: Identifier = seeds::names::TOPIC.try_into().unwrap(); + let consumer_id: Identifier = "numeric_nan_consumer".try_into().unwrap(); + let mut received = None; + + for _ in 0..POLL_ATTEMPTS { + if let Ok(polled) = client + .poll_messages( + &stream_id, + &topic_id, + None, + &Consumer::new(consumer_id.clone()), + &PollingStrategy::next(), + 1, + true, + ) + .await + { + for message in polled.messages { + if let Ok(record) = serde_json::from_slice::(&message.payload) { + received = Some(record); + break; + } + } + if received.is_some() { + break; + } + } + sleep(Duration::from_millis(POLL_INTERVAL_MS)).await; + } + + let received = received.expect("NUMERIC NaN tracking row should be delivered"); + assert_eq!( + received["data"]["tracking_value"], + serde_json::json!(TRACKING_VALUE), + "NUMERIC NaN should remain a string in the payload" + ); + + let mut remaining_rows = fixture.count_rows(&pool).await; + for _ in 0..POLL_ATTEMPTS { + if remaining_rows == 0 { + break; + } + sleep(Duration::from_millis(POLL_INTERVAL_MS)).await; + remaining_rows = fixture.count_rows(&pool).await; + } + assert_eq!( + remaining_rows, 0, + "NUMERIC NaN boundary should allow ACK cleanup" + ); + + pool.close().await; +} + #[iggy_harness( server(connectors_runtime(config_path = "tests/connectors/postgres/source.toml")), seed = seeds::connector_stream diff --git a/core/integration/tests/connectors/postgres/postgres_source_cdc.rs b/core/integration/tests/connectors/postgres/postgres_source_cdc.rs index b9a485e832..26abb8f7c7 100644 --- a/core/integration/tests/connectors/postgres/postgres_source_cdc.rs +++ b/core/integration/tests/connectors/postgres/postgres_source_cdc.rs @@ -15,13 +15,18 @@ // specific language governing permissions and limitations // under the License. -use super::{POLL_ATTEMPTS, POLL_INTERVAL_MS}; +use super::{ + API_KEY, DEFAULT_SLOT, POLL_ATTEMPTS, POLL_INTERVAL_MS, SOURCE_KEY, source_stats, + wait_for_source_errors, wait_for_source_status, +}; use crate::connectors::create_test_messages; -use crate::connectors::fixtures::{PostgresOps, PostgresSourceCdcFixture, PostgresSourceOps}; +use crate::connectors::fixtures::{ + PostgresOps, PostgresSourceCdcFixture, PostgresSourceCdcSlowPollFixture, PostgresSourceOps, +}; use iggy::prelude::IggyClient; use iggy_common::MessageClient; use iggy_common::{Consumer, Identifier, PollingStrategy}; -use iggy_connector_sdk::api::{ConnectorStatus, SourceInfoResponse}; +use iggy_connector_sdk::api::ConnectorStatus; use integration::harness::seeds; use integration::iggy_harness; use reqwest::Client; @@ -29,10 +34,6 @@ use serde::Deserialize; use std::time::Duration; use tokio::time::sleep; -const API_KEY: &str = "test-api-key"; -const SOURCE_KEY: &str = "postgres"; -const DEFAULT_SLOT: &str = "iggy_slot"; - #[derive(Debug, Deserialize)] struct CdcRecord { table_name: String, @@ -46,9 +47,28 @@ async fn poll_cdc_records( topic_id: &Identifier, consumer_id: &Identifier, want: usize, +) -> Vec { + poll_cdc_records_with_attempts( + client, + stream_id, + topic_id, + consumer_id, + want, + POLL_ATTEMPTS, + ) + .await +} + +async fn poll_cdc_records_with_attempts( + client: &IggyClient, + stream_id: &Identifier, + topic_id: &Identifier, + consumer_id: &Identifier, + want: usize, + attempts: usize, ) -> Vec { let mut received = Vec::new(); - for _ in 0..POLL_ATTEMPTS { + for _ in 0..attempts { if let Ok(polled) = client .poll_messages( stream_id, @@ -75,6 +95,31 @@ async fn poll_cdc_records( received } +async fn slot_contains_change(pool: &sqlx::PgPool, expected_value: &str) -> bool { + for attempt in 0..POLL_ATTEMPTS { + match sqlx::query_scalar::<_, String>( + "SELECT data FROM pg_logical_slot_peek_changes($1, NULL, NULL)", + ) + .bind(DEFAULT_SLOT) + .fetch_all(pool) + .await + { + Ok(changes) => { + return changes.iter().any(|change| change.contains(expected_value)); + } + Err(sqlx::Error::Database(ref database_error)) + if attempt + 1 < POLL_ATTEMPTS + && database_error.code().as_deref() == Some(PG_OBJECT_IN_USE) => + { + sleep(Duration::from_millis(POLL_INTERVAL_MS)).await; + } + Err(error) => panic!("CDC replication slot should be readable: {error}"), + } + } + + panic!("CDC replication slot remained active after {POLL_ATTEMPTS} attempts"); +} + // End-to-end CDC coverage against a real wal_level=logical container: // INSERT, UPDATE, PK-changing UPDATE, DELETE, a rolled-back transaction // (must produce nothing), an untracked table (must be filtered out), a @@ -231,25 +276,264 @@ async fn cdc_source_captures_insert_update_delete( pool.close().await; } -async fn wait_for_source_status( - http: &Client, - api_url: &str, - expected: ConnectorStatus, -) -> SourceInfoResponse { +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/postgres/source.toml")), + seed = seeds::connector_stream +)] +async fn idle_cdc_source_advances_slot_to_current_wal( + harness: &TestHarness, + fixture: PostgresSourceCdcFixture, +) { + let state_path = harness + .connectors_runtime() + .expect("connectors runtime") + .state_path() + .join("source_postgres.state"); + let pool = fixture.create_pool().await.expect("Failed to create pool"); + fixture.create_table(&pool).await; + + let api_url = harness + .connectors_runtime() + .expect("connectors runtime") + .http_url(); + wait_for_source_status(&Client::new(), &api_url, ConnectorStatus::Running).await; + + sqlx::query("CHECKPOINT") + .execute(&pool) + .await + .expect("Failed to generate WAL without a logical table change"); + let target_lsn: String = sqlx::query_scalar("SELECT pg_current_wal_flush_lsn()::text") + .fetch_one(&pool) + .await + .expect("Failed to read current WAL flush LSN"); + for _ in 0..POLL_ATTEMPTS { - if let Ok(resp) = http - .get(format!("{api_url}/sources/{SOURCE_KEY}")) - .header("api-key", API_KEY) - .send() - .await - && let Ok(info) = resp.json::().await - && info.status == expected - { - return info; + let reached: bool = sqlx::query_scalar( + "SELECT confirmed_flush_lsn >= $2::pg_lsn FROM pg_replication_slots WHERE slot_name = $1", + ) + .bind(DEFAULT_SLOT) + .bind(&target_lsn) + .fetch_one(&pool) + .await + .expect("Failed to read replication slot position"); + if reached { + assert!( + !state_path.exists(), + "Idle CDC polling should not rewrite an unchanged checkpoint" + ); + pool.close().await; + return; + } + sleep(Duration::from_millis(POLL_INTERVAL_MS)).await; + } + + panic!("Idle CDC slot did not advance to WAL flush LSN {target_lsn}"); +} + +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/postgres/source.toml")), + seed = seeds::connector_stream +)] +async fn given_cdc_change_when_iggy_crashes_should_advance_slot_only_after_redelivery( + harness: &mut TestHarness, + fixture: PostgresSourceCdcFixture, +) { + let pool = fixture.create_pool().await.expect("Failed to create pool"); + fixture.create_table(&pool).await; + + harness + .server_mut() + .stop_dependents() + .expect("Failed to stop connectors runtime"); + harness + .server_mut() + .connectors_runtime_mut() + .expect("connectors runtime") + .set_iggy_connection_options("reconnection_retries=0"); + harness + .server_mut() + .start_dependents() + .await + .expect("Failed to restart connectors runtime"); + + let api_url = harness + .connectors_runtime() + .expect("connectors runtime") + .http_url(); + let http = Client::new(); + wait_for_source_status(&http, &api_url, ConnectorStatus::Running).await; + let errors_before_failure = source_stats(&http, &api_url) + .await + .expect("PostgreSQL source stats should be present") + .errors; + harness.kill_node(0).expect("Failed to kill Iggy server"); + + let [expected] = create_test_messages(1).try_into().unwrap(); + fixture + .insert_row( + &pool, + expected.id as i32, + &expected.name, + expected.count as i32, + expected.amount, + expected.active, + expected.timestamp, + ) + .await; + + let failed_source = wait_for_source_errors(&http, &api_url, errors_before_failure + 2).await; + assert_eq!(failed_source.status, ConnectorStatus::Error); + assert!( + slot_contains_change(&pool, &expected.name).await, + "NACKed CDC change must remain available in the replication slot" + ); + + harness + .server_mut() + .stop_dependents() + .expect("Failed to stop connectors runtime"); + harness + .restart_node(0) + .expect("Failed to restart Iggy server"); + harness + .server_mut() + .connectors_runtime_mut() + .expect("connectors runtime") + .clear_iggy_connection_options(); + harness + .server_mut() + .start_dependents() + .await + .expect("Failed to restart connectors runtime"); + + let client = harness.root_client().await.unwrap(); + let stream_id: Identifier = seeds::names::STREAM.try_into().unwrap(); + let topic_id: Identifier = seeds::names::TOPIC.try_into().unwrap(); + let consumer_id: Identifier = "cdc_send_failure_consumer".try_into().unwrap(); + let received = poll_cdc_records(&client, &stream_id, &topic_id, &consumer_id, 1).await; + + assert_eq!(received.len(), 1, "CDC change should be redelivered"); + assert_eq!(received[0].operation_type, "INSERT"); + assert_eq!(received[0].data["id"], serde_json::json!(expected.id)); + + let mut change_remains = slot_contains_change(&pool, &expected.name).await; + for _ in 0..POLL_ATTEMPTS { + if !change_remains { + break; } sleep(Duration::from_millis(POLL_INTERVAL_MS)).await; + change_remains = slot_contains_change(&pool, &expected.name).await; } - panic!("Source connector did not reach {expected:?} status in time"); + assert!( + !change_remains, + "ACKed CDC change should be consumed from the replication slot" + ); + wait_for_source_status(&http, &api_url, ConnectorStatus::Running).await; + + pool.close().await; +} + +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/postgres/source.toml")), + seed = seeds::connector_stream +)] +async fn given_delivery_failure_when_iggy_restarts_should_redeliver_cdc_without_runtime_restart( + harness: &mut TestHarness, + fixture: PostgresSourceCdcSlowPollFixture, +) { + const REDELIVERY_ATTEMPTS: usize = POLL_ATTEMPTS * 3; + + let pool = fixture.create_pool().await.expect("Failed to create pool"); + fixture.create_table(&pool).await; + + harness + .server_mut() + .stop_dependents() + .expect("Failed to stop connectors runtime"); + harness + .server_mut() + .connectors_runtime_mut() + .expect("connectors runtime") + .set_iggy_connection_options("reconnection_retries=0"); + harness + .server_mut() + .start_dependents() + .await + .expect("Failed to restart connectors runtime"); + + let api_url = harness + .connectors_runtime() + .expect("connectors runtime") + .http_url(); + let http = Client::new(); + wait_for_source_status(&http, &api_url, ConnectorStatus::Running).await; + let errors_before_failure = source_stats(&http, &api_url) + .await + .expect("PostgreSQL source stats should be present") + .errors; + + harness.kill_node(0).expect("Failed to kill Iggy server"); + + let [expected] = create_test_messages(1).try_into().unwrap(); + fixture + .insert_row( + &pool, + expected.id as i32, + &expected.name, + expected.count as i32, + expected.amount, + expected.active, + expected.timestamp, + ) + .await; + + let failed_source = wait_for_source_errors(&http, &api_url, errors_before_failure + 1).await; + assert_eq!(failed_source.status, ConnectorStatus::Error); + assert!( + slot_contains_change(&pool, &expected.name).await, + "NACKed CDC change must remain available in the replication slot" + ); + + harness + .restart_node(0) + .expect("Failed to restart only the Iggy server"); + + let client = harness.root_client().await.unwrap(); + let stream_id: Identifier = seeds::names::STREAM.try_into().unwrap(); + let topic_id: Identifier = seeds::names::TOPIC.try_into().unwrap(); + let consumer_id: Identifier = "cdc_nack_survival_consumer".try_into().unwrap(); + let received = poll_cdc_records_with_attempts( + &client, + &stream_id, + &topic_id, + &consumer_id, + 1, + REDELIVERY_ATTEMPTS, + ) + .await; + + assert_eq!(received.len(), 1, "CDC change should be redelivered"); + assert_eq!(received[0].operation_type, "INSERT"); + assert_eq!(received[0].data["id"], serde_json::json!(expected.id)); + + let mut change_remains = slot_contains_change(&pool, &expected.name).await; + for _ in 0..REDELIVERY_ATTEMPTS { + if !change_remains { + break; + } + sleep(Duration::from_millis(POLL_INTERVAL_MS)).await; + change_remains = slot_contains_change(&pool, &expected.name).await; + } + assert!( + !change_remains, + "ACKed CDC change should be consumed from the replication slot" + ); + wait_for_source_status(&http, &api_url, ConnectorStatus::Running).await; + + pool.close().await; } async fn get_active_source_config(http: &Client, api_url: &str) -> serde_json::Value { @@ -340,8 +624,8 @@ async fn delete_source_config_version(http: &Client, api_url: &str, version: u64 ); } -// The connector calls pg_logical_slot_get_changes on a fixed poll interval and -// briefly holds the slot active during each call. A drop landing in that window +// The connector peeks changes and advances the slot on a fixed poll interval. +// Both operations briefly hold the slot active. A drop landing in that window // gets ERROR 55006 (slot is active for PID ...), so retry past transient hits // instead of dropping while the poller is guaranteed stopped. const PG_OBJECT_IN_USE: &str = "55006"; @@ -376,11 +660,9 @@ async fn drop_replication_slot_retrying(pool: &sqlx::PgPool, slot: &str) { // connector that silently drops every change or emits wrong data - the // same silent-death shape as the slot mismatch above. Config is fixed // one field at a time until restart succeeds and CDC resumes. -// 3. Changes written while the connector is down (the slot retains WAL -// regardless of consumer state) - not the at-least-once crash window -// where the slot has already been consumed but send/state-persist -// hasn't happened yet. That gap remains open until the slot-peek/LSN -// work lands. +// 3. Changes written while the connector is down. The slot retains WAL +// regardless of consumer state, then the connector peeks and advances +// it only after Iggy acknowledges the recovered batch. #[iggy_harness( server(connectors_runtime(config_path = "tests/connectors/postgres/source_cdc_restart.toml")), seed = seeds::connector_stream diff --git a/core/integration/tests/connectors/postgres/restart.rs b/core/integration/tests/connectors/postgres/restart.rs index fd226785a5..bd58d3547c 100644 --- a/core/integration/tests/connectors/postgres/restart.rs +++ b/core/integration/tests/connectors/postgres/restart.rs @@ -15,7 +15,7 @@ // specific language governing permissions and limitations // under the License. -use super::{POLL_ATTEMPTS, POLL_INTERVAL_MS, TEST_MESSAGE_COUNT}; +use super::{API_KEY, POLL_ATTEMPTS, POLL_INTERVAL_MS, TEST_MESSAGE_COUNT}; use crate::connectors::fixtures::{PostgresOps, PostgresSinkFixture}; use crate::connectors::{TestMessage, create_test_messages}; use bytes::Bytes; @@ -29,7 +29,6 @@ use reqwest::Client; use std::time::Duration; use tokio::time::sleep; -const API_KEY: &str = "test-api-key"; const SINK_TABLE: &str = "iggy_messages"; const SINK_KEY: &str = "postgres"; From e0a8196efd7238a98ab9f4228ac8fb2455b5f408 Mon Sep 17 00:00:00 2001 From: Jorge Polanco <55784702+Jorge-Polanco-Roque@users.noreply.github.com> Date: Mon, 14 Sep 2026 07:41:14 -0600 Subject: [PATCH 131/182] feat(python): expose update_user options (#4173) Closes #4164. --------- Co-authored-by: Piotr Gankiewicz --- foreign/python/apache_iggy.pyi | 4 ++++ foreign/python/src/client.rs | 20 +++++++++-------- foreign/python/tests/test_user.py | 36 +++++++++++++++++++++++++++++++ 3 files changed, 51 insertions(+), 9 deletions(-) diff --git a/foreign/python/apache_iggy.pyi b/foreign/python/apache_iggy.pyi index e628a7c4d3..579fca81c7 100644 --- a/foreign/python/apache_iggy.pyi +++ b/foreign/python/apache_iggy.pyi @@ -1246,6 +1246,7 @@ class IggyClient: user_id: builtins.str | builtins.int, username: builtins.str | None = None, status: UserStatus | None = None, + options: builtins.dict[builtins.str, builtins.str] | None = None, ) -> collections.abc.Awaitable[None]: r""" Update a user by unique ID or username. @@ -1254,6 +1255,9 @@ class IggyClient: user_id: User identifier as `str | int`. username: New username as `str | None`; unchanged when `None`. status: New status as `UserStatus | None`; unchanged when `None`. + options: Reserved for future use. Additional option keys as + `dict[str, str] | None`, forwarded to the server. No user update + option key exists yet, so a current server rejects every key. Returns: An awaitable that resolves to `None` when the user is updated. diff --git a/foreign/python/src/client.rs b/foreign/python/src/client.rs index d75f7c350b..4c5799c17f 100644 --- a/foreign/python/src/client.rs +++ b/foreign/python/src/client.rs @@ -354,6 +354,9 @@ impl IggyClient { /// user_id: User identifier as `str | int`. /// username: New username as `str | None`; unchanged when `None`. /// status: New status as `UserStatus | None`; unchanged when `None`. + /// options: Reserved for future use. Additional option keys as + /// `dict[str, str] | None`, forwarded to the server. No user update + /// option key exists yet, so a current server rejects every key. /// /// Returns: /// An awaitable that resolves to `None` when the user is updated. @@ -361,7 +364,7 @@ impl IggyClient { /// Raises: /// ValueError: If a string identifier is invalid. /// RuntimeError: If the request fails. - #[pyo3(signature = (user_id, username=None, status=None))] + #[pyo3(signature = (user_id, username=None, status=None, options=None))] #[gen_stub(override_return_type(type_repr="collections.abc.Awaitable[None]", imports=("collections.abc")))] fn update_user<'a>( &self, @@ -369,20 +372,19 @@ impl IggyClient { user_id: PyIdentifier, #[gen_stub(override_type(type_repr = "builtins.str | None"))] username: Option, #[gen_stub(override_type(type_repr = "UserStatus | None"))] status: Option, + #[gen_stub(override_type(type_repr = "builtins.dict[builtins.str, builtins.str] | None"))] + options: Option>, ) -> PyResult> { let user_id = Identifier::try_from(user_id)?; let status = status.map(UserStatus::from); + let update_options = UserUpdateOptions { + raw: options.unwrap_or_default(), + }; let inner = self.inner.clone(); future_into_py(py, async move { inner - .update_user( - &user_id, - username.as_deref(), - status, - // Users have no option keys yet. - &UserUpdateOptions::default(), - ) + .update_user(&user_id, username.as_deref(), status, &update_options) .await .map_err(to_runtime_error)?; Ok(()) @@ -590,7 +592,7 @@ impl IggyClient { /// `manage_streams` or per-stream `manage_stream` permission, the /// stream does not exist, the new name is invalid or already used, or /// the request fails. - #[pyo3(signature = (stream_id, name, options = None))] + #[pyo3(signature = (stream_id, name, options=None))] #[gen_stub(override_return_type(type_repr="collections.abc.Awaitable[None]", imports=("collections.abc")))] fn update_stream<'a>( &self, diff --git a/foreign/python/tests/test_user.py b/foreign/python/tests/test_user.py index a15b714b1d..f6a328ca1f 100644 --- a/foreign/python/tests/test_user.py +++ b/foreign/python/tests/test_user.py @@ -361,6 +361,42 @@ async def test_update_username_and_status_together( await iggy_client.delete_user(created.id) + @pytest.mark.asyncio + async def test_update_user_with_empty_options_succeeds( + self, iggy_client: IggyClient, unique_name + ): + """Test update_user accepts an empty options map.""" + username, password = unique_credentials(unique_name) + new_username = unique_name(max_bytes=MAX_USERNAME_BYTES) + created = await iggy_client.create_user(username, password) + + await iggy_client.update_user(created.id, username=new_username, options={}) + + user = await iggy_client.get_user(created.id) + assert user is not None + assert user.username == new_username + + await iggy_client.delete_user(created.id) + + @pytest.mark.asyncio + async def test_update_user_forwards_options( + self, iggy_client: IggyClient, unique_name + ): + """Test update_user forwards option keys to the server.""" + username, password = unique_credentials(unique_name) + created = await iggy_client.create_user(username, password) + + with pytest.raises(RuntimeError) as rejection: + await iggy_client.update_user( + created.id, + options={"unknown": "value"}, + ) + + # Binary transports carry the code alone, so the key itself is empty. + assert "Unsupported option key" in str(rejection.value) + + await iggy_client.delete_user(created.id) + @pytest.mark.asyncio async def test_update_user_with_no_fields_is_a_noop( self, iggy_client: IggyClient, unique_name From b68a506be65b0a559db522106f9f87875f63686a Mon Sep 17 00:00:00 2001 From: Hubert Gruszecki Date: Mon, 14 Sep 2026 16:56:23 +0200 Subject: [PATCH 132/182] fix(connectors): validate sink writes and startup readiness (#4154) --- Cargo.lock | 2 + .../connectors/quickwit_sink.toml | 2 +- .../sinks/clickhouse_sink/README.md | 48 ++-- .../sinks/clickhouse_sink/config.toml | 13 +- .../sinks/clickhouse_sink/src/binary.rs | 124 ++++++++- core/connectors/sinks/delta_sink/README.md | 17 +- core/connectors/sinks/delta_sink/config.toml | 2 +- core/connectors/sinks/doris_sink/README.md | 26 +- core/connectors/sinks/doris_sink/config.toml | 1 - core/connectors/sinks/doris_sink/src/lib.rs | 5 +- .../sinks/elasticsearch_sink/Cargo.toml | 3 + .../sinks/elasticsearch_sink/README.md | 4 +- .../sinks/elasticsearch_sink/src/lib.rs | 239 ++++++++++++++++-- core/connectors/sinks/iceberg_sink/README.md | 69 +++-- core/connectors/sinks/influxdb_sink/README.md | 89 +++++-- .../connectors/sinks/influxdb_sink/src/lib.rs | 58 ++++- .../sinks/influxdb_sink/src/protocol.rs | 30 +-- .../sinks/meilisearch_sink/README.md | 58 +++-- core/connectors/sinks/mongodb_sink/README.md | 74 ++++-- core/connectors/sinks/mongodb_sink/src/lib.rs | 9 - core/connectors/sinks/postgres_sink/README.md | 107 +++++++- .../connectors/sinks/postgres_sink/src/lib.rs | 31 ++- core/connectors/sinks/quickwit_sink/README.md | 25 +- .../sinks/quickwit_sink/config.toml | 2 +- .../connectors/sinks/quickwit_sink/src/lib.rs | 200 +++++++++++++-- core/connectors/sinks/s3_sink/Cargo.toml | 1 + core/connectors/sinks/s3_sink/README.md | 67 +++-- core/connectors/sinks/s3_sink/src/client.rs | 42 ++- core/connectors/sinks/s3_sink/src/sink.rs | 116 ++++++++- .../connectors/sinks/surrealdb_sink/README.md | 103 ++++++-- .../sinks/surrealdb_sink/config.toml | 1 - .../sinks/surrealdb_sink/src/lib.rs | 9 +- .../elasticsearch/elasticsearch_sink.rs | 79 ++++++ .../tests/connectors/fixtures/mod.rs | 4 +- .../tests/connectors/fixtures/mongodb/sink.rs | 2 - .../connectors/fixtures/surrealdb/mod.rs | 4 +- .../connectors/fixtures/surrealdb/sink.rs | 52 +++- .../connectors/influxdb/influxdb_sink.rs | 6 +- .../tests/connectors/mongodb/mongodb_sink.rs | 4 +- .../connectors/surrealdb/surrealdb_sink.rs | 33 ++- 40 files changed, 1415 insertions(+), 346 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 0a71fdfffc..0580a25af6 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -7161,6 +7161,7 @@ dependencies = [ "simd-json", "tokio", "tracing", + "wiremock", ] [[package]] @@ -7465,6 +7466,7 @@ dependencies = [ "simd-json", "tokio", "tracing", + "wiremock", ] [[package]] diff --git a/core/connectors/runtime/example_config/connectors/quickwit_sink.toml b/core/connectors/runtime/example_config/connectors/quickwit_sink.toml index 27a3c75607..d13d02f20d 100644 --- a/core/connectors/runtime/example_config/connectors/quickwit_sink.toml +++ b/core/connectors/runtime/example_config/connectors/quickwit_sink.toml @@ -58,7 +58,7 @@ verbose_logging = false # max_retries = 3 # retry_delay = "1s" # retry_max_delay = "5s" -# Total readiness probes including the first; 1 disables retries. +# Total attempts per readiness check including the first; 1 disables retries. # max_open_retries = 10 # open_retry_max_delay = "30s" # timeout = "30s" diff --git a/core/connectors/sinks/clickhouse_sink/README.md b/core/connectors/sinks/clickhouse_sink/README.md index 52fed58bea..941d3a6375 100644 --- a/core/connectors/sinks/clickhouse_sink/README.md +++ b/core/connectors/sinks/clickhouse_sink/README.md @@ -51,23 +51,16 @@ verbose_logging = false | `password` | string | `""` | ClickHouse password | | `insert_format` | string | `"json_each_row"` | Insert format: `json_each_row`, `row_binary`, or `string` | | `string_format` | string | `"json_each_row"` | ClickHouse format for `string` mode: `json_each_row`, `csv`, or `tsv` | -| `timeout_seconds` | u64 | `30` | HTTP request timeout | -| `max_retries` | u32 | `3` | Max retry attempts on transient errors | -| `retry_delay` | u64 | `1` | Delay between retries, in seconds | +| `timeout_seconds` | u64 | `30` | HTTP request timeout in seconds | +| `max_retries` | u32 | `3` | Total attempts for startup requests and transient insert errors; at least one even when `0` | +| `retry_delay` | u64 | `1` | Base for exponential retry delay, in seconds | | `verbose_logging` | bool | `false` | Log inserts at info level instead of debug | -> **TODO:** `database` and `table` values are interpolated directly into SQL. Currently only -> single quotes are escaped; backslashes pass through unchanged, which can misparse string -> literals if a value ends with `\`. A future improvement should validate both fields against a -> strict allowlist (`^[A-Za-z_][A-Za-z0-9_]*$`) at config load and escape backslashes in SQL -> string literals. Deferred because these sinks run in operator-controlled environments where -> config values are trusted. - ## Insert Formats ### `json_each_row` (Default) -Accepts messages with a `Payload::Json` payload. Each message is sent as a JSON object on its own line using ClickHouse's `JSONEachRow` format. ClickHouse handles type coercion from the JSON values to the column types, so the table can have any schema. +Accepts messages with a `Payload::Json` payload. Each payload is serialized on its own line using ClickHouse's `JSONEachRow` format. Send JSON objects whose fields and values are compatible with the existing table and its ClickHouse input settings. The connector does not validate JSON rows against the table schema before sending them. ```toml [plugin_config] @@ -80,9 +73,9 @@ insert_format = "json_each_row" Accepts messages with a `Payload::Json` payload. At startup the connector fetches the table schema from `system.columns` and validates that all column types are supported. Messages are then serialised to ClickHouse's `RowBinaryWithDefaults` binary format, which is more efficient than JSON for large volumes. -Requires ClickHouse 23.7 or newer, when `RowBinaryWithDefaults` was introduced. Older servers reject the format; use `json_each_row` instead. +Requires [ClickHouse 23.7 or newer](https://presentations.clickhouse.com/2023-release-23.7/index.html), when `RowBinaryWithDefaults` was introduced. Older servers reject the format; use `json_each_row` instead. -The table must already exist. Columns with an ordinary `DEFAULT` expression can be omitted from the message — the connector emits a `0x01` prefix byte to signal that the default should be used. `MATERIALIZED`, `ALIAS`, and `EPHEMERAL` columns are not insertable and are dropped from the schema entirely. +The table must already exist. Columns with an ordinary `DEFAULT` expression can be omitted from the message - the connector emits a `0x01` prefix byte to signal that the default should be used. An explicit JSON `null` requires a nullable column and is stored as `NULL`, even when that column has a default. Missing columns without defaults must be nullable. The connector excludes `MATERIALIZED`, `ALIAS`, and `EPHEMERAL` columns from its insert schema. The schema is captured once at startup and never refreshed. Do not `ALTER TABLE` the target while the connector runs. See [Schema changes while running](#schema-changes-while-running). @@ -99,7 +92,7 @@ insert_format = "row_binary" ### `string` -Accepts messages with a `Payload::Text` payload and passes them through to ClickHouse without modification. Use `string_format` to tell ClickHouse which format to expect. +Accepts messages with a `Payload::Text` payload and appends a newline to each payload that does not already end with one. Set the stream `schema = "text"` and use `string_format` to tell ClickHouse which format to expect. ```toml [plugin_config] @@ -170,17 +163,18 @@ string_format = "csv" ## Reliability -The connector retries failed inserts up to `max_retries` times, starting from `retry_delay`. Retryable HTTP errors and network/timeout errors both back off exponentially with full jitter, so instances spread their retries instead of hammering a recovering server in lockstep. Non-retryable errors fail immediately. The startup ping and schema fetch use the same jittered backoff. +Insert requests retry HTTP 408, 429, and 5xx responses, plus network and timeout errors. Other unsuccessful HTTP statuses fail immediately. `max_retries` is the total attempt limit, with at least one attempt even when set to `0`. Before retry number `n` (starting at 1), the delay is sampled from zero through `min(retry_delay * 2^n, 60)` seconds. +With the defaults, there are at most three attempts and the first retry waits between zero and two seconds. The startup ping and, in `row_binary` mode, schema fetch use the same limit and backoff but retry every error. On shutdown the connector logs the total number of messages processed. ### Bad rows in a batch -A message whose payload type does not match the chosen format (for example a text payload in JSON mode) is always skipped with a warning. The rest of the batch is still sent. +A message whose payload type does not match the chosen format (for example a text payload in JSON mode) is skipped with an error log. The rest of the batch is still sent. -The `rowbinary` format has one extra case. It turns each row into binary and writes it straight into the batch buffer, so a row that cannot be converted (a value that does not fit the target column) cannot be skipped cleanly — a half-written row would corrupt the rows after it. In that case the **whole batch fails** on the first bad row and is retried as a unit per the rules above. +The `row_binary` format fails the whole batch on the first JSON row whose values cannot be converted to the target columns. This occurs before any insert request, so the batch does not enter the plugin's insert retry loop. No partial binary row is sent. -If a single malformed row keeps failing, every retry of that batch will fail too. Fix or remove the bad message at the source, or switch to the `json` / `string` format, which skip bad rows instead of failing the batch. +A batch with no serializable payloads returns success without an insert. In `json_each_row` and `string` modes, ClickHouse still validates the submitted data and can reject an insert containing a malformed row. ### Schema changes while running @@ -188,7 +182,7 @@ In `row_binary` mode the table schema is fetched once at startup and cached for An `ALTER TABLE` that runs while the connector is live breaks that assumption. Adding, dropping, or reordering a column shifts the byte layout by one or more columns. Depending on how the shifted bytes decode, ClickHouse either rejects the batch as malformed or, worse, stores it silently with values landing in the wrong columns. Nothing detects this at runtime, so the corruption is easy to miss. -Only `row_binary` is affected. The `json_each_row` and `string` (`json_each_row` string) formats are self-describing and map values by field name, so they tolerate schema changes. +The `json_each_row` format, including `string` mode with `string_format = "json_each_row"`, maps values by field name, but changed column names, types, or constraints can still make inserts fail. The plain `CSV` and `TSV` string formats use the current table column order. Until the hardening below lands, treat the `row_binary` schema as fixed for the connector's lifetime: **restart the connector after any `ALTER TABLE`** on the target table so it re-fetches the schema. @@ -197,20 +191,16 @@ Two planned fixes remove the restriction: 1. **Explicit column list in the INSERT.** Emitting `INSERT INTO db.table (col1, col2, ...) FORMAT RowBinaryWithDefaults` binds the stream to column *names* instead of table position. ClickHouse then routes each value by name, applies `DEFAULT` for columns the connector omits, and returns a hard error (instead of silently corrupting rows) when a named column has been dropped or renamed. This makes added and reordered columns safe and turns the remaining drift into a visible failure. 2. **Refresh the schema on a failed insert.** When an insert fails with a data error, re-fetch the schema from `system.columns` and rebuild the column list before the batch is retried, letting the connector recover from a drop or rename on its own rather than failing every retry against a stale snapshot. -### Delivery semantics: at-least-once - -This connector provides **at-least-once** delivery — not exactly-once. Retries resend the full batch body without an `insert_deduplication_token`, so if the server applied a batch but the acknowledgement was lost in transit (network drop, timeout), the retry will insert the same rows again. +### Delivery semantics -**Affected table engines:** +The runtime uses consumer auto-commit and does not replay a failed sink batch. End-to-end at-least-once delivery is therefore not guaranteed; see [sink guide](https://iggy.apache.org/docs/connectors/sinks/sink#sample-implementation). -- `MergeTree` — no deduplication at all; duplicate rows will be stored. -- `ReplicatedMergeTree` — has implicit block-level deduplication based on the data checksum (controlled by `replicated_deduplication_window`, default 100 blocks), which will suppress duplicates in the common retry case as long as the window has not been exceeded. +Plugin retries resend the same batch without an `insert_deduplication_token`, so a lost acknowledgement can also produce duplicate rows. ClickHouse deduplication depends on the table engine, query settings, identical retry data, and the retained deduplication window: -If your workload cannot tolerate duplicate rows, either: +- `ReplicatedMergeTree` enables a deduplication log by default, controlled by `replicated_deduplication_window` and `replicated_deduplication_window_seconds`. +- Non-replicated `MergeTree` can also deduplicate when `non_replicated_deduplication_window` is positive; its default is zero. -1. Use a `ReplicatedMergeTree` table and keep `max_retries` low enough that retries stay within the deduplication window, or -2. Use a `CollapsingMergeTree` / `ReplacingMergeTree` and apply deduplication at query time, or -3. Accept duplicates at write time and deduplicate with `DISTINCT` or `GROUP BY` in your queries. +See [ClickHouse insert deduplication](https://clickhouse.com/docs/concepts/features/operations/insert/deduplicating-inserts-on-retries) for the settings and limits. A small retry count alone does not ensure that a retry remains within the window, because other inserts can evict its deduplication record. ## Testing diff --git a/core/connectors/sinks/clickhouse_sink/config.toml b/core/connectors/sinks/clickhouse_sink/config.toml index e8eb284828..3bb217cb7c 100644 --- a/core/connectors/sinks/clickhouse_sink/config.toml +++ b/core/connectors/sinks/clickhouse_sink/config.toml @@ -37,16 +37,13 @@ database = "default" username = "default" password = "" table = "events" -# TODO: validate database/table with a strict allowlist (^[A-Za-z_][A-Za-z0-9_]*$) and -# escape backslashes in SQL string literals (currently only ' is escaped). Deferred -# because sinks run in operator-controlled environments where config values are trusted. # Insert format: "json_each_row" (default), "row_binary", or "string" -# json_each_row — accepts Payload::Json; ClickHouse handles type coercion -# row_binary — accepts Payload::Json; connector validates + serialises +# json_each_row - accepts Payload::Json; ClickHouse handles type coercion +# row_binary - accepts Payload::Json; connector validates + serialises # to RowBinaryWithDefaults (table must exist, schema is # fetched at startup; fails if table has unsupported types) -# string — accepts Payload::Text; raw passthrough +# string - accepts Payload::Text; appends a newline when absent insert_format = "json_each_row" # Only relevant when insert_format = "string": @@ -54,6 +51,6 @@ insert_format = "json_each_row" # string_format = "csv" timeout_seconds = 30 -max_retries = 3 -retry_delay = 1 # seconds +max_retries = 3 # total attempts, including the initial request +retry_delay = 1 # base delay in seconds, before exponential backoff and jitter verbose_logging = false diff --git a/core/connectors/sinks/clickhouse_sink/src/binary.rs b/core/connectors/sinks/clickhouse_sink/src/binary.rs index 14af11ff04..227eba0e67 100644 --- a/core/connectors/sinks/clickhouse_sink/src/binary.rs +++ b/core/connectors/sinks/clickhouse_sink/src/binary.rs @@ -486,7 +486,7 @@ fn coerce_i64(value: &OwnedValue) -> Result { }), OwnedValue::Static(simd_json::StaticNode::F64(f)) => { let n = *f as i64; - if n as f64 != *f { + if !f.is_finite() || *f < i64::MIN as f64 || *f >= i64::MAX as f64 || n as f64 != *f { error!("Float {f} is not a finite whole number within i64 range"); return Err(Error::InvalidRecord); } @@ -512,7 +512,7 @@ fn coerce_u64(value: &OwnedValue) -> Result { }), OwnedValue::Static(simd_json::StaticNode::F64(f)) => { let n = *f as u64; - if n as f64 != *f { + if !f.is_finite() || *f < 0.0 || *f >= u64::MAX as f64 || n as f64 != *f { error!("Float {f} is not a finite whole number within u64 range"); return Err(Error::InvalidRecord); } @@ -579,7 +579,12 @@ fn coerce_decimal(value: &OwnedValue, precision: u8, scale: u8) -> Result= i128::MAX as f64 { + error!("Decimal overflow for Decimal({precision}, {scale})"); + return Err(Error::InvalidRecord); + } + Ok(scaled as i128) } other => { error!("Cannot coerce {other:?} to Decimal({precision}, {scale})"); @@ -608,6 +613,15 @@ fn parse_decimal_str(s: &str, scale: u8) -> Result { None => (s, ""), }; + if (int_str.is_empty() && frac_str.is_empty()) + || !int_str + .bytes() + .chain(frac_str.bytes()) + .all(|byte| byte.is_ascii_digit()) + { + return Err(()); + } + let int_val: i128 = if int_str.is_empty() { 0 } else { @@ -790,7 +804,12 @@ fn strip_timezone(s: &str) -> (&str, i64) { match (it.next(), it.next()) { (Some(h_s), Some(m_s)) => { if let (Ok(h), Ok(m)) = (h_s.parse::(), m_s.parse::()) { - Some(sign * (h * 3600 + m * 60)) + match (h.checked_mul(3600), m.checked_mul(60)) { + (Some(hours), Some(minutes)) => hours + .checked_add(minutes) + .and_then(|seconds| seconds.checked_mul(sign)), + _ => None, + } } else { None } @@ -874,6 +893,65 @@ mod tests { } // ── primitives ─────────────────────────────────────────────────────────── + #[test] + fn given_boundary_floats_when_serializing_numbers_should_preserve_values() { + for (column_type, value, expected) in [ + ( + ChType::Int64, + i64::MIN as f64, + i64::MIN.to_le_bytes().to_vec(), + ), + ( + ChType::UInt64, + (1u64 << 63) as f64, + (1u64 << 63).to_le_bytes().to_vec(), + ), + ( + ChType::Decimal(38, 0), + i128::MIN as f64, + i128::MIN.to_le_bytes().to_vec(), + ), + ] { + let mut buffer = Vec::new(); + serialize_value(&json_f64(value), &column_type, &mut buffer).unwrap(); + assert_eq!(buffer, expected, "value: {value}"); + } + } + + #[test] + fn given_overflowing_float_when_serializing_integer_should_reject() { + let mut accepted_values = Vec::new(); + for (column_type, value) in [ + (ChType::Int64, i64::MAX as f64), + (ChType::UInt64, u64::MAX as f64), + ] { + let mut row = simd_json::owned::Object::new(); + row.insert("value".into(), json_f64(value)); + let result = serialize_row( + &OwnedValue::Object(Box::new(row)), + &[col("value", column_type, false)], + &mut Vec::new(), + ); + if !matches!(result, Err(Error::InvalidRecord)) { + accepted_values.push(value); + } + } + assert!(accepted_values.is_empty(), "accepted: {accepted_values:?}"); + } + + #[test] + fn given_overflowing_float_when_serializing_decimal_should_reject() { + for value in [i128::MAX as f64, f64::MAX, -f64::MAX, f64::NAN] { + let mut buffer = Vec::new(); + let result = serialize_value(&json_f64(value), &ChType::Decimal(38, 0), &mut buffer); + assert!( + matches!(result, Err(Error::InvalidRecord)), + "value: {value}" + ); + assert!(buffer.is_empty()); + } + } + #[test] fn serialize_int32_little_endian() { let mut buf = vec![]; @@ -1307,6 +1385,19 @@ mod tests { assert_eq!(buf, 1i32.to_le_bytes()); } + #[test] + fn given_overflowing_timezone_when_serializing_datetime_should_reject() { + for offset in [format!("{}:00", i64::MAX), format!("00:{}", i64::MAX)] { + let value = format!("1970-01-01T00:00:00+{offset}"); + let result = + serialize_value(&json_str(&value), &ChType::DateTime64(3), &mut Vec::new()); + assert!( + matches!(result, Err(Error::InvalidRecord)), + "value: {value}" + ); + } + } + #[test] fn serialize_datetime_from_iso8601_utc_string() { let mut buf = vec![]; @@ -1370,6 +1461,31 @@ mod tests { } // ── decimal ────────────────────────────────────────────────────────────── + #[test] + fn given_multibyte_decimal_when_serializing_should_return_error() { + let value = format!("0.{}é", "1".repeat(37)); + let result = serialize_value(&json_str(&value), &ChType::Decimal(38, 2), &mut Vec::new()); + assert!(matches!(result, Err(Error::InvalidRecord))); + } + + #[test] + fn given_malformed_decimal_when_serializing_should_reject() { + let mut accepted_values = Vec::new(); + for value in [ + ".".into(), + "--1".into(), + "1.-2".into(), + format!("0.{}x", "1".repeat(38)), + ] { + let result = + serialize_value(&json_str(&value), &ChType::Decimal(38, 2), &mut Vec::new()); + if !matches!(result, Err(Error::InvalidRecord)) { + accepted_values.push(value); + } + } + assert!(accepted_values.is_empty(), "accepted: {accepted_values:?}"); + } + #[test] fn serialize_decimal32_scale2() { let mut buf = vec![]; diff --git a/core/connectors/sinks/delta_sink/README.md b/core/connectors/sinks/delta_sink/README.md index 84a9964c17..7a98642e49 100644 --- a/core/connectors/sinks/delta_sink/README.md +++ b/core/connectors/sinks/delta_sink/README.md @@ -8,6 +8,8 @@ The Delta Lake Sink Connector allows you to consume messages from Iggy topics an - **Intelligent type coercion** to match Delta table schemas (e.g. ISO 8601 strings to timestamps) - **Transactional writes** with atomic flush-and-commit operations +The table must already exist. The connector appends each successful nonempty batch in one Delta transaction and keeps its schema snapshot until restart. The plugin has no failed-batch retry loop; the Delta library can retry eligible commit conflicts and storage requests. Write or commit errors clear the writer buffers and return an error. The runtime uses consumer auto-commit and does not replay failed sink batches, so end-to-end at-least-once delivery is not guaranteed. + ## Configuration example ### Local filesystem @@ -38,7 +40,6 @@ table_uri = "az://my-container/delta-tables/users" storage_backend_type = "azure" azure_storage_account_name = "mystorageaccount" azure_storage_account_key = "account-key" -azure_storage_sas_token = "sas-token" azure_container_name = "my-container" ``` @@ -55,7 +56,7 @@ gcs_service_account_key = '{"type": "service_account", "project_id": "...", ...} ### Core -- **table_uri** (required): Path or URI to the Delta table. Supported schemes: `file://`, `s3://`, `az://`, `gs://`. +- **table_uri** (required): Absolute URI to an existing Delta table. Use `file:///...` for a local path; bare filesystem paths are not accepted. Supported schemes: `file://`, `s3://`, `az://`, `gs://`. - **storage_backend_type** (optional): The cloud storage backend to use. One of `"s3"`, `"azure"`, or `"gcs"`. Omit for local filesystem tables. ### AWS S3 @@ -65,12 +66,12 @@ Required when `storage_backend_type = "s3"`. - **aws_s3_access_key**: AWS access key ID. - **aws_s3_secret_key**: AWS secret access key. - **aws_s3_region**: AWS region (e.g. `us-east-1`). -- **aws_s3_endpoint_url**: S3 endpoint URL. Useful for S3-compatible services like MinIO. -- **aws_s3_allow_http**: Set to `true` to allow HTTP connections (for local development). +- **aws_s3_endpoint_url** (optional): S3 endpoint URL. Useful for S3-compatible services like MinIO. +- **aws_s3_allow_http** (optional, default `false`): Set to `true` to allow HTTP connections (for local development). ### Azure Blob Storage -Required when `storage_backend_type = "azure"`. +When `storage_backend_type = "azure"`, provide the account name, container name, and exactly one of the account key or SAS token. Providing both authentication fields is an error. - **azure_storage_account_name**: Azure storage account name. - **azure_storage_account_key**: Azure storage account key. @@ -87,6 +88,6 @@ Required when `storage_backend_type = "gcs"`. The connector automatically coerces JSON values to match the Delta table schema: -- **Timestamp fields**: ISO 8601 / RFC 3339 formatted strings (e.g. `"2021-11-11T22:11:58Z", "2021-11-11 22:11:58"`) are converted to microsecond timestamps. Numeric timestamps pass through unchanged. -- **String fields**: Non-string values (numbers, booleans, objects, arrays) are converted to their string representation. -- **Nested fields**: Coercions are applied recursively to nested structs and arrays. +- **Timestamp fields**: ISO 8601 / RFC 3339 formatted strings (e.g. `"2021-11-11T22:11:58Z", "2021-11-11 22:11:58"`) are converted to microsecond timestamps. Integer epoch-microsecond timestamps pass through unchanged. Space-separated timestamps without an offset are interpreted as UTC. Invalid timestamp strings fail the batch. +- **String fields**: Non-null, non-string values (numbers, booleans, objects, arrays) are converted to their string representation. +- **Nested fields**: Coercions cover nested structs, arrays of strings or timestamps, and arrays of structs. Nested arrays, maps, and variant columns pass through without these coercions. Nulls remain null. diff --git a/core/connectors/sinks/delta_sink/config.toml b/core/connectors/sinks/delta_sink/config.toml index 27a455593a..87d7e1f8b2 100644 --- a/core/connectors/sinks/delta_sink/config.toml +++ b/core/connectors/sinks/delta_sink/config.toml @@ -31,7 +31,7 @@ batch_length = 100 poll_interval = "5ms" consumer_group = "delta_sink_connector" -# table_uri: Path or URI of the Delta table to write to. +# table_uri: Absolute URI of an existing Delta table; bare paths are not accepted. # - Local: "file:///path/to/delta_table" # - S3: "s3://bucket/path/to/delta_table" # - Azure: "az://container/path/to/delta_table" diff --git a/core/connectors/sinks/doris_sink/README.md b/core/connectors/sinks/doris_sink/README.md index bf47169fb8..cc4bec4500 100644 --- a/core/connectors/sinks/doris_sink/README.md +++ b/core/connectors/sinks/doris_sink/README.md @@ -6,8 +6,8 @@ The Doris sink connector consumes JSON messages from Iggy streams and writes the - The target Doris **database and table must be pre-created** before enabling the sink. The connector never issues DDL. - `database` and `table` config values must match `[A-Za-z0-9_]+`. Anything else is rejected at startup with `Error::InvalidConfigValue` — this also prevents path traversal in the constructed `/api/{db}/{table}/_stream_load` URL. -- Messages must arrive with `Payload::Json` (i.e. the configured stream schema is `json`). If a non-JSON payload reaches the connector it logs at `error!` and aborts the whole poll; since the consumer offset is already committed at poll time the batch is not replayed — effectively silent data loss — so the upstream schema must be guaranteed JSON. (Under `schema = "json"` the SDK drops non-JSON before the connector sees it, so this abort is a defensive guard.) -- The Iggy message JSON shape must match the target table columns. Use the optional `columns` plugin setting if the column order differs from the JSON keys. +- Messages must arrive with `Payload::Json` (i.e. the configured stream schema is `json`). If a non-JSON payload reaches the connector it logs at `error!` and aborts the whole poll; since the consumer offset is already committed at poll time the batch is not replayed - data loss despite the error log - so the upstream schema must be guaranteed JSON. (Under `schema = "json"` the SDK drops non-JSON before the connector sees it, so this abort is a defensive guard.) +- The Iggy message JSON shape must match the target table columns. JSON fields map by name. Use the optional `columns` plugin setting for field mappings or derived expressions. ## How it works @@ -16,7 +16,8 @@ The Doris sink connector consumes JSON messages from Iggy streams and writes the - `hash16` is a single 64-bit blake3 hash computed over the *raw* (un-sanitized), length-prefixed `(label_prefix, table, stream, topic)` tuple. Identities that sanitize to the same string therefore get distinct labels, whether the collision is in the names (`events.v1` vs `events_v1`) or in two tenants' prefixes that truncate alike (`prod_events_us_east_1` vs `..._2`). Length prefixes prevent boundary-shift aliasing (`("ab","c")` ≠ `("a","bc")`). The target table participates because Doris labels are scoped to a database, not a table. - The total label is bounded under Doris's 128-char cap regardless of input length (worst case 120 chars). - - Doris dedupes loads by label inside its `label_keep_max_second` window. The in-request retry (step 6) re-PUTs a transiently-failed batch under the same label, so a prior attempt that actually landed (e.g. a `2xx` with a missing or unreadable body) is absorbed, not doubled. This protects **in-request retry only**: the runtime commits the offset before `consume()` runs and discards its return, so a failure outliving the retry budget or a crash mid-load is **at-most-once**. + - Doris dedupes loads while it retains their labels (see Operational guidance). The in-request retry (step 6) re-PUTs a transiently-failed batch under the same label, so a prior attempt that actually landed (e.g. a `2xx` with a missing or unreadable body) is absorbed, not doubled. + The runtime uses consumer auto-commit before `consume()` finishes and does not replay a failed poll, so a failure outliving the retry budget or a crash mid-load can lose data. Expired labels or changed chunk boundaries can also permit duplicate loads during a redrive. 3. It `PUT`s the batch to `{fe_url}/api/{database}/{table}/_stream_load` with HTTP Basic auth, `Expect: 100-continue`, `label: