diff --git a/Cargo.lock b/Cargo.lock index 7e146a36..c58647d7 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -883,8 +883,7 @@ dependencies = [ [[package]] name = "ant-protocol" version = "3.1.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cb1f2e060c4283457a718c094d5fd594efd1d4eeec7cc6ca3ea113df17cce382" +source = "git+https://github.com/WithAutonomi/ant-protocol?rev=ab459e52b3db19c0db15ed33a97fe6282dfd4ae0#ab459e52b3db19c0db15ed33a97fe6282dfd4ae0" dependencies = [ "blake3", "bytes", diff --git a/Cargo.toml b/Cargo.toml index baabb808..a1d0f97f 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -229,6 +229,13 @@ webrtc-direct = [ "saorsa-transport/webrtc-direct", ] +[patch.crates-io] +# The published 3.1.0 carries pointers (ADR-0016) but not the final state this +# change relies on (ADR-0018), so this is the only entry that has to leave the +# release baseline. A rev, not a branch, so the pin is immutable. Drop it once a +# release includes the companion change. Versions are bumped at release. +ant-protocol = { git = "https://github.com/WithAutonomi/ant-protocol", rev = "ab459e52b3db19c0db15ed33a97fe6282dfd4ae0" } + [profile.release] lto = true codegen-units = 1 diff --git a/docs/adr/ADR-0018-pointer-transfer-by-final-redirection.md b/docs/adr/ADR-0018-pointer-transfer-by-final-redirection.md new file mode 100644 index 00000000..efa664fd --- /dev/null +++ b/docs/adr/ADR-0018-pointer-transfer-by-final-redirection.md @@ -0,0 +1,319 @@ +# ADR-0018: Pointer ownership transfer by final redirection + +- **Status:** Proposed +- **Date:** 2026-09-29 +- **Decision owners:** Anselme (@grumbach) +- **Reviewers:** TBD +- **Supersedes:** none. Amends ADR-0016's merge rule at the final counter. +- **Superseded by:** none +- **Related:** ADR-0016 (pointers), ADR-0005 (repair quorum); V2-1354; WithAutonomi/ant-protocol#40, WithAutonomi/ant-node#239, WithAutonomi/ant-client#210 + +## Context + +ADR-0016 fixes a pointer's owner key at creation and offers handover only as +indirection: the owner points the pointer at a pointer the recipient owns. The +former owner keeps its key, so that is a revocable forwarding and not a sale. +The one thing that could make it stick is the counter running out, and under +ADR-0016 it does not: + +```text +1. larger counter +2. smaller target bytes +``` + +At `counter == u64::MAX` no counter is larger, but rule 2 still applies. A +former owner who handed the address over at `u64::MAX` can sign another state +at `u64::MAX` whose target sorts first — about two tries of grinding — and it +displaces the handover on every node, deterministically. ADR-0016 says so +("does not even freeze the pointer") and tells owners to migrate *before* the +final counter, which is exactly the revocable forwarding again. + +People want to hand pointers over — a name, an application's root, a published +handle — while every reader keeps using the same address. The address derives +from the owner key, so it cannot follow a new key. What the address *resolves +to* can. + +## Decision Drivers + +- The address readers use must not change. +- No new record type, field, message or storage format; nothing a node has to + interpret beyond the counter it already compares. +- Once a node holds a transfer, no arrival may move it off, the former + owner's included. +- Forks the owner can still make must be detectable by any reader, and must + not be able to take a transfer back from the nodes that hold it. +- Below the final counter nothing changes: ADR-0016's convergence argument + still holds there. + +## Considered Options + +1. **ADR-0016 as is.** Handover by revocable forwarding. Not a transfer. +2. **Certificates and epochs** (ADR-0016's withdrawn alternative): a transfer + certificate chain, an admission index per epoch and a 5-of-7 branch quorum. + Real re-keying, at the cost of a second record type, lineage walks, Sybil + exposure in ownership and a stuck-not-reversed failure mode under churn. +3. **Final state resolved by target, as today.** Grindable in about two tries. +4. **Final state resolved by payment time or order.** A pre-buy defeats it: + pay early, withhold, publish later. Clock trust besides. +5. **A final state is replaced by nothing; the first one a node takes is the + one it keeps.** Clients settle the owner's only remaining fork by the close + group's majority, and nodes look before taking one. + +## Decision + +We choose option 5. + +### Merge + +One rule ahead of ADR-0016's two: + +```text +0. a final state (counter == u64::MAX) is replaced by nothing +1. larger counter +2. smaller target bytes +``` + +Below the final counter the order is ADR-0016's, total and deterministic. At +it, two *different* final states are unordered, so each node keeps whichever it +took first. `replaces` stays a strict partial order — never both ways round, +transitive — and every pair of distinct states is ordered except two final +ones. + +This lives in `ant_protocol::pointer::PointerState::replaces`, so the node's +store, its admission gate, fresh offers, repair and hints all take it without a +line of their own. + +### What this changes in ADR-0016 + +ADR-0016 is left as written. Where the two disagree, this ADR holds, at the +final counter only: + +- ADR-0016's merge rule gains rule 0 above. +- "An owner who jumps straight to `u64::MAX` ... does not even freeze the + pointer" no longer holds: a state at `u64::MAX` is the last one the pointer + holds on every node that takes it. +- "Ownership cannot change ... a revocable forwarding state, not a sale" no + longer holds for a forwarding signed at the final counter: no node that holds + it gives it up. The owner key still cannot change, and a forwarding below the + final counter is still revocable. + +### Transfer + +A transfer is the final state whose target is another pointer: + +```text +counter = u64::MAX +target = (Pointer, recipient) recipient = the new owner's pointer address +``` + +`Pointer::transfer_to` signs it; `transferred_to()` recognises it. A final state +with any other target freezes the pointer — nobody receives it. A pointer target +below the final counter is still ADR-0016's revocable forwarding. + +Readers already follow pointer targets (`pointer_resolve`), so a transferred +address resolves through the recipient's pointer, which only the recipient can +move. The recipient's pointer should serve that one address: a pointer's +address derives from its owner key, so everything handed to the same recipient +pointer resolves to the same place. A recipient uses a fresh key per received +pointer. + +### The fork that remains + +Only the owner can sign a final state, so only the owner can fork one. It +keeps the earlier record and its key, so it can sign a second final state at +once or much later, and any node the first has not reached will take it. Each +node keeps its first. Nothing local can settle that, and this design does not +try. What it guarantees instead: + +- **No fork after the fact.** Once a node holds a final state, no arrival — + client PUT, fresh offer or repair — moves it off it. A second final state can + only land on a node that holds none. +- **Nodes look before a final state.** Before taking a final state it does not + hold, from a client or from a fresh offer, a node asks its close group which + state each holds. A peer claiming a *different* final state is asked for the + record, and if it is the state claimed and verifies — only the owner could + have signed it — the node + refuses, answering `Stale` with the state the group proved. A claim alone + refuses nothing, so one dishonest peer cannot block a transfer. Each peer's + question and fetch run as one pipeline, all at once, so a peer that claims a + rival and then stalls its fetch cannot hide another peer's proof until the + budget runs out. The look runs after payment is verified, asks only peers + that have sent a pointer message, and is bounded at four seconds; silence + proves nothing and the write goes ahead. This is what closes the gap the + merge rule leaves: a node that joined the group after the transfer would + otherwise take a second final state on the merge rule alone. +- **A look costs one round, once.** A payment proof, once verified, is cached, + so replaying one paid final state costs its sender nothing after the first + time. A node therefore remembers each final state a look proved, up to two + per address and 16,384 addresses, oldest forgotten first, and never lets a + later proof replace an earlier one; it refuses a replayed loser from that + memory, ahead of the signature check and without asking anyone. A look that + found nothing answers the same state again for two seconds without asking, + enough for replays queued behind it, and not at all once the write it + cleared has failed. + Looks for one address wait their turn, so replays queued behind a look find + its answer, and a burst costs one round; at most 64 run at once, and one + that cannot start within two seconds answers the PUT with an error and + drops the fresh offer, neither taking nor refusing the state for good; the + client retries such an error with the proof it already paid for. Looks are + spread over the address's last byte, since the addresses one node holds + share their leading bits. + Two seconds of waiting and four of looking add at most six seconds to a + PUT after its payment is verified; a PUT whose payment check is itself slow + can outlast the client's ten-second store timeout, and the client then + retries it as a missing answer with the proof it already paid for. Flooding + past all this needs a new paid final state per round. +- **A node restores its own final state without looking.** A node that lost + the file of a final state it held is admitted that exact state again and + nothing else (ADR-0016's lost-record rule), so taking it back is a restore, + not a new final state. It asks nobody: a peer on the other side of a fork + would otherwise keep it from restoring its own copy for good. +- **Readers see the majority, and see forks.** A client read that meets a final + state is settled only once one final state is held by a majority of the close + group, and returns that one. If two different final states are seen and + neither has a majority, the read fails as forked rather than guess. One final + state below a majority with no rival is returned once corroborated, as any + state is: it is a transfer still spreading. +- **Recipients check before they rely on it.** `pointer_finality` asks the + whole group and answers `Open`, `Settling` (one final state, short of a + majority), `Final` (one final state, a majority holds it, no rival seen) or + `Forked` (rivals seen; the majority's state if there is one). A recipient + treats the transfer as done only on `Final`; anything else is the former + owner's to explain. + +### Replication + +- **Repair** is the merge rule over quorum-backed states, so a node holding a + final state adopts nothing else, and a node holding none adopts the final + state a quorum holds. Between two final states a group wide enough to back + both at quorum — never a seven-node group at four — the larger side is + adopted, not whichever answered first. It must be strictly larger even + counting every peer that did not answer for its rival, so a tie, seen or + possibly hidden behind silent peers, adopts neither until the group + settles. +- **Hints** for a different final state are dropped by a node holding one, + since the hint cannot replace it. No refetch loop. +- **Possession.** A member holding a *different final* state is not penalised + when checked for the one this node offered: it holds what the merge rule told + it to, and the fork is the owner's. It is logged at warn. A member holding + nothing is penalised as before. + +### Client + +- `pointer_transfer` refuses before paying if the pointer is already final, if + the recipient is the pointer itself, if the recipient pointer does not exist, + or if the recipient's chain leads back to this address. It then signs, + pays, stores and reports `pointer_finality`. +- `pointer_update` on a final pointer refuses before paying, as it always did + when the counter could not advance. +- `pointer_controller` follows transfers — final pointer-targeted states only — + to the pointer whose owner now decides what the address resolves to. + +## Consequences + +### Positive + +- Pointers can be handed over for good, and the address readers use never + changes. +- No new record, field, message or storage format. The wire is untouched; the + only protocol change is one comparison. +- Once the handover lands on a node, the former owner has no move there: no + larger counter exists and no equal one replaces. +- A handover can be chained: the recipient can transfer its own pointer on. +- ADR-0016's "revocable forwarding" and "migrate before the final counter" + caveats are gone: migration *is* the final update. + +### Negative / Trade-offs + +- **A race is a permanent fork.** An owner who signs two final states and races + them leaves each node on its first, forever. With a majority on one side, + reads return that side and flag the fork to anyone who checks finality. With + no majority, reads of that pointer fail as forked, and nothing — not the + owner, not repair — can fix it. Only the owner can do this, and only to its + own pointer, as only the owner can lose its key. +- **A transfer is irreversible.** One sent to the wrong recipient pointer is + gone. The client checks the recipient exists and does not loop back; it + cannot check intent. +- **The look before a final state is not a lock.** It is fail-open on silence + and bounded in time, and it only runs when a node has learned which peers + understand pointers, which takes a sync round after joining. A node asked in + that first round, or one whose group cannot answer in four seconds, takes a + second final state on the merge rule alone, and so does a node running with + no replication, a local devnet or one whose replication engine failed to + start, since there is nothing to ask the group with. Reads still return the + majority side; the residual risk is a minority fork that `pointer_finality` + reports. +- Under a flood of distinct paid final states a node answers some honest final + PUTs with an error rather than look for them late. The client retries them + with the proof it already paid for; a write that still falls short fails, + and the owner tries again. +- A final PUT costs its node one state query per capable close-group peer, and + a fetch per claimed rival, before it commits. +- **Mixed fleets.** A node on ADR-0016's rule still lets a smaller-target final + state displace the first. Until a close group's majority runs this rule a + former owner can win back the nodes that do not. Reads follow the majority, so + the handover holds wherever most of the group has upgraded; recipients should + wait for the upgrade before relying on a transfer. Nothing on the wire tells + the two rules apart: the pointer format version is unchanged, so no node can + refuse to replicate with a peer on the old rule. +- Transferred reads take one extra hop, and a chain of transfers one per hop, + bounded by the client's resolve depth. + +### Neutral / Operational + +- A node refusing a final state because its group proved another logs it at + info with both state ids; a possession check that finds the other side of a + fork logs at warn. Either is the owner equivocating. +- Pricing, payments, audits and storage commitments are unchanged: a final + state is a state, paid, stored, committed and audited like any other. + +## Validation + +- Protocol: a final state is replaced by no counter and no other final state, + whatever its target; below the final counter the order is unchanged; + `replaces` is a strict partial order over a set including final states, total + except between two of them; a fold keeps the first final state it meets; + `transfer_to` signs at the final counter to the recipient, refuses to sign past + a final record, and only a final pointer target counts as a transfer. +- Node, store: a stored transfer is not displaced by a final state whose target + sorts first, nor by any lower counter. +- Node, properties: with one final state among the records every delivery order + converges on it, exhaustively over all permutations; with two, the first + delivered is kept. +- Node, request handler: a final state the group proves already superseded is + refused and named, and nothing is written; one nobody contradicts is taken; + the group is asked only about a final state the node lacks; two final states + raced to two nodes leave each on its first; a node restores its own lost + final state without asking, and still refuses the rival; a busy look neither + takes nor refuses a final state for good; a proven loser is refused again + before its signature is checked, and without asking. +- Node, the look: a peer that claims a rival and stalls its fetch does not hide + another peer's proof; a claim the served record does not back is no proof; + proven final states refuse others, not themselves, and the oldest is + forgotten first past the cap; a peer that serves a different rival than it + claimed proves nothing; a second proof at an address never replaces + the first; replays queued behind a look reuse its answer, and a proof + answers every replay of the loser; a clear answer is dropped when the write + it cleared fails; looks for addresses sharing their leading bits run side by + side. +- Node, repair: a node holding a final state adopts nothing else; of two final + states with quorum, the larger side is adopted in either answer order, and + a tie adopts neither in either order, nor does a side that silent peers + could still tie. +- Node, live network: a transfer written to one node reaches the group and no + node takes a second final state; a node that missed the transfer refuses a + different one because a peer serves the one it holds, and still takes the + group's own; the possession check does not penalise the other side of a fork + and still penalises a member holding nothing. +- Client: reads over a group holding a transfer, a majority fork, a no-majority + fork and a transfer still spreading; the finality check for each; transfer + preflight refusals; end to end against a local testnet with real settlement, + a transfer is final, readers are redirected to the recipient's pointer, the + recipient moves it, and the former owner's attempt to take it back is refused + by every node. + +## Notes for AI-assisted work + +AI tools may help draft this ADR, but **must not mark it Accepted without human +review**. Accepted ADRs are immutable: create a new superseding ADR rather than +editing an Accepted ADR. diff --git a/docs/adr/ADR-0019-pointer-audits-serve-the-record-round-one-bound.md b/docs/adr/ADR-0019-pointer-audits-serve-the-record-round-one-bound.md new file mode 100644 index 00000000..b9215a4b --- /dev/null +++ b/docs/adr/ADR-0019-pointer-audits-serve-the-record-round-one-bound.md @@ -0,0 +1,165 @@ +# ADR-0019: Pointer audits serve the record round 1 bound + +- **Status:** Proposed +- **Date:** 2026-09-29 +- **Decision owners:** Anselme (@grumbach) +- **Reviewers:** +- **Supersedes:** none. It amends one point of ADR-0016, "Updates between the + rounds", and leaves the rest of ADR-0016 as written. +- **Superseded by:** none +- **Related:** ADR-0002 (audit), ADR-0009 (audit families), ADR-0016 (pointers) + +## Context + +A storage audit is two rounds (ADR-0002, ADR-0009). Round 1 reports, for every +leaf of the audited subtree, a root over the bytes the node holds, keyed by the +audit's fresh nonce. Round 2 then opens a few of those leaves, and the node must +serve bytes that reproduce the root round 1 reported. + +For a pointer (ADR-0016) those bytes are the whole signed record, and the owner +may replace the record at any time with a paid update. ADR-0016 covers one +update between the rounds: the store keeps the record an update replaced for +five minutes, and round 2 serves it beside the new one. It accepts the case of +two updates: + +> Two updates to one pointer inside the same audit would fail an honest holder, +> and would take the owner two paid updates within seconds of each other. + +Two things make that worth closing rather than accepting. + +- **The failure is charged to the wrong party.** The auditor reports + `DigestMismatch`, a confirmed failure, and the holder takes the trust penalty + for its owner's activity. Nothing the holder did was wrong. +- **It can be aimed.** An owner who is also a close-group auditor of the holder + knows exactly when its round 1 has been answered, and two paid updates cost + two writes, about 0.013 ANT each on the 990-node pointer testnet. ADR-0016 + already notes that an owner can grind a node id beside its own pointer. That + turns an accepted edge case into a cheap way to penalise a chosen honest + neighbour. + +The cause is narrow. Round 2 serves the record held now and the one last +replaced, because the node has not kept what round 1 bound and so cannot tell +which record it owes. After two updates neither of those is the one round 1 +read. + +## Decision Drivers + +- An honest holder must not take a confirmed failure because its owner updated + the pointer, however often. +- No wire change and no change to what the auditor accepts. Pointers have not + shipped in any release yet, but the smaller change is still the better one. +- Memory stays bounded against an auditor that opens many round-1 sessions, and + running out of it must never turn into a confirmed failure either. + +## Considered Options + +1. **Keep ADR-0016 as written.** Cheapest, and it leaves the failure above. +2. **Serve every record held since round 1.** It needs a larger per-item cap + than `MAX_POINTER_RECORDS_PER_ITEM`, so the auditor's check changes, and any + cap is still one more paid update away from failing. +3. **Remember what round 1 bound, and serve that record.** Round 1 already + reports a nonced root per pointer leaf. The node keeps those roots in the + round-1 session it already holds, and round 2 serves the one record, + current or replaced, whose root matches. + +## Decision + +We will take option 3. + +- **The session keeps round 1's roots.** When a round-1 proof is about to be + sent, the single-use session it opens keeps the nonced root reported for each + pointer leaf, keyed by address: 64 bytes of key and root a pointer, and + nothing for chunks. A session whose proof then fails to send keeps them until + it expires, as it keeps its place today. +- **The store keeps every replaced record, for longer.** Every record an update + replaces is kept in memory, not only the last one, for ten minutes rather + than five. A round 1 can read a pointer at its start and take as long as the + auditor waits for the largest subtree (1,024 leaves), about seven minutes + with the default configuration, before its session even opens, and round 2 + then has the session's two minutes. A test ties the ten minutes to those two + figures for the default configuration; an auditor configured to wait longer + than that can outlast the record. The overall cap stays 2,048 records, about + 11 MB, oldest first wherever it is, and records past the ten minutes are + dropped at each prune pass even when no update comes to drop them. +- **Round 2 serves the record that matches.** Among the record held now and the + replaced records kept for that address, it serves the one whose nonced root, + under the audit's own nonce, is the root round 1 reported. That is a single + record, so the auditor's check and the item cap are unchanged. +- **When it cannot, it says so.** If the root matches nothing kept, because the + record aged out or was evicted, round 2 is rejected as `Transient`, as for a + local read error. That is the auditor's timeout lane: no trust penalty, but + the auditor forgets the holder's standing as a proven holder of every key + under the pinned commitment, until the holder passes again. A node that no + longer holds the pointer reports it absent, which is a confirmed failure, as + before, whatever replaced records it still keeps: those prove what round 1 + read, not that the pointer is still held. +- **The roots are bounded, by admission.** Every live session together keeps + at most `MAX_SESSION_POINTER_BINDINGS` (65,536) roots, 4 MiB of payload + before the maps' own overhead. A round 1 whose roots would not fit withholds + its proof and answers `Transient` instead, which puts it in the auditor's + timeout lane with no trust penalty; staying silent would read as a peer that + did not answer. Roots are never stripped from a live session to make room: + its round 2 is owed them. A whole session can still be evicted when the + session count reaches `MAX_SUBTREE_SESSIONS`, as before this change, and its + round 2 then goes to the timeout lane. Without a root for a pointer, round 2 + would reject it as `Transient` rather than guess, since the replaced records + kept are capped and an empty history proves nothing, but a session this node + opened always holds a root for every pointer it proved. + +## Consequences + +### Positive + +- Updates between the rounds no longer fail an honest holder, however many. + What remains are local limits, and none is a confirmed failure: the bound + record evicted by more than 2,048 paid updates across the node's pointers + inside ten minutes is reported as `Transient`, and so is a round 1 over the + roots budget. +- Round 2 serves one pointer record where it could serve two, so it is smaller. +- The auditor, the wire format and the subtree-audit protocol id are unchanged. + +### Negative / Trade-offs + +- Round-1 sessions carry state they did not before, bounded by the budget + above. Auditors that open pointer-heavy sessions faster than they complete + can fill it, and later round 1s are then answered `Transient` until it + drains. That costs the holder those audits' credit, not trust, much as the + round-1 concurrency and work budgets already can. +- A responder that returns `Transient` is not proved wrong. That was already + so, since any responder can report a local read error, so this gives a + dishonest node no answer it did not have. +- One address can hold many replaced records inside the window, and round 2 + hashes each candidate it checks for that address. The global cap bounds that + work to about 11 MB of keyed BLAKE3 per opened pointer, and filling it takes + that many paid updates. +- Replaced records are kept twice as long, so under a high update rate the + 2,048-record cap is reached sooner. The memory bound itself is unchanged. + +### Neutral / Operational + +- A restart drops every session, as before, so a round 2 that follows one goes + to the graced timeout lane, as it already did. + +## Validation + +- `several_updates_between_the_rounds_do_not_fail_an_honest_holder`: three + updates between the rounds fail with `DigestMismatch` before this change and + pass after it, and round 2 serves exactly the record round 1 read. +- `round_two_serves_the_record_round_one_read_across_several_updates` (e2e): + both rounds sent over QUIC to a live node, with three updates between them. + It fails if the node stops handing round 1's roots to its session. +- `a_bound_record_no_longer_held_is_unavailable_not_failed`, + `without_the_bound_root_a_pointer_is_unavailable_not_failed`, + `subtree_session_carries_pointer_bindings_within_the_budget`, + `past_the_cap_the_oldest_replaced_record_anywhere_goes_first` and + `a_replaced_record_outlives_the_slowest_audit` cover the limits above. +- In production, a pointer holder's `DigestMismatch` rate should not rise with + the update rate of the pointers it holds. A rise in `Transient` round-2 + rejections naming a pointer means the retention or session budget is being + reached. + +## Notes for AI-assisted work + +AI tools may help draft this ADR, but **must not mark it Accepted without human +review**. Accepted ADRs are immutable: create a new superseding ADR rather than +editing an Accepted ADR. diff --git a/docs/adr/README.md b/docs/adr/README.md index 5061984e..fb204dc8 100644 --- a/docs/adr/README.md +++ b/docs/adr/README.md @@ -36,3 +36,5 @@ See [`TOOLING.md`](./TOOLING.md) for `adrs`, `adr-kit`, and AI harness setup. - [ADR-0013: Settlement version and pre-payment compatibility](./ADR-0013-settlement-version-and-pre-payment-compatibility.md) - [ADR-0015: Direct browser clients over WebRTC Direct](./ADR-0015-direct-browser-clients-over-webrtc-direct.md) - [ADR-0016: Pointers — paid mutable references with an immutable owner](./ADR-0016-pointers-immutable-owner.md) +- [ADR-0018: Pointer ownership transfer by final redirection](./ADR-0018-pointer-transfer-by-final-redirection.md) +- [ADR-0019: Pointer audits serve the record round 1 bound](./ADR-0019-pointer-audits-serve-the-record-round-one-bound.md) diff --git a/src/node.rs b/src/node.rs index 9fe7e254..cfd51c3f 100644 --- a/src/node.rs +++ b/src/node.rs @@ -272,11 +272,10 @@ impl NodeBuilder { }; // ADR-0016: pointers replicate through the same engine. The PUT handler - // hands each newly stored paid state to it on this channel. + // hands each newly stored paid state to it, and asks it before taking + // a final state (ADR-0018). if let Some(service) = protocol.pointer_service() { - let (writes, fresh_writes) = tokio::sync::mpsc::unbounded_channel(); - service.attach_fresh_writes(writes); - engine.with_pointers(service.store().clone(), fresh_writes); + engine.with_pointer_service(service); } // ADR-0004: wire the engine's commitment state as the quote generator's diff --git a/src/pointer/mod.rs b/src/pointer/mod.rs index c4787f39..b859231d 100644 --- a/src/pointer/mod.rs +++ b/src/pointer/mod.rs @@ -3,10 +3,18 @@ //! Implements `docs/adr/ADR-0016-pointers-immutable-owner.md`. //! //! A pointer is a mutable, owner-signed reference stored at an address derived -//! from the owner's public key. Ownership is fixed at creation: there is no -//! transfer, no lineage, no certificates and no key rotation. That choice is -//! what lets the design be this small — the owner key is inlined in the -//! record, so validating a pointer needs nothing but the pointer. +//! from the owner's public key. The owner key is fixed at creation: there is no +//! lineage, no certificates and no key rotation. That choice is what lets the +//! design be this small — the owner key is inlined in the record, so +//! validating a pointer needs nothing but the pointer. +//! +//! What the address resolves to can still be handed over for good +//! (`docs/adr/ADR-0018-pointer-transfer-by-final-redirection.md`): a state at +//! the final counter is replaced by nothing, so once a node holds one pointing +//! at the new owner's pointer, the former owner cannot take it back there. The +//! store gets that from the merge rule; the service adds one look at the close +//! group before taking a final state, so a node that had not heard of the +//! first refuses a second one when a peer proves the first in time. //! //! # What lives here //! @@ -53,8 +61,8 @@ pub mod store; pub use ant_protocol::pointer::{ pointer_address, state_id_for_body, ParsedPointer, Pointer, PointerError, PointerState, - PointerTarget, PointerTargetKind, DATA_TYPE_POINTER, POINTER_BODY_LEN, POINTER_FORMAT_VERSION, - POINTER_WIRE_LEN, TARGET_WIRE_LEN, + PointerTarget, PointerTargetKind, DATA_TYPE_POINTER, FINAL_COUNTER, POINTER_BODY_LEN, + POINTER_FORMAT_VERSION, POINTER_WIRE_LEN, TARGET_WIRE_LEN, }; -pub use service::PointerService; +pub use service::{FinalStateWitness, FinalityCheck, PointerService}; pub use store::{PointerStore, PutOutcome}; diff --git a/src/pointer/service.rs b/src/pointer/service.rs index bdda6c54..8a2993eb 100644 --- a/src/pointer/service.rs +++ b/src/pointer/service.rs @@ -16,11 +16,22 @@ //! are both 32 bytes from the same range; the domain separator makes a //! collision infeasible, not impossible. A node that holds one kind at an //! address refuses the other rather than silently choosing. +//! 4. **A look before a final state.** A state at the final counter is +//! replaced by nothing, so a node that takes one can never be corrected. +//! Before taking one it neither holds nor lost, the node asks its close +//! group whether a *different* final state is already held there, and +//! refuses if a peer proves one with the signed record (ADR-0018). That is +//! what keeps a former owner from handing an address over a second time to +//! a node that had not heard of the first, whenever a peer can prove the +//! first in time. A proven loser is remembered and refused again before its +//! signature is checked. //! //! # Order of work //! //! ```text -//! parse → compare with what is held → verify signature → check payment → commit +//! parse → compare with what is held → (final state only) a proven loser? +//! → verify signature → check payment +//! → (final state only) ask the close group → commit //! ``` //! //! The comparison precedes the signature check, so re-submitting a state the @@ -35,6 +46,7 @@ use ant_protocol::chunk::{ XorName, }; use bytes::Bytes; +use futures::future::BoxFuture; use parking_lot::RwLock; use saorsa_core::P2PNode; use tokio::sync::mpsc; @@ -46,7 +58,46 @@ use crate::pointer::store::{Inspected, PointerStore, PutOutcome}; use crate::replication::admission; use crate::replication::pointer::PointerFreshWrite; use crate::storage::{ChunkStore, SELF_CLOSENESS_GATE_WIDTH}; -use ant_protocol::pointer::POINTER_WIRE_LEN; +use ant_protocol::pointer::{PointerState, POINTER_WIRE_LEN}; + +/// Where a node looks, before it takes a final state, for a different final +/// state its close group already holds (ADR-0018). +/// +/// A trait rather than the replication engine itself, because the engine is +/// built after this service and needs a running P2P node, and because what the +/// service decides from the answer is worth testing without one. +pub trait FinalStateWitness: Send + Sync { + /// Ask the close group whether a final state other than `state` is + /// already held at `state.address`. + fn check_final<'a>(&'a self, state: &'a PointerState) -> BoxFuture<'a, FinalityCheck>; + + /// A final state at `state.address`, other than `state`, that an earlier + /// check already proved, without asking anyone. + /// + /// Cheap enough to consult ahead of the signature check, so a final state + /// that lost once is refused again for nothing, however often the same + /// paid record is replayed. + fn proven_conflict(&self, state: &PointerState) -> Option; + + /// Forget that a look found `state` clear: the write it cleared did not + /// land, so a retry must look again rather than reuse the answer. + fn forget_clear(&self, state: &PointerState); +} + +/// What a close group said about a final state a node is about to take. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum FinalityCheck { + /// No peer proved a different final state. Also the answer when the group + /// could not be asked in time: silence is not evidence, so it never + /// refuses a write. + Clear, + /// A different final state a peer holds, proven by the signed record, + /// which only the owner could have made. + Conflict(PointerState), + /// Too many checks were running to start this one in time. The state is + /// neither taken nor refused for good, and the sender may try again. + Busy, +} /// Handles pointer requests against a [`PointerStore`]. #[derive(Clone)] @@ -70,6 +121,10 @@ pub struct PointerService { /// handler before it has a running P2P node. `None` in unit tests that /// never attach one, exactly as the chunk path does. p2p_node: Arc>>>, + /// Asked before a final state is taken (ADR-0018). Attached with the + /// replication engine; `None` where nothing replicates, and then a final + /// state is taken on the merge rule alone. + final_witness: Arc>>>, } impl std::fmt::Debug for PointerService { @@ -92,6 +147,7 @@ impl PointerService { payments: None, fresh_writes: Arc::new(RwLock::new(None)), p2p_node: Arc::new(RwLock::new(None)), + final_witness: Arc::new(RwLock::new(None)), } } @@ -125,6 +181,11 @@ impl PointerService { *self.fresh_writes.write() = Some(writes); } + /// Ask `witness` before taking a final state this node does not hold. + pub fn attach_final_state_witness(&self, witness: Arc) { + *self.final_witness.write() = Some(witness); + } + /// The store this service fronts. #[must_use] pub const fn store(&self) -> &PointerStore { @@ -162,10 +223,11 @@ impl PointerService { } }; - let address = parsed.state().address; - let state_id = parsed.state().state_id; + let state = *parsed.state(); + let address = state.address; + let state_id = state.state_id; - if let Some(refusal) = self.admit(parsed.state()).await { + if let Some(refusal) = self.admit(&state).await { return refusal; } @@ -191,6 +253,13 @@ impl PointerService { } } + // Last, because it costs the close group a round trip, and only for a + // paid final state: if a peer proves a different final state is + // already held, this one lost the race to it everywhere that matters. + if let Some(refusal) = self.final_refusal(&state).await { + return refusal; + } + // Charge the bytes this write will take, and hold the charge until it // lands. Checking capacity and then writing is the race the file store // exists to close: concurrent writers all pass one cached measurement @@ -202,6 +271,7 @@ impl PointerService { Ok(reservation) => Some(reservation), Err(e) => { debug!("Rejecting pointer PUT for {}: {e}", hex::encode(address)); + self.forget_clear(&state); return PointerPutResponse::Error(ProtocolError::StorageFailed(e.to_string())); } }, @@ -211,7 +281,7 @@ impl PointerService { // release it the moment this future is dropped, while the blocking // transaction it started runs on and publishes the file. let record_bytes = request.record.to_vec(); - match self.store.commit(record, reservation).await { + let response = match self.store.commit(record, reservation).await { Ok(PutOutcome::Changed) => { // Offer it to the rest of the close group, proof included, so // every member holds it whichever of them the client reached. @@ -243,6 +313,24 @@ impl PointerService { warn!("Pointer commit failed for {}: {e}", hex::encode(address)); PointerPutResponse::Error(ProtocolError::StorageFailed(e.to_string())) } + }; + // A look that cleared a final state is reused briefly for replays of + // it; a write that did not land must not leave that answer behind. + if !matches!( + response, + PointerPutResponse::Success { .. } | PointerPutResponse::Unchanged { .. } + ) { + self.forget_clear(&state); + } + response + } + + /// Forget a clear look for `state`, if a witness took one. + fn forget_clear(&self, state: &PointerState) { + if state.is_terminal() { + if let Some(witness) = self.final_witness.read().as_ref() { + witness.forget_clear(state); + } } } @@ -251,10 +339,7 @@ impl PointerService { /// `Some(response)` means refuse. Ordered cheapest first, and all of it /// ahead of the signature check, so a forged record for an address this /// node does not serve buys no cryptography. - async fn admit( - &self, - state: &ant_protocol::pointer::PointerState, - ) -> Option { + async fn admit(&self, state: &PointerState) -> Option { let address = state.address; // Any state the merge rule prefers to what this node knows, whatever @@ -269,9 +354,15 @@ impl PointerService { hex::encode(address), state.counter ); + // Named as the state this node knows, even one whose file it + // lost, so the sender can see what it must catch up to. return Some(PointerPutResponse::Stale { address, - state_id: self.store.state_id(&address).unwrap_or_default(), + state_id: self + .store + .remembered(&address) + .map(|known| known.state_id) + .unwrap_or_default(), }); } @@ -316,9 +407,81 @@ impl PointerService { ))); } } + + // A final state an earlier check proved lost to another is refused + // again here, before it buys a signature check or a round trip. + if self.needs_final_check(state) { + let proven = self + .final_witness + .read() + .as_ref() + .and_then(|witness| witness.proven_conflict(state)); + if let Some(conflict) = proven { + return Some(PointerPutResponse::Stale { + address, + state_id: conflict.state_id, + }); + } + } None } + /// Whether `state` needs the close group's word before this node takes + /// it: it is final, and this node neither holds nor remembers a final + /// state at its address. + /// + /// Nothing else is asked about. A lower counter can be replaced, so taking + /// one blind costs nothing a later state cannot fix. A node that holds a + /// final state has had its answer from the merge rule before this runs: + /// the same state is unchanged and any other is stale. And a node that + /// lost the file of a final state it held is admitted only that state + /// again ([`PointerStore::admits`]), which restores what it already took + /// rather than taking a new one. Asking there would let any peer on the + /// other side of a fork keep it from restoring its own copy. + fn needs_final_check(&self, state: &PointerState) -> bool { + state.is_terminal() + && !self + .store + .remembered(&state.address) + .is_some_and(|known| known.is_terminal()) + } + + /// The answer to a paid final state the close group proves already lost, + /// or that could not be checked in time; `None` to take it. + async fn final_refusal(&self, state: &PointerState) -> Option { + if !self.needs_final_check(state) { + return None; + } + let witness = self.final_witness.read().as_ref().map(Arc::clone)?; + match witness.check_final(state).await { + FinalityCheck::Clear => None, + // Named as the state the group proved, as for any stale arrival. + FinalityCheck::Conflict(conflict) => { + info!( + "Refusing final pointer state {} at {}: the close group already holds \ + final state {}, so the owner has finalized it before", + hex::encode(state.state_id), + hex::encode(state.address), + hex::encode(conflict.state_id) + ); + Some(PointerPutResponse::Stale { + address: state.address, + state_id: conflict.state_id, + }) + } + FinalityCheck::Busy => { + debug!( + "Deferring final pointer state {} at {}: too many finality checks running", + hex::encode(state.state_id), + hex::encode(state.address) + ); + Some(PointerPutResponse::Error(ProtocolError::Internal( + "too many finality checks running; try again".to_string(), + ))) + } + } + } + /// Handle a pointer GET. pub async fn handle_get(&self, request: PointerGetRequest) -> PointerGetResponse { match self.store.get(&request.address).await { @@ -399,8 +562,9 @@ fn cross_kind_refusal(address: XorName, chunk_present: Result) -> Option

(MlDsaPublicKey, MlDsaSecretKey) { ml_dsa_65().generate_keypair_from_seed(&[seed; 32]) @@ -781,6 +945,372 @@ mod tests { assert_eq!(held.counter(), u64::MAX); } + /// A close group whose answer to the finality question is fixed, and + /// which counts how often it was asked. + struct StubWitness { + answer: FinalityCheck, + proven: Option, + asked: AtomicUsize, + forgotten: AtomicUsize, + } + + impl StubWitness { + fn answering(answer: FinalityCheck) -> Arc { + Arc::new(Self { + answer, + proven: None, + asked: AtomicUsize::new(0), + forgotten: AtomicUsize::new(0), + }) + } + + /// One that already proved `conflict` in an earlier check. + fn having_proven(conflict: PointerState) -> Arc { + Arc::new(Self { + answer: FinalityCheck::Clear, + proven: Some(conflict), + asked: AtomicUsize::new(0), + forgotten: AtomicUsize::new(0), + }) + } + + fn asked(&self) -> usize { + self.asked.load(Ordering::SeqCst) + } + } + + impl FinalStateWitness for StubWitness { + fn check_final<'a>(&'a self, _state: &'a PointerState) -> BoxFuture<'a, FinalityCheck> { + self.asked.fetch_add(1, Ordering::SeqCst); + let answer = self.answer; + Box::pin(async move { answer }) + } + + fn proven_conflict(&self, state: &PointerState) -> Option { + self.proven + .filter(|proven| proven.state_id != state.state_id) + } + + fn forget_clear(&self, _state: &PointerState) { + self.forgotten.fetch_add(1, Ordering::SeqCst); + } + } + + fn final_state(seed: u8, target_byte: u8) -> Pointer { + let (pk, sk) = keypair(seed); + let target = PointerTarget::new(PointerTargetKind::Pointer, [target_byte; 32]); + Pointer::sign(&sk, &pk, FINAL_COUNTER, target).expect("sign") + } + + #[tokio::test] + async fn a_final_state_the_group_proves_is_already_superseded_is_refused() { + // The node has the pointer's ordinary state and has never seen a final + // one, so the merge rule alone would take this. The group proves the + // owner already finalized it elsewhere, so it is refused, named as the + // state that got there first, and nothing is written. + let (service, _dir) = service().await; + let current = signed(1, 3, 1); + service.handle_put(put(¤t)).await; + + let established = final_state(1, 0xAA); + let late = final_state(1, 0x01); + let witness = StubWitness::answering(FinalityCheck::Conflict(established.state())); + service.attach_final_state_witness(witness.clone()); + + match service.handle_put(put(&late)).await { + PointerPutResponse::Stale { address, state_id } => { + assert_eq!(address, late.address()); + assert_eq!( + state_id, + established.state_id(), + "the first final state is named" + ); + } + other => panic!("a second final state must be refused, got {other:?}"), + } + assert_eq!(witness.asked(), 1); + assert_eq!( + service.store().state_id(&late.address()), + Some(current.state_id()), + "nothing was written" + ); + } + + #[tokio::test] + async fn a_final_state_nobody_contradicts_is_taken() { + // Silence is not evidence: a group that proves nothing lets the final + // state through on the merge rule. + let (service, _dir) = service().await; + let witness = StubWitness::answering(FinalityCheck::Clear); + service.attach_final_state_witness(witness.clone()); + + let transfer = final_state(2, 0x42); + assert!(matches!( + service.handle_put(put(&transfer)).await, + PointerPutResponse::Success { .. } + )); + assert_eq!(witness.asked(), 1); + assert_eq!( + service.store().state_id(&transfer.address()), + Some(transfer.state_id()) + ); + } + + #[tokio::test] + async fn the_group_is_asked_only_about_a_final_state_the_node_lacks() { + // A lower counter can always be replaced later, so it is taken without + // a round trip. A node already holding a final state has its answer + // from the merge rule before any round trip: the same state is + // unchanged, and any other is stale. + let (service, _dir) = service().await; + let witness = StubWitness::answering(FinalityCheck::Clear); + service.attach_final_state_witness(witness.clone()); + + for counter in [0u64, 1, u64::MAX - 1] { + assert!(matches!( + service.handle_put(put(&signed(3, counter, 1))).await, + PointerPutResponse::Success { .. } + )); + } + assert_eq!(witness.asked(), 0, "no final state yet, no question"); + + let first = final_state(3, 0x10); + assert!(matches!( + service.handle_put(put(&first)).await, + PointerPutResponse::Success { .. } + )); + assert_eq!(witness.asked(), 1); + + let second = final_state(3, 0x01); + match service.handle_put(put(&second)).await { + PointerPutResponse::Stale { state_id, .. } => { + assert_eq!(state_id, first.state_id()); + } + other => panic!("a second final state must be stale, got {other:?}"), + } + assert!(matches!( + service.handle_put(put(&final_state(3, 0x10))).await, + PointerPutResponse::Unchanged { .. } + )); + assert_eq!(witness.asked(), 1, "a held final state answers for itself"); + } + + #[tokio::test] + async fn a_node_restores_its_own_lost_final_state_without_asking() { + // The node took a final state, then lost the file. The group holds + // the other side of a fork, which a check would prove. Asking would + // keep the node from restoring what it already took, for good; the + // lost state itself is the one thing it is admitted, so nobody is + // asked, and the rival is still refused. + let (service, _dir) = service().await; + let own = final_state(5, 0x10); + let rival = final_state(5, 0x01); + assert!(matches!( + service.handle_put(put(&own)).await, + PointerPutResponse::Success { .. } + )); + std::fs::remove_file(service.store().file_for(&own.address())).expect("remove"); + assert!(service + .store() + .get(&own.address()) + .await + .expect("get") + .is_none()); + + let witness = StubWitness::answering(FinalityCheck::Conflict(rival.state())); + service.attach_final_state_witness(witness.clone()); + assert!(matches!( + service.handle_put(put(&own)).await, + PointerPutResponse::Success { .. } + )); + assert_eq!(witness.asked(), 0, "restoring its own state asks nobody"); + assert_eq!( + service.store().state_id(&own.address()), + Some(own.state_id()) + ); + std::fs::remove_file(service.store().file_for(&own.address())).expect("remove"); + assert!(service + .store() + .get(&own.address()) + .await + .expect("get") + .is_none()); + match service.handle_put(put(&rival)).await { + PointerPutResponse::Stale { state_id, .. } => { + assert_eq!( + state_id, + own.state_id(), + "the rival is told which state won" + ); + } + other => panic!("the rival must be stale, got {other:?}"), + } + } + + /// A group that answers clear while, meanwhile, a rival final state + /// lands on this node: the race a clear look can lose. + struct RacingWitness { + store: PointerStore, + rival: Pointer, + forgotten: AtomicUsize, + } + + impl FinalStateWitness for RacingWitness { + fn check_final<'a>(&'a self, _state: &'a PointerState) -> BoxFuture<'a, FinalityCheck> { + Box::pin(async move { + self.store + .put_bytes(&self.rival.to_bytes()) + .await + .expect("the rival lands"); + FinalityCheck::Clear + }) + } + + fn proven_conflict(&self, _state: &PointerState) -> Option { + None + } + + fn forget_clear(&self, _state: &PointerState) { + self.forgotten.fetch_add(1, Ordering::SeqCst); + } + } + + #[tokio::test] + async fn a_cleared_final_state_whose_write_fails_is_looked_for_again() { + // The look found nothing, but a rival final state landed while it + // looked, so the write it cleared comes back stale. The clear answer + // must not outlive that write: a retry has to look again. + let (landing, _other) = service().await; + let (service, _dir) = service().await; + let witness = Arc::new(RacingWitness { + store: service.store().clone(), + rival: final_state(8, 0x01), + forgotten: AtomicUsize::new(0), + }); + service.attach_final_state_witness(witness.clone()); + assert!(matches!( + service.handle_put(put(&final_state(8, 0x42))).await, + PointerPutResponse::Stale { .. } + )); + assert_eq!(witness.forgotten.load(Ordering::SeqCst), 1); + + // A write that lands keeps its answer. + let witness = StubWitness::answering(FinalityCheck::Clear); + landing.attach_final_state_witness(witness.clone()); + assert!(matches!( + landing.handle_put(put(&final_state(9, 0x42))).await, + PointerPutResponse::Success { .. } + )); + assert_eq!(witness.forgotten.load(Ordering::SeqCst), 0); + } + + #[tokio::test] + async fn a_busy_check_neither_takes_nor_refuses_a_final_state_for_good() { + let (service, _dir) = service().await; + let current = signed(6, 3, 1); + service.handle_put(put(¤t)).await; + let witness = StubWitness::answering(FinalityCheck::Busy); + service.attach_final_state_witness(witness.clone()); + + let transfer = final_state(6, 0x42); + assert!(matches!( + service.handle_put(put(&transfer)).await, + PointerPutResponse::Error(ProtocolError::Internal(_)) + )); + assert_eq!(witness.asked(), 1); + assert_eq!( + service.store().state_id(&transfer.address()), + Some(current.state_id()), + "nothing was written" + ); + assert!( + service.store().admits(&transfer.state()), + "and the state can still be taken later" + ); + } + + #[tokio::test] + async fn a_proven_conflict_is_refused_again_before_any_signature_check() { + // Replaying a paid final state that already lost costs a node + // nothing: the refusal comes from what an earlier check proved, ahead + // of the signature check and without asking the group. A record + // whose signature does not even verify shows the order. + let (unwitnessed, _other) = service().await; + let (service, _dir) = service().await; + let established = final_state(7, 0xAA); + let late = final_state(7, 0x01); + let witness = StubWitness::having_proven(established.state()); + service.attach_final_state_witness(witness.clone()); + + let mut forged = late.to_bytes(); + if let Some(last) = forged.last_mut() { + *last ^= 0xFF; + } + match service + .handle_put(PointerPutRequest::new(Bytes::from(forged.clone()))) + .await + { + PointerPutResponse::Stale { state_id, .. } => { + assert_eq!(state_id, established.state_id()); + } + other => panic!("a proven loser must be refused as stale, got {other:?}"), + } + assert_eq!(witness.asked(), 0, "and nobody is asked again"); + assert!( + matches!( + unwitnessed + .handle_put(PointerPutRequest::new(Bytes::from(forged.clone()))) + .await, + PointerPutResponse::Error(_) + ), + "the record really does fail its signature check" + ); + + // The proven state itself is not refused by its own proof. + assert!(matches!( + service.handle_put(put(&established)).await, + PointerPutResponse::Success { .. } + )); + } + + #[tokio::test] + async fn two_final_states_raced_to_two_nodes_leave_each_on_its_first() { + // The one fork ADR-0018 allows: the owner signs two final states and + // sends one to each node before either has heard of the other. Each + // keeps what it took first and refuses the other, whichever sorts + // first. Nothing on a node can settle it; a reader decides by how many + // of the group hold each side. + let (first_node, _a) = service().await; + let (second_node, _b) = service().await; + let one = final_state(4, 0x09); + let other = final_state(4, 0x01); + + assert!(matches!( + first_node.handle_put(put(&one)).await, + PointerPutResponse::Success { .. } + )); + assert!(matches!( + second_node.handle_put(put(&other)).await, + PointerPutResponse::Success { .. } + )); + assert!(matches!( + first_node.handle_put(put(&other)).await, + PointerPutResponse::Stale { .. } + )); + assert!(matches!( + second_node.handle_put(put(&one)).await, + PointerPutResponse::Stale { .. } + )); + assert_eq!( + first_node.store().state_id(&one.address()), + Some(one.state_id()) + ); + assert_eq!( + second_node.store().state_id(&other.address()), + Some(other.state_id()) + ); + } + #[test] fn a_pointer_is_refused_where_a_chunk_already_sits() { // The two kinds share one 32-byte address space. A collision is diff --git a/src/pointer/store.rs b/src/pointer/store.rs index 3628d173..a0c64e7a 100644 --- a/src/pointer/store.rs +++ b/src/pointer/store.rs @@ -45,7 +45,9 @@ //! lock file under `{root}/pointers/` keeps two processes from keeping two //! indexes over one set of files. -use std::collections::HashMap; +use std::collections::{HashMap, VecDeque}; + +use bytes::Bytes; use std::fs::{File, OpenOptions}; use std::io::{Read, Write}; use std::path::{Path, PathBuf}; @@ -81,16 +83,35 @@ const SHARD_COUNT: u16 = 256; /// /// A storage audit binds the record a node holds in its first round and asks /// for it in the second (ADR-0016). An owner updating the pointer between the -/// two would otherwise fail the honest node that took the update, so the -/// record the update replaced stays servable for longer than an audit session -/// lives. -pub const SUPERSEDED_RETENTION: Duration = Duration::from_mins(5); - -/// Most replaced records kept at once, about 11 MB at the cap. Past it the -/// oldest goes first; reaching it inside [`SUPERSEDED_RETENTION`] takes that -/// many paid updates to pointers this node holds. +/// two would otherwise fail the honest node that took the update, so every +/// record an update replaces stays servable, however many updates follow it +/// (ADR-0019), for longer than the slowest audit takes with the default +/// configuration: a round 1 over the largest subtree an auditor will wait +/// for, then the session its round 2 must arrive within. Round 1 can read a +/// pointer at its very start and take that long to finish, so the time counts +/// from the read, not from the session. +pub const SUPERSEDED_RETENTION: Duration = Duration::from_mins(10); + +/// Most replaced records kept at once, across every address, about 11 MB at +/// the cap. Past it the oldest goes first; reaching it inside +/// [`SUPERSEDED_RETENTION`] takes that many paid updates to pointers this +/// node holds. const MAX_SUPERSEDED: usize = 2048; +/// Every record a recent update replaced, by address, oldest first, each with +/// when it was replaced (see [`SUPERSEDED_RETENTION`]). Held as [`Bytes`] so +/// handing them out copies nothing. +type Superseded = HashMap>; + +/// Drop every kept record older than [`SUPERSEDED_RETENTION`] at `now`, and +/// every address left with none. +fn drop_expired(superseded: &mut Superseded, now: Instant) { + superseded.retain(|_, kept| { + kept.retain(|(at, _)| now.saturating_duration_since(*at) < SUPERSEDED_RETENTION); + !kept.is_empty() + }); +} + /// The name of the shard directory `address` lives in: its last byte in hex. fn shard_name(address: &XorName) -> String { let last = address.last().copied().unwrap_or_default(); @@ -258,9 +279,8 @@ struct Inner { generation: AtomicU64, /// What the store has done, for telemetry. counters: Counters, - /// The record each recent update replaced, by address, with when (see - /// [`SUPERSEDED_RETENTION`]). - superseded: Mutex)>>, + /// Every record a recent update replaced (see [`Superseded`]). + superseded: Mutex, /// Held for the store's lifetime; releasing it releases the directory. _lock_file: File, } @@ -519,17 +539,43 @@ impl PointerStore { .map(|entry| entry.state.state_id) } - /// The record an update at `address` replaced within the last - /// [`SUPERSEDED_RETENTION`], if any: what a storage audit that bound it - /// before the update is still owed. + /// The record the last update at `address` replaced, within the last + /// [`SUPERSEDED_RETENTION`], if any. See [`Self::superseded_all`] for + /// every one kept. #[must_use] pub fn superseded(&self, address: &XorName) -> Option> { + self.superseded_all(address) + .into_iter() + .next() + .map(|bytes| bytes.to_vec()) + } + + /// Every record updates at `address` replaced within the last + /// [`SUPERSEDED_RETENTION`], newest first: what a storage audit that bound + /// one of them before the updates is still owed (ADR-0019). + #[must_use] + pub fn superseded_all(&self, address: &XorName) -> Vec { self.inner .superseded .lock() .get(address) - .filter(|(at, _)| at.elapsed() < SUPERSEDED_RETENTION) - .map(|(_, bytes)| bytes.clone()) + .map(|kept| { + kept.iter() + .rev() + .filter(|(at, _)| at.elapsed() < SUPERSEDED_RETENTION) + .map(|(_, bytes)| bytes.clone()) + .collect() + }) + .unwrap_or_default() + } + + /// Forget every replaced record older than [`SUPERSEDED_RETENTION`]. + /// + /// Updates do this as they keep a record; this is for when none come, so + /// memory goes back within a pass of the retention running out rather + /// than at the next update, which may never come. + pub fn drop_expired_superseded(&self) { + drop_expired(&mut self.inner.superseded.lock(), Instant::now()); } /// The bytes of the record held at `address`, read from disk without @@ -562,6 +608,16 @@ impl PointerStore { .map(|entry| entry.state) } + /// The state this node holds at `address`, or held until it lost the + /// file. + /// + /// Unlike [`Self::state`], a lost record still answers: it is the one + /// state, besides a newer one, that [`Self::admits`] lets restore it. + #[must_use] + pub fn remembered(&self, address: &XorName) -> Option { + self.snapshot(address).map(|entry| entry.state) + } + /// Whether a paid PUT of `state` may be taken. /// /// Any state the merge rule prefers to what is held, whatever its counter. @@ -738,23 +794,59 @@ impl PointerStore { } impl Inner { - /// Keep `bytes` as the record just replaced at `address`, dropping what - /// has aged out and, past the cap, the oldest. + /// Keep `bytes` beside whatever else was recently replaced at `address`, + /// dropping what has aged out. + /// + /// Nothing is evicted to make room here: the update that replaced it may + /// yet fail, and a record evicted for an update that never happened would + /// be lost for nothing. [`Self::trim_superseded`] does that once it has. fn keep_superseded(&self, address: XorName, bytes: Vec) { let now = Instant::now(); let mut superseded = self.superseded.lock(); - superseded.retain(|_, (at, _)| now.duration_since(*at) < SUPERSEDED_RETENTION); - while superseded.len() >= MAX_SUPERSEDED { + drop_expired(&mut superseded, now); + superseded + .entry(address) + .or_default() + .push_back((now, Bytes::from(bytes))); + } + + /// Past the cap, drop the oldest kept records, wherever they are. + fn trim_superseded(&self) { + let mut superseded = self.superseded.lock(); + let mut held: usize = superseded.values().map(VecDeque::len).sum(); + while held > MAX_SUPERSEDED { + // Each address keeps its records oldest first, so the oldest + // anywhere is the first of one of them. let Some(oldest) = superseded .iter() - .min_by_key(|(_, (at, _))| *at) - .map(|(address, _)| *address) + .filter_map(|(address, kept)| kept.front().map(|(at, _)| (*at, *address))) + .min() + .map(|(_, address)| address) else { break; }; - superseded.remove(&oldest); + if let Some(kept) = superseded.get_mut(&oldest) { + kept.pop_front(); + if kept.is_empty() { + superseded.remove(&oldest); + } + } + held = held.saturating_sub(1); + } + drop(superseded); + } + + /// Take back the record [`Self::keep_superseded`] just kept for `address`, + /// when the update that replaced it did not happen after all. Called under + /// the index lock that kept it, so nothing was kept for `address` since. + fn unkeep_superseded(&self, address: &XorName) { + let mut superseded = self.superseded.lock(); + if let Some(kept) = superseded.get_mut(address) { + kept.pop_back(); + if kept.is_empty() { + superseded.remove(address); + } } - superseded.insert(address, (now, bytes)); } /// Remove the file and the index entry for `address` under one lock, so a @@ -845,9 +937,21 @@ impl Inner { // have committed while it ran. let replacing = index.get(&address).is_some_and(|entry| entry.on_disk); let outcome = match index.get(&address) { - // Nothing held, or a record this node lost: either way the - // write must happen, whatever state it carries. - None | Some(IndexEntry { on_disk: false, .. }) => PutOutcome::Changed, + // Nothing held: any state lands. + None => PutOutcome::Changed, + // A record this node lost: the lost state itself lands, to + // restore it, and so does anything that replaces it, but + // nothing older, so an arrival verified before the loss cannot + // roll the node back past it (the rule `admits` states). + Some(entry @ IndexEntry { on_disk: false, .. }) => { + if entry.state.state_id == record.state_id() + || record.state().replaces(&entry.state) + { + PutOutcome::Changed + } else { + PutOutcome::Stale + } + } Some(entry) if entry.state.state_id == record.state_id() => PutOutcome::Unchanged, Some(entry) if record.state().replaces(&entry.state) => PutOutcome::Changed, Some(_) => PutOutcome::Stale, @@ -863,18 +967,26 @@ impl Inner { } // What this replaces, read under the lock so it is the record the - // index names. Kept for an audit that bound it; a record that - // cannot be read is not kept, and an audit owed it fails as it - // would have on the lost file. - let previous = if replacing { - read_record_file(&path).ok().flatten() - } else { - None - }; + // index names. Kept for an audit that bound it, and kept before + // the rename makes the new record visible, so a round 2 reading + // the new record always finds the old one kept beside it. A record + // that cannot be read is not kept, and an audit owed it fails as + // it would have on the lost file. + let mut kept = false; + if replacing { + if let Some(previous) = read_record_file(&path).ok().flatten() { + self.keep_superseded(address, previous); + kept = true; + } + } // The rename is the commit point: nothing fallible happens between // it and the index update, and both are under this one lock. if let Err(e) = rename_with_retry(&temp, &path) { + // Nothing was replaced, so nothing replaced is kept. + if kept { + self.unkeep_superseded(&address); + } if std::fs::remove_file(&temp).is_err() { settle(reservation); } @@ -884,11 +996,11 @@ impl Inner { path.display() ))); } + if kept { + self.trim_superseded(); + } let generation = self.generation.fetch_add(1, Ordering::Relaxed); index.insert(address, IndexEntry::of(record, generation)); - if let Some(previous) = previous { - self.keep_superseded(address, previous); - } // A file is on the disk now. Charge it whether or not this replaced // one: telling those apart would mean trusting an observation taken // before the rename, and that observation can be wrong in the one @@ -1276,9 +1388,8 @@ mod tests { let (store, _dir) = store().await; let first = signed(1, 1, 1); store.put_bytes(&first.to_bytes()).await.expect("put"); - assert_eq!( - store.superseded(&first.address()), - None, + assert!( + store.superseded_all(&first.address()).is_empty(), "a creation replaces nothing" ); @@ -1288,8 +1399,8 @@ mod tests { PutOutcome::Changed ); assert_eq!( - store.superseded(&first.address()), - Some(first.to_bytes()), + store.superseded_all(&first.address()), + vec![Bytes::from(first.to_bytes())], "the replaced record, exactly as it was held" ); assert_eq!( @@ -1300,7 +1411,116 @@ mod tests { // A stale arrival replaces nothing, so it keeps nothing. store.put_bytes(&first.to_bytes()).await.expect("put"); - assert_eq!(store.superseded(&first.address()), Some(first.to_bytes())); + assert_eq!( + store.superseded_all(&first.address()), + vec![Bytes::from(first.to_bytes())] + ); + + // A later update keeps the one before it too, newest first: an audit + // may have bound either. + let third = signed(1, 3, 3); + store.put_bytes(&third.to_bytes()).await.expect("put"); + assert_eq!( + store.superseded_all(&first.address()), + vec![ + Bytes::from(second.to_bytes()), + Bytes::from(first.to_bytes()) + ] + ); + assert_eq!( + store.superseded(&first.address()), + Some(second.to_bytes()), + "the earlier accessor still answers with the record last replaced" + ); + } + + /// Past the cap the oldest record kept goes first, whichever address it + /// is for, and an address whose last record goes is forgotten. + /// Keeping a record evicts nothing: an update that then fails takes back + /// what it kept and leaves every other kept record where it was. + #[tokio::test] + async fn a_record_kept_for_an_update_that_fails_costs_no_other_record() { + let (store, _dir) = store().await; + let (owed, failing) = ([1u8; 32], [2u8; 32]); + store.inner.keep_superseded(owed, vec![1]); + for i in 0..MAX_SUPERSEDED - 1 { + store + .inner + .keep_superseded(failing, (i as u64).to_le_bytes().to_vec()); + } + // The cap is full. The next update keeps its record first ... + store.inner.keep_superseded(failing, vec![0xFF]); + // ... and its rename fails, so it takes that record back. + store.inner.unkeep_superseded(&failing); + assert_eq!( + store.superseded_all(&owed), + vec![Bytes::from(vec![1])], + "the record an audit may be owed is still kept" + ); + assert_eq!(store.superseded_all(&failing).len(), MAX_SUPERSEDED - 1); + } + + /// A replaced record past its retention is forgotten even if no update + /// comes to do it. + #[tokio::test] + async fn expired_replaced_records_are_dropped_without_another_update() { + let (store, _dir) = store().await; + let old = Instant::now() + .checked_sub(SUPERSEDED_RETENTION + Duration::from_secs(1)) + .expect("the clock is past one retention"); + store + .inner + .superseded + .lock() + .insert([3u8; 32], VecDeque::from([(old, Bytes::from(vec![3]))])); + store.inner.keep_superseded([4u8; 32], vec![4]); + store.drop_expired_superseded(); + assert!( + store.superseded_all(&[3u8; 32]).is_empty() + && !store.inner.superseded.lock().contains_key(&[3u8; 32]), + "the expired record is gone" + ); + assert!( + store.inner.superseded.lock().contains_key(&[4u8; 32]), + "a fresh one stays" + ); + } + + #[tokio::test] + async fn past_the_cap_the_oldest_replaced_record_anywhere_goes_first() { + let (store, _dir) = store().await; + let (first, second) = ([1u8; 32], [2u8; 32]); + let keep = |address: XorName, bytes: Vec| { + store.inner.keep_superseded(address, bytes); + store.inner.trim_superseded(); + }; + keep(first, vec![1]); + for i in 0..MAX_SUPERSEDED - 1 { + keep(second, (i as u64).to_le_bytes().to_vec()); + } + assert_eq!(store.superseded_all(&first), vec![Bytes::from(vec![1])]); + assert_eq!(store.superseded_all(&second).len(), MAX_SUPERSEDED - 1); + + // One more: the first address held the oldest record, so it goes. + keep(second, vec![0xFF]); + let evicted = store.superseded_all(&first); + assert!( + evicted.is_empty(), + "expected the oldest address evicted, got {evicted:?}" + ); + assert!(!store.inner.superseded.lock().contains_key(&first)); + assert_eq!(store.superseded_all(&second).len(), MAX_SUPERSEDED); + + // And the next goes from the front of the second address's history. + keep(first, vec![2]); + let kept = store.superseded_all(&second); + assert_eq!(kept.len(), MAX_SUPERSEDED - 1); + assert_eq!(kept.first(), Some(&Bytes::from(vec![0xFF])), "newest first"); + assert_eq!( + kept.last(), + Some(&Bytes::from(1u64.to_le_bytes().to_vec())), + "the oldest of them went" + ); } #[tokio::test] @@ -1754,6 +1974,33 @@ mod tests { assert!(store.get(&record.address()).await.expect("get").is_some()); } + /// The commit itself holds the line `admits` states for a lost record, so + /// an older state that reaches it without that gate, or was admitted + /// before the loss, cannot roll the node back. + #[tokio::test] + async fn a_commit_after_a_loss_takes_nothing_older_than_what_was_lost() { + let (store, _dir) = store().await; + let held = signed(1, 3, 1); + store.put_bytes(&held.to_bytes()).await.expect("put"); + std::fs::remove_file(store.file_for(&held.address())).expect("remove"); + assert!(store.get(&held.address()).await.expect("get").is_none()); + + assert_eq!( + store + .put_bytes(&signed(1, 2, 2).to_bytes()) + .await + .expect("put"), + PutOutcome::Stale, + "an older state must not replace the one that was lost" + ); + assert!(!store.contains(&held.address()), "nothing was written"); + assert_eq!( + store.put_bytes(&held.to_bytes()).await.expect("put"), + PutOutcome::Changed, + "the lost state itself is restored" + ); + } + #[tokio::test] async fn a_lost_record_is_restored_and_cannot_be_rolled_back() { // The node remembers what it lost. Without that the address would look diff --git a/src/replication/config.rs b/src/replication/config.rs index 276ba1bb..f4463db3 100644 --- a/src/replication/config.rs +++ b/src/replication/config.rs @@ -258,6 +258,17 @@ pub const SUBTREE_SESSION_TTL: Duration = Duration::from_mins(2); /// peers open sessions; oldest are evicted past this). pub const MAX_SUBTREE_SESSIONS: usize = 4 * MAX_CONCURRENT_SUBTREE_ROUND1 * 256; +/// Most pointer bindings every live round-1 session holds together +/// (ADR-0019): 4 MiB of keys and roots at the cap, before the maps' own +/// overhead. +/// +/// A session keeps one per pointer its round 1 proved, and a round-1 subtree +/// can hold about a thousand leaves, so [`MAX_SUBTREE_SESSIONS`] full sessions +/// would otherwise hold two million. A round 1 whose bindings would not fit +/// withholds its proof and answers `Transient`, and no session already +/// answered gives its bindings up. +pub const MAX_SESSION_POINTER_BINDINGS: usize = 1 << 16; + /// Sustained rate at which the responder-wide round-1 work budget refills, in /// bytes of chunk content per second. /// diff --git a/src/replication/mod.rs b/src/replication/mod.rs index 73d91f45..12f4ab7c 100644 --- a/src/replication/mod.rs +++ b/src/replication/mod.rs @@ -68,6 +68,7 @@ use crate::payment::{ MIN_PAYMENT_PROOF_SIZE_BYTES, }; use crate::pointer::store::PointerStore; +use crate::pointer::{FinalStateWitness, PointerService}; use crate::replication::audit::AuditTickResult; use crate::replication::audit_coordinator::AuditChallengeCoordinator; use crate::replication::audit_metrics::{ @@ -82,9 +83,9 @@ use crate::replication::config::{ max_parallel_fetch, storage_admission_width, ReplicationConfig, MAX_AUDIT_RESPONSES_PER_PEER, MAX_CONCURRENT_AUDIT_RESPONSES, MAX_CONCURRENT_REPLICATION_SENDS, MAX_DIGEST_AUDIT_RESPONSES_PER_PEER, MAX_INCOMING_VERIFICATION_KEYS, - MAX_SUBTREE_ROUND1_PER_PEER, MAX_SUBTREE_SESSIONS, MAX_VERIFICATION_KEYS_PER_CYCLE, - REPLICATION_PROTOCOL_ID, SUBTREE_AUDIT_PROTOCOL_ID, SUBTREE_ROUND1_WORK_BURST_BYTES, - SUBTREE_ROUND1_WORK_REFILL_BYTES_PER_SEC, SUBTREE_SESSION_TTL, + MAX_SESSION_POINTER_BINDINGS, MAX_SUBTREE_ROUND1_PER_PEER, MAX_SUBTREE_SESSIONS, + MAX_VERIFICATION_KEYS_PER_CYCLE, REPLICATION_PROTOCOL_ID, SUBTREE_AUDIT_PROTOCOL_ID, + SUBTREE_ROUND1_WORK_BURST_BYTES, SUBTREE_ROUND1_WORK_REFILL_BYTES_PER_SEC, SUBTREE_SESSION_TTL, }; use crate::replication::paid_list::PaidList; use crate::replication::protocol::{ @@ -94,6 +95,7 @@ use crate::replication::protocol::{ use crate::replication::quorum::KeyVerificationOutcome; use crate::replication::recent_provers::RecentProvers; use crate::replication::scheduling::{CapacityDisplacement, DeferralOutcome, ReplicationQueues}; +use crate::replication::storage_commitment_audit::PointerBindings; use crate::replication::types::{ AuditFailureReason, BootstrapClaimObservation, BootstrapState, FailureEvidence, NeighborSyncState, PeerSyncRecord, PresenceEvidence, RepairProofs, VerificationEntry, @@ -2217,6 +2219,10 @@ impl ReplicationEngine { /// Replicate pointers too (ADR-0016): the records in `store`, and the /// fresh writes the pointer PUT handler sends on `fresh_writes`. /// + /// The PUT handler is not wired to ask this engine anything before it + /// takes a final state (ADR-0018); [`Self::with_pointer_service`] wires + /// both. + /// /// Call before [`Self::start`]. pub fn with_pointers( &mut self, @@ -2237,6 +2243,22 @@ impl ReplicationEngine { self.pointer_fresh_rx = Some(fresh_writes); } + /// Replicate the pointers `service` stores (ADR-0016), and wire the + /// service to this engine both ways: it hands each newly stored paid + /// state here to be offered on, and asks here before it takes a final + /// state (ADR-0018). + /// + /// Call before [`Self::start`]. + pub fn with_pointer_service(&mut self, service: &PointerService) { + let (writes, fresh_writes) = mpsc::unbounded_channel(); + service.attach_fresh_writes(writes); + self.with_pointers(service.store().clone(), fresh_writes); + if let Some(replication) = &self.pointers { + let witness: Arc = replication.clone(); + service.attach_final_state_witness(witness); + } + } + /// The pointer replication, when enabled. Tests use it to drive rounds. #[cfg(any(test, feature = "test-utils"))] #[must_use] @@ -4624,6 +4646,9 @@ struct SubtreeSession { commitment_hash: [u8; 32], nonce: [u8; 32], inserted: Instant, + /// What round 1 bound for each pointer it proved, so round 2 serves that + /// record however many updates land in between (ADR-0019). + pointer_bindings: PointerBindings, } /// Responder-wide token bucket over the chunk bytes round-1 proof building may @@ -4787,16 +4812,31 @@ impl SubtreeRound1Limiter { /// Record a single-use session once a round-1 proof is built and about to be /// sent, so the matching round 2 is admitted exactly once. + /// + /// The session keeps what round 1 bound for each pointer it proved, while + /// every live session together holds no more than + /// [`MAX_SESSION_POINTER_BINDINGS`] of them. A session whose bindings do + /// not fit is not opened and `false` is returned, so its proof is not + /// sent and round 1 is answered `Transient` instead: the bindings of + /// sessions already answered are never given up, because their round 2 is + /// owed them (ADR-0019). async fn open_session( &self, source: PeerId, challenge_id: u64, commitment_hash: [u8; 32], nonce: [u8; 32], - ) { + pointer_bindings: PointerBindings, + ) -> bool { let now = Instant::now(); let mut sessions = self.sessions.write().await; sessions.retain(|_, e| now.duration_since(e.inserted) < SUBTREE_SESSION_TTL); + // Checked before anything is evicted, so a session refused here costs + // no other session its place. + let held: usize = sessions.values().map(|e| e.pointer_bindings.len()).sum(); + if held.saturating_add(pointer_bindings.len()) > MAX_SESSION_POINTER_BINDINGS { + return false; + } if sessions.len() >= MAX_SUBTREE_SESSIONS { if let Some(oldest) = sessions .iter() @@ -4812,21 +4852,24 @@ impl SubtreeRound1Limiter { commitment_hash, nonce, inserted: now, + pointer_bindings, }, ); + true } - /// Atomically consume the round-2 session for this exchange. `true` iff a + /// Atomically consume the round-2 session for this exchange. `Some` iff a /// live session matching `(source, challenge_id, commitment_hash, nonce)` - /// existed (and is now removed); a miss silently drops round 2 to the graced - /// timeout lane (sessions are ephemeral and can be lost across a restart). + /// existed (and is now removed), carrying what its round 1 bound for each + /// pointer; a miss silently drops round 2 to the graced timeout lane + /// (sessions are ephemeral and can be lost across a restart). async fn consume_session( &self, source: &PeerId, challenge_id: u64, commitment_hash: &[u8; 32], nonce: &[u8; 32], - ) -> bool { + ) -> Option { let mut sessions = self.sessions.write().await; let matches = sessions.get(&(*source, challenge_id)).is_some_and(|e| { Instant::now().duration_since(e.inserted) < SUBTREE_SESSION_TTL @@ -4834,17 +4877,21 @@ impl SubtreeRound1Limiter { && &e.nonce == nonce }); if matches { - sessions.remove(&(*source, challenge_id)); + sessions + .remove(&(*source, challenge_id)) + .map(|e| e.pointer_bindings) + } else { + None } - matches } } /// Outcome of admitting a round-2 slice challenge. enum SliceAdmission { /// Admitted: the guard holds the global permit and the per-peer slot, and - /// the single-use round-1 session has been consumed. - Admitted(AuditResponderGuard), + /// the single-use round-1 session has been consumed, yielding what its + /// round 1 bound for each pointer. + Admitted(AuditResponderGuard, PointerBindings), /// Refused at a responder ceiling. The round-1 session is left INTACT. Capacity(AuditResponderAdmissionFailure), /// No live round-1 session matched this challenge. @@ -4885,7 +4932,7 @@ async fn admit_slice_challenge( Ok(guard) => guard, Err(failure) => return SliceAdmission::Capacity(failure), }; - if !round1 + let Some(pointer_bindings) = round1 .consume_session( source, challenge.challenge_id, @@ -4893,13 +4940,13 @@ async fn admit_slice_challenge( &challenge.nonce, ) .await - { + else { // Release the permit and per-peer slot before the caller replies: no // chunk work follows, so holding them would shrink the pool for nothing. drop(guard); return SliceAdmission::NoSession; - } - SliceAdmission::Admitted(guard) + }; + SliceAdmission::Admitted(guard, pointer_bindings) } /// Try to admit one audit-responder task for `source`: take a global permit AND @@ -5234,16 +5281,42 @@ async fn handle_replication_message( // A round-1 proof authorizes exactly one matching round 2: open a // single-use session so a slice challenge cannot be served without // a live round-1 exchange. + let mut response = response; if let crate::replication::protocol::SubtreeAuditResponse::Proof { .. } = &response { - subtree_round1 + let opened = subtree_round1 .open_session( source, challenge.challenge_id, challenge.expected_commitment_hash, challenge.nonce, + storage_commitment_audit::pointer_bindings(&response), ) .await; + // A proof round 2 could not be answered for is not sent. + // The node says so instead, as for a local read error: the + // auditor's timeout lane, with no trust penalty, where + // silence would read as a peer that did not answer + // (ADR-0019). + if !opened { + protocol::record_audit_drop(protocol::AuditDropKind::Subtree); + warn!( + target: "ant_node::replication::audit_responder", + event = "admission_dropped", + kind = "subtree", + responder_class = class.as_str(), + source = %source, + challenge_id = challenge.challenge_id, + request_response = rr_message_id.is_some(), + reason = "pointer_binding_budget", + "Audit responder admission dropped" + ); + response = crate::replication::protocol::SubtreeAuditResponse::Rejected { + challenge_id: challenge.challenge_id, + kind: protocol::RejectKind::Transient, + reason: "pointer binding budget full".to_string(), + }; + } } let response_kind = subtree_audit_response_kind(&response); let work_items = subtree_audit_response_work_items(&response); @@ -5286,7 +5359,7 @@ async fn handle_replication_message( "Audit challenge received: kind=slice source={source} request_response={}", rr_message_id.is_some(), ); - let guard = match admit_slice_challenge( + let (guard, pointer_bindings) = match admit_slice_challenge( &ctx.audit_responder_semaphore, &ctx.audit_responder_inflight, &ctx.subtree_round1, @@ -5295,7 +5368,7 @@ async fn handle_replication_message( ) .await { - SliceAdmission::Admitted(guard) => guard, + SliceAdmission::Admitted(guard, pointer_bindings) => (guard, pointer_bindings), SliceAdmission::Capacity(failure) => { protocol::record_audit_drop(protocol::AuditDropKind::Slice); audit_metrics::record_admission_drop(class); @@ -5391,10 +5464,11 @@ async fn handle_replication_message( let worker_started = Instant::now(); let processing_started = Instant::now(); let response = - storage_commitment_audit::handle_subtree_slice_challenge_with_pointers( + storage_commitment_audit::handle_subtree_slice_challenge_with_pointer_bindings( &challenge, &storage, pointer_store.as_ref(), + &pointer_bindings, p2p_node.peer_id(), bootstrapping, Some(&my_commitment_state), @@ -5549,23 +5623,27 @@ async fn handle_replication_message( } ReplicationMessageBody::PointerFetchRequest(request) => { if let Some(pointers) = &ctx.pointers { - pointers.serve_fetch_detached( - *source, - request, - msg.request_id, - rr_message_id.map(ToOwned::to_owned), - ); + pointers + .serve_fetch_detached( + *source, + request, + msg.request_id, + rr_message_id.map(ToOwned::to_owned), + ) + .await; } Ok(()) } ReplicationMessageBody::PointerStateRequest(request) => { if let Some(pointers) = &ctx.pointers { - pointers.serve_state_detached( - *source, - request, - msg.request_id, - rr_message_id.map(ToOwned::to_owned), - ); + pointers + .serve_state_detached( + *source, + request, + msg.request_id, + rr_message_id.map(ToOwned::to_owned), + ) + .await; } Ok(()) } @@ -6536,12 +6614,6 @@ async fn dispatch_neighbor_sync_request( received_at: Instant, rr_message_id: Option<&str>, ) -> Result<()> { - // A peer syncing with us gets our pointer hints as well — including a - // node that is bootstrapping, which is how it learns the pointers it - // should hold. - if let Some(pointers) = &ctx.pointers { - pointers.push_hints_detached(vec![source]); - } let guard = match admit_bounded_responder( &ctx.neighbor_sync_responder_admission_semaphore, &ctx.neighbor_sync_responder_inflight, @@ -6565,7 +6637,7 @@ async fn dispatch_neighbor_sync_request( return Ok(()); } }; - + let pointers = ctx.pointers.clone(); let worker_semaphore = Arc::clone(&ctx.neighbor_sync_responder_worker_semaphore); let p2p_node = Arc::clone(&ctx.p2p_node); let storage = Arc::clone(&ctx.storage); @@ -6598,6 +6670,14 @@ async fn dispatch_neighbor_sync_request( ); return; } + // A peer whose sync request is admitted and fresh gets our pointer + // hints as well, including a node that is bootstrapping, which is how + // it learns the pointers it should hold. Only such a request: each + // answer scans every pointer held, and a refused or stale one must + // cost nothing. + if let Some(pointers) = &pointers { + pointers.answer_sync_with_hints(source); + } if let Err(e) = handle_neighbor_sync_request( &source, &request, @@ -10723,15 +10803,156 @@ mod tests { // Session: opened by round 1, consumed exactly once by the matching round 2. let hash = [7u8; 32]; let nonce = [9u8; 32]; - limiter.open_session(peer, 42, hash, nonce).await; + assert!( + limiter + .open_session(peer, 42, hash, nonce, PointerBindings::new()) + .await + ); // Wrong nonce / commitment does not match. - assert!(!limiter.consume_session(&peer, 42, &hash, &[0u8; 32]).await); - assert!(!limiter.consume_session(&peer, 42, &[0u8; 32], &nonce).await); + assert!(limiter + .consume_session(&peer, 42, &hash, &[0u8; 32]) + .await + .is_none()); + assert!(limiter + .consume_session(&peer, 42, &[0u8; 32], &nonce) + .await + .is_none()); // A round 2 with no prior round 1 (wrong challenge_id) misses. - assert!(!limiter.consume_session(&peer, 99, &hash, &nonce).await); + assert!(limiter + .consume_session(&peer, 99, &hash, &nonce) + .await + .is_none()); // The matching round 2 consumes it — and only once (single-use). - assert!(limiter.consume_session(&peer, 42, &hash, &nonce).await); - assert!(!limiter.consume_session(&peer, 42, &hash, &nonce).await); + assert!(limiter + .consume_session(&peer, 42, &hash, &nonce) + .await + .is_some()); + assert!(limiter + .consume_session(&peer, 42, &hash, &nonce) + .await + .is_none()); + } + + // A session carries what its round 1 bound for each pointer to round 2, + // and every live session together stays under the binding budget. A + // session that does not fit is refused, and no session already opened + // gives up its bindings or its place for it. + #[tokio::test] + async fn subtree_session_carries_pointer_bindings_within_the_budget() { + let limiter = SubtreeRound1Limiter::new(Duration::ZERO, 1); + let (hash, nonce) = ([1u8; 32], [2u8; 32]); + let bindings = |from: u64, count: usize| -> PointerBindings { + (from..) + .take(count) + .map(|i| { + let mut key = [0u8; 32]; + key[..8].copy_from_slice(&i.to_le_bytes()); + (key, [0xAB; 32]) + }) + .collect() + }; + + let half = MAX_SESSION_POINTER_BINDINGS / 2; + let first = bindings(0, half); + assert!( + limiter + .open_session(test_peer(1), 1, hash, nonce, first.clone()) + .await + ); + assert!( + limiter + .open_session(test_peer(2), 2, hash, nonce, bindings(1 << 40, half)) + .await + ); + // The budget is full: one more binding is refused, but a round 1 with + // no pointers in it still opens. + assert!( + !limiter + .open_session(test_peer(3), 3, hash, nonce, bindings(1 << 41, 1)) + .await + ); + assert!( + limiter + .open_session(test_peer(4), 4, hash, nonce, PointerBindings::new()) + .await + ); + + assert_eq!( + limiter + .consume_session(&test_peer(1), 1, &hash, &nonce) + .await, + Some(first.clone()), + "an opened session keeps every binding it was given" + ); + assert!( + limiter + .consume_session(&test_peer(3), 3, &hash, &nonce) + .await + .is_none(), + "a refused session was never opened" + ); + assert_eq!( + limiter + .consume_session(&test_peer(2), 2, &hash, &nonce) + .await + .map(|b| b.len()), + Some(half) + ); + + // Consumed sessions free their room. + assert!( + limiter + .open_session(test_peer(5), 5, hash, nonce, bindings(1 << 42, half)) + .await + ); + } + + // A session refused for the binding budget is refused before the session + // cap evicts anything: with both full, the refusal costs no session its + // place. + #[tokio::test] + async fn a_session_refused_for_bindings_evicts_no_other_session() { + let limiter = SubtreeRound1Limiter::new(Duration::ZERO, 1); + let (hash, nonce) = ([1u8; 32], [2u8; 32]); + let full: PointerBindings = (0u64..) + .take(MAX_SESSION_POINTER_BINDINGS) + .map(|i| { + let mut key = [0u8; 32]; + key[..8].copy_from_slice(&i.to_le_bytes()); + (key, [0xCD; 32]) + }) + .collect(); + assert!( + limiter + .open_session(test_peer(0), 0, hash, nonce, full) + .await + ); + for id in 1..MAX_SUBTREE_SESSIONS as u64 { + assert!( + limiter + .open_session(test_peer(1), id, hash, nonce, PointerBindings::new()) + .await + ); + } + + let one: PointerBindings = std::iter::once(([0xEE; 32], [0xEE; 32])).collect(); + assert!( + !limiter + .open_session(test_peer(2), u64::MAX, hash, nonce, one) + .await + ); + assert_eq!( + limiter.sessions.read().await.len(), + MAX_SUBTREE_SESSIONS, + "every session is still there" + ); + assert!( + limiter + .consume_session(&test_peer(0), 0, &hash, &nonce) + .await + .is_some(), + "including the oldest" + ); } // The concurrency pool and the per-peer cooldown are both keyed by peer id, @@ -10945,7 +11166,11 @@ mod tests { let (id, hash, nonce) = (77u64, [3u8; 32], [4u8; 32]); let challenge = slice_challenge(id, hash, nonce); - round1.open_session(peer, id, hash, nonce).await; + assert!( + round1 + .open_session(peer, id, hash, nonce, PointerBindings::new()) + .await + ); // Saturate this peer's share so the next admission must be refused. let mut hold = Vec::new(); @@ -10970,7 +11195,7 @@ mod tests { let retried = admit_slice_challenge(&semaphore, &inflight, &round1, &peer, &challenge).await; assert!( - matches!(retried, SliceAdmission::Admitted(_)), + matches!(retried, SliceAdmission::Admitted(..)), "the round-1 session must survive a capacity refusal so the retry succeeds" ); diff --git a/src/replication/pointer.rs b/src/replication/pointer.rs index df62a0f4..442699fa 100644 --- a/src/replication/pointer.rs +++ b/src/replication/pointer.rs @@ -24,7 +24,13 @@ //! they hold that state or a newer one by returning a valid record. //! - **Possession.** Some minutes after offering a fresh state, the node asks //! each close-group member for the record. A member that is still responsible -//! and cannot produce that state or a newer one is penalised. +//! and cannot produce that state or a newer one is penalised — unless it holds +//! a *different final* state, which is a fork its owner made and not a failure +//! to store. +//! - **Finality.** Before taking a final state it does not hold, from a client +//! or a fresh offer, a node asks the close group whether a different final +//! state is already held, and refuses if a peer proves one with the signed +//! record (ADR-0018). See [`PointerReplication::check_final`]. //! //! Requests go only to peers that have sent a pointer message themselves (see //! [`PointerReplication::is_capable`]). A peer built before pointers cannot @@ -33,18 +39,21 @@ //! neighbour-sync round pushes hints, empty or not, so capability is learned //! within a cycle. -use std::collections::{HashMap, HashSet}; +use std::collections::{HashMap, HashSet, VecDeque}; +use std::future::Future; use std::sync::Arc; use std::time::{Duration, Instant}; use ant_protocol::pointer::{Pointer, PointerState, POINTER_WIRE_LEN}; +use futures::future::BoxFuture; use futures::stream::{self, StreamExt}; use parking_lot::Mutex; use rand::Rng; use saorsa_core::identity::PeerId; use saorsa_core::{P2PNode, TrustEvent}; -use tokio::sync::{mpsc, RwLock, Semaphore}; -use tokio::task::JoinHandle; +use tokio::sync::{mpsc, OwnedSemaphorePermit, RwLock, Semaphore}; +use tokio::task::{spawn_blocking, JoinHandle}; +use tokio::time::Instant as TokioInstant; use tokio_util::sync::CancellationToken; use tokio_util::task::TaskTracker; @@ -54,7 +63,8 @@ use crate::payment::{ PaymentVerifier, VerificationContext, MAX_PAYMENT_PROOF_SIZE_BYTES, MIN_PAYMENT_PROOF_SIZE_BYTES, }; -use crate::pointer::store::{Inspected, PointerStore}; +use crate::pointer::store::{Inspected, PointerStore, PutOutcome}; +use crate::pointer::{FinalStateWitness, FinalityCheck}; use crate::replication::admission; use crate::replication::commitment_state::ResponderCommitmentState; use crate::replication::config::{ @@ -79,6 +89,49 @@ const MAX_CONCURRENT_OFFERS: usize = 16; /// Fetch and state requests served at once. const MAX_CONCURRENT_SERVES: usize = 32; +/// Fetch and state requests admitted at once, served or waiting to be. Past +/// it a request is dropped at admission rather than queued, as a chunk fetch +/// is, so a flood costs a bounded number of tasks. +const MAX_SERVES_OUTSTANDING: usize = MAX_CONCURRENT_SERVES * 4; + +/// Of those, how many one peer may have: twice the [`MAX_REQUESTS_PER_PEER`] +/// an honest node lets itself have outstanding at any one peer, so its +/// requests are never dropped, even those arriving before the reply to the +/// last has released its place here, while one flooding peer cannot take the +/// room every other peer's requests need. +const MAX_SERVES_OUTSTANDING_PER_PEER: u32 = 16; + +/// Pointer requests this node has outstanding at any one peer, across repair, +/// possession checks and pruning together. As many as a repair asks at once, +/// so repair is not slowed; the rest wait their turn rather than go out and +/// be dropped by the peer's per-peer allowance, which would make an honest +/// peer look as though it had not answered. +const MAX_REQUESTS_PER_PEER: usize = VERIFICATION_CONCURRENCY; + +// An honest node's requests to one peer, repair included, never exceed what +// that peer admits from it, with room for a reply still releasing its place. +const _: () = assert!( + MAX_SERVES_OUTSTANDING_PER_PEER as usize >= 2 * MAX_REQUESTS_PER_PEER, + "a peer's serve allowance must cover what an honest node asks of it" +); + +/// Most peers remembered as speaking pointers. Only routed peers are, and a +/// routing table removal forgets one, so this only bounds what removals missed +/// (a lagging event stream) could leave behind. Capability forgotten this way +/// is learned again from the peer's next hint push. +const MAX_CAPABLE_PEERS: usize = 4096; + +/// Most hint pushes answering peers' sync requests at once. Each scans every +/// pointer this node holds, so a peer's sync request that finds them all +/// running gets no hints this time, and the next sync round delivers them. +const MAX_CONCURRENT_SYNC_ANSWERS: usize = 8; + +/// Most peers remembered as having been answered with hints. Only routing-table +/// peers are answered, so this only bounds what a lagging table could leave +/// behind; past it the peer answered longest ago is forgotten, never a new one +/// refused. +const MAX_ANSWERED_SYNC_PEERS: usize = 4096; + /// How often the pending hints are looked at. const VERIFICATION_TICK: Duration = Duration::from_millis(500); @@ -102,9 +155,90 @@ const UNDECIDED_BACKOFF: Duration = Duration::from_secs(30); /// Most prune candidates examined per pass. const MAX_PRUNE_CANDIDATES_PER_PASS: usize = 256; +/// How long a node spends asking its close group for a conflicting final +/// state before taking one (ADR-0018). +/// +/// It runs inside a client's PUT, after payment has been verified, so it adds +/// to whatever that verification took; it is kept short so a PUT whose +/// payment is quick stays inside the client's ten-second store timeout. A +/// group that has not answered by then is taken to hold nothing that +/// conflicts: silence is never a vote, here as anywhere else. +pub const FINAL_STATE_CHECK_BUDGET: Duration = Duration::from_secs(4); + +/// How long a finality check waits for its turn before it gives up and +/// answers [`FinalityCheck::Busy`]. +/// +/// With [`FINAL_STATE_CHECK_BUDGET`] after it, a check adds at most six +/// seconds to a PUT after its payment is verified. +pub const FINAL_STATE_CHECK_WAIT: Duration = Duration::from_secs(2); + +/// How many finality checks run at once. +/// +/// Each runs on its own share of the address space. Two checks for one +/// address never run together, so a burst of replays of one final state +/// costs the group one round of questions. +pub const FINAL_STATE_CHECK_STRIPES: usize = 64; + +/// How many addresses a node remembers proven final states for, oldest +/// forgotten first. +/// +/// A proof is a final state the close group was shown to hold, so it never +/// goes stale; the cap only bounds memory, at most two states of about 200 +/// bytes each per address. +const MAX_PROVEN_FINALS: usize = 16_384; + +/// Proven final states remembered per address. Two different ones are enough +/// to refuse every final state there: each conflicts with the other, and any +/// third with both. +const MAX_PROOFS_PER_ADDRESS: usize = 2; + +/// How long a look that found no conflict answers again for the same final +/// state without asking anyone. +/// +/// As long as a replay queued behind it can have waited for its turn, and no +/// longer, since a rival could reach the group after it. The answer is also +/// forgotten as soon as the write it cleared fails. +pub const FINAL_STATE_CLEAR_REUSE: Duration = FINAL_STATE_CHECK_WAIT; + +/// Most clear looks remembered for reuse. Each is at most +/// [`FINAL_STATE_CLEAR_REUSE`] old, so this only bounds a burst. +const MAX_CLEAR_LOOKS: usize = 4096; + /// One peer's answer about one address: the state it holds there, if any. type StateAnswer = ((PeerId, XorName), Option); +/// Whether `record` is exactly the state a quorum backed. +/// +/// The whole state, not only its identifier: the backed state is built from +/// peers' unauthenticated summaries, and a summary can pair a real state's +/// identifier with some other counter or target. +fn backs(record: &Pointer, wanted: &PointerState) -> bool { + record.state() == *wanted +} + +/// One peer's state response, read against the addresses it was asked about. +/// +/// An explicit "nothing held" is an answer. A summary about some other +/// address is no answer at all, and is left out, so that it does not count as +/// a peer that answered. +fn read_state_answers( + peer: PeerId, + addresses: &[XorName], + states: Vec>, +) -> Vec { + addresses + .iter() + .zip(states) + .filter_map(|(address, summary)| match summary { + None => Some(((peer, *address), None)), + Some(summary) if summary.address == *address => { + Some(((peer, *address), Some(PointerState::from(summary)))) + } + Some(_) => None, + }) + .collect() +} + /// A pointer state this node accepted from a paying client, to be offered to /// the rest of its close group. pub struct PointerFreshWrite { @@ -158,37 +292,63 @@ pub(crate) fn evaluate( if group_size == 0 || quorum_needed == 0 { return Verdict::Undecided; } + // Votes count together only for the same whole state. A summary is a + // peer's unauthenticated claim: one naming a real state's identifier with + // some other counter must not absorb, or reshape, the honest votes for it. let mut by_state: Vec<(PointerState, Vec)> = Vec::new(); for (peer, answer) in answers { let Some(state) = answer else { continue }; - match by_state - .iter_mut() - .find(|(known, _)| known.state_id == state.state_id) - { + match by_state.iter_mut().find(|(known, _)| known == state) { Some((_, holders)) => holders.push(*peer), None => by_state.push((*state, vec![*peer])), } } let beats_held = |state: &PointerState| held.is_none_or(|held| state.replaces(held)); + // The merge winner, and between two final states — which the merge rule + // leaves unordered — the one more of the group holds. Only a group wider + // than twice the quorum can back two at once, and then arrival order must + // not be what decides. + let prefer = |candidate: &(PointerState, Vec), + current: &(PointerState, Vec)| { + candidate.0.replaces(¤t.0) + || (!current.0.replaces(&candidate.0) && candidate.1.len() > current.1.len()) + }; let best = by_state .iter() .filter(|(state, holders)| holders.len() >= quorum_needed && beats_held(state)) .fold( None::<&(PointerState, Vec)>, |best, candidate| match best { - Some(current) if !candidate.0.replaces(¤t.0) => Some(current), + Some(current) if !prefer(candidate, current) => Some(current), _ => Some(candidate), }, ); + let unanswered = group_size.saturating_sub(answers.len()); if let Some((state, holders)) = best { + // A final state is adopted only as the strictly larger side, counting + // every peer that did not answer for its rival: a final state is + // never replaced, so two repairing nodes that each saw part of a tie + // would otherwise adopt opposite sides for good. That covers a rival + // seen among the answers and one only silent peers could hold. + if state.is_terminal() { + let rival_could_match = by_state.iter().any(|(other, others)| { + other.state_id != state.state_id + && beats_held(other) + && !other.replaces(state) + && !state.replaces(other) + && others.len().saturating_add(unanswered) >= holders.len() + }); + if rival_could_match || unanswered >= holders.len() { + return Verdict::Undecided; + } + } return Verdict::Adopt { state: *state, holders: holders.clone(), }; } - let unanswered = group_size.saturating_sub(answers.len()); let largest = by_state .iter() .filter(|(state, _)| beats_held(state)) @@ -202,6 +362,100 @@ pub(crate) fn evaluate( } } +/// What asking a peer for a pointer record came to. +enum Fetched { + /// A record that verifies and belongs at the address. + Record(Pointer), + /// A record that does not, which the peer has already been penalised for. + Invalid, + /// No record: no answer, an empty one, or one about another address. + Nothing, + /// A record this node could not check, or a request it never sent, for a + /// reason of its own. + LocalFailure, +} + +/// What asking a peer came to (see [`PointerReplication::ask`]). +enum Asked { + /// The peer's answer. + Answered(ReplicationMessageBody), + /// Sent, and no usable answer came back. + Silent, + /// Never sent: this node is shutting down, or its own requests to the + /// peer stayed busy for the whole timeout. + NotSent, +} + +/// One of `permits`, if one comes free within `timeout` and before +/// `shutdown`. Shutdown wins when both are ready at once. +async fn acquire_or_give_up( + permits: Arc, + timeout: Duration, + shutdown: &CancellationToken, +) -> Option { + let permit = tokio::select! { + biased; + () = shutdown.cancelled() => None, + permit = tokio::time::timeout(timeout, permits.acquire_owned()) => permit.ok()?.ok(), + }?; + (!shutdown.is_cancelled()).then_some(permit) +} + +/// What a possession check makes of one peer's answer about `fresh`. +#[derive(Debug, PartialEq, Eq)] +enum Possession { + /// The peer served `fresh` or a state that replaces it. + Holds, + /// The peer serves a final state other than `fresh`, which is final too: + /// the owner signed both and each node kept the one it took first, as + /// this one did. The peer holds what the merge rule told it to hold, so it + /// is not penalised for the owner's fork. Carries the peer's state id. + Forked(XorName), + /// The peer could not produce `fresh` or anything newer. + Missing, + /// Nothing to judge the peer on: it served a record that does not verify, + /// which the fetch has already charged it for, or this node could not + /// check what it served. + NotJudged, +} + +/// Judge one peer's answer in a possession check for `fresh`. +fn judge_possession(fetched: Fetched, fresh: &PointerState) -> Possession { + match fetched { + Fetched::Record(record) => { + let state = record.state(); + if state.state_id == fresh.state_id || state.replaces(fresh) { + Possession::Holds + } else if state.is_terminal() && fresh.is_terminal() { + Possession::Forked(state.state_id) + } else { + Possession::Missing + } + } + Fetched::Nothing => Possession::Missing, + Fetched::Invalid | Fetched::LocalFailure => Possession::NotJudged, + } +} + +/// Whether nothing but the map holds a peer's outbound permits. +/// +/// Every holder, a request waiting for a permit or one holding it, has a +/// clone, and clones are only made under the map's lock. So an entry nothing +/// else holds can be dropped without a later request getting a second set of +/// permits beside one still in use, which would let this node exceed +/// [`MAX_REQUESTS_PER_PEER`] at that peer. +fn outbound_idle(permits: &Arc) -> bool { + Arc::strong_count(permits) == 1 +} + +/// How many members a repair counts its quorum over: the whole close group, +/// however few of them this node can see (ADR-0016). A member it cannot see is +/// unanswered, so a thin routing table leaves a repair undecided instead of +/// shrinking the quorum to what little it can see. +fn repair_width(seen: usize, close_group_size: usize) -> usize { + seen.max(close_group_size) +} + /// Pointer replication for one node. See the module documentation. pub struct PointerReplication { store: PointerStore, @@ -214,16 +468,214 @@ pub struct PointerReplication { send_semaphore: Arc, offer_permits: Arc, serve_permits: Arc, + /// Admission ahead of `serve_permits`: see [`MAX_SERVES_OUTSTANDING`]. + serve_admission: Arc, + /// Requests each peer has admitted and not yet had answered. + serve_inflight: Arc>>, + /// This node's own requests outstanding at each peer (see + /// [`MAX_REQUESTS_PER_PEER`]). + outbound: Mutex>>, /// Peers that have sent a pointer message: the only ones asked anything. capable: Mutex>, /// Hinted states awaiting verification, by address. pending: Mutex>, /// When each held address was first seen continuously out of range. out_of_range: Mutex>, + /// The bounds on looks before a final state (ADR-0018). + finality: FinalityLooks, + /// Hint pushes answering sync requests that may run at once. + sync_answers: Arc, + /// When each peer was last answered with hints. + answered_syncs: Mutex, shutdown: CancellationToken, tracker: TaskTracker, } +/// Final states a finality check proved the close group holds, remembered so +/// that a paid final state which lost to one is refused again without asking +/// anyone. +#[derive(Default)] +struct ProvenFinals { + /// Up to [`MAX_PROOFS_PER_ADDRESS`] different proven states per address. + by_address: HashMap>, + /// Addresses in the order they were first proven, oldest first. + order: VecDeque, +} + +impl ProvenFinals { + /// A proven final state at `state.address` other than `state`. + fn conflict(&self, state: &PointerState) -> Option { + self.by_address + .get(&state.address)? + .iter() + .find(|proven| proven.state_id != state.state_id) + .copied() + } + + /// Remember `proven` beside what is already proven at its address, never + /// in place of it: a proof replaced would let the state it disproved in. + /// Forgets the oldest addresses past the cap. + fn remember(&mut self, proven: PointerState) { + if let Some(states) = self.by_address.get_mut(&proven.address) { + let known = states.iter().any(|state| state.state_id == proven.state_id); + if !known && states.len() < MAX_PROOFS_PER_ADDRESS { + states.push(proven); + } + } else { + self.by_address.insert(proven.address, vec![proven]); + self.order.push_back(proven.address); + } + while self.by_address.len() > MAX_PROVEN_FINALS { + let Some(oldest) = self.order.pop_front() else { + break; + }; + self.by_address.remove(&oldest); + } + } +} + +/// The bounds on looks before a final state (ADR-0018), apart from the +/// network so they can be tested without one. +/// +/// A look's answer is kept either way: a proof for good, a clear look briefly. +/// Looks for one address take turns, so replays queued behind a look find its +/// answer rather than asking again; at most [`FINAL_STATE_CHECK_STRIPES`] +/// run at once. +struct FinalityLooks { + proven: Mutex, + /// When each final state was last looked for and found clear, on the + /// runtime's clock, which the look's own time bounds use as well. + clear: Mutex>, + /// One turn per share of the address space. + turns: Vec>, +} + +impl FinalityLooks { + fn new() -> Self { + Self { + proven: Mutex::new(ProvenFinals::default()), + clear: Mutex::new(HashMap::new()), + turns: (0..FINAL_STATE_CHECK_STRIPES) + .map(|_| tokio::sync::Mutex::new(())) + .collect(), + } + } + + /// A proven final state that conflicts with `state`, asking nobody. + fn proven_conflict(&self, state: &PointerState) -> Option { + self.proven.lock().conflict(state) + } + + /// Whether a look found `state` clear recently enough to answer again. + fn recently_clear(&self, state: &PointerState, now: TokioInstant) -> bool { + self.clear + .lock() + .get(&(state.address, state.state_id)) + .is_some_and(|at| now.saturating_duration_since(*at) < FINAL_STATE_CLEAR_REUSE) + } + + /// Forget a clear look for `state`: the write it cleared did not land, so + /// a retry must look again. + fn forget_clear(&self, state: &PointerState) { + self.clear.lock().remove(&(state.address, state.state_id)); + } + + fn note_clear(&self, state: &PointerState, now: TokioInstant) { + let mut clear = self.clear.lock(); + if clear.len() >= MAX_CLEAR_LOOKS { + clear.retain(|_, at| now.saturating_duration_since(*at) < FINAL_STATE_CLEAR_REUSE); + } + if clear.len() < MAX_CLEAR_LOOKS { + clear.insert((state.address, state.state_id), now); + } + } + + /// Answer whether `state` may be taken, running `look` only when neither + /// a proof nor a recent clear look answers it already. + async fn check(&self, state: &PointerState, look: Look) -> FinalityCheck + where + Look: FnOnce() -> Fut + Send, + Fut: Future> + Send, + { + if !state.is_terminal() { + return FinalityCheck::Clear; + } + if let Some(conflict) = self.proven_conflict(state) { + return FinalityCheck::Conflict(conflict); + } + // By the last byte: the addresses one node is responsible for share + // their leading bits, so the first byte would put them all on one turn. + let stripe = usize::from(state.address.last().copied().unwrap_or_default()) + % FINAL_STATE_CHECK_STRIPES; + let Some(turn) = self.turns.get(stripe) else { + return FinalityCheck::Busy; + }; + let Ok(_turn) = tokio::time::timeout(FINAL_STATE_CHECK_WAIT, turn.lock()).await else { + return FinalityCheck::Busy; + }; + // A look that held the turn may have answered this meanwhile. + if let Some(conflict) = self.proven_conflict(state) { + return FinalityCheck::Conflict(conflict); + } + if self.recently_clear(state, TokioInstant::now()) { + return FinalityCheck::Clear; + } + match tokio::time::timeout(FINAL_STATE_CHECK_BUDGET, look()).await { + Ok(Some(record)) => { + let conflict = record.state(); + self.proven.lock().remember(conflict); + FinalityCheck::Conflict(conflict) + } + Ok(None) => { + self.note_clear(state, TokioInstant::now()); + FinalityCheck::Clear + } + // Silence proves nothing, and is not kept for reuse either: the + // next look may be answered. + Err(_) => { + debug!( + "Close group of pointer {} did not answer the finality check in time", + hex::encode(state.address) + ); + FinalityCheck::Clear + } + } + } +} + +/// When each peer was last sent hints in answer to its own sync request. +#[derive(Default)] +struct AnsweredSyncs { + at: HashMap, +} + +impl AnsweredSyncs { + /// Whether `peer` may be answered at `now`, and if so note it: not if it + /// was answered less than `spacing` ago. A full map forgets the peers + /// answered longest ago rather than refuse anyone new. + fn admit(&mut self, peer: PeerId, now: Instant, spacing: Duration) -> bool { + let recent = |at: &Instant| now.saturating_duration_since(*at) < spacing; + if self.at.get(&peer).is_some_and(recent) { + return false; + } + if self.at.len() >= MAX_ANSWERED_SYNC_PEERS && !self.at.contains_key(&peer) { + self.at.retain(|_, at| recent(at)); + } + if self.at.len() >= MAX_ANSWERED_SYNC_PEERS && !self.at.contains_key(&peer) { + let oldest = self + .at + .iter() + .min_by_key(|(_, at)| **at) + .map(|(peer, _)| *peer); + if let Some(oldest) = oldest { + self.at.remove(&oldest); + } + } + self.at.insert(peer, now); + true + } +} + impl PointerReplication { /// Pointer replication over `store`, sharing the engine's resources. #[allow(clippy::too_many_arguments)] @@ -248,9 +700,15 @@ impl PointerReplication { send_semaphore, offer_permits: Arc::new(Semaphore::new(MAX_CONCURRENT_OFFERS)), serve_permits: Arc::new(Semaphore::new(MAX_CONCURRENT_SERVES)), + serve_admission: Arc::new(Semaphore::new(MAX_SERVES_OUTSTANDING)), + serve_inflight: Arc::new(RwLock::new(HashMap::new())), + outbound: Mutex::new(HashMap::new()), capable: Mutex::new(HashSet::new()), pending: Mutex::new(HashMap::new()), out_of_range: Mutex::new(HashMap::new()), + finality: FinalityLooks::new(), + sync_answers: Arc::new(Semaphore::new(MAX_CONCURRENT_SYNC_ANSWERS)), + answered_syncs: Mutex::new(AnsweredSyncs::default()), shutdown, tracker, } @@ -265,8 +723,22 @@ impl PointerReplication { // Capability // ----------------------------------------------------------------------- - fn mark_capable(&self, peer: &PeerId) { - self.capable.lock().insert(*peer); + /// Remember that `peer` speaks pointers, if it is in the routing table. + /// + /// Only a routed peer is remembered: it is the only kind this node ever + /// asks, and the routing table's removals are what forget it again, so the + /// set cannot outgrow the table however many identities send a message. + async fn mark_capable(&self, peer: &PeerId) { + if !self.p2p.dht_manager().is_in_routing_table(peer).await { + return; + } + let mut capable = self.capable.lock(); + if capable.len() >= MAX_CAPABLE_PEERS && !capable.contains(peer) { + if let Some(evicted) = capable.iter().next().copied() { + capable.remove(&evicted); + } + } + capable.insert(*peer); } /// Whether `peer` has sent a pointer message, and may therefore be asked. @@ -277,6 +749,24 @@ impl PointerReplication { /// Forget a peer that left the routing table. pub(crate) fn forget_peer(&self, peer: &PeerId) { self.capable.lock().remove(peer); + let mut outbound = self.outbound.lock(); + if outbound.get(peer).is_some_and(outbound_idle) { + outbound.remove(peer); + } + } + + /// The permits bounding this node's requests to `peer`. + fn outbound_permits(&self, peer: &PeerId) -> Arc { + let mut outbound = self.outbound.lock(); + if outbound.len() >= MAX_CAPABLE_PEERS && !outbound.contains_key(peer) { + // Peers nothing is outstanding at hold no state worth keeping. + outbound.retain(|_, permits| !outbound_idle(permits)); + } + Arc::clone( + outbound + .entry(*peer) + .or_insert_with(|| Arc::new(Semaphore::new(MAX_REQUESTS_PER_PEER))), + ) } /// Addresses currently awaiting verification. Tests only. @@ -311,19 +801,41 @@ impl PointerReplication { body: ReplicationMessageBody, timeout: Duration, ) -> Option { + match self.ask(peer, body, timeout).await { + Asked::Answered(body) => Some(body), + Asked::Silent | Asked::NotSent => None, + } + } + + /// Ask `peer`, saying whether a missing answer was the peer's silence or + /// this node never sending the request at all. + /// + /// The request waits for one of this node's own permits for `peer` first, + /// but no longer than `timeout` and not past shutdown: a request that + /// could not be sent in that time is dropped as not sent, never left + /// waiting, and never held against the peer. + async fn ask(&self, peer: &PeerId, body: ReplicationMessageBody, timeout: Duration) -> Asked { + let Some(_permit) = + acquire_or_give_up(self.outbound_permits(peer), timeout, &self.shutdown).await + else { + return Asked::NotSent; + }; let msg = ReplicationMessage { request_id: rand::thread_rng().gen::(), body, }; - let bytes = msg.encode().ok()?; - let response = self + let Ok(bytes) = msg.encode() else { + return Asked::NotSent; + }; + let Ok(response) = self .p2p .send_request(peer, REPLICATION_PROTOCOL_ID, bytes, timeout) .await - .ok()?; + else { + return Asked::Silent; + }; ReplicationMessage::decode(&response.data) - .ok() - .map(|msg| msg.body) + .map_or(Asked::Silent, |msg| Asked::Answered(msg.body)) } async fn penalise(&self, peer: &PeerId) { @@ -339,31 +851,59 @@ impl PointerReplication { /// back: a valid signature and the right address. Anything else counts /// against the peer. async fn fetch_record(&self, peer: &PeerId, address: &XorName) -> Option { - let body = self - .request( + match self.fetch_outcome(peer, address).await { + Fetched::Record(record) => Some(record), + Fetched::Invalid | Fetched::Nothing | Fetched::LocalFailure => None, + } + } + + /// [`Self::fetch_record`], saying whether a failure was a record that did + /// not verify, which has already cost the peer, or no record at all. + async fn fetch_outcome(&self, peer: &PeerId, address: &XorName) -> Fetched { + let body = match self + .ask( peer, ReplicationMessageBody::PointerFetchRequest(PointerFetchRequest { address: *address, }), self.config.fetch_request_timeout, ) - .await?; + .await + { + Asked::Answered(body) => body, + Asked::Silent => return Fetched::Nothing, + // Never asked: nothing to judge the peer on. + Asked::NotSent => return Fetched::LocalFailure, + }; let ReplicationMessageBody::PointerFetchResponse(response) = body else { - return None; + return Fetched::Nothing; }; if response.address != *address { - return None; + return Fetched::Nothing; } - let bytes = response.record?; - match Pointer::from_bytes(&bytes) { - Ok(record) if record.address() == *address => Some(record), + let Some(bytes) = response.record else { + return Fetched::Nothing; + }; + // Parsing verifies the ML-DSA signature, milliseconds of CPU a peer + // can demand once per record it serves: off the async executor, as + // every other pointer signature check is (ADR-0016). + let Ok(parsed) = spawn_blocking(move || Pointer::from_bytes(&bytes)).await else { + // The check itself failed here, which says nothing about the peer. + warn!( + "Could not verify the pointer record {peer} served for {}", + hex::encode(address) + ); + return Fetched::LocalFailure; + }; + match parsed { + Ok(record) if record.address() == *address => Fetched::Record(record), Ok(_) | Err(_) => { debug!( "Peer {peer} served an invalid pointer record for {}", hex::encode(address) ); self.penalise(peer).await; - None + Fetched::Invalid } } } @@ -386,6 +926,38 @@ impl PointerReplication { .spawn(async move { this.push_hints(&peers).await }); } + /// Answer a peer's sync request, admitted and still fresh, with this + /// node's pointer hints, in the background: how a peer, including one + /// still bootstrapping, learns the pointers it should hold. + /// + /// Each answer scans every pointer held, so only a routing-table peer is + /// answered, at most once per shortest sync interval, which an honest + /// peer never syncs faster than, and at most + /// [`MAX_CONCURRENT_SYNC_ANSWERS`] answers run at once. A request that + /// finds them all busy gets no hints this time; the next sync round sends + /// them anyway. + pub(crate) fn answer_sync_with_hints(self: &Arc, peer: PeerId) { + let Ok(permit) = Arc::clone(&self.sync_answers).try_acquire_owned() else { + debug!("Not answering {peer}'s sync with pointer hints: too many answers running"); + return; + }; + let this = Arc::clone(self); + self.tracker.spawn(async move { + let _permit = permit; + if !this.p2p.dht_manager().is_in_routing_table(&peer).await { + return; + } + let admitted = this.answered_syncs.lock().admit( + peer, + Instant::now(), + this.config.neighbor_sync_interval_min, + ); + if admitted { + this.push_hints(&[peer]).await; + } + }); + } + /// Push hints to `peers` and wait until they are sent. This is what each /// neighbour-sync round runs in the background; tests call it to drive a /// round. @@ -433,7 +1005,7 @@ impl PointerReplication { /// Take in hints from `source`: queue each hinted state this node should /// hold and lacks, or holds an older state than. pub(crate) async fn handle_hints(&self, source: PeerId, hints: Vec) { - self.mark_capable(&source); + self.mark_capable(&source).await; if hints.is_empty() { return; } @@ -506,12 +1078,22 @@ impl PointerReplication { }) } - /// Verify and fetch whatever is due now. Also the tests' way to drive it. - pub async fn verify_due(&self) { + /// Whether this node should look for anything to repair now. + async fn may_repair(&self) -> bool { // As for chunks (ADR-0011): a node that cannot store what it would // fetch does not spend the network's time finding it. The hints wait, // and come back each round if they are dropped meanwhile. if self.chunks.capacity_verdict() == CapacityVerdict::Full { + return false; + } + // A node still bootstrapping sees too little of its close group to + // judge what a quorum of it holds. The hints wait for it. + !*self.is_bootstrapping.read().await + } + + /// Verify and fetch whatever is due now. Also the tests' way to drive it. + pub async fn verify_due(&self) { + if !self.may_repair().await { return; } let now = Instant::now(); @@ -561,11 +1143,17 @@ impl PointerReplication { }) .collect(); let held = self.store.state(address); + // The quorum is counted over the whole close group, as + // ADR-0016 has it: a member this node cannot see, because its + // routing table is thin, counts as unanswered, never as a + // smaller group. Otherwise one peer in a thin view could vote + // alone for an owner-signed state nobody paid to store. + let width = repair_width(group.len(), self.config.close_group_size); let verdict = evaluate( held.as_ref(), - group.len(), + width, &answered, - self.config.quorum_needed(group.len()), + self.config.quorum_needed(width), ); (*address, verdict) }) @@ -641,17 +1229,7 @@ impl PointerReplication { let Some(ReplicationMessageBody::PointerStateResponse(response)) = body else { return Vec::new(); }; - addresses - .iter() - .zip(response.states) - .map(|(address, summary)| { - // An answer about a different address is no answer. - let state = summary - .filter(|summary| summary.address == *address) - .map(PointerState::from); - ((peer, *address), state) - }) - .collect() + read_state_answers(peer, &addresses, response.states) }) .buffer_unordered(VERIFICATION_CONCURRENCY) .collect() @@ -671,9 +1249,10 @@ impl PointerReplication { let Some(record) = self.fetch_record(holder, &address).await else { continue; }; - // The quorum backed this state. A holder that has moved on since - // serves a different one, which that quorum says nothing about. - if record.state_id() != wanted.state_id { + // The quorum backed this state, exactly. A holder that has moved + // on since serves a different one, which that quorum says nothing + // about. + if !backs(&record, &wanted) { continue; } match self.store_verified(record, None).await { @@ -708,28 +1287,67 @@ impl PointerReplication { return Ok(false); } let reservation = self.chunks.reserve(POINTER_WIRE_LEN as u64)?; - self.store.commit(record, Some(reservation)).await?; - Ok(true) + // Held now, whether this wrote it or it was already here; a commit + // that found a state it loses to is not. + let outcome = self.store.commit(record, Some(reservation)).await?; + Ok(!matches!(outcome, PutOutcome::Stale)) } // ----------------------------------------------------------------------- // Serving // ----------------------------------------------------------------------- + /// Admit a fetch or state request from `source`, or drop it. + /// + /// Admitted before a task exists, fairly across peers, as a chunk fetch + /// is: each admitted request would otherwise be a task waiting on a serve + /// permit with nothing bounding how many, and one peer could take every + /// place. + async fn admit_serve(&self, source: &PeerId, kind: &str) -> Option { + match super::admit_bounded_responder( + &self.serve_admission, + &self.serve_inflight, + source, + MAX_SERVES_OUTSTANDING, + MAX_SERVES_OUTSTANDING_PER_PEER, + ) + .await + { + Ok(guard) => Some(guard), + Err(failure) => { + debug!("Dropping a pointer {kind} request from {source}: {failure}"); + None + } + } + } + /// Answer a fetch request in the background. - pub(crate) fn serve_fetch_detached( + pub(crate) async fn serve_fetch_detached( self: &Arc, source: PeerId, request: PointerFetchRequest, request_id: u64, rr_message_id: Option, ) { - self.mark_capable(&source); + let Some(guard) = self.admit_serve(&source, "fetch").await else { + return; + }; let this = Arc::clone(self); self.tracker.spawn(async move { - let Ok(_permit) = this.serve_permits.acquire().await else { + let _guard = guard; + // Shutting down ends the wait, and the work behind it. + let permit = tokio::select! { + biased; + () = this.shutdown.cancelled() => return, + permit = this.serve_permits.acquire() => permit, + }; + let Ok(_permit) = permit else { return; }; + if this.shutdown.is_cancelled() { + return; + } + this.mark_capable(&source).await; // `get` verifies the signature before serving, so a record damaged // on this disk is never handed on. let record = match this.store.get(&request.address).await { @@ -751,19 +1369,41 @@ impl PointerReplication { } /// Answer a state request in the background, from the index. - pub(crate) fn serve_state_detached( + pub(crate) async fn serve_state_detached( self: &Arc, source: PeerId, request: PointerStateRequest, request_id: u64, rr_message_id: Option, ) { - self.mark_capable(&source); + // An honest peer never asks about more than this at once. The decoded + // request would otherwise be held by its task for as long as it waits, + // and the wire allows one far larger. + if request.addresses.len() > MAX_POINTER_STATE_REQUEST_ADDRESSES { + debug!( + "Dropping a pointer state request from {source}: {} addresses", + request.addresses.len() + ); + return; + } + let Some(guard) = self.admit_serve(&source, "state").await else { + return; + }; let this = Arc::clone(self); self.tracker.spawn(async move { - let Ok(_permit) = this.serve_permits.acquire().await else { + let _guard = guard; + let permit = tokio::select! { + biased; + () = this.shutdown.cancelled() => return, + permit = this.serve_permits.acquire() => permit, + }; + let Ok(_permit) = permit else { return; }; + if this.shutdown.is_cancelled() { + return; + } + this.mark_capable(&source).await; let states = request .addresses .iter() @@ -875,7 +1515,6 @@ impl PointerReplication { source: PeerId, offer: PointerFreshOffer, ) { - self.mark_capable(&source); let Ok(permit) = Arc::clone(&self.offer_permits).try_acquire_owned() else { info!("Dropping a fresh pointer offer from {source}: too many in flight"); return; @@ -883,6 +1522,7 @@ impl PointerReplication { let this = Arc::clone(self); self.tracker.spawn(async move { let _permit = permit; + this.mark_capable(&source).await; this.accept_offer(source, offer).await; }); } @@ -929,6 +1569,16 @@ impl PointerReplication { { return; } + // A final state already proven to have lost is dropped before it + // buys a signature check. + let needs_final_check = state.is_terminal() + && !self + .store + .remembered(&state.address) + .is_some_and(|known| known.is_terminal()); + if needs_final_check && self.finality.proven_conflict(&state).is_some() { + return; + } let record = match self.store.verify(parsed).await { Ok(record) => record, Err(e) => { @@ -953,10 +1603,21 @@ impl PointerReplication { ); return; } - match self + // A final state this node neither holds nor remembers is looked for + // in the group first, exactly as a client PUT of one is: the peer + // offering it may simply be the side of a race this node has not + // heard the other side of. A node that lost its own final state is + // admitted only that state, so restoring it asks nobody. + if needs_final_check && !self.offered_final_is_clear(&source, &state).await { + return; + } + let stored = self .store_verified(record, Some(self.config.paid_list_close_group_size)) - .await - { + .await; + if !matches!(stored, Ok(true)) && needs_final_check { + self.finality.forget_clear(&state); + } + match stored { Ok(true) => debug!( "Stored fresh pointer {} from {source}", hex::encode(state.address) @@ -969,6 +1630,116 @@ impl PointerReplication { } } + // ----------------------------------------------------------------------- + // Finality + // ----------------------------------------------------------------------- + + /// Whether a freshly offered final state may be taken: no peer proved a + /// different one, and the look was not too busy to run. + async fn offered_final_is_clear(&self, source: &PeerId, state: &PointerState) -> bool { + match self.check_final(state).await { + FinalityCheck::Clear => true, + FinalityCheck::Conflict(conflict) => { + info!( + "Refusing fresh final pointer state {} at {} from {source}: the close \ + group already holds final state {}", + hex::encode(state.state_id), + hex::encode(state.address), + hex::encode(conflict.state_id) + ); + false + } + FinalityCheck::Busy => { + debug!( + "Dropping fresh final pointer state {} at {} from {source}: too many \ + finality checks running", + hex::encode(state.state_id), + hex::encode(state.address) + ); + false + } + } + } + + /// Ask the close group whether a final state at `state.address`, other + /// than `state`, is already held, and have a peer that says so prove it + /// by serving the signed record. + /// + /// Asked before this node takes a final state it does not hold. A final + /// state is replaced by nothing, so a node that took a second one would + /// hold it for good; asking first is what keeps a former owner from + /// finalizing an address again on a node that had not yet heard it was + /// final, such as one that joined the group since. + /// + /// A peer's word is not enough to refuse: a state summary is a claim + /// anyone can make, and one dishonest peer could otherwise block every + /// handover. A signed record is not a claim. Only the owner can sign a + /// final state, so one that verifies is the owner's own proof that it + /// finalized the pointer before. + /// + /// A proof is remembered, beside any other for the address, so the same + /// loser is refused again without a question; a clear look is reused for + /// the same state for [`FINAL_STATE_CLEAR_REUSE`]. Checks for one address + /// wait their turn, so a burst of replays queued behind a look costs that + /// one round, and at most [`FINAL_STATE_CHECK_STRIPES`] run at once; one + /// that cannot start within [`FINAL_STATE_CHECK_WAIT`] answers `Busy`. + /// Once started it is bounded by [`FINAL_STATE_CHECK_BUDGET`], and only + /// peers that have sent a pointer message are asked. A group that cannot + /// be asked in time finds nothing, and the write goes ahead on the merge + /// rule alone: a race is a fork the client detects, not one this can + /// prevent. + pub async fn check_final(&self, state: &PointerState) -> FinalityCheck { + self.finality + .check(state, || self.find_conflicting_final(state)) + .await + } + + /// The body of [`Self::check_final`], without its bounds. + async fn find_conflicting_final(&self, state: &PointerState) -> Option { + let self_id = *self.p2p.peer_id(); + let address = state.address; + let peers: Vec = self + .p2p + .dht_manager() + .find_closest_nodes_local(&address, self.config.close_group_size) + .await + .into_iter() + .map(|node| node.peer_id) + .filter(|peer| *peer != self_id && self.is_capable(peer)) + .collect(); + first_proven_final( + peers, + state, + |peer| async move { self.ask_state(&peer, &address).await }, + |peer| async move { self.fetch_record(&peer, &address).await }, + ) + .await + } + + /// Ask one peer which state it holds at `address`. + async fn ask_state(&self, peer: &PeerId, address: &XorName) -> Option { + let body = self + .request( + peer, + ReplicationMessageBody::PointerStateRequest(PointerStateRequest { + addresses: vec![*address], + }), + FINAL_STATE_CHECK_BUDGET, + ) + .await?; + let ReplicationMessageBody::PointerStateResponse(response) = body else { + return None; + }; + // An answer about a different address is no answer. + response + .states + .into_iter() + .next() + .flatten() + .filter(|summary| summary.address == *address) + .map(PointerState::from) + } + // ----------------------------------------------------------------------- // Possession // ----------------------------------------------------------------------- @@ -998,37 +1769,56 @@ impl PointerReplication { /// that cannot produce `fresh` or a state that replaces it. pub async fn check_possession(&self, fresh: PointerState, peers: &[PeerId]) { let self_id = *self.p2p.peer_id(); - let group: HashSet = self - .p2p - .dht_manager() - .find_closest_nodes_local_with_self(&fresh.address, self.config.close_group_size) - .await - .into_iter() - .map(|node| node.peer_id) - .collect(); for peer in peers { // A peer that has left the group owes nothing, and one that cannot // be asked is never judged on its silence. - if *peer == self_id || !group.contains(peer) || !self.is_capable(peer) { + if *peer == self_id || !self.owes(peer, &fresh.address).await { continue; } - let holds = self - .fetch_record(peer, &fresh.address) - .await - .is_some_and(|record| { - let state = record.state(); - state.state_id == fresh.state_id || state.replaces(&fresh) - }); - if !holds { - warn!( - "Peer {peer} does not hold pointer {} it was offered", - hex::encode(fresh.address) - ); - self.penalise(peer).await; + let fetched = self.fetch_outcome(peer, &fresh.address).await; + match judge_possession(fetched, &fresh) { + Possession::Missing => {} + // Said loudly, because a read of this pointer now depends on + // which side most of the group is on. + Possession::Forked(held) => { + warn!( + "Pointer {} is forked: peer {peer} holds final state {}, this node \ + offered final state {}", + hex::encode(fresh.address), + hex::encode(held), + hex::encode(fresh.state_id) + ); + continue; + } + Possession::Holds | Possession::NotJudged => continue, + } + // Asking can wait on this peer's other requests and on the peers + // before it, and it may have left the group meanwhile. It is + // judged on what it owes now, not on what it owed when this began. + if !self.owes(peer, &fresh.address).await { + continue; } + warn!( + "Peer {peer} does not hold pointer {} it was offered", + hex::encode(fresh.address) + ); + self.penalise(peer).await; } } + /// Whether `peer` is responsible for `address` in this node's view, and + /// can be asked about it. + async fn owes(&self, peer: &PeerId, address: &XorName) -> bool { + self.is_capable(peer) + && self + .p2p + .dht_manager() + .find_closest_nodes_local_with_self(address, self.config.close_group_size) + .await + .iter() + .any(|node| node.peer_id == *peer) + } + // ----------------------------------------------------------------------- // Pruning // ----------------------------------------------------------------------- @@ -1047,6 +1837,9 @@ impl PointerReplication { allow_remote: bool, commitment_state: Option<&ResponderCommitmentState>, ) { + // Replaced records kept for audits go when their retention does, + // whether or not another update comes along to drop them. + self.store.drop_expired_superseded(); let committed = |address: &XorName| commitment_state.is_some_and(|cs| cs.is_held(address)); let self_id = *self.p2p.peer_id(); let retention = storage_admission_width(self.config.close_group_size); @@ -1160,11 +1953,71 @@ impl PointerReplication { } } +impl FinalStateWitness for PointerReplication { + fn check_final<'a>(&'a self, state: &'a PointerState) -> BoxFuture<'a, FinalityCheck> { + Box::pin(Self::check_final(self, state)) + } + + fn proven_conflict(&self, state: &PointerState) -> Option { + self.finality.proven_conflict(state) + } + + fn forget_clear(&self, state: &PointerState) { + self.finality.forget_clear(state); + } +} + +/// The first final state other than `state` that one of `peers` claims with +/// `ask` and then proves with `fetch`. +/// +/// Each peer's question and fetch run as one pipeline, all of them at once, +/// and the first proof wins. A peer that claims a rival and then stalls its +/// fetch holds up nobody else's proof: were the fetches made one at a time +/// after the claims, that one peer would hold the check until its budget ran +/// out, and a finality check that runs out finds nothing. +async fn first_proven_final( + peers: Vec, + state: &PointerState, + ask: Ask, + fetch: Fetch, +) -> Option +where + Ask: Fn(PeerId) -> AskFut + Sync, + AskFut: Future> + Send, + Fetch: Fn(PeerId) -> FetchFut + Sync, + FetchFut: Future> + Send, +{ + let rival = |held: &PointerState| held.is_terminal() && held.state_id != state.state_id; + let (ask, fetch, rival) = (&ask, &fetch, &rival); + let width = peers.len().max(1); + let mut proofs = stream::iter(peers) + .map(|peer| async move { + let held = ask(peer).await?; + if !rival(&held) { + return None; + } + let record = fetch(peer).await?; + // The record must be the state the peer claimed: a claim the + // served record does not back is no proof, even if what it serves + // is some other rival. + (backs(&record, &held) && rival(&record.state())).then_some(record) + }) + .buffer_unordered(width); + while let Some(proof) = proofs.next().await { + if proof.is_some() { + return proof; + } + } + None +} + #[cfg(test)] #[allow(clippy::unwrap_used, clippy::expect_used, clippy::panic)] mod tests { use super::*; - use ant_protocol::pointer::{PointerTarget, PointerTargetKind}; + use ant_protocol::pointer::{PointerTarget, PointerTargetKind, FINAL_COUNTER}; + use saorsa_pqc::api::sig::ml_dsa_65; + use std::sync::atomic::{AtomicUsize, Ordering}; fn state(counter: u64, target: u8, id: u8) -> PointerState { PointerState { @@ -1183,6 +2036,163 @@ mod tests { list.iter().map(|(p, s)| (peer(*p), *s)).collect() } + /// A peer's outbound permits are dropped only once nothing holds them, so + /// a request in flight keeps the one set later requests share. + fn record(counter: u64, target: u8) -> Pointer { + let (pk, sk) = ml_dsa_65().generate_keypair_from_seed(&[3; 32]); + let target = PointerTarget::new(PointerTargetKind::Chunk, [target; 32]); + Pointer::sign(&sk, &pk, counter, target).expect("sign") + } + + #[tokio::test(start_paused = true)] + async fn a_request_that_cannot_be_sent_in_time_is_given_up_not_queued() { + let permits = Arc::new(Semaphore::new(1)); + let shutdown = CancellationToken::new(); + let held = acquire_or_give_up(Arc::clone(&permits), Duration::from_secs(1), &shutdown) + .await + .expect("a free permit is taken"); + + // Busy past the timeout: given up, not left waiting. + assert!( + acquire_or_give_up(Arc::clone(&permits), Duration::from_secs(1), &shutdown) + .await + .is_none() + ); + + // Shutdown ends a wait at once, however long it could still run. + shutdown.cancel(); + let waited = tokio::time::Instant::now(); + assert!( + acquire_or_give_up(Arc::clone(&permits), Duration::from_secs(3600), &shutdown) + .await + .is_none() + ); + assert!(waited.elapsed() < Duration::from_secs(1)); + drop(held); + + // With a permit free and shutdown already under way, nothing is sent, + // every time. + for _ in 0..64 { + assert!( + acquire_or_give_up(Arc::clone(&permits), Duration::from_secs(1), &shutdown) + .await + .is_none() + ); + } + } + + #[test] + fn a_possession_check_judges_each_answer_once() { + let fresh = record(5, 1); + let judge = |fetched| judge_possession(fetched, &fresh.state()); + assert_eq!(judge(Fetched::Record(fresh.clone())), Possession::Holds); + assert_eq!( + judge(Fetched::Record(record(6, 9))), + Possession::Holds, + "a newer state is as good" + ); + assert_eq!(judge(Fetched::Record(record(4, 1))), Possession::Missing); + assert_eq!(judge(Fetched::Nothing), Possession::Missing); + // The fetch already charged the peer for an invalid record; charging + // it again as missing would count one bad answer twice. + assert_eq!(judge(Fetched::Invalid), Possession::NotJudged); + // A failure on this node says nothing about the peer. + assert_eq!(judge(Fetched::LocalFailure), Possession::NotJudged); + + // Two different final states: the peer kept the one it took first, as + // the merge rule tells it to, so it is not charged for the owner's fork. + let taken = final_record(3, 0x01); + let rival = final_record(3, 0x02); + assert_eq!( + judge_possession(Fetched::Record(rival.clone()), &taken.state()), + Possession::Forked(rival.state().state_id) + ); + assert_eq!( + judge_possession(Fetched::Record(record(6, 9)), &taken.state()), + Possession::Missing, + "only a final state stands in for another final state" + ); + } + + #[test] + fn a_peer_is_answered_with_hints_once_per_sync_interval() { + let spacing = Duration::from_secs(600); + let start = Instant::now(); + let mut answered = AnsweredSyncs::default(); + assert!(answered.admit(peer(1), start, spacing)); + assert!( + !answered.admit(peer(1), start + Duration::from_secs(1), spacing), + "a second request inside the interval gets no second scan" + ); + assert!( + answered.admit(peer(2), start, spacing), + "other peers are not held up" + ); + assert!(answered.admit(peer(1), start + spacing, spacing)); + + // A full map forgets the peer answered longest ago; a new peer is + // never refused for it. + let mut full = AnsweredSyncs::default(); + let mut first = None; + for i in 0..MAX_ANSWERED_SYNC_PEERS { + let id = u32::try_from(i).expect("fits"); + let mut bytes = [0u8; 32]; + if let Some(prefix) = bytes.get_mut(..4) { + prefix.copy_from_slice(&id.to_be_bytes()); + } + let at = start + Duration::from_millis(u64::from(id)); + let each = PeerId::from_bytes(bytes); + first.get_or_insert(each); + assert!(full.admit(each, at, spacing)); + } + let late = start + Duration::from_secs(10); + assert!( + full.admit(peer(0xEE), late, spacing), + "a new peer is answered" + ); + assert_eq!(full.at.len(), MAX_ANSWERED_SYNC_PEERS); + let first = first.expect("one was admitted"); + assert!(!full.at.contains_key(&first), "the oldest was forgotten"); + } + + #[tokio::test] + async fn outbound_permits_in_use_are_never_dropped() { + let permits = Arc::new(Semaphore::new(MAX_REQUESTS_PER_PEER)); + assert!(outbound_idle(&permits)); + let held = Arc::clone(&permits) + .acquire_owned() + .await + .expect("a permit"); + assert!(!outbound_idle(&permits), "a held permit keeps the set"); + drop(held); + assert!(outbound_idle(&permits)); + } + + /// A node that sees one member of the close group, as one still filling + /// its routing table does, cannot adopt a state on that member's word: the + /// members it cannot see count as unanswered. + #[test] + fn a_thin_view_of_the_group_cannot_adopt_on_one_vote() { + let config = ReplicationConfig::default(); + let unpaid = state(9, 1, 90); + let width = repair_width(1, config.close_group_size); + let got = evaluate( + None, + width, + &answers(&[(1, Some(unpaid))]), + config.quorum_needed(width), + ); + assert!( + matches!(got, Verdict::Undecided), + "one visible peer must not decide, got {got:?}" + ); + assert_eq!( + repair_width(9, config.close_group_size), + 9, + "a view wider than the configured group is counted as it is" + ); + } + #[test] fn a_state_a_quorum_holds_is_adopted_from_its_holders() { let newer = state(3, 1, 30); @@ -1297,8 +2307,394 @@ mod tests { } } + #[test] + fn a_node_holding_a_final_state_adopts_nothing_else() { + // Not even another final state the whole group holds: the merge rule + // replaces a final state with nothing, and repair is the merge rule + // applied to what the group says. + let held = state(u64::MAX, 9, 90); + let other = state(u64::MAX, 1, 91); + let got = evaluate( + Some(&held), + 7, + &answers(&[ + (1, Some(other)), + (2, Some(other)), + (3, Some(other)), + (4, Some(other)), + (5, Some(other)), + ]), + 4, + ); + assert_eq!(got, Verdict::Refused); + } + + #[test] + fn of_two_final_states_with_quorum_the_one_more_hold_is_adopted() { + // Two final states are unordered, so only a group wide enough to back + // both can get here. Arrival order must not pick; the larger side does. + let fewer = state(u64::MAX, 1, 70); + let more = state(u64::MAX, 9, 71); + for order in [[fewer, more], [more, fewer]] { + let mut list = Vec::new(); + let mut next = 1u8; + for candidate in order { + let holders = if candidate.state_id == more.state_id { + 5 + } else { + 4 + }; + for _ in 0..holders { + list.push((next, Some(candidate))); + next += 1; + } + } + // Everyone answered, so the four cannot become five. + match evaluate(None, 9, &answers(&list), 4) { + Verdict::Adopt { state, holders } => { + assert_eq!(state.state_id, more.state_id); + assert_eq!(holders.len(), 5); + } + other => panic!("expected the larger side, got {other:?}"), + } + } + } + + #[test] + fn a_silent_peer_that_could_tie_two_final_states_blocks_adoption() { + // Eight peers, quorum four: four hold one final state, three another, + // one is silent. The silent one could hold the second, making a tie, + // so neither side is adopted, whichever answered first. + let four = state(u64::MAX, 1, 90); + let three = state(u64::MAX, 9, 91); + for order in [[four, three], [three, four]] { + let mut list = Vec::new(); + let mut next = 1u8; + for candidate in order { + let count = if candidate.state_id == four.state_id { + 4 + } else { + 3 + }; + for _ in 0..count { + list.push((next, Some(candidate))); + next += 1; + } + } + assert_eq!(evaluate(None, 8, &answers(&list), 4), Verdict::Undecided); + } + + // A lone final state with quorum, and as many silent peers as hold + // it, could meet an unseen rival: undecided too. One fewer silent + // peer and it is adopted. + let alone = state(u64::MAX, 5, 92); + let held: Vec<(u8, Option)> = (1..=4).map(|p| (p, Some(alone))).collect(); + assert_eq!(evaluate(None, 8, &answers(&held), 4), Verdict::Undecided); + assert!(matches!( + evaluate(None, 7, &answers(&held), 4), + Verdict::Adopt { .. } + )); + } + + #[test] + fn two_final_states_backed_equally_are_adopted_by_neither_order() { + // A group of eight with a quorum of four can back two final states + // four to four. Whichever answered first must not decide. + let one = state(u64::MAX, 1, 80); + let other = state(u64::MAX, 9, 81); + for order in [[one, other], [other, one]] { + let mut list = Vec::new(); + let mut next = 1u8; + for candidate in order { + for _ in 0..4 { + list.push((next, Some(candidate))); + next += 1; + } + } + assert_eq!(evaluate(None, 8, &answers(&list), 4), Verdict::Undecided); + } + } + + #[test] + fn a_forged_summary_does_not_absorb_the_votes_for_a_final_state() { + // One peer names final state A's identifier with a lower counter. It + // must stand alone, not carry A's three honest votes as a non-final + // state of four and so slip past the guard for final states. + let a = state(u64::MAX, 1, 93); + let forged = PointerState { + counter: u64::MAX - 1, + ..a + }; + let b = state(u64::MAX, 9, 94); + let mut list = vec![(1u8, Some(forged))]; + list.extend((2..=4).map(|p| (p, Some(a)))); + list.extend((5..=7).map(|p| (p, Some(b)))); + assert!(!matches!( + evaluate(None, 8, &answers(&list), 4), + Verdict::Adopt { .. } + )); + } + + #[tokio::test] + async fn a_claim_that_names_a_rival_with_another_target_proves_nothing() { + // The claim reuses a real rival's identifier with some other target; + // the peer then serves the real rival. The record is not what was + // claimed, so it is no proof. + let taking = final_record(18, 0x01); + let rival = final_record(18, 0xAA); + let mut claim = rival.state(); + claim.target = PointerTarget::new(PointerTargetKind::Chunk, [0xCC; 32]); + let found = first_proven_final( + vec![peer(1)], + &taking.state(), + |_| async move { Some(claim) }, + |_| { + let record = rival.clone(); + async move { Some(record) } + }, + ) + .await; + assert!(found.is_none()); + } + + #[test] + fn a_record_backs_only_the_whole_state_a_quorum_named() { + let record = final_record(17, 0x01); + assert!(backs(&record, &record.state())); + let forged = PointerState { + counter: 7, + ..record.state() + }; + assert!( + !backs(&record, &forged), + "a summary pairing the record's identifier with another counter is not backed" + ); + } + + #[test] + fn a_summary_for_another_address_is_no_answer_at_all() { + let (a, b, c) = ([1u8; 32], [2u8; 32], [3u8; 32]); + let mut elsewhere = state(3, 1, 95); + elsewhere.address = [9u8; 32]; + let mut here = state(3, 1, 96); + here.address = c; + let read = read_state_answers( + peer(1), + &[a, b, c], + vec![None, Some(elsewhere.into()), Some(here.into())], + ); + assert_eq!(read, vec![((peer(1), a), None), ((peer(1), c), Some(here))]); + } + #[test] fn a_group_nobody_can_be_asked_in_is_undecided() { assert_eq!(evaluate(None, 0, &[], 0), Verdict::Undecided); } + + fn final_record(seed: u8, target: u8) -> Pointer { + let (pk, sk) = ml_dsa_65().generate_keypair_from_seed(&[seed; 32]); + let target = PointerTarget::new(PointerTargetKind::Pointer, [target; 32]); + Pointer::sign(&sk, &pk, FINAL_COUNTER, target).expect("sign") + } + + #[tokio::test(start_paused = true)] + async fn a_claimant_that_stalls_its_fetch_does_not_hide_a_rival_another_peer_proves() { + // One peer claims a rival final state first and then never serves + // it; an honest peer claims the same rival a moment later and serves + // it at once. The honest proof must land inside the budget. + let taking = final_record(9, 0x01); + let rival = final_record(9, 0xAA); + let (staller, honest) = (peer(1), peer(2)); + let found = tokio::time::timeout( + FINAL_STATE_CHECK_BUDGET, + first_proven_final( + vec![staller, honest], + &taking.state(), + |asked| { + let claim = rival.state(); + async move { + if asked == honest { + tokio::time::sleep(Duration::from_millis(100)).await; + } + Some(claim) + } + }, + |asked| { + let record = rival.clone(); + async move { + if asked == staller { + std::future::pending::<()>().await; + } + Some(record) + } + }, + ), + ) + .await; + assert_eq!( + found.ok().flatten().map(|record| record.state_id()), + Some(rival.state_id()), + "the honest peer's proof was hidden behind the stalled fetch" + ); + } + + #[tokio::test] + async fn only_a_different_final_state_that_is_served_is_a_proof() { + let taking = final_record(10, 0x01); + let rival = final_record(10, 0xAA); + let (same, unbacked, open) = (peer(1), peer(2), peer(3)); + let found = first_proven_final( + vec![same, unbacked, open], + &taking.state(), + |asked| { + let claim = if asked == same { + taking.state() + } else if asked == unbacked { + rival.state() + } else { + state(3, 1, 1) + }; + async move { Some(claim) } + }, + // The peer claiming a rival serves the state being taken instead. + |_| { + let record = taking.clone(); + async move { Some(record) } + }, + ) + .await; + assert!( + found.is_none(), + "a claim the record does not back is no proof" + ); + } + + #[tokio::test] + async fn a_peer_that_serves_a_different_rival_than_it_claimed_proves_nothing() { + let taking = final_record(16, 0x01); + let claimed = final_record(16, 0xAA); + let served = final_record(16, 0xBB); + let found = first_proven_final( + vec![peer(1)], + &taking.state(), + |_| { + let claim = claimed.state(); + async move { Some(claim) } + }, + |_| { + let record = served.clone(); + async move { Some(record) } + }, + ) + .await; + assert!(found.is_none()); + } + + #[test] + fn a_proven_final_state_refuses_others_and_forgets_the_oldest_past_the_cap() { + let mut proven = ProvenFinals::default(); + let held = final_record(11, 0xAA).state(); + let other = final_record(11, 0x01).state(); + proven.remember(held); + assert_eq!(proven.conflict(&other), Some(held)); + assert_eq!( + proven.conflict(&held), + None, + "a proof is no conflict with itself" + ); + + for i in 0..MAX_PROVEN_FINALS { + let mut filler = held; + filler.address = [0; 32]; + if let Some(slot) = filler.address.get_mut(..8) { + slot.copy_from_slice(&(i as u64 + 1).to_be_bytes()); + } + proven.remember(filler); + } + assert_eq!(proven.by_address.len(), MAX_PROVEN_FINALS); + assert_eq!(proven.order.len(), MAX_PROVEN_FINALS); + assert_eq!( + proven.conflict(&other), + None, + "the oldest proof is forgotten first" + ); + } + + #[test] + fn a_second_proof_never_replaces_the_first() { + // Both sides of a fork proven at one address: each is refused by the + // other, so a later look that proved the other side cannot let back + // in the state the first look disproved. + let mut proven = ProvenFinals::default(); + let a = final_record(12, 0xAA).state(); + let b = final_record(12, 0xBB).state(); + let c = final_record(12, 0xCC).state(); + proven.remember(a); + proven.remember(b); + proven.remember(c); + assert_eq!(proven.conflict(&b), Some(a)); + assert_eq!(proven.conflict(&a), Some(b)); + assert!(proven.conflict(&c).is_some(), "a third is refused by both"); + assert_eq!( + proven.by_address.get(&a.address).map(Vec::len), + Some(MAX_PROOFS_PER_ADDRESS) + ); + } + + #[tokio::test(start_paused = true)] + async fn looks_for_addresses_sharing_leading_bits_run_side_by_side() { + // The addresses a node is responsible for share their leading bits. + // Two such looks, each longer than a check waits for its turn, must + // both run rather than one wait and come back Busy. + let looks = FinalityLooks::new(); + let mut first = final_record(14, 0x01).state(); + let mut second = final_record(15, 0x01).state(); + first.address = [0xAB; 32]; + second.address = [0xAB; 32]; + if let Some(last) = second.address.last_mut() { + *last = 0xAC; + } + let slow = || async { + tokio::time::sleep(FINAL_STATE_CHECK_WAIT + Duration::from_secs(1)).await; + None + }; + let (a, b) = tokio::join!(looks.check(&first, slow), looks.check(&second, slow)); + assert_eq!(a, FinalityCheck::Clear); + assert_eq!(b, FinalityCheck::Clear, "the second look was held up"); + } + + #[tokio::test(start_paused = true)] + async fn replays_queued_behind_a_look_reuse_its_answer() { + let looks = FinalityLooks::new(); + let taking = final_record(13, 0x01); + let asked = AtomicUsize::new(0); + let look = || { + asked.fetch_add(1, Ordering::SeqCst); + async { + tokio::time::sleep(Duration::from_millis(500)).await; + None + } + }; + let state = taking.state(); + let checks = (0..8).map(|_| looks.check(&state, look)); + let answers = futures::future::join_all(checks).await; + assert!(answers.iter().all(|answer| *answer == FinalityCheck::Clear)); + assert_eq!(asked.load(Ordering::SeqCst), 1, "eight replays, one look"); + + // Past the reuse window the group is asked again. + tokio::time::sleep(FINAL_STATE_CLEAR_REUSE).await; + assert_eq!(looks.check(&state, look).await, FinalityCheck::Clear); + assert_eq!(asked.load(Ordering::SeqCst), 2); + + // A proof answers every replay of the loser without a look. + let loser = final_record(13, 0x02).state(); + let rival = final_record(13, 0xAA); + let proved = looks.check(&loser, || async { Some(rival.clone()) }).await; + assert_eq!(proved, FinalityCheck::Conflict(rival.state())); + assert_eq!( + looks.check(&loser, look).await, + FinalityCheck::Conflict(rival.state()) + ); + assert_eq!(asked.load(Ordering::SeqCst), 2); + } } diff --git a/src/replication/protocol.rs b/src/replication/protocol.rs index b70ae9a7..ee1c1696 100644 --- a/src/replication/protocol.rs +++ b/src/replication/protocol.rs @@ -637,6 +637,23 @@ pub(crate) fn log_traffic_summary() { get_commitment_by_pin_response_rx_count = rc(16), "replication traffic summary (cumulative)" ); + crate::logging::info!( + target: "ant_node::replication::traffic", + group = 4, + pointer_fresh_offer_tx_bytes = tb(17), pointer_fresh_offer_tx_count = tc(17), + pointer_fresh_offer_rx_bytes = rb(17), pointer_fresh_offer_rx_count = rc(17), + pointer_hints_tx_bytes = tb(18), pointer_hints_tx_count = tc(18), + pointer_hints_rx_bytes = rb(18), pointer_hints_rx_count = rc(18), + pointer_fetch_request_tx_bytes = tb(19), pointer_fetch_request_tx_count = tc(19), + pointer_fetch_request_rx_bytes = rb(19), pointer_fetch_request_rx_count = rc(19), + pointer_fetch_response_tx_bytes = tb(20), pointer_fetch_response_tx_count = tc(20), + pointer_fetch_response_rx_bytes = rb(20), pointer_fetch_response_rx_count = rc(20), + pointer_state_request_tx_bytes = tb(21), pointer_state_request_tx_count = tc(21), + pointer_state_request_rx_bytes = rb(21), pointer_state_request_rx_count = rc(21), + pointer_state_response_tx_bytes = tb(22), pointer_state_response_tx_count = tc(22), + pointer_state_response_rx_bytes = rb(22), pointer_state_response_rx_count = rc(22), + "replication traffic summary (cumulative)" + ); } // --------------------------------------------------------------------------- @@ -1185,7 +1202,7 @@ pub enum AuditResponse { /// commitment, or a [`SubtreeAuditResponse::Rejected`] if it genuinely cannot /// (for a recently gossiped pinned commitment a rejection is a confirmed /// failure, since the responder retains its recently gossiped commitments for a -/// bounded TTL window). +/// bounded TTL window, unless it is [`RejectKind::Transient`]). #[derive(Debug, Clone, Serialize, Deserialize)] pub struct SubtreeAuditChallenge { /// Unique challenge identifier. @@ -1272,9 +1289,11 @@ pub enum RejectKind { /// retention and in-window auditing this is provable repudiation of a root /// the node published → CONFIRMED failure. UnknownCommitment, - /// A transient, recoverable local condition (e.g. a storage read error), - /// emitted only after the responder's read retries failed. Routed to the - /// timeout lane (holder credit revoked, no trust penalty). + /// A transient, recoverable local condition: a storage read error the + /// responder's read retries did not clear, a round 1 refused because its + /// pointer roots would not fit the session budget, or a pointer record + /// round 1 bound that is no longer kept (ADR-0019). Routed to the timeout + /// lane (holder credit revoked, no trust penalty). Transient, /// Any other rejection (wrong target peer, no commitment state, malformed /// proof plan, oversized slice challenge, …). CONFIRMED failure. @@ -1374,16 +1393,18 @@ pub enum SubtreeSliceItem { PointerRecord { /// The requested key: the pointer's address. key: XorName, - /// The record held now, and the one an update replaced since round 1 - /// if there was one, each in its canonical encoding. At most - /// [`MAX_POINTER_RECORDS_PER_ITEM`]: round 1 bound one of them, and - /// the responder cannot tell which without keeping round 1's answer. + /// The record round 1 read, in its canonical encoding, found by the + /// nonced root round 1 reported over it (ADR-0019). At most + /// [`MAX_POINTER_RECORDS_PER_ITEM`], and the auditor accepts whichever + /// reproduces that root. records: Vec>, }, } -/// Most records one [`SubtreeSliceItem::PointerRecord`] may carry: the one -/// held now and the one it replaced. +/// Most records one [`SubtreeSliceItem::PointerRecord`] may carry. +/// +/// A responder that keeps what round 1 bound serves one. Before it did, it +/// served the record held now and the one an update last replaced (ADR-0016). pub const MAX_POINTER_RECORDS_PER_ITEM: usize = 2; /// Response to a [`SubtreeSliceChallenge`] (round 2). diff --git a/src/replication/storage_commitment_audit.rs b/src/replication/storage_commitment_audit.rs index d2835610..5df180f5 100644 --- a/src/replication/storage_commitment_audit.rs +++ b/src/replication/storage_commitment_audit.rs @@ -14,6 +14,7 @@ use std::sync::Arc; use std::time::{Duration, Instant}; use crate::logging::{debug, info, warn}; +use bytes::Bytes; use rand::Rng; use crate::ant_protocol::XorName; @@ -748,6 +749,27 @@ const _: () = assert!( "a replaced pointer record must outlive the audit session that may be owed it" ); +/// What round 1 bound for each committed pointer it proved, by address. +/// +/// Each is the nonced root round 1 reported over the record it read +/// (ADR-0019). Round 2 is owed that record, whatever the pointer holds by the +/// time it asks. +pub type PointerBindings = HashMap; + +/// The pointer bindings a round-1 response reports. Only a proof reports any. +#[must_use] +pub fn pointer_bindings(response: &SubtreeAuditResponse) -> PointerBindings { + match response { + SubtreeAuditResponse::Proof { proof, .. } => proof + .leaves + .iter() + .filter(|leaf| is_pointer_leaf(leaf)) + .map(|leaf| (leaf.key, leaf.nonced_root)) + .collect(), + _ => PointerBindings::new(), + } +} + /// Whether a round-1 leaf commits a pointer (ADR-0016): committed under /// [`pointer_leaf_hash`] of its key, at the fixed record length. fn is_pointer_leaf(leaf: &SubtreeLeaf) -> bool { @@ -977,7 +999,7 @@ pub(crate) fn verify_slice_response( // belong at the committed address, and be the record round 1 bound its // nonced root over, which the responder had to read before it knew what // would be sampled. An update between the rounds does not fail an - // honest holder: it serves the record it held then beside the new one. + // honest holder: it serves the record round 1 read (ADR-0019). if is_pointer_leaf(leaf) { if let Err(reason) = verify_pointer_item(nonce, challenged_peer_bytes, leaf, items) { return AuditVerdict::Fail(reason); @@ -1729,10 +1751,11 @@ pub async fn handle_subtree_slice_challenge( is_bootstrapping: bool, commitment_state: Option<&Arc>, ) -> SubtreeSliceResponse { - handle_subtree_slice_challenge_with_pointers( + serve_slice_challenge( challenge, storage, None, + None, self_peer_id, is_bootstrapping, commitment_state, @@ -1741,8 +1764,13 @@ pub async fn handle_subtree_slice_challenge( } /// [`handle_subtree_slice_challenge`] for a node that also commits pointers -/// (ADR-0016): a committed pointer is answered with its whole signed record. -#[allow(clippy::too_many_lines)] +/// (ADR-0016), without what round 1 bound for them. +/// +/// Kept for callers of the earlier signature, and answering as it did: a +/// committed pointer is served as the record held now and the newest one an +/// update replaced, and one no longer held is absent. That fails an honest holder after two updates between +/// the rounds, which is what ADR-0019 fixes; the engine uses +/// [`handle_subtree_slice_challenge_with_pointer_bindings`]. pub async fn handle_subtree_slice_challenge_with_pointers( challenge: &SubtreeSliceChallenge, storage: &ChunkStore, @@ -1750,6 +1778,54 @@ pub async fn handle_subtree_slice_challenge_with_pointers( self_peer_id: &PeerId, is_bootstrapping: bool, commitment_state: Option<&Arc>, +) -> SubtreeSliceResponse { + serve_slice_challenge( + challenge, + storage, + pointers, + None, + self_peer_id, + is_bootstrapping, + commitment_state, + ) + .await +} + +/// [`handle_subtree_slice_challenge`] for a node that also commits pointers +/// (ADR-0016): a committed pointer is answered with its whole signed record, +/// the one `bound` says round 1 read (ADR-0019). +pub async fn handle_subtree_slice_challenge_with_pointer_bindings( + challenge: &SubtreeSliceChallenge, + storage: &ChunkStore, + pointers: Option<&PointerStore>, + bound: &PointerBindings, + self_peer_id: &PeerId, + is_bootstrapping: bool, + commitment_state: Option<&Arc>, +) -> SubtreeSliceResponse { + serve_slice_challenge( + challenge, + storage, + pointers, + Some(bound), + self_peer_id, + is_bootstrapping, + commitment_state, + ) + .await +} + +/// The body of the slice-challenge handlers. `bound` is `None` for the +/// earlier entry point, which serves pointers as it did before ADR-0019. +#[allow(clippy::too_many_lines, clippy::too_many_arguments)] +async fn serve_slice_challenge( + challenge: &SubtreeSliceChallenge, + storage: &ChunkStore, + pointers: Option<&PointerStore>, + bound: Option<&PointerBindings>, + self_peer_id: &PeerId, + is_bootstrapping: bool, + commitment_state: Option<&Arc>, ) -> SubtreeSliceResponse { if is_bootstrapping { return SubtreeSliceResponse::Bootstrapping { @@ -1882,10 +1958,10 @@ pub async fn handle_subtree_slice_challenge_with_pointers( for key in key_order { let indices = indices_by_key.remove(&key).unwrap_or_default(); if built.tree().commits_pointer(&key) { - // The bytes held now, exactly as round 1 read them, and the record - // an update replaced since, if any: round 1 bound one of the two. - // The auditor verifies whichever it checks, so nothing is verified - // here. + // The record round 1 read, found by the root it reported over it + // among what is held now and every record updates have replaced + // since. The auditor verifies whatever is served, so nothing is + // verified here. let Some(store) = pointers else { items.push(SubtreeSliceItem::Absent { key }); continue; @@ -1900,11 +1976,44 @@ pub async fn handle_subtree_slice_challenge_with_pointers( } } }; - let records: Vec> = current.into_iter().chain(store.superseded(&key)).collect(); - if records.is_empty() { - items.push(SubtreeSliceItem::Absent { key }); - } else { - items.push(SubtreeSliceItem::PointerRecord { key, records }); + let Some(bound) = bound else { + // As before ADR-0019, the record held and the newest one an + // update replaced; but a node that no longer holds the pointer + // is absent, whatever replaced records it still keeps. + match current { + Some(current) => { + let mut records = vec![current]; + records.extend(store.superseded(&key)); + items.push(SubtreeSliceItem::PointerRecord { key, records }); + } + None => items.push(SubtreeSliceItem::Absent { key }), + } + continue; + }; + match pointer_records( + challenge, + &key, + bound.get(&key), + current, + store.superseded_all(&key), + ) { + PointerServe::Record(record) => { + items.push(SubtreeSliceItem::PointerRecord { + key, + records: vec![record], + }); + } + PointerServe::Absent => items.push(SubtreeSliceItem::Absent { key }), + PointerServe::Unavailable => { + return SubtreeSliceResponse::Rejected { + challenge_id: challenge.challenge_id, + kind: RejectKind::Transient, + reason: format!( + "cannot serve the record round 1 read for pointer {}", + hex::encode(key) + ), + } + } } continue; } @@ -1921,6 +2030,59 @@ pub async fn handle_subtree_slice_challenge_with_pointers( } } +/// What round 2 serves for a committed pointer. +enum PointerServe { + /// The record round 1 read. + Record(Vec), + /// The pointer is no longer held, which is a lost pointer, whatever + /// replaced records this node still keeps in memory. + Absent, + /// The pointer is held, but this node cannot serve the record round 1 + /// read: it has aged out or been evicted to keep memory bounded, or no + /// root for it was kept, which a session this node opened never lacks. A + /// local limit, not a lost pointer, so it is reported as one rather than + /// proved wrong. + Unavailable, +} + +/// Choose what round 2 serves for the pointer at `key` (ADR-0019), from the +/// record held now and the records updates replaced: the one that reproduces +/// the root round 1 reported. +/// +/// Without that root nothing is served for a pointer still held. Neither the +/// record held now nor the replaced records kept can show that no update came +/// since round 1 read it, since the kept ones are capped, so serving one would +/// be a guess that fails the node when it is wrong. +fn pointer_records( + challenge: &SubtreeSliceChallenge, + key: &XorName, + bound: Option<&[u8; 32]>, + current: Option>, + replaced: Vec, +) -> PointerServe { + // A node that no longer holds the pointer has lost it. Records an update + // replaced are kept only to prove what round 1 read; they do not make up + // for the pointer itself being gone. + let Some(current) = current else { + return PointerServe::Absent; + }; + let Some(root) = bound else { + return PointerServe::Unavailable; + }; + let reproduces = |record: &[u8]| { + nonced_block_root(&challenge.nonce, &challenge.challenged_peer_id, key, record) == *root + }; + if reproduces(¤t) { + return PointerServe::Record(current); + } + replaced + .into_iter() + .find(|record| reproduces(record)) + .map_or(PointerServe::Unavailable, |record| { + PointerServe::Record(Vec::from(record)) + }) +} + /// Outcome of serving all requested openings for one committed key. enum KeyServe { /// Openings built for this key; append to the response. @@ -2749,7 +2911,9 @@ mod tests { mod pointer_audit_tests { use super::*; use crate::replication::commitment::MerkleTree; + use crate::replication::commitment::MAX_COMMITMENT_KEY_COUNT; use crate::replication::commitment_state::BuiltCommitment; + use crate::replication::subtree::max_subtree_leaves; use crate::storage::ChunkStoreConfig; use ant_protocol::pointer::{PointerTarget, PointerTargetKind}; use saorsa_pqc::api::sig::ml_dsa_65; @@ -2849,11 +3013,40 @@ mod pointer_audit_tests { .response } + /// Round 2 as the engine serves it: with what round 1 bound for each + /// pointer it opens, as the live session carries it. async fn round2( &self, nonce: [u8; 32], openings: &[(SubtreeLeaf, u32)], ) -> Vec { + let bound = openings + .iter() + .map(|(leaf, _)| leaf) + .filter(|leaf| is_pointer_leaf(leaf)) + .map(|leaf| (leaf.key, leaf.nonced_root)) + .collect(); + self.round2_bound(nonce, openings, &bound).await + } + + async fn round2_bound( + &self, + nonce: [u8; 32], + openings: &[(SubtreeLeaf, u32)], + bound: &PointerBindings, + ) -> Vec { + match self.round2_response(nonce, openings, bound).await { + SubtreeSliceResponse::Items { items, .. } => items, + other => panic!("expected items, got {other:?}"), + } + } + + async fn round2_response( + &self, + nonce: [u8; 32], + openings: &[(SubtreeLeaf, u32)], + bound: &PointerBindings, + ) -> SubtreeSliceResponse { let challenge = SubtreeSliceChallenge { challenge_id: CHALLENGE_ID, nonce, @@ -2867,19 +3060,16 @@ mod pointer_audit_tests { }) .collect(), }; - match handle_subtree_slice_challenge_with_pointers( + handle_subtree_slice_challenge_with_pointer_bindings( &challenge, &self.storage, Some(&self.pointers), + bound, &self.peer, false, Some(&self.state), ) .await - { - SubtreeSliceResponse::Items { items, .. } => items, - other => panic!("expected items, got {other:?}"), - } } /// Round 1 as the auditor sees it: the proof, checked against the pin. @@ -2978,6 +3168,247 @@ mod pointer_audit_tests { ); } + /// A record an update replaces is kept for longer than the slowest audit + /// can take: a round 1 over the largest subtree an auditor waits for, + /// which may have read the pointer at its start, then the session its + /// round 2 must arrive within. + #[test] + fn a_replaced_record_outlives_the_slowest_audit_by_default() { + let largest = + usize::try_from(max_subtree_leaves(MAX_COMMITMENT_KEY_COUNT)).expect("fits a usize"); + let slowest = + ReplicationConfig::default().audit_response_timeout(largest) + SUBTREE_SESSION_TTL; + assert!( + SUPERSEDED_RETENTION > slowest, + "{SUPERSEDED_RETENTION:?} must outlive {slowest:?}" + ); + } + + /// The earlier entry point knows nothing of round 1's roots, and answers + /// as it did before them: an unchanged pointer is served, and passes, + /// rather than reported as a transient failure. + #[tokio::test] + async fn the_earlier_entry_point_still_serves_an_unchanged_pointer() { + let responder = Responder::new(24, 24).await; + let nonce = mixed_nonce(responder.committed().tree()); + let openings = openings(&responder.proved_leaves(nonce).await); + let items = legacy_round2(&responder, nonce, &openings).await; + assert!(items + .iter() + .any(|item| matches!(item, SubtreeSliceItem::PointerRecord { .. }))); + assert!(matches!( + verify_slice_response(&openings, &nonce, &responder.peer_bytes, &items), + AuditVerdict::Pass { .. } + )); + } + + /// After one update between the rounds, the earlier entry point serves + /// the record held and the one it replaced, newest first, and passes. + #[tokio::test] + async fn the_earlier_entry_point_still_serves_across_one_update() { + let responder = Responder::new(24, 24).await; + let nonce = mixed_nonce(responder.committed().tree()); + let openings = openings(&responder.proved_leaves(nonce).await); + let updated = first_pointer(&openings); + let owner = (0..24u8) + .find(|owner| pointer(*owner, 1).address() == updated) + .expect("the opened pointer is one of ours"); + let replaced = responder + .pointers + .record_bytes(&updated) + .await + .expect("read") + .expect("held"); + let newer = pointer(owner, 2).to_bytes(); + responder.pointers.put_bytes(&newer).await.expect("update"); + + let items = legacy_round2(&responder, nonce, &openings).await; + let served = items.iter().find_map(|item| match item { + SubtreeSliceItem::PointerRecord { key, records } if *key == updated => { + Some(records.clone()) + } + _ => None, + }); + assert_eq!( + served, + Some(vec![newer, replaced]), + "the record held, then the one it replaced" + ); + assert!(matches!( + verify_slice_response(&openings, &nonce, &responder.peer_bytes, &items), + AuditVerdict::Pass { .. } + )); + } + + /// Through the earlier entry point too, a pointer updated and then lost + /// is absent, not proved by the record the update replaced. + #[tokio::test] + async fn the_earlier_entry_point_reports_a_lost_pointer_absent() { + let responder = Responder::new(24, 24).await; + let nonce = mixed_nonce(responder.committed().tree()); + let openings = openings(&responder.proved_leaves(nonce).await); + let lost = first_pointer(&openings); + let owner = (0..24u8) + .find(|owner| pointer(*owner, 1).address() == lost) + .expect("the opened pointer is one of ours"); + responder + .pointers + .put_bytes(&pointer(owner, 2).to_bytes()) + .await + .expect("update"); + assert!(responder.pointers.delete(&lost).await.expect("delete")); + + let items = legacy_round2(&responder, nonce, &openings).await; + assert!(items + .iter() + .any(|item| matches!(item, SubtreeSliceItem::Absent { key } if *key == lost))); + assert_eq!( + verify_slice_response(&openings, &nonce, &responder.peer_bytes, &items), + AuditVerdict::Fail(AuditFailureReason::KeyAbsent) + ); + } + + /// Round 2 through the earlier entry point, which takes no bindings. + async fn legacy_round2( + responder: &Responder, + nonce: [u8; 32], + openings: &[(SubtreeLeaf, u32)], + ) -> Vec { + let challenge = SubtreeSliceChallenge { + challenge_id: CHALLENGE_ID, + nonce, + challenged_peer_id: responder.peer_bytes, + expected_commitment_hash: responder.committed().hash(), + openings: openings + .iter() + .map(|(leaf, block_index)| SubtreeSliceOpening { + key: leaf.key, + block_index: *block_index, + }) + .collect(), + }; + let response = handle_subtree_slice_challenge_with_pointers( + &challenge, + &responder.storage, + Some(&responder.pointers), + &responder.peer, + false, + Some(&responder.state), + ) + .await; + let SubtreeSliceResponse::Items { items, .. } = response else { + panic!("expected items, got {response:?}"); + }; + items + } + + /// Round 1 binds each pointer it proves, and nothing else. + #[tokio::test] + async fn round_one_binds_exactly_the_pointers_it_proves() { + let responder = Responder::new(24, 24).await; + let nonce = mixed_nonce(responder.committed().tree()); + let response = responder.round1(nonce).await; + let SubtreeAuditResponse::Proof { proof, .. } = &response else { + panic!("expected a proof, got {response:?}"); + }; + let expected: PointerBindings = proof + .leaves + .iter() + .filter(|leaf| responder.committed().tree().commits_pointer(&leaf.key)) + .map(|leaf| (leaf.key, leaf.nonced_root)) + .collect(); + assert!(!expected.is_empty(), "the subtree holds a pointer"); + assert_eq!(pointer_bindings(&response), expected); + assert!( + pointer_bindings(&SubtreeAuditResponse::Bootstrapping { + challenge_id: CHALLENGE_ID + }) + .is_empty(), + "only a proof binds anything" + ); + } + + /// Without the root round 1 reported, which a session this node opened + /// never lacks but a direct caller can, a pointer still held is reported + /// as a transient failure, updated or not: the node cannot show which + /// record round 1 read, and guessing would risk a confirmed failure it did + /// not earn. + #[tokio::test] + async fn without_the_bound_root_a_pointer_is_unavailable_not_failed() { + let responder = Responder::new(24, 24).await; + let nonce = mixed_nonce(responder.committed().tree()); + let openings = openings(&responder.proved_leaves(nonce).await); + let unavailable = |response: SubtreeSliceResponse| { + matches!( + response, + SubtreeSliceResponse::Rejected { + kind: RejectKind::Transient, + .. + } + ) + }; + + assert!(unavailable( + responder + .round2_response(nonce, &openings, &PointerBindings::new()) + .await + )); + + let updated = first_pointer(&openings); + let owner = (0..24u8) + .find(|owner| pointer(*owner, 1).address() == updated) + .expect("the opened pointer is one of ours"); + responder + .pointers + .put_bytes(&pointer(owner, 2).to_bytes()) + .await + .expect("update"); + assert!(unavailable( + responder + .round2_response(nonce, &openings, &PointerBindings::new()) + .await + )); + } + + /// A root round 1 reported that nothing held now reproduces, as when the + /// record it read has been evicted to keep memory bounded, is reported as + /// a transient failure while the pointer is still held, and as absent + /// once nothing at all is. + #[tokio::test] + async fn a_bound_record_no_longer_held_is_unavailable_not_failed() { + let responder = Responder::new(24, 24).await; + let nonce = mixed_nonce(responder.committed().tree()); + let openings = openings(&responder.proved_leaves(nonce).await); + let target = first_pointer(&openings); + let evicted: PointerBindings = openings + .iter() + .map(|(leaf, _)| leaf) + .filter(|leaf| is_pointer_leaf(leaf)) + .map(|leaf| (leaf.key, [0xEE; 32])) + .collect(); + + assert!(matches!( + responder.round2_response(nonce, &openings, &evicted).await, + SubtreeSliceResponse::Rejected { + kind: RejectKind::Transient, + .. + } + )); + + for (leaf, _) in openings.iter().filter(|(leaf, _)| is_pointer_leaf(leaf)) { + responder.pointers.delete(&leaf.key).await.expect("delete"); + } + let items = responder.round2_bound(nonce, &openings, &evicted).await; + assert!(items + .iter() + .any(|item| matches!(item, SubtreeSliceItem::Absent { key } if *key == target))); + assert_eq!( + verify_slice_response(&openings, &nonce, &responder.peer_bytes, &items), + AuditVerdict::Fail(AuditFailureReason::KeyAbsent), + "a pointer lost outright is still a confirmed failure" + ); + } + /// The commitment binds which pointers are held, not their state, so an /// owner updating a pointer mid-audit cannot fail the node holding it. #[tokio::test] @@ -3003,6 +3434,59 @@ mod pointer_audit_tests { )); } + /// However many paid updates land between the rounds, the record round 1 + /// bound is the one round 2 serves: the owner's activity is not the + /// holder's failure. + #[tokio::test] + async fn several_updates_between_the_rounds_do_not_fail_an_honest_holder() { + let responder = Responder::new(24, 24).await; + let nonce = mixed_nonce(responder.committed().tree()); + let openings = openings(&responder.proved_leaves(nonce).await); + let updated = first_pointer(&openings); + + let owner = (0..24u8) + .find(|owner| pointer(*owner, 1).address() == updated) + .expect("the opened pointer is one of ours"); + for counter in 2..=4 { + responder + .pointers + .put_bytes(&pointer(owner, counter).to_bytes()) + .await + .expect("update"); + } + + let items = responder.round2(nonce, &openings).await; + assert!( + matches!( + verify_slice_response(&openings, &nonce, &responder.peer_bytes, &items), + AuditVerdict::Pass { .. } + ), + "three updates between the rounds must not fail the holder, got {:?}", + verify_slice_response(&openings, &nonce, &responder.peer_bytes, &items) + ); + let served = items.iter().find_map(|item| match item { + SubtreeSliceItem::PointerRecord { key, records } if *key == updated => Some(records), + _ => None, + }); + assert_eq!( + served, + Some(&vec![pointer_record_bound_in_round_one(&responder, owner)]), + "exactly the record round 1 read, and nothing else" + ); + } + + /// The record a responder held for `owner`'s pointer before any update: + /// counter 1, as [`Responder::new`] stored it. + fn pointer_record_bound_in_round_one(responder: &Responder, owner: u8) -> Vec { + let address = pointer(owner, 1).address(); + responder + .pointers + .superseded_all(&address) + .last() + .map(|record| record.to_vec()) + .expect("the first record is kept") + } + #[tokio::test] async fn a_node_that_lost_a_committed_pointer_fails_round_one() { let responder = Responder::new(24, 24).await; @@ -3028,6 +3512,40 @@ mod pointer_audit_tests { } } + /// A node that lost the pointer is absent, even while it still keeps in + /// memory the record an update replaced, and even when that is the very + /// record round 1 read: the replaced records prove what round 1 read, not + /// that the node still holds the pointer. + #[tokio::test] + async fn a_lost_pointer_is_absent_whatever_replaced_records_remain() { + let responder = Responder::new(24, 24).await; + let nonce = mixed_nonce(responder.committed().tree()); + let openings = openings(&responder.proved_leaves(nonce).await); + let lost = first_pointer(&openings); + let owner = (0..24u8) + .find(|owner| pointer(*owner, 1).address() == lost) + .expect("the opened pointer is one of ours"); + responder + .pointers + .put_bytes(&pointer(owner, 2).to_bytes()) + .await + .expect("update"); + assert!(responder.pointers.delete(&lost).await.expect("delete")); + assert!( + !responder.pointers.superseded_all(&lost).is_empty(), + "the record round 1 read is still kept in memory" + ); + + let items = responder.round2(nonce, &openings).await; + assert!(items + .iter() + .any(|item| matches!(item, SubtreeSliceItem::Absent { key } if *key == lost))); + assert_eq!( + verify_slice_response(&openings, &nonce, &responder.peer_bytes, &items), + AuditVerdict::Fail(AuditFailureReason::KeyAbsent) + ); + } + #[tokio::test] async fn a_pointer_lost_after_round_one_is_admitted_absent() { let responder = Responder::new(24, 24).await; @@ -3080,14 +3598,16 @@ mod pointer_audit_tests { async fn a_relay_that_fetches_records_only_in_round_two_fails() { let responder = Responder::new(24, 24).await; let nonce = mixed_nonce(responder.committed().tree()); - let mut leaves = responder.proved_leaves(nonce).await; + let genuine = responder.proved_leaves(nonce).await; + let held = openings(&genuine); // What a relay can say in round 1 without the bytes. + let mut leaves = genuine; for leaf in leaves.iter_mut().filter(|leaf| is_pointer_leaf(leaf)) { leaf.nonced_root = [0u8; 32]; } let openings = openings(&leaves); // And in round 2 it serves the genuine records, fetched on demand. - let items = responder.round2(nonce, &openings).await; + let items = responder.round2(nonce, &held).await; assert_eq!( verify_slice_response(&openings, &nonce, &responder.peer_bytes, &items), AuditVerdict::Fail(AuditFailureReason::DigestMismatch) @@ -3121,8 +3641,8 @@ mod pointer_audit_tests { ); } - /// A pointer item may carry the record held now and the one it replaced, - /// never more. + /// A pointer item may carry at most two records. This build serves one; + /// two is what a responder that kept no round-1 roots served (ADR-0016). #[tokio::test] async fn a_pointer_item_with_more_than_two_records_is_malformed() { let responder = Responder::new(24, 24).await; diff --git a/tests/e2e/pointer_replication.rs b/tests/e2e/pointer_replication.rs index c031d719..8548a7c0 100644 --- a/tests/e2e/pointer_replication.rs +++ b/tests/e2e/pointer_replication.rs @@ -17,9 +17,17 @@ use ant_node::pointer::PointerStore; use ant_node::replication::audit::AuditTickResult; use ant_node::replication::commitment::pointer_leaf_hash; use ant_node::replication::commitment_state::{BuiltCommitment, ResponderCommitmentState}; +use ant_node::replication::config::SUBTREE_AUDIT_PROTOCOL_ID; use ant_node::replication::pointer::{PointerFreshWrite, PointerReplication}; +use ant_node::replication::protocol::{ + ReplicationMessage, ReplicationMessageBody, SubtreeAuditChallenge, SubtreeAuditResponse, + SubtreeSliceChallenge, SubtreeSliceItem, SubtreeSliceOpening, SubtreeSliceResponse, +}; +use ant_node::replication::slice::nonced_block_root; use ant_node::ReplicationConfig; -use ant_protocol::pointer::{Pointer, PointerState, PointerTarget, PointerTargetKind}; +use ant_protocol::pointer::{ + Pointer, PointerState, PointerTarget, PointerTargetKind, FINAL_COUNTER, +}; use bytes::Bytes; use saorsa_core::identity::PeerId; use saorsa_pqc::api::sig::{ml_dsa_65, MlDsaPublicKey, MlDsaSecretKey}; @@ -33,6 +41,9 @@ const SETTLE: Duration = Duration::from_secs(30); /// How often to look while waiting. const POLL: Duration = Duration::from_millis(200); +/// The id a hand-driven storage audit uses for both of its rounds. +const CHALLENGE_ID: u64 = 0x5EED; + /// A proof the receivers never parse: the state is pre-marked as paid in each /// verifier's cache, so verification answers from the cache. const DUMMY_PROOF: [u8; 64] = [0x01; 64]; @@ -51,6 +62,18 @@ fn signed(pk: &MlDsaPublicKey, sk: &MlDsaSecretKey, counter: u64, target: u8) -> .expect("sign") } +/// The final state that hands `pk`'s pointer over to the pointer at +/// `recipient` (ADR-0018). +fn transfer(pk: &MlDsaPublicKey, sk: &MlDsaSecretKey, recipient: u8) -> Pointer { + Pointer::sign( + sk, + pk, + FINAL_COUNTER, + PointerTarget::new(PointerTargetKind::Pointer, [recipient; 32]), + ) + .expect("sign") +} + fn store(node: &TestNode) -> PointerStore { node.ant_protocol .as_ref() @@ -452,6 +475,177 @@ async fn the_possession_check_penalises_only_a_member_that_dropped_the_record() harness.teardown().await.expect("teardown"); } +/// A transfer written to one node reaches the whole group, and after that no +/// node takes a second final state — not even one whose target sorts first, +/// which the previous merge rule let displace the first everywhere (ADR-0018). +#[tokio::test] +#[serial] +async fn a_transfer_reaches_the_group_and_no_node_takes_a_second_one() { + let harness = TestHarness::setup_minimal().await.expect("setup"); + harness.warmup_dht().await.expect("warmup"); + + let (pk, sk) = owner(); + let created = signed(&pk, &sk, 0, 1); + let handed_over = transfer(&pk, &sk, 0x77); + let take_back = transfer(&pk, &sk, 0x01); + assert!(take_back.target().to_bytes() < handed_over.target().to_bytes()); + for record in [&created, &handed_over, &take_back] { + mark_paid(&harness, record); + } + + put(harness.test_node(1).expect("node"), &created).await; + for i in 0..harness.node_count() { + assert!( + wait_for(&harness, i, &created).await, + "node {i} lacks the create" + ); + } + match put(harness.test_node(2).expect("node"), &handed_over).await { + PointerPutResponse::Success { state_id, .. } => { + assert_eq!(state_id, handed_over.state_id()); + } + other => panic!("the transfer was refused: {other:?}"), + } + for i in 0..harness.node_count() { + assert!( + wait_for(&harness, i, &handed_over).await, + "node {i} never received the transfer" + ); + } + + // The former owner, paid up, tries to take it back on every node. + for i in 0..harness.node_count() { + match put(harness.test_node(i).expect("node"), &take_back).await { + PointerPutResponse::Stale { state_id, .. } => assert_eq!( + state_id, + handed_over.state_id(), + "node {i} must name the transfer it holds" + ), + other => panic!("node {i} took a second final state: {other:?}"), + } + assert!(holds(harness.test_node(i).expect("node"), &handed_over)); + } + + harness.teardown().await.expect("teardown"); +} + +/// A node that never heard of a transfer — one that joined after it, or lost +/// its copy — would take any final state on the merge rule alone. Before it +/// does, it asks its group, and a peer that serves the transfer it holds is +/// proof enough to refuse the second one. +#[tokio::test] +#[serial] +async fn a_node_that_missed_the_transfer_refuses_another_its_group_proves() { + let harness = TestHarness::setup_minimal().await.expect("setup"); + harness.warmup_dht().await.expect("warmup"); + + let (pk, sk) = owner(); + let created = signed(&pk, &sk, 0, 1); + let handed_over = transfer(&pk, &sk, 0x77); + let second = transfer(&pk, &sk, 0x01); + mark_paid(&harness, &second); + mark_paid(&harness, &handed_over); + + // Everyone holds the create, and learns everyone else understands + // pointers while there is nothing newer to hint. + let everyone: Vec = (0..harness.node_count()).collect(); + for i in &everyone { + store(harness.test_node(*i).expect("node")) + .put_bytes(&created.to_bytes()) + .await + .expect("put"); + } + exchange_hints(&harness, &everyone, &everyone).await; + + // The transfer lands everywhere but one node, written straight into the + // stores so nothing replicates or hints it there. + let unaware = 4; + for i in others(&harness, &[unaware]) { + store(harness.test_node(i).expect("node")) + .put_bytes(&handed_over.to_bytes()) + .await + .expect("put"); + } + let node = harness.test_node(unaware).expect("node"); + assert!( + holds(node, &created), + "the unaware node has only the create" + ); + + match put(node, &second).await { + PointerPutResponse::Stale { state_id, .. } => assert_eq!( + state_id, + handed_over.state_id(), + "the refusal names the transfer the group proved" + ), + other => panic!("the unaware node took a second final state: {other:?}"), + } + assert!(holds(node, &created), "nothing was written"); + + // The transfer the group holds is not refused: it is no conflict. + match put(node, &handed_over).await { + PointerPutResponse::Success { state_id, .. } => { + assert_eq!(state_id, handed_over.state_id()); + } + other => panic!("the group's own transfer was refused: {other:?}"), + } + + harness.teardown().await.expect("teardown"); +} + +/// Two different final states are a fork only the owner can make, by racing +/// them. A member holding the other side took what reached it first, as the +/// merge rule says; the possession check does not penalise it for the owner's +/// fork, while it still penalises a member that holds nothing. +#[tokio::test] +#[serial] +async fn the_possession_check_does_not_penalise_the_other_side_of_a_fork() { + let harness = TestHarness::setup_minimal().await.expect("setup"); + harness.warmup_dht().await.expect("warmup"); + + let (pk, sk) = owner(); + let one_side = transfer(&pk, &sk, 0x77); + let other_side = transfer(&pk, &sk, 0x01); + let checker = 3; + let forked = 1; + let dropper = 2; + for i in others(&harness, &[forked, dropper]) { + store(harness.test_node(i).expect("node")) + .put_bytes(&one_side.to_bytes()) + .await + .expect("put"); + } + store(harness.test_node(forked).expect("node")) + .put_bytes(&other_side.to_bytes()) + .await + .expect("put"); + + let everyone: Vec = (0..harness.node_count()).collect(); + exchange_hints(&harness, &everyone, &[checker]).await; + + let checker_node = harness.test_node(checker).expect("node"); + let checker_p2p = checker_node.p2p_node.as_ref().expect("p2p"); + let forked_peer = peer(harness.test_node(forked).expect("node")); + let dropper_peer = peer(harness.test_node(dropper).expect("node")); + let forked_before = checker_p2p.peer_trust(&forked_peer); + let dropper_before = checker_p2p.peer_trust(&dropper_peer); + + replication(checker_node) + .check_possession(one_side.state(), &[forked_peer, dropper_peer]) + .await; + + assert!( + checker_p2p.peer_trust(&forked_peer) >= forked_before, + "the member holding the other final state was penalised" + ); + assert!( + checker_p2p.peer_trust(&dropper_peer) < dropper_before, + "the member holding nothing was not penalised" + ); + + harness.teardown().await.expect("teardown"); +} + /// A network whose close group is two nodes, so that in five nodes some node /// is always outside the retention width (two plus the margin of two) for any /// address, with no pruning hysteresis. @@ -632,13 +826,23 @@ async fn commit_pointers( auditor: usize, count: usize, ) -> Vec { - let holder_node = harness.test_node(holder).expect("holder"); let records: Vec = (0..count) .map(|_| { let (pk, sk) = owner(); signed(&pk, &sk, 1, 1) }) .collect(); + commit_records(harness, holder, auditor, records).await +} + +/// [`commit_pointers`] over records the caller signed. +async fn commit_records( + harness: &TestHarness, + holder: usize, + auditor: usize, + records: Vec, +) -> Vec { + let holder_node = harness.test_node(holder).expect("holder"); for record in &records { store(holder_node) .put_bytes(&record.to_bytes()) @@ -704,6 +908,144 @@ async fn a_node_holding_its_committed_pointers_passes_the_storage_audit() { harness.teardown().await.expect("teardown"); } +/// Several paid updates between the two rounds of a storage audit do not fail +/// the node holding the pointer: round 2 serves the record round 1 read, found +/// by the root round 1 reported over it (ADR-0019). Driven one round at a time +/// against the holder's live engine, so it is the round-1 session that carries +/// what round 1 bound across to round 2. +#[tokio::test] +#[serial] +async fn round_two_serves_the_record_round_one_read_across_several_updates() { + let harness = TestHarness::setup_small().await.expect("setup"); + harness.warmup_dht().await.expect("warmup"); + let (holder, auditor) = (7, 8); + let owners: Vec<(MlDsaPublicKey, MlDsaSecretKey)> = (0..24).map(|_| owner()).collect(); + let records = owners.iter().map(|(pk, sk)| signed(pk, sk, 1, 1)).collect(); + commit_records(&harness, holder, auditor, records).await; + + let holder_node = harness.test_node(holder).expect("holder"); + let holder_peer = peer(holder_node); + let committed = commitments(holder_node) + .current() + .expect("a current commitment"); + let pointer_keys = committed.pointer_leaf_keys(); + let auditor_p2p = harness + .test_node(auditor) + .expect("auditor") + .p2p_node + .as_ref() + .expect("p2p") + .clone(); + let ask = |body: ReplicationMessageBody| { + let auditor_p2p = Arc::clone(&auditor_p2p); + async move { + let request = ReplicationMessage { + request_id: CHALLENGE_ID, + body, + } + .encode() + .expect("encode"); + let response = auditor_p2p + .send_request( + &holder_peer, + SUBTREE_AUDIT_PROTOCOL_ID, + request, + Duration::from_secs(60), + ) + .await + .expect("a response"); + ReplicationMessage::decode_subtree_audit_response(&response.data) + .expect("decode") + .body + } + }; + + // Round 1: the holder binds the record it holds for each pointer. + let nonce = [0x5A; 32]; + let round1 = ask(ReplicationMessageBody::SubtreeAuditChallenge( + SubtreeAuditChallenge { + challenge_id: CHALLENGE_ID, + nonce, + challenged_peer_id: *holder_peer.as_bytes(), + expected_commitment_hash: committed.hash(), + }, + )) + .await; + let ReplicationMessageBody::SubtreeAuditResponse(SubtreeAuditResponse::Proof { proof, .. }) = + round1 + else { + panic!("expected a round-1 proof, got {round1:?}"); + }; + let opened: Vec<_> = proof + .leaves + .iter() + .filter(|leaf| pointer_keys.contains(&leaf.key)) + .take(5) + .cloned() + .collect(); + assert!(!opened.is_empty(), "round 1 proved a pointer"); + + // Between the rounds, the owner of every pointer about to be opened + // updates it three times, each a state the holder accepts. + let holder_store = store(holder_node); + for leaf in &opened { + let (pk, sk) = owners + .iter() + .find(|(pk, sk)| signed(pk, sk, 1, 1).address() == leaf.key) + .expect("an owner for every committed pointer"); + for counter in 2..=4 { + holder_store + .put_bytes(&signed(pk, sk, counter, 2).to_bytes()) + .await + .expect("update"); + } + } + + // Round 2: each opened pointer is proved by the record round 1 read. + let round2 = ask(ReplicationMessageBody::SubtreeSliceChallenge( + SubtreeSliceChallenge { + challenge_id: CHALLENGE_ID, + nonce, + challenged_peer_id: *holder_peer.as_bytes(), + expected_commitment_hash: committed.hash(), + openings: opened + .iter() + .map(|leaf| SubtreeSliceOpening { + key: leaf.key, + block_index: 0, + }) + .collect(), + }, + )) + .await; + let ReplicationMessageBody::SubtreeSliceResponse(SubtreeSliceResponse::Items { items, .. }) = + round2 + else { + panic!("expected round-2 items, got {round2:?}"); + }; + for leaf in &opened { + let served = items + .iter() + .find_map(|item| match item { + SubtreeSliceItem::PointerRecord { key, records } if *key == leaf.key => { + Some(records.as_slice()) + } + _ => None, + }) + .expect("a record for every opened pointer"); + assert!( + served.iter().any(|record| { + nonced_block_root(&nonce, holder_peer.as_bytes(), &leaf.key, record) + == leaf.nonced_root + && Pointer::from_bytes(record).is_ok_and(|p| p.address() == leaf.key) + }), + "round 2 must serve the record round 1 bound, after three updates" + ); + } + + harness.teardown().await.expect("teardown"); +} + /// A node that dropped the pointers it committed to fails the storage audit, /// exactly as a node that dropped its chunks does. #[tokio::test] diff --git a/tests/e2e/testnet.rs b/tests/e2e/testnet.rs index 6e878709..a20c225f 100644 --- a/tests/e2e/testnet.rs +++ b/tests/e2e/testnet.rs @@ -1371,11 +1371,11 @@ impl TestNetwork { .await { Ok(mut engine) => { - // Pointers replicate through the same engine (ADR-0016). + // Pointers replicate through the same engine (ADR-0016), + // which the service also asks before a final state + // (ADR-0018). if let Some(service) = protocol.pointer_service() { - let (writes, fresh_writes) = tokio::sync::mpsc::unbounded_channel(); - service.attach_fresh_writes(writes); - engine.with_pointers(service.store().clone(), fresh_writes); + engine.with_pointer_service(service); } let dht_events = p2p.dht_manager().subscribe_events(); engine.start(dht_events); diff --git a/tests/pointer_convergence.rs b/tests/pointer_convergence.rs index 4e1864aa..6f011c53 100644 --- a/tests/pointer_convergence.rs +++ b/tests/pointer_convergence.rs @@ -6,11 +6,18 @@ //! These are the property tests behind that claim, plus the two anti-abuse //! properties the merge rule exists to provide: one payment funds one state, //! and no re-signature of a stored state can displace it. +//! +//! ADR-0018 carves out one exception: a final state (counter `u64::MAX`) is +//! replaced by nothing, so two *different* final states are unordered and a +//! node keeps the first one it took. The claim holds for every set with at +//! most one final state in it; with two, the first final state delivered wins, +//! and nothing delivered after it moves a node off it. #![allow(clippy::unwrap_used, clippy::expect_used, clippy::panic)] use std::collections::BTreeSet; +use ant_node::pointer::store::{PointerStore, PutOutcome}; use ant_protocol::pointer::{Pointer, PointerTarget, PointerTargetKind}; use proptest::prelude::*; use saorsa_pqc::api::sig::{ @@ -438,8 +445,6 @@ fn the_wire_format_is_what_the_adr_says() { /// layer does, which is where the cost would actually have been paid. #[tokio::test] async fn sixty_four_signatures_buy_exactly_one_write() { - use ant_node::pointer::store::{PointerStore, PutOutcome}; - let dir = tempfile::tempdir().expect("tempdir"); let store = PointerStore::new(dir.path()).await.expect("open store"); @@ -559,11 +564,10 @@ fn an_unknown_version_is_refused_rather_than_accepted_at_its_own_price() { } } -/// At `u64::MAX` no *counter* can out-rank the winner, but a smaller *target* -/// still can. That asymmetry is why migration has to happen before the terminal -/// update rather than on it. +/// At `u64::MAX` nothing out-ranks the held state: no counter is larger, and +/// an equal counter no longer resolves by target (ADR-0018). #[test] -fn a_terminal_counter_cannot_be_out_counted_only_out_targeted() { +fn a_final_state_is_out_ranked_by_nothing() { let terminal = signed(13, u64::MAX, 5, PointerTargetKind::Chunk); assert!(terminal.is_terminal()); assert!(terminal.next_counter().is_err(), "no successor exists"); @@ -575,91 +579,137 @@ fn a_terminal_counter_cannot_be_out_counted_only_out_targeted() { assert!(terminal.replaces(&earlier)); } - // The order does not degenerate there: equal-counter conflicts at the - // maximum still resolve deterministically, so replicas cannot split. + // Nor does another final state, whichever way its target sorts: the first + // one a node took is the one it keeps. let low = signed(13, u64::MAX, 1, PointerTargetKind::Chunk); let high = signed(13, u64::MAX, 2, PointerTargetKind::Chunk); - assert!(low.replaces(&high)); + assert!(!low.replaces(&high)); assert!(!high.replaces(&low)); - assert_eq!(winner(&[&high, &low]).state_id(), low.state_id()); + assert_eq!(winner(&[&high, &low]).state_id(), high.state_id()); assert_eq!(winner(&[&low, &high]).state_id(), low.state_id()); } -/// Why migration must happen *before* the terminal update. -/// -/// At `u64::MAX` the counter can no longer advance, but the pointer is not -/// frozen: the merge order still resolves equal counters by target bytes, and -/// *smaller* target bytes win. So a migration written at the terminal counter -/// can still be displaced — by the owner, or by anyone replaying an older -/// signed record of theirs with a smaller target. The only safe migration is -/// one made while a successor counter still exists, because a strictly larger -/// counter is the one move nothing can answer. +/// A transfer is final on a node's disk: once stored, the former owner cannot +/// take the address back — not with a later counter, which does not exist, and +/// not with a final state whose target sorts first, which the previous rule +/// would have let displace it. #[tokio::test] -async fn migration_must_happen_before_the_terminal_update() { - use ant_node::pointer::store::{PointerStore, PutOutcome}; - use ant_protocol::pointer::PointerTarget; - use saorsa_pqc::api::sig::ml_dsa_65; - +async fn a_stored_transfer_cannot_be_taken_back() { let dir = tempfile::tempdir().expect("tempdir"); let store = PointerStore::new(dir.path()).await.expect("store"); let (pk, sk) = ml_dsa_65().generate_keypair_from_seed(&[14u8; 32]); - let sign_at = |counter: u64, target: PointerTarget| { - Pointer::sign(&sk, &pk, counter, target).expect("sign") - }; - - // One update short of the end: a successor counter still exists. - let penultimate = sign_at( - u64::MAX - 1, + let current = Pointer::create( + &sk, + &pk, PointerTarget::new(PointerTargetKind::Chunk, [0x10u8; 32]), - ); - assert!(!penultimate.is_terminal()); + ) + .expect("create"); assert_eq!( - store.put_bytes(&penultimate.to_bytes()).await.expect("put"), + store.put_bytes(¤t.to_bytes()).await.expect("put"), PutOutcome::Changed ); - // The safe migration: spend the last counter. A strictly larger counter - // beats every target, so nothing at u64::MAX - 1 can answer it. - let migration = sign_at( - penultimate.next_counter().expect("successor exists"), - PointerTarget::new(PointerTargetKind::Pointer, [0x80u8; 32]), - ); + // The handover: one final state, pointing at the recipient's pointer. + let recipient = [0x80u8; 32]; + let transfer = current.transfer_to(&sk, recipient).expect("transfer"); + assert_eq!(transfer.transferred_to(), Some(recipient)); assert_eq!( - store.put_bytes(&migration.to_bytes()).await.expect("put"), + store.put_bytes(&transfer.to_bytes()).await.expect("put"), PutOutcome::Changed ); - assert!(migration.is_terminal()); - assert!(migration.next_counter().is_err()); - // Now the danger. The counter is spent, so the only remaining moves are to - // strictly smaller target bytes — and they still win. - let smaller_target = sign_at( + // A former owner grinding a target that sorts first. Under the previous + // rule this displaced the transfer; now it is stale. + let take_back = Pointer::sign( + &sk, + &pk, u64::MAX, - PointerTarget::new(PointerTargetKind::Chunk, [0x01u8; 32]), - ); - assert!( - smaller_target.to_bytes() != migration.to_bytes(), - "a genuinely different state" - ); + PointerTarget::new(PointerTargetKind::Chunk, [0x00u8; 32]), + ) + .expect("sign"); + assert!(take_back.target().to_bytes() < transfer.target().to_bytes()); assert_eq!( - store - .put_bytes(&smaller_target.to_bytes()) - .await - .expect("put"), - PutOutcome::Changed, - "a terminal pointer is NOT frozen: a smaller target still displaces it" + store.put_bytes(&take_back.to_bytes()).await.expect("put"), + PutOutcome::Stale, + "a second final state must not displace the first" ); - // Larger target bytes cannot claw it back: the move is one-way. + // Nor any lower counter, and the held state is the transfer throughout. + for counter in [0u64, 1, u64::MAX - 1] { + let older = Pointer::sign( + &sk, + &pk, + counter, + PointerTarget::new(PointerTargetKind::Chunk, [0x00u8; 32]), + ) + .expect("sign"); + assert_eq!( + store.put_bytes(&older.to_bytes()).await.expect("put"), + PutOutcome::Stale + ); + } assert_eq!( - store.put_bytes(&migration.to_bytes()).await.expect("put"), - PutOutcome::Stale, - "and the displaced migration can never be restored" + store.state_id(&transfer.address()), + Some(transfer.state_id()) ); +} + +proptest! { + #![proptest_config(ProptestConfig::with_cases(16))] + + /// With one final state among the records, every delivery order still + /// ends on it: rule 0 only leaves two *final* states unordered. + #[test] + fn one_final_state_wins_in_every_delivery_order( + counters in prop::collection::vec(0u64..4, 1..5), + targets in prop::collection::vec(0u8..4, 1..5), + final_target in 0u8..4, + ) { + let mut records: Vec = counters + .iter() + .zip(targets.iter()) + .map(|(counter, target)| signed(1, *counter, *target, PointerTargetKind::Chunk)) + .collect(); + let finalized = signed(1, u64::MAX, final_target, PointerTargetKind::Pointer); + records.push(finalized.clone()); + + let mut order: Vec<&Pointer> = records.iter().collect(); + let mut permutations = 0usize; + permute(&mut order, 0, &mut |candidate| { + prop_assert_eq!(winner(candidate).state_id(), finalized.state_id()); + Ok(()) + }, &mut permutations)?; + prop_assert_eq!(permutations, factorial(records.len())); + } - // Which is the whole point: a migration made at the terminal counter is - // not final, so it has to be made earlier, where the counter still answers. - assert!(smaller_target.replaces(&migration)); - assert!(!migration.replaces(&smaller_target)); + /// With two different final states, the one delivered first is what a + /// node keeps, whatever else arrives before, between or after them. + #[test] + fn the_first_final_state_delivered_is_kept( + counters in prop::collection::vec(0u64..4, 1..4), + targets in prop::collection::vec(0u8..4, 1..4), + first_target in 0u8..8, + second_target in 8u8..16, + ) { + let mut records: Vec = counters + .iter() + .zip(targets.iter()) + .map(|(counter, target)| signed(1, *counter, *target, PointerTargetKind::Chunk)) + .collect(); + records.push(signed(1, u64::MAX, first_target, PointerTargetKind::Pointer)); + records.push(signed(1, u64::MAX, second_target, PointerTargetKind::Pointer)); + + let mut order: Vec<&Pointer> = records.iter().collect(); + let mut permutations = 0usize; + permute(&mut order, 0, &mut |candidate| { + let first_final = candidate + .iter() + .find(|record| record.is_terminal()) + .expect("two finals were delivered"); + prop_assert_eq!(winner(candidate).state_id(), first_final.state_id()); + Ok(()) + }, &mut permutations)?; + prop_assert_eq!(permutations, factorial(records.len())); + } }