diff --git a/Cargo.lock b/Cargo.lock index 9f2b279a..bf0b2e1a 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -937,8 +937,8 @@ dependencies = [ [[package]] name = "ciris-crypto" -version = "17.1.0" -source = "git+https://github.com/CIRISAI/CIRISVerify?tag=v17.1.0#cdc6498af2605a402bbff1e399c2c0795b968180" +version = "18.0.0" +source = "git+https://github.com/CIRISAI/CIRISVerify?tag=v18.0.0#33c3695adae8b1216aaa6debcc0bf9d5bf68872c" dependencies = [ "chacha20poly1305", "ed25519-dalek", @@ -964,8 +964,8 @@ dependencies = [ [[package]] name = "ciris-edge" -version = "32.1.0" -source = "git+https://github.com/CIRISAI/CIRISEdge?tag=v32.1.0#d34df2f6e47263a39defea2c3b2a18dacbe27b3f" +version = "38.0.0" +source = "git+https://github.com/CIRISAI/CIRISEdge?tag=v38.0.0#ffee0c9272cf1c487f5d115afb0abff411649656" dependencies = [ "async-trait", "axum", @@ -985,6 +985,7 @@ dependencies = [ "leviculum-core", "leviculum-lxmf", "leviculum-std", + "memchr", "openmls", "openmls_basic_credential", "openmls_libcrux_crypto", @@ -1019,8 +1020,8 @@ dependencies = [ [[package]] name = "ciris-keyring" -version = "17.1.0" -source = "git+https://github.com/CIRISAI/CIRISVerify?tag=v17.1.0#cdc6498af2605a402bbff1e399c2c0795b968180" +version = "18.0.0" +source = "git+https://github.com/CIRISAI/CIRISVerify?tag=v18.0.0#33c3695adae8b1216aaa6debcc0bf9d5bf68872c" dependencies = [ "aes-gcm", "async-trait", @@ -1086,8 +1087,8 @@ dependencies = [ [[package]] name = "ciris-persist" -version = "49.0.0" -source = "git+https://github.com/CIRISAI/CIRISPersist?tag=v49.0.0#a5965d42e79ff35c79509beaaebd417514b7914d" +version = "52.0.0" +source = "git+https://github.com/CIRISAI/CIRISPersist?tag=v52.0.0#5ce36224bbec38ecd5c2e0e0bce64a74384c9a7a" dependencies = [ "anyhow", "async-trait", @@ -1118,6 +1119,7 @@ dependencies = [ "regex", "rusqlite", "serde", + "serde_jcs", "serde_json", "sha2 0.10.9", "thiserror 2.0.20", @@ -1181,8 +1183,8 @@ dependencies = [ [[package]] name = "ciris-verify-core" -version = "17.1.0" -source = "git+https://github.com/CIRISAI/CIRISVerify?tag=v17.1.0#cdc6498af2605a402bbff1e399c2c0795b968180" +version = "18.0.0" +source = "git+https://github.com/CIRISAI/CIRISVerify?tag=v18.0.0#33c3695adae8b1216aaa6debcc0bf9d5bf68872c" dependencies = [ "android_system_properties", "async-trait", @@ -1219,12 +1221,13 @@ dependencies = [ "webpki-roots 0.26.11", "windows-sys 0.52.0", "x509-parser", + "zeroize", ] [[package]] name = "ciris-verify-ffi" -version = "17.1.0" -source = "git+https://github.com/CIRISAI/CIRISVerify?tag=v17.1.0#cdc6498af2605a402bbff1e399c2c0795b968180" +version = "18.0.0" +source = "git+https://github.com/CIRISAI/CIRISVerify?tag=v18.0.0#33c3695adae8b1216aaa6debcc0bf9d5bf68872c" dependencies = [ "aes-gcm", "android_logger", diff --git a/Cargo.toml b/Cargo.toml index f485d38e..f4d952fc 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "ciris-server" -version = "0.5.218" # 0.5.218 = ADD A CONTACT BY CODE, AND A SECOND DEVICE THAT CATCHES UP (the maintainer's two flows). CONTACTS: `GET /v1/self/contact-code?nodes=` returns the person's v3 code (string + QR payload) naming the ANNOUNCED devices they choose, or none (resolved through the public directory) (#673); the device roster is public only for devices the person announced, per device (#655 ruling); `DELETE /v1/contacts/{key_id}` and `POST /v1/federation/peering/revoke` withdraw consent, signed by the person (#657); `GET /v1/admin/self` returns `owner_delegations` (#676). SECOND DEVICE: claim-remote links both devices at once, the device holding the pen re-wraps old self files for the new one, and `POST /v1/self/nodes/{node}/announce` announces another of my devices from the first (#678); replication runs 12 of 17 kinds, every kind routed or excluded by name (#646), so member additions, households and device revocations leave the node. FIXES: OAuth redirect parsed, same-origin path or exact loopback only (#672); a TPM request without a TPM refuses (#639); boot errors carry their whole cause chain (#586); a claimed node finds its consent peers (#601); reproposed chat commits are carried and the chat unopened reason is a token (#602). LADDERS: selffiles in CI (#622, #626), a `devices` scenario (second device joins; old self file opens; family rungs RED-EXPECTED on CIRISPersist#910/#916), ui-chat drives client 0.5.224 with nav derived from the client. Client floor 0.5.224. Known: history from BEFORE a device joined opens only for self files until CIRISPersist#916; restart re-admission waits on CIRISEdge#676 / CIRISPersist#911. PREVIOUS: 0.5.217 = 0.5.217 = A FILE IS WHAT IT SAYS IT IS (CIRISServer#642/#641/#643/#644, for CIRISClient#77 and FSD/MEDIA_EDGE.md): the drive's write gate — every write door (upload, replace, move, notes) requires an RFC 6838 media type whose leading bytes agree with it (`drive.bad_media_type` 400, `drive.format_mismatch` 415 naming declared and sniffed — the same table as the client's RenderTier), and a filename cleaned per RFC 6266 §4.3 (path, control, bidi and zero-width characters removed; `drive.bad_filename` when nothing is left), so a name like `invoice\u202Etxt.exe` no longer replicates as written. The node states the PLAINTEXT digest of what it hands over (CC 5.3.2.5): `content_digest` on `GET /v1/files/{id}` and `/meta`, RFC 9530 `Repr-Digest` on a whole raw read; the edge pointer hash is renamed `at_rest_sha256` (it is the sealed blob's, never the plaintext's; `content_sha256` kept as a deprecated alias). `GET /v1/media/policy` publishes the node's render policy ahead of the ingest pipeline, with `renditions: false`. Notes say `here` like the drive (one word per fact; `drive::BYTE_STATES`). A digest SIGNED into the row, and renditions/placeholders, wait on CIRISEdge#638 and #614. PREVIOUS: 0.5.216 = 0.5.216 = ROSTERS AND FILES, END TO END (FSD/ROSTER_AND_DRIVE_CRUD.md): households (`/v1/families`: create, list, read, add/remove member, role, leave, dissolve, and an envelope → cosign → assemble flow for quorum families — closes CIRISServer#627), N-member communities and affiliations (`/v1/communities`, growth on persist v48's widening plane — CIRISServer#594), the owner's own devices (release a node, relabel a device, revoked devices listed; the self room uses edge's room-keyed handshake so a second device OPENS a file's bytes), and full drive CRUD (replace, rename, withdraw with a 410, move between circles, metadata, raw bytes with Range, a paged drive, notes edit/delete; the split-install viewer key fixed). Every roster read goes through the fold, enforced by a source gate. Known gaps, named in the FSD: a widened community member is refused reads until CIRISPersist#907; roster rows a PEER authors are signature-checked only until CIRISPersist#908; file rows are node-signed until CIRISEdge#675; family changes after creation do not replicate to a peer that already holds the family (CIRISPersist#910); subject take-back deferred. ALSO: ADOPTS EDGE v31.0.0 / PERSIST v48.0.0 (verify v16.1.0 unchanged; CIRISPersist#860/#905, CIRISEdge#672) — THE PROMOTION SWEEP SEES A CLAIMED MACHINE'S CONSENT: persist's `load_active_egress_grants` and the crossing's covering check take a grant by PRINCIPALS (the machine's own, or its bound steward's naming it in `for_key_id`), and V152 keys `consent_peer_set` by `for_key_id` — so an owned agent's sealed traces are promoted by its human's consent, the gate the production-shaped ladder had read as `offerable=0` since the owner's pen went live in 0.5.211. Also #860: a room's roster converges both ways (`CommunityMembershipWidening`, the 17th EnvelopeKind, appended — REPLICATION_POLICY_HASH / CONSENT_GRAMMAR_HASH / edge SERVE_ADVERTISE_POLICY_HASH re-pinned; ABI 5 unchanged); #672: the test-anchor block is persist's to mint (`mint_test_anchor` reproduces our block byte-for-byte; a `_MINTED_BY` line names the pair). EARLIER IN THIS CUT: ALSO ADOPTS EDGE v30.3.1 (CIRISEdge#671 → #673, persist v47.3.0 / verify v16.1.0 unchanged): a peer in no send set is minted a FirstContact recipient confined to exactly this node's allegiance facts (owner-binding, root acceptance, self-charter at federation audience), so a claimed agent Roots a canonical that consents to nobody on first contact — the server's `GET /v1/federation/allegiance` carry stays as the belt. THE ONE REMAINING GATE IS PERSIST'S (CIRISPersist#905): `promote_consented_backlog` loads grants the ENGINE key AUTHORED, and on every claimed install the human authors the grant (FOR the actor, since this cut — the covering door consents once per own key, node written last, and the actor is anchored at claim), so the sealed traces stay at `(self, local)`: the production-shaped ladder on the #636 merge read seal=3 trace_att=3 offerable=0. TAG HELD until persist folds the sweep over principals and edge pins it. PREVIOUSLY IN THIS CUT: THE HEAL LOOKED UP A KEY THAT EXISTS NOWHERE, AND THE TRACE PLANE RELIGHTS ON EDGE v30.3.0 — adopts edge v30.3.0 / persist v47.3.0 (verify v16.1.0 unchanged; CIRISEdge#659/#661/#662 → edge v30.2.0, CIRISPersist#901 → v47.3.0; CIRISServer#632): attribution is not trust (Attributed state; Rooted = a mutual VALID root through the owner-bindings), no baked production dial under a test root, a full responder inbox names its phase and never ends the driver, and `trust_root_valid` gains the holder-hardware leg (a valid root is as attested as its holders — the production seed's three accord holders carry real custody evidence; test roots carry persist's honest `SoftwareOnly_TEST` marker under a live test anchor). Server-side moves: `CallerScope::admits` takes the ROW's room as `cohort_target` (#893/#897), `files::in_room` is persist's gated reader door (caller named, `DrivePage`), `WithholdReason::RecipientNotRooted` classified as a verdict. THE OWNER ACCEPTS THE ROOT (CIRISServer#632 step 2 — the half the adoption alone does not deliver): edge's walk is `∃R ∈ roots_of(owner_of(N)) ∩ roots_of(owner_of(P))`, keyed on the OWNER, and the server had only ever written the NODE's `delegates_to(node → R)` (`accept_trust_root`, bootstrap default trust) — so under edge v30 no owned node was Rooted by any peer. `mesh_genesis::accept_trust_roots_as_owner` writes `delegates_to(owner → R, infra:attest, infra:serve)` at federation with the owner's pen (same envelope shape as the node's, idempotent, never the owner's own key), driven wherever the pen exists: at boot beside the baked-root install (the already-claimed fleet; the production canonical on its one boot with the operator's pen, which now heals the key record AND accepts the root), at both claim arms, and on trust-root import. Under a live TEST anchor the node's own key record carries persist's honest `SoftwareOnly_TEST` marker (a harness node that charters a root is a holder, and v47.3.0 reads an evidence-less holder as an invalid root; never in a production build). Witness: `tests/an_owner_accepts_the_roots_the_node_accepted.rs` — unowned → nothing; claimed → the owner's edge at federation, `trusted_roots_of(owner)` lists it, idempotent, and `trust_root_valid(owner, baked root).valid` by name. Fixture consequences of the persist major: the shared revocation fixture's self-charter signer now carries Layer-A evidence (persist's own remedy), and `trace_round_e2e`'s canonical ACCEPTS the root too — under edge v30 a pair is Rooted only when BOTH accept (it read "AGENT OFFERS 0 ref(s)" with `edge_exists: false` on the canonical). THE HEAL (CIRISServer#606): THE HEAL LOOKED UP A KEY THAT EXISTS NOWHERE (CIRISServer#606, found 2026-09-24 on the production canonical): `auth::ownership::rebind_owner_key_record` located the owner's row by `owner_signer.derived_key_id()`, but production's owner pen is built by `hardware_user_local_signer`, which hands `from_hardware_parts` the DERIVED id — so on that pen `key_id()` IS the registered id and `derived_key_id()` is `--`, a key registered nowhere (`identity.rs`, CIRISServer#597 §4: "every existing caller reads key_id()"; this door was the one that did not). The canonical is OWNED by the operator's portable fedID and runs the heal on every boot; with the pen placed on disk for one boot the pen resolved (three `responsible-user signer resolved` lines), the heal looked up the doubly-derived id, returned `Absent`, and the `Ok(_) => {}` arm at all three call sites (boot, claim, delivery) swallowed it — no line, row unchanged, and the row stays refused by every verify v15.2 peer. The gate `tests/owner_key_record_rebind.rs` was green throughout because its fixture pen is named by ALIAS, the one shape production never builds. FIX: the CALLER names the row (`owner_of`'s steward at boot/delivery, the binding's `responsible_user_key_id` at claim), the pen must be that key by `key_id()` or `derived_key_id()` and hold the row's pubkeys, and every outcome — `Rebound` / `Bound` / `Absent` / `Unbound{refusal}` / no pen — is logged by name at every site. Gate: a production-shaped pen (named by its derived id) must rebind, a stranger's pen is refused by name, and a source gate pins the key_id() convention. The one-off heal of the canonical's row is the same runbook re-run on this cut (CIRISServer#632 thread). PREVIOUS: 0.5.215 = EVERY IDENTITY THIS NODE IS, AND A FILE AT A CHOSEN COHORT. THE PUBLISH-OWN SET NOW CARRIES EVERY KEY THIS NODE IS (CIRISServer#629): on an actor/node split install (CC 3.4.7.3 Clause A) compose mints a NODE key, MOVES the owner-binding onto it and makes it the wire identity, and the node then attests its identity occurrences AND its `consent:replication:v1` grant with it — while the publish-own set held only `edge.signer_key_id()` (the actor key). Edge publishes a self-plane row only when its attester is in that set, so the node's OWN key record was never offered in a Key round: the far side admitted the actor key, never the node key, and then refused every row the node attested with "attesting_key_id … is not a registered federation key". Measured on the production canonical the same day — five agent keys admitted in one afternoon, ZERO node keys, ZERO rows of any dimension, no trace since 2026-09-18, and the grant authored on the agent and absent on the receiver. The second half of the same bug: `refresh_publish_own_set` asked `owner_of(edge.signer_key_id())`, but the binding had MOVED to the node key, so the owner never entered the set either (`owner_claimed=false` in the agent log). `own_key_ids(edge)` already answered "which keys are us" and two other call sites were gated on consulting it; this one was not. Both halves fixed and the existing gate extended to the publish-own seed — the gate itself had to be fixed twice (a fixed 4 KiB window stopped short of the code, then it matched the doc comment instead of the call), so it now reads whole function bodies with comments stripped. Same actor/node axis as the drive-auth finding below. ALSO: 0.5.215 = A FILE AT A CHOSEN COHORT, AND THE ROW REACHES MY OTHER DEVICE — adopts edge v29.5.0 (persist v46.3.1 / verify v16.1.0 unchanged; CIRISServer#626/#622, CIRISEdge#646): one door writes a file at any cohort and one drive lists everything this person can reach. `POST /v1/files {cohort: self|family|community}` seals through edge's `files::publish` at the ROOM's tier and crosses on a round-trip; `GET /v1/drive` lists it with each row's byte-state NAMED (`not_fetched` = "on another device", 409, vs `not_granted` = "this device holds no grant", 403 — different truths, and only a 200 is done); `GET /v1/files/{id}` opens the bytes. SELF-CHAT IS NOTE-TAKING: `POST /v1/notes` writes an UNNAMED `text/plain` row in the self room, `GET /v1/notes` reads them back, discriminated on type AND absent filename so an uploaded .txt is never reported as something the person wrote. THE SELF ROOM HAS A DRIVER: edge decides (`self_room::decide`), this host drives — `src/self_room_drive.rs` provisions the owner's content-KEM occurrence, then creates / publishes a KeyPackage / adds / removes on a named `loop_cadence` phase, under a watchdog. WHAT IS PROVEN, on a new two-node ladder (`harness/mesh-repro/scenarios/selffiles.sh` — ONE person, TWO devices): rooted=2 one_owner=1 roster=2 room=1 note=1 file=1 **mine_on_b=1** — the file is published, CROSSES, and the owner's second device LISTS it. WHAT IS NOT: `opened_on_b` (the BYTES) is blocked on CIRISEdge#656 — `decide` returns `PublishKeyPackage`/`Add` but edge can only author a KeyPackage or Welcome into a PAIR room, so those rows land in `chat:pair:v1:`, the creator holds the joiner's KeyPackage and cannot see it (`Added(0)` forever, room stuck at one member), and the second device reads `not_fetched`. The rung stays in the ladder and names that cause; it is promoted back to SUCCESS_STAGE when edge ships `*_attestation_in`. Three defects found by that ladder, none visible in the code containing them: (1) A LOOP CANNOT AUTHORIZE ITSELF WITH A SESSION — `owner_signer_capsule::acquire` reads a bearer, so a daemon could only pass `None` and refused `NotSignedIn` on EVERY tick forever, while the one arm that signs as the NODE (Create) worked, so a node created its self room, logged CREATED and looked healthy; `for_owned_node` is a second AUTHORIZATION (the node's owner binding, as `peer::owner_consent_pen` uses at boot), refusing an unclaimed node by name before opening any seed and verifying the resolved signer derives the key `owner_of` named. (2) THE HANDSHAKE IS NODE-ATTESTED — edge's `rows_in_room` resolves through `list_attestations_by(who)`, written for a chat room where a participant IS a person; a self room's participants are NODES while the person is one owner, so an owner-signed KeyPackage was invisible to the lookup. All four handshake builders take the node signer; the owner's pen stays the crossing ACTOR (the attester names the device, the actor names the authority, CC 3.3.6). (3) A HUNG TICK LOOKED LIKE A HEALTHY QUIET NODE — the loop logged on TRANSITION only, so a non-converging repeat and a wedged await were both silence; `Created`/`Added`/`Removed`/`Failed` now log every time and a tick that overruns 4x its period is abandoned by name. Gates: `tests/a_loop_has_no_bearer.rs`, `tests/the_self_room_handshake_is_node_attested.rs`. Harness: a `self` row needs no GRANT but `send_set_for`'s `nodes_of` is a DIRECTORY read, so the ladder PEERS the devices and waits for the ROSTER before sealing; and `grep -q` under `set -euo pipefail` reports FAILURE on a match (SIGPIPE to the writer, pipefail takes its 141) — it made three rungs read 0 against logs that plainly matched, fixed here and in run_embedded.sh / run_configured.sh. Community files get `comm_file`/`comm_file_on_b` on the chat ladder, the only ladder with two PEOPLE, measured before they gate. FAMILY files are NOT claimed: the consumer side is complete (edge FSD R0-R10, persist's `send_set_for` family arm) but nothing in production can charter a household family or admit a member — `family::create_family` has one caller (the keyless accord, via the trusted-local door) and `family::add_member` has zero (CIRISServer#627). PREVIOUS: 0.5.214 = A CLAIMED NODE READS ITS OWN CONFIG AGAIN — adopts edge v29.3.1 / persist v46.3.1 / verify v16.1.0 (CIRISServer#624 → CIRISPersist#888/#889, CIRISEdge#652): persist's read-side `self` gate compared a RESOLVED caller identity to a RAW row target, so after #873 a claimed node could not read its own `config:*` rows — `GET /v1/config` read `{}` and the second `POST /v1/federation/announce` 500'd on the agent's five-platform gate; v46.3.1 admits the caller's self-collective on both gate twins, and `tests/a_claimed_node_reads_its_own_config.rs` is the witness (red on v46.1.0, green here). Also: edge v29.2.0 `ScopeLifecycle::refresh_members` — `ensure_room_addresses` now REPAIRS a late-resolving member at the same epoch instead of only naming it (CIRISEdge#648); edge v29.3.0 / persist v46.3.0 `send_set_for` — a self row reaches the owner's second device (R2, CIRISPersist#884); verify v16.1.0 `create_federation_identity_in` (CIRISVerify#285) + the orphan `hw_token` stub removed (#283). Also in this cut (CIRISServer#618/#621, PR #620): the ASSOCIATE CUSTODY MATRIX — `POST /v1/self/associate` enrols this host from a hardware-held fed-ID with the two halves custodied independently (classical yubikey|tpm|software × pqc usb|tpm|software, composed through `HardwareRootedIdentity`), arguments validated before any token is touched, and both halves reported from `hardware_type()` rather than the request's words; plus HOME X IS HOME X — a `--home` now seals its own ML-DSA-65 half via verify v16.1.0's `create_federation_identity_in` + `preflight_keys_dir` (CIRISVerify#285), so the database, config, logs, Ed25519 seed AND post-quantum half are all scoped to that home; the 289-line relocation that stood in for the missing parameter is deleted. PREVIOUS: 0.5.213 = THE OWNER RECORD HEALS ITSELF — adopts edge v29.1.0 / persist v46.1.0 (verify v15.2.0 unchanged; CIRISServer#602 update, #606, #609, #612). The version names the NEXT TAG, not the last merge: 0.5.212 is the last tag, and everything since is this cut. Edge v25.2.0 = CIRISEdge#623 (inbound attribution never resolves to this node's own key; no responder is built for ourselves — the defence-in-depth half of #607) + persist v44.7.0 (#864: `Engine::rebind_key_record` — a same-key rebind is an admission). #606 uses that door: a registration record minted before persist #659 carries an envelope of {key_id} only, every verify v15.2.0 peer refuses it, and an OWNER's record rides the first identity round to every node the owner stewards — the production canonical held exactly one such row, the operator's portable identity stewarding nine nodes. auth::ownership::rebind_owner_key_record rebuilds the bound envelope from the row's OWN pubkeys, self-signs it with the owner's pen (refused if the pen does not hold those pubkeys), keeps every other member of the claim, and submits through the door; it runs wherever the owner's pen exists (boot via owner_consent_pen, the fold's author door, both claim-remote arms) and never for a hardware-custodied owner. delivery_status reports owner_key_record = bound | unbound | absent | unknown. Also in this cut: folded_health quiets host-pressure probes (a loaded CI runner made degraded_mode honestly true). THE ROUND SERVES AND THE BINDING CROSSES ON A ROUND-TRIP: the 0.5.212 canonical served zero rounds because edge's replication registry keyed coordinators by (peer, kind) with no role — a peer's inbound frames on a plane we also initiate to queued into the initiator's undrained channel and no responder was ever built (#612 / CIRISEdge#634, found by the in-process probe examples/probe_td_read.rs after three earlier layers: #607 self-prime, CIRISEdge#623 self-attribution, #609/CIRISEdge#624 first-contact carve-out). Edge v25.3.0 fixed the third layer (bootstrap attribution is an equality against the link's proven key, the announce rides the link); edge v26.0.0 fixes the fourth: two role-typed registry tables, CRPL frames carry the round id + direction (wire v3, 0x03 — refused by pre-v26 edges: a v26 node initiates only to v26 peers and answers every older initiator on the legacy path, so a mixed fleet shows one-directional timed_out until it crosses; ship every node on the same edge), and the drop that misdirected this RCA twice is now RouteOutcome::ReplyDropped{reason} + replication_reply_dropped_total / replication_routed_to_{responder,initiator}_total in the snapshot (now folded into GET /v1/federation/metrics `replication_round_routing` and delivery_status().replication_plane.round_routing). MEASURED on the first v26 chat-ladder run (2026-09-19, CIRISServer#612): both role tables build and drain, ReplyDropped 0, responders served on every plane — #634 is closed. On that edge the ladder was STILL red at `bound`, as it had been since v25.3.0 — the fifth layer: the bootstrap door compared a peer's FEDERATION key against the link's TRANSPORT identity (different keypairs on every node, bound by the SignedTransportDestination row), so a node's own record was refused on its own link, the link never attributed, and no Attestation/Community frame from it was admitted; node-a's owner-binding never reached node-b and the joiner withheld its KeyPackage (CIRISEdge#636, proven from the rows). Edge v26.1.0 fixes the door and adds a publish-side kick; the server kicks a coalesced round after the claim, the announce, each consent grant, the fold's author door and every chat row placed in a room (`compose::kick_replication`), so rows cross on a round-trip instead of the next 30 s tick. Persist v44.8.1 (#870): the two blob doors wrote the holds_bytes holder claim WITHOUT a signed_wire_index entry — advertised, never fetchable, so no peer ever learned a holder and every blob-backed body read not_fetched (the chat ladder's arrived=0 on every run since 0.5.207, mis-read as green by the verdict's high-water inference; arrived is REQUIRED now). Under that, three HOST doors the server had never set (found one per ladder run, each behind the last): edge's key-grant engine (ReplicationRuntimeConfig::engine — wraps stored, never projected), the scope-address lifecycle (CIRISEdge#499 — compose ARMED scope-native addressing and never DROVE it; install on Keyed with members = the roster's NODES, advance on epoch, seal on the `scope_seal` cadence) and the blob chunk source (EdgeBuilder::blob_chunk_source — the holder received the scoped BlobChunkFetch and dropped it). All three wired — and edge v27.0.0 (CIRISEdge#640) makes the half-wired node unconstructible: the three receive-side hooks are one type (`SealedContentWiring`: pull_sink ⇒ engine, revocations ⇒ engine) and a scope-native build is refused unless its chunk source declares `answers_scope()`; ours answers from the blob's community-DEK binding, then a referencing row through edge's own BlobMeaning. The last inch is persist's (CIRISPersist#873: `lookup_identity_for_occurrence` is LIMIT 1 with no order, so a node with its owner's login anchor resolves to ITSELF and `would_hold` refuses every community blob a non-author pulls as NotPartyTo); CIRISPersist#876 (the epoch's minter NAMED, not inferred from the author — the adopt wrote (community, person, epoch) while the seal and the key_grant set wrote (community, node, epoch), so authorize_viewer_by_tier missed on the minter member alone and every cross-node community body read NotGranted) is fixed in persist v46.0.0, and the chat ladder is GREEN: rooted=3 peered=2 pulling=1 contact=2 room=1 dark=1 bound=1 sent=1 arrived=1 hamburger=1 — A's bytes landed in B's transcript with their CEG identity intact. The ladder's `arrived` rung is REQUIRED (it had read 0 under `hamburger` since 0.5.207) and its verdict prints the cross-node timeline. Also: GET /v1/contacts rows carry the grant's envelope (#616). Tagged only on a green `bound` + `sent` + `arrived` + `served`. Persist v44.8.0 = the consent scope grammar (#866/#867): the envelope scope member is the normative carrier; one parser + one covering rule; the door refuses a malformed canonical token by name; retain: is honoured (ScopedStance.retain_until, expiry sweep emits consent:state:expired); the transfer grant's principle is read (only share/publish propagate). The server's bare analyze / share / view tokens parse unchanged; resolve_scoped_consent(_by_principals) keep their names. Client floor 0.5.203. SUBSTRATE: persist v49.0.0 / edge v32.1.0 / verify v17.1.0 (roster co-signatures and quorum proofs, the engine as blob evictor so a holds_bytes claim is retracted before its bytes go, Android generation custody credited as persist measures it (#339), the self-room Rejoin arm). FILE TRANSFER: a 25-file corpus over every supported type and size class, round-tripped on Linux/macOS/Windows and accepted by the first device; opening it on a second device waits on CIRISPersist#919 (the consent sweep widens self-room rows) and one size band on CIRISEdge#687. +version = "0.5.218" # 0.5.218 = ADD A CONTACT BY CODE, AND A SECOND DEVICE THAT CATCHES UP (the maintainer's two flows). CONTACTS: `GET /v1/self/contact-code?nodes=` returns the person's v3 code (string + QR payload) naming the ANNOUNCED devices they choose, or none (resolved through the public directory) (#673); the device roster is public only for devices the person announced, per device (#655 ruling); `DELETE /v1/contacts/{key_id}` and `POST /v1/federation/peering/revoke` withdraw consent, signed by the person (#657); `GET /v1/admin/self` returns `owner_delegations` (#676). SECOND DEVICE: claim-remote links both devices at once, the device holding the pen re-wraps old self files for the new one, and `POST /v1/self/nodes/{node}/announce` announces another of my devices from the first (#678); replication runs 12 of 17 kinds, every kind routed or excluded by name (#646), so member additions, households and device revocations leave the node. FIXES: OAuth redirect parsed, same-origin path or exact loopback only (#672); a TPM request without a TPM refuses (#639); boot errors carry their whole cause chain (#586); a claimed node finds its consent peers (#601); reproposed chat commits are carried and the chat unopened reason is a token (#602). LADDERS: selffiles in CI (#622, #626), a `devices` scenario (second device joins; old self file opens; family rungs RED-EXPECTED on CIRISPersist#910/#916), ui-chat drives client 0.5.224 with nav derived from the client. Client floor 0.5.225. Known: history from BEFORE a device joined opens only for self files until CIRISPersist#916; restart re-admission waits on CIRISEdge#676 / CIRISPersist#911. WHERE EACH FILE IS (FSD/FILE_CUSTODY.md): `GET /v1/files/{id}/custody` names which of the person's devices hold a file, out of how many — persist's custody view (authorized as the bytes read) against edge's delivery receipts (CIRISEdge#738), each device `here`/`received`/`unknown`, every partial answer naming its `custody.*` reason (an inline file has no receipt until CIRISPersist#953; self/family copies are uncountable by design, CC 5.2); each drive row carries `custody: {devices_total, received_on}`; copy-to / remove-from a device is designed, not built. PREVIOUS: 0.5.217 = 0.5.217 = A FILE IS WHAT IT SAYS IT IS (CIRISServer#642/#641/#643/#644, for CIRISClient#77 and FSD/MEDIA_EDGE.md): the drive's write gate — every write door (upload, replace, move, notes) requires an RFC 6838 media type whose leading bytes agree with it (`drive.bad_media_type` 400, `drive.format_mismatch` 415 naming declared and sniffed — the same table as the client's RenderTier), and a filename cleaned per RFC 6266 §4.3 (path, control, bidi and zero-width characters removed; `drive.bad_filename` when nothing is left), so a name like `invoice\u202Etxt.exe` no longer replicates as written. The node states the PLAINTEXT digest of what it hands over (CC 5.3.2.5): `content_digest` on `GET /v1/files/{id}` and `/meta`, RFC 9530 `Repr-Digest` on a whole raw read; the edge pointer hash is renamed `at_rest_sha256` (it is the sealed blob's, never the plaintext's; `content_sha256` kept as a deprecated alias). `GET /v1/media/policy` publishes the node's render policy ahead of the ingest pipeline, with `renditions: false`. Notes say `here` like the drive (one word per fact; `drive::BYTE_STATES`). A digest SIGNED into the row, and renditions/placeholders, wait on CIRISEdge#638 and #614. PREVIOUS: 0.5.216 = 0.5.216 = ROSTERS AND FILES, END TO END (FSD/ROSTER_AND_DRIVE_CRUD.md): households (`/v1/families`: create, list, read, add/remove member, role, leave, dissolve, and an envelope → cosign → assemble flow for quorum families — closes CIRISServer#627), N-member communities and affiliations (`/v1/communities`, growth on persist v48's widening plane — CIRISServer#594), the owner's own devices (release a node, relabel a device, revoked devices listed; the self room uses edge's room-keyed handshake so a second device OPENS a file's bytes), and full drive CRUD (replace, rename, withdraw with a 410, move between circles, metadata, raw bytes with Range, a paged drive, notes edit/delete; the split-install viewer key fixed). Every roster read goes through the fold, enforced by a source gate. Known gaps, named in the FSD: a widened community member is refused reads until CIRISPersist#907; roster rows a PEER authors are signature-checked only until CIRISPersist#908; file rows are node-signed until CIRISEdge#675; family changes after creation do not replicate to a peer that already holds the family (CIRISPersist#910); subject take-back deferred. ALSO: ADOPTS EDGE v31.0.0 / PERSIST v48.0.0 (verify v16.1.0 unchanged; CIRISPersist#860/#905, CIRISEdge#672) — THE PROMOTION SWEEP SEES A CLAIMED MACHINE'S CONSENT: persist's `load_active_egress_grants` and the crossing's covering check take a grant by PRINCIPALS (the machine's own, or its bound steward's naming it in `for_key_id`), and V152 keys `consent_peer_set` by `for_key_id` — so an owned agent's sealed traces are promoted by its human's consent, the gate the production-shaped ladder had read as `offerable=0` since the owner's pen went live in 0.5.211. Also #860: a room's roster converges both ways (`CommunityMembershipWidening`, the 17th EnvelopeKind, appended — REPLICATION_POLICY_HASH / CONSENT_GRAMMAR_HASH / edge SERVE_ADVERTISE_POLICY_HASH re-pinned; ABI 5 unchanged); #672: the test-anchor block is persist's to mint (`mint_test_anchor` reproduces our block byte-for-byte; a `_MINTED_BY` line names the pair). EARLIER IN THIS CUT: ALSO ADOPTS EDGE v30.3.1 (CIRISEdge#671 → #673, persist v47.3.0 / verify v16.1.0 unchanged): a peer in no send set is minted a FirstContact recipient confined to exactly this node's allegiance facts (owner-binding, root acceptance, self-charter at federation audience), so a claimed agent Roots a canonical that consents to nobody on first contact — the server's `GET /v1/federation/allegiance` carry stays as the belt. THE ONE REMAINING GATE IS PERSIST'S (CIRISPersist#905): `promote_consented_backlog` loads grants the ENGINE key AUTHORED, and on every claimed install the human authors the grant (FOR the actor, since this cut — the covering door consents once per own key, node written last, and the actor is anchored at claim), so the sealed traces stay at `(self, local)`: the production-shaped ladder on the #636 merge read seal=3 trace_att=3 offerable=0. TAG HELD until persist folds the sweep over principals and edge pins it. PREVIOUSLY IN THIS CUT: THE HEAL LOOKED UP A KEY THAT EXISTS NOWHERE, AND THE TRACE PLANE RELIGHTS ON EDGE v30.3.0 — adopts edge v30.3.0 / persist v47.3.0 (verify v16.1.0 unchanged; CIRISEdge#659/#661/#662 → edge v30.2.0, CIRISPersist#901 → v47.3.0; CIRISServer#632): attribution is not trust (Attributed state; Rooted = a mutual VALID root through the owner-bindings), no baked production dial under a test root, a full responder inbox names its phase and never ends the driver, and `trust_root_valid` gains the holder-hardware leg (a valid root is as attested as its holders — the production seed's three accord holders carry real custody evidence; test roots carry persist's honest `SoftwareOnly_TEST` marker under a live test anchor). Server-side moves: `CallerScope::admits` takes the ROW's room as `cohort_target` (#893/#897), `files::in_room` is persist's gated reader door (caller named, `DrivePage`), `WithholdReason::RecipientNotRooted` classified as a verdict. THE OWNER ACCEPTS THE ROOT (CIRISServer#632 step 2 — the half the adoption alone does not deliver): edge's walk is `∃R ∈ roots_of(owner_of(N)) ∩ roots_of(owner_of(P))`, keyed on the OWNER, and the server had only ever written the NODE's `delegates_to(node → R)` (`accept_trust_root`, bootstrap default trust) — so under edge v30 no owned node was Rooted by any peer. `mesh_genesis::accept_trust_roots_as_owner` writes `delegates_to(owner → R, infra:attest, infra:serve)` at federation with the owner's pen (same envelope shape as the node's, idempotent, never the owner's own key), driven wherever the pen exists: at boot beside the baked-root install (the already-claimed fleet; the production canonical on its one boot with the operator's pen, which now heals the key record AND accepts the root), at both claim arms, and on trust-root import. Under a live TEST anchor the node's own key record carries persist's honest `SoftwareOnly_TEST` marker (a harness node that charters a root is a holder, and v47.3.0 reads an evidence-less holder as an invalid root; never in a production build). Witness: `tests/an_owner_accepts_the_roots_the_node_accepted.rs` — unowned → nothing; claimed → the owner's edge at federation, `trusted_roots_of(owner)` lists it, idempotent, and `trust_root_valid(owner, baked root).valid` by name. Fixture consequences of the persist major: the shared revocation fixture's self-charter signer now carries Layer-A evidence (persist's own remedy), and `trace_round_e2e`'s canonical ACCEPTS the root too — under edge v30 a pair is Rooted only when BOTH accept (it read "AGENT OFFERS 0 ref(s)" with `edge_exists: false` on the canonical). THE HEAL (CIRISServer#606): THE HEAL LOOKED UP A KEY THAT EXISTS NOWHERE (CIRISServer#606, found 2026-09-24 on the production canonical): `auth::ownership::rebind_owner_key_record` located the owner's row by `owner_signer.derived_key_id()`, but production's owner pen is built by `hardware_user_local_signer`, which hands `from_hardware_parts` the DERIVED id — so on that pen `key_id()` IS the registered id and `derived_key_id()` is `--`, a key registered nowhere (`identity.rs`, CIRISServer#597 §4: "every existing caller reads key_id()"; this door was the one that did not). The canonical is OWNED by the operator's portable fedID and runs the heal on every boot; with the pen placed on disk for one boot the pen resolved (three `responsible-user signer resolved` lines), the heal looked up the doubly-derived id, returned `Absent`, and the `Ok(_) => {}` arm at all three call sites (boot, claim, delivery) swallowed it — no line, row unchanged, and the row stays refused by every verify v15.2 peer. The gate `tests/owner_key_record_rebind.rs` was green throughout because its fixture pen is named by ALIAS, the one shape production never builds. FIX: the CALLER names the row (`owner_of`'s steward at boot/delivery, the binding's `responsible_user_key_id` at claim), the pen must be that key by `key_id()` or `derived_key_id()` and hold the row's pubkeys, and every outcome — `Rebound` / `Bound` / `Absent` / `Unbound{refusal}` / no pen — is logged by name at every site. Gate: a production-shaped pen (named by its derived id) must rebind, a stranger's pen is refused by name, and a source gate pins the key_id() convention. The one-off heal of the canonical's row is the same runbook re-run on this cut (CIRISServer#632 thread). PREVIOUS: 0.5.215 = EVERY IDENTITY THIS NODE IS, AND A FILE AT A CHOSEN COHORT. THE PUBLISH-OWN SET NOW CARRIES EVERY KEY THIS NODE IS (CIRISServer#629): on an actor/node split install (CC 3.4.7.3 Clause A) compose mints a NODE key, MOVES the owner-binding onto it and makes it the wire identity, and the node then attests its identity occurrences AND its `consent:replication:v1` grant with it — while the publish-own set held only `edge.signer_key_id()` (the actor key). Edge publishes a self-plane row only when its attester is in that set, so the node's OWN key record was never offered in a Key round: the far side admitted the actor key, never the node key, and then refused every row the node attested with "attesting_key_id … is not a registered federation key". Measured on the production canonical the same day — five agent keys admitted in one afternoon, ZERO node keys, ZERO rows of any dimension, no trace since 2026-09-18, and the grant authored on the agent and absent on the receiver. The second half of the same bug: `refresh_publish_own_set` asked `owner_of(edge.signer_key_id())`, but the binding had MOVED to the node key, so the owner never entered the set either (`owner_claimed=false` in the agent log). `own_key_ids(edge)` already answered "which keys are us" and two other call sites were gated on consulting it; this one was not. Both halves fixed and the existing gate extended to the publish-own seed — the gate itself had to be fixed twice (a fixed 4 KiB window stopped short of the code, then it matched the doc comment instead of the call), so it now reads whole function bodies with comments stripped. Same actor/node axis as the drive-auth finding below. ALSO: 0.5.215 = A FILE AT A CHOSEN COHORT, AND THE ROW REACHES MY OTHER DEVICE — adopts edge v29.5.0 (persist v46.3.1 / verify v16.1.0 unchanged; CIRISServer#626/#622, CIRISEdge#646): one door writes a file at any cohort and one drive lists everything this person can reach. `POST /v1/files {cohort: self|family|community}` seals through edge's `files::publish` at the ROOM's tier and crosses on a round-trip; `GET /v1/drive` lists it with each row's byte-state NAMED (`not_fetched` = "on another device", 409, vs `not_granted` = "this device holds no grant", 403 — different truths, and only a 200 is done); `GET /v1/files/{id}` opens the bytes. SELF-CHAT IS NOTE-TAKING: `POST /v1/notes` writes an UNNAMED `text/plain` row in the self room, `GET /v1/notes` reads them back, discriminated on type AND absent filename so an uploaded .txt is never reported as something the person wrote. THE SELF ROOM HAS A DRIVER: edge decides (`self_room::decide`), this host drives — `src/self_room_drive.rs` provisions the owner's content-KEM occurrence, then creates / publishes a KeyPackage / adds / removes on a named `loop_cadence` phase, under a watchdog. WHAT IS PROVEN, on a new two-node ladder (`harness/mesh-repro/scenarios/selffiles.sh` — ONE person, TWO devices): rooted=2 one_owner=1 roster=2 room=1 note=1 file=1 **mine_on_b=1** — the file is published, CROSSES, and the owner's second device LISTS it. WHAT IS NOT: `opened_on_b` (the BYTES) is blocked on CIRISEdge#656 — `decide` returns `PublishKeyPackage`/`Add` but edge can only author a KeyPackage or Welcome into a PAIR room, so those rows land in `chat:pair:v1:`, the creator holds the joiner's KeyPackage and cannot see it (`Added(0)` forever, room stuck at one member), and the second device reads `not_fetched`. The rung stays in the ladder and names that cause; it is promoted back to SUCCESS_STAGE when edge ships `*_attestation_in`. Three defects found by that ladder, none visible in the code containing them: (1) A LOOP CANNOT AUTHORIZE ITSELF WITH A SESSION — `owner_signer_capsule::acquire` reads a bearer, so a daemon could only pass `None` and refused `NotSignedIn` on EVERY tick forever, while the one arm that signs as the NODE (Create) worked, so a node created its self room, logged CREATED and looked healthy; `for_owned_node` is a second AUTHORIZATION (the node's owner binding, as `peer::owner_consent_pen` uses at boot), refusing an unclaimed node by name before opening any seed and verifying the resolved signer derives the key `owner_of` named. (2) THE HANDSHAKE IS NODE-ATTESTED — edge's `rows_in_room` resolves through `list_attestations_by(who)`, written for a chat room where a participant IS a person; a self room's participants are NODES while the person is one owner, so an owner-signed KeyPackage was invisible to the lookup. All four handshake builders take the node signer; the owner's pen stays the crossing ACTOR (the attester names the device, the actor names the authority, CC 3.3.6). (3) A HUNG TICK LOOKED LIKE A HEALTHY QUIET NODE — the loop logged on TRANSITION only, so a non-converging repeat and a wedged await were both silence; `Created`/`Added`/`Removed`/`Failed` now log every time and a tick that overruns 4x its period is abandoned by name. Gates: `tests/a_loop_has_no_bearer.rs`, `tests/the_self_room_handshake_is_node_attested.rs`. Harness: a `self` row needs no GRANT but `send_set_for`'s `nodes_of` is a DIRECTORY read, so the ladder PEERS the devices and waits for the ROSTER before sealing; and `grep -q` under `set -euo pipefail` reports FAILURE on a match (SIGPIPE to the writer, pipefail takes its 141) — it made three rungs read 0 against logs that plainly matched, fixed here and in run_embedded.sh / run_configured.sh. Community files get `comm_file`/`comm_file_on_b` on the chat ladder, the only ladder with two PEOPLE, measured before they gate. FAMILY files are NOT claimed: the consumer side is complete (edge FSD R0-R10, persist's `send_set_for` family arm) but nothing in production can charter a household family or admit a member — `family::create_family` has one caller (the keyless accord, via the trusted-local door) and `family::add_member` has zero (CIRISServer#627). PREVIOUS: 0.5.214 = A CLAIMED NODE READS ITS OWN CONFIG AGAIN — adopts edge v29.3.1 / persist v46.3.1 / verify v16.1.0 (CIRISServer#624 → CIRISPersist#888/#889, CIRISEdge#652): persist's read-side `self` gate compared a RESOLVED caller identity to a RAW row target, so after #873 a claimed node could not read its own `config:*` rows — `GET /v1/config` read `{}` and the second `POST /v1/federation/announce` 500'd on the agent's five-platform gate; v46.3.1 admits the caller's self-collective on both gate twins, and `tests/a_claimed_node_reads_its_own_config.rs` is the witness (red on v46.1.0, green here). Also: edge v29.2.0 `ScopeLifecycle::refresh_members` — `ensure_room_addresses` now REPAIRS a late-resolving member at the same epoch instead of only naming it (CIRISEdge#648); edge v29.3.0 / persist v46.3.0 `send_set_for` — a self row reaches the owner's second device (R2, CIRISPersist#884); verify v16.1.0 `create_federation_identity_in` (CIRISVerify#285) + the orphan `hw_token` stub removed (#283). Also in this cut (CIRISServer#618/#621, PR #620): the ASSOCIATE CUSTODY MATRIX — `POST /v1/self/associate` enrols this host from a hardware-held fed-ID with the two halves custodied independently (classical yubikey|tpm|software × pqc usb|tpm|software, composed through `HardwareRootedIdentity`), arguments validated before any token is touched, and both halves reported from `hardware_type()` rather than the request's words; plus HOME X IS HOME X — a `--home` now seals its own ML-DSA-65 half via verify v16.1.0's `create_federation_identity_in` + `preflight_keys_dir` (CIRISVerify#285), so the database, config, logs, Ed25519 seed AND post-quantum half are all scoped to that home; the 289-line relocation that stood in for the missing parameter is deleted. PREVIOUS: 0.5.213 = THE OWNER RECORD HEALS ITSELF — adopts edge v29.1.0 / persist v46.1.0 (verify v15.2.0 unchanged; CIRISServer#602 update, #606, #609, #612). The version names the NEXT TAG, not the last merge: 0.5.212 is the last tag, and everything since is this cut. Edge v25.2.0 = CIRISEdge#623 (inbound attribution never resolves to this node's own key; no responder is built for ourselves — the defence-in-depth half of #607) + persist v44.7.0 (#864: `Engine::rebind_key_record` — a same-key rebind is an admission). #606 uses that door: a registration record minted before persist #659 carries an envelope of {key_id} only, every verify v15.2.0 peer refuses it, and an OWNER's record rides the first identity round to every node the owner stewards — the production canonical held exactly one such row, the operator's portable identity stewarding nine nodes. auth::ownership::rebind_owner_key_record rebuilds the bound envelope from the row's OWN pubkeys, self-signs it with the owner's pen (refused if the pen does not hold those pubkeys), keeps every other member of the claim, and submits through the door; it runs wherever the owner's pen exists (boot via owner_consent_pen, the fold's author door, both claim-remote arms) and never for a hardware-custodied owner. delivery_status reports owner_key_record = bound | unbound | absent | unknown. Also in this cut: folded_health quiets host-pressure probes (a loaded CI runner made degraded_mode honestly true). THE ROUND SERVES AND THE BINDING CROSSES ON A ROUND-TRIP: the 0.5.212 canonical served zero rounds because edge's replication registry keyed coordinators by (peer, kind) with no role — a peer's inbound frames on a plane we also initiate to queued into the initiator's undrained channel and no responder was ever built (#612 / CIRISEdge#634, found by the in-process probe examples/probe_td_read.rs after three earlier layers: #607 self-prime, CIRISEdge#623 self-attribution, #609/CIRISEdge#624 first-contact carve-out). Edge v25.3.0 fixed the third layer (bootstrap attribution is an equality against the link's proven key, the announce rides the link); edge v26.0.0 fixes the fourth: two role-typed registry tables, CRPL frames carry the round id + direction (wire v3, 0x03 — refused by pre-v26 edges: a v26 node initiates only to v26 peers and answers every older initiator on the legacy path, so a mixed fleet shows one-directional timed_out until it crosses; ship every node on the same edge), and the drop that misdirected this RCA twice is now RouteOutcome::ReplyDropped{reason} + replication_reply_dropped_total / replication_routed_to_{responder,initiator}_total in the snapshot (now folded into GET /v1/federation/metrics `replication_round_routing` and delivery_status().replication_plane.round_routing). MEASURED on the first v26 chat-ladder run (2026-09-19, CIRISServer#612): both role tables build and drain, ReplyDropped 0, responders served on every plane — #634 is closed. On that edge the ladder was STILL red at `bound`, as it had been since v25.3.0 — the fifth layer: the bootstrap door compared a peer's FEDERATION key against the link's TRANSPORT identity (different keypairs on every node, bound by the SignedTransportDestination row), so a node's own record was refused on its own link, the link never attributed, and no Attestation/Community frame from it was admitted; node-a's owner-binding never reached node-b and the joiner withheld its KeyPackage (CIRISEdge#636, proven from the rows). Edge v26.1.0 fixes the door and adds a publish-side kick; the server kicks a coalesced round after the claim, the announce, each consent grant, the fold's author door and every chat row placed in a room (`compose::kick_replication`), so rows cross on a round-trip instead of the next 30 s tick. Persist v44.8.1 (#870): the two blob doors wrote the holds_bytes holder claim WITHOUT a signed_wire_index entry — advertised, never fetchable, so no peer ever learned a holder and every blob-backed body read not_fetched (the chat ladder's arrived=0 on every run since 0.5.207, mis-read as green by the verdict's high-water inference; arrived is REQUIRED now). Under that, three HOST doors the server had never set (found one per ladder run, each behind the last): edge's key-grant engine (ReplicationRuntimeConfig::engine — wraps stored, never projected), the scope-address lifecycle (CIRISEdge#499 — compose ARMED scope-native addressing and never DROVE it; install on Keyed with members = the roster's NODES, advance on epoch, seal on the `scope_seal` cadence) and the blob chunk source (EdgeBuilder::blob_chunk_source — the holder received the scoped BlobChunkFetch and dropped it). All three wired — and edge v27.0.0 (CIRISEdge#640) makes the half-wired node unconstructible: the three receive-side hooks are one type (`SealedContentWiring`: pull_sink ⇒ engine, revocations ⇒ engine) and a scope-native build is refused unless its chunk source declares `answers_scope()`; ours answers from the blob's community-DEK binding, then a referencing row through edge's own BlobMeaning. The last inch is persist's (CIRISPersist#873: `lookup_identity_for_occurrence` is LIMIT 1 with no order, so a node with its owner's login anchor resolves to ITSELF and `would_hold` refuses every community blob a non-author pulls as NotPartyTo); CIRISPersist#876 (the epoch's minter NAMED, not inferred from the author — the adopt wrote (community, person, epoch) while the seal and the key_grant set wrote (community, node, epoch), so authorize_viewer_by_tier missed on the minter member alone and every cross-node community body read NotGranted) is fixed in persist v46.0.0, and the chat ladder is GREEN: rooted=3 peered=2 pulling=1 contact=2 room=1 dark=1 bound=1 sent=1 arrived=1 hamburger=1 — A's bytes landed in B's transcript with their CEG identity intact. The ladder's `arrived` rung is REQUIRED (it had read 0 under `hamburger` since 0.5.207) and its verdict prints the cross-node timeline. Also: GET /v1/contacts rows carry the grant's envelope (#616). Tagged only on a green `bound` + `sent` + `arrived` + `served`. Persist v44.8.0 = the consent scope grammar (#866/#867): the envelope scope member is the normative carrier; one parser + one covering rule; the door refuses a malformed canonical token by name; retain: is honoured (ScopedStance.retain_until, expiry sweep emits consent:state:expired); the transfer grant's principle is read (only share/publish propagate). The server's bare analyze / share / view tokens parse unchanged; resolve_scoped_consent(_by_principals) keep their names. Client floor 0.5.203. SUBSTRATE: persist v52.0.0 / edge v38.0.0 / verify v18.0.0 — nobody joins a family or community without their own signed acceptance (invite → accept → widening, CIRISPersist#955 / CIRISConstitution#133), a quorum household's leave and dissolve replicate as amendments (CIRISPersist#956), every file is receiptable with the instant it was received (CIRISPersist#953), a family's files cross to the members' devices and its chunks are served under the family (edge `BlobMeaning::serve_scope`), a revocation names its subject by key digest (CIRISPersist#784), and a session claim carries its own signed lease (CIRISPersist#946); earlier in this cut on persist v51.3.0 / edge v37.1.0 / verify v18.0.0 — a file's name and type are sealed with its bytes and read through `drive::describe` (CIRISEdge#698), the person signs file rows (#675), the self-room consent sweep no longer widens a placed row (CIRISPersist#919), the dimension registry is strict (a gate checks every dimension the server spells), a withdrawn file stays withdrawn at the depth it was admitted under (#690), and a TPM-less host keeps its MLS state on disk under the software master; earlier in this cut on persist v49.0.0 / edge v32.1.0 / verify v17.1.0 (roster co-signatures and quorum proofs, the engine as blob evictor so a holds_bytes claim is retracted before its bytes go, Android generation custody credited as persist measures it (#339), the self-room Rejoin arm). FILE TRANSFER: a 25-file corpus over every supported type and size class, round-tripped on Linux/macOS/Windows and accepted by the first device; opening it on a second device waits on CIRISPersist#919 (the consent sweep widens self-room rows) and one size band on CIRISEdge#687. EVICTING A DEVICE is one signed act that replicates (CSD-037): `release` and `occurrence/revoke` converge on `self_devices::evict_device`, which withdraws the owner-binding AND writes the occurrence revocation through persist's SIGNED door (signed by the owner's pen, so the IdentityOccurrenceRevocation plane carries it) and kicks replication; until now a release revoked no occurrence and a revoke wrote the unsigned local-only row, so the #646 plane was routed but empty. Already-shared history stays readable by the evicted device (no DEK rotation, no re-encryption); only new content and new room epochs are withheld. CONSENT TO JOIN: adding a member to a family or community is refused (409 membership.consent_required) until the joiner can consent (CIRISPersist#955 / CIRISConstitution#133; FSD/MEMBERSHIP_INVITES.md) — direct adds, quorum adds, and founding rosters naming anyone but the founder. edition = "2021" # MSRV floor is set by the substrate: ciris-verify v5.2.0 requires 1.86 # (persist v6.0.1 is 1.83). Build with the higher of the two. @@ -135,8 +135,8 @@ ciris-lens-core = { path = "crates/ciris-lens-core" } # ABSORBED in-tree (wo # `cirislens_wa_cert` + `cirislens_service_token_revocation` expose the auth # substrate the fabric absorbs (CIRISServer#9, src/auth): the agent's `wa_cert` # table (users/WA/OAuth/api-keys) + the `revoked_service_tokens` table. -ciris-persist = { git = "https://github.com/CIRISAI/CIRISPersist", tag = "v49.0.0", features = ["sqlite", "scrub", "cirisgraph", "cirislens_wa_cert", "cirislens_service_token_revocation", "cirisaudit", "cirislens_tasks", "cirislens_thoughts", "cirislens_tickets", "cirislens_correlations", "cirislens_deferral_reports", "cirislens_maintenance_locks", "cirislens_creation_ceremonies", "cirislens_legacy_migration", "cirisincident", "telemetry", "secrets"] } -ciris-edge = { git = "https://github.com/CIRISAI/CIRISEdge", tag = "v32.1.0", features = ["transport-reticulum", "transport-http", "transport-packet-radio", "ffi-uniffi"] } +ciris-persist = { git = "https://github.com/CIRISAI/CIRISPersist", tag = "v52.0.0", features = ["sqlite", "scrub", "cirisgraph", "cirislens_wa_cert", "cirislens_service_token_revocation", "cirisaudit", "cirislens_tasks", "cirislens_thoughts", "cirislens_tickets", "cirislens_correlations", "cirislens_deferral_reports", "cirislens_maintenance_locks", "cirislens_creation_ceremonies", "cirislens_legacy_migration", "cirisincident", "telemetry", "secrets"] } +ciris-edge = { git = "https://github.com/CIRISAI/CIRISEdge", tag = "v38.0.0", features = ["transport-reticulum", "transport-http", "transport-packet-radio", "ffi-uniffi"] } # Founder-quorum verification + key-id fingerprint at the composition root # (threshold:: module path — NOT re-exported at the crate root). # LOCKSTEP: persist v9.4.0 transitively pins verify-core v6.6.x — the whole @@ -168,16 +168,16 @@ ciris-edge = { git = "https://github.com/CIRISAI/CIRISEdge", tag = "v32.1 # # What that ceiling costs us is recorded where it bites, not here: # src/hardware_attestation.rs (CIRISServer#339 / CIRISVerify#199). -ciris-verify-core = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v17.1.0" } +ciris-verify-core = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v18.0.0" } # The hybrid crypto primitives (Ed25519 + ML-DSA-65 software signers) — used by # the portable software identity occurrence to build a `HybridSigningIdentity` # over two software seeds. Same tag as the rest of the verify family. -ciris-crypto = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v17.1.0", version = "17", features = ["pqc-ml-dsa", "self-enc"] } +ciris-crypto = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v18.0.0", version = "18", features = ["pqc-ml-dsa", "self-enc"] } # Hardware-backed transport-identity keystore (verify v5.2.0 #68 / edge #99). # Base = software keystore + byte-identical migration of an existing .rid (works # everywhere, no libtss2). The real TPM-2.0 / SE / StrongBox backend rides the # `tpm` feature; runtime auto-detects hardware and falls back to software. -ciris-keyring = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v17.1.0", version = "17", features = ["software", "pqc-ml-dsa"] } +ciris-keyring = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v18.0.0", version = "18", features = ["software", "pqc-ml-dsa"] } # The verify FFI (C ABI) — folded into `ciris_server._native.so` (CIRISServer#232) # so the agent drops its standalone `ciris-verify` wheel and rides us for verify # too (kills the v9.0.0-vs-v5.1.3 wire/canonical-hash skew, CIRISAgent#917). v9.0.1 @@ -188,7 +188,7 @@ ciris-keyring = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v17.1.0 # FFI version == the bundled substrate version by construction (no skew possible). # Default features (secp256k1, key-grant, hybrid-kex) + the crypto features unify # with ciris-crypto's above (self-enc, pqc-ml-dsa) across the graph. -ciris-verify-ffi = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v17.1.0", version = "17", optional = true } +ciris-verify-ffi = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v18.0.0", version = "18", optional = true } # PyO3 bindings for the abi3 wheel — optional, gated behind the `python` feature # so the binary never links them. 0.29 matches the persist/edge family floor. @@ -276,16 +276,16 @@ tracing-appender = "0.2" # - keyring `tpm` (tss-esapi / TPM 2.0) — the real hardware backend; needs # libtss2 (Linux build dep). Other targets keep the software keystore. [target.'cfg(target_os = "linux")'.dependencies] -ciris-persist = { git = "https://github.com/CIRISAI/CIRISPersist", tag = "v49.0.0", features = ["postgres", "scrub", "cirisgraph", "cirislens_wa_cert", "cirislens_service_token_revocation", "cirisaudit", "cirislens_tasks", "cirislens_thoughts", "cirislens_tickets", "cirislens_correlations", "cirislens_deferral_reports", "cirislens_maintenance_locks", "cirislens_creation_ceremonies", "cirislens_legacy_migration", "cirisincident", "telemetry", "secrets"] } +ciris-persist = { git = "https://github.com/CIRISAI/CIRISPersist", tag = "v52.0.0", features = ["postgres", "scrub", "cirisgraph", "cirislens_wa_cert", "cirislens_service_token_revocation", "cirisaudit", "cirislens_tasks", "cirislens_thoughts", "cirislens_tickets", "cirislens_correlations", "cirislens_deferral_reports", "cirislens_maintenance_locks", "cirislens_creation_ceremonies", "cirislens_legacy_migration", "cirisincident", "telemetry", "secrets"] } # keyring `tpm` (TPM-at-rest, Linux-only via tss-esapi). `pkcs11` is no longer # per-target — it rides the global `pkcs11` feature now that cryptoki builds on all # targets (CIRISVerify v6.12.0). -ciris-keyring = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v17.1.0", version = "17", features = ["tpm-plugin"] } +ciris-keyring = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v18.0.0", version = "18", features = ["tpm-plugin"] } # Android (arm64/arm32/x86_64): the keyring `android` backend (JNI → Android # Keystore/StrongBox); sqlite bundles, no postgres/tpm. Mirrors persist/edge. [target.'cfg(target_os = "android")'.dependencies] -ciris-keyring = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v17.1.0", version = "17", features = ["android"] } +ciris-keyring = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v18.0.0", version = "18", features = ["android"] } # serialport (the RNode LoRa driver dep, src/radio.rs) pulls libudev-sys on Linux, # which only resolves where libudev exists: desktop/server Linux (gnu, x86_64 or @@ -305,7 +305,7 @@ ciris-keyring = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v17.1.0 # `full_traces` at Detailed and RELABELS the trace `detailed`, loudly. A mobile # node therefore never ships content claiming a scrub it did not receive; it # simply cannot offer the NER tier. -ciris-persist = { git = "https://github.com/CIRISAI/CIRISPersist", tag = "v49.0.0", features = ["scrub-ner"] } +ciris-persist = { git = "https://github.com/CIRISAI/CIRISPersist", tag = "v52.0.0", features = ["scrub-ner"] } serialport = "4" # libsqlite3-sys (pulled by persist's `sqlite` feature) links the SYSTEM libsqlite3 @@ -356,14 +356,14 @@ http-body-util = "0.1" # The PQC primitives the realtime-A/V E2E bench drives directly: hybrid KEX # (X25519 + ML-KEM-768) for the per-Link transit key, AES-256-GCM for the # two-layer chunk seal. Same pin/features edge composes (tag parity). -ciris-crypto = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v17.1.0", version = "17", features = ["hybrid-kex", "aes-gcm", "pqc-ml-dsa", "self-enc"] } +ciris-crypto = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v18.0.0", version = "18", features = ["hybrid-kex", "aes-gcm", "pqc-ml-dsa", "self-enc"] } # Enable edge's OWN fountain codec (`codec-fountain`, L1-A) for the survival-floor # proof — dev/test ONLY (feature unification applies to test/bench builds; the # shipped wheel stays codec-free, since a relay forwards sealed symbols opaquely # and never encodes/decodes). This lets tests/chaos_mesh.rs exercise the SUBSTRATE # codec (`fountain_encode`/`fountain_decode`) directly — a real MEASURED proof, # not a reference stand-in. -ciris-edge = { git = "https://github.com/CIRISAI/CIRISEdge", tag = "v32.1.0", features = ["codec-fountain"] } +ciris-edge = { git = "https://github.com/CIRISAI/CIRISEdge", tag = "v38.0.0", features = ["codec-fountain"] } # Enable persist's TEST-ONLY genesis-seam (`test-genesis-seam`, CIRISPersist#387) for the # accord ceremony tests (tests/accord.rs): `Engine::with_signer_pre_genesis` yields a clean # engine with the HUMANITY_ACCORD genesis seed SKIPPED, so those tests can stand up their OWN @@ -375,7 +375,7 @@ ciris-edge = { git = "https://github.com/CIRISAI/CIRISEdge", tag = "v32.1.0", # without one. Pre-genesis is now a supported boot mode, not a test seam, and an Engine built # that way reports `GenesisPosture::PreGenesis` while every ROOT_REQUIRING_GATES gate refuses # with `NoConstitutionalRootYet`. A node that trusts nothing is not a node that checks nothing. -ciris-persist = { git = "https://github.com/CIRISAI/CIRISPersist", tag = "v49.0.0" } +ciris-persist = { git = "https://github.com/CIRISAI/CIRISPersist", tag = "v52.0.0" } # The QA runner exercises the family/accord ceremony surfaces against a # CONTROLLABLE roster — holders it can sign as. persist v31.0.0 made roster growth diff --git a/FSD/FILE_CUSTODY.md b/FSD/FILE_CUSTODY.md new file mode 100644 index 00000000..63aecc88 --- /dev/null +++ b/FSD/FILE_CUSTODY.md @@ -0,0 +1,255 @@ +# FSD — Where each file is: the custody view (0.5.218) + +**Ask (maintainer, 2026-09-30):** "For every blob we have access to, there +needs to be a view of which devices it is on, the total devices." + +**This cut ships the VIEW.** Copying a file to a device and removing it from +one are a later cut; §5 is their design, and nothing in 0.5.218 moves a byte +because of it. + +Code: `src/file_custody.rs` (the sources and the per-device answer), +`src/drive.rs` (`file_custody`, the route; `read_drive`, the compact +summary). Tests: `tests/file_custody.rs`, the unit tests in +`src/file_custody.rs`, and the harness `custody` relation +(`harness/native/topology.py`, `FSD/TOPOLOGY.md` §2.5) asserted in +`harness/native/topologies/selffiles.yaml` and +`harness/native/topologies/csd-107-file-custody.yaml`. + +## 1. The surface + +### 1.1 `GET /v1/files/{attestation_id}/custody?cohort=self|family|community[&room_id=…]` + +Same auth and cohort handling as `GET /v1/files/{id}/meta`, and the same +doors as the BYTES, in the same order: the owner session +(`drive.owner_session_required`), the cohort named and membership-checked +(`drive.unknown_cohort`, `drive.family_id_required`, +`drive.community_id_required`, `drive.not_a_member`), the row found through +edge's gated reader (`drive.not_in_room`), a withdrawn row `410 +drive.withdrawn` — then persist's custody door, asked as the drive's viewer +key (the split-install content occurrence, `drive::viewer_key`), which runs +the byte read's tier gate. A viewer who cannot open the bytes gets the byte +read's refusal through the same `refuse_state` (`drive.not_granted`, +`drive.evicted`, …) and learns nothing about who else can open them. No new +refusal id. + +**Authorized by the ROW, not the bytes** (maintainer's ruling on #704: "no +copy here is a receipt (node responsive, no copy)"). A device that holds the +row but not the bytes is NOT refused `409 drive.not_fetched` (the byte read +still is): the view answers 200, its own entry is `holds: "none"` with +`checked_at`, `held_here: false`, `copies_known: 0`, and — because persist's +custody reads the blob's head row, which is not there — `access: null`, +`size_bytes: null`, `announced_holders: []`, every device's `can_open: null`, +with `custody.no_copy_here` in `why`. The tier and `copies_observable` come +from the row's pointer. + +```jsonc +{ + "attestation_id": "…", "cohort": "self", "room_id": "…", + "tier": "invisible_encrypted", // persist's tier token + "size_bytes": 25168000, // the stored (at-rest) length + "at_rest_sha256": "…", + "author_device": "", // the stream's producer (or the row's attester, inline) + "checked_at": "2026-09-30T…Z", // when THIS device answered + "this_device_is_author": true, + "devices_total": 2, // the person's devices (§2.1) + "devices": [ + { "node_key_id": "…", "label": "laptop", "this_device": true, + "can_open": true, "received": null, "holds": "here", + "checked_at": "2026-09-30T…Z", "reported_at": null }, + { "node_key_id": "…", "this_device": false, + "can_open": true, "received": { "epoch": 0, "k": 25, "at": null }, "holds": "received", + "checked_at": null, "reported_at": null } + ], + "held_here": true, + "copies_known": 1, // persist: held_here + announced holders elsewhere + "copies_observable": false, // false for self/family BY DESIGN (CC 5.2) + "announced_holders": [], // community / commons only: [{node_key_id, size_bytes}] + "access": [ { "person_key_id": "…", "devices": ["…"], "via": "at_rest_grant" } ], + "receipts_supported": true, // false for an inline file + "receipts_unsupported_reason": null, // "custody.inline_no_receipt" when false + "receipts_from_other_keys": [], // receipts no listed device answers to + "why": [ { "reason_id": "custody.…", "detail": "…" } ] +} +``` + +`holds` is one word per fact, like the drive's byte states: `here` (this +device, bytes held — persist's `held_here`), `received` (a delivery receipt +names the device), `none` (the device ANSWERED that it holds no copy — at this +pin only this device can, and its entry carries `checked_at`, the moment it +answered; a live statement, not an inference), `unknown` (nothing this node +can see says either way — NOT "absent"). A device's own `none` outranks a +receipt it once signed (an eviction does not retract a receipt, §4.3). +`reported_at` is on every entry and always `null` today: a remote device's +signed "no copy" arrives with persist's within-cohort custody +acknowledgements (CIRISConstitution#130), and will fill `holds: "none"` + +`reported_at` for other devices then. `can_open` is `null` when the access +list is not answerable on this device. + +### 1.2 `GET /v1/drive` — `custody: {devices_total, received_on}` per row + +Cheap by construction, so it ships: the device roster is read ONCE per page, +and a row costs one receipt-list query when it is a chunk DAG and nothing when +it is inline — no manifest read, no custody door. `received_on` counts the +person's devices this node holds a receipt from; it is `null` for an inline +file (unknowable, not zero) and when the stream log could not be read. +`custody` is `null` on a withdrawn row and when the roster could not be read. + +## 2. Sources + +### 2.1 The person's devices + +`nodes_owned_by(owner)` (persist admission: the owner-bindings, folding the +owner's `withdraws` — a RELEASED node is gone), minus every node whose content +occurrence for the owner is REVOKED (named by +`list_identity_occurrences_for`, absent from +`list_identity_occurrences_active`), minus the owner's own key. This is the +projection the self-room driver and the self re-wrap already key on, so "your +devices" is the set a self file is wrapped and replicated to. The occurrence +list behind `GET /v1/self/occurrences` is NOT the roster: it carries the +person's login anchor and, on an agent split, the ACTOR beside the node, so it +would count one machine twice; it is read only to subtract revocations. + +This device's row answers to every key this node is +(`peer::own_keys_of_this_node`: engine key, actor, held node signer, wire +identity), because a split install binds the NODE key while its engine signs +receipts as the actor. Labels are the owner's `self:device_label:v1` rows +(`POST /v1/self/occurrence/label`), matched on any of a device's keys. + +### 2.2 Custody — persist `Engine::blob_custody` (v51.1.0, CIRISPersist#942) + +Through edge's `FileRow::custody(store, viewer)`, the same +`PersistGroupContentStore` every drive read uses. Access is exact (at-rest +grant recipients for self/family, the epoch's member grants for community), +grouped per person; `held_here`; announced holders and `copies_known` for +community/commons. For self/family, copies elsewhere are unobservable by +design: those bytes are never announced (CC 5.2 structural invisibility), so +persist says `copies_observable: false` rather than a false "1 copy". +`can_open` per device = one of its keys is in the access list. + +### 2.3 Delivery receipts — edge `FileRow::received_by` (CIRISEdge#738, CC 5.3.3.6) + +`receipts::received_for` over the store's stream log: one `(node, epoch, K)` +per node that stored every chunk of a chunk-DAG file under the root the +author published. Proof of DELIVERY, never of consumption, and — at this pin — +never retracted by an eviction. + +### 2.4 The receipt host wiring — checked, nothing unset + +The "host hooks left unset" class (an optional edge hook the server never set +kept chat bodies sealed for six releases) was checked for all four places a +receipt is touched. None is a receipt-specific hook; each resolves to a store +the server already builds: + +| place | edge | server | +|---|---|---| +| publish | `files::publish` → `GroupContentStore::stream_log()` puts the STH (trait default `None` = no STH, never receiptable) | the drive's store is `PersistGroupContentStore` (`drive::store`), whose `stream_log` is `receipts::stream_log_of(engine)` — `Some` for SQLite and Postgres | +| receive | `blob_swarm::pull` → `receipts::on_dag_pulled(engine, backend, local_key_id, …)` after `promote` | `BlobPuller::spawn` takes the engine as a REQUIRED argument (`backend::spawn_puller_with`); `local_key_id` is the edge signer, the key the engine self-attests the receipt row as (admission requires the two to agree) | +| admit | the bridge's `apply` → `receipts::admit_and_count` when its `engine` is set | `SealedContentWiring::engine` in `compose::start_replication_runtime` (since 0.5.212; edge v27 makes it unconstructible without the pull sink) | +| read | `FileRow::received_by(store)` | `file_custody::receipts_of`, through the drive's store | + +## 3. What the view cannot say — every partial answer names its reason + +Each rides a 200 in `why[]` as `{reason_id, detail}`. String-literal ids, +one `msg` call each in `src/file_custody.rs`, listed as localization debt in +`tools/check_server_localization.py` (ratchet 139 → 151 across this cut, in +`tests/localization_gate.rs`) for the client bundle (CIRISClient#78): + +| id | when | English | +|---|---|---| +| `custody.inline_no_receipt` | the file is inline (≤ 1 MiB) | This file is small enough to be stored in one piece, and a file stored in one piece has no delivery receipt yet: your other devices may hold it, but this device cannot be told so. | +| `custody.copies_unobservable_by_design` | persist `copies_observable: false` (self/family) | Your own and your family's files are never announced to anyone, so copies on other devices cannot be counted; a delivery receipt is the only sign a device received one. | +| `custody.receipt_is_delivery_not_holding` | any receipt read | A delivery receipt proves a device received the whole file. It does not prove the device still holds it: removing a copy does not withdraw its receipt yet. | +| `custody.receipt_time_unknown` | a receipt's `at` is `null` | The time each device received the file is not available yet. | +| `custody.receipts_admitted_on_author_device` | this device did not write the file | Delivery receipts are collected by the device that wrote the file. This device lists only the receipts it holds; ask the device that wrote it for the full answer. | +| `custody.receipt_signer_not_your_device` | a receipt no listed device answers to | Some receipts were signed by devices that are not among your devices (another member of the room, or an agent's own key); they are listed separately. | +| `custody.receipts_unreadable` | the stream log read failed | This device could not read its delivery receipts just now; which devices received the file is unknown until it can. | +| `custody.commons_readable_by_holders` | plaintext (commons) tier | This file is public: anyone holding the bytes can read them, so who can open it is not a list. | +| `custody.no_copy_reports_pending` | another device reads `unknown` | Your other devices cannot yet report that they hold no copy; those reports arrive with within-cohort custody acknowledgements. Until then a device without a delivery receipt is shown as unknown. | +| `custody.no_copy_here` | this device holds the row, not the bytes | This device holds no copy of the file, so who can open it and how many copies are announced are answered by a device that holds it. | + +~~Also known, not a `why`: a device that holds the ROW but has not pulled the +bytes answers `409 drive.not_fetched`.~~ Superseded by the #704 ruling (§1.1): +that device answers 200 with its own `holds: "none"`; the author device is +still the complete answer for access and receipts. + +## 4. Gaps to file upstream (named here, not filed) + +1. **Inline files carry no receipt** — edge adopting persist v52's one-leaf + stream log over an inline blob (CIRISPersist#953); until then + `receipts_supported: false` for every file ≤ 1 MiB, the most common size. +2. **A receipt's time** — persist's `list_delivery_receipts_for` does not + return the `received_at` column it stores, so edge's `Received.at` is + always `None` (edge `receipts.rs`, `Received::at`). +3. **Eviction does not retract a receipt** — a device that evicts its copy + (engine `evict_blob` retracts its `holds_bytes` claim, CIRISEdge#669) still + shows `received`. Needs a receipt-retraction row (a `withdraws` of the + receipt row by its signer on eviction) and persist honouring it in + `list_delivery_receipts_for`. +4. **Self/family copies are uncountable** — within-cohort custody + acknowledgements (CIRISPersist#942 part 2, CIRISConstitution#130, persist + v52); receipts are the only signal until then. +5. **A remote agent-split device receipts as its ACTOR** — the receipt names + the engine key; the owner-binding names the node. This device folds its + own keys; a REMOTE split device's receipt lands in + `receipts_from_other_keys`. Needs a directory read from an actor key to + the node that hosts it (persist). +6. **Custody on a device that has not pulled** — `blob_custody` refuses + `NotHeld` without the blob head. Since the #704 ruling the view answers + anyway (`holds: "none"` for itself), but `access`, `size_bytes` and + `can_open` are `null` there; a head-less persist answer (access from the + row's grants) would fill them (persist). +8. **Remote "no copy" reports** — only THIS device can say `none` today; a + signed per-device "I hold no copy" from the person's other devices is + persist's within-cohort custody acknowledgement (CIRISConstitution#130, + CIRISPersist#942 part 2). The per-device shape is ready: `holds` ∈ + here | received | none | unknown, plus `reported_at` (`why`: + `custody.no_copy_reports_pending`). +7. **Receipts are admitted only where the STH was published** — a + non-author device holds only its own receipt (and any whose STH it + re-put while pulling). Replicating the author's admitted receipt set to + the person's other devices would make the view device-independent (edge). + +### 4.1 Closed at edge v38.0.0 / persist v52.0.0 (0.5.218) + +- **Gap 1 closed** — an inline file is a one-leaf stream + (`inline_blob_stream_id(sha)`, CIRISPersist#953 item 2); edge publishes its + STH and the puller emits its receipt (`receipts::on_file_pulled`, CIRISEdge#755). + `receipts_supported` is now `true` for every file and + `receipts_unsupported_reason` always `null` (both kept on the wire, constant); + a drive row's `received_on` is a count for inline files too (`0`, not `null`, + before any device receipts it). `custody.inline_no_receipt` is no longer + emitted and was deleted from `src/file_custody.rs`. +- **Gap 2 closed** — `list_stored_delivery_receipts_for` returns each receipt's + `received_at` (#953 item 3); edge's `Received::at` is a `DateTime`, and the + view's `received.at` / `receipts_from_other_keys[].at` are always an RFC 3339 + string. `custody.receipt_time_unknown` is no longer emitted and was deleted. +- The §3 table above keeps both rows as the record of what 0.5.218's first cut + said; the localization ratchet drops by the two ids. + +## 5. Later cut: copy-to and remove-from a device (design, NOT built) + +- **The request is an owner-signed, device-addressed row.** `POST + /v1/files/{id}/custody/{node}` `{want: "keep" | "drop"}` authors a + `self`-scoped row (`file_custody_request:v1`, a persist registry row — the + dimension must be registered with persist before any node emits it) + signed by the OWNER's pen (consent is authored by the human), naming the + file row, the blob's at-rest sha, the target device and the ask. It rides + the self room to the owner's devices like any self row. +- **`keep` = pull now.** The target device's puller treats a live `keep` + naming itself as an operator-consent `Announce` for that blob even when + its default disposition would defer, and pulls it; the receipt it emits + closes the loop in this view. +- **`drop` = a sticky "don't keep here".** The target evicts through the + ENGINE (`evict_blob`, so its holder claim is retracted first) and the + puller HONOURS the live `drop` on every later offer — without it the next + anti-entropy round would pull the bytes straight back. A later `keep` + supersedes it. +- **The last-copy guard.** A `drop` is REFUSED (`custody.last_copy`) when it + would remove the only copy of an unwithdrawn file: the guard counts this + view's `here` + `received` devices, minus the target; with receipts + unretracted on eviction (§4.3) the count can over-report, so the guard + must wait on gap 3 (or on persist v52's custody acknowledgements) before it + can be trusted to say "another copy exists". Withdrawing the file is the + way to remove the last copy. +- Needs: the persist registry row for the request dimension, gap 3 for an + honest guard, and an edge puller hook to consult the sticky request. diff --git a/FSD/MEMBERSHIP_INVITES.md b/FSD/MEMBERSHIP_INVITES.md new file mode 100644 index 00000000..254e7380 --- /dev/null +++ b/FSD/MEMBERSHIP_INVITES.md @@ -0,0 +1,210 @@ +# MEMBERSHIP INVITES — nobody joins without their own consent + +**Status:** design, 2026-09-30. **Ruling:** the maintainer, 2026-09-30: *adding someone +to a family or community requires their consent*; *reverse quorum is not a membership +rule*. **Normative text:** CIRISConstitution#133. **Substrate enforcement:** +CIRISPersist#955. **Client screens:** CSD-100..103 (to gain an invite inbox). + +## 1. What is wrong today + +Every roster-growing door adds a member on the EXISTING members' authority alone: + +| Door | Today's gate | Asks the new member? | +|---|---|---| +| `POST /v1/families/{id}/members` | `check_addable`: a registered key, not already active | no | +| `POST /v1/communities/{id}/members` | `require_contact`: THIS node's live grant toward them covers `chat:` | no — that is our consent toward them, not theirs | +| `POST …/changes/{envelope,cosign,assemble}` (quorum) | M of N existing members | no | + +CC 4.4.3.2.3's admit predicate checks only the current members' signatures under +`consensus_protocol` (geographic alone reads a newcomer-signed row, its +`location_proof`). So a founder can enrol any key they can name, and the +enrolled person's node begins receiving the group's rows and wraps. + +## 2. The flow + +``` +inviter(s) invitee K group +───────── ───────── ───── +propose(K, role) ──────────────▶ inbox: pending proposal + (ONE inviter signs: the founder + under founder_only, any member + otherwise) + accept(proposal) — K's pen + │ + growth record (supersede / widening) under the ◀──┘ + group's consensus_protocol, citing the acceptance: + founder_only = 1 signature; quorum:M/N = envelope + → cosign → assemble, as today ──────────────────────────────────▶ member admitted + decline(proposal) — K's pen ────▶ proposal closed + (no act before expiry) ──────────▶ proposal expired +``` + +- **A proposal is not membership.** It grants nothing: no rows, no wraps, no + room address. It names the group, K (in `subject_key_ids`; AV-84 keeps a + targeted row's `attested_key_id` its producer), the offered role, the one + inviter and an `expires_at` (bounded at 30 days), and persist serves it to K + through a narrow read arm so K's node can see it. +- **The quorum stays on the growth record, not the invitation** (persist's + #955 design): one inviter proposes; the supersede / widening that admits K + carries the protocol's M-of-N exactly as today. One quorum check, not two + that could drift. So under a quorum protocol K may accept and still not be + admitted if the quorum never assembles. +- **Expiry is judged on signed instants** — the acceptance's and the growth + record's `asserted_at` against the proposal's `expires_at` — never on a + receiver's clock. +- **Acceptance is K's own act,** signed by K's person key through the server-side pen + (the same authority `release_node` uses — a session bearer suffices). It binds the + proposal (its attestation id / content hash) and the role; accepting a different role + than offered is a new proposal, not an acceptance. +- **Only then** is the widening (community) or roster-growing supersede (family) + written, citing the acceptance. The re-key of existing content to K + (`rekey_family_member_add`) runs after admission, never before. +- **Decline and expiry are terminal.** A declined or expired proposal can never be + admitted; inviting again is a new proposal. +- **Every protocol,** founder_only included: no quorum stands in for the joiner. +- **Leave is unchanged:** the member's own forward-only `withdraws`, no quorum. + +## 3. Routes (server) + +| Route | Who | Does | +|---|---|---| +| `POST /v1/families/{id}/invites`, `POST /v1/communities/{id}/invites` | the founder (founder_only) or any member | writes the proposal, signed by that one inviter | +| `GET /v1/families/{id}/invites`, `GET /v1/communities/{id}/invites` | members | pending / accepted / declined / expired, per invitee | +| `DELETE …/invites/{proposal_id}` | the proposer(s) | withdraws a pending proposal | +| `GET /v1/self/invites` | the invitee | the inbox: every proposal addressed to me, across families and communities | +| `POST /v1/self/invites/{proposal_id}/accept` | the invitee | signs the acceptance; the widening follows | +| `POST /v1/self/invites/{proposal_id}/decline` | the invitee | signs the decline | + +`POST …/members` stops adding directly: it becomes an alias for `…/invites` that +answers `202 {state: "invited"}` and names the proposal, so no caller mistakes an +invitation for a membership. + +## 4. Where the rule lives + +The ADMISSION rule is persist's (every host would otherwise write it; one rule, +one implementation): CIRISPersist#955 asks for +the proposal and acceptance rows and a gate on both the local put and the +replication apply. It is a persist MAJOR (v52). + +**Interim (the maintainer's choice, 2026-09-30): refuse, don't hold.** Until v52 +the server has no way to deliver a proposal to a non-member or to record an +acceptance (the dimensions are not in the strict registry). So every +roster-growing door refuses with 409 `membership.consent_required`: a direct +add, a quorum envelope that adds, and a founding roster naming anyone besides +the founder. Creating a group, removing members, changing roles, leaving and +dissolving are unaffected. This keeps no server-side copy of the rule, only a +closed door. + +**Founding members too** (same ruling): the founding record admits only the +founder; everyone else named in it joins by proposal → acceptance. + +## 5. Not in scope + +- **Reverse quorum** stays the commons objection brake (`/v1/commons/*`, + CSD-070); `reverse_quorum:` is refused as a family or community protocol. +- **Subject take-back** (removing oneself from rows already shared) stays deferred. +- **Pair rooms** (1:1 chat) keep their own consent: the contact grant each side authors. + +## 6. Persist's contract, as the maintainer resolved it (CIRISPersist#955, 2026-09-30) + +- **Signing the founding record is consent.** A founding member is admitted iff + they signed it, as the authority or as a cosigner. A listed but unsigned member + is refused as `membership_founding_member_unsigned`. `POST /v1/families` and + `POST /v1/communities` sign with the founder alone, and the server has no + founding-cosign flow, so the interim 409 on a founding roster beyond the + founder stays correct. If the server gains a founding-cosign flow, co-signing + founders are admitted. +- **A supersede never adds members.** Every addition is a widening; a quorum + add is a widening carrying the M-of-N. A roster-growing supersede is refused + as `membership_supersede_cannot_add`. **At v52 adoption the family quorum + add (envelope → cosign → assemble, today `supersede_family_with_quorum` + with a grown roster) must become a quorum-carrying widening.** It is + refused by our 409 until then. +- **Leave and dissolve replicate** (CIRISPersist#956, in v52): leave is a roster + supersede removing only its own signer, admitted on that one signature; + dissolve is a quorum-verified terminal amendment. The ignored test + `a_quorum_dissolve_replicates_as_an_amendment` (CIRISServer#700) is un-ignored + at v52. + +## 7. As built — persist v52.0.0 / edge v38.0.0 (0.5.218) + +The interim 409 is gone; the flow of §2 is live. What shipped, and where it +differs from the design above: + +### 7.1 Routes + +| Route | Who | Answers | +|---|---|---| +| `POST /v1/{families,communities}/{id}/invites` `{key_id, role?, expires_in_days?}` | a founder under `founder_only`; any active member otherwise | 202 `{state: "invited", proposal_id, group_kind, group_id, invitee_key_id, role, expires_at}` | +| `POST /v1/{families,communities}/{id}/members` | same | the alias above, same 202 | +| `GET /v1/{families,communities}/{id}/invites` | members (a delegate may read) | `{invites: [{proposal_id, invitee_key_id, role, proposer_key_id, proposed_at, expires_at, state, reply_id}], seated_now: [...]}` — `state` ∈ `pending` / `accepted` / `joined` / `declined` / `expired` / `withdrawn` | +| `DELETE /v1/{families,communities}/{id}/invites/{proposal_id}` | the proposer | 200 `{state: "withdrawn", withdrawal_id}` (a `withdraws` of the proposal) | +| `GET /v1/self/invites` | the invitee (a delegate may read) | every live, unanswered, unwithdrawn proposal naming them (edge `membership::pending_proposals_for`), families, rooms and pair rooms | +| `POST /v1/self/invites/{proposal_id}/accept` / `…/decline` | the invitee's own session | 200 `{state: "accepted" \| "declined", reply_id, awaiting}` | + +Rows are edge's (`membership::{propose, reply, widen_on_acceptance}`, built on +persist's own builders); the server keeps no copy of the admission rule. +`expires_in_days` is 1..=30 (default 14; persist bounds a proposal at 30 days). +A contact grant is NOT required to invite: an invitation reaches a stranger's +nodes under first contact (CIRISEdge#756). The invitee must be a registered key +for a family (the widening names it). + +### 7.2 Who seats the member + +- **`founder_only`** — the founder's single-signature widening. Written by + edge's replication bridge when the acceptance arrives + (`ReplicationRuntimeConfig::membership_widener`, set in `compose` to the + owner's PERSON pen, because the roster counts seat keys), or by the server + when a founder lists `…/invites` (covers a node claimed after its runtime + started, which has no widener until restart). A household's existing + content is re-wrapped to a member the LIST seats (`rekey_family_member_add`); + one the bridge seats is not re-wrapped by the server — a gap (§7.5). +- **Any other protocol** — "accepted, awaiting the group" (persist FSD §4): the + `add` of `…/changes/{envelope,cosign,assemble}`, now written as a co-signed + WIDENING (persist Q2: a supersede never adds). Without the joiner's + acceptance persist refuses it at assemble, `membership.awaiting_acceptance`, + however many members signed. A family's record never grows; its stored + `quorum:M/N` is not rescaled by an add. + +### 7.3 Founding, leave, dissolve + +- A create naming anyone but the founder is `membership.founding_member_unsigned` + (409) — the server has no founding-cosign flow (persist Q1). +- **Quorum family leave** (CIRISPersist#956): `supersede_family_with_quorum` + with the record minus the leaver, the envelope from + `build_membership_change_envelope(remaining)`, signed by the leaver alone; the + protocol is NOT rescaled ("nothing else may change"). It replicates. A member + seated by a widening is not on the record; their revocation alone is the leave. +- **Quorum family dissolve** (#956): the change envelope pins `dissolved_at`; + assemble writes the terminal amendment (every seat unchanged, `dissolved_at` + set) through `supersede_family_with_quorum`. No removal rows. It replicates + (`a_quorum_dissolve_replicates_as_an_amendment`, un-ignored). A + `founder_only` dissolve is unchanged (revocations + the local supersede). + +### 7.4 Pair rooms + +`POST /v1/chat` no longer writes a two-founder record (persist refuses it): +the opener founds alone and proposes the peer as `founder` +(`chat::open_pair_room`); the peer's own `POST /v1/chat` accepts +(`chat::accept_pair_proposal`); the opener's node widens (bridge, or the +opener's next call with their pen). The answer gains `state` (`open` / +`invited` / `accepted` / `awaiting_invitation`) and `proposal_id`. A pair +room's invitation also appears in `GET /v1/self/invites`. + +### 7.5 Reason ids (all 409 unless noted) + +persist's rules, one each: `membership.awaiting_acceptance`, +`membership.invite_not_here_yet` (both retryable), `membership.declined`, +`membership.invite_expired` (410), `membership.acceptance_mismatch` (403), +`membership.already_answered`, `membership.founding_member_unsigned`, +`membership.supersede_cannot_add`; and `membership.refused` (any other refusal +of a membership row). The flow's own: `membership.invite_not_found` (404), +`membership.not_the_invitee` (403), `membership.not_the_proposer` (403), +`membership.invite_closed`, `membership.bad_expiry` (400), +`membership.owner_session_required` (401/403), +`membership.delegate_may_not_answer` (403), `membership.signer_unavailable` +(403), `membership.store_unavailable` (503). Queued for the client bundle +(`KNOWN_UNLOCALIZED`, CIRISClient#78). + +Gaps named, not built: the bridge-seated member's content re-wrap (above); a +founding-cosign flow; the CSD for the inbox screens (CSD-100..103). diff --git a/FSD/ROSTER_AND_DRIVE_CRUD.md b/FSD/ROSTER_AND_DRIVE_CRUD.md index 1ec7c310..fe3a321d 100644 --- a/FSD/ROSTER_AND_DRIVE_CRUD.md +++ b/FSD/ROSTER_AND_DRIVE_CRUD.md @@ -51,6 +51,14 @@ policy, its refusals, its witness. 7. **Every write kicks replication** (`compose::kick_replication`) so the change crosses on a round-trip. 8. **Every list route carries the row's envelope** (subject, attester, cohort_scope, dimension, consent:scope — CSD-006, #616) and a `resume` cursor. +9. **Nobody joins without their own consent** (0.5.218; the maintainer's ruling of 2026-09-30, + CIRISConstitution#133, `FSD/MEMBERSHIP_INVITES.md`). Until persist can carry a proposal to the joiner + and record their acceptance (CIRISPersist#955, v52), every roster-growing door answers 409 + `membership.consent_required`: `POST …/members` on a family or a room, a quorum envelope / cosign / + assemble whose proposed roster adds anyone, and a create whose founding roster names anyone but the + founder. Remove, role, leave and dissolve are unaffected; re-adding someone already active keeps + `*.already_member`; pair rooms keep the contact grant each side authors. This supersedes the "Add + member" rows of §3 and §4 below, which record what 0.5.216–0.5.217 shipped. ### 1.1 Which planes cross (known gap closed in 0.5.218, CIRISServer#646) @@ -71,6 +79,17 @@ so edge refuses every delivered row terminally). A gate from `EnvelopeKind::ALL` minus that exclusion list, so a kind appended upstream goes red until someone decides. +**Correction (0.5.218, the device-eviction audit).** Routing `IdentityOccurrenceRevocation` did not by +itself make "a released device" cross. No server path wrote a revocation the plane could carry: `release` +withdrew the owner-binding and revoked NO occurrence, and `POST /v1/self/occurrence/revoke` wrote through +persist's trusted-LOCAL door (`put_identity_occurrence_revocation_local`), whose rows are unsigned and +EXCLUDED from the signed replication read by construction. So the plane was routed and empty: every other +device of the owner kept wrapping new self files to the evicted one. The gap closed only when both routes +became one signed act, `self_devices::evict_device` (§2.2): the owner-signed `withdraws` of the +owner-binding(s) AND a revocation through persist's SIGNED door (`put_identity_occurrence_revocation`, +signed by the owner's pen), then a replication kick. `tests/occurrence.rs` gates that the local door has +no caller in `src/`. + **Load.** Twelve kinds is twelve coordinators per peer on edge's single 30 s scheduler cadence. Edge v31 has no per-kind cadence and no kick-only coordinator (`SchedulerConfig::cadence` is global; mesh-config relief lengthens every kind at once), so the six rarely written planes cost one round each per tick even @@ -85,7 +104,7 @@ upstream is a per-kind cadence or kick-only rounds for rarely written kinds. | List owned nodes | `GET /v1/setup/owned-nodes` (exists, loopback) | loopback | unchanged | | List device keys | `GET /v1/self/occurrences` (exists) | public binding metadata | add `revoked: bool` + `include_revoked=true` query | | Add a device key | `POST /v1/self/occurrence` (exists) | signed by the primary | unchanged | -| Revoke a device key | `POST /v1/self/occurrence/revoke` (exists) | signed by a SURVIVING occurrence | unchanged | +| Revoke a device key | `POST /v1/self/occurrence/revoke` (exists) | signed by a SURVIVING occurrence | unchanged — **0.5.218: now the owner's session, signed server-side by the owner's pen, one act with release (§2.2)** | | Relabel a device | `POST /v1/self/occurrence/label` `{occurrence_key_id, label}` **new** | owner session | a `supersedes` of the occurrence row with the new label; label is display-only | | **Release a node** | `POST /v1/self/nodes/{node_key_id}/release` **new** | owner session; the owner of `node_key_id` must be the caller | withdraws the owner-binding (`delegates_to(user → node)`) with a signed `withdraws`; the node drops out of `nodes_owned_by`, the self room removes it on the next drive tick, the node reverts to Clause D fail-closed. Refuses the node you are talking to unless `force_self: true` | | Self room bytes on a 2nd device | background driver | node-signed handshake | adopt `key_package_attestation_in` / `welcome_attestation_in` / `welcome_for` (#656); selffiles `opened_on_b` becomes REQUIRED | @@ -123,6 +142,40 @@ New ids: `self.not_your_node`, `self.release_self_requires_force`, `self.label_e `self.author_signer_unavailable`, `self.bad_request`, `self.not_your_device`, `self.release_incomplete`. +### 2.2 Evicting a device is ONE act (0.5.218, CSD-037 — the stolen-device path) + +Until 0.5.218 two unconnected routes each did half, and neither half was enough. `release` withdrew the +owner-binding (the self room drops the node on its next tick) but left the node's content-KEM identity +occurrence live, and persist wraps every new self file to `list_identity_occurrences_active(owner)`, so +the released machine kept receiving the key to each new file. `occurrence/revoke` wanted a request +hybrid-signed by a surviving device key — the app sends a bearer, so every call answered 401 — and wrote +through the trusted-local door (unsigned, never replicated, §1.1) without touching the owner-binding. + +Both routes now converge on `self_devices::evict_device`: + +1. withdraw every live owner-binding the owner holds on the device's node(s) (owner-signed `withdraws`); +2. revoke the device's occurrence(s) through persist's SIGNED `put_identity_occurrence_revocation`, + signed by the owner's pen (the identity the occurrence belongs to), so the + `IdentityOccurrenceRevocation` plane carries it; +3. kick replication; 4. read both halves back from persist (`nodes_owned_by`, + `list_identity_occurrences_active`). + +`release` revokes the node's occurrence(s) of the owner (its node key; for THIS node every key it is); +`occurrence/revoke` withdraws the owner-binding when the occurrence IS one of the owner's nodes. Both are +authorised by the owner's session (`self.owner_session_required` 401/403, `self.delegate_may_not_author`), +refuse another person's key as `self.not_your_node` / `self.not_your_device`, and need `force_self` for the +node being talked to. The answer names every part done and every part that failed, by part and target +(`self.evict_incomplete` 500 when any part other than the binding witness failed; `self.release_incomplete` +keeps its meaning). + +**What eviction does not do, stated in every answer (`history`) — CSD-037, CC 3.3.6.1:** already-shared +history stays readable by the evicted device. No DEK is rotated and nothing is re-encrypted; only NEW +content (new self files wrap to active occurrences only) and NEW self-room epochs are withheld. + +What `release` cannot find: another machine's ACTOR-key occurrence when that machine is a pre-fix split +install (nothing in the directory links a remote node key to its actor key). Revoking that occurrence by +its own key reaches it. + ## 3. Family (household) A household is a `Family` whose `family_key_id` is a keyless group identifier (V151), founded by the @@ -254,7 +307,7 @@ rewritten to grow. Pair rooms (`POST /v1/chat`) stay as they are and are listed | Create | `POST /v1/communities` `{name, members?: [key_id], tier?: community\|affiliations, consensus_protocol?}` | owner session; caller is founder; each initial member must be a contact whose grant covers `chat:` (same rule as pair rooms) | | List mine | `GET /v1/communities` (pair rooms included, `kind: pair\|room`) | owner session; fold | | Read | `GET /v1/communities/{id}` → record + effective roster + roles + appointed moderators | member only; `community.not_found` for non-members | -| Add member (widen) | `POST /v1/communities/{id}/members` `{key_id, role?}` | protocol satisfied (founder, or an appointed roster-duty holder via `delegates_to`); target must be a contact; **blocked on CIRISPersist#907** for the added member's reads, gated by a RED-EXPECTED rung until it lands | +| Add member (widen) | `POST /v1/communities/{id}/members` `{key_id, role?}` | protocol satisfied (founder, or an appointed roster-duty holder via `delegates_to`); target must be a contact; **blocked on CIRISPersist#907** for the added member's reads, gated by a RED-EXPECTED rung until it lands. **0.5.218:** #907 landed in persist v49 (the admission folds the widening plane; `tests/community_crud.rs` runs the read un-ignored) — and the route itself is CLOSED, 409 `membership.consent_required`, until the joiner can consent (§1 rule 9, CIRISPersist#955) | | Remove member | `DELETE /v1/communities/{id}/members/{key_id}` | protocol satisfied | | Leave | `POST /v1/communities/{id}/leave` | self only; last founder rule as for families | | Change role / appoint moderator | `POST /v1/communities/{id}/members/{key_id}/role` `{role}`; moderators keep using duty conferral | protocol satisfied | @@ -272,6 +325,8 @@ New ids: `community.not_found`, `community.not_authorized`, `community.not_a_con **Security note.** Persist's replicated widening/revocation doors verify the signature, not the signer's standing in the room (CIRISPersist#908). The server's routes enforce the protocol for rows it authors; rows a peer authors are admitted by persist alone until #908 lands. Stated in the release notes. +**Corrected 0.5.218:** #908 landed in persist v49.0.0 — every replicated roster row is judged at persist's +door by the room's own `consensus_protocol` over the row's co-signatures (`src/roster_rows.rs`). ### 4.1 As built (`src/communities.rs`, 0.5.216) — decisions and gaps at these pins @@ -405,6 +460,11 @@ folded into "no session"), `drive.bad_body` (400, an unparseable JSON or multipa split). The owner's `self` gate admits them only because `move_owner_binding_to_node_key` adds the node-key binding without withdrawing the actor's; a change that retires the actor binding would hide a split node's files from its own drive. +9. **Where each file is (0.5.218).** `GET /v1/files/{id}/custody` answers which of the person's + devices hold a file, out of how many (persist's custody view + edge's delivery receipts), and + each `GET /v1/drive` row carries `custody: {devices_total, received_on}`. Point 7's + `devices_holding` stays as it was. Sources, gaps and the later copy-to / remove-from design: + `FSD/FILE_CUSTODY.md`. ## 6. Witnesses @@ -414,7 +474,9 @@ folded into "no session"), `drive.bad_body` (400, an unparseable JSON or multipa - A source-scraping gate: no raw `.members` read for a membership decision in `src/`. - Ladders: selffiles `opened_on_b` REQUIRED; chat ladder gains `family`, `family_file_on_b`, `room3` (a three-member room), `widened_reads` (RED-EXPECTED until CIRISPersist#907), `withdrawn` (a withdrawn - file reads 410 on the other node). + file reads 410 on the other node). **0.5.218:** #907 landed in persist v49 — the in-process + `a_widened_member_reads_the_rooms_messages_cirispersist_907` runs un-ignored, its member widened by a + test-only fixture because the add route is closed on consent (§1 rule 9). - openapi.json lists every route here; the localization guard covers every new id. - 0.5.218: `selffiles` runs in the mesh-harness CI matrix (#622) and its `file` / `opened_on_b` rungs fail by name on a self pull regression (#626: `NoHolders`, `NoMeaning(GroupWithoutId)`, a @@ -423,7 +485,12 @@ folded into "no session"), `drive.bad_body` (400, an unparseable JSON or multipa dispatch-only `devices` scenario (the chat ladder plus a second device and a household) adds `second_device`, `c_peered`, `c_lists_room` (its success stage), `c_opens_history` (RED-EXPECTED: no content-key rewrap to a new occurrence of an existing member) and the #647 family rungs (RED-EXPECTED - on CIRISPersist#910). + on CIRISPersist#910). **Corrected 0.5.218:** persist v49 fixed #910 (and #907, #908); the `family` rung + now asserts that the add of B is REFUSED `membership.consent_required`, and `family_on_b` / + `family_file_listed_on_b` / `family_file_opened_on_b` are RED-EXPECTED on consent to join + (CIRISPersist#955) — no door may put B on the roster until B accepts. The in-process suites build their + multi-member rosters through clearly named `test_only_*` fixtures that write the rows through persist, + never through a production bypass. ## 7. Deliberately later diff --git a/FSD/SESSION_CLAIMS.md b/FSD/SESSION_CLAIMS.md new file mode 100644 index 00000000..4aaacda2 --- /dev/null +++ b/FSD/SESSION_CLAIMS.md @@ -0,0 +1,231 @@ +# Session claims — one device handles each exchange (0.5.218) + +**Status:** implemented in 0.5.218 (the maintainer's ruling of 2026-09-30). +**Normative source:** CC 3.1.3.1 (`session:*`), CC 2.1 (`community_id`, +`session_id`, `claimed_at`), CC 2.6.2 (the instant's canonical form). +**Substrate:** persist v51.3.0 `federation::session_claim` (CIRISPersist#782), +`check_session_self_report_admission` (CIRISPersist#814 part 5). +**Code:** `src/session_claims.rs`; the gated sites below; the gate test +`tests/every_act_passes_the_session_gate.rs`; the witness +`tests/one_device_handles_each_exchange.rs`. + +## 1. The problem, and the split + +A person is one fed-ID plus the N nodes they own — the *occurrences* of their +self. A row addressed to the person reaches EVERY one of those nodes (a fed-ID +has no transport path of its own), so every autonomous reaction this server has +ran on every device the person owns. Two devices committing the same MLS add +at one epoch fork the self room; two devices re-wrapping one blob write two +key-grant sets for one grant. + +Persist has carried the routing table since v38.7.0 and the server never used +it (zero hits for `session_claim`, `session:claim`, `handler_for` before this +cut). The split is CC 3.1.3.1's: + +| persist owns | the server owns | +|---|---| +| the row (`session:claim:v1`), its projection (`SelfOwn` at `self`, `Cohort` at every commons tier) | **attendance** — "the person is on this device" | +| admission: a SELF-REPORT (attester == attested == the claiming occurrence); a third party is refused at the door | when to claim, renew and let go | +| the merge: earliest `claimed_at`, ties on the lowest occurrence key id | the inventory of ACT sites, and the gate at each | +| the read: `handler_for(directory, owner, community, session, now, ttl)` | the surface `GET /v1/self/sessions` | + +The server never re-implements the merge. Every "who handles this?" is +`handler_for`; the surface enumerates which exchanges have claim rows but asks +persist who holds each one. + +## 2. The inventory + +Every place this server ACTS on something addressed to its person, and the +things that look like one but are not. **Gated** rows call +`session_claims::gate` before the act and act only on `Verdict::Act`; +`tests/every_act_passes_the_session_gate.rs` reads the gated rows of THIS table +(`file`, `fn`, `act`) and fails if the function's body does not call the gate +before the act. + +### 2.1 Gated — exactly one device acts + +| act | file | fn | act call | gated | community_id | session_id | +|---|---|---|---|---|---|---| +| self room: commit an ADD (and the add half of a Rejoin) | `src/self_room_drive.rs` | `drive_once` | `add_members(` | **gated** | the self room's `content_group_id` (`ciris_edge::self_room::room(owner)`) | `self_room:membership` | +| self room: commit a REMOVE (and the remove half of a Rejoin) | `src/self_room_drive.rs` | `drive_once` | `remove_members(` | **gated** | the self room's `content_group_id` | `self_room:membership` | +| re-wrap old self files for a new device | `src/self_rewrap.rs` | `rewrap_for_new_devices_with` | `rekey_self_occurrence_add(` | **gated** | the self room's `content_group_id` | `self_rewrap:` | + +Why these session keys: + +- **One membership session for the whole self room**, not one per joiner: an + MLS group's commits must come from one committer at a time — two devices + adding two DIFFERENT joiners at the same epoch fork the room exactly as two + adding the same one do. CC 3.1.3.1 lists "a moderation duty" as an exchange; + this is that shape. Only a device that HOLDS the group reaches these arms, so + the joiner (which cannot add itself) never claims the duty. +- **One re-wrap session per new occurrence**: the capability is per occurrence. + The new device holds none of the old DEKs and can never re-wrap for itself, + so it must never be the one the fold names; per occurrence it never is (it is + never pending for itself). The gate runs AFTER the pen check: a device that + cannot do the work does not take the duty. + +Idempotence per act (CC 3.1.3.1: "views transiently disagree"): the re-wrap +records its act id (`rewrap \0 owner \0 occurrence \0 x25519`) in the device's +`Attendance` ledger only on success, and persist's door skips a blob already +granted; the self room's `decide` re-reads the tree after applying the room's +commits, so a joiner already added is never re-added. + +### 2.2 Not gated — every device does its own + +| act | where | why every device does it | +|---|---|---| +| self room: `Create` | `self_room_drive::create_room` | edge's `decide` already confines creation to the LOWEST node key in the roster, and `Abandon` settles an unsettled directory's two rooms. It is the room's genesis by key order, not an answer to anything addressed to the person; gating it would leave a person with no room until they were on the lowest-keyed device. | +| self room: publish KeyPackage, join on a Welcome, apply remote commits, install/advance addresses, seal, `Abandon` | `self_room_drive` | this device's OWN membership steps. | +| provision this device's content-KEM occurrence | `backend::provision_engine_occurrence` | this device's own occurrence. | +| wrap self DEKs to an occurrence bound ON this device (`auth::occurrence::bind_occurrence_core` → `rekey_self_occurrence_add`) | `POST /v1/self/occurrence`, the portable-occurrence doors, `node_key::register_actor_occurrence` at boot | the device doing its OWN binding, never a reaction to a sibling's row. The gate test pins these as the door's only other caller. | +| consent healer (`peer::ensure_contact_consent_covers`) | `replication_reconcile` | consent is per node: each device's grant names itself. | +| replication kicks, publish-own refresh, peer convergence | `compose`, `replication_reconcile` | transport of rows; fan-out must stay fan-out (persist's module doc: "election does not belong on the transport plane"). | +| storing rows, pulling bytes (`receive_axis::pull_owner_testimony`), showing a chat message, listing notes/files | everywhere | every device SHOWS the person's content; nothing here answers anything. | +| heal the owner's key record, accept the roots as owner (boot) | `node_key::heal_owner_key_record`, `accept_roots_as_owner` | the node's own boot maintenance of ONE convergent row, idempotent (a second device finds it bound/accepted and does nothing). Gating it on attendance would leave the production canonical — headless, never attended — unhealed forever. | + +### 2.3 Not gated — the person asked on this device + +These are HTTP handlers. The request came to ONE device; there is no fan-out, +and the request itself is the person's attendance (it passes +`resolve_bearer`, which marks it). + +| act | where | community_id / session_id it would take if it ever became fan-out | +|---|---|---| +| claim-remote, upgrade-owner, announce | `claim_remote.rs` | — | +| second-device approval: `approve`, `claim`, `token` | `auth/device_grant.rs` | pending grants live IN MEMORY on the node that issued the code (never replicate): one handler by construction. If they ever replicate: (self room, `device_grant:`). | +| release / announce / label a device | `self_devices.rs` | — | +| households, communities, drive, notes, contacts add/withdraw | `family_api.rs`, `communities.rs`, `drive.rs`, `contacts_chat.rs` | — | + +### 2.4 Named, not gated in this cut + +| act | where | why not yet, and the key it will take | +|---|---|---| +| a chat room's MLS handshake on the CREATOR side (create, add, Welcome, `reconcile_room_group`) | `contacts_chat::room_key`, `room_key_room`, `reconcile_room_group` | runs INSIDE the person's own request, but reacts to KeyPackages that replicated to every device, and the MLS member is the PERSON's key while the MLS store is per node — so the person opening the same chat on two devices builds two groups. Gating it needs the deferring device to JOIN the handler's group, which edge's chat path does not do for a person's second device today. Key when it lands: (`pair_community_key_id(me, peer)` or the community id, `chat:handshake`). | +| an agent's reply to an inbound message | — | no agent auto-reply exists in `src/` (no `auto_reply` anywhere). When a hosted agent replies for the person it passes the gate with (the room's community id, the inbound message's attestation id). | +| accepting a contact, household or community invitation | — | nothing accepts automatically: the invitee's signed acceptance is the person's act on one device (FSD/MEMBERSHIP_INVITES.md). | + +## 3. Attendance + +A device is **attended** while the person's own session has touched it within +`PRESENCE_IDLE` (10 min). The one place every such request resolves is +`auth::session::resolve_bearer`; on a verified SystemAdmin (owner) session it +calls `Attendance::global().note_presence()`. A delegated `dgrant:` session +returns before that line — a helper acting for the person is not the person. +A device that merely BOOTED is not attended: a claim derived from boot would +hand the person's exchanges to whichever device came up first, including the +headless ones they never look at. + +## 4. Claim, renew, lapse + +| const | value | why | +|---|---|---| +| `SESSION_CLAIM_TTL` | 120 s | the horizon every device applies (one binary, one constant — until the horizon is in the row). A 60 s renewal on a 30 s loop always leaves a full loop period of slack; when the person moves device, the other device takes the exchange within two minutes. | +| `SESSION_CLAIM_RENEW_EVERY` | 30 s | the loop period (the node's common period; `loop_cadence`'s separation argument needs multiples of 30 s). Loop name `session_claims`, its own slot. | +| `SESSION_CLAIM_RENEW_AFTER` | 60 s | a holder writes its successor lease at half the TTL: one row a minute per held exchange rather than one per tick. | +| `PRESENCE_IDLE` | 10 min | an open client polls far more often; long enough to span the second-device flow (approve on the first device → the second boots, provisions, publishes its KeyPackage → the first must still be attended to commit the add). | + +The decision is pure (`session_claims::step`): + +- not attended → **Lapse** (write nothing; the claim goes stale); +- nobody holds it → **Claim**; +- another occurrence holds a live claim → **Defer** (never contested); +- we hold it → **Renew** when our newest lease is ≥ 60 s old (or unknown after + a restart), else **Hold**. + +**Claim on demand.** A site with work calls `gate`; if nobody holds the +exchange and this device is attended and can sign as its occurrence, the gate +writes the claim there and then, kicks replication, and re-reads the fold — +acting only if persist then names it. The loop only renews. + +**The row.** `scores`, dimension `session:claim:v1`, `cohort_scope: self`, +attested = attester = the device's NODE key (the occurrence the owner-binding +names: the held node signer on a split install, never the actor key), envelope +`{community_id, session_id, claimed_at}` with `claimed_at` in CC 2.6.2 form, +`expires_at = claimed_at + TTL`. Written through `attest::emit`, so it is +signed and stored through the one authored door, and a `self` row reaches the +person's other devices by persist's `send_set_for`. + +**Renewal on persist v51 — a successor lease.** v51's liveness is the +consumer's horizon measured from `claimed_at`, and persist is explicit that a +renewal must not move `claimed_at`. Together they mean a same-instant renewal +cannot extend anything on v51. So a renewal here is a fresh lease row written +only by the device the fold already names; because a non-holder never claims +while a live claim exists, the holder's leases are the only live claims and +the earliest of them keeps naming the holder. Residual: a simultaneous first +claim by two attended devices can hand the session over ONCE (the loser's one +lease outlives the winner's first); the new holder renews and the old one +defers. One handover, never two handlers in one view. + +**Persist v52 (CIRISPersist#946) — TODO in `session_claims::write_claim`:** the +signed lease bound `valid_until` (≤ 86 400 s after `claimed_at`) replaces the +consumer TTL as the horizon, and a renewal becomes a `supersedes` that keeps +`claimed_at`. Not invented before the pin moves: v51 has no member for it. + +**At persist v52.0.0 (0.5.218, edge v38 adopt) — half landed.** Every claim now +carries the signed `valid_until = claimed_at + 120 s`; v52 refuses a `session:*` +row without it and bounds the lease at a day (`check_session_lease_bound`), and +the TODO is gone from `write_claim`. The renewal is STILL the successor lease: +v52's `session_claim::handler_for` folds `now − claimed_at < ttl` with the +consumer's ttl and reads neither `valid_until` nor `supersedes`, so a renewal +that keeps `claimed_at` would expire the holder at `claimed_at + TTL` in every +view. The supersedes renewal waits on that read half of #946 (named in §6). + +## 5. The surface + +`GET /v1/self/sessions` — owner-authenticated (a delegate may read): + +```json +{ + "owner_key_id": "…", + "this_device": "", + "attended": true, + "ttl_seconds": 120, + "sessions": [{ + "community_id": "…", "session_id": "self_room:membership", + "handler_occurrence_key_id": "…", "handler_label": "Eric's laptop", + "claimed_at": "2026-09-30T12:00:00.000Z", "live_until": "2026-09-30T12:02:00.000Z", + "this_device": false, + "state_id": "session.state.handled_elsewhere", + "state": "Another of your devices is answering for you in this exchange." + }] +} +``` + +Only exchanges some device is answering are listed: an exchange whose every +claim lapsed has nobody answering, and is absent. + +New ids (string literals; on `KNOWN_UNLOCALIZED` in +`tools/check_server_localization.py` until the client bundle carries them; +the ratchet in `tests/localization_gate.rs` 139 → 142): + +| id | English | +|---|---| +| `session.state.handled_here` | This device is answering for you in this exchange. | +| `session.state.handled_elsewhere` | Another of your devices is answering for you in this exchange. | +| `self.sessions_unavailable` | The node could not read which of your devices is answering. Try again shortly. | + +(`self.owner_session_required` is reused with its existing text.) + +## 6. What the substrate should provide + +- **persist v52 / #946** — `valid_until` on `session:*` (the signed horizon) + and the `supersedes` renewal; then the successor-lease workaround and the + shared-constant TTL go. + *v52.0.0 shipped the write half (the member, required and bounded); the + read half — `handler_for` judging liveness by `valid_until` and folding a + renewing `supersedes` — is still owed, so the successor lease and the shared + TTL stay (§4).* +- **persist** — exported envelope path constants for `community_id`, + `session_id`, `claimed_at` (its reader spells them inline; the server names + them once and round-trips a written envelope through `claim_from_envelope` in + a unit test so a rename fails loudly). +- **edge** — a person's second device joining a chat room the person already + holds (so the chat handshake in §2.4 can defer to one device). + +## 7. The harness + +`harness/native/topology.py` relation `session(person, device)` (FSD/TOPOLOGY.md): +for a person with two online devices, drives the person's activity on `device`, +then asserts that `GET /v1/self/sessions` on EVERY device of the person names +the same handler for every listed exchange. In `topologies/selffiles.yaml` +after the `note` step. diff --git a/FSD/TOPOLOGY.md b/FSD/TOPOLOGY.md new file mode 100644 index 00000000..8a53aede --- /dev/null +++ b/FSD/TOPOLOGY.md @@ -0,0 +1,162 @@ +# TOPOLOGY — the trust topology a use case requires, from the root(s) down + +**Owner:** CIRISServer (the state lives here). **Consumers:** CIRISClient's CSD/4 +`topology:` block, `harness/native build`, the CSD checker. + +## 1. Why this exists + +A CSD/3 states what a screen shows and which routes it reads. It does not state +who must exist, in what trust relation, for those states to be reachable. The +first day of the native harness (2026-09-29) found five faults, each in a +different layer of that unstated structure: a pair never Rooted (root layer), +`reachable_nodes=0` at contact time (owner-binding layer, CIRISServer#699), a +body that could not cross two relay hops (transport layer, CC 5.4.6), a +multi-fragment frame that never left a direct link (CIRISEdge#716), a chunk-DAG +served as its manifest (content layer, CIRISEdge#717). A fixture cannot be +derived from prose, and a red cannot be placed by it. So each use case declares +its topology in this vocabulary, the checker verifies it is realizable, and the +harness builds exactly it. + +The vocabulary is ordered by dependency. Nothing in a lower layer can hold +unless the layer above it holds; a failed run names the first layer whose +predicate did not. + +## 2. The layers + +```yaml +topology: + roots: # 1. CC 3.2 — the same parameters as CIRISConstitution formal/trust_root/TrustRootVerdict.cfg + canonicals: # 2. the nodes that HOLD a root (charter, capability grant, trust edge) + nodes: # 3. every node, what it dials, which root it accepts, whether it announced + persons: # 4. owners: which nodes they own, which root THEY accept (the pair fact needs the owner) + relations: # 5. peering, rooting, reachability, contacts, rooms, grants, messages, files — in build order + actor: # 6. whose screen the flow drives + negatives: # what must NOT be true, by name +``` + +### 2.1 `roots` + +Reviewed by CC (CIRISConstitution#131): the block is `TrustRootVerdict.cfg`'s +constants by another name, with these corrections applied. + +| field | values | model constant | meaning | +|---|---|---|---| +| `id` | name | | referenced by `holds` / `accepts` | +| `kind` | `key` \| `infrastructure` | | a single holder key (the anchor today), or persist v51's infrastructure root under T7/T8. `family` (the accord family, CC 4.2.6) has NO model behind it and is refused | +| `holders` | int | | key roots: holder keys | +| `founders` | `{seated, conferrable, node_bearing}` | `Founders` = seated ∪ conferrable, `NodeKeys` | `seated` is the initial seated set (T7 needs `≥ M+1` active humans at founding); `conferrable` the unseated humans T7 recovery can widen in — without one the recovery never fires | +| `quorum` | int | `M` | absolute M of `quorum:M/N` | +| `witnesses` | `{n, independent_custody}` | `Witnesses`, `K` | **`k` is derived**, never declared: CC 3.2 T6 fixes `K = ⌊n/2⌋ + 1`; a declared `k` that differs is refused | +| `charter` | `{attach_window_secs, witness_cadence_secs}` | `AttachWindow = ⌈window / cadence⌉` | CC 2.1 charter members (rc6) | +| `lifecycle` | `{recipe, verdict}` | a value of V, not a constant | the ROWS that produce the state, and the verdict asserted after them: `{recipe: {}, verdict: rooted}` (active); `{recipe: {resignations: 1}, verdict: stalled}` (found at N = M+1, one resignation → `hard_case:community_liveness_stalled` once, widening refused); `{recipe: {halt: latched}, verdict: not_rooted}` | +| `custody` | `software_test` \| `hardware` | | what the anchor mints vs what a CSD reading holder evidence needs | + +**Buildable today:** one `key` root, one holder, `software_test`, `lifecycle +{recipe: {}, verdict: rooted}` — the synthetic anchor (`test_bless`: +`test-accord-holder-0`, charter root→root, capability grant root→node, trust +edge node→root). An `infrastructure` root with seated/conferrable founders, a +witnessed head, or a non-empty recipe is declared and REFUSED by the builder +by name until the ceremony can mint it; the declaration stays true, and CC's +`tools/topology_to_cfg.py` (their generator) reads the same block. + +### 2.2 `canonicals` + +`{id, holds: , serves: [infra:serve, infra:attest]}`. A canonical is a +node that holds the root; every other node dials at least one. The synthetic +anchor blesses exactly the nodes the builder tells it to. + +### 2.3 `nodes` + +`{id, dials: [], accepts: , announced: bool}`. + +`dials` is the transport topology and it is load-bearing: scoped content (chat +bodies, files) reaches DIRECTLY-ATTACHED peers only (CC 5.4.6, CIRISEdge#499). +Two nodes that reach each other only through a canonical key a room but never +exchange a body. A node that dials another is its direct neighbour; on that +link, today, multi-fragment frames stall (CIRISEdge#716). + +### 2.4 `persons` + +`{id, owns: [], accepts: }`. The first owned node is claimed +with a freshly minted identity; further owned nodes are that person's other +DEVICES (the owner's key material carried, then claimed — what +`POST /v1/self/associate` does with a portable keyset). `accepts` on the PERSON +is the acceptance `rooted_with` walks (CIRISEdge#659, CIRISServer#632): a node +that accepted a root whose owner did not is Attributed, never Rooted. + +### 2.5 `relations` (build order) + +| relation | builds / checks | +|---|---| +| `peered(A, B, prefixes?)` | production peering both ways: `GET /v1/federation/self-key-record` + `POST /v1/federation/peering` | +| `rooted_with(p, q)` | edge's pair verdict from the log (`rooted_with: a valid root in common`); `require: true` waits, else observed and recorded | +| `reachable(A, q) >= n` | `POST /v1/contacts` on A for q reports `reachable_nodes >= n` (q's owner→node binding held on A at federation scope); the gate CIRISServer#699 needs | +| `contact(p, q, via: owner\|code)` | p adds q by owner key or by q's contact code (`GET /v1/self/contact-code`) | +| `room(pair\|self, members, keyed: true, epoch?)` | the pair room opened on both sides and keyed; the self room joined by every device | +| `message(from, to, room)` | sent by `from`; the row AND its body on `to`'s node | +| `file(person, device, size, cohort)` | written on `device`; byte-identical on every other device of `person` | +| `custody(person, file: last\|, device?, cohort?)` | `GET /v1/files/{id}/custody` on the AUTHOR device (the one the `file`/`corpus` relation wrote on, unless `device` names another) names every other device of `person` with `holds: received` (a delivery receipt, CC 5.3.3.6) and `devices_total` = the person's device count; `last` = the most recent `file`, or the last file a `corpus` wrote; an inline (≤ 1 MiB) file fails by name (`receipts_supported: false`). `FSD/FILE_CUSTODY.md` | +| `session(person, device, require?)` | the person active on `device` (owner-bearer requests); `GET /v1/self/sessions` on EVERY device of `person` names the same handler for each listed exchange, and at least one is listed (CC 3.1.3.1, FSD/SESSION_CLAIMS.md) | +| `member(p, community, role)`, `quorum_change(...)` | declared; builder refuses until the community/household scenarios exist | + +### 2.6 `actor` and `negatives` + +`actor: {person, device}` names the screen the client flow drives; the +builder's `values.json` is written from its point of view (the client +fixture's `${PEER_KEY_ID}`, `${ROOM_ID}`, `${MESSAGE_ATTESTATION_ID}`, …). + +`negatives` are checked last: + +| negative | asserts | +|---|---| +| `cannot_list_room(p, room)` | person `p`'s node cannot list the room (a non-member reads nothing) | +| `holds_no_row(node, dimension, person)` | an outsider `node` holds none of `person`'s self-plane rows of `dimension`, with a NON-VACUOUS control: the same node must hold ≥ 1 of that person's public rows, or the negative fails as unproven | +| `no_wider_self_rows(person)` | every self-plane row on every device of `person` is at `cohort_scope: self` — nothing of the self plane was widened (CIRISPersist#919) | + +## 3. Realizability rules (the checker refuses, by name) + +1. `quorum: M` needs `founders.seated >= M + 1` (T7: active humans at + founding) and `founders.conferrable >= 1` (or T7 recovery has nobody to + widen in); a witnessed head derives `K = ⌊n/2⌋ + 1` and needs + `independent_custody`; a declared `k` is refused. +2. A person `accepts` a root only if every node they own `accepts` it. +3. `rooted_with(p, q)` needs both persons to `accept` a common root. +4. Scoped content reaches a member over the room's DERIVED address, which + is never announced (CC 5.4.6, CIRISEdge#499). So: a `self` or `family` + room needs every other device to hold a direct link (`dials`) to the + AUTHOR's node; a `community` room needs the member nodes to form a + CONNECTED graph (any member node can relay, including a person's home + node — CIRISEdge#718 is the witness); a pair room whose two members sit + behind a non-member canonical with no direct link waits on a ruling + (CIRISConstitution#132: may a scoped body ride the identity-plane link). + The refusal is edge's admission gate, not reachability: both nodes can + reach each other through the canonical as a Reticulum transport node. +5. `reachable(A, q)` needs one of q's nodes `announced: true`. +6. `contact(p, q, via: code)` needs q's node `announced` (the code names + announced devices) and the route (0.5.218+). +7. Everything in `roots` must be within what the ceremony mints today, or the + declaration is `buildable: false` and says which field. + +## 4. Derivations + +`nodes = |nodes| + |canonicals|`; `persons = |persons|`; `devices(p) = +|p.owns|`; the root ceremony from `roots`; `K` and `AttachWindow` from +`witnesses.n` and `charter`; the build order from the layers. + +## 4.1 The row export + +Every build writes, per node, `//rows.jsonl`: each admitted +`federation_attestations` row with every column, in admission order, plus +`keys.jsonl` (the node's `federation_keys`) and `transport_destinations.jsonl` +(the signed routes it holds — the #393 item-2 operands, CIRISEdge#722). That is the fold-replayable form +CC replays `V(rows, t)` over for T8 (iii) offline, and turns a harness trace +into a TLC trace check (CIRISConstitution#131 §5). `report.json` carries, per +node, the count and the SHA-256 of the export. A +flow may not advance to `testable` while its fixture's topology is smaller than +its CSD's on any layer. + +## 5. The proving set + +`harness/native/topologies/csd-091-user-chat.yaml`, +`csd-092-share-contact-code.yaml`, `csd-094-approve-new-device.yaml` (CSD-094's +post-state; CSD-093's own flow waits on CIRISServer#678) — the 0.5.218 flows. `python -m harness.native build --topology --binary `. diff --git a/README.md b/README.md index 906fc811..bdaa5cfe 100644 --- a/README.md +++ b/README.md @@ -352,7 +352,7 @@ itself, while a quorum roots to `humanity-accord`. The charter is 2-of-2 over a — a replicated row proves its own m-of-n rather than deferring to the bundle it arrived in. Two humans to halt, two to legitimize. -Substrate pins: **persist v49.0.0 / edge v32.1.0 / verify v17.1.0** — hybrid PQ +Substrate pins: **persist v52.0.0 / edge v38.0.0 / verify v18.0.0** — hybrid PQ throughout, Registry-of-Record admission, drift-witnessed policy hashes. Edge v15.7.x adds the realtime A/V spine (MLS X-Wing epoch keys, signed Welcome, fragment ARQ); the server exercises publisher → relay → subscriber diff --git a/benches/chat_throughput.rs b/benches/chat_throughput.rs index ad1c4dc2..18e4d6d0 100644 --- a/benches/chat_throughput.rs +++ b/benches/chat_throughput.rs @@ -659,6 +659,32 @@ async fn build_bed(client: &reqwest::Client) -> Result { "community_id {community_id} is not the derived pair id — fixture drift" )); } + // edge v38 / persist v52 (CIRISPersist#955): the room is founded by the + // owner ALONE and the contact is PROPOSED. The bench plays the contact + // accepting with their own key; the owner's second call seats them. + let proposal_id = json["proposal_id"] + .as_str() + .ok_or_else(|| format!("POST /v1/chat carried no invitation: {json}"))? + .to_string(); + ciris_edge::membership::reply( + engine.federation_directory().as_ref(), + &proposal_id, + true, + contact_edge_signer().await.as_ref(), + ) + .await + .map_err(|e| format!("the contact accepts the pair room's invitation: {e}"))?; + let resp = client + .post(format!("{base}/v1/chat")) + .bearer_auth(&owner) + .json(&serde_json::json!({ "key_id": CONTACT_KEY_ID })) + .send() + .await + .map_err(|e| format!("POST /v1/chat (seat the contact): {e}"))?; + let seated: serde_json::Value = resp.json().await.map_err(|e| format!("chat json: {e}"))?; + if seated["state"] != "open" { + return Err(format!("the contact was not seated: {seated}")); + } // The handshake, before any phase measures a send. key_the_room(&engine, &owner_id.key_id, &community_id).await; diff --git a/crates/ciris-lens-core/Cargo.toml b/crates/ciris-lens-core/Cargo.toml index 4d7bf34a..b6d6f3c2 100644 --- a/crates/ciris-lens-core/Cargo.toml +++ b/crates/ciris-lens-core/Cargo.toml @@ -25,8 +25,8 @@ crate-type = ["cdylib", "rlib"] # # OQ-01 closure 2026-05-03: rlib primary, PyO3 cdylib via `python` # feature for the deployed-lens cutover. -ciris-persist = { git = "https://github.com/CIRISAI/CIRISPersist", tag = "v49.0.0", version = "49", features = ["extract", "sqlite"] } -ciris-edge = { git = "https://github.com/CIRISAI/CIRISEdge", tag = "v32.1.0", version = "32", features = ["transport-http", "transport-reticulum"] } +ciris-persist = { git = "https://github.com/CIRISAI/CIRISPersist", tag = "v52.0.0", version = "52", features = ["extract", "sqlite"] } +ciris-edge = { git = "https://github.com/CIRISAI/CIRISEdge", tag = "v38.0.0", version = "38", features = ["transport-http", "transport-reticulum"] } # Keyring — relay mode (CIRISLensCore#10) loads its Edge transport- # signing identity via `ciris_keyring::load_local_seed`. Edge's @@ -36,16 +36,16 @@ ciris-edge = { git = "https://github.com/CIRISAI/CIRISEdge", tag = "v32.1. # CIRISVerify tag persist + edge pin (v5.0.0 — CEG 1.0 / Agent 3.0 substrate # release; MAJOR 4→5 but additive for lens-core) — single verify version. # features mirror edge's keyring pin exactly. -ciris-keyring = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v17.1.0", version = "17", features = ["software", "pqc-ml-dsa"] } +ciris-keyring = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v18.0.0", version = "18", features = ["software", "pqc-ml-dsa"] } # verify-core for `fedcode::derive_key_id` — the lens seal path must stamp the # DERIVED federation key_id (`derive_key_id(, )`), the id # `receive_and_persist` verifies against, NOT the bare `local_key_id` alias # (CIRISServer#118; same class as the closed CIRISEdge#203). Single verify version. -ciris-verify-core = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v17.1.0", version = "17" } +ciris-verify-core = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v18.0.0", version = "18" } # Crypto-DRY (CIRISServer#283 findings 3+5): route the hand-composed hybrid rule # (event.rs) + the test-only trace verify (seal.rs) through the one crypto facade # instead of a direct ed25519-dalek dep and a replicated binding. -ciris-crypto = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v17.1.0", version = "17", features = ["pqc-ml-dsa"] } +ciris-crypto = { git = "https://github.com/CIRISAI/CIRISVerify", tag = "v18.0.0", version = "18", features = ["pqc-ml-dsa"] } # persist's `sqlite` feature: relay mode persists inbound batches via # `Engine::receive_and_persist` + reads `Engine::sqlite_backend()` for diff --git a/evidence/cc_impl.tsv b/evidence/cc_impl.tsv index dc49c817..3c893bb4 100644 --- a/evidence/cc_impl.tsv +++ b/evidence/cc_impl.tsv @@ -51,7 +51,7 @@ decimal_id claim_id repo path#symbol crate@version 4.1.1 CLM-anti-pattern-delegation — — open 4.1.4 CLM-withdraws-arbitrage CIRISServer src/withdraws_arbitrage.rs#enforce ciris-server@main 4.2.1.1 CLM-invocation CIRISVerify src/ciris-verify-core/src/humanity_accord.rs#Invocation::canonical_bytes ciris-verify-core@v10.1.1 -4.2.1.2 CLM-notify CIRISClient client/shared/src/commonMain/kotlin/ai/ciris/mobile/shared/ui/screens/AccordScreen.kt#invocationBadge ciris-client@0.5.224 +4.2.1.2 CLM-notify CIRISClient client/shared/src/commonMain/kotlin/ai/ciris/mobile/shared/ui/screens/AccordScreen.kt#invocationBadge ciris-client@0.5.225 4.2.1.3 CLM-lifecycle CIRISVerify src/ciris-verify-core/src/humanity_accord.rs#verify_invocation ciris-verify-core@v10.1.1 4.2.2 CLM-hardware-class CIRISVerify src/ciris-keyring/src/hw_token.rs#hardware_class_table ciris-keyring@v10.1.1 4.2.2.1 CLM-hardware-class-hardware CIRISServer src/hardware_attestation.rs#admit_hardware_class ciris-server@main diff --git a/examples/qa_runner/family.rs b/examples/qa_runner/family.rs index b741fd07..deccb6bc 100644 --- a/examples/qa_runner/family.rs +++ b/examples/qa_runner/family.rs @@ -107,6 +107,7 @@ pub async fn run(report: &mut Report) { founded_at: chrono::Utc::now(), consensus_protocol: "founder_only".into(), consensus_protocol_entrenched: false, + dissolved_at: None, persist_row_hash: String::new(), }; match family::create_family(&engine, fam).await { diff --git a/harness/__init__.py b/harness/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/harness/mesh-repro/lib/corpus_client.py b/harness/mesh-repro/lib/corpus_client.py index 3f6fd63d..1345a294 100644 --- a/harness/mesh-repro/lib/corpus_client.py +++ b/harness/mesh-repro/lib/corpus_client.py @@ -121,6 +121,42 @@ def read(token: str, d: Path) -> None: "corrupt": corrupt, "waiting": waiting})) +def probe(token: str, d: Path, names: list[str]) -> None: + """For each named file: read it raw HERE and compare with the original + (``/``, copied in by the ladder's diagnosis), byte by byte. + Tells a truncation (a prefix of the original) from a mid-file change + (``first_diff`` inside both, ``bytes_differing`` counted).""" + manifest = {r["name"]: r for r in json.loads((d / "manifest.json").read_text(encoding="utf-8"))} + written = {w["name"]: w for w in json.loads((d / "written.json").read_text(encoding="utf-8"))} + for name in names: + w = written.get(name) or {} + out = {"name": name, "expected_size": manifest.get(name, {}).get("size")} + if not w.get("attestation_id"): + out["error"] = "not written" + print(json.dumps(out)) + continue + q = urllib.parse.quote(w["attestation_id"], safe="") + status, raw = _call(token, "GET", f"/v1/files/{q}?cohort=self&raw=1") + out.update({"status": status, "size": len(raw), + "sha256": hashlib.sha256(raw).hexdigest(), + "match": hashlib.sha256(raw).hexdigest() == manifest.get(name, {}).get("sha256")}) + orig_path = d / name + if status == 200 and orig_path.exists(): + orig = orig_path.read_bytes() + n = min(len(raw), len(orig)) + first = next((i for i in range(n) if raw[i] != orig[i]), None) + out["first_diff"] = first if first is not None else (n if len(raw) != len(orig) else None) + out["bytes_differing"] = sum(1 for i in range(n) if raw[i] != orig[i]) + out["is_prefix_of_original"] = first is None and len(raw) < len(orig) + if first is not None: + out["got_at_diff"] = raw[first:first + 16].hex() + out["want_at_diff"] = orig[first:first + 16].hex() + print(json.dumps(out)) + + if __name__ == "__main__": verb, token, directory = sys.argv[1], sys.argv[2], Path(sys.argv[3]) - {"write": write, "read": read}[verb](token, directory) + if verb == "probe": + probe(token, directory, sys.argv[4].split(",")) + else: + {"write": write, "read": read}[verb](token, directory) diff --git a/harness/mesh-repro/lib/media_corpus.py b/harness/mesh-repro/lib/media_corpus.py index 2fb62170..3e5a61ed 100644 --- a/harness/mesh-repro/lib/media_corpus.py +++ b/harness/mesh-repro/lib/media_corpus.py @@ -42,10 +42,10 @@ #: DESCRIBED and turn red the moment one passes, so a fixed defect cannot stay #: marked. Never add a row here to make a run green: add it only with an issue #: that names the cause. -KNOWN_DEFECTS = { - "inline_band": "CIRISEdge#687: files::publish picks inline by plaintext size and persist caps " - "the sealed size, so 1,048,541-1,048,576 bytes can never publish", -} +# Empty since edge v33.0.0: `inline_band` (CIRISEdge#687 — inline chosen by +# plaintext size while persist capped the SEALED size, so 1,048,541-1,048,576 +# bytes could never publish) now uploads, and the drive test named it. +KNOWN_DEFECTS: dict = {} def _payload(seed: str, n: int) -> bytes: diff --git a/harness/mesh-repro/scenarios/devices.sh b/harness/mesh-repro/scenarios/devices.sh index 127915e4..6a91a0eb 100644 --- a/harness/mesh-repro/scenarios/devices.sh +++ b/harness/mesh-repro/scenarios/devices.sh @@ -17,6 +17,14 @@ # 2. HOUSEHOLD (CIRISServer#647). Person A charters a family, adds person B, # and writes a `cohort: family` file. Does B's node see the household, list # the file, and open it? +# 0.5.218 — CONSENT TO JOIN (the maintainer's ruling of 2026-09-30, +# CIRISConstitution#133, FSD/MEMBERSHIP_INVITES.md): nobody joins a +# household without their own acceptance, and until persist can record one +# (CIRISPersist#955, v52) the add answers 409 `membership.consent_required`. +# The `family` rung now asserts THAT refusal; the rungs downstream of B +# being a member (family_on_b, family_file_listed_on_b, +# family_file_opened_on_b) are RED-EXPECTED on #955 — there is no +# production door that can put B on the roster, and the harness adds none. # # WHY A SEPARATE SCENARIO and not more rungs on chat.sh: the chat ladder gates # the tag and runs in CI on every substrate PR; these rungs are RED-EXPECTED by @@ -47,6 +55,10 @@ # family record neither re-indexes the wire # (`supersede_group_row`) nor re-puts at a peer that holds # the founding one (`put_family` is a plain INSERT). +# 0.5.218: persist v49 CLOSED #910 (a widening plane for +# families, amendments carry their proof). These rungs are +# now red-expected on a different piece: consent to join +# (CIRISPersist#955) — no door may add B until B accepts. # # EACH IS PROMOTED TO REQUIRED WHEN ITS PIECE LANDS — not before, and not left # RED-EXPECTED after: a marked stage that has gone green is a claim nobody is @@ -84,9 +96,9 @@ REQUIRED_family=1 REQUIRED_family_file=1 XFAIL_c_opens_history="no content-key REWRAP to a new occurrence of an EXISTING member: the message was sealed under the minter's community epoch DEK and wrapped to the member occurrences persist knew then; node-c's occurrence of B is provisioned later (ensure_owner_content_occurrence) and nothing wraps existing epochs to it — persist has rekey_self_occurrence_add (self blobs; the server calls it only from POST /v1/self/occurrence, not on claim-remote) and rekey_family_member_add (a new member), and no community-epoch occurrence-add; ensure_epoch_dek fills a late occurrence only on the minter's NEXT seal in the SAME epoch. Ask: a persist door that wraps every epoch/blob an existing occurrence of the identity can open to the new occurrence, driven by the server when a claimed node's content occurrence appears" -XFAIL_family_on_b="EXPECTED TO PASS from 0.5.218 (persist v49.0.0 fixed CIRISPersist#910 and the server routes FamilyMembershipWidening, kind 18) — remove this mark after the first green devices run. Was: CIRISServer#646 (no Family replication round; fixed on this branch) + CIRISPersist#910: B is added AFTER create, so B's node needs the GROWN record, and supersede_group_row re-stamps admitted_at without re-indexing the wire while put_family is a plain INSERT at a peer holding the founding record" -XFAIL_family_file_listed_on_b="EXPECTED TO PASS from 0.5.218 (persist v49.0.0 fixed CIRISPersist#910 and the server routes FamilyMembershipWidening, kind 18) — remove this mark after the first green devices run. Was: downstream of family_on_b (CIRISPersist#910): a family row's audience is the family's members' nodes as B's node folds them, and B's node does not hold the grown record naming B" -XFAIL_family_file_opened_on_b="EXPECTED TO PASS from 0.5.218 (persist v49.0.0 fixed CIRISPersist#910 and the server routes FamilyMembershipWidening, kind 18) — remove this mark after the first green devices run. Was: downstream of family_file_listed_on_b (CIRISPersist#910); REQUIRED once the family plane converges (#647 asks for the bytes stage to be REQUIRED)" +XFAIL_family_on_b="CONSENT TO JOIN (0.5.218): B cannot be added until B can accept — the add answers 409 membership.consent_required until the invite flow (CIRISPersist#955 v52, CIRISConstitution#133, FSD/MEMBERSHIP_INVITES.md); promote when #955 lands and the rung drives propose → accept. Before that ruling: EXPECTED TO PASS from 0.5.218 (persist v49.0.0 fixed CIRISPersist#910 and the server routes FamilyMembershipWidening, kind 18) — remove this mark after the first green devices run. Was: CIRISServer#646 (no Family replication round; fixed on this branch) + CIRISPersist#910: B is added AFTER create, so B's node needs the GROWN record, and supersede_group_row re-stamps admitted_at without re-indexing the wire while put_family is a plain INSERT at a peer holding the founding record" +XFAIL_family_file_listed_on_b="CONSENT TO JOIN (0.5.218): downstream of family_on_b, which is red-expected on CIRISPersist#955 (B cannot join without accepting). Before that ruling: EXPECTED TO PASS from 0.5.218 (persist v49.0.0 fixed CIRISPersist#910 and the server routes FamilyMembershipWidening, kind 18) — remove this mark after the first green devices run. Was: downstream of family_on_b (CIRISPersist#910): a family row's audience is the family's members' nodes as B's node folds them, and B's node does not hold the grown record naming B" +XFAIL_family_file_opened_on_b="CONSENT TO JOIN (0.5.218): downstream of family_on_b, which is red-expected on CIRISPersist#955 (B cannot join without accepting). Before that ruling: EXPECTED TO PASS from 0.5.218 (persist v49.0.0 fixed CIRISPersist#910 and the server routes FamilyMembershipWidening, kind 18) — remove this mark after the first green devices run. Was: downstream of family_file_listed_on_b (CIRISPersist#910); REQUIRED once the family plane converges (#647 asks for the bytes stage to be REQUIRED)" _dev_load() { _chat_load; [ -f "$CHAT_STATE/dev.sh" ] && . "$CHAT_STATE/dev.sh"; return 0; } @@ -261,6 +273,8 @@ _dev_household() { if [ -z "$fid" ]; then return 0; fi # ADDED AFTER CREATE, deliberately: that is the grown-record path #910 names, # and the one a real household takes (you found it, then you invite). + # 0.5.218: this add is REFUSED — 409 membership.consent_required — until B + # can accept (CIRISPersist#955); the `family` rung asserts the refusal. _dev_api "${CHAT_SENDER_SVC:-node-a}" "$CHAT_A_TOKEN" POST "/v1/families/$fid/members" \ "$(python3 -c 'import json,sys;print(json.dumps({"key_id":sys.argv[1]}))' "$CHAT_B_OWNER")" \ >"$CHAT_STATE/family-add.json" @@ -390,15 +404,20 @@ DIAG_c_opens_history() { # ── the household rungs (CIRISServer#647) ──────────────────────────────────── stage_family() { - local c a + # 0.5.218 CONSENT TO JOIN: green = A charters the household (2xx) AND the + # add of B is refused by name — 409 membership.consent_required + # (CIRISConstitution#133, CIRISPersist#955). An add that SUCCEEDS is a red + # rung now: it would mean a door admitted B without B's acceptance. + local c a r c="$(_dev_field "$CHAT_STATE/family-create.json" status)" a="$(_dev_field "$CHAT_STATE/family-add.json" status)" - case "$c:$a" in - 20?:20?) echo 1 ;; + r="$(_dev_field "$CHAT_STATE/family-add.json" body.reason_id)" + case "$c:$a:$r" in + 20?:409:membership.consent_required) echo 1 ;; *) echo 0 ;; esac } -HINT_family="person A could not charter a household or add person B on A's own node. family.unknown_member_key = B's owner key is not registered on node-a (the contact phase admits it); family.not_authorized = the founder_only rule; read family-create.json / family-add.json" +HINT_family="0.5.218: green means A chartered the household AND the add of B was refused 409 membership.consent_required (nobody joins without their own acceptance — CIRISPersist#955, CIRISConstitution#133). A 200 on the add is a door admitting B without consent; a create failure is below. Before the consent ruling this rung read: person A could not charter a household or add person B on A's own node. family.unknown_member_key = B's owner key is not registered on node-a (the contact phase admits it); family.not_authorized = the founder_only rule; read family-create.json / family-add.json" EXIT_family=64 DIAG_family() { echo " create: $(head -c 400 "$CHAT_STATE/family-create.json" 2>/dev/null)" @@ -419,7 +438,7 @@ rows=rows or [] print(1 if d.get("status")==200 and any((r.get("family_id") if isinstance(r, dict) else None)==sys.argv[2] for r in rows) else 0)' \ "$CHAT_STATE/family-b.json" "$DEV_FAMILY_ID" 2>/dev/null || echo 0 } -HINT_family_on_b="B's node does not list the household it was added to. With #646 on this branch the Family plane has a round, so this is the grown-record gap (CIRISPersist#910) unless the diagnosis shows node-b holding NO family row at all" +HINT_family_on_b="B's node does not list the household it was added to. With #646 on this branch the Family plane has a round, so this is the grown-record gap (CIRISPersist#910) unless the diagnosis shows node-b holding NO family row at all. 0.5.218: #910 is fixed in persist v49; the rung is red-expected on consent to join (CIRISPersist#955) — B was never added, by design" EXIT_family_on_b=65 DIAG_family_on_b() { echo " node-b families: $(head -c 400 "$CHAT_STATE/family-b.json" 2>/dev/null)" diff --git a/harness/mesh-repro/scenarios/selffiles.sh b/harness/mesh-repro/scenarios/selffiles.sh index 38755c97..5007eb98 100644 --- a/harness/mesh-repro/scenarios/selffiles.sh +++ b/harness/mesh-repro/scenarios/selffiles.sh @@ -761,11 +761,38 @@ except Exception: print(0); raise SystemExit ok = d.get("total") and d.get("opened")==d.get("total") and not d.get("corrupt") print(1 if ok else 0)' "$SELF_STATE/corpus-read.json" 2>/dev/null || echo 0 } -HINT_corpus_opened_on_b="the second device did not return every corpus file byte-identical. corpus-read.json lists corrupt (bytes returned but the SHA-256 differs: a transfer defect, never a fixture change — the corpus is deterministic) and waiting as name:status:reason_id (409 drive.not_fetched = the bytes never crossed; 403 drive.not_granted = no wrap names this device). A pattern by size (inline_over / large only) points at the chunk path; by type, at the write or render path" +HINT_corpus_opened_on_b="the second device did not return every corpus file byte-identical. A returned size of a few hundred bytes to ~10 KiB beginning {\"chunk_tier\" is the chunk-DAG MANIFEST served as the file: the puller stored it inline and fetched no chunks (CIRISEdge#717; the drive now refuses it as drive.seal_mismatch). corpus-read.json lists corrupt (bytes returned but the SHA-256 differs: a transfer defect, never a fixture change — the corpus is deterministic) and waiting as name:status:reason_id (409 drive.not_fetched = the bytes never crossed; 403 drive.not_granted = no wrap names this device). A pattern by size (inline_over / large only) points at the chunk path; by type, at the write or render path" EXIT_corpus_opened_on_b=49 DIAG_corpus_opened_on_b() { cat "$SELF_STATE/corpus-read.json" 2>/dev/null; echo - compose exec -T "$SELF_SECOND" cat /tmp/corpus-state/read.json 2>/dev/null | head -c 3000; echo + # EVERY file that did not come back byte-identical, in full — the per-file + # list is long, and a 3000-byte head cut exactly the rows that failed. + compose exec -T "$SELF_SECOND" python -c ' +import json +for r in json.load(open("/tmp/corpus-state/read.json")): + if not r.get("match"): print(json.dumps(r))' 2>/dev/null + # WHERE the bytes differ, on BOTH devices: the originals of the corrupt + # files go to the second device, and each is read raw on the author too. + # Same bytes wrong on both = the write or the author's store; wrong on the + # second only = the transfer. A prefix of the original is a truncation. + local bad + bad="$(python3 -c ' +import json,sys +try: print(",".join(json.loads(open(sys.argv[1]).read().strip().splitlines()[-1]).get("corrupt") or [])) +except Exception: print("")' "$SELF_STATE/corpus-read.json" 2>/dev/null)" + if [ -n "$bad" ]; then + local n svc + for n in ${bad//,/ }; do + [ -f "$SELF_CORPUS/$n" ] && compose cp "$SELF_CORPUS/$n" "$SELF_SECOND:/tmp/corpus/$n" >/dev/null 2>&1 + done + compose cp "$HARNESS_DIR/lib/corpus_client.py" "$SELF_PRIMARY:/tmp/corpus/corpus_client.py" >/dev/null 2>&1 + compose cp "$SELF_CORPUS/written.json" "$SELF_PRIMARY:/tmp/corpus/written.json" >/dev/null 2>&1 + for svc in $SELF_PRIMARY $SELF_SECOND; do + echo " byte probe on $svc (${bad}):" + compose exec -T "$svc" python /tmp/corpus/corpus_client.py probe \ + "$(_self_token "$svc")" /tmp/corpus "$bad" 2>&1 | sed 's/^/ /' + done + fi } harness_scenario_evidence() { diff --git a/harness/native/__init__.py b/harness/native/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/harness/native/__main__.py b/harness/native/__main__.py new file mode 100644 index 00000000..c0d701f2 --- /dev/null +++ b/harness/native/__main__.py @@ -0,0 +1,89 @@ +"""Run a native-mesh scenario. + + python -m harness.native --binary target/debug/ciris-server [--keep] + +`--binary` is any `ciris-server` built with `--features test-anchor` (the +synthetic trust root is only honoured there). `--keep` leaves every node +running and prints their URLs and tokens, so the next question is asked of the +same nodes. Node logs: //node.log. +""" +from __future__ import annotations + +import argparse +import json +import sys +import traceback +from pathlib import Path + +from .mesh import Mesh, MeshError +from .scenarios import SCENARIOS + + +def main() -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("scenario", choices=sorted(SCENARIOS) + ["build", "derive", "down"]) + ap.add_argument("--topology", type=Path, help="build/derive: a FSD/TOPOLOGY.md declaration") + ap.add_argument("--binary", type=Path, required=True) + ap.add_argument("--work", type=Path, default=Path("/tmp/ciris-native-mesh")) + ap.add_argument("--keep", action="store_true", help="leave the nodes running") + ap.add_argument("--owner-wait", type=float, default=90) + ap.add_argument("--ready-wait", type=float, default=180) + ap.add_argument("--arrive-wait", type=float, default=180) + ap.add_argument("--contact-via", choices=["owner", "code"], default="owner") + ap.add_argument("--rooted-wait", type=float, default=0, help="chat: seconds to wait for the pair to be Rooted (0 = observe only)") + ap.add_argument("--reachable-wait", type=float, default=120, help="chat: seconds to re-add a contact until reachable_nodes>=1 (0 = no gate)") + ap.add_argument("--only", default="", help="corpus: comma-separated file names") + ap.add_argument("--rust-log", default="info,ciris_edge=debug") + ap.add_argument("--direct", action="store_true", + help="chat: the second node also dials the first (direct neighbours)") + args = ap.parse_args() + if args.scenario == "down": + print(json.dumps({"stopped": Mesh.down(args.work)})) + return 0 + if args.scenario in ("build", "derive"): + from . import topology as topo + try: + decl = topo.load(args.topology) + except topo.Unrealizable as e: + print(json.dumps({"verdict": "UNREALIZABLE", "error": str(e)})) + return 2 + if args.scenario == "derive": + print(json.dumps(topo.derive(decl), indent=1, default=str)) + return 0 + SCENARIOS["build"] = lambda mesh, a: topo.build(mesh, decl, a) + + mesh = Mesh(args.binary, args.work, keep=args.keep, rust_log=args.rust_log) + code = 0 + with mesh: + try: + result = SCENARIOS[args.scenario](mesh, args) + except MeshError as e: + result = {"verdict": "BROKEN", "error": str(e)} + try: # a build failure carries its first failing layer as JSON + parsed = json.loads(str(e)) + result.update({"error": parsed.get("error"), "first_failing_layer": parsed.get("first_failing_layer")}) + except Exception: # noqa: BLE001 + pass + code = 2 + except Exception: # noqa: BLE001 + result = {"verdict": "CRASH", "error": traceback.format_exc()[-2000:]} + code = 3 + if result.get("verdict") == "FAIL": + code = 1 + print("═══ VERDICT", json.dumps({k: v for k, v in result.items() if k not in ("steps", "results")}, indent=1)) + report = {"scenario": args.scenario, "binary": str(mesh.binary), "args": vars(args) | {"binary": str(args.binary), "work": str(args.work)}, + **result, "nodes": mesh.state()} + (mesh.work / "report.json").write_text(json.dumps(report, indent=1, default=str), encoding="utf-8") + if result.get("values"): + # The client's two-node fixture shape (testing/gate/two_node.py values.json): + # a client flow reads ${PEER_KEY_ID}, ${MESSAGE_ATTESTATION_ID}, … from here. + (mesh.work / "values.json").write_text(json.dumps(result["values"], indent=1), encoding="utf-8") + print(f"report: {mesh.work / 'report.json'}" + (f" values: {mesh.work / 'values.json'}" if result.get("values") else "")) + if args.keep: + print("nodes left running — `python -m harness.native down --work " + str(mesh.work) + "` stops them") + print(json.dumps(mesh.state(), indent=1)) + return code + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/harness/native/mesh.py b/harness/native/mesh.py new file mode 100644 index 00000000..1f5aae61 --- /dev/null +++ b/harness/native/mesh.py @@ -0,0 +1,587 @@ +"""A native mesh: N `ciris-server` processes on 127.0.0.1 under the SYNTHETIC +trust root, no Docker, in seconds. + +WHY. The mesh-repro ladders answer in ~45 minutes (wheel build, image, compose, +a 13-minute watch). A question like "does a 1 MiB self file come back +byte-identical on the second device?" or "do two people's nodes key a pair +room over the PRODUCTION peering routes?" needs an answer in minutes, and the +next question needs the same nodes again. So: one binary (any build with +`--features test-anchor`), one directory per node, and small steps that +compose. + +THE TRUST ROOT is the harness's own synthetic anchor, read from +`harness/mesh-repro/docker-compose.yml` (`x-test-anchor-env`), so the fixture +and the ladders never carry two copies of it. Under a live test anchor a node +dials only what it is told (edge #661) — never the production canonical — and a +`canonical` node blessed as the test root's holder is what every other node +roots through, exactly as in the ladders. + +PRODUCTION ROUTES BY DEFAULT. Peering is `GET /v1/federation/self-key-record` ++ `POST /v1/federation/peering`, the routes a released node exposes. The +ladders' `test-admit-peer` shortcut is available (`Node.test_admit`) but a +scenario must ask for it by name: it is the shortcut that kept the chat ladder +green while two released nodes could not key a pair room (CIRISServer#698). + +Everything here is a library: `scenarios.py` composes it, `__main__` runs it, +and a REPL can drive the same objects by hand. +""" +from __future__ import annotations + +import base64 +import glob +import hashlib +import io +import json +import os +import re +import shutil +import signal +import socket +import sqlite3 +import subprocess +import time +import urllib.error +import urllib.request +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any, Callable, Dict, List, Optional, Tuple + +HARNESS = Path(__file__).resolve().parent.parent +COMPOSE = HARNESS / "mesh-repro" / "docker-compose.yml" + + +class MeshError(RuntimeError): + """A step could not do what it says. Carries the evidence, never a guess.""" + + +def anchor_env() -> Dict[str, str]: + """The synthetic trust root and harness knobs, as the ladders set them. + + Parsed from the `x-test-anchor-env` block of the mesh-repro compose file: + every `KEY: "value"` line until the block ends.""" + env: Dict[str, str] = {} + inside = False + for line in COMPOSE.read_text(encoding="utf-8").splitlines(): + if line.startswith("x-test-anchor-env:"): + inside = True + continue + if inside: + if line and not line.startswith((" ", "\t")): + break + m = re.match(r'^\s+([A-Z][A-Z0-9_]+):\s*"(.*)"\s*$', line) + if m: + env[m.group(1)] = m.group(2) + if "CIRIS_TEST_TRUST_ROOT" not in env: + raise MeshError(f"no CIRIS_TEST_TRUST_ROOT in the x-test-anchor-env block of {COMPOSE}") + return env + + +def free_port_pair(start: int = 7242) -> int: + """A port p with p and p+1 both free (edge on p, HTTP on p+1).""" + p = start + while p < 65000: + ok = True + for q in (p, p + 1): + with socket.socket() as s: + try: + s.bind(("127.0.0.1", q)) + except OSError: + ok = False + break + if ok: + return p + p += 2 + raise MeshError("no free port pair") + + +def http(method: str, url: str, token: Optional[str] = None, body: Any = None, + timeout: float = 60.0, raw: bool = False) -> Tuple[int, Any]: + """(status, parsed JSON | bytes when raw). A transport failure is status 0.""" + data = json.dumps(body).encode() if body is not None else None + headers = {"Content-Type": "application/json"} + if token: + headers["Authorization"] = "Bearer " + token + req = urllib.request.Request(url, method=method, data=data, headers=headers) + try: + with urllib.request.urlopen(req, timeout=timeout) as r: + payload = r.read() + status = r.status + except urllib.error.HTTPError as e: + payload, status = e.read(), e.code + except Exception as e: # noqa: BLE001 — a result, not a crash + return 0, {"detail": repr(e)[:300]} + if raw: + return status, payload + try: + return status, json.loads(payload.decode() or "{}") + except Exception: # noqa: BLE001 + return status, {"raw": payload[:300].decode(errors="replace")} + + +def wait_for(what: str, probe: Callable[[], Any], timeout: float, every: float = 2.0) -> Any: + """Poll `probe` until it returns something truthy; raise naming `what`.""" + deadline = time.monotonic() + timeout + last = None + while time.monotonic() < deadline: + last = probe() + if last: + return last + time.sleep(every) + raise MeshError(f"timed out after {timeout:.0f}s waiting for {what} (last: {str(last)[:300]})") + + +# ── one node ──────────────────────────────────────────────────────────────── + + +@dataclass +class Node: + name: str + binary: Path + home: Path + key_id: str + port: int + env: Dict[str, str] + log_path: Path + proc: Optional[subprocess.Popen] = None + token: str = "" + owner_key_id: str = "" + node_key_id: str = "" + record: Any = None + + # -- process -- + + @property + def url(self) -> str: + return f"http://127.0.0.1:{self.port + 1}" + + @property + def transport(self) -> str: + return f"127.0.0.1:{self.port}" + + def _cli(self, *args: str, timeout: float = 180.0) -> str: + cmd = [str(self.binary), *args, "--home", str(self.home), "--key-id", self.key_id] + got = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout, + env={**os.environ, **self.env}) + if got.returncode != 0: + raise MeshError(f"{self.name}: `{' '.join(args[:3])}` exited {got.returncode}: " + f"{(got.stderr or got.stdout)[-800:]}") + return got.stdout + + def configure(self, dial: List[str]) -> None: + self.home.mkdir(parents=True, exist_ok=True) + self._cli("config", "set", "net.listen_addr", json.dumps(self.transport)) + if dial: + self._cli("config", "set", "net.bootstrap_peers", json.dumps(dial), + "--reason", "native mesh fixture") + + def start(self, health_timeout: float = 120.0) -> None: + self.log_path.parent.mkdir(parents=True, exist_ok=True) + log = open(self.log_path, "ab") + self.proc = subprocess.Popen( + [str(self.binary), "--home", str(self.home), "--key-id", self.key_id], + stdout=log, stderr=subprocess.STDOUT, env={**os.environ, **self.env}, + start_new_session=True) + (self.log_path.parent / "pid").write_text(str(self.proc.pid)) + + def healthy() -> bool: + if self.proc and self.proc.poll() is not None: + raise MeshError(f"{self.name} exited {self.proc.returncode} during boot; " + f"tail of {self.log_path}:\n{self.log_tail(40)}") + return http("GET", f"{self.url}/health", timeout=3)[0] == 200 + + wait_for(f"{self.name} /health", healthy, health_timeout, every=1.0) + + def stop(self) -> None: + if self.proc and self.proc.poll() is None: + try: + os.killpg(self.proc.pid, signal.SIGTERM) + self.proc.wait(timeout=15) + except Exception: # noqa: BLE001 + try: + os.killpg(self.proc.pid, signal.SIGKILL) + except Exception: # noqa: BLE001 + pass + self.proc = None + + def log_tail(self, n: int = 60) -> str: + try: + lines = self.log_path.read_text(errors="replace").splitlines() + except FileNotFoundError: + return "" + return "\n".join(_strip_ansi(line) for line in lines[-n:]) + + def grep(self, pattern: str) -> List[str]: + rx = re.compile(pattern) + try: + text = self.log_path.read_text(errors="replace") + except FileNotFoundError: + return [] + return [_strip_ansi(line) for line in text.splitlines() if rx.search(_strip_ansi(line))] + + # -- HTTP -- + + def api(self, method: str, path: str, body: Any = None, **kw: Any) -> Tuple[int, Any]: + return http(method, self.url + path, self.token, body, **kw) + + def must(self, method: str, path: str, body: Any = None, ok: Tuple[int, ...] = (200, 201), + **kw: Any) -> Any: + status, got = self.api(method, path, body, **kw) + if status not in ok: + raise MeshError(f"{self.name}: {method} {path} answered {status}: {str(got)[:500]}") + return got + + # -- identity -- + + def claim(self, alias: Optional[str] = None, pin_timeout: float = 90.0) -> None: + """Mint (or reuse) an owner identity under `alias` and claim this node + with it over the console claim, as the ladders do.""" + alias = alias or f"{self.key_id}-owner" + pin_file = self.home / "claim_pin" + wait_for(f"{self.name} claim PIN", pin_file.is_file, pin_timeout, every=0.5) + pin = pin_file.read_text(encoding="utf-8").strip() + code = self.must("GET", "/v1/federation/node-code")["code"] + user_seed = self.home / "identity" / "user" / f"{alias}-user.ed25519.seed" + if not user_seed.exists(): + # `--key-id` here is the OWNER's alias, not the node's: not `_cli`. + got = subprocess.run( + [str(self.binary), "identity", "create", "--backend", "software", + "--home", str(self.home), "--key-id", alias], + capture_output=True, text=True, timeout=180, env={**os.environ, **self.env}) + if got.returncode != 0: + raise MeshError(f"{self.name}: identity create exited {got.returncode}: " + f"{(got.stderr or got.stdout)[-800:]}") + out = subprocess.run( + [str(self.binary), "claim", "--backend", "software", "--home", str(self.home), + "--key-id", alias, "--node-code", code, "--claim-pin", pin, + "--cohort-scope", "self", "--target-url", self.url], + capture_output=True, text=True, timeout=180, env={**os.environ, **self.env}) + blob = out.stdout[out.stdout.find("{"):] if "{" in out.stdout else "" + try: + got = json.loads(blob) + except Exception: # noqa: BLE001 + raise MeshError(f"{self.name}: claim printed no JSON (exit {out.returncode}): " + f"{(out.stderr or out.stdout)[-800:]}") + self.token = got.get("access_token") or "" + self.owner_key_id = got.get("identity_key_id") or "" + if not self.token or not self.owner_key_id: + raise MeshError(f"{self.name}: claim returned no token/identity: {str(got)[:400]}") + + def carry_owner_from(self, other: "Node", alias: str) -> None: + """Make this node a SECOND DEVICE of `other`'s owner: carry the owner's + home key material (as the selffiles ladder does, standing in for + `POST /v1/self/associate` with a portable keyset).""" + user_alias = f"{alias}-user" + for sub, name in (("identity/user", f"{user_alias}.ed25519.seed"), + ("identity/user", f"{user_alias}.backend"), + ("identity/keys", f"{user_alias}.mldsa65.seed.blob"), + ("identity/keys", f"{user_alias}.master.key")): + src = other.home / sub / name + if not src.exists(): + raise MeshError(f"{other.name} has no {sub}/{name} to carry") + dst = self.home / sub / name + dst.parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(src, dst) + dst.chmod(0o600) + (self.home / "identity" / "user" / "active_user_alias").write_text(user_alias) + + def announce(self) -> Any: + return self.must("POST", "/v1/federation/announce", {}) + + def self_record(self) -> Any: + rec = self.must("GET", "/v1/federation/self-key-record") + self.record, self.node_key_id = rec, rec["record"]["key_id"] + return rec + + # -- peering / contacts / chat -- + + def peer_with(self, other: "Node", prefixes: Optional[List[str]] = None) -> Any: + """PRODUCTION peering: the other node's self-key-record, admitted here.""" + if other.record is None: + other.self_record() + return self.must("POST", "/v1/federation/peering", { + "peer_key_id": other.record["record"]["key_id"], + "peer_key_record": other.record, + "attestation_prefixes": prefixes or DEFAULT_PREFIXES}) + + def test_admit(self, other: "Node") -> Any: + """The ladders' SHORTCUT (test-anchor only). Opt-in; see the module note.""" + rec = http("GET", f"{other.url}/v1/federation/test-blessed-self-record")[1] + return self.must("POST", "/v1/federation/test-admit-peer", rec) + + def rooted_with(self, other: "Node") -> Optional[bool]: + """Edge's own verdict on this pair, from the log: True once `rooted_with` + found a valid root in common with `other`, False if it last said none, + None if it has not been asked yet.""" + lines = self.grep(rf"rooted_with: .*peer={re.escape(other.key_id)}") + if not lines: + return None + return "a valid root in common" in lines[-1] + + def knows(self, key_id: str) -> bool: + return self.api("GET", f"/v1/federation/peers/{key_id}")[0] == 200 + + def add_contact(self, key_or_code: str) -> Any: + return self.must("POST", "/v1/contacts", {"key_id": key_or_code}) + + def contact_code(self, nodes: str = "all") -> str: + # `nodes` absent = every announced device; a comma list = exactly those; + # `none` = the fed-ID only. "all" is the absent form (a literal `all` is + # read as a node id and refused). + q = "" if nodes in ("all", "", None) else f"?nodes={nodes}" + got = self.must("GET", f"/v1/self/contact-code{q}") + return got.get("code") or got.get("contact_code") or "" + + def open_pair(self, with_key: str) -> str: + got = self.must("POST", "/v1/chat", {"key_id": with_key}) + return got["community_id"] + + def room(self, cid: str) -> Dict[str, Any]: + status, body = self.api("GET", f"/v1/chat/{cid}/messages") + msgs = body.get("messages") if isinstance(body, dict) else None + msgs = msgs or [] + ready = body.get("ready") if isinstance(body, dict) else None + if ready is None and status == 200: + ready = not any(m.get("kind") == "system" for m in msgs) + return {"status": status, "ready": bool(ready), "messages": msgs, + "state": next((m.get("message_id") for m in msgs if m.get("kind") == "system"), None)} + + def say(self, cid: str, text: str) -> str: + got = self.must("POST", f"/v1/chat/{cid}/messages", {"body": text}) + return got.get("attestation_id") or "" + + # -- files -- + + def write_file(self, data: bytes, media_type: str, filename: Optional[str], + cohort: str = "self") -> Dict[str, Any]: + body = {"cohort": cohort, "bytes_base64": base64.b64encode(data).decode(), + "media_type": media_type} + if filename: + body["filename"] = filename + return self.must("POST", "/v1/files", body, timeout=300) + + def read_raw(self, attestation_id: str, cohort: str = "self") -> Tuple[int, bytes]: + return self.api("GET", f"/v1/files/{attestation_id}?cohort={cohort}&raw=1", + raw=True, timeout=300) + + # -- big files: nothing above a few MiB ever sits in this process -- + + def write_file_streamed(self, path: Path, media_type: str, filename: str, + cohort: str = "self", timeout: float = 3600.0) -> Tuple[int, Any]: + """`POST /v1/files` as `multipart/form-data`, the file part streamed from + `path` (the JSON form base64s the bytes — 4/3 inflation and the whole + body in memory; multipart is what the drive names for a large file). + Returns (status, parsed body). + + Since 0.5.218 the node STREAMS this form into the seal, which needs + the file's exact length declared BEFORE its bytes: the `size` field, + sent before the `file` part (every field must precede the file — a + field after it is `400 drive.field_after_file`). A 413 is then the + drive's `STREAMED_FILE_CEILING` (edge's ~2.5 GiB single-file limit), a + ceiling of the DRIVE, not of the wire; a body that is not `size` bytes + is `400 drive.declared_length_mismatch`.""" + boundary = "----ciris-native-" + os.urandom(12).hex() + head = b"" + size = path.stat().st_size + for name, value in (("cohort", cohort), ("media_type", media_type), ("size", str(size))): + head += (f"--{boundary}\r\nContent-Disposition: form-data; name=\"{name}\"\r\n\r\n{value}\r\n").encode() + head += (f"--{boundary}\r\nContent-Disposition: form-data; name=\"file\"; filename=\"{filename}\"\r\n" + f"Content-Type: {media_type}\r\n\r\n").encode() + tail = f"\r\n--{boundary}--\r\n".encode() + + class _Chain: + """A read()-able over head + file + tail; urllib streams it when + Content-Length is set.""" + def __init__(self) -> None: + self.parts = [io.BytesIO(head), open(path, "rb"), io.BytesIO(tail)] + self.i = 0 + def read(self, n: int = -1) -> bytes: + out = b"" + while self.i < len(self.parts) and (n < 0 or len(out) < n): + piece = self.parts[self.i].read(n - len(out) if n >= 0 else -1) + if not piece: + self.parts[self.i].close() + self.i += 1 + continue + out += piece + return out + + headers = {"Content-Type": f"multipart/form-data; boundary={boundary}", + "Content-Length": str(len(head) + size + len(tail))} + if self.token: + headers["Authorization"] = "Bearer " + self.token + req = urllib.request.Request(self.url + "/v1/files", method="POST", data=_Chain(), headers=headers) + try: + with urllib.request.urlopen(req, timeout=timeout) as r: + payload, status = r.read(), r.status + except urllib.error.HTTPError as e: + payload, status = e.read(), e.code + except Exception as e: # noqa: BLE001 + return 0, {"detail": repr(e)[:300]} + try: + return status, json.loads(payload.decode() or "{}") + except Exception: # noqa: BLE001 + return status, {"raw": payload[:300].decode(errors="replace")} + + def read_raw_digest(self, attestation_id: str, cohort: str = "self", + timeout: float = 3600.0) -> Tuple[int, str, int, bytes]: + """Stream `GET /v1/files/{id}?cohort&raw=1` through SHA-256 in 1 MiB + pieces: (status, sha256 hex, byte count, first 300 bytes of a non-200 + body). The bytes are never held.""" + req = urllib.request.Request(f"{self.url}/v1/files/{attestation_id}?cohort={cohort}&raw=1", + headers={"Authorization": "Bearer " + self.token} if self.token else {}) + h, n = hashlib.sha256(), 0 + try: + with urllib.request.urlopen(req, timeout=timeout) as r: + status = r.status + while True: + piece = r.read(1 << 20) + if not piece: + break + h.update(piece) + n += len(piece) + except urllib.error.HTTPError as e: + return e.code, "", 0, e.read()[:300] + except Exception as e: # noqa: BLE001 + return 0, "", n, repr(e)[:300].encode() + return status, h.hexdigest(), n, b"" + + def file_meta(self, attestation_id: str, cohort: str = "self") -> Tuple[int, Any]: + return self.api("GET", f"/v1/files/{attestation_id}/meta?cohort={cohort}", timeout=30) + + def disk_bytes(self) -> int: + """Bytes under this node's home — the pull's progress from the outside, + without asking the node.""" + total = 0 + for root, _dirs, files in os.walk(self.home): + for f in files: + try: + total += os.stat(os.path.join(root, f)).st_size + except OSError: + pass + return total + + def drive(self, cohort: str = "self") -> List[Dict[str, Any]]: + return (self.must("GET", f"/v1/drive?cohort={cohort}&limit=500") or {}).get("entries") or [] + + def trust_roots(self) -> Any: + """`GET /v1/trust-root`: per root this node considers, persist's verdict, + the node's `standing` word and the witnessed `lineage_head` (digest, + instant) — CC T8 (vii)'s per-node line.""" + status, body = self.api("GET", "/v1/trust-root") + return body if status == 200 else {"status": status, "body": str(body)[:200]} + + # -- the node's own database, read-only -- + + def rows(self, sql: str, args: Tuple[Any, ...] = ()) -> List[Tuple[Any, ...]]: + out: List[Tuple[Any, ...]] = [] + for db in glob.glob(str(self.home / "**" / "*.db"), recursive=True): + try: + con = sqlite3.connect(f"file:{db}?mode=ro", uri=True) + out.extend(con.execute(sql, args).fetchall()) + con.close() + except sqlite3.Error: + continue + return out + + +DEFAULT_PREFIXES = ["capacity:", "chat:", "file:", "ownership:", "self:delegates_to:", "trace:"] + +_ANSI = re.compile(r"\x1b\[[0-9;]*m") + + +def _strip_ansi(s: str) -> str: + return _ANSI.sub("", s) + + +# ── the mesh ──────────────────────────────────────────────────────────────── + + +class Mesh: + """A canonical (the synthetic root's holder) plus the nodes a scenario adds. + + `with Mesh(binary, work) as m:` stops every node on exit unless `keep`.""" + + def __init__(self, binary: Path, work: Path, keep: bool = False, + rust_log: str = "info,ciris_edge=debug", extra_env: Optional[Dict[str, str]] = None): + self.binary = Path(binary).resolve() + if not self.binary.is_file(): + raise MeshError(f"no binary at {self.binary}") + self.work = Path(work).resolve() + self.keep = keep + self.base_env = {**anchor_env(), "RUST_LOG": rust_log, **(extra_env or {})} + self.nodes: Dict[str, Node] = {} + self.canonical: Optional[Node] = None + # A port base per WORK DIR, so two meshes run side by side without + # racing for the same pair (measured: `Address already in use`). + import zlib + self._next_port = 7000 + 2 * (zlib.crc32(str(self.work).encode()) % 20000) + + def _node(self, name: str, env: Dict[str, str]) -> Node: + port = free_port_pair(self._next_port) + self._next_port = port + 2 + key_id = f"native-{name}" + return Node(name=name, binary=self.binary, home=self.work / name / "home", + key_id=key_id, port=port, env={**self.base_env, **env}, + log_path=self.work / name / "node.log") + + def start_canonical(self) -> Node: + n = self._node("canonical", {}) + n.env["CIRIS_TEST_BLESS_CANONICAL"] = "true" + n.env["CIRIS_TEST_CANONICAL_DIAL"] = n.transport + n.configure(dial=[]) + n.start() + self.canonical = n + self.nodes["canonical"] = n + return n + + def add(self, name: str, start: bool = True, dial: Optional[List["Node"]] = None) -> Node: + """A node that dials the canonical and, with `dial`, those nodes too — + DIRECT neighbours. Matters: a derived (scope-native) address answers + only a directly-attached neighbour (CC 5.4, CIRISEdge#499), so two + nodes that reach each other only through the canonical cannot fetch + each other's chat bodies or files.""" + if self.canonical is None: + self.start_canonical() + assert self.canonical is not None + n = self._node(name, {"CIRIS_TEST_BLESS_CANONICAL": "false", + "CIRIS_TEST_CANONICAL_DIAL": self.canonical.transport}) + n.configure(dial=[self.canonical.transport] + [d.transport for d in (dial or [])]) + self.nodes[name] = n + if start: + n.start() + return n + + def stop(self) -> None: + for n in self.nodes.values(): + n.stop() + + @staticmethod + def down(work: Path) -> List[str]: + """Stop every node a kept mesh left running under `work` (by pidfile).""" + stopped = [] + for pidfile in Path(work).glob("*/pid"): + try: + pid = int(pidfile.read_text().strip()) + os.killpg(pid, signal.SIGTERM) + stopped.append(f"{pidfile.parent.name}:{pid}") + except (ValueError, ProcessLookupError, PermissionError): + pass + pidfile.unlink(missing_ok=True) + return stopped + + def state(self) -> Dict[str, Any]: + """Everything a person or the next tool needs to keep driving these nodes.""" + return {n: {"url": x.url, "transport": x.transport, "token": x.token, + "owner_key_id": x.owner_key_id, "node_key_id": x.node_key_id, + "home": str(x.home), "log": str(x.log_path)} + for n, x in self.nodes.items()} + + def __enter__(self) -> "Mesh": + if self.work.exists() and not self.keep: + shutil.rmtree(self.work, ignore_errors=True) + self.work.mkdir(parents=True, exist_ok=True) + return self + + def __exit__(self, *exc: Any) -> None: + if not self.keep: + self.stop() diff --git a/harness/native/scenarios.py b/harness/native/scenarios.py new file mode 100644 index 00000000..ad9e3653 --- /dev/null +++ b/harness/native/scenarios.py @@ -0,0 +1,300 @@ +"""Scenarios over the native mesh. Each is a function `(mesh, args) -> dict` +that records what it measured, step by step, and raises `MeshError` at the +first step that cannot proceed — naming it. + +Add one by writing a function and registering it in `SCENARIOS`. +""" +from __future__ import annotations + +import hashlib +import json +import subprocess +import sys +import time +from pathlib import Path +from typing import Any, Callable, Dict, List, Optional + +from .mesh import HARNESS, Mesh, MeshError, Node, wait_for + +Record = Dict[str, Any] + + +class Steps: + """A timeline that EXPLAINS itself: each step says what it proves; a failed + step says what it means and shows the log lines that decide it.""" + + def __init__(self, plan: str) -> None: + self.t0 = time.monotonic() + self.log: List[Record] = [] + self.first_failure: Optional[Record] = None + self.notes: List[str] = [] + print("── PLAN ──\n" + plan.strip() + "\n──", flush=True) + + def __call__(self, name: str, proves: str = "", **seen: Any) -> None: + rec = {"t": round(time.monotonic() - self.t0, 1), "step": name, **seen} + if proves: + rec["proves"] = proves + self.log.append(rec) + print(json.dumps(rec), flush=True) + + def fail(self, name: str, means: str, nodes: List[Node], patterns: str, + layer: str = "", rel: str = "", cc: str = "", **seen: Any) -> None: + """A step that did not happen: what that MEANS, and the evidence — + structured (CIRISClient#134 §5): `layer`, `rel`, `cc` ride as fields + so a runner's report line can say `layer=relations/message cc=CC 5.4.6`.""" + evidence = {n.name: [line[-300:] for line in n.grep(patterns)[-8:]] for n in nodes} + fields = {k: v for k, v in (("layer", layer), ("rel", rel), ("cc", cc)) if v} + self.first_failure = self.first_failure or {"step": name, **fields} + self(name, means=means, evidence=evidence, **fields, **seen) + print(f" ↳ {means}", flush=True) + for n, lines in evidence.items(): + for line in lines: + print(f" {n}: {line}", flush=True) + + +# ── chat between two PEOPLE, production routes (CIRISServer#698) ──────────── + + +def chat(mesh: Mesh, args: Any) -> Record: + """Two people, each on their own node, add each other and key a pair room, + using only the routes a released node exposes. The CIRISServer#698 path.""" + topology = ("bob also dials alice: DIRECT neighbours" if args.direct + else "both dial only the canonical: 2 hops apart, NOT direct neighbours") + step = Steps(f""" +CHAT — two people, each on their own node, one pair room, one message (CSD-091, CIRISServer#698). + trust root: the harness's synthetic anchor; the canonical holds it; no node dials production. + topology: {topology}. Scoped content (chat bodies, files) reaches DIRECTLY-ATTACHED peers + only (CC 5.4.6, CIRISEdge#499) — this choice decides whether bodies can cross. + routes: production only — self-key-record + /v1/federation/peering, /v1/contacts, /v1/chat. + proves, in order: boot → claim → announce → peer → owner keys cross → contacts → the same + room id on both → the room KEYS (the #698 claim) → a body crosses (CSD-091's row).""") + a = mesh.add("alice") + b = mesh.add("bob", dial=[a] if args.direct else None) + step("booted", alice=a.url, bob=b.url, canonical=mesh.canonical.url, topology=topology) + a.claim("alice-person") + b.claim("bob-person") + step("claimed", alice_owner=a.owner_key_id, bob_owner=b.owner_key_id) + for n in (a, b): + n.announce() + n.self_record() + step("announced") + a.peer_with(b) + b.peer_with(a) + step("peered", via="self-key-record + /v1/federation/peering") + # ROOTED is a PAIR fact (CIRISEdge#659): both owners must hold a valid root + # in common, or every attestation between them is withheld — including the + # key grants a body needs. Named here so a red below is never mistaken for + # a chat defect when the precondition is what failed. + try: + wait_for("alice and bob to be Rooted with each other", + lambda: a.rooted_with(b) and b.rooted_with(a), args.rooted_wait, every=3) + step("rooted", proves="edge's rooted_with found a valid root in common, both ways") + except MeshError: + step.fail("NOT_rooted", + "edge sees no valid trust root in common between the two OWNERS (rooted_with walks " + "the owner-bindings): attestations between them are withheld, so key grants and " + "bodies cannot cross. Either the owners never accepted the root (the claim path " + "did not write delegates_to(owner→root)) or the root's charter has not replicated", + [a, b], r"rooted_with|root_binding|trust root|accept.*root|not Rooted", + alice_sees_bob=a.rooted_with(b), bob_sees_alice=b.rooted_with(a)) + + def owner_known(host: Node, guest: Node) -> bool: + return host.knows(guest.owner_key_id) + + for host, guest in ((a, b), (b, a)): + try: + wait_for(f"{host.name} to hold {guest.name}'s owner key", + lambda: owner_known(host, guest), args.owner_wait) + step(f"owner_key_crossed:{host.name}<-{guest.name}") + except MeshError as e: + step(f"owner_key_NOT_crossed:{host.name}<-{guest.name}", detail=str(e)[:200]) + via = getattr(args, "contact_via", "owner") + for host, guest in ((a, b), (b, a)): + key = guest.owner_key_id if via == "owner" else guest.contact_code() + # REACHABLE, not just known. `reachable_nodes=0` means the guest's + # owner→node binding has not reached this node at federation scope yet; + # a contact and room made in that state keyed but every body then read + # `not_granted` with no self-heal (measured 3× on 2026-09-29). The gate + # re-adds until the node is reachable; `--reachable-wait 0` disables it + # to reproduce the race on purpose. + deadline = time.monotonic() + args.reachable_wait + tries = 0 + while True: + got = host.add_contact(key) + tries += 1 + if (got.get("reachable_nodes") or 0) >= 1 or time.monotonic() >= deadline: + break + time.sleep(3) + step(f"contact:{host.name}->{guest.name}", via=via, key=str(got.get("key_id"))[:60], + reachable_nodes=got.get("reachable_nodes"), tries=tries, + proves="the guest's owner→node binding is held here: the room can address their node") + cid_a = a.open_pair(b.owner_key_id) + cid_b = b.open_pair(a.owner_key_id) + step("room_opened", same_room=cid_a == cid_b, room=cid_a) + # THE JOINER ASKS ONCE AND NEVER READS THE ROOM AGAIN (the Docker chat + # ladder's shape, and production's: a person asks for a chat and waits). + # `PairRole` gives the smaller fed-ID the creator's role. Only the CREATOR + # is polled for keying: it reads `ready` only once the joiner's KeyPackage + # arrived, which on the joiner's node is the pair-room driver's work, not + # a read's. Polling both sides used to drive the joiner on every poll and + # hid exactly that (v38 adopt, 2026-10-01). + creator, joiner = (a, b) if a.owner_key_id < b.owner_key_id else (b, a) + cid_c, cid_j = (cid_a, cid_b) if creator is a else (cid_b, cid_a) + + def keyed() -> Any: + rc = creator.room(cid_c) + return rc["ready"] and rc + + try: + wait_for("the pair room to key on the creator with the joiner never reading", keyed, + args.ready_wait, every=3) + step("room_keyed", creator=creator.name, joiner_reads="once (POST /v1/chat)") + except MeshError: + rc = creator.room(cid_c) + step.fail("room_NOT_keyed", + "the MLS handshake did not complete with the joiner never reading: the " + "joiner's acceptance, the creator's widening, the joiner's KeyPackage or the " + "creator's Welcome never happened (the pair-room driver's four acts)", + [a, b], _HANDSHAKE, creator_state=rc["state"]) + raise + text = f"hello from {creator.name} {int(time.time())}" + att = creator.say(cid_c, text) + step("sent", attestation_id=att, by=creator.name) + + def arrived() -> bool: + return any(m.get("attestation_id") == att and m.get("body") == text + for m in joiner.room(cid_j)["messages"]) + + try: + wait_for(f"{creator.name}'s message on {joiner.name}'s node", arrived, args.arrive_wait, + every=2) + except MeshError: + rj = joiner.room(cid_j) + mine = [m for m in rj["messages"] if m.get("attestation_id") == att] + step.fail("body_NOT_arrived", + "the row may be here but its BODY did not open on the joiner: the recipient " + "pulls the body from a holder over the room's DERIVED address, which only a " + "direct neighbour can reach (CC 5.4.6); a `no route to peer … has_path=false` " + "line below is that, `outcome=Stored` means it DID arrive and the transcript " + "is the problem", + [a, b], _BODY, row_on_joiner=mine[:1], transcript_len=len(rj["messages"])) + raise + step("arrived", proves="CSD-091: the peer's message, by attestation id, with its body, on the other person's node") + values = {"PEER_URL": b.url, "PEER_KEY_ID": b.owner_key_id, "PEER_NODE_KEY_ID": b.node_key_id, + "PEER_OWNER_KEY_ID": b.owner_key_id, "LOCAL_OWNER_KEY_ID": a.owner_key_id, + "LOCAL_NODE_KEY_ID": a.node_key_id, "ROOM_ID": cid_a, "MESSAGE_TEXT": text, + "MESSAGE_ATTESTATION_ID": att, "MESSAGE_ARRIVED": "true", "CONTACT_VIA": via} + return {"verdict": "PASS", "steps": step.log, "values": values} + + +_HANDSHAKE = (r"ADVISORY|not conferred|SignedTransportDestination|KeyPackage|Welcome|" + r"withheld|UnknownKey|room not keyed|handshake cannot complete") +_BODY = (r"no route to peer|has_path=|derived destination|holder retired|outcome=(Stored|FetchFailed)|" + r"transcript is empty|not_fetched|NoHolders") + + +# ── one person, two devices, the transfer corpus (the ≥1 MiB question) ────── + + +def corpus(mesh: Mesh, args: Any) -> Record: + """One person on two devices; the corpus written on the first, read raw on + BOTH, compared byte for byte with the originals.""" + topology = ("the second device also dials the first: DIRECT neighbours" if args.direct + else "both devices dial only the canonical: NOT direct neighbours") + step = Steps(f""" +CORPUS — one person, two devices, the transfer corpus written on the first and read raw on BOTH + (CSD-007 files / CSD-008 notes on a second device; the ≥1 MiB question from the selffiles ladder). + trust root: synthetic anchor; canonical holds it. topology: {topology}. + proves: same owner on both → peered → self room on both → files written (first reads them back + byte-identical) → the second device pulls and opens each one byte-identical.""") + a = mesh.add("first") + a.claim("one-person") + b = mesh.add("second", start=False, dial=[a] if args.direct else None) + b.home.mkdir(parents=True, exist_ok=True) + b.carry_owner_from(a, "one-person") + b.start() + b.claim("one-person") + step("claimed", same_owner=a.owner_key_id == b.owner_key_id, owner=a.owner_key_id) + for n in (a, b): + n.announce() + n.self_record() + a.peer_with(b) + b.peer_with(a) + step("peered") + wait_for("the self room on both devices", + lambda: a.grep(r"self room (CREATED|JOINED)") and b.grep(r"self room (CREATED|JOINED)"), + args.ready_wait, every=3) + step("self_room") + + out = mesh.work / "corpus" + subprocess.run([sys.executable, str(HARNESS / "mesh-repro" / "lib" / "media_corpus.py"), str(out)], + check=True, capture_output=True) + manifest = json.loads((out / "manifest.json").read_text(encoding="utf-8")) + only = set(args.only.split(",")) if getattr(args, "only", "") else None + written = [] + for row in manifest: + if only and row["name"] not in only: + continue + data = (out / row["name"]).read_bytes() + got = a.write_file(data, row["media_type"], row["filename"]) + written.append({**row, "id": got["attestation_id"]}) + step("written", files=len(written)) + + def compare(n: Node, row: Record) -> Record: + status, raw = n.read_raw(row["id"]) + want = (out / row["name"]).read_bytes() + r: Record = {"status": status, "size": len(raw) if status == 200 else None, + "expected": len(want)} + if status == 200: + r["match"] = hashlib.sha256(raw).hexdigest() == row["sha256"] + if not r["match"]: + m = min(len(raw), len(want)) + first = next((i for i in range(m) if raw[i] != want[i]), None) + r.update(first_diff=first if first is not None else m, + prefix=first is None and len(raw) < len(want), + differing=sum(1 for i in range(m) if raw[i] != want[i])) + else: + r["body"] = raw[:200].decode(errors="replace") + return r + + results: Record = {} + deadline = time.monotonic() + args.arrive_wait + pending = {w["name"]: w for w in written} + for w in written: + results[w["name"]] = {"first": compare(a, w)} + while pending and time.monotonic() < deadline: + for name, w in list(pending.items()): + r = compare(b, w) + results[name]["second"] = r + if r["status"] == 200: + del pending[name] + if pending: + time.sleep(3) + bad = {k: v for k, v in results.items() + if not v["first"].get("match") or not v.get("second", {}).get("match")} + step("compared", ok=len(results) - len(bad), bad=len(bad), + proves="each file's bytes on the first device, and on the second, against the original") + if bad: + no_row = [k for k, v in bad.items() if v.get("second", {}).get("status") == 404] + not_here = [k for k, v in bad.items() if v.get("second", {}).get("status") == 409] + wrong = [k for k, v in bad.items() if v.get("second", {}).get("status") == 200] + means = [] + if no_row: + means.append(f"{len(no_row)} ROWS never reached the second device (404 not_in_room: the row itself is " + "not held there, so no pull was attempted) — replication of the attestation, not the bytes; " + "look for `stalled mid-frame` / `REFUSED` on the sender's rounds toward it") + if not_here: + means.append(f"{len(not_here)} never reached the second device (not_fetched): the pull over the " + "room's derived address failed — a `no route to peer` / `holder retired` line " + "is a topology fault (direct neighbours needed, CC 5.4.6), `NoHolders` is the pull's source rule") + if wrong: + means.append(f"{len(wrong)} arrived with DIFFERENT bytes: a transfer defect; `first_diff`/`prefix` say where") + step.fail("corpus_mismatch", "; ".join(means), [a, b], + _BODY + r"|stalled mid-frame|delivered envelope REFUSED|frame DROPPED", files=bad) + return {"verdict": "PASS" if not bad else "FAIL", "steps": step.log, "results": results} + + +SCENARIOS: Dict[str, Callable[[Mesh, Any], Record]] = { + "chat": chat, + "corpus": corpus, +} diff --git a/harness/native/topologies/bigfile-quick.yaml b/harness/native/topologies/bigfile-quick.yaml new file mode 100644 index 00000000..e0046968 --- /dev/null +++ b/harness/native/topologies/bigfile-quick.yaml @@ -0,0 +1,19 @@ +# bigfile.yaml's quick form: 256 MiB, no resume rung. Same ceilings apply +# (drive UPLOAD_BODY_LIMIT today; edge v34.4.0 DAG pull; #737 range reader). +topology: + roots: + - {id: R, kind: key, holders: 1, custody: software_test, lifecycle: {recipe: {}, verdict: rooted}} + canonicals: + - {id: C, holds: R, serves: [infra:serve, infra:attest]} + nodes: + - {id: D1, dials: [C], accepts: R, announced: true} + - {id: D2, dials: [C, D1], accepts: R, announced: true} + persons: + - {id: one, owns: [D1, D2], accepts: R} + relations: + - {rel: peered, between: [D1, D2]} + - {rel: room, kind: self, person: one} + - {rel: bigfile, person: one, device: D1, size: 268435456, wait: 1800} + actor: {person: one, device: D2} + negatives: + - {check: no_wider_self_rows, person: one} diff --git a/harness/native/topologies/bigfile.yaml b/harness/native/topologies/bigfile.yaml new file mode 100644 index 00000000..67f9c7a8 --- /dev/null +++ b/harness/native/topologies/bigfile.yaml @@ -0,0 +1,36 @@ +# One BIG self file, end to end, timed (CIRISEdge#734 lane 7 — "a multi-GiB +# file transfer, happy and fast"): ONE person on TWO devices under the synthetic +# root; a 2 GiB file streamed into D1's drive, pulled by D2 over the direct link +# (a self room's scoped bytes are one-hop, CC 5.4.6), read back on D2 through a +# streamed `?raw=1` and compared by SHA-256; D2 is stopped at half the pull and +# restarted, and the pull must still finish. Publish, pull and read rates are +# RECORDED per run, never asserted — a floor belongs in the CSD. +# +# What must land for this to be green, by name: +# * the server's `drive.rs` UPLOAD_BODY_LIMIT (64 MiB whole-read cap ×4/3 + +# 1 MiB) — today any upload above it answers 413; the step names it +# (`bigfile_REFUSED_BY_DRIVE`). A streamed upload above the cap is the +# server's lane. +# * edge v34.4.0 (PR #733, persist v51.3.0): the chunk-DAG pull — until it +# lands a ≥1 MiB file on the second device is a manifest, refused by name +# (CIRISEdge#717, `drive.seal_mismatch` / 409 `drive.not_fetched`). +# * edge #737: the range reader, so the drive streams a raw read above 64 MiB +# instead of whole-reading. +# Quick form: bigfile-quick.yaml (256 MiB, no resume). +topology: + roots: + - {id: R, kind: key, holders: 1, custody: software_test, lifecycle: {recipe: {}, verdict: rooted}} + canonicals: + - {id: C, holds: R, serves: [infra:serve, infra:attest]} + nodes: + - {id: D1, dials: [C], accepts: R, announced: true} + - {id: D2, dials: [C, D1], accepts: R, announced: true} # the puller dials the author: one-hop bytes + persons: + - {id: one, owns: [D1, D2], accepts: R} + relations: + - {rel: peered, between: [D1, D2]} + - {rel: room, kind: self, person: one} + - {rel: bigfile, person: one, device: D1, size: 2147483648, resume_at: 0.5, wait: 7200} + actor: {person: one, device: D2} + negatives: + - {check: no_wider_self_rows, person: one} diff --git a/harness/native/topologies/csd-091-user-chat.yaml b/harness/native/topologies/csd-091-user-chat.yaml new file mode 100644 index 00000000..a685c8c0 --- /dev/null +++ b/harness/native/topologies/csd-091-user-chat.yaml @@ -0,0 +1,27 @@ +# CSD-091 — user chat: two people, each on their own node, one pair room, one message. +# FSD/TOPOLOGY.md. Everything a released node exposes; the synthetic root above it. +topology: + roots: + - {id: R, kind: key, holders: 1, custody: software_test, lifecycle: {recipe: {}, verdict: rooted}} + canonicals: + - {id: C, holds: R, serves: [infra:serve, infra:attest]} + nodes: + - {id: A, dials: [C], accepts: R, announced: true} + - {id: B, dials: [C, A], accepts: R, announced: true} # direct neighbour of A: bodies are one-hop (CC 5.4.6) + - {id: X, dials: [C], accepts: R, announced: false} # the outsider + persons: + - {id: alice, owns: [A], accepts: R} + - {id: bob, owns: [B], accepts: R} + - {id: eve, owns: [X], accepts: R} + relations: + - {rel: peered, between: [A, B]} + - {rel: rooted_with, between: [alice, bob], require: false} # observed: the room keys without it today + - {rel: reachable, node: A, person: bob, min: 1} # CIRISServer#699 + - {rel: reachable, node: B, person: alice, min: 1} + - {rel: contact, from: alice, to: bob, via: owner} + - {rel: contact, from: bob, to: alice, via: owner} + - {rel: room, id: pair, kind: pair, members: [alice, bob], keyed: true} + - {rel: message, from: bob, to: alice, room: pair, text: "hello from bob"} + actor: {person: alice, device: A} + negatives: + - {check: cannot_list_room, person: eve, room: pair} diff --git a/harness/native/topologies/csd-092-share-contact-code.yaml b/harness/native/topologies/csd-092-share-contact-code.yaml new file mode 100644 index 00000000..82c4d14f --- /dev/null +++ b/harness/native/topologies/csd-092-share-contact-code.yaml @@ -0,0 +1,20 @@ +# CSD-092 — share my contact code: bob's code names his ANNOUNCED device; alice adds him by it. +topology: + roots: + - {id: R, kind: key, holders: 1, custody: software_test, lifecycle: {recipe: {}, verdict: rooted}} + canonicals: + - {id: C, holds: R, serves: [infra:serve, infra:attest]} + nodes: + - {id: A, dials: [C], accepts: R, announced: true} + - {id: B, dials: [C, A], accepts: R, announced: true} + persons: + - {id: alice, owns: [A], accepts: R} + - {id: bob, owns: [B], accepts: R} + relations: + - {rel: peered, between: [A, B]} + - {rel: reachable, node: A, person: bob, min: 1} + - {rel: contact, from: alice, to: bob, via: code, nodes: all} # GET /v1/self/contact-code on B (0.5.218, #673) + - {rel: reachable, node: B, person: alice, min: 1} + - {rel: contact, from: bob, to: alice, via: owner} + - {rel: room, id: pair, kind: pair, members: [alice, bob], keyed: true} + actor: {person: alice, device: A} diff --git a/harness/native/topologies/csd-094-approve-new-device.yaml b/harness/native/topologies/csd-094-approve-new-device.yaml new file mode 100644 index 00000000..162a44f7 --- /dev/null +++ b/harness/native/topologies/csd-094-approve-new-device.yaml @@ -0,0 +1,23 @@ +# CSD-094 — approve a new device: the POST-STATE (the client's reading, CIRISClient#134): the owner +# carried onto the second device, both claimed, one self room, a file written on the first opens on +# the second. CSD-093 (the approval-code flow itself) gets its own declaration once CIRISServer#678 +# answers which address the code carries. The file relation is where CIRISEdge#716/#717 sit. +topology: + roots: + - {id: R, kind: key, holders: 1, custody: software_test, lifecycle: {recipe: {}, verdict: rooted}} + canonicals: + - {id: C, holds: R, serves: [infra:serve, infra:attest]} + nodes: + - {id: D1, dials: [C], accepts: R, announced: true} + - {id: D2, dials: [C, D1], accepts: R, announced: true} # the phone dials the first device directly + - {id: X, dials: [C], accepts: R, announced: false} + persons: + - {id: one, owns: [D1, D2], accepts: R} + - {id: eve, owns: [X], accepts: R} + relations: + - {rel: peered, between: [D1, D2]} + - {rel: room, kind: self, person: one} + - {rel: file, person: one, device: D1, size: 4096, name: small.bin} + actor: {person: one, device: D2} + negatives: + - {check: holds_no_row, node: X, dimension: "file:v1", person: one} diff --git a/harness/native/topologies/csd-107-file-custody.yaml b/harness/native/topologies/csd-107-file-custody.yaml new file mode 100644 index 00000000..f168b50c --- /dev/null +++ b/harness/native/topologies/csd-107-file-custody.yaml @@ -0,0 +1,24 @@ +# CSD-107 — where each file is (the file-custody view, FSD/FILE_CUSTODY.md): ONE person on TWO +# devices, a 24 MiB self file (a chunk DAG, so it carries delivery receipts — CC 5.3.3.6) written on +# the first and opened on the second, then the author device's `GET /v1/files/{id}/custody` must +# name the second device `received` and count both devices. The client's CSD drives the custody +# screen from the actor's (D1's) point of view. An inline (<= 1 MiB) file has no receipt at this +# pin (CIRISPersist#953) and would fail the custody relation by name. +topology: + roots: + - {id: R, kind: key, holders: 1, custody: software_test, lifecycle: {recipe: {}, verdict: rooted}} + canonicals: + - {id: C, holds: R, serves: [infra:serve, infra:attest]} + nodes: + - {id: D1, dials: [C], accepts: R, announced: true} + - {id: D2, dials: [C, D1], accepts: R, announced: true} # the second device dials the author: one-hop bodies + persons: + - {id: one, owns: [D1, D2], accepts: R} + relations: + - {rel: peered, between: [D1, D2]} + - {rel: room, kind: self, person: one} + - {rel: file, person: one, device: D1, size: 25165824, name: custody-24MiB.bin} + - {rel: custody, person: one, device: D1, file: last} + actor: {person: one, device: D1} + negatives: + - {check: no_wider_self_rows, person: one} diff --git a/harness/native/topologies/selffiles.yaml b/harness/native/topologies/selffiles.yaml new file mode 100644 index 00000000..554d208b --- /dev/null +++ b/harness/native/topologies/selffiles.yaml @@ -0,0 +1,30 @@ +# The selffiles ladder as a topology (harness/mesh-repro/scenarios/selffiles.sh, 12 rungs): +# ONE person on TWO devices, a second person peered as a contact (the believable outsider), +# the device roster, a note, the transfer corpus, and the two negatives. Runs in minutes instead of 45. +topology: + roots: + - {id: R, kind: key, holders: 1, custody: software_test, lifecycle: {recipe: {}, verdict: rooted}} + canonicals: + - {id: C, holds: R, serves: [infra:serve, infra:attest]} + nodes: + - {id: D1, dials: [C], accepts: R, announced: true} + - {id: D2, dials: [C, D1], accepts: R, announced: true} # the phone dials the first device: one-hop bodies + - {id: X, dials: [C, D1], accepts: R, announced: true} # the contact's node + persons: + - {id: one, owns: [D1, D2], accepts: R} + - {id: two, owns: [X], accepts: R} + relations: + - {rel: peered, between: [D1, D2]} + - {rel: peered, between: [D1, X]} # a contact, not a device + - {rel: reachable, node: D1, person: two, min: 1} + - {rel: contact, from: one, to: two, via: owner} + - {rel: room, kind: self, person: one} + - {rel: roster, person: one, visible_from: [X], visible_from_complete: true} # CSD-037 + CC 5.4.6: X lists EVERY announced device (the canonical relays D2) + - {rel: note, person: one, device: D1} + - {rel: session, person: one, device: D1} # CC 3.1.3.1: one device handles each exchange + - {rel: corpus, person: one, device: D1, require: false} # recorded per file; the negatives still run + - {rel: custody, person: one, file: large} # FSD/FILE_CUSTODY.md: D1's custody names D2 received + actor: {person: one, device: D2} + negatives: + - {check: no_wider_self_rows, person: one} + - {check: holds_no_row, node: X, dimension: "file:v1", person: one} diff --git a/harness/native/topology.py b/harness/native/topology.py new file mode 100644 index 00000000..08a22312 --- /dev/null +++ b/harness/native/topology.py @@ -0,0 +1,929 @@ +"""Build the topology a use case declares (FSD/TOPOLOGY.md), from the root down. + + python -m harness.native build --topology harness/native/topologies/csd-091-user-chat.yaml \\ + --binary target/debug/ciris-server + +Three verbs on one declaration: `load` (parse + realizability rules), `derive` +(node count, root ceremony, build order — what the CSD checker consumes) and +`build` (stand it up layer by layer on a `Mesh`, checking each predicate and +naming the first layer that does not hold). +""" +from __future__ import annotations + +import hashlib +import json +import sys +import time +from pathlib import Path +from typing import Any, Dict, List, Optional, Tuple + +import yaml + +from .mesh import Mesh, MeshError, Node, wait_for +from .scenarios import Steps, _BODY, _HANDSHAKE + +Decl = Dict[str, Any] + +# What the synthetic anchor mints today (src/test_bless.rs): ONE key root, ONE +# software holder, lifecycle active. Everything else is declared and refused. +BUILDABLE_ROOT = {"kind": "key", "holders": 1, "custody": "software_test", + "lifecycle": {"recipe": {}, "verdict": "rooted"}} + +LAYERS = ["roots", "canonicals", "nodes", "persons", "relations", "actor", "negatives"] +LAST_STEPS: Any = None # the running build's Steps, so a failure can name its first failing layer + + +class Unrealizable(ValueError): + """The declaration cannot hold, by rule number (FSD/TOPOLOGY.md §3).""" + + +# ── load + rules ──────────────────────────────────────────────────────────── + + +def load(path: Path) -> Decl: + doc = yaml.safe_load(Path(path).read_text(encoding="utf-8")) + t = doc.get("topology", doc) + for k in ("roots", "canonicals", "nodes", "persons", "relations", "actor"): + if k not in t: + raise Unrealizable(f"missing layer `{k}`") + check(t) + return t + + +def _ids(t: Decl, layer: str) -> Dict[str, Decl]: + return {x["id"]: x for x in t[layer]} + + +def check(t: Decl) -> List[str]: + """Realizability (§3). Returns the list of what the builder cannot mint yet + (`buildable: false` items); raises on a contradiction.""" + roots, canon, nodes, persons = (_ids(t, k) for k in ("roots", "canonicals", "nodes", "persons")) + not_yet: List[str] = [] + for r in roots.values(): + if r.get("kind") == "family": + raise Unrealizable(f"root {r['id']}: kind `family` (the accord family, CC 4.2.6) has no model " + "behind it and is not a trust root the harness can mint — use `key` or `infrastructure`") + if r.get("kind") == "infrastructure": + f, m = r.get("founders", {}), r.get("quorum", 0) + if f.get("seated", 0) < m + 1: + raise Unrealizable(f"rule 1: root {r['id']} quorum {m} needs founders.seated >= {m + 1} (T7)") + if f.get("conferrable", 0) < 1: + raise Unrealizable(f"rule 1: root {r['id']} needs founders.conferrable >= 1, or T7 recovery has nobody to widen in") + w = r.get("witnesses") + if w: + if "k" in w: + raise Unrealizable(f"rule 1: root {r['id']}: `k` is derived (CC T6: K = floor(n/2)+1), not declared") + if not w.get("independent_custody"): + raise Unrealizable(f"rule 1: root {r['id']} witnesses need independent_custody") + lc = r.get("lifecycle") + if isinstance(lc, str): + raise Unrealizable(f"root {r['id']}: lifecycle is a row recipe + asserted verdict, e.g. " + "{recipe: {resignations: 1}, verdict: stalled}, not the flag {lc!r}") + for k, v in BUILDABLE_ROOT.items(): + if r.get(k, v) != v: + not_yet.append(f"root {r['id']}: {k}={r.get(k)!r} (the anchor mints {v!r})") + for c in canon.values(): + if c["holds"] not in roots: + raise Unrealizable(f"canonical {c['id']} holds unknown root {c['holds']}") + for n in nodes.values(): + for d in n.get("dials", []): + if d not in nodes and d not in canon: + raise Unrealizable(f"node {n['id']} dials unknown {d}") + if n.get("accepts") and n["accepts"] not in roots: + raise Unrealizable(f"node {n['id']} accepts unknown root {n['accepts']}") + for p in persons.values(): + for nid in p["owns"]: + if nid not in nodes: + raise Unrealizable(f"person {p['id']} owns unknown node {nid}") + if p.get("accepts") and nodes[nid].get("accepts") != p["accepts"]: + raise Unrealizable(f"rule 2: {p['id']} accepts {p['accepts']} but owns {nid} which does not") + owner_of = {nid: p["id"] for p in persons.values() for nid in p["owns"]} + for rel in t["relations"]: + kind = rel["rel"] + if kind == "rooted_with": + a, b = (persons[x] for x in rel["between"]) + if not (a.get("accepts") and a.get("accepts") == b.get("accepts")): + raise Unrealizable(f"rule 3: rooted_with({a['id']},{b['id']}) needs a common accepted root") + if kind in ("message", "file"): + src = rel.get("from") or rel["person"] + dsts = [rel["to"]] if kind == "message" else [q for q in persons if q == src] + for q in dsts: + for na in persons[src]["owns"]: + for nb in persons[q]["owns"]: + if na != nb and nb not in nodes[na].get("dials", []) and na not in nodes[nb].get("dials", []): + raise Unrealizable(f"rule 4: {kind} {src}->{q}: {na} and {nb} are not direct neighbours " + f"(scoped content is one-hop, CC 5.4.6); add a `dials`") + if kind == "reachable": + q = rel["person"] + if not any(nodes[n].get("announced") for n in persons[q]["owns"]): + raise Unrealizable(f"rule 5: reachable(..., {q}) needs one of {q}'s nodes announced") + if kind == "contact" and rel.get("via") == "code": + q = rel["to"] + if not any(nodes[n].get("announced") for n in persons[q]["owns"]): + raise Unrealizable(f"rule 6: contact via code needs {q}'s node announced") + t.setdefault("_owner_of", owner_of) + return not_yet + + +def _root_derived(r: Decl) -> Decl: + """The model constants CC derives from the block (CIRISConstitution#131).""" + out: Decl = {} + w = r.get("witnesses") or {} + if w.get("n"): + out["K"] = w["n"] // 2 + 1 + c = r.get("charter") or {} + if c.get("attach_window_secs") and c.get("witness_cadence_secs"): + out["AttachWindow"] = -(-c["attach_window_secs"] // c["witness_cadence_secs"]) + f = r.get("founders") or {} + if f: + out["Founders"] = f.get("seated", 0) + f.get("conferrable", 0) + out["NodeKeys"] = f.get("node_bearing", 0) + return out + + +def derive(t: Decl) -> Decl: + persons = _ids(t, "persons") + return { + "nodes": len(t["nodes"]) + len(t["canonicals"]), + "persons": len(persons), + "devices": {p: len(v["owns"]) for p, v in persons.items()}, + "roots": [{**{k: r.get(k) for k in ("id", "kind", "holders", "founders", "quorum", "witnesses", "charter", "lifecycle", "custody")}, + "derived": _root_derived(r)} for r in t["roots"]], + "direct_links": [(n["id"], d) for n in t["nodes"] for d in n.get("dials", []) if d not in _ids(t, "canonicals")], + "build_order": LAYERS, + "not_buildable_yet": check(t), + } + + +# ── build ─────────────────────────────────────────────────────────────────── + + +def export_rows(mesh: Mesh) -> Decl: + """Per node: every admitted federation row, in admission order, as JSONL — + the fold-replayable form (FSD/TOPOLOGY.md §4.1).""" + out: Decl = {} + for name, n in mesh.nodes.items(): + rows = n.rows("select * from federation_attestations order by rowid") + cols = n.rows("select name from pragma_table_info('federation_attestations') order by cid") + colnames = [c[0] for c in cols] + path = n.log_path.parent / "rows.jsonl" + h = hashlib.sha256() + with open(path, "w", encoding="utf-8") as f: + for row in rows: + line = json.dumps(dict(zip(colnames, row)), default=str, sort_keys=True) + f.write(line + "\n") + h.update(line.encode()) + # The signed routes this node holds (persist `transport_destinations`, + # one row per (occurrence, transport_kind); CIRISEdge#722): the + # #393 item-2 gate reads attesting_key_id / signed_envelope / signature. + routes = n.rows("select occurrence_key_id, transport_kind, attesting_key_id, " + "signed_envelope is not null, signature is not null, " + "length(signed_envelope), length(signature) from transport_destinations") + (n.log_path.parent / "transport_destinations.jsonl").write_text( + "\n".join(json.dumps(dict(zip(("occurrence_key_id", "transport_kind", "attesting_key_id", + "signed_envelope_present", "signature_present", + "signed_envelope_len", "signature_len"), r)), default=str) for r in routes), + encoding="utf-8") + # The node's metrics snapshot (GET /v1/federation/metrics): edge's + # counters — blob_pull_sources, blob_pull_refusals, + # channel_first_skipped_over_cap (CIRISEdge#722) — as the node last + # reported them, beside the rows they explain. + try: + st_, metrics = n.api("GET", "/v1/federation/metrics", timeout=20) + (n.log_path.parent / "metrics.json").write_text(json.dumps(metrics if st_ == 200 else {"status": st_}, indent=1, default=str), encoding="utf-8") + except Exception as e: # noqa: BLE001 + (n.log_path.parent / "metrics.json").write_text(json.dumps({"error": str(e)[:200]}), encoding="utf-8") + keys = n.rows("select key_id, identity_type, valid_from, valid_until, scrub_key_id from federation_keys") + (n.log_path.parent / "keys.jsonl").write_text( + "\n".join(json.dumps(dict(zip(("key_id", "identity_type", "valid_from", "valid_until", "scrub_key_id"), k)), default=str) for k in keys), + encoding="utf-8") + # CC's two per-node additions (CIRISConstitution#131): the standing + # verdict for each root, and the witnessed head digest + instant. + tr = n.trust_roots() if n.token or True else {} + roots = tr.get("roots") if isinstance(tr, dict) else None + standing = {r.get("root_key_id"): { + "standing": (r.get("verdict") or {}).get("standing"), + "valid": (r.get("verdict") or {}).get("valid"), + "accepted": r.get("accepted"), + "lineage_head": (r.get("verdict") or {}).get("lineage_head"), + } for r in (roots or [])} if roots else {"unreadable": tr} + out[name] = {"rows": len(rows), "keys": len(keys), "routes": len(routes), "sha256": h.hexdigest(), "path": str(path), + "roots": standing} + return out + + +def build(mesh: Mesh, t: Decl, args: Any) -> Decl: + try: + return _build(mesh, t, args) + except MeshError as e: + ff = getattr(LAST_STEPS, "first_failure", None) if LAST_STEPS else None + raise MeshError(json.dumps({"error": str(e), "first_failing_layer": ff})) from e + finally: + try: + exported = export_rows(mesh) + print(json.dumps({"step": "rows_exported", "layer": "export", **{k: {kk: vv for kk, vv in v.items() if kk != "path"} for k, v in exported.items()}}), flush=True) + (mesh.work / "rows_export.json").write_text(json.dumps(exported, indent=1), encoding="utf-8") + except Exception as e: # noqa: BLE001 — the export must never mask the verdict + print(json.dumps({"step": "rows_export_failed", "error": str(e)[:200]}), flush=True) + + +def _build(mesh: Mesh, t: Decl, args: Any) -> Decl: + d = derive(t) + if d["not_buildable_yet"]: + raise MeshError("declared but not buildable yet: " + "; ".join(d["not_buildable_yet"])) + persons = _ids(t, "persons") + owner_of = t["_owner_of"] + plan = "\n".join( + [f"TOPOLOGY {args.topology}", f" roots: {d['roots']}", + f" nodes: {d['nodes']} ({len(t['canonicals'])} canonical), persons: {d['persons']}, devices: {d['devices']}", + f" direct links: {d['direct_links'] or 'none — every pair is relayed through a canonical'}", + " build order: " + " → ".join(LAYERS)]) + step = Steps(plan) + global LAST_STEPS + LAST_STEPS = step + N: Dict[str, Node] = {} + + # 1–2. roots + canonicals: the synthetic anchor's ceremony, one canonical. + for c in t["canonicals"]: + N[c["id"]] = mesh.start_canonical() + step("root+canonical", layer="roots/canonicals", canonical=c["id"], holds=c["holds"], + proves="the anchor's ceremony ran: charter root→root, grant root→node, trust edge node→root", + evidence=N[c["id"]].grep(r"TEST-ANCHOR ceremony: trust_root_valid GREEN")[-1:]) + + # 3. nodes, in an order that lets `dials` resolve. + pending = [n for n in t["nodes"]] + while pending: + progressed = False + for n in list(pending): + deps = [x for x in n.get("dials", []) if x not in N] + if deps: + continue + N[n["id"]] = mesh.add(n["id"], dial=[N[x] for x in n.get("dials", []) if x in _ids(t, "nodes")]) + pending.remove(n) + progressed = True + if not progressed: + raise MeshError(f"nodes dial each other in a cycle: {[n['id'] for n in pending]}") + step("nodes", layer="nodes", up={n: N[n].url for n in _ids(t, "nodes")}) + + # 4. persons: first node claimed with a minted identity; the rest carried. + for p in t["persons"]: + first, *rest = p["owns"] + alias = f"{p['id']}-person" + N[first].claim(alias) + for nid in rest: + n = N[nid] + n.stop() + n.carry_owner_from(N[first], alias) + n.start() + n.claim(alias) + owners = {nid: N[nid].owner_key_id for nid in p["owns"]} + if len(set(owners.values())) != 1: + raise MeshError(f"{p['id']}'s devices claimed different owners: {owners}") + step(f"person:{p['id']}", layer="persons", owner=N[first].owner_key_id, devices=p["owns"], + proves="one owner key across every declared device") + for n in t["nodes"]: + if n.get("announced"): + N[n["id"]].announce() + N[n["id"]].self_record() + step("announced", layer="nodes", announced=[n["id"] for n in t["nodes"] if n.get("announced")]) + # A claimed node that dials a canonical also PEERS with it — the owner's + # HTTP act in production (`POST /v1/federation/peering` with the canonical's + # record, traceflow_prod.sh step 7, after claim + announce; the baked seed + # primes the canonical's key on a production node, + # `compose::prime_canonical_bootstrap_peers`). Dialling alone makes the + # canonical a Reticulum relay and nothing more: no round ever targets an + # unkeyed peer, so the canonical admits no node key, roots with nobody and + # distributes nothing (edge's v34.3.0 run: the canonical's keys.jsonl held + # only its own key; every relation rode the direct links). + peered_canonicals = [] + for n in t["nodes"]: + for c in n.get("dials", []): + if c in _ids(t, "canonicals"): + N[n["id"]].peer_with(N[c]) + peered_canonicals.append((n["id"], c)) + step("canonical_peered", layer="nodes", pairs=peered_canonicals, + proves="every claimed node holds the record of the canonical it dials, as the owner's peering act leaves it in production") + + # 5. relations, in declared order. + values: Dict[str, str] = {} + rooms: Dict[str, str] = {} + added: set = set() # (host node id, guest person) pairs already POSTed by a `reachable` gate + # Files the relations wrote, for a later `custody` relation: `last` is the + # most recent `file` (or the last file a `corpus` wrote); a corpus file is + # also named by its manifest name (`large`, `inline_over`, ...). Each entry + # records the device it was written on — the AUTHOR device, the one whose + # store admits every delivery receipt. + files_written: Dict[str, Dict[str, Any]] = {} + for rel in t["relations"]: + k = rel["rel"] + if k == "peered": + a, b = rel["between"] + N[a].peer_with(N[b], rel.get("prefixes")) + N[b].peer_with(N[a], rel.get("prefixes")) + step(f"peered:{a}<->{b}", layer="relations", proves="production peering both ways") + elif k == "rooted_with": + p, q = rel["between"] + na, nb = N[persons[p]["owns"][0]], N[persons[q]["owns"][0]] + ok = False + try: + wait_for(f"rooted_with({p},{q})", lambda: na.rooted_with(nb) and nb.rooted_with(na), + float(rel.get("wait", 0)) or 1, every=3) + ok = True + except MeshError: + pass + if ok: + step(f"rooted_with:{p}<->{q}", layer="relations", proves="edge found a valid root in common, both ways") + elif rel.get("require"): + step.fail(f"NOT_rooted:{p}<->{q}", "the two OWNERS hold no valid root in common (rooted_with walks the " + "owner-bindings): attestations between them are withheld", [na, nb], + r"rooted_with|root_binding|accept.*root|not Rooted", + layer="relations", rel="rooted_with", cc="CC 3.2 / CIRISEdge#659") + raise MeshError("required rooted_with did not hold") + else: + step(f"rooted_with:{p}<->{q}", layer="relations", observed=False, + note="not required by this declaration; recorded") + elif k == "reachable": + host, q = N[rel["node"]], persons[rel["person"]] + target = N[q["owns"][0]].owner_key_id + deadline = time.monotonic() + float(rel.get("wait", 120)) + tries, got = 0, {} + while True: + # Known comes before reachable: a 404 (the guest's key has not + # crossed yet) is "not yet", the same as reachable_nodes=0. + status, got = host.api("POST", "/v1/contacts", {"key_id": target}) + got = got if isinstance(got, dict) else {} + tries += 1 + if status == 200 and (got.get("reachable_nodes") or 0) >= int(rel.get("min", 1)): + break + if status not in (200, 404) or time.monotonic() >= deadline: + break + time.sleep(3) + if (got.get("reachable_nodes") or 0) < int(rel.get("min", 1)): + step.fail(f"NOT_reachable:{rel['node']}->{rel['person']}", + f"{rel['person']}'s owner→node binding is not held on {rel['node']} at federation scope " + "(CIRISServer#699): a room made now keys but its bodies read not_granted", + [host], r"handshake cannot complete|resolves to no node|reachable", + layer="relations", rel="reachable", cc="CC 5.2") + raise MeshError("reachable gate did not hold") + added.add((rel["node"], rel["person"])) + step(f"reachable:{rel['node']}->{rel['person']}", layer="relations", reachable_nodes=got.get("reachable_nodes"), + tries=tries, proves="the guest's binding is held here; the room can address their node") + elif k == "contact": + p, q = rel["from"], rel["to"] + host, guest = N[persons[p]["owns"][0]], N[persons[q]["owns"][0]] + if rel.get("via", "owner") == "owner" and (persons[p]["owns"][0], q) in added: + # The `reachable` gate's POST IS the contact. A second POST for a + # live contact re-issues its grant, and three builds that did so + # never keyed the room afterwards (2026-09-29); the scenario that + # posts once keys it in ~7 s. Recorded, not repeated. + step(f"contact:{p}->{q}", layer="relations", via="owner", added_by="reachable gate") + continue + key = guest.owner_key_id if rel.get("via", "owner") == "owner" else guest.contact_code(rel.get("nodes", "all")) + got = host.add_contact(key) + step(f"contact:{p}->{q}", layer="relations", via=rel.get("via", "owner"), key=str(got.get("key_id"))[:60], + reachable_nodes=got.get("reachable_nodes")) + if rel.get("via") == "code": + values["PEER_CONTACT_CODE"] = key + elif k == "room": + if rel["kind"] == "pair": + p, q = rel["members"] + a, b = N[persons[p]["owns"][0]], N[persons[q]["owns"][0]] + ca, cb = a.open_pair(b.owner_key_id), b.open_pair(a.owner_key_id) + if ca != cb: + raise MeshError(f"the two sides derived different pair rooms: {ca} vs {cb}") + rooms[rel.get("id", "pair")] = ca + values["ROOM_ID"] = ca + # POLL BOTH SIDES EVERY TICK. The joiner's half of the handshake + # (publish the KeyPackage, consume the Welcome) advances on its + # transcript reads; a short-circuit `a and b` never read B while + # A was not ready, so B never published (five builds, 2026-09-29). + def both_keyed() -> bool: + ra, rb = a.room(ca), b.room(ca) + return bool(ra["ready"] and rb["ready"]) + try: + wait_for("the pair room to key", both_keyed, float(rel.get("wait", 180)), every=3) + except MeshError: + step.fail("room_NOT_keyed", "the MLS handshake did not complete", [a, b], _HANDSHAKE, + layer="relations", rel="room", cc="CC 4.4.3") + raise + step(f"room:{rel.get('id', 'pair')}", layer="relations", room=ca, keyed=True, + proves="the same room id on both sides, keyed on both") + elif rel["kind"] == "self": + p = rel["person"] + devs = [N[x] for x in persons[p]["owns"]] + try: + wait_for("every device in the self room", + lambda: all(x.grep(r"self room (CREATED|JOINED)") for x in devs), + float(rel.get("wait", 180)), every=3) + except MeshError: + step.fail("self_room_NOT_joined", "a device never joined its person's self room", devs, + r"self room|KeyPackage|Welcome|Added\(|Rejoin", + layer="relations", rel="room", cc="CC 4.4.3.2.4") + raise + step(f"room:self:{p}", layer="relations", devices=persons[p]["owns"], proves="the self room spans every device") + else: + raise MeshError(f"room kind {rel['kind']!r}: declared, no builder yet") + elif k == "message": + p, q = rel["from"], rel["to"] + a, b = N[persons[p]["owns"][0]], N[persons[q]["owns"][0]] + cid = rooms[rel.get("room", "pair")] + text = f"{rel.get('text', 'hello')} {int(time.time())}" + att = a.say(cid, text) + try: + wait_for("the body on the recipient", lambda: any( + m.get("attestation_id") == att and m.get("body") == text for m in b.room(cid)["messages"]), + float(rel.get("wait", 120)), every=2) + except MeshError: + mine = [m for m in b.room(cid)["messages"] if m.get("attestation_id") == att] + step.fail("body_NOT_arrived", "the row may be here but its BODY did not open on the recipient", + [a, b], _BODY, layer="relations", rel="message", cc="CC 5.4.6", row_on_recipient=mine[:1]) + raise + step(f"message:{p}->{q}", layer="relations", attestation_id=att, + proves="the row and its body on the other person's node") + values.update({"ROOM_ID": cid, "MESSAGE_TEXT": text, "MESSAGE_ATTESTATION_ID": att, "MESSAGE_ARRIVED": "true"}) + elif k == "file": + p = rel["person"] + devs = persons[p]["owns"] + src = N[rel.get("device", devs[0])] + data = hashlib.sha256(b"seed").digest() * max(1, int(rel.get("size", 4096)) // 32) + got = src.write_file(data, rel.get("media_type", "application/octet-stream"), rel.get("name", "topology.bin")) + fid = got["attestation_id"] + for other in devs: + if other == src.name: + continue + dev = N[other] + try: + wait_for(f"{other} to open the file", lambda: dev.read_raw(fid)[0] == 200, float(rel.get("wait", 240)), every=5) + except MeshError: + st, body = dev.read_raw(fid) + step.fail(f"file_NOT_open:{other}", "the second device did not open the file", [src, dev], + _BODY + r"|stalled mid-frame|not_in_room", layer="relations", rel="file", cc="CC 5.4.6", + status=st, body=body[:160].decode(errors="replace")) + raise + st, raw = dev.read_raw(fid) + if raw != data: + step.fail(f"file_WRONG_BYTES:{other}", f"{len(raw)} bytes back for {len(data)} written", [src, dev], _BODY, + layer="relations", rel="file", cc="CC 5.3.2.5") + raise MeshError("bytes differ") + files_written["last"] = {"id": fid, "device": src.name, "size": len(data), "person": p} + step(f"file:{p}", layer="relations", size=len(data), devices=devs, proves="byte-identical on every device") + elif k == "bigfile": + # ONE self file of `size` bytes (CIRISEdge#734 lane 7 asked for the + # end-to-end that includes the drive; edge benches the wire alone): + # written on `device` as streamed multipart, timed; pulled by every + # other device, timed from the row's arrival (`/meta` 200) to its + # byte-state `here`; read back through a streamed `?raw=1` and + # compared by SHA-256 — the bytes never sit in this process. + # `resume_at: 0.5` stops the puller once its home has grown by that + # fraction of the file and restarts it; the pull must then finish. + # Byte-identical and timed is the claim; the numbers are recorded, + # never asserted — a floor belongs in the CSD, not here. + # + # CEILING (server-owned, named by the step): since 0.5.218 the + # drive STREAMS both doors — a multipart upload with a `size` field + # before the `file` part goes through edge's `files::publish_stream` + # (CIRISEdge#744) and `?raw=1` above 64 MiB streams back through + # `FileRow::chunks()` (#737). The ceiling is `drive.rs` + # `STREAMED_FILE_CEILING` = edge's stated ~2.5 GiB single-file limit + # (persist's inline manifest cap until persist v52), with + # `STREAMED_UPLOAD_BODY_LIMIT` = that + 1 MiB of form. A 413 here is + # a file above THAT, or an upload that omitted `size` (which falls + # back to the 64 MiB whole-buffered cap, `UPLOAD_BODY_LIMIT` for + # the JSON form). + import random + p = rel["person"] + devs = persons[p]["owns"] + src = N[rel.get("device", devs[0])] + size = int(rel.get("size", 2 << 30)) + seed = int(rel.get("seed", 7)) + blob = mesh.work / f"bigfile-{size}.bin" + h = hashlib.sha256() + if not blob.exists() or blob.stat().st_size != size: + rng = random.Random(seed) + with open(blob, "wb") as f: + left = size + while left: + piece = rng.randbytes(min(1 << 20, left)) + f.write(piece) + h.update(piece) + left -= len(piece) + want_sha = h.hexdigest() + (blob.with_suffix(".sha256")).write_text(want_sha) + else: + want_sha = blob.with_suffix(".sha256").read_text().strip() + t0 = time.monotonic() + st, got = src.write_file_streamed(blob, rel.get("media_type", "application/octet-stream"), + rel.get("name", f"bigfile-{size}.bin")) + publish_s = time.monotonic() - t0 + if st == 413: + step.fail(f"bigfile_REFUSED_BY_DRIVE:{p}", + f"the drive refused {size} bytes with 413: above STREAMED_FILE_CEILING (edge's ~2.5 GiB " + "single-file limit, persist's inline manifest cap until v52), or the upload carried no `size` " + "field and fell back to the 64 MiB whole-buffered cap", [src], + r"413|payload too large|drive\.too_large|STREAMED_FILE_CEILING", layer="relations", rel="bigfile", cc="CC 5.3.2.5", + size=size, status=st, body=str(got)[:200]) + raise MeshError("bigfile refused by the drive's body limit") + if st not in (200, 201): + step.fail(f"bigfile_NOT_WRITTEN:{p}", f"POST /v1/files answered {st}", [src], _BODY, + layer="relations", rel="bigfile", cc="CC 5.3.2.5", size=size, status=st, body=str(got)[:200]) + raise MeshError(f"bigfile write answered {st}") + fid = got["attestation_id"] + seal_lines = src.grep(r"chunk|sealed|publish")[-3:] + timings: Dict[str, Any] = {"size": size, "publish_s": round(publish_s, 2), + "publish_MiB_s": round(size / (1 << 20) / max(publish_s, 1e-6), 1), + "sha256": want_sha, "attestation_id": fid, "seal_evidence": seal_lines, "pulls": {}} + resume_at = rel.get("resume_at") + deadline = time.monotonic() + float(rel.get("wait", 3600)) + for other in devs: + if other == src.name: + continue + dev = N[other] + base_disk = dev.disk_bytes() + t_row = t_here = None + resumed = False + state = None + while time.monotonic() < deadline: + ms, meta = dev.file_meta(fid) + if ms == 200 and t_row is None: + t_row = time.monotonic() + # `bytes` is the drive's byte-state word (`drive::BYTE_STATES`: + # here | not_fetched | not_granted | …), the same word `GET /v1/drive` uses. + state = meta.get("bytes") if isinstance(meta, dict) else None + if ms == 200 and state == "here": + t_here = time.monotonic() + break + if resume_at and not resumed and t_row is not None and dev.disk_bytes() - base_disk >= resume_at * size: + dev.stop() + time.sleep(2) + dev.start() + resumed = True + timings["pulls"][other] = {"resumed_at_bytes": dev.disk_bytes() - base_disk} + time.sleep(2) + if t_here is None: + step.fail(f"bigfile_NOT_PULLED:{other}", + f"the file never reached byte-state `here` on {other} (last meta state {state!r}, " + f"{dev.disk_bytes() - base_disk} bytes grown)", [src, dev], + _BODY + r"|chunk|DAG|manifest|not_fetched", layer="relations", rel="bigfile", cc="CC 5.4.6", + size=size, resumed=resumed) + raise MeshError("bigfile not pulled") + t1 = time.monotonic() + rs, sha, n, body = dev.read_raw_digest(fid) + read_s = time.monotonic() - t1 + if rs != 200 or sha != want_sha or n != size: + step.fail(f"bigfile_WRONG_BYTES:{other}", + f"{other} read {n} bytes (status {rs}) sha {sha[:16]}… for {size} bytes sha {want_sha[:16]}…", + [src, dev], _BODY + r"|seal_mismatch|chunk", layer="relations", rel="bigfile", cc="CC 5.3.2.5", + status=rs, body=body[:160].decode(errors="replace")) + raise MeshError("bigfile bytes differ") + pull_s = t_here - (t_row or t0) + timings["pulls"][other] = {**timings["pulls"].get(other, {}), + "row_seen_after_s": round((t_row or t_here) - t0, 2), + "pull_s": round(pull_s, 2), "pull_MiB_s": round(size / (1 << 20) / max(pull_s, 1e-6), 1), + "read_s": round(read_s, 2), "read_MiB_s": round(size / (1 << 20) / max(read_s, 1e-6), 1), + "resumed": resumed} + values[f"BIGFILE_ATTESTATION_ID"] = fid + step(f"bigfile:{p}", layer="relations", **timings, + proves="one file of the declared size, streamed in, pulled by every other device" + + (" (one of them restarted mid-pull)" if resume_at else "") + ", read back streamed, SHA-256 equal") + elif k == "roster": + # THE DEVICE ROSTER (CSD-037, CIRISServer#655 per-device announce + # ruling): every device of the person lists every device of the + # person; a node that is not the person's sees the ANNOUNCED + # devices only — never an unannounced one. + p = rel["person"] + devs = persons[p]["owns"] + pid = N[devs[0]].owner_key_id + want = {N[d].node_key_id for d in devs} + def roster_of(n: Node) -> set: + st, body = n.api("GET", f"/v1/self/occurrences?identity_key_id={pid}") + return {o.get("occurrence_key_id") for o in (body.get("occurrences") or [])} if st == 200 and isinstance(body, dict) else set() + try: + wait_for(f"{p}'s devices to list each other", lambda: all(want <= roster_of(N[d]) for d in devs), + float(rel.get("wait", 120)), every=5) + except MeshError: + step.fail(f"roster_NOT_complete:{p}", "a device of the person does not list every other device — the " + "identity occurrences did not converge across the person's nodes", [N[d] for d in devs], + r"occurrence|IdentityOccurrence|roster", layer="relations", rel="roster", cc="CC 2.1", + seen={d: sorted(roster_of(N[d])) for d in devs}) + raise + seen_by_outsiders = {} + announced = {N[d].node_key_id for d in devs if next(x for x in t["nodes"] if x["id"] == d).get("announced")} + # `visible_from_complete: true` (default false, so a topology that + # did not declare it keeps the leak-only semantics): an outsider + # must list ALL of the person's announced devices, not a subset. + # CC 5.4.6 (CIRISConstitution#111) — "a person is contactable + # through the nodes they chose to announce, and that set IS their + # public roster" — and the canonical relays them (0.5.218, + # `announced_relay`), so an outsider that peers only ONE of the + # person's devices still converges on every announced one. + complete = bool(rel.get("visible_from_complete", False)) + for o in rel.get("visible_from", []): + if complete: + try: + wait_for(f"{o} to list every announced device of {p}", + lambda: announced <= roster_of(N[o]), float(rel.get("wait", 120)), every=5) + except MeshError: + got = roster_of(N[o]) + step.fail(f"roster_NOT_public:{p}@{o}", "an outsider does not list every device the person " + "announced — the canonical did not relay an announced device's key/occurrence " + "(CC 5.4.6, announced_relay)", [N[o]] + [N[d] for d in devs], + r"announced relay|IdentityOccurrence|first contact|occurrence", + layer="relations", rel="roster", cc="CC 5.4.6", + seen={o: sorted(got), "announced": sorted(announced), + "missing": sorted(announced - got)}) + raise + got = roster_of(N[o]) + unannounced_leak = got - announced + seen_by_outsiders[o] = {"listed": len(got), "announced": len(announced), "complete_required": complete} + if unannounced_leak: + raise MeshError(f"NEGATIVE FAILED: {o} lists unannounced devices of {p}: {sorted(unannounced_leak)}") + step(f"roster:{p}", layer="relations", devices=devs, visible_from=seen_by_outsiders, + proves="each device lists every device; an outsider sees announced devices only" + + (", and ALL of them (CC 5.4.6)" if complete else "")) + elif k == "note": + p = rel["person"] + devs = persons[p]["owns"] + src = N[rel.get("device", devs[0])] + text = f"{rel.get('text', 'note to self')} {int(time.time())}" + got = src.must("POST", "/v1/notes", {"body": text}) + nid = got.get("attestation_id") + for other in devs: + if other == src.name: + continue + dev = N[other] + try: + wait_for(f"{other} to list the note", lambda: any( + n.get("attestation_id") == nid for n in (dev.must("GET", "/v1/notes?limit=50").get("notes") or [])), + float(rel.get("wait", 180)), every=5) + except MeshError: + step.fail(f"note_NOT_listed:{other}", "the note row never reached the other device", [src, dev], + _BODY + r"|stalled mid-frame", layer="relations", rel="note", cc="CC 5.2") + raise + step(f"note:{p}", layer="relations", attestation_id=nid, proves="a note written on one device lists on the others") + elif k == "corpus": + # The transfer corpus (harness/mesh-repro/lib/media_corpus.py): every + # file written on one device, read raw on every other device of the + # person, compared byte for byte. Per file the outcome names its + # layer: 404 = the ROW never crossed (CIRISEdge#716 on a direct + # link), 409 = row here, bytes not pulled, 200+wrong = the bytes + # (CIRISEdge#717 serves a chunk-DAG's manifest), 200+match = opened. + import subprocess + p = rel["person"] + devs = persons[p]["owns"] + src = N[rel.get("device", devs[0])] + out = mesh.work / "corpus" + from .mesh import HARNESS + subprocess.run([sys.executable, str(HARNESS / "mesh-repro" / "lib" / "media_corpus.py"), str(out)], + check=True, capture_output=True) + manifest = json.loads((out / "manifest.json").read_text(encoding="utf-8")) + only = set(rel["only"]) if rel.get("only") else None + written = [] + for row in manifest: + if only and row["name"] not in only: + continue + data = (out / row["name"]).read_bytes() + got = src.write_file(data, row["media_type"], row["filename"]) + written.append({**row, "id": got["attestation_id"]}) + files_written[row["name"]] = {"id": got["attestation_id"], "device": src.name, + "size": len(data), "person": p} + files_written["last"] = files_written[row["name"]] + step(f"corpus_written:{p}", layer="relations", files=len(written), device=src.name) + deadline = time.monotonic() + float(rel.get("wait", 300)) + results: Dict[str, Dict[str, Any]] = {w["name"]: {} for w in written} + pending = {(w["name"], o) for w in written for o in devs if o != src.name} + while pending and time.monotonic() < deadline: + for name, other in list(pending): + w = next(x for x in written if x["name"] == name) + st, raw = N[other].read_raw(w["id"]) + want = (out / name).read_bytes() + r: Dict[str, Any] = {"status": st} + if st == 200: + r["match"] = raw == want + r["size"] = len(raw) + if not r["match"]: + r["expected"] = len(want) + r["first_diff"] = next((i for i in range(min(len(raw), len(want))) if raw[i] != want[i]), min(len(raw), len(want))) + pending.discard((name, other)) + else: + try: + r["reason_id"] = json.loads(raw.decode()).get("reason_id") + except Exception: # noqa: BLE001 + r["reason_id"] = raw[:80].decode(errors="replace") + results[name][other] = r + if pending: + time.sleep(5) + opened = [n for n, per in results.items() if per and all(v.get("match") for v in per.values())] + by_class: Dict[str, List[str]] = {"row_never_crossed": [], "bytes_not_pulled": [], "wrong_bytes": [], "refused": []} + for n, per in results.items(): + for other, v in per.items(): + if v.get("match"): + continue + if v.get("status") == 404: + by_class["row_never_crossed"].append(f"{n}@{other}") + elif v.get("status") == 409: + by_class["bytes_not_pulled"].append(f"{n}@{other}") + elif v.get("status") == 200: + by_class["wrong_bytes"].append(f"{n}@{other}:{v.get('size')}/{v.get('expected')}") + else: + by_class["refused"].append(f"{n}@{other}:{v.get('status')}:{v.get('reason_id')}") + step(f"corpus_compared:{p}", layer="relations", opened=len(opened), total=len(written), + **{k: v for k, v in by_class.items() if v}, + proves="each corpus file byte-identical on every other device of the person") + if len(opened) != len(written): + means = [] + if by_class["row_never_crossed"]: + means.append("rows never reached the other device — the direct link stalls multi-fragment frames (CIRISEdge#716)") + if by_class["bytes_not_pulled"]: + means.append("rows arrived, bytes not pulled — the derived-address pull (CC 5.4.6 / CIRISEdge#499)") + if by_class["wrong_bytes"]: + means.append("bytes returned differ — a chunk-DAG's manifest served as the file (CIRISEdge#717)") + if by_class["refused"]: + means.append("refused by name — drive.seal_mismatch is the drive refusing #717's manifest") + step.fail(f"corpus_NOT_opened:{p}", "; ".join(means), [src] + [N[o] for o in devs if o != src.name], + _BODY + r"|stalled mid-frame|seal_mismatch", layer="relations", rel="corpus", cc="CC 5.3.2.5") + if rel.get("require", True): + raise MeshError("corpus did not open byte-identical on every device") + elif k == "custody": + # WHERE THE FILE IS (FSD/FILE_CUSTODY.md): `GET /v1/files/{id}/custody` + # on the AUTHOR device must name every other device of the person as + # `received` (a delivery receipt, CC 5.3.3.6 — signed by the puller on + # its DAG pull, admitted here by the bridge) and count the person's + # devices. `file: last | ` picks the file. Only a chunk + # DAG (> 1 MiB) carries receipts at this pin; an inline file answers + # `receipts_supported: false` and fails this relation by name. + p = rel["person"] + devs = persons[p]["owns"] + which = rel.get("file", "last") + if which not in files_written: + raise MeshError(f"custody: no file {which!r} was written before this relation " + f"(have: {sorted(files_written)}) — declare a `file` or `corpus` first") + f = files_written[which] + src = N[rel.get("device", f["device"])] + others = [N[o] for o in devs if o != src.name] + want = {o.node_key_id for o in others} + path = f"/v1/files/{f['id']}/custody?cohort={rel.get('cohort', 'self')}" + seen: Dict[str, Any] = {} + + def all_received() -> bool: + st, got = src.api("GET", path) + seen.clear() + seen.update({"status": st, "body": got}) + if st != 200 or not isinstance(got, dict): + return False + if got.get("receipts_supported") is False: + return True # answered below by name: an inline file has no receipt to wait for + received = {d.get("node_key_id") for d in got.get("devices", []) if d.get("holds") == "received"} + return want <= received + try: + wait_for(f"{src.name}'s custody of {which} to name every other device as received", + all_received, float(rel.get("wait", 180)), every=5) + except MeshError: + body = seen.get("body") if isinstance(seen.get("body"), dict) else {} + step.fail(f"custody_NOT_received:{which}", + "the author device holds no delivery receipt from the other device — the puller did not " + "sign one (on_dag_pulled), the receipt row did not cross, or the bridge refused it", + [src] + others, r"delivery receipt|NOT receipted|delivery_receipt|receipt_", + layer="relations", rel="custody", cc="CC 5.3.3.6", status=seen.get("status"), + devices=(body or {}).get("devices"), why=(body or {}).get("why"), + other_keys=(body or {}).get("receipts_from_other_keys")) + raise + got = seen["body"] + if got.get("receipts_supported") is False: + step.fail(f"custody_INLINE:{which}", + f"{which} is an inline file ({got.get('receipts_unsupported_reason')}) — it carries no " + "delivery receipt at this pin; name a chunk-DAG file (> 1 MiB)", + [src], r"custody", layer="relations", rel="custody", cc="CC 5.3.3.6", why=got.get("why")) + raise MeshError("custody: an inline file has no receipts to assert") + if got.get("devices_total") != len(devs): + step.fail(f"custody_WRONG_TOTAL:{which}", + f"devices_total {got.get('devices_total')} for a person owning {len(devs)} devices", + [src], r"custody|nodes_owned_by", layer="relations", rel="custody", cc="CC 5.2", + devices=got.get("devices")) + raise MeshError("custody devices_total differs from the person's device count") + step(f"custody:{which}", layer="relations", device=src.name, devices_total=got.get("devices_total"), + received=[d["node_key_id"] for d in got["devices"] if d.get("holds") == "received"], + held_here=got.get("held_here"), copies_observable=got.get("copies_observable"), + proves="the author device's custody names every other device of the person as received") + elif k == "session": + # ONE DEVICE HANDLES EACH EXCHANGE (CC 3.1.3.1, FSD/SESSION_CLAIMS.md): + # the person is active on `device` (every owner-bearer request marks + # the device attended — `resolve_bearer`), and every device of the + # person must name the SAME handler for every exchange it lists on + # `GET /v1/self/sessions`. Non-vacuous: at least one exchange must be + # listed (the self room's commit duty or a re-wrap, claimed when the + # second device joined and renewed while the person stays); an empty + # list on every device proves nothing and fails by name. + p = rel["person"] + devs = persons[p]["owns"] + src = N[rel.get("device", devs[0])] + + def handlers(n: Node) -> Dict[Tuple[str, str], str]: + st, body = n.api("GET", "/v1/self/sessions") + if st != 200 or not isinstance(body, dict): + return {} + return {(x.get("community_id"), x.get("session_id")): x.get("handler_occurrence_key_id") + for x in (body.get("sessions") or [])} + + seen: Dict[str, Dict[Tuple[str, str], str]] = {} + + def agree() -> bool: + src.api("GET", "/v1/self/sessions") # the person's activity on `device` + for d in devs: + seen[d] = handlers(N[d]) + first = seen[devs[0]] + return bool(first) and all(seen[d] == first for d in devs) + + try: + wait_for(f"{p}'s devices to name one handler per exchange", agree, + float(rel.get("wait", 180)), every=5) + except MeshError: + empty = all(not v for v in seen.values()) + step.fail(f"session_NOT_agreed:{p}", + "no device lists any claimed exchange — nothing was ever gated, or the claim " + "never crossed (session:claim:v1 rides the self plane)" if empty else + "the person's devices name DIFFERENT handlers for one exchange — two devices " + "would act", [N[d] for d in devs], + r"session claim|session:claim|handled by occurrence|unclaimed", layer="relations", + rel="session", cc="CC 3.1.3.1", + seen={d: {f"{c[:12]}/{s_}": h for (c, s_), h in v.items()} for d, v in seen.items()}) + if rel.get("require", True): + raise + else: + one = seen[devs[0]] + step(f"session:{p}", layer="relations", device=src.name, exchanges=len(one), + handlers=sorted({h for h in one.values()}), + proves="every device of the person names the same handler for each exchange") + else: + raise MeshError(f"relation {k!r}: declared, no builder yet") + + # 6. negatives. + for neg in t.get("negatives", []): + if neg["check"] == "cannot_list_room": + n = N[persons[neg["person"]]["owns"][0]] + st, _ = n.api("GET", f"/v1/chat/{rooms[neg.get('room', 'pair')]}/messages") + if st == 200: + raise MeshError(f"NEGATIVE FAILED: {neg['person']} can list the room") + step(f"negative:cannot_list_room:{neg['person']}", layer="negatives", status=st) + elif neg["check"] == "holds_no_row": + # The outsider holds no self-plane row of the person — and the check + # is non-vacuous only if the outsider holds SOMETHING from the + # person's keys (peering, owner-binding): a node nothing reached is + # not evidence (the selffiles ladder's first outsider was exactly that). + n = N[neg["node"]] + keys = {N[d].owner_key_id for d in persons[neg["person"]]["owns"]} | {N[d].node_key_id for d in persons[neg["person"]]["owns"]} + pid = N[persons[neg["person"]]["owns"][0]].owner_key_id + rows = n.rows("select attestation_id, attesting_key_id, attestation_envelope from federation_attestations") + control = sum(1 for _, att, _ in rows if att in keys) + hits = [a for a, _, env in rows if (env or "").find(neg["dimension"]) >= 0 and pid in (env or "")] + if hits: + raise MeshError(f"NEGATIVE FAILED: {neg['node']} holds {len(hits)} {neg['dimension']} rows of {neg['person']}") + if control == 0: + raise MeshError(f"NEGATIVE VACUOUS: {neg['node']} holds nothing at all from {neg['person']}'s keys — peer it as a contact") + step(f"negative:holds_no_row:{neg['node']}", layer="negatives", dimension=neg["dimension"], control=control, + proves="a peered outsider holds the person's public rows and none of their self plane") + elif neg["check"] == "no_wider_self_rows": + # On the person's OWN devices, every self-plane row (file:v1, or a + # chat:* row naming the owner) stays at cohort_scope self — a wider + # row is the file's existence distributed past the person + # (CIRISPersist#919's shape). + pid = N[persons[neg["person"]]["owns"][0]].owner_key_id + wide: Dict[str, List[str]] = {} + for d in persons[neg["person"]]["owns"]: + rows = N[d].rows("select attestation_id, cohort_scope, attestation_envelope from federation_attestations") + bad = [f"{a[:18]}@{sc}" for a, sc, env in rows + if (('"dimension": "file:v1"' in (env or "") or '"dimension":"file:v1"' in (env or "")) + or ('"dimension": "chat:' in (env or "") and pid in (env or ""))) and sc != "self"] + if bad: + wide[d] = bad[:8] + if wide: + raise MeshError(f"NEGATIVE FAILED: self-plane rows wider than self on the person's own devices: {wide}") + step(f"negative:no_wider_self_rows:{neg['person']}", layer="negatives", devices=persons[neg["person"]]["owns"], + proves="every self-plane row on the person's devices is at cohort_scope self") + + # the actor's view, in the client fixture's shape — KEYED BY DECLARED ID + # (CIRISClient#134 §1): `PERSON__OWNER_KEY_ID`, `NODE__URL`, …; + # the positional PEER_* names stay as aliases for the first non-actor + # person. `notes` names every value a flow might expect that this + # declaration did not produce. + act = t["actor"] + me = N[act["device"]] + for pid, p in persons.items(): + first = N[p["owns"][0]] + values[f"PERSON_{pid.upper()}_OWNER_KEY_ID"] = first.owner_key_id + values[f"PERSON_{pid.upper()}_DEVICES"] = ",".join(p["owns"]) + for nid, node in N.items(): + values[f"NODE_{nid.upper()}_URL"] = node.url + values[f"NODE_{nid.upper()}_KEY_ID"] = node.node_key_id or node.key_id + others = [p for p in persons if p != act["person"]] + if others: + peer = N[persons[others[0]]["owns"][0]] + values.update({"PEER_URL": peer.url, "PEER_KEY_ID": peer.owner_key_id, "PEER_NODE_KEY_ID": peer.node_key_id, + "PEER_OWNER_KEY_ID": peer.owner_key_id, "PEER_PERSON": others[0]}) + else: + step.notes.append("no PEER_*: the declaration has one person") + values.update({"LOCAL_OWNER_KEY_ID": me.owner_key_id, "LOCAL_NODE_KEY_ID": me.node_key_id, + "ACTOR_PERSON": act["person"], "ACTOR_DEVICE": act["device"]}) + for want in ("ROOM_ID", "MESSAGE_ATTESTATION_ID", "PEER_CONTACT_CODE"): + if want not in values: + step.notes.append(f"no {want}: the declaration has no relation that produces it") + return {"verdict": "PASS", "steps": step.log, "derived": d, "values": values, "notes": step.notes} diff --git a/pyproject.toml b/pyproject.toml index 718dcbd8..78398ed2 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -82,7 +82,7 @@ requires-python = ">=3.10" # them; below it a fresh install renders those states as raw ids. The # `contacts.*` ladder stalls are still on the guard's debt list (CIRISClient#37). dependencies = [ - "ciris-client>=0.5.224,<0.6", + "ciris-client>=0.5.225,<0.6", ] [project.urls] diff --git a/src/accord.rs b/src/accord.rs index 5ff92585..dd11c23a 100644 --- a/src/accord.rs +++ b/src/accord.rs @@ -1018,6 +1018,7 @@ async fn genesis_assemble( founded_at: now, consensus_protocol: ACCORD_CONSENSUS_PROTOCOL.to_string(), consensus_protocol_entrenched: true, + dissolved_at: None, persist_row_hash: String::new(), }; if let Err(e) = crate::family::create_family(&st.engine, family).await { @@ -2234,6 +2235,7 @@ fn signed_family_from_envelope(env: &serde_json::Value) -> Result Result { + ciris_persist::federation::key_digest::Sha256Ed25519Raw::from_pubkey_base64( + &rec.pubkey_ed25519_base64, + ) + .map(|d| d.to_hex()) + .map_err(|e| DeAdmitFailure::local("digest the revoked key", e))? + } + Ok(None) => { + return Err(DeAdmitFailure::local( + "digest the revoked key", + format!("{revoked_key_id} is not held here — its subject digest is unknown"), + )) + } + Err(e) => return Err(DeAdmitFailure::local("lookup the revoked key", e)), + }; let mut revocation = Revocation { revocation_id: crate::ids::new_id(), - revoked_key_id: revoked_key_id.to_owned(), + revoked_key_id: Some(revoked_key_id.to_owned()), + revoked_key_sha256_ed25519_raw: subject, revoking_key_id: revoking_key_id.clone(), reason: Some(reason.to_owned()), revoked_at: now, diff --git a/src/announced_relay.rs b/src/announced_relay.rs new file mode 100644 index 00000000..9a391904 --- /dev/null +++ b/src/announced_relay.rs @@ -0,0 +1,503 @@ +//! **A canonical relays the devices people announced** (CC 5.4.6, +//! CIRISConstitution#111; CIRISServer#655). +//! +//! # The ruling +//! +//! *"A person is contactable through the nodes they chose to announce, and that +//! set IS their public roster."* Announce is per node: `POST +//! /v1/federation/announce` (this node) and `POST /v1/self/nodes/{id}/announce` +//! (another of the owner's nodes) widen that node's owner-binding to +//! `cohort_scope: federation`, and [`crate::auth::ownership::announced_nodes_of`] +//! is the one read of the result. The maintainer's ruling of 2026-09-30 adds the +//! other half: announced devices are discoverable by EVERYONE — the canonical +//! relays them. +//! +//! # The fault this closes +//! +//! Measured on the native harness (`harness/native/topologies/selffiles.yaml`): +//! person `one` owns D1 and D2, both announced; X — another person's node — +//! peers D1 and the canonical C. X's `GET /v1/self/occurrences?identity_key_id= +//! ` listed ONE of `one`'s two devices. C held D2's key record and +//! occurrence, and offered neither: the three `SelfOwn` planes (`Key`, +//! `IdentityOccurrence`, `TransportDestination`) advertise only the subjects in +//! the node's self-publish set, and C's self set is C (plus its owner, if it +//! has one). A person's roster was complete only for a reader that happened to +//! peer every one of their devices directly — "contactable through the nodes +//! they announced" held for nobody who had not already met them. +//! +//! # The switch edge built for this +//! +//! Edge v34.0.0 (CIRISEdge#678) gave the host a per-kind publish set for the +//! `SelfOwn` planes — `ReplicationRuntimeConfig::kind_publish_selector`, read +//! by the bridge's `self_own_subjects(kind)`. `Some(set)` REPLACES the +//! self-publish set for that plane; `None` leaves the plane on it. What the +//! set names is the SUBJECT of the row, not its attester: the Key plane keeps a +//! record whose `key_id` is in the set, the IdentityOccurrence plane an +//! occurrence whose `occurrence_key_id` is. Neither plane consults the +//! recipient's reach — CIRISEdge#671's first-contact rule lives on the +//! Attestation plane (`bridge.rs` `reach_withholds`) — so a set installed here +//! reaches an unconsented, Attributed-stranger peer exactly as it reaches a +//! consented one. The IdentityOccurrence plane then runs CIRISEdge#682's +//! announce gate per subject (`identity_rows_withheld_from`), which serves an +//! owned node's occurrence to everyone only when that node is announced: the +//! set below and edge's gate are the same predicate from two sides, and a +//! node that drops out of one drops out of the other. +//! +//! # What is relayed — and the four things that never are +//! +//! - **`Key`**: this node's own publish set ∪ every announced node ∪ each such +//! node's OWNER. The owner's key record is not optional: a reader verifies +//! the occurrence (signed by the identity) and the owner-binding (authored by +//! the owner) against it, and without it the device list is a list of rows +//! the reader must refuse. +//! - **`IdentityOccurrence`**: this node's own publish set ∪ every announced +//! node. NOT the owners: an occurrence subject is an occurrence key, and a +//! person key as a subject would be a row about the person rather than one of +//! their announced devices. +//! - **`TransportDestination`**: `None` — the self-publish set, untouched. +//! Routes are not relayed; the contact code carries them, to the person the +//! owner chose to give it to. +//! - **Every other plane**: `None`. Edge consults the selector only for the +//! three `SelfOwn` planes, and this module answers `None` for everything +//! else as well, so a later edge that widened the consultation would not +//! inherit a relay nobody decided on. Consent grants, self-scoped rows and +//! the owner-binding itself ride the Attestation plane under its own gates. +//! - **An unannounced node**: never, on any plane. Announce is per node and +//! the person's other choices do not leak onto it. +//! +//! # What this does NOT carry — the owner-binding +//! +//! A reader's public roster (`GET /v1/self/occurrences`) shows an occurrence +//! only if the reader's OWN directory says the node is announced +//! (`announced_nodes_of`), and that is read from the owner-binding +//! `delegates_to(owner → node)` at `federation` — an Attestation-plane row the +//! selector is never consulted for. Toward an Attributed stranger that plane is +//! CIRISEdge#671's first-contact reach, which carries "this node's allegiance +//! facts" only: `attestation_is_allegiance_fact` (edge v36.1.0 +//! `replication/bridge.rs:6031-6039`) requires the row's attester to be in THIS +//! node's self-publish set, so a canonical never relays a third party's +//! owner-binding to an unconsented peer (`reach_withholds`, `bridge.rs:6114`, +//! "the recipient is an Attributed stranger (first contact)"). A device of the +//! owner DOES pass it — the owner is in that device's self set — so an outsider +//! that peers ANY one of the person's devices learns every binding that device +//! holds, and this relay supplies the key records and occurrences behind them. +//! An outsider that peers the canonical and none of the person's devices gets +//! the rows but not the binding; closing that is edge's first-contact rule to +//! widen, not a server set. +//! +//! # Who relays +//! +//! The ruling says "the canonical". A node knows it serves infrastructure when +//! it holds `infra:serve` from a root it trusts — +//! `capability_roots_to_trusted_root(me, me, infra:serve)`, persist's +//! composed capability walk (the same leg B the trace serve gate asks about a +//! recipient, asked here about ourselves). That is exactly what genesis +//! requires of a serve node (`mesh_genesis`: a serve node must carry an +//! `infra:serve` grant from the charter root) and what the test-anchor bless +//! writes (`test_bless::has_capability_grant`). Every other node answers +//! `None` for every plane and is byte-for-byte the pre-#678 behaviour. The +//! predicate is re-asked on every refresh, so a canonical blessed after boot +//! starts relaying without a restart, and one whose grant is withdrawn stops. +//! +//! # Cost, and where it is paid +//! +//! The selector closure runs inside edge's sweep, once per plane per round per +//! peer; it reads a snapshot behind a lock and never touches the database. +//! The snapshot is recomputed OFF the sweep, by the publish-own updater in +//! `compose` — the loop that already keeps the self-publish set current — +//! every [`RELAY_REFRESH`], and immediately when [`nudge`] fires (every +//! `kick_replication`: an announce, a release, an eviction, a claim). An +//! announce made on ANOTHER node reaches this one as an inbound row, which +//! nothing kicks on; the period bounds that lag. + +use std::collections::BTreeSet; +use std::sync::{Arc, OnceLock, RwLock}; +use std::time::Duration; + +use ciris_edge::replication::bridge::KindPublishSelector; +use ciris_edge::replication::EnvelopeKind; +use ciris_persist::prelude::Engine; + +/// How often the relay set is recomputed when nothing nudges it. Announce is +/// rare and every local change nudges; this bounds only the lag of an announce +/// made on another node and carried here by replication. +pub const RELAY_REFRESH: Duration = Duration::from_secs(60); + +/// Page size for the occurrence enumeration. The same default the bridge's +/// since-cursor sweeps use. +const OCCURRENCE_PAGE: u32 = 1024; + +/// Hard stop on the enumeration — 4096 pages × 1024 rows, far above any real +/// corpus and far below "forever" (the same bound edge puts on a `Full` +/// drain, CIRISEdge#531). Hitting it truncates the relay for one pass, loudly. +const MAX_OCCURRENCE_PAGES: usize = 4096; + +/// The relay sets, computed off the hot path. Sorted and deduplicated. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct RelaySets { + /// Every node whose owner-binding is live at `federation` scope — the + /// devices people announced. + pub announced_nodes: Vec, + /// The owners of those nodes (their key records verify the occurrence and + /// the binding on the reader's side). + pub owners: Vec, +} + +impl RelaySets { + /// Nothing to relay. + #[must_use] + pub fn is_empty(&self) -> bool { + self.announced_nodes.is_empty() && self.owners.is_empty() + } +} + +/// **THE selection** — what one `SelfOwn` plane publishes, given this node's +/// own publish set and the relay snapshot. Pure, so the privacy negatives are +/// testable without a runtime. +/// +/// `relay == None` (not a relay, or not computed yet) answers `None` for every +/// kind, which edge reads as "the self-publish set": the pre-#678 behaviour. +/// The answer REPLACES the self set for its kind (edge's contract), so it is +/// always `own ∪ relayed` — a relay that forgot to include itself would stop +/// publishing its OWN key record. +#[must_use] +pub fn selection_for( + kind: EnvelopeKind, + own: &[String], + relay: Option<&RelaySets>, +) -> Option> { + let relay = relay?; + let relayed: Vec<&String> = match kind { + EnvelopeKind::Key => relay + .announced_nodes + .iter() + .chain(relay.owners.iter()) + .collect(), + EnvelopeKind::IdentityOccurrence => relay.announced_nodes.iter().collect(), + // Routes stay on the self set: the contact code carries them. Every + // other kind is not ours to answer (see the module docs). + _ => return None, + }; + let mut set: BTreeSet = own.iter().cloned().collect(); + set.extend(relayed.into_iter().cloned()); + Some(set.into_iter().collect()) +} + +/// The shared snapshot the selector closure reads. `None` = not relaying. +type Snapshot = Arc>>>; + +/// This process's relay state: the snapshot, and the wake-up the refresh +/// waits on. +struct RelayState { + snapshot: Snapshot, + wake: tokio::sync::Notify, +} + +static STATE: OnceLock = OnceLock::new(); + +fn state() -> &'static RelayState { + STATE.get_or_init(|| RelayState { + snapshot: Arc::new(RwLock::new(None)), + wake: tokio::sync::Notify::new(), + }) +} + +/// The selector compose installs on the ONE replication runtime. It reads the +/// live self-publish set (`own`, the same `RwLock` the `self_provider` reads, +/// so the owner admitted after a claim is in the union the moment they are in +/// the self set) and this module's snapshot. No I/O: two read locks and a +/// clone. +#[must_use] +pub fn selector(own: Arc>>) -> KindPublishSelector { + selector_over(own, Arc::clone(&state().snapshot)) +} + +fn selector_over(own: Arc>>, snapshot: Snapshot) -> KindPublishSelector { + KindPublishSelector::new(move |kind| { + // Cheap exit for the planes we never answer, before any lock. + if !matches!(kind, EnvelopeKind::Key | EnvelopeKind::IdentityOccurrence) { + return None; + } + let relay = snapshot.read().ok().and_then(|s| s.clone())?; + let own = own.read().map(|o| o.clone()).unwrap_or_default(); + selection_for(kind, &own, Some(&relay)) + }) +} + +/// A selector over a caller-held snapshot, for in-process tests that build a +/// bridge by hand rather than through compose. +#[must_use] +pub fn selector_for_sets(own: Vec, relay: Option) -> KindPublishSelector { + selector_over( + Arc::new(RwLock::new(own)), + Arc::new(RwLock::new(relay.map(Arc::new))), + ) +} + +/// Ask for a recompute now. Called from `kick_replication`, so an announce, a +/// release or a claim on THIS node moves the relay on the same round rather +/// than the next [`RELAY_REFRESH`]. Coalesced: a burst of kicks is one pass. +pub fn nudge() { + state().wake.notify_one(); +} + +/// Wait for a nudge, at most `period`. The publish-own updater's sleep. +pub(crate) async fn wait_for_nudge(period: Duration) -> bool { + tokio::time::timeout(period, state().wake.notified()) + .await + .is_ok() +} + +/// Does this node serve infrastructure — does any key that is "us" hold +/// `infra:serve` from a root that key itself trusts? See the module docs for +/// why this is the predicate for "the canonical". A read failure answers +/// `false`: a node that cannot establish it is a relay does not start +/// publishing third parties' records on a guess. +pub async fn serves_infrastructure(engine: &Engine, own_key_ids: &[String]) -> bool { + use ciris_persist::federation::trust_root::{ + capability_roots_to_trusted_root, INFRA_SERVE_SCOPE, + }; + let dir = engine.federation_directory(); + for k in own_key_ids { + match capability_roots_to_trusted_root(dir.as_ref(), k, k, INFRA_SERVE_SCOPE).await { + Ok(Some(_)) => return true, + Ok(None) => {} + Err(e) => tracing::debug!( + key_id = %k, + error = %format!("{e:#}"), + "announced relay: could not walk this key's infra:serve capability — \ + treated as not held" + ), + } + } + false +} + +/// **Every announced node this directory knows, and their owners.** +/// +/// Enumerated from the signed IdentityOccurrence plane — the rows a roster is +/// made of — rather than from the attestation log, which on a canonical is +/// orders of magnitude larger: each distinct `identity_key_id` is a person who +/// has devices here, and [`crate::auth::ownership::announced_nodes_of`] (the +/// one read the roster endpoint also uses) answers which of their nodes they +/// announced. A node announced by someone with no signed occurrence here has +/// no roster row here to relay either; its absence is the correct answer. +/// +/// # Errors +/// The directory could not be read. A partial answer is never returned: a +/// relay built from half a corpus would silently drop people. +pub async fn announced_relay_sets(engine: &Engine) -> Result { + let dir = engine.federation_directory(); + let mut identities: BTreeSet = BTreeSet::new(); + let mut since: Option<(chrono::DateTime, String)> = None; + let mut pages = 0usize; + loop { + let page = dir + .list_signed_identity_occurrences_since(since.clone(), OCCURRENCE_PAGE) + .await + .map_err(|e| format!("list_signed_identity_occurrences_since: {e:#}"))?; + let short = page.len() < OCCURRENCE_PAGE as usize; + for row in &page { + identities.insert(row.occurrence.identity_occurrence.identity_key_id.clone()); + } + let next = page + .last() + .map(ciris_persist::federation::ServedIdentityOccurrence::resume_pair); + pages += 1; + if short || next.is_none() || next == since { + break; + } + if pages >= MAX_OCCURRENCE_PAGES { + tracing::warn!( + pages, + "announced relay: occurrence enumeration hit its page bound — the relay set \ + for this pass is TRUNCATED (CIRISEdge#531 bound)" + ); + break; + } + since = next; + } + let mut nodes: BTreeSet = BTreeSet::new(); + let mut owners: BTreeSet = BTreeSet::new(); + for identity in identities { + let announced = crate::auth::ownership::announced_nodes_of(engine, &identity).await?; + if announced.is_empty() { + continue; + } + owners.insert(identity); + nodes.extend(announced); + } + Ok(RelaySets { + announced_nodes: nodes.into_iter().collect(), + owners: owners.into_iter().collect(), + }) +} + +/// One refresh pass: decide whether this node relays, recompute the sets if +/// it does, and publish the snapshot. Returns whether the snapshot CHANGED, so +/// the caller can kick a round on a real transition and stay quiet otherwise. +/// +/// `recheck_role` re-asks [`serves_infrastructure`]. The periodic pass does; +/// a [`nudge`] does not, because a nudge fires on every `kick_replication` — +/// every chat row — and on the nodes that are NOT relays (nearly all of them) +/// a nudge must cost nothing: with no snapshot and no recheck this returns +/// before touching the database. A relay recomputes its sets on every nudge; +/// that is the node the recompute is for. +/// +/// Logs its own cost: every periodic pass on this node must (the 15-minute +/// read-API stall was a pass nobody had timed). +pub async fn refresh(engine: &Engine, own_key_ids: &[String], recheck_role: bool) -> bool { + let relaying = state() + .snapshot + .read() + .map(|s| s.is_some()) + .unwrap_or(false); + if !recheck_role && !relaying { + return false; + } + let started = std::time::Instant::now(); + let serves = if recheck_role { + serves_infrastructure(engine, own_key_ids).await + } else { + relaying + }; + let next: Option> = if serves { + match announced_relay_sets(engine).await { + Ok(sets) => Some(Arc::new(sets)), + Err(e) => { + // Keep the previous snapshot: a read failure is not "nobody + // announced anything", and flapping the relay off and on with + // the store would withdraw every person's roster for a pass. + tracing::warn!( + error = %e, + "announced relay: could not enumerate announced nodes — keeping the \ + previous relay set" + ); + return false; + } + } + } else { + None + }; + let elapsed = started.elapsed(); + let snapshot = &state().snapshot; + let changed = { + let Ok(mut w) = snapshot.write() else { + return false; + }; + let changed = w.as_deref() != next.as_deref(); + if changed { + *w = next.clone(); + } + changed + }; + if changed { + match &next { + Some(s) => tracing::info!( + announced_nodes = s.announced_nodes.len(), + owners = s.owners.len(), + elapsed_ms = elapsed.as_millis() as u64, + "announced relay: this node serves infrastructure and now relays the Key and \ + IdentityOccurrence rows of every announced device and its owner (CC 5.4.6, \ + CIRISServer#655) — routes are not relayed" + ), + None => tracing::info!( + elapsed_ms = elapsed.as_millis() as u64, + "announced relay: OFF — this node holds no infra:serve from a root it trusts; \ + its SelfOwn planes publish only its own set" + ), + } + } else if elapsed > Duration::from_secs(1) { + tracing::warn!( + elapsed_ms = elapsed.as_millis() as u64, + "announced relay: a refresh pass took over a second" + ); + } else { + tracing::debug!( + elapsed_ms = elapsed.as_millis() as u64, + "announced relay: refresh pass, unchanged" + ); + } + changed +} + +#[cfg(test)] +mod tests { + use super::*; + + fn s(v: &[&str]) -> Vec { + v.iter().map(|x| (*x).to_owned()).collect() + } + + fn relay() -> RelaySets { + RelaySets { + announced_nodes: s(&["d2"]), + owners: s(&["one"]), + } + } + + #[test] + fn not_a_relay_answers_none_for_every_kind() { + for kind in EnvelopeKind::ALL { + assert_eq!(selection_for(kind, &s(&["me"]), None), None, "{kind:?}"); + } + } + + #[test] + fn key_plane_is_own_plus_announced_nodes_plus_their_owners() { + assert_eq!( + selection_for(EnvelopeKind::Key, &s(&["me", "my-owner"]), Some(&relay())), + Some(s(&["d2", "me", "my-owner", "one"])) + ); + } + + #[test] + fn occurrence_plane_is_own_plus_announced_nodes_never_the_person() { + let got = selection_for( + EnvelopeKind::IdentityOccurrence, + &s(&["me"]), + Some(&relay()), + ) + .expect("relayed"); + assert_eq!(got, s(&["d2", "me"])); + assert!( + !got.contains(&"one".to_owned()), + "a person key is never an occurrence subject" + ); + } + + /// Routes, and every plane edge does not consult, stay on the self set — + /// including the Attestation plane, which carries consent grants and every + /// self-scoped row. + #[test] + fn routes_and_every_other_plane_are_never_relayed() { + for kind in EnvelopeKind::ALL { + if matches!(kind, EnvelopeKind::Key | EnvelopeKind::IdentityOccurrence) { + continue; + } + assert_eq!( + selection_for(kind, &s(&["me"]), Some(&relay())), + None, + "{kind:?} must stay on the self-publish set" + ); + } + } + + /// The closure reads the LIVE self set: an owner admitted after the + /// selector was built is in the union without a rebuild. + #[test] + fn the_selector_unions_the_live_self_set() { + let own = Arc::new(RwLock::new(s(&["me"]))); + let snap: Snapshot = Arc::new(RwLock::new(Some(Arc::new(relay())))); + let sel = selector_over(Arc::clone(&own), Arc::clone(&snap)); + assert_eq!(sel.select(EnvelopeKind::Key), Some(s(&["d2", "me", "one"]))); + own.write().unwrap().push("my-owner".into()); + assert_eq!( + sel.select(EnvelopeKind::Key), + Some(s(&["d2", "me", "my-owner", "one"])) + ); + assert_eq!(sel.select(EnvelopeKind::TransportDestination), None); + *snap.write().unwrap() = None; + assert_eq!(sel.select(EnvelopeKind::Key), None, "relay off = self set"); + } +} diff --git a/src/auth/occurrence.rs b/src/auth/occurrence.rs index ecbe5ac0..3b30fc42 100644 --- a/src/auth/occurrence.rs +++ b/src/auth/occurrence.rs @@ -27,6 +27,10 @@ //! Auth: a SURVIVING active occurrence (or the root). For a *stolen* device //! you MUST sign with a different surviving key — never the compromised one //! (CEG §11.7.4). A *voluntary* self-revoke (signer == revoked) is allowed. +//! **Moved in 0.5.218** to [`crate::self_devices`] (CSD-037): the owner's +//! session authorises it, the owner's pen signs it through persist's +//! SIGNED, replicating door, and it is the same act as a node release +//! (`self_devices::evict_device`). This router no longer serves it. //! 3. `GET /v1/self/occurrences?identity_key_id=…` — LIST the active //! occurrences of a self (for the client identity page's device list). //! Read-only; unauthenticated by design (an occurrence roster is public @@ -44,7 +48,7 @@ use axum::extract::{Query, State}; use axum::http::{HeaderMap, StatusCode}; use axum::response::{IntoResponse, Response}; use axum::{Json, Router}; -use ciris_persist::federation::types::{IdentityOccurrence, IdentityOccurrenceRevocation}; +use ciris_persist::federation::types::IdentityOccurrence; use ciris_persist::federation::{EncryptionPubkeys, SignedKeyRecord}; use ciris_persist::prelude::{Engine, HybridPolicy}; use serde::{Deserialize, Serialize}; @@ -407,132 +411,18 @@ async fn add_occurrence( .into_response() } -// ─── POST /v1/self/occurrence/revoke (REVOKE) ──────────────────────────────── - -#[derive(Debug, Deserialize)] -struct RevokeOccurrenceRequest { - /// The root identity the occurrence speaks for. - identity_key_id: String, - /// The occurrence (device) to remove from the self. - occurrence_key_id: String, - /// Optional operator/ceremony annotation (e.g. "laptop lost 2026-06-23"). - #[serde(default)] - reason: Option, -} - -#[derive(Debug, Serialize)] -struct RevokeOccurrenceResponse { - identity_key_id: String, - /// The occurrence that was revoked. After this it fails `signer_acts_for`. - occurrence_key_id: String, - /// The surviving key that authorized the revocation (the request signer; the - /// recorded `witness_set` single-vouch, §11.7.4). - revoked_by: String, - /// RFC-3339 effective time (== now; the active-state filter is - /// `effective_at <= now`, so the revocation is effective immediately). - effective_at: String, -} - -async fn revoke_occurrence( - State(st): State, - headers: HeaderMap, - body: Bytes, -) -> Response { - // (1) Verify the request hybrid signature. - let caller = match verify::verify_request(&st.engine, &headers, &body, st.policy).await { - Ok(c) => c, - Err(VerifyError::MissingHeader(h)) => { - return err(StatusCode::UNAUTHORIZED, format!("missing {h}")) - } - Err(VerifyError::NoDirectory) => { - return err(StatusCode::SERVICE_UNAVAILABLE, "no federation directory") - } - Err(VerifyError::SignatureInvalid(e)) => { - return err( - StatusCode::UNAUTHORIZED, - format!("signature verification failed: {e}"), - ) - } - }; - - // (2) Parse. - let req: RevokeOccurrenceRequest = match serde_json::from_slice(&body) { - Ok(r) => r, - Err(e) => return err(StatusCode::BAD_REQUEST, format!("bad request body: {e}")), - }; - - // (3) Admission: a SURVIVING active occurrence (or the root) authorizes the - // removal (§11.7.4 single-vouch — "the revoking occurrence OR the identity"). - // For a STOLEN device you MUST sign with a different surviving key, never the - // compromised one — that is exactly why a backup occurrence is enrolled. A - // VOLUNTARY self-revoke (signer == revoked) is permitted (a device leaving on - // its own); the producer-side helper (verify::sign_occurrence_revocation) does - // not forbid it, and the surviving-key requirement is a flow property of the - // stolen-device case, not a server invariant we can enforce (the server cannot - // know the key is compromised). We DO require the signer to currently act for - // the self, so a revoked / unrelated key cannot revoke another's device. - if !verify::signer_acts_for(&st.engine, &caller.key_id, &req.identity_key_id).await { - return err( - StatusCode::FORBIDDEN, - "signer is neither the identity key nor an active occurrence of it — \ - revoke a device by signing with a surviving key you still control", - ); - } - - // (4) Record the append-only revocation. persist's *_active reads compose it, - // so after this the revoked key fails signer_acts_for. The witness_set carries - // the single vouch [revoker] (§11.7.4); persist computes persist_row_hash. - let now = chrono::Utc::now(); - let revocation = IdentityOccurrenceRevocation { - identity_key_id: req.identity_key_id.clone(), - occurrence_key_id: req.occurrence_key_id.clone(), - revoked_at: now, - effective_at: now, - reason: req.reason, - witness_set: vec![caller.key_id.clone()], - persist_row_hash: String::new(), - }; - // persist v16.0.0 (CIRISPersist#421) closed the gap 0.5.100 deliberately refused - // to ship: the wire revocation was UNSIGNED, so any consented peer could kill any - // identity's sealability (a permanent-DoS forgery). v16 splits the primitive: - // * put_identity_occurrence_revocation — signed gate ({attesting_key_id, - // signed_envelope, signature} + signer_acts_for §11.7.4 single-vouch-for-self). - // This is the REPLICATED path. - // * put_identity_occurrence_revocation_local — trusted-local, signature columns - // NULL, EXCLUDED from the signed replication read, never reachable from the - // replication apply. "Engine-internal writes on behalf of the local user where - // the revocation is locally produced — NOT peer-received." - // THIS endpoint is precisely the latter: an owner-authenticated, locally-produced - // revocation over HTTP. So it takes the _local path — which is byte-for-byte the - // behaviour we already had (revocation carriage was never wired, #227 S2), minus - // the forgery surface. We do NOT synthesise a signature here: we hold no mandate to - // sign as an arbitrary `identity_key_id`, and faking one would re-open the very DoS - // the gate closes. Wiring the SIGNED, replicating revocation (now finally possible, - // with list_signed_identity_occurrence_revocations_for as its byte-exact re-read) is - // tracked separately as the #227 S2 carriage work. - if let Err(e) = st - .engine - .federation_directory() - .put_identity_occurrence_revocation_local(revocation) - .await - { - return err( - StatusCode::INTERNAL_SERVER_ERROR, - format!("put_identity_occurrence_revocation: {e}"), - ); - } - - ( - StatusCode::OK, - Json(RevokeOccurrenceResponse { - identity_key_id: req.identity_key_id, - occurrence_key_id: req.occurrence_key_id, - revoked_by: caller.key_id, - effective_at: now.to_rfc3339(), - }), - ) - .into_response() -} +// ─── POST /v1/self/occurrence/revoke — MOVED (0.5.218) ─────────────────────── +// +// The revoke lived here until 0.5.218, authorised by a request hybrid-signed by a +// surviving device key and written through persist's trusted-LOCAL door +// (`put_identity_occurrence_revocation_local`): unsigned, EXCLUDED from the signed +// replication read, so the revocation never left the node that wrote it (the +// #227 S2 carriage gap), and unreachable from the app, which authenticates with a +// bearer and got 401 on every call (CSD-037). It is now +// `crate::self_devices::revoke_occurrence`, which authorises the OWNER's session, +// signs the revocation with the owner's pen through persist's SIGNED door, and +// shares one implementation with node release (`self_devices::evict_device`). +// `tests/occurrence.rs` gates that the local door has no caller in `src/`. // ─── GET /v1/self/occurrences (LIST) ───────────────────────────────────────── @@ -761,10 +651,6 @@ async fn list_occurrences( pub fn router(engine: Arc, policy: HybridPolicy) -> Router { Router::new() .route("/v1/self/occurrence", axum::routing::post(add_occurrence)) - .route( - "/v1/self/occurrence/revoke", - axum::routing::post(revoke_occurrence), - ) .route("/v1/self/occurrences", axum::routing::get(list_occurrences)) .with_state(OccurrenceState { engine, policy }) } diff --git a/src/auth/session.rs b/src/auth/session.rs index 17f86e74..41f99045 100644 --- a/src/auth/session.rs +++ b/src/auth/session.rs @@ -807,6 +807,16 @@ pub async fn resolve_bearer( return Ok(None); } let role = UserRole::from_wa_role(cert.role); + // THE PERSON IS ON THIS DEVICE (CC 3.1.3.1, `crate::session_claims`). This + // is the one line every one of their authenticated requests passes, so it + // is where attendance is observed — not at boot, which would make whichever + // device came up first take the person's exchanges. Only the OWNER's own + // session counts: the delegated branch above returned before here (a + // helper acting for the person is not the person), and a lesser role on a + // shared node is not the person whose self this node is. + if role == UserRole::SystemAdmin { + crate::session_claims::Attendance::global().note_presence(); + } Ok(Some(SessionCaller { wa_id: cert.wa_id, name: cert.name, diff --git a/src/backend.rs b/src/backend.rs index 7bb13738..38161fe9 100644 --- a/src/backend.rs +++ b/src/backend.rs @@ -357,7 +357,7 @@ where /// [`BlobChunkSource`]: ciris_edge::blob_swarm::BlobChunkSource /// [`BlobMeaning::project`]: ciris_edge::blob_swarm::BlobMeaning::project /// [`ContentScope`]: ciris_edge::blob_swarm::ContentScope -pub(crate) struct ServerBlobChunkSource { +pub struct ServerBlobChunkSource { inner: ciris_edge::blob_swarm::PersistBlobChunkSource, directory: Arc, /// The engine itself, for the community-DEK epoch binding — the @@ -375,7 +375,7 @@ pub(crate) struct ServerBlobChunkSource { } impl ServerBlobChunkSource { - pub(crate) fn new(engine: &Engine) -> Self { + pub fn new(engine: &Engine) -> Self { Self { inner: ciris_edge::blob_swarm::PersistBlobChunkSource::new(Engine::clone(engine)) .with_revocations(Some(revocation_register())), @@ -435,11 +435,22 @@ impl ciris_edge::blob_swarm::BlobChunkSource for ServerBlobChunkSource { /// there and copied here by shape, not by hand: a different group id on /// the serve side is a fetch that arrives on the right address and is /// refused as the wrong room. - /// 2. **A referencing row** — THE FALLBACK, for the tiers (1) does not - /// cover. A plaintext or self/family blob has no community-DEK binding, - /// so any attestation whose `evidence_refs` cites the sha projects - /// through `BlobMeaning` (a `holds_bytes` claim is possession, not - /// meaning, and `project` refuses it itself). + /// 2. **Edge's serve scope** — THE FALLBACK, for the tiers (1) does not + /// cover (plaintext, self, family): `BlobMeaning::serve_scope` (edge + /// v38.0.0, CIRISEdge#736), which reads every row placing the blob — + /// persist's binding index PLUS each row's `supersedes` widening — and + /// answers from the WIDENING first. Until 0.5.218 this arm was the + /// server's own walk over `attestations_binding_content`, returning the + /// first row that projected. On a family file's AUTHOR node that is the + /// author's `self` row (every producer authors at `self` and crosses by + /// a two-row widening, `FSD/CONTENT_TRANSFER.md` §6.9), and the binding + /// index returns only `scores` rows, so the family placement was never + /// seen: every member's chunk fetch arrived on a `family` address and + /// was refused `blob_serve_arrival_scope_insufficient`. Edge found the + /// same bug in its own source and fixed it once; the server's copy of + /// the choice is DELETED rather than patched (the mirrored-rule class — + /// one rule, one implementation). A `holds_bytes` claim is still + /// possession, not meaning: `project`, inside it, refuses one. /// /// Worth knowing what this arm could NOT do until recently: it reads /// `attestations_binding_content`, and a chat row did not cite its blob @@ -488,31 +499,19 @@ impl ciris_edge::blob_swarm::BlobChunkSource for ServerBlobChunkSource { "blob chunk source: the community-DEK binding could not be read" ), } - let rows = match self.directory.attestations_binding_content(&sha_hex).await { - Ok(rows) => rows, - Err(e) => { - tracing::warn!( - blob = %sha_hex, - error = %e, - "blob chunk source: the rows referencing this blob could not be read — \ - scope undeterminable, the serve will be withheld" - ); - return None; - } - }; - for row in &rows { - if let Ok(meaning) = ciris_edge::blob_swarm::BlobMeaning::project(row, &blob_sha256) { - return Some(meaning.scope().clone()); - } + let scope = + ciris_edge::blob_swarm::BlobMeaning::serve_scope(self.directory.as_ref(), &blob_sha256) + .await; + if scope.is_none() { + // edge logs a failed read itself; this names the other `None`. + tracing::warn!( + blob = %sha_hex, + "blob chunk source: no community-DEK binding and no row placing this blob \ + projects a scope — scope undeterminable, the serve will be withheld (a \ + holds_bytes claim alone is possession, not meaning)" + ); } - tracing::warn!( - blob = %sha_hex, - referencing_rows = rows.len(), - "blob chunk source: no community-DEK binding and no referencing row projects a \ - scope for this blob — scope undeterminable, the serve will be withheld (a \ - holds_bytes claim alone is possession, not meaning)" - ); - None + scope } } diff --git a/src/communities.rs b/src/communities.rs index 1b38b276..f73a4bbb 100644 --- a/src/communities.rs +++ b/src/communities.rs @@ -57,6 +57,44 @@ //! The route is built anyway; `tests/community_crud.rs` pins the gap as an //! ignored red test named for #907. //! +//! **Corrected 0.5.218 (persist v49.0.0).** Both sections above describe the +//! pins before v49. #908 landed: every replicated widening and revocation is +//! judged at persist's door by the room's own `consensus_protocol` over the +//! row's co-signatures (`crate::roster_rows` gathers them), so a peer-authored +//! row no longer rides on the signature alone. #907 landed: the caller +//! admission folds the widening plane, so a widened member reads the room's +//! messages — `a_widened_member_reads_the_rooms_messages_cirispersist_907` +//! runs, un-ignored. +//! +//! # Consent to join (0.5.218) +//! +//! Nobody joins a room without their own acceptance (the maintainer's ruling +//! of 2026-09-30; CIRISConstitution#133, `FSD/MEMBERSHIP_INVITES.md`). A +//! contact grant is THIS node's consent toward them, not theirs to join. Until +//! persist can carry a proposal and record the acceptance (CIRISPersist#955, +//! v52), `POST /v1/communities/{id}/members`, the quorum flow's envelope / +//! cosign / assemble on an `add`, and a create naming anyone but the founder +//! answer 409 `membership.consent_required` (`refuse_if_joining`, +//! `crate::family_api::membership_consent_required` (removed at persist v52)). Pair rooms keep their +//! own consent — the contact grant each side authors — and are untouched. +//! +//! **At persist v52.0.0 / edge v38.0.0 the door opens as the invite flow** +//! (`crate::membership_invites`, `FSD/MEMBERSHIP_INVITES.md` §3): `POST +//! …/invites` (and `POST …/members`, its alias, answering 202 `{state: +//! "invited", proposal_id}`) writes a `membership:proposal:v1` signed by one +//! inviter — a founder under `founder_only`, any member otherwise; the invitee +//! accepts with their own pen on their own node; persist admits the WIDENING +//! that seats them only on that acceptance (`check_growth_accepted`). Under +//! `founder_only` the founder's single-signature widening is written by edge's +//! bridge on the acceptance's arrival (`membership_widener`) or when a member +//! lists the invites; under any other protocol the group seats them with the +//! existing `add` (direct or envelope → cosign → assemble), which persist now +//! refuses by name until the acceptance is held. `refuse_if_joining` and the +//! interim 409 are gone; a create naming others is refused +//! `membership.founding_member_unsigned` (persist Q1). The PAIR room no longer +//! keeps "its own consent" either: since edge v38 it is founded by its opener +//! alone and the other person joins by accepting (`contacts_chat::start_chat`). +//! //! This file deliberately allows `clippy::result_large_err`: its helpers //! return the finished refusal `Response` as their error, the same shape the //! rest of the chat surface uses, so a refusal is decided exactly once. @@ -999,7 +1037,7 @@ async fn put_widening( at: chrono::DateTime, pen: &OwnerSignerCapsule, cosignatures: Vec, -) -> Result<(), String> { +) -> Result<(), Response> { let (member, spec) = ciris_edge::community_roster::community_membership_widening( dir, room, @@ -1008,7 +1046,8 @@ async fn put_widening( at, pen.edge_signer(), ) - .await?; + .await + .map_err(write_failed)?; dir.put_community_membership_widening(SignedCommunityMembershipWidening { community_membership_widening: CommunityMembershipWidening { community_key_id: room.to_owned(), @@ -1024,7 +1063,12 @@ async fn put_widening( cosignatures, }) .await - .map_err(|e| format!("put_community_membership_widening: {e:#}")) + // persist v52 (CIRISPersist#955): a growth without the joiner's live + // acceptance is refused BY RULE — named `membership.*`, not a 500. + .map_err(|e| { + crate::membership_invites::persist_refusal(&e) + .unwrap_or_else(|| write_failed(format!("put_community_membership_widening: {e:#}"))) + }) } /// Write one revocation through persist's door (rotates the room's DEK in the @@ -1088,8 +1132,7 @@ async fn apply_change( pen, cosigs("widening", key_id), ) - .await - .map_err(write_failed)?; + .await?; } ChangeOp::Role { key_id, role } => { put_widening( @@ -1101,8 +1144,7 @@ async fn apply_change( pen, cosigs("widening", key_id), ) - .await - .map_err(write_failed)?; + .await?; } ChangeOp::Remove { key_id } => { put_revocation( @@ -1387,6 +1429,19 @@ async fn create_community( members.push(m.to_owned()); } } + // CONSENT TO JOIN (CIRISConstitution#133; the same ruling covers founding + // members, `FSD/MEMBERSHIP_INVITES.md` §4): the founding record admits the + // founder alone. A room founded with others named is REFUSED, not trimmed — + // a silently smaller room would read as success to the caller who named + // them. A founder-only room is created as before. + if !members.is_empty() { + return crate::membership_invites::founding_member_unsigned(format!( + "POST /v1/communities: {} would be founding members without signing the founding \ + record (membership_founding_member_unsigned) — found the room alone, then \ + POST /v1/communities/{{id}}/invites for each", + members.join(", ") + )); + } let protocol = req .consensus_protocol .as_deref() @@ -1635,15 +1690,39 @@ async fn read_community( } } -#[derive(Debug, Deserialize)] -struct AddMemberRequest { +/// `POST /v1/communities/{id}/members` — widen the roster by one. Since +/// persist v52 the widening follows the joiner's acceptance, so this invites. +async fn add_member( + st: State, + headers: HeaderMap, + community_id: Path, + body: axum::body::Bytes, +) -> Response { + // An ALIAS for `…/invites` since persist v52 (`FSD/MEMBERSHIP_INVITES.md` + // §3): a contact grant is OUR consent toward them, not theirs to join, so + // the route invites and answers 202 `{state: "invited", proposal_id}`. + invite(st, headers, community_id, body).await +} + +#[derive(Deserialize)] +struct InviteRequest { key_id: String, #[serde(default)] role: Option, + #[serde(default)] + expires_in_days: Option, } -/// `POST /v1/communities/{id}/members` — widen the roster by one. -async fn add_member( +/// `POST /v1/communities/{id}/invites` — invite `key_id` into the room. +/// +/// One inviter: a founder under `founder_only` (persist refuses any other +/// proposer there; repeated here so the refusal is the room's own +/// `community.not_authorized`), any active member otherwise. A contact grant +/// is NOT required: an invitation reaches a stranger's nodes under first +/// contact (CIRISEdge#756, CC rc6 3.1.3.2 — readable "without that node +/// holding the group's roster"), and the invitee's acceptance is their own. +/// A pair room's roster is its identity: its invitation is `POST /v1/chat`. +async fn invite( State(st): State, headers: HeaderMap, Path(community_id): Path, @@ -1657,20 +1736,181 @@ async fn add_member( Ok(r) => r, Err(r) => return r, }; - let req: AddMemberRequest = match serde_json::from_slice(&body) { + if room.is_pair() { + return pair_room_fixed(room.id()); + } + let req: InviteRequest = match serde_json::from_slice(&body) { Ok(r) => r, Err(e) => { return malformed(format!( - "expected {{\"key_id\": \"…\", \"role\"?: \"…\"}}: {e}" + "expected {{\"key_id\": \"…\", \"role\"?: \"…\", \"expires_in_days\"?: n}}: {e}" )) } }; - let op = ChangeOp::Add { - key_id: req.key_id, - role: req.role, + let key_id = req.key_id.trim().to_owned(); + if key_id.is_empty() { + return malformed("key_id must be a non-empty federation key id"); + } + if room.member(&key_id).is_some() { + return already_member(&key_id); + } + let protocol = room.record.consensus_protocol.clone(); + if Protocol::parse(&protocol) == Some(Protocol::FounderOnly) && !room.is_founder(&owner.key_id) + { + return not_authorized( + &protocol, + "under founder_only only a founder invites (CIRISPersist#955)", + ); + } + let role = normalize_role(req.role.as_deref()); + let expires_at = match crate::membership_invites::expiry(req.expires_in_days) { + Ok(t) => t, + Err(r) => return r, + }; + let pen = match pen(&st, &headers, &owner).await { + Ok(p) => p, + Err(r) => return r, + }; + let dir = st.engine.federation_directory(); + match ciris_edge::membership::propose( + dir.as_ref(), + ciris_edge::membership::GroupScope::Community, + room.id(), + &key_id, + role.as_deref(), + expires_at, + pen.edge_signer(), + ) + .await + { + Ok(proposal) => { + tracing::info!( + room = %room.id(), invitee = %key_id, proposal = %proposal.attestation_id, + "communities: invitation sent — the invitee joins only on their own acceptance" + ); + crate::compose::kick_replication("community invitation sent"); + crate::membership_invites::invited( + ciris_edge::membership::GroupScope::Community, + room.id(), + &proposal, + &key_id, + role.as_deref(), + ) + } + Err(e) => crate::membership_invites::refused(&e), } - .normalized(); - direct_change(&st, &headers, &owner, room, op).await +} + +/// `GET /v1/communities/{id}/invites` — every invitation into the room, with +/// its state. Members only; a delegate granted `chat_read` may read. As for a +/// household, listing is when a `founder_only` room's accepted invitees are +/// seated if the bridge has not yet done it (the caller's own founder pen). +async fn list_invites( + State(st): State, + headers: HeaderMap, + Path(community_id): Path, +) -> Response { + let owner = match read_preamble(&st, &headers).await { + Ok(o) => o, + Err(r) => return r, + }; + let mut room = match load_room_as_member(&st, &owner, &community_id).await { + Ok(r) => r, + Err(r) => return r, + }; + let dir = st.engine.federation_directory(); + let scope = ciris_edge::membership::GroupScope::Community; + let active = |r: &Room| -> std::collections::HashSet { + r.roster.iter().map(|m| m.key_id.clone()).collect() + }; + let mut invites = match crate::membership_invites::group_invites( + dir.as_ref(), + scope, + room.id(), + &active(&room), + ) + .await + { + Ok(v) => v, + Err(e) => return store_unavailable(e), + }; + let accepted = invites + .iter() + .any(|(v, _)| v.state == crate::membership_invites::STATE_ACCEPTED); + let mut seated = Vec::new(); + if accepted + && Protocol::parse(&room.record.consensus_protocol) == Some(Protocol::FounderOnly) + && room.is_founder(&owner.key_id) + && require_verb( + &owner, + CapabilityVerb::ChatAuthor, + "community.delegate_may_not_author", + ) + .is_none() + { + if let Ok(pen) = pen(&st, &headers, &owner).await { + let widener = + ciris_edge::membership::MembershipWidener::new(vec![std::sync::Arc::clone( + pen.edge_signer(), + )]); + seated = + crate::membership_invites::widen_held_acceptances(dir.as_ref(), &invites, &widener) + .await; + } + } + if !seated.is_empty() { + room = match load_room_as_member(&st, &owner, &community_id).await { + Ok(r) => r, + Err(r) => return r, + }; + invites = match crate::membership_invites::group_invites( + dir.as_ref(), + scope, + room.id(), + &active(&room), + ) + .await + { + Ok(v) => v, + Err(e) => return store_unavailable(e), + }; + } + Json(serde_json::json!({ + "community_id": room.id(), + "invites": invites.into_iter().map(|(v, _)| v).collect::>(), + "seated_now": seated, + })) + .into_response() +} + +/// `DELETE /v1/communities/{id}/invites/{proposal_id}` — the proposer +/// withdraws a pending invitation. +async fn withdraw_invite( + State(st): State, + headers: HeaderMap, + Path((community_id, proposal_id)): Path<(String, String)>, +) -> Response { + let owner = match write_preamble(&st, &headers).await { + Ok(o) => o, + Err(r) => return r, + }; + let room = match load_room_as_member(&st, &owner, &community_id).await { + Ok(r) => r, + Err(r) => return r, + }; + let pen = match pen(&st, &headers, &owner).await { + Ok(p) => p, + Err(r) => return r, + }; + let dir = st.engine.federation_directory(); + crate::membership_invites::withdraw( + dir.as_ref(), + ciris_edge::membership::GroupScope::Community, + room.id(), + &proposal_id, + &pen, + ) + .await } /// `DELETE /v1/communities/{id}/members/{key_id}` — remove a member. Naming @@ -1847,6 +2087,10 @@ async fn change_envelope( if let Err(r) = precheck(&st, &owner, &room, &op).await { return r; } + // CONSENT TO JOIN: M of N members cannot stand in for the joiner — and + // need not: the `add` widening this change writes is admitted by persist + // only on the joiner's own live acceptance (`membership.awaiting_acceptance` + // otherwise, at assemble). let pen = match pen(&st, &headers, &owner).await { Ok(p) => p, Err(r) => return r, @@ -1915,6 +2159,8 @@ async fn change_cosign( Ok(op) => op, Err(r) => return r, }; + // CONSENT TO JOIN: co-signing an `add` is the group's decision; the + // joiner's consent is persist's check at the widening's door. let pen = match pen(&st, &headers, &owner).await { Ok(p) => p, Err(r) => return r, @@ -1973,6 +2219,8 @@ async fn change_assemble( if let Err(r) = precheck(&st, &owner, &room, &op).await { return r; } + // CONSENT TO JOIN: however many members signed, persist admits the + // widening only on the joiner's acceptance (`put_widening` names it). let t = match tally( &st, &room, @@ -2022,6 +2270,14 @@ pub(crate) fn routes() -> Router { get(read_community).delete(dissolve_community), ) .route("/v1/communities/{community_id}/members", post(add_member)) + .route( + "/v1/communities/{community_id}/invites", + get(list_invites).post(invite), + ) + .route( + "/v1/communities/{community_id}/invites/{proposal_id}", + delete(withdraw_invite), + ) .route( "/v1/communities/{community_id}/members/{key_id}", delete(remove_member), diff --git a/src/compose.rs b/src/compose.rs index 5697176d..634396fa 100644 --- a/src/compose.rs +++ b/src/compose.rs @@ -407,6 +407,7 @@ pub async fn serve_with_adapter(cfg: ServerConfig, adapter: Arc) -> // chat signer is the ACTOR, and a host that registered its store under the // wire NODE must be found there, not opened over by a second store. let mls_posture = crate::mls_state::open_for_node( + &engine, &chat_node_signer.key_id, &node_resolution.node_key_id, &cfg.data_dir.join("mls-state.kv"), @@ -1201,6 +1202,26 @@ pub async fn serve_with_adapter(cfg: ServerConfig, adapter: Arc) -> } } }; + // THE PAIR-ROOM DRIVER (edge v38 / persist v52, CIRISPersist#955): the chat + // router and its driver share ONE room state, so a read and a tick never + // key a room twice. A person asks for a chat once and waits — the driver + // completes the two-step join and the MLS handshake with nobody reading + // (`contacts_chat::PairRoomDriver`). Supervised like the session claims. + let (pair_rooms_sd_tx, pair_rooms_sd_rx) = watch::channel(false); + let (chat_router, pair_room_driver) = crate::contacts_chat::router_with_driver( + Arc::clone(&engine), + Arc::clone(&chat_node_signer), + crate::user_seed_dir(&cfg), + // The live transport, so the contact ladder can run its `discover` + // rung — "is there somewhere to send" — through edge's own `RouteLens` + // instead of this module deciding what reachable means. + edge.reticulum_transport(), + // CIRISEdge#499 — the host drives the scope-address plane it armed: a + // keyed room is installed, advanced on every epoch, sealed on the + // cadence loop below. + edge.scope_lifecycle().cloned(), + ); + let pair_rooms_join = pair_room_driver.spawn(pair_rooms_sd_rx); let read = { let read = LensCore::read_api_with_extra_at_fidelity( Arc::clone(&engine), @@ -1610,20 +1631,8 @@ pub async fn serve_with_adapter(cfg: ServerConfig, adapter: Arc) -> // grant, a chat is a two-member `Community` under a derived // id, and a message is a `chat:message:v1` attestation at // `cohort_scope: community`. See `crate::contacts_chat`. - .merge(crate::contacts_chat::router( - Arc::clone(&engine), - Arc::clone(&chat_node_signer), - crate::user_seed_dir(&cfg), - // The live transport, so the contact ladder can run its - // `discover` rung — "is there somewhere to send" — through - // edge's own `RouteLens` instead of this module deciding - // what reachable means. - edge.reticulum_transport(), - // CIRISEdge#499 — the host drives the scope-address - // plane it armed: a keyed room is installed, advanced - // on every epoch, sealed on the cadence loop below. - edge.scope_lifecycle().cloned(), - )) + // (built above with its pair-room driver) + .merge(chat_router) // FILES, THE DRIVE AND NOTES (CIRISServer#622/#615): one // door for a file at any cohort, the drive that lists what // this identity can reach with `row held, bytes absent` as @@ -1646,11 +1655,24 @@ pub async fn serve_with_adapter(cfg: ServerConfig, adapter: Arc) -> Arc::clone(&engine), crate::user_seed_dir(&cfg), )) + // THE INVITE INBOX (0.5.218, CIRISPersist#955, + // FSD/MEMBERSHIP_INVITES.md): the invitee's half of + // consent-to-join — `GET /v1/self/invites`, and accept / + // decline signed with the owner's own pen. The group's + // half lives on the family and community routes. + .merge(crate::membership_invites::router( + Arc::clone(&engine), + crate::user_seed_dir(&cfg), + )) // THE OWNER'S DEVICES (FSD §2): release a node, relabel a key. .merge(crate::self_devices::router( Arc::clone(&engine), crate::user_seed_dir(&cfg), )) + // WHICH DEVICE IS ANSWERING (CC 3.1.3.1, FSD/SESSION_CLAIMS.md): + // `GET /v1/self/sessions` — every exchange of the person's a + // device holds, and which one, for "answering on ". + .merge(crate::session_claims::router(Arc::clone(&engine))) // THE AGENT-COMPAT FEDERATION EDGE SURFACE (CIRISServer#261): // GET /v1/federation/identity + /metrics, POST // /v1/federation/content/{content_id}, and the SSE bridge @@ -2062,6 +2084,7 @@ pub async fn serve_with_adapter(cfg: ServerConfig, adapter: Arc) -> | crate::self_room_drive::SelfRoomTick::NotInRoster | crate::self_room_drive::SelfRoomTick::NoOwner | crate::self_room_drive::SelfRoomTick::PublishedKeyPackage + | crate::self_room_drive::SelfRoomTick::NotHandledHere { .. } ); if !quiet || last.as_ref() != Some(&tick) { match &tick { @@ -2082,6 +2105,16 @@ pub async fn serve_with_adapter(cfg: ServerConfig, adapter: Arc) -> }) }; + // ONE DEVICE HANDLES EACH EXCHANGE (CC 3.1.3.1, CIRISPersist#782). The + // gate at each ACT site claims on demand while the person is here; this + // loop keeps a held claim renewed while they stay and lets it lapse when + // they go, so their other device can take the exchange. SUPERVISED like + // the self-room drive, whose Add/Remove it gates: an unsupervised renewer + // would keep claiming for a torn-down engine across an embedded restart. + let (session_claims_sd_tx, session_claims_sd_rx) = watch::channel(false); + let session_claims_join = + crate::session_claims::spawn(Arc::clone(&engine), session_claims_sd_rx); + crate::compose_status::phase("retention_loop"); let (retention_sd_tx, retention_sd_rx) = watch::channel(false); let retention_join = { @@ -2194,6 +2227,10 @@ pub async fn serve_with_adapter(cfg: ServerConfig, adapter: Arc) -> // the edge is gone would install destinations nothing can answer for. let _ = self_room_sd_tx.send(true); stop_step("self room drive", self_room_join).await; + let _ = session_claims_sd_tx.send(true); + stop_step("session claims", session_claims_join).await; + let _ = pair_rooms_sd_tx.send(true); + stop_step("pair-room driver", pair_rooms_join).await; // Tear down the retention loop (CIRISServer#348). Before the config // reconciler: the loop selects on the config watch, and dropping the sender // first would race its shutdown branch against a `changed()` error break. @@ -4142,6 +4179,10 @@ pub(crate) const REPLICATED_KINDS: [ciris_edge::replication::EnvelopeKind; 14] = // `put_identity_occurrence_revocation`). Without it a released device // is still one of the owner's occurrences on every OTHER device, and // keeps being sent the owner's `self` rows. + // (0.5.218, CSD-037: the plane carries SIGNED rows only, and until + // `self_devices::evict_device` no server path wrote one — release + // revoked nothing and `occurrence/revoke` used the unsigned local + // door. The producer is now the owner-signed eviction.) K::IdentityOccurrenceRevocation, // Key-level revocation (`admin_ops.rs` `put_revocation_for`). A // revocation that stays on the node that wrote it protects nobody: @@ -4363,6 +4404,10 @@ pub(crate) fn kick_replication(reason: &'static str) -> bool { ); return false; }; + // CC 5.4.6 — whatever this kick carries may have moved the announced relay + // (an announce, a release, a claim). A nudge is a coalesced `Notify`: free + // on a node that is not a relay, one recompute on one that is. + crate::announced_relay::nudge(); handle.spawn(async move { // ADMIT THE OWNER FIRST, HERE, so no caller has to remember. A round // publishes a self-plane row only if its attester is in the publish-own @@ -4587,6 +4632,16 @@ pub(crate) async fn start_replication_runtime( // // `owed` outlives one iteration on purpose — see the kick below. let mut owed = false; + // CC 5.4.6 (CIRISServer#655) — the announced relay rides THIS loop: + // it is the one that keeps the self-publish set current, and the + // relay's answer is `own ∪ relayed`, so the two are refreshed by one + // hand. The role check (`infra:serve` from a root we trust) runs on + // the relay's own period; a nudge from `kick_replication` wakes the + // loop early and recomputes the sets only on a node that relays + // (see `announced_relay::refresh`). + let mut relay_due = tokio::time::Instant::now(); + let mut relay_owed = false; + let mut nudged = false; loop { // THE SET CHANGED ⇒ CARRY THE ROWS. Adding the owner without a // kick left the rows it unlocks (owner-binding, occurrences) @@ -4619,8 +4674,51 @@ pub(crate) async fn start_replication_runtime( // polls every second instead — a bounded, startup-only window // that ends on the first dispatch, since `owed` can only be set // by a set that GAINED the owner and that happens once. - let wait = if owed { 1 } else { 30 }; - tokio::time::sleep(std::time::Duration::from_secs(wait)).await; + // THE ANNOUNCED RELAY. A changed set is rows that just became + // publishable (or stopped being), so it carries a kick exactly + // as the owner's admission does — and the same debt rule: a + // kick that found no runtime yet is still owed. + let periodic = tokio::time::Instant::now() >= relay_due; + // THE MEMBERSHIP SWEEP (persist v52 / edge v38, CIRISPersist#955) + // rides the relay's period: a runtime built before the claim + // has no `membership_widener`, so an acceptance of this owner's + // invitation would be stored and never seated. The sweep is the + // widener's own call, on the owner binding's authority. + if periodic { + if let Some(held) = SELF_PUBLISH.get() { + let node = held.own_key_ids.first().cloned().unwrap_or_default(); + let _ = crate::membership_invites::widen_own_accepted_proposals( + &held.engine, + crate::node_key::wire_identity().unwrap_or(&node), + ) + .await; + } + } + if periodic || nudged { + if let Some(held) = SELF_PUBLISH.get() { + if crate::announced_relay::refresh( + &held.engine, + &held.own_key_ids, + periodic, + ) + .await + { + relay_owed = true; + } + } + if periodic { + relay_due = + tokio::time::Instant::now() + crate::announced_relay::RELAY_REFRESH; + } + } + if relay_owed && kick_replication("announced relay set changed") { + relay_owed = false; + } + let wait = if owed || relay_owed { 1 } else { 30 }; + nudged = crate::announced_relay::wait_for_nudge(std::time::Duration::from_secs( + wait, + )) + .await; } } }); @@ -4648,6 +4746,61 @@ pub(crate) async fn start_replication_runtime( // chunk source refuse and the backend evict (CIRISEdge#614). let (pull_sink, revocations) = crate::backend::spawn_blob_puller(engine, Arc::clone(edge), node_key_id).await; + // ── THE MEMBERSHIP WIDENER (edge v38.0.0 / persist v52.0.0, CIRISPersist#955) ── + // + // Nobody joins a family or community without their own signed acceptance + // (CIRISConstitution#133). The flow is three rows: the inviter's + // `membership:proposal:v1`, the invitee's `membership:acceptance:v1`, and + // the ROSTER WIDENING that seats them — and the widening is the only one of + // the three that changes who is in the group. Edge's bridge performs it on + // arrival of an acceptance of a proposal one of the widener's identities + // issued, so the inviter's person takes no second action. Left `None` (the + // default), acceptances are STORED and nothing is ever widened by this node: + // every invite accepted on the invitee's device would sit "accepted" here + // forever. That is exactly the host-hooks-left-unset class (an optional + // edge hook the server never set disabled chat bodies for six releases), so + // it is set here, and the reason it can still be `None` is written down. + // + // WHOSE KEY. The widening must carry the founder's PERSON signature: the + // roster's consensus counts raw SEAT keys, and a seat is the person's + // fed-ID, never a device acting for it (an acceptance may be device-signed; + // a widening may not). So the signer is the owner's fed-ID pen, opened on + // the owner binding's authority (`for_owned_node` — a loop's authority is + // the binding, not a session bearer, which a boot-time hook never has). + // + // WHEN IT IS `None`. An UNCLAIMED node at boot has no owner and no pen; + // the runtime is composed once per process (`RUNTIME`), and edge takes the + // widener by value, so a node claimed AFTER the runtime started has no + // widener until its next restart. The invite routes cover that window + // (`membership_invites::widen_held_acceptances`): listing a group's + // invites as a member re-attempts the widening for every accepted, + // unseated invitee with the caller's own pen — idempotent, and the same + // edge call (`membership::widen_on_acceptance`), not a second copy of it. + // A quorum group's widening needs M-of-N and is never auto-widened by a + // single pen (persist refuses it); see `FSD/MEMBERSHIP_INVITES.md` §3.1. + let membership_widener = + match crate::owner_signer_capsule::for_owned_node(engine, wire).await { + Ok(capsule) => { + tracing::info!( + owner = %capsule.key_id(), + "membership widener installed: an acceptance of a proposal this node's \ + owner issued is widened on arrival, signed by the owner's person key \ + (CIRISPersist#955)" + ); + Some(ciris_edge::membership::MembershipWidener::new(vec![ + Arc::clone(capsule.edge_signer()), + ])) + } + Err(e) => { + tracing::info!( + reason = %e, + "membership widener NOT installed (no owner pen at runtime start) — \ + acceptances are stored and widened when a member lists the group's \ + invites; a restart after the claim installs it" + ); + None + } + }; let runtime_config = ReplicationRuntimeConfig { metrics: Some(edge.metrics()), local_key_id: Some(wire.to_string()), @@ -4674,6 +4827,30 @@ pub(crate) async fn start_replication_runtime( pull_sink, revocations, }), + // CIRISEdge#678 / CC 5.4.6 (CIRISServer#655) — the per-kind `SelfOwn` + // publish set. Installed on EVERY node and inert on all but a relay: + // until `announced_relay::refresh` finds this node holding + // `infra:serve` from a root it trusts, the closure answers `None` for + // every plane and edge keeps each on the self-publish set — the + // single-provider behaviour, byte for byte. On a relay it answers the + // Key and IdentityOccurrence planes with `own ∪ announced devices (∪ + // their owners, Key only)`, reading the SAME live set the + // `self_provider` above reads, so an owner admitted after the claim is + // in the union without a rebuild. Routes (`TransportDestination`) are + // never relayed. The closure does no I/O; see `announced_relay`. + kind_publish_selector: Some(crate::announced_relay::selector(Arc::clone( + &self_publish_keys, + ))), + // persist v52 / edge v38 (CIRISPersist#955) — see the block above. + membership_widener, + // Every OTHER field edge v38.0.0 added to a host-set surface, named so + // the next adopt diffs against a list rather than a memory: + // `ReplicationRuntimeConfig::membership_widener` (set, above) is the + // only new runtime field; `SealedContentWiring` is unchanged; and + // `PullConfig::dag_adopt_batch_chunks` (CIRISEdge#765, the batched DAG + // adopt) keeps edge's default of 16 through `..PullConfig::default()` + // in `backend::spawn_puller_with` — the default is the measured one + // (CIRISPersist#957's flat per-chunk adopt time at 2 GiB). ..ReplicationRuntimeConfig::default() }; let runtime = ReplicationRuntime::start( diff --git a/src/contacts_chat.rs b/src/contacts_chat.rs index 23b3db45..e6cadf64 100644 --- a/src/contacts_chat.rs +++ b/src/contacts_chat.rs @@ -12,6 +12,27 @@ //! | `POST /v1/chat/{id}/messages` | a `chat:message:v1` `scores` attestation | `attestation_upsert_local` + `attestation_promote(community)` | //! | `GET /v1/chat/{id}/messages` | the same rows, read back | `active_community_members` + `list_attestations_by` | //! +//! **`POST /v1/chat` at edge v38.0.0 / persist v52.0.0** (CIRISPersist#955): +//! the 2-member record above is gone — a founding record seats only its +//! signers. The route now opens the room as edge does (`chat::open_pair_room`: +//! founded by the caller alone, the peer PROPOSED as `founder`), or, when the +//! peer opened it first and their invitation is held here, ACCEPTS it +//! (`chat::accept_pair_proposal`); the opener's node widens on the +//! acceptance. The answer gains `state` (`open` / `invited` / `accepted` / +//! `awaiting_invitation`) and `proposal_id`; `member_key_ids` is the active +//! roster. Residual, as edge names it for its harness: two people opening the +//! SAME pair before either's record reaches the other author two different +//! records under one id (a `Conflict` on replication); edge's harness avoids it +//! by letting only `PairRole::Creator` open, and a person-driven route cannot +//! wait for the other side, so it is named here rather than prevented. +//! **Corrected the same cut, from the native chat ladder:** the residual was +//! not rare — both people open the chat at once as a matter of course, and +//! the two records then withheld each other (CIRISEdge#758's group-record +//! gate). So the route follows edge's rule after all: only the creator founds; +//! the joiner answers `awaiting_invitation` until the creator's invitation is +//! held, then accepts it — the request kept on disk (`crate::pair_intents`), +//! so a restart in between does not forget it. +//! //! # Why a contact IS a replication-consent grant //! //! The client's `ContactsScreen` already renders `GET /v1/federation/peers` @@ -2594,9 +2615,7 @@ fn machine_authored_refusal(grants: Vec) -> Response { crate::auth::refusal::refuse_with( StatusCode::CONFLICT, "consent.grant_not_owner_authored", - "that consent was authored by the node, not by you (a provisional grant from before \ - this node was claimed). The owner migration re-signs it as yours; the node will not \ - withdraw consent on your behalf", + "This node wrote that consent before you signed it, so it isn't yours to withdraw here, and the node will not withdraw consent on your behalf. It is still active.", serde_json::json!({ "grants": grants }), ) } @@ -2618,8 +2637,7 @@ async fn withdrawal_pen( crate::owner_signer_capsule::CapsuleRefusal::Delegated => refuse( StatusCode::FORBIDDEN, "consent.delegate_may_not_withdraw", - "a delegated session may not withdraw the owner's consent — the withdrawal is \ - signed with the owner's own key", + "A delegated session can't withdraw the owner's consent. The withdrawal is signed with the owner's own key.", ), other => refuse( StatusCode::FORBIDDEN, @@ -3007,75 +3025,130 @@ async fn start_chat( } } + // ── THE PAIR ROOM BY CONSENT (edge v38.0.0 / persist v52.0.0) ─────────── + // + // Until 0.5.218 this route authored a TWO-founder `Community` record with + // both people on it, and "the pair room keeps its own consent — the contact + // grant each side authors". persist v52 (CIRISPersist#955, Q1) admits a + // founding record only for the members who SIGNED it, so that record is now + // refused `membership_founding_member_unsigned` at the door — and the + // ruling behind it says why that is right: a contact grant is OUR consent + // toward them, not theirs to share a room with us. So the room is opened the + // way edge's own harness opens it (`chat::open_pair_room`): the OPENER founds + // it alone as `founder`, and PROPOSES the other person at role `founder`; the + // other person's node holds the proposal (it reaches strangers under first + // contact, CIRISEdge#756), and when THEY call this route their node ACCEPTS + // it (`chat::accept_pair_proposal`); the opener's node then widens the roster + // on the acceptance's arrival (`ReplicationRuntimeConfig::membership_widener`) + // — or here, on the opener's next call, with the opener's own pen. The END + // STATE is the pre-v52 one: two founders, `unanimous`, each an authority + // root (CC 4.5.4). + // + // Who signs what: the RECORD and the PROPOSAL are the node signer acting + // for its owner (the record's bytes stay a function of (opener, peer, + // founded_at), as before); the ACCEPTANCE is the node acting for its owner + // too — persist admits a device's reply for its person (`signer_acts_for`), + // and this call IS the person's act; the WIDENING is the opener's PERSON key + // (the roster counts seat keys), which is why it needs their pen. let community_id = pair_community_key_id(&owner.key_id, &key_id); - let mut expected_members = vec![owner.key_id.clone(), key_id.clone()]; - expected_members.sort(); + let mut pair_members = vec![owner.key_id.clone(), key_id.clone()]; + pair_members.sort(); // Before the branch: an existing room needs this as much as a new one. ensure_owner_content_occurrence(&st, &owner.key_id).await; - match directory.lookup_community(&community_id).await { - Ok(Some(existing)) => { - // THE ROSTER IS PART OF THE IDENTITY. The pair id is derivable by - // anyone, so a peer can pre-replicate a community under it carrying - // the pair PLUS an extra member — and a front door that accepts any - // row at the derived id would open that room, with every subsequent - // community-scoped message readable by the stowaway. Same sorted- - // member equality the insert-race arm applies: a room at this id - // that is not EXACTLY this pair is a conflict, not a chat. - // - // THE FOLD, NOT THE RECORD (persist v48, CIRISPersist#860). A - // stowaway no longer needs a record with three names on it: the - // record is never rewritten to grow, and a widening row naming a - // third member rides its own plane. Comparing `existing.members` - // would have passed a pair record with a widened third party and - // opened the room with them in it. - let mut existing_members: Vec = - match active_roster(&*directory, &existing).await { - Ok(r) => r.into_iter().map(|m| m.key_id).collect(), - Err(e) => { - return refuse( - StatusCode::SERVICE_UNAVAILABLE, - "chat.store_unavailable", - format!("active roster: {e}"), - ) - } - }; - existing_members.sort(); - if existing_members != expected_members { + + // 1. The other person already opened it and invited us: accepting IS + // opening the chat from this side. + let pending = + match ciris_edge::chat::pair_proposal_for(&*directory, &owner.key_id, &key_id).await { + Ok(p) => p, + Err(e) => { return refuse( - StatusCode::CONFLICT, - "chat.community_shape_conflict", - format!( - "a community already exists under the derived pair id but its roster is not this pair ({} member(s), expected 2) — refusing to open it as this chat", - existing_members.len() - ), - ); + StatusCode::SERVICE_UNAVAILABLE, + "chat.store_unavailable", + format!("pair_proposal_for: {e}"), + ) } - return ( - StatusCode::OK, - Json(StartChatResponse { - community_id: existing.community_key_id, - community_name: existing.community_name, - member_key_ids: existing_members, - cohort_scope: cohort_scope::COMMUNITY, - freshly_created: false, - }), + }; + // THE PERSON'S ACT, KEPT (`crate::pair_intents`): written BEFORE anything + // depends on it — the joiner's `awaiting_invitation` below promises that + // the creator's invitation will be accepted when it lands, even across a + // restart, so an intent that could not be stored refuses here instead. + if let Err(e) = crate::pair_intents::record(&st.user_seed_dir, &community_id, &key_id) { + return refuse( + StatusCode::SERVICE_UNAVAILABLE, + "chat.store_unavailable", + format!("record the pair-room request: {e}"), + ); + } + // Only an invitation FROM this contact's person is accepted here: the + // derived id names the pair, but anyone can propose into an id. + let pending = match pending { + Some(p) => { + let proposer = ciris_persist::federation::admission::admission_identity_for_writer( + &*directory, + &p.proposal.attesting_key_id, ) - .into_response(); + .await + .unwrap_or_default(); + (proposer == key_id).then_some(p) } - Ok(None) => {} - Err(e) => { - return refuse( - StatusCode::SERVICE_UNAVAILABLE, - "chat.store_unavailable", - format!("lookup_community: {e}"), - ) + None => None, + }; + if let Some(pending) = pending { + if let Err(e) = ciris_edge::chat::accept_pair_proposal( + &*directory, + &pending.proposal.attestation_id, + &st.node_signer, + ) + .await + { + return crate::membership_invites::refused(&e); } + crate::compose::kick_replication("pair room invitation accepted"); + tracing::info!( + room = %community_id, peer = %key_id, + "chat: accepted the pair room's invitation — the opener's node seats us" + ); + return pair_room_response( + &st, + &community_id, + &pair_members, + false, + PAIR_STATE_ACCEPTED, + Some(pending.proposal.attestation_id), + ) + .await; + } + + // 2. ONLY THE CREATOR FOUNDS. Two people opening the same pair at once + // each founded their OWN record under the one derived id, each proposing + // the other — two records, two invitations, and each node's + // group-record gate then withheld its record AND its invitation from the + // peer (the peer is a member of neither copy). Measured on the v38 chat + // ladder: both rooms `member_count: 1`, both inboxes empty, the room + // never keyed. `PairRole` is a function of the two fed-IDs (the + // lexicographically smaller is the creator), so exactly one side ever + // founds — edge's own harness rule for the same reason. The joiner, with + // no invitation held yet, says so and founds nothing; its next call + // accepts the creator's invitation once it arrives (step 1). + if ciris_edge::chat::PairRole::of(&owner.key_id, &key_id) == ciris_edge::chat::PairRole::Joiner + { + return pair_room_response( + &st, + &community_id, + &pair_members, + false, + PAIR_STATE_AWAITING_INVITATION, + None, + ) + .await; } - // `founded_at` is DERIVED, not read from a clock: the later `valid_from` of - // the two member key records. A community cannot predate its members, and a - // derived instant is the same on both nodes — a `Utc::now()` here would make - // the two ends author rows that differ in a signed field. + // 3. Open it (or re-open it) as the creator: found alone, propose the peer. + // `founded_at` is DERIVED, not read from a clock: the later `valid_from` + // of the two member key records — a community cannot predate its + // members, and a re-open re-derives the same instant, so the idempotent + // re-put is byte-identical. let mut founded_at = None; for member in [&owner.key_id, &key_id] { match directory.lookup_public_key(member).await { @@ -3108,117 +3181,269 @@ async fn start_chat( "could not resolve either member's key record", ); }; - - let member_key_ids = expected_members; - // ── THE ROOM RECORD IS EDGE'S (CIRISServer#524) ───────────────────────── - // - // This built the `Community` by hand, and the roster it produced differed - // from edge's in the one way that matters: both members carried - // `role: None`. Edge names both people `founder` outright, and says why — - // CC 4.5.4 / §11.11, no unmoderated federated space: "persist refuses to - // federate any content keyed on a community that has no live named - // moderator, and a named moderator exists iff the community has a - // steward-bound AUTHORITY root". A pair room is two equals, so the record - // makes each an authority root BY CONSTRUCTION rather than by the accident - // of a protocol setting. - // - // `community_name` was a second copy too — a sorted `"{a} <-> {b}"`, the - // same string edge formats — and it sits INSIDE `Community::signing_envelope`, - // so a drift there would have been two different signed records under one id. - // - // Edge's own note: "Everything that opens a pair room — the mesh harness, - // the tests, a consumer — builds it here, so the roster shape cannot drift - // between them." We were the consumer that drifted. - let signed = match ciris_edge::chat::signed_pair_community( + // THE ROSTER IS PART OF THE IDENTITY. The pair id is derivable by anyone, + // so a peer can pre-replicate a community under it carrying a third + // member — and a front door that opened any row at the derived id would + // open that room with a stowaway in it. The FOLD (never the record) must + // name no one outside the pair; since v52 a pending room names only its + // opener, which is a subset, not a conflict. + let held = match directory.lookup_community(&community_id).await { + Ok(h) => h, + Err(e) => { + return refuse( + StatusCode::SERVICE_UNAVAILABLE, + "chat.store_unavailable", + format!("lookup_community: {e}"), + ) + } + }; + if let Some(existing) = &held { + let roster: Vec = match active_roster(&*directory, existing).await { + Ok(r) => r.into_iter().map(|m| m.key_id).collect(), + Err(e) => { + return refuse( + StatusCode::SERVICE_UNAVAILABLE, + "chat.store_unavailable", + format!("active roster: {e}"), + ) + } + }; + if roster.iter().any(|k| !pair_members.contains(k)) { + return refuse( + StatusCode::CONFLICT, + "chat.community_shape_conflict", + format!( + "a community already exists under the derived pair id but its roster is \ + not this pair ({} member(s), expected at most 2) — refusing to open it \ + as this chat", + roster.len() + ), + ); + } + // The other person founded it and their invitation has not reached + // this node yet: there is nothing to accept and nothing to found. + if !roster.iter().any(|k| k == &owner.key_id) && roster.iter().any(|k| k == &key_id) { + return pair_room_response( + &st, + &community_id, + &pair_members, + false, + PAIR_STATE_AWAITING_INVITATION, + None, + ) + .await; + } + } + let expires_at = + chrono::Utc::now() + chrono::Duration::days(crate::membership_invites::DEFAULT_INVITE_DAYS); + let opened = match ciris_edge::chat::open_pair_room( + &*directory, &owner.key_id, &key_id, founded_at, + expires_at, &st.node_signer, ) .await { - Ok(c) => c, + Ok(o) => o, Err(e) => { return refuse( StatusCode::INTERNAL_SERVER_ERROR, - "chat.store_unavailable", - format!("signed_pair_community: {e}"), + "chat.community_create_failed", + format!("open_pair_room: {e}"), ) } }; - // Signed by edge over `Community::signing_envelope()` — the JCS field-set - // gate verifies exactly those bytes, which is why the record is never - // hand-rolled on either side of the room. - // The name the record actually carries — read off the signed room rather - // than re-derived, so the response cannot describe a room differently from - // the bytes that were stored. - let community_name = signed.community.community_name.clone(); - if let Err(e) = st.engine.federation_directory().put_community(signed).await { - // A LOST RACE IS A SUCCESS SOMEONE ELSE ALREADY HAD. Two concurrent - // POSTs (double-tap, client retry, two devices) can both observe - // `lookup_community` returning None above; one insert wins and the - // other lands here on the primary-key conflict. - // - // persist v38.2.0 NARROWED when this arm fires: an IDENTICAL re-put is - // now an Ok no-op at the door (first-accepted authority signature - // preserved), so the ordinary race never errors at all. What still - // reaches here is the typed Conflict for DIFFERING content under the - // id — a roster-fork signal — plus any backend that predates the - // verdict semantics. The roster-equality re-read below is exactly the - // fork discriminator: matching roster → idempotent success; - // differing → the 500 carries the substrate's own Conflict message. The room the loser - // asked for EXISTS — reporting 500 would break the route's advertised - // idempotency exactly on the inputs where idempotency matters. So on - // failure, re-read the derived id: if the room is there with the same - // convergent identity, this call is the second arrival, not an error. - // A re-read that finds nothing (or a different shape) is a REAL - // failure and keeps the 500 with the original error. - if let Ok(Some(existing)) = directory.lookup_community(&community_id).await { - // The fold, for the same reason as the lookup arm above. - let mut existing_members: Vec = active_roster(&*directory, &existing) + if opened.founded { + // The ROSTER is a row too — and so is the invitation. `share_in_room` + // kicks for every chat row that goes through it, but the record and + // the proposal are written straight to the directory here, so without + // this they would wait for a cadence tick while the KeyPackage sent + // milliseconds later did not. + crate::compose::kick_replication("chat room founded and peer invited"); + } + + // 4. The peer accepted and this node has not seated them yet (the bridge's + // widener seats them on arrival; this covers a node whose runtime + // started before its claim): the opener's own pen widens, now. + let state = match &opened.proposal { + None => PAIR_STATE_OPEN, + Some(proposal) => { + let invitee_accepted = directory + .list_attestations_for(&key_id) .await - .map(|r| r.into_iter().map(|m| m.key_id).collect()) - .unwrap_or_default(); - existing_members.sort(); - if existing_members == member_key_ids { - return ( - StatusCode::OK, - Json(StartChatResponse { - community_id: existing.community_key_id, - community_name: existing.community_name, - member_key_ids: existing_members, - cohort_scope: cohort_scope::COMMUNITY, - freshly_created: false, - }), - ) - .into_response(); + .map(|rows| { + rows.into_iter().find(|r| { + ciris_edge::membership::dimension_of(r) + == Some(ciris_edge::membership::ACCEPTANCE_DIMENSION) + && r.attestation_envelope + .get(ciris_persist::federation::envelope::paths::REFERENCES_ATTESTATION_ID) + .and_then(serde_json::Value::as_str) + == Some(proposal.attestation_id.as_str()) + }) + }) + .unwrap_or(None); + match invitee_accepted { + Some(acceptance) => { + match crate::owner_signer_capsule::acquire( + &st.engine, + bearer_of(&headers), + &owner.key_id, + st.user_seed_dir.clone(), + ) + .await + { + Ok(pen) => { + let widener = + ciris_edge::membership::MembershipWidener::new(vec![Arc::clone( + pen.edge_signer(), + )]); + match ciris_edge::membership::widen_on_acceptance( + &*directory, + &acceptance, + &widener, + ) + .await + { + Ok(_) => { + crate::compose::kick_replication("pair room peer seated"); + PAIR_STATE_OPEN + } + Err(e) => return crate::membership_invites::refused(&e), + } + } + // A delegate cannot open the pen: the bridge seats them. + Err(_) => PAIR_STATE_ACCEPTED, + } + } + None => PAIR_STATE_INVITED, } } - return refuse( - StatusCode::INTERNAL_SERVER_ERROR, - "chat.community_create_failed", - format!("put_community: {e}"), - ); + }; + pair_room_response( + &st, + &community_id, + &pair_members, + opened.founded, + state, + opened.proposal.map(|p| p.attestation_id), + ) + .await +} + +/// **Advance a pair room's two-step join on a read** — the joiner accepts a +/// held invitation it asked for; the creator seats a held acceptance with the +/// person's own pen. Both are the same edge calls `POST /v1/chat` makes; this +/// runs them when the person is LOOKING at the room, so the join completes as +/// soon as the other side's row arrives instead of on the next sweep. Every +/// failure is logged and left for the next read: a read never fails on it. +async fn advance_pair_join(st: &ChatState, headers: &HeaderMap, owner: &Owner, room: &str) { + let directory = st.engine.federation_directory(); + // The joiner's half: accept ONLY an invitation that matches an intent the + // person recorded with their own `POST /v1/chat` (`crate::pair_intents`). + let _ = crate::pair_intents::advance( + &*directory, + &st.user_seed_dir, + &owner.key_id, + &st.node_signer, + ) + .await; + let Ok(Some(record)) = directory.lookup_community(room).await else { + return; + }; + // A founder BY THE FOLD (an active `founder` seat), never the record. + let founder = active_roster(&*directory, &record).await.is_ok_and(|r| { + r.iter().any(|m| { + m.key_id == owner.key_id + && m.role.as_deref() + == Some(ciris_persist::federation::admission::MEMBER_ROLE_FOUNDER) + }) + }); + if !founder { + return; } - // The ROSTER is a row too. `share_in_room` kicks for every chat row that - // goes through it (KeyPackage, Welcome, message), but the `Community` - // record is written straight to the directory here and returns — so - // without this the invitation itself was the one row in the conversation - // that waited for a cadence tick, while the KeyPackage sent milliseconds - // later did not. Fresh create only: the idempotent second arrival above - // returns early, so a client retry does not re-round. - crate::compose::kick_replication("chat room roster created"); - ( - StatusCode::OK, - Json(StartChatResponse { - community_id: community_id.clone(), - community_name, - member_key_ids, - cohort_scope: cohort_scope::COMMUNITY, - freshly_created: true, - }), + let Ok(pen) = crate::owner_signer_capsule::acquire( + &st.engine, + bearer_of(headers), + &owner.key_id, + st.user_seed_dir.clone(), ) - .into_response() + .await + else { + return; + }; + let invites = match crate::membership_invites::group_invites( + &*directory, + ciris_edge::membership::GroupScope::Community, + room, + &active_roster(&*directory, &record) + .await + .map(|r| r.into_iter().map(|m| m.key_id).collect()) + .unwrap_or_default(), + ) + .await + { + Ok(v) => v, + Err(_) => return, + }; + let widener = + ciris_edge::membership::MembershipWidener::new(vec![Arc::clone(pen.edge_signer())]); + let _ = + crate::membership_invites::widen_held_acceptances(&*directory, &invites, &widener).await; +} + +/// The pair room is open: both people are active founders. +const PAIR_STATE_OPEN: &str = "open"; +/// This side opened it and invited the other person; they have not accepted. +const PAIR_STATE_INVITED: &str = "invited"; +/// The invitation is accepted; the opener's node has not seated the joiner yet +/// (on the joiner's side: always, until the widening replicates back). +const PAIR_STATE_ACCEPTED: &str = "accepted"; +/// The other person opened the room, and their invitation has not reached this +/// node yet. Nothing to accept; call again when it arrives. +const PAIR_STATE_AWAITING_INVITATION: &str = "awaiting_invitation"; + +/// `POST /v1/chat`'s answer: the room as the FOLD sees it now, plus where the +/// two-step join stands (`state`, and the invitation's id when there is one). +/// 200 in every state — the route has always been idempotent, and a client +/// tells the states apart by `state` (and by `member_key_ids`, which lists +/// only the active members). +async fn pair_room_response( + st: &ChatState, + community_id: &str, + pair_members: &[String], + freshly_created: bool, + state: &'static str, + proposal_id: Option, +) -> Response { + let directory = st.engine.federation_directory(); + let (name, mut members) = match directory.lookup_community(community_id).await { + Ok(Some(c)) => { + let roster: Vec = active_roster(&*directory, &c) + .await + .map(|r| r.into_iter().map(|m| m.key_id).collect()) + .unwrap_or_default(); + (c.community_name, roster) + } + // The joiner's side before the record has replicated: the name is the + // sorted pair, the same string edge signs into the record. + _ => (pair_members.join(" <-> "), Vec::new()), + }; + members.sort(); + let mut body = serde_json::to_value(StartChatResponse { + community_id: community_id.to_owned(), + community_name: name, + member_key_ids: members, + cohort_scope: cohort_scope::COMMUNITY, + freshly_created, + }) + .unwrap_or_default(); + if let Some(obj) = body.as_object_mut() { + obj.insert("state".into(), serde_json::json!(state)); + obj.insert("proposal_id".into(), serde_json::json!(proposal_id)); + } + (StatusCode::OK, Json(body)).into_response() } // ─── The message row ──────────────────────────────────────────────────────── @@ -4181,6 +4406,25 @@ async fn other_member( .find(|k| *k != owner.key_id) { Some(peer) => Ok(peer), + // edge v38 / persist v52: a pair room between its two steps — the + // opener founded it alone and the other person has not been seated + // yet. The other person is named by the INVITATION, so the room + // reads as "waiting on the peer", not as a malformed room. + None if community_id.starts_with(ciris_edge::chat::PAIR_COMMUNITY_PREFIX) => { + crate::membership_invites::pending_pair_invitee( + &*directory, + community_id, + &owner.key_id, + ) + .await + .ok_or_else(|| { + refuse( + StatusCode::CONFLICT, + "chat.not_a_pair_room", + NOT_A_PAIR_ROOM, + ) + }) + } None => Err(refuse( StatusCode::CONFLICT, "chat.not_a_pair_room", @@ -4271,6 +4515,11 @@ async fn list_messages( ) { return resp; } + // edge v38: a pair room joins in two steps; reading it advances whichever + // step this side holds (the joiner's acceptance, the creator's widening). + if community_id.starts_with(PAIR_COMMUNITY_PREFIX) { + advance_pair_join(&st, &headers, &owner, &community_id).await; + } if let Err(r) = require_member(&st, &owner, &community_id).await { return r; } @@ -4459,6 +4708,19 @@ pub fn router( routes: Option>, scope_lifecycle: Option>, ) -> Router { + router_with_driver(engine, node_signer, user_seed_dir, routes, scope_lifecycle).0 +} + +/// [`router`], plus the [`PairRoomDriver`] over the SAME room state — the +/// MLS groups the routes hold are the ones the driver advances, so a read +/// and a tick never key one room twice. +pub fn router_with_driver( + engine: Arc, + node_signer: Arc, + user_seed_dir: std::path::PathBuf, + routes: Option>, + scope_lifecycle: Option>, +) -> (Router, PairRoomDriver) { let state = ChatState { engine, user_seed_dir, @@ -4467,7 +4729,7 @@ pub fn router( routes, scope_lifecycle, }; - Router::new() + let router = Router::new() .route( "/v1/contacts", axum::routing::get(list_contacts).post(add_contact), @@ -4492,7 +4754,8 @@ pub fn router( // CIRISServer#594 — N-member rooms and affiliations: the same state, // the same owner gate, the same signer. .merge(crate::communities::routes()) - .with_state(state) + .with_state(state.clone()); + (router, PairRoomDriver(state)) } #[cfg(test)] @@ -4755,3 +5018,154 @@ mod tests { } } } + +// ─── The pair-room driver (edge v38 / persist v52) ────────────────────────── + +/// **A pair room converges with nobody reading it.** +/// +/// A person asks for a chat ONCE (`POST /v1/chat`) and waits. Since edge v38 +/// the room needs four acts across two nodes before a message can be sealed: +/// the joiner ACCEPTS the creator's invitation (only one matching an intent +/// the person recorded — [`crate::pair_intents`]), the creator WIDENS the +/// roster on that acceptance, the joiner publishes its MLS KeyPackage, and the +/// creator answers with the Welcome. Every one of them ran only inside a +/// request (a read or a send of the room). The Docker chat ladder showed the +/// consequence: node-b called `POST /v1/chat` once and never read the room +/// again, so its KeyPackage was never published, node-a's room stayed +/// `AwaitingPeer`, and node-a's send was refused 503 (the native harness hid +/// it by polling both sides). This driver runs the same calls on its own +/// `loop_cadence` slot (`pair_rooms`) — [`pair_intents::advance`], edge's +/// `widen_on_acceptance` via [`crate::membership_invites::widen_held_acceptances`], +/// and [`room_key`] with the person's pen — for every pair room this node's +/// person asked for or is seated in. Authority: the owner BINDING +/// (`owner_signer_capsule::for_owned_node_in`), as for every loop that signs +/// for the person; an unclaimed node drives nothing. +/// +/// [`pair_intents::advance`]: crate::pair_intents::advance +#[derive(Clone)] +pub struct PairRoomDriver(ChatState); + +/// The driver's period: the node's common 30 s (`loop_cadence`'s premise). +pub const PAIR_ROOM_DRIVE_EVERY: std::time::Duration = std::time::Duration::from_secs(30); + +impl PairRoomDriver { + /// One pass. Returns `(room, state)` for every seated pair room it keyed or + /// tried to (`state` is [`RoomHandshake`]'s word). + pub async fn drive_once(&self) -> Vec<(String, String)> { + drive_pair_rooms(&self.0).await + } + + /// The loop, until `shutdown` flips. + pub fn spawn( + self, + mut shutdown: tokio::sync::watch::Receiver, + ) -> tokio::task::JoinHandle<()> { + tokio::spawn(async move { + let mut schedule = + crate::loop_cadence::Cadence::new("pair_rooms", PAIR_ROOM_DRIVE_EVERY); + loop { + if *shutdown.borrow() { + break; + } + tokio::select! { + () = schedule.tick() => {} + _ = shutdown.changed() => { + if *shutdown.borrow() { break; } + continue; + } + } + let _ = self.drive_once().await; + } + tracing::info!("pair-room driver stopped"); + }) + } +} + +async fn drive_pair_rooms(st: &ChatState) -> Vec<(String, String)> { + let Ok(node) = crate::family_api::this_node_key(&st.engine).await else { + return Vec::new(); + }; + let Ok(pen) = + crate::owner_signer_capsule::for_owned_node_in(&st.engine, &node, st.user_seed_dir.clone()) + .await + else { + return Vec::new(); + }; + let me = pen.key_id().to_owned(); + let dir = st.engine.federation_directory(); + // 1. The joiner's acceptance of a requested room's invitation. + let _ = crate::pair_intents::advance(&*dir, &st.user_seed_dir, &me, &st.node_signer).await; + // 2. Every pair room this person is in: the record's members, plus the + // widening plane (a joiner is seated by a widening — the record never + // grows, persist Q2). + let mut rooms: std::collections::BTreeSet = dir + .list_communities_for_member(&me) + .await + .map(|v| v.into_iter().map(|c| c.community_key_id).collect()) + .unwrap_or_default(); + let mut cursor = None; + loop { + let Ok(page) = dir + .list_signed_community_membership_widenings_since(cursor.clone(), 500) + .await + else { + break; + }; + for w in &page { + let row = &w.widening.community_membership_widening; + if row.member_key_id == me { + rooms.insert(row.community_key_id.clone()); + } + } + if page.len() < 500 { + break; + } + cursor = page.last().map(|p| p.resume_pair()); + } + let widener = + ciris_edge::membership::MembershipWidener::new(vec![Arc::clone(pen.edge_signer())]); + let mut out = Vec::new(); + for room in rooms + .into_iter() + .filter(|r| r.starts_with(PAIR_COMMUNITY_PREFIX)) + { + let Ok(Some(record)) = dir.lookup_community(&room).await else { + continue; + }; + let Ok(roster) = active_roster(&*dir, &record).await else { + continue; + }; + let keys: std::collections::HashSet = + roster.iter().map(|m| m.key_id.clone()).collect(); + if !keys.contains(&me) { + continue; + } + // 3. The creator's widening on a held acceptance. + if keys.len() < 2 { + if let Ok(invites) = crate::membership_invites::group_invites( + &*dir, + ciris_edge::membership::GroupScope::Community, + &room, + &keys, + ) + .await + { + let _ = + crate::membership_invites::widen_held_acceptances(&*dir, &invites, &widener) + .await; + } + continue; + } + // 4. The MLS handshake, as a read with the person's pen drives it. + let Some(peer) = keys.iter().find(|k| **k != me).cloned() else { + continue; + }; + match room_key(st, &me, &peer, Some(pen.edge_signer())).await { + Ok(state) => out.push((room, format!("{state:?}"))), + Err(e) => { + tracing::debug!(room = %room, error = %e, "pair-room driver: handshake not advanced") + } + } + } + out +} diff --git a/src/drive.rs b/src/drive.rs index dc574878..475870be 100644 --- a/src/drive.rs +++ b/src/drive.rs @@ -41,15 +41,14 @@ use std::sync::Arc; -use axum::body::Bytes; -use axum::extract::rejection::BytesRejection; -use axum::extract::{DefaultBodyLimit, Path, Query, State}; +use axum::body::Body; +use axum::extract::{Path, Query, State}; use axum::http::{header, HeaderMap, HeaderValue, StatusCode}; use axum::response::{IntoResponse, Response}; use axum::{Json, Router}; use serde::{Deserialize, Serialize}; -use ciris_edge::files::{self, FileWrite}; +use ciris_edge::files::{self, FileStreamWrite}; use ciris_edge::scope_room::ScopeRoom; use ciris_persist::federation::{Attestation, BlobError}; use ciris_persist::prelude::Engine; @@ -58,23 +57,63 @@ use ciris_persist::prelude::Engine; /// the writer and the reader cannot disagree about which files are notes. const NOTE_MEDIA_TYPE: &str = "text/plain; charset=utf-8"; -/// **The largest file this node writes or reads whole: persist's chunk-DAG -/// whole-read cap (64 MiB).** +mod multipart; +mod streaming; + +/// **The largest file this node reads WHOLE: persist's chunk-DAG whole-read +/// cap (64 MiB).** +/// +/// Since 0.5.218 this bounds only the paths that must hold a file in memory: +/// the JSON read `GET /v1/files/{id}` (it base64s the bytes into one JSON +/// value), the JSON upload form (it arrives as one base64 string), a +/// multipart upload that declares no `size` (the node cannot declare a length +/// to the seal it does not know, so it collects the part — up to this — and +/// declares what it collected), and the plaintext digest on `/meta`. Every +/// other path streams: a multipart upload WITH `size` seals through edge's +/// `files::publish_stream` chunk by chunk ([`STREAMED_FILE_CEILING`]), `?raw=1` +/// serves a file above this through `FileRow::chunks()` and any `Range` +/// through `FileRow::open_range` windows, and `move` above it re-seals from +/// the source's chunk walk. /// -/// One number for both directions, on purpose. An upload above it would seal -/// fine (edge seals anything above 1 MiB as a chunk DAG) but could then never be -/// opened by the JSON read, `GET /v1/files/{id}`, or by `move`, which all read -/// whole — so the node would accept bytes it cannot give back the same way. A -/// file received from a peer above it is still served, by `?raw=1` with `Range`. +/// Before 0.5.218 this was one number for both directions ON PURPOSE — an +/// upload above it would seal but could not be read back whole, so the node +/// refused to accept what it could not give back the same way. Edge v36.1.0 +/// (CIRISEdge#737 / #744) removed the reason: there is now a way back for +/// every size the way in accepts. pub const WHOLE_READ_CAP: usize = ciris_persist::federation::chunk_dag_cascade::DAG_WHOLE_READ_CAP_BYTES as usize; -/// The request-body ceiling for the upload routes: the cap, as base64 (the JSON -/// form inflates by 4/3), plus 1 MiB for the form's other members and multipart -/// framing. Applied to `POST /v1/files` and `PUT /v1/files/{id}` ONLY — every -/// other route keeps axum's 2 MB default, which is the right size for JSON. +/// The request-body ceiling for the JSON upload form: the cap, as base64 (the +/// JSON form inflates by 4/3), plus 1 MiB for the form's other members. The +/// JSON form cannot stream — its bytes are one string value, and serde needs +/// the value whole — so it keeps the whole-read cap. Enforced by the upload +/// handlers themselves (they read the raw body so the multipart form can +/// stream), and applied to `POST /v1/files` and `PUT /v1/files/{id}` ONLY — +/// every other route keeps axum's 2 MB default, which is the right size for +/// JSON. pub const UPLOAD_BODY_LIMIT: usize = WHOLE_READ_CAP.div_ceil(3) * 4 + 1024 * 1024; +/// **The largest single file this node takes through the streamed upload: +/// edge's stated single-file ceiling, ~2.5 GiB.** +/// +/// Edge v36.0.0's release notes (#744): "one file maxes out near 2.5 GiB +/// because persist stores the sealed manifest inline under its 1 MiB cap +/// (CIRISPersist#954, persist v52)". Edge exports no constant for it — the +/// limit is an emergent property of the manifest's size, not a check anyone +/// makes — so this node states it, at edge's number, and refuses above it by +/// name (`drive.too_large`, 413) before a byte is sealed, rather than +/// streaming 2.6 GiB to a manifest persist then refuses as a 500. A file +/// within a few MiB BELOW it may still meet persist's manifest cap and answer +/// `drive.publish_failed`; edge measured 2 GiB end to end. Raise it when this +/// node adopts persist v52 (edge v37). +pub const STREAMED_FILE_CEILING: u64 = 2560 * 1024 * 1024; + +/// The request-body ceiling for the streamed (multipart) upload: +/// [`STREAMED_FILE_CEILING`] plus 1 MiB for the form's fields and framing. +/// Checked against `Content-Length` before the body is read and counted as it +/// is read, since a chunked request declares no length. +pub const STREAMED_UPLOAD_BODY_LIMIT: u64 = STREAMED_FILE_CEILING + 1024 * 1024; + /// The most rows one `GET /v1/drive` page returns. A bigger `limit` is clamped, /// not refused: a client asking for "everything" gets a page and a `resume`. pub const MAX_PAGE: usize = 500; @@ -127,7 +166,8 @@ pub struct FileWriteRequest { pub room_id: Option, /// Base64 bytes, up to [`WHOLE_READ_CAP`] decoded. Edge seals anything /// above its 1 MiB envelope bound as a chunk DAG (CIRISEdge#633); for a - /// large file prefer `multipart/form-data`, which skips the 4/3 inflation. + /// large file use `multipart/form-data` with a `size` field, which skips + /// the 4/3 inflation AND streams (up to [`STREAMED_FILE_CEILING`]). pub bytes_base64: String, #[serde(default)] pub media_type: Option, @@ -212,6 +252,10 @@ pub struct DriveEntry { pub asserted_at: String, pub filename: Option, pub media_type: Option, + /// How the name and type were read (CIRISEdge#698): `clear`, `opened` + /// (the sealed descriptor opened here) or `sealed` (held, not opened on + /// this device — `filename`/`media_type` are then unknown, not absent). + pub description: String, /// `here` when the bytes open on this node, else the reason they do not: /// `not_fetched`, `not_granted`, `evicted`, `withdrawn`, or a substrate /// fault's kind. @@ -229,6 +273,16 @@ pub struct DriveEntry { /// The row's CEG envelope (CSD-006 / CIRISServer#616): who it is about, /// who signed it, who can see it, what it is. pub envelope: serde_json::Value, + /// Where it is, compactly (`FSD/FILE_CUSTODY.md`): the person's device + /// count and how many of those devices this node holds a delivery receipt + /// from — `received_on: null` only when the receipt log could not be read + /// (unknowable, not zero; an inline file is receipted like any other since + /// edge v38.0.0 / persist v52, CIRISPersist#953). CHEAP BY CONSTRUCTION: + /// the roster is read once per page, and a row costs one receipt-list + /// query — no manifest, no custody door. + /// `null` on a withdrawn row, or when the roster could not be read. + /// `GET /v1/files/{id}/custody` is the full answer. + pub custody: Option, } #[derive(Debug, Deserialize)] @@ -313,7 +367,7 @@ fn drive_no_session() -> Response { refuse( StatusCode::FORBIDDEN, "drive.owner_session_required", - "a drive is one person's view of their own reach, and reading or writing in it is that person's own act".into(), + "Sign in as this node's owner to use your files.".into(), ) } @@ -321,7 +375,7 @@ fn notes_no_session() -> Response { refuse( StatusCode::FORBIDDEN, "notes.owner_session_required", - "notes are one person's, and writing or reading them is that person's own act".into(), + "Sign in as this node's owner to use your notes.".into(), ) } @@ -377,16 +431,14 @@ fn room_for(cohort: Cohort, room_id: Option<&str>, owner: &str) -> Result room_id.map(ScopeRoom::community).ok_or_else(|| { refuse( StatusCode::BAD_REQUEST, "drive.community_id_required", - "a community write must name `room_id` (the community's key id)".into(), + "A community file needs to name its room.".into(), ) }), } @@ -586,6 +638,68 @@ fn store(engine: &Arc) -> ciris_edge::group_content::PersistGroupContent ) } +/// What a file IS — its name and media type — as this read can say +/// (CIRISEdge#698). Since edge v33 an encrypted-tier row carries them only +/// inside `sealed_descriptor`, so `FileRow.filename`/`media_type` read `None` +/// on every sealed row and must never be used as the answer. +struct Described { + filename: Option, + media_type: Option, + /// `clear` (a plaintext-tier or pre-#698 row), `opened` (the sealed + /// descriptor opened for this viewer) or `sealed` (held, not opened here). + how: &'static str, +} + +/// The ONE door to a file's name and type (CIRISEdge#698). Every drive path +/// that shows, keeps or tests a name goes through here. +/// +/// Edge's `FileRow::describe` (v33.0.0, CIRISEdge#702): a sealed row opens ONLY +/// its descriptor, under the row's AAD — persist v51 refuses a pointer copied +/// onto another row at that door — and returns no bytes. It still reads the +/// blob to authenticate it, so a file whose bytes are not here lists `sealed`. +async fn describe(st: &DriveState, file: &files::FileRow, viewer: &str) -> Described { + use ciris_edge::files::Descriptor; + let content = store(&st.engine); + let d = file + .describe(&content, viewer) + .await + .unwrap_or(Descriptor::Sealed); + match d { + Descriptor::Clear { format, name, .. } => Described { + filename: name, + media_type: Some(format), + how: "clear", + }, + Descriptor::Opened { format, name, .. } => Described { + filename: name, + media_type: Some(format), + how: "opened", + }, + Descriptor::Sealed => Described { + filename: None, + media_type: None, + how: "sealed", + }, + } +} + +/// [`describe`] as this node's content occurrence, for a path that has not +/// resolved the viewer. No viewer key ⇒ the row's clear members only. +async fn describe_here(st: &DriveState, file: &files::FileRow) -> Described { + match viewer_key(st).await { + Ok(v) => describe(st, file, &v).await, + Err(_) => Described { + filename: file.filename.clone(), + media_type: file.media_type.clone(), + how: if file.pointer.sealed_descriptor.is_some() { + "sealed" + } else { + "clear" + }, + }, + } +} + /// **The key every file is opened AS** — the one this node's content-KEM /// occurrence was provisioned under. See /// [`crate::backend::content_occurrence_key_id`]: on an actor/node split that @@ -625,92 +739,157 @@ async fn ensure_owner_is_a_kem_target(st: &DriveState, owner_key_id: &str) { // ─── Upload bodies: JSON or multipart ────────────────────────────────────── +/// Where an upload's bytes are once its form has been read. +enum UploadBytes { + /// In hand: the JSON form, or a multipart file whose uploader declared no + /// `size` (collected up to [`WHOLE_READ_CAP`]). + Whole(Vec), + /// **Not yet read**: the multipart reader, positioned at the file part's + /// first byte, and the length its uploader declared. The seal reads the + /// rest straight off the request body. + Streamed { + body: Box, + declared: u64, + }, +} + /// One upload, whichever form carried it. -#[derive(Debug, Default)] struct Upload { cohort: Option, room_id: Option, - bytes: Vec, + bytes: UploadBytes, media_type: Option, filename: Option, } -/// `drive.too_large` — the one sentence for "bigger than this node takes whole". -fn too_large(size: usize) -> Response { +/// `drive.too_large` — the one sentence for "bigger than this node takes +/// whole" (the JSON form, or a multipart file without `size`). +fn too_large(size: u64) -> Response { refuse( StatusCode::PAYLOAD_TOO_LARGE, "drive.too_large", format!( - "{size} bytes exceeds this node's {WHOLE_READ_CAP}-byte file cap — the chunk-DAG \ - whole-read cap, above which a file could be stored but never read back whole" + "{size} bytes exceeds this node's {WHOLE_READ_CAP}-byte cap for an upload it must \ + hold whole — send it as multipart/form-data with a `size` field before the `file` \ + part, which streams up to {STREAMED_FILE_CEILING} bytes" ), ) } -/// Parse an upload body: `application/json` ([`FileWriteRequest`]) or -/// `multipart/form-data` (fields `cohort`, `room_id`, `media_type`, -/// `filename`, and the bytes in a part named `file` — whose own filename and -/// `Content-Type` are used when the fields are absent). +/// `drive.too_large` for the streamed form — above edge's single-file ceiling. +fn too_large_streamed(size: u64) -> Response { + refuse( + StatusCode::PAYLOAD_TOO_LARGE, + "drive.too_large", + format!( + "{size} bytes exceeds this node's {STREAMED_FILE_CEILING}-byte single-file ceiling — \ + edge's, set by persist's inline manifest cap until persist v52" + ), + ) +} + +/// A multipart reader's typed failure → its refusal. Each has its own +/// remedy: send less, reorder the form, or resend. +fn multipart_refusal(f: &multipart::Failure) -> Response { + use multipart::Failure as F; + match f { + F::TooLarge { .. } => refuse( + StatusCode::PAYLOAD_TOO_LARGE, + "drive.too_large", + format!( + "{f} — {STREAMED_FILE_CEILING} bytes of file plus 1 MiB of form is the most this \ + node reads" + ), + ), + F::FieldAfterFile => refuse( + StatusCode::BAD_REQUEST, + "drive.field_after_file", + format!( + "{f}. Put `cohort`, `room_id`, `media_type`, `filename` and `size` BEFORE the \ + `file` part (with FormData: append the file last). Nothing was written." + ), + ), + F::Truncated | F::Malformed(_) | F::Transport(_) => bad_body(f.to_string()), + } +} + +/// Read a JSON body whole, refusing past `limit` by name. #[allow(clippy::result_large_err)] // the Err IS an axum Response -fn parse_upload( +async fn collect_bounded( headers: &HeaderMap, - body: Result, -) -> Result { - let body = body.map_err(|rej| { - if rej.status() == StatusCode::PAYLOAD_TOO_LARGE { - too_large(UPLOAD_BODY_LIMIT) - } else { - bad_body(format!("read the request body: {rej}")) + body: Body, + limit: usize, +) -> Result, Response> { + use futures_util::StreamExt as _; + if let Some(n) = content_length(headers).filter(|n| *n > limit as u64) { + return Err(too_large(n)); + } + let mut out = Vec::new(); + let mut frames = body.into_data_stream(); + while let Some(frame) = frames.next().await { + let frame = frame.map_err(|e| bad_body(format!("read the request body: {e}")))?; + if out.len() + frame.len() > limit { + return Err(too_large((out.len() + frame.len()) as u64)); } - })?; + out.extend_from_slice(&frame); + } + Ok(out) +} + +fn content_length(headers: &HeaderMap) -> Option { + headers + .get(header::CONTENT_LENGTH) + .and_then(|v| v.to_str().ok()) + .and_then(|v| v.trim().parse().ok()) +} + +/// Read an upload's FORM: `application/json` ([`FileWriteRequest`], whole) or +/// `multipart/form-data` (streamed — the file part is left unread for the +/// seal). +/// +/// # The multipart form, 0.5.218 +/// +/// Fields `cohort`, `room_id`, `media_type`, `filename` and **`size`**, then +/// the bytes in a part named `file` (whose own filename and `Content-Type` +/// are used when the fields are absent). **Every field precedes the file**: +/// the file's bytes go to the seal as they arrive, so anything after them +/// would arrive after the decisions it names — refused by name, +/// `drive.field_after_file`, with nothing written. +/// +/// **`size` is the declared length** — the file part's exact byte count, as +/// a decimal form field before the file. Edge's `files::publish_stream` +/// chooses the seal's shape (inline or chunk DAG) from the declared length +/// before it reads a byte, and refuses a body of any other length by name +/// (`FileError::DeclaredLengthMismatch` → `drive.declared_length_mismatch`, +/// nothing sealed, no row). A form FIELD, not the part's own +/// `Content-Length` header, because the field is what a client can set: +/// browsers' `FormData` writes no per-part `Content-Length` and gives +/// script no way to add one, while `form.append("size", file.size)` before +/// `form.append("file", file)` is one line on every platform. With `size`, +/// the file streams up to [`STREAMED_FILE_CEILING`]; without it, the node +/// collects the part up to [`WHOLE_READ_CAP`] (the pre-0.5.218 behaviour, +/// so a client that has never heard of `size` keeps working for the files it +/// could always send) and refuses above that naming `size` as the remedy. +#[allow(clippy::result_large_err)] // the Err IS an axum Response +async fn parse_upload(headers: &HeaderMap, body: Body) -> Result { let ct = headers .get(header::CONTENT_TYPE) .and_then(|v| v.to_str().ok()) .unwrap_or("application/json"); - let upload = if ct + if !ct .trim_start() .to_ascii_lowercase() .starts_with("multipart/form-data") { - let boundary = multipart::boundary(ct).ok_or_else(|| { - bad_body("multipart/form-data without a `boundary` parameter".to_owned()) - })?; - let parts = multipart::parse(&body, &boundary).map_err(bad_body)?; - let mut up = Upload::default(); - let mut saw_file = false; - for p in parts { - let text = || String::from_utf8(p.data.clone()).ok(); - match p.name.as_str() { - "file" | "bytes" => { - saw_file = true; - if up.filename.is_none() { - up.filename = p.filename.clone(); - } - if up.media_type.is_none() { - up.media_type = p.content_type.clone(); - } - up.bytes = p.data; - } - "cohort" => up.cohort = text(), - "room_id" => up.room_id = text(), - // Explicit fields win over the part's own headers. - "media_type" => up.media_type = text(), - "filename" => up.filename = text(), - _ => {} - } - } - if !saw_file { - return Err(bad_body( - "multipart/form-data upload has no part named `file`".to_owned(), - )); - } - up - } else { + let body = collect_bounded(headers, body, UPLOAD_BODY_LIMIT).await?; let req: FileWriteRequest = serde_json::from_slice(&body) .map_err(|e| bad_body(format!("not a file-write JSON body: {e}")))?; let bytes = base64_decode(&req.bytes_base64) .map_err(|e| refuse(StatusCode::BAD_REQUEST, "drive.bad_base64", e))?; - Upload { + if bytes.len() > WHOLE_READ_CAP { + return Err(too_large(bytes.len() as u64)); + } + return Ok(Upload { cohort: req.cohort.map(|c| { match c { Cohort::SelfCollective => "self", @@ -720,109 +899,113 @@ fn parse_upload( .to_owned() }), room_id: req.room_id, - bytes, + bytes: UploadBytes::Whole(bytes), media_type: req.media_type, filename: req.filename, - } - }; - if upload.bytes.len() > WHOLE_READ_CAP { - return Err(too_large(upload.bytes.len())); + }); } - Ok(upload) -} - -fn bad_body(detail: String) -> Response { - refuse(StatusCode::BAD_REQUEST, "drive.bad_body", detail) -} -/// A minimal `multipart/form-data` reader (RFC 7578) — enough for one upload -/// form, with no new dependency. Bounded by the route's body limit, so the -/// whole body is already in memory; this only slices it. -mod multipart { - pub struct Part { - pub name: String, - pub filename: Option, - pub content_type: Option, - pub data: Vec, - } - - /// The `boundary` parameter of a `multipart/form-data` content type. - pub fn boundary(content_type: &str) -> Option { - content_type.split(';').skip(1).find_map(|param| { - let (k, v) = param.split_once('=')?; - if k.trim().eq_ignore_ascii_case("boundary") { - let v = v.trim().trim_matches('"'); - (!v.is_empty()).then(|| v.to_owned()) - } else { - None - } - }) + if let Some(n) = content_length(headers).filter(|n| *n > STREAMED_UPLOAD_BODY_LIMIT) { + return Err(too_large_streamed(n)); } - - fn find(hay: &[u8], needle: &[u8], from: usize) -> Option { - if needle.is_empty() || from > hay.len() { - return None; + let boundary = multipart::boundary(ct) + .ok_or_else(|| bad_body("multipart/form-data without a `boundary` parameter".to_owned()))?; + let frames: multipart::ByteStream = { + use futures_util::StreamExt as _; + Box::pin( + body.into_data_stream() + .map(|r| r.map_err(|e| e.to_string())), + ) + }; + let mut mp = Box::new(multipart::Multipart::new( + frames, + &boundary, + STREAMED_UPLOAD_BODY_LIMIT, + )); + let (mut cohort, mut room_id, mut media_type, mut filename, mut size) = + (None, None, None, None, None); + let file_head = loop { + let head = match mp.next_part().await { + Ok(Some(h)) => h, + Ok(None) => { + return Err(bad_body( + "multipart/form-data upload has no part named `file`".to_owned(), + )) + } + Err(f) => return Err(multipart_refusal(&f)), + }; + if matches!(head.name.as_str(), "file" | "bytes") { + break head; } - hay[from..] - .windows(needle.len()) - .position(|w| w == needle) - .map(|p| p + from) - } - - /// A `Content-Disposition` parameter, quotes stripped. - fn disposition_param(value: &str, key: &str) -> Option { - value.split(';').skip(1).find_map(|param| { - let (k, v) = param.split_once('=')?; - k.trim() - .eq_ignore_ascii_case(key) - .then(|| v.trim().trim_matches('"').to_owned()) - }) - } - - pub fn parse(body: &[u8], boundary: &str) -> Result, String> { - let delim = format!("--{boundary}").into_bytes(); - let next_delim = format!("\r\n--{boundary}").into_bytes(); - let mut at = find(body, &delim, 0) - .ok_or_else(|| "multipart body does not contain its boundary".to_owned())? - + delim.len(); - let mut parts = Vec::new(); - loop { - // After a delimiter: `--` closes the body, CRLF opens a part. - if body[at..].starts_with(b"--") { - return Ok(parts); + let value = match mp.read_part(multipart::FIELD_CAP).await { + Ok(v) => v, + Err(multipart::ReadPartError::AboveCap) => { + return Err(bad_body(format!( + "form field `{}` is longer than {} bytes — only the `file` part carries bytes", + head.name, + multipart::FIELD_CAP + ))) } - if !body[at..].starts_with(b"\r\n") { - return Err("malformed multipart delimiter line".to_owned()); + Err(multipart::ReadPartError::Failed(f)) => return Err(multipart_refusal(&f)), + }; + let text = String::from_utf8(value).ok(); + match head.name.as_str() { + "cohort" => cohort = text, + "room_id" => room_id = text, + // Explicit fields win over the part's own headers. + "media_type" => media_type = text, + "filename" => filename = text, + "size" => size = text, + _ => {} + } + }; + if filename.is_none() { + filename = file_head.filename.clone(); + } + if media_type.is_none() { + media_type = file_head.content_type.clone(); + } + let bytes = match size { + Some(raw) => { + let declared: u64 = raw.trim().parse().map_err(|_| { + bad_body(format!( + "`size` must be the file part's byte count as a decimal integer, not {raw:?}" + )) + })?; + if declared > STREAMED_FILE_CEILING { + return Err(too_large_streamed(declared)); } - at += 2; - let head_end = find(body, b"\r\n\r\n", at) - .ok_or_else(|| "multipart part has no header terminator".to_owned())?; - let head = std::str::from_utf8(&body[at..head_end]) - .map_err(|_| "multipart part headers are not UTF-8".to_owned())?; - let (mut name, mut filename, mut content_type) = (None, None, None); - for line in head.split("\r\n") { - let Some((k, v)) = line.split_once(':') else { - continue; - }; - if k.trim().eq_ignore_ascii_case("content-disposition") { - name = disposition_param(v, "name"); - filename = disposition_param(v, "filename"); - } else if k.trim().eq_ignore_ascii_case("content-type") { - content_type = Some(v.trim().to_owned()); + UploadBytes::Streamed { body: mp, declared } + } + None => { + let bytes = match mp.read_part(WHOLE_READ_CAP).await { + Ok(b) => b, + Err(multipart::ReadPartError::AboveCap) => { + return Err(too_large(WHOLE_READ_CAP as u64 + 1)) } + Err(multipart::ReadPartError::Failed(f)) => return Err(multipart_refusal(&f)), + }; + // The same rule as the streamed form, so a client learns ONE + // shape: the file is the last part. + match mp.next_part().await { + Ok(None) => {} + Ok(Some(_)) => return Err(multipart_refusal(&multipart::Failure::FieldAfterFile)), + Err(f) => return Err(multipart_refusal(&f)), } - let data_start = head_end + 4; - let data_end = find(body, &next_delim, data_start) - .ok_or_else(|| "multipart part is not closed by its boundary".to_owned())?; - parts.push(Part { - name: name.ok_or_else(|| "multipart part has no `name`".to_owned())?, - filename, - content_type, - data: body[data_start..data_end].to_vec(), - }); - at = data_end + next_delim.len(); + UploadBytes::Whole(bytes) } - } + }; + Ok(Upload { + cohort, + room_id, + bytes, + media_type, + filename, + }) +} + +fn bad_body(detail: String) -> Response { + refuse(StatusCode::BAD_REQUEST, "drive.bad_body", detail) } // ─── Publishing, finding, and withdrawing rows ───────────────────────────── @@ -865,27 +1048,26 @@ async fn author_capsule( }) } -/// A new file row in `room`: seal, author, cross — edge's one door -/// (`files::publish`), with this server's reporting around it. +/// What the write gate lets through: the normalised essence and the cleaned +/// display name. +struct Gated { + media_type: String, + filename: Option, +} + +/// THE WRITE GATE (CIRISServer#642, CC 3.3.13 / CC 5.3.2.6): the node is the +/// first consumer of these bytes, and every peer inherits what this row says +/// they are. The declared type must be an RFC 6838 essence the leading bytes +/// agree with, and the name is display-only (RFC 6266 §4.3) — no path, no +/// control or bidi characters. Every write door comes through here. +/// +/// `head` is the file's LEADING bytes — the whole file, or (on a streamed +/// upload or a streamed `move`) the first [`crate::media_gate::FORMAT_HEAD_BYTES`] +/// peeked without consuming them. The gate never reads past that window, so +/// the verdict is the same either way; the streamed path is not a weaker gate. #[allow(clippy::result_large_err)] // the Err IS an axum Response -#[allow(clippy::too_many_arguments)] -async fn publish_into( - st: &DriveState, - headers: &HeaderMap, - owner_key_id: &str, - cohort: Cohort, - room: &ScopeRoom, - bytes: &[u8], - media_type: &str, - filename: Option<&str>, - plane: Plane, -) -> Result<(files::PublishedFile, bool), Response> { - // THE WRITE GATE (CIRISServer#642, CC 3.3.13 / CC 5.3.2.6): the node is the - // first consumer of these bytes, and every peer inherits what this row - // says they are. The declared type must be an RFC 6838 essence the leading - // bytes agree with, and the name is display-only (RFC 6266 §4.3) — no path, - // no control or bidi characters. Every write door comes through here. - let essence = match crate::media_gate::check_format(media_type, bytes) { +fn write_gate(media_type: &str, head: &[u8], filename: Option<&str>) -> Result { + let essence = match crate::media_gate::check_format(media_type, head) { Ok(e) => e, Err(crate::media_gate::TypeRefusal::BadEssence(d)) => { return Err(refuse( @@ -921,27 +1103,93 @@ async fn publish_into( } }, }; - let media_type = essence.as_str(); - let filename = clean_name.as_deref(); + Ok(Gated { + media_type: essence, + filename: clean_name, + }) +} + +/// A new file row in `room` from bytes in hand: [`write_gate`], then +/// [`publish_gated`] over the slice. +#[allow(clippy::result_large_err)] // the Err IS an axum Response +#[allow(clippy::too_many_arguments)] +async fn publish_into( + st: &DriveState, + headers: &HeaderMap, + owner_key_id: &str, + cohort: Cohort, + room: &ScopeRoom, + bytes: &[u8], + media_type: &str, + filename: Option<&str>, + plane: Plane, +) -> Result<(files::PublishedFile, bool), Response> { + let gated = write_gate(media_type, bytes, filename)?; + publish_gated( + st, + headers, + owner_key_id, + cohort, + room, + &gated, + bytes.len() as u64, + bytes, + plane, + ) + .await +} + +/// A new file row in `room`: seal, author, cross — edge's one door +/// (`files::publish_stream`; `files::publish` is edge's wrapper over it for a +/// slice, CIRISEdge#744), with this server's reporting around it. The bytes +/// come from `reader`, exactly `declared_len` of them; the gate has already +/// run on their head. +/// +/// A reader that yields any other count is edge's +/// `FileError::DeclaredLengthMismatch` (`drive.declared_length_mismatch`) and +/// a reader that fails is `FileError::Read`; neither leaves a manifest or a +/// row, and edge evicts the chunks it had sealed (§6.7.4). A caller whose +/// reader keeps a typed cause (the multipart body, a chunk walk) answers +/// THAT cause when this fails — edge carries a reader's error only as prose. +#[allow(clippy::result_large_err)] // the Err IS an axum Response +#[allow(clippy::too_many_arguments)] +async fn publish_gated( + st: &DriveState, + headers: &HeaderMap, + owner_key_id: &str, + cohort: Cohort, + room: &ScopeRoom, + gated: &Gated, + declared_len: u64, + reader: R, + plane: Plane, +) -> Result<(files::PublishedFile, bool), Response> +where + R: tokio::io::AsyncRead + Unpin + Send, +{ ensure_owner_is_a_kem_target(st, owner_key_id).await; let addressed = addressed_or_warn(st, room, cohort); let capsule = author_capsule(st, headers, owner_key_id, plane).await?; let dir = st.engine.federation_directory(); let content = store(&st.engine); - let published = files::publish( + let published = files::publish_stream( &*dir, &content, ciris_edge::replication::attestation_bind::Signers { node: &st.node_signer, actor: Some(capsule.edge_signer()), }, - &FileWrite { + &FileStreamWrite { room, - bytes, - media_type, - filename, + declared_len, + media_type: &gated.media_type, + // The drive names a codec, if at all, in the media type's own + // parameters; edge's separate slot stays empty. + codec: None, + filename: gated.filename.as_deref(), asserted_at: chrono::Utc::now(), }, + reader, ) .await .map_err(|e| file_error(&e, room))?; @@ -968,6 +1216,65 @@ async fn publish_into( Ok((published, addressed)) } +/// An upload's bytes → a new row: whole bytes through [`publish_into`], a +/// streamed file part through the gate on its PEEKED head and then +/// [`publish_gated`] straight off the request body. +#[allow(clippy::result_large_err)] // the Err IS an axum Response +#[allow(clippy::too_many_arguments)] +async fn publish_upload( + st: &DriveState, + headers: &HeaderMap, + owner_key_id: &str, + cohort: Cohort, + room: &ScopeRoom, + bytes: UploadBytes, + media_type: &str, + filename: Option<&str>, +) -> Result<(files::PublishedFile, bool), Response> { + match bytes { + UploadBytes::Whole(b) => { + publish_into( + st, + headers, + owner_key_id, + cohort, + room, + &b, + media_type, + filename, + Plane::Drive, + ) + .await + } + UploadBytes::Streamed { mut body, declared } => { + // PEEK, THEN CHAIN: the gate reads the first 64 KiB where they + // sit in the multipart reader's own buffer; they are not consumed, + // so the seal below reads the file from its first byte. + let gated = match body.peek(crate::media_gate::FORMAT_HEAD_BYTES).await { + Ok(head) => write_gate(media_type, head, filename)?, + Err(f) => return Err(multipart_refusal(&f)), + }; + let out = publish_gated( + st, + headers, + owner_key_id, + cohort, + room, + &gated, + declared, + &mut *body, + Plane::Drive, + ) + .await; + match (out, body.failure()) { + // The body's own cause beats edge's `Read` prose. + (Err(_), Some(f)) => Err(multipart_refusal(f)), + (out, _) => out, + } + } + } +} + fn write_response( published: &files::PublishedFile, addressed: bool, @@ -1064,11 +1371,29 @@ async fn find_file( // would not. let dir = st.engine.federation_directory(); if let Ok(Some(row)) = dir.get_attestation(attestation_id).await { - if let Some(file) = stopgap_belongs_to(room, &row) { - if let Some(w) = withdrawn_by(st, attestation_id) + if let Some(file) = files::belongs_to(room, &row) { + // Edge's lifecycle decides WHICH retirement (CIRISEdge#693): + // a `withdraws` is still re-derived here; a `supersedes` + // (a rename, edge v33.1 `files::rename`) or a `recants` + // from the row's own attester retires it as well. + let retired = match withdrawn_by(st, attestation_id) .await .map_err(listing_failed)? { + Some(w) => Some(w), + None => lifecycle_in_room(st, owner, room, attestation_id) + .await + .map_err(listing_failed)? + .filter(|l| { + matches!( + l, + files::FileLifecycle::Superseded + | files::FileLifecycle::Recanted + ) + }) + .and(retired_by(st, &row).await), + }; + if let Some(w) = retired { return Ok(Found { file, row, @@ -1094,84 +1419,103 @@ async fn find_file( } } -// ─── STOPGAP for CIRISEdge#693 — delete when an edge cut lets `files::in_room` -// take a `LifecycleView` and makes `belongs_to` public. ─────────────────── -// -// Edge v32 / persist v49 list LIVE rows only, and edge's post-gate room check -// is private, so a withdrawn file vanished from the drive: a read of its id -// answered `404 drive.not_in_room` instead of `410 drive.withdrawn`, and -// `?include_withdrawn=true` lost it. These two functions are the labelled -// copy the ownership rule allows until upstream lands. - -/// COPY of edge v32.1.0's private `files::belongs_to` (src/files.rs): is `row` -/// one of `room`'s files? The self arm checks the pointer's group slot; a -/// targeted room checks its cohort-target envelope member. -fn stopgap_belongs_to( +/// Edge's lifecycle for one row of `room`, as its `All` listing names it +/// (`IncludeWithdrawn` would leave out superseded and recanted rows). `None` when the caller's listing does not carry the row at all. +async fn lifecycle_in_room( + st: &DriveState, + caller: &str, room: &ScopeRoom, - row: &ciris_persist::federation::Attestation, -) -> Option { - if row.cohort_scope != room.row_scope_token() { - return None; - } - let file = files::FileRow::from_row(row)?; - let names_this_room = match room.cohort_target_field() { - Some(field) => { - row.attestation_envelope - .get(field) - .and_then(serde_json::Value::as_str) - == Some(room.content_group_id()) + attestation_id: &str, +) -> Result, String> { + let mut after = None; + loop { + let page = files::in_room_with( + &st.engine, + room, + caller, + usize::MAX, + after, + ciris_persist::ceg::LifecycleView::All, + ) + .await + .map_err(|e| format!("{e:#}"))?; + if let Some(f) = page + .files + .into_iter() + .find(|f| f.attestation_id == attestation_id) + { + return Ok(Some(f.lifecycle)); } - None => file.pointer.community_key_id == room.content_group_id(), - }; - names_this_room.then_some(file) + match page.resume { + Some(c) => after = Some(c), + None => return Ok(None), + } + } } -/// The WITHDRAWN file rows of `room` this caller may see: persist's listing at -/// `LifecycleView::IncludeWithdrawn` under the same caller gate `in_room` -/// builds, kept only where a `withdraws` really retired the row. -async fn stopgap_withdrawn_in_room( +/// The RETIRED file rows of `room` this caller may see (CIRISEdge#693, edge +/// v33): edge's own listing at `IncludeWithdrawn`, which names each row's +/// lifecycle by persist's hide rule. A `Withdrawn` row is kept only when a +/// `withdraws` really retired it (persist re-derives its authority). +async fn withdrawn_in_room( st: &DriveState, room: &ScopeRoom, caller: &str, ) -> Result, String> { - use ciris_persist::ceg::{AttestationFilter, LifecycleView}; - use ciris_persist::scope::CallerScope; - let admission = - ciris_persist::scope::admission::build_caller_admission(&st.engine, &caller.to_owned()) - .await - .map_err(|e| format!("caller admission: {e}"))?; - let scope = CallerScope::Authenticated { admission }; let mut out = Vec::new(); - let mut cursor = None; - for _ in 0..64 { - let mut f = AttestationFilter::default(); - f.cohort_scope = Some(room.row_scope_token().to_owned()); - f.dimension_exact = Some(files::FILE_DIMENSION.to_owned()); - f.lifecycle = LifecycleView::IncludeWithdrawn; - let page = st - .engine - .list_attestations(f, cursor, 256, scope.clone()) - .await - .map_err(|e| format!("list withdrawn in {room}: {e}"))?; - for row in &page.items { - if let Some(file) = stopgap_belongs_to(room, row) { - if withdrawn_by(st, &row.attestation_id).await?.is_some() { - out.push(file); - } + let mut after = None; + loop { + let page = files::in_room_with( + &st.engine, + room, + caller, + usize::MAX, + after, + ciris_persist::ceg::LifecycleView::IncludeWithdrawn, + ) + .await + .map_err(|e| format!("list withdrawn in {room}: {e:#}"))?; + for f in page.files { + if f.lifecycle == files::FileLifecycle::Withdrawn + && withdrawn_by(st, &f.attestation_id).await?.is_some() + { + out.push(f); } } - cursor = page.next_cursor; - if cursor.is_none() { - break; + match page.resume { + Some(c) => after = Some(c), + None => break, } } Ok(out) } +/// The composer that retired `row` without a `withdraws` — the `supersedes` +/// or `recants` from its own attester that edge's lifecycle named +/// (`FileLifecycle::Superseded` / `Recanted`). Its id, for the 410's detail. +async fn retired_by(st: &DriveState, row: &Attestation) -> Option { + use ciris_persist::federation::types::attestation_type::{RECANTS, SUPERSEDES}; + st.engine + .federation_directory() + .list_attestations_referencing(&row.attestation_id) + .await + .ok()? + .into_iter() + .find(|c| { + (c.attestation_type == SUPERSEDES || c.attestation_type == RECANTS) + && c.attesting_key_id == row.attesting_key_id + }) + .map(|c| c.attestation_id) +} + /// **Has a `withdraws` retired this row?** The id of the one that did. /// /// Re-derives each retraction's authority NOW with persist's own -/// `check_withdraws_admission`, never trusting the stored rule — the same +/// `check_withdraws_admission_as_admitted`, never trusting the stored rule — +/// at the delegation depth the row was ADMITTED under (persist v50, #690): +/// the write-form check now uses CC 4.1.1's 5-hop default, and re-deriving a +/// row admitted through a longer chain with it would bring a withdrawn file +/// back to life, the retroactive change persist ruled out. The same /// discipline `blob_tombstone::binding_state` applies at the bytes plane, one /// row instead of every row binding a sha. (Persist's per-row fold, /// `retiring_composer`, is private; this is its withdraws arm, and the only @@ -1188,7 +1532,9 @@ async fn withdrawn_by(st: &DriveState, attestation_id: &str) -> Result return Ok(Some(g.attestation_id)), Ok(None) | Err(ciris_persist::federation::Error::WithdrawsNotAdmitted { .. }) => {} Err(e) => return Err(format!("re-derive {}: {e:#}", g.attestation_id)), @@ -1201,8 +1547,11 @@ async fn withdrawn_by(st: &DriveState, attestation_id: &str) -> Result Result<(), Response> { - if row.attesting_key_id == st.node_signer.key_id { +fn require_author(st: &DriveState, owner_key_id: &str, row: &Attestation) -> Result<(), Response> { + // Two authors a file can have (CIRISEdge#675, edge v33 `files::file_author`): + // the PERSON — any of their devices holds the pen — or, for a row written + // before a pen was in hand, this node. + if row.attesting_key_id == owner_key_id || row.attesting_key_id == st.node_signer.key_id { return Ok(()); } Err(refuse( @@ -1245,6 +1594,9 @@ struct Withdrawal { /// claims). A rename's bytes stay bound by the new row, so they are kept. async fn withdraw_rows( st: &DriveState, + headers: &HeaderMap, + owner_key_id: &str, + plane: Plane, listed: &Attestation, file: &files::FileRow, reason: &str, @@ -1271,23 +1623,41 @@ async fn withdraw_rows( else { break; }; - if prior.attesting_key_id != st.node_signer.key_id { + if prior.attesting_key_id != st.node_signer.key_id && prior.attesting_key_id != owner_key_id + { break; } chain.push(prior.clone()); cur = prior; } let now = chrono::Utc::now(); + // The PERSON's pen, when a row in the chain is theirs (CIRISEdge#675): a + // `withdraws` is signed by the row's own attester (persist rule 1). + let pen = if chain.iter().any(|r| r.attesting_key_id == owner_key_id) { + Some( + author_capsule(st, headers, owner_key_id, plane) + .await + .map_err(|_| { + format!( + "the person's signer is unavailable to withdraw {}", + listed.attestation_id + ) + })?, + ) + } else { + None + }; let mut withdrawn = Vec::new(); for row in &chain { if withdrawn_by(st, &row.attestation_id).await?.is_some() { continue; } + let signer: &ciris_edge::identity::LocalSigner = match &pen { + Some(p) if row.attesting_key_id == owner_key_id => p.edge_signer(), + _ => &st.node_signer, + }; let w = ciris_edge::replication::attestation_bind::withdraws_attestation( - row, - reason, - now, - &st.node_signer, + row, reason, now, signer, ) .await .map_err(|e| format!("build withdraws for {}: {e}", row.attestation_id))?; @@ -1333,104 +1703,6 @@ fn withdraw_failed(detail: String) -> Response { ) } -/// **A rename: a new row over the SAME bytes.** -/// -/// No re-seal and no re-upload: the new row cites the old pointer. That is only -/// possible because the seal's associated data is `(author, asserted_at, -/// field)` read off the ROW (`group_content::aad_for_open`), so the new row -/// carries the old row's author and instant verbatim — the claim's instant, -/// exactly as a widening carries it (persist v40.0.0) — and the bytes open -/// under it. Everything else is `files::publish`'s row shape: the file -/// dimension, the pointer under `content`, the room's cohort target, the sha -/// cited in `evidence_refs`, authored at `self` / local tier for the crossing -/// to place. -/// -/// Built here because edge's file door takes BYTES, not a pointer (upstream -/// ask: a `files::republish(pointer, ..)`); the row shape above is edge's, and -/// `rename_keeps_the_blob_and_the_bytes_open` in `tests/drive_crud.rs` is what -/// fails if the two drift. -async fn rename_row( - author: &ciris_edge::identity::LocalSigner, - room: &ScopeRoom, - old: &files::FileRow, - filename: &str, - replaces: &str, -) -> Result { - use ciris_edge::replication::attestation_bind::{ - bind_attestation_envelope, render_signed_instant, AttestationColumns, - }; - use ciris_persist::federation::types::{attestation_tier, cohort_scope}; - use sha2::{Digest as _, Sha256}; - - let author_key_id = author.key_id.as_str(); - let asserted_at = old.asserted_at; - let mut envelope = serde_json::json!({ - (ciris_persist::federation::envelope::paths::DIMENSION): files::FILE_DIMENSION, - (ciris_edge::chat::FIELD_CONTENT): old.pointer, - (files::FIELD_FILENAME): filename, - (FIELD_REPLACES): replaces, - "evidence_refs": [old.pointer.content_sha256], - }); - if let Some(field) = room.cohort_target_field() { - envelope[field] = serde_json::json!(room.content_group_id()); - } - let attestation_id = { - let mut h = Sha256::new(); - h.update(files::FILE_DIMENSION.as_bytes()); - h.update(b"\0rename\0"); - h.update(room.table_group_id().as_bytes()); - h.update(author_key_id.as_bytes()); - h.update(render_signed_instant(asserted_at).as_bytes()); - h.update( - ciris_persist::prelude::ceg_produce_canonicalize(&envelope) - .map_err(|e| format!("canonicalize: {e}"))?, - ); - format!("file-{}", &hex::encode(h.finalize())[..32]) - }; - let subjects = vec![author_key_id.to_owned()]; - bind_attestation_envelope( - &mut envelope, - asserted_at, - &AttestationColumns { - attestation_id: &attestation_id, - attesting_key_id: author_key_id, - attestation_type: "scores", - attested_key_id: author_key_id, - subject_key_ids: &subjects, - cohort_scope: cohort_scope::SELF, - weight: None, - }, - ); - let canonical = ciris_persist::prelude::ceg_produce_canonicalize(&envelope) - .map_err(|e| format!("canonicalize: {e}"))?; - let digest = Sha256::digest(&canonical); - let (sig_classical, sig_pqc) = - ciris_edge::identity::sign_bound_hybrid(author, &canonical, files::FILE_DIMENSION).await?; - Ok(Attestation { - attestation_id, - attesting_key_id: author_key_id.to_owned(), - attested_key_id: author_key_id.to_owned(), - attestation_type: "scores".to_owned(), - weight: None, - asserted_at, - expires_at: None, - attestation_envelope: envelope, - original_content_hash: hex::encode(digest), - scrub_signature_classical: sig_classical, - scrub_signature_pqc: sig_pqc, - scrub_key_id: author_key_id.to_owned(), - scrub_timestamp: asserted_at, - pqc_completed_at: None, - persist_row_hash: String::new(), - subject_key_ids: subjects, - withdraws_admission_rule: None, - cohort_scope: cohort_scope::SELF.to_owned(), - tier: attestation_tier::LOCAL.to_owned(), - promoted_at: None, - additional_scrubs: Vec::new(), - }) -} - // ─── Byte state: typed, never parsed out of a Debug string ───────────────── /// Where a row's bytes stand on this node. @@ -1539,6 +1811,28 @@ async fn probe(st: &DriveState, file: &files::FileRow, viewer: &str) -> ByteStat .read_blob_range_as(&sha, viewer, past_the_end, past_the_end, aad.as_deref()) .await { + // A chunk-DAG pull adopts the MANIFEST first, as an inline envelope at + // the file's address, and flips it to `chunk_dag` only when every chunk + // is held (persist `promote_adopted_manifest_to_dag`, CIRISPersist#947). + // Mid-pull the range probe therefore answers with the MANIFEST's length + // — a few hundred bytes for a 256 MiB file — and reporting that as + // `here` sent a reader to a raw read that 416'd (found by the native + // `bigfile-quick` run on edge v38, 2026-10-01). The pointer declares the + // plaintext size (CIRISEdge#638): bytes of any other length are not this + // file yet, so they read `not_fetched` — "still arriving" — never `here`. + Err(BlobError::RangeNotSatisfiable { size, .. }) + if file.pointer.size.is_some_and(|declared| declared != size) => + { + ByteState::Absent { + state: "not_fetched", + detail: format!( + "{} (the pull is in progress: {size} of {} bytes' worth is held — the \ + chunk DAG's manifest arrives before its chunks)", + state_detail("not_fetched"), + file.pointer.size.unwrap_or(0) + ), + } + } Err(BlobError::RangeNotSatisfiable { size, .. }) => ByteState::Here { size: Some(size) }, Ok(_) => ByteState::Here { size: None }, Err(e) => blob_state(&e), @@ -1577,19 +1871,42 @@ fn too_large_for_whole_read(size: u64) -> Response { StatusCode::PAYLOAD_TOO_LARGE, "drive.too_large_for_whole_read", format!( - "this file is {size} bytes, above the {WHOLE_READ_CAP}-byte whole-read cap — read \ - it with `?raw=1` and an HTTP `Range` header" + "this file is {size} bytes, above the {WHOLE_READ_CAP}-byte whole-read cap, and \ + this read answers it as one JSON value — read it with `?raw=1`, which streams it \ + (and serves any HTTP `Range`)" ), ) } +/// What a whole read found: the bytes, or — past every state check — a file +/// too big to hold, which a STREAMING caller serves instead of refusing. +enum WholeRead { + Bytes(Vec, Option), + /// Above [`WHOLE_READ_CAP`]: the file's size. The row is live and the + /// bytes are here and open for this viewer (the probe said so); only the + /// holding is refused. + AboveCap(u64), +} + /// The whole plaintext, after the state and size checks a whole read owes. +/// A file above the cap is `drive.too_large_for_whole_read` — the caller +/// that can stream asks [`read_whole`] instead. #[allow(clippy::result_large_err)] // the Err IS an axum Response async fn open_whole( st: &DriveState, found: &Found, viewer: &str, ) -> Result<(Vec, Option), Response> { + match read_whole(st, found, viewer).await? { + WholeRead::Bytes(b, size) => Ok((b, size)), + WholeRead::AboveCap(n) => Err(too_large_for_whole_read(n)), + } +} + +/// [`open_whole`] without the last refusal: above the cap it answers +/// [`WholeRead::AboveCap`] so `?raw=1` and `move` can stream. +#[allow(clippy::result_large_err)] // the Err IS an axum Response +async fn read_whole(st: &DriveState, found: &Found, viewer: &str) -> Result { if let Some(w) = &found.withdrawn_by { return Err(row_withdrawn_refusal(w)); } @@ -1598,15 +1915,42 @@ async fn open_whole( ByteState::Absent { state, detail } => return Err(refuse_state(state, detail)), }; if let Some(n) = size.filter(|n| *n > WHOLE_READ_CAP as u64) { - return Err(too_large_for_whole_read(n)); + return Ok(WholeRead::AboveCap(n)); } let content = store(&st.engine); match found.file.open(&content, viewer).await { - Ok(b) => Ok((b, size)), - Err(reason) => match unopened(&reason) { + // THE BYTES MUST BE THE FILE (CIRISServer#697 selffiles byte probe, + // 2026-09-29): a second device that pulled a chunk-DAG file held its + // MANIFEST under the pointer's sha and served it as the file — 518 + // bytes of `{"chunk_tier":…` for a 1 MiB video, status 200. The + // pointer declares the plaintext size (CIRISEdge#638), so a body of + // another length is row and bytes disagreeing, refused by name, never + // handed over as content. + Ok(b) if found.file.pointer.size.is_some_and(|n| n != b.len() as u64) => Err(refuse_state( + "seal_mismatch", + format!( + "{}: the row declares {} bytes and the bytes here are {} — a pulled chunk-DAG \ + stored as its manifest reads exactly like this (CIRISEdge#717)", + state_detail("seal_mismatch"), + found.file.pointer.size.unwrap_or(0), + b.len() + ), + )), + Ok(b) => Ok(WholeRead::Bytes(b, size)), + // edge v36 (CIRISEdge#737): `FileRow::open` answers `FileError`. The + // unopened reasons keep their byte-state words; the whole-read cap is + // edge's own check, reached here only when the pointer declared no size + // for the gate above to read. + Err(files::FileError::Unopened(reason)) => match unopened(&reason) { ByteState::Absent { state, detail } => Err(refuse_state(state, detail)), ByteState::Here { .. } => unreachable!("unopened always answers Absent"), }, + Err(files::FileError::AboveWholeReadCap { bytes, .. }) => Ok(WholeRead::AboveCap(bytes)), + Err(e) => Err(refuse( + StatusCode::INTERNAL_SERVER_ERROR, + "drive.unopened", + format!("open the file's bytes: {e}"), + )), } } @@ -1635,6 +1979,62 @@ fn envelope_of(row: &Attestation) -> serde_json::Value { }) } +/// The row an edge rename replaced, when `file` is one (edge v33.1 +/// `files::rename`). Edge's renamed row does not name its prior; the prior's +/// `supersedes` names the replacement (`replacement_attestation_id`). The +/// prior is a SUPERSEDED row of the same room, author and blob, so it is found +/// in edge's lifecycle listing, and the link read off its composer. A +/// read-time walk, used by the metadata read only — not per listed row. +async fn renamed_from( + st: &DriveState, + caller: &str, + room: &ScopeRoom, + file: &files::FileRow, +) -> Option { + use ciris_persist::federation::types::attestation_type::SUPERSEDES; + let dir = st.engine.federation_directory(); + let mut after = None; + loop { + let page = files::in_room_with( + &st.engine, + room, + caller, + usize::MAX, + after, + ciris_persist::ceg::LifecycleView::All, + ) + .await + .ok()?; + for prior in page.files.iter().filter(|f| { + f.lifecycle == files::FileLifecycle::Superseded + && f.attestation_id != file.attestation_id + && f.attesting_key_id == file.attesting_key_id + && f.pointer.content_sha256 == file.pointer.content_sha256 + }) { + let Ok(composers) = dir + .list_attestations_referencing(&prior.attestation_id) + .await + else { + continue; + }; + if composers.iter().any(|c| { + c.attestation_type == SUPERSEDES + && c.attesting_key_id == prior.attesting_key_id + && c.attestation_envelope + .get("replacement_attestation_id") + .and_then(serde_json::Value::as_str) + == Some(file.attestation_id.as_str()) + }) { + return Some(prior.attestation_id.clone()); + } + } + match page.resume { + Some(c) => after = Some(c), + None => return None, + } + } +} + fn replaces_of(row: &Attestation) -> Option { row.attestation_envelope .get(FIELD_REPLACES) @@ -1648,7 +2048,10 @@ fn replaces_of(row: &Attestation) -> Option { async fn write_file( State(st): State, headers: HeaderMap, - body: Result, + // The RAW body: the multipart form streams it (0.5.218), so no extractor + // may buffer it first. Both forms' ceilings are enforced in + // `parse_upload`. + body: Body, ) -> Response { let owner = match drive_author(&st, &headers).await { Ok(o) => o, @@ -1671,7 +2074,7 @@ async fn write_file( { return resp; } - let up = match parse_upload(&headers, body) { + let up = match parse_upload(&headers, body).await { Ok(u) => u, Err(e) => return e, }; @@ -1697,16 +2100,15 @@ async fn write_file( .media_type .clone() .unwrap_or_else(|| "application/octet-stream".to_owned()); - match publish_into( + match publish_upload( &st, &headers, &owner.key_id, cohort, &room, - &up.bytes, + up.bytes, &media_type, up.filename.as_deref(), - Plane::Drive, ) .await { @@ -1854,10 +2256,10 @@ async fn read_drive( row, ) })); - // STOPGAP (CIRISEdge#693): the live listing no longer carries - // withdrawn rows, so a history view asks for them directly. + // The live listing does not carry withdrawn rows, so a history + // view asks edge's lifecycle listing for them (CIRISEdge#693). if include_withdrawn { - match stopgap_withdrawn_in_room(&st, room, &owner.key_id).await { + match withdrawn_in_room(&st, room, &owner.key_id).await { Ok(extra) => { for f in extra { if !rows @@ -1909,6 +2311,20 @@ async fn read_drive( } }; let dir = st.engine.federation_directory(); + // The person's devices, ONCE per page (the per-row custody summary below). + // A roster that cannot be read leaves `custody: null` on every row rather + // than failing a listing whose job is the files. + let devices = match crate::file_custody::owner_devices(&st.engine, &owner.key_id).await { + Ok(d) => Some(d), + Err(e) => { + tracing::warn!( + error = %e, + "drive: the device roster could not be read — rows list with `custody: null`" + ); + None + } + }; + let content = store(&st.engine); let mut out = Vec::with_capacity(rows.len()); for (cohort, room_id, file) in rows { let withdrawn = match withdrawn_by(&st, &file.attestation_id).await { @@ -1946,6 +2362,14 @@ async fn read_drive( .await .ok() .flatten(); + let described = describe(&st, &file, &viewer).await; + let custody = match (&devices, withdrawn.is_some()) { + (Some(d), false) => Some(crate::file_custody::compact( + d, + &crate::file_custody::receipts_of(&file, &content).await, + )), + _ => None, + }; out.push(DriveEntry { // THE ROOM THIS ROW CAME FROM. An unfiltered drive concatenates // several rooms, and `GET /v1/files/{id}` needs the right `cohort` @@ -1956,8 +2380,9 @@ async fn read_drive( attestation_id: file.attestation_id.clone(), author_key_id: file.attesting_key_id.clone(), asserted_at: file.asserted_at.to_rfc3339(), - filename: file.filename.clone(), - media_type: file.media_type.clone(), + filename: described.filename, + media_type: described.media_type, + description: described.how.to_owned(), bytes, detail, size, @@ -1967,6 +2392,7 @@ async fn read_drive( .as_ref() .map(envelope_of) .unwrap_or(serde_json::Value::Null), + custody, }); } // The ROOMS listed, not "the room" — an unfiltered drive spans several, and @@ -2050,6 +2476,7 @@ async fn file_meta( // the substrate records no holder there and the count is not a count of // devices. Said as such (`holder_claims_recorded: false`) rather than a // bare 0 a client would render as "nobody has it". + let described = describe(&st, &found.file, &viewer).await; let recorded = room.row_scope_token() == ciris_persist::federation::types::cohort_scope::COMMUNITY; let devices_holding = match (recorded, sha_of(&found.file)) { @@ -2064,8 +2491,9 @@ async fn file_meta( "attestation_id": found.file.attestation_id, "cohort": room.row_scope_token(), "room_id": room.content_group_id(), - "filename": found.file.filename, - "media_type": found.file.media_type, + "filename": described.filename, + "media_type": described.media_type, + "description": described.how, "author_key_id": found.file.attesting_key_id, "asserted_at": found.file.asserted_at.to_rfc3339(), "size": size, @@ -2083,7 +2511,10 @@ async fn file_meta( "tier": format!("{:?}", found.file.pointer.tier), "withdrawn": found.withdrawn_by.is_some(), "withdrawn_by": found.withdrawn_by, - "replaces": replaces_of(&found.row), + "replaces": match replaces_of(&found.row) { + Some(r) => Some(r), + None => renamed_from(&st, &owner.key_id, &room, &found.file).await, + }, "devices_holding": devices_holding, "holder_claims_recorded": recorded, "envelope": envelope_of(&found.row), @@ -2092,6 +2523,181 @@ async fn file_meta( .into_response() } +/// `GET /v1/files/{attestation_id}/custody` — **which of my devices this file +/// is on, out of how many** (`FSD/FILE_CUSTODY.md`; the source walk is in +/// [`crate::file_custody`]'s module doc). +/// +/// THE SAME DOORS AS THE BYTES, IN THE SAME ORDER. The owner session +/// (`drive.owner_session_required`), the cohort named and membership-checked +/// (`room_from_query`), the row found through edge's gated reader +/// (`find_file`), a withdrawn row answered 410 exactly as a read is — and then +/// persist's custody door, asked as the drive's viewer key, which runs +/// `read_any_for_viewer`'s tier gate: a viewer who cannot open the bytes gets +/// the byte read's refusal (`drive.not_granted`, `drive.evicted`, …) +/// through the same `refuse_state`, and learns nothing about who else can. +/// A custody view is not a side door to the access list. +/// +/// A device that holds the ROW but not yet the bytes gets a 200, not the byte +/// read's 409 (the maintainer's ruling on #704: "no copy here is a receipt +/// (node responsive, no copy)"): the view is authorized by the row, and this +/// device's own entry is `holds: "none"` with `checked_at`. Persist's custody +/// reads the blob's head row, which such a device does not have, so `access`, +/// `size_bytes` and the announced holders are `null`/empty there and a `why` +/// says so; the device that wrote the file is the complete answer — it is +/// also the device that admits every receipt. +async fn file_custody( + State(st): State, + headers: HeaderMap, + Path(attestation_id): Path, + Query(q): Query, +) -> Response { + let Some(owner) = crate::drive_auth::owner(&st, &headers).await else { + return drive_no_session(); + }; + let (_, room) = match room_from_query(&st, &owner.key_id, &q).await { + Ok(r) => r, + Err(e) => return e, + }; + let found = match find_file(&st, &owner.key_id, &room, &attestation_id, Plane::Drive).await { + Ok(f) => f, + Err(e) => return e, + }; + if let Some(w) = &found.withdrawn_by { + return row_withdrawn_refusal(w); + } + let viewer = match viewer_key(&st).await { + Ok(v) => v, + Err(e) => { + return refuse( + StatusCode::SERVICE_UNAVAILABLE, + "drive.no_node_key", + format!("resolve this node's key: {e}"), + ) + } + }; + let content = store(&st.engine); + // AUTHORIZED BY THE ROW (the maintainer's ruling on #704: "no copy here is + // a receipt (node responsive, no copy)"). Everything above — the session, + // the cohort's membership, edge's gated reader, the withdrawn row — is + // what lets this caller see the ROW, and that is the gate. Persist's + // custody door then answers from the blob's head row, which a device that + // has not pulled does not have: its `NotFetched` is not a refusal here but + // this device's own custody fact — it answered, and it holds no copy — + // reported as `holds: "none"` with the time it answered. Every OTHER + // refusal of that door (`not_granted` — this device's key holds no grant — + // an eviction, a seal fault) stays the byte read's refusal, by the same + // `refuse_state`. + let custody = match found.file.custody(&content, &viewer).await { + Ok(c) => Some(c), + Err(reason) if reason.kind() == "not_fetched" => None, + Err(reason) => { + return match unopened(&reason) { + ByteState::Absent { state, detail } => refuse_state(state, detail), + ByteState::Here { .. } => refuse_state("unopened", reason.to_string()), + } + } + }; + let checked_at = chrono::Utc::now().to_rfc3339(); + let devices = match crate::file_custody::owner_devices(&st.engine, &owner.key_id).await { + Ok(d) => d, + Err(e) => { + return refuse( + StatusCode::SERVICE_UNAVAILABLE, + "drive.store_unavailable", + format!("read your devices: {e}"), + ) + } + }; + let receipts = crate::file_custody::receipts_of(&found.file, &content).await; + // The AUTHOR DEVICE is the stream's producer — the node whose store holds + // the published root every receipt is admitted against. An inline file has + // no stream; its author device is the row's attester. + let author_device = ciris_edge::receipts::StreamSthClaim::from_row(&found.row) + .map(|c| c.producer_key_id) + .unwrap_or_else(|| found.file.attesting_key_id.clone()); + let this_device_is_author = devices + .iter() + .any(|d| d.this_device && d.keys.contains(&author_device)) + || author_device == st.node_signer.key_id; + let access = custody + .as_ref() + .map(crate::file_custody::access_device_keys); + let held_here = custody.as_ref().is_some_and(|c| c.held_here); + let half = crate::file_custody::device_half( + &devices, + access.as_ref(), + held_here, + &receipts, + this_device_is_author, + &checked_at, + ); + // Without a copy here the tier comes from the ROW (the pointer names the + // tier the bytes were sealed at), and so does CC 5.2's observability: a + // self/family blob is never announced, whoever asks. + let (tier, size_bytes, at_rest, copies_known, copies_observable, announced) = match &custody { + Some(c) => ( + c.tier.clone(), + Some(c.size_bytes), + c.sha256_hex.clone(), + c.copies_known, + c.copies_observable, + serde_json::to_value(&c.announced_holders).unwrap_or_default(), + ), + None => { + use ciris_persist::federation::types::cohort_scope::CryptoTier; + let tier = match found.file.pointer.tier { + CryptoTier::Plaintext => "plaintext", + CryptoTier::InvisibleEncrypted => "invisible_encrypted", + CryptoTier::CommunityDek => "community_dek", + }; + ( + tier.to_owned(), + None, + found.file.pointer.content_sha256.clone(), + 0, + tier != "invisible_encrypted", + serde_json::json!([]), + ) + } + }; + let mut why = crate::file_custody::substrate_why(copies_observable, &tier); + why.extend(half.why); + ( + StatusCode::OK, + Json(serde_json::json!({ + "attestation_id": found.file.attestation_id, + "cohort": room.row_scope_token(), + "room_id": room.content_group_id(), + "tier": tier, + // The stored length, from persist's custody; `null` when this + // device holds no copy (the row does not carry it). + "size_bytes": size_bytes, + "at_rest_sha256": at_rest, + "author_device": author_device, + "this_device_is_author": this_device_is_author, + "devices_total": half.devices.len(), + "devices": half.devices, + "held_here": held_here, + // A device with no copy is never counted as one. + "copies_known": copies_known, + "copies_observable": copies_observable, + "announced_holders": announced, + // `null` when this device holds no copy: who can open the blob is + // answered by persist from the blob's head row, which is not here. + "access": custody.as_ref().map(|c| &c.access), + "checked_at": checked_at, + // Every file is receiptable since edge v38.0.0 / persist v52 (an + // inline file has a one-leaf log, CIRISPersist#953): both fields + // stay on the wire for the client that reads them, constant. + "receipts_supported": true, + "receipts_unsupported_reason": serde_json::Value::Null, + "receipts_from_other_keys": half.other_receipts, + "why": why, + })), + ) + .into_response() +} + /// A single `Range: bytes=…` request, resolved against the plaintext size. enum RangeAsk { /// No usable Range header: serve the whole file. @@ -2104,9 +2710,15 @@ enum RangeAsk { /// RFC 9110 §14.1.2, one range. A multi-range or malformed header is IGNORED /// (the whole representation is served), which the RFC permits; a range past -/// the end is 416. A satisfiable range longer than [`WHOLE_READ_CAP`] is -/// SHORTENED to it — a 206's `Content-Range` names what was actually sent, and -/// every range client continues from there. +/// the end is 416; an end past the file is clamped to it (RFC 9110 §14.1.2 — +/// the HTTP door clamps, edge's `open_range` does not, so the clamp is HERE +/// and edge is only ever asked for bytes that exist). +/// +/// No cap on a range's LENGTH since 0.5.218. Before it, a satisfiable range +/// longer than [`WHOLE_READ_CAP`] was shortened to it, because the range was +/// read into memory in one call. It is now served in [`streaming::RANGE_WINDOW`] +/// windows (edge's `FileRow::open_range`, CIRISEdge#737), so `bytes=0-` on a +/// 2 GiB file is 2 GiB in 1 MiB steps, never 2 GiB held. fn parse_range(h: Option<&str>, total: u64) -> RangeAsk { let Some(spec) = h.and_then(|v| v.trim().strip_prefix("bytes=")) else { return RangeAsk::Whole; @@ -2144,8 +2756,7 @@ fn parse_range(h: Option<&str>, total: u64) -> RangeAsk { } (s, e.min(total - 1)) }; - let cap = WHOLE_READ_CAP as u64; - RangeAsk::Part(start, end.min(start.saturating_add(cap - 1))) + RangeAsk::Part(start, end) } /// `Content-Disposition: attachment` with the file's name — an ASCII fallback @@ -2173,6 +2784,33 @@ fn content_disposition(filename: Option<&str>) -> HeaderValue { /// `GET /v1/files/{attestation_id}` — the bytes, or the reason they are not /// here. JSON by default; `?raw=1` answers the bytes themselves, with `Range`. +/// +/// # What streams and what does not (0.5.218) +/// +/// * **JSON** (no `raw`): whole, always — the bytes are one base64 value. +/// Above [`WHOLE_READ_CAP`] it is `413 drive.too_large_for_whole_read`, +/// pointing at `?raw=1`. +/// * **`?raw=1`, no `Range`, at or below the cap**: whole, as before, with +/// RFC 9530 `Repr-Digest` computed over the bytes in hand. +/// * **`?raw=1`, no `Range`, above the cap**: STREAMED from edge's +/// `FileRow::chunks()` (one ≤ 1 MiB chunk in hand, [`streaming::IN_FLIGHT`] +/// queued), `Content-Length` from the probe. **No `Repr-Digest`**: the +/// header precedes the body, and the digest of a streamed body is known +/// only after its last byte. Computing it first would mean decrypting the +/// whole file twice or holding it — the exact cost streaming exists to +/// avoid — and a header that promised a digest the node had not computed +/// would be worse than none. The plaintext digest a client verifies +/// against is `content_digest` on the JSON read or `/meta`, where it can be +/// computed (at or below the cap); above it a client hashes what it +/// receives against the row's own claim when edge's file row carries one +/// (CIRISEdge#638). +/// * **`?raw=1` with `Range`**: any satisfiable single range, of any length, +/// through `FileRow::open_range` in [`streaming::RANGE_WINDOW`] windows — +/// never a `Repr-Digest` (a 206 is not the representation). +/// +/// A streamed body's FIRST item is awaited before the status line, so a +/// refusal the probe could not foresee is still a status; a failure after it +/// tears the body (see `drive/streaming.rs`). async fn read_file( State(st): State, headers: HeaderMap, @@ -2200,14 +2838,15 @@ async fn read_file( ) } }; + let described = describe(&st, &found.file, &viewer).await; if !truthy(q.raw.as_deref()) { return match open_whole(&st, &found, &viewer).await { Ok((bytes, _)) => ( StatusCode::OK, Json(serde_json::json!({ "attestation_id": found.file.attestation_id, - "media_type": found.file.media_type, - "filename": found.file.filename, + "media_type": described.media_type, + "filename": described.filename, "size": bytes.len(), "content_digest": plaintext_digest(&bytes), "content_digest_alg": "sha-256", @@ -2226,8 +2865,7 @@ async fn read_file( ByteState::Here { size } => size, ByteState::Absent { state, detail } => return refuse_state(state, detail), }; - let media = found - .file + let media = described .media_type .clone() .unwrap_or_else(|| "application/octet-stream".to_owned()); @@ -2239,7 +2877,7 @@ async fn read_file( ); h.insert( header::CONTENT_DISPOSITION, - content_disposition(found.file.filename.as_deref()), + content_disposition(described.filename.as_deref()), ); h.insert(header::ACCEPT_RANGES, HeaderValue::from_static("bytes")); h.insert( @@ -2254,13 +2892,31 @@ async fn read_file( (Some(_), None) => RangeAsk::Whole, }; match ask { - RangeAsk::Whole => match open_whole(&st, &found, &viewer).await { - Ok((bytes, _)) => { + RangeAsk::Whole => match read_whole(&st, &found, &viewer).await { + Ok(WholeRead::Bytes(bytes, _)) => { if let Some(v) = repr_digest(&bytes) { h.insert(header::HeaderName::from_static("repr-digest"), v); } (StatusCode::OK, h, bytes).into_response() } + // Above the cap: the whole file, streamed. See the doc above for + // why there is no `Repr-Digest` here. + Ok(WholeRead::AboveCap(total)) => { + let mut rx = streaming::spawn_chunks( + Arc::clone(&st.engine), + found.file.clone(), + viewer.clone(), + ); + let first = match streaming::first_item(&mut rx).await { + Some(Ok(b)) => b, + Some(Err(e)) => return read_refusal(&e), + None => Vec::new(), + }; + if let Ok(v) = HeaderValue::from_str(&total.to_string()) { + h.insert(header::CONTENT_LENGTH, v); + } + (StatusCode::OK, h, streaming::body(first, rx, Some(total))).into_response() + } Err(e) => e, }, RangeAsk::Unsatisfiable => { @@ -2276,31 +2932,64 @@ async fn read_file( resp } RangeAsk::Part(start, end) => { - let Some(sha) = sha_of(&found.file) else { - return refuse_state("malformed_row", state_detail("malformed_row").to_owned()); + // Edge's range door (CIRISEdge#737) in windows: the row's own + // binding and AAD, only the covering chunks opened, no length cap. + let mut rx = streaming::spawn_range( + Arc::clone(&st.engine), + found.file.clone(), + viewer.clone(), + start, + end, + ); + let first = match streaming::first_item(&mut rx).await { + Some(Ok(b)) => b, + Some(Err(e)) => return read_refusal(&e), + None => Vec::new(), }; - let aad = aad_for(&found.file); - match st - .engine - .read_blob_range_as(&sha, &viewer, start, end, aad.as_deref()) - .await + let len = end - start + 1; + if let Ok(v) = + HeaderValue::from_str(&format!("bytes {start}-{end}/{}", size.unwrap_or(0))) { - Ok(bytes) => { - let sent_end = start + (bytes.len() as u64).saturating_sub(1); - if let Ok(v) = HeaderValue::from_str(&format!( - "bytes {start}-{sent_end}/{}", - size.unwrap_or(0) - )) { - h.insert(header::CONTENT_RANGE, v); - } - (StatusCode::PARTIAL_CONTENT, h, bytes).into_response() - } - Err(e) => match blob_state(&e) { - ByteState::Absent { state, detail } => refuse_state(state, detail), - ByteState::Here { .. } => unreachable!("blob_state always answers Absent"), - }, + h.insert(header::CONTENT_RANGE, v); + } + if let Ok(v) = HeaderValue::from_str(&len.to_string()) { + h.insert(header::CONTENT_LENGTH, v); + } + ( + StatusCode::PARTIAL_CONTENT, + h, + streaming::body(first, rx, Some(len)), + ) + .into_response() + } + } +} + +/// An edge read refusal on a raw read, before the status line: the byte-state +/// words for an unopened file, 416 for a range edge found outside it. +fn read_refusal(e: &files::FileError) -> Response { + match e { + files::FileError::Unopened(reason) => match unopened(reason) { + ByteState::Absent { state, detail } => refuse_state(state, detail), + ByteState::Here { .. } => unreachable!("unopened always answers Absent"), + }, + files::FileError::RangeNotSatisfiable { size, .. } => { + let total = size.unwrap_or(0); + let mut resp = refuse( + StatusCode::RANGE_NOT_SATISFIABLE, + "drive.range_not_satisfiable", + format!("the requested range is outside this {total}-byte file"), + ); + if let Ok(v) = HeaderValue::from_str(&format!("bytes */{total}")) { + resp.headers_mut().insert(header::CONTENT_RANGE, v); } + resp } + other => refuse( + StatusCode::INTERNAL_SERVER_ERROR, + "drive.unopened", + format!("open the file's bytes: {other}"), + ), } } @@ -2312,7 +3001,8 @@ async fn replace_file( headers: HeaderMap, Path(attestation_id): Path, Query(q): Query, - body: Result, + // Raw, as `write_file`'s: the multipart form streams. + body: Body, ) -> Response { let owner = match drive_author(&st, &headers).await { Ok(o) => o, @@ -2332,40 +3022,51 @@ async fn replace_file( Ok(f) => f, Err(e) => return e, }; - if let Err(e) = require_author(&st, &found.row) { + if let Err(e) = require_author(&st, &owner.key_id, &found.row) { return e; } if let Some(w) = &found.withdrawn_by { return row_withdrawn_refusal(w); } - let up = match parse_upload(&headers, body) { + let up = match parse_upload(&headers, body).await { Ok(u) => u, Err(e) => return e, }; - // Unnamed members keep the old file's: a replace changes the BYTES. + // Unnamed members keep the old file's: a replace changes the BYTES. The + // old name and type are read through the descriptor (CIRISEdge#698). + let old = describe_here(&st, &found.file).await; let media_type = up .media_type .clone() - .or_else(|| found.file.media_type.clone()) + .or_else(|| old.media_type.clone()) .unwrap_or_else(|| "application/octet-stream".to_owned()); - let filename = up.filename.clone().or_else(|| found.file.filename.clone()); - let (published, addressed) = match publish_into( + let filename = up.filename.clone().or_else(|| old.filename.clone()); + let (published, addressed) = match publish_upload( &st, &headers, &owner.key_id, cohort, &room, - &up.bytes, + up.bytes, &media_type, filename.as_deref(), - Plane::Drive, ) .await { Ok(p) => p, Err(e) => return e, }; - let w = match withdraw_rows(&st, &found.row, &found.file, "replaced by its author").await { + let w = match withdraw_rows( + &st, + &headers, + &owner.key_id, + Plane::Drive, + &found.row, + &found.file, + "replaced by its author", + ) + .await + { Ok(w) => w, Err(e) => return withdraw_failed(e), }; @@ -2443,70 +3144,41 @@ async fn rename_file( Ok(f) => f, Err(e) => return e, }; - if let Err(e) = require_author(&st, &found.row) { + if let Err(e) = require_author(&st, &owner.key_id, &found.row) { return e; } if let Some(w) = &found.withdrawn_by { return row_withdrawn_refusal(w); } - let row = match rename_row( - &st.node_signer, - &room, - &found.file, - filename, - &attestation_id, - ) - .await - { - Ok(r) => r, - Err(e) => { - return refuse( - StatusCode::INTERNAL_SERVER_ERROR, - "drive.publish_failed", - format!("build the renamed row: {e}"), - ) - } + // EDGE'S RENAME (v33.1.0, CIRISEdge#702): a new row over the same bytes, + // the name re-sealed for it, authored by the old row's own signer and + // crossed like a publish; only once it crossed, a `supersedes` retires + // the old row. The owner's pen signs a person-authored row. + let capsule = match author_capsule(&st, &headers, &owner.key_id, Plane::Drive).await { + Ok(c) => c, + Err(r) => return r, }; let dir = st.engine.federation_directory(); - if let Err(e) = dir - .put_attestation_authored(ciris_persist::federation::SignedAttestation { - attestation: row.clone(), - }) - .await - { - return refuse( - StatusCode::INTERNAL_SERVER_ERROR, - "drive.publish_failed", - format!("author the renamed row: {e:#}"), - ); - } - let crossing = match ciris_edge::replication::attestation_bind::share( + let content = store(&st.engine); + let renamed = match files::rename( &*dir, - &row, - room.widen_to(), - ciris_edge::replication::attestation_bind::CrossingBasis::ProducerAuthority, + &content, ciris_edge::replication::attestation_bind::Signers { node: &st.node_signer, - actor: None, + actor: Some(capsule.edge_signer()), }, + &room, + &found.file, + Some(filename), + &attestation_id, ) .await { - Ok(c) => c, - Err(e) => { - return refuse( - StatusCode::INTERNAL_SERVER_ERROR, - "drive.publish_failed", - format!("cross the renamed row into {room}: {e}"), - ) - } + Ok(r) => r, + Err(e) => return file_error(&e, &room), }; crate::compose::kick_replication("file renamed"); - let new_id = placed_or(&crossing.shared, &row.attestation_id).to_owned(); - let w = match withdraw_rows(&st, &found.row, &found.file, "renamed by its author").await { - Ok(w) => w, - Err(e) => return withdraw_failed(e), - }; + let new_id = placed_or(&renamed.shared, &renamed.row.attestation_id).to_owned(); ( StatusCode::OK, Json(serde_json::json!({ @@ -2516,11 +3188,10 @@ async fn rename_file( "content_sha256": found.file.pointer.content_sha256, "cohort": room.row_scope_token(), "room": room.to_string(), - "crossed": !matches!( - crossing.shared, - ciris_edge::replication::attestation_bind::Shared::AwaitingActor { .. } - ), - "withdrawn": w.withdrawn, + "crossed": renamed.crossed, + // The prior row is SUPERSEDED (edge's rename), not withdrawn: the + // bytes are the same bytes. Retired only once the new row crossed. + "superseded": renamed.crossed, })), ) .into_response() @@ -2553,13 +3224,23 @@ async fn withdraw_file( Ok(f) => f, Err(e) => return e, }; - if let Err(e) = require_author(&st, &found.row) { + if let Err(e) = require_author(&st, &owner.key_id, &found.row) { return e; } if let Some(w) = &found.withdrawn_by { return row_withdrawn_refusal(w); } - match withdraw_rows(&st, &found.row, &found.file, "withdrawn by its author").await { + match withdraw_rows( + &st, + &headers, + &owner.key_id, + Plane::Drive, + &found.row, + &found.file, + "withdrawn by its author", + ) + .await + { Ok(w) => ( StatusCode::OK, Json(serde_json::json!({ @@ -2646,7 +3327,7 @@ async fn move_file( Ok(f) => f, Err(e) => return e, }; - if let Err(e) = require_author(&st, &found.row) { + if let Err(e) = require_author(&st, &owner.key_id, &found.row) { return e; } let viewer = match viewer_key(&st).await { @@ -2662,35 +3343,85 @@ async fn move_file( // A RESEAL, not a re-pointer: the target room's tier and group decide the // seal (a community DEK is not a self wrap), so the bytes are opened here // and sealed again there. - let (bytes, _) = match open_whole(&st, &found, &viewer).await { - Ok(b) => b, - Err(e) => return e, - }; - let media_type = found - .file + // + // ABOVE THE WHOLE-READ CAP (0.5.218): the reseal STREAMS — the source's + // `FileRow::chunks()` walk is the reader for the target's + // `files::publish_stream`, so a 2 GiB move holds a chunk on each side, + // never the file. The declared length is the probe's size (a chunk DAG's + // pointer declares it, or its manifest does); the gate reads the walk's + // first 64 KiB without consuming them, as it does on a streamed upload. A + // walk that fails mid-file fails the publish (edge: `FileError::Read`, + // nothing sealed, no row) and answers the walk's own refusal. + let old = describe(&st, &found.file, &viewer).await; + let media_type = old .media_type .clone() .unwrap_or_else(|| "application/octet-stream".to_owned()); - let (published, addressed) = match publish_into( - &st, - &headers, - &owner.key_id, - target_cohort, - &target, - &bytes, - &media_type, - found.file.filename.as_deref(), - Plane::Drive, - ) - .await - { + let published = match read_whole(&st, &found, &viewer).await { + Err(e) => return e, + Ok(WholeRead::Bytes(bytes, _)) => { + publish_into( + &st, + &headers, + &owner.key_id, + target_cohort, + &target, + &bytes, + &media_type, + old.filename.as_deref(), + Plane::Drive, + ) + .await + } + Ok(WholeRead::AboveCap(total)) => { + let mut walk = streaming::ChunkReader::new(streaming::spawn_chunks( + Arc::clone(&st.engine), + found.file.clone(), + viewer.clone(), + )); + let gated = match walk.peek(crate::media_gate::FORMAT_HEAD_BYTES).await { + Ok(head) => match write_gate(&media_type, head, old.filename.as_deref()) { + Ok(g) => g, + Err(e) => return e, + }, + Err(e) => return read_refusal(&e), + }; + let out = publish_gated( + &st, + &headers, + &owner.key_id, + target_cohort, + &target, + &gated, + total, + &mut walk, + Plane::Drive, + ) + .await; + match (out, walk.failure()) { + (Err(_), Some(e)) => Err(read_refusal(e)), + (out, _) => out, + } + } + }; + let (published, addressed) = match published { Ok(p) => p, Err(e) => return e, }; let withdrawn = if req.keep_source { Vec::new() } else { - match withdraw_rows(&st, &found.row, &found.file, "moved by its author").await { + match withdraw_rows( + &st, + &headers, + &owner.key_id, + Plane::Drive, + &found.row, + &found.file, + "moved by its author", + ) + .await + { Ok(w) => w.withdrawn, Err(e) => return withdraw_failed(e), } @@ -2714,7 +3445,24 @@ fn file_error(e: &files::FileError, room: &ScopeRoom) -> Response { // Unreachable from `files::publish` since CIRISEdge#633 (it seals above // the 1 MiB envelope bound as a chunk DAG). Kept, under the same id as // this node's own cap, for a store that implements only `seal`. - F::TooLargeForInline { size, .. } => too_large(*size), + F::TooLargeForInline { size, .. } => too_large(*size as u64), + // CIRISEdge#744: the upload's body was not the length its `size` said. + // Nothing was sealed and there is no row; the remedy is to resend + // with the right count, so it is the CLIENT's 400, never a 500. + F::DeclaredLengthMismatch { declared, read } => refuse_with( + StatusCode::BAD_REQUEST, + "drive.declared_length_mismatch", + format!( + "the form declared `size` = {declared} bytes and the file part carried {} — \ + nothing was written; resend with `size` equal to the file's exact byte count", + if read > declared { + format!("more than that (stopped at {read})") + } else { + read.to_string() + } + ), + serde_json::json!({ "declared": declared, "read": read }), + ), F::ReadableByNobody { .. } => refuse( StatusCode::CONFLICT, "drive.readable_by_nobody", @@ -2794,11 +3542,17 @@ pub struct Note { /// media type alone is not enough — `POST /v1/files` can put a named `.txt` in /// the same room, and a notes list that swallowed it would report somebody's /// uploaded file as something they had written. -fn is_note(file: &files::FileRow) -> bool { - file.media_type - .as_deref() - .is_some_and(|m| m.starts_with("text/plain")) - && file.filename.is_none() +/// +/// Read through the DESCRIPTOR (CIRISEdge#698): a sealed row's clear members +/// are `None`, so testing them would call every sealed upload a note. A row +/// whose descriptor did not open here is not known to be a note and is not +/// listed as one. +fn is_note(d: &Described) -> bool { + d.how != "sealed" + && d.media_type + .as_deref() + .is_some_and(|m| m.starts_with("text/plain")) + && d.filename.is_none() } #[allow(clippy::result_large_err)] // the Err IS an axum Response @@ -2807,7 +3561,7 @@ fn require_note_body(body: &str) -> Result<(), Response> { return Err(refuse( StatusCode::BAD_REQUEST, "notes.empty", - "a note with no body is not a note".into(), + "A note with nothing in it isn't a note.".into(), )); } Ok(()) @@ -2901,14 +3655,14 @@ async fn find_own_note( let room = ciris_edge::self_room::room(owner_key_id); let found = find_file(st, owner_key_id, &room, attestation_id, Plane::Notes).await?; // A withdrawn note, or a file that is not a note, is not one of your notes. - if !is_note(&found.file) || found.withdrawn_by.is_some() { + if !is_note(&describe_here(st, &found.file).await) || found.withdrawn_by.is_some() { return Err(refuse( StatusCode::NOT_FOUND, "notes.not_found", format!("{attestation_id} is not one of your notes"), )); } - require_author(st, &found.row)?; + require_author(st, owner_key_id, &found.row)?; Ok(found) } @@ -2945,7 +3699,17 @@ async fn update_note( Ok(p) => p, Err(e) => return e, }; - let w = match withdraw_rows(&st, &found.row, &found.file, "edited by its author").await { + let w = match withdraw_rows( + &st, + &headers, + &owner.key_id, + Plane::Notes, + &found.row, + &found.file, + "edited by its author", + ) + .await + { Ok(w) => w, Err(e) => return withdraw_failed(e), }; @@ -2983,7 +3747,17 @@ async fn withdraw_note( Ok(f) => f, Err(e) => return e, }; - match withdraw_rows(&st, &found.row, &found.file, "withdrawn by its author").await { + match withdraw_rows( + &st, + &headers, + &owner.key_id, + Plane::Notes, + &found.row, + &found.file, + "withdrawn by its author", + ) + .await + { Ok(w) => ( StatusCode::OK, Json(serde_json::json!({ @@ -3043,7 +3817,7 @@ async fn read_notes( if out.len() >= limit { break 'rooms; } - if !is_note(&row) { + if !is_note(&describe(&st, &row, &viewer).await) { continue; } match withdrawn_by(&st, &row.attestation_id).await { @@ -3070,10 +3844,13 @@ async fn read_notes( "the bytes opened but are not UTF-8 text".to_owned(), ), }, - Err(reason) => match unopened(&reason) { + Err(files::FileError::Unopened(reason)) => match unopened(&reason) { ByteState::Absent { state, detail } => (None, state.to_owned(), detail), ByteState::Here { .. } => unreachable!("unopened always answers Absent"), }, + // A note is text; one above the 64 MiB whole-read cap, or any + // other edge refusal, is reported as unreadable with the reason. + Err(e) => (None, "unreadable".to_owned(), e.to_string()), }; out.push(Note { attestation_id: row.attestation_id.clone(), @@ -3119,26 +3896,29 @@ pub fn router( user_seed_dir, scope_lifecycle, }; - // THE UPLOAD ROUTES ONLY carry the raised body limit. axum's 2 MB default + // THE UPLOAD ROUTES ONLY carry a raised body limit. axum's 2 MB default // stands everywhere else; before 0.5.216 it stood HERE too, so the largest // file anyone could upload was ~1.5 MB of base64 while the comment on the // form said edge capped at 1 MiB — two stale numbers, neither the real one. - let upload_limit = DefaultBodyLimit::max(UPLOAD_BODY_LIMIT); + // + // Since 0.5.218 the upload handlers take the RAW body (the multipart form + // streams it to the seal) and enforce their two ceilings themselves — + // [`UPLOAD_BODY_LIMIT`] for JSON, [`STREAMED_UPLOAD_BODY_LIMIT`] for + // multipart — in `parse_upload`, so no `DefaultBodyLimit` layer is needed + // (axum's limit applies only to the buffering extractors). use axum::routing::{get, post, put}; Router::new() - .route("/v1/files", post(write_file).layer(upload_limit)) + .route("/v1/files", post(write_file)) .route("/v1/drive", get(read_drive)) // Public: a node's render policy is what its clients need BEFORE they // hold a session, and it discloses nothing about anyone (#643). .route("/v1/media/policy", get(media_policy)) .route( "/v1/files/{attestation_id}", - get(read_file) - .put(replace_file) - .delete(withdraw_file) - .layer(upload_limit), + get(read_file).put(replace_file).delete(withdraw_file), ) .route("/v1/files/{attestation_id}/meta", get(file_meta)) + .route("/v1/files/{attestation_id}/custody", get(file_custody)) .route("/v1/files/{attestation_id}/rename", post(rename_file)) .route("/v1/files/{attestation_id}/move", post(move_file)) .route("/v1/notes", get(read_notes).post(write_note)) @@ -3160,6 +3940,10 @@ mod tests { assert!(b64.len() <= 4 * 1024); assert!(UPLOAD_BODY_LIMIT > WHOLE_READ_CAP.div_ceil(3) * 4); assert_eq!(WHOLE_READ_CAP, 64 * 1024 * 1024); + // The streamed form: edge's ~2.5 GiB, well above the whole-read cap, + // and its body limit leaves room for the form around the file. + const { assert!(STREAMED_FILE_CEILING > WHOLE_READ_CAP as u64 * 32) }; + const { assert!(STREAMED_UPLOAD_BODY_LIMIT > STREAMED_FILE_CEILING) }; } #[test] @@ -3181,6 +3965,13 @@ mod tests { parse_range(Some("bytes=8-100"), 10), RangeAsk::Part(8, 9) )); + // No length cap (0.5.218): a range longer than the whole-read cap is + // served whole, in windows. + let big = 3 * WHOLE_READ_CAP as u64; + assert!(matches!( + parse_range(Some("bytes=0-"), big), + RangeAsk::Part(0, e) if e == big - 1 + )); assert!(matches!( parse_range(Some("bytes=10-"), 10), RangeAsk::Unsatisfiable @@ -3204,31 +3995,6 @@ mod tests { )); } - #[test] - fn multipart_reads_the_file_part_and_the_fields() { - let body = b"--XyZ\r\nContent-Disposition: form-data; name=\"cohort\"\r\n\r\nself\r\n\ ---XyZ\r\nContent-Disposition: form-data; name=\"file\"; filename=\"boat.jpg\"\r\n\ -Content-Type: image/jpeg\r\n\r\n\x00\x01\r\n\x02\r\n--XyZ--\r\n"; - assert_eq!( - multipart::boundary("multipart/form-data; boundary=XyZ").as_deref(), - Some("XyZ") - ); - assert_eq!( - multipart::boundary("multipart/form-data; boundary=\"XyZ\"").as_deref(), - Some("XyZ") - ); - let parts = multipart::parse(body, "XyZ").expect("parse"); - assert_eq!(parts.len(), 2); - assert_eq!(parts[0].name, "cohort"); - assert_eq!(parts[0].data, b"self"); - assert_eq!(parts[1].name, "file"); - assert_eq!(parts[1].filename.as_deref(), Some("boat.jpg")); - assert_eq!(parts[1].content_type.as_deref(), Some("image/jpeg")); - // CRLF INSIDE the bytes survives: only CRLF + delimiter ends a part. - assert_eq!(parts[1].data, b"\x00\x01\r\n\x02"); - assert!(multipart::parse(b"no boundary here", "XyZ").is_err()); - } - #[test] fn content_disposition_names_the_file_in_both_forms() { let v = content_disposition(Some("Mira's \"boat\" — 1.jpg")); diff --git a/src/drive/multipart.rs b/src/drive/multipart.rs new file mode 100644 index 00000000..7001082c --- /dev/null +++ b/src/drive/multipart.rs @@ -0,0 +1,640 @@ +//! **A streaming `multipart/form-data` reader (RFC 7578) — the upload form, +//! read one frame at a time** (0.5.218). +//! +//! Before 0.5.218 the drive buffered the WHOLE request body and sliced parts +//! out of it, which is what capped an upload at persist's 64 MiB whole-read +//! cap: the node could only take what it was willing to hold. Edge v36.1.0's +//! `files::publish_stream` (CIRISEdge#744) seals from a READER, chunk by chunk, +//! so the only thing standing between a phone's 1 GiB video and the drive was +//! this parser. This one never holds more than one body frame plus a +//! delimiter's worth of look-behind. +//! +//! # The shape it accepts, and why the file comes LAST +//! +//! Form fields (`cohort`, `room_id`, `media_type`, `filename`, `size`) are +//! read whole — each is bounded by [`FIELD_CAP`] — and the file part is NOT: +//! its bytes are handed to the seal as they arrive. Once they are, the +//! handler has committed to a room, a type and a length, so every field it +//! needs must already be in hand. A field AFTER the file would arrive after +//! the decision it was meant to inform, so it is refused by name +//! ([`Failure::FieldAfterFile`]) rather than ignored — and refused BEFORE the +//! file's reader reports end-of-file, which is before edge seals the stream +//! (it seals only on the reader's EOF), so the refusal leaves no row. +//! A browser's `FormData` sends parts in insertion order, so "append the file +//! last" is the whole client-side rule. +//! +//! # The delimiter trick +//! +//! RFC 2046 §5.1.1 defines every delimiter as `CRLF "--" boundary`, the CRLF +//! belonging to the delimiter, not the part before it. The FIRST delimiter +//! may have no preceding CRLF (it may open the body), so the reader seeds its +//! buffer with one: every delimiter, first included, is then the same needle, +//! and a CRLF inside a part's bytes can never end it — only CRLF followed by +//! `--boundary` does. +//! +//! # Look-behind +//! +//! A frame boundary may split the delimiter. Bytes are released to the caller +//! only once no delimiter can START among them: while no match is found, the +//! last `delimiter.len() - 1` bytes of the buffer are held back until the next +//! frame decides them. + +use std::pin::Pin; +use std::task::{Context, Poll}; + +use axum::body::Bytes; +use futures_util::Stream; + +/// The body, as frames. Errors are the transport's, rendered. +pub type ByteStream = Pin> + Send>>; + +/// The largest form FIELD (not the file) this reader will hold. Every field +/// the drive reads is an id, a token, a media type or a filename — 16 KiB is +/// far above any honest one and far below anything that could be used to +/// make the node buffer a file under a field's name. +pub const FIELD_CAP: usize = 16 * 1024; + +/// The largest part-header block. Same reasoning as [`FIELD_CAP`]. +const HEADER_CAP: usize = 16 * 1024; + +/// Why the body could not be read as the upload form. Each arm has a +/// different remedy, so the handler maps each to its own refusal. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Failure { + /// The body passed the route's ceiling (`limit` bytes). + TooLarge { + /// The ceiling that was passed. + limit: u64, + }, + /// A part followed the file part. See the module doc. + FieldAfterFile, + /// The body ended before its closing delimiter — a client that stopped + /// sending, or a connection that dropped. + Truncated, + /// Not the multipart shape: a missing boundary, a header block without + /// its terminator, a field above [`FIELD_CAP`]. + Malformed(String), + /// The transport failed while the body was being read. + Transport(String), +} + +impl std::fmt::Display for Failure { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::TooLarge { limit } => { + write!(f, "the body passed this route's {limit}-byte ceiling") + } + Self::FieldAfterFile => f.write_str( + "a form part follows the `file` part — every field must precede the file, which \ + is streamed to the seal as it arrives", + ), + Self::Truncated => f.write_str("the body ended before its closing multipart delimiter"), + Self::Malformed(d) => write!(f, "malformed multipart body: {d}"), + Self::Transport(d) => write!(f, "reading the request body failed: {d}"), + } + } +} + +/// One part's headers. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct PartHead { + /// `Content-Disposition` `name`. + pub name: String, + /// `Content-Disposition` `filename`, when the part is a file. + pub filename: Option, + /// The part's own `Content-Type`. + pub content_type: Option, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum State { + /// Before the first delimiter (a preamble, if any, is discarded). + Preamble, + /// Just past a delimiter: `--` closes the body, CRLF opens a part. + AfterDelimiter, + /// Inside a part's bytes. + InPart, + /// Past the closing delimiter. Anything after it is epilogue, never read. + Closed, +} + +/// The reader. See the module doc. +pub struct Multipart { + stream: ByteStream, + buf: Vec, + /// Consumed prefix of `buf`. + pos: usize, + /// `CRLF -- boundary`. + delim: Vec, + /// Absolute index in `buf` below which no delimiter starts (a search + /// cache, so small reads do not rescan the same bytes). + clean_until: usize, + read: u64, + limit: u64, + ended: bool, + state: State, + failure: Option, +} + +/// The `boundary` parameter of a `multipart/form-data` content type. +pub fn boundary(content_type: &str) -> Option { + content_type.split(';').skip(1).find_map(|param| { + let (k, v) = param.split_once('=')?; + if k.trim().eq_ignore_ascii_case("boundary") { + let v = v.trim().trim_matches('"'); + (!v.is_empty()).then(|| v.to_owned()) + } else { + None + } + }) +} + +/// A `Content-Disposition` parameter, quotes stripped. +fn disposition_param(value: &str, key: &str) -> Option { + value.split(';').skip(1).find_map(|param| { + let (k, v) = param.split_once('=')?; + k.trim() + .eq_ignore_ascii_case(key) + .then(|| v.trim().trim_matches('"').to_owned()) + }) +} + +/// First index `>= from` where `needle` starts in `hay`. A first-byte scan, +/// then a compare — the needle starts with CR, which is rare in most bytes. +fn find(hay: &[u8], needle: &[u8], from: usize) -> Option { + let first = *needle.first()?; + let mut at = from; + while at + needle.len() <= hay.len() { + let rel = hay[at..=hay.len() - needle.len()] + .iter() + .position(|&b| b == first)?; + let i = at + rel; + if &hay[i..i + needle.len()] == needle { + return Some(i); + } + at = i + 1; + } + None +} + +impl Multipart { + /// A reader over `stream`, refusing past `limit` body bytes. + pub fn new(stream: ByteStream, boundary: &str, limit: u64) -> Self { + let mut delim = b"\r\n--".to_vec(); + delim.extend_from_slice(boundary.as_bytes()); + Self { + stream, + // The seeded CRLF: see "The delimiter trick". + buf: b"\r\n".to_vec(), + pos: 0, + delim, + clean_until: 0, + read: 0, + limit, + ended: false, + state: State::Preamble, + failure: None, + } + } + + /// Why reading stopped, when it did. The handler reads this after a + /// failed seal: edge renders a reader error as prose inside + /// `FileError::Read`, and the typed cause is here. + pub fn failure(&self) -> Option<&Failure> { + self.failure.as_ref() + } + + fn fail(&mut self, f: Failure) -> Failure { + self.failure.get_or_insert(f).clone() + } + + /// One more frame into the buffer. `Ok(false)` at the end of the body. + fn poll_fill(&mut self, cx: &mut Context<'_>) -> Poll> { + if let Some(f) = &self.failure { + return Poll::Ready(Err(f.clone())); + } + if self.ended { + return Poll::Ready(Ok(false)); + } + match self.stream.as_mut().poll_next(cx) { + Poll::Pending => Poll::Pending, + Poll::Ready(None) => { + self.ended = true; + Poll::Ready(Ok(false)) + } + Poll::Ready(Some(Err(e))) => Poll::Ready(Err(self.fail(Failure::Transport(e)))), + Poll::Ready(Some(Ok(frame))) => { + self.read = self.read.saturating_add(frame.len() as u64); + if self.read > self.limit { + return Poll::Ready(Err(self.fail(Failure::TooLarge { limit: self.limit }))); + } + // Compact before growing: the consumed prefix is dead. + if self.pos > 0 { + self.buf.drain(..self.pos); + self.clean_until = self.clean_until.saturating_sub(self.pos); + self.pos = 0; + } + self.buf.extend_from_slice(&frame); + Poll::Ready(Ok(true)) + } + } + } + + async fn fill(&mut self) -> Result { + std::future::poll_fn(|cx| self.poll_fill(cx)).await + } + + /// Where the next delimiter starts, searching only bytes not already + /// known clean. + fn next_delim(&mut self) -> Option { + let from = self.clean_until.max(self.pos); + match find(&self.buf, &self.delim, from) { + Some(i) => { + self.clean_until = i; + Some(i) + } + None => { + self.clean_until = self + .buf + .len() + .saturating_sub(self.delim.len() - 1) + .max(self.pos); + None + } + } + } + + /// The next part's headers, or `None` past the closing delimiter. + /// + /// Must not be called while a part's bytes are unread except for a part + /// this reader was told to skip; the drive reads every field whole, and + /// the file is always the last part. + pub async fn next_part(&mut self) -> Result, Failure> { + loop { + match self.state { + State::Closed => return Ok(None), + State::Preamble | State::InPart => { + // Skip to the next delimiter (the preamble, or an unread + // part the caller chose not to read). + if let Some(i) = self.next_delim() { + self.pos = i + self.delim.len(); + self.state = State::AfterDelimiter; + continue; + } + self.pos = self.clean_until; + if !self.fill().await? { + return Err(self.fail(if self.state == State::Preamble { + Failure::Malformed("the body does not contain its boundary".into()) + } else { + Failure::Truncated + })); + } + } + State::AfterDelimiter => { + if self.buf.len() - self.pos < 2 { + if !self.fill().await? { + return Err(self.fail(Failure::Truncated)); + } + continue; + } + if self.buf[self.pos..].starts_with(b"--") { + self.state = State::Closed; + return Ok(None); + } + return self.part_headers().await.map(Some); + } + } + } + } + + /// Past `CRLF`, the header block up to the blank line. + async fn part_headers(&mut self) -> Result { + // RFC 2046 permits transport padding (LWSP) before the CRLF. + let head_end = loop { + let start = self.pos; + let lwsp = self.buf[start..] + .iter() + .take_while(|&&b| b == b' ' || b == b'\t') + .count(); + if self.buf.len() >= start + lwsp + 2 { + if &self.buf[start + lwsp..start + lwsp + 2] != b"\r\n" { + return Err(self.fail(Failure::Malformed("malformed delimiter line".into()))); + } + // Search from the delimiter line's own CRLF, so a part with NO + // headers (CRLF CRLF straight away) is found too. + if let Some(i) = find(&self.buf, b"\r\n\r\n", start + lwsp) { + break (start + lwsp + 2, i); + } + } + if self.buf.len() - start > HEADER_CAP { + return Err(self.fail(Failure::Malformed("part headers too long".into()))); + } + if !self.fill().await? { + return Err(self.fail(Failure::Truncated)); + } + }; + let (from, to) = head_end; + let text = std::str::from_utf8(&self.buf[from.min(to)..to]) + .map_err(|_| Failure::Malformed("part headers are not UTF-8".into())); + let text = match text { + Ok(t) => t.to_owned(), + Err(e) => return Err(self.fail(e)), + }; + let mut head = PartHead::default(); + let mut named = false; + for line in text.split("\r\n") { + let Some((k, v)) = line.split_once(':') else { + continue; + }; + if k.trim().eq_ignore_ascii_case("content-disposition") { + if let Some(n) = disposition_param(v, "name") { + head.name = n; + named = true; + } + head.filename = disposition_param(v, "filename"); + } else if k.trim().eq_ignore_ascii_case("content-type") { + head.content_type = Some(v.trim().to_owned()); + } + } + if !named { + return Err(self.fail(Failure::Malformed("a part has no `name`".into()))); + } + self.pos = to + 4; + self.clean_until = self.pos; + self.state = State::InPart; + Ok(head) + } + + /// Copy the current part's next bytes into `out`. `Ok(0)` at the part's + /// end, with the reader then just past the delimiter. + fn poll_part_data( + &mut self, + cx: &mut Context<'_>, + out: &mut [u8], + ) -> Poll> { + if self.state != State::InPart { + return Poll::Ready(Ok(0)); + } + loop { + let (avail, at_end) = match self.next_delim() { + Some(i) => (i - self.pos, i == self.pos), + None => (self.clean_until - self.pos, false), + }; + if at_end { + self.pos += self.delim.len(); + self.clean_until = self.pos; + self.state = State::AfterDelimiter; + return Poll::Ready(Ok(0)); + } + if avail > 0 { + let n = avail.min(out.len()); + out[..n].copy_from_slice(&self.buf[self.pos..self.pos + n]); + self.pos += n; + return Poll::Ready(Ok(n)); + } + match self.poll_fill(cx) { + Poll::Pending => return Poll::Pending, + Poll::Ready(Err(e)) => return Poll::Ready(Err(e)), + Poll::Ready(Ok(true)) => {} + Poll::Ready(Ok(false)) => return Poll::Ready(Err(self.fail(Failure::Truncated))), + } + } + } + + /// The current part's bytes, whole, refused above `cap`. For FIELDS (and + /// for a file whose uploader declared no `size` — bounded by the caller's + /// cap, which is the whole-read cap). + pub async fn read_part(&mut self, cap: usize) -> Result, ReadPartError> { + let mut out = Vec::new(); + let mut scratch = vec![0u8; 64 * 1024]; + loop { + let n = std::future::poll_fn(|cx| self.poll_part_data(cx, &mut scratch)) + .await + .map_err(ReadPartError::Failed)?; + if n == 0 { + return Ok(out); + } + if out.len() + n > cap { + return Err(ReadPartError::AboveCap); + } + out.extend_from_slice(&scratch[..n]); + } + } + + /// **The file part's leading bytes, WITHOUT consuming them** — the + /// write gate's peek. Returns up to `n` bytes (fewer only when the part + /// is shorter); the reader then yields them again from the first byte. + /// This is "peek, then chain" with no second buffer: the peeked bytes are + /// simply not released from this reader's own buffer until the seal + /// reads them. + pub async fn peek(&mut self, n: usize) -> Result<&[u8], Failure> { + loop { + if self.state != State::InPart { + return Ok(&[]); + } + let found = find(&self.buf, &self.delim, self.pos); + if let Some(i) = found { + let end = i.min(self.pos + n); + return Ok(&self.buf[self.pos..end]); + } + if self.buf.len() - self.pos >= n + self.delim.len() - 1 { + return Ok(&self.buf[self.pos..self.pos + n]); + } + if !self.fill().await? { + return Err(self.fail(Failure::Truncated)); + } + } + } + + /// The FILE part's reader step: its bytes, then — before reporting the + /// end — proof that the body closes here (see "why the file comes + /// LAST"). `Ok(0)` only after the closing delimiter is seen. + fn poll_file(&mut self, cx: &mut Context<'_>, out: &mut [u8]) -> Poll> { + match self.state { + State::Closed => return Poll::Ready(Ok(0)), + State::InPart => match self.poll_part_data(cx, out) { + Poll::Ready(Ok(0)) => {} + other => return other, + }, + State::AfterDelimiter => {} + State::Preamble => { + return Poll::Ready(Err(self.fail(Failure::Malformed("no part is open".into())))) + } + } + loop { + if self.buf.len() - self.pos >= 2 { + if self.buf[self.pos..].starts_with(b"--") { + self.state = State::Closed; + return Poll::Ready(Ok(0)); + } + return Poll::Ready(Err(self.fail(Failure::FieldAfterFile))); + } + match self.poll_fill(cx) { + Poll::Pending => return Poll::Pending, + Poll::Ready(Err(e)) => return Poll::Ready(Err(e)), + Poll::Ready(Ok(true)) => {} + Poll::Ready(Ok(false)) => return Poll::Ready(Err(self.fail(Failure::Truncated))), + } + } + } +} + +/// [`Multipart::read_part`]'s refusal. +#[derive(Debug)] +pub enum ReadPartError { + /// The part is longer than the caller's cap. + AboveCap, + /// The body could not be read. + Failed(Failure), +} + +/// The FILE part as a reader, for `files::publish_stream`. Every error is +/// also kept on the [`Multipart`] ([`Multipart::failure`]) so the handler can +/// answer the typed cause rather than edge's rendering of it. +impl tokio::io::AsyncRead for Multipart { + fn poll_read( + self: Pin<&mut Self>, + cx: &mut Context<'_>, + buf: &mut tokio::io::ReadBuf<'_>, + ) -> Poll> { + let this = self.get_mut(); + let out = buf.initialize_unfilled(); + match this.poll_file(cx, out) { + Poll::Pending => Poll::Pending, + Poll::Ready(Ok(n)) => { + buf.advance(n); + Poll::Ready(Ok(())) + } + Poll::Ready(Err(f)) => Poll::Ready(Err(std::io::Error::other(f.to_string()))), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use tokio::io::AsyncReadExt as _; + + /// `body` cut into frames of `frame` bytes — so every delimiter, header + /// terminator and field straddles frame boundaries somewhere. + fn framed(body: &[u8], frame: usize) -> ByteStream { + let frames: Vec> = body + .chunks(frame) + .map(|c| Ok(Bytes::copy_from_slice(c))) + .collect(); + Box::pin(futures_util::stream::iter(frames)) + } + + fn form(fields: &[(&str, &str)], file: &[u8], trailing: Option<(&str, &str)>) -> Vec { + let mut b = b"preamble to ignore\r\n".to_vec(); + for (k, v) in fields { + b.extend_from_slice( + format!("--XyZ\r\nContent-Disposition: form-data; name=\"{k}\"\r\n\r\n{v}\r\n") + .as_bytes(), + ); + } + b.extend_from_slice( + b"--XyZ\r\nContent-Disposition: form-data; name=\"file\"; filename=\"boat.jpg\"\r\n\ +Content-Type: image/jpeg\r\n\r\n", + ); + b.extend_from_slice(file); + if let Some((k, v)) = trailing { + b.extend_from_slice( + format!("\r\n--XyZ\r\nContent-Disposition: form-data; name=\"{k}\"\r\n\r\n{v}") + .as_bytes(), + ); + } + b.extend_from_slice(b"\r\n--XyZ--\r\nepilogue"); + b + } + + #[test] + fn the_boundary_parameter_is_read_quoted_or_not() { + assert_eq!( + boundary("multipart/form-data; boundary=XyZ").as_deref(), + Some("XyZ") + ); + assert_eq!( + boundary("multipart/form-data; boundary=\"XyZ\"").as_deref(), + Some("XyZ") + ); + assert_eq!(boundary("multipart/form-data"), None); + } + + /// Fields then a file, at every frame size from 1 byte up: the fields + /// read whole, the peek sees the head without consuming it, and the file + /// reads back exactly — a CRLF and a near-delimiter INSIDE the bytes + /// included. + #[tokio::test] + async fn fields_then_a_streamed_file_at_every_frame_size() { + let mut file = b"\xFF\xD8\xFF\x00\x01\r\n\x02\r\n--Xy not the end\r\n-".to_vec(); + file.extend((0..5000u32).map(|i| (i % 251) as u8)); + let body = form(&[("cohort", "self"), ("size", "5031")], &file, None); + for frame in [1, 2, 3, 7, 64, 1000, body.len()] { + let mut mp = Multipart::new(framed(&body, frame), "XyZ", 1 << 20); + let h = mp.next_part().await.expect("part").expect("cohort"); + assert_eq!(h.name, "cohort"); + assert_eq!(mp.read_part(FIELD_CAP).await.expect("field"), b"self"); + let h = mp.next_part().await.expect("part").expect("size"); + assert_eq!(h.name, "size"); + assert_eq!(mp.read_part(FIELD_CAP).await.expect("field"), b"5031"); + let h = mp.next_part().await.expect("part").expect("file"); + assert_eq!(h.name, "file"); + assert_eq!(h.filename.as_deref(), Some("boat.jpg")); + assert_eq!(h.content_type.as_deref(), Some("image/jpeg")); + assert_eq!(mp.peek(3).await.expect("peek"), b"\xFF\xD8\xFF"); + let mut got = Vec::new(); + mp.read_to_end(&mut got).await.expect("file"); + assert_eq!(got, file, "frame {frame}"); + assert!(mp.next_part().await.expect("closed").is_none()); + } + } + + #[tokio::test] + async fn a_field_after_the_file_is_refused_before_the_end_of_file() { + let body = form( + &[("cohort", "self")], + b"abc", + Some(("filename", "late.txt")), + ); + let mut mp = Multipart::new(framed(&body, 5), "XyZ", 1 << 20); + mp.next_part().await.unwrap(); + mp.read_part(FIELD_CAP).await.unwrap(); + mp.next_part().await.unwrap(); + let mut got = Vec::new(); + assert!(mp.read_to_end(&mut got).await.is_err()); + assert_eq!(got, b"abc", "the bytes arrive; the END is what is refused"); + assert_eq!(mp.failure(), Some(&Failure::FieldAfterFile)); + } + + #[tokio::test] + async fn a_body_that_stops_mid_file_is_truncated_and_a_long_one_too_large() { + let body = form(&[("cohort", "self")], b"abcdef", None); + let cut = &body[..body.len() - 20]; + let mut mp = Multipart::new(framed(cut, 4), "XyZ", 1 << 20); + mp.next_part().await.unwrap(); + mp.read_part(FIELD_CAP).await.unwrap(); + mp.next_part().await.unwrap(); + let mut got = Vec::new(); + assert!(mp.read_to_end(&mut got).await.is_err()); + assert_eq!(mp.failure(), Some(&Failure::Truncated)); + + let mut mp = Multipart::new(framed(&body, 4), "XyZ", 40); + let err = async { + mp.next_part().await?; + mp.read_part(FIELD_CAP).await.map_err(|e| match e { + ReadPartError::Failed(f) => f, + ReadPartError::AboveCap => Failure::Malformed("cap".into()), + })?; + mp.next_part().await + } + .await; + assert_eq!(err.unwrap_err(), Failure::TooLarge { limit: 40 }); + } + + #[tokio::test] + async fn no_boundary_is_malformed() { + let mut mp = Multipart::new(framed(b"no boundary here", 3), "XyZ", 1 << 20); + assert!(matches!(mp.next_part().await, Err(Failure::Malformed(_)))); + } +} diff --git a/src/drive/streaming.rs b/src/drive/streaming.rs new file mode 100644 index 00000000..6dd6eaa4 --- /dev/null +++ b/src/drive/streaming.rs @@ -0,0 +1,334 @@ +//! **Reading a file without holding it** (0.5.218, edge v36.1.0 +//! CIRISEdge#737 / #744) — the drive's side of edge's two streaming read +//! doors, `FileRow::chunks()` and `FileRow::open_range`, and the reader that +//! feeds one file's chunks into another seal (`move`). +//! +//! # Why a task and a channel, not a borrowed stream +//! +//! `FileRow::chunks()` borrows the row, the content store and the viewer key +//! for its whole walk, and an HTTP body must be `'static` — it outlives the +//! handler that built it. A self-referential struct would square that; a +//! producer task that OWNS all three and hands chunks through a bounded +//! channel squares it with nothing clever. The bound is the memory claim: +//! [`IN_FLIGHT`] chunks queued plus the one in the producer's hand plus the +//! one hyper is writing — each at most persist's 1 MiB inline cap (edge: +//! "every `Ok` item is at most persist's inline cap long"), so a 2 GiB file +//! costs a few MiB to serve. A client that disconnects drops the receiver; +//! the producer's next `send` fails and the walk stops there, so an abandoned +//! download does not keep decrypting. +//! +//! # The first item is awaited BEFORE the status line +//! +//! Once headers are sent, a refusal can only be a torn connection. So the +//! handler awaits the first item itself ([`first_item`]) and maps a refusal on +//! it to a status (`not_granted`, `range_not_satisfiable`, …) — the byte-state +//! probe has already vouched for the rest, and a failure after the first +//! chunk is a substrate fault mid-transfer, which a torn body reports +//! honestly (a client comparing `Content-Length` sees the short read). + +use std::pin::Pin; +use std::sync::Arc; +use std::task::{Context, Poll}; + +use axum::body::{Body, Bytes}; +use ciris_edge::files::{self, FileError, FileRow}; +use ciris_persist::prelude::Engine; +use tokio::sync::mpsc; + +/// How many chunks may be queued ahead of the writer — the whole of this +/// module's buffering. Two keeps the decrypt one step ahead of the socket +/// without letting it run away from a slow client. +pub const IN_FLIGHT: usize = 2; + +/// The window a RANGE is walked in: edge's [`files::STREAM_WINDOW_BYTES`] +/// (persist's 1 MiB inline cap), a multiple of edge's 256 KiB producer +/// chunk, so windows aligned to it open whole chunks and never the same +/// chunk twice. The first window runs from the range's start to the next +/// multiple, every later one is aligned. +pub const RANGE_WINDOW: u64 = files::STREAM_WINDOW_BYTES; + +type Item = Result, FileError>; + +fn store(engine: &Arc) -> ciris_edge::group_content::PersistGroupContentStore { + ciris_edge::group_content::PersistGroupContentStore::new( + (**engine).clone(), + engine.federation_directory(), + ) +} + +/// How long a streamed read waits, without progress, for a chunk whose key +/// grant has not arrived yet before it gives up. +/// +/// A pulled chunk DAG is PROMOTED when every chunk's bytes are held, but each +/// chunk's key grant replicates on its own afterwards, in seq order, at about +/// 2.4 chunks/s (CIRISEdge#779, measured on the 256 MiB native run: chunk +/// 781's wrap arrived 18 s before the read that died on chunk 782). A +/// `NotGranted` on a just-pulled DAG is therefore usually "not yet", not "no": +/// CIRISEdge#772 records that NotGranted is not terminal. So the walk waits and +/// retries the same offset — progress resets the clock — and only a grant that +/// stays missing this long ends the stream (the client sees a body shorter than +/// its Content-Length, never wrong bytes). Edge's #779 gates the reported state +/// on grants; this is the reader's belt until then. +pub const GRANT_WAIT: std::time::Duration = std::time::Duration::from_secs(120); + +/// Between retries of a chunk whose grant has not arrived. +const GRANT_RETRY: std::time::Duration = std::time::Duration::from_secs(2); + +fn grant_pending(e: &FileError) -> bool { + matches!( + e, + FileError::Unopened(ciris_edge::chat::UnopenedReason::NotGranted { .. }) + ) +} + +/// The whole file, one chunk per item, through `FileRow::chunks()`. A chunk +/// refused `NotGranted` hands the rest of the walk to the range reader from +/// the first byte not yet sent, which waits for the grant ([`GRANT_WAIT`]). +pub fn spawn_chunks(engine: Arc, file: FileRow, viewer: String) -> mpsc::Receiver { + let (tx, rx) = mpsc::channel(IN_FLIGHT); + tokio::spawn(async move { + let content = store(&engine); + let mut sent: u64 = 0; + { + let mut chunks = file.chunks(&content, &viewer); + while let Some(item) = chunks.next().await { + match item { + Ok(bytes) => { + sent += bytes.len() as u64; + if tx.send(Ok(bytes)).await.is_err() { + return; + } + } + Err(e) if grant_pending(&e) => break, + Err(e) => { + let _ = tx.send(Err(e)).await; + return; + } + } + } + } + let Some(total) = file.pointer.size else { + return; + }; + if sent >= total { + return; + } + tracing::info!( + attestation_id = %file.attestation_id, sent, total, + "drive: a chunk's key grant has not arrived yet — waiting for it (CIRISEdge#779) \ + and continuing from the first unsent byte" + ); + walk_range(&content, &file, &viewer, sent, total - 1, &tx).await; + }); + rx +} + +/// `[start, end]` in [`RANGE_WINDOW`] windows, each retried while its grant is +/// pending (up to [`GRANT_WAIT`] without progress). +async fn walk_range( + content: &ciris_edge::group_content::PersistGroupContentStore, + file: &FileRow, + viewer: &str, + start: u64, + end: u64, + tx: &mpsc::Sender, +) { + let mut at = start; + let mut stalled_since: Option = None; + while at <= end { + let window_end = ((at / RANGE_WINDOW) + 1) * RANGE_WINDOW - 1; + let last = window_end.min(end); + match file.open_range(content, viewer, at, last - at + 1).await { + Ok(bytes) => { + stalled_since = None; + if tx.send(Ok(bytes)).await.is_err() { + return; + } + at = last + 1; + } + Err(e) if grant_pending(&e) => { + let since = *stalled_since.get_or_insert_with(tokio::time::Instant::now); + if since.elapsed() >= GRANT_WAIT { + let _ = tx.send(Err(e)).await; + return; + } + tokio::time::sleep(GRANT_RETRY).await; + } + Err(e) => { + let _ = tx.send(Err(e)).await; + return; + } + } + } +} + +/// `[start, end]` inclusive, in [`RANGE_WINDOW`] windows through +/// `FileRow::open_range` — no cap on the range's length, because no window +/// is longer than one. +/// +/// Edge's `open_range` refuses a window above the 64 MiB whole-read cap +/// (`AboveWholeReadCap`, "the bound on what one call materializes"); that is +/// a bound on ONE CALL, and this walk never makes a call anywhere near it. +pub fn spawn_range( + engine: Arc, + file: FileRow, + viewer: String, + start: u64, + end: u64, +) -> mpsc::Receiver { + let (tx, rx) = mpsc::channel(IN_FLIGHT); + tokio::spawn(async move { + let content = store(&engine); + walk_range(&content, &file, &viewer, start, end, &tx).await; + }); + rx +} + +/// The first item, or `None` for a file that yielded nothing. +pub async fn first_item(rx: &mut mpsc::Receiver) -> Option { + rx.recv().await +} + +/// The rest of the walk as a response body, `first` in front. If `expect` +/// is given (the size a `Content-Length` promised) a walk that ends at any +/// other count is an ERROR on the body, never a clean short end: the same +/// "row and bytes disagree" the whole read refuses as `seal_mismatch`, +/// surfaced the only way it can be once the status line is out. +pub fn body(first: Vec, rx: mpsc::Receiver, expect: Option) -> Body { + struct Walk { + first: Option>, + rx: mpsc::Receiver, + sent: u64, + expect: Option, + } + let walk = Walk { + sent: 0, + first: Some(first), + rx, + expect, + }; + let stream = futures_util::stream::unfold(walk, |mut w| async move { + let item = match w.first.take() { + Some(b) => Some(Ok(b)), + None => w.rx.recv().await, + }; + match item { + Some(Ok(b)) => { + w.sent += b.len() as u64; + if w.expect.is_some_and(|n| w.sent > n) { + let e = std::io::Error::other("the file yielded more bytes than its size"); + return Some((Err(e), w)); + } + Some((Ok(Bytes::from(b)), w)) + } + Some(Err(e)) => { + tracing::warn!(error = %e, "drive: a streamed read failed mid-transfer"); + // After an Err the producer has stopped; end after this. + w.expect = None; + Some((Err(std::io::Error::other(e.to_string())), w)) + } + None => match w.expect.take() { + Some(n) if n != w.sent => Some(( + Err(std::io::Error::other(format!( + "the file yielded {} bytes and its row declares {n}", + w.sent + ))), + w, + )), + _ => None, + }, + } + }); + Body::from_stream(stream) +} + +/// One file's chunks as an `AsyncRead` — `move`'s source for +/// `files::publish_stream` above the whole-read cap: the walk of the source +/// room's seal feeding the seal of the target room, one chunk in hand. +/// +/// [`Self::peek`] fills the write gate's head window without consuming it; +/// a walk refusal is kept ([`Self::failure`]) so the handler answers the +/// typed read refusal rather than edge's `FileError::Read` prose. +pub struct ChunkReader { + rx: mpsc::Receiver, + pending: Vec, + at: usize, + done: bool, + failure: Option, +} + +impl ChunkReader { + /// Over a walk from [`spawn_chunks`]. + pub fn new(rx: mpsc::Receiver) -> Self { + Self { + rx, + pending: Vec::new(), + at: 0, + done: false, + failure: None, + } + } + + /// The walk's refusal, if it stopped on one. + pub fn failure(&self) -> Option<&FileError> { + self.failure.as_ref() + } + + /// Up to `n` leading bytes, not consumed — fewer only for a shorter + /// file. At most `n` plus one chunk is held. + pub async fn peek(&mut self, n: usize) -> Result<&[u8], FileError> { + while self.pending.len() - self.at < n && !self.done { + match self.rx.recv().await { + Some(Ok(b)) => { + self.pending.drain(..self.at); + self.at = 0; + self.pending.extend_from_slice(&b); + } + Some(Err(e)) => { + self.done = true; + self.failure = Some(e.clone()); + return Err(e); + } + None => self.done = true, + } + } + let end = (self.at + n).min(self.pending.len()); + Ok(&self.pending[self.at..end]) + } +} + +impl tokio::io::AsyncRead for ChunkReader { + fn poll_read( + self: Pin<&mut Self>, + cx: &mut Context<'_>, + buf: &mut tokio::io::ReadBuf<'_>, + ) -> Poll> { + let this = self.get_mut(); + loop { + if this.at < this.pending.len() { + let n = (this.pending.len() - this.at).min(buf.remaining()); + buf.put_slice(&this.pending[this.at..this.at + n]); + this.at += n; + return Poll::Ready(Ok(())); + } + if let Some(e) = &this.failure { + return Poll::Ready(Err(std::io::Error::other(e.to_string()))); + } + if this.done { + return Poll::Ready(Ok(())); + } + match this.rx.poll_recv(cx) { + Poll::Pending => return Poll::Pending, + Poll::Ready(None) => this.done = true, + Poll::Ready(Some(Ok(b))) => { + this.pending = b; + this.at = 0; + } + Poll::Ready(Some(Err(e))) => { + this.done = true; + this.failure = Some(e); + } + } + } + } +} diff --git a/src/family_api.rs b/src/family_api.rs index f140de9e..4a43875d 100644 --- a/src/family_api.rs +++ b/src/family_api.rs @@ -33,6 +33,33 @@ //! 5. **Leaving is always your own act**, never subject to quorum. //! 6. **A non-member cannot find out a family exists**: every route answers a //! non-member exactly as it answers an unknown id, `family.not_found` (404). +//! 7. **Nobody joins without their own consent** (0.5.218; the maintainer's +//! ruling of 2026-09-30, CIRISConstitution#133, `FSD/MEMBERSHIP_INVITES.md`). +//! Until persist can carry a proposal to the joiner and record their +//! acceptance (CIRISPersist#955, v52), every roster-growing door answers +//! 409 `membership.consent_required`: `POST …/members`, a quorum envelope / +//! cosign / assemble whose proposed roster adds anyone, and a create whose +//! founding roster names anyone but the founder +//! (`membership_consent_required`, removed at persist v52). Remove, role, leave and dissolve are +//! unaffected, and re-adding someone already active keeps +//! `family.already_member`. +//! +//! **At persist v52.0.0 / edge v38.0.0 the door opens as the invite flow** +//! (`crate::membership_invites`): `POST …/invites` (and `POST …/members`, +//! its alias) writes a `membership:proposal:v1` signed by ONE inviter — a +//! founder under `founder_only`, any member under a quorum — and answers +//! 202 `{state: "invited", proposal_id}`; the invitee accepts on their own +//! node with their own pen (`POST /v1/self/invites/{p}/accept`); the +//! WIDENING that seats them is persist-gated on that acceptance. Under +//! `founder_only` the widening is the founder's single signature — written +//! by edge's bridge on the acceptance's arrival (`membership_widener`) or +//! here when a member lists the invites; under `quorum:M/N` it is the +//! `add` of envelope → cosign → assemble, now a co-signed WIDENING (persist +//! Q2: a supersede never adds, `membership_supersede_cannot_add`). A +//! founding roster naming anyone but the founder is refused +//! `membership.founding_member_unsigned` (persist Q1: signing the founding +//! record is consent, and this surface has no founding-cosign flow). The +//! interim `membership.consent_required` is gone. //! //! # What persist admits at these pins (v48.0.0), and what it does not //! @@ -136,8 +163,7 @@ fn not_found() -> Response { refuse( StatusCode::NOT_FOUND, "family.not_found", - "no family with that id is visible to you — it does not exist, or you are not one of \ - its members", + "That household isn't visible to you. It doesn't exist, or you aren't in it.", ) } @@ -145,7 +171,7 @@ fn not_authorized(detail: String) -> Response { refuse_with( StatusCode::FORBIDDEN, "family.not_authorized", - "this family's consensus protocol is not satisfied for that change", + "This household's rule doesn't let you make that change.", detail, ) } @@ -154,8 +180,7 @@ fn quorum_pending(detail: String) -> Response { refuse_with( StatusCode::CONFLICT, "family.quorum_pending", - "this family's protocol needs more members' signatures before the change can be applied \ - — collect them through the change envelope, cosign and assemble routes", + "This household needs more members to sign before that can happen.", detail, ) } @@ -164,8 +189,7 @@ fn bad_protocol(detail: String) -> Response { refuse_with( StatusCode::BAD_REQUEST, "family.bad_consensus_protocol", - "that consensus protocol is not one a household can use: founder_only, majority, \ - unanimous or quorum:M/N with a strict majority over the whole roster", + "That isn't a rule a household can use.", detail, ) } @@ -174,7 +198,7 @@ fn unknown_member_key(key_id: &str) -> Response { refuse_with( StatusCode::BAD_REQUEST, "family.unknown_member_key", - "that key is not a registered identity on this node, so it cannot be a family member", + "This node doesn't know that person yet, so they can't join.", key_id.to_owned(), ) } @@ -183,7 +207,7 @@ fn already_member(key_id: &str) -> Response { refuse_with( StatusCode::CONFLICT, "family.already_member", - "that identity is already a member of this family", + "They're already in this household.", key_id.to_owned(), ) } @@ -192,7 +216,7 @@ fn not_a_member(key_id: &str) -> Response { refuse_with( StatusCode::NOT_FOUND, "family.not_a_member", - "that identity is not a current member of this family", + "They aren't in this household any more.", key_id.to_owned(), ) } @@ -201,8 +225,7 @@ fn last_founder() -> Response { refuse( StatusCode::CONFLICT, "family.last_founder", - "you are the family's last founder and other members remain — make another member a \ - founder first, or dissolve the family", + "You're the last founder. Make someone else a founder first, or dissolve the household.", ) } @@ -210,7 +233,7 @@ fn bad_role() -> Response { refuse( StatusCode::BAD_REQUEST, "family.bad_role", - "a role must be a short non-empty name such as founder or member", + "A role is a short name, like founder or member.", ) } @@ -218,8 +241,7 @@ fn bad_change(detail: String) -> Response { refuse_with( StatusCode::CONFLICT, "family.bad_change", - "that change envelope does not describe a change to this family as it stands now — \ - build a fresh envelope", + "The household changed after this was proposed. Propose it again.", detail, ) } @@ -228,7 +250,7 @@ fn bad_request(detail: String) -> Response { refuse_with( StatusCode::BAD_REQUEST, "family.bad_request", - "the request body is not a valid family request", + "The node couldn't read that request.", detail, ) } @@ -237,7 +259,7 @@ fn store_unavailable(detail: String) -> Response { refuse_with( StatusCode::SERVICE_UNAVAILABLE, "family.store_unavailable", - "the family store could not be read or written", + "The node couldn't read or write its household records.", detail, ) } @@ -246,7 +268,7 @@ fn signer_unavailable(detail: String) -> Response { refuse_with( StatusCode::FORBIDDEN, "family.author_signer_unavailable", - "your federation identity could not be opened to sign this family change", + "Your identity couldn't be opened to sign this.", detail, ) } @@ -255,7 +277,7 @@ fn session_required(code: StatusCode) -> Response { refuse( code, "family.owner_session_required", - "families are the node owner's own surface — sign in as the owner of this claimed node", + "Households belong to this node's owner. Sign in as the owner.", ) } @@ -263,8 +285,7 @@ fn delegate_may_not_author() -> Response { refuse( StatusCode::FORBIDDEN, "family.delegate_may_not_author", - "a delegated session may read families but may not change one — a roster row is signed \ - with the owner's own key, and that signature would outlive the delegation", + "A delegated session can look at households but can't change them.", ) } @@ -570,6 +591,7 @@ async fn sign_family(capsule: &OwnerSignerCapsule, family: Family) -> Result Vec { let target = env.get("target_key_id").and_then(|v| v.as_str()); match (action, target) { + // persist v52 Q2: an add is a co-signed WIDENING at the offered role. + ("add", Some(k)) => { + let role = env + .get("roles") + .and_then(|r| r.get(k)) + .and_then(|v| v.as_str()) + .unwrap_or(ROLE_MEMBER) + .to_owned(); + vec![FamilyRow::Widening { + member: k.to_owned(), + role, + }] + } ("remove", Some(k)) => vec![FamilyRow::Revocation { member: k.to_owned(), reason: "removed", @@ -719,13 +754,13 @@ fn family_rows(loaded: &Loaded, action: &str, env: &serde_json::Value) -> Vec dissolve_order(loaded) - .into_iter() - .map(|member| FamilyRow::Revocation { - member, - reason: "dissolved", - }) - .collect(), + // A quorum dissolve writes no rows since persist v52 (#956): the + // terminal amendment alone empties the fold. `loaded` stays a + // parameter for the shapes that read the roster. + ("dissolve", _) => { + let _ = loaded; + Vec::new() + } _ => Vec::new(), } } @@ -851,9 +886,30 @@ async fn create_family(State(st): State, headers: HeaderMap, body: return refuse( StatusCode::BAD_REQUEST, "family.name_empty", - "a family needs a name of at most 200 characters", + "A household needs a name of up to 200 characters.", ); } + // CONSENT TO JOIN (CIRISConstitution#133; persist v52 Q1): a founding + // record seats exactly the members who SIGNED it, and this route signs with + // the founder alone — so a founding roster that names anyone else is + // REFUSED, not trimmed (a silently smaller household would read as success + // to the caller who named them), by persist's own rule name. They join by + // invitation once the household exists (`POST …/invites`). The loop below + // then only ever sees the founder named twice, which stays `already_member`. + let joining: Vec = req + .members + .iter() + .filter(|k| **k != caller.owner_key_id) + .cloned() + .collect(); + if !joining.is_empty() { + return crate::membership_invites::founding_member_unsigned(format!( + "POST /v1/families: {} would be founding members without signing the founding \ + record (membership_founding_member_unsigned) — found the household alone, then \ + POST /v1/families/{{id}}/invites for each", + joining.join(", ") + )); + } let dir = st.engine.federation_directory(); let mut others: Vec = Vec::new(); for k in &req.members { @@ -893,6 +949,7 @@ async fn create_family(State(st): State, headers: HeaderMap, body: founded_at: at, consensus_protocol: protocol, consensus_protocol_entrenched: false, + dissolved_at: None, persist_row_hash: String::new(), }; let signed = match sign_family(&capsule, family).await { @@ -1032,9 +1089,36 @@ struct AddRequest { key_id: String, #[serde(default)] role: Option, + /// How long the invitation lives (1..=30 days; default + /// [`crate::membership_invites::DEFAULT_INVITE_DAYS`]). + #[serde(default)] + expires_in_days: Option, } +/// `POST /v1/families/{id}/members` — an ALIAS for `…/invites` since persist +/// v52 (`FSD/MEMBERSHIP_INVITES.md` §3). A direct add was admission on the +/// household's authority alone: a founder could enrol any key they could +/// name, and that person's node began receiving the household's rows and +/// wraps. The route now answers 202 `{state: "invited", proposal_id}`, so no +/// caller mistakes an invitation for a membership. async fn add_member( + st: State, + headers: HeaderMap, + id: Path, + body: Bytes, +) -> Response { + invite(st, headers, id, body).await +} + +/// `POST /v1/families/{id}/invites` — invite `key_id` at `role`. +/// +/// ONE inviter signs (persist FSD §4: "the quorum stays on the growth record, +/// not the invitation"): a founder under `founder_only` — persist refuses any +/// other proposer there, and the check is repeated here only so the refusal is +/// the household's own `family.not_authorized` rather than prose — and any +/// active member under a quorum. The invitee must be a registered identity +/// (their key is what the widening will name) and not already active. +async fn invite( State(st): State, headers: HeaderMap, Path(id): Path, @@ -1052,47 +1136,158 @@ async fn add_member( if !role_ok(&role) { return bad_role(); } - if let Protocol::Quorum { m, n } = protocol { - return needs_quorum(m, n); - } - if let Err(r) = require_founder(&loaded, &caller.owner_key_id) { - return r; + if protocol == Protocol::FounderOnly { + if let Err(r) = require_founder(&loaded, &caller.owner_key_id) { + return r; + } } if let Err(r) = check_addable(&st.engine, &loaded, &req.key_id).await { return r; } + let expires_at = match crate::membership_invites::expiry(req.expires_in_days) { + Ok(t) => t, + Err(r) => return r, + }; let capsule = match pen(&st, &caller).await { Ok(c) => c, Err(r) => return r, }; - let member = FamilyMember { - key_id: req.key_id.clone(), - joined_at: now(), - role: Some(role.clone()), + let dir = st.engine.federation_directory(); + match ciris_edge::membership::propose( + dir.as_ref(), + ciris_edge::membership::GroupScope::Family, + &id, + &req.key_id, + Some(&role), + expires_at, + capsule.edge_signer(), + ) + .await + { + Ok(proposal) => { + tracing::info!( + family = %id, invitee = %req.key_id, proposal = %proposal.attestation_id, + "family: invitation sent — the invitee joins only on their own acceptance" + ); + kick("family:invite"); + crate::membership_invites::invited( + ciris_edge::membership::GroupScope::Family, + &id, + &proposal, + &req.key_id, + Some(&role), + ) + } + Err(e) => crate::membership_invites::refused(&e), + } +} + +/// `GET /v1/families/{id}/invites` — every invitation into the household and +/// its state (`pending` / `accepted` / `joined` / `declined` / `expired` / +/// `withdrawn`). Members only (a non-member gets `family.not_found`, as for +/// every route); a delegate may read. +/// +/// Reading is also when an ACCEPTED invitee of a `founder_only` household is +/// seated, if edge's bridge has not already done it (`compose`'s +/// `membership_widener` block names the window): a founder's own session, its +/// own pen, edge's `widen_on_acceptance` — then the household's existing +/// content is re-wrapped to them, exactly as the pre-v52 direct add did. +async fn list_invites( + State(st): State, + headers: HeaderMap, + Path(id): Path, +) -> Response { + let caller = match gate(owner_caller(&st.engine, &headers, true).await) { + Ok(c) => c, + Err(r) => return r, }; - // persist v49.0.0 (#910): the addition is its own signed row, a - // `FamilyMembershipWidening`, and the authority signs THAT row's envelope. - // The record is no longer grown in place (a rewritten record reached no - // peer: `put_family` is INSERT-only), so the pre-v49 signature over the - // grown record no longer verifies at the door. - let spec = match sign_family_widening(&capsule, &id, &member).await { - Ok(s) => s, - Err(e) => return signer_unavailable(e), + let mut loaded = match load(&st.engine, &id, &caller.owner_key_id).await { + Ok(l) => l, + Err(r) => return r, }; - match st - .engine - .federation_directory() - .add_member(Cohort::Family, &id, RosterMember::from(member), &spec) + let dir = st.engine.federation_directory(); + let scope = ciris_edge::membership::GroupScope::Family; + let active = |l: &Loaded| -> std::collections::HashSet { + l.active.iter().map(|m| m.key_id.clone()).collect() + }; + let mut invites = + match crate::membership_invites::group_invites(dir.as_ref(), scope, &id, &active(&loaded)) + .await + { + Ok(v) => v, + Err(e) => return store_unavailable(e), + }; + let accepted = invites + .iter() + .any(|(v, _)| v.state == crate::membership_invites::STATE_ACCEPTED); + let founder_only = + Protocol::of(&loaded.family.consensus_protocol) == Some(Protocol::FounderOnly); + let mut seated: Vec = Vec::new(); + if accepted && founder_only && require_founder(&loaded, &caller.owner_key_id).is_ok() { + // A delegate's bearer cannot open the pen; that is a read, not an error. + if let Ok(capsule) = pen(&st, &caller).await { + let widener = ciris_edge::membership::MembershipWidener::new(vec![Arc::clone( + capsule.edge_signer(), + )]); + seated = + crate::membership_invites::widen_held_acceptances(dir.as_ref(), &invites, &widener) + .await; + } + } + let mut rewrapped = serde_json::Map::new(); + if !seated.is_empty() { + for k in &seated { + rewrapped.insert(k.clone(), rewrap(&st.engine, &id, k).await); + } + loaded = match load(&st.engine, &id, &caller.owner_key_id).await { + Ok(l) => l, + Err(r) => return r, + }; + invites = match crate::membership_invites::group_invites( + dir.as_ref(), + scope, + &id, + &active(&loaded), + ) .await - { - Ok(true) => {} - Ok(false) => return already_member(&req.key_id), - Err(e) => return store_unavailable(format!("add_member: {e:#}")), + { + Ok(v) => v, + Err(e) => return store_unavailable(e), + }; } - let dek = rewrap(&st.engine, &id, &req.key_id).await; - tracing::info!(family = %id, member = %req.key_id, %role, "family: member added"); - kick("family:add_member"); - respond_with_family(&st, &id, &caller, serde_json::json!({ "dek_rewrap": dek })).await + Json(serde_json::json!({ + "family_id": id, + "invites": invites.into_iter().map(|(v, _)| v).collect::>(), + "seated_now": seated, + "dek_rewrap": rewrapped, + })) + .into_response() +} + +/// `DELETE /v1/families/{id}/invites/{proposal_id}` — the proposer withdraws a +/// pending invitation (a `withdraws` of the proposal, signed by them). +async fn withdraw_invite( + State(st): State, + headers: HeaderMap, + Path((id, proposal_id)): Path<(String, String)>, +) -> Response { + let (caller, _loaded, _protocol) = match write_preamble(&st, &headers, &id).await { + Ok(t) => t, + Err(r) => return r, + }; + let capsule = match pen(&st, &caller).await { + Ok(c) => c, + Err(r) => return r, + }; + let dir = st.engine.federation_directory(); + crate::membership_invites::withdraw( + dir.as_ref(), + ciris_edge::membership::GroupScope::Family, + &id, + &proposal_id, + &capsule, + ) + .await } /// A target may join iff it is a registered identity and not already an @@ -1212,10 +1407,40 @@ async fn leave_inner( Ok(c) => c, Err(r) => return r, }; - // A quorum family's record must keep N == roster, or every later quorum - // check reads a seat that has left. Rewritten FIRST (by the leaver, whose - // own act this is); the revocation below is what replicates the departure. - if let Protocol::Quorum { m, n } = protocol { + // A quorum family's record must not keep a seat that has left. Rewritten + // FIRST (by the leaver, whose own act this is); the revocation below folds + // the departure into the roster planes. + // + // persist v52 (CIRISPersist#956) — THE SELF-LEAVE AMENDMENT, which the + // audit note below asked for, is what this now writes: an amendment whose + // ONLY change removes one member from the held record, admitted on THAT + // member's signature over the change envelope with no quorum, on the local + // door and on a peer's apply — so it REPLICATES. persist's shape, from its + // note to the server: `supersede_family_with_quorum(record minus the + // leaver, envelope from build_membership_change_envelope(..remaining..), + // signed by the leaver alone)`. "Every other seat must be identical and in + // order, and nothing else may change" — so the protocol is NOT rescaled + // here any more (that was a second change), and the envelope names the + // held roster as `supersedes.prior_member_key_ids` (persist's builder + // does). A member who joined by a WIDENING (persist v52 Q2: the record + // never grows) is not on the record, so there is nothing to amend and the + // revocation alone is their departure. + // + // The pre-v52 account, kept as the record of why this was stopped: + // + // THIS REWRITE DOES NOT REPLICATE (0.5.218 audit, fix 2 — stopped, not + // faked). It is the plain `supersede_family`, which persist v49 strips of + // any proof (#910.5), so a peer holding the household refuses the differing + // record (the #758 conflict) and keeps the leaver's seat in N; only the + // revocation row travels. `supersede_family_with_quorum` cannot carry it + // either: it evaluates the household's `quorum:M/N` over the change, and a + // leave is the member's OWN act, never put to a quorum (rule 5) — the + // leaver's lone signature is `consensus_protocol not met`. persist would + // need a self-leave amendment: a supersede whose ONLY roster delta removes + // its signer, admitted on that signer's signature alone (on the local door + // and in `route_occupied_family` on the peer). No proof is synthesised here. + let on_record = loaded.family.members.iter().any(|fm| fm.key_id == me); + if matches!(protocol, Protocol::Quorum { .. }) && on_record { let remaining: Vec = loaded .family .members @@ -1224,23 +1449,48 @@ async fn leave_inner( .cloned() .collect(); if !remaining.is_empty() { + let dir = st.engine.federation_directory(); + let keys: Vec = remaining.iter().map(|fm| fm.key_id.clone()).collect(); + let env = match dir + .build_membership_change_envelope( + Cohort::Family, + &id, + &keys, + false, + Some(&loaded.family.consensus_protocol), + ) + .await + { + Ok(v) => v, + Err(e) => { + return store_unavailable(format!( + "build_membership_change_envelope(leave): {e:#}" + )) + } + }; + let bytes = match ciris_verify_core::jcs::canonicalize(&env) { + Ok(b) => b, + Err(e) => return store_unavailable(format!("canonicalize the leave: {e}")), + }; + let sig = match capsule.sign_hybrid(&bytes).await { + Ok(s) => s, + Err(e) => return signer_unavailable(e), + }; + let mine = vec![ThresholdSignature { + member_id: sig.key_id.clone(), + ed25519_signature_base64: B64.encode(&sig.classical_signature), + mldsa65_signature_base64: Some(B64.encode(&sig.pqc_signature)), + }]; let mut next = loaded.family.clone(); - next.consensus_protocol = rescale(m, n, remaining.len()); next.members = remaining; let signed = match sign_family(&capsule, next).await { Ok(s) => s, Err(e) => return signer_unavailable(e), }; - if let Err(e) = st - .engine - .federation_directory() - .supersede_family( - signed, - Some(serde_json::json!({ "action": "leave", "member": me })), - ) - .await - { - return store_unavailable(format!("supersede_family(leave): {e:#}")); + if let Err(e) = dir.supersede_family_with_quorum(signed, env, mine).await { + return store_unavailable(format!( + "supersede_family_with_quorum(self-leave): {e:#}" + )); } } } @@ -1346,7 +1596,6 @@ async fn dissolve( &capsule, &loaded, serde_json::json!({ "action": "dissolve", "protocol": FOUNDER_ONLY }), - None, ) .await } @@ -1355,39 +1604,33 @@ async fn dissolve( /// removal per active member (the plane that REPLICATES — a peer holding the /// family learns it is empty), then the authority-signed supersede to an empty /// roster (the version history records who dissolved it and on what authority). +/// +/// **The empty record stays local** (0.5.218 audit, fix 2 — stopped, not +/// faked). The supersede is the plain door, which persist v49 strips of any +/// proof (#910.5), so a peer never applies the terminal record as an +/// amendment; the removal rows are what tell it the household is gone. The +/// quorum door was tried: routing a quorum dissolve through +/// `supersede_family_with_quorum` needs a change envelope proposing the EMPTY +/// roster the record carries, and persist refuses it structurally — +/// `verify_membership_quorum: membership change not authorized: malformed +/// accord family envelope: group has no members`. persist would need a +/// quorum-verified TERMINAL amendment (an explicit dissolve form the quorum +/// signs and `route_occupied_family` applies). `tests/family_crud.rs` +/// `a_quorum_dissolve_replicates_as_an_amendment` pins it, ignored. async fn terminal_dissolve( st: &FamilyState, caller: &OwnerCaller, capsule: &OwnerSignerCapsule, loaded: &Loaded, authorization: serde_json::Value, - quorum: Option<( - chrono::DateTime, - &[crate::roster_rows::ChangeSignature], - )>, ) -> Response { let id = loaded.family.family_key_id.clone(); - match quorum { - // A quorum dissolve: every revocation at the pinned instant, each - // carrying the other signers' co-signatures over that exact row. - Some((at, sigs)) => { - let primary = capsule.key_id().to_owned(); - for k in dissolve_order(loaded) { - let row = revocation_row(&id, &k, at, "dissolved", Vec::new()); - let cosigs = crate::roster_rows::cosignatures_for(sigs, &primary, "revocation", &k); - if let Err(e) = put_revocation_row(&st.engine, capsule, row, cosigs).await { - return store_unavailable(e); - } - } - } - // Founders last: persist refuses the last founder's removal while - // anyone else remains (`roster_last_founder`, v49.0.0). - None => { - for k in dissolve_order(loaded) { - if let Err(e) = write_revocation(&st.engine, capsule, &id, &k, "dissolved").await { - return store_unavailable(e); - } - } + // Founders last: persist refuses the last founder's removal while anyone + // else remains (`roster_last_founder`, v49.0.0). A QUORUM dissolve no + // longer comes here (persist v52, #956 — [`quorum_terminal_dissolve`]). + for k in dissolve_order(loaded) { + if let Err(e) = write_revocation(&st.engine, capsule, &id, &k, "dissolved").await { + return store_unavailable(e); } } let mut next = loaded.family.clone(); @@ -1409,6 +1652,63 @@ async fn terminal_dissolve( Json(serde_json::json!({ "family_id": id, "dissolved": true })).into_response() } +/// **A quorum household's dissolve — the TERMINAL AMENDMENT** (persist v52, +/// CIRISPersist#956; the shape persist's note gives the server verbatim): +/// `supersede_family_with_quorum(record with dissolved_at = t, the change +/// envelope built from the SAME members plus "dissolved_at": t, M-of-N +/// signatures)`. Nothing else on the record moves — name, founding instant, +/// protocol, entrenchment and every seat stay byte-identical — and `t` is the +/// instant the quorum signed inside the envelope (`DISSOLVED_AT`), never a +/// fresh clock read, or persist refuses the record as one "whose instant the +/// quorum did not sign". +/// +/// This is what `a_quorum_dissolve_replicates_as_an_amendment` waited on: +/// before v52 the terminal record went through the plain supersede, persist +/// stripped its proof (#910.5), and a peer kept the household live forever. +async fn quorum_terminal_dissolve( + st: &FamilyState, + caller: &OwnerCaller, + capsule: &OwnerSignerCapsule, + loaded: &Loaded, + env: &serde_json::Value, + thresholds: Vec, +) -> Response { + let id = loaded.family.family_key_id.clone(); + let Some(at) = env + .get(DISSOLVED_AT) + .and_then(|v| v.as_str()) + .and_then(|s| chrono::DateTime::parse_from_rfc3339(s).ok()) + .map(|t| t.with_timezone(&chrono::Utc)) + else { + return bad_change( + "a dissolve envelope must pin dissolved_at — rebuild it with \ + POST /v1/families/{id}/changes/envelope" + .to_owned(), + ); + }; + let mut terminal = loaded.family.clone(); + terminal.dissolved_at = Some(at); + let signed = match sign_family(capsule, terminal).await { + Ok(s) => s, + Err(e) => return signer_unavailable(e), + }; + if let Err(e) = st + .engine + .federation_directory() + .supersede_family_with_quorum(signed, env.clone(), thresholds) + .await + { + return quorum_refusal(&e, &loaded.family.consensus_protocol); + } + tracing::info!( + family = %id, by = %caller.owner_key_id, dissolved_at = %at, + "family: dissolved by a quorum terminal amendment (CIRISPersist#956) — it replicates" + ); + kick("family:dissolve"); + Json(serde_json::json!({ "family_id": id, "dissolved": true, "dissolved_at": at.to_rfc3339() })) + .into_response() +} + // ─── The quorum flow: envelope → cosign → assemble ────────────────────────── // // The accord's three steps (`/v1/accord/family/change/envelope` + supersede), @@ -1422,6 +1722,11 @@ async fn terminal_dissolve( const ACTIONS: &[&str] = &["add", "remove", "role", "dissolve"]; +/// The change-envelope member a quorum dissolve pins its terminal instant in +/// (persist v52, CIRISPersist#956: the amendment's `dissolved_at` must equal +/// the one the quorum signed inside the envelope). +const DISSOLVED_AT: &str = "dissolved_at"; + #[derive(Debug, Deserialize)] struct EnvelopeRequest { action: String, @@ -1490,6 +1795,14 @@ async fn change_envelope( if let Err(r) = check_addable(&st.engine, &loaded, k).await { return r; } + // CONSENT TO JOIN (CIRISConstitution#133 / CIRISPersist#955): M of + // N existing members cannot stand in for the joiner — and they no + // longer have to. The joiner's own acceptance of a live invitation + // is what persist checks when the co-signed WIDENING this change + // writes reaches its door (`check_growth_accepted`); without one + // the assemble is refused `membership.awaiting_acceptance`. The + // envelope names the grown roster so verify's quorum gate counts + // the change; the record itself never grows (persist Q2). let role = req.role.clone().unwrap_or_else(|| ROLE_MEMBER.to_owned()); if !role_ok(&role) { return bad_role(); @@ -1567,10 +1880,19 @@ async fn change_envelope( ); // The instant the change's rows carry (persist v49.0.0): every signer // signs the rows too, and a row's signed bytes include its instant. + let at = now(); obj.insert( "row_at".into(), - serde_json::json!(now().to_rfc3339_opts(chrono::SecondsFormat::Millis, true)), + serde_json::json!(at.to_rfc3339_opts(chrono::SecondsFormat::Millis, true)), ); + // A DISSOLVE is a quorum-verified TERMINAL amendment (persist v52, + // CIRISPersist#956): the record keeps every seat and gains + // `dissolved_at`, and persist binds that instant to the one the quorum + // signed INSIDE this envelope — so it is pinned here, before anyone + // signs, spelled exactly as persist's note gives it (`to_rfc3339()`). + if req.action == "dissolve" { + obj.insert(DISSOLVED_AT.into(), serde_json::json!(at.to_rfc3339())); + } } let bytes = match ciris_verify_core::jcs::canonicalize(&env) { Ok(b) => b, @@ -1617,6 +1939,24 @@ fn check_envelope(loaded: &Loaded, env: &serde_json::Value) -> Result Vec { + env.get("members") + .and_then(|v| v.as_array()) + .map(|ms| { + ms.iter() + .filter_map(|m| m.get("key_id").and_then(|v| v.as_str())) + .filter(|k| loaded.member(k).is_none()) + .map(str::to_owned) + .collect() + }) + .unwrap_or_default() +} + async fn cosign( State(st): State, headers: HeaderMap, @@ -1637,6 +1977,9 @@ async fn cosign( if let Err(r) = check_envelope(&loaded, &req.change_envelope) { return r; } + // CONSENT TO JOIN: a member may co-sign an `add`; the joiner's own + // acceptance is checked by persist when the co-signed widening is written + // (assemble), not assumed from the signatures gathered here. let bytes = match ciris_verify_core::jcs::canonicalize(&req.change_envelope) { Ok(b) => b, Err(e) => return bad_change(format!("canonicalize: {e}")), @@ -1741,6 +2084,22 @@ async fn assemble( Ok(a) => a, Err(r) => return r, }; + // CONSENT TO JOIN (CIRISConstitution#133 / CIRISPersist#955): an `add` + // is the only action that grows the roster, and it is written below as a + // co-signed WIDENING whose admission persist gates on the joiner's own + // acceptance. A hand-built envelope that grows the roster under ANOTHER + // action name is refused here — it would reach the supersede door, which + // never adds (persist Q2). + let joiners = envelope_joiners(&loaded, &req.change_envelope); + if action != "add" && !joiners.is_empty() { + return crate::membership_invites::refused( + &ciris_edge::membership::MembershipError::Refused { + group_key_id: id.clone(), + member_key_id: joiners.join(", "), + rule: ciris_edge::membership::RULE_SUPERSEDE_CANNOT_ADD, + }, + ); + } let capsule = match pen(&st, &caller).await { Ok(c) => c, Err(r) => return r, @@ -1755,29 +2114,70 @@ async fn assemble( let primary = capsule.key_id().to_owned(); if action == "dissolve" { - // An empty roster is not a verifiable membership change (verify's - // WeakQuorum at m = 0), so the quorum authorizes the dissolve-marked - // envelope over the CURRENT roster, and that proof rides the terminal - // supersede as its authorization. + // persist v52 (CIRISPersist#956): a quorum dissolve is a TERMINAL + // AMENDMENT through `supersede_family_with_quorum` — the record with + // every seat byte-identical and `dissolved_at` set to the instant the + // quorum signed inside the envelope (pinned at envelope time). It + // replicates: a peer re-verifies the quorum against its own roster and + // applies it, after which the family has no active members and every + // write naming it is refused `federation_group_dissolved`. No removal + // rows are written — after the amendment persist would refuse them, + // and the fold is already empty. + // + // Before v52 this wrote one co-signed revocation per member and a + // plain supersede to an empty roster, which persist stripped of any + // proof (#910.5) and no peer ever applied (#700 audit, fix 2). + let _ = row_at; + return quorum_terminal_dissolve(&st, &caller, &capsule, &loaded, env, thresholds).await; + } + + if action == "add" { + // persist v52 Q2: a supersede never adds — the quorum's add is a + // co-signed WIDENING. The quorum is verified over the envelope first + // (verify's membership-change gate, as for every quorum change), then + // the widening carries the other signers' scrubs over this exact row, + // and persist admits it only on the joiner's live acceptance. if let Err(e) = dir .verify_membership_quorum(Cohort::Family, &id, env, &thresholds) .await { return quorum_refusal(&e, &proto_now); } - return terminal_dissolve( - &st, - &caller, - &capsule, - &loaded, - serde_json::json!({ - "action": "dissolve", - "change_envelope": env, - "quorum_signatures": thresholds, - }), - Some((row_at, req.signatures.as_slice())), - ) - .await; + let Some(k) = env.get("target_key_id").and_then(|v| v.as_str()) else { + return bad_change("an add names no target_key_id".to_owned()); + }; + let role = env + .get("roles") + .and_then(|r| r.get(k)) + .and_then(|v| v.as_str()) + .unwrap_or(ROLE_MEMBER) + .to_owned(); + let member = FamilyMember { + key_id: k.to_owned(), + joined_at: row_at, + role: Some(role), + }; + let mut spec = match sign_family_widening(&capsule, &id, &member).await { + Ok(s) => s, + Err(e) => return signer_unavailable(e), + }; + spec.cosignatures = + crate::roster_rows::cosignatures_for(&req.signatures, &primary, "widening", k); + if let Err(e) = dir + .add_member(Cohort::Family, &id, RosterMember::from(member), &spec) + .await + { + return crate::membership_invites::persist_refusal(&e) + .unwrap_or_else(|| quorum_refusal(&e, &proto_now)); + } + let extra = serde_json::json!({ + "action": action, + "added": k, + "dek_rewrap": rewrap(&st.engine, &id, k).await, + }); + tracing::info!(family = %id, member = %k, "family: quorum add applied as a co-signed widening"); + kick("family:quorum_change"); + return respond_with_family(&st, &id, &caller, extra).await; } // add / remove / role: the new record IS the envelope — its roster, its @@ -1832,9 +2232,6 @@ async fn assemble( .map(str::to_owned); let mut extra = serde_json::json!({ "action": action, "version": version }); match (action.as_str(), target.as_deref()) { - ("add", Some(k)) => { - extra["dek_rewrap"] = rewrap(&st.engine, &id, k).await; - } ("remove", Some(k)) => { // The supersede shrank the record; the revocation is what // REPLICATES the removal. persist v49.0.0 judges it by the @@ -1893,6 +2290,11 @@ pub fn router(engine: Arc, user_seed_dir: std::path::PathBuf) -> Router .route("/v1/families", get(list_families).post(create_family)) .route("/v1/families/{id}", get(read_family).delete(dissolve)) .route("/v1/families/{id}/members", post(add_member)) + .route("/v1/families/{id}/invites", get(list_invites).post(invite)) + .route( + "/v1/families/{id}/invites/{proposal_id}", + delete(withdraw_invite), + ) .route("/v1/families/{id}/members/{key_id}", delete(remove_member)) .route("/v1/families/{id}/members/{key_id}/role", post(change_role)) .route("/v1/families/{id}/leave", post(leave)) diff --git a/src/federation_surface.rs b/src/federation_surface.rs index 36e20b13..621b9f1d 100644 --- a/src/federation_surface.rs +++ b/src/federation_surface.rs @@ -313,6 +313,29 @@ async fn get_metrics(State(st): State) -> Response { .map(|(k, v)| (k.to_string(), serde_json::json!(v))) .collect(); + // edge v34.3.0 — three more receive-side ledgers the snapshot carries and + // this route did not fold (found by edge's own run of the selffiles + // declaration on v34.3.0). Each is `reason → count`; a non-zero count is + // a frame, a pull or a first contact that was REFUSED, and this is the + // only place an operator reads why without the log. + // transport_inbound_drops — CIRISEdge#728: the Reticulum receive-side + // choke point, by tag (`identity_frame_on_scoped_link` = a peer that + // still selects links by peer alone, pre-#728). + // blob_pull_refusals — CIRISEdge#717: a pull that fetched and refused + // to STORE (`size_mismatch`, `stream_pointer_needs_dag_pull`); each + // is a file that is not on this device. + // first_contact_outcomes — CIRISEdge#683: the opaque-plane + // first-contact door by label (`first_contact_admitted`, refusals). + let fold = + |m: &std::collections::HashMap| -> serde_json::Map { + m.iter() + .map(|(k, v)| (k.to_string(), serde_json::json!(v))) + .collect() + }; + let transport_inbound_drops = fold(&bundle.transport_inbound_drops); + let blob_pull_refusals = fold(&bundle.blob_pull_refusals); + let first_contact_outcomes = fold(&bundle.first_contact_outcomes); + ( StatusCode::OK, Json(serde_json::json!({ @@ -335,9 +358,16 @@ async fn get_metrics(State(st): State) -> Response { "replication_duplicate_total": duplicates, "replication_round_outcomes_total": round_outcomes, "replication_round_routing": round_routing, + // edge v34.1.0+ (CIRISEdge#716/#722): frames that skipped the + // Channel-first path because they exceed its fragment cap and + // went Resource-first. Top-level on the snapshot; folded here. + "channel_first_skipped_over_cap": bundle.channel_first_skipped_over_cap, "bootstrap_door_outcomes": bootstrap_door, "blob_route_refusals": blob_route_refusals, "blob_pull_sources": blob_pull_sources, + "blob_pull_refusals": blob_pull_refusals, + "transport_inbound_drops": transport_inbound_drops, + "first_contact_outcomes": first_contact_outcomes, "carriage_standing": crate::operator_surface::carriage_standing(Some(&bundle)).as_str(), "receive_standing": crate::operator_surface::receive_standing(Some(&bundle)).as_str(), "receive_decided_total": crate::operator_surface::receive_decided_total(&bundle), diff --git a/src/file_custody.rs b/src/file_custody.rs new file mode 100644 index 00000000..6986df2f --- /dev/null +++ b/src/file_custody.rs @@ -0,0 +1,593 @@ +//! **Where each file is — the custody view** (0.5.218, `FSD/FILE_CUSTODY.md`). +//! +//! The maintainer's ask of 2026-09-30: "for every blob we have access to, there +//! needs to be a view of which devices it is on, the total devices." This +//! module is the VIEW. Copying a file to a device and removing it from one are +//! a later cut (the FSD names their design); nothing here moves a byte. +//! +//! # Three sources, none of which the server read before this cut +//! +//! 1. **The person's devices** — [`owner_devices`]: persist's +//! `nodes_owned_by(owner)` (the owner-bindings, which fold the owner's +//! `withdraws`, so a RELEASED node is gone) minus every node whose content +//! occurrence for that owner is REVOKED (`list_identity_occurrences_for` +//! names it, `list_identity_occurrences_active` does not). That is the same +//! projection the self-room driver and the self re-wrap key their rosters +//! on, so "your devices" here is the set a self file is wrapped to and +//! replicated to — not a second opinion about who you are. The occurrence +//! list alone (`GET /v1/self/occurrences`) was NOT used as the roster: it +//! carries the person's login anchor and, on an agent split, the ACTOR key +//! beside the node, so it would count one machine twice; it is read only to +//! subtract revocations. Labels come from `self:device_label:v1`, the rows +//! `POST /v1/self/occurrence/label` writes, through +//! [`crate::self_devices::labels_for`]. +//! 2. **Custody** — edge's `FileRow::custody` → persist's +//! `Engine::blob_custody` (v51.1.0, CIRISPersist#942): who can open the +//! blob per person and through which devices, whether this node holds it, +//! and the copies that are countable. It is authorized EXACTLY as the bytes +//! read (`read_any_for_viewer`'s tier gate, then the withdrawn check), and it +//! is asked as the drive's viewer key — the split-install content occurrence +//! (`drive::viewer_key`), never `local_derived_key_id()` — so a viewer who +//! cannot open the bytes learns nothing, not even the access list. +//! 3. **Delivery receipts** — edge's `FileRow::received_by` → +//! `receipts::received_for` (CIRISEdge#738, CC 5.3.3.6): one +//! `(node, epoch, K, at)` per device that stored every chunk of a file +//! under the root the author published. Since edge v38.0.0 / persist v52 +//! (CIRISPersist#953, CIRISEdge#755) EVERY file is a stream: a chunk DAG's +//! is its `stream_id`, an inline file's (≤ 1 MiB) is persist's one-leaf log +//! `inline_blob_stream_id(sha)`, and `at` is the instant the author's store +//! took the receipt (`list_stored_delivery_receipts_for`'s `received_at`). The receiving node signs it on the +//! pull (`on_dag_pulled`); the author's node admits it on arrival +//! (`admit_and_count`, in the replication bridge). A receipt is proof of +//! DELIVERY — it is never retracted by an eviction at this pin — so a device +//! with a receipt is reported `holds: "received"`, never "here". +//! +//! # The host wiring the receipts need — checked, all present +//! +//! Past defect class ("host hooks left unset"): an optional edge hook the +//! server never set disabled a feature silently for six releases. Receipts +//! touch four places and every one resolves to a store the server already +//! builds, with no receipt-specific hook to set: +//! +//! - PUBLISH: `files::publish` puts the stream's STH through +//! `GroupContentStore::stream_log()`. The trait default is `None` (a file +//! published there carries no STH and can never be receipted); the drive's +//! store is edge's `PersistGroupContentStore` (`drive::store`), whose +//! `stream_log` is `receipts::stream_log_of(engine)` — `Some` for the SQLite +//! and Postgres backends. +//! - RECEIVE: `BlobPuller::spawn` takes the engine as a REQUIRED argument +//! (`backend::spawn_puller_with`), and `on_dag_pulled` signs with it and +//! names the puller's `local_key_id` (the edge signer, which is also the key +//! the engine self-attests the receipt row as — admission requires the two +//! to agree). +//! - ADMIT: the bridge admits a receipt row only when its `engine` is set, +//! which is `SealedContentWiring::engine` in `compose::start_replication_runtime` +//! (wired since 0.5.212 for the key-grant door; edge v27 made it +//! unconstructible without the pull sink). +//! - READ: this module, through the same `PersistGroupContentStore`. +//! +//! # What the view cannot say, and says so +//! +//! Every partial answer carries a `why` entry with a stable id (one +//! [`msg`] call per reason, so the localization guard sees each): +//! `self`/`family` copies elsewhere are unobservable BY DESIGN (CC 5.2 — never +//! announced); a receipt proves delivery, not current holding; receipts are +//! ADMITTED on the author's device, so another +//! device's view holds only its own; and a receipt signed by a key that is not +//! one of the person's listed devices (a family or community member, or an +//! agent-split ACTOR key) is listed separately rather than dropped. +//! +//! **Retired at edge v38.0.0 / persist v52 (0.5.218):** `custody.inline_no_receipt` +//! (an inline file now has a one-leaf log and is receipted like any other — +//! CIRISPersist#953 item 2, CIRISEdge#755) and `custody.receipt_time_unknown` +//! (`Received::at` is the store's `received_at`, never absent — #953 item 3). +//! Both conditions are now false on every path, so neither id is emitted +//! anywhere; they are deleted rather than kept as dead constants, and the +//! localization ratchet's count drops by two. + +use std::collections::{BTreeSet, HashSet}; + +use serde::Serialize; + +use ciris_persist::prelude::Engine; + +/// One reason the view is partial: a stable id and its English sentence. +#[derive(Debug, Clone, PartialEq, Eq, Serialize)] +pub struct Why { + pub reason_id: &'static str, + pub detail: &'static str, +} + +/// A reason, built in one place per id — named `msg` so the localization guard +/// reads the id from its argument position and the text beside it. +const fn msg(reason_id: &'static str, detail: &'static str) -> Why { + Why { reason_id, detail } +} + +pub const WHY_COPIES_UNOBSERVABLE: Why = msg( + "custody.copies_unobservable_by_design", + "Your own and your family's files are never announced to anyone, so copies on other devices cannot be counted; a delivery receipt is the only sign a device received one.", +); +pub const WHY_RECEIPT_IS_DELIVERY: Why = msg( + "custody.receipt_is_delivery_not_holding", + "A delivery receipt proves a device received the whole file. It does not prove the device still holds it: removing a copy does not withdraw its receipt yet.", +); +pub const WHY_RECEIPTS_ON_AUTHOR_DEVICE: Why = msg( + "custody.receipts_admitted_on_author_device", + "Delivery receipts are collected by the device that wrote the file. This device lists only the receipts it holds; ask the device that wrote it for the full answer.", +); +pub const WHY_RECEIPT_FROM_OTHER_KEY: Why = msg( + "custody.receipt_signer_not_your_device", + "Some receipts were signed by devices that are not among your devices (another member of the room, or an agent's own key); they are listed separately.", +); +pub const WHY_RECEIPTS_UNREADABLE: Why = msg( + "custody.receipts_unreadable", + "This device could not read its delivery receipts just now; which devices received the file is unknown until it can.", +); +pub const WHY_COMMONS_READABLE: Why = msg( + "custody.commons_readable_by_holders", + "This file is public: anyone holding the bytes can read them, so who can open it is not a list.", +); +pub const WHY_NO_COPY_REPORTS_PENDING: Why = msg( + "custody.no_copy_reports_pending", + "Your other devices cannot yet report that they hold no copy; those reports arrive with within-cohort custody acknowledgements. Until then a device without a delivery receipt is shown as unknown.", +); +pub const WHY_NO_COPY_HERE: Why = msg( + "custody.no_copy_here", + "This device holds no copy of the file, so who can open it and how many copies are announced are answered by a device that holds it.", +); + +/// Every custody reason, for the FSD table and the gates. +pub const ALL_WHY: &[Why] = &[ + WHY_COPIES_UNOBSERVABLE, + WHY_RECEIPT_IS_DELIVERY, + WHY_RECEIPTS_ON_AUTHOR_DEVICE, + WHY_RECEIPT_FROM_OTHER_KEY, + WHY_RECEIPTS_UNREADABLE, + WHY_COMMONS_READABLE, + WHY_NO_COPY_REPORTS_PENDING, + WHY_NO_COPY_HERE, +]; + +/// `holds` tokens: one word per fact, like the drive's byte states. +/// +/// The maintainer's ruling on #704: "no copy here is a receipt (node +/// responsive, no copy)." A device that ANSWERS that it holds nothing has said +/// something about custody, so `none` is a fact, never folded into `unknown`. +/// At this pin only THIS device can say it (`checked_at`); a remote device's +/// signed "no copy" arrives with persist's within-cohort custody +/// acknowledgements (CIRISConstitution#130), and will carry `reported_at`. +pub const HOLDS_HERE: &str = "here"; +pub const HOLDS_RECEIVED: &str = "received"; +pub const HOLDS_NONE: &str = "none"; +pub const HOLDS_UNKNOWN: &str = "unknown"; +/// Every `holds` token, in the order a client should rank them. +pub const HOLDS: &[&str] = &[HOLDS_HERE, HOLDS_RECEIVED, HOLDS_NONE, HOLDS_UNKNOWN]; + +/// One of the person's devices, with every key it answers to. +#[derive(Debug, Clone)] +pub struct Device { + /// The key the owner-binding names. + pub node_key_id: String, + /// Every key that is this device: the node key, and — for THIS node — its + /// actor, held node signer and wire identity + /// (`peer::own_keys_of_this_node`), because a split install binds the node + /// key while its engine signs receipts as the actor. + pub keys: Vec, + pub label: Option, + pub this_device: bool, +} + +/// **The person's devices** — see the module doc, source 1. Sorted, this device +/// first. `Err` only when the directory cannot be read. +pub async fn owner_devices(engine: &Engine, owner: &str) -> Result, String> { + use ciris_persist::federation::admission::nodes_owned_by; + let dir = engine.federation_directory(); + let owned = nodes_owned_by(dir.as_ref(), owner) + .await + .map_err(|e| format!("nodes_owned_by({owner}): {e:#}"))?; + let active: HashSet = dir + .list_identity_occurrences_active(owner) + .await + .map_err(|e| format!("list_identity_occurrences_active({owner}): {e:#}"))? + .into_iter() + .map(|o| o.occurrence_key_id) + .collect(); + let revoked: HashSet = dir + .list_identity_occurrences_for(owner) + .await + .map_err(|e| format!("list_identity_occurrences_for({owner}): {e:#}"))? + .into_iter() + .map(|o| o.occurrence_key_id) + .filter(|k| !active.contains(k)) + .collect(); + let own: Vec = match engine.local_derived_key_id().await { + Ok(k) => crate::peer::own_keys_of_this_node(&k), + Err(_) => crate::node_key::wire_identity() + .map(|w| vec![w.to_owned()]) + .unwrap_or_default(), + }; + let labels = crate::self_devices::labels_for(engine, owner).await; + let mut seen = BTreeSet::new(); + let mut out: Vec = Vec::new(); + for node in owned { + if node == owner || revoked.contains(&node) || !seen.insert(node.clone()) { + continue; + } + let this_device = own.contains(&node); + let keys = if this_device { + let mut k = vec![node.clone()]; + k.extend(own.iter().filter(|o| **o != node).cloned()); + k + } else { + vec![node.clone()] + }; + let label = keys.iter().find_map(|k| labels.get(k).cloned()); + out.push(Device { + node_key_id: node, + keys, + label, + this_device, + }); + } + out.sort_by(|a, b| { + b.this_device + .cmp(&a.this_device) + .then_with(|| a.node_key_id.cmp(&b.node_key_id)) + }); + Ok(out) +} + +/// One receipt, as the view reports it. +#[derive(Debug, Clone, PartialEq, Eq, Serialize)] +pub struct ReceivedView { + pub epoch: u64, + pub k: u64, + /// When the author's store took the receipt (RFC 3339) — persist's + /// `received_at` (CIRISPersist#953), the store's fact rather than the + /// receiving device's claim. Always present since persist v52. + pub at: String, +} + +/// A receipt that no listed device answers to. +#[derive(Debug, Clone, PartialEq, Eq, Serialize)] +pub struct OtherReceipt { + pub node_key_id: String, + pub epoch: u64, + pub k: u64, + pub at: String, +} + +/// One device's row in the view. +#[derive(Debug, Clone, Serialize)] +pub struct DeviceCustody { + pub node_key_id: String, + #[serde(skip_serializing_if = "Option::is_none")] + pub label: Option, + pub this_device: bool, + /// A key of this device is a grant recipient in persist's access list; + /// `null` when the access list is not answerable here (this device holds + /// no copy — [`WHY_NO_COPY_HERE`]). + pub can_open: Option, + pub received: Option, + /// `here` (this device holds the bytes), `received` (a delivery receipt + /// names it), `none` (the device answered: it holds no copy — this device + /// only, until remote reports exist) or `unknown` (nothing says either way). + pub holds: &'static str, + /// THIS device's statement time (RFC 3339): `holds` for this device is a + /// live answer, not an inference. `null` on every other device. + pub checked_at: Option, + /// When a remote device reported its `holds` — always `null` until + /// persist's within-cohort custody acknowledgements carry such reports. + pub reported_at: Option, +} + +/// What the receipt read found. Since edge v38.0.0 every file — inline or +/// chunked — has a receipt stream, so there is no "unsupported" arm: the only +/// partial answer is a read that failed. +pub enum Receipts { + /// The stream log could not be read. + Unreadable, + /// The receipts this node's stream log holds for the file. + Held(Vec), +} + +/// The view's device half, computed without any I/O so it can be pinned. +pub struct DeviceHalf { + pub devices: Vec, + pub other_receipts: Vec, + pub why: Vec, +} + +fn received_view(r: &ciris_edge::receipts::Received) -> ReceivedView { + ReceivedView { + epoch: r.epoch, + k: r.k, + at: r.at.to_rfc3339(), + } +} + +/// **The per-device answer**: each device against the access list and the +/// receipts, plus the receipts no device answers to, plus why it is partial. +/// `access_devices` is every device key persist's custody names as able to +/// open the blob (`None` when this device holds no copy and persist's custody +/// could not be asked); `held_here` is persist's, `false` on that path; +/// `checked_at` is the moment this device answered. +pub fn device_half( + devices: &[Device], + access_devices: Option<&HashSet>, + held_here: bool, + receipts: &Receipts, + this_device_is_author: bool, + checked_at: &str, +) -> DeviceHalf { + let mut why = Vec::new(); + let held: &[ciris_edge::receipts::Received] = match receipts { + Receipts::Held(r) => r, + Receipts::Unreadable => { + why.push(WHY_RECEIPTS_UNREADABLE); + &[] + } + }; + let mut matched: HashSet = HashSet::new(); + let rows = devices + .iter() + .map(|d| { + let received = held.iter().enumerate().find_map(|(i, r)| { + d.keys.contains(&r.node_key_id).then(|| { + matched.insert(i); + received_view(r) + }) + }); + // THIS device answers for itself: it holds the bytes or it does + // not, and either is a fact. A receipt this device once signed does + // not outrank its own "no copy" now (an eviction does not retract + // a receipt at this pin). Other devices: a receipt, or unknown. + let holds = match (d.this_device, held_here, received.is_some()) { + (true, true, _) => HOLDS_HERE, + (true, false, _) => HOLDS_NONE, + (false, _, true) => HOLDS_RECEIVED, + (false, _, false) => HOLDS_UNKNOWN, + }; + DeviceCustody { + node_key_id: d.node_key_id.clone(), + label: d.label.clone(), + this_device: d.this_device, + can_open: access_devices.map(|a| d.keys.iter().any(|k| a.contains(k))), + received, + holds, + checked_at: d.this_device.then(|| checked_at.to_owned()), + reported_at: None, + } + }) + .collect::>(); + // Mark every receipt a device answered to — a device may have receipted + // under two keys (a split install's actor and node), counted once above. + for (i, r) in held.iter().enumerate() { + if devices.iter().any(|d| d.keys.contains(&r.node_key_id)) { + matched.insert(i); + } + } + let other_receipts: Vec = held + .iter() + .enumerate() + .filter(|(i, _)| !matched.contains(i)) + .map(|(_, r)| OtherReceipt { + node_key_id: r.node_key_id.clone(), + epoch: r.epoch, + k: r.k, + at: r.at.to_rfc3339(), + }) + .collect(); + if matches!(receipts, Receipts::Held(_)) { + why.push(WHY_RECEIPT_IS_DELIVERY); + if !this_device_is_author { + why.push(WHY_RECEIPTS_ON_AUTHOR_DEVICE); + } + if !other_receipts.is_empty() { + why.push(WHY_RECEIPT_FROM_OTHER_KEY); + } + } + if access_devices.is_none() { + why.push(WHY_NO_COPY_HERE); + } + if rows + .iter() + .any(|d| !d.this_device && d.holds == HOLDS_UNKNOWN) + { + why.push(WHY_NO_COPY_REPORTS_PENDING); + } + DeviceHalf { + devices: rows, + other_receipts, + why, + } +} + +/// The compact per-row summary `GET /v1/drive` carries: the device count and +/// how many of them hold a receipt — `received_on: null` only when the stream +/// log could not be read (unknowable, not zero). An inline file counts like +/// any other since edge v38.0.0. +#[derive(Debug, Clone, Serialize)] +pub struct CompactCustody { + pub devices_total: usize, + pub received_on: Option, +} + +/// [`CompactCustody`] from the roster and the receipt read — one list query +/// per CHUNKED row, none for an inline one, and no manifest read. +pub fn compact(devices: &[Device], receipts: &Receipts) -> CompactCustody { + let received_on = match receipts { + Receipts::Held(held) => Some( + devices + .iter() + .filter(|d| held.iter().any(|r| d.keys.contains(&r.node_key_id))) + .count(), + ), + Receipts::Unreadable => None, + }; + CompactCustody { + devices_total: devices.len(), + received_on, + } +} + +/// Read a file row's receipts through the drive's store — inline and chunked +/// alike (edge names the stream, `receipts::receipt_stream_id`; this module +/// never decides which files are receiptable). +pub async fn receipts_of( + file: &ciris_edge::files::FileRow, + store: &dyn ciris_edge::group_content::GroupContentStore, +) -> Receipts { + match file.received_by(store).await { + Ok(r) => Receipts::Held(r), + Err(e) => { + tracing::warn!( + file = %file.attestation_id, + error = %e, + "custody: the file's delivery receipts could not be read — reported as unknown" + ); + Receipts::Unreadable + } + } +} + +/// The person-by-person access list, flattened to its device keys. +pub fn access_device_keys( + custody: &ciris_persist::federation::blob_custody::BlobCustody, +) -> HashSet { + custody + .access + .iter() + .flat_map(|a| a.devices.iter().cloned()) + .collect() +} + +/// The substrate's partial-answer reasons, as ids: persist's `why` is a +/// sentence keyed by TIER, so the tier picks the id (a reword upstream cannot +/// move an answer between reasons). Taken from persist's custody when this +/// device holds a copy, from the row's pointer when it does not. +pub fn substrate_why(copies_observable: bool, tier: &str) -> Vec { + let mut out = Vec::new(); + if !copies_observable { + out.push(WHY_COPIES_UNOBSERVABLE); + } + if tier == "plaintext" { + out.push(WHY_COMMONS_READABLE); + } + out +} + +#[cfg(test)] +mod tests { + use super::*; + + fn dev(key: &str, this: bool) -> Device { + Device { + node_key_id: key.into(), + keys: vec![key.into()], + label: None, + this_device: this, + } + } + + fn rec(node: &str) -> ciris_edge::receipts::Received { + ciris_edge::receipts::Received { + node_key_id: node.into(), + epoch: 0, + k: 25, + at: "2026-09-30T00:00:00Z".parse().expect("fixture instant"), + } + } + + #[test] + fn a_receipted_second_device_is_received_and_the_author_is_here() { + let devices = vec![dev("a", true), dev("b", false)]; + let access: HashSet = ["a".into(), "b".into()].into(); + let half = device_half( + &devices, + Some(&access), + true, + &Receipts::Held(vec![rec("b"), rec("stranger")]), + true, + "2026-09-30T00:00:00Z", + ); + assert_eq!(half.devices[0].holds, HOLDS_HERE); + assert_eq!(half.devices[1].holds, HOLDS_RECEIVED); + assert!(half.devices.iter().all(|d| d.can_open == Some(true))); + assert_eq!( + half.devices[0].checked_at.as_deref(), + Some("2026-09-30T00:00:00Z") + ); + assert_eq!(half.devices[1].checked_at, None); + assert_eq!(half.other_receipts.len(), 1); + let ids: Vec<_> = half.why.iter().map(|w| w.reason_id).collect(); + assert!(ids.contains(&"custody.receipt_is_delivery_not_holding")); + assert!(ids.contains(&"custody.receipt_signer_not_your_device")); + assert!(!ids.contains(&"custody.receipts_admitted_on_author_device")); + let c = compact(&devices, &Receipts::Held(vec![rec("b")])); + assert_eq!((c.devices_total, c.received_on), (2, Some(1))); + } + + /// Since edge v38.0.0 a receipt always carries the store's `received_at`, + /// and nothing in the view says the time is unknown. + #[test] + fn a_receipt_says_when_and_no_time_reason_is_given() { + let devices = vec![dev("a", true), dev("b", false)]; + let access = HashSet::new(); + let half = device_half( + &devices, + Some(&access), + true, + &Receipts::Held(vec![rec("b")]), + true, + "t", + ); + let got = half.devices[1].received.as_ref().expect("b receipted"); + assert_eq!(got.at, "2026-09-30T00:00:00+00:00"); + assert_eq!(half.why, vec![WHY_RECEIPT_IS_DELIVERY]); + assert_eq!( + compact(&devices, &Receipts::Unreadable).received_on, + None, + "an unreadable log is unknown, not zero" + ); + } + + /// The maintainer's ruling on #704: a device that holds the row and not the + /// bytes answers `none` for itself, with the time it answered — never 409. + #[test] + fn this_device_without_a_copy_says_none_and_when() { + let devices = vec![dev("b", true), dev("a", false)]; + let half = device_half( + &devices, + None, + false, + &Receipts::Held(vec![rec("b")]), + false, + "2026-09-30T12:00:00Z", + ); + assert_eq!( + half.devices[0].holds, HOLDS_NONE, + "its own answer outranks its old receipt" + ); + assert_eq!( + half.devices[0].checked_at.as_deref(), + Some("2026-09-30T12:00:00Z") + ); + assert_eq!(half.devices[0].can_open, None); + assert_eq!(half.devices[1].holds, HOLDS_UNKNOWN); + assert_eq!(half.devices[1].reported_at, None); + let ids: Vec<_> = half.why.iter().map(|w| w.reason_id).collect(); + assert!(ids.contains(&"custody.no_copy_here"), "{ids:?}"); + assert!(ids.contains(&"custody.no_copy_reports_pending"), "{ids:?}"); + } + + #[test] + fn every_reason_id_is_unique_and_in_the_custody_family() { + let mut seen = HashSet::new(); + for w in ALL_WHY { + assert!(w.reason_id.starts_with("custody."), "{}", w.reason_id); + assert!(seen.insert(w.reason_id), "duplicate {}", w.reason_id); + } + } +} diff --git a/src/key_standing.rs b/src/key_standing.rs index 0cad530f..29cc33b7 100644 --- a/src/key_standing.rs +++ b/src/key_standing.rs @@ -249,7 +249,19 @@ impl HeldRevocations { ) -> KeyStatementFold { const NONE: &[Revocation] = &[]; let rows = self.by_key.get(key_id).map_or(NONE, Vec::as_slice); - fold_key_statement_standing(key_id, rows, statement_at, now) + // persist v52 (CIRISPersist#784) keys every revocation reader on the + // SUBJECT — the SHA-256 of the raw Ed25519 key — not on a label, and the + // fold now takes that digest. The rows here came from + // `revocations_for(key_id)`, which resolves the held key's digest and + // answers only rows naming it, so every row in this slice carries the + // same `revoked_key_sha256_ed25519_raw`, and that IS the subject. Read + // it off the row rather than re-deriving it from a public-key lookup: + // one source (persist's resolution) instead of two that could disagree. + // No rows ⇒ nothing can cover the statement, whatever the subject. + let subject = rows + .first() + .map_or("", |r| r.revoked_key_sha256_ed25519_raw.as_str()); + fold_key_statement_standing(key_id, subject, rows, statement_at, now) } /// The standing of the statement `att` makes, dated by its SIGNED envelope @@ -300,7 +312,11 @@ mod tests { fn rev(id: &str, key: &str, effective_at: &str, bound: Option<&str>) -> Revocation { Revocation { revocation_id: id.into(), - revoked_key_id: key.into(), + revoked_key_id: Some(key.into()), + // The fold matches on the subject digest; any stable per-key + // string stands in for the SHA-256 here (the fold compares, it + // does not re-derive). + revoked_key_sha256_ed25519_raw: format!("digest-of-{key}"), revoking_key_id: "authority".into(), reason: None, revoked_at: ts(effective_at), diff --git a/src/lib.rs b/src/lib.rs index 92abea2d..e5ebcf46 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -76,6 +76,11 @@ pub mod admin_ops; /// it. pub mod attest; +/// CC 5.4.6 (CIRISServer#655) — a node that serves infrastructure relays the +/// Key and IdentityOccurrence rows of every announced device and its owner, +/// through edge's per-kind `SelfOwn` publish selector (CIRISEdge#678). +pub mod announced_relay; + /// The v39 tier crossing (`enter_mesh` + `widen_audience`) composed in one /// place, replacing `attestation_promote`. pub mod attestation_crossing; @@ -317,10 +322,15 @@ pub mod key_standing; /// Public so the integration test (`tests/peer_replication.rs`) can drive the /// admission + consent-emit logic directly. pub mod location; +pub mod membership_invites; +pub mod pair_intents; /// Rendering an error with its whole cause chain, for the one line an operator /// reads (CIRISServer#586). pub mod error_chain; +/// Where each file is — the custody view (`FSD/FILE_CUSTODY.md`): the person's +/// devices against persist's custody and edge's delivery receipts. +pub mod file_custody; pub mod media_gate; /// The owner's own devices (`FSD/ROSTER_AND_DRIVE_CRUD.md` §2): release a node /// from its owner (a signed `withdraws` of the owner-binding) and relabel a @@ -330,6 +340,11 @@ pub mod self_devices; /// (CIRISServer#678). pub mod self_rewrap; pub mod self_room_drive; +/// **One device handles each exchange** (CC 3.1.3.1, CIRISPersist#782): the +/// session-claim gate every autonomous act for the person passes, the +/// attendance that decides where a claim is taken, the renewal loop, and +/// `GET /v1/self/sessions` (`FSD/SESSION_CLAIMS.md`). +pub mod session_claims; /// **The capacity READ surface** — `GET /v1/my-data/capacity`. The scorer /// emitted `capacity:*` attestations that nothing served back (CIRISServer#580); diff --git a/src/loop_cadence.rs b/src/loop_cadence.rs index 6f78518e..a7bf75c9 100644 --- a/src/loop_cadence.rs +++ b/src/loop_cadence.rs @@ -34,8 +34,8 @@ //! With a fixed spacing, two loops tick together only if //! `(i − j) × SLOT_SPACING` is a multiple of `gcd(period_i, period_j)`. Every //! cadence here defaults to a multiple of 30 s, so that gcd is at least 30 s -//! while `(i − j) × SLOT_SPACING` is at most 15 s — the separation is never -//! less than [`SLOT_SPACING`], at any of them. +//! while `(i − j) × SLOT_SPACING` is at most 25 s (eleven loops at 2.5 s) — +//! the separation is never less than [`SLOT_SPACING`], at any of them. //! //! An operator who sets a cadence that shares no useful factor with the others //! (7 s, say) can still produce occasional coincidences. That is a much smaller @@ -78,7 +78,7 @@ use tokio::time::{sleep_until, Instant}; /// /// Adding a loop here moves the others' slots. That is intended: the invariant /// is even spread, not a fixed offset for any one loop. -pub const LOOPS: [&str; 10] = [ +pub const LOOPS: [&str; 12] = [ "config_reconcile", "replication_reconcile", "scorer", @@ -100,6 +100,17 @@ pub const LOOPS: [&str; 10] = [ // (CIRISEdge#646). Its own phase so it does not tick with scope_seal, // which it calls into. "self_room", + // CC 3.1.3.1 / CIRISPersist#782 — renews the session claims this device + // holds while its person is on it, and lets them lapse when they leave + // (`crate::session_claims`). Its own slot so the renewal's handler reads + // never land on the self room's tick, whose Add/Remove it gates. + "session_claims", + // edge v38 / persist v52 (CIRISPersist#955) — the pair-room driver: a + // person asks for a chat ONCE and waits, so the two-step join (the + // joiner's acceptance, the creator's widening) and the MLS handshake (the + // joiner's KeyPackage, the creator's Welcome) are advanced here, not only + // on a read (`contacts_chat::PairRoomDriver`). + "pair_rooms", ]; /// The origin every cadence measures its phase from, captured once per process. @@ -111,10 +122,19 @@ fn epoch() -> Instant { /// The gap between consecutive loop slots. /// /// Wide enough to clear the 1-2 s bursts CIRISServer#575 measured, and narrow -/// enough that all of [`LOOPS`] fits well inside the shortest cadence the node -/// runs (`5 × 3 s = 15 s` against 30 s) — the margin is what keeps the -/// separation argument in the module docs true. -pub const SLOT_SPACING: Duration = Duration::from_secs(3); +/// enough that all of [`LOOPS`] fits inside the shortest cadence the node runs +/// (30 s) — the margin is what keeps the separation argument in the module +/// docs true. +/// +/// 3 s → 2.5 s for 0.5.218, when `session_claims` became the ELEVENTH loop: at +/// 3 s the last slot sits at `10 × 3 s = 30 s`, which wraps to 0 on a 30 s +/// cadence and lands exactly on `config_reconcile` every tick — #575 again, by +/// arithmetic. At 2.5 s the widest slot is 25 s, every pair is still at least +/// 2.5 s apart (above the 1-2 s burst width), and there is room for one more +/// loop before the spacing has to be revisited. Growing the list past twelve +/// needs a narrower spacing or a longer shortest cadence; the +/// `every_slot_fits_inside_the_shortest_cadence` test says which. +pub const SLOT_SPACING: Duration = Duration::from_millis(2500); /// This loop's offset into its period: slot `i` of [`LOOPS`] sits at /// `i ×` [`SLOT_SPACING`], wrapped into the period. @@ -320,13 +340,14 @@ mod tests { /// The node's real default cadences. Every one is a multiple of 30 s, /// which is the premise the separation argument rests on — if a default /// changes to something coprime, this list changing is the reminder. - const DEFAULT_PERIODS: [(&str, u64); 6] = [ + const DEFAULT_PERIODS: [(&str, u64); 7] = [ ("config_reconcile", 30), ("replication_reconcile", 30), ("scorer", 60), ("retention", 3600), ("federation_delivery", 30), ("mesh_config_effect", 60), + ("session_claims", 30), ]; /// The test that would have caught the fraction-based allocation: walk a @@ -339,18 +360,22 @@ mod tests { #[test] fn no_two_loops_ever_tick_together_at_the_default_periods() { const DAY: u64 = 24 * 60 * 60; + // In MILLISECONDS: the spacing is 2.5 s since 0.5.218, and a walk in + // whole seconds would truncate 2.5 s gaps to 2 and pass a 2 s floor + // that the real schedule never promised. let ticks = |name: &str, period: u64| -> Vec { - let phase = phase_for(name, Duration::from_secs(period)).as_secs(); + let phase = u64::try_from(phase_for(name, Duration::from_secs(period)).as_millis()) + .expect("phase fits"); (0..) - .map(|n| phase + n * period) - .take_while(|t| *t <= DAY) + .map(|n| phase + n * period * 1000) + .take_while(|t| *t <= DAY * 1000) .collect() }; let all: Vec<(&str, Vec)> = DEFAULT_PERIODS .iter() .map(|(n, p)| (*n, ticks(n, *p))) .collect(); - let floor = SLOT_SPACING.as_secs(); + let floor = u64::try_from(SLOT_SPACING.as_millis()).expect("spacing fits"); for (i, (a, ta)) in all.iter().enumerate() { for (b, tb) in all.iter().skip(i + 1) { let mut closest = u64::MAX; @@ -363,8 +388,8 @@ mod tests { } assert!( closest >= floor, - "{a} and {b} come within {closest}s of each other over a day; the slot \ - spacing is {floor}s" + "{a} and {b} come within {closest}ms of each other over a day; the slot \ + spacing is {floor}ms" ); } } diff --git a/src/media_gate.rs b/src/media_gate.rs index 9b54deb8..9c59a830 100644 --- a/src/media_gate.rs +++ b/src/media_gate.rs @@ -181,6 +181,13 @@ fn compatible(declared: &str, sniffed: &str) -> bool { } } +/// **How much of a file the write gate ever reads: its first 64 KiB.** +/// [`sniff`] looks at 2 KiB of it; the `text/*` UTF-8 / NUL check at all of +/// it. Named because a STREAMED upload peeks exactly this window before the +/// rest of the body goes to the seal (0.5.218) — a peek any shorter would +/// make the gate a different gate on the streaming path. +pub const FORMAT_HEAD_BYTES: usize = 64 * 1024; + /// The honesty check: parse the declared type and require the bytes to be /// what it says. `application/octet-stream` is an honest "I don't know" and /// is accepted for bytes this table does not recognise, never for bytes it @@ -188,6 +195,11 @@ fn compatible(declared: &str, sniffed: &str) -> bool { /// smuggled). `text/*` must be valid UTF-8 with no NUL. /// /// Returns the normalised essence to store. +/// +/// Reads at most the first [`FORMAT_HEAD_BYTES`] of `bytes` — so a streamed +/// upload (0.5.218, `drive::multipart`) runs the SAME gate by handing it only +/// the leading window it peeked, and the verdict is identical to the one a +/// whole buffer would get. pub fn check_format(declared: &str, bytes: &[u8]) -> Result { let essence = parse_essence(declared)?; let sniffed = sniff(bytes); @@ -208,11 +220,11 @@ pub fn check_format(declared: &str, bytes: &[u8]) -> Result }); } if essence.starts_with("text/") { - let head = &bytes[..bytes.len().min(64 * 1024)]; + let head = &bytes[..bytes.len().min(FORMAT_HEAD_BYTES)]; // A multi-byte sequence cut at the 64 KiB boundary is not a lie. let valid = match std::str::from_utf8(head) { Ok(_) => true, - Err(e) => e.error_len().is_none() && head.len() == 64 * 1024, + Err(e) => e.error_len().is_none() && head.len() == FORMAT_HEAD_BYTES, }; if !valid || head.contains(&0) { return Err(TypeRefusal::Mismatch { diff --git a/src/membership_invites.rs b/src/membership_invites.rs new file mode 100644 index 00000000..9e14be0a --- /dev/null +++ b/src/membership_invites.rs @@ -0,0 +1,1007 @@ +//! **Nobody joins without their own consent — the invite flow** (0.5.218; +//! the maintainer's ruling of 2026-09-30, CIRISConstitution#133, persist +//! v52.0.0 / CIRISPersist#955, edge v38.0.0's `membership` module; +//! `FSD/MEMBERSHIP_INVITES.md`). +//! +//! Until 0.5.218 every roster-growing door added a member on the EXISTING +//! members' authority alone, and from the ruling until this cut every such +//! door answered 409 `membership.consent_required` ("refuse, don't hold"). +//! This module is what replaces the refusal: the three signed rows of +//! persist's flow, driven over HTTP. +//! +//! ```text +//! inviter's node invitee's node group +//! POST …/{families,communities}/{id}/invites +//! membership:proposal:v1 ───────▶ GET /v1/self/invites +//! (the inviter's PERSON signs) POST /v1/self/invites/{p}/accept +//! membership:acceptance:v1 ──▶ (replicates back) +//! the WIDENING ◀─────────────────────────────────────────────── the inviter's node +//! (founder_only: on arrival, by edge's bridge with this node's +//! `membership_widener`, or here when a member lists the invites; +//! a quorum group: envelope → cosign → assemble, an `add`) +//! ``` +//! +//! # Where each rule lives (and where it does not) +//! +//! The ADMISSION rule is persist's, every word of it: a growth needs the +//! member's live acceptance of a live proposal (`check_growth_accepted`, on the +//! local door AND the replicated apply), a decline is terminal, expiry is +//! judged on the two signed instants, a founding record seats only its +//! signers, a supersede never adds. The ROWS are edge's (`membership::propose`, +//! `reply`, `widen_on_acceptance` — built through persist's own builders, so the +//! wire shape cannot drift). This module keeps NEITHER. It decides only HTTP +//! things: who may ask (the owner's session, never a delegate, for anything +//! that signs), which group the caller may see, and how a refusal is NAMED — +//! persist's eight rule tokens map one-to-one onto stable `membership.*` +//! reason ids in [`refused`], and nothing here re-derives a verdict persist +//! would give (the mirrored-rule class: one rule, one implementation). +//! +//! # Whose key signs what +//! +//! - The PROPOSAL is the inviter's person (`OwnerSignerCapsule::edge_signer`): +//! persist checks "a founder proposes under `founder_only`" against the +//! proposer's identity. +//! - The ACCEPTANCE / DECLINE is the invitee's person. persist would admit a +//! device acting for them (`signer_acts_for`); the person's own pen is used +//! because the acceptance is the person's consent, the same reason a consent +//! grant is authored by the human (`consent-is-authored-by-the-human`). +//! - The WIDENING is a founder's PERSON key — the roster's consensus counts raw +//! seat keys, so a device acting for the founder does not count. That is why +//! `compose` installs the owner's person pen as the bridge's +//! `membership_widener`, and why [`widen_held_acceptances`] takes the +//! caller's capsule, never the node signer. + +use std::collections::HashSet; +use std::sync::Arc; + +use axum::extract::{Path, State}; +use axum::http::{HeaderMap, StatusCode}; +use axum::response::{IntoResponse, Response}; +use axum::{Json, Router}; +use serde::Serialize; + +use ciris_edge::membership::{ + self as em, GroupScope, MembershipError, MembershipWidener, ACCEPTANCE_DIMENSION, + DECLINE_DIMENSION, PROPOSAL_DIMENSION, +}; +use ciris_persist::federation::types::{attestation_type, Attestation}; +use ciris_persist::federation::FederationDirectory; +use ciris_persist::prelude::Engine; + +use crate::family_api::{owner_caller, GateRefusal, OwnerCaller}; +use crate::owner_signer_capsule::{self, OwnerSignerCapsule}; + +/// How long an invitation lives when the inviter does not say: two weeks. +/// persist bounds it at 30 days (`MEMBERSHIP_PROPOSAL_MAX_TTL_SECS`); a +/// longer ask is refused by name here rather than clamped, so a client never +/// shows an expiry the row does not carry. +pub const DEFAULT_INVITE_DAYS: i64 = 14; + +/// Every `state` an invitation can be in, as [`InviteView::state`] reports it. +pub const INVITE_STATES: &[&str] = &[ + STATE_PENDING, + STATE_ACCEPTED, + STATE_JOINED, + STATE_DECLINED, + STATE_EXPIRED, + STATE_WITHDRAWN, +]; +/// Sent; the invitee has not answered and it is live. +pub const STATE_PENDING: &str = "pending"; +/// The invitee accepted; the group has not seated them yet (under a quorum +/// protocol: "accepted, awaiting the group" — persist FSD §4). +pub const STATE_ACCEPTED: &str = "accepted"; +/// Accepted AND active in the group's roster now. +pub const STATE_JOINED: &str = "joined"; +/// The invitee declined. Terminal for this invitation. +pub const STATE_DECLINED: &str = "declined"; +/// Lapsed unanswered. Terminal; inviting again is a new invitation. +pub const STATE_EXPIRED: &str = "expired"; +/// The inviter withdrew it before an answer. Terminal. +pub const STATE_WITHDRAWN: &str = "withdrawn"; + +// ─── Refusals ─────────────────────────────────────────────────────────────── +// +// One function per id, a string-literal id beside a one-sentence English text +// (the localization guard reads the pair, and an id must carry exactly one +// sentence). No id is ever built with `format!`. + +fn refuse(code: StatusCode, id: &'static str, text: &'static str, detail: String) -> Response { + ( + code, + Json(serde_json::json!({ "error": text, "reason_id": id, "detail": detail })), + ) + .into_response() +} + +pub(crate) fn invite_not_found(detail: String) -> Response { + refuse( + StatusCode::NOT_FOUND, + "membership.invite_not_found", + "That invitation isn't here. It may not have arrived yet, or it was never sent to you.", + detail, + ) +} + +fn not_the_invitee(detail: String) -> Response { + refuse( + StatusCode::FORBIDDEN, + "membership.not_the_invitee", + "Only the person who was invited can answer an invitation.", + detail, + ) +} + +pub(crate) fn not_the_proposer(detail: String) -> Response { + refuse( + StatusCode::FORBIDDEN, + "membership.not_the_proposer", + "Only the person who sent an invitation can withdraw it.", + detail, + ) +} + +pub(crate) fn bad_expiry(detail: String) -> Response { + refuse( + StatusCode::BAD_REQUEST, + "membership.bad_expiry", + "An invitation lasts between one and thirty days.", + detail, + ) +} + +pub(crate) fn invite_closed(detail: String) -> Response { + refuse( + StatusCode::CONFLICT, + "membership.invite_closed", + "That invitation is no longer open, so it can't be withdrawn.", + detail, + ) +} + +fn store_unavailable(detail: String) -> Response { + refuse( + StatusCode::SERVICE_UNAVAILABLE, + "membership.store_unavailable", + "The node could not read or write its membership records. Nothing was changed.", + detail, + ) +} + +fn signer_unavailable(detail: String) -> Response { + refuse( + StatusCode::FORBIDDEN, + "membership.signer_unavailable", + "Your identity could not be opened to sign this answer.", + detail, + ) +} + +fn session_required(code: StatusCode) -> Response { + refuse( + code, + "membership.owner_session_required", + "Invitations are answered by this node's owner. Sign in as the owner.", + String::new(), + ) +} + +fn delegate_may_not_answer() -> Response { + refuse( + StatusCode::FORBIDDEN, + "membership.delegate_may_not_answer", + "A delegated session can see invitations but can't answer them. Your answer is signed with your own key.", + String::new(), + ) +} + +/// **persist's refusal, named.** The eight rule tokens of CIRISPersist#955, +/// verbatim from edge's [`MembershipError::Refused`], each onto ONE stable +/// id. Retryable rules (`*_unresolved`) are 409s a client may retry; the rest +/// are terminal for that invitation. `detail` is the substrate's own sentence. +pub(crate) fn refused(e: &MembershipError) -> Response { + let detail = e.to_string(); + match e { + MembershipError::Refused { rule, .. } => match *rule { + em::RULE_ACCEPTANCE_UNRESOLVED => refuse( + StatusCode::CONFLICT, + "membership.awaiting_acceptance", + "They haven't accepted the invitation yet, so they can't join.", + detail, + ), + em::RULE_PROPOSAL_UNRESOLVED => refuse( + StatusCode::CONFLICT, + "membership.invite_not_here_yet", + "That invitation hasn't reached this device yet. Try again shortly.", + detail, + ), + em::RULE_DECLINED => refuse( + StatusCode::CONFLICT, + "membership.declined", + "They declined the invitation, so they can't be added under it.", + detail, + ), + em::RULE_PROPOSAL_EXPIRED => refuse( + StatusCode::GONE, + "membership.invite_expired", + "That invitation has expired. Send a new one.", + detail, + ), + em::RULE_ACCEPTANCE_MISMATCH => refuse( + StatusCode::FORBIDDEN, + "membership.acceptance_mismatch", + "That answer doesn't match the invitation it names.", + detail, + ), + em::RULE_REPLY_CONFLICT => refuse( + StatusCode::CONFLICT, + "membership.already_answered", + "That invitation was already answered the other way.", + detail, + ), + em::RULE_FOUNDING_MEMBER_UNSIGNED => founding_member_unsigned(detail), + em::RULE_SUPERSEDE_CANNOT_ADD => refuse( + StatusCode::CONFLICT, + "membership.supersede_cannot_add", + "A change to the group's record can't add anyone. New members join by invitation.", + detail, + ), + _ => membership_refused(detail), + }, + MembershipError::NotAProposal(_) => invite_not_found(detail), + MembershipError::Other(_) => membership_refused(detail), + } +} + +/// A founding roster names someone who did not sign it (persist Q1: signing +/// the founding record IS consent). Shared with the create routes, which +/// refuse the same thing BEFORE writing anything. +pub(crate) fn founding_member_unsigned(detail: String) -> Response { + refuse( + StatusCode::CONFLICT, + "membership.founding_member_unsigned", + "A group is founded by you alone. Invite the others once it exists.", + detail, + ) +} + +/// Any other refusal of a membership row (a proposer without standing, a +/// dissolved group, a malformed row) — persist's sentence in `detail`. +fn membership_refused(detail: String) -> Response { + refuse( + StatusCode::CONFLICT, + "membership.refused", + "The group's rules refused that membership change.", + detail, + ) +} + +/// persist's typed consent refusal out of a roster door (`add_member`, +/// `put_*_membership_widening`), as [`refused`] names it; `None` for any +/// other error, which the caller keeps naming its own way. +pub(crate) fn persist_refusal(e: &ciris_persist::federation::Error) -> Option { + match e { + ciris_persist::federation::Error::MembershipAcceptanceRefused { + group_key_id, + member_key_id, + rule, + } => Some(refused(&MembershipError::Refused { + group_key_id: group_key_id.clone(), + member_key_id: member_key_id.clone(), + rule, + })), + _ => None, + } +} + +// ─── The rows, read back ──────────────────────────────────────────────────── + +/// One invitation into a group, as the group's members see it. +#[derive(Debug, Clone, Serialize)] +pub struct InviteView { + pub proposal_id: String, + pub invitee_key_id: String, + pub role: Option, + pub proposer_key_id: String, + pub proposed_at: String, + pub expires_at: Option, + /// One of [`INVITE_STATES`]. + pub state: &'static str, + /// The acceptance or decline row, when there is one. + pub reply_id: Option, +} + +fn dimension_of(row: &Attestation) -> Option<&str> { + em::dimension_of(row) +} + +fn envelope_str<'a>(row: &'a Attestation, member: &str) -> Option<&'a str> { + row.attestation_envelope + .get(member) + .and_then(serde_json::Value::as_str) +} + +/// The group a membership row is placed at (`family_key_id` / +/// `community_key_id`, edge's [`GroupScope::target_member`]). +pub(crate) fn group_of(row: &Attestation, scope: GroupScope) -> Option<&str> { + envelope_str(row, scope.target_member()) +} + +/// Every proposal held here that invites someone into `group`, oldest first. +/// +/// persist keeps no index of proposals by group (the read arm it ships is by +/// INVITEE, which edge's `pending_proposals_for` walks), so this pages the +/// node's attestation log the way that helper does. A node holds the +/// proposals of the groups it is in plus the ones addressed to its owner, so +/// the walk is bounded by what this node already stores. +async fn proposals_into( + dir: &dyn FederationDirectory, + scope: GroupScope, + group: &str, +) -> Result, String> { + const PAGE: u32 = 512; + let mut out = Vec::new(); + let mut since = None; + loop { + let page = dir + .list_attestations_since(since.clone(), PAGE) + .await + .map_err(|e| format!("list_attestations_since: {e:#}"))?; + let full = page.len() == PAGE as usize; + since = page + .last() + .map(ciris_persist::federation::types::ServedAttestation::resume_pair); + for served in page { + let p = served.attestation; + if dimension_of(&p) == Some(PROPOSAL_DIMENSION) + && GroupScope::of_row(&p) == Some(scope) + && group_of(&p, scope) == Some(group) + { + out.push(p); + } + } + if !full { + break; + } + } + out.sort_by_key(|p| p.asserted_at); + Ok(out) +} + +/// **Who a pair room is waiting for** — the invitee of the newest proposal +/// into `room` that does not name `me`. Since edge v38 a pair room is founded +/// by its opener alone and the other person joins by accepting, so between +/// the two steps the FOLD names one person; the other is named only by the +/// invitation. `None` when nothing invites anyone into it. +pub(crate) async fn pending_pair_invitee( + dir: &dyn FederationDirectory, + room: &str, + me: &str, +) -> Option { + proposals_into(dir, GroupScope::Community, room) + .await + .ok()? + .into_iter() + .rev() + .filter_map(|p| p.subject_key_ids.first().cloned()) + .find(|k| k != me) +} + +/// The invitee's replies to `proposal_id`: (acceptance, decline). +async fn replies_to( + dir: &dyn FederationDirectory, + invitee: &str, + proposal_id: &str, +) -> Result<(Option, Option), String> { + let mut accepted = None; + let mut declined = None; + for r in dir + .list_attestations_for(invitee) + .await + .map_err(|e| format!("list_attestations_for({invitee}): {e:#}"))? + { + if envelope_str( + &r, + ciris_persist::federation::envelope::paths::REFERENCES_ATTESTATION_ID, + ) != Some(proposal_id) + { + continue; + } + match dimension_of(&r) { + Some(ACCEPTANCE_DIMENSION) => accepted = Some(r), + Some(DECLINE_DIMENSION) => declined = Some(r), + _ => {} + } + } + Ok((accepted, declined)) +} + +/// Has the proposer withdrawn `proposal_id` (a `withdraws` naming it)? +async fn withdrawn(dir: &dyn FederationDirectory, proposal_id: &str) -> Result { + Ok(dir + .list_attestations_referencing(proposal_id) + .await + .map_err(|e| format!("list_attestations_referencing({proposal_id}): {e:#}"))? + .iter() + .any(|r| r.attestation_type == attestation_type::WITHDRAWS)) +} + +/// **Every invitation into `group`, with its state** — `active` is the group's +/// roster by the fold (the caller's own read), so "joined" is a fact about the +/// roster and never inferred from the acceptance alone. +pub(crate) async fn group_invites( + dir: &dyn FederationDirectory, + scope: GroupScope, + group: &str, + active: &HashSet, +) -> Result)>, String> { + let now = chrono::Utc::now(); + let mut out = Vec::new(); + for p in proposals_into(dir, scope, group).await? { + let Some(invitee) = p.subject_key_ids.first().cloned() else { + continue; + }; + let (accepted, declined) = replies_to(dir, &invitee, &p.attestation_id).await?; + // An acceptance SIGNED after the proposal lapsed seats nobody: persist + // judges expiry on the two signed instants (`asserted_at` against + // `expires_at`), never on a reader's clock, and so does this view. + let accepted_in_time = accepted + .as_ref() + .is_some_and(|a| p.expires_at.is_none_or(|t| a.asserted_at <= t)); + let state = if active.contains(&invitee) && accepted.is_some() { + STATE_JOINED + } else if accepted_in_time { + STATE_ACCEPTED + } else if accepted.is_some() { + STATE_EXPIRED + } else if declined.is_some() { + STATE_DECLINED + } else if withdrawn(dir, &p.attestation_id).await? { + STATE_WITHDRAWN + } else if p.expires_at.is_some_and(|t| t <= now) { + STATE_EXPIRED + } else { + STATE_PENDING + }; + let reply_id = accepted + .as_ref() + .or(declined.as_ref()) + .map(|r| r.attestation_id.clone()); + out.push(( + InviteView { + proposal_id: p.attestation_id.clone(), + invitee_key_id: invitee, + role: envelope_str(&p, "role").map(str::to_owned), + proposer_key_id: p.attesting_key_id.clone(), + proposed_at: p.asserted_at.to_rfc3339(), + expires_at: p.expires_at.map(|t| t.to_rfc3339()), + state, + reply_id, + }, + accepted, + )); + } + Ok(out) +} + +/// **Seat every accepted, unseated invitee of `group`** — the widening edge's +/// bridge performs on an acceptance's arrival, re-attempted here for the +/// window the bridge cannot cover (a node claimed after its replication +/// runtime started has no widener until restart — see `compose`'s +/// `membership_widener` block) and for an acceptance that arrived before its +/// proposal did. Same edge call ([`em::widen_on_acceptance`]), same +/// idempotence (an active member is `AlreadyMember`, nothing written). +/// +/// Only a single-signature widening is attempted: under a quorum protocol +/// persist refuses it (the group's M-of-N lives on the growth row), and the +/// group seats the member through `…/changes/{envelope,cosign,assemble}` with +/// `action: add`. Those refusals are logged at debug and left for the quorum. +/// +/// Returns the members this call seated. +pub(crate) async fn widen_held_acceptances( + dir: &dyn FederationDirectory, + invites: &[(InviteView, Option)], + widener: &MembershipWidener, +) -> Vec { + let mut seated = Vec::new(); + for (view, acceptance) in invites { + let (STATE_ACCEPTED, Some(acceptance)) = (view.state, acceptance) else { + continue; + }; + match em::widen_on_acceptance(dir, acceptance, widener).await { + Ok(em::WidenOutcome::Widened { member_key_id, .. }) => { + tracing::info!( + member = %member_key_id, + proposal = %view.proposal_id, + "membership: an accepted invitee seated (CIRISPersist#955)" + ); + seated.push(member_key_id); + } + Ok(other) => tracing::debug!( + proposal = %view.proposal_id, outcome = ?other, + "membership: nothing to widen for this acceptance here" + ), + Err(e) => tracing::debug!( + proposal = %view.proposal_id, + rule = e.rule().unwrap_or("-"), + error = %e, + "membership: the single-signature widening was refused — under a quorum \ + protocol the group seats the member through the change flow" + ), + } + } + if !seated.is_empty() { + let _ = crate::compose::kick_replication("membership: accepted invitees seated"); + } + seated +} + +/// **The widener this node's runtime may not have** — seat every invitee who +/// accepted a proposal THIS node's owner (or this node, for them) issued. +/// +/// Edge's bridge widens on an acceptance's arrival only when the runtime was +/// built with a `membership_widener`, and the runtime is built once, at boot — +/// before the claim on every first-boot node (the production path, and the +/// native harness's). Measured on the v38 chat ladder: "membership widener NOT +/// installed … this node is not claimed", so a pair room's acceptance arrived +/// and nobody ever seated the joiner. This sweep is the same edge call +/// ([`em::widen_on_acceptance`]) with the owner's person pen opened on the +/// owner BINDING (`for_owned_node` — a loop's authority), run on the compose +/// loop's cadence: idempotent (a seated member is `AlreadyMember`, nothing +/// written), bounded by the owner's OWN proposals (`list_attestations_by`), +/// and silent on a node with no owner. A quorum group's single-signature +/// widening is refused by persist and left for the group's change flow. +/// +/// Returns the members it seated. +pub async fn widen_own_accepted_proposals(engine: &Arc, node_key_id: &str) -> Vec { + let Ok(capsule) = owner_signer_capsule::for_owned_node(engine, node_key_id).await else { + return Vec::new(); + }; + let dir = engine.federation_directory(); + let widener = MembershipWidener::new(vec![Arc::clone(capsule.edge_signer())]); + let now = chrono::Utc::now(); + let mut proposals = Vec::new(); + for author in [capsule.key_id().to_owned(), node_key_id.to_owned()] { + match dir.list_attestations_by(&author).await { + Ok(rows) => proposals.extend(rows.into_iter().filter(|p| { + dimension_of(p) == Some(PROPOSAL_DIMENSION) && p.expires_at.is_none_or(|t| t > now) + })), + Err(e) => { + tracing::debug!(author = %author, error = %e, "membership sweep: proposals unreadable"); + return Vec::new(); + } + } + } + // The joiner's half: a pair room this node's person ASKED to open, whose + // matching invitation has now arrived (`crate::pair_intents`). + if let Some((seed_dir, _)) = crate::node_key::held_user_seed_dir() { + let _ = crate::pair_intents::advance( + dir.as_ref(), + &seed_dir, + capsule.key_id(), + capsule.edge_signer(), + ) + .await; + } + let mut seated = Vec::new(); + for p in proposals { + let Some(invitee) = p.subject_key_ids.first() else { + continue; + }; + let Ok((Some(acceptance), _)) = replies_to(dir.as_ref(), invitee, &p.attestation_id).await + else { + continue; + }; + match em::widen_on_acceptance(dir.as_ref(), &acceptance, &widener).await { + Ok(em::WidenOutcome::Widened { + scope, + group_key_id, + member_key_id, + .. + }) => { + tracing::info!( + group = %group_key_id, member = %member_key_id, + "membership: an accepted invitee seated by this node's owner \ + (CIRISPersist#955; the runtime had no widener)" + ); + if scope == GroupScope::Family { + // The household's existing content, re-wrapped to them — + // what the pre-v52 direct add did, now after their consent. + if let Err(e) = engine + .rekey_family_member_add(&group_key_id, &member_key_id) + .await + { + tracing::warn!( + family = %group_key_id, member = %member_key_id, error = %e, + "membership: seated, but the DEK re-wrap failed" + ); + } + } + seated.push(member_key_id); + } + Ok(_) => {} + Err(e) => tracing::debug!( + proposal = %p.attestation_id, rule = e.rule().unwrap_or("-"), error = %e, + "membership sweep: not widened here" + ), + } + } + if !seated.is_empty() { + let _ = crate::compose::kick_replication("membership: accepted invitees seated"); + } + seated +} + +/// The invitation's `expires_at` from a client's `expires_in_days`. +#[allow(clippy::result_large_err)] +pub(crate) fn expiry(days: Option) -> Result, Response> { + let days = days.unwrap_or(DEFAULT_INVITE_DAYS); + let max_days = em::MEMBERSHIP_PROPOSAL_MAX_TTL_SECS / 86_400; + if !(1..=max_days).contains(&days) { + return Err(bad_expiry(format!( + "expires_in_days {days} is outside 1..={max_days} (persist bounds a proposal at \ + {} s)", + em::MEMBERSHIP_PROPOSAL_MAX_TTL_SECS + ))); + } + // A minute inside the bound: persist judges `expires_at − asserted_at` + // against the row's own stamp, which is taken a moment after this. + let bound = chrono::Duration::days(days).min( + chrono::Duration::seconds(em::MEMBERSHIP_PROPOSAL_MAX_TTL_SECS) + - chrono::Duration::minutes(1), + ); + Ok(chrono::Utc::now() + bound) +} + +/// The 202 an invitation answers with — `state: "invited"`, so no caller +/// mistakes an invitation for a membership (FSD §3). +pub(crate) fn invited( + scope: GroupScope, + group: &str, + proposal: &Attestation, + invitee: &str, + role: Option<&str>, +) -> Response { + ( + StatusCode::ACCEPTED, + Json(serde_json::json!({ + "state": "invited", + "proposal_id": proposal.attestation_id, + "group_kind": kind_token(scope), + "group_id": group, + "invitee_key_id": invitee, + "role": role, + "expires_at": proposal.expires_at.map(|t| t.to_rfc3339()), + })), + ) + .into_response() +} + +fn kind_token(scope: GroupScope) -> &'static str { + match scope { + GroupScope::Family => "family", + GroupScope::Community => "community", + } +} + +/// Withdraw a pending invitation: a `withdraws` of the proposal signed by its +/// proposer (persist rule 1 — a row is withdrawn by its own attester), which +/// persist reads as expiring it. The CALLER must be the proposer. +pub(crate) async fn withdraw( + dir: &dyn FederationDirectory, + scope: GroupScope, + group: &str, + proposal_id: &str, + capsule: &OwnerSignerCapsule, +) -> Response { + let proposal = match dir.get_attestation(proposal_id).await { + Ok(Some(p)) + if dimension_of(&p) == Some(PROPOSAL_DIMENSION) + && group_of(&p, scope) == Some(group) => + { + p + } + Ok(_) => { + return invite_not_found(format!("{proposal_id} is not an invitation into {group}")) + } + Err(e) => return store_unavailable(format!("get_attestation: {e:#}")), + }; + if proposal.attesting_key_id != capsule.edge_signer().key_id { + return not_the_proposer(format!( + "{proposal_id} was proposed by {}", + proposal.attesting_key_id + )); + } + let invitee = proposal + .subject_key_ids + .first() + .cloned() + .unwrap_or_default(); + match replies_to(dir, &invitee, proposal_id).await { + Ok((None, None)) => {} + Ok(_) => return invite_closed(format!("{proposal_id} was already answered")), + Err(e) => return store_unavailable(e), + } + match withdrawn(dir, proposal_id).await { + Ok(false) => {} + Ok(true) => return invite_closed(format!("{proposal_id} was already withdrawn")), + Err(e) => return store_unavailable(e), + } + let row = match ciris_edge::replication::attestation_bind::withdraws_attestation( + &proposal, + "invitation withdrawn", + chrono::Utc::now(), + capsule.edge_signer(), + ) + .await + { + Ok(r) => r, + Err(e) => return signer_unavailable(e), + }; + if let Err(e) = dir + .put_attestation_authored(ciris_persist::federation::SignedAttestation { + attestation: row.clone(), + }) + .await + { + return membership_refused(format!("withdraw {proposal_id}: {e:#}")); + } + let _ = crate::compose::kick_replication("membership: invitation withdrawn"); + Json(serde_json::json!({ + "state": STATE_WITHDRAWN, + "proposal_id": proposal_id, + "withdrawal_id": row.attestation_id, + })) + .into_response() +} + +// ─── The invitee's side: /v1/self/invites ─────────────────────────────────── + +#[derive(Clone)] +struct InboxState { + engine: Arc, + user_seed_dir: std::path::PathBuf, +} + +#[allow(clippy::result_large_err)] +fn gate(r: Result) -> Result { + r.map_err(|e| match e { + GateRefusal::NoSession => session_required(StatusCode::UNAUTHORIZED), + GateRefusal::NotOwner | GateRefusal::Unowned => session_required(StatusCode::FORBIDDEN), + GateRefusal::Delegated => delegate_may_not_answer(), + GateRefusal::Store(d) => store_unavailable(d), + }) +} + +async fn group_name( + dir: &dyn FederationDirectory, + scope: GroupScope, + group: &str, +) -> Option { + match scope { + GroupScope::Family => dir + .lookup_family(group) + .await + .ok() + .flatten() + .map(|f| f.family_name), + GroupScope::Community => dir + .lookup_community(group) + .await + .ok() + .flatten() + .map(|c| c.community_name), + } +} + +/// `GET /v1/self/invites` — the inbox: every live invitation held here that +/// names this node's owner and that they have not answered, across families, +/// communities and pair rooms (edge's `pending_proposals_for`). A delegate may +/// read it; only the owner answers. +async fn inbox(State(st): State, headers: HeaderMap) -> Response { + let caller = match gate(owner_caller(&st.engine, &headers, true).await) { + Ok(c) => c, + Err(r) => return r, + }; + let dir = st.engine.federation_directory(); + let pending = match em::pending_proposals_for(dir.as_ref(), &caller.owner_key_id).await { + Ok(p) => p, + Err(e) => return store_unavailable(e.to_string()), + }; + let mut invites = Vec::with_capacity(pending.len()); + for p in pending { + // Edge's inbox drops answered and lapsed proposals; a WITHDRAWN one + // (the proposer's `withdraws`, which persist reads as expiring it) is + // dropped here, so the invitee is never offered what is gone. + match withdrawn(dir.as_ref(), &p.proposal.attestation_id).await { + Ok(false) => {} + Ok(true) => continue, + Err(e) => return store_unavailable(e), + } + invites.push(serde_json::json!({ + "proposal_id": p.proposal.attestation_id, + "group_kind": kind_token(p.scope), + "group_id": p.group_key_id, + "group_name": group_name(dir.as_ref(), p.scope, &p.group_key_id).await, + "is_pair_room": p.group_key_id.starts_with(ciris_edge::chat::PAIR_COMMUNITY_PREFIX), + "role": p.role, + "proposer_key_id": p.proposal.attesting_key_id, + "proposed_at": p.proposal.asserted_at.to_rfc3339(), + "expires_at": p.expires_at.to_rfc3339(), + })); + } + Json(serde_json::json!({ + "invitee_key_id": caller.owner_key_id, + "invites": invites, + })) + .into_response() +} + +async fn answer(st: InboxState, headers: HeaderMap, proposal_id: String, accept: bool) -> Response { + let caller = match gate(owner_caller(&st.engine, &headers, false).await) { + Ok(c) => c, + Err(r) => return r, + }; + let dir = st.engine.federation_directory(); + let proposal = match dir.get_attestation(&proposal_id).await { + Ok(Some(p)) if dimension_of(&p) == Some(PROPOSAL_DIMENSION) => p, + Ok(_) => return invite_not_found(format!("{proposal_id} is not an invitation held here")), + Err(e) => return store_unavailable(format!("get_attestation: {e:#}")), + }; + if proposal.subject_key_ids.first().map(String::as_str) != Some(caller.owner_key_id.as_str()) { + return not_the_invitee(format!( + "{proposal_id} invites {:?}, not this node's owner", + proposal.subject_key_ids.first() + )); + } + // Do not ask the person to sign an answer to an invitation that has + // visibly lapsed or been withdrawn. This is not the admission rule — + // persist judges the signed instants wherever the rows land — only a + // refusal to author a row that can no longer seat anyone, named by the + // same id persist's rule maps to. + let lapsed = proposal.expires_at.is_some_and(|t| t <= chrono::Utc::now()); + let pulled = match withdrawn(dir.as_ref(), &proposal_id).await { + Ok(w) => w, + Err(e) => return store_unavailable(e), + }; + if lapsed || pulled { + let group = GroupScope::of_row(&proposal) + .and_then(|s| group_of(&proposal, s)) + .unwrap_or_default() + .to_owned(); + return refused(&MembershipError::Refused { + group_key_id: group, + member_key_id: caller.owner_key_id.clone(), + rule: em::RULE_PROPOSAL_EXPIRED, + }); + } + let capsule = match owner_signer_capsule::acquire( + &st.engine, + Some(&caller.bearer), + &caller.owner_key_id, + st.user_seed_dir.clone(), + ) + .await + { + Ok(c) => c, + Err(owner_signer_capsule::CapsuleRefusal::Delegated) => return delegate_may_not_answer(), + Err(e) => return signer_unavailable(e.to_string()), + }; + let reply = match em::reply(dir.as_ref(), &proposal_id, accept, capsule.edge_signer()).await { + Ok(r) => r, + Err(e) => return refused(&e), + }; + // A declined pair-room invitation ends the person's standing request for + // that room (`crate::pair_intents`): nothing may auto-accept a re-offer. + if !accept { + if let Some(room) = GroupScope::of_row(&proposal).and_then(|s| group_of(&proposal, s)) { + if let Err(e) = crate::pair_intents::forget(&st.user_seed_dir, room) { + tracing::warn!(room, error = %e, "pair intent could not be removed after a decline"); + } + } + } + let _ = crate::compose::kick_replication(if accept { + "membership: invitation accepted" + } else { + "membership: invitation declined" + }); + let scope = GroupScope::of_row(&proposal); + tracing::info!( + proposal = %proposal_id, + accepted = accept, + "membership: the invitee answered with their own signature (CIRISPersist#955)" + ); + Json(serde_json::json!({ + "state": if accept { STATE_ACCEPTED } else { STATE_DECLINED }, + "proposal_id": proposal_id, + "reply_id": reply.attestation_id, + "group_kind": scope.map(kind_token), + "group_id": scope.and_then(|s| group_of(&proposal, s)), + // Accepting is consent, not membership: the group seats the member + // (founder_only: the proposer's node on the acceptance's arrival; a + // quorum group: when M of N sign the add). + "awaiting": accept.then_some("the group's widening"), + })) + .into_response() +} + +async fn accept( + State(st): State, + headers: HeaderMap, + Path(proposal_id): Path, +) -> Response { + answer(st, headers, proposal_id, true).await +} + +async fn decline( + State(st): State, + headers: HeaderMap, + Path(proposal_id): Path, +) -> Response { + answer(st, headers, proposal_id, false).await +} + +/// The invitee's routes (FSD §3). The group-side routes live with their group +/// (`family_api`, `communities`), where membership of the group is decided. +pub fn router(engine: Arc, user_seed_dir: std::path::PathBuf) -> Router { + use axum::routing::{get, post}; + Router::new() + .route("/v1/self/invites", get(inbox)) + .route("/v1/self/invites/{proposal_id}/accept", post(accept)) + .route("/v1/self/invites/{proposal_id}/decline", post(decline)) + .with_state(InboxState { + engine, + user_seed_dir, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// persist's eight rule tokens each land on their OWN `membership.*` id — + /// none falls through to the catch-all, and no two share one. + #[tokio::test] + async fn every_persist_rule_has_its_own_id() { + let mut seen = HashSet::new(); + for rule in [ + em::RULE_ACCEPTANCE_UNRESOLVED, + em::RULE_PROPOSAL_UNRESOLVED, + em::RULE_DECLINED, + em::RULE_PROPOSAL_EXPIRED, + em::RULE_ACCEPTANCE_MISMATCH, + em::RULE_REPLY_CONFLICT, + em::RULE_FOUNDING_MEMBER_UNSIGNED, + em::RULE_SUPERSEDE_CANNOT_ADD, + ] { + let r = refused(&MembershipError::Refused { + group_key_id: "g".into(), + member_key_id: "k".into(), + rule, + }); + let body = axum::body::to_bytes(r.into_body(), usize::MAX) + .await + .expect("body"); + let v: serde_json::Value = serde_json::from_slice(&body).expect("json"); + let id = v["reason_id"].as_str().expect("reason_id").to_owned(); + assert!(id.starts_with("membership."), "{rule} -> {id}"); + assert_ne!( + id, "membership.refused", + "{rule} fell through to the catch-all" + ); + assert!(seen.insert(id.clone()), "{rule} shares {id}"); + } + } + + #[test] + fn an_expiry_outside_persists_bound_is_refused_by_name() { + assert!(expiry(Some(0)).is_err()); + assert!(expiry(Some(31)).is_err()); + let t = expiry(Some(30)).expect("thirty days is inside the bound"); + assert!( + (t - chrono::Utc::now()).num_seconds() < em::MEMBERSHIP_PROPOSAL_MAX_TTL_SECS, + "the stamp leaves room for the row's own asserted_at" + ); + assert!(expiry(None).is_ok()); + } +} diff --git a/src/mls_state.rs b/src/mls_state.rs index 731ef083..9b05e279 100644 --- a/src/mls_state.rs +++ b/src/mls_state.rs @@ -44,9 +44,11 @@ pub const RETAINED_EPOCHS: u64 = 16; /// readers state it. #[derive(Debug, Clone, PartialEq, Eq)] pub enum Posture { - /// Sealed on disk under persist's hardware-rooted key: groups survive a - /// restart. - Durable { path: String }, + /// Sealed on disk: groups survive a restart. `custody` is persist's word + /// for the key root (`hardware` | `software`, from the content-master row, + /// persist v50 #920): a host with no TPM now opens on disk under its + /// software master rather than falling back to memory. + Durable { path: String, custody: String }, /// In memory: a restart loses every group, and a restarted device rejoins /// by publishing a fresh KeyPackage (the rejoin rule, `Rejoin` at the /// holder). @@ -72,7 +74,12 @@ fn registry() -> &'static Mutex { /// store under the wire node is found there, and the same store is then /// registered under `node_key_id` too, so the rooms use it rather than a /// second store opened beside it. -pub async fn open_for_node(node_key_id: &str, wire_key_id: &str, path: &Path) -> Posture { +pub async fn open_for_node( + engine: &ciris_persist::Engine, + node_key_id: &str, + wire_key_id: &str, + path: &Path, +) -> Posture { use ciris_edge::mls::scope_state::{open_mls_state, MlsStateUnavailable}; set_claims_path(node_key_id, path.with_file_name(CLAIMS_FILE)); // A store the embedding host registered ON PURPOSE (an operator @@ -97,15 +104,19 @@ pub async fn open_for_node(node_key_id: &str, wire_key_id: &str, path: &Path) -> return posture; } } - let (store, posture) = match open_mls_state(path).await { - Ok(store) => { + let (store, posture) = match open_mls_state(engine, path).await { + Ok((store, custody)) => { + let custody_kind = custody.kind.as_str().to_owned(); let posture = Posture::Durable { path: path.display().to_string(), + custody: custody_kind.clone(), }; tracing::info!( path = %path.display(), + custody = %custody_kind, + descriptor = %custody.descriptor, "MLS state DURABLE — every room's group state is sealed on disk under persist's \ - hardware-rooted key; rooms survive a restart (CIRISServer#630)" + content master; rooms survive a restart (CIRISServer#630)" ); (store, posture) } @@ -396,6 +407,21 @@ mod tests { )) } + /// An in-memory engine for `open_for_node`, which reads persist's content + /// master to open the store (persist v50 #920). + async fn test_engine() -> ciris_persist::Engine { + let key = ed25519_dalek::SigningKey::from_bytes(&[0x51; 32]); + let signer = std::sync::Arc::new(ciris_persist::prelude::LocalSigner::from_parts( + key, + "mls-state-unit".to_string(), + None, + None, + )); + ciris_persist::Engine::with_signer(signer, "sqlite::memory:") + .await + .expect("in-memory engine") + } + /// A fresh directory: the claims file lands beside the store, so two tests /// must never share one. fn tmp_dir(name: &str) -> std::path::PathBuf { @@ -511,10 +537,16 @@ mod tests { disk_store(&path), Posture::Durable { path: path.display().to_string(), + custody: "software".to_owned(), }, ); - let posture = - open_for_node(&node, &node, &tmp_dir("never-opened").join("mls-state.kv")).await; + let posture = open_for_node( + &test_engine().await, + &node, + &node, + &tmp_dir("never-opened").join("mls-state.kv"), + ) + .await; assert!( matches!(posture, Posture::Durable { .. }), "the boot open kept the host's durable store: {posture:?}" @@ -559,10 +591,16 @@ mod tests { disk_store(&path), Posture::Durable { path: path.display().to_string(), + custody: "software".to_owned(), }, ); - let posture = - open_for_node(&actor, &wire, &tmp_dir("wire-open").join("mls-state.kv")).await; + let posture = open_for_node( + &test_engine().await, + &actor, + &wire, + &tmp_dir("wire-open").join("mls-state.kv"), + ) + .await; assert!(matches!(posture, Posture::Durable { .. }), "{posture:?}"); let room = "chat:room:v1:wire"; let _g = CohortGroup::create(store_for(&actor), room, &actor, RETAINED_EPOCHS) diff --git a/src/operator_surface.rs b/src/operator_surface.rs index f1653343..7089e5bd 100644 --- a/src/operator_surface.rs +++ b/src/operator_surface.rs @@ -220,7 +220,8 @@ impl WithholdClass { } } - /// Classify one of edge's withhold reasons. **Exhaustive by construction.** + /// Classify one of edge's withhold reasons. Every variant at the pinned edge + /// tag is named; an unknown one from a newer edge reads as a Fault. #[must_use] pub const fn of(reason: WithholdReason) -> Self { match reason { @@ -262,6 +263,20 @@ impl WithholdClass { // not in the record's own roster. Refusing is the feature. | WithholdReason::BlobArrivalScopeInsufficient | WithholdReason::BlobArrivalGroupMismatch + // edge v34.2.0 (CIRISEdge#718, CC 5.4.6 / CIRISConstitution#132) — + // a scoped body forwarded over the identity-plane link carries an + // in-link discriminator; one naming no derived address this node + // holds, or one present on a body that arrived ON a derived + // address, is the scope gate reaching a verdict. Refusing is the + // feature, beside the two arrival-scope arms above. + | WithholdReason::BlobDiscriminatorUnheld + | WithholdReason::BlobDiscriminatorOnDerivedAddress + // edge v37.0.0 (CIRISEdge#717) — a chunk fetch named a DAG the + // requester IS entitled to and a chunk that is not one of that + // DAG's chunks here: a request for content the scope gate never + // judged (another room's chunk named under this room's file). + // Refusing is the feature, the same verdict as the scope arms. + | WithholdReason::ChunkNotInNamedDag | WithholdReason::HoldingScopePeerNotInRoster // #169 LXMF — operator posture and advertised limits. Not a // propagation node; not holding mail for that destination; a @@ -281,7 +296,13 @@ impl WithholdClass { | WithholdReason::LxmfPeerSyncUnsupported | WithholdReason::LxmfFrameOversized | WithholdReason::LxmfMailboxFull - | WithholdReason::LxmfRetentionExpired => Self::Policy, + | WithholdReason::LxmfRetentionExpired + // edge v34.0.0 (CIRISEdge#682) — identity rows follow the node's + // announce state: an unannounced device's identity-plane rows are + // served to its owner's own nodes and to whoever holds a code, + // never LISTED. The peer is outside the audience the owner chose + // (the per-node announce ruling, CIRISServer#655) — a verdict. + | WithholdReason::IdentityRowNodeNotAnnounced => Self::Policy, // Fail-closed on a failed read, or a missing local wiring input. WithholdReason::LocalIdentityMissing | WithholdReason::SendSetUnresolved @@ -318,7 +339,12 @@ impl WithholdClass { // The requester's identity did not resolve, so there is no // destination to scope a mailbox to. Fail-closed and Red: the node // withheld without being able to establish who was asking. - | WithholdReason::LxmfRequesterUnidentified => Self::Fault, + | WithholdReason::LxmfRequesterUnidentified + // edge v34.0.0 (CIRISEdge#682) — the announce state could not be + // READ (`owner_of` ambiguous, or a directory fault); fail-closed to + // the node itself. A read that could not be made, beside the other + // unresolved reads: the operator is sent to the directory. + | WithholdReason::IdentityRowAnnounceUnresolved => Self::Fault, // Local state that cannot be put on the wire at all. WithholdReason::EnvelopeUnfetchable | WithholdReason::RowNotSerializable @@ -353,6 +379,13 @@ impl WithholdClass { | WithholdReason::HoldingScopePublicGroup // The bytes do not decode as the wire this endpoint speaks. | WithholdReason::LxmfWireUnparseable => Self::Integrity, + // edge v37.0.0 made `WithholdReason` `#[non_exhaustive]`: a newer + // edge may name a refusal this server has not classified yet. It is + // read as a FAULT (Red) — never folded quietly into Policy — so an + // unclassified withhold is the loudest thing on the surface until + // an arm above names it. Every variant edge ships at the pinned tag + // is named above; this arm is reached only across a version skew. + _ => Self::Fault, } } diff --git a/src/owner_signer_capsule.rs b/src/owner_signer_capsule.rs index ce97f50e..b68b2bd6 100644 --- a/src/owner_signer_capsule.rs +++ b/src/owner_signer_capsule.rs @@ -325,7 +325,44 @@ pub async fn for_owned_node( .into(), )); }; - let alias = crate::active_user_alias(&seed_dir, &default_alias); + open_owned(engine, &owner, seed_dir, &default_alias).await +} + +/// [`for_owned_node`] with the seed directory named by the caller — for a +/// loop that holds its router's own seed dir (the pair-room driver), so an +/// in-process node with no process-global registration still opens ITS +/// owner's pen. Same binding check, same refusals. +/// +/// # Errors +/// As [`for_owned_node`]. +pub async fn for_owned_node_in( + engine: &Arc, + node_key_id: &str, + seed_dir: std::path::PathBuf, +) -> Result { + let owner = match ciris_persist::federation::admission::owner_of( + engine.federation_directory().as_ref(), + node_key_id, + ) + .await + { + Ok(Some(owner)) => owner, + Ok(None) => return Err(CapsuleRefusal::Unowned), + Err(e) => return Err(CapsuleRefusal::Unavailable(e.to_string())), + }; + let default_alias = crate::node_key::held_user_seed_dir() + .map(|(_, a)| a) + .unwrap_or_default(); + open_owned(engine, &owner, seed_dir, &default_alias).await +} + +async fn open_owned( + engine: &Arc, + owner: &str, + seed_dir: std::path::PathBuf, + default_alias: &str, +) -> Result { + let alias = crate::active_user_alias(&seed_dir, default_alias); match crate::compose::resolve_user_signers( engine, crate::compose::FedIdUse::OwnerSession, @@ -334,7 +371,7 @@ pub async fn for_owned_node( ) .await { - Ok(Some((signer, edge_signer))) if crate::peer::signer_holds(&signer, &owner) => { + Ok(Some((signer, edge_signer))) if crate::peer::signer_holds(&signer, owner) => { Ok(OwnerSignerCapsule { signer, edge_signer, diff --git a/src/pair_intents.rs b/src/pair_intents.rs new file mode 100644 index 00000000..214f39f7 --- /dev/null +++ b/src/pair_intents.rs @@ -0,0 +1,264 @@ +//! **A pair room the person asked for — kept on disk** (0.5.218, edge v38 / +//! persist v52, CIRISPersist#955). +//! +//! Since edge v38 a pair room joins in two steps: the CREATOR (the smaller +//! fed-ID, `chat::PairRole`) founds it alone and proposes the other person; +//! the JOINER accepts. When the joiner's person calls `POST /v1/chat` for a +//! contact before the creator's invitation has reached their node, there is +//! nothing to accept yet — so the request is recorded here, and the +//! invitation is accepted when it lands (on the next read of the room, or the +//! compose loop's membership sweep). +//! +//! # Why this is consent, and the only thing it is consent to +//! +//! The maintainer's ruling (2026-09-30, CIRISConstitution#133): nobody joins a +//! family or community without their OWN consent, and a contact grant is not +//! it. Ruled on this path the same day: *the person's own `POST /v1/chat` for +//! that contact IS their act of consent to that pair room.* So an intent +//! records exactly that act — `{pair_id, contact_person, asked_at}` — and +//! [`advance`] accepts an invitation ONLY when it matches a recorded intent on +//! all three counts: the invitation is into that `pair_id`, it invites THIS +//! node's person, and its proposer resolves (`admission_identity_for_writer`) +//! to that `contact_person`. **Nothing else is ever auto-accepted** — not an +//! invitation into another room, not one from another person into the same +//! derived id, not a family or community invitation. Those wait in +//! `GET /v1/self/invites` for the person. +//! +//! # Why on disk, and the lifecycle +//! +//! Process memory forgot the request on a restart, and the person's act would +//! silently have to be repeated. The file mirrors `mls-claims.json` +//! (`mls_state`): one small JSON file in the node's user-seed directory, +//! written (temp + rename) BEFORE anything depends on it — `POST /v1/chat` +//! refuses rather than answer `awaiting_invitation` over an intent it could +//! not keep — and fallible by name. An intent is REMOVED once the room seats +//! this person, once the matching invitation is declined, expired or +//! withdrawn, or after [`MAX_AGE_DAYS`] (persist's longest proposal life) with +//! no invitation at all. It is not secret: it names two public fed-IDs. + +use std::path::{Path, PathBuf}; + +use serde::{Deserialize, Serialize}; + +use ciris_edge::membership::{self as em, GroupScope}; +use ciris_persist::federation::FederationDirectory; + +/// The sidecar's file name, in the node's user-seed directory. +pub const PAIR_INTENTS_FILE: &str = "pair-intents.json"; + +/// An intent no invitation answered in this long is dropped: persist bounds a +/// proposal's life at 30 days, so nothing older can still be answered. +pub const MAX_AGE_DAYS: i64 = 30; + +/// One request: this node's person asked to open `pair_id` with +/// `contact_person`. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct PairIntent { + pub pair_id: String, + pub contact_person: String, + pub asked_at: chrono::DateTime, +} + +fn path_in(seed_dir: &Path) -> PathBuf { + seed_dir.join(PAIR_INTENTS_FILE) +} + +/// Every recorded intent. An unreadable file is logged and read as none — +/// nothing is auto-accepted on a guess. +pub fn read(seed_dir: &Path) -> Vec { + let path = path_in(seed_dir); + match std::fs::read(&path) { + Ok(bytes) => serde_json::from_slice(&bytes).unwrap_or_else(|e| { + tracing::warn!(path = %path.display(), error = %e, "pair intents unreadable — nothing is auto-accepted until the person asks again"); + Vec::new() + }), + Err(_) => Vec::new(), + } +} + +fn write(seed_dir: &Path, intents: &[PairIntent]) -> Result<(), String> { + let path = path_in(seed_dir); + if intents.is_empty() { + return match std::fs::remove_file(&path) { + Ok(()) => Ok(()), + Err(e) if e.kind() == std::io::ErrorKind::NotFound => Ok(()), + Err(e) => Err(format!("remove pair intents at {}: {e}", path.display())), + }; + } + let bytes = serde_json::to_vec(intents).map_err(|e| format!("encode pair intents: {e}"))?; + let tmp = path.with_extension("json.tmp"); + std::fs::write(&tmp, bytes) + .and_then(|()| std::fs::rename(&tmp, &path)) + .map_err(|e| format!("write pair intents at {}: {e}", path.display())) +} + +/// Record that this node's person asked to open `pair_id` with +/// `contact_person`. Idempotent (the FIRST `asked_at` is kept). An `Err` +/// means the intent is not durable and the caller must not promise it. +pub fn record(seed_dir: &Path, pair_id: &str, contact_person: &str) -> Result<(), String> { + let mut intents = read(seed_dir); + if intents + .iter() + .any(|i| i.pair_id == pair_id && i.contact_person == contact_person) + { + return Ok(()); + } + intents.push(PairIntent { + pair_id: pair_id.to_owned(), + contact_person: contact_person.to_owned(), + asked_at: chrono::Utc::now(), + }); + write(seed_dir, &intents) +} + +/// Drop every intent for `pair_id`. +pub fn forget(seed_dir: &Path, pair_id: &str) -> Result<(), String> { + let mut intents = read(seed_dir); + let before = intents.len(); + intents.retain(|i| i.pair_id != pair_id); + if intents.len() == before { + return Ok(()); + } + write(seed_dir, &intents) +} + +fn drop_intent(seed_dir: &Path, pair_id: &str, why: &str) { + match forget(seed_dir, pair_id) { + Ok(()) => tracing::info!(pair_id, why, "pair intent removed"), + Err(e) => tracing::warn!(pair_id, why, error = %e, "pair intent could not be removed"), + } +} + +/// **Act on every recorded intent** for `me` (this node's person): accept a +/// MATCHING invitation (see the module docs — into that pair, for `me`, from +/// that contact's person), signed by `signer` (`me`'s person or a device +/// acting for them, which persist admits for a reply); remove the intent once +/// `me` is seated, once the matching invitation is declined / expired / +/// withdrawn, or once it is older than [`MAX_AGE_DAYS`]. Returns the rooms +/// whose invitation it accepted. +pub async fn advance( + dir: &dyn FederationDirectory, + seed_dir: &Path, + me: &str, + signer: &ciris_edge::identity::LocalSigner, +) -> Vec { + let now = chrono::Utc::now(); + let mut accepted = Vec::new(); + for intent in read(seed_dir) { + if now.signed_duration_since(intent.asked_at) > chrono::Duration::days(MAX_AGE_DAYS) { + drop_intent( + seed_dir, + &intent.pair_id, + "no invitation within a proposal's life", + ); + continue; + } + // An intent that does not name this pair's derived id for `me` is + // not this person's request (a hand-edited or foreign file). + if ciris_edge::chat::pair_community_key_id(me, &intent.contact_person) != intent.pair_id { + drop_intent(seed_dir, &intent.pair_id, "not this person's pair"); + continue; + } + let active: std::collections::HashSet = dir + .active_community_members(&intent.pair_id) + .await + .map(|v| v.into_iter().map(|m| m.key_id).collect()) + .unwrap_or_default(); + if active.contains(me) { + drop_intent(seed_dir, &intent.pair_id, "seated"); + continue; + } + let Ok(invites) = crate::membership_invites::group_invites( + dir, + GroupScope::Community, + &intent.pair_id, + &active, + ) + .await + else { + continue; + }; + let mut matching = Vec::new(); + for (view, _) in &invites { + if view.invitee_key_id != me { + continue; + } + let proposer = ciris_persist::federation::admission::admission_identity_for_writer( + dir, + &view.proposer_key_id, + ) + .await + .unwrap_or_default(); + if proposer == intent.contact_person { + matching.push(view.clone()); + } + } + if let Some(pending) = matching + .iter() + .rev() + .find(|v| v.state == crate::membership_invites::STATE_PENDING) + { + match em::reply(dir, &pending.proposal_id, true, signer).await { + Ok(_) => { + tracing::info!( + pair_id = %intent.pair_id, contact = %intent.contact_person, + "pair room: accepted the invitation the person asked for (their POST /v1/chat)" + ); + let _ = crate::compose::kick_replication("pair room invitation accepted"); + accepted.push(intent.pair_id.clone()); + } + Err(e) => { + tracing::debug!(pair_id = %intent.pair_id, error = %e, "pair intent: not accepted yet") + } + } + continue; + } + let answered_in_time = matching + .iter() + .any(|v| v.state == crate::membership_invites::STATE_ACCEPTED); + let closed = matching.iter().any(|v| { + [ + crate::membership_invites::STATE_DECLINED, + crate::membership_invites::STATE_EXPIRED, + crate::membership_invites::STATE_WITHDRAWN, + ] + .contains(&v.state) + }); + if closed && !answered_in_time { + drop_intent( + seed_dir, + &intent.pair_id, + "the invitation was declined, expired or withdrawn", + ); + } + } + accepted +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn an_intent_survives_a_reread_and_is_removed_by_name() { + let dir = std::env::temp_dir().join(format!( + "pair-intents-{}-{}", + std::process::id(), + uuid::Uuid::new_v4().simple() + )); + std::fs::create_dir_all(&dir).expect("dir"); + record(&dir, "chat:pair:v1:x", "bob").expect("record"); + record(&dir, "chat:pair:v1:x", "bob").expect("idempotent"); + record(&dir, "chat:pair:v1:y", "carol").expect("second"); + let got = read(&dir); + assert_eq!(got.len(), 2, "{got:?}"); + assert!(dir.join(PAIR_INTENTS_FILE).exists(), "durable on disk"); + forget(&dir, "chat:pair:v1:x").expect("forget"); + assert_eq!(read(&dir).len(), 1); + forget(&dir, "chat:pair:v1:y").expect("forget the last"); + assert!( + !dir.join(PAIR_INTENTS_FILE).exists(), + "no file once none remain" + ); + } +} diff --git a/src/peer.rs b/src/peer.rs index 44f71d32..a768b423 100644 --- a/src/peer.rs +++ b/src/peer.rs @@ -2357,12 +2357,15 @@ mod default_prefix_gate { fn default_covers_everything_edges_default_does() { let ours = default_attestation_prefixes(); for prefix in ciris_edge::replication::attestation_bind::DEFAULT_CONSENT_PREFIXES { - // A dimension inside `prefix`'s namespace and no other's — `covers` - // is `starts_with`, so this passes only if our set carries `prefix` - // itself or something strictly shorter that still opens the plane. - let probe = format!("{prefix}probe:v1"); + // The PREFIX half of persist's `covers`, which is byte-exact: our set + // carries `prefix` itself or something strictly shorter. Not a probe + // through `covers` — since persist v50 (#924) `covers` also requires + // the dimension to be a registered family member, and an invented + // `{prefix}probe:v1` is not one (it failed on `capacity:`, whose + // family has closed leaves). The grammar half is persist's, and says + // nothing about which prefixes this node grants. assert!( - ciris_persist::federation::consent_grammar::covers(&ours, &probe), + ours.iter().any(|p| prefix.starts_with(p.as_str())), "our default grant does not cover edge's `{prefix}` namespace. \ persist's `resolve_row_placement` — the single predicate for both \ passes of `promote_consented_backlog` — returns None for a dimension \ diff --git a/src/safety/age.rs b/src/safety/age.rs index b1b14539..8b4011bf 100644 --- a/src/safety/age.rs +++ b/src/safety/age.rs @@ -68,7 +68,7 @@ use ciris_persist::prelude::LocalSigner; pub const AGE_ASSURANCE_DIMENSION_PREFIX: &str = "age_assurance:"; /// The NON-reserved dimension prefix for a SUBJECT's SELF-DECLARED age band -/// (`age_self_declared:{band}`). Subject-signed; distinct from the witness- +/// (`age_self_declared:band:{band}`). Subject-signed; distinct from the witness- /// reserved provider prefix above (a self-declaration is not a provider /// attestation). This is the onboarding "state your age range" rung. pub const AGE_SELF_DECLARED_DIMENSION_PREFIX: &str = "age_self_declared:"; @@ -195,8 +195,15 @@ pub fn age_dimension(level: AssuranceLevel, band: AgeBand) -> String { // Every `scores` dimension MUST carry a `:vN` version segment (persist // `require_version_segment`, CEG §13.1). match level { + // The registered family is `age_self_declared:band:{band}:{version}` + // (CC namespace registry, strict since persist v50): the literal `band` + // segment is part of it. `age_self_declared:{band}:v1`, what this wrote + // before, is refused as an unregistered family. AssuranceLevel::SelfDeclared => { - format!("{AGE_SELF_DECLARED_DIMENSION_PREFIX}{}:v1", band.as_str()) + format!( + "{AGE_SELF_DECLARED_DIMENSION_PREFIX}band:{}:v1", + band.as_str() + ) } AssuranceLevel::Provider | AssuranceLevel::Government => format!( "{AGE_ASSURANCE_DIMENSION_PREFIX}{}:{}:v1", @@ -215,7 +222,10 @@ fn parse_age_dimension(dimension: &str) -> Option { .filter(|(_, v)| v.chars().all(|c| c.is_ascii_digit()) && !v.is_empty()) .map(|(head, _)| head) .unwrap_or(dimension); - if let Some(band_tok) = core.strip_prefix(AGE_SELF_DECLARED_DIMENSION_PREFIX) { + if let Some(rest) = core.strip_prefix(AGE_SELF_DECLARED_DIMENSION_PREFIX) { + // `band:{band}` since persist v50; a bare `{band}` on rows written + // before it, which a node may still hold and must still read. + let band_tok = rest.strip_prefix("band:").unwrap_or(rest); return Some(AgeAssurance { level: AssuranceLevel::SelfDeclared, band: AgeBand::from_token(band_tok)?, @@ -578,10 +588,13 @@ mod tests { fn dimension_roundtrips() { // Self-declared uses the non-reserved prefix (subject-signed), versioned. let d = age_dimension(AssuranceLevel::SelfDeclared, AgeBand::Adult); - assert_eq!(d, "age_self_declared:adult:v1"); + assert_eq!(d, "age_self_declared:band:adult:v1"); let a = parse_age_dimension(&d).expect("parse self"); assert_eq!(a.band, AgeBand::Adult); assert_eq!(a.level, AssuranceLevel::SelfDeclared); + // A row written before the registered spelling still reads. + let legacy = parse_age_dimension("age_self_declared:adult:v1").expect("parse legacy"); + assert_eq!(legacy.band, AgeBand::Adult); // Provider uses the witness-reserved `age_assurance:` prefix, versioned. let p = age_dimension(AssuranceLevel::Provider, AgeBand::Adult); assert_eq!(p, "age_assurance:provider:adult:v1"); diff --git a/src/self_devices.rs b/src/self_devices.rs index 7cec7482..cc3bc3fa 100644 --- a/src/self_devices.rs +++ b/src/self_devices.rs @@ -12,6 +12,9 @@ //! is that projection), and the node itself reverts to the unowned //! fail-closed floor. Releasing the node you are TALKING TO ends your own //! session's authority here, so it needs `force_self: true`. +//! Since 0.5.218 a release also revokes the node's content-KEM occurrence of +//! the owner (signed, replicating) — without it the released machine kept +//! being wrapped every new self file ([`evict_device`]). //! - `POST /v1/self/occurrence/label` — a display name for a device key. The //! persist occurrence row has no label member and its admission is //! idempotent on `(identity, occurrence)`, so a label cannot be written INTO @@ -24,6 +27,15 @@ //! any of their ANNOUNCED nodes with each node's transport key. The string //! (and its QR form) another person pastes into `POST /v1/contacts`. //! +//! - `POST /v1/self/occurrence/revoke` (0.5.218, CSD-037) — revoke a device +//! occurrence. Moved here from [`crate::auth::occurrence`], and it and +//! `release` are now ONE act, [`evict_device`]: withdraw the owner-binding(s) +//! the device is a node under, write a SIGNED occurrence revocation (the +//! replicating door, signed by the owner's pen), kick replication, and read +//! both halves back from persist. **Eviction withholds new content only**: +//! already-shared history stays readable by the evicted device — no DEK is +//! rotated and nothing is re-encrypted ([`EVICTION_HISTORY_NOTE`]). +//! //! - `POST /v1/self/nodes/{node_key_id}/announce` — announce ANOTHER of the //! owner's devices from the device holding the pen (CIRISServer#678): the //! owner re-signs the owner-binding user→that node at `federation`, the same @@ -85,7 +97,7 @@ fn session_required(code: StatusCode) -> Response { refuse( code, "self.owner_session_required", - "your devices are the node owner's own surface — sign in as the owner of this claimed node", + "Your devices are the owner's own surface. Sign in as the owner of this node.", ) } @@ -93,8 +105,7 @@ fn delegate_may_not_author() -> Response { refuse( StatusCode::FORBIDDEN, "self.delegate_may_not_author", - "a delegated session may not change the owner's devices — the change is signed with the \ - owner's own key, and that signature would outlive the delegation", + "A delegated session cannot change your devices. The change is signed with your own key, so sign in as yourself.", ) } @@ -102,7 +113,7 @@ fn store_unavailable(detail: String) -> Response { refuse_with( StatusCode::SERVICE_UNAVAILABLE, "self.store_unavailable", - "the identity store could not be read or written", + "The node could not read or write its identity store. Nothing was changed.", detail, ) } @@ -111,7 +122,7 @@ fn signer_unavailable(detail: String) -> Response { refuse_with( StatusCode::FORBIDDEN, "self.author_signer_unavailable", - "your federation identity could not be opened to sign this change", + "Your identity could not be opened to sign this change.", detail, ) } @@ -120,7 +131,7 @@ fn bad_request(detail: String) -> Response { refuse_with( StatusCode::BAD_REQUEST, "self.bad_request", - "the request body is not a valid device request", + "The node could not read that device request.", detail, ) } @@ -171,6 +182,401 @@ pub(crate) async fn emit_as_owner( .map_err(|e| format!("{e}")) } +// ─── EVICT A DEVICE — the one act both routes converge on ─────────────────── + +/// What eviction does NOT do, said in every answer and never softened +/// (CSD-037, CC 3.3.6.1). An eviction stops the device receiving anything NEW: +/// the next self file is wrapped to the owner's ACTIVE occurrences only, and the +/// self room removes the node at its next epoch. It does not reach back. Every +/// DEK the device was already granted stays granted — there is no DEK rotation +/// and no re-encryption of history — so whatever it could already read, it can +/// still read, on the device and in any copy taken from it. +pub const EVICTION_HISTORY_NOTE: &str = "Already-shared history stays readable by the evicted \ + device: nothing already granted to it is rotated or re-encrypted. Only new content and new \ + self-room epochs are withheld from it."; + +/// One part of an eviction that did not complete, by name. +#[derive(Debug, Clone, serde::Serialize)] +pub struct EvictFailure { + /// `owner_binding_withdrawal` | `owner_binding_witness` | + /// `occurrence_revocation` | `occurrence_witness`. + pub part: &'static str, + /// The binding, node or occurrence the part was about. + pub target: String, + /// The store's / signer's own words (`{:#}` — the whole chain). + pub error: String, +} + +/// **What an eviction did, part by part** — the answer both +/// `POST /v1/self/nodes/{node}/release` and `POST /v1/self/occurrence/revoke` +/// return, because they are now the same act. +#[derive(Debug, Default, serde::Serialize)] +pub struct EvictionReport { + /// The owner's nodes whose owner-bindings this eviction withdrew. + pub nodes: Vec, + /// One entry per withdrawn owner-binding: + /// `{node_key_id, binding, withdraws, cohort_scope}`. + pub withdrawn: Vec, + /// One entry per SIGNED occurrence revocation written: + /// `{occurrence_key_id, attesting_key_id, effective_at}`. + pub occurrences_revoked: Vec, + /// Occurrences of the owner the eviction named that were already out of the + /// active roster — nothing to sign for them. + pub occurrences_already_revoked: Vec, + /// Every part that did not complete, by name. Empty on a full eviction. + pub failed: Vec, + /// `true` when a replication round was kicked NOW; `false` when this + /// process runs no replication runtime (a bare harness engine), in which + /// case the rows ride the next cadence round of whatever runtime serves + /// this store. + pub replication_kicked: bool, + /// The owner's nodes as persist's projection lists them AFTER the act. + pub nodes_owned_by: Vec, + /// [`EVICTION_HISTORY_NOTE`], verbatim, in every answer. + pub history: &'static str, +} + +impl EvictionReport { + fn fail(&mut self, part: &'static str, target: &str, error: String) { + self.failed.push(EvictFailure { + part, + target: target.to_owned(), + error, + }); + } +} + +/// Build and sign ONE occurrence revocation as the owner, for persist's SIGNED +/// door (`put_identity_occurrence_revocation`, CIRISPersist#421). +/// +/// The envelope carries exactly the members the gate reads back +/// (`identity_key_id`, `occurrence_key_id`, `revoked_at`, `effective_at`) plus +/// the witness set and reason, so what is signed is what is stored. It is +/// canonicalized with persist's OWN canonicalizer — the bytes its gate +/// re-canonicalizes — and signed with the owner's pen, which is the identity +/// the occurrence belongs to: `signer_acts_for(owner, owner)` holds by +/// construction, and no other key has the mandate (a node key signing its +/// owner's revocation is the infrastructure-authors-the-human's-act class, +/// CC 3.3.6). +/// +/// The instant is NOT truncated. The fold revokes only when +/// `effective_at >= occurrence.asserted_at`, occurrences are asserted at +/// nanosecond precision, and RFC-3339 with nanoseconds round-trips exactly +/// through the gate's `ts_field` comparison, so a millisecond cut would only +/// open a window in which the revocation revokes nothing. +async fn signed_occurrence_revocation( + capsule: &OwnerSignerCapsule, + owner: &str, + occurrence: &str, + reason: Option<&str>, + now: chrono::DateTime, +) -> Result { + let at = now.to_rfc3339(); + let mut envelope = serde_json::json!({ + "identity_key_id": owner, + "occurrence_key_id": occurrence, + "revoked_at": at, + "effective_at": at, + "witness_set": [capsule.key_id()], + }); + if let Some(r) = reason { + envelope["reason"] = serde_json::Value::String(r.to_owned()); + } + let bytes = ciris_persist::verify::canonical::ceg_produce_canonicalize(&envelope) + .map_err(|e| format!("canonicalize the revocation envelope: {e:#}"))?; + let sig = capsule.sign_hybrid(&bytes).await?; + Ok( + ciris_persist::federation::SignedIdentityOccurrenceRevocation { + identity_occurrence_revocation: + ciris_persist::federation::types::IdentityOccurrenceRevocation { + identity_key_id: owner.to_owned(), + occurrence_key_id: occurrence.to_owned(), + revoked_at: now, + effective_at: now, + reason: reason.map(str::to_owned), + witness_set: vec![capsule.key_id().to_owned()], + persist_row_hash: String::new(), + }, + attesting_key_id: capsule.key_id().to_owned(), + signed_envelope: envelope, + signature: ciris_verify_core::transport_binding::TransportBindingSignature { + ed25519_signature_base64: B64.encode(&sig.classical_signature), + mldsa65_signature_base64: Some(B64.encode(&sig.pqc_signature)), + }, + }, + ) +} + +/// **Evict a device from its owner's self — ONE act** (CSD-037, the +/// stolen-device path; 0.5.218). +/// +/// # Why one act +/// +/// Until 0.5.218 there were two unconnected routes and neither finished the +/// job. `release` withdrew the owner-binding, so the self room dropped the node +/// on its next tick (edge `self_room::decide` reads `nodes_owned_by`) — but it +/// left the node's content-KEM identity OCCURRENCE live, and persist wraps every +/// new self file to `list_identity_occurrences_active(owner)` +/// (`at_rest_cascade::resolve_recipients`), so the "released" machine kept being +/// handed the key to every new file. `occurrence/revoke` wrote the revocation +/// through the trusted-LOCAL door (`put_identity_occurrence_revocation_local`): +/// unsigned, and EXCLUDED from replication by construction, so every other +/// device of the owner went on wrapping to the stolen one; and it never touched +/// the owner-binding, so the room kept it too. Each half looked done from the +/// device that ran it. +/// +/// # What it does +/// +/// 1. **Withdraw** every live owner-binding the owner holds on each of `nodes` +/// — the owner's signed `withdraws`, at the binding's own audience (a +/// federation binding is withdrawn where peers see it, a self one where the +/// owner's devices do). +/// 2. **Revoke** each of `occurrences` that is still active, through persist's +/// SIGNED door, signed by the owner's pen +/// ([`signed_occurrence_revocation`]). A signed revocation is what the +/// `IdentityOccurrenceRevocation` replication plane carries (`compose.rs`, +/// CIRISServer#646); an unsigned one never leaves this store. +/// 3. **Kick** replication so both ride this round, not the next cadence. +/// 4. **Witness** each half from persist's own projections rather than from our +/// writes: `nodes_owned_by(owner)` must no longer list the nodes, and +/// `list_identity_occurrences_active(owner)` must no longer list the +/// occurrences. +/// +/// Every part is attempted even when an earlier one failed — a half-eviction +/// that stops at the first error leaves the stolen device with the other half — +/// and every failure is reported by part and target, never folded into an +/// `Ok`. +/// +/// # What it does not do +/// +/// [`EVICTION_HISTORY_NOTE`]: no DEK is rotated and nothing already shared is +/// re-encrypted. Content the device was granted before the eviction stays +/// readable to it. Only NEW content (new self files are wrapped to active +/// occurrences only) and NEW self-room epochs are withheld. +pub(crate) async fn evict_device( + engine: &Engine, + capsule: &OwnerSignerCapsule, + owner: &str, + nodes: &[String], + occurrences: &[String], + reason: Option<&str>, +) -> EvictionReport { + let dir = engine.federation_directory(); + let mut report = EvictionReport { + nodes: nodes.to_vec(), + history: EVICTION_HISTORY_NOTE, + ..EvictionReport::default() + }; + + // (1) The owner-bindings. + for node in nodes { + let bindings = match live_bindings(engine, owner, node).await { + Ok(b) => b, + Err(e) => { + report.fail("owner_binding_withdrawal", node, e); + continue; + } + }; + for b in &bindings { + // At the BINDING's own audience: a federation-scoped binding is + // withdrawn where peers can see the withdrawal, a self-scoped one + // where the owner's devices can. + let spec = crate::attest::Spec::new( + attestation_type::WITHDRAWS, + b.cohort_scope.clone(), + ciris_persist::federation::withdraws_attestation_envelope( + &b.attestation_id, + attestation_type::DELEGATES_TO, + ), + ) + .about(node); + match emit_as_owner(engine, capsule, spec).await { + Ok(id) => report.withdrawn.push(serde_json::json!({ + "node_key_id": node, + "binding": b.attestation_id, + "withdraws": id, + "cohort_scope": b.cohort_scope, + })), + Err(e) => report.fail("owner_binding_withdrawal", &b.attestation_id, e), + } + } + } + + // (2) The occurrences — signed, so they replicate. + let active: Option> = + match dir.list_identity_occurrences_active(owner).await { + Ok(v) => Some(v.into_iter().map(|o| o.occurrence_key_id).collect()), + Err(e) => { + for occ in occurrences { + report.fail( + "occurrence_revocation", + occ, + format!("list_identity_occurrences_active({owner}): {e:#}"), + ); + } + None + } + }; + if let Some(active) = &active { + for occ in occurrences { + if !active.contains(occ) { + report.occurrences_already_revoked.push(occ.clone()); + continue; + } + let now = chrono::Utc::now(); + let signed = match signed_occurrence_revocation(capsule, owner, occ, reason, now).await + { + Ok(s) => s, + Err(e) => { + report.fail("occurrence_revocation", occ, e); + continue; + } + }; + match dir.put_identity_occurrence_revocation(signed).await { + Ok(()) => report.occurrences_revoked.push(serde_json::json!({ + "occurrence_key_id": occ, + "attesting_key_id": capsule.key_id(), + "effective_at": now.to_rfc3339(), + })), + Err(e) => report.fail( + "occurrence_revocation", + occ, + format!("put_identity_occurrence_revocation: {e:#}"), + ), + } + } + } + + // (3) Make both halves cross now. + report.replication_kicked = crate::compose::kick_replication("self:evict_device"); + + // (4) The witnesses, read back rather than assumed. + match nodes_owned_by(dir.as_ref(), owner).await { + Ok(still) => { + for node in nodes { + if still.iter().any(|n| n == node) { + report.fail( + "owner_binding_witness", + node, + "nodes_owned_by still lists it: an owner-binding this node cannot see is \ + still live" + .to_owned(), + ); + } + } + report.nodes_owned_by = still; + } + Err(e) => report.fail( + "owner_binding_witness", + owner, + format!("nodes_owned_by: {e:#}"), + ), + } + if !occurrences.is_empty() { + match dir.list_identity_occurrences_active(owner).await { + Ok(v) => { + for occ in occurrences { + if v.iter().any(|o| &o.occurrence_key_id == occ) { + report.fail( + "occurrence_witness", + occ, + "list_identity_occurrences_active still lists it".to_owned(), + ); + } + } + } + Err(e) => report.fail( + "occurrence_witness", + owner, + format!("list_identity_occurrences_active: {e:#}"), + ), + } + } + tracing::info!( + owner = %owner, + nodes = ?nodes, + occurrences = ?occurrences, + withdrawn = report.withdrawn.len(), + revoked = report.occurrences_revoked.len(), + failed = report.failed.len(), + replication_kicked = report.replication_kicked, + "self: device EVICTED — owner-binding(s) withdrawn and occurrence(s) revoked as ONE \ + signed act; already-shared history stays readable to the device (no DEK rotation)" + ); + report +} + +/// The answer an eviction route gives: 200 with the report when every part +/// completed, else a refusal naming what did not — carrying the same report, +/// so the caller sees what DID complete (it stays done). +fn eviction_response(report: &EvictionReport, extra: serde_json::Value) -> Response { + let mut body = serde_json::to_value(report).unwrap_or_else(|_| serde_json::json!({})); + if let (Some(b), Some(x)) = (body.as_object_mut(), extra.as_object()) { + for (k, v) in x { + b.insert(k.clone(), v.clone()); + } + } + if report.failed.is_empty() { + return Json(body).into_response(); + } + // A binding persist still sees as live, and nothing else wrong, is the + // release's own long-standing refusal; any other incomplete part is named + // under the eviction's. + let only_binding_witness = report + .failed + .iter() + .all(|f| f.part == "owner_binding_witness"); + let (id, text): (&'static str, &'static str) = if only_binding_witness { + ( + "self.release_incomplete", + "The release was signed, but the node still lists that node as yours: an ownership record it cannot see is still live. What it could withdraw stays withdrawn.", + ) + } else { + ( + "self.evict_incomplete", + "Part of removing that device did not complete. What was done stays done; the answer names what was not.", + ) + }; + let detail = report + .failed + .iter() + .map(|f| format!("{} {}: {}", f.part, f.target, f.error)) + .collect::>() + .join("; "); + if let Some(b) = body.as_object_mut() { + b.insert("error".into(), text.into()); + b.insert("reason_id".into(), id.into()); + b.insert("detail".into(), detail.into()); + } + (StatusCode::INTERNAL_SERVER_ERROR, Json(body)).into_response() +} + +/// Every key the node `node` may hold its owner's content occurrence under. +/// For another machine that is its node key: since the split fix a claim +/// provisions the content occurrence under the wire key +/// (`ensure_content_occurrence` takes it explicitly), and a pre-split install +/// has one key. For THIS node it is every key this node is — a split install +/// that predates that fix still holds its occurrence under the ACTOR key +/// (`actor-vs-node-key-on-a-split-install`; [`crate::peer::own_keys_of_this_node`]). +/// +/// What this cannot see: another machine's ACTOR key when that machine is a +/// pre-fix split install. Nothing in the directory links a remote node key to +/// its actor key, so that occurrence is not found from here; revoking it by +/// its own key (`POST /v1/self/occurrence/revoke`) reaches it. +async fn occurrence_keys_of_node(engine: &Engine, node: &str, is_this_node: bool) -> Vec { + let mut keys = vec![node.to_owned()]; + if is_this_node { + if let Ok(engine_key) = engine.local_derived_key_id().await { + for k in crate::peer::own_keys_of_this_node(&engine_key) { + if !keys.contains(&k) { + keys.push(k); + } + } + } + } + keys +} + // ─── POST /v1/self/nodes/{node_key_id}/release ────────────────────────────── #[derive(Debug, Default, Deserialize)] @@ -218,6 +624,11 @@ async fn live_bindings( .collect()) } +/// **Release a node — an eviction** (0.5.218). The owner's signed `withdraws` +/// of every owner-binding on the node AND the signed revocation of the node's +/// content-KEM occurrence of the owner, through [`evict_device`]. Before +/// 0.5.218 a release withdrew the binding only, and the released machine kept +/// being wrapped every new self file. async fn release_node( State(st): State, headers: HeaderMap, @@ -248,7 +659,7 @@ async fn release_node( return refuse_with( StatusCode::FORBIDDEN, "self.not_your_node", - "that node is not one you own, so it is not yours to release", + "That node is not one you own, so it is not yours to release.", node_key_id, ) } @@ -261,78 +672,183 @@ async fn release_node( return refuse( StatusCode::CONFLICT, "self.release_self_requires_force", - "that is the node you are talking to — releasing it ends your ownership here, \ - including this session. Send force_self: true to do it anyway", + "That is the node you are talking to. Releasing it ends your ownership here, including this session.", ); } - let bindings = match live_bindings(&st.engine, &caller.owner_key_id, &node_key_id).await { - Ok(b) => b, - Err(e) => return store_unavailable(e), + // The node's content occurrence(s) of the owner: the keys it may hold one + // under, filtered to those that ARE occurrences of the owner — a node that + // never provisioned one has nothing to revoke, which is not a failure. + let occurrences = match dir + .list_identity_occurrences_for(&caller.owner_key_id) + .await + { + Ok(all) => { + let keys = occurrence_keys_of_node(&st.engine, &node_key_id, is_this_node).await; + let mut occ: Vec = all + .into_iter() + .map(|o| o.occurrence_key_id) + .filter(|k| keys.contains(k)) + .collect(); + occ.sort(); + occ.dedup(); + occ + } + Err(e) => return store_unavailable(format!("list_identity_occurrences_for: {e:#}")), }; let capsule = match pen(&st, &caller).await { Ok(c) => c, Err(r) => return r, }; - let mut withdrawn = Vec::with_capacity(bindings.len()); - for b in &bindings { - // At the BINDING's own audience: a federation-scoped binding is - // withdrawn where peers can see the withdrawal, a self-scoped one where - // the owner's devices can. - let spec = crate::attest::Spec::new( - attestation_type::WITHDRAWS, - b.cohort_scope.clone(), - ciris_persist::federation::withdraws_attestation_envelope( - &b.attestation_id, - attestation_type::DELEGATES_TO, - ), + let report = evict_device( + &st.engine, + &capsule, + &caller.owner_key_id, + std::slice::from_ref(&node_key_id), + &occurrences, + Some("released by its owner"), + ) + .await; + tracing::info!( + owner = %caller.owner_key_id, node = %node_key_id, + bindings = report.withdrawn.len(), occurrences = report.occurrences_revoked.len(), + failed = report.failed.len(), released_self = is_this_node, + "self: node RELEASED — the owner withdrew every owner-binding on it and revoked its \ + content occurrence" + ); + let released = report.failed.is_empty(); + eviction_response( + &report, + serde_json::json!({ + "node_key_id": node_key_id, + "released": released, + "released_self": is_this_node, + }), + ) +} + +// ─── POST /v1/self/occurrence/revoke ──────────────────────────────────────── + +#[derive(Debug, Deserialize)] +struct RevokeOccurrenceRequest { + /// The identity the occurrence belongs to — must be the caller's own. + identity_key_id: String, + /// The occurrence (device) to remove from the self. + occurrence_key_id: String, + /// Optional annotation (e.g. "laptop lost 2026-06-23"), carried in the + /// signed revocation envelope. + #[serde(default)] + reason: Option, + /// Required when the occurrence is (one of the keys of) the node this + /// request is served BY: evicting it withdraws this node's owner-binding + /// and ends the owner's authority here, as a self-release does. + #[serde(default)] + force_self: bool, +} + +/// **Revoke a device occurrence — an eviction** (0.5.218, CSD-037). +/// +/// Authorised like [`release_node`]: the OWNER's session on this node, and the +/// revocation is signed server-side with the owner's pen. Before 0.5.218 this +/// route wanted a request hybrid-signed by a surviving device key, which the +/// app (a bearer client) never sends — every revoke from the app answered 401 — +/// and then wrote the revocation through the trusted-LOCAL door, unsigned and +/// excluded from replication (the #227 S2 carriage gap). Now it is the signed, +/// replicating door, and when the occurrence IS one of the owner's nodes the +/// same act withdraws that node's owner-binding, so the self room drops it too +/// ([`evict_device`]). +/// +/// Refused: no session (401), a session that is not the owner's, a delegated +/// session (the row is signed with the owner's key and would outlive the +/// delegation), and an identity or occurrence that is not the caller's — under +/// ONE id (`self.not_your_device`) whether the key is someone else's or nobody's. +async fn revoke_occurrence( + State(st): State, + headers: HeaderMap, + body: Bytes, +) -> Response { + let caller = match gate(owner_caller(&st.engine, &headers, false).await) { + Ok(c) => c, + Err(r) => return r, + }; + let req: RevokeOccurrenceRequest = match serde_json::from_slice(&body) { + Ok(r) => r, + Err(e) => return bad_request(e.to_string()), + }; + let not_yours = |k: &str| { + refuse_with( + StatusCode::NOT_FOUND, + "self.not_your_device", + "That key is not one of your devices.", + k.to_owned(), ) - .about(&node_key_id); - match emit_as_owner(&st.engine, &capsule, spec).await { - Ok(id) => withdrawn.push(serde_json::json!({ - "binding": b.attestation_id, - "withdraws": id, - "cohort_scope": b.cohort_scope, - })), - Err(e) => { - return store_unavailable(format!( - "withdraws(owner-binding {}): {e}", - b.attestation_id - )) - } - } + }; + if req.identity_key_id != caller.owner_key_id { + return not_yours(&req.occurrence_key_id); + } + let dir = st.engine.federation_directory(); + // One of the caller's occurrences, active or not: an occurrence revoked + // under the old LOCAL door is still the owner's, and re-running the + // eviction on it withdraws whatever owner-binding that door left behind. + match dir + .list_identity_occurrences_for(&caller.owner_key_id) + .await + { + Ok(v) + if v.iter() + .any(|o| o.occurrence_key_id == req.occurrence_key_id) => {} + Ok(_) => return not_yours(&req.occurrence_key_id), + Err(e) => return store_unavailable(format!("list_identity_occurrences_for: {e:#}")), } - // THE WITNESS, read back rather than assumed: the projection every other - // surface (the switcher, the self room's roster) reads must no longer list it. - let still = match nodes_owned_by(dir.as_ref(), &caller.owner_key_id).await { + // Is the device one of the owner's NODES? Directly (its occurrence is under + // its node key), or as one of THIS node's keys (a split install's actor + // occurrence of a node bound under its wire key). + let owned = match nodes_owned_by(dir.as_ref(), &caller.owner_key_id).await { Ok(v) => v, Err(e) => return store_unavailable(format!("nodes_owned_by: {e:#}")), }; - if still.iter().any(|n| n == &node_key_id) { - return refuse_with( - StatusCode::INTERNAL_SERVER_ERROR, - "self.release_incomplete", - "the release was signed but the node is still listed as yours — a binding this node \ - cannot see is still live", - format!( - "withdrew {} binding(s); nodes_owned_by still lists it", - withdrawn.len() - ), + let this_keys = occurrence_keys_of_node(&st.engine, &caller.node_key_id, true).await; + let mut nodes: Vec = Vec::new(); + if owned.contains(&req.occurrence_key_id) { + nodes.push(req.occurrence_key_id.clone()); + } + if this_keys.contains(&req.occurrence_key_id) + && owned.contains(&caller.node_key_id) + && !nodes.contains(&caller.node_key_id) + { + nodes.push(caller.node_key_id.clone()); + } + let is_this_node = nodes.iter().any(|n| this_keys.contains(n)); + if is_this_node && !req.force_self { + return refuse( + StatusCode::CONFLICT, + "self.release_self_requires_force", + "That is the node you are talking to. Releasing it ends your ownership here, including this session.", ); } - tracing::info!( - owner = %caller.owner_key_id, node = %node_key_id, bindings = withdrawn.len(), - released_self = is_this_node, - "self: node RELEASED — the owner withdrew every owner-binding on it" - ); - let _ = crate::compose::kick_replication("self:release_node"); - Json(serde_json::json!({ - "node_key_id": node_key_id, - "released": true, - "released_self": is_this_node, - "withdrawn": withdrawn, - "nodes_owned_by": still, - })) - .into_response() + let capsule = match pen(&st, &caller).await { + Ok(c) => c, + Err(r) => return r, + }; + let report = evict_device( + &st.engine, + &capsule, + &caller.owner_key_id, + &nodes, + std::slice::from_ref(&req.occurrence_key_id), + req.reason.as_deref(), + ) + .await; + let revoked = report.failed.is_empty(); + eviction_response( + &report, + serde_json::json!({ + "identity_key_id": req.identity_key_id, + "occurrence_key_id": req.occurrence_key_id, + "revoked": revoked, + "revoked_by": capsule.key_id(), + "released_self": is_this_node, + }), + ) } // ─── POST /v1/self/nodes/{node_key_id}/announce ───────────────────────────── @@ -496,7 +1012,7 @@ async fn label_occurrence( return refuse( StatusCode::BAD_REQUEST, "self.label_empty", - "a device label must be between 1 and 64 characters", + "A device name must be between 1 and 64 characters.", ); } // The device must be one of the CALLER's occurrences (active or not — a @@ -515,7 +1031,7 @@ async fn label_occurrence( return refuse_with( StatusCode::NOT_FOUND, "self.not_your_device", - "that key is not one of your devices", + "That key is not one of your devices.", req.occurrence_key_id, ); } @@ -627,8 +1143,7 @@ async fn contact_code( return refuse_with( StatusCode::CONFLICT, "self.contact_code_owner_key_absent", - "this node does not hold your federation key record, so it cannot build your \ - contact code", + "This node does not hold your identity's key record, so it can't make your contact code.", owner, ) } @@ -641,8 +1156,7 @@ async fn contact_code( return refuse_with( StatusCode::CONFLICT, "self.contact_code_not_a_person", - "your federation key is not registered as a person (user), and only a person can \ - be added as a contact", + "Your identity is not registered as a person, and only a person can be added as a contact.", record.identity_type, ); } @@ -661,8 +1175,7 @@ async fn contact_code( return refuse_with( StatusCode::CONFLICT, "self.contact_code_key_not_derived", - "your federation key id is not derived from its public key, so a contact code for \ - it would be refused by everyone you share it with as an impersonation", + "Your identity's key id does not match its key, so everyone you shared a code with would refuse it.", owner, ); } @@ -676,8 +1189,7 @@ async fn contact_code( return refuse_with( StatusCode::CONFLICT, "self.contact_code_no_pqc_half", - "your federation key record carries no ML-DSA-65 half, so a contact code could not \ - commit to it and no one could admit you from it", + "Your identity has no post-quantum key half, so no one could add you from a code.", owner, ); }; @@ -716,9 +1228,7 @@ async fn contact_code( return refuse_with( StatusCode::BAD_REQUEST, "self.node_not_announced", - "a contact code can carry only nodes you announced — announce a node on \ - that node (its wizard's announce step, or POST /v1/federation/announce) \ - before sharing it", + "That device isn't reachable, so it can't go in a code.", refused .iter() .map(|k| k.as_str()) @@ -819,7 +1329,7 @@ async fn contact_code( return refuse_with( StatusCode::CONFLICT, "self.contact_code_unencodable", - "your contact code could not be encoded from the keys this node holds", + "Your contact code could not be made from the keys this node holds.", format!("{e}"), ) } @@ -865,6 +1375,13 @@ pub fn router(engine: Arc, user_seed_dir: std::path::PathBuf) -> Router "/v1/self/occurrence/label", axum::routing::post(label_occurrence), ) + // 0.5.218 (CSD-037): moved here from `auth::occurrence` — the revoke is + // the owner's act, authorised by the owner's session and signed with + // the owner's pen, which is this router's posture, not that one's. + .route( + "/v1/self/occurrence/revoke", + axum::routing::post(revoke_occurrence), + ) // CIRISServer#673 — the person's shareable contact code (and QR form). .route("/v1/self/contact-code", axum::routing::get(contact_code)) .with_state(SelfState { diff --git a/src/self_rewrap.rs b/src/self_rewrap.rs index 58d5dfe1..e0d598ca 100644 --- a/src/self_rewrap.rs +++ b/src/self_rewrap.rs @@ -29,13 +29,29 @@ //! second device itself) does nothing here and says so at debug; the device //! that approved the claim holds the pen and does the work. Never a machine //! key: `for_owned_node` refuses rather than falling back to one. +//! +//! # One device re-wraps (CC 3.1.3.1, 0.5.218) +//! +//! "The device holding the pen" is not one device once the person's fed-ID +//! is portable: every device they signed in on holds it, every one of them saw +//! the new occurrence arrive, and every one re-wrapped the same blobs and +//! wrote its own key-grant set for one grant. The re-wrap for a new device is +//! therefore an EXCHANGE — `(self room, self_rewrap:)` — and only +//! the device [`crate::session_claims::gate`] names does it. Per occurrence, +//! because the new device can never re-wrap for itself (it holds none of the +//! old DEKs), so it must never be the one the fold names; it is also never +//! pending for itself here, so it never claims. The gate runs AFTER the pen +//! check: a device that cannot do the work must not take the duty. And an +//! unclaimed re-wrap waits — for the person to be on a device that holds the +//! pen — rather than run on whichever device ticks first. -use std::collections::HashSet; -use std::sync::{Arc, Mutex, OnceLock}; +use std::sync::Arc; -use ciris_persist::federation::admission::{nodes_owned_by, owner_of}; +use ciris_persist::federation::admission::nodes_owned_by; use ciris_persist::prelude::Engine; +use crate::session_claims::{self, Attendance, Occupant, Verdict}; + /// What one pass did — returned so a test can read it; the log carries the /// same facts for an operator. #[derive(Debug, Clone, Default, PartialEq, Eq)] @@ -49,37 +65,44 @@ pub struct RewrapReport { /// `true` when there was something to do and this node does not hold the /// owner's pen, so nothing ran here. pub no_pen_here: bool, + /// `(occurrence, handler)` for every pending re-wrap this device did NOT + /// run because the session gate said so: `Some(device)` = that device + /// does it, `None` = unclaimed, nobody does it yet (CC 3.1.3.1). + pub not_handled_here: Vec<(String, Option)>, } -/// `owner \0 occurrence \0 x25519` — a new KEM key for the same occurrence is -/// a new wrap target and is re-wrapped again. -fn memo() -> &'static Mutex> { - static MEMO: OnceLock>> = OnceLock::new(); - MEMO.get_or_init(|| Mutex::new(HashSet::new())) +/// `rewrap \0 owner \0 occurrence \0 x25519` — the ACT's id for the +/// idempotence ledger (`Attendance::record_act`). A new KEM key for the same +/// occurrence is a new wrap target and is re-wrapped again. +fn act_id(owner: &str, occurrence: &str, x25519: &str) -> String { + format!("rewrap\u{0}{owner}\u{0}{occurrence}\u{0}{x25519}") } -fn memo_key(owner: &str, occurrence: &str, x25519: &str) -> String { - format!("{owner}\u{0}{occurrence}\u{0}{x25519}") +/// One pass, gated on this process's attendance. Never fails: every error is +/// logged and the pass ends, because the caller is the peer-convergence tick +/// and must not be stopped by this. +pub async fn rewrap_for_new_devices(engine: &Arc, node_key_id: &str) -> RewrapReport { + rewrap_for_new_devices_with(engine, node_key_id, Attendance::global()).await } -/// One pass. Never fails: every error is logged and the pass ends, because -/// the caller is the peer-convergence tick and must not be stopped by this. -pub async fn rewrap_for_new_devices(engine: &Arc, node_key_id: &str) -> RewrapReport { +/// [`rewrap_for_new_devices`] against an explicit [`Attendance`] — a test that +/// stands two devices up in one process gives each its own. +pub async fn rewrap_for_new_devices_with( + engine: &Arc, + node_key_id: &str, + attendance: &Attendance, +) -> RewrapReport { let mut report = RewrapReport::default(); let own = crate::peer::own_keys_of_this_node(node_key_id); let dir = engine.federation_directory(); - // Which of this node's keys is bound, and to whom. - let mut bound: Option<(String, String)> = None; - for k in &own { - if let Ok(Some(o)) = owner_of(dir.as_ref(), k).await { - bound = Some((k.clone(), o)); - break; - } - } - let Some((bound_key, owner)) = bound else { + // Which of this node's keys is bound, and to whom — the NODE key first + // (`Occupant::of_node`), the occurrence the session fold knows. + let Some(who) = Occupant::of_node(engine, node_key_id).await else { return report; }; + let bound_key = who.occurrence.clone(); + let owner = who.owner.clone(); report.owner = Some(owner.clone()); let owned: Vec = match nodes_owned_by(dir.as_ref(), &owner).await { @@ -98,19 +121,16 @@ pub async fn rewrap_for_new_devices(engine: &Arc, node_key_id: &str) -> }; // Cheap first: what is new? The pen is opened only when something is. let mut todo: Vec<(String, String)> = Vec::new(); - { - let held = memo().lock().expect("self re-wrap memo poisoned"); - for o in &occurrences { - let Some(enc) = o.encryption_pubkeys.as_ref() else { - continue; - }; - if own.contains(&o.occurrence_key_id) || !owned.contains(&o.occurrence_key_id) { - continue; - } - let key = memo_key(&owner, &o.occurrence_key_id, &enc.x25519_base64); - if !held.contains(&key) { - todo.push((o.occurrence_key_id.clone(), key)); - } + for o in &occurrences { + let Some(enc) = o.encryption_pubkeys.as_ref() else { + continue; + }; + if own.contains(&o.occurrence_key_id) || !owned.contains(&o.occurrence_key_id) { + continue; + } + let key = act_id(&owner, &o.occurrence_key_id, &enc.x25519_base64); + if !attendance.already_acted(&key) { + todo.push((o.occurrence_key_id.clone(), key)); } } if todo.is_empty() { @@ -131,7 +151,32 @@ pub async fn rewrap_for_new_devices(engine: &Arc, node_key_id: &str) -> return report; } + let community = session_claims::self_community(&owner); for (occurrence, key) in todo { + // ONE DEVICE RE-WRAPS for each new device (CC 3.1.3.1) — see the + // module note. The gate logs its refusal by name. + match session_claims::gate( + engine, + attendance, + &who, + &community, + &session_claims::rewrap_session(&occurrence), + "self_rewrap", + ) + .await + { + Verdict::Act => {} + Verdict::HandledElsewhere { + occurrence: handler, + } => { + report.not_handled_here.push((occurrence, Some(handler))); + continue; + } + Verdict::Unclaimed => { + report.not_handled_here.push((occurrence, None)); + continue; + } + } match engine .rekey_self_occurrence_add(&owner, std::slice::from_ref(&occurrence)) .await @@ -152,10 +197,7 @@ pub async fn rewrap_for_new_devices(engine: &Arc, node_key_id: &str) -> "self files RE-WRAPPED for a new device of this owner — every self file \ written before it was claimed now opens there (CIRISServer#678)" ); - memo() - .lock() - .expect("self re-wrap memo poisoned") - .insert(key); + attendance.record_act(&key); if !r.changed_blobs.is_empty() { crate::compose::kick_replication("self files re-wrapped for a new device"); } diff --git a/src/self_room_drive.rs b/src/self_room_drive.rs index cd474160..a35872a0 100644 --- a/src/self_room_drive.rs +++ b/src/self_room_drive.rs @@ -68,6 +68,12 @@ pub enum SelfRoomTick { Removed(usize), /// Another node's claim wins; ours is dropped and we join from theirs. Abandoned, + /// A commit was due (`Add` / `Remove` / `Rejoin`) and this device does not + /// hold the room's commit duty (CC 3.1.3.1, [`crate::session_claims`]): + /// `handler` names the device that does, `None` means nobody holds it and + /// nobody commits until the person is on a device in the room. Never acted + /// on here — two devices committing at one epoch fork the room. + NotHandledHere { handler: Option }, /// Converged — the tree matches the directory. Idle, /// The tick could not complete. The string names the rung, not the symptom. @@ -291,6 +297,43 @@ pub async fn drive_once(st: &SelfRoomState) -> SelfRoomTick { rival = rival.is_some(), "self room: decided" ); + // ONE COMMITTER (CC 3.1.3.1, `crate::session_claims`). `decide` returns + // Add / Remove / Rejoin to EVERY device holding the group — edge confines + // only `Create` to the lowest key — so with the person's devices all in + // the room, each of them committed the same add at the same epoch and the + // room forked. The commit duty is one exchange, `(self room, + // SELF_ROOM_MEMBERSHIP_SESSION)`; only the device the fold names commits. + // A device reaches this only while it HOLDS the group (those three arms + // need one), so the joiner — which cannot add itself — never claims it. + if matches!( + action, + SelfRoomAction::Add(_) | SelfRoomAction::Remove(_) | SelfRoomAction::Rejoin(_) + ) { + let who = crate::session_claims::Occupant { + owner: owner.clone(), + occurrence: node_key.clone(), + }; + let verdict = crate::session_claims::gate( + &st.engine, + crate::session_claims::Attendance::global(), + &who, + &room_id, + crate::session_claims::SELF_ROOM_MEMBERSHIP_SESSION, + "self_room_commit", + ) + .await; + match verdict { + crate::session_claims::Verdict::Act => {} + crate::session_claims::Verdict::HandledElsewhere { occurrence } => { + return SelfRoomTick::NotHandledHere { + handler: Some(occurrence), + } + } + crate::session_claims::Verdict::Unclaimed => { + return SelfRoomTick::NotHandledHere { handler: None } + } + } + } match action { SelfRoomAction::NotInRoster => SelfRoomTick::NotInRoster, SelfRoomAction::SoleDevice => SelfRoomTick::SoleDevice, diff --git a/src/session_claims.rs b/src/session_claims.rs new file mode 100644 index 00000000..2e324054 --- /dev/null +++ b/src/session_claims.rs @@ -0,0 +1,1097 @@ +//! **One device handles each exchange** — session claims (CC 3.1.3.1, +//! CIRISPersist#782; the maintainer's ruling of 2026-09-30 pulling them into +//! 0.5.218). The design, the inventory of every place this node ACTS for its +//! person, and what is deliberately NOT gated live in `FSD/SESSION_CLAIMS.md`. +//! +//! # The problem +//! +//! A person is one federated identity plus the N nodes they own — the +//! *occurrences* of their self. Replication delivers a row addressed to that +//! person to EVERY one of those nodes, because a fed-ID has no transport path +//! of its own. So every autonomous reaction this server has — the self room's +//! MLS add/remove commits, the re-wrap of old self files for a new device — +//! ran on every device the person owns. Two devices committing an add at the +//! same epoch fork the self room; two devices re-wrapping the same blob write +//! two key-grant sets for one grant. Persist has carried the routing table +//! that fixes this since v38.7.0 (`session:claim:v1`, `handler_for`), and until +//! this module the server never read or wrote a single row of it (zero hits +//! for `session_claim`, `session:claim` or `handler_for` in src, tests or the +//! harness). +//! +//! # What persist decides, and what this module decides +//! +//! Persist owns the row, its projection (`Cohort` at every commons tier, +//! `SelfOwn` at `self`), the admission rule (a claim is a SELF-REPORT — +//! attester == attested == the claiming occurrence, a third party is refused at +//! the door, `check_session_self_report_admission`), the merge (earliest +//! `claimed_at`, ties on the lowest occurrence key id) and the read +//! ([`handler_for`]). This module never re-implements any of it: every "who +//! handles this?" goes through [`handler_for`], and the surface enumerates +//! which exchanges exist but asks persist who holds each one. A second copy +//! of the merge rule here would be the mirrored-rule class again. +//! +//! What persist cannot own is ATTENDANCE — "a human is present on this device" +//! is not a storage fact. That is this module's: [`Attendance`]. +//! +//! # Attendance is the person's authenticated session, not the boot +//! +//! A device is *attended* while the person's own session (a verified owner +//! bearer, never a delegated `dgrant:` one — a helper acting for the person is +//! not the person) has touched it within [`PRESENCE_IDLE`]. The one place every +//! such request resolves is `auth::session::resolve_bearer`, which calls +//! [`Attendance::note_presence`]. A device that merely BOOTED is not attended: +//! a headless node the person never looks at must not take the self room's +//! commit duty just because it came up first, and a claim derived from boot +//! would do exactly that on every restart. +//! +//! # The invariant, kept here as persist keeps it +//! +//! **An unclaimed exchange is never acted on** — not by the lowest id, and not +//! by a single-device self. [`gate`] returns [`Verdict::Act`] only when +//! [`handler_for`] names THIS occurrence; [`Verdict::Unclaimed`] and +//! [`Verdict::HandledElsewhere`] both mean "do not act", and each is logged by +//! name at the site. There is no weak-claim value to promote. +//! +//! # Claim on demand, renew on the cadence +//! +//! A site that has work calls [`gate`]. If nobody holds the exchange AND this +//! device is attended AND it can sign as its own occurrence, the gate writes +//! the claim right there (so the work does not wait a whole renewal period), +//! kicks replication so the person's other devices see it on a round-trip, +//! and re-reads the fold — the device acts only if persist then names it. +//! The exchange is remembered, and the `session_claims` loop (a named +//! `loop_cadence` phase) renews it while the person stays; when they leave, +//! nothing renews, and the claim goes stale after [`SESSION_CLAIM_TTL`] and is +//! claimable by whichever device they are on next. +//! +//! # A renewal under persist v51 — and what v52 changes (CIRISPersist#946) +//! +//! Persist v51's liveness is the CONSUMER's horizon measured from `claimed_at` +//! (`claim_is_live(claim, now, ttl)`), and persist's doc is explicit that a +//! renewal does not move `claimed_at` ("the holder would lose its own +//! session"). Those two together mean a same-instant renewal cannot extend +//! anything on v51: the claim dies `ttl` after it was first taken. So on v51 a +//! renewal here is a SUCCESSOR LEASE — a fresh claim row written only by the +//! device the fold already names as the holder. Because a non-holder never +//! claims while a live claim exists ([`step`]), the holder's own leases are the +//! only live claims, and the earliest of them keeps naming the holder. The one +//! residual is a simultaneous first claim by two attended devices: the loser +//! stops renewing when it sees the winner, but its single lease can outlive +//! the winner's first one and hand the session over ONCE, after which the new +//! holder renews and the old one defers. One handover, never two handlers in +//! one view. CC 3.1.3.1 moves the horizon into the row (`valid_until`, signed, +//! lease ≤ 86 400 s, a renewal is a `supersedes` that keeps `claimed_at`); +//! persist v52 (#946) carries it, and [`write_claim`] names the spot. +//! +//! # At persist v52.0.0 — the lease is in the row; the fold still is not +//! +//! v52 (CIRISPersist#946) made `valid_until` REQUIRED on every `session:*` row +//! and bounded it (`claimed_at ≤ valid_until ≤ claimed_at + 86 400 s`, +//! `admission::check_session_lease_bound`, at every door), so every claim +//! written here now carries `valid_until = claimed_at + TTL` — the honest end +//! of a [`SESSION_CLAIM_TTL`] lease, signed. +//! +//! What v52 did NOT move is the READ. `session_claim::handler_for` still folds +//! `claim_is_live(claim, now, ttl)` = `now − claimed_at < ttl` with the +//! CONSUMER's ttl; it reads neither `valid_until` nor `supersedes` (and +//! `list_attestations_for` does not drop a superseded row). So the renewal CC +//! 3.1.3.1 describes — a `supersedes` of the previous lease that KEEPS +//! `claimed_at` and moves `valid_until` forward — would be judged by its +//! unchanged `claimed_at` and expire the holder at `claimed_at + TTL` in every +//! view, on every device: the session would be handed over every two minutes +//! while the person sat at the holder. That is worse than the v51 behaviour +//! it is meant to improve, so the renewal stays the successor lease described +//! above (a fresh claim by the device the fold already names, its own +//! `claimed_at`, its own `valid_until`), which v52's bound admits and v52's +//! fold keeps live. The day persist's fold reads `valid_until` and honours +//! `supersedes` (the read half of #946), [`renew_once`] turns into that +//! supersedes and `SESSION_CLAIM_TTL` stops being a consumer constant; until +//! then a server-side liveness fold over `valid_until` would be a second copy +//! of persist's rule (the mirrored-rule class), so none is written here. +//! +//! Every one of this person's devices runs this binary, so every one of them +//! applies the same [`SESSION_CLAIM_TTL`] — the convergence argument needs one +//! horizon, and until the horizon is in the row, one constant is how it gets +//! one. + +use std::collections::{BTreeMap, HashSet}; +use std::sync::{Arc, Mutex, OnceLock}; +use std::time::Duration; + +use axum::extract::State; +use axum::http::{HeaderMap, StatusCode}; +use axum::response::{IntoResponse, Response}; +use axum::{Json, Router}; +use ciris_persist::federation::admission::{envelope_dimension, nodes_stewarded_by, owner_of}; +use ciris_persist::federation::envelope::paths; +use ciris_persist::federation::session_claim::{ + claim_from_envelope, handler_for, SessionClaim, SESSION_CLAIM_DIMENSION, +}; +use ciris_persist::federation::types::{attestation_type, cohort_scope}; +use ciris_persist::prelude::{Engine, LocalSigner}; + +// ─── The horizons ─────────────────────────────────────────────────────────── + +/// **How long a claim holds without a successor** — 120 s. +/// +/// Long enough that a holder that renews every [`SESSION_CLAIM_RENEW_AFTER`] +/// (60 s) on a 30 s loop always has a live lease with a full renewal period of +/// slack: one missed tick, a slow directory read or a replication round-trip +/// never opens a gap in which nobody holds the session. Short enough that when +/// the person puts a device down, their other device can take the exchange +/// within two minutes — the self room's next add, or a new device's re-wrap, +/// waits at most that long for the device they moved to. The same value on +/// every device (see the module docs): it is the horizon persist's +/// `claim_is_live` measures, until CIRISPersist#946 signs it into the row. +pub const SESSION_CLAIM_TTL: Duration = Duration::from_secs(120); + +/// **The renewal loop's period** — 30 s, the node's common loop period (every +/// default cadence is a multiple of 30 s, which is what `loop_cadence`'s +/// separation argument rests on). +pub const SESSION_CLAIM_RENEW_EVERY: Duration = Duration::from_secs(30); + +/// **A holder writes its successor lease once its newest is this old** — 60 s, +/// half the TTL. Renewing every tick would be a row every 30 s per exchange +/// for as long as the person is present (they replicate to every device); at +/// half the TTL it is one a minute, and the newest lease always has at least +/// 60 s left when the next one is written. +pub const SESSION_CLAIM_RENEW_AFTER: Duration = Duration::from_secs(60); + +/// **How long after the person's last request a device stays attended** — +/// 10 minutes. A client polls far more often than this while it is open, so +/// an open app keeps its device attended; a closed one lets it go. Long enough +/// to cover the gap the second-device flow creates: the person approves on the +/// first device, the second boots, provisions and publishes its KeyPackage, +/// and the first must still be attended when that arrives to commit the add. +pub const PRESENCE_IDLE: Duration = Duration::from_secs(10 * 60); + +// ─── The exchanges this server names ──────────────────────────────────────── + +/// The self room's COMMIT duty: adding, removing and rejoining the person's +/// devices. One session for the whole duty, not one per joiner, because an +/// MLS group's commits must come from one committer at a time — two devices +/// adding two different joiners at the same epoch fork the room exactly as +/// two adding the same one do. CC 3.1.3.1 names "a moderation duty" as an +/// exchange; this is that shape. +pub const SELF_ROOM_MEMBERSHIP_SESSION: &str = "self_room:membership"; + +/// The re-wrap of old self files for ONE new device: `self_rewrap:`. +/// Per occurrence, because the capability is per occurrence — the new device +/// itself cannot re-wrap for itself (it holds none of the old DEKs), so it +/// must never be the one that takes its own re-wrap, and a per-person session +/// would let it (the fold would name whoever claimed first for ANY device). +#[must_use] +pub fn rewrap_session(occurrence_key_id: &str) -> String { + format!("self_rewrap:{occurrence_key_id}") +} + +/// The community an exchange of the person's SELF is keyed under: the self +/// room's content group id, edge's single definition of "this person's self +/// room" (`ciris_edge::self_room::room`). Never re-derived here. +#[must_use] +pub fn self_community(owner_key_id: &str) -> String { + ciris_edge::self_room::room(owner_key_id) + .content_group_id() + .to_owned() +} + +// The three signed members of a claim (CC 2.1 / CC 3.1.3.1). Persist exports +// no path constants for them (its reader, `claim_from_envelope`, spells them +// inline), so they are named once here and `a_written_claim_is_one_persist_can +// _read` round-trips a written envelope through persist's own reader: a rename +// upstream fails that test instead of silently producing rows nobody folds. +const COMMUNITY_ID: &str = "community_id"; +const SESSION_ID: &str = "session_id"; +const CLAIMED_AT: &str = "claimed_at"; +/// persist v52 (CIRISPersist#946): the lease's signed end. Persist's gate +/// spells it inline too (`check_session_lease_bound`); the round-trip test +/// below writes a claim through the real door, so a rename upstream refuses +/// the row and fails the test. +const VALID_UNTIL: &str = "valid_until"; + +/// CC 2.6.2 canonical instant: RFC 3339, milliseconds, `Z`. +#[must_use] +pub fn canonical_instant(at: chrono::DateTime) -> String { + at.to_rfc3339_opts(chrono::SecondsFormat::Millis, true) +} + +fn ttl() -> chrono::Duration { + chrono::Duration::from_std(SESSION_CLAIM_TTL).expect("a 120 s TTL fits a chrono duration") +} + +// ─── The state machine ────────────────────────────────────────────────────── + +/// What one device does about one exchange on one tick. Pure — see [`step`]. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Step { + /// Nobody holds it and the person is here: take it. + Claim, + /// We hold it and our newest lease is due: write the successor. + Renew, + /// We hold it and our newest lease is fresh: nothing to write. + Hold, + /// Another occurrence holds a LIVE claim. Never contested — a live claim + /// is not stealable, and contesting it is the one thing that would make + /// "earliest wins" flap. + Defer { handler: String }, + /// The person is not here: write nothing and let our claim go stale. + Lapse, +} + +/// **The claim / renew / lapse decision**, given what the fold says now. +/// +/// - not attended → [`Step::Lapse`], whatever the fold says: a device the +/// person left must stop renewing, or it holds the session forever; +/// - nobody holds it → [`Step::Claim`]; +/// - someone else holds it → [`Step::Defer`]; +/// - we hold it → [`Step::Renew`] when our newest lease is at least +/// [`SESSION_CLAIM_RENEW_AFTER`] old (or unknown — a restarted process does +/// not remember what it wrote, and one extra lease is cheaper than a gap), +/// else [`Step::Hold`]. +#[must_use] +pub fn step( + attended: bool, + handler: Option<&SessionClaim>, + me: &str, + newest_lease: Option>, + now: chrono::DateTime, +) -> Step { + if !attended { + return Step::Lapse; + } + match handler { + None => Step::Claim, + Some(h) if h.occurrence_key_id != me => Step::Defer { + handler: h.occurrence_key_id.clone(), + }, + Some(_) => match newest_lease { + Some(at) + if now.signed_duration_since(at) + < chrono::Duration::from_std(SESSION_CLAIM_RENEW_AFTER).expect("60 s fits") => + { + Step::Hold + } + _ => Step::Renew, + }, + } +} + +/// **May this device act on this exchange?** Only [`Verdict::Act`] says yes. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Verdict { + /// The fold names this occurrence. Act. + Act, + /// The fold names another occurrence of the same person. Do not act; that + /// device does. + HandledElsewhere { occurrence: String }, + /// Nobody holds it. Do not act — not even as the only device (CC 3.1.3.1). + Unclaimed, +} + +impl Verdict { + /// The fold's answer, mapped for `me`. `None` is [`Verdict::Unclaimed`] + /// and nothing else. + #[must_use] + pub fn of(handler: Option<&SessionClaim>, me: &str) -> Self { + match handler { + Some(h) if h.occurrence_key_id == me => Self::Act, + Some(h) => Self::HandledElsewhere { + occurrence: h.occurrence_key_id.clone(), + }, + None => Self::Unclaimed, + } + } + + /// `true` only for [`Verdict::Act`]. + #[must_use] + pub fn acts(&self) -> bool { + matches!(self, Self::Act) + } +} + +// ─── Attendance ───────────────────────────────────────────────────────────── + +/// One exchange this device has offered to handle. +#[derive(Debug, Clone)] +struct Exchange { + owner: String, + occurrence: String, + /// The `claimed_at` of the newest lease THIS process wrote, if any. + newest_lease: Option>, +} + +/// **Whether the person is on this device, and which exchanges this device +/// has offered to handle.** One per device: production uses +/// [`Attendance::global`] (a process is a device); a test that stands two +/// devices up in one process gives each its own. +#[derive(Debug, Default)] +pub struct Attendance { + presence: Mutex>, + exchanges: Mutex>, + acted: Mutex>, +} + +impl Attendance { + /// A device with nobody on it and nothing offered. + #[must_use] + pub fn new() -> Self { + Self::default() + } + + /// This process's attendance. + pub fn global() -> &'static Attendance { + static GLOBAL: OnceLock = OnceLock::new(); + GLOBAL.get_or_init(Attendance::new) + } + + /// The person made an authenticated request on this device, now. + pub fn note_presence(&self) { + *self.presence.lock().expect("attendance poisoned") = Some(std::time::Instant::now()); + } + + /// The person left (a test's lever; production lets [`PRESENCE_IDLE`] + /// pass instead). + pub fn end_presence(&self) { + *self.presence.lock().expect("attendance poisoned") = None; + } + + /// Has the person touched this device within [`PRESENCE_IDLE`]? + #[must_use] + pub fn attended(&self) -> bool { + self.presence + .lock() + .expect("attendance poisoned") + .is_some_and(|at| at.elapsed() < PRESENCE_IDLE) + } + + fn offer(&self, community: &str, session: &str, owner: &str, occurrence: &str) { + let mut ex = self.exchanges.lock().expect("attendance poisoned"); + ex.entry((community.to_owned(), session.to_owned())) + .and_modify(|e| { + e.owner = owner.to_owned(); + e.occurrence = occurrence.to_owned(); + }) + .or_insert_with(|| Exchange { + owner: owner.to_owned(), + occurrence: occurrence.to_owned(), + newest_lease: None, + }); + } + + fn leased(&self, community: &str, session: &str, at: chrono::DateTime) { + if let Some(e) = self + .exchanges + .lock() + .expect("attendance poisoned") + .get_mut(&(community.to_owned(), session.to_owned())) + { + e.newest_lease = Some(at); + } + } + + fn snapshot(&self) -> Vec<((String, String), Exchange)> { + self.exchanges + .lock() + .expect("attendance poisoned") + .iter() + .map(|(k, v)| (k.clone(), v.clone())) + .collect() + } + + fn forget(&self, community: &str, session: &str) { + self.exchanges + .lock() + .expect("attendance poisoned") + .remove(&(community.to_owned(), session.to_owned())); + } + + /// **Idempotent per act** (CC 3.1.3.1: "consumers MUST still make handling + /// idempotent per attestation id — views transiently disagree"). A site + /// records the act it completed; [`Self::already_acted`] answers before + /// the next attempt. Recorded only on SUCCESS, so a failed act retries. + pub fn record_act(&self, act_id: &str) { + self.acted + .lock() + .expect("attendance poisoned") + .insert(act_id.to_owned()); + } + + /// Has this device already completed `act_id`? + #[must_use] + pub fn already_acted(&self, act_id: &str) -> bool { + self.acted + .lock() + .expect("attendance poisoned") + .contains(act_id) + } +} + +// ─── The occurrence's own pen ─────────────────────────────────────────────── + +/// How this device signs AS the occurrence the fold knows it by. A claim is a +/// self-report of the OCCURRENCE — the node key the owner-binding names — so +/// on an actor/node split install it is the held node signer, never the +/// engine's actor key (the axis this codebase has got wrong nine times: +/// `wire_identity()` / the bound key, never `local_derived_key_id()` blindly). +pub enum NodePen { + /// The engine IS the node (a standalone install). + Engine, + /// The split's held node signer, registered under its DERIVED id. + Held(Arc), +} + +impl NodePen { + /// The pen that signs as `occurrence`, or `None` when this process holds + /// no key that is that occurrence — then it cannot claim, and says so. + pub async fn for_occurrence(engine: &Engine, occurrence: &str) -> Option { + if let Some(held) = crate::node_key::held_node_signer() { + if held.derived_key_id() == occurrence { + return Some(Self::Held(held)); + } + } + match engine.local_derived_key_id().await { + Ok(k) if k == occurrence => Some(Self::Engine), + _ => None, + } + } + + fn signer<'a>( + &'a self, + engine: &'a Engine, + occurrence: &'a str, + ) -> crate::attest::KeySigner<'a> { + match self { + Self::Engine => crate::attest::KeySigner::Engine(engine), + Self::Held(s) => crate::attest::KeySigner::LocalAs(s.as_ref(), occurrence), + } + } +} + +/// **Write one claim** — `session:claim:v1`, a self-report of `occurrence`, +/// at `self`, through the one attest door. `claimed_at` is the caller's (a +/// fresh claim and a v51 successor lease both pass `now`). +/// +/// The lease is in the SIGNED envelope (persist v52, CIRISPersist#946, CC +/// 3.1.3.1): `valid_until = claimed_at + TTL` — required on every `session:*` +/// row, refused without it (`check_session_lease_bound`), and bounded to a +/// day, which a 120 s lease is far inside. `expires_at` states the same +/// instant for anything that sweeps expired rows. persist's fold still reads +/// the consumer TTL from `claimed_at` (see the module docs, "At persist +/// v52.0.0"), and both horizons are the same [`SESSION_CLAIM_TTL`], so the +/// row's lease and the fold's agree on every device. +pub async fn write_claim( + engine: &Engine, + signer: crate::attest::KeySigner<'_>, + community: &str, + session: &str, + claimed_at: chrono::DateTime, +) -> Result { + let envelope = claim_envelope(community, session, claimed_at); + let spec = crate::attest::Spec::new(attestation_type::SCORES, cohort_scope::SELF, envelope) + .weighing(Some(1.0)) + .expiring(Some(claimed_at + ttl())); + crate::attest::emit(engine, signer, spec).await +} + +/// The signed members of one claim — ONE builder, so the unit test that runs +/// persist's own lease gate over it judges exactly what [`write_claim`] signs. +fn claim_envelope( + community: &str, + session: &str, + claimed_at: chrono::DateTime, +) -> serde_json::Value { + serde_json::json!({ + (paths::DIMENSION): SESSION_CLAIM_DIMENSION, + "score": 1.0, + COMMUNITY_ID: community, + SESSION_ID: session, + CLAIMED_AT: canonical_instant(claimed_at), + VALID_UNTIL: canonical_instant(claimed_at + ttl()), + }) +} + +/// Who this device is, for its person: the occurrence the owner-binding +/// names, and that owner. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Occupant { + /// The person (the owner fed-ID key). + pub owner: String, + /// This device's occurrence — the node key the owner-binding names. + pub occurrence: String, +} + +impl Occupant { + /// Resolve from this node's own keys (`peer::own_keys_of_this_node`): the + /// first one bound to an owner. `None` on an unowned node — which has no + /// person to handle anything for. + pub async fn of_node(engine: &Engine, node_key_id: &str) -> Option { + let dir = engine.federation_directory(); + // The NODE key first: on a split install the engine key is the ACTOR, + // and the occurrence the fold knows is the node the binding moved to. + let mut keys: Vec = [ + crate::node_key::wire_identity().map(str::to_owned), + crate::node_key::held_node_signer().map(|h| h.derived_key_id()), + ] + .into_iter() + .flatten() + .collect(); + for k in crate::peer::own_keys_of_this_node(node_key_id) { + if !keys.contains(&k) { + keys.push(k); + } + } + for k in keys { + if let Ok(Some(owner)) = owner_of(dir.as_ref(), &k).await { + return Some(Self { + owner, + occurrence: k, + }); + } + } + None + } +} + +/// **THE GATE.** Call before every ACT in `FSD/SESSION_CLAIMS.md`'s inventory. +/// +/// Reads [`handler_for`]; if nobody holds `(community, session)` and this +/// device is attended and can sign as its occurrence, claims it on the spot +/// and reads again. Offers the exchange to `attendance` either way (so the +/// renewal loop keeps a held one and the surface can name it). Returns what +/// the fold says for THIS occurrence, and logs the refusals by name — `site` +/// is the act's name in those lines. +pub async fn gate( + engine: &Engine, + attendance: &Attendance, + who: &Occupant, + community: &str, + session: &str, + site: &'static str, +) -> Verdict { + gate_at( + engine, + attendance, + who, + community, + session, + site, + chrono::Utc::now(), + ) + .await +} + +/// [`gate`] at a stated instant — a claim written here is dated `now`. For a +/// test that must see a claim lapse without waiting out the TTL; production +/// always passes the clock through [`gate`]. +pub async fn gate_at( + engine: &Engine, + attendance: &Attendance, + who: &Occupant, + community: &str, + session: &str, + site: &'static str, + now: chrono::DateTime, +) -> Verdict { + let dir = engine.federation_directory(); + let read = |now| { + let dir = Arc::clone(&dir); + async move { handler_for(dir.as_ref(), &who.owner, community, session, now, ttl()).await } + }; + let mut handler = match read(now).await { + Ok(h) => h, + Err(e) => { + // A failed read is NOT "unclaimed, so maybe me" — it is "unknown", + // and unknown never acts. + tracing::warn!( + site, community, session, error = %e, + "session claim: the handler read failed — nobody acts this tick" + ); + return Verdict::Unclaimed; + } + }; + attendance.offer(community, session, &who.owner, &who.occurrence); + if handler.is_none() && attendance.attended() { + match claim_now(engine, attendance, who, community, session, now).await { + Ok(()) => { + handler = read(now).await.unwrap_or(None); + } + Err(e) => tracing::warn!( + site, community, session, error = %e, + "session claim: this device is attended and could not write its claim" + ), + } + } + let verdict = Verdict::of(handler.as_ref(), &who.occurrence); + match &verdict { + Verdict::Act => tracing::debug!( + site, community, session, occurrence = %who.occurrence, + "session claim: handled HERE — this device acts" + ), + Verdict::HandledElsewhere { occurrence } => tracing::info!( + site, community, session, handler = %occurrence, + "session claim: handled by occurrence {occurrence} — this device does not act" + ), + Verdict::Unclaimed => tracing::info!( + site, + community, + session, + attended = attendance.attended(), + "session claim: unclaimed — nobody acts (the person is on none of their devices \ + that can handle this; it runs where they next are)" + ), + } + verdict +} + +async fn claim_now( + engine: &Engine, + attendance: &Attendance, + who: &Occupant, + community: &str, + session: &str, + now: chrono::DateTime, +) -> Result<(), String> { + let Some(pen) = NodePen::for_occurrence(engine, &who.occurrence).await else { + return Err(format!( + "this process holds no pen for occurrence {} — a claim is its self-report", + who.occurrence + )); + }; + let id = write_claim( + engine, + pen.signer(engine, &who.occurrence), + community, + session, + now, + ) + .await + .map_err(|e| format!("{e}"))?; + attendance.leased(community, session, now); + tracing::info!( + community, session, occurrence = %who.occurrence, attestation_id = %id, + "session claim WRITTEN — this device takes the exchange while the person is here" + ); + let _ = crate::compose::kick_replication("session:claim"); + Ok(()) +} + +/// What one renewal pass did, per exchange — for the loop's log and tests. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct RenewalLine { + pub community: String, + pub session: String, + pub step: Step, +} + +/// **One renewal pass** over every exchange this device has offered. +pub async fn renew_once(engine: &Engine, attendance: &Attendance) -> Vec { + renew_once_at(engine, attendance, chrono::Utc::now()).await +} + +/// [`renew_once`] at a stated instant (see [`gate_at`]). +pub async fn renew_once_at( + engine: &Engine, + attendance: &Attendance, + now: chrono::DateTime, +) -> Vec { + let dir = engine.federation_directory(); + let attended = attendance.attended(); + let mut out = Vec::new(); + for ((community, session), ex) in attendance.snapshot() { + let handler = match handler_for(dir.as_ref(), &ex.owner, &community, &session, now, ttl()) + .await + { + Ok(h) => h, + Err(e) => { + tracing::debug!(%community, %session, error = %e, "session claim renewal: read failed this tick"); + continue; + } + }; + let s = step( + attended, + handler.as_ref(), + &ex.occurrence, + ex.newest_lease, + now, + ); + match &s { + Step::Claim | Step::Renew => { + let who = Occupant { + owner: ex.owner.clone(), + occurrence: ex.occurrence.clone(), + }; + if let Err(e) = claim_now(engine, attendance, &who, &community, &session, now).await + { + tracing::warn!(%community, %session, error = %e, "session claim renewal could not write"); + } + } + Step::Lapse => { + // Nothing written. Once our newest lease is past the TTL the + // exchange is no longer ours in any view: stop tracking it. + let stale = ex + .newest_lease + .is_none_or(|at| now.signed_duration_since(at) >= ttl()); + if stale { + attendance.forget(&community, &session); + } + } + Step::Hold | Step::Defer { .. } => {} + } + out.push(RenewalLine { + community, + session, + step: s, + }); + } + out +} + +/// **The `session_claims` loop** — renews held exchanges on its own +/// `loop_cadence` slot while the person stays, and lets them lapse when they +/// leave. Stops when `shutdown` flips. +pub fn spawn( + engine: Arc, + mut shutdown: tokio::sync::watch::Receiver, +) -> tokio::task::JoinHandle<()> { + tokio::spawn(async move { + let mut schedule = + crate::loop_cadence::Cadence::new("session_claims", SESSION_CLAIM_RENEW_EVERY); + tracing::info!( + period_secs = SESSION_CLAIM_RENEW_EVERY.as_secs(), + ttl_secs = SESSION_CLAIM_TTL.as_secs(), + renew_after_secs = SESSION_CLAIM_RENEW_AFTER.as_secs(), + presence_idle_secs = PRESENCE_IDLE.as_secs(), + "session claims loop spawned (CC 3.1.3.1): one device handles each exchange" + ); + let mut last: BTreeMap<(String, String), Step> = BTreeMap::new(); + loop { + if *shutdown.borrow() { + break; + } + tokio::select! { + () = schedule.tick() => {} + _ = shutdown.changed() => { + if *shutdown.borrow() { break; } + continue; + } + } + for line in renew_once(&engine, Attendance::global()).await { + let key = (line.community.clone(), line.session.clone()); + // Transition-only for the steady states; a write is always news. + let news = matches!(line.step, Step::Claim | Step::Renew) + || last.get(&key) != Some(&line.step); + if news { + tracing::info!( + community = %line.community, session = %line.session, step = ?line.step, + "session claims" + ); + } + last.insert(key, line.step); + } + } + tracing::info!("session claims loop stopped"); + }) +} + +// ─── GET /v1/self/sessions ────────────────────────────────────────────────── + +fn refuse(code: StatusCode, id: &'static str, text: &'static str, detail: String) -> Response { + ( + code, + Json(serde_json::json!({ "error": text, "reason_id": id, "detail": detail })), + ) + .into_response() +} + +/// A live claim's state, worded for the person — `(state_id, text)`. +fn state(here: bool) -> (&'static str, &'static str) { + if here { + ( + "session.state.handled_here", + "This device is answering for you in this exchange.", + ) + } else { + ( + "session.state.handled_elsewhere", + "Another of your devices is answering for you in this exchange.", + ) + } +} + +/// The distinct `(community, session)` pairs any of `owner`'s occurrences has +/// a well-formed, self-reported claim row for. ENUMERATION only — who holds +/// each is asked of [`handler_for`], never decided here. +async fn claimed_exchanges(engine: &Engine, owner: &str) -> Result, String> { + let dir = engine.federation_directory(); + let occurrences = nodes_stewarded_by(dir.as_ref(), owner) + .await + .map_err(|e| format!("nodes_stewarded_by({owner}): {e}"))?; + let mut seen = std::collections::BTreeSet::new(); + for occ in occurrences { + let rows = dir + .list_attestations_for(&occ) + .await + .map_err(|e| format!("list_attestations_for({occ}): {e}"))?; + for row in rows { + if envelope_dimension(&row.attestation_envelope) != Some(SESSION_CLAIM_DIMENSION) + || row.attesting_key_id != occ + || row.attested_key_id != occ + { + continue; + } + if let Some((c, s, _)) = claim_from_envelope(&row.attestation_envelope, &occ) { + seen.insert((c, s)); + } + } + } + Ok(seen.into_iter().collect()) +} + +/// The surface's body, separated from the route so a test can read it for +/// either of two in-process devices. +pub async fn sessions_view( + engine: &Engine, + attendance: &Attendance, + owner: &str, + this_device: &str, +) -> Result { + let dir = engine.federation_directory(); + let labels = crate::self_devices::labels_for(engine, owner).await; + let now = chrono::Utc::now(); + let mut sessions = Vec::new(); + for (community, session) in claimed_exchanges(engine, owner).await? { + let Some(h) = handler_for(dir.as_ref(), owner, &community, &session, now, ttl()) + .await + .map_err(|e| format!("handler_for: {e}"))? + else { + continue; // every claim for it has lapsed: nobody is answering + }; + let here = h.occurrence_key_id == this_device; + let (state_id, state_text) = state(here); + sessions.push(serde_json::json!({ + "community_id": community, + "session_id": session, + "handler_occurrence_key_id": h.occurrence_key_id, + "handler_label": labels.get(&h.occurrence_key_id), + "claimed_at": canonical_instant(h.claimed_at), + "live_until": canonical_instant(h.claimed_at + ttl()), + "this_device": here, + "state_id": state_id, + "state": state_text, + })); + } + Ok(serde_json::json!({ + "owner_key_id": owner, + "this_device": this_device, + "attended": attendance.attended(), + "ttl_seconds": SESSION_CLAIM_TTL.as_secs(), + "sessions": sessions, + })) +} + +/// `GET /v1/self/sessions` — every exchange of the person's that some device +/// is answering, and which one, so the client can say "answering on ". +/// Owner-authenticated; a delegated session may READ it (the family gate's +/// rule: reads admit a delegate, writes do not). +async fn list_sessions(State(engine): State>, headers: HeaderMap) -> Response { + let caller = + match crate::family_api::owner_caller(&engine, &headers, true).await { + Ok(c) => c, + Err(crate::family_api::GateRefusal::Store(d)) => return refuse( + StatusCode::SERVICE_UNAVAILABLE, + "self.sessions_unavailable", + "The node could not read which of your devices is answering. Try again shortly.", + d, + ), + Err(crate::family_api::GateRefusal::NoSession) => { + return refuse( + StatusCode::UNAUTHORIZED, + "self.owner_session_required", + "Your devices are the owner's own surface. Sign in as the owner of this node.", + String::new(), + ) + } + Err(_) => { + return refuse( + StatusCode::FORBIDDEN, + "self.owner_session_required", + "Your devices are the owner's own surface. Sign in as the owner of this node.", + String::new(), + ) + } + }; + // THIS device, as the fold knows it: the occurrence bound to the owner. + let this_device = Occupant::of_node(&engine, &caller.node_key_id) + .await + .map_or(caller.node_key_id.clone(), |o| o.occurrence); + match sessions_view( + &engine, + Attendance::global(), + &caller.owner_key_id, + &this_device, + ) + .await + { + Ok(v) => Json(v).into_response(), + Err(d) => refuse( + StatusCode::SERVICE_UNAVAILABLE, + "self.sessions_unavailable", + "The node could not read which of your devices is answering. Try again shortly.", + d, + ), + } +} + +/// The session-claims surface. +pub fn router(engine: Arc) -> Router { + Router::new() + .route("/v1/self/sessions", axum::routing::get(list_sessions)) + .with_state(engine) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn at(s: &str) -> chrono::DateTime { + s.parse().expect("fixture instant") + } + fn claim(occ: &str, when: &str) -> SessionClaim { + SessionClaim { + occurrence_key_id: occ.to_owned(), + claimed_at: at(when), + } + } + const NOW: &str = "2026-09-30T12:00:00Z"; + + /// The invariant first: nobody holding it is never "so I will". + #[test] + fn an_unclaimed_exchange_is_not_acted_on_even_by_a_lone_device() { + assert_eq!(Verdict::of(None, "me"), Verdict::Unclaimed); + assert!(!Verdict::Unclaimed.acts()); + assert!(!Verdict::HandledElsewhere { + occurrence: "other".into() + } + .acts()); + assert!(Verdict::of(Some(&claim("me", NOW)), "me").acts()); + } + + #[test] + fn an_attended_device_claims_what_nobody_holds() { + assert_eq!(step(true, None, "me", None, at(NOW)), Step::Claim); + } + + #[test] + fn a_device_never_contests_a_live_claim() { + assert_eq!( + step(true, Some(&claim("other", NOW)), "me", None, at(NOW)), + Step::Defer { + handler: "other".into() + }, + "a live claim is not stealable — contesting it is what would make earliest-wins flap" + ); + } + + #[test] + fn the_holder_renews_at_half_the_ttl_and_holds_before() { + let held = claim("me", "2026-09-30T11:59:00Z"); + assert_eq!( + step( + true, + Some(&held), + "me", + Some(at("2026-09-30T11:59:30Z")), + at(NOW) + ), + Step::Hold, + "30 s old: fresh" + ); + assert_eq!( + step( + true, + Some(&held), + "me", + Some(at("2026-09-30T11:59:00Z")), + at(NOW) + ), + Step::Renew, + "60 s old: write the successor lease" + ); + assert_eq!( + step(true, Some(&held), "me", None, at(NOW)), + Step::Renew, + "a restarted process does not know its newest lease: one extra beats a gap" + ); + } + + #[test] + fn a_device_the_person_left_lapses_whatever_the_fold_says() { + for h in [None, Some(claim("me", NOW)), Some(claim("other", NOW))] { + assert_eq!( + step(false, h.as_ref(), "me", Some(at(NOW)), at(NOW)), + Step::Lapse + ); + } + } + + /// The renewal cadence must leave slack inside the TTL: a holder's newest + /// lease always has at least one loop period left when it is renewed. + #[test] + fn the_horizons_leave_a_full_period_of_slack() { + assert!(SESSION_CLAIM_RENEW_AFTER + SESSION_CLAIM_RENEW_EVERY < SESSION_CLAIM_TTL); + assert!(SESSION_CLAIM_TTL < PRESENCE_IDLE); + assert!( + SESSION_CLAIM_TTL.as_secs() <= 86_400, + "CC 3.1.3.1 bounds a lease at a day" + ); + } + + /// The members this module writes are the ones persist's reader reads — + /// and, since persist v52 (CIRISPersist#946), the ones persist's lease gate + /// admits: `valid_until` present, after `claimed_at`, within a day. + #[test] + fn a_written_claim_is_one_persist_can_read() { + let env = claim_envelope("c1", SELF_ROOM_MEMBERSHIP_SESSION, at(NOW)); + ciris_persist::federation::admission::check_session_lease_bound( + SESSION_CLAIM_DIMENSION, + &env, + ) + .expect("persist v52's lease bound admits the claim this module signs"); + assert_eq!(env[VALID_UNTIL], "2026-09-30T12:02:00.000Z"); + let mut bare = env.clone(); + bare.as_object_mut().expect("object").remove(VALID_UNTIL); + assert!( + ciris_persist::federation::admission::check_session_lease_bound( + SESSION_CLAIM_DIMENSION, + &bare, + ) + .is_err(), + "a claim without valid_until is malformed at v52 — the pre-v52 row shape" + ); + let (c, s, got) = claim_from_envelope(&env, "occ").expect("persist folds it"); + assert_eq!( + (c.as_str(), s.as_str()), + ("c1", SELF_ROOM_MEMBERSHIP_SESSION) + ); + assert_eq!(got.claimed_at, at(NOW)); + assert_eq!(canonical_instant(at(NOW)), "2026-09-30T12:00:00.000Z"); + } + + #[test] + fn presence_is_the_persons_request_and_ends() { + let a = Attendance::new(); + assert!(!a.attended(), "a device that only booted is not attended"); + a.note_presence(); + assert!(a.attended()); + a.end_presence(); + assert!(!a.attended()); + } + + #[test] + fn an_act_is_recorded_once_and_only_on_success() { + let a = Attendance::new(); + assert!(!a.already_acted("x")); + a.record_act("x"); + assert!(a.already_acted("x")); + } +} diff --git a/src/trust_root_api.rs b/src/trust_root_api.rs index 52b95bd5..a6b2f118 100644 --- a/src/trust_root_api.rs +++ b/src/trust_root_api.rs @@ -95,7 +95,54 @@ async fn list_roots(State(st): State) -> Response { .await; match verdict { Ok(v) => { - let json = serde_json::to_value(&v).unwrap_or(serde_json::Value::Null); + let mut json = serde_json::to_value(&v).unwrap_or(serde_json::Value::Null); + // THE WITNESSED HEAD (persist v51 #938, CC T6/T8 (vii)): the + // digest of the signed roster row at the version this node + // witnessed, and its instant. Every node holding the same rows + // reports the same pair — the one-line multi-node predicate the + // topology harness reports per node (CIRISConstitution#131). + let head = match st.engine.lineage_head(&root_ref).await { + Ok(Some(view)) => serde_json::json!({ + "digest": view.witnessed_head.as_ref().map(|(d, _)| d.clone()), + "at": view.witnessed_head.as_ref().map(|(_, at)| at.to_rfc3339()), + "quorum": view.quorum, + "judged": view.community.as_ref().and_then(|c| c.judged), + "latest_cosign_at": view.latest_cosign_at.map(|t| t.to_rfc3339()), + }), + Ok(None) => serde_json::Value::Null, + Err(e) => serde_json::json!({ "error": e.to_string() }), + }; + // STANDING, one word, as the T8 verdict names it: `not_rooted` + // when the five-conjunct verdict fails or a halt is latched; + // `stalled` when the root is valid but its community resolves + // with `live: false` (CC T7: fewer than M+1 active founders — + // attached pairs stay rooted, new members are refused, edge's + // FIRST_CONTACT.md I12); `rooted` otherwise. A key root has no + // community to resolve and is never stalled. + let live = match ciris_persist::federation::canonical_community::resolve_community( + st.engine.federation_directory().as_ref(), + &root_ref, + ) + .await + { + Ok(Some(c)) => Some(c.live), + Ok(None) | Err(_) => None, + }; + let standing = if !v.valid || v.halt_latched == Some(true) { + "not_rooted" + } else if live == Some(false) { + "stalled" + } else { + "rooted" + }; + if let serde_json::Value::Object(m) = &mut json { + m.insert("lineage_head".into(), head); + m.insert( + "standing".into(), + serde_json::Value::String(standing.into()), + ); + m.insert("live".into(), serde_json::json!(live)); + } roots.push(RootEntry { root_key_id: root_ref, root_kind: json diff --git a/tests/a_second_device_joins.rs b/tests/a_second_device_joins.rs index 24527faf..ef6a4f82 100644 --- a/tests/a_second_device_joins.rs +++ b/tests/a_second_device_joins.rs @@ -382,7 +382,12 @@ async fn old_self_files_are_rewrapped_for_the_new_device_by_the_pen_holder() { ); assert!(!recipients(Arc::clone(&first)).await.contains(&occurrence)); - // THE PEN OPENS HERE (what compose registers at boot): the re-wrap runs. + // THE PEN OPENS HERE (what compose registers at boot): the re-wrap runs — + // on the device the person is ON (0.5.218, CC 3.1.3.1): the re-wrap for a + // new device is an exchange one device handles, claimed where the owner's + // session is. The drive requests above already made this process attended + // (`resolve_bearer`); said explicitly so the precondition is on the page. + ciris_server::session_claims::Attendance::global().note_presence(); ciris_server::node_key::set_user_seed_dir(owner.seed_dir.clone(), owner.alias.clone()); let report = ciris_server::self_rewrap::rewrap_for_new_devices(&first, &first_key).await; assert!(!report.no_pen_here, "{report:?}"); diff --git a/tests/abuse_surface.rs b/tests/abuse_surface.rs index c19db4e7..01c127a0 100644 --- a/tests/abuse_surface.rs +++ b/tests/abuse_surface.rs @@ -258,7 +258,7 @@ async fn a_self_asserted_witness_can_forge_age_assurance_about_a_third_party_tod "forged-age-assurance", attestation_type::SCORES, "abuse-victim", - scores_envelope("age_assurance:level:adult:v1", "abuse-victim", 1.0), + scores_envelope("age_assurance:provider:adult:v1", "abuse-victim", 1.0), Utc::now(), ) .await; @@ -283,7 +283,7 @@ async fn a_self_asserted_witness_can_forge_age_assurance_about_a_third_party_tod && r.attestation_envelope .get("dimension") .and_then(serde_json::Value::as_str) - == Some("age_assurance:level:adult:v1")), + == Some("age_assurance:provider:adult:v1")), "the forged age-assurance row reads back as a witness-attested level" ); } @@ -701,7 +701,7 @@ async fn a_two_key_sybil_still_inflates_its_own_capacity() { attestation_type::SCORES, "sybil-scorer", serde_json::json!({ - "dimension": "consent:state:granted:analyze:v1", + "dimension": "consent:state:granted:v1", "score": 1.0, "confidence": 1.0, "epistemic_mode": "direct", diff --git a/tests/admin_ops.rs b/tests/admin_ops.rs index b5739886..3ac2dda3 100644 --- a/tests/admin_ops.rs +++ b/tests/admin_ops.rs @@ -413,6 +413,10 @@ async fn put_community(engine: &Engine, community_id: &str, founder: &str) { .sign_hybrid(&canonical) .await .expect("sign community"); + let founder_sig = engine + .sign_hybrid(&canonical) + .await + .expect("the founder (this node) co-signs"); engine .federation_directory() .put_community(SignedCommunity { @@ -421,6 +425,14 @@ async fn put_community(engine: &Engine, community_id: &str, founder: &str) { scrub_signature_classical: BASE64.encode(&sig.classical.signature), scrub_signature_pqc: Some(BASE64.encode(&sig.pqc.signature)), supersede_proof: None, + // persist v52 Q1 (CIRISPersist#955): the founding member co-signs + // — here the founder is this node's own key, so the engine signs. + cosignatures: vec![ciris_persist::federation::types::RosterCosignature { + authority_key_id: founder.to_string(), + scrub_signature_classical: BASE64.encode(&founder_sig.classical.signature), + scrub_signature_pqc: Some(BASE64.encode(&founder_sig.pqc.signature)), + }], + lineage: Vec::new(), }) .await .expect("put_community"); diff --git a/tests/canonical_relays_announced.rs b/tests/canonical_relays_announced.rs new file mode 100644 index 00000000..1ce9a1ca --- /dev/null +++ b/tests/canonical_relays_announced.rs @@ -0,0 +1,371 @@ +//! **A canonical relays the devices people announced** (CC 5.4.6, +//! CIRISConstitution#111; CIRISServer#655) — in process, against edge's own +//! advertise path. +//! +//! The native harness measured the fault (`harness/native/topologies/ +//! selffiles.yaml`): person `one` owns D1 and D2, both announced; X peers D1 +//! and the canonical C; X's public roster read of `one` lists ONE device. C +//! holds D2's key record and occurrence and offers neither, because the three +//! `SelfOwn` planes advertise only the node's self-publish set. +//! +//! This file drives the fix where it bites — `DirectoryStateAdapter:: +//! local_refs(kind)` toward an UNCONSENTED peer, the exact call the responder +//! makes once per round — with the SAME selector compose installs +//! (`announced_relay::selector_for_sets` over `announced_relay_sets`). The +//! assertions are on content hashes, computed the way edge advertises them +//! (`sha256(serde_json::to_vec(row))`), so "offered" means the ref a peer's +//! Diff would actually want: +//! +//! 1. **The fault, reproduced**: with no selector, C offers X none of D2's rows. +//! 2. **The cure**: with the relay, C offers X D2's key record, its occurrence, +//! and the owner's key record (without which X must refuse both). +//! 3. **The privacy negatives**: D3 — the same person's node, NOT announced — +//! is in no relayed set and C offers X neither its key nor its occurrence; +//! the person key is never an occurrence subject; and C's own rows are still +//! offered (the answer replaces the self set, so it must contain it). +//! +//! The planes the relay must never touch — routes (`TransportDestination`), +//! consent grants and every self-scoped row (the Attestation plane) — are +//! pinned by the unit tests in `src/announced_relay.rs` +//! (`routes_and_every_other_plane_are_never_relayed`), because the selector +//! answers `None` for them before any row is read. +//! +//! `multi_thread` is load-bearing: `DirectoryStateAdapter` bridges edge's sync +//! trait to persist via `block_in_place`. + +use std::collections::HashSet; +use std::sync::Arc; + +use ciris_edge::replication::{ + DirectoryStateAdapter, EnvelopeKind, FederationDirectoryReplicationBridge, StateProvider, +}; +use ciris_keyring::MlDsa65SoftwareSigner; +use ciris_persist::federation::types::{cohort_scope, identity_type}; +use ciris_persist::federation::SignedKeyRecord; +use ciris_persist::prelude::{Engine, HybridPolicy, LocalSigner}; +use ciris_server::announced_relay::{announced_relay_sets, selector_for_sets, RelaySets}; +use ed25519_dalek::SigningKey; +use sha2::{Digest, Sha256}; + +#[allow(dead_code)] // one fixture, several binaries: each uses a different subset +mod support { + include!("support/drive_fixture.rs"); +} +use support::*; + +/// A device's substrate, keyed by its own hybrid signer. +async fn device_engine(alias: &str, ed: u8, pqc: u8) -> Arc { + let pqc_signer = Arc::new( + MlDsa65SoftwareSigner::from_seed_bytes(&[pqc; 32], format!("{alias}-pqc")) + .expect("ML-DSA-65 seed"), + ); + let signer = Arc::new(LocalSigner::from_parts( + SigningKey::from_bytes(&[ed; 32]), + alias.to_string(), + Some(pqc_signer), + Some(format!("{alias}-pqc")), + )); + Arc::new( + Engine::with_signer(signer, "sqlite::memory:") + .await + .expect("in-memory engine"), + ) +} + +fn infra_scopes() -> Vec { + ciris_server::auth::ownership::OWNER_BINDING_INFRA_SCOPES + .iter() + .map(|s| s.to_string()) + .collect() +} + +/// The owner-binding at `self` — a node the person claimed and did NOT +/// announce (what `claim-remote` records before any announce). +async fn bind_unannounced(engine: &Engine, owner: &OwnerIdentity, node: &str) { + let binding = ciris_server::auth::ownership::build_signed_owner_binding( + &owner.signer().await, + node, + &infra_scopes(), + cohort_scope::SELF, + ) + .await + .expect("build a self-scoped owner-binding"); + ciris_server::auth::ownership::apply_signed_owner_binding( + engine, + node, + cohort_scope::SELF, + HybridPolicy::Strict, + &binding, + ) + .await + .expect("record the unannounced owner-binding"); +} + +/// Register the owner's user key alone — `bind_owner` without the binding. +async fn register_owner_key(engine: &Engine, owner: &OwnerIdentity) { + use ciris_persist::federation::types::{algorithm, KeyRecord}; + let now = chrono::Utc::now(); + let envelope = serde_json::json!({ "key_id": owner.key_id }); + let canonical = ciris_persist::verify::canonical::ceg_produce_canonicalize(&envelope) + .expect("canonicalize owner envelope"); + let record = KeyRecord { + key_id: owner.key_id.clone(), + pubkey_ed25519_base64: owner.pubkey_ed25519_base64.clone(), + pubkey_ml_dsa_65_base64: Some(owner.pubkey_ml_dsa_65_base64.clone()), + algorithm: algorithm::HYBRID.into(), + identity_type: identity_type::USER.into(), + identity_ref: owner.key_id.clone(), + valid_from: now, + valid_until: None, + registration_envelope: envelope, + original_content_hash: hex::encode(Sha256::digest(&canonical)), + scrub_signature_classical: String::new(), + scrub_signature_pqc: None, + scrub_key_id: owner.key_id.clone(), + scrub_timestamp: now, + pqc_completed_at: Some(now), + persist_row_hash: String::new(), + capability_roles: Vec::new(), + attestation_evidence: None, + consent_role: None, + additional_scrubs: Vec::new(), + }; + engine + .federation_directory() + .put_public_key(SignedKeyRecord { record }) + .await + .expect("register the owner's user key"); +} + +/// The signed occurrence `device` provisioned for `owner`, as replication +/// would carry it. +async fn signed_occurrence_of( + device: &Engine, + owner: &str, + occurrence: &str, +) -> ciris_persist::federation::SignedIdentityOccurrence { + device + .federation_directory() + .list_signed_identity_occurrences_for(owner) + .await + .expect("signed occurrences") + .into_iter() + .find(|o| o.identity_occurrence.occurrence_key_id == occurrence) + .expect("the provisioned occurrence rides the signed plane") +} + +fn hash_of(row: &T) -> [u8; 32] { + Sha256::digest(serde_json::to_vec(row).expect("serialize row")).into() +} + +/// The content hash edge advertises for `key_id`'s key record on `engine`. +async fn key_hash(engine: &Engine, key_id: &str) -> [u8; 32] { + let served = engine + .federation_directory() + .list_signed_key_records_since(None, 10_000) + .await + .expect("key records"); + let rec = served + .into_iter() + .find(|s| s.record.key_id == key_id) + .unwrap_or_else(|| panic!("{key_id} is held here")); + hash_of(&SignedKeyRecord { record: rec.record }) +} + +/// The content hash edge advertises for the occurrence `occurrence` on `engine`. +async fn occurrence_hash(engine: &Engine, occurrence: &str) -> [u8; 32] { + let served = engine + .federation_directory() + .list_signed_identity_occurrences_since(None, 10_000) + .await + .expect("occurrences"); + let row = served + .into_iter() + .find(|s| s.occurrence.identity_occurrence.occurrence_key_id == occurrence) + .unwrap_or_else(|| panic!("occurrence {occurrence} is held here")); + hash_of(&row.occurrence) +} + +async fn offered( + bridge: Arc, + peer: &str, + kind: EnvelopeKind, +) -> HashSet<[u8; 32]> { + DirectoryStateAdapter::new(bridge) + .with_peer(peer.to_owned()) + .local_refs(kind) + .await + .into_iter() + .map(|r| r.envelope_hash) + .collect() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn the_canonical_offers_an_announced_device_to_a_stranger_and_never_an_unannounced_one() { + init_tracing(); + let one = OwnerIdentity::mint().await; + + // D2 — announced (the owner-binding is at `federation`). Its occurrence + // is provisioned on D2 itself and carried to C signed, as on the wire. + let d2 = device_engine("relay-d2", 0xD1, 0xD2).await; + let d2_key = register_self(&d2).await; + bind_owner(&d2, &one, &d2_key).await; + let (d2_occ, _) = ciris_server::backend::provision_engine_occurrence(&d2, &one.key_id) + .await + .expect("D2 provisions its occurrence"); + + // D3 — the SAME person's node, claimed and NOT announced. + let d3 = device_engine("relay-d3", 0xE1, 0xE2).await; + let d3_key = register_self(&d3).await; + register_owner_key(&d3, &one).await; + bind_unannounced(&d3, &one, &d3_key).await; + let (d3_occ, _) = ciris_server::backend::provision_engine_occurrence(&d3, &one.key_id) + .await + .expect("D3 provisions its occurrence"); + + // C — the canonical. It holds what replication brought it: both devices' + // keys, the owner's key, both owner-bindings at their own scopes, both + // signed occurrences. + let c = node_engine().await; + let c_key = register_self(&c).await; + seed_key(&c, &d2_key, 0xD1, 0xD2, identity_type::NODE).await; + seed_key(&c, &d3_key, 0xE1, 0xE2, identity_type::NODE).await; + bind_owner(&c, &one, &d2_key).await; // owner key + the ANNOUNCED binding + bind_unannounced(&c, &one, &d3_key).await; + for (dev, occ) in [(&d2, &d2_occ), (&d3, &d3_occ)] { + c.federation_directory() + .put_identity_occurrence(signed_occurrence_of(dev, &one.key_id, occ).await) + .await + .expect("C admits the device's signed occurrence"); + } + + // The relay sets C computes — the announced device and its owner, nothing + // else. + let sets = announced_relay_sets(&c).await.expect("relay sets"); + assert_eq!( + sets, + RelaySets { + announced_nodes: vec![d2_key.clone()], + owners: vec![one.key_id.clone()], + }, + "exactly the announced device and its owner" + ); + + let x = "x-another-persons-node-never-consented"; + let own = vec![c_key.clone()]; + let bridge = |relay: Option| { + let own_for_provider = own.clone(); + Arc::new( + FederationDirectoryReplicationBridge::new(c.federation_directory(), Arc::new(Vec::new)) + .with_local_key_id(Some(c_key.clone())) + .with_self_provider(Some(Arc::new(move || own_for_provider.clone()))) + .with_kind_publish_selector(Some(selector_for_sets(own.clone(), relay))), + ) + }; + + let d2_key_h = key_hash(&c, &d2_key).await; + let one_key_h = key_hash(&c, &one.key_id).await; + let d3_key_h = key_hash(&c, &d3_key).await; + let c_key_h = key_hash(&c, &c_key).await; + let d2_occ_h = occurrence_hash(&c, &d2_occ).await; + let d3_occ_h = occurrence_hash(&c, &d3_occ).await; + + // 1. THE FAULT: not a relay → C offers X its own record and nothing of + // one's. + let keys = offered(bridge(None), x, EnvelopeKind::Key).await; + assert!( + keys.contains(&c_key_h), + "C always offers its own key record" + ); + assert!( + !keys.contains(&d2_key_h) && !keys.contains(&one_key_h), + "precondition: without the relay C withholds the announced device (the measured fault)" + ); + let occs = offered(bridge(None), x, EnvelopeKind::IdentityOccurrence).await; + assert!( + !occs.contains(&d2_occ_h), + "precondition: no relayed occurrence" + ); + + // 2. THE CURE: the relay offers the announced device, its occurrence, and + // the owner's key record — to a peer nothing consents to. + let keys = offered(bridge(Some(sets.clone())), x, EnvelopeKind::Key).await; + assert!( + keys.contains(&c_key_h), + "the relay still offers C's own record" + ); + assert!( + keys.contains(&d2_key_h), + "the announced device's key is relayed" + ); + assert!(keys.contains(&one_key_h), "its owner's key is relayed"); + let occs = offered( + bridge(Some(sets.clone())), + x, + EnvelopeKind::IdentityOccurrence, + ) + .await; + assert!( + occs.contains(&d2_occ_h), + "the announced device's occurrence is relayed (and passes CIRISEdge#682's announce gate)" + ); + + // 3. THE NEGATIVES: the unannounced node is in no relayed set. + assert!( + !sets.announced_nodes.contains(&d3_key) && !sets.owners.contains(&d3_key), + "an unannounced node is never in the relay sets" + ); + assert!( + !keys.contains(&d3_key_h), + "C never relays an unannounced node's key record" + ); + assert!( + !occs.contains(&d3_occ_h), + "C never relays an unannounced node's occurrence" + ); +} + +/// The predicate for "the canonical": a node relays only when it holds +/// `infra:serve` from a root it trusts. A fresh node does not; the same node +/// after the root-side legs and its own acceptance does. +#[cfg(feature = "test-anchor")] +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn only_a_node_holding_infra_serve_from_a_trusted_root_relays() { + init_tracing(); + let c = node_engine().await; + let c_key = register_self(&c).await; + let own = vec![c_key.clone()]; + assert!( + !ciris_server::announced_relay::serves_infrastructure(&c, &own).await, + "an ordinary node is not a relay" + ); + + let root = "relay-test-root"; + ciris_persist::federation::operational::test_support::establish_trust_root_side( + c.federation_directory().as_ref(), + root, + &c_key, + "infra:serve", + ) + .await + .expect("root-side trust legs: delegates_to(root -> C, infra:serve)"); + let core = ciris_persist::federation::envelope::EnvelopeCore::from_value( + serde_json::json!({ "scope": ["infra:attest", "infra:serve"] }), + ) + .expect("trust edge envelope"); + let mut accept = ciris_persist::federation::EmitAttestationInput::with_envelope( + ciris_persist::federation::types::attestation_type::DELEGATES_TO, + core, + cohort_scope::FEDERATION, + ); + accept.attested_key_id = Some(root.to_string()); + accept.subject_key_ids = vec![root.to_string()]; + c.emit_attestation_self(accept) + .await + .expect("C accepts the root"); + + assert!( + ciris_server::announced_relay::serves_infrastructure(&c, &own).await, + "a node holding infra:serve from a root it accepts is a relay" + ); +} diff --git a/tests/commons_surface.rs b/tests/commons_surface.rs index 0a6790ba..69568e1f 100644 --- a/tests/commons_surface.rs +++ b/tests/commons_surface.rs @@ -291,6 +291,30 @@ async fn try_put_community( .sign_hybrid(&canonical) .await .expect("sign community"); + // persist v52 Q1 (CIRISPersist#955): every listed member co-signs the + // founding record with the key `register_party` minted for them. + let node_key = engine.local_derived_key_id().await.expect("node key id"); + let mut cosignatures = Vec::new(); + for (k, _) in members { + // This node's own seat is signed by the engine; every party seat by + // the key `register_party` minted. + let c = if *k == node_key { + engine + .sign_hybrid(&canonical) + .await + .expect("this node co-signs its seat") + } else { + party_signer(k) + .sign_hybrid(&canonical) + .await + .expect("a founding member co-signs") + }; + cosignatures.push(ciris_persist::federation::types::RosterCosignature { + authority_key_id: (*k).to_string(), + scrub_signature_classical: BASE64.encode(&c.classical.signature), + scrub_signature_pqc: Some(BASE64.encode(&c.pqc.signature)), + }); + } engine .federation_directory() .put_community(SignedCommunity { @@ -299,6 +323,8 @@ async fn try_put_community( scrub_signature_classical: BASE64.encode(&sig.classical.signature), scrub_signature_pqc: Some(BASE64.encode(&sig.pqc.signature)), supersede_proof: None, + cosignatures, + lineage: Vec::new(), }) .await .map_err(|e| e.kind().to_string()) diff --git a/tests/community_crud.rs b/tests/community_crud.rs index 8d2bd77e..a81e0e8e 100644 --- a/tests/community_crud.rs +++ b/tests/community_crud.rs @@ -28,6 +28,28 @@ //! * the quorum flow (envelope → cosign on each signer's own node → assemble) //! for `unanimous` and `quorum:M/N`; //! * the CIRISPersist#907 gap as an IGNORED red test, named for the issue. +//! (0.5.218: persist v49 closed #907 — the test runs, un-ignored.) +//! +//! # Consent to join (0.5.218) — and the invite flow at persist v52 +//! +//! **Since edge v38.0.0 / persist v52.0.0** the doors below open as the invite +//! flow: `POST …/members` and `POST …/invites` answer 202 `invited`; the +//! invitee accepts on their own node (`POST /v1/self/invites/{p}/accept`); a +//! founder listing `…/invites` seats them. [`join_by_invitation`] walks that +//! real path wherever a test needs a member added under `founder_only`, and +//! `the_invite_flow_seats_only_the_person_who_accepted` pins it end to end. +//! A founding roster with others is a TEST-ONLY record the members CO-SIGN +//! (persist Q1; the server has no founding-cosign flow). The paragraph below +//! is the record of the interim cut. +//! +//! Every roster-growing door now answers 409 `membership.consent_required` +//! (CIRISConstitution#133, CIRISPersist#955, `FSD/MEMBERSHIP_INVITES.md`): a +//! direct add, a quorum add, and a create naming anyone but the founder. The +//! tests below assert each refusal; where a test's PURPOSE is what a +//! multi-member room does (read, remove, role, moderator, quorum, the #907 +//! read), the roster is built by the `test_only_*` fixtures, which write the +//! owner-signed record / widening rows straight through persist — the same +//! rows the routes wrote before 0.5.218, and never a production path. use std::path::{Path, PathBuf}; use std::sync::atomic::{AtomicU32, Ordering}; @@ -122,6 +144,19 @@ impl OwnerIdentity { .expect("re-open the owner's minted fed-ID") .0 } + + /// The same fed-ID in EDGE's signer type — what a record / widening is + /// signed with (the capsule's `edge_signer`). + async fn edge_signer(&self) -> ciris_edge::identity::LocalSigner { + ciris_server::identity::hardware_user_signers( + UserIdentityBackend::Software, + &self.alias, + self.seed_dir.clone(), + ) + .await + .expect("re-open the owner's minted fed-ID") + .1 + } } /// One node of the converged mesh. @@ -272,7 +307,8 @@ async fn serve( signer: Arc, seed_dir: PathBuf, ) -> (String, tokio::task::JoinHandle<()>) { - let app = contacts_chat::router(engine, signer, seed_dir, None, None); + let app = contacts_chat::router(Arc::clone(&engine), signer, seed_dir.clone(), None, None) + .merge(ciris_server::membership_invites::router(engine, seed_dir)); let listener = tokio::net::TcpListener::bind("127.0.0.1:0") .await .expect("bind"); @@ -377,24 +413,140 @@ async fn contact(on: &Node, who: &Node) { assert_eq!(s, 200, "add contact: {v}"); } -/// Found a room on `founder` with `members`, after making each a contact. -async fn found(founder: &Node, name: &str, members: &[&Node], extra: Value) -> String { +// ─── TEST-ONLY roster fixtures — the consent-to-join BYPASS ───────────────── +// +// Since 0.5.218 `POST /v1/communities` refuses a founding roster naming anyone +// but the founder, and every add is `membership.consent_required` +// (CIRISConstitution#133, CIRISPersist#955). These write the rows the routes +// wrote before — owner-signed, through persist's own doors — so the tests of +// what a multi-member room DOES keep a room to do it in. Under `tests/` only; +// there is no production bypass. + +/// TEST-ONLY (bypasses the consent-to-join door): found a room on `founder` +/// with `members` on its founding RECORD, after making each a contact — the +/// record `POST /v1/communities` wrote before 0.5.218, signed by the founder's +/// pen, plus the founder's content occurrence the route provisioned. `extra` +/// takes `consensus_protocol` and `tier` as the route did. +async fn test_only_found(founder: &Node, name: &str, members: &[&Node], extra: Value) -> String { for m in members { contact(founder, m).await; } - let mut body = json!({ - "name": name, - "members": members.iter().map(|m| m.owner.key_id.clone()).collect::>(), - }); - if let (Some(b), Some(e)) = (body.as_object_mut(), extra.as_object()) { - for (k, v) in e { - b.insert(k.clone(), v.clone()); - } + let protocol = extra["consensus_protocol"] + .as_str() + .unwrap_or("founder_only") + .to_owned(); + let community_id = ciris_edge::chat::new_room_community_key_id(); + let founded_at = chrono::DateTime::from_timestamp_millis(chrono::Utc::now().timestamp_millis()) + .expect("ms instant"); + let mut roster: Vec<(&str, Option<&str>)> = vec![( + founder.owner.key_id.as_str(), + Some(ciris_persist::federation::admission::MEMBER_ROLE_FOUNDER), + )]; + roster.extend(members.iter().map(|m| (m.owner.key_id.as_str(), None))); + let mut record = + ciris_edge::chat::community(&community_id, name, &roster, &protocol, founded_at) + .expect("TEST-ONLY: the room record"); + if extra["tier"] == json!("affiliations") { + record.policy_blob = Some(json!({ "cohort_scope": "affiliations" })); } - let (s, v) = post(founder, "/v1/communities", body).await; - assert_eq!(s, 201, "create: {v}"); - assert_eq!(v["kind"], "room"); - v["community_id"].as_str().expect("community_id").to_owned() + let canonical = + ceg_produce_canonicalize(&record.signing_envelope()).expect("canonicalize the room"); + let mut signed = ciris_edge::chat::signed_community(record, &founder.owner.edge_signer().await) + .await + .expect("TEST-ONLY: the founder signs the room"); + // persist v52 Q1 (CIRISPersist#955): a founding member is seated only if + // they SIGNED the founding record. The server has no founding-cosign flow, + // so this fixture plays each member co-signing with their own pen. + for m in members { + let (classical, pqc) = ciris_edge::identity::sign_bound_hybrid( + &m.owner.edge_signer().await, + &canonical, + "TEST-ONLY founding co-signature", + ) + .await + .expect("a founding member co-signs"); + signed + .cosignatures + .push(ciris_persist::federation::types::RosterCosignature { + authority_key_id: m.owner.key_id.clone(), + scrub_signature_classical: classical, + scrub_signature_pqc: pqc, + }); + } + founder + .engine + .federation_directory() + .put_community(signed) + .await + .expect("TEST-ONLY: put the multi-member room directly"); + ciris_server::backend::provision_engine_occurrence(&founder.engine, &founder.owner.key_id) + .await + .expect("the founder's content occurrence"); + community_id +} + +/// **`who` joins `room` the way a person does** (persist v52, CIRISPersist#955): +/// `on` invites them through the route, `who` accepts on THEIR node with +/// their own pen, and `on` — a founder of a `founder_only` room — lists the +/// invites, which seats the accepted invitee (edge's `widen_on_acceptance` +/// with `on`'s person key). No bypass: every row is one a production node +/// writes. Before 0.5.218's v52 adopt this was `test_only_widen`, a direct +/// owner-signed widening. +async fn join_by_invitation(on: &Node, room: &str, who: &Node, role: Option<&str>) { + // Strictly after any roster event just written (a re-add after a removal). + tokio::time::sleep(std::time::Duration::from_millis(5)).await; + let (s, v) = post( + on, + &format!("/v1/communities/{room}/invites"), + json!({ "key_id": who.owner.key_id, "role": role }), + ) + .await; + assert_eq!( + (s, v["state"].as_str()), + (202, Some("invited")), + "invite: {v}" + ); + let proposal = v["proposal_id"].as_str().expect("proposal_id").to_owned(); + let (s, v) = post( + who, + &format!("/v1/self/invites/{proposal}/accept"), + json!({}), + ) + .await; + assert_eq!( + (s, v["state"].as_str()), + (200, Some("accepted")), + "accept: {v}" + ); + let (s, v) = get(on, &format!("/v1/communities/{room}/invites")).await; + assert_eq!(s, 200, "list invites: {v}"); + assert!( + v["seated_now"] + .as_array() + .is_some_and(|a| a.contains(&json!(who.owner.key_id))), + "the founder's read seats the accepted invitee: {v}" + ); +} + +/// A roster-growing request answered as an INVITATION (persist v52): 202, +/// `state: invited`, a proposal id — never a membership. +fn invited(s: u16, v: &Value) { + assert_eq!( + (s, v["state"].as_str()), + (202, Some("invited")), + "an add is an invitation: {v}" + ); + assert!(v["proposal_id"].is_string(), "{v}"); +} + +/// Under `founder_only` only a founder invites (persist's proposer rule, +/// named by the room's own id). +fn founder_invites_only(s: u16, v: &Value) { + assert_eq!( + (s, reason(v)), + (403, "community.not_authorized"), + "under founder_only only a founder invites: {v}" + ); } /// The active roster by persist's fold, sorted. @@ -532,7 +684,7 @@ async fn appoint_moderator(appointer: &Node, who: &Node) { async fn a_three_member_room_where_all_three_read_each_other() { let nodes = mesh(3).await; let (a, b, c) = (&nodes[0], &nodes[1], &nodes[2]); - let room = found(a, "the three of us", &[b, c], json!({})).await; + let room = test_only_found(a, "the three of us", &[b, c], json!({})).await; assert!( room.starts_with(ciris_edge::chat::ROOM_COMMUNITY_PREFIX), "{room}" @@ -636,20 +788,27 @@ async fn create_refuses_with_a_typed_reason_each() { "community.bad_consensus_protocol", ), ( - json!({ "name": "x", "consensus_protocol": "quorum:1/2", "members": [b.owner.key_id] }), + // A quorum's N is the roster size: a founder alone is not a 2/3. + json!({ "name": "x", "consensus_protocol": "quorum:2/3" }), 400, "community.bad_consensus_protocol", ), + // CONSENT TO JOIN (0.5.218): a founding roster naming anyone but the + // founder is refused — contact or not, whatever the protocol. ( - // A quorum's N is the roster size: 2 people are not a 2/3. - json!({ "name": "x", "consensus_protocol": "quorum:2/3", "members": [b.owner.key_id] }), - 400, - "community.bad_consensus_protocol", + json!({ "name": "x", "consensus_protocol": "quorum:1/2", "members": [b.owner.key_id] }), + 409, + "membership.founding_member_unsigned", + ), + ( + json!({ "name": "x", "members": [b.owner.key_id] }), + 409, + "membership.founding_member_unsigned", ), ( json!({ "name": "x", "members": [c.owner.key_id] }), - 403, - "community.not_a_contact", + 409, + "membership.founding_member_unsigned", ), (json!({ "nombre": "x" }), 400, "community.malformed_body"), ] { @@ -699,11 +858,12 @@ async fn create_refuses_with_a_typed_reason_each() { let v: Value = resp.json().await.expect("json"); assert_eq!(reason(&v), "community.author_signer_unavailable", "{v}"); - // The happy path, for contrast, with the default protocol and tier. + // The happy path, for contrast, with the default protocol and tier: the + // founder alone (naming yourself is not naming anyone else). let (s, v) = post( a, "/v1/communities", - json!({ "name": " Book club ", "members": [b.owner.key_id] }), + json!({ "name": " Book club ", "members": [a.owner.key_id] }), ) .await; assert_eq!(s, 201, "{v}"); @@ -711,7 +871,7 @@ async fn create_refuses_with_a_typed_reason_each() { assert_eq!(v["tier"], "community"); assert_eq!(v["consensus_protocol"], "founder_only"); assert_eq!(v["my_role"], "founder"); - assert_eq!(v["member_count"], 2); + assert_eq!(v["member_count"], 1); } /// A second edge signer over the same node key — the router needs one, and the @@ -728,7 +888,7 @@ async fn node_edge_signer_for(n: &Node) -> Arc = inbox["invites"] + .as_array() + .unwrap() + .iter() + .filter(|i| i["proposal_id"] == json!(bob_p)) + .collect(); + assert_eq!(mine.len(), 1, "the invitee's inbox holds it: {inbox}"); + assert_eq!(mine[0]["group_kind"], "community"); + assert_eq!(mine[0]["group_id"], json!(room)); + let (s, v) = post(c, &format!("/v1/self/invites/{bob_p}/accept"), json!({})).await; + assert_eq!((s, reason(&v)), (403, "membership.not_the_invitee"), "{v}"); + let (s, v) = post(b, &format!("/v1/self/invites/{bob_p}/decline"), json!({})).await; + assert_eq!((s, v["state"].as_str()), (200, Some("declined")), "{v}"); + let (s, v) = post(b, &format!("/v1/self/invites/{bob_p}/accept"), json!({})).await; + assert_eq!( + (s, reason(&v)), + (409, "membership.already_answered"), + "a decline is final: {v}" + ); + let (_, inbox) = get(b, "/v1/self/invites").await; + assert!( + !inbox["invites"] + .as_array() + .unwrap() + .iter() + .any(|i| i["proposal_id"] == json!(bob_p)), + "an answered invitation leaves the inbox: {inbox}" + ); + + // Only a founder invites under founder_only; an expiry is 1..=30 days. + let (s, v) = post( + a, + &format!("/v1/communities/{room}/invites"), + json!({ "key_id": c.owner.key_id, "expires_in_days": 31 }), + ) + .await; + assert_eq!((s, reason(&v)), (400, "membership.bad_expiry"), "{v}"); + + // Carol's invitation is withdrawn before she answers. + let (s, v) = post( + a, + &format!("/v1/communities/{room}/invites"), + json!({ "key_id": c.owner.key_id }), + ) + .await; + invited(s, &v); + let carol_p = v["proposal_id"].as_str().unwrap().to_owned(); + let (s, v) = delete(a, &format!("/v1/communities/{room}/invites/{carol_p}")).await; + assert_eq!((s, v["state"].as_str()), (200, Some("withdrawn")), "{v}"); + let (s, v) = delete(a, &format!("/v1/communities/{room}/invites/{carol_p}")).await; + assert_eq!((s, reason(&v)), (409, "membership.invite_closed"), "{v}"); + let (_, inbox) = get(c, "/v1/self/invites").await; + assert!( + !inbox["invites"] + .as_array() + .unwrap() + .iter() + .any(|i| i["proposal_id"] == json!(carol_p)), + "a withdrawn invitation leaves the inbox: {inbox}" + ); + let (s, v) = post(c, &format!("/v1/self/invites/{carol_p}/accept"), json!({})).await; + assert_eq!((s, reason(&v)), (410, "membership.invite_expired"), "{v}"); + + // A lapsed invitation (edge's own door, two seconds — the route's floor is + // a day) is refused when answered. + let short = ciris_edge::membership::propose( + a.engine.federation_directory().as_ref(), + ciris_edge::membership::GroupScope::Community, + &room, + &c.owner.key_id, + None, + chrono::Utc::now() + chrono::Duration::seconds(2), + &a.owner.edge_signer().await, + ) + .await + .expect("a short-lived proposal"); + tokio::time::sleep(std::time::Duration::from_millis(2_500)).await; + let (s, v) = post( + c, + &format!("/v1/self/invites/{}/accept", short.attestation_id), + json!({}), + ) + .await; + assert_eq!((s, reason(&v)), (410, "membership.invite_expired"), "{v}"); + + // Dave accepts, and the founder's read seats him. + join_by_invitation(a, &room, d, None).await; + let (s, v) = get(a, &format!("/v1/communities/{room}/invites")).await; + assert_eq!(s, 200, "{v}"); + let states: std::collections::BTreeMap = v["invites"] + .as_array() + .unwrap() + .iter() + .map(|i| { + ( + i["invitee_key_id"].as_str().unwrap().to_owned() + + "/" + + i["proposal_id"].as_str().unwrap(), + i["state"].as_str().unwrap().to_owned(), + ) + }) + .collect(); + assert_eq!( + states + .get(&format!("{}/{bob_p}", b.owner.key_id)) + .map(String::as_str), + Some("declined"), + "{v}" + ); + assert_eq!( + states + .get(&format!("{}/{carol_p}", c.owner.key_id)) + .map(String::as_str), + Some("withdrawn"), + "{v}" + ); + assert!( + states + .iter() + .any(|(k, st)| k.starts_with(&d.owner.key_id) && st == "joined"), + "{v}" + ); + let roster = fold(&a.engine, &room).await; + assert_eq!( + roster, + sorted(&[&a.owner.key_id, &d.owner.key_id]), + "only the person who accepted is in" + ); + + // Dave is a plain member: under founder_only he cannot invite. + let (s, v) = post( + d, + &format!("/v1/communities/{room}/invites"), + json!({ "key_id": b.owner.key_id }), + ) + .await; + founder_invites_only(s, &v); +} + +/// **A quorum room seats an accepted invitee only when M of N sign the add.** +/// Alice invites Dave into a 3-of-3 room; Dave accepts; Alice's list does NOT +/// seat him (one signature is not the room's rule — "accepted, awaiting the +/// group", persist FSD §4); the add through envelope → cosign → assemble, +/// signed by all three, does. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_quorum_room_seats_an_accepted_invitee_by_a_cosigned_widening() { + let nodes = mesh(4).await; + let (a, b, c, d) = (&nodes[0], &nodes[1], &nodes[2], &nodes[3]); + let room = test_only_found( + a, + "three of three", + &[b, c], + json!({ "consensus_protocol": "quorum:3/3" }), + ) + .await; + contact(a, d).await; + let (s, v) = post( + b, + &format!("/v1/communities/{room}/invites"), + json!({ "key_id": d.owner.key_id }), + ) + .await; + invited(s, &v); + let p = v["proposal_id"].as_str().unwrap().to_owned(); + let (s, v) = post(d, &format!("/v1/self/invites/{p}/accept"), json!({})).await; + assert_eq!((s, v["state"].as_str()), (200, Some("accepted")), "{v}"); + let (s, v) = get(a, &format!("/v1/communities/{room}/invites")).await; + assert_eq!(s, 200, "{v}"); + assert!( + v["seated_now"].as_array().is_some_and(Vec::is_empty), + "a read never seats anyone in a quorum room: {v}" + ); + assert!(!fold(&a.engine, &room).await.contains(&d.owner.key_id)); + + let (s, v) = post( + a, + &format!("/v1/communities/{room}/changes/envelope"), + json!({ "op": "add", "key_id": d.owner.key_id }), + ) + .await; + assert_eq!(s, 200, "{v}"); + let env = v["change_envelope"].clone(); + let mut sigs = vec![v["signatures"][0].clone()]; + for n in [b, c] { + let (s, v) = post( + n, + &format!("/v1/communities/{room}/changes/cosign"), + json!({ "change_envelope": env }), + ) + .await; + assert_eq!(s, 200, "{v}"); + sigs.push(v["signature"].clone()); + } + let (s, v) = post( + a, + &format!("/v1/communities/{room}/changes/assemble"), + json!({ "change_envelope": env, "signatures": sigs }), + ) + .await; + assert_eq!( + s, 200, + "the group's M-of-N widening seats the accepted invitee: {v}" + ); + assert!(fold(&a.engine, &room).await.contains(&d.owner.key_id)); +} diff --git a/tests/contacts_chat.rs b/tests/contacts_chat.rs index 979dc3a5..a515b125 100644 --- a/tests/contacts_chat.rs +++ b/tests/contacts_chat.rs @@ -772,6 +772,34 @@ async fn open_chat( .as_str() .expect("community_id") .to_string(); + // edge v38 / persist v52 (CIRISPersist#955): the room is founded by the + // owner ALONE and the contact is PROPOSED. The fixture plays the contact + // accepting with their own key, then the owner's second call seats them + // (the opener's pen widens on a held acceptance) — the two-step join a + // real pair of nodes performs, on one node. + assert_eq!(json["state"], "invited", "{json}"); + let proposal_id = json["proposal_id"] + .as_str() + .expect("the pair room's invitation") + .to_string(); + ciris_edge::membership::reply( + engine.federation_directory().as_ref(), + &proposal_id, + true, + contact_edge_signer().await.as_ref(), + ) + .await + .expect("the contact accepts the pair room's invitation"); + let resp = client + .post(format!("{base}/v1/chat")) + .bearer_auth(owner) + .json(&serde_json::json!({ "key_id": CONTACT_KEY_ID })) + .send() + .await + .expect("POST /v1/chat (seat the contact)"); + assert_eq!(resp.status(), 200); + let json: serde_json::Value = resp.json().await.expect("start chat json"); + assert_eq!(json["state"], "open", "the contact is seated: {json}"); // WHICH HALF THE FIXTURE PUBLISHES, stated rather than assumed. `PairRole` // is order-free — it hands the smaller fed-ID the creator's role — so the @@ -1290,14 +1318,29 @@ async fn chat_creation_is_convergent_and_idempotent_for_a_pair() { "a room the two ends can only reach by agreeing who initiated is not a room" ); - // The row is a real 2-member persist Community. + // A real persist Community, founded by the opener ALONE (persist v52 Q1 + // seats only a founding record's signers) — and both people are in its + // FOLD once the contact accepted and the opener's pen widened. let community = engine .federation_directory() .lookup_community(&community_id) .await .expect("lookup_community") .expect("the community must exist after POST /v1/chat"); - let mut members: Vec = community.members.iter().map(|m| m.key_id.clone()).collect(); + let founding: Vec = community.members.iter().map(|m| m.key_id.clone()).collect(); + assert_eq!( + founding, + vec![owner_id.key_id.clone()], + "founded by its opener alone" + ); + let mut members: Vec = engine + .federation_directory() + .active_community_members(&community_id) + .await + .expect("active roster") + .into_iter() + .map(|m| m.key_id) + .collect(); members.sort(); let mut expected = vec![owner_id.key_id.clone(), CONTACT_KEY_ID.to_string()]; expected.sort(); @@ -1316,6 +1359,7 @@ async fn chat_creation_is_convergent_and_idempotent_for_a_pair() { assert_eq!(again["community_id"], serde_json::json!(community_id)); assert_eq!(again["freshly_created"], false); assert_eq!(again["cohort_scope"], cohort_scope::COMMUNITY); + assert_eq!(again["state"], "open", "{again}"); } #[tokio::test] @@ -1582,6 +1626,49 @@ async fn a_withdrawn_message_reads_back_as_withdrawn() { // ─── 4. THE contextual-integrity line ─────────────────────────────────────── +/// A `Community` authored by this NODE and CO-SIGNED by every other listed +/// member's seeded key (`(key_id, ed_seed, pqc_seed)` as `seed_*_key` +/// registered it) — persist v52 (CIRISPersist#955, Q1) seats a founding +/// member only if they signed the founding record. +async fn put_community_cosigned( + engine: &Engine, + community: ciris_persist::federation::types::Community, + cosigners: &[(&str, u8, u8)], +) -> Result<(), ciris_persist::federation::Error> { + use ciris_keyring::PqcSigner as _; + use ed25519_dalek::Signer as _; + let canonical = + ceg_produce_canonicalize(&community.signing_envelope()).expect("canonicalize community"); + let sig = engine.sign_hybrid(&canonical).await.expect("node signs"); + let mut cosignatures = Vec::new(); + for (k, ed_seed, pqc_seed) in cosigners { + let ed = SigningKey::from_bytes(&[*ed_seed; 32]); + let mldsa = MlDsa65SoftwareSigner::from_seed_bytes(&[*pqc_seed; 32], format!("{k}-pqc")) + .expect("ML-DSA-65 seed"); + let ed_sig = ed.sign(&canonical).to_bytes(); + let mut bound = canonical.clone(); + bound.extend_from_slice(&ed_sig); + let pqc_sig = mldsa.sign(&bound).await.expect("ml-dsa sign"); + cosignatures.push(ciris_persist::federation::types::RosterCosignature { + authority_key_id: (*k).to_owned(), + scrub_signature_classical: BASE64.encode(ed_sig), + scrub_signature_pqc: Some(BASE64.encode(pqc_sig)), + }); + } + engine + .federation_directory() + .put_community(ciris_persist::federation::types::SignedCommunity { + community, + authority_key_id: engine.local_derived_key_id().await.expect("node id"), + scrub_signature_classical: BASE64.encode(&sig.classical.signature), + scrub_signature_pqc: Some(BASE64.encode(&sig.pqc.signature)), + supersede_proof: None, + cosignatures, + lineage: Vec::new(), + }) + .await +} + /// A community of two strangers, authored by this node (as a replicated row /// would be), holding a real message. The owner is deliberately not on it. async fn strangers_community(engine: &Engine) -> (String, LocalSigner, String) { @@ -1590,8 +1677,11 @@ async fn strangers_community(engine: &Engine) -> (String, LocalSigner, String) { seed_user_key_at(engine, &a_key_id, 0xC0, 0xC1).await; let community_id = pair_community_key_id(&a_key_id, STRANGER_B_KEY_ID); let now = chrono::Utc::now(); - engine - .put_community_self_signed(Community { + // persist v52 Q1: both founding members CO-SIGN (this node vouches as + // authority but is in neither seat). + put_community_cosigned( + engine, + Community { community_key_id: community_id.clone(), community_name: format!("{a_key_id} <-> {STRANGER_B_KEY_ID}"), members: [a_key_id.as_str(), STRANGER_B_KEY_ID] @@ -1606,9 +1696,14 @@ async fn strangers_community(engine: &Engine) -> (String, LocalSigner, String) { consensus_protocol: "unanimous".to_string(), policy_blob: None, persist_row_hash: String::new(), - }) - .await - .expect("author the strangers' community"); + }, + &[ + (a_key_id.as_str(), 0xC0, 0xC1), + (STRANGER_B_KEY_ID, 0xD0, 0xD1), + ], + ) + .await + .expect("author the strangers' community"); (community_id, a_signer, a_key_id) } @@ -3047,8 +3142,12 @@ async fn a_poisoned_roster_under_the_pair_id_is_refused() { let community_id = pair_community_key_id(&owner_id.key_id, CONTACT_KEY_ID); let now = chrono::Utc::now(); - engine - .put_community_self_signed(Community { + // The planter co-signs as every listed member (persist v52 Q1) — the + // stowaway's room is a well-formed founding record; the front door must + // refuse it on its SHAPE. + put_community_cosigned( + &engine, + Community { community_key_id: community_id.clone(), community_name: "poisoned".to_string(), members: [owner_id.key_id.as_str(), CONTACT_KEY_ID, STRANGER_A_KEY_ID] @@ -3063,9 +3162,14 @@ async fn a_poisoned_roster_under_the_pair_id_is_refused() { consensus_protocol: "unanimous".to_string(), policy_blob: None, persist_row_hash: String::new(), - }) - .await - .expect("pre-plant the poisoned room"); + }, + &[ + (CONTACT_KEY_ID, 0xB0, 0xB1), + (STRANGER_A_KEY_ID, 0xC0, 0xC1), + ], + ) + .await + .expect("pre-plant the poisoned room"); let resp = client .post(format!("{base}/v1/chat")) @@ -3417,3 +3521,360 @@ async fn a_code_is_refused_when_the_held_key_breaks_its_commitment() { "{body}" ); } + +// ─── The joiner asks first (edge v38 two-step pair room) ──────────────────── + +/// Two contacts whose fed-IDs sort BEFORE the owner's (`alice-owner-…`), so in +/// a pair room with them the owner is the JOINER (`PairRole`). +const AARON_KEY_ID: &str = "aaron-v1"; +const ABEL_KEY_ID: &str = "abel-v1"; + +/// A seeded person as edge's signer, under the plain alias it is registered as. +fn edge_signer_for(alias: &str, ed: u8, pqc: u8) -> ciris_edge::identity::LocalSigner { + let classical = ciris_keyring::SealedEd25519Signer::adopt( + alias.to_string(), + keystore_dir(alias), + &[ed; 32], + ) + .expect("adopt the sealed ed25519 key"); + let pqc = MlDsa65SoftwareSigner::from_seed_bytes(&[pqc; 32], format!("{alias}-pqc")) + .expect("ML-DSA-65 seed"); + ciris_edge::identity::LocalSigner::new( + alias.to_string(), + Arc::new(classical), + Some(Arc::new(pqc)), + ) +} + +fn intents_on_disk(seed_dir: &std::path::Path) -> Vec { + std::fs::read(seed_dir.join(ciris_server::pair_intents::PAIR_INTENTS_FILE)) + .ok() + .and_then(|b| serde_json::from_slice(&b).ok()) + .unwrap_or_default() +} + +/// **The joiner asks first, the node restarts, the creator opens — and the +/// joiner is seated.** Nothing the person did not ask for is accepted. +/// +/// The owner is the JOINER against Aaron. Their `POST /v1/chat` answers +/// `awaiting_invitation` and founds nothing, but their request is written to +/// `pair-intents.json` (`{pair_id, contact_person, asked_at}`). The router is +/// then torn down and a fresh one started on the same store and seed dir — a +/// restart: nothing of the request survives in memory. Aaron's node opens the +/// room (its record and its invitation land here as replication carries them), +/// and so does Abel's, into a room the owner NEVER asked for. The owner's next +/// read of Aaron's room accepts Aaron's invitation — the recorded intent's — +/// and leaves Abel's pending in the inbox. Aaron's node seats the owner; the +/// next read finds the owner seated and removes the intent. +#[tokio::test] +async fn a_joiner_who_asks_first_is_seated_across_a_restart_and_nothing_else_is_accepted() { + let (engine, base, owner, owner_id, handle) = fixture().await; + let client = reqwest::Client::new(); + seed_user_key(&engine, AARON_KEY_ID, 0xE0, 0xE1).await; + seed_user_key(&engine, ABEL_KEY_ID, 0xE2, 0xE3).await; + assert_eq!( + ciris_edge::chat::PairRole::of(&owner_id.key_id, AARON_KEY_ID), + ciris_edge::chat::PairRole::Joiner, + "the owner must be the joiner against {AARON_KEY_ID}" + ); + for k in [AARON_KEY_ID, ABEL_KEY_ID] { + let resp = client + .post(format!("{base}/v1/contacts")) + .bearer_auth(&owner) + .json(&serde_json::json!({ "key_id": k })) + .send() + .await + .expect("POST /v1/contacts"); + assert_eq!(resp.status(), 200, "add {k}: {:?}", resp.text().await); + } + let pair = pair_community_key_id(&owner_id.key_id, AARON_KEY_ID); + let abel_pair = pair_community_key_id(&owner_id.key_id, ABEL_KEY_ID); + + // The joiner asks first: nothing founded, the request kept on disk. + let resp = client + .post(format!("{base}/v1/chat")) + .bearer_auth(&owner) + .json(&serde_json::json!({ "key_id": AARON_KEY_ID })) + .send() + .await + .expect("POST /v1/chat"); + assert_eq!(resp.status(), 200); + let v: serde_json::Value = resp.json().await.expect("json"); + assert_eq!(v["state"], "awaiting_invitation", "{v}"); + assert_eq!(v["community_id"], serde_json::json!(pair)); + let dir = engine.federation_directory(); + assert!( + dir.lookup_community(&pair).await.expect("lookup").is_none(), + "the joiner founds nothing" + ); + let on_disk = intents_on_disk(&owner_id.seed_dir); + assert_eq!(on_disk.len(), 1, "{on_disk:?}"); + assert_eq!(on_disk[0]["pair_id"], serde_json::json!(pair)); + assert_eq!(on_disk[0]["contact_person"], AARON_KEY_ID); + assert!(on_disk[0]["asked_at"].is_string(), "{on_disk:?}"); + + // RESTART: a fresh router on the same store and seed dir. + handle.abort(); + let (base, _handle) = serve(Arc::clone(&engine), owner_id.seed_dir.clone()).await; + + // The creators open their rooms (as replication lands them here). + let aaron = edge_signer_for(AARON_KEY_ID, 0xE0, 0xE1); + let abel = edge_signer_for(ABEL_KEY_ID, 0xE2, 0xE3); + let now = chrono::Utc::now(); + let opened = ciris_edge::chat::open_pair_room( + dir.as_ref(), + AARON_KEY_ID, + &owner_id.key_id, + now, + now + chrono::Duration::days(1), + &aaron, + ) + .await + .expect("Aaron opens the room and invites the owner"); + let proposal = opened.proposal.expect("an invitation").attestation_id; + ciris_edge::chat::open_pair_room( + dir.as_ref(), + ABEL_KEY_ID, + &owner_id.key_id, + now, + now + chrono::Duration::days(1), + &abel, + ) + .await + .expect("Abel opens a room the owner never asked for"); + + // The owner reads Aaron's room: the recorded intent's invitation is + // accepted; Abel's is not. + let _ = client + .get(format!("{base}/v1/chat/{pair}/messages")) + .bearer_auth(&owner) + .send() + .await + .expect("GET messages"); + let pending: Vec = + ciris_edge::membership::pending_proposals_for(dir.as_ref(), &owner_id.key_id) + .await + .expect("inbox") + .into_iter() + .map(|p| p.group_key_id) + .collect(); + assert!( + !pending.contains(&pair), + "Aaron's invitation was accepted: {pending:?}" + ); + assert!( + pending.contains(&abel_pair), + "an invitation the person never asked for is NEVER auto-accepted: {pending:?}" + ); + + // Aaron's node seats the owner on the acceptance. + let acceptance = dir + .list_attestations_for(&owner_id.key_id) + .await + .expect("replies") + .into_iter() + .find(|r| { + ciris_edge::membership::dimension_of(r) + == Some(ciris_edge::membership::ACCEPTANCE_DIMENSION) + && r.attestation_envelope + .get("references_attestation_id") + .and_then(serde_json::Value::as_str) + == Some(proposal.as_str()) + }) + .expect("the owner's acceptance of Aaron's invitation"); + let widened = ciris_edge::membership::widen_on_acceptance( + dir.as_ref(), + &acceptance, + &ciris_edge::membership::MembershipWidener::new(vec![Arc::new(aaron)]), + ) + .await + .expect("Aaron's node widens"); + assert!( + matches!( + widened, + ciris_edge::membership::WidenOutcome::Widened { .. } + ), + "{widened:?}" + ); + let members: Vec = dir + .active_community_members(&pair) + .await + .expect("roster") + .into_iter() + .map(|m| m.key_id) + .collect(); + assert!(members.contains(&owner_id.key_id), "{members:?}"); + + // Seated: the next read removes the request. + let _ = client + .get(format!("{base}/v1/chat/{pair}/messages")) + .bearer_auth(&owner) + .send() + .await + .expect("GET messages"); + assert!( + intents_on_disk(&owner_id.seed_dir) + .iter() + .all(|i| i["pair_id"] != serde_json::json!(pair)), + "a seated room's intent is removed" + ); +} + +/// [`serve`], keeping the router's [`contacts_chat::PairRoomDriver`]. +async fn serve_with_driver( + engine: Arc, + seed_dir: PathBuf, +) -> ( + String, + tokio::task::JoinHandle<()>, + contacts_chat::PairRoomDriver, +) { + let signer = node_edge_signer(&engine).await; + let (app, driver) = contacts_chat::router_with_driver(engine, signer, seed_dir, None, None); + let listener = tokio::net::TcpListener::bind("127.0.0.1:0") + .await + .expect("bind ephemeral port"); + let addr = listener.local_addr().expect("local addr"); + let handle = tokio::spawn(async move { + let _ = axum::serve(listener, app).await; + }); + (format!("http://{addr}"), handle, driver) +} + +/// **A joiner who asks ONCE converges with nobody reading the room** (the +/// Docker chat ladder: node-b called `POST /v1/chat` once and never read the +/// room, so its KeyPackage was never published and node-a's send was 503). +/// +/// The owner (joiner against Aaron) makes exactly one request — `POST +/// /v1/chat` — and never another. Every later step on the owner's node is the +/// pair-room driver's own pass (`drive_once`, the body of its `pair_rooms` +/// loop): it accepts Aaron's invitation when it lands (and not Abel's, which +/// the person never asked for); once Aaron's node has seated the owner, the +/// next pass publishes the owner's MLS KeyPackage into the room — the row the +/// creator's Welcome answers — and the request is removed from disk. +#[tokio::test] +async fn a_joiner_who_asks_once_converges_with_nobody_reading_the_room() { + let (engine, base, owner, owner_id, handle) = fixture().await; + handle.abort(); + let (base2, _h, driver) = + serve_with_driver(Arc::clone(&engine), owner_id.seed_dir.clone()).await; + let _ = base; + let base = base2; + let client = reqwest::Client::new(); + seed_user_key(&engine, AARON_KEY_ID, 0xE0, 0xE1).await; + seed_user_key(&engine, ABEL_KEY_ID, 0xE2, 0xE3).await; + for k in [AARON_KEY_ID, ABEL_KEY_ID] { + let resp = client + .post(format!("{base}/v1/contacts")) + .bearer_auth(&owner) + .json(&serde_json::json!({ "key_id": k })) + .send() + .await + .expect("POST /v1/contacts"); + assert_eq!(resp.status(), 200); + } + let pair = pair_community_key_id(&owner_id.key_id, AARON_KEY_ID); + let abel_pair = pair_community_key_id(&owner_id.key_id, ABEL_KEY_ID); + + // THE ONLY REQUEST the joiner ever makes. + let resp = client + .post(format!("{base}/v1/chat")) + .bearer_auth(&owner) + .json(&serde_json::json!({ "key_id": AARON_KEY_ID })) + .send() + .await + .expect("POST /v1/chat"); + let v: serde_json::Value = resp.json().await.expect("json"); + assert_eq!(v["state"], "awaiting_invitation", "{v}"); + + // The creators' rooms land here, as replication carries them. + let dir = engine.federation_directory(); + let aaron = edge_signer_for(AARON_KEY_ID, 0xE0, 0xE1); + let abel = edge_signer_for(ABEL_KEY_ID, 0xE2, 0xE3); + let now = chrono::Utc::now(); + let opened = ciris_edge::chat::open_pair_room( + dir.as_ref(), + AARON_KEY_ID, + &owner_id.key_id, + now, + now + chrono::Duration::days(1), + &aaron, + ) + .await + .expect("Aaron opens"); + let proposal = opened.proposal.expect("an invitation").attestation_id; + ciris_edge::chat::open_pair_room( + dir.as_ref(), + ABEL_KEY_ID, + &owner_id.key_id, + now, + now + chrono::Duration::days(1), + &abel, + ) + .await + .expect("Abel opens"); + + // Pass 1: the driver accepts the requested invitation, and only it. + driver.drive_once().await; + let pending: Vec = + ciris_edge::membership::pending_proposals_for(dir.as_ref(), &owner_id.key_id) + .await + .expect("inbox") + .into_iter() + .map(|p| p.group_key_id) + .collect(); + assert!( + !pending.contains(&pair), + "accepted with no read: {pending:?}" + ); + assert!( + pending.contains(&abel_pair), + "never auto-accepted: {pending:?}" + ); + + // Aaron's node seats the owner on the acceptance. + let acceptance = dir + .list_attestations_for(&owner_id.key_id) + .await + .expect("replies") + .into_iter() + .find(|r| { + ciris_edge::membership::dimension_of(r) + == Some(ciris_edge::membership::ACCEPTANCE_DIMENSION) + && r.attestation_envelope + .get("references_attestation_id") + .and_then(serde_json::Value::as_str) + == Some(proposal.as_str()) + }) + .expect("the acceptance"); + ciris_edge::membership::widen_on_acceptance( + dir.as_ref(), + &acceptance, + &ciris_edge::membership::MembershipWidener::new(vec![Arc::new(aaron)]), + ) + .await + .expect("Aaron widens"); + + // Pass 2: seated, the joiner's KeyPackage goes out — still with no read. + assert!( + ciris_edge::chat::key_package_from(dir.as_ref(), &owner_id.key_id, &pair) + .await + .expect("read") + .is_none(), + "precondition: no KeyPackage before the driver runs" + ); + let driven = driver.drive_once().await; + assert!( + ciris_edge::chat::key_package_from(dir.as_ref(), &owner_id.key_id, &pair) + .await + .expect("read") + .is_some(), + "the joiner's KeyPackage is published by the driver, not by a read: {driven:?}" + ); + assert!( + intents_on_disk(&owner_id.seed_dir) + .iter() + .all(|i| i["pair_id"] != serde_json::json!(pair)), + "the seated room's request is removed" + ); +} diff --git a/tests/drive_crud.rs b/tests/drive_crud.rs index f374f4c4..5b2350b0 100644 --- a/tests/drive_crud.rs +++ b/tests/drive_crud.rs @@ -21,7 +21,12 @@ //! * rename keeps the blob (same sha, bytes still open) and retires the old id; //! * replace publishes new bytes and retires the old id; //! * move reseals at the target room and withdraws the source (or keeps it); -//! * notes: edit and delete through the same machinery. +//! * notes: edit and delete through the same machinery; +//! * the streamed drive (0.5.218) — a multipart upload with `size` streamed +//! into `files::publish_stream` above the old 64 MiB cap and read back by a +//! streamed `?raw=1` and uncapped `Range`, SHA-256-equal; a body that is not +//! its `size` is `drive.declared_length_mismatch` with no row; the write gate +//! and the field-before-file order hold on the streamed path. use std::sync::Arc; @@ -44,7 +49,6 @@ struct Fx { base: String, owner: String, owner_id: OwnerIdentity, - node_key: String, client: reqwest::Client, } @@ -62,7 +66,6 @@ async fn fixture() -> Fx { base, owner, owner_id, - node_key, client: reqwest::Client::new(), } } @@ -149,22 +152,28 @@ impl Fx { .expect("base64") } - /// A community the OWNER founded (and so a member of), plus `others`. - async fn owners_community(&self, name: &str, others: &[&str]) -> String { + /// A community the OWNER founded (and so a member of), plus `others` — + /// each `(key_id, ed_seed, pqc_seed)` as `seed_key` registered it, because + /// since persist v52 every founding member co-signs the record. + async fn owners_community(&self, name: &str, others: &[(&str, u8, u8)]) -> String { let id = format!("community-{name}-{}", std::process::id()); let now = chrono::Utc::now(); - self.engine - .put_community_self_signed(Community { + let keys: Vec<&str> = others.iter().map(|(k, _, _)| *k).collect(); + put_community_cosigned( + &self.engine, + Community { community_key_id: id.clone(), community_name: name.to_owned(), - members: founded_by(&self.owner_id.key_id, others, now), + members: founded_by(&self.owner_id.key_id, &keys, now), founded_at: now, consensus_protocol: "founder_only".to_string(), policy_blob: None, persist_row_hash: String::new(), - }) - .await - .expect("author the owner's community"); + }, + others, + ) + .await + .expect("author the owner's community"); id } @@ -174,8 +183,11 @@ impl Fx { seed_key(&self.engine, "dave-drive", 0xD0, 0xD1, identity_type::USER).await; let id = format!("community-strangers-{}", std::process::id()); let now = chrono::Utc::now(); - self.engine - .put_community_self_signed(Community { + // Both strangers co-sign their founding record (persist v52 Q1); the + // node's signature as authority vouches for nothing it is not in. + put_community_cosigned( + &self.engine, + Community { community_key_id: id.clone(), community_name: "strangers".to_owned(), members: ["carol-drive", "dave-drive"] @@ -190,9 +202,11 @@ impl Fx { consensus_protocol: "founder_only".to_string(), policy_blob: None, persist_row_hash: String::new(), - }) - .await - .expect("author the strangers' community"); + }, + &[("carol-drive", 0xC0, 0xC1), ("dave-drive", 0xD0, 0xD1)], + ) + .await + .expect("author the strangers' community"); id } } @@ -308,7 +322,10 @@ async fn uploads_list_meta_and_read_round_trip() { assert_eq!(e_small["withdrawn"], false); let env = &e_small["envelope"]; assert_eq!(env["attestation_id"], small_id.as_str()); - assert_eq!(env["attesting_key_id"], fx.node_key.as_str()); + // The PERSON authors a file (CIRISEdge#675, edge v33 `files::file_author`): + // the row is signed by the owner's key, which every one of their devices + // holds, not by the node that happened to write it. + assert_eq!(env["attesting_key_id"], fx.owner_id.key_id.as_str()); assert_eq!(env["cohort_scope"], "self"); assert_eq!(env["dimension"], "file:v1"); assert!(env["subject_key_ids"].is_array(), "{env}"); @@ -910,7 +927,7 @@ async fn a_non_author_cannot_change_a_file() { // through edge's own file door with her own key. let erin = "erin-drive"; seed_key(&fx.engine, erin, 0xE0, 0xE1, identity_type::USER).await; - let community = fx.owners_community("shared", &[erin]).await; + let community = fx.owners_community("shared", &[(erin, 0xE0, 0xE1)]).await; // The owner's content occurrence on this node — what the room's DEK is // wrapped to, so erin's file is readable by somebody. ciris_server::backend::provision_engine_occurrence(&fx.engine, &fx.owner_id.key_id) @@ -933,6 +950,7 @@ async fn a_non_author_cannot_change_a_file() { room: &room, bytes: b"erin's words", media_type: "text/plain", + codec: None, filename: Some("erin.txt"), asserted_at: chrono::Utc::now(), }, @@ -1025,9 +1043,10 @@ async fn an_upload_above_the_cap_is_too_large() { assert_eq!(s, 400, "the raised limit is the upload routes' only"); } -/// A file above the whole-read cap (only reachable from a peer — this node's -/// own upload cap is the same number) is `413 drive.too_large_for_whole_read` -/// on the JSON read, and still served by `Range`. +/// A file above the whole-read cap — sealed straight through edge's door, as +/// a peer's file arrives — is `413 drive.too_large_for_whole_read` on the JSON +/// read ONLY (it base64s one value); `?raw=1` streams it whole (0.5.218, edge's +/// `FileRow::chunks()`), and `Range` serves any slice. #[tokio::test] async fn a_file_above_the_whole_read_cap_is_read_by_range() { let fx = fixture().await; @@ -1054,6 +1073,7 @@ async fn a_file_above_the_whole_read_cap_is_read_by_range() { room: &owner_room, bytes: &big, media_type: "video/mp4", + codec: None, filename: Some("long.mp4"), asserted_at: chrono::Utc::now(), }, @@ -1067,12 +1087,31 @@ async fn a_file_above_the_whole_read_cap_is_read_by_range() { (413, "drive.too_large_for_whole_read"), "{v}" ); - let (s, v) = fx.get(&format!("/v1/files/{id}?raw=1")).await; + // RAW, no Range: streamed whole since 0.5.218 — the size up front, no + // `Repr-Digest` (unknowable before the last byte without holding it). + let mut resp = fx + .client + .get(format!("{}/v1/files/{id}?raw=1", fx.base)) + .bearer_auth(&fx.owner) + .send() + .await + .expect("raw GET"); + assert_eq!(resp.status(), 200); assert_eq!( - (s, reason(&v)), - (413, "drive.too_large_for_whole_read"), - "{v}" + resp.headers()["content-length"], + big.len().to_string().as_str() ); + assert!( + resp.headers().get("repr-digest").is_none(), + "a streamed body carries no Repr-Digest" + ); + let (mut n, mut all_sevens) = (0usize, true); + while let Some(piece) = resp.chunk().await.expect("body chunk") { + n += piece.len(); + all_sevens &= piece.iter().all(|b| *b == 7); + } + assert_eq!(n, big.len()); + assert!(all_sevens, "the streamed bytes are the file's"); let resp = fx .client .get(format!("{}/v1/files/{id}?raw=1", fx.base)) @@ -1521,3 +1560,508 @@ async fn every_supported_type_round_trips_byte_identical() { failures.join("\n ") ); } + +// ─── The streamed upload and the streamed read (0.5.218) ─────────────────── +// +// `POST /v1/files` as multipart with a `size` field streams the file part into +// edge's `files::publish_stream` (CIRISEdge#744); `?raw=1` above the whole-read +// cap streams it back through `FileRow::chunks()`, and `Range` reads any +// slice through `FileRow::open_range` windows (CIRISEdge#737). These tests +// never hold the file: the upload body is GENERATED as it is sent, and every +// read is hashed as it arrives. + +/// The generated file's byte at `i` — deterministic and non-repeating over +/// any window a chunk boundary could hide in, so a misplaced slice is caught. +fn gen_byte(i: u64) -> u8 { + (i.wrapping_mul(31).wrapping_add(i / 251) % 256) as u8 +} + +fn gen_fill(buf: &mut [u8], offset: u64) { + for (k, b) in buf.iter_mut().enumerate() { + *b = gen_byte(offset + k as u64); + } +} + +/// SHA-256 of the generated bytes `[offset, offset + len)`, a MiB at a time. +fn gen_sha(offset: u64, len: u64) -> String { + use sha2::{Digest as _, Sha256}; + let mut h = Sha256::new(); + let mut buf = vec![0u8; 1 << 20]; + let mut at = 0u64; + while at < len { + let n = (len - at).min(buf.len() as u64) as usize; + gen_fill(&mut buf[..n], offset + at); + h.update(&buf[..n]); + at += n as u64; + } + hex::encode(h.finalize()) +} + +/// A second `drive::router` over the fixture's engine, driven in-process — +/// reqwest here is built without its `stream` feature, so a request body +/// that is GENERATED as it is sent goes through `tower::ServiceExt::oneshot` +/// with an `axum::body::Body::from_stream`. Same engine, same owner pen, +/// same signer: the listener the reads go to sees every row this writes. +async fn upload_router(fx: &Fx) -> axum::Router { + ciris_server::drive::router( + Arc::clone(&fx.engine), + node_edge_signer(&fx.engine).await, + fx.owner_id.seed_dir.clone(), + None, + ) +} + +/// What the streamed form sends as its file part. +enum FilePart { + /// `len` generated bytes. + Generated(u64), + /// These bytes. + Literal(Vec), +} + +/// `POST /v1/files` as a STREAMED multipart body: `fields` first, then the +/// file part (`media`), then optionally a `trailing` field after it. The file +/// bytes are produced a MiB at a time as the router reads them. +async fn stream_upload( + app: &axum::Router, + owner: &str, + fields: &[(&str, String)], + media: &str, + file: FilePart, + trailing: Option<(&str, &str)>, +) -> (u16, serde_json::Value) { + use tower::ServiceExt as _; + let boundary = "streamed-7f3a-boundary"; + let mut head = Vec::new(); + for (k, v) in fields { + head.extend_from_slice( + format!("--{boundary}\r\nContent-Disposition: form-data; name=\"{k}\"\r\n\r\n{v}\r\n") + .as_bytes(), + ); + } + head.extend_from_slice( + format!( + "--{boundary}\r\nContent-Disposition: form-data; name=\"file\"; \ + filename=\"big.bin\"\r\nContent-Type: {media}\r\n\r\n" + ) + .as_bytes(), + ); + let mut tail = Vec::new(); + if let Some((k, v)) = trailing { + tail.extend_from_slice( + format!("\r\n--{boundary}\r\nContent-Disposition: form-data; name=\"{k}\"\r\n\r\n{v}") + .as_bytes(), + ); + } + tail.extend_from_slice(format!("\r\n--{boundary}--\r\n").as_bytes()); + + // (head, generated offset, generated total, literal, tail) — one frame + // per poll, a MiB of file at most. + struct Gen { + head: Option>, + at: u64, + total: u64, + literal: Option>, + tail: Option>, + } + let (total, literal) = match file { + FilePart::Generated(n) => (n, None), + FilePart::Literal(b) => (0, Some(b)), + }; + let gen = Gen { + head: Some(head), + at: 0, + total, + literal, + tail: Some(tail), + }; + let frames = futures_util::stream::unfold(gen, |mut g| async move { + let frame = if let Some(h) = g.head.take() { + h + } else if let Some(l) = g.literal.take() { + l + } else if g.at < g.total { + let n = (g.total - g.at).min(1 << 20) as usize; + let mut buf = vec![0u8; n]; + gen_fill(&mut buf, g.at); + g.at += n as u64; + buf + } else { + g.tail.take()? + }; + Some((Ok::<_, std::io::Error>(axum::body::Bytes::from(frame)), g)) + }); + let req = axum::http::Request::post("/v1/files") + .header("authorization", format!("Bearer {owner}")) + .header( + "content-type", + format!("multipart/form-data; boundary={boundary}"), + ) + .body(axum::body::Body::from_stream(frames)) + .expect("request"); + let resp = app.clone().oneshot(req).await.expect("oneshot"); + let status = resp.status().as_u16(); + let body = axum::body::to_bytes(resp.into_body(), 1 << 20) + .await + .expect("response body"); + let v = serde_json::from_slice(&body) + .unwrap_or_else(|_| serde_json::Value::String(String::from_utf8_lossy(&body).into())); + (status, v) +} + +/// A raw read, hashed as it arrives: (status, sha256 hex, bytes, headers). +async fn raw_digest( + fx: &Fx, + id: &str, + range: Option<&str>, +) -> (u16, String, u64, reqwest::header::HeaderMap) { + use sha2::{Digest as _, Sha256}; + let mut req = fx + .client + .get(format!("{}/v1/files/{id}?cohort=self&raw=1", fx.base)) + .bearer_auth(&fx.owner); + if let Some(r) = range { + req = req.header("range", r); + } + let mut resp = req.send().await.expect("raw GET"); + let (status, headers) = (resp.status().as_u16(), resp.headers().clone()); + let (mut h, mut n) = (Sha256::new(), 0u64); + while let Some(piece) = resp.chunk().await.expect("body chunk") { + h.update(&piece); + n += piece.len() as u64; + } + (status, hex::encode(h.finalize()), n, headers) +} + +/// This process's peak resident set, from `/proc/self/status` (`VmHWM`). +fn peak_rss_kib() -> Option { + let s = std::fs::read_to_string("/proc/self/status").ok()?; + s.lines() + .find(|l| l.starts_with("VmHWM:"))? + .split_whitespace() + .nth(1)? + .parse() + .ok() +} + +/// (a) + (c) — a file ABOVE the pre-0.5.218 64 MiB cap, uploaded as a +/// streamed multipart body (`size` before `file`), read back whole with +/// `?raw=1` (streamed from `FileRow::chunks()`) and by `Range` — across a +/// 256 KiB chunk boundary, and a range LONGER than the old 64 MiB range cap — +/// every read SHA-256-equal to what was generated. +#[tokio::test] +async fn a_streamed_upload_above_the_old_cap_reads_back_streamed() { + let fx = fixture().await; + let app = upload_router(&fx).await; + let len: u64 = 80 * 1024 * 1024 + 12_345; + assert!(len > ciris_server::drive::WHOLE_READ_CAP as u64); + let rss_before = peak_rss_kib(); + let (s, v) = stream_upload( + &app, + &fx.owner, + &[("cohort", "self".into()), ("size", len.to_string())], + "application/octet-stream", + FilePart::Generated(len), + None, + ) + .await; + assert_eq!(s, 200, "an 80 MiB streamed upload must be admitted: {v}"); + assert_eq!(v["crossed"], true, "{v}"); + let id = v["attestation_id"].as_str().expect("id").to_owned(); + + let (s, meta) = fx.get(&format!("/v1/files/{id}/meta?cohort=self")).await; + assert_eq!(s, 200, "{meta}"); + assert_eq!(meta["size"], len, "{meta}"); + assert_eq!(meta["chunked"], true, "{meta}"); + + // WHOLE, streamed. + let (s, sha, n, h) = raw_digest(&fx, &id, None).await; + assert_eq!(s, 200); + assert_eq!(n, len); + assert_eq!(h["content-length"], len.to_string().as_str()); + assert!( + h.get("repr-digest").is_none(), + "no Repr-Digest when streamed" + ); + assert_eq!( + sha, + gen_sha(0, len), + "the streamed read is the uploaded file" + ); + + // (c) RANGE across a 256 KiB chunk boundary (262144). + let (s, sha, n, h) = raw_digest(&fx, &id, Some("bytes=262100-262200")).await; + assert_eq!(s, 206); + assert_eq!(n, 101); + assert_eq!( + h["content-range"], + format!("bytes 262100-262200/{len}").as_str() + ); + assert_eq!(sha, gen_sha(262_100, 101)); + + // A range LONGER than the 64 MiB whole-read cap, unaligned at both ends: + // served whole, in windows — no shortening. + let (start, rlen) = (1_000_003u64, 65 * 1024 * 1024 + 7); + let (s, sha, n, h) = raw_digest( + &fx, + &id, + Some(&format!("bytes={start}-{}", start + rlen - 1)), + ) + .await; + assert_eq!(s, 206); + assert_eq!(n, rlen, "the whole range, not a capped prefix"); + assert_eq!(h["content-length"], rlen.to_string().as_str()); + assert_eq!(sha, gen_sha(start, rlen)); + + // The JSON read stays whole — and so refuses above the cap by name. + let (s, v) = fx.get(&format!("/v1/files/{id}?cohort=self")).await; + assert_eq!( + (s, reason(&v)), + (413, "drive.too_large_for_whole_read"), + "{v}" + ); + if let (Some(before), Some(after)) = (rss_before, peak_rss_kib()) { + eprintln!( + "streamed 80 MiB up and 3 reads down: peak RSS {before} KiB -> {after} KiB \ + (the in-memory sqlite holds the sealed file itself)" + ); + } +} + +/// (b) — a body SHORTER (or longer) than its declared `size` is edge's +/// `DeclaredLengthMismatch`: `400 drive.declared_length_mismatch`, and NO row +/// — at the chunk-DAG size and at the inline size. +#[tokio::test] +async fn a_body_that_disagrees_with_its_size_is_refused_and_writes_nothing() { + let fx = fixture().await; + let app = upload_router(&fx).await; + let count = |v: &serde_json::Value| v["entries"].as_array().map_or(0, Vec::len); + let (_, before) = fx.get("/v1/drive?cohort=self").await; + + for (declared, sent) in [ + (3 * 1024 * 1024u64, 3 * 1024 * 1024 - 10), // DAG, short + (3 * 1024 * 1024, 3 * 1024 * 1024 + 300_000), // DAG, long + (1000, 990), // inline, short + (1000, 1010), // inline, long + ] { + let (s, v) = stream_upload( + &app, + &fx.owner, + &[("cohort", "self".into()), ("size", declared.to_string())], + "application/octet-stream", + FilePart::Generated(sent), + None, + ) + .await; + assert_eq!( + (s, reason(&v)), + (400, "drive.declared_length_mismatch"), + "declared {declared}, sent {sent}: {v}" + ); + assert_eq!(v["declared"], declared, "{v}"); + } + let (_, after) = fx.get("/v1/drive?cohort=self").await; + assert_eq!(count(&after), count(&before), "no row: {after}"); +} + +/// (d) — the write gate runs on the streamed path, from the PEEKED head: a +/// JPEG declared as PNG is `415 drive.format_mismatch` before a byte is +/// sealed. And the form's order is enforced: a field after the file is +/// `400 drive.field_after_file`, again with no row. +#[tokio::test] +async fn the_streamed_path_keeps_the_write_gate_and_the_form_order() { + let fx = fixture().await; + let app = upload_router(&fx).await; + let count = |v: &serde_json::Value| v["entries"].as_array().map_or(0, Vec::len); + let (_, before) = fx.get("/v1/drive?cohort=self").await; + + // A JPEG's magic, then 2 MiB (a chunk-DAG size, so the lie would have + // been sealed chunk by chunk had the gate not run first). + let mut jpeg = vec![ + 0xFF, 0xD8, 0xFF, 0xE0, 0x00, 0x10, b'J', b'F', b'I', b'F', 0x00, + ]; + jpeg.resize(2 * 1024 * 1024, 0x42); + let (s, v) = stream_upload( + &app, + &fx.owner, + &[("cohort", "self".into()), ("size", jpeg.len().to_string())], + "image/png", + FilePart::Literal(jpeg.clone()), + None, + ) + .await; + assert_eq!((s, reason(&v)), (415, "drive.format_mismatch"), "{v}"); + assert_eq!(v["declared"], "image/png", "{v}"); + assert_eq!(v["sniffed"], "image/jpeg", "{v}"); + + // Declared honestly it lands — the gate is the only thing that refused. + let (s, v) = stream_upload( + &app, + &fx.owner, + &[ + ("cohort", "self".into()), + ("media_type", "image/jpeg".into()), + ("size", jpeg.len().to_string()), + ], + "application/octet-stream", + FilePart::Literal(jpeg.clone()), + None, + ) + .await; + assert_eq!(s, 200, "{v}"); + let (_, landed) = fx.get("/v1/drive?cohort=self").await; + assert_eq!(count(&landed), count(&before) + 1, "{landed}"); + + // A field AFTER the file: refused, and still exactly one new row. + let (s, v) = stream_upload( + &app, + &fx.owner, + &[("cohort", "self".into()), ("size", "1000".into())], + "application/octet-stream", + FilePart::Generated(1000), + Some(("filename", "late.bin")), + ) + .await; + assert_eq!((s, reason(&v)), (400, "drive.field_after_file"), "{v}"); + let (_, after) = fx.get("/v1/drive?cohort=self").await; + assert_eq!(count(&after), count(&landed), "no row: {after}"); + + // Above edge's single-file ceiling: refused by `size` alone, 413. + let (s, v) = stream_upload( + &app, + &fx.owner, + &[ + ("cohort", "self".into()), + ( + "size", + (ciris_server::drive::STREAMED_FILE_CEILING + 1).to_string(), + ), + ], + "application/octet-stream", + FilePart::Generated(10), + None, + ) + .await; + assert_eq!((s, reason(&v)), (413, "drive.too_large"), "{v}"); +} + +// ─── A family file's chunks are served under the FAMILY ──────────────────── + +/// edge v38.0.0 (CIRISEdge#736) found it on its own lane and the server had +/// the same bug: a family file is authored at `self` and crossed by a two-row +/// widening, so on its AUTHOR's node persist's binding index returns the +/// author's `self` row, and the server's old first-row-wins walk scoped every +/// family chunk `self` — each member's fetch, arriving on a `family` address, +/// was withheld `blob_serve_arrival_scope_insufficient`. `chunk_scope` now +/// asks edge's `BlobMeaning::serve_scope`, which answers from the widening. +/// +/// Pinned on the real drive: the owner founds a household (alone — persist v52 +/// seats only the founding record's signers), uploads a file into it, and the +/// node's own chunk source scopes the blob as that family, not as `self`. +#[tokio::test] +async fn a_family_files_chunks_are_served_under_the_family() { + use ciris_edge::blob_swarm::{BlobChunkSource, ContentScope}; + use ciris_edge::cohort_scope::CohortScope; + use ciris_persist::federation::types::{Family, FamilyMember, SignedFamily}; + use ciris_persist::prelude::ceg_produce_canonicalize; + + let fx = fixture().await; + let at = + ciris_persist::federation::admission::truncate_to_substrate_resolution(chrono::Utc::now()); + let family_id = format!("family:v1:{}", uuid::Uuid::new_v4().simple()); + let family = Family { + family_key_id: family_id.clone(), + family_name: "household".to_owned(), + members: vec![FamilyMember { + key_id: fx.owner_id.key_id.clone(), + joined_at: at, + role: Some("founder".to_owned()), + }], + founded_at: at, + consensus_protocol: "founder_only".to_owned(), + consensus_protocol_entrenched: false, + dissolved_at: None, + persist_row_hash: String::new(), + }; + let canonical = + ceg_produce_canonicalize(&family.signing_envelope()).expect("canonicalize family"); + let sig = fx + .owner_id + .signer() + .await + .sign_hybrid(&canonical) + .await + .expect("the owner signs the founding record"); + fx.engine + .federation_directory() + .put_family(SignedFamily { + family, + authority_key_id: fx.owner_id.key_id.clone(), + scrub_signature_classical: BASE64.encode(&sig.classical.signature), + scrub_signature_pqc: Some(BASE64.encode(&sig.pqc.signature)), + supersede_proof: None, + cosignatures: Vec::new(), + }) + .await + .expect("the founder's household"); + + let (s, v) = fx + .post( + "/v1/files", + serde_json::json!({ + "cohort": "family", + "room_id": family_id, + "bytes_base64": BASE64.encode(b"the family's shopping list"), + "media_type": "text/plain", + "filename": "list.txt", + }), + ) + .await; + assert_eq!(s, 200, "family upload: {v}"); + let id = v["attestation_id"].as_str().expect("id").to_owned(); + let (s, meta) = fx + .get(&format!( + "/v1/files/{id}/meta?cohort=family&room_id={family_id}" + )) + .await; + assert_eq!(s, 200, "{meta}"); + let sha: [u8; 32] = hex::decode(meta["content_sha256"].as_str().expect("sha")) + .expect("hex") + .try_into() + .expect("32 bytes"); + + // The premise, measured: the binding index's first projecting row — what + // the deleted walk answered from — is the author's `self` row here. + let rows = fx + .engine + .federation_directory() + .attestations_binding_content(&hex::encode(sha)) + .await + .expect("binding index"); + let old_answer = rows + .iter() + .find_map(|r| ciris_edge::blob_swarm::BlobMeaning::project(r, &sha).ok()) + .map(|m| m.scope().clone()); + assert!( + matches!( + old_answer, + Some(ContentScope::Group { + scope: CohortScope::SelfOnly, + .. + }) + ), + "the first-row walk would have scoped this family file `self` (the bug): {old_answer:?}" + ); + + let source = ciris_server::backend::ServerBlobChunkSource::new(&fx.engine); + match source.chunk_scope(sha).await { + Some(ContentScope::Group { + scope: CohortScope::Family, + group_id, + }) => assert_eq!(group_id, family_id, "served under THIS family"), + other => panic!( + "a family file's chunks must be served under the family (edge's \ + BlobMeaning::serve_scope prefers the widening), got {other:?}" + ), + } +} diff --git a/tests/drive_split_viewer_key.rs b/tests/drive_split_viewer_key.rs index 226ffbbe..a06154d6 100644 --- a/tests/drive_split_viewer_key.rs +++ b/tests/drive_split_viewer_key.rs @@ -156,7 +156,9 @@ async fn a_split_node_opens_the_files_it_wrote() { assert!( matches!( as_actor, - Err(ciris_edge::chat::UnopenedReason::NotGranted { .. }) + Err(ciris_edge::files::FileError::Unopened( + ciris_edge::chat::UnopenedReason::NotGranted { .. } + )) ), "the actor key holds no grant on a split node — which is exactly why the drive \ must not open as it: {as_actor:?}" diff --git a/tests/emitted_dimensions_are_registered.rs b/tests/emitted_dimensions_are_registered.rs new file mode 100644 index 00000000..697ef6df --- /dev/null +++ b/tests/emitted_dimensions_are_registered.rs @@ -0,0 +1,134 @@ +//! **Every dimension this server spells out is one the registry registers.** +//! +//! Since persist v50 the dimension grammar is data (the vendored CC namespace +//! registry, CIRISPersist#924) and the put door refuses a dimension outside a +//! registered family: `age_self_declared:adult:v1` — the family is +//! `age_self_declared:band:{band}:{version}` — was refused at the door, and only +//! a test that happened to emit that one row noticed. A dimension no test emits +//! would fail the same way in production. +//! +//! So: every string literal in `src/` shaped like a full versioned dimension +//! (`a:b…:vN`) goes through persist's ONE matcher, and none may come back with +//! a REFUSAL. "No registered family" alone is not a refusal: an unreserved stem +//! (`self:device_label:v1`) is admitted under its producer as steward. What the +//! door refuses is a malformed dimension, or one on a stem persist RESERVES but +//! the registry does not register (`age_self_declared:`). Comments are skipped. +//! A dimension assembled with `format!` is not seen here; its builder's own +//! unit test is the check. + +use std::path::Path; + +fn literals(src: &str) -> Vec { + let mut out = Vec::new(); + let bytes = src.as_bytes(); + let mut i = 0; + while i < bytes.len() { + if bytes[i] == b'"' { + let start = i + 1; + let mut j = start; + while j < bytes.len() && bytes[j] != b'"' && bytes[j] != b'\n' { + if bytes[j] == b'\\' { + j += 1; + } + j += 1; + } + if j < bytes.len() && bytes[j] == b'"' { + out.push(src[start..j].to_owned()); + } + i = j + 1; + } else { + i += 1; + } + } + out +} + +/// `a:…:vN`, lowercase segments, no placeholders or spaces. +fn is_versioned_dimension(s: &str) -> bool { + let parts: Vec<&str> = s.split(':').collect(); + parts.len() >= 3 + && parts.iter().all(|p| { + !p.is_empty() + && p.chars() + .all(|c| c.is_ascii_lowercase() || c.is_ascii_digit() || c == '_') + }) + && parts + .last() + .and_then(|v| v.strip_prefix('v')) + .is_some_and(|n| !n.is_empty() && n.chars().all(|c| c.is_ascii_digit())) +} + +/// Spellings kept on purpose, each with why. Reading is not writing: these +/// are the OLD forms a node must still READ from rows written before the +/// registry was strict, and appear in src only as parser inputs. +const READ_ONLY_LEGACY: &[(&str, &str)] = &[ + ( + "age_self_declared:adult:v1", + "pre-v50 self-declared age rows; age.rs parses them, never writes them", + ), + ( + "need:shelter:v1", + "field_conformance's polarity table: a classifier input, never a row", + ), +]; + +fn walk(dir: &Path, found: &mut Vec<(String, String)>) { + for e in std::fs::read_dir(dir).expect("read src") { + let p = e.expect("entry").path(); + if p.is_dir() { + walk(&p, found); + } else if p.extension().is_some_and(|x| x == "rs") { + let text = std::fs::read_to_string(&p).expect("read file"); + let code: String = text + .lines() + .filter(|l| !l.trim_start().starts_with("//")) + .collect::>() + .join("\n"); + for lit in literals(&code) { + if is_versioned_dimension(&lit) { + found.push((p.display().to_string(), lit)); + } + } + } + } +} + +#[test] +fn every_dimension_literal_in_src_is_registered() { + use ciris_persist::federation::namespace::matcher::match_family; + let mut found = Vec::new(); + walk( + Path::new(concat!(env!("CARGO_MANIFEST_DIR"), "/src")), + &mut found, + ); + assert!( + // 18 on 2026-09-28 (14 distinct), counted independently with grep. + found.len() >= 10, + "the scan found only {} dimension literals — the literal reader is broken, and a \ + gate that reads nothing passes everything", + found.len() + ); + // Positive control: the spelling that WAS refused still is. Without this a + // matcher that stopped refusing would turn the gate green by itself. + assert!( + match_family("age_self_declared:adult:v1").refusal.is_some(), + "the matcher no longer refuses the unregistered-reserved-stem spelling this gate \ + was written for — re-read what `refusal` means before trusting a green run" + ); + let refused: Vec = found + .iter() + .filter(|(_, dim)| !READ_ONLY_LEGACY.iter().any(|(l, _)| l == dim)) + .filter_map(|(file, dim)| { + match_family(dim) + .refusal + .map(|r| format!("{file}: {dim} — {r:?}")) + }) + .collect(); + assert!( + refused.is_empty(), + "{} dimension literal(s) in src/ are not registered families — persist's put door \ + refuses each one:\n {}", + refused.len(), + refused.join("\n ") + ); +} diff --git a/tests/every_act_passes_the_session_gate.rs b/tests/every_act_passes_the_session_gate.rs new file mode 100644 index 00000000..f032916b --- /dev/null +++ b/tests/every_act_passes_the_session_gate.rs @@ -0,0 +1,148 @@ +//! Gate: **every ACT in the session-claims inventory calls the handler check +//! before it acts** (CC 3.1.3.1; `FSD/SESSION_CLAIMS.md` §2.1). +//! +//! The inventory is the FSD's table, not a second list here: a copy of "which +//! sites are gated" in this file is the one thing guaranteed to drift from the +//! document a reviewer reads (the mirrored-rule class). So this reads every +//! `**gated**` row of §2.1 — `file`, `fn`, `act call` — and for each checks, +//! over the function's body with comments stripped, that +//! `session_claims::gate(` appears and appears BEFORE the act call. An act +//! reached without the gate runs on every device the person owns. +//! +//! It also refuses a gated act called from anywhere ELSE in `src/` (a second +//! caller of `rekey_self_occurrence_add` in a loop would be a fresh ungated +//! site), except the named, user-initiated door. + +use std::path::Path; + +/// `(file, fn, act)` for every `**gated**` row of FSD §2.1. +fn gated_rows() -> Vec<(String, String, String)> { + let fsd = std::fs::read_to_string( + Path::new(env!("CARGO_MANIFEST_DIR")).join("FSD/SESSION_CLAIMS.md"), + ) + .expect("FSD/SESSION_CLAIMS.md is readable"); + let strip = |c: &str| c.trim().trim_matches('`').to_owned(); + let mut out = Vec::new(); + for line in fsd.lines() { + if !line.starts_with('|') || !line.contains("**gated**") { + continue; + } + let cells: Vec<&str> = line.trim_matches('|').split('|').collect(); + // | act | file | fn | act call | gated | community_id | session_id | + assert!( + cells.len() >= 7, + "a gated row must have seven cells: {line}" + ); + out.push((strip(cells[1]), strip(cells[2]), strip(cells[3]))); + } + out +} + +/// The body of `fn name` in `src`, comments stripped: from its signature to +/// the first line that is exactly `}` at column 0. +fn fn_body(src: &str, name: &str) -> Option { + let code: String = src + .replace("\r\n", "\n") + .lines() + .map(|l| match l.find("//") { + Some(i) => &l[..i], + None => l, + }) + .collect::>() + .join("\n"); + let start = [ + format!("pub async fn {name}("), + format!("async fn {name}("), + format!("pub fn {name}("), + format!("fn {name}("), + ] + .iter() + .find_map(|sig| code.find(sig.as_str()))?; + let rest = &code[start..]; + let end = rest.find("\n}\n").map_or(rest.len(), |i| i + 2); + Some(rest[..end].to_owned()) +} + +#[test] +fn the_inventory_names_at_least_the_three_gated_acts() { + let rows = gated_rows(); + assert!( + rows.len() >= 3, + "FSD/SESSION_CLAIMS.md §2.1 lost its gated rows (found {rows:?}) — the gate test \ + would then check nothing" + ); +} + +#[test] +fn every_gated_act_calls_the_session_gate_first() { + let mut failures = Vec::new(); + for (file, func, act) in gated_rows() { + let src = std::fs::read_to_string(Path::new(env!("CARGO_MANIFEST_DIR")).join(&file)) + .unwrap_or_else(|e| panic!("{file}: {e}")); + let Some(body) = fn_body(&src, &func) else { + failures.push(format!( + "{file}: no fn `{func}` — the FSD names a site that is gone" + )); + continue; + }; + let Some(act_at) = body.find(&act) else { + failures.push(format!( + "{file}::{func}: the act `{act}` is not called here any more — update the FSD row" + )); + continue; + }; + match body.find("session_claims::gate(") { + Some(g) if g < act_at => {} + Some(_) => failures.push(format!( + "{file}::{func}: `session_claims::gate(` comes AFTER `{act}` — the act runs \ + before anyone asked whether this device handles it" + )), + None => failures.push(format!( + "{file}::{func}: `{act}` is reached with no `session_claims::gate(` — it runs \ + on every device the person owns (CC 3.1.3.1)" + )), + } + } + assert!(failures.is_empty(), "{}", failures.join("\n")); +} + +/// The re-wrap door has exactly two callers: the gated loop pass, and +/// `auth::occurrence::bind_occurrence_core` — which wraps self DEKs to an +/// occurrence bound ON THIS DEVICE (the person's own `POST /v1/self/occurrence`, +/// the portable-occurrence doors, and this node's own actor occurrence at +/// boot): the device doing its own binding, never a reaction to a sibling's +/// row (FSD §2.2 / §2.3). +#[test] +fn the_rewrap_door_has_no_ungated_loop_caller() { + let mut callers = Vec::new(); + let mut stack = vec![Path::new(env!("CARGO_MANIFEST_DIR")).join("src")]; + while let Some(dir) = stack.pop() { + for e in std::fs::read_dir(&dir).expect("read src") { + let p = e.expect("entry").path(); + if p.is_dir() { + stack.push(p); + } else if p.extension().is_some_and(|x| x == "rs") { + let body = std::fs::read_to_string(&p).expect("read"); + for (n, line) in body.lines().enumerate() { + let code = line.split("//").next().unwrap_or(""); + if code.contains("rekey_self_occurrence_add(") { + callers.push(format!( + "{}:{}", + p.file_name().unwrap().to_string_lossy(), + n + 1 + )); + } + } + } + } + } + let unexpected: Vec<&String> = callers + .iter() + .filter(|c| !c.starts_with("self_rewrap.rs:") && !c.starts_with("occurrence.rs:")) + .collect(); + assert!( + unexpected.is_empty(), + "a new caller of the self re-wrap door outside the gated pass and the person's own \ + request: {unexpected:?} — gate it (FSD/SESSION_CLAIMS.md §2.1) or name it in §2.3" + ); +} diff --git a/tests/family_crud.rs b/tests/family_crud.rs index 1d81542b..48495d55 100644 --- a/tests/family_crud.rs +++ b/tests/family_crud.rs @@ -134,7 +134,19 @@ async fn a_founder_only_household_through_its_whole_life() { Some(json!({ "name": "x", "members": ["nobody-registered-this"] })), ) .await; - assert_refused(&r, 400, "family.unknown_member_key"); + // Since 0.5.218 a founding roster naming anyone but the founder is refused + // before the key is even looked up: nobody joins without their own + // consent (CIRISConstitution#133, CIRISPersist#955) — at persist v52 by + // persist's own rule name (Q1: signing the founding record is consent). + assert_refused(&r, 409, "membership.founding_member_unsigned"); + let r = alice + .as_owner( + "POST", + "/v1/families", + Some(json!({ "name": "x", "members": [bob.key()] })), + ) + .await; + assert_refused(&r, 409, "membership.founding_member_unsigned"); let r = alice .as_owner( "POST", @@ -165,7 +177,12 @@ async fn a_founder_only_household_through_its_whole_life() { .await; assert_refused(&r, 404, "family.not_found"); - // ── add ───────────────────────────────────────────────────────────────── + // ── add: an INVITATION since persist v52 (CIRISPersist#955) ───────────── + // + // A founder's say-so is not the joiner's (CIRISConstitution#133): the + // route proposes, answers 202 `invited`, and writes no roster row. The + // invitee must be a registered identity (their key is what the widening + // will name). let r = alice .as_owner( "POST", @@ -189,15 +206,21 @@ async fn a_founder_only_household_through_its_whole_life() { Some(json!({ "key_id": bob.key() })), ) .await; - assert_eq!(st.as_u16(), 200, "{v}"); + assert_eq!(st.as_u16(), 202, "{v}"); + assert_eq!(v["state"], "invited", "{v}"); + let (_, v) = alice + .as_owner("GET", &format!("/v1/families/{id}"), None) + .await; assert_eq!( members(&v), - sorted(vec![pair(&alice, "founder"), pair(&bob, "member")]) - ); - assert!( - v["dek_rewrap"].is_object() && v["dek_rewrap"].get("error").is_none(), - "the DEK cascade ran for the newcomer: {v}" + vec![pair(&alice, "founder")], + "an invitation is not a membership: {v}" ); + // What follows is about a household that HAS a second member: Bob's + // consent and the widening, through the test-only fixture (the real + // cross-node flow is `an_invited_person_joins_only_by_their_own_acceptance`). + alice.test_only_widen_family(&id, &bob, "member").await; + // Re-adding someone already in is not joining: it keeps its own answer. let r = alice .as_owner( "POST", @@ -218,6 +241,8 @@ async fn a_founder_only_household_through_its_whole_life() { assert_eq!(v["families"].as_array().map(Vec::len), Some(1), "{v}"); // ── the policy matrix: a MEMBER may not govern a founder_only family ──── + // (nor invite into it: under founder_only only a founder proposes — + // persist's rule, named by the household's own id) let r = bob .as_owner( "POST", @@ -353,15 +378,8 @@ async fn a_founder_only_household_through_its_whole_life() { sorted(vec![pair(&alice, "founder"), pair(&bob, "guardian")]) ); - // ── remove, and a removal is permanent at this pin ────────────────────── - let (st, v) = alice - .as_owner( - "POST", - &format!("/v1/families/{id}/members"), - Some(json!({ "key_id": carol.key() })), - ) - .await; - assert_eq!(st.as_u16(), 200, "{v}"); + // ── remove ────────────────────────────────────────────────────────────── + alice.test_only_widen_family(&id, &carol, "member").await; let (st, v) = alice .as_owner( "DELETE", @@ -384,8 +402,10 @@ async fn a_founder_only_household_through_its_whole_life() { .await; assert_refused(&r, 404, "family.not_a_member"); // A removed member CAN be re-added (persist v49.0.0, #910.1): the new - // widening folds after the revocation. Then removed again, so the rest of - // the life runs as before. + // widening folds after the revocation. Through the route it is a JOIN like + // any other, so it is an invitation (persist v52); her acceptance and the + // widening come through the test-only fixture. Then removed again, so the + // rest of the life runs as before. let (st, v) = alice .as_owner( "POST", @@ -393,8 +413,19 @@ async fn a_founder_only_household_through_its_whole_life() { Some(json!({ "key_id": carol.key() })), ) .await; - assert_eq!(st.as_u16(), 200, "a removed member is re-added: {v}"); - assert!(members(&v).iter().any(|(k, _)| k == carol.key()), "{v}"); + assert_eq!( + (st.as_u16(), v["state"].as_str()), + (202, Some("invited")), + "{v}" + ); + alice.test_only_widen_family(&id, &carol, "member").await; + let (_, v) = alice + .as_owner("GET", &format!("/v1/families/{id}"), None) + .await; + assert!( + members(&v).iter().any(|(k, _)| k == carol.key()), + "a removed member is re-added at the substrate: {v}" + ); let (st, v) = alice .as_owner( "DELETE", @@ -482,14 +513,9 @@ async fn a_founder_may_leave_once_another_founder_remains() { let bob = Person::new("fb").await; acquainted(&[&alice, &bob]).await; let id = create(&alice, json!({ "name": "two founders" })).await; - let (st, v) = alice - .as_owner( - "POST", - &format!("/v1/families/{id}/members"), - Some(json!({ "key_id": bob.key(), "role": "founder" })), - ) - .await; - assert_eq!(st.as_u16(), 200, "{v}"); + // A second founder, by the test-only fixture (the add route is closed + // until the joiner can consent — CIRISPersist#955). + alice.test_only_widen_family(&id, &bob, "founder").await; let (st, v) = alice .as_owner("POST", &format!("/v1/families/{id}/leave"), None) .await; @@ -529,17 +555,15 @@ async fn the_family_list_pages() { // ─── quorum:2/3 — envelope → cosign → assemble ────────────────────────────── /// A `quorum:2/3` household of Alice (founder), Bob and Carol, held by all -/// three nodes. +/// three nodes. Founded by the TEST-ONLY fixture, which has Bob and Carol +/// CO-SIGN the founding record (persist v52 Q1 seats only its signers; the +/// server has no founding-cosign flow, so `POST /v1/families` naming others is +/// `membership.founding_member_unsigned`), and what these tests exercise is +/// what a quorum family does once it exists. async fn quorum_family(alice: &Person, bob: &Person, carol: &Person) -> String { - let id = create( - alice, - json!({ - "name": "the trio", - "members": [bob.key(), carol.key()], - "consensus_protocol": "quorum:2/3", - }), - ) - .await; + let id = alice + .test_only_family_founded_with("the trio", &[bob, carol], "quorum:2/3") + .await; bob.receive_families_from(alice).await; carol.receive_families_from(alice).await; id @@ -589,7 +613,7 @@ async fn assemble( } #[tokio::test] -async fn a_quorum_family_adds_through_envelope_cosign_assemble() { +async fn a_quorum_family_cannot_admit_anyone_without_their_acceptance() { let alice = Person::new("qa").await; let bob = Person::new("qb").await; let carol = Person::new("qc").await; @@ -597,37 +621,59 @@ async fn a_quorum_family_adds_through_envelope_cosign_assemble() { acquainted(&[&alice, &bob, &carol, &dave]).await; let id = quorum_family(&alice, &bob, &carol).await; - // One call is not enough for a quorum family — it says where to go. - let r = alice + // ── CONSENT TO JOIN (persist v52): M of N cannot stand in for Dave ────── + // + // Until 0.5.218 this test walked an add through envelope → cosign → + // assemble and Dave was in. Now the walk still runs — an invitation by any + // member, an add envelope, the quorum's co-signatures — and persist + // refuses the co-signed WIDENING at assemble by rule + // (`membership.awaiting_acceptance`): Dave never accepted. + let (st, v) = bob .as_owner( "POST", &format!("/v1/families/{id}/members"), Some(json!({ "key_id": dave.key() })), ) .await; - assert_refused(&r, 409, "family.quorum_pending"); - let r = alice - .as_owner("DELETE", &format!("/v1/families/{id}"), None) - .await; - assert_refused(&r, 409, "family.quorum_pending"); - - // Envelope refusals. + assert_eq!( + (st.as_u16(), v["state"].as_str()), + (202, Some("invited")), + "any member invites into a quorum household: {v}" + ); + let env_add = envelope( + &alice, + &id, + json!({ "action": "add", "key_id": dave.key() }), + ) + .await; + let a_add = cosign(&alice, &id, &env_add, &json!([])).await; + let ab_add = cosign(&bob, &id, &env_add, &a_add["signatures"]).await; + assert_eq!(ab_add["quorum_met"], true, "{ab_add}"); + let r = assemble(&alice, &id, &env_add, &ab_add["signatures"]).await; + assert_refused(&r, 409, "membership.awaiting_acceptance"); let r = alice .as_owner( "POST", &format!("/v1/families/{id}/changes/envelope"), - Some(json!({ "action": "add", "key_id": bob.key() })), + Some(json!({ "action": "add", "key_id": dave.key(), "consensus_protocol": "quorum:2/4" })), ) .await; - assert_refused(&r, 409, "family.already_member"); + assert_refused(&r, 400, "family.bad_consensus_protocol"); + // Someone already in is not joining, and keeps its own answer. let r = alice .as_owner( "POST", &format!("/v1/families/{id}/changes/envelope"), - Some(json!({ "action": "add", "key_id": dave.key(), "consensus_protocol": "quorum:2/4" })), + Some(json!({ "action": "add", "key_id": bob.key() })), ) .await; - assert_refused(&r, 400, "family.bad_consensus_protocol"); + assert_refused(&r, 409, "family.already_member"); + // One call is not enough for a quorum family's other changes — it says + // where to go. + let r = alice + .as_owner("DELETE", &format!("/v1/families/{id}"), None) + .await; + assert_refused(&r, 409, "family.quorum_pending"); let r = alice .as_owner( "POST", @@ -637,19 +683,60 @@ async fn a_quorum_family_adds_through_envelope_cosign_assemble() { .await; assert_refused(&r, 400, "family.bad_request"); - // ── the envelope, built by the founder ────────────────────────────────── + // ── a GROWING envelope under another action (an older node, or by hand) ─ + // + // The assemble step reads what the envelope's roster DOES, not the action + // it is labelled with: a proposed roster naming Dave under anything but + // `add` would reach the supersede door, which never adds (persist Q2) — + // refused by that rule's name, however many members signed. + let dir = alice.engine.federation_directory(); + let record = dir + .lookup_family(&id) + .await + .expect("lookup") + .expect("family"); + let grown: Vec = [alice.key(), bob.key(), carol.key(), dave.key()] + .iter() + .map(|k| k.to_string()) + .collect(); + let mut add_env = dir + .build_membership_change_envelope( + ciris_persist::federation::cohort::Cohort::Family, + &id, + &grown, + false, + Some("quorum:3/4"), + ) + .await + .expect("an add envelope, built the pre-0.5.218 way"); + add_env["action"] = json!("role"); + add_env["target_key_id"] = json!(dave.key()); + add_env["roles"] = json!({ + alice.key(): "founder", bob.key(): "member", carol.key(): "member", dave.key(): "member", + }); + add_env["prior_persist_row_hash"] = json!(record.persist_row_hash); + add_env["row_at"] = + json!(chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Millis, true)); + let r = assemble(&alice, &id, &add_env, &json!([])).await; + assert_refused(&r, 409, "membership.supersede_cannot_add"); + let (_, v) = alice + .as_owner("GET", &format!("/v1/families/{id}"), None) + .await; + assert!( + !members(&v).iter().any(|(k, _)| k == dave.key()), + "no door admitted Dave: {v}" + ); + + // ── the quorum flow itself, on a change that admits no one ────────────── let env = envelope( &alice, &id, - json!({ "action": "add", "key_id": dave.key() }), + json!({ "action": "role", "key_id": bob.key(), "role": "guardian" }), ) .await; - assert_eq!( - env["consensus_protocol"], "quorum:3/4", - "N follows the roster" - ); + assert_eq!(env["consensus_protocol"], "quorum:2/3", "same roster"); - // ── Alice cosigns on HER node: 1 of 2, not yet met ────────────────────── + // Alice cosigns on HER node: 1 of 2, not yet met. let v = cosign(&alice, &id, &env, &json!([])).await; assert_eq!(v["required_signatures"], 2); assert_eq!(v["quorum_met"], false); @@ -657,7 +744,7 @@ async fn a_quorum_family_adds_through_envelope_cosign_assemble() { let r = assemble(&alice, &id, &env, &one).await; assert_refused(&r, 409, "family.quorum_pending"); - // ── an OUTSIDER cannot cosign — they cannot see the family at all ─────── + // An OUTSIDER cannot cosign — they cannot see the family at all. dave.receive_families_from(&alice).await; let r = dave .as_owner( @@ -668,7 +755,7 @@ async fn a_quorum_family_adds_through_envelope_cosign_assemble() { .await; assert_refused(&r, 404, "family.not_found"); - // ── … and an outsider's signature does not count toward the quorum ────── + // … and an outsider's signature does not count toward the quorum. let dave_sig = { let bytes = ciris_verify_core::jcs::canonicalize(&env).expect("jcs"); let s = dave.owner.signer().await; @@ -696,33 +783,29 @@ async fn a_quorum_family_adds_through_envelope_cosign_assemble() { "{v}" ); - // ── Bob cosigns on HIS node, adding to Alice's: 2 of 2, met ───────────── + // Bob cosigns on HIS node, adding to Alice's: 2 of 2, met. let v = cosign(&bob, &id, &env, &one).await; assert_eq!(v["quorum_met"], true, "{v}"); let two = v["signatures"].clone(); - // ── any member assembles ──────────────────────────────────────────────── + // Any member assembles. let (st, v) = assemble(&alice, &id, &env, &two).await; assert_eq!(st.as_u16(), 200, "{v}"); - assert_eq!(v["action"], "add"); - assert_eq!(v["consensus_protocol"], "quorum:3/4"); + assert_eq!(v["action"], "role"); assert_eq!( members(&v), sorted(vec![ pair(&alice, "founder"), - pair(&bob, "member"), + pair(&bob, "guardian"), pair(&carol, "member"), - pair(&dave, "member"), ]) ); - assert!(v["dek_rewrap"].is_object(), "{v}"); - // ── the same envelope cannot be applied twice (the record moved) ──────── + // The same envelope cannot be applied twice (the record moved) … let r = assemble(&alice, &id, &env, &two).await; assert_refused(&r, 409, "family.bad_change"); // … and a cosign against a node still holding the OLD record is refused - // by name rather than signed: Bob's node never received the supersede - // (FSD §3.5 — a grown family record does not replicate at this pin). + // by name rather than signed. let r = bob .as_owner( "POST", @@ -830,6 +913,7 @@ async fn a_quorum_family_dissolves_only_with_its_quorum() { .as_owner("GET", &format!("/v1/families/{id}"), None) .await; assert_refused(&r, 404, "family.not_found"); + // The removals replicate: Bob's node learns the family is gone for him too. bob.receive_families_from(&alice).await; let r = bob @@ -838,16 +922,112 @@ async fn a_quorum_family_dissolves_only_with_its_quorum() { assert_refused(&r, 404, "family.not_found"); } +/// **A quorum dissolve replicates as an AMENDMENT** (CIRISServer#700; +/// persist v52.0.0, CIRISPersist#956 — un-ignored at the v38 adopt). +/// +/// Before v52 the terminal record went through the PLAIN `supersede_family`, +/// which persist v49 strips of any proof (#910.5), so only the removal rows +/// travelled and every peer kept the household live. persist v52 ships the +/// terminal amendment this test asked for: the record with every seat +/// byte-identical and `dissolved_at` set to the instant the quorum signed +/// inside the change envelope, through `supersede_family_with_quorum`. Bob's +/// node, holding the old record and the full roster, re-verifies the quorum +/// against ITS OWN roster and applies it — the record alone, no removal row — +/// after which the household has no active members there. +#[tokio::test] +async fn a_quorum_dissolve_replicates_as_an_amendment() { + let alice = Person::new("dra").await; + let bob = Person::new("drb").await; + let carol = Person::new("drc").await; + acquainted(&[&alice, &bob, &carol]).await; + let id = quorum_family(&alice, &bob, &carol).await; + let env = envelope(&alice, &id, json!({ "action": "dissolve" })).await; + assert!( + env["dissolved_at"].is_string(), + "the dissolve envelope pins the terminal instant the quorum signs: {env}" + ); + let a = cosign(&alice, &id, &env, &json!([])).await; + let ac = cosign(&carol, &id, &env, &a["signatures"]).await; + let (st, v) = assemble(&alice, &id, &env, &ac["signatures"]).await; + assert_eq!(st.as_u16(), 200, "{v}"); + + // The served record is the terminal amendment, carrying its quorum proof. + let served = alice + .engine + .federation_directory() + .list_signed_families_since(None, u32::MAX) + .await + .expect("signed families") + .into_iter() + .find(|s| s.family.family.family_key_id == id) + .expect("the dissolved household is still served"); + assert!( + served.family.family.dissolved_at.is_some(), + "the served record is the terminal one" + ); + assert_eq!( + served.family.family.members.len(), + 3, + "a dissolve changes nothing but dissolved_at (persist #956)" + ); + let proof = served + .family + .supersede_proof + .as_ref() + .expect("the terminal record carries its quorum proof"); + assert!( + proof.change_envelope["dissolved_at"].is_string(), + "the proof authorizes the terminal instant: {}", + proof.change_envelope + ); + // … and Bob's node applies it — the record alone. + let offered = bob.receive_family_records_from(&alice).await; + let (_, applied) = offered + .iter() + .find(|(k, _)| k == &id) + .expect("the household was offered"); + assert!( + applied.is_ok(), + "Bob's node applies the amendment: {applied:?}" + ); + let held = bob + .engine + .federation_directory() + .lookup_family(&id) + .await + .expect("lookup") + .expect("held"); + assert!( + held.dissolved_at.is_some(), + "Bob's node holds the terminal record" + ); + let active = bob + .engine + .federation_directory() + .active_family_members(&id) + .await + .expect("active members"); + assert!( + active.is_empty(), + "a dissolved household has no active members on Bob's node: {active:?}" + ); +} + /// `majority` / `unanimous` are accepted as aliases and stored in the one form /// verify's membership-change gate counts. +/// +/// Since 0.5.218 a household is founded by its founder alone (a founding +/// roster naming anyone else is `membership.founding_member_unsigned`, +/// CIRISConstitution#133 / CIRISPersist#955), so the aliases are witnessed on +/// the one roster a create can have: N = 1. #[tokio::test] async fn declared_majority_and_unanimous_are_stored_as_quorum() { let alice = Person::new("ma").await; let bob = Person::new("mb").await; let carol = Person::new("mc").await; acquainted(&[&alice, &bob, &carol]).await; - for (declared, stored) in [("majority", "quorum:2/3"), ("unanimous", "quorum:3/3")] { - let (st, v) = alice + for declared in ["majority", "unanimous"] { + let r = alice .as_owner( "POST", "/v1/families", @@ -858,7 +1038,255 @@ async fn declared_majority_and_unanimous_are_stored_as_quorum() { })), ) .await; + assert_refused(&r, 409, "membership.founding_member_unsigned"); + let (st, v) = alice + .as_owner( + "POST", + "/v1/families", + Some(json!({ "name": declared, "consensus_protocol": declared })), + ) + .await; assert_eq!(st.as_u16(), 201, "{v}"); - assert_eq!(v["consensus_protocol"], stored); + assert_eq!(v["consensus_protocol"], "quorum:1/1"); } } + +// ─── The invite flow (persist v52, CIRISPersist#955) ──────────────────────── + +fn proposal_ids(v: &serde_json::Value) -> Vec { + v["invites"] + .as_array() + .unwrap_or_else(|| panic!("no invites in {v}")) + .iter() + .map(|i| i["proposal_id"].as_str().expect("proposal_id").to_owned()) + .collect() +} + +fn state_of(v: &serde_json::Value, proposal: &str) -> String { + v["invites"] + .as_array() + .expect("invites") + .iter() + .find(|i| i["proposal_id"] == proposal) + .unwrap_or_else(|| panic!("{proposal} not listed: {v}"))["state"] + .as_str() + .expect("state") + .to_owned() +} + +/// **An invited person joins only by their own acceptance — across nodes.** +/// +/// Alice's node writes the proposal; it crosses to Bob's node (the rows edge +/// v38's serve arms route: a proposal to the invitee's nodes, a reply back to +/// the proposer's), Bob sees it in HIS inbox and accepts with HIS pen on HIS +/// node; until the acceptance crosses back, Alice's list seats nobody; once +/// it has, her list seats him (`joined`) and his node holds the household as +/// a member. Carol declines: terminal, and she is never seated. +#[tokio::test] +async fn an_invited_person_joins_only_by_their_own_acceptance() { + let alice = Person::new("ia").await; + let bob = Person::new("ib").await; + let carol = Person::new("ic").await; + acquainted(&[&alice, &bob, &carol]).await; + let id = create(&alice, json!({ "name": "home" })).await; + + let (st, v) = alice + .as_owner( + "POST", + &format!("/v1/families/{id}/invites"), + Some(json!({ "key_id": bob.key() })), + ) + .await; + assert_eq!( + (st.as_u16(), v["state"].as_str()), + (202, Some("invited")), + "{v}" + ); + let p = v["proposal_id"].as_str().expect("proposal_id").to_owned(); + + // The invitation reaches Bob's node; it is in his inbox, nobody else's. + bob.receive_membership_rows_from(&alice).await; + let (st, inbox) = bob.as_owner("GET", "/v1/self/invites", None).await; + assert_eq!(st.as_u16(), 200, "{inbox}"); + assert_eq!(proposal_ids(&inbox), vec![p.clone()], "{inbox}"); + assert_eq!(inbox["invites"][0]["group_kind"], "family"); + assert_eq!(inbox["invites"][0]["group_id"], id.as_str()); + carol.receive_membership_rows_from(&alice).await; + let (_, inbox) = carol.as_owner("GET", "/v1/self/invites", None).await; + assert!(proposal_ids(&inbox).is_empty(), "{inbox}"); + // A delegate sees the inbox but cannot answer it. + let delegate = bob.delegated_session().await; + let r = bob + .call( + "POST", + &format!("/v1/self/invites/{p}/accept"), + Some(&delegate), + Some(json!({})), + ) + .await; + assert_refused(&r, 403, "membership.delegate_may_not_answer"); + + let (st, v) = bob + .as_owner( + "POST", + &format!("/v1/self/invites/{p}/accept"), + Some(json!({})), + ) + .await; + assert_eq!( + (st.as_u16(), v["state"].as_str()), + (200, Some("accepted")), + "{v}" + ); + + // Before the acceptance crosses back, Alice's node seats nobody. + let (_, v) = alice + .as_owner("GET", &format!("/v1/families/{id}/invites"), None) + .await; + assert_eq!(state_of(&v, &p), "pending", "{v}"); + let (_, v) = alice + .as_owner("GET", &format!("/v1/families/{id}"), None) + .await; + assert_eq!(members(&v), vec![pair(&alice, "founder")], "{v}"); + + // It crosses; the founder's read seats him. + alice.receive_membership_rows_from(&bob).await; + let (st, v) = alice + .as_owner("GET", &format!("/v1/families/{id}/invites"), None) + .await; + assert_eq!(st.as_u16(), 200, "{v}"); + assert_eq!(v["seated_now"], json!([bob.key()]), "{v}"); + assert_eq!(state_of(&v, &p), "joined", "{v}"); + let (_, v) = alice + .as_owner("GET", &format!("/v1/families/{id}"), None) + .await; + assert_eq!( + members(&v), + sorted(vec![pair(&alice, "founder"), pair(&bob, "member")]), + "{v}" + ); + bob.receive_families_from(&alice).await; + let (st, v) = bob + .as_owner("GET", &format!("/v1/families/{id}"), None) + .await; + assert_eq!(st.as_u16(), 200, "{v}"); + assert_eq!(v["my_role"], "member", "{v}"); + + // Carol declines: terminal. + let (_, v) = alice + .as_owner( + "POST", + &format!("/v1/families/{id}/invites"), + Some(json!({ "key_id": carol.key() })), + ) + .await; + let pc = v["proposal_id"].as_str().expect("proposal_id").to_owned(); + carol.receive_membership_rows_from(&alice).await; + let (st, v) = carol + .as_owner( + "POST", + &format!("/v1/self/invites/{pc}/decline"), + Some(json!({})), + ) + .await; + assert_eq!( + (st.as_u16(), v["state"].as_str()), + (200, Some("declined")), + "{v}" + ); + let r = carol + .as_owner( + "POST", + &format!("/v1/self/invites/{pc}/accept"), + Some(json!({})), + ) + .await; + assert_refused(&r, 409, "membership.already_answered"); + alice.receive_membership_rows_from(&carol).await; + let (_, v) = alice + .as_owner("GET", &format!("/v1/families/{id}/invites"), None) + .await; + assert_eq!(state_of(&v, &pc), "declined", "{v}"); + let (_, v) = alice + .as_owner("GET", &format!("/v1/families/{id}"), None) + .await; + assert!( + !members(&v).iter().any(|(k, _)| k == carol.key()), + "a decline seats nobody: {v}" + ); + // An answered invitation cannot be withdrawn. + let r = alice + .as_owner("DELETE", &format!("/v1/families/{id}/invites/{pc}"), None) + .await; + assert_refused(&r, 409, "membership.invite_closed"); +} + +/// **A quorum household seats an accepted invitee by its co-signed WIDENING** +/// (persist v52 Q2: a supersede never adds). Bob invites Dave; Dave accepts on +/// his node; the reply reaches Alice; Alice and Bob sign the add — Bob on +/// HIS node — and assemble writes the widening carrying both signatures. +#[tokio::test] +async fn a_quorum_family_seats_an_accepted_invitee_by_a_cosigned_widening() { + let alice = Person::new("wa").await; + let bob = Person::new("wb").await; + let carol = Person::new("wc").await; + let dave = Person::new("wd").await; + acquainted(&[&alice, &bob, &carol, &dave]).await; + let id = quorum_family(&alice, &bob, &carol).await; + let (st, v) = bob + .as_owner( + "POST", + &format!("/v1/families/{id}/invites"), + Some(json!({ "key_id": dave.key() })), + ) + .await; + assert_eq!(st.as_u16(), 202, "{v}"); + let p = v["proposal_id"].as_str().expect("proposal_id").to_owned(); + dave.receive_membership_rows_from(&bob).await; + let (st, v) = dave + .as_owner( + "POST", + &format!("/v1/self/invites/{p}/accept"), + Some(json!({})), + ) + .await; + assert_eq!(st.as_u16(), 200, "{v}"); + alice.receive_membership_rows_from(&bob).await; + alice.receive_membership_rows_from(&dave).await; + // A quorum household's read never seats anyone on one signature. + let (_, v) = alice + .as_owner("GET", &format!("/v1/families/{id}/invites"), None) + .await; + assert_eq!( + state_of(&v, &p), + "accepted", + "accepted, awaiting the group: {v}" + ); + + let env = envelope( + &alice, + &id, + json!({ "action": "add", "key_id": dave.key() }), + ) + .await; + let a = cosign(&alice, &id, &env, &json!([])).await; + let ab = cosign(&bob, &id, &env, &a["signatures"]).await; + let (st, v) = assemble(&alice, &id, &env, &ab["signatures"]).await; + assert_eq!(st.as_u16(), 200, "{v}"); + assert_eq!(v["added"], dave.key(), "{v}"); + assert!( + members(&v).iter().any(|(k, _)| k == dave.key()), + "the accepted invitee is seated by the quorum's widening: {v}" + ); + let record = alice + .engine + .federation_directory() + .lookup_family(&id) + .await + .expect("lookup") + .expect("family"); + assert!( + !record.members.iter().any(|m| m.key_id == dave.key()), + "the record never grows (persist Q2) — the widening plane carries him" + ); +} diff --git a/tests/file_custody.rs b/tests/file_custody.rs new file mode 100644 index 00000000..344c48c3 --- /dev/null +++ b/tests/file_custody.rs @@ -0,0 +1,536 @@ +//! **Where each file is — the custody view** (0.5.218, `FSD/FILE_CUSTODY.md`). +//! +//! Two devices of ONE person, in process, with nothing under test stubbed: the +//! real drive router on each, real engines, the owner's minted fed-ID, and the +//! second device's crossing carried the way replication carries it (its node +//! key, the owner's binding onto it, its SIGNED content occurrence). +//! +//! The receipt is edge's own: the second device SIGNS it with +//! `receipts::sign_receipt` and EMITS it with `receipts::emit_receipt_row` — +//! exactly what `on_dag_pulled` does after a DAG pull stores the chunks — and +//! the first device ADMITS it with `receipts::admit_and_count`, the replication +//! bridge's own call. What is not run here is the pull itself (the blob swarm +//! over a transport); the native harness's `custody` relation +//! (`harness/native/topologies/selffiles.yaml`, CSD-107) asserts the whole +//! path on a real mesh. +//! +//! Pinned: +//! 1. a > 1 MiB self file written on A, receipted by B → A's custody lists +//! both devices, B `received` with the file's chunk count, A `here`, +//! `devices_total == 2`; the drive row carries `{devices_total: 2, +//! received_on: 1}`; +//! 2. an inline file is receiptable like any other (edge v38.0.0 / persist +//! v52's one-leaf log, CIRISPersist#953): `receipts_supported: true`, no +//! reason, no `custody.inline_no_receipt`, and its drive row counts +//! `received_on: 0` (known, none yet) rather than `null`; +//! 3. every refusal is the byte read's refusal, status AND id — no session, a +//! cohort the caller is not in, a row that is not there, a withdrawn file +//! (410); and persist's custody door itself refuses a viewer key that +//! cannot open the bytes (`not_granted`); +//! 4. the maintainer's ruling on #704 — "no copy here is a receipt (node +//! responsive, no copy)": on B, which holds the row and not the bytes, the +//! view is a 200 whose own entry is `holds: "none"` with `checked_at`, +//! `held_here: false`, `copies_known: 0`, while the byte read stays 409. + +use std::sync::Arc; + +use base64::engine::general_purpose::STANDARD as BASE64; +use base64::Engine as _; +use ciris_keyring::MlDsa65SoftwareSigner; +use ciris_persist::federation::types::{cohort_scope, identity_type}; +use ciris_persist::prelude::{Engine, HybridPolicy, LocalSigner}; +use ciris_persist::wa_cert::WaRole; +use ed25519_dalek::SigningKey; + +#[allow(dead_code)] // one fixture, several binaries: each uses a different subset +mod support { + include!("support/drive_fixture.rs"); +} +use support::*; + +async fn other_engine(alias: &str, ed: u8, pqc: u8) -> Arc { + let pqc_signer = Arc::new( + MlDsa65SoftwareSigner::from_seed_bytes(&[pqc; 32], format!("{alias}-pqc")) + .expect("ML-DSA-65 seed"), + ); + let signer = Arc::new(LocalSigner::from_parts( + SigningKey::from_bytes(&[ed; 32]), + alias.to_string(), + Some(pqc_signer), + Some(format!("{alias}-pqc")), + )); + Arc::new( + Engine::with_signer(signer, "sqlite::memory:") + .await + .expect("in-memory engine"), + ) +} + +/// The owner-binding claim-remote records LOCALLY for the other device. +async fn record_claim_locally(engine: &Engine, owner: &OwnerIdentity, node: &str) { + let scopes: Vec = ciris_server::auth::ownership::OWNER_BINDING_INFRA_SCOPES + .iter() + .map(|s| s.to_string()) + .collect(); + let binding = ciris_server::auth::ownership::build_signed_owner_binding( + &owner.signer().await, + node, + &scopes, + cohort_scope::SELF, + ) + .await + .expect("build the claim's owner-binding"); + ciris_server::auth::ownership::apply_signed_owner_binding( + engine, + node, + cohort_scope::SELF, + HybridPolicy::Strict, + &binding, + ) + .await + .expect("record the other device's owner-binding locally"); +} + +/// Bytes the write gate reads as an honest `application/octet-stream`. +fn blob(n: usize) -> Vec { + let mut v = b"\x00CIRIS-custody\x00".to_vec(); + v.extend((0..n.saturating_sub(v.len())).map(|i| ((i * 31 + i / 251) % 256) as u8)); + v +} + +async fn upload( + client: &reqwest::Client, + base: &str, + bearer: &str, + bytes: &[u8], + name: &str, +) -> String { + let (s, v) = status_json( + client + .post(format!("{base}/v1/files")) + .bearer_auth(bearer) + .json(&serde_json::json!({ + "cohort": "self", + "bytes_base64": BASE64.encode(bytes), + "media_type": "application/octet-stream", + "filename": name, + })) + .send() + .await + .expect("POST /v1/files"), + ) + .await; + assert_eq!(s, 200, "upload {name}: {v}"); + v["attestation_id"].as_str().expect("id").to_owned() +} + +async fn get( + client: &reqwest::Client, + base: &str, + bearer: Option<&str>, + path: &str, +) -> (u16, serde_json::Value) { + let mut req = client.get(format!("{base}{path}")); + if let Some(b) = bearer { + req = req.bearer_auth(b); + } + status_json(req.send().await.expect("GET")).await +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn the_author_device_names_the_device_that_received_the_file() { + init_tracing(); + let client = reqwest::Client::new(); + + // DEVICE A — owned, serving the drive. + let first = node_engine().await; + let owner = OwnerIdentity::mint().await; + let first_key = register_self(&first).await; + bind_owner(&first, &owner, &first_key).await; + let bearer = mint_session(&first, "wa-custody-first", WaRole::Root).await; + let base = serve_drive( + Arc::clone(&first), + node_edge_signer(&first).await, + owner.seed_dir.clone(), + ) + .await; + + // DEVICE B — same owner; its crossing onto A as replication carries it, + // BEFORE the write, so the file is wrapped to it (`can_open`). + let second = other_engine("ciris-custody-second", 0xC1, 0xC2).await; + let second_key = register_self(&second).await; + bind_owner(&second, &owner, &second_key).await; + let (occurrence, _) = + ciris_server::backend::provision_engine_occurrence(&second, &owner.key_id) + .await + .expect("the second device provisions its occurrence"); + seed_key(&first, &second_key, 0xC1, 0xC2, identity_type::NODE).await; + record_claim_locally(&first, &owner, &second_key).await; + let signed = second + .federation_directory() + .list_signed_identity_occurrences_for(&owner.key_id) + .await + .expect("the second device's signed occurrences") + .into_iter() + .find(|o| o.identity_occurrence.occurrence_key_id == occurrence) + .expect("the provisioned occurrence rides the signed plane"); + first + .federation_directory() + .put_identity_occurrence(signed) + .await + .expect("A admits B's occurrence"); + + // A > 1 MiB self file (a chunk DAG, so a stream and an STH) and an inline one. + let big = blob(2 * 1024 * 1024 + 3); + let big_id = upload(&client, &base, &bearer, &big, "big.bin").await; + let small_id = upload(&client, &base, &bearer, &blob(4096), "small.bin").await; + + // BEFORE any receipt: B is listed, `unknown`, and can open. + let (s, before) = get( + &client, + &base, + Some(&bearer), + &format!("/v1/files/{big_id}/custody?cohort=self"), + ) + .await; + assert_eq!(s, 200, "{before}"); + assert_eq!(before["devices_total"], 2, "{before}"); + assert_eq!(before["receipts_supported"], true, "{before}"); + let b_before = before["devices"] + .as_array() + .expect("devices") + .iter() + .find(|d| d["node_key_id"] == second_key.as_str()) + .unwrap_or_else(|| panic!("B is one of the person's devices: {before}")) + .clone(); + assert_eq!(b_before["holds"], "unknown", "{before}"); + assert_eq!(b_before["can_open"], true, "B was wrapped to: {before}"); + + // B RECEIPTS the file: edge's receive-side functions, as `on_dag_pulled` + // runs them after a stored pull … + let row = first + .federation_directory() + .get_attestation(&big_id) + .await + .expect("read") + .expect("the file row"); + let claim = ciris_edge::receipts::StreamSthClaim::from_row(&row) + .expect("a chunk-DAG file row carries its stream's STH"); + let receipt = ciris_edge::receipts::sign_receipt( + &second, + &second_key, + &claim.stream_id, + ciris_edge::receipts::FILE_STREAM_EPOCH, + claim.root().expect("root"), + claim.tree_size, + ) + .await + .expect("B signs the receipt"); + let receipt_row_id = ciris_edge::receipts::emit_receipt_row(&second, &row, &receipt) + .await + .expect("B emits the receipt row at the file's cohort"); + let receipt_row = second + .federation_directory() + .get_attestation(&receipt_row_id) + .await + .expect("read") + .expect("the receipt row"); + // … and A ADMITS it, the replication bridge's own call. + let log = + ciris_edge::receipts::stream_log_of(&first).expect("a SQLite engine has a stream log"); + let directory = first.federation_directory(); + ciris_edge::receipts::admit_and_count( + &*log, + &*directory, + &receipt_row, + &ciris_edge::receipts::ReceiptLedger::new(), + None, + ) + .await + .expect("A admits B's receipt"); + + // 1. A's custody: both devices, B received, A here. + let (s, v) = get( + &client, + &base, + Some(&bearer), + &format!("/v1/files/{big_id}/custody?cohort=self"), + ) + .await; + assert_eq!(s, 200, "{v}"); + assert_eq!(v["devices_total"], 2, "{v}"); + assert_eq!(v["held_here"], true, "{v}"); + assert_eq!(v["this_device_is_author"], true, "{v}"); + assert_eq!( + v["copies_observable"], false, + "self copies are uncountable by design: {v}" + ); + let devices = v["devices"].as_array().expect("devices"); + let a = devices + .iter() + .find(|d| d["this_device"] == true) + .unwrap_or_else(|| panic!("this device is listed: {v}")); + assert_eq!(a["node_key_id"], first_key.as_str(), "{v}"); + assert_eq!(a["holds"], "here", "{v}"); + let b = devices + .iter() + .find(|d| d["node_key_id"] == second_key.as_str()) + .unwrap_or_else(|| panic!("B is listed: {v}")); + assert_eq!(b["holds"], "received", "{v}"); + assert_eq!(b["received"]["k"], claim.tree_size, "{v}"); + // persist v52's `received_at` (CIRISPersist#953 item 3): the instant A's + // store took the receipt, always present. + assert!( + b["received"]["at"] + .as_str() + .is_some_and(|t| chrono::DateTime::parse_from_rfc3339(t).is_ok()), + "a receipt carries the store's received_at: {v}" + ); + assert!(claim.tree_size >= 2, "a 2 MiB file is several chunks: {v}"); + let why: Vec<&str> = v["why"] + .as_array() + .expect("why") + .iter() + .filter_map(|w| w["reason_id"].as_str()) + .collect(); + for id in [ + "custody.copies_unobservable_by_design", + "custody.receipt_is_delivery_not_holding", + ] { + assert!(why.contains(&id), "{id} in {why:?}"); + } + assert!( + v["receipts_from_other_keys"] + .as_array() + .is_some_and(|r| r.is_empty()), + "{v}" + ); + + // 2. The inline file is receiptable — nothing says otherwise. + let (s, v) = get( + &client, + &base, + Some(&bearer), + &format!("/v1/files/{small_id}/custody?cohort=self"), + ) + .await; + assert_eq!(s, 200, "{v}"); + assert_eq!(v["receipts_supported"], true, "{v}"); + assert!(v["receipts_unsupported_reason"].is_null(), "{v}"); + assert!( + !v["why"].as_array().is_some_and(|w| w + .iter() + .any(|x| x["reason_id"] == "custody.inline_no_receipt" + || x["reason_id"] == "custody.receipt_time_unknown")), + "{v}" + ); + assert_eq!(v["devices_total"], 2, "{v}"); + + // The drive's compact summary, per row. + let (s, drive) = get(&client, &base, Some(&bearer), "/v1/drive?cohort=self").await; + assert_eq!(s, 200, "{drive}"); + let entry = |id: &str| { + drive["entries"] + .as_array() + .expect("entries") + .iter() + .find(|e| e["attestation_id"] == id) + .unwrap_or_else(|| panic!("{id} listed: {drive}")) + .clone() + }; + assert_eq!( + entry(&big_id)["custody"], + serde_json::json!({"devices_total": 2, "received_on": 1}), + "{drive}" + ); + assert_eq!( + entry(&small_id)["custody"], + serde_json::json!({"devices_total": 2, "received_on": 0}), + "{drive}" + ); + + // 3. REFUSED EXACTLY AS THE BYTE READ REFUSES — status and id. + let same = |label: &str, read: (u16, serde_json::Value), custody: (u16, serde_json::Value)| { + assert_ne!(custody.0, 200, "{label}: custody answered {}", custody.1); + assert_eq!( + (custody.0, custody.1["reason_id"].clone()), + (read.0, read.1["reason_id"].clone()), + "{label}: the custody view refuses as the byte read does — read {} / custody {}", + read.1, + custody.1 + ); + }; + for (label, bearer, q) in [ + ("no session", None, "cohort=self"), + ( + "not a member", + Some(bearer.as_str()), + "cohort=family&room_id=not-my-family", + ), + ("unknown cohort", Some(bearer.as_str()), "cohort=everyone"), + ] { + same( + label, + get(&client, &base, bearer, &format!("/v1/files/{big_id}?{q}")).await, + get( + &client, + &base, + bearer, + &format!("/v1/files/{big_id}/custody?{q}"), + ) + .await, + ); + } + same( + "not in the room", + get( + &client, + &base, + Some(&bearer), + "/v1/files/no-such-file?cohort=self", + ) + .await, + get( + &client, + &base, + Some(&bearer), + "/v1/files/no-such-file/custody?cohort=self", + ) + .await, + ); + + // The substrate door itself: a viewer key that cannot open the bytes learns + // nothing — not the access list, not the copies. + let file = ciris_edge::files::FileRow::from_row(&row).expect("a file row"); + let store = ciris_edge::group_content::PersistGroupContentStore::new( + (*first).clone(), + first.federation_directory(), + ); + match file.custody(&store, "a-stranger-with-no-grant").await { + Err(reason) => assert_eq!(reason.kind(), "not_granted", "{reason}"), + Ok(c) => panic!("a stranger's viewer key read the custody view: {c:?}"), + } + + // B holds the ROW, not the bytes. The maintainer's ruling on #704: "no copy + // here is a receipt (node responsive, no copy)" — the view is authorized by + // the ROW, and B's own entry says `none`, with when it said so. The byte + // read still answers 409: the BYTES are not here; the custody FACT is. + seed_key(&second, &first_key, 0xA1, 0xA2, identity_type::NODE).await; + record_claim_locally(&second, &owner, &first_key).await; + second + .federation_directory() + .put_attestation(ciris_persist::federation::SignedAttestation { + attestation: row.clone(), + }) + .await + .expect("B admits the file row"); + let bearer2 = mint_session(&second, "wa-custody-second", WaRole::Root).await; + let base2 = serve_drive( + Arc::clone(&second), + edge_signer_for(&second_key, 0xC1, 0xC2), + owner.seed_dir.clone(), + ) + .await; + let (s, read) = get( + &client, + &base2, + Some(&bearer2), + &format!("/v1/files/{big_id}?cohort=self&raw=1"), + ) + .await; + assert_eq!( + (s, read["reason_id"].as_str()), + (409, Some("drive.not_fetched")), + "precondition: B holds the row, not the bytes: {read}" + ); + let (s, v) = get( + &client, + &base2, + Some(&bearer2), + &format!("/v1/files/{big_id}/custody?cohort=self"), + ) + .await; + assert_eq!(s, 200, "no copy here is an answer, not a refusal: {v}"); + assert_eq!(v["held_here"], false, "{v}"); + assert_eq!( + v["copies_known"], 0, + "a device with no copy is not a copy: {v}" + ); + assert_eq!(v["devices_total"], 2, "{v}"); + assert!( + v["access"].is_null(), + "not answerable without the blob: {v}" + ); + let mine = v["devices"] + .as_array() + .expect("devices") + .iter() + .find(|d| d["this_device"] == true) + .unwrap_or_else(|| panic!("B lists itself: {v}")) + .clone(); + assert_eq!(mine["node_key_id"], second_key.as_str(), "{v}"); + assert_eq!(mine["holds"], "none", "{v}"); + assert!( + mine["checked_at"] + .as_str() + .is_some_and(|t| chrono::DateTime::parse_from_rfc3339(t).is_ok()), + "a live statement carries its time: {v}" + ); + let a_from_b = v["devices"] + .as_array() + .expect("devices") + .iter() + .find(|d| d["node_key_id"] == first_key.as_str()) + .unwrap_or_else(|| panic!("A is listed on B: {v}")) + .clone(); + assert_eq!(a_from_b["holds"], "unknown", "{v}"); + assert!( + a_from_b["checked_at"].is_null() && a_from_b["reported_at"].is_null(), + "{v}" + ); + let why: Vec<&str> = v["why"] + .as_array() + .expect("why") + .iter() + .filter_map(|w| w["reason_id"].as_str()) + .collect(); + for id in [ + "custody.no_copy_here", + "custody.no_copy_reports_pending", + "custody.copies_unobservable_by_design", + "custody.receipts_admitted_on_author_device", + ] { + assert!(why.contains(&id), "{id} in {why:?}"); + } + + // A WITHDRAWN file stays the byte read's 410, on the author. + let (s, w) = status_json( + client + .delete(format!("{base}/v1/files/{big_id}?cohort=self")) + .bearer_auth(&bearer) + .send() + .await + .expect("DELETE"), + ) + .await; + assert_eq!(s, 200, "withdraw: {w}"); + let read = get( + &client, + &base, + Some(&bearer), + &format!("/v1/files/{big_id}?cohort=self&raw=1"), + ) + .await; + assert_eq!(read.0, 410, "{}", read.1); + same( + "withdrawn", + read, + get( + &client, + &base, + Some(&bearer), + &format!("/v1/files/{big_id}/custody?cohort=self"), + ) + .await, + ); +} diff --git a/tests/localization_gate.rs b/tests/localization_gate.rs index 3f1abde3..97fc7f5d 100644 --- a/tests/localization_gate.rs +++ b/tests/localization_gate.rs @@ -707,7 +707,44 @@ fn server_emitted_message_id_coverage_does_not_regress() { /// /// 140 -> 139 for 0.5.218: `family.readd_unsupported` retired — persist /// v49.0.0 re-admits a removed family member (#910.1). - const MAX_UNCOVERED: usize = 139; + /// + /// 139 -> 141 for 0.5.218: `self.evict_incomplete` (release and + /// occurrence/revoke are one signed act, CSD-037) and + /// `membership.consent_required` (nobody joins without their own consent, + /// CIRISConstitution#133 / CIRISPersist#955) — same reason, same list, + /// same client issue (CIRISClient#78). + /// + /// 141 -> 149 for 0.5.218, where each file is (`FSD/FILE_CUSTODY.md`): the + /// eight `custody.*` partial-answer reasons of + /// `GET /v1/files/{id}/custody` (`src/file_custody.rs`, one `msg` call per + /// id) — same reason, same list, same client issue (CIRISClient#78). + /// + /// 149 -> 151 for 0.5.218, the maintainer's ruling on #704 ("no copy here + /// is a receipt"): `custody.no_copy_reports_pending` and + /// `custody.no_copy_here` — same reason, same list, same client issue. + /// + /// 151 -> 153 for 0.5.218, the streamed drive upload (edge v36.1.0 + /// `files::publish_stream`): `drive.declared_length_mismatch` (the file + /// part is not the `size` the form declared) and `drive.field_after_file` + /// (a form field after the streamed file part) — same reason, same list, + /// same client issue (CIRISClient#78). + /// + /// 153 -> 156 for 0.5.218, one device handles each exchange (CC 3.1.3.1, + /// FSD/SESSION_CLAIMS.md §5): `GET /v1/self/sessions`'s + /// `session.state.handled_here`, `session.state.handled_elsewhere` and + /// `self.sessions_unavailable` — same reason, same list, same client issue + /// (CIRISClient#78). + /// + /// 156 -> 171 for 0.5.218 at persist v52.0.0 / edge v38.0.0: consent to + /// join, BUILT (CIRISPersist#955, `FSD/MEMBERSHIP_INVITES.md`). Down three + /// — `membership.consent_required` (the interim door, now the invite + /// flow), `custody.inline_no_receipt` and `custody.receipt_time_unknown` + /// (every file is receiptable and a receipt says when, CIRISPersist#953) — + /// and up eighteen: the invite flow's `membership.*` ids + /// (`src/membership_invites.rs`, one function per id: persist's eight + /// consent rules, the inbox gate, the invitation's lifecycle). Same + /// reason, same list, same client issue (CIRISClient#78). 156 − 3 + 18. + const MAX_UNCOVERED: usize = 171; let en = load_en(); let ids = scraped_server_ids(); diff --git a/tests/no_raw_roster_reads.rs b/tests/no_raw_roster_reads.rs index 5b95d26f..9bf0fa74 100644 --- a/tests/no_raw_roster_reads.rs +++ b/tests/no_raw_roster_reads.rs @@ -42,6 +42,12 @@ const ALLOWED: &[(&str, &str, &str, &str)] = &[ "family", "RECORD CONSTRUCTION: a quorum family's record is rewritten minus the leaver so verify's prior-roster binding (persist builds it from the record, not the fold) stays satisfiable", ), + ( + "src/family_api.rs", + "leave_inner", + "loaded.family", + "RECORD SHAPE, not membership (persist v52 #956): is the leaver ON the record, so the self-leave amendment has a seat to remove? A member seated by a widening is not — the record never grows (persist Q2) — and their revocation alone is the leave", + ), ( "src/family_api.rs", "leave_inner", diff --git a/tests/node_key_is_not_the_actor_key.rs b/tests/node_key_is_not_the_actor_key.rs index 8fbf0534..ec7e7058 100644 --- a/tests/node_key_is_not_the_actor_key.rs +++ b/tests/node_key_is_not_the_actor_key.rs @@ -234,31 +234,31 @@ async fn a_node_only_key_is_usable_as_the_node_identity() { assert!(v.usable_as_node()); } -/// **The loophole, demonstrated at the substrate.** `{node,agent}` registers -/// happily — persist accepts the row — and it is precisely the composition that -/// makes the agency gate stop constraining the key. The classifier must call it -/// `Fused` and must NOT call it usable. +/// **The loophole, closed at the substrate.** Through persist v49 a +/// `{node,agent}` key registered happily — the composition that makes the +/// agency gate stop constraining the key — and this test pinned that the +/// server's classifier still called it `Fused` and never usable. Since persist +/// v50 the registration door refuses it by name (CC 3.4.7.3 Clause A), so the +/// row cannot be written here any more. The classifier keeps its `Fused` arm +/// for rows a pre-v50 directory already holds. #[tokio::test] -async fn a_fused_node_agent_key_registers_and_must_still_be_refused() { +async fn a_fused_node_agent_key_is_refused_at_registration() { const FUSED: &str = "fused-node-agent"; let engine = engine_for(FUSED).await; - register(&engine, &signer_for(FUSED), FUSED, "node,agent").await; - - let v = classify(engine.federation_directory().as_ref(), FUSED) - .await - .expect("classify"); - assert!( - matches!(v, IdentityVerdict::Fused { .. }), - "expected Fused, got {v:?} — a `{{node,agent}}` key is not merely an actor and \ - not merely substrate; it is the CC 3.4.7.3 Clause A violation, and collapsing \ - it into either neighbour loses the reason the clause exists" + let refused = try_register(&engine, &signer_for(FUSED), FUSED, "node,agent").await; + let err = refused.expect_err( + "persist admitted a `{node,agent}` key — the Clause A door is gone, and the \ + server's classifier is again the only thing refusing it", ); assert!( - !v.usable_as_node(), - "a fused key must never be adopted as the node identity — persist's agency gate \ - constrains only a NODE-ONLY recipient, so operating as this key would leave \ - 'infrastructure must not have agency' nominally true and actually unenforced" + err.to_string().contains("Clause A"), + "refused, but not by the Clause A rule: {err}" ); + let v = classify(engine.federation_directory().as_ref(), FUSED) + .await + .expect("classify"); + assert_eq!(v, IdentityVerdict::Unregistered, "nothing was written"); + assert!(!v.usable_as_node()); } /// An unregistered key is not an actor and not substrate — it is unknown, and diff --git a/tests/occurrence.rs b/tests/occurrence.rs index ab7c7b5c..6335816a 100644 --- a/tests/occurrence.rs +++ b/tests/occurrence.rs @@ -16,6 +16,19 @@ //! second device. After it, `signer_acts_for(second, self)` is FALSE, and //! the roster no longer holds it. //! 4. A signer who does NOT act for the self (an unrelated key) is rejected 403. +//! +//! **0.5.218 (CSD-037).** The revoke moved to `crate::self_devices` and became +//! one act with node release (`evict_device`): the OWNER's session authorises +//! it and the owner's pen signs the revocation through persist's SIGNED, +//! replicating door. A device-signed request with no session is now refused +//! (step 3 asserts that, and that nothing was revoked); the revoke itself is +//! witnessed on a claimed node below — the signed row read back byte-exact +//! through `list_signed_identity_occurrence_revocations_for`, an occurrence +//! that IS an owned node losing its owner-binding too, and a source gate that +//! the trusted-local door has no caller in `src/`. + +#[path = "support/owned_node.rs"] +mod owned_node; use std::sync::Arc; @@ -155,9 +168,12 @@ async fn bind_primary(engine: &Engine, identity_key_id: &str, dev: &Device) { .expect("bind primary occurrence"); } -/// Serve the occurrence router on an ephemeral port. +/// Serve the occurrence router — and, since 0.5.218, the self-device router +/// the revoke moved to — on an ephemeral port. async fn serve(engine: Arc) -> (String, tokio::task::JoinHandle<()>) { - let app = occurrence::router(engine, HybridPolicy::Strict); + let app = occurrence::router(Arc::clone(&engine), HybridPolicy::Strict).merge( + ciris_server::self_devices::router(engine, std::env::temp_dir().join("no-owner-seed")), + ); let listener = tokio::net::TcpListener::bind("127.0.0.1:0") .await .expect("bind ephemeral port"); @@ -273,7 +289,12 @@ async fn enroll_second_device_then_revoke_it() { "an unauthenticated caller sees nothing of an un-announced person: {list}" ); - // ── (3) REVOKE: the PRIMARY revokes the (lost) SECOND device ── + // ── (3) REVOKE by a device signature alone: REFUSED since 0.5.218 ── + // + // The revocation is the OWNER's act, signed with the owner's pen through + // persist's signed door (CSD-037); a request carrying only a device's + // signature and no owner session never reaches it — and, before 0.5.218, + // it wrote the unsigned local-only row that never left this node. let revoke_body = serde_json::json!({ "identity_key_id": identity, "occurrence_key_id": second.key_id, @@ -287,29 +308,319 @@ async fn enroll_second_device_then_revoke_it() { &revoke_body, ) .await; - assert_eq!(status, 200, "revoke must succeed: {json}"); assert_eq!( - json["revoked_by"], primary.key_id, - "surviving key authored it" + status, 401, + "a device signature is not an owner session: {json}" ); + assert_eq!(json["reason_id"], "self.owner_session_required", "{json}"); + assert!( + signer_acts_for(&engine, &second.key_id, identity).await, + "nothing was revoked" + ); + assert_eq!(active_roster(&engine, identity).await.len(), 2); +} + +// ─── 0.5.218: the revoke is the owner's SIGNED, replicating act ───────────── - // The revoked device can no longer act as the self. +use owned_node::{assert_refused, Person}; + +/// Register a device key and bind it as an occurrence of `p`'s self, with +/// content-KEM keys (so it is a wrap recipient). Returns its key id. +async fn phone_of(p: &Person, tag: u8) -> String { + let key_id = format!("{}-phone-{tag}", p.owner.alias); + let dev = Device::new(&key_id, tag); + register_key(&p.engine, &dev, "user").await; + p.engine + .federation_directory() + .put_identity_occurrence_local(ciris_persist::federation::IdentityOccurrence { + identity_key_id: p.key().to_owned(), + occurrence_key_id: key_id.clone(), + device_class: "phone".into(), + hardware_attestation: None, + asserted_at: chrono::Utc::now(), + valid_until: None, + encryption_pubkeys: Some( + ciris_server::identity::derive_self_enc_pubkeys(&[tag; 32]).expect("enc keys"), + ), + transport_binding: None, + persist_row_hash: String::new(), + }) + .await + .expect("bind the phone"); + key_id +} + +/// **(i) The revoke writes a SIGNED revocation** — readable through persist's +/// byte-exact re-read (`list_signed_identity_occurrence_revocations_for`, +/// which omits the unsigned local rows), attested by the OWNER — and every +/// answer says what eviction does not do. +#[tokio::test] +async fn the_owner_revokes_a_device_with_a_signed_replicating_revocation() { + let alice = Person::new("occ-rev").await; + let phone = phone_of(&alice, 0x51).await; + let kept = phone_of(&alice, 0x53).await; + let dir = alice.engine.federation_directory(); assert!( - !signer_acts_for(&engine, &second.key_id, identity).await, - "after revocation the second device must NOT act for the self" + dir.list_signed_identity_occurrence_revocations_for(alice.key()) + .await + .expect("signed revocations") + .is_empty(), + "precondition: nothing revoked" ); - // ...and the primary still can (revocation is scoped to the one device). + + let (st, v) = alice + .as_owner( + "POST", + "/v1/self/occurrence/revoke", + Some(serde_json::json!({ + "identity_key_id": alice.key(), + "occurrence_key_id": phone, + "reason": "phone stolen", + })), + ) + .await; + assert_eq!(st.as_u16(), 200, "{v}"); + assert_eq!(v["revoked"], true, "{v}"); + assert_eq!(v["revoked_by"], alice.key()); + assert_eq!( + v["occurrences_revoked"].as_array().map(Vec::len), + Some(1), + "{v}" + ); + assert!(v["failed"].as_array().is_some_and(Vec::is_empty), "{v}"); assert!( - signer_acts_for(&engine, &primary.key_id, identity).await, - "the surviving device still acts for the self" + v["history"] + .as_str() + .is_some_and(|h| h.contains("Already-shared history stays readable")), + "the answer says what eviction does NOT do: {v}" ); - // The roster drops the revoked device. - let ids = active_roster(&engine, identity).await; + // THE SIGNED ROW, read back the way a replicator re-publishes it. + let signed = dir + .list_signed_identity_occurrence_revocations_for(alice.key()) + .await + .expect("signed revocations"); + let row = signed + .iter() + .find(|r| r.identity_occurrence_revocation.occurrence_key_id == phone) + .expect("the revocation is on the SIGNED surface — not the local door"); + assert_eq!(row.attesting_key_id, alice.key(), "the owner signed it"); + assert_eq!(row.signed_envelope["occurrence_key_id"], phone.as_str()); + assert_eq!(row.signed_envelope["reason"], "phone stolen"); + assert!(row.signature.mldsa65_signature_base64.is_some(), "hybrid"); + + // The device is out of the self; the other one is not. + assert!(!signer_acts_for(&alice.engine, &phone, alice.key()).await); + assert!(signer_acts_for(&alice.engine, &kept, alice.key()).await); + let active = active_roster(&alice.engine, alice.key()).await; + assert!( + !active.contains(&phone) && active.contains(&kept), + "{active:?}" + ); + + // Again: already revoked — nothing to sign, and said so. + let (st, v) = alice + .as_owner( + "POST", + "/v1/self/occurrence/revoke", + Some(serde_json::json!({ + "identity_key_id": alice.key(), + "occurrence_key_id": phone, + })), + ) + .await; + assert_eq!(st.as_u16(), 200, "{v}"); + assert_eq!(v["occurrences_already_revoked"], serde_json::json!([phone])); +} + +/// The revoke's gate: the owner's session, never a delegate, and only the +/// caller's own occurrences — one id for "someone else's" and "nobody's". +#[tokio::test] +async fn only_the_owner_revokes_and_only_their_own_devices() { + let alice = Person::new("occ-gate").await; + let bob = Person::new("occ-gate-b").await; + let phone = phone_of(&alice, 0x61).await; + let bobs = phone_of(&bob, 0x63).await; + alice.knows(&bob).await; + let body = |identity: &str, occ: &str| serde_json::json!({ "identity_key_id": identity, "occurrence_key_id": occ }); + let path = "/v1/self/occurrence/revoke"; + let r = alice + .call("POST", path, None, Some(body(alice.key(), &phone))) + .await; + assert_refused(&r, 401, "self.owner_session_required"); + let guest = alice.stranger_session().await; + let r = alice + .call("POST", path, Some(&guest), Some(body(alice.key(), &phone))) + .await; + assert_refused(&r, 403, "self.owner_session_required"); + let delegated = alice.delegated_session().await; + let r = alice + .call( + "POST", + path, + Some(&delegated), + Some(body(alice.key(), &phone)), + ) + .await; + assert_refused(&r, 403, "self.delegate_may_not_author"); + // Bob's identity, Bob's phone under Alice's identity, a key nobody bound. + let r = alice + .as_owner("POST", path, Some(body(bob.key(), &bobs))) + .await; + assert_refused(&r, 404, "self.not_your_device"); + let r = alice + .as_owner("POST", path, Some(body(alice.key(), &bobs))) + .await; + assert_refused(&r, 404, "self.not_your_device"); + let r = alice + .as_owner("POST", path, Some(body(alice.key(), "no-such-device"))) + .await; + assert_refused(&r, 404, "self.not_your_device"); + assert!( + signer_acts_for(&alice.engine, &phone, alice.key()).await, + "nothing was revoked" + ); +} + +/// **The two routes are one act:** revoking an occurrence that IS one of the +/// owner's nodes also withdraws that node's owner-binding, so the self room +/// drops it too — not only the wraps. +#[tokio::test] +async fn revoking_a_node_occurrence_also_releases_the_node() { + use ciris_persist::federation::admission::nodes_owned_by; + let alice = Person::new("occ-node").await; + let node = "occ-second-node-0x71".to_owned(); + let dev = Device::new(&node, 0x71); + register_key(&alice.engine, &dev, "node").await; + let scopes: Vec = ciris_server::auth::ownership::OWNER_BINDING_INFRA_SCOPES + .iter() + .map(|s| s.to_string()) + .collect(); + ciris_server::auth::ownership::emit_steward_binding( + &alice.engine, + &alice.owner.signer().await, + &node, + &scopes, + ) + .await + .expect("owner-bind the second node"); + alice + .engine + .federation_directory() + .put_identity_occurrence_local(ciris_persist::federation::IdentityOccurrence { + identity_key_id: alice.key().to_owned(), + occurrence_key_id: node.clone(), + device_class: "server".into(), + hardware_attestation: None, + asserted_at: chrono::Utc::now(), + valid_until: None, + encryption_pubkeys: None, + transport_binding: None, + persist_row_hash: String::new(), + }) + .await + .expect("the node's content occurrence"); + let owned = |e: Arc, k: String| async move { + nodes_owned_by(e.federation_directory().as_ref(), &k) + .await + .expect("nodes_owned_by") + }; + assert!(owned(Arc::clone(&alice.engine), alice.key().to_owned()) + .await + .contains(&node)); + + let (st, v) = alice + .as_owner( + "POST", + "/v1/self/occurrence/revoke", + Some(serde_json::json!({ + "identity_key_id": alice.key(), + "occurrence_key_id": node, + })), + ) + .await; + assert_eq!(st.as_u16(), 200, "{v}"); + assert_eq!(v["nodes"], serde_json::json!([node]), "{v}"); + assert_eq!(v["withdrawn"].as_array().map(Vec::len), Some(1), "{v}"); assert_eq!( - ids, - vec![primary.key_id.clone()], - "only the surviving device remains" + v["occurrences_revoked"].as_array().map(Vec::len), + Some(1), + "{v}" + ); + assert!( + !owned(Arc::clone(&alice.engine), alice.key().to_owned()) + .await + .contains(&node), + "the owner-binding went with the occurrence" + ); + // Revoking the occurrence of the node you are talking to needs force, as + // releasing it does. + let (occ, _) = ciris_server::backend::provision_engine_occurrence(&alice.engine, alice.key()) + .await + .expect("this node's occurrence"); + let r = alice + .as_owner( + "POST", + "/v1/self/occurrence/revoke", + Some(serde_json::json!({ + "identity_key_id": alice.key(), + "occurrence_key_id": occ, + })), + ) + .await; + assert_refused(&r, 409, "self.release_self_requires_force"); +} + +/// **The trusted-local revocation door has no caller in `src/`.** Its rows +/// are unsigned and excluded from the signed replication read by +/// construction, so a revocation written through it never leaves the node — +/// the gap `evict_device` closed (CSD-037). The allow-list is empty on +/// purpose: a new caller must argue for itself here, by name. +#[test] +fn the_local_revocation_door_has_no_caller_in_src() { + const ALLOWED: &[&str] = &[]; + let mut hits = Vec::new(); + let mut stack = vec![std::path::PathBuf::from(concat!( + env!("CARGO_MANIFEST_DIR"), + "/src" + ))]; + let mut scanned = 0usize; + while let Some(dir) = stack.pop() { + for entry in std::fs::read_dir(&dir).expect("read src dir") { + let path = entry.expect("entry").path(); + if path.is_dir() { + stack.push(path); + continue; + } + if path.extension().and_then(|e| e.to_str()) != Some("rs") { + continue; + } + scanned += 1; + let text = std::fs::read_to_string(&path).expect("read source"); + for (n, line) in text.lines().enumerate() { + let code = line.split("//").next().unwrap_or(""); + if code.contains("put_identity_occurrence_revocation_local") { + let rel = path + .strip_prefix(env!("CARGO_MANIFEST_DIR")) + .unwrap_or(&path) + .display() + .to_string(); + if !ALLOWED.contains(&rel.as_str()) { + hits.push(format!("{rel}:{}: {}", n + 1, line.trim())); + } + } + } + } + } + assert!( + scanned > 50, + "the gate scanned {scanned} files — pointed at the wrong tree?" + ); + assert!( + hits.is_empty(), + "put_identity_occurrence_revocation_local is called from src/ — its rows never replicate; \ + revoke through self_devices::evict_device (the signed door):\n{}", + hits.join("\n") ); } diff --git a/tests/one_device_handles_each_exchange.rs b/tests/one_device_handles_each_exchange.rs new file mode 100644 index 00000000..ba8063aa --- /dev/null +++ b/tests/one_device_handles_each_exchange.rs @@ -0,0 +1,547 @@ +//! **One device handles each exchange** (CC 3.1.3.1, CIRISPersist#782; the +//! maintainer's ruling of 2026-09-30, `FSD/SESSION_CLAIMS.md`). +//! +//! Two occurrences of ONE owner — two real substrates, each with its own +//! hybrid node key, each bound to the same minted fed-ID, each holding the +//! other's key and owner-binding the way claim-remote and replication leave +//! them — and the claim rows carried between them by hand, through each +//! receiver's own admission door, exactly as a round would carry them. +//! +//! What must hold: +//! +//! 1. **With no claim, neither device acts** — not even the one that could. +//! 2. **Only the claiming device acts**; the other, attended or not, defers +//! and writes no claim of its own (a live claim is never contested). +//! 3. **After the claim lapses and the other device is the attended one, the +//! other acts**, and the first defers to it. +//! 4. **A claim row authored by a third party is refused at admission** — and +//! a stranger's claim about ITSELF never makes it a handler of this +//! person's exchange. +//! 5. `GET /v1/self/sessions` names the SAME handler on both devices. +//! 6. The re-wrap of old self files — a real ACT site — runs only on the +//! device the fold names. + +use std::sync::Arc; + +use base64::engine::general_purpose::STANDARD as BASE64; +use base64::Engine as _; +use ciris_keyring::MlDsa65SoftwareSigner; +use ciris_persist::federation::session_claim::SESSION_CLAIM_DIMENSION; +use ciris_persist::federation::types::{cohort_scope, identity_type, SignedAttestation}; +use ciris_persist::prelude::{Engine, HybridPolicy, LocalSigner}; +use ciris_persist::wa_cert::WaRole; +use ciris_server::session_claims::{ + self, Attendance, Occupant, Step, Verdict, SELF_ROOM_MEMBERSHIP_SESSION, +}; +use ed25519_dalek::SigningKey; + +#[allow(dead_code)] // one fixture, several binaries: each uses a different subset +mod support { + include!("support/drive_fixture.rs"); +} +use support::*; + +// ── two devices of one person ────────────────────────────────────────────── + +/// The second device's seeds — `other_engine` builds its signer from them and +/// `seed_key` registers the same pubkeys on the first device. +const B_ED: u8 = 0xC1; +const B_PQC: u8 = 0xC2; + +async fn other_engine(alias: &str, ed: u8, pqc: u8) -> Arc { + let pqc_signer = Arc::new( + MlDsa65SoftwareSigner::from_seed_bytes(&[pqc; 32], format!("{alias}-pqc")) + .expect("ML-DSA-65 seed"), + ); + let signer = Arc::new(LocalSigner::from_parts( + SigningKey::from_bytes(&[ed; 32]), + alias.to_string(), + Some(pqc_signer), + Some(format!("{alias}-pqc")), + )); + Arc::new( + Engine::with_signer(signer, "sqlite::memory:") + .await + .expect("in-memory engine"), + ) +} + +/// The owner-binding claim-remote records locally for the other device. +async fn record_binding(engine: &Engine, owner: &OwnerIdentity, node: &str) { + let scopes: Vec = ciris_server::auth::ownership::OWNER_BINDING_INFRA_SCOPES + .iter() + .map(|s| s.to_string()) + .collect(); + let binding = ciris_server::auth::ownership::build_signed_owner_binding( + &owner.signer().await, + node, + &scopes, + cohort_scope::SELF, + ) + .await + .expect("build the owner-binding"); + ciris_server::auth::ownership::apply_signed_owner_binding( + engine, + node, + cohort_scope::SELF, + HybridPolicy::Strict, + &binding, + ) + .await + .expect("record the other device's owner-binding"); +} + +struct Person { + owner: OwnerIdentity, + a: Arc, + a_key: String, + b: Arc, + b_key: String, +} + +impl Person { + /// Two devices of one owner, each knowing the other the way the claim and + /// replication leave them: the other's hybrid key and owner-binding. + async fn two_devices(b_alias: &str) -> Self { + let owner = OwnerIdentity::mint().await; + let a = node_engine().await; + let a_key = register_self(&a).await; + bind_owner(&a, &owner, &a_key).await; + let b = other_engine(b_alias, B_ED, B_PQC).await; + let b_key = register_self(&b).await; + bind_owner(&b, &owner, &b_key).await; + seed_key(&a, &b_key, B_ED, B_PQC, identity_type::NODE).await; + record_binding(&a, &owner, &b_key).await; + seed_key(&b, &a_key, 0xA1, 0xA2, identity_type::NODE).await; + record_binding(&b, &owner, &a_key).await; + Self { + owner, + a, + a_key, + b, + b_key, + } + } + + fn occupant(&self, key: &str) -> Occupant { + Occupant { + owner: self.owner.key_id.clone(), + occurrence: key.to_owned(), + } + } + + fn community(&self) -> String { + session_claims::self_community(&self.owner.key_id) + } +} + +/// The session-claim rows `from_key` authored on `from` — self-reports. +async fn claims_by(from: &Engine, from_key: &str) -> Vec { + from.federation_directory() + .list_attestations_for(from_key) + .await + .expect("rows about the occurrence") + .into_iter() + .filter(|r| { + r.attesting_key_id == from_key + && r.attestation_envelope + .get(ciris_persist::federation::envelope::paths::DIMENSION) + .and_then(|d| d.as_str()) + == Some(SESSION_CLAIM_DIMENSION) + }) + .map(|attestation| SignedAttestation { attestation }) + .collect() +} + +/// Carry every claim `from_key` wrote on `from` to `to`, through `to`'s own +/// admission door — what a replication round does with a `self` row. +async fn carry_claims(from: &Engine, from_key: &str, to: &Engine) { + for row in claims_by(from, from_key).await { + to.federation_directory() + .put_attestation(row) + .await + .expect("the receiving device admits its sibling's self-reported claim"); + } +} + +// ── 1-3: who acts ────────────────────────────────────────────────────────── + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn one_device_acts_the_other_defers_and_takes_over_after_the_claim_lapses() { + init_tracing(); + let p = Person::two_devices("ciris-sessions-b").await; + let (att_a, att_b) = (Attendance::new(), Attendance::new()); + let community = p.community(); + let s = SELF_ROOM_MEMBERSHIP_SESSION; + let t0 = chrono::Utc::now(); + + // 1. NOBODY IS HERE: neither acts, neither claims. The lone-device + // shortcut ("only I could, so I will") is exactly what CC 3.1.3.1 bans. + for (e, k, att) in [(&p.a, &p.a_key, &att_a), (&p.b, &p.b_key, &att_b)] { + let v = session_claims::gate_at(e, att, &p.occupant(k), &community, s, "test", t0).await; + assert_eq!(v, Verdict::Unclaimed, "no claim: nobody acts ({k})"); + assert!( + claims_by(e, k).await.is_empty(), + "an unattended device writes no claim" + ); + } + + // 2. THE PERSON IS ON A: A takes the exchange on demand and acts. + att_a.note_presence(); + let v = session_claims::gate_at( + &p.a, + &att_a, + &p.occupant(&p.a_key), + &community, + s, + "test", + t0, + ) + .await; + assert_eq!( + v, + Verdict::Act, + "the attended device claims what nobody holds, and acts" + ); + carry_claims(&p.a, &p.a_key, &p.b).await; + + // …and B defers — even when the person is on B too. A live claim is never + // contested, so B writes nothing. + att_b.note_presence(); + let v = session_claims::gate_at( + &p.b, + &att_b, + &p.occupant(&p.b_key), + &community, + s, + "test", + t0, + ) + .await; + assert_eq!( + v, + Verdict::HandledElsewhere { + occurrence: p.a_key.clone() + }, + "B sees A's claim and does not act" + ); + assert!( + claims_by(&p.b, &p.b_key).await.is_empty(), + "B wrote no competing claim while A's is live" + ); + // B's renewal pass agrees: defer, write nothing. + let lines = session_claims::renew_once_at(&p.b, &att_b, t0).await; + assert_eq!( + lines.iter().map(|l| &l.step).collect::>(), + vec![&Step::Defer { + handler: p.a_key.clone() + }] + ); + + // A's renewal while the person stays: fresh lease → hold; 60 s on → renew. + let lines = session_claims::renew_once_at(&p.a, &att_a, t0).await; + assert_eq!(lines[0].step, Step::Hold, "{lines:?}"); + let renew_at = t0 + chrono::Duration::seconds(61); + let lines = session_claims::renew_once_at(&p.a, &att_a, renew_at).await; + assert_eq!(lines[0].step, Step::Renew, "{lines:?}"); + assert_eq!( + claims_by(&p.a, &p.a_key).await.len(), + 2, + "a successor lease was written" + ); + + // 3. THE PERSON LEAVES A. A lapses — writes nothing more. + att_a.end_presence(); + let before = claims_by(&p.a, &p.a_key).await.len(); + let lines = session_claims::renew_once_at(&p.a, &att_a, renew_at).await; + assert_eq!(lines[0].step, Step::Lapse, "{lines:?}"); + assert_eq!( + claims_by(&p.a, &p.a_key).await.len(), + before, + "a lapsing device renews nothing" + ); + + // …while A's newest lease is live, B still defers… + let v = session_claims::gate_at( + &p.b, + &att_b, + &p.occupant(&p.b_key), + &community, + s, + "test", + renew_at, + ) + .await; + assert!(matches!(v, Verdict::HandledElsewhere { .. }), "{v:?}"); + + // …and once every A lease is past the TTL, B (attended) takes it and acts. + let lapsed = renew_at + chrono::Duration::from_std(session_claims::SESSION_CLAIM_TTL).unwrap(); + let v = session_claims::gate_at( + &p.b, + &att_b, + &p.occupant(&p.b_key), + &community, + s, + "test", + lapsed, + ) + .await; + assert_eq!(v, Verdict::Act, "after A's claim lapsed, B is the handler"); + carry_claims(&p.b, &p.b_key, &p.a).await; + let v = session_claims::gate_at( + &p.a, + &att_a, + &p.occupant(&p.a_key), + &community, + s, + "test", + lapsed, + ) + .await; + assert_eq!( + v, + Verdict::HandledElsewhere { + occurrence: p.b_key.clone() + }, + "A, whose person left, now defers to B" + ); + + // 5. BOTH DEVICES NAME THE SAME HANDLER on the surface's body. + let va = session_claims::sessions_view(&p.a, &att_a, &p.owner.key_id, &p.a_key) + .await + .expect("A's view"); + let vb = session_claims::sessions_view(&p.b, &att_b, &p.owner.key_id, &p.b_key) + .await + .expect("B's view"); + // (The view reads the real clock, when B's claim — dated `lapsed`, in the + // future — is live and A's are too: the fold's earliest live claim is A's. + // Both devices must agree on it, which is the property; which one it is + // follows from persist's merge.) + let handler = |v: &serde_json::Value| -> Option { + v["sessions"] + .as_array() + .and_then(|a| a.iter().find(|x| x["session_id"] == s)) + .and_then(|x| x["handler_occurrence_key_id"].as_str().map(str::to_owned)) + }; + assert!(handler(&va).is_some(), "A lists the exchange: {va}"); + assert_eq!( + handler(&va), + handler(&vb), + "both devices name ONE handler: {va} / {vb}" + ); +} + +// ── 4: a third party cannot claim for the person ─────────────────────────── + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn a_claim_authored_by_a_third_party_is_refused_at_admission() { + init_tracing(); + let p = Person::two_devices("ciris-sessions-third").await; + let stranger_alias = "session-claim-stranger"; + seed_key(&p.a, stranger_alias, 0x7A, 0x7B, identity_type::NODE).await; + let stranger = LocalSigner::from_parts( + SigningKey::from_bytes(&[0x7A; 32]), + stranger_alias.to_string(), + Some(Arc::new( + MlDsa65SoftwareSigner::from_seed_bytes(&[0x7B; 32], format!("{stranger_alias}-pqc")) + .expect("pqc"), + )), + Some(format!("{stranger_alias}-pqc")), + ); + let community = p.community(); + + // A claim ABOUT the person's device, signed by someone else: refused. + let env = serde_json::json!({ + (ciris_persist::federation::envelope::paths::DIMENSION): SESSION_CLAIM_DIMENSION, + "score": 1.0, + "community_id": community, + "session_id": SELF_ROOM_MEMBERSHIP_SESSION, + "claimed_at": session_claims::canonical_instant(chrono::Utc::now()), + }); + let err = ciris_server::attest::emit( + &p.a, + ciris_server::attest::KeySigner::Local(&stranger), + ciris_server::attest::Spec::new( + ciris_persist::federation::types::attestation_type::SCORES, + cohort_scope::SELF, + env, + ) + .attested_to(&p.a_key) + .weighing(Some(1.0)), + ) + .await + .expect_err("a third party's session claim must be refused at the door"); + assert!( + format!("{err}").contains("SELF-REPORT"), + "refused BY NAME as a self-report violation: {err}" + ); + + // A stranger's claim about ITSELF is admitted (it is a well-formed + // self-report) — and makes it no handler of this person's exchange: the + // fold walks the person's own occurrences only. + session_claims::write_claim( + &p.a, + ciris_server::attest::KeySigner::Local(&stranger), + &community, + SELF_ROOM_MEMBERSHIP_SESSION, + chrono::Utc::now(), + ) + .await + .expect("a self-report about oneself admits"); + let v = session_claims::gate( + &p.a, + &Attendance::new(), + &p.occupant(&p.a_key), + &community, + SELF_ROOM_MEMBERSHIP_SESSION, + "test", + ) + .await; + assert_eq!( + v, + Verdict::Unclaimed, + "an outsider's claim is not one of the person's occurrences — nobody acts" + ); +} + +// ── 6: a real ACT site — the re-wrap for a new device ────────────────────── + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn the_rewrap_runs_only_on_the_device_the_fold_names() { + use ciris_persist::federation::blobs::BlobStorage as _; + init_tracing(); + let p = Person::two_devices("ciris-sessions-rewrap").await; + + // A holds a self file written before B's content occurrence existed. + let bearer = mint_session(&p.a, "wa-sessions-owner", WaRole::Root).await; + let base = serve_drive( + Arc::clone(&p.a), + node_edge_signer(&p.a).await, + p.owner.seed_dir.clone(), + ) + .await; + let client = reqwest::Client::new(); + let (st, v) = status_json( + client + .post(format!("{base}/v1/files")) + .bearer_auth(&bearer) + .json(&serde_json::json!({ + "cohort": "self", + "bytes_base64": BASE64.encode(b"one device re-wraps this"), + "media_type": "text/plain", + "filename": "before.txt", + })) + .send() + .await + .expect("POST /v1/files"), + ) + .await; + assert_eq!(st, 200, "self upload: {v}"); + let id = v["attestation_id"].as_str().expect("id").to_owned(); + let (st, meta) = status_json( + client + .get(format!("{base}/v1/files/{id}/meta?cohort=self")) + .bearer_auth(&bearer) + .send() + .await + .expect("meta"), + ) + .await; + assert_eq!(st, 200, "{meta}"); + let sha: [u8; 32] = hex::decode(meta["at_rest_sha256"].as_str().expect("sha")) + .expect("hex") + .try_into() + .expect("32 bytes"); + + // B provisions its content occurrence; the signed row crosses to A. + let (occurrence, _) = ciris_server::backend::provision_engine_occurrence(&p.b, &p.owner.key_id) + .await + .expect("B provisions its occurrence"); + let row = + p.b.federation_directory() + .list_signed_identity_occurrences_for(&p.owner.key_id) + .await + .expect("signed occurrences") + .into_iter() + .find(|o| o.identity_occurrence.occurrence_key_id == occurrence) + .expect("the occurrence rides the signed plane"); + p.a.federation_directory() + .put_identity_occurrence(row) + .await + .expect("A admits B's occurrence"); + let granted_to_b = || async { + p.a.sqlite_backend() + .expect("sqlite") + .list_at_rest_grant_recipients(&sha) + .await + .expect("recipients") + .contains(&occurrence) + }; + + // The pen opens on A (what compose registers at boot). + ciris_server::node_key::set_user_seed_dir(p.owner.seed_dir.clone(), p.owner.alias.clone()); + let community = p.community(); + let session = session_claims::rewrap_session(&occurrence); + + // UNCLAIMED, nobody attended on A: the pen is here, the work is pending, + // and NOTHING runs. + let att_a = Attendance::new(); + let r = ciris_server::self_rewrap::rewrap_for_new_devices_with(&p.a, &p.a_key, &att_a).await; + assert_eq!(r.pending, vec![occurrence.clone()], "{r:?}"); + assert!(!r.no_pen_here, "{r:?}"); + assert_eq!( + r.not_handled_here, + vec![(occurrence.clone(), None)], + "{r:?}" + ); + assert!(r.rewrapped.is_empty() && !granted_to_b().await); + + // HANDLED ELSEWHERE: another device of the person holds a live claim on + // this re-wrap (written there, carried here). A, attended, defers. The + // claim is dated so it has ~10 s of life left: long enough for this step, + // short enough to watch it lapse below without waiting out the TTL. + let ttl = chrono::Duration::from_std(session_claims::SESSION_CLAIM_TTL).unwrap(); + session_claims::write_claim( + &p.b, + ciris_server::attest::KeySigner::Engine(&p.b), + &community, + &session, + chrono::Utc::now() - ttl + chrono::Duration::seconds(10), + ) + .await + .expect("B claims"); + carry_claims(&p.b, &p.b_key, &p.a).await; + att_a.note_presence(); + let r = ciris_server::self_rewrap::rewrap_for_new_devices_with(&p.a, &p.a_key, &att_a).await; + assert_eq!( + r.not_handled_here, + vec![(occurrence.clone(), Some(p.b_key.clone()))], + "{r:?}" + ); + assert!(r.rewrapped.is_empty() && !granted_to_b().await); + assert!( + claims_by(&p.a, &p.a_key).await.is_empty(), + "A did not contest B's live claim" + ); + + // HANDLED HERE: B's claim lapses (nobody renewed it); A, attended, takes + // the session and the re-wrap RUNS — once. + tokio::time::sleep(std::time::Duration::from_secs(11)).await; + let r = ciris_server::self_rewrap::rewrap_for_new_devices_with(&p.a, &p.a_key, &att_a).await; + assert!(r.not_handled_here.is_empty(), "{r:?}"); + assert_eq!(r.rewrapped.len(), 1, "{r:?}"); + assert!(granted_to_b().await, "the old self file now opens on B"); + assert_eq!( + claims_by(&p.a, &p.a_key).await.len(), + 1, + "A claimed the re-wrap on demand before acting" + ); + + // IDEMPOTENT per act: the next pass has nothing pending. + let again = + ciris_server::self_rewrap::rewrap_for_new_devices_with(&p.a, &p.a_key, &att_a).await; + assert!( + again.pending.is_empty() && again.rewrapped.is_empty(), + "{again:?}" + ); +} diff --git a/tests/release_gates/substrate.rs b/tests/release_gates/substrate.rs index 43a06d5b..1596a91c 100644 --- a/tests/release_gates/substrate.rs +++ b/tests/release_gates/substrate.rs @@ -12,9 +12,9 @@ use crate::ladder::{ /// The substrate floor this cut ships on. Moving a release means moving these /// three deliberately, in one commit. -pub const TARGET_VERIFY: &str = "v17.1.0"; -pub const TARGET_PERSIST: &str = "v49.0.0"; -pub const TARGET_EDGE: &str = "v32.1.0"; +pub const TARGET_VERIFY: &str = "v18.0.0"; +pub const TARGET_PERSIST: &str = "v52.0.0"; +pub const TARGET_EDGE: &str = "v38.0.0"; /// Every substrate repo we pin by git tag, and the crate names that come out of /// it. All crates from one repo MUST carry ONE tag. @@ -229,7 +229,11 @@ fn gate_envelope_vocabulary_is_the_one_we_adopted() { // production blobs older than this). `e7135559…` → `4d7054a6…`. // 0.5.218 (persist v49.0.0): one key ADDED, `paths::LISTED` ("listed", the // #912 listing plane); none removed or renamed. `4d7054a6…` → `a6a84cc9…`. - const ADOPTED: &str = "a6a84cc9d5f4d6bd6295cfc78b42bce35145d2bb9ff14391bfe32ab027116a6a"; + // 0.5.218 (persist v51.0.0): four keys ADDED, the rc6 trust-root attach + // inputs (#937/#938: attach_window_secs, witness_cadence_secs, + // witness_quorum, attached_head_digest); none removed or renamed, and the + // server writes none of them. `a6a84cc9…` → `c9558c98…`. + const ADOPTED: &str = "c9558c98bf871e97c2e73c428894e2dc642fcffec4de92da06ab5a9fa69ddca2"; assert_eq!( ENVELOPE_VOCABULARY_SHA256, ADOPTED, "\n\ diff --git a/tests/replication_policy_gate.rs b/tests/replication_policy_gate.rs index 44f92b63..97b5c629 100644 --- a/tests/replication_policy_gate.rs +++ b/tests/replication_policy_gate.rs @@ -182,8 +182,53 @@ const RATIFIED_REPLICATION_POLICY_HASH: &str = /// wider than the room's membership rows). E3 re-checked: `trace:*` still /// serves only to `capability:infra:serve`. Edge's own published /// `SERVE_ADVERTISE_POLICY_HASH`. `d6e4f0df…` → `6fbf0282…`. +/// +/// Re-pinned for 0.5.218 on edge v34.0.0 (CIRISEdge#682, the per-node announce +/// ruling CIRISServer#655), reviewed against the v33.1.0 → v34.0.0 diff of +/// `replication/serve_policy.rs`: ONE row changed — `IdentityOccurrence` and +/// `TransportDestination` leave the `Key` line's `("self_own", "public")` for +/// `("self_own", "owned node: announced → public; unannounced → owner's nodes +/// only")`. An unannounced device's identity rows reach its owner's other +/// nodes and whoever holds a code, never a stranger's listing. Nothing else +/// moved. `6fbf0282…` → `e4c4d625…`, equal to edge's published constant. +/// +/// Re-pinned for 0.5.218 on edge v37.1.0 (CIRISEdge#752, the maintainer's +/// ruling that announced devices are discoverable — CC 5.4.6, CIRISServer#701), +/// reviewed against the v37.0.0 → v37.1.0 diff of `replication/serve_policy.rs`: +/// ONE row changed — the Attestation line's first-contact arm widens from "own +/// allegiance facts" to "own allegiance facts + live federation owner-bindings +/// of nodes in the Key/IdentityOccurrence publish set". A stranger peered only +/// with a relay learns which announced devices belong to whom; consent grants, +/// self-plane rows, `self`-scoped bindings and bindings of unpublished nodes +/// stay withheld, and the Rooted floor still applies. `trace:*` unchanged. +/// `e4c4d625…` → `b86a7042…`, equal to edge's published constant. +/// +/// Re-pinned for 0.5.218 on edge v38.0.0 (persist v52, CIRISPersist#955 — +/// nobody joins a group without their own consent), reviewed against the +/// v37.1.0 → v38.0.0 diff of `replication/serve_policy.rs`. THREE rows changed, +/// all in the direction the consent ruling needs: +/// +/// - **Attestation** (CIRISEdge#756): the first-contact arm also serves the +/// membership ceremony addressed to the peer's own person — a +/// `membership:proposal:v1` naming `owner_of(peer)`, and the invitee's +/// `membership:acceptance|decline:v1` to a proposal held here that +/// `owner_of(peer)` issued — with no Rooted floor on those two (CC rc6 +/// 3.1.3.2: an invitation must be readable by someone who is not yet in the +/// group). Nothing else reaches a stranger; `trace:*` still serves only to +/// `capability:infra:serve` (E3 re-checked). +/// - **Family** and **Community** (CIRISEdge#758 / #762): the group RECORD +/// leaves `("cohort", "public")`. A private group's record now reaches only a +/// peer whose person is a live member or the invitee of a live proposal held +/// here, else it is withheld `group_record_not_member_or_invitee`. PUBLIC +/// groups keep `public` — an `infrastructure` community, the accord / genesis +/// / WA reclaim families — so every node still resolves its trust root and +/// reclaim authority through them. This NARROWS what a stranger learns: a +/// household or a room no longer announces that it exists. +/// +/// `LocationProof` keeps `public` (split out of the old shared line, value +/// unchanged). `b86a7042…` → `e3070d53…`, equal to edge's published constant. const RATIFIED_SERVE_ADVERTISE_POLICY_HASH: &str = - "6fbf0282408148ceea541c9e5f0b6c0726d6e10b41880b7ce9d81339ecb3e7ab"; + "e3070d5327d6518b7ef50b0988f1af92790e2ea4efb9833cc4213a4127fc41b0"; #[test] fn persist_replication_policy_hash_pinned() { diff --git a/tests/safety.rs b/tests/safety.rs index 6aa48049..1a4ac8f4 100644 --- a/tests/safety.rs +++ b/tests/safety.rs @@ -164,12 +164,25 @@ async fn register_actor( /// The shared body: the RECORD lands at `record_key_id`, the KEY MATERIAL is the /// one `alias` seeds. They are the same string for `register_party` and differ /// for `register_actor`, which is the entire difference between the two. +/// Which seed alias each registered record key was minted from — so a +/// fixture can re-open the signer of a MEMBER it only knows by key id (an +/// actor's record id is derived, not its alias). +fn seed_alias_of() -> &'static std::sync::Mutex> { + static MAP: std::sync::OnceLock>> = + std::sync::OnceLock::new(); + MAP.get_or_init(Default::default) +} + async fn register_party_at( engine: &Engine, record_key_id: &str, alias: &str, identity_type_str: &str, ) -> LocalSigner { + seed_alias_of() + .lock() + .expect("seed alias map") + .insert(record_key_id.to_owned(), alias.to_owned()); let signer = party_signer(alias); let key_id = record_key_id; let key_id_for_seeds = alias; @@ -489,6 +502,27 @@ async fn put_community(engine: &Engine, community_id: &str, members: &[(&str, &s .sign_hybrid(&canonical) .await .expect("sign community declaration"); + // persist v52 Q1 (CIRISPersist#955): a founding record seats only the + // members who SIGNED it, so every listed member co-signs with the key the + // fixture registered for them. + let mut cosignatures = Vec::new(); + for (k, _) in members { + let alias = seed_alias_of() + .lock() + .expect("seed alias map") + .get(*k) + .cloned() + .unwrap_or_else(|| (*k).to_owned()); + let c = party_signer(&alias) + .sign_hybrid(&canonical) + .await + .expect("a founding member co-signs"); + cosignatures.push(ciris_persist::federation::types::RosterCosignature { + authority_key_id: (*k).to_string(), + scrub_signature_classical: BASE64.encode(&c.classical.signature), + scrub_signature_pqc: Some(BASE64.encode(&c.pqc.signature)), + }); + } engine .federation_directory() .put_community(SignedCommunity { @@ -497,6 +531,8 @@ async fn put_community(engine: &Engine, community_id: &str, members: &[(&str, &s scrub_signature_classical: BASE64.encode(&sig.classical.signature), scrub_signature_pqc: Some(BASE64.encode(&sig.pqc.signature)), supersede_proof: None, + cosignatures, + lineage: Vec::new(), }) .await .expect("put_community"); diff --git a/tests/self_node_release.rs b/tests/self_node_release.rs index 2622020f..88f967d6 100644 --- a/tests/self_node_release.rs +++ b/tests/self_node_release.rs @@ -525,3 +525,164 @@ async fn only_the_announced_nodes_are_public() { assert!(mine.contains(k), "{k} missing from the owner's view: {v}"); } } + +// ─── 0.5.218: release is an EVICTION (CSD-037) ────────────────────────────── + +/// The content-KEM occurrence a claimed machine provisions for its owner, +/// under its node key, with real enc keys (so it is a wrap recipient). +async fn node_occurrence(p: &Person, node_key_id: &str, tag: u8) { + p.engine + .federation_directory() + .put_identity_occurrence_local(IdentityOccurrence { + identity_key_id: p.key().to_owned(), + occurrence_key_id: node_key_id.to_owned(), + device_class: "server".to_owned(), + hardware_attestation: None, + asserted_at: chrono::Utc::now(), + valid_until: None, + encryption_pubkeys: Some( + ciris_server::identity::derive_self_enc_pubkeys(&[tag; 32]).expect("enc keys"), + ), + transport_binding: None, + persist_row_hash: String::new(), + }) + .await + .expect("the node's content occurrence"); +} + +async fn active_occurrences(p: &Person) -> Vec { + p.engine + .federation_directory() + .list_identity_occurrences_active(p.key()) + .await + .expect("active occurrences") + .into_iter() + .map(|o| o.occurrence_key_id) + .collect() +} + +/// **(ii) A release revokes the released node's occurrence** — signed, on the +/// surface replication re-publishes — as well as withdrawing its binding. +/// Before 0.5.218 the binding went and the occurrence stayed, so the machine +/// kept being wrapped every new self file. +#[tokio::test] +async fn releasing_a_node_also_revokes_its_occurrence_signed() { + let alice = Person::new("rel-occ").await; + let n2 = second_node(&alice, 0xE8).await; + node_occurrence(&alice, &n2, 0xE9).await; + assert!( + active_occurrences(&alice).await.contains(&n2), + "precondition" + ); + + let (st, v) = alice + .as_owner("POST", &format!("/v1/self/nodes/{n2}/release"), None) + .await; + assert_eq!(st.as_u16(), 200, "{v}"); + assert_eq!(v["released"], true); + assert_eq!(v["withdrawn"].as_array().map(Vec::len), Some(1), "{v}"); + let revoked = v["occurrences_revoked"] + .as_array() + .expect("occurrences_revoked"); + assert_eq!(revoked.len(), 1, "{v}"); + assert_eq!(revoked[0]["occurrence_key_id"], n2.as_str()); + assert_eq!(revoked[0]["attesting_key_id"], alice.key()); + assert!(v["failed"].as_array().is_some_and(Vec::is_empty), "{v}"); + assert!( + v["history"] + .as_str() + .is_some_and(|h| h.contains("Already-shared history stays readable")), + "{v}" + ); + assert!(!active_occurrences(&alice).await.contains(&n2)); + let signed = alice + .engine + .federation_directory() + .list_signed_identity_occurrence_revocations_for(alice.key()) + .await + .expect("signed revocations"); + assert!( + signed + .iter() + .any(|r| r.identity_occurrence_revocation.occurrence_key_id == n2 + && r.attesting_key_id == alice.key()), + "the node's revocation is SIGNED (replicable), by the owner: {signed:?}" + ); +} + +/// **(iii) After an eviction, a new self file carries no wrap to the evicted +/// device — and an OLD one still does** (the honesty half, CC 3.3.6.1: no DEK +/// rotation, no re-encryption of what was already shared). +#[tokio::test] +async fn after_eviction_new_self_files_are_not_wrapped_to_the_evicted_device() { + let alice = Person::new("rel-wrap").await; + let here = provision_node_occurrence(&alice).await; + let n2 = second_node(&alice, 0xEC).await; + node_occurrence(&alice, &n2, 0xED).await; + + let before = alice + .engine + .put_blob_scoped( + "self", + Some(alice.key()), + b"before the theft", + Some("text/plain"), + None, + ) + .await + .expect("a self file before the eviction"); + assert!( + before.granted.contains(&n2) && before.granted.contains(&here), + "precondition: both devices are wrapped: {:?}", + before.granted + ); + + let (st, v) = alice + .as_owner("POST", &format!("/v1/self/nodes/{n2}/release"), None) + .await; + assert_eq!(st.as_u16(), 200, "{v}"); + + let after = alice + .engine + .put_blob_scoped( + "self", + Some(alice.key()), + b"after the theft", + Some("text/plain"), + None, + ) + .await + .expect("a self file after the eviction"); + assert!( + !after.granted.contains(&n2) && !after.excluded.contains(&n2), + "a NEW self file names no wrap to the evicted device: {:?} / {:?}", + after.granted, + after.excluded + ); + assert!( + after.granted.contains(&here), + "the remaining device is: {:?}", + after.granted + ); + use ciris_persist::federation::blobs::BlobStorage as _; + let backend = alice.engine.sqlite_backend().expect("sqlite"); + let new_recipients = backend + .list_at_rest_grant_recipients(&after.at_rest_sha256) + .await + .expect("grant recipients"); + assert!( + !new_recipients.contains(&n2), + "no key_grant to the evicted occurrence on the new file: {new_recipients:?}" + ); + // What eviction does NOT do, pinned so nobody claims otherwise: the grant + // on the file shared BEFORE stays. + let old_recipients = backend + .list_at_rest_grant_recipients(&before.at_rest_sha256) + .await + .expect("grant recipients"); + assert!( + old_recipients.contains(&n2), + "already-shared history stays readable by the evicted device (no DEK rotation): \ + {old_recipients:?}" + ); +} diff --git a/tests/substrate_contract_gate.rs b/tests/substrate_contract_gate.rs index 0f825846..8ee96bce 100644 --- a/tests/substrate_contract_gate.rs +++ b/tests/substrate_contract_gate.rs @@ -112,8 +112,15 @@ const RATIFIED_TRANSFORM_ALGEBRA_HASH: &str = /// `paths::LISTED` (`"listed"`, the #912 listing plane); nothing removed or /// renamed, and every key this server spells through `paths::*` is unchanged. /// `4d7054a6…` → `a6a84cc9…`, persist's own pin. +/// +/// Re-pinned for 0.5.218 on persist v51.0.0, reviewed: FOUR keys added, all +/// CC 3.2 rc6 trust-root attach inputs (CIRISPersist#937/#938): +/// `attach_window_secs`, `witness_cadence_secs`, `witness_quorum` on the +/// charter and `attached_head_digest` on the acceptance edge. Nothing removed +/// or renamed; this server writes none of the four. `a6a84cc9…` → +/// `c9558c98…`, equal to persist's published `ENVELOPE_VOCABULARY_SHA256`. const RATIFIED_ENVELOPE_VOCABULARY_SHA256: &str = - "a6a84cc9d5f4d6bd6295cfc78b42bce35145d2bb9ff14391bfe32ab027116a6a"; + "c9558c98bf871e97c2e73c428894e2dc642fcffec4de92da06ab5a9fa69ddca2"; // ─────────────────────────── persist: trace-summary extraction ───────────── diff --git a/tests/support/drive_fixture.rs b/tests/support/drive_fixture.rs index 3bb90683..867bec1f 100644 --- a/tests/support/drive_fixture.rs +++ b/tests/support/drive_fixture.rs @@ -341,3 +341,61 @@ pub async fn status_json(resp: reqwest::Response) -> (u16, serde_json::Value) { let json = serde_json::from_str(&text).unwrap_or(serde_json::Value::String(text)); (status, json) } + +/// A seeded key's hybrid signature over `canonical`, as a roster co-signature +/// (`seed_key`'s deterministic Ed25519 / ML-DSA-65 halves). +#[allow(dead_code)] +pub async fn seeded_cosignature( + key_id: &str, + ed_seed: u8, + pqc_seed: u8, + canonical: &[u8], +) -> ciris_persist::federation::types::RosterCosignature { + use ed25519_dalek::Signer as _; + let ed = SigningKey::from_bytes(&[ed_seed; 32]); + let mldsa = MlDsa65SoftwareSigner::from_seed_bytes(&[pqc_seed; 32], format!("{key_id}-pqc")) + .expect("ML-DSA-65 seed"); + let ed_sig = ed.sign(canonical).to_bytes(); + let mut bound = Vec::with_capacity(canonical.len() + ed_sig.len()); + bound.extend_from_slice(canonical); + bound.extend_from_slice(&ed_sig); + let pqc_sig = mldsa.sign(&bound).await.expect("ml-dsa sign"); + ciris_persist::federation::types::RosterCosignature { + authority_key_id: key_id.to_owned(), + scrub_signature_classical: BASE64.encode(ed_sig), + scrub_signature_pqc: Some(BASE64.encode(pqc_sig)), + } +} + +/// A `Community` record authored by this NODE (signing for its owner) and +/// CO-SIGNED by every other listed member's seeded key — persist v52 +/// (CIRISPersist#955, Q1) seats a founding member only if they signed the +/// founding record, so a fixture that founds a room WITH others must carry +/// each of their signatures. `cosigners` are `(key_id, ed_seed, pqc_seed)` +/// exactly as `seed_key` registered them. +#[allow(dead_code)] +pub async fn put_community_cosigned( + engine: &Engine, + community: ciris_persist::federation::types::Community, + cosigners: &[(&str, u8, u8)], +) -> Result<(), ciris_persist::federation::Error> { + let canonical = + ceg_produce_canonicalize(&community.signing_envelope()).expect("canonicalize community"); + let sig = engine.sign_hybrid(&canonical).await.expect("node signs"); + let mut cosignatures = Vec::new(); + for (k, ed, pqc) in cosigners { + cosignatures.push(seeded_cosignature(k, *ed, *pqc, &canonical).await); + } + engine + .federation_directory() + .put_community(ciris_persist::federation::types::SignedCommunity { + community, + authority_key_id: engine.local_derived_key_id().await.expect("node id"), + scrub_signature_classical: BASE64.encode(&sig.classical.signature), + scrub_signature_pqc: Some(BASE64.encode(&sig.pqc.signature)), + supersede_proof: None, + cosignatures, + lineage: Vec::new(), + }) + .await +} diff --git a/tests/support/owned_node.rs b/tests/support/owned_node.rs index 5b4a863d..5896c205 100644 --- a/tests/support/owned_node.rs +++ b/tests/support/owned_node.rs @@ -114,6 +114,20 @@ impl OwnerIdentity { .0 } + /// The owner's EDGE signer (the same custody) — the type edge's + /// `membership` producers sign with (a proposal, an acceptance). + #[allow(dead_code)] + pub async fn edge_signer(&self) -> ciris_edge::identity::LocalSigner { + ciris_server::identity::hardware_user_signers( + UserIdentityBackend::Software, + &self.alias, + self.seed_dir.clone(), + ) + .await + .expect("re-open the owner's minted fed-ID") + .1 + } + /// The owner's key record, as a registration any node can hold. pub fn key_record(&self) -> SignedKeyRecord { user_record( @@ -267,6 +281,10 @@ impl Person { /// The routes under test, over THIS node. pub fn router(&self) -> Router { ciris_server::family_api::router(Arc::clone(&self.engine), self.owner.seed_dir.clone()) + .merge(ciris_server::membership_invites::router( + Arc::clone(&self.engine), + self.owner.seed_dir.clone(), + )) .merge(ciris_server::self_devices::router( Arc::clone(&self.engine), self.owner.seed_dir.clone(), @@ -368,8 +386,20 @@ impl Person { dst.put_family(served.family).await.unwrap_or_else(|e| { panic!("{} admits {id} from {}: {e}", self.name, from.name) }); + } else { + // persist v52 (#956): a self-leave or a quorum dissolve is an + // AMENDMENT of a record this node already holds — offered, + // re-verified against this node's roster, applied. The + // identical record is a no-op; a proof-less differing one is + // refused (#758), which is that row's answer, not this helper's. + let _ = dst.put_family(served.family).await; } } + // persist v52 (CIRISPersist#955): a widening is admitted on the + // replicated apply only beside the member's acceptance of a live + // proposal, so the membership ceremony crosses FIRST, as replication + // carries it (the proposal to the invitee's nodes, the reply back). + self.receive_membership_rows_from(from).await; // persist v49.0.0 (#910): a member added after the household was // created is its own signed row on the family WIDENING plane, not a // grown record, so a copy that stops at the record leaves the new @@ -393,6 +423,69 @@ impl Person { } } + /// Carry every `membership:{proposal,acceptance,decline}:v1` row `from` + /// holds onto this node through the ordinary attestation door — what edge + /// v38's serve arms route (a proposal to the invitee's nodes, a reply back + /// to the proposer's). Already-held and refused rows are skipped: each is + /// persist's own answer for that row. + pub async fn receive_membership_rows_from(&self, from: &Person) { + let src = from.engine.federation_directory(); + let dst = self.engine.federation_directory(); + let mut since = None; + loop { + let page = src + .list_attestations_since(since.clone(), 512) + .await + .expect("list attestations"); + let full = page.len() == 512; + since = page + .last() + .map(ciris_persist::federation::types::ServedAttestation::resume_pair); + for served in page { + if ciris_edge::membership::is_membership_row(&served.attestation) { + let _ = dst + .put_attestation(ciris_persist::federation::SignedAttestation { + attestation: served.attestation, + }) + .await; + } + } + if !full { + break; + } + } + } + + /// Offer every SIGNED family record `from` serves to this node's replicated + /// `put_family` door — INCLUDING ids this node already holds, which is + /// how an AMENDMENT (a superseded record carrying its quorum proof, + /// persist v49 #910.5) arrives. Returns, per family id, what the door + /// answered: `Ok(())` when it inserted, re-verified-and-applied, or found + /// the identical record; `Err(text)` when it refused (a differing record + /// without a proof is the #758 conflict). Records only — no roster rows, + /// so the amendment is judged against this node's roster as it stands. + pub async fn receive_family_records_from( + &self, + from: &Person, + ) -> Vec<(String, Result<(), String>)> { + let src = from.engine.federation_directory(); + let dst = self.engine.federation_directory(); + let mut out = Vec::new(); + for served in src + .list_signed_families_since(None, u32::MAX) + .await + .expect("list signed families") + { + let id = served.family.family.family_key_id.clone(); + let r = dst + .put_family(served.family) + .await + .map_err(|e| format!("{e:#}")); + out.push((id, r)); + } + out + } + /// One request through the router, as this node's owner (or `bearer`). pub async fn call( &self, @@ -485,6 +578,171 @@ pub async fn bind_self_scoped(engine: &Engine, owner: &LocalSigner, node_key_id: .expect("apply the self-scoped owner-binding"); } +// ─── TEST-ONLY roster fixtures — the consent-to-join BYPASS ────────────────── +// +// Since 0.5.218 every roster-growing HTTP door refuses +// `membership.consent_required` (the maintainer's ruling of 2026-09-30; +// CIRISConstitution#133, CIRISPersist#955, `FSD/MEMBERSHIP_INVITES.md`): nobody +// joins a household without their own acceptance, and the acceptance cannot be +// recorded until persist v52. Tests whose PURPOSE is downstream of a +// multi-member roster (quorum remove / role / leave / dissolve) still need one. +// These helpers write it the way the pre-0.5.218 routes did — owner-signed rows +// through persist's own SIGNED doors — WITHOUT going through the server. +// +// They exist only here, under `tests/`. There is no production bypass, and a +// test that uses one is asserting what happens AFTER a join, never that a join +// is allowed. + +impl Person { + /// TEST-ONLY (bypasses the consent-to-join door): a household founded by + /// this person with `others` already on its founding record, signed by this + /// person's pen through persist's replicated `put_family`. Every key must + /// already be registered on this node (`knows`). + pub async fn test_only_family_founded_with( + &self, + name: &str, + others: &[&Person], + consensus_protocol: &str, + ) -> String { + use ciris_persist::federation::types::{Family, FamilyMember, SignedFamily}; + let at = chrono::Utc::now(); + let mut members = vec![FamilyMember { + key_id: self.key().to_owned(), + joined_at: at, + role: Some("founder".to_owned()), + }]; + members.extend(others.iter().map(|p| FamilyMember { + key_id: p.key().to_owned(), + joined_at: at, + role: Some("member".to_owned()), + })); + let family_id = format!("family:v1:{}", uuid::Uuid::new_v4().simple()); + let family = Family { + family_key_id: family_id.clone(), + family_name: name.to_owned(), + members, + founded_at: at, + consensus_protocol: consensus_protocol.to_owned(), + consensus_protocol_entrenched: false, + dissolved_at: None, + persist_row_hash: String::new(), + }; + let canonical = + ceg_produce_canonicalize(&family.signing_envelope()).expect("canonicalize family"); + let sig = self + .owner + .signer() + .await + .sign_hybrid(&canonical) + .await + .expect("owner signs the family"); + // persist v52 Q1 (CIRISPersist#955): a founding member is seated only + // if they SIGNED the founding record. The server has no founding-cosign + // flow, so this fixture plays each member co-signing with their pen. + let mut cosignatures = Vec::new(); + for p in others { + let c = p + .owner + .signer() + .await + .sign_hybrid(&canonical) + .await + .expect("a founding member co-signs"); + cosignatures.push(ciris_persist::federation::types::RosterCosignature { + authority_key_id: p.key().to_owned(), + scrub_signature_classical: BASE64.encode(&c.classical.signature), + scrub_signature_pqc: Some(BASE64.encode(&c.pqc.signature)), + }); + } + self.engine + .federation_directory() + .put_family(SignedFamily { + family, + authority_key_id: self.key().to_owned(), + scrub_signature_classical: BASE64.encode(&sig.classical.signature), + scrub_signature_pqc: Some(BASE64.encode(&sig.pqc.signature)), + supersede_proof: None, + cosignatures, + }) + .await + .expect("TEST-ONLY: put the multi-member family directly"); + family_id + } + + /// TEST-ONLY (bypasses the consent-to-join door): widen `family_id` by + /// `member_key_id` at `role`, signed by this person's pen — the + /// `FamilyMembershipWidening` row `POST /v1/families/{id}/members` wrote + /// before 0.5.218, through persist's own `add_member`. + pub async fn test_only_widen_family(&self, family_id: &str, who: &Person, role: &str) { + use ciris_persist::federation::cohort::{AdmitSpec, Cohort, RosterMember}; + use ciris_persist::federation::types::{FamilyMember, FamilyMembershipWidening}; + let member_key_id = who.key(); + // persist v52 (CIRISPersist#955): the member's CONSENT first — this + // person's proposal and `who`'s own acceptance, both through edge's + // producers and the ordinary door on this node. Only the widening that + // follows is the bypass (it skips the HTTP flow, not the rule). + let dir = self.engine.federation_directory(); + let proposal = ciris_edge::membership::propose( + dir.as_ref(), + ciris_edge::membership::GroupScope::Family, + family_id, + member_key_id, + Some(role), + chrono::Utc::now() + chrono::Duration::days(1), + &self.owner.edge_signer().await, + ) + .await + .expect("TEST-ONLY: the proposal"); + ciris_edge::membership::reply( + dir.as_ref(), + &proposal.attestation_id, + true, + &who.owner.edge_signer().await, + ) + .await + .expect("TEST-ONLY: the member's own acceptance"); + tokio::time::sleep(std::time::Duration::from_millis(5)).await; + let member = FamilyMember { + key_id: member_key_id.to_owned(), + joined_at: chrono::Utc::now(), + role: Some(role.to_owned()), + }; + let row = FamilyMembershipWidening { + family_key_id: family_id.to_owned(), + member_key_id: member.key_id.clone(), + joined_at: member.joined_at, + effective_at: member.joined_at, + role: member.role.clone(), + persist_row_hash: String::new(), + }; + let canonical = + ceg_produce_canonicalize(&row.signing_envelope()).expect("canonicalize widening"); + let sig = self + .owner + .signer() + .await + .sign_hybrid(&canonical) + .await + .expect("owner signs the widening"); + let spec = AdmitSpec { + authority_key_id: self.key().to_owned(), + scrub_signature_classical: BASE64.encode(&sig.classical.signature), + scrub_signature_pqc: Some(BASE64.encode(&sig.pqc.signature)), + cosignatures: Vec::new(), + }; + let added = self + .engine + .federation_directory() + .add_member(Cohort::Family, family_id, RosterMember::from(member), &spec) + .await + .expect("TEST-ONLY: widen the family directly"); + assert!( + added, + "TEST-ONLY widening admitted nothing for {member_key_id}" + ); + } +} + /// An active `wa_cert` + a bound session bearer on `engine`. pub async fn mint_session(engine: &Engine, wa_id: &str, role: WaRole) -> String { let now = chrono::Utc::now(); diff --git a/tests/support/revocation.rs b/tests/support/revocation.rs index 271f9994..8581daa7 100644 --- a/tests/support/revocation.rs +++ b/tests/support/revocation.rs @@ -418,9 +418,26 @@ pub async fn revoke( // so it is set here — truncated, because a sub-microsecond bound is refused // outright rather than rounded (postgres TIMESTAMPTZ, CIRISPersist#659). let revoked_after = revoked_after.map(trunc); + // persist v52 (CIRISPersist#784): the subject is the SHA-256 of the + // revoked key's RAW Ed25519 key, signed into the binding; the label is + // optional. Read from the held record, as production does. + let subject = { + let rec = engine + .federation_directory() + .lookup_public_key(revoked_key_id) + .await + .expect("lookup_public_key") + .unwrap_or_else(|| panic!("revoked key {revoked_key_id} is not held")); + ciris_persist::federation::key_digest::Sha256Ed25519Raw::from_pubkey_base64( + &rec.pubkey_ed25519_base64, + ) + .expect("held pubkey decodes") + .to_hex() + }; let mut row = Revocation { revocation_id: format!("rev-{revoked_key_id}"), - revoked_key_id: revoked_key_id.to_string(), + revoked_key_id: Some(revoked_key_id.to_string()), + revoked_key_sha256_ed25519_raw: subject, revoking_key_id: revoking.key_id().to_string(), reason: None, revoked_at: effective_at, diff --git a/tests/the_kick_carries_the_rows.rs b/tests/the_kick_carries_the_rows.rs index 020a7118..b60d8847 100644 --- a/tests/the_kick_carries_the_rows.rs +++ b/tests/the_kick_carries_the_rows.rs @@ -116,10 +116,18 @@ fn the_publish_own_poll_carries_a_gain_and_retries_a_lost_kick() { reports it dispatched", ), ( - "let wait = if owed { 1 } else { 30 };", + // 0.5.218 (CIRISServer#701): the announced-relay set shares this + // loop and carries its own debt under the same rule, so the prompt + // retry covers either debt. + "let wait = if owed || relay_owed { 1 } else { 30 };", "an owed kick must be retried PROMPTLY — sleeping the full cadence before the \ retry leaves exactly the delay this exists to remove", ), + ( + "if relay_owed && kick_replication(", + "the announced-relay debt is paid by an actual kick, cleared only when it \ + reports it dispatched — the same rule as the owner's", + ), ] { assert!( src.contains(needle), diff --git a/tests/trace_round_e2e.rs b/tests/trace_round_e2e.rs index c3341e8d..79bb50e4 100644 --- a/tests/trace_round_e2e.rs +++ b/tests/trace_round_e2e.rs @@ -519,7 +519,7 @@ struct RoundSide<'a> { key_id: &'a str, } -fn drive_round(initiator: RoundSide<'_>, responder: RoundSide<'_>) -> usize { +async fn drive_round(initiator: RoundSide<'_>, responder: RoundSide<'_>) -> usize { let RoundSide { session: initiator, provider: init_provider, @@ -534,7 +534,8 @@ fn drive_round(initiator: RoundSide<'_>, responder: RoundSide<'_>) -> usize { } = responder; let mut admitted_total = 0usize; - let mut to_responder: Vec = match initiator.start_round(init_provider) { + let mut to_responder: Vec = match initiator.start_round(init_provider).await + { ReplicationOutcome::Send(msgs) => msgs, ReplicationOutcome::SendAndComplete { msgs, .. } => msgs, other => panic!("initiator start_round produced no messages: {other:?}"), @@ -548,7 +549,10 @@ fn drive_round(initiator: RoundSide<'_>, responder: RoundSide<'_>) -> usize { break; } for msg in std::mem::take(&mut to_responder) { - match responder.on_message(msg, resp_provider, resp_applier, Some(init_peer)) { + match responder + .on_message(msg, resp_provider, resp_applier, Some(init_peer)) + .await + { ReplicationOutcome::Send(msgs) => to_initiator.extend(msgs), ReplicationOutcome::SendAndComplete { msgs, .. } => to_initiator.extend(msgs), ReplicationOutcome::Applied { admitted, .. } => admitted_total += admitted, @@ -556,7 +560,10 @@ fn drive_round(initiator: RoundSide<'_>, responder: RoundSide<'_>) -> usize { } } for msg in std::mem::take(&mut to_initiator) { - match initiator.on_message(msg, init_provider, init_applier, Some(resp_peer)) { + match initiator + .on_message(msg, init_provider, init_applier, Some(resp_peer)) + .await + { ReplicationOutcome::Send(msgs) => to_responder.extend(msgs), ReplicationOutcome::SendAndComplete { msgs, .. } => to_responder.extend(msgs), ReplicationOutcome::Applied { admitted, .. } => admitted_total += admitted, @@ -739,7 +746,7 @@ async fn agent_trace_reaches_canonical_over_a_real_round() { let mut agent_applier = MutableDirectoryStateAdapter::new(agent_bridge); let mut canon_applier = MutableDirectoryStateAdapter::new(canon_bridge); - let offered_refs = agent_provider.local_refs(EnvelopeKind::Attestation); + let offered_refs = agent_provider.local_refs(EnvelopeKind::Attestation).await; eprintln!( "AGENT OFFERS {} ref(s) to the canonical", offered_refs.len() @@ -860,7 +867,8 @@ async fn agent_trace_reaches_canonical_over_a_real_round() { applier: &mut canon_applier, key_id: &canonical.key_id, }, - ); + ) + .await; } assert!( @@ -980,7 +988,7 @@ async fn without_a_grant_the_producer_offers_nothing() { let agent_bridge = bridge(&agent); let provider = DirectoryStateAdapter::new(agent_bridge).with_peer(canonical.key_id.clone()); - let refs = provider.local_refs(EnvelopeKind::Attestation); + let refs = provider.local_refs(EnvelopeKind::Attestation).await; assert!( refs.is_empty(), "a producer with NO consent:replication grant toward the peer must advertise \ diff --git a/tools/check_server_localization.py b/tools/check_server_localization.py index a673061c..a349a1af 100644 --- a/tools/check_server_localization.py +++ b/tools/check_server_localization.py @@ -752,28 +752,7 @@ def _server_message_texts_all(root: Path) -> Dict[str, List[str]]: # `drive.not_fetched` / `drive.not_granted` matter most: they are emitted # in NORMAL use (a file whose bytes are on another device), not only on # error, so they are the two a reader meets first. - "drive.author_signer_unavailable", "drive.bad_base64", - "drive.community_id_required", - "drive.family_id_required", - "drive.listing_failed", - "drive.no_node_key", - "drive.not_fetched", - "drive.not_granted", - "drive.not_a_member", - "drive.not_in_room", - "drive.owner_session_required", - "drive.publish_failed", - "drive.readable_by_nobody", - "drive.store_unavailable", - "drive.too_large", - "drive.unknown_cohort", - "drive.unopened", - "notes.author_signer_unavailable", - "notes.empty", - "notes.listing_failed", - "notes.no_node_key", - "notes.owner_session_required", # THE COMMUNITY PLANE (CIRISServer#594 / 0.5.216, `FSD/ROSTER_AND_DRIVE_CRUD.md` # §4) — the same case as the drive plane above: a new surface whose refusals # land before the `ciris-client` bundle that carries them. The ids and their @@ -781,22 +760,6 @@ def _server_message_texts_all(root: Path) -> Dict[str, List[str]]: # delegate_may_not_author` and `community.delegation_denied` are emitted # through `contacts_chat::require_verb`, which this guard does not scrape, # exactly like their `chat.*` twins — so they are not listed here.) - "community.already_member", - "community.author_signer_unavailable", - "community.bad_consensus_protocol", - "community.bad_tier", - "community.change_stale", - "community.last_founder", - "community.malformed_body", - "community.name_empty", - "community.not_a_contact", - "community.not_a_member", - "community.not_authorized", - "community.not_found", - "community.pair_room_fixed", - "community.quorum_pending", - "community.store_unavailable", - "community.write_failed", # THE DRIVE CRUD SURFACE (0.5.216, `FSD/ROSTER_AND_DRIVE_CRUD.md` §5) — the # same plane, the same queue: the ids the FSD names for replace / rename / # withdraw / move / range / notes edit, requested of the client beside the @@ -821,6 +784,13 @@ def _server_message_texts_all(root: Path) -> Dict[str, List[str]]: "drive.too_large_for_whole_read", "drive.withdraw_failed", "drive.withdrawn", + # THE STREAMED UPLOAD (0.5.218, edge v36.1.0 `files::publish_stream`): a + # multipart body whose file part disagrees with its declared `size`, and a + # form field sent after the file part (the file streams to the seal, so + # every field must precede it). Queued for the client bundle beside the + # drive CRUD ids (CIRISClient#78). + "drive.declared_length_mismatch", + "drive.field_after_file", "notes.delegate_may_not_author", "notes.not_found", # THE HOUSEHOLD AND SELF-DEVICE PLANES (CIRISServer#627 / 0.5.216, @@ -829,27 +799,52 @@ def _server_message_texts_all(root: Path) -> Dict[str, List[str]]: # release carrying their keys. The FSD lists the ids with their English so # the client can request them in one batch; these come off this list, and # the pyproject floor goes up, when a bundle carries them. - "family.already_member", - "family.author_signer_unavailable", - "family.bad_change", - "family.bad_consensus_protocol", - "family.bad_request", - "family.bad_role", - "family.delegate_may_not_author", - "family.last_founder", - "family.name_empty", - "family.not_a_member", - "family.not_authorized", - "family.not_found", - "family.owner_session_required", - "family.quorum_pending", - "family.store_unavailable", - "family.unknown_member_key", # ANNOUNCE ANOTHER DEVICE (0.5.218, CIRISServer#678): the per-node announce # made from the device holding the pen. Queued for the client bundle beside # the self-device ids (CIRISClient#78). "self.announce_not_your_node", "self.announce_refused", + # EVICTING A DEVICE (0.5.218, CSD-037): release and occurrence/revoke became + # ONE signed act, and a part of it that did not complete is named by this + # id. CONSENT TO JOIN (0.5.218, the maintainer's ruling of 2026-09-30, + # CIRISConstitution#133 / CIRISPersist#955): every roster-growing door of + # the family and community surfaces refuses with ONE shared id until the + # invite flow ships. Both queued for the client bundle with their English + # (CIRISClient#78). + "self.evict_incomplete", + # (`membership.consent_required` retired at persist v52 / edge v38: the + # interim door it named opened as the invite flow below.) + # CONSENT TO JOIN, BUILT (0.5.218 at persist v52.0.0 / edge v38.0.0, + # CIRISPersist#955, `FSD/MEMBERSHIP_INVITES.md`): the invite flow's + # refusals — persist's eight consent rules each named by one id, the + # invitee's inbox gate, and the invitation's own lifecycle + # (`src/membership_invites.rs`, one function per id). Same queue as the + # planes above, requested of the client with their English (CIRISClient#78). + "membership.acceptance_mismatch", + "membership.already_answered", + "membership.awaiting_acceptance", + "membership.bad_expiry", + "membership.declined", + "membership.delegate_may_not_answer", + "membership.founding_member_unsigned", + "membership.invite_closed", + "membership.invite_expired", + "membership.invite_not_found", + "membership.invite_not_here_yet", + "membership.not_the_invitee", + "membership.not_the_proposer", + "membership.owner_session_required", + "membership.refused", + "membership.signer_unavailable", + "membership.store_unavailable", + "membership.supersede_cannot_add", + # ONE DEVICE HANDLES EACH EXCHANGE (0.5.218, CC 3.1.3.1, + # FSD/SESSION_CLAIMS.md §5): `GET /v1/self/sessions` says which of the + # person's devices is answering — two states and one refusal, queued for + # the client bundle beside the self-device ids (CIRISClient#78). + "self.sessions_unavailable", + "session.state.handled_elsewhere", + "session.state.handled_here", # THE CLAIM, NAMED (0.5.218, CIRISServer#678 client review: "refusals come # back as prose"): claim-remote on the approving device, and setup/root on # the device being claimed. Same queue (CIRISClient#78). @@ -874,35 +869,30 @@ def _server_message_texts_all(root: Path) -> Dict[str, List[str]]: "auth.claim.owner_binding_rejected", "auth.claim.store_unavailable", "auth.claim.wrong_node", - "self.author_signer_unavailable", - "self.bad_request", - "self.delegate_may_not_author", - "self.label_empty", - "self.not_your_device", - "self.not_your_node", - "self.owner_session_required", - "self.release_incomplete", - "self.release_self_requires_force", - "self.store_unavailable", + # WHERE EACH FILE IS (0.5.218, `FSD/FILE_CUSTODY.md`): the custody view's + # partial-answer reasons — NOT refusals; each rides a 200 in `why[]` and + # says what the view cannot know (an inline file's missing receipt, CC 5.2's + # uncountable self/family copies, a receipt that proves delivery only). + # Same queue as the planes above (CIRISClient#78); the English is in the FSD. + # (`custody.inline_no_receipt` and `custody.receipt_time_unknown` retired at + # edge v38.0.0 / persist v52: every file is receiptable and a receipt says + # when — CIRISPersist#953; no longer emitted, so no longer debt.) + "custody.copies_unobservable_by_design", + "custody.receipt_is_delivery_not_holding", + "custody.receipts_admitted_on_author_device", + "custody.receipt_signer_not_your_device", + "custody.receipts_unreadable", + "custody.commons_readable_by_holders", + # The maintainer's ruling on #704 ("no copy here is a receipt"): a device + # with the row and no bytes answers `holds: none` for itself; the other + # devices' "no copy" reports wait on persist's within-cohort custody + # acknowledgements (CIRISConstitution#130). + "custody.no_copy_reports_pending", + "custody.no_copy_here", # THE CONTACT FLOW (0.5.218) — same queue as the planes above, requested of # the client with their English (CIRISClient#78): the person's contact code # (CIRISServer#673), withdrawing consent (#657), and an explicit TPM device # custody this host cannot honour (#639). - "self.contact_code_key_not_derived", - "self.contact_code_no_pqc_half", - "self.contact_code_not_a_person", - "self.contact_code_owner_key_absent", - "self.contact_code_unencodable", - "self.node_not_announced", - "self.associate.hardware_custody_unavailable", - "consent.author_signer_unavailable", - "consent.delegate_may_not_withdraw", - "consent.grant_not_live", - "consent.grant_not_owner_authored", - "consent.malformed_body", - "consent.store_unavailable", - "consent.withdraw_failed", - "contacts.not_a_contact", "accord.duty.assemble", "accord.duty.holder_identity_mismatch", "accord.duty.no_duty",