diff --git a/README.md b/README.md index 77a2476b6..59a79bdde 100644 --- a/README.md +++ b/README.md @@ -1539,7 +1539,16 @@ bus.listen( ``` Retryable failures (e.g. transient `NotFound`) are nacked for redelivery; the runner -never silently acks a handler error. +never silently acks a handler error. NATS JetStream NAKs back off by delivery count +(50 ms doubling to 5 s by default) so a repeatedly failing message cannot monopolize +a consumer; retries stay unlimited. A deterministic, durably recorded rejection is a +permanent error, not a retryable one. + +A service can place independent route bundles in named delivery lanes +(`Service::lane(name, routes)`). Each lane keeps broker order and runs +concurrently with the others, and a delivery is settled only after every lane +finished it. See [consumer delivery lanes](docs/consumer-delivery-lanes.md) for +the contract and when a route may leave the default lane. ### Transport boundaries (producer vs consumer) diff --git a/distributed_cli/src/contracts/tests.rs b/distributed_cli/src/contracts/tests.rs index 1c05512d7..2edef446b 100644 --- a/distributed_cli/src/contracts/tests.rs +++ b/distributed_cli/src/contracts/tests.rs @@ -1259,7 +1259,7 @@ fn migration_inventory_is_deterministic_and_preserves_runtime_order() { .iter() .map(|migration| migration.version) .collect::>(); - assert_eq!(versions, vec![1, 2, 3, 4, 5, 6, 7, 8]); + assert_eq!(versions, vec![1, 2, 3, 4, 5, 6, 7, 8, 9]); assert_eq!( inventory.canonical_bytes().expect("canonical inventory"), inventory diff --git a/distributed_cli/tests/cli_lifecycle.rs b/distributed_cli/tests/cli_lifecycle.rs index eda70d1a4..72d2363c4 100644 --- a/distributed_cli/tests/cli_lifecycle.rs +++ b/distributed_cli/tests/cli_lifecycle.rs @@ -286,6 +286,7 @@ exit 1 fs::write(path, serde_json::to_vec_pretty(&config).unwrap()).unwrap(); } +#[track_caller] fn wait_until(timeout: Duration, predicate: impl Fn() -> bool) { let started = std::time::Instant::now(); while !predicate() { diff --git a/docs/consumer-delivery-lanes.md b/docs/consumer-delivery-lanes.md new file mode 100644 index 000000000..94878d7e6 --- /dev/null +++ b/docs/consumer-delivery-lanes.md @@ -0,0 +1,99 @@ +# Consumer delivery lanes and retry backoff + +One durable consumer (`listen`/`subscribe` group) delivers messages in broker +order and, by default, runs every registered route for a message before it +receives the next one. That is the simplest ordering contract, but it couples +latency across unrelated work: a process-manager policy that only turns a fact +into the next idempotent command waits behind every projection and slow +external effect registered for earlier messages. It also lets one message +that keeps failing monopolize the consumer. + +## Retry backoff for retryable failures + +A retryable failure NAKs the delivery so the broker redelivers it later. On +NATS JetStream the NAK carries a delay derived from the delivery count: +`base * 2^(delivered - 1)`, capped (defaults 50 ms and 5 s; +`NatsBus::with_nack_backoff` / `NatsJetStreamSource::with_nack_backoff`). +The first retry is still prompt. A message that fails every time stops +being redelivered ahead of newer messages in a hot loop. Retries remain +unlimited and the message is still retained; only the redelivery time moves. +Setting the base to zero restores an immediate NAK. + +This is not a substitute for classifying failures correctly. A command +rejection that is durably recorded under a deterministic command identity +replays the same rejection on every retry. The handler must return a +permanent error, which the configured failure policy (dead-letter by default) +settles and records in transport metrics. Transient infrastructure failures +stay retryable. + +## Lanes + +A router may partition its routes into named **lanes**. `Service::lane(name, +routes)` registers a route bundle in a lane. Plain `Service::routes` uses the +default lane. Lanes are opt-in. A service that registers no named lane runs +exactly as before. + +When a router has more than one lane and the source settles each delivery +independently (`MessageSource::settles_independently`, true for NATS +JetStream), the runner: + +1. receives messages in broker order and hands each one to every lane that + has a route for it; +2. runs each lane's messages strictly in receive order, one at a time, with + the lane's routes in registration order. This is the same order a + single-lane consumer gives those routes; +3. runs different lanes concurrently, so a slow route in one lane cannot delay + another lane; +4. settles a delivery only after every lane that received it has finished it. + Any retryable lane failure NAKs the delivery, and redelivery runs all of its + lanes again. A permanent failure applies the failure policy once; +5. bounds the number of received but unsettled deliveries + (`RunOptions::with_lane_window`, default 16). When the window is full the + runner stops receiving until a delivery settles; +6. stops a lane at its first stop-class failure (`FailurePolicy::Stop` or a + retain-and-stop error such as `ApplicationReloading`). That delivery is + settled as in sequential mode. Later deliveries queued for the halted lane + are NAKed, not skipped. Other in-flight deliveries finish and settle, then + the run returns the error. + +The default lane keeps its existing behavior, including route-order +dependencies between its routes within one delivery. + +Sources whose acknowledgement is positional (Kafka offset commits) or +lease-based (SQL table rows) report `settles_independently() == false`. For +them the runner keeps the sequential loop and dispatches every lane's routes +in order for each message. Lanes then change nothing about delivery. + +### What a lane must satisfy + +Put a route bundle in its own lane only when **no route in another lane +depends on its effects within the same delivery, and it depends on none of +theirs**. Typical examples are process-manager policies that read only the +delivered fact and send idempotent commands. A route that reads a projection +written by an earlier route for the same message must stay in that route's +lane. + +Lanes also reorder effects **across deliveries**: one lane may finish later +messages while another lane is still working on earlier ones. A route whose +correctness (or whose downstream readers' correctness) needs another lane's +effects from an *earlier* message — for example, a process policy that +completes a workflow the UI then reads through a projection maintained in +another lane — must share that lane, or its readers must tolerate the +projection arriving later. Forge hit this: provisioning completed before the +read-access projection existed, producing transient 404s, until the +derivation moved into the process lane. + +Lanes do not weaken existing guarantees: + +- **Ordering:** each lane observes deliveries in broker order. A redelivery + (after NAK or ack-wait expiry) can arrive after later messages. That is + already true for a single-lane consumer. +- **At-least-once:** a delivery is acknowledged only after all of its lanes + succeeded. A crash or failure before that redelivers it to every lane, so + every route must stay idempotent, as today. +- **Failure policy and DLQ:** unchanged, applied once per delivery after all + lanes finish. +- **Ack deadline:** a delivery waiting in a busy lane still counts against the + broker's ack wait. Keep the window small enough that a lane's backlog clears + well within it. An expired delivery is redelivered, which is safe under the + idempotency requirement. diff --git a/docs/external-facts.md b/docs/external-facts.md new file mode 100644 index 000000000..e3cdef685 --- /dev/null +++ b/docs/external-facts.md @@ -0,0 +1,78 @@ +# Authenticated external facts + +`DomainEventOccurrence::capture_external` represents an authenticated fact from +an external ledger, webhook or other durable source. It does not create an +aggregate, command receipt or business authorization. The adapter must verify +the source before calling the typed constructor. Deserializing an occurrence +alone does not authenticate its origin. + +The source identity consists of producer, stream, numeric position and member +key. Members of one external transaction share its position, with distinct keys. +The logical occurrence ID depends only on that identity. Changing its descriptor, +body, timestamp or metadata preserves the ID and must fail the existing durable +input fingerprint fence. Retries must reproduce the same canonical bytes. When +the source has no timestamp, an adapter may explicitly use the Unix epoch as an +unknown-time sentinel; it must not present adapter receipt time as source time. + +An `external_snapshot` projection applies full-row source snapshots. It orders +each external stream/member independently of actual broker delivery positions; +it does not invent an aggregate sequence. Existing aggregate projection and +command APIs retain their authority and fencing rules. Derived external facts +retain their source provenance, but are not accepted as direct source snapshots. + +NATS publishes external facts with a content-bound broker dedup key and a +separate reserved logical occurrence ID header. Thus identical retries within +the broker dedup window collapse, but altered bytes reach durable validation. +Different logical source facts with identical bodies never collapse. Retained +archive reads verify these headers and preserve external occurrences for replay. +Malformed or ambiguous identity headers fail permanently before dispatch and +remain unacknowledged. The supervisor sees the error; the adapter does not +silently terminate the message or advance progress past it. + +An identical logical input redelivered at a new broker cursor advances only the +delivery checkpoint within its execution generation. It does not repeat row +changes, observations or business effects. A new execution generation applies +its first delivery normally. Original and alias cursor bindings remain immutable, +and one canonical topology-wide message identity arbitrates concurrent partition +claims. Migration 0009 adds delivery aliases without deleting canonical identity +or failure records. SQL mutation continues to use the existing partition lock. + +Permanent input identities and their aliases must be retained for the replay +and source-conflict horizon. A bounded broker dedup window is not such a fence. +An ingress that acknowledges an external cursor must first establish its own +required durable qualification (for example, a history projection atomically +committed with the protocol fingerprint), and retain source replay until then. +Publishing is not approval, and waiting for all UI consumers is unnecessary. + +Offline snapshot rebuild preserves these same identities. It distinguishes +original aggregate positions, original external stream/position/member keys, +and derived occurrence IDs; unrelated external facts cannot collide at empty +aggregate fields. Duplicate logical IDs must retain identical canonical bytes. +Aggregate projections still require a complete original aggregate sequence +prefix. External positions may be sparse or shared by different members, so +the source adapter must certify the complete retained source range instead of +inventing a contiguous aggregate history. Rebuild still rejects omitted stored +source versions, changed content at one source position, a different source +claiming an existing row, and derived facts used as snapshot authority. It does +not run business handlers, republish events, or reset delivery cursors. + +Some original aggregate records are intentionally private and therefore absent +from the public archive. Public-archive-only rebuild remains fail-closed at such +a gap. An offline adapter can instead supply `AggregateRebuildCoverage`, built +from a complete original quiescent event-store stream, its independently read +head, an explicit authored private-contract inventory, and the retained public +occurrences. Missing or reordered records, unknown private contracts, missing +publications, changed public identities, and cross-stream substitution fail. +Every witnessed public occurrence must remain byte-identical in the rebuild +history. No private record is fabricated as a public event. +This constructor deliberately supports only authored single-publication streams +(one ordinal-zero occurrence per non-private record). The adapter must establish +that emitter contract; arbitrary one-to-many publication completeness cannot be +inferred from archive absence and is not covered by this API. + +This is an operator trust boundary, not cryptographic authentication of an +arbitrary export: retain the original source provenance and review the artifact +digest. The digest prevents substitution of the reviewed artifact. It does not +prove arbitrary private payloads reconstruct a public JSON state. Authored +event-to-public contract mappings and typed projection body validation remain +required; an absent broker message never makes a source record private. diff --git a/docs/gateway/live-sharing.md b/docs/gateway/live-sharing.md index 404f7bf58..531b281a1 100644 --- a/docs/gateway/live-sharing.md +++ b/docs/gateway/live-sharing.md @@ -37,6 +37,8 @@ If a slow consumer cannot preserve all evidence within its queue, it receives `LIVE_RESET_REQUIRED`; a blocked socket is closed so it must reconnect. No latest-value replacement silently discards confirmation proof. Group deadline, upstream loss and incomplete/invalid origin envelopes also require recovery. +An origin failure frame (an error-only live execution result) carries no evidence +and is not fanned out; the group's consumers receive `LIVE_RESET_REQUIRED`. Dropping a consumer releases only its lease. Last leave aborts the actual upstream stream, socket and origin change-feed receiver, including a pending origin SQL diff --git a/docs/live-query-delivery.md b/docs/live-query-delivery.md index 78931e7af..aa373488d 100644 --- a/docs/live-query-delivery.md +++ b/docs/live-query-delivery.md @@ -12,6 +12,47 @@ Every live response declares `extensions.distributed.live.mode`: - `resumable`: a result with matching, nonempty index and cursor vectors. Existing resume validation, reset, replay and causal reconciliation rules apply. +A failed execution has no result and therefore no `live` metadata. It is sent +as the terminal failure frame described below. + +## Failed live executions + +A live execution that fails (for example a storage error or statement timeout) +has no result. The server sends exactly one **failure frame** and then completes +the operation. The failure ends that subscription: the server sends no more +frames for it and never attaches a later execution's metadata to the failure +frame. A failure frame has: + +- `data` that is `null` or absent; +- a nonempty `errors` array, where each entry is an object with a string + `message`; +- a valid base `extensions.distributed` envelope (`protocolVersion`, + `schemaHash`, `authorizationGeneration`, `cacheScope`, `operation`, and + `trustedPresets` when the surface has any), with no `snapshot`, `live` or + `command`. + +The client checks the envelope's binding, schema, operation and authorization +generation in the same way as for any other response. The failure frame then +works like an HTTP error response: it admits nothing. It writes no data or +membership, advances no cursor or operation generation, takes no ownership and +confirms no command. The client shows the original GraphQL errors for that +operation and keeps any previously admitted data readable. It closes the +failed stream. After a bounded backoff (1s, doubling to at most 30s), it opens +a fresh subscription for the operation's current watches. That subscription +resumes from the last admitted cursors, if there are any. The backoff resets +when an admitted frame arrives, and that frame also replaces the errors. This is +how a live query recovers after its storage comes back, without a page reload. +Disposing the last watch or ending the authorization generation cancels a +pending reopen. + +Only an error-only frame is a failure frame. A live frame that carries non-null +`data` (including partial data with errors) without both `snapshot` and `live` +is invalid, as are frames with `errors` that are absent, empty or malformed and +lack live metadata, and frames with a `snapshot` or `live` but not both. A +response without the `extensions.distributed` envelope is invalid as before. +Intermediaries that cannot relay a failure frame, such as shared gateway live +fan-out, end the consumer with `LIVE_RESET_REQUIRED` instead. + Snapshot delivery does not relax read permissions or invent causal evidence. Changes affecting only denied rows must not produce activity frames. A row leaving the authorized result disappears from that operation's membership; @@ -30,6 +71,24 @@ empty result acquire rows without waiting for another page load. It does not override an independently active live stream or query ownership acquired after the subscription started; those results have no safe cross-stream ordering. +Independent streams often share an index without disagreeing about it. A +layout and a page may both select +`repositories(where: { id: { _eq: $id } }, limit: 1)` and then different +relationships below that row. A snapshot frame whose shared indexes have the +same membership as the independent owner's is admitted. Each such index must +contain the same normalized record keys in the same order and the same null +value, and the owner's index must be complete and not stale. The owner keeps +those shared indexes, and the frame writes only its other indexes and its +records. The resulting graph is exactly the frame's server result, so nothing +is fabricated and no cross-stream order is assumed. + +The whole frame is rejected as before if any shared membership differs or +cannot be established. That includes GraphQL errors, operation-local embedded +rows and rows without record evidence. The subscription is then reopened after +another operation next writes one of those shared indexes. The fresh result is +checked against the owner's new membership, so a stream whose frame arrived +before the owner caught up does not wait for its own next server change. + When the last watch for a live operation is disposed, its local ownership is retired at a monotonically increasing local boundary. A later live subscription may take over an incomparable shared index only when that subscription started diff --git a/docs/live-retired-owner-handoff.md b/docs/live-retired-owner-handoff.md new file mode 100644 index 000000000..f0c46f441 --- /dev/null +++ b/docs/live-retired-owner-handoff.md @@ -0,0 +1,21 @@ +# Snapshot stream handoff after owner disposal + +A snapshot live stream can begin while a previous route still owns a shared +relationship index. Its buffered frames must not acquire that graph merely +because the old route later disposes: their start fence predates disposal. +However, retaining that fence forever leaves the new route permanently pending. + +When an incoming snapshot is blocked by a retired owner's index, reopen that +contending operation at a fresh local start fence. Reject the triggering old +frame and all subsequent callbacks from its retired transport. Only a new +server-authorized initial frame may take over the graph. A still-active owner +continues to block; normal same-scope hydration must not restart retained layout +subscriptions. No application polling, forced refresh, or optimistic-state +suppression is part of this recovery. + +Runtime evidence: Forge's creation page received completed/ready rows, but the +prior NewRepository operation owned a shared topology relationship at revision +24, retired at 27, after the new subscription had already started. Entity clocks +advanced while the complete-empty root index remained fenced. The existing +pre-disposal protocol test covers safety; a fresh-receiver continuation covers +liveness after the same boundary. diff --git a/docs/protocol-manifest-reuse.md b/docs/protocol-manifest-reuse.md new file mode 100644 index 000000000..93a610319 --- /dev/null +++ b/docs/protocol-manifest-reuse.md @@ -0,0 +1,44 @@ +# Immutable protocol manifest reuse + +Runtime contract: a selected client surface export owns immutable service, +role/application selection, surface IR and execution limits. Its derived manifest +can be initialized once and shared across clones. Public callers still receive +independently mutable manifest copies. Errors are deterministic for that export +and may be retained; creating another export/engine creates a separate cache. + +The engine must retain the exact exports already validated during protocol +construction, including distinct role and application identities. Request seeds +may borrow their compiled manifests, but must still validate principal, asserted +roles, resolved preset values, authorization generation, visibility surface, +cache-scope HMAC and issuance time for each request. No session, token, result row, +authorization decision or request seed is cached. + +Observed baseline in Forge: authenticated document TTFB around 2.9–4.1 seconds, +simple authenticated GraphQL queries around 1.1 seconds. Native sampling identifies +ProtocolProjectionRequestSeed::new → export.manifest → projection manifest +lowering as repeated CPU work. Validation must cover cache/clone concurrency, +mutable-return independence, role/application/limits isolation and existing +protocol privacy/authority tests, followed by same-route runtime measurements. +No deadline, authentication, SSR or live-subscription behavior may be weakened. + +## Measured local validation + +Same retained Forge stack, same authenticated user and paths, September 21: + +| Warm document (DOM ready) | Before | After | +| --- | ---: | ---: | +| Dashboard | 3140 ms | 116 ms | +| Personal repository, including default-ref redirect | 7445 ms | 291 ms | +| Organization People | 3350 ms | 192 ms | +| ChangeSets | 4123 ms | 97 ms | + +Three direct authenticated GraphQL queries fell from 1143/1135/1072 ms to +22/15/14 ms. These are local observations, not performance assertions or an SLA. +The first repository document after the development server restart still took +18.7 seconds; the warm comparison does not hide that cold-development cost. + +Framework library: 1052 passed, 3 explicit ignored. GraphQL/identity/causal HTTP +integration targets: 43 passed. Added tests exercise concurrent first use, +independent mutable returns, role/application/limit separation, rebuilt-engine +isolation, and actual repeated protocol accumulators retaining/releasing shared +metadata without retaining request authority. Existing auth/privacy checks remain. diff --git a/docs/reloading-event-delivery.md b/docs/reloading-event-delivery.md new file mode 100644 index 000000000..c22c35ddf --- /dev/null +++ b/docs/reloading-event-delivery.md @@ -0,0 +1,15 @@ +# Event delivery during application reload + +A supervisor generation gate is temporary infrastructure state, not a domain +rejection. Both direct command admission and bus dispatch return typed +`ApplicationReloading` while closed. Bus conversion classifies it as retryable +and retains/stops the receive loop: NAK the exact delivery, then surface the +error to the host's bounded restart policy. It must never Ack, Term or apply +ordinary permanent-failure policy. After activation the same delivery can run. + +Business validation/authorization failures retain their existing permanent +classification and configured settlement policy. No gate bypass or implicit +success is introduced. The NAK uses the transport's ordinary retry spacing +(NATS JetStream: delivery-count backoff, see +[consumer delivery lanes and retry backoff](consumer-delivery-lanes.md)); the +gate adds no delay of its own. diff --git a/js/src/replica/distributed-replica/constants.ts b/js/src/replica/distributed-replica/constants.ts index 4bb101799..35cdaaabc 100644 --- a/js/src/replica/distributed-replica/constants.ts +++ b/js/src/replica/distributed-replica/constants.ts @@ -11,3 +11,6 @@ export const EMPTY_CACHE_SNAPSHOT = Object.freeze({ records: Object.freeze([]), indexes: Object.freeze([]) }); +/** Reopen backoff after a live failure frame (`docs/live-query-delivery.md`). */ +export const LIVE_FAILURE_RETRY_BASE_MS = 1_000; +export const LIVE_FAILURE_RETRY_MAX_MS = 30_000; diff --git a/js/src/replica/distributed-replica/helpers.ts b/js/src/replica/distributed-replica/helpers.ts index e67e19ce0..d8ebd69d6 100644 --- a/js/src/replica/distributed-replica/helpers.ts +++ b/js/src/replica/distributed-replica/helpers.ts @@ -105,6 +105,41 @@ export function prepareRecordEvidence( }; } +/** + * Membership a result would write for one index: normalized record keys in + * response order. `null` means it cannot be established from the response + * alone (blocked by errors, missing, embedded, or without record evidence). + */ +export type ReplicaIndexMembership = { + readonly records: readonly string[]; + readonly nullValue: boolean; +}; + +export type ReplicaIndexMemberships = Map; + +function recordMembership( + memberships: ReplicaIndexMemberships | undefined, + key: string, + membership: ReplicaIndexMembership | null +): void { + if (memberships === undefined) return; + const previous = memberships.get(key); + if (previous === undefined) { + memberships.set(key, membership); + return; + } + // One response reaching an index twice must agree with itself. + if ( + previous === null || + membership === null || + previous.nullValue !== membership.nullValue || + previous.records.length !== membership.records.length || + previous.records.some((record, ordinal) => record !== membership.records[ordinal]) + ) { + memberships.set(key, null); + } +} + export function replicaResultIndexKeys< TData, TVariables extends GraphqlVariables @@ -112,7 +147,8 @@ export function replicaResultIndexKeys< artifact: ReplicaOperationArtifact, variables: TVariables, envelope: ReplicaResultEnvelope, - snapshot: DistributedQuerySnapshot + snapshot: DistributedQuerySnapshot, + memberships?: ReplicaIndexMemberships ): ReadonlySet { const keys = new Set(); if ( @@ -156,6 +192,7 @@ export function replicaResultIndexKeys< root.responseKey ) ) { + recordMembership(memberships, rootKey, null); continue; } const value = envelope.data[root.responseKey]; @@ -163,6 +200,7 @@ export function replicaResultIndexKeys< value === null && resultPathHasErrors(errorPaths, rootPath) ) { + recordMembership(memberships, rootKey, null); continue; } collectResultBranchIndexKeys( @@ -174,7 +212,8 @@ export function replicaResultIndexKeys< variables, errorPaths, evidencePaths, - keys + keys, + memberships ); } return keys; @@ -189,11 +228,23 @@ export function collectResultBranchIndexKeys( variables: GraphqlVariables, errorPaths: readonly (readonly (string | number)[])[], evidencePaths: ReadonlySet, - keys: Set + keys: Set, + memberships?: ReplicaIndexMemberships ): void { - if (value === null || value === undefined) return; + if (value === null) { + recordMembership( + memberships, + enclosingIndexKey, + selection.nullable ? { records: [], nullValue: true } : null + ); + return; + } + if (value === undefined) { + recordMembership(memberships, enclosingIndexKey, null); + return; + } if (selection.cardinality === 'one') { - collectResultObjectIndexKeys( + const recordKey = collectResultObjectIndexKeys( artifactId, selection.selection, value, @@ -203,14 +254,28 @@ export function collectResultBranchIndexKeys( variables, errorPaths, evidencePaths, - keys + keys, + memberships + ); + recordMembership( + memberships, + enclosingIndexKey, + recordKey === undefined ? null : { records: [recordKey], nullValue: false } ); return; } - if (!Array.isArray(value)) return; + if (!Array.isArray(value)) { + recordMembership(memberships, enclosingIndexKey, null); + return; + } + const records: string[] = []; + let certain = true; for (const [ordinal, entry] of value.entries()) { - if (entry === null || entry === undefined) continue; - collectResultObjectIndexKeys( + if (entry === null || entry === undefined) { + certain = false; + continue; + } + const recordKey = collectResultObjectIndexKeys( artifactId, selection.selection, entry, @@ -220,9 +285,17 @@ export function collectResultBranchIndexKeys( variables, errorPaths, evidencePaths, - keys + keys, + memberships ); + if (recordKey === undefined) certain = false; + else records.push(recordKey); } + recordMembership( + memberships, + enclosingIndexKey, + certain ? { records, nullValue: false } : null + ); } export function collectResultObjectIndexKeys( @@ -235,10 +308,11 @@ export function collectResultObjectIndexKeys( variables: GraphqlVariables, errorPaths: readonly (readonly (string | number)[])[], evidencePaths: ReadonlySet, - keys: Set -): void { + keys: Set, + memberships?: ReplicaIndexMemberships +): string | undefined { if (resultPathBlocked(errorPaths, path) || !isReplicaResultObject(value)) { - return; + return undefined; } const fields = new Map(); for (const member of selection.members) { @@ -260,6 +334,7 @@ export function collectResultObjectIndexKeys( } } let parentKey: string; + let normalized = false; if ( selection.storage.kind === 'normalized' && evidencePaths.has(responsePathKey(path.map(String))) @@ -268,7 +343,8 @@ export function collectResultObjectIndexKeys( const value = fields.get(field); return value === undefined || value === null ? [] : [value]; }); - if (identity.length !== selection.storage.identityFields.length) return; + if (identity.length !== selection.storage.identityFields.length) return undefined; + normalized = true; parentKey = replicaRecordKey( { id: selection.storage.model, @@ -300,6 +376,7 @@ export function collectResultObjectIndexKeys( resultPathBlocked(errorPaths, branchPath) || !Object.prototype.hasOwnProperty.call(value, member.responseKey) ) { + recordMembership(memberships, branchKey, null); continue; } const branchValue = value[member.responseKey]; @@ -307,6 +384,7 @@ export function collectResultObjectIndexKeys( branchValue === null && resultPathHasErrors(errorPaths, branchPath) ) { + recordMembership(memberships, branchKey, null); continue; } collectResultBranchIndexKeys( @@ -318,9 +396,12 @@ export function collectResultObjectIndexKeys( variables, errorPaths, evidencePaths, - keys + keys, + memberships ); } + // Embedded rows are operation-local; they never share an index membership. + return normalized ? parentKey : undefined; } export function resultPathBlocked( @@ -736,6 +817,30 @@ export function stableErrors( return deepEqual(current, next) ? current : freezeErrors(next); } +/** + * GraphQL payload of a live failure frame (`docs/live-query-delivery.md`): no + * result data and a nonempty list of errors that each carry a string message. + * Callers also require absent snapshot, live and command metadata. Partial + * data with errors is never a failure payload. + */ +export function isGraphqlFailurePayload(envelope: { + readonly data?: unknown; + readonly errors?: unknown; +}): envelope is { readonly data?: null; readonly errors: readonly GqlError[] } { + return ( + (envelope.data === undefined || envelope.data === null) && + Array.isArray(envelope.errors) && + envelope.errors.length > 0 && + envelope.errors.every( + (error: unknown) => + error !== null && + typeof error === 'object' && + !Array.isArray(error) && + typeof (error as { message?: unknown }).message === 'string' + ) + ); +} + export function graphqlError(error: unknown): GqlError { return Object.freeze({ message: error instanceof Error ? error.message : String(error), diff --git a/js/src/replica/distributed-replica/impl-fetch-live.ts b/js/src/replica/distributed-replica/impl-fetch-live.ts index 5211aeee4..e8ba4bbde 100644 --- a/js/src/replica/distributed-replica/impl-fetch-live.ts +++ b/js/src/replica/distributed-replica/impl-fetch-live.ts @@ -1,4 +1,4 @@ -import type { GraphqlVariables } from '../../types.js'; +import type { GqlError, GraphqlVariables } from '../../types.js'; import { type DistributedLiveCursor, type DistributedProtocolEnvelope @@ -10,8 +10,13 @@ import type { ReplicaTransport, ReplicaWriteSource } from '../types.js'; +import { + LIVE_FAILURE_RETRY_BASE_MS, + LIVE_FAILURE_RETRY_MAX_MS +} from './constants.js'; import { graphqlError, + isGraphqlFailurePayload, replicaClientRequestExtensions, stableErrors } from './helpers.js'; @@ -68,6 +73,7 @@ export function closeActiveTransports(host: FetchLiveHost): void { for (const [key, entry] of host.lives) { retireLiveProtocol(host, key); entry.active = false; + cancelLiveRetry(entry); try { entry.unsubscribe(); } catch { @@ -311,7 +317,7 @@ export function retainLive( return; } try { - host.writeCanonicalResult( + const distributed = host.writeCanonicalResult( watch.artifact, watch.variables, result, @@ -319,9 +325,19 @@ export function retainLive( undefined, projectionGeneration ); + // Accepted without a snapshot, an error-only frame is the + // receipt-only failure frame: it admitted nothing. + if ( + distributed.snapshot === undefined && + isGraphqlFailurePayload(result) + ) { + failLive(host, watch, entry, result.errors); + return; + } state.live = 'active'; entry.operationGeneration = host.operationGeneration(watch.key); + entry.failures = undefined; } catch (error) { state.live = 'error'; state.errors = stableErrors(state.errors, [graphqlError(error)]); @@ -433,6 +449,7 @@ export function restartLive(host: FetchLiveHost, key: string): void { const count = previous.count; retireLiveProtocol(host, key); previous.active = false; + cancelLiveRetry(previous); host.lives.delete(key); try { previous.unsubscribe(); @@ -473,12 +490,69 @@ export function releaseLive(host: FetchLiveHost, key: string): void { if (entry.count > 0) return; retireLiveProtocol(host, key); entry.active = false; + cancelLiveRetry(entry); host.lives.delete(key); entry.unsubscribe(); host.queryState(key).live = 'off'; host.emitState(key, false); } +/** + * Surface a live failure frame and close its terminal stream + * (`docs/live-query-delivery.md`). The frame was receipt-only: no data, + * cursor, operation generation or ownership changed. The inactive entry stays + * registered, as in fallbackFromLive, so ingestion's resumeLiveWatches cannot + * bypass the backoff; the timer reopens the operation's current watches. + */ +function failLive( + host: FetchLiveHost, + watch: ReplicaWatchState, + entry: LiveEntry, + errors: readonly GqlError[] +): void { + if (!entry.active || host.lives.get(watch.key) !== entry) return; + retireLiveProtocol(host, watch.key); + entry.active = false; + const unsubscribe = entry.unsubscribe; + entry.unsubscribe = () => undefined; + try { + unsubscribe(); + } catch { + // The failed stream is fenced; transport cleanup is best effort. + } + const failures = (entry.failures ?? 0) + 1; + entry.failures = failures; + const delay = Math.min( + LIVE_FAILURE_RETRY_BASE_MS * 2 ** Math.min(failures - 1, 30), + LIVE_FAILURE_RETRY_MAX_MS + ); + entry.retry = setTimeout(() => { + entry.retry = undefined; + if ( + host.lives.get(watch.key) !== entry || + entry.active || + entry.protocolGeneration !== host.protocolGenerationSequence() + ) { + return; + } + restartLive(host, watch.key); + const replacement = host.lives.get(watch.key); + if (replacement !== undefined && replacement !== entry) { + replacement.failures = failures; + } + }, delay); + const state = host.queryState(watch.key); + state.live = 'error'; + state.errors = stableErrors(state.errors, errors); + host.emitState(watch.key, false); +} + +function cancelLiveRetry(entry: LiveEntry): void { + if (entry.retry === undefined) return; + clearTimeout(entry.retry); + entry.retry = undefined; +} + /** * Mark protocol state as no longer backed by a live transport. The state is * retained for cache and hydration bookkeeping, but a later live stream may diff --git a/js/src/replica/distributed-replica/impl.ts b/js/src/replica/distributed-replica/impl.ts index a18077834..dfb3dd46f 100644 --- a/js/src/replica/distributed-replica/impl.ts +++ b/js/src/replica/distributed-replica/impl.ts @@ -140,10 +140,12 @@ import { indexKeyFromTarget, indexMaintenanceSnapshot, indexSemanticLayer, + isGraphqlFailurePayload, operationKey, prepareRecordEvidence, protocolOperationSource, replicaResultIndexKeys, + type ReplicaIndexMemberships, reportSafely, reportUnhandledObserverError, snapshotFrom, @@ -1153,8 +1155,20 @@ export class DistributedReplicaImpl implements DistributedReplicaApi { if (live !== undefined && snapshot === undefined) { protocolInvalid('extensions.distributed.snapshot'); } + // A failed live execution has no result: an error-only frame with only + // the base envelope. It is receipt-only, like an HTTP error response, + // and must not carry a command receipt or create stream state. + const liveFailure = + source === 'live' && + snapshot === undefined && + live === undefined && + isGraphqlFailurePayload(envelope); + if (liveFailure && distributed.command !== undefined) { + protocolInvalid('extensions.distributed.command'); + } if ( source === 'live' && + !liveFailure && (envelope.data !== undefined || envelope.errors !== undefined) && (snapshot === undefined || live === undefined) ) { @@ -1167,7 +1181,7 @@ export class DistributedReplicaImpl implements DistributedReplicaApi { } const operationSource = protocolOperationSource(source); const operationState = - operation === undefined + operation === undefined || liveFailure ? undefined : this.#operationProtocol(key, operation, operationSource); if (snapshot === undefined || operationState === undefined) { @@ -1202,18 +1216,35 @@ export class DistributedReplicaImpl implements DistributedReplicaApi { handoff && activeState !== undefined ? compareSnapshotToOperationState(activeState, snapshot) : 'fresh'; - const sharedDisposition = this.#sharedIndexDisposition( + const incomingMemberships: ReplicaIndexMemberships = new Map(); + const fencedSharedDisposition = this.#sharedIndexDisposition( key, replicaResultIndexKeys( artifact, stableVariables, envelope, - snapshot + snapshot, + incomingMemberships ), snapshot, requestRevision, source ); + /* + * A snapshot frame that agrees with an independent owner on every + * index they share is exactly its server result once admitted: the + * owner keeps those indexes and this frame writes only the rest + * (docs/live-query-delivery.md). Disagreement keeps the atomic fence. + */ + const fencedIndexKeys = + snapshotLive ? fencedSharedDisposition.fencedIndexKeys : undefined; + const sharedMembershipAdmitted = + fencedIndexKeys !== undefined && + (envelope.errors ?? []).length === 0 && + this.#fencedMembershipsMatch(fencedIndexKeys, incomingMemberships); + const sharedDisposition: SharedIndexDisposition = sharedMembershipAdmitted + ? { compared: false } + : fencedSharedDisposition; // A non-resumable stream has no causal vector to compare. An HTTP // refresh may replace it only if the request began after its last // accepted membership. fetchWatch also fences intervening live frames. @@ -1473,7 +1504,10 @@ export class DistributedReplicaImpl implements DistributedReplicaApi { let summary: ReturnType; try { const update = (writer: BaseCacheWriter) => { - const guarded = this.#guardIndexWriter(writer); + const guarded = this.#guardIndexWriter( + writer, + sharedMembershipAdmitted ? fencedIndexKeys : undefined + ); this.#applyTombstoneEvidence( guarded, recordEvidence.tombstones, @@ -1614,6 +1648,30 @@ export class DistributedReplicaImpl implements DistributedReplicaApi { if (source !== 'live' && sourceSwitched) { this.#restartLive(key); } + if (source === 'live') { + const liveEntry = this.#lives.get(key); + if (liveEntry !== undefined) { + liveEntry.fencedIndexKeys = + fencedIndexKeys !== undefined && !sharedMembershipAdmitted + ? fencedIndexKeys + : undefined; + } + } + if (writeIndexes) { + this.#reopenStreamsFencedBy( + key, + summary.indexKeys.filter( + (indexKey) => + !(sharedMembershipAdmitted && fencedIndexKeys!.has(indexKey)) + ) + ); + } + if (source === 'live' && sharedDisposition.restartAfterRetirement) { + // This receiver began before a shared owner retired. Its buffered + // frame stays fenced, but a fresh receiver starts after that boundary + // and can obtain an authoritative replacement without polling. + this.#restartLive(key); + } this.#trustedPresets = nextTrustedPresets; this.#protocolGeneration = nextProtocolGeneration; this.#resumeLiveWatches(); @@ -1806,7 +1864,12 @@ export class DistributedReplicaImpl implements DistributedReplicaApi { } } - #guardIndexWriter(writer: BaseCacheWriter): BaseCacheWriter { + #guardIndexWriter( + writer: BaseCacheWriter, + ownedElsewhere?: ReadonlySet + ): BaseCacheWriter { + // Equal shared memberships stay with their independent owner. + const skip = (key: string): boolean => ownedElsewhere?.has(key) === true; return { recordClock: (key) => writer.recordClock(key), writeRecord: (write) => writer.writeRecord(write), @@ -1814,6 +1877,7 @@ export class DistributedReplicaImpl implements DistributedReplicaApi { writer.tombstoneRecord(key, revision, incarnation), discardRecord: (key) => writer.discardRecord(key), writeIndex: (write) => { + if (skip(write.key)) return false; const recordsForIndex = this.#membershipFences.get(write.key); if (write.complete === true && recordsForIndex !== undefined) { const visible = @@ -1841,11 +1905,59 @@ export class DistributedReplicaImpl implements DistributedReplicaApi { return wrote; }, markIndexStale: (key, reason, revision) => - writer.markIndexStale(key, reason, revision), - deleteIndex: (key, revision) => writer.deleteIndex(key, revision) + !skip(key) && writer.markIndexStale(key, reason, revision), + deleteIndex: (key, revision) => + !skip(key) && writer.deleteIndex(key, revision) }; } + #fencedMembershipsMatch( + fencedIndexKeys: ReadonlySet, + incoming: ReplicaIndexMemberships + ): boolean { + return this.#engine.readConfirmed((reader) => { + for (const indexKey of fencedIndexKeys) { + const membership = incoming.get(indexKey); + const current = reader.index(indexKey); + if ( + membership === undefined || + membership === null || + current === undefined || + !current.complete || + (current.staleRevision !== undefined && + compareCanonicalDecimalStrings(current.staleRevision, current.revision) > 0) || + (current.metadata?.nullValue === true) !== membership.nullValue || + current.records.length !== membership.records.length || + current.records.some( + (record, ordinal) => record !== membership.records[ordinal] + ) + ) { + return false; + } + } + return true; + }); + } + + /** + * Reopen live streams whose last frame was fenced by a shared index that + * another operation has now rewritten, so a fresh authoritative result is + * compared against the new owner membership. + */ + #reopenStreamsFencedBy(writerKey: string, writtenKeys: readonly string[]): void { + if (writtenKeys.length === 0) return; + const reopen: string[] = []; + for (const [liveKey, entry] of this.#lives) { + if (liveKey === writerKey || !entry.active) continue; + const fenced = entry.fencedIndexKeys; + if (fenced === undefined) continue; + if (writtenKeys.some((indexKey) => fenced.has(indexKey))) { + reopen.push(liveKey); + } + } + for (const liveKey of reopen) this.#restartLive(liveKey); + } + #flushDeferredMembershipConfirms(): void { for (const commandId of [...this.#deferredMembershipConfirms]) { if (this.#commandHasMembershipFence(commandId)) continue; @@ -2327,37 +2439,40 @@ export class DistributedReplicaImpl implements DistributedReplicaApi { let lower = false; let higher = false; let incomparable = false; + let restartAfterRetirement = false; + const fencedIndexKeys = new Set(); let equalRevision: string | undefined; let latestOwnerRevision: string | undefined; for (const [key, group] of this.#operationProtocols) { if (key === currentKey) continue; for (const state of [group.query, group.live]) { if (state?.indexRevision === undefined) continue; - let ownsIncomingIndex = false; + const ownedIncomingKeys: string[] = []; for (const indexKey of state.indexKeys) { if ( incomingIndexKeys.has(indexKey) && confirmedRevisions.get(indexKey) === state.indexRevision ) { - ownsIncomingIndex = true; - break; + ownedIncomingKeys.push(indexKey); } } - if (!ownsIncomingIndex) continue; + if (ownedIncomingKeys.length === 0) continue; if ( !snapshot.indexesComparable && state === group.live && state.retiredAtRevision !== undefined && - liveStart !== undefined && - compareCanonicalDecimalStrings( - liveStart, - state.retiredAtRevision - ) > 0 + liveStart !== undefined ) { // A disposed stream is no longer an owner. Its boundary is // still retained so a stream that started before disposal // cannot win merely because the old transport later closed. - continue; + if (compareCanonicalDecimalStrings(liveStart, state.retiredAtRevision) > 0) { + continue; + } + // Reopen at most once per observed retirement boundary: the + // replacement's allocated start is strictly newer. Active + // siblings never trigger this recovery and remain fenced. + restartAfterRetirement = true; } latestOwnerRevision = latestOwnerRevision === undefined || @@ -2396,6 +2511,7 @@ export class DistributedReplicaImpl implements DistributedReplicaApi { compareCanonicalDecimalStrings(liveStart, state.indexRevision) > 0 ) continue; incomparable = true; + for (const indexKey of ownedIncomingKeys) fencedIndexKeys.add(indexKey); continue; } compared = true; @@ -2429,7 +2545,12 @@ export class DistributedReplicaImpl implements DistributedReplicaApi { * explicit synchronous ingress retains its caller-defined order. */ if (requestRevision === undefined && source === 'live') { - return { compared: true, disposition: 'lower' }; + return { + compared: true, + disposition: 'lower', + ...(restartAfterRetirement ? { restartAfterRetirement: true } : {}), + ...(!snapshot.indexesComparable ? { fencedIndexKeys } : {}) + }; } if (requestRevision === undefined) return { compared: false }; return latestOwnerRevision !== undefined && diff --git a/js/src/replica/distributed-replica/types.ts b/js/src/replica/distributed-replica/types.ts index 4ae8faade..686038609 100644 --- a/js/src/replica/distributed-replica/types.ts +++ b/js/src/replica/distributed-replica/types.ts @@ -33,6 +33,15 @@ export type LiveEntry = { /** Local fence for taking over query snapshots that preceded this stream. */ startRevision: string; operationGeneration?: number; + /** Consecutive failed live executions; drives the reopen backoff. */ + failures?: number; + /** Pending reopen of this inactive entry after a live failure frame. */ + retry?: ReturnType; + /** + * Shared indexes whose independent owner disagreed with this stream's last + * frame. Another operation rewriting one of them reopens this stream. + */ + fencedIndexKeys?: ReadonlySet; }; export type ProtocolGeneration = { @@ -169,6 +178,13 @@ export type SharedIndexDisposition = { readonly compared: boolean; readonly disposition?: 'equal' | 'higher' | 'lower'; readonly indexRevision?: string; + readonly restartAfterRetirement?: boolean; + /** + * Incoming indexes owned by an independent live stream or later query + * owner. A snapshot frame may be admitted without writing them when its + * membership for each is identical to the owner's. + */ + readonly fencedIndexKeys?: ReadonlySet; }; export type CapturedReplicaOptimisticOperation = diff --git a/js/tests/replica-protocol.test.mjs b/js/tests/replica-protocol.test.mjs index 42095e281..f7955b84f 100644 --- a/js/tests/replica-protocol.test.mjs +++ b/js/tests/replica-protocol.test.mjs @@ -1945,6 +1945,209 @@ test('snapshot live cannot supersede an incomparable sibling index owner', async assert.equal(unsubscribeCount, 1); }); +// A layout and a page commonly open independent snapshot streams over the same +// root row and select different relationships below it. +const GamesRootLive = Object.freeze({ + ...GamesWithOwner, + id: 'query:games-root-live', + document: 'query GamesRoot { games { id owner_id } }', + protocol: Object.freeze({ + ...GamesWithOwner.protocol, + operation: 'query:games-root-live' + }), + live: Object.freeze({ + id: 'live:games-root', + document: 'subscription GamesRootLive { games { id owner_id } }' + }), + roots: Object.freeze([ + Object.freeze({ + ...GamesWithOwner.roots[0], + selection: Object.freeze({ + ...GamesWithOwner.roots[0].selection, + members: Object.freeze( + GamesWithOwner.roots[0].selection.members.filter( + (member) => member.kind === 'scalar' + ) + ) + }) + }) + ]) +}); + +function sharedGamesFrame({ artifact, games, revision = '1', withOwner }) { + const records = games.flatMap((game, index) => [ + { + path: ['games', String(index)], + model: 'GameView', + scopeToken: `record:${game.id}`, + incarnation: '1', + revision, + tombstone: false + }, + ...(withOwner + ? [{ + path: ['games', String(index), 'owner'], + model: 'UserView', + scopeToken: `record:${game.owner_id}`, + incarnation: '1', + revision, + tombstone: false + }] + : []) + ]); + return { + data: { + games: games.map((game) => withOwner + ? { id: game.id, owner_id: game.owner_id, owner: { id: game.owner_id, name: game.name } } + : { id: game.id, owner_id: game.owner_id }) + }, + extensions: { + distributed: { + protocolVersion: DISTRIBUTED_PROTOCOL_VERSION, + schemaHash: 'schema-a', + authorizationGeneration: 'auth-1', + cacheScope: 'cache:a', + operation: artifact.live.id, + snapshot: { + scopeToken: `snapshot:${artifact.id}`, + recordsComplete: true, + indexesComparable: false, + records, + indexes: [], + observations: [] + }, + live: { mode: 'snapshot', reset: true, cursors: [] } + } + } + }; +} + +function sharedGamesHarness() { + const subscriptions = []; + const replica = createDistributedReplica({ + transport: { + fetch() { + return new Promise(() => {}); + }, + subscribe(request, observer) { + const subscription = { operation: request.operationId, observer, closed: false }; + subscriptions.push(subscription); + return () => { + subscription.closed = true; + }; + } + } + }); + const latest = (operation) => + subscriptions.filter((entry) => entry.operation === operation).at(-1); + return { replica, subscriptions, latest }; +} + +test('snapshot live admits a frame whose shared membership matches an independent owner', () => { + const { replica, latest } = sharedGamesHarness(); + const layout = replica.watch(GamesRootLive, {}, { live: true }); + latest('live:games-root').observer.next(sharedGamesFrame({ + artifact: GamesRootLive, games: [{ id: 'game-1', owner_id: 'user-1' }] + })); + const page = replica.watch(GamesWithOwnerLiveOperation, {}, { live: true }); + const pageStream = latest('live:games-with-owner'); + pageStream.observer.next(sharedGamesFrame({ + artifact: GamesWithOwnerLiveOperation, withOwner: true, + games: [{ id: 'game-1', owner_id: 'user-1', name: 'Owner' }] + })); + assert.deepEqual(replica.read(GamesWithOwnerLiveOperation, {}).data.games, [ + { id: 'game-1', owner_id: 'user-1', owner: { id: 'user-1', name: 'Owner' } } + ], 'the first page frame renders although the layout owns the shared root'); + + // A later page frame changes only its own relationship below the shared row. + pageStream.observer.next(sharedGamesFrame({ + artifact: GamesWithOwnerLiveOperation, withOwner: true, revision: '2', + games: [{ id: 'game-1', owner_id: 'user-2', name: 'Second' }] + })); + assert.deepEqual(replica.read(GamesWithOwnerLiveOperation, {}).data.games, [ + { id: 'game-1', owner_id: 'user-2', owner: { id: 'user-2', name: 'Second' } } + ], 'a live update to a relationship-joined row is delivered'); + + // The admitted page stream did not take the shared root from the layout. + latest('live:games-root').observer.next(sharedGamesFrame({ + artifact: GamesRootLive, revision: '3', + games: [{ id: 'game-1', owner_id: 'user-2' }, { id: 'game-3', owner_id: 'user-3' }] + })); + assert.deepEqual( + replica.read(GamesRootLive, {}).data.games.map((game) => game.id), + ['game-1', 'game-3'] + ); + pageStream.observer.next(sharedGamesFrame({ + artifact: GamesWithOwnerLiveOperation, withOwner: true, revision: '3', + games: [ + { id: 'game-1', owner_id: 'user-2', name: 'Second' }, + { id: 'game-3', owner_id: 'user-3', name: 'Third' } + ] + })); + assert.deepEqual( + replica.read(GamesWithOwnerLiveOperation, {}).data.games.map((game) => game.owner.name), + ['Second', 'Third'] + ); + assert.equal(page.get().live, 'active'); + assert.equal(pageStream.closed, false, 'agreeing frames never reopen the stream'); + page.destroy(); + layout.destroy(); +}); + +test('snapshot live keeps a disagreeing frame fenced and reopens after the owner catches up', () => { + const { replica, subscriptions, latest } = sharedGamesHarness(); + const layout = replica.watch(GamesRootLive, {}, { live: true }); + latest('live:games-root').observer.next(sharedGamesFrame({ + artifact: GamesRootLive, games: [{ id: 'game-1', owner_id: 'user-1' }] + })); + const page = replica.watch(GamesWithOwnerLiveOperation, {}, { live: true }); + const firstPageStream = latest('live:games-with-owner'); + // The page stream observed a newer membership than the layout owner. + firstPageStream.observer.next(sharedGamesFrame({ + artifact: GamesWithOwnerLiveOperation, withOwner: true, revision: '2', + games: [ + { id: 'game-1', owner_id: 'user-1', name: 'Owner' }, + { id: 'game-2', owner_id: 'user-2', name: 'Second' } + ] + })); + assert.deepEqual( + replica.read(GamesRootLive, {}).data.games.map((game) => game.id), + ['game-1'], + 'a disagreeing frame cannot replace the independent owner membership' + ); + assert.equal(firstPageStream.closed, false); + const subscriptionCount = subscriptions.length; + + latest('live:games-root').observer.next(sharedGamesFrame({ + artifact: GamesRootLive, revision: '2', + games: [{ id: 'game-1', owner_id: 'user-1' }, { id: 'game-2', owner_id: 'user-2' }] + })); + assert.equal(firstPageStream.closed, true, 'the owner rewrite reopens the fenced stream'); + assert.equal(subscriptions.length, subscriptionCount + 1); + const reopened = latest('live:games-with-owner'); + assert.notEqual(reopened, firstPageStream); + + // A late frame from the closed stream stays fenced. + firstPageStream.observer.next(sharedGamesFrame({ + artifact: GamesWithOwnerLiveOperation, withOwner: true, revision: '2', + games: [{ id: 'game-1', owner_id: 'user-1', name: 'Late' }] + })); + reopened.observer.next(sharedGamesFrame({ + artifact: GamesWithOwnerLiveOperation, withOwner: true, revision: '2', + games: [ + { id: 'game-1', owner_id: 'user-1', name: 'Owner' }, + { id: 'game-2', owner_id: 'user-2', name: 'Second' } + ] + })); + assert.deepEqual( + replica.read(GamesWithOwnerLiveOperation, {}).data.games.map((game) => game.owner.name), + ['Owner', 'Second'] + ); + assert.equal(page.get().live, 'active'); + page.destroy(); + layout.destroy(); +}); + test('snapshot live replaces SSR membership, updates and removes rows, and fences disposal', () => { let observer; const replica = createDistributedReplica({ transport: { @@ -2022,13 +2225,14 @@ test('snapshot live takes over a nested graph from a disposed page subscription' test('snapshot live keeps a stream started before disposal behind the retired owner', () => { const observers = []; + const closed = []; const previousPage = { ...FeaturedGamesWithOwner, live: { id: 'live:featured-owner-before-disposal', document: 'subscription FeaturedOwnerBeforeDisposal { featuredGames { id owner { id name } } }' } }; const replica = createDistributedReplica({ transport: { fetch() { throw new Error('complete snapshot must not force HTTP fallback'); }, - subscribe(_request, observer) { observers.push(observer); return () => {}; } + subscribe(_request, observer) { observers.push(observer); return () => { closed.push(observer); }; } } }); const oldFrame = (ownerName) => gamesFrame({ artifact: previousPage, responseKey: 'featuredGames', operation: previousPage.live.id, @@ -2045,22 +2249,50 @@ test('snapshot live keeps a stream started before disposal behind the retired ow empty.extensions.distributed.snapshot.records = []; replica.writeResult(GamesWithOwnerLiveOperation, {}, empty, 'ssr'); const current = replica.watch(GamesWithOwnerLiveOperation, {}, { live: true }); + // The contender disagrees with the active owner's shared owner membership; + // agreeing frames are admitted (docs/live-query-delivery.md). observers[1].next(gamesFrame({ artifact: GamesWithOwnerLiveOperation, responseKey: 'games', operation: GamesWithOwnerLiveOperation.live.id, position: '3', - ownerId: 'user-1', ownerName: 'started before disposal', indexesComparable: false, + ownerId: 'user-2', ownerName: 'started before disposal', indexesComparable: false, live: { mode: 'snapshot', reset: true, cursors: [] } })); assert.deepEqual(current.get().data.games, []); + const layout = replica.watch(Todos, {}, { live: true }); + observers[2].next(wireFrame({ operation: Todos.live.id, indexesComparable: false, + live: { mode: 'snapshot', reset: true }, rows: [{ id: 'retained', title: 'layout' }] })); oldWatch.destroy(); observers[1].next(gamesFrame({ artifact: GamesWithOwnerLiveOperation, responseKey: 'games', operation: GamesWithOwnerLiveOperation.live.id, position: '4', - ownerId: 'user-1', ownerName: 'still started before disposal', indexesComparable: false, + ownerId: 'user-2', ownerName: 'still started before disposal', indexesComparable: false, live: { mode: 'snapshot', reset: true, cursors: [] } })); assert.deepEqual(current.get().data.games, []); + assert.equal(observers.length, 4, 'retired ownership must reopen only the contender after the disposal fence'); + assert.deepEqual(closed, [observers[0], observers[1]]); + assert.equal(layout.get().live, 'active'); + observers[1].next(gamesFrame({ + artifact: GamesWithOwnerLiveOperation, responseKey: 'games', + operation: GamesWithOwnerLiveOperation.live.id, position: '5', + ownerId: 'user-2', ownerName: 'queued old receiver', indexesComparable: false, + live: { mode: 'snapshot', reset: true, cursors: [] } + })); + assert.deepEqual(current.get().data.games, [], 'old receiver stays fenced after reopening'); + observers[3].next(gamesFrame({ + artifact: GamesWithOwnerLiveOperation, responseKey: 'games', + operation: GamesWithOwnerLiveOperation.live.id, position: '6', + ownerId: 'user-1', ownerName: 'fresh authoritative receiver', indexesComparable: false, + live: { mode: 'snapshot', reset: true, cursors: [] } + })); + assert.equal(current.get().data.games[0].owner.name, 'fresh authoritative receiver'); + assert.equal(observers.length, 4, 'the same retirement boundary cannot reopen repeatedly'); + observers[2].next(wireFrame({ operation: Todos.live.id, indexesComparable: false, + live: { mode: 'snapshot', reset: true }, revision: '2', rows: [{ id: 'retained', title: 'continued layout' }] })); + assert.equal(layout.get().data.todos[0].title, 'continued layout'); current.destroy(); + layout.destroy(); + assert.deepEqual(closed, [observers[0], observers[1], observers[3], observers[2]]); }); test('reopening a retired live owner restores its ownership fence', () => { @@ -2069,8 +2301,8 @@ test('reopening a retired live owner restores its ownership fence', () => { ...FeaturedGamesWithOwner, live: { id: 'live:featured-owner-reopened', document: 'subscription FeaturedOwnerReopened { featuredGames { id owner { id name } } }' } }; - const frame = (artifact, responseKey, operation, position, ownerName) => gamesFrame({ - artifact, responseKey, operation, position, ownerId: 'user-1', ownerName, + const frame = (artifact, responseKey, operation, position, ownerName, ownerId = 'user-1') => gamesFrame({ + artifact, responseKey, operation, position, ownerId, ownerName, indexesComparable: false, live: { mode: 'snapshot', reset: true, cursors: [] } }); const replica = createDistributedReplica({ transport: { @@ -2099,7 +2331,9 @@ test('reopening a retired live owner restores its ownership fence', () => { 'games', GamesWithOwnerLiveOperation.live.id, '4', - 'current contender' + 'current contender', + // Disagrees with the reopened owner; an agreeing frame is admitted. + 'user-2' )); assert.deepEqual(current.get().data.games, []); reopened.destroy(); @@ -2182,7 +2416,8 @@ test('two watches retire shared live ownership only after final disposal', () => observers[1].next(gamesFrame({ artifact: GamesWithOwnerLiveOperation, responseKey: 'games', operation: GamesWithOwnerLiveOperation.live.id, position: '3', - ownerId: 'user-1', ownerName: 'before final release', indexesComparable: false, + // Disagrees with the owner; an agreeing frame is admitted. + ownerId: 'user-2', ownerName: 'before final release', indexesComparable: false, live: { mode: 'snapshot', reset: true, cursors: [] } })); assert.deepEqual(startedBeforeFinalRelease.get().data.games, []); @@ -2191,10 +2426,11 @@ test('two watches retire shared live ownership only after final disposal', () => observers[1].next(gamesFrame({ artifact: GamesWithOwnerLiveOperation, responseKey: 'games', operation: GamesWithOwnerLiveOperation.live.id, position: '4', - ownerId: 'user-1', ownerName: 'still before final release', indexesComparable: false, + ownerId: 'user-2', ownerName: 'still before final release', indexesComparable: false, live: { mode: 'snapshot', reset: true, cursors: [] } })); assert.deepEqual(startedBeforeFinalRelease.get().data.games, []); + assert.equal(observers.length, 3, 'the final release enables one fresh contender receiver'); startedBeforeFinalRelease.destroy(); const afterFinalRelease = replica.watch( @@ -2202,7 +2438,8 @@ test('two watches retire shared live ownership only after final disposal', () => {}, { live: true } ); - observers[2].next(gamesFrame({ + assert.equal(observers.length, 4, 'reopening after disposal creates another receiver'); + observers[3].next(gamesFrame({ artifact: GamesWithOwnerLiveOperation, responseKey: 'games', operation: GamesWithOwnerLiveOperation.live.id, position: '5', ownerId: 'user-1', ownerName: 'after final release', indexesComparable: false, @@ -2573,6 +2810,251 @@ test('terminal live errors close their stream and allow a later HTTP retry', asy assert.equal(unsubscribeCount, 2); }); +/** Error-only live failure frame: base envelope, no snapshot/live/command. */ +function liveFailureFrame(options = {}) { + return { + ...(options.omitData ? {} : { data: null }), + ...(options.omitErrors + ? {} + : { + errors: options.errors ?? [ + { + message: 'statement timeout', + path: ['todos'], + extensions: { code: 'TIMEOUT' } + } + ] + }), + ...(options.data === undefined ? {} : { data: options.data }), + extensions: { + distributed: { + protocolVersion: DISTRIBUTED_PROTOCOL_VERSION, + schemaHash: 'schema-a', + authorizationGeneration: 'auth-1', + cacheScope: 'cache:a', + operation: options.operation ?? 'live:todos', + ...options.distributed + } + } + }; +} + +function liveFailureHarness() { + const fetches = []; + const subscriptions = []; + let unsubscribeCount = 0; + const replica = createDistributedReplica({ + transport: { + fetch() { + let resolve; + const promise = new Promise((done) => { + resolve = done; + }); + fetches.push({ resolve }); + return promise; + }, + subscribe(request, observer) { + subscriptions.push({ request, observer }); + return () => { + unsubscribeCount += 1; + }; + } + } + }); + return { + replica, + fetches, + subscriptions, + unsubscribes: () => unsubscribeCount + }; +} + +test('live failure frames surface the GraphQL error, admit nothing, and reopen with backoff', async (t) => { + t.mock.timers.enable({ apis: ['setTimeout'] }); + const { replica, fetches, subscriptions, unsubscribes } = liveFailureHarness(); + const watch = replica.watch(Todos, {}, { live: true }); + await Promise.resolve(); + assert.equal(subscriptions.length, 1); + subscriptions[0].observer.next( + wireFrame({ + operation: 'live:todos', + position: '2', + rows: [{ id: 'todo-live', title: 'admitted before failure' }], + recordScope: 'record:live', + live: { mode: 'resumable', reset: true } + }) + ); + const admitted = [{ id: 'todo-live', title: 'admitted before failure' }]; + assert.deepEqual(watch.get().data.todos, admitted); + assert.deepEqual(watch.get().errors, []); + const fetchesBeforeFailure = fetches.length; + + subscriptions[0].observer.next(liveFailureFrame()); + assert.equal(watch.get().errors[0].message, 'statement timeout'); + assert.equal(watch.get().errors[0].extensions.code, 'TIMEOUT'); + assert.equal(watch.get().live, 'error'); + assert.deepEqual(watch.get().data.todos, admitted); + assert.deepEqual(replica.read(Todos, {}).data.todos, admitted); + // The failure is terminal: the stream is closed and the server's trailing + // complete neither falls back to HTTP nor reopens outside the backoff. + assert.equal(unsubscribes(), 1); + subscriptions[0].observer.complete(); + assert.equal(fetches.length, fetchesBeforeFailure); + assert.equal(watch.get().live, 'error'); + // Receipt-only ingestion (here a failing HTTP read) resumes live watches; + // the failed one keeps waiting for its backoff. + replica.writeResult( + Todos, + {}, + liveFailureFrame({ operation: 'query:todos' }), + 'network' + ); + assert.equal(subscriptions.length, 1); + assert.deepEqual(replica.read(Todos, {}).data.todos, admitted); + + t.mock.timers.tick(999); + assert.equal(subscriptions.length, 1); + t.mock.timers.tick(1); + assert.equal(subscriptions.length, 2); + // No cursor or ownership moved: the reopen resumes at the admitted cursor. + assert.deepEqual(subscriptions[1].request.resume, [ + { projection: 'todos-projector', position: '2', token: 'resume:2' } + ]); + assert.equal(watch.get().errors[0].message, 'statement timeout'); + + subscriptions[1].observer.next(liveFailureFrame({ omitData: true })); + assert.equal(unsubscribes(), 2); + t.mock.timers.tick(1999); + assert.equal(subscriptions.length, 2, 'consecutive failures double the backoff'); + t.mock.timers.tick(1); + assert.equal(subscriptions.length, 3); + + subscriptions[2].observer.next( + wireFrame({ + operation: 'live:todos', + position: '3', + rows: [{ id: 'todo-live', title: 'recovered' }], + recordScope: 'record:live', + live: { mode: 'resumable' } + }) + ); + assert.deepEqual(watch.get().errors, []); + assert.equal(watch.get().live, 'active'); + assert.deepEqual(watch.get().data.todos, [ + { id: 'todo-live', title: 'recovered' } + ]); + + subscriptions[2].observer.next(liveFailureFrame()); + t.mock.timers.tick(1000); + assert.equal(subscriptions.length, 4, 'an admitted frame resets the backoff'); + + subscriptions[3].observer.next(liveFailureFrame()); + watch.destroy(); + t.mock.timers.tick(30_000); + assert.equal(subscriptions.length, 4, 'disposing the last watch cancels the reopen'); +}); + +test('authorization invalidation cancels a pending live failure reopen', async (t) => { + t.mock.timers.enable({ apis: ['setTimeout'] }); + const { replica, subscriptions } = liveFailureHarness(); + const watch = replica.watch(Todos, {}, { live: true }); + await Promise.resolve(); + subscriptions[0].observer.next(liveFailureFrame()); + assert.equal(watch.get().errors[0].message, 'statement timeout'); + replica.invalidateAuthorization(); + t.mock.timers.tick(30_000); + assert.equal(subscriptions.length, 1); + watch.destroy(); +}); + +test('live frames that are not error-only failures stay strict and admit nothing', async () => { + const cases = [ + { + name: 'partial data with errors and no live metadata', + frame: liveFailureFrame({ + data: { todos: [{ id: 'todo-partial', title: 'partial' }] } + }), + path: 'extensions.distributed.live' + }, + { + name: 'data-bearing snapshot without live metadata', + frame: wireFrame({ + operation: 'live:todos', + position: '9', + rows: [{ id: 'todo-unscoped', title: 'unscoped' }] + }), + path: 'extensions.distributed.live' + }, + { + name: 'empty errors', + frame: liveFailureFrame({ errors: [] }), + path: 'extensions.distributed.live' + }, + { + name: 'null data without errors', + frame: liveFailureFrame({ omitErrors: true }), + path: 'extensions.distributed.live' + }, + { + name: 'errors without a string message', + frame: liveFailureFrame({ errors: [{ message: 42 }] }), + path: 'extensions.distributed.live' + }, + { + name: 'command receipt on a failure frame', + frame: liveFailureFrame({ + distributed: { command: commandReceipt() } + }), + path: 'extensions.distributed.command' + }, + { + name: 'failure frame for another operation', + frame: liveFailureFrame({ operation: 'live:todos-other' }), + path: 'extensions.distributed.operation' + }, + { + name: 'failure frame without the distributed envelope', + frame: { + data: null, + errors: [{ message: 'statement timeout' }] + }, + path: 'extensions.distributed' + } + ]; + for (const { name, frame, path } of cases) { + const { replica, fetches, subscriptions, unsubscribes } = + liveFailureHarness(); + const watch = replica.watch(Todos, {}, { live: true }); + await Promise.resolve(); + subscriptions[0].observer.next( + wireFrame({ + operation: 'live:todos', + position: '2', + rows: [{ id: 'todo-live', title: 'admitted' }], + recordScope: 'record:live', + live: { mode: 'resumable', reset: true } + }) + ); + const fetchCount = fetches.length; + subscriptions[0].observer.next(frame); + assert.equal(watch.get().live, 'error', name); + assert.equal( + watch.get().errors[0].message, + `Invalid Distributed GraphQL protocol envelope at ${path}`, + name + ); + assert.deepEqual( + replica.read(Todos, {}).data.todos, + [{ id: 'todo-live', title: 'admitted' }], + name + ); + assert.equal(unsubscribes(), 0, `${name}: not a terminal failure frame`); + assert.equal(fetches.length, fetchCount, name); + assert.equal(subscriptions.length, 1, name); + watch.destroy(); + } +}); + test('live advancement fences an overlapping refresh while a later clean refresh succeeds', async () => { const fetches = []; const subscriptions = []; diff --git a/migrations/inventory.json b/migrations/inventory.json index 25b8cc134..5a16b6163 100644 --- a/migrations/inventory.json +++ b/migrations/inventory.json @@ -96,6 +96,18 @@ "path": "migrations/postgres/0008_external_command_binding.sql", "sha256": "2f93c5685cde1d5cafd444d3a28dbcbf01f16cd44a566c7af61a7bd2a793a163" } + }, + { + "version": 9, + "description": "projection delivery aliases", + "sqlite": { + "path": "migrations/sqlite/0009_projection_delivery_aliases.sql", + "sha256": "b7fafebbe898fe7d05b6fb385eb6db4fd850fc9d314357bb91582fa5c00f38a4" + }, + "postgres": { + "path": "migrations/postgres/0009_projection_delivery_aliases.sql", + "sha256": "19845d4bfcb30d1bf43cb485a6bb9f765a6c1db7abf74fb004046457af83a2b9" + } } ] } diff --git a/migrations/postgres/0009_projection_delivery_aliases.sql b/migrations/postgres/0009_projection_delivery_aliases.sql new file mode 100644 index 000000000..2eee335d6 --- /dev/null +++ b/migrations/postgres/0009_projection_delivery_aliases.sql @@ -0,0 +1,15 @@ +-- Alternate broker delivery positions for an already immutable logical input. +-- Keep the canonical globally unique message binding and all historical rows. +CREATE TABLE projection_input_delivery_aliases ( + topology_hash bytea NOT NULL, + partition_hash bytea NOT NULL, + source_hash bytea NOT NULL, + source_partition_hash bytea NOT NULL, + source_epoch text NOT NULL, + source_position bigint NOT NULL CHECK (source_position >= 0), + message_id text NOT NULL, + PRIMARY KEY (topology_hash, partition_hash, source_hash, source_partition_hash, source_epoch, source_position), + FOREIGN KEY (topology_hash, message_id) + REFERENCES projection_input_identities (topology_hash, message_id) +); + diff --git a/migrations/sqlite/0009_projection_delivery_aliases.sql b/migrations/sqlite/0009_projection_delivery_aliases.sql new file mode 100644 index 000000000..c9f391bec --- /dev/null +++ b/migrations/sqlite/0009_projection_delivery_aliases.sql @@ -0,0 +1,15 @@ +-- Alternate broker delivery positions for an already immutable logical input. +-- Keep the canonical globally unique message binding and all historical rows. +CREATE TABLE projection_input_delivery_aliases ( + topology_hash BLOB NOT NULL, + partition_hash BLOB NOT NULL, + source_hash BLOB NOT NULL, + source_partition_hash BLOB NOT NULL, + source_epoch text NOT NULL, + source_position bigint NOT NULL CHECK (source_position >= 0), + message_id text NOT NULL, + PRIMARY KEY (topology_hash, partition_hash, source_hash, source_partition_hash, source_epoch, source_position), + FOREIGN KEY (topology_hash, message_id) + REFERENCES projection_input_identities (topology_hash, message_id) +); + diff --git a/src/bus/mod.rs b/src/bus/mod.rs index 07ca9d09a..c6d1b4bad 100644 --- a/src/bus/mod.rs +++ b/src/bus/mod.rs @@ -130,7 +130,10 @@ pub use knative::knative_triggers; #[cfg(feature = "http")] pub use knative_bus::KnativeBus; #[cfg(feature = "nats")] -pub use nats::{NatsJetStreamSource, NatsPublisher, NatsReceived}; +pub use nats::{ + NackBackoff, NatsJetStreamSource, NatsPublisher, NatsReceived, DEFAULT_NACK_BACKOFF_BASE, + DEFAULT_NACK_BACKOFF_MAX, +}; #[cfg(feature = "nats")] pub use nats_bus::{NatsBus, NatsBusConnect}; #[cfg(feature = "rabbitmq")] @@ -162,8 +165,8 @@ pub use ordered_delivery::OrderedDelivery; #[cfg(feature = "postgres")] pub use postgres_bus::{LogReceived, PostgresBus, QueueReceived}; pub use publisher::MessagePublisher; -pub use router::MessageRouter; -pub use run_options::{ConsumerDeliveryMode, InboxHook, NoInbox, RunOptions}; +pub use router::{LaneSet, MessageRouter}; +pub use run_options::{ConsumerDeliveryMode, InboxHook, NoInbox, RunOptions, DEFAULT_LANE_WINDOW}; pub use runner::run_source; pub use source::{MessageSource, ReceivedMessage}; #[cfg(feature = "sqlite")] diff --git a/src/bus/nats.rs b/src/bus/nats.rs index f18b72a30..ceebfd21e 100644 --- a/src/bus/nats.rs +++ b/src/bus/nats.rs @@ -3,9 +3,10 @@ //! Maps the canonical [`Message`] onto NATS JetStream: [`NatsPublisher`] publishes //! to a subject (waiting for the JetStream publish ack — the durable publish //! threshold), and [`NatsJetStreamSource`] pulls from a durable consumer and -//! settles via JetStream ack semantics (ack→`Ack`, nack→`Nak`, dead-letter/park→ -//! `Term`). The stable message id rides as the `Nats-Msg-Id` header so JetStream -//! dedup and downstream `Message.id` agree. +//! settles via JetStream ack semantics (ack→`Ack`, nack→`Nak` with a +//! delivery-count [`NackBackoff`] delay, dead-letter/park→`Term`). Aggregate message IDs use `Nats-Msg-Id` unchanged. External facts +//! retain their logical source identity separately; their broker dedup ID also +//! binds content so altered retries reach the durable conflict fence. //! //! Requires the `nats` feature. Integration-tested in `tests/nats_transport` //! against a JetStream-enabled server (see `compose.yaml`). @@ -30,6 +31,148 @@ const MESSAGE_ID_HEADER: &str = "Nats-Msg-Id"; const MESSAGE_KIND_HEADER: &str = "X-Sourced-Kind"; /// Header carrying the canonical payload media type. const CONTENT_TYPE_HEADER: &str = "Content-Type"; +/// External occurrences separate immutable fact identity from broker dedup. +const LOGICAL_ID_HEADER: &str = "X-Distributed-Occurrence-Id"; + +fn external_dedup_id(payload: &[u8]) -> String { + use sha2::{Digest, Sha256}; + let mut digest = Sha256::new(); + digest.update(b"distributed.nats.external-occurrence.v1\0"); + // Canonical occurrence bytes include the immutable logical source identity. + digest.update(payload); + format!("external:sha256:{:x}", digest.finalize()) +} + +fn external_occurrence( + message: &Message, +) -> Result, TransportError> { + let Ok(event) = crate::DomainEventOccurrence::from_canonical_bytes(&message.payload) else { + return Ok(None); + }; + if event.external_source().is_none() { + return Ok(None); + } + if message.kind != super::MessageKind::Event + || message.id() != Some(event.id()) + || message.name() != event.descriptor().name + { + return Err(TransportError::permanent( + "external occurrence differs from transport identity", + )); + } + Ok(Some(event)) +} + +pub(super) fn archived_identity_matches( + event: &crate::DomainEventOccurrence, + payload: &[u8], + headers: &async_nats::HeaderMap, +) -> bool { + decode_headers( + event.descriptor().name.to_string(), + payload.to_vec(), + Some(headers), + ) + .is_ok_and(|message| { + message.kind == super::MessageKind::Event && message.id() == Some(event.id()) + }) +} + +fn unambiguous_identity_headers(headers: &async_nats::HeaderMap) -> bool { + [LOGICAL_ID_HEADER, MESSAGE_ID_HEADER, MESSAGE_KIND_HEADER] + .iter() + .all(|reserved| { + headers + .iter() + .filter(|(key, _)| key.to_string().eq_ignore_ascii_case(reserved)) + .map(|(_, values)| values.len()) + .sum::() + <= 1 + }) +} + +fn decode_headers( + name: String, + payload: Vec, + headers: Option<&async_nats::HeaderMap>, +) -> Result { + if headers.is_some_and(|headers| !unambiguous_identity_headers(headers)) { + return Err(TransportError::permanent( + "ambiguous occurrence identity header", + )); + } + let values: Vec<_> = headers + .into_iter() + .flat_map(|headers| headers.iter()) + .flat_map(|(key, values)| { + values + .iter() + .map(move |value| (key.to_string(), value.to_string())) + }) + .collect(); + // External publishers always emit the exact kind. Do not let the generic + // transport's permissive unknown-kind default turn poison into an event. + if crate::DomainEventOccurrence::from_canonical_bytes(&payload) + .is_ok_and(|event| event.external_source().is_some()) + && !values + .iter() + .any(|(key, value)| key.eq_ignore_ascii_case(MESSAGE_KIND_HEADER) && value == "event") + { + return Err(TransportError::permanent( + "external occurrence event kind is missing or invalid", + )); + } + let mut message = message_from_wire( + name, + payload, + Some(MESSAGE_ID_HEADER), + MESSAGE_KIND_HEADER, + values, + ); + take_content_type(&mut message); + restore_external_identity(&mut message)?; + Ok(message) +} + +fn restore_external_identity(message: &mut Message) -> Result<(), TransportError> { + let ids: Vec<_> = message + .metadata + .iter() + .filter(|(key, _)| key.eq_ignore_ascii_case(LOGICAL_ID_HEADER)) + .map(|(_, value)| value.clone()) + .collect(); + message + .metadata + .retain(|(key, _)| !key.eq_ignore_ascii_case(LOGICAL_ID_HEADER)); + if ids.is_empty() { + if crate::DomainEventOccurrence::from_canonical_bytes(&message.payload) + .is_ok_and(|event| event.external_source().is_some()) + { + return Err(TransportError::permanent( + "external occurrence logical identity is missing", + )); + } + return Ok(()); + } + if ids.len() != 1 || message.id() != Some(external_dedup_id(&message.payload).as_str()) { + return Err(TransportError::permanent( + "invalid external occurrence transport identity", + )); + } + let event = crate::DomainEventOccurrence::from_canonical_bytes(&message.payload) + .map_err(|_| TransportError::permanent("invalid external occurrence payload"))?; + if event.external_source().is_none() + || event.id() != ids[0] + || event.descriptor().name != message.name() + || message.kind != super::MessageKind::Event + { + return Err(TransportError::permanent( + "external occurrence differs from transport identity", + )); + } + message.id = Some(ids[0].clone()); + Ok(()) +} /// Publishes canonical messages to a NATS JetStream subject. /// @@ -78,15 +221,23 @@ impl MessagePublisher for NatsPublisher { let subject = self.subject(&message); let mut headers = async_nats::HeaderMap::new(); for (key, value) in &message.metadata { - if [MESSAGE_ID_HEADER, MESSAGE_KIND_HEADER, CONTENT_TYPE_HEADER] - .iter() - .any(|reserved| key.eq_ignore_ascii_case(reserved)) + if [ + MESSAGE_ID_HEADER, + MESSAGE_KIND_HEADER, + CONTENT_TYPE_HEADER, + LOGICAL_ID_HEADER, + ] + .iter() + .any(|reserved| key.eq_ignore_ascii_case(reserved)) { continue; } headers.insert(key.as_str(), value.as_str()); } - if let Some(id) = message.id() { + if let Some(event) = external_occurrence(&message)? { + headers.insert(MESSAGE_ID_HEADER, external_dedup_id(&message.payload)); + headers.insert(LOGICAL_ID_HEADER, event.id()); + } else if let Some(id) = message.id() { headers.insert(MESSAGE_ID_HEADER, id); } headers.insert(MESSAGE_KIND_HEADER, message.kind.as_str()); @@ -109,12 +260,65 @@ impl MessagePublisher for NatsPublisher { } } +/// Default first delay for a retryable NAK; doubles per delivery attempt. +pub const DEFAULT_NACK_BACKOFF_BASE: Duration = Duration::from_millis(50); +/// Default ceiling for a retryable NAK delay. +pub const DEFAULT_NACK_BACKOFF_MAX: Duration = Duration::from_secs(5); + +/// Redelivery delay requested with a retryable NAK. +/// +/// `base * 2^(delivered - 1)`, capped at `max`. A zero `base` requests +/// immediate redelivery (the broker's own policy). Retries stay unlimited; +/// only their spacing changes, so one failing delivery cannot monopolize a +/// consumer ahead of newer messages. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct NackBackoff { + base: Duration, + max: Duration, +} + +impl NackBackoff { + /// Backoff starting at `base` and capped at `max` (at least `base`). + pub fn new(base: Duration, max: Duration) -> Self { + Self { + base, + max: max.max(base), + } + } + + /// Immediate NAK without a delay. + pub fn immediate() -> Self { + Self::new(Duration::ZERO, Duration::ZERO) + } + + /// Delay for a delivery that has been delivered `delivered` times (1-based). + pub fn delay(&self, delivered: u64) -> Option { + if self.base.is_zero() { + return None; + } + let doublings = delivered.saturating_sub(1).min(30) as u32; + Some( + self.base + .checked_mul(1u32 << doublings) + .unwrap_or(self.max) + .min(self.max), + ) + } +} + +impl Default for NackBackoff { + fn default() -> Self { + Self::new(DEFAULT_NACK_BACKOFF_BASE, DEFAULT_NACK_BACKOFF_MAX) + } +} + /// A pull-based JetStream source bound to a durable consumer. pub struct NatsJetStreamSource { consumer: Consumer, fetch_timeout: Duration, strip_prefix: Option, idle_poll: Duration, + nack_backoff: NackBackoff, } impl NatsJetStreamSource { @@ -125,9 +329,16 @@ impl NatsJetStreamSource { fetch_timeout: Duration::from_millis(500), strip_prefix: None, idle_poll: Duration::ZERO, + nack_backoff: NackBackoff::default(), } } + /// Space retryable NAK redeliveries (see [`NackBackoff`]). + pub fn with_nack_backoff(mut self, backoff: NackBackoff) -> Self { + self.nack_backoff = backoff; + self + } + /// How long `recv` waits for a message before returning `Ok(None)`. pub fn with_fetch_timeout(mut self, timeout: Duration) -> Self { self.fetch_timeout = timeout; @@ -204,6 +415,11 @@ impl MessageSource for NatsJetStreamSource { "nats" } + fn settles_independently(&self) -> bool { + // JetStream explicit acks settle each delivery by its own reply subject. + true + } + async fn recv(&mut self) -> Result, TransportError> { loop { let mut batch = self @@ -217,10 +433,11 @@ impl MessageSource for NatsJetStreamSource { match batch.next().await { Some(Ok(message)) => { - return Ok(Some(NatsReceived::from_jetstream( - message, - self.strip_prefix.as_deref(), - ))) + // Fail closed before dispatch without acknowledging or overriding + // the supervisor's permanent-error policy. The durable delivery is + // retained for operator repair; no later cursor is falsely sealed. + return NatsReceived::from_jetstream(message, self.strip_prefix.as_deref()) + .map(|received| Some(received.with_nack_backoff(self.nack_backoff))); } Some(Err(err)) => return Err(retryable("nats batch message", err)), None if self.idle_poll.is_zero() => return Ok(None), @@ -235,37 +452,37 @@ pub struct NatsReceived { raw: jetstream::Message, message: Message, ordered: Option, + nack_backoff: NackBackoff, } impl NatsReceived { - fn from_jetstream(raw: jetstream::Message, strip_prefix: Option<&str>) -> Self { + fn from_jetstream( + raw: jetstream::Message, + strip_prefix: Option<&str>, + ) -> Result { let name = strip_address_prefix(raw.subject.to_string(), strip_prefix); let payload = raw.payload.to_vec(); - let headers: Vec<(String, String)> = raw - .headers - .as_ref() - .into_iter() - .flat_map(|headers| headers.iter()) - .filter_map(|(key, values)| { - values - .last() - .map(|value| (key.to_string(), value.to_string())) - }) - .collect(); - let mut message = message_from_wire( - name, - payload, - Some(MESSAGE_ID_HEADER), - MESSAGE_KIND_HEADER, - headers, - ); - take_content_type(&mut message); + let message = decode_headers(name, payload, raw.headers.as_ref())?; let ordered = jetstream_ordered(&raw); - Self { + Ok(Self { raw, message, ordered, - } + nack_backoff: NackBackoff::default(), + }) + } + + fn with_nack_backoff(mut self, backoff: NackBackoff) -> Self { + self.nack_backoff = backoff; + self + } + + /// Delivery attempt count reported by JetStream (1 for the first delivery). + fn delivered(&self) -> u64 { + self.raw + .info() + .map(|info| u64::try_from(info.delivered).unwrap_or(1)) + .unwrap_or(1) } async fn settle(self, kind: AckKind) -> Result<(), TransportError> { @@ -324,8 +541,10 @@ impl ReceivedMessage for NatsReceived { } async fn nack(self, _reason: &str) -> Result<(), TransportError> { - // Nak with no delay: JetStream redelivers per the consumer policy. - self.settle(AckKind::Nak(None)).await + // Redeliver after a delivery-count backoff so a message that keeps + // failing cannot be redelivered ahead of newer messages in a hot loop. + let delay = self.nack_backoff.delay(self.delivered()); + self.settle(AckKind::Nak(delay)).await } async fn dead_letter(self, _reason: &str) -> Result<(), TransportError> { @@ -344,6 +563,128 @@ mod tests { use super::*; use crate::bus::MessageKind; + #[derive(serde::Serialize, serde::Deserialize, crate::DomainEvent)] + #[domain_event(name = "external.balance_recorded", version = 1)] + struct Balance { + value: u64, + } + + fn external_message(position: u64, value: u64) -> Message { + let event = crate::DomainEventOccurrence::capture_external( + crate::ExternalEventSource { + producer: "ledger.adapter".into(), + stream: "account-one".into(), + position, + key: "balance".into(), + }, + std::time::UNIX_EPOCH, + Default::default(), + &Balance { value }, + ) + .unwrap(); + Message::new( + event.descriptor().name.to_string(), + MessageKind::Event, + event.canonical_bytes().unwrap(), + ) + .with_id(event.id()) + } + + #[test] + fn external_nats_identity_binds_logical_fact_and_content_without_changing_aggregate_ids() { + let first = external_message(1, 10); + let altered = external_message(1, 11); + let next = external_message(2, 10); + assert_eq!(first.id(), altered.id()); + assert_ne!( + external_dedup_id(&first.payload), + external_dedup_id(&altered.payload) + ); + assert_ne!( + external_dedup_id(&first.payload), + external_dedup_id(&next.payload) + ); + let mut wire = first.clone(); + wire.id = Some(external_dedup_id(&wire.payload)); + wire.metadata + .push((LOGICAL_ID_HEADER.into(), first.id().unwrap().into())); + restore_external_identity(&mut wire).unwrap(); + assert_eq!(wire.id, first.id); + assert_eq!(wire.payload, first.payload); + assert_eq!(wire.metadata, first.metadata); + let mut ordinary = Message::new("record.changed", MessageKind::Event, b"{}".to_vec()) + .with_id("aggregate-message"); + assert!(external_occurrence(&ordinary).unwrap().is_none()); + restore_external_identity(&mut ordinary).unwrap(); + assert_eq!(ordinary.id(), Some("aggregate-message")); + } + + #[test] + fn external_nats_forged_or_ambiguous_logical_headers_fail_closed() { + let first = external_message(1, 10); + for (logical, broker) in [ + ("forged".to_string(), external_dedup_id(&first.payload)), + (first.id().unwrap().to_string(), "forged-broker".to_string()), + ] { + let mut wire = first.clone(); + wire.id = Some(broker); + wire.metadata.push((LOGICAL_ID_HEADER.into(), logical)); + assert!(restore_external_identity(&mut wire).is_err()); + } + let mut repeated = first.clone(); + repeated.id = Some(external_dedup_id(&first.payload)); + repeated.metadata = vec![ + (LOGICAL_ID_HEADER.into(), first.id().unwrap().into()), + (LOGICAL_ID_HEADER.to_lowercase(), first.id().unwrap().into()), + ]; + assert!(restore_external_identity(&mut repeated).is_err()); + } + + #[test] + fn retained_archive_rejects_same_ambiguity_as_live_delivery() { + let message = external_message(1, 10); + let event = crate::DomainEventOccurrence::from_canonical_bytes(&message.payload).unwrap(); + let mut headers = async_nats::HeaderMap::new(); + headers.insert(MESSAGE_ID_HEADER, external_dedup_id(&message.payload)); + headers.insert(LOGICAL_ID_HEADER, event.id()); + headers.insert(MESSAGE_KIND_HEADER, "event"); + assert!(archived_identity_matches( + &event, + &message.payload, + &headers + )); + for name in [ + LOGICAL_ID_HEADER.to_string(), + LOGICAL_ID_HEADER.to_lowercase(), + MESSAGE_ID_HEADER.to_string(), + MESSAGE_ID_HEADER.to_lowercase(), + ] { + let mut forged = headers.clone(); + forged.append(name, "conflicting-extra-value"); + assert!(!unambiguous_identity_headers(&forged)); + assert!(!archived_identity_matches( + &event, + &message.payload, + &forged + )); + } + } + + #[test] + fn nack_backoff_doubles_from_base_and_caps() { + let backoff = NackBackoff::default(); + assert_eq!(backoff.delay(1), Some(Duration::from_millis(50))); + assert_eq!(backoff.delay(2), Some(Duration::from_millis(100))); + assert_eq!(backoff.delay(4), Some(Duration::from_millis(400))); + assert_eq!(backoff.delay(8), Some(Duration::from_secs(5))); + assert_eq!(backoff.delay(u64::MAX), Some(Duration::from_secs(5))); + // Delivery counts from a broker that never reports one still back off. + assert_eq!(backoff.delay(0), Some(Duration::from_millis(50))); + assert_eq!(NackBackoff::immediate().delay(9), None); + let custom = NackBackoff::new(Duration::from_millis(10), Duration::from_millis(5)); + assert_eq!(custom.delay(3), Some(Duration::from_millis(10))); + } + #[test] fn content_type_selection_removes_every_case_variant() { let mut message = Message::new("example.recorded", MessageKind::Event, Vec::new()) diff --git a/src/bus/nats_bus.rs b/src/bus/nats_bus.rs index 6f0a9feb3..67999d174 100644 --- a/src/bus/nats_bus.rs +++ b/src/bus/nats_bus.rs @@ -31,7 +31,7 @@ use async_nats::jetstream; use async_nats::jetstream::consumer::pull::Config as PullConfig; use async_nats::jetstream::stream::{Config as StreamConfig, Stream}; -use super::nats::{NatsJetStreamSource, NatsPublisher}; +use super::nats::{NackBackoff, NatsJetStreamSource, NatsPublisher}; use super::{ retryable, run_source, Bus, BusConsumer, BusTopologyConfig, MessagePublisher, MessageRouter, RunOptions, TransportError, @@ -49,6 +49,7 @@ pub struct NatsBus { topology: BusTopologyConfig, fetch_timeout: Duration, idle_poll: Duration, + nack_backoff: NackBackoff, } /// Awaitable builder returned by [`NatsBus::connect`]. @@ -113,6 +114,7 @@ impl NatsBus { topology: BusTopologyConfig::default(), fetch_timeout: DEFAULT_FETCH_TIMEOUT, idle_poll: Duration::ZERO, + nack_backoff: NackBackoff::default(), } } @@ -175,6 +177,12 @@ impl NatsBus { self } + /// Space retryable NAK redeliveries for `listen`/`subscribe` sources. + pub fn with_nack_backoff(mut self, backoff: NackBackoff) -> Self { + self.nack_backoff = backoff; + self + } + /// Sanitize the group into a valid NATS consumer-name token. Consumer names /// cannot contain `.`, `*`, `>`, or whitespace, so map them to `_`. fn durable_base(group: &str) -> String { @@ -290,7 +298,11 @@ impl NatsBus { TransportError::permanent(format!("invalid archived occurrence: {e}")) })?; if event.descriptor().name != name - || message.headers.get("Nats-Msg-Id").map(|v| v.as_str()) != Some(event.id()) + || !super::nats::archived_identity_matches( + &event, + &message.payload, + &message.headers, + ) { return Err(TransportError::permanent( "archived occurrence differs from its transport identity", @@ -361,7 +373,8 @@ impl NatsBus { Ok(NatsJetStreamSource::new(consumer) .with_fetch_timeout(self.fetch_timeout) .with_strip_prefix(strip_prefix) - .with_idle_poll(self.idle_poll)) + .with_idle_poll(self.idle_poll) + .with_nack_backoff(self.nack_backoff)) } /// Shared consume path for `listen` (commands) and `subscribe` (events): diff --git a/src/bus/router.rs b/src/bus/router.rs index 0ad35dd94..78df6b245 100644 --- a/src/bus/router.rs +++ b/src/bus/router.rs @@ -61,4 +61,81 @@ pub trait MessageRouter: Send + Sync { ) -> impl Future> + Send { self.dispatch(message) } + + /// Number of independent delivery lanes (at least 1). + /// + /// Lanes are opt-in route groups that a receive loop may run concurrently + /// while each lane keeps delivery order. See + /// `docs/consumer-delivery-lanes.md`. The default single lane preserves the + /// sequential contract. + fn delivery_lanes(&self) -> usize { + 1 + } + + /// Lanes that have a handler for `(kind, name)`. Empty means unhandled. + fn lanes_for(&self, kind: MessageKind, name: &str) -> LaneSet { + if self.handles(kind, name) { + LaneSet::single(0) + } else { + LaneSet::EMPTY + } + } + + /// Run only the routes of `lane` for one delivered message. + /// + /// Routers that report more than one lane must override this. The default + /// is correct only for the single default lane. + fn dispatch_lane( + &self, + message: &Message, + ordered: Option<&OrderedDelivery>, + _lane: usize, + ) -> impl Future> + Send { + self.dispatch_ordered(message, ordered) + } +} + +/// A set of delivery lane indexes (at most [`LaneSet::MAX_LANES`]). +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Hash)] +pub struct LaneSet(u64); + +impl LaneSet { + /// Highest supported number of lanes for one router. + pub const MAX_LANES: usize = 64; + /// No lanes: the router does not handle the message. + pub const EMPTY: Self = Self(0); + + /// A set containing only `lane`. + pub fn single(lane: usize) -> Self { + Self::EMPTY.with(lane) + } + + /// Add `lane` to the set. + /// + /// # Panics + /// When `lane >= MAX_LANES`. + pub fn with(self, lane: usize) -> Self { + assert!(lane < Self::MAX_LANES, "delivery lane index out of range"); + Self(self.0 | (1 << lane)) + } + + /// Whether the set is empty. + pub fn is_empty(self) -> bool { + self.0 == 0 + } + + /// Number of lanes in the set. + pub fn len(self) -> usize { + self.0.count_ones() as usize + } + + /// Whether `lane` is in the set. + pub fn contains(self, lane: usize) -> bool { + lane < Self::MAX_LANES && self.0 & (1 << lane) != 0 + } + + /// Lane indexes in ascending order. + pub fn iter(self) -> impl Iterator { + (0..Self::MAX_LANES).filter(move |lane| self.contains(*lane)) + } } diff --git a/src/bus/run_options.rs b/src/bus/run_options.rs index 2a98ae450..3ed4e58ad 100644 --- a/src/bus/run_options.rs +++ b/src/bus/run_options.rs @@ -70,14 +70,21 @@ pub struct RunOptions { pub delivery_mode: ConsumerDeliveryMode, /// What the runner does with a permanent handler/transport failure. pub failure_policy: FailurePolicy, + /// Maximum received-but-unsettled deliveries when a router runs several + /// delivery lanes concurrently. Ignored for single-lane routers. + pub lane_window: usize, } +/// Default bound on unsettled deliveries for a multi-lane run. +pub const DEFAULT_LANE_WINDOW: usize = 16; + impl Default for RunOptions { /// Idempotent delivery with the default [`FailurePolicy`]. fn default() -> Self { Self { delivery_mode: ConsumerDeliveryMode::default(), failure_policy: FailurePolicy::default(), + lane_window: DEFAULT_LANE_WINDOW, } } } @@ -98,6 +105,7 @@ impl RunOptions { Self { delivery_mode: ConsumerDeliveryMode::Inbox(hook), failure_policy: FailurePolicy::default(), + lane_window: DEFAULT_LANE_WINDOW, } } @@ -107,6 +115,13 @@ impl RunOptions { self } + /// Bound the received-but-unsettled deliveries of a multi-lane run. + /// Values below 1 are raised to 1. + pub fn with_lane_window(mut self, window: usize) -> Self { + self.lane_window = window.max(1); + self + } + /// Whether this run dispatches directly without an inbox. pub fn is_idempotent(&self) -> bool { matches!(self.delivery_mode, ConsumerDeliveryMode::Idempotent) diff --git a/src/bus/runner/lane_tests.rs b/src/bus/runner/lane_tests.rs new file mode 100644 index 000000000..e81faba11 --- /dev/null +++ b/src/bus/runner/lane_tests.rs @@ -0,0 +1,548 @@ +//! Delivery-lane runner contract (`docs/consumer-delivery-lanes.md`). +//! +//! Runtime-free like the sequential runner tests: a busy-poll executor with a +//! no-op waker. Gated handlers wake themselves so `FuturesUnordered` re-polls +//! them, and `block_on` fails after a bounded number of polls instead of +//! hanging, so a lane that is blocked behind another lane is a test failure. +use super::run_source; +use crate::bus::source::{MessageSource, ReceivedMessage}; +use crate::bus::{ + FailurePolicy, LaneSet, Message, MessageKind, MessageRouter, RunOptions, TransportError, +}; +use std::collections::VecDeque; +use std::future::{poll_fn, Future}; +use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; +use std::sync::{Arc, Mutex}; +use std::task::Poll; + +const POLL_LIMIT: usize = 200_000; + +fn block_on(future: F) -> F::Output { + use std::ptr; + use std::task::{Context, RawWaker, RawWakerVTable, Waker}; + + const VTABLE: RawWakerVTable = RawWakerVTable::new( + |_| RawWaker::new(ptr::null(), &VTABLE), + |_| {}, + |_| {}, + |_| {}, + ); + let waker = unsafe { Waker::from_raw(RawWaker::new(ptr::null(), &VTABLE)) }; + let mut cx = Context::from_waker(&waker); + let mut future = std::pin::pin!(future); + for _ in 0..POLL_LIMIT { + if let Poll::Ready(output) = future.as_mut().poll(&mut cx) { + return output; + } + } + panic!("runner made no progress: a lane is blocked behind another lane"); +} + +/// Resolve once `ready` returns true, yielding (and self-waking) until then. +async fn wait_until(ready: impl Fn() -> bool) { + poll_fn(|cx| { + if ready() { + Poll::Ready(()) + } else { + cx.waker().wake_by_ref(); + Poll::Pending + } + }) + .await +} + +#[derive(Clone, Debug, PartialEq, Eq)] +enum Event { + Received(String), + Ran(usize, String), + Ack(String), + Nack(String), + DeadLetter(String), +} + +#[derive(Default)] +struct Log(Mutex>); + +impl Log { + fn push(&self, event: Event) { + self.0.lock().unwrap().push(event); + } + fn events(&self) -> Vec { + self.0.lock().unwrap().clone() + } + fn ran(&self, lane: usize) -> Vec { + self.events() + .into_iter() + .filter_map(|event| match event { + Event::Ran(l, id) if l == lane => Some(id), + _ => None, + }) + .collect() + } + fn position(&self, wanted: &Event) -> usize { + self.events() + .iter() + .position(|event| event == wanted) + .unwrap_or_else(|| panic!("missing {wanted:?} in {:?}", self.events())) + } +} + +struct Received { + message: Message, + log: Arc, + unsettled: Arc, +} + +impl Received { + fn id(&self) -> String { + self.message.id().unwrap().to_owned() + } + fn settle(self, event: Event) -> Result<(), TransportError> { + self.unsettled.fetch_sub(1, Ordering::SeqCst); + self.log.push(event); + Ok(()) + } +} + +impl ReceivedMessage for Received { + fn message(&self) -> &Message { + &self.message + } + async fn ack(self) -> Result<(), TransportError> { + let id = self.id(); + self.settle(Event::Ack(id)) + } + async fn nack(self, _reason: &str) -> Result<(), TransportError> { + let id = self.id(); + self.settle(Event::Nack(id)) + } + async fn dead_letter(self, _reason: &str) -> Result<(), TransportError> { + let id = self.id(); + self.settle(Event::DeadLetter(id)) + } +} + +struct Source { + queue: VecDeque, + log: Arc, + independent: bool, + unsettled: Arc, + max_unsettled: Arc, +} + +impl MessageSource for Source { + type Received = Received; + + fn settles_independently(&self) -> bool { + self.independent + } + + async fn recv(&mut self) -> Result, TransportError> { + let Some(message) = self.queue.pop_front() else { + return Ok(None); + }; + let unsettled = self.unsettled.fetch_add(1, Ordering::SeqCst) + 1; + self.max_unsettled.fetch_max(unsettled, Ordering::SeqCst); + self.log + .push(Event::Received(message.id().unwrap().to_owned())); + Ok(Some(Received { + message, + log: self.log.clone(), + unsettled: self.unsettled.clone(), + })) + } +} + +type Behavior = Arc Result<(), TransportError> + Send + Sync>; +type Gate = Arc bool + Send + Sync>; + +/// Lane 0 = "projection" (default), lane 1 = "process". Every message named +/// `both` reaches both lanes; `process.only` reaches lane 1; `unhandled` none. +struct LaneRouter { + log: Arc, + behavior: Behavior, + gate: Gate, +} + +impl MessageRouter for LaneRouter { + fn consumer_group(&self) -> Option<&str> { + Some("lane-test") + } + fn handles(&self, kind: MessageKind, name: &str) -> bool { + !self.lanes_for(kind, name).is_empty() + } + fn subscription_plan(&self) -> crate::bus::SubscriptionPlan { + crate::bus::SubscriptionPlan::default() + } + async fn dispatch(&self, message: &Message) -> Result<(), TransportError> { + for lane in self.lanes_for(message.kind, message.name()).iter() { + self.dispatch_lane(message, None, lane).await?; + } + Ok(()) + } + fn delivery_lanes(&self) -> usize { + 2 + } + fn lanes_for(&self, _kind: MessageKind, name: &str) -> LaneSet { + match name { + "both" => LaneSet::single(0).with(1), + "process.only" => LaneSet::single(1), + _ => LaneSet::EMPTY, + } + } + async fn dispatch_lane( + &self, + message: &Message, + _ordered: Option<&crate::bus::OrderedDelivery>, + lane: usize, + ) -> Result<(), TransportError> { + let gate = self.gate.clone(); + let message_for_gate = message.clone(); + wait_until(move || gate(lane, &message_for_gate)).await; + self.log + .push(Event::Ran(lane, message.id().unwrap().to_owned())); + (self.behavior)(lane, message) + } +} + +fn message(name: &str, id: &str) -> Message { + Message::new(name, MessageKind::Event, b"{}".to_vec()).with_id(id) +} + +struct Harness { + log: Arc, + max_unsettled: Arc, + outcome: Result<(), TransportError>, +} + +fn run_lanes( + messages: Vec, + independent: bool, + options: RunOptions, + behavior: Behavior, + gate: impl Fn(&Arc) -> Gate, +) -> Harness { + let log = Arc::new(Log::default()); + let max_unsettled = Arc::new(AtomicUsize::new(0)); + let router = Arc::new(LaneRouter { + log: log.clone(), + behavior, + gate: gate(&log), + }); + let source = Source { + queue: messages.into_iter().collect(), + log: log.clone(), + independent, + unsettled: Arc::new(AtomicUsize::new(0)), + max_unsettled: max_unsettled.clone(), + }; + let outcome = block_on(run_source(router, source, options)); + Harness { + log, + max_unsettled, + outcome, + } +} + +fn ok() -> Behavior { + Arc::new(|_, _| Ok(())) +} + +fn open() -> impl Fn(&Arc) -> Gate { + |_| Arc::new(|_, _| true) +} + +/// The slow lane-0 route for `m1` finishes only after the process lane has +/// handled every later message, which is impossible if lanes were sequential. +#[test] +fn slow_projection_lane_does_not_delay_process_lane() { + let messages = vec![ + message("both", "m1"), + message("both", "m2"), + message("process.only", "m3"), + ]; + let harness = run_lanes(messages, true, RunOptions::idempotent(), ok(), |log| { + let log = log.clone(); + Arc::new(move |lane, message| { + lane != 0 || message.id() != Some("m1") || log.ran(1).len() == 3 + }) + }); + harness.outcome.unwrap(); + let log = &harness.log; + assert_eq!(log.ran(1), ["m1", "m2", "m3"], "process lane keeps order"); + assert_eq!(log.ran(0), ["m1", "m2"], "projection lane keeps order"); + // m3 (process only) settles before m1, whose slow lane was still running. + assert!(log.position(&Event::Ack("m3".into())) < log.position(&Event::Ack("m1".into()))); + // Every delivery is acknowledged only after all of its lanes ran. + for id in ["m1", "m2"] { + let ack = log.position(&Event::Ack(id.into())); + assert!(log.position(&Event::Ran(0, id.into())) < ack); + assert!(log.position(&Event::Ran(1, id.into())) < ack); + } +} + +#[test] +fn retryable_failure_in_one_lane_nacks_after_every_lane_finished() { + let harness = run_lanes( + vec![message("both", "m1"), message("both", "m2")], + true, + RunOptions::idempotent(), + Arc::new(|lane, message| { + if lane == 1 && message.id() == Some("m1") { + Err(TransportError::retryable("cell unavailable")) + } else { + Ok(()) + } + }), + open(), + ); + harness.outcome.unwrap(); + let log = &harness.log; + let nack = log.position(&Event::Nack("m1".into())); + assert!(log.position(&Event::Ran(0, "m1".into())) < nack); + assert!(log.position(&Event::Ran(1, "m1".into())) < nack); + assert!(!log.events().contains(&Event::Ack("m1".into()))); + assert!(log.events().contains(&Event::Ack("m2".into()))); +} + +#[test] +fn permanent_failure_applies_failure_policy_once_per_delivery() { + let harness = run_lanes( + vec![message("both", "m1")], + true, + RunOptions::idempotent(), + Arc::new(|lane, _| { + if lane == 0 { + Err(TransportError::permanent("rejected")) + } else { + Ok(()) + } + }), + open(), + ); + harness.outcome.unwrap(); + let settled: Vec<_> = harness + .log + .events() + .into_iter() + .filter(|event| matches!(event, Event::Ack(_) | Event::Nack(_) | Event::DeadLetter(_))) + .collect(); + assert_eq!(settled, [Event::DeadLetter("m1".into())]); +} + +#[test] +fn retryable_wins_over_permanent_so_a_recoverable_delivery_is_retained() { + let harness = run_lanes( + vec![message("both", "m1")], + true, + RunOptions::idempotent(), + Arc::new(|lane, _| { + if lane == 0 { + Err(TransportError::permanent("rejected")) + } else { + Err(TransportError::retryable("transient")) + } + }), + open(), + ); + harness.outcome.unwrap(); + assert!(harness.log.events().contains(&Event::Nack("m1".into()))); + assert!(!harness + .log + .events() + .contains(&Event::DeadLetter("m1".into()))); +} + +#[test] +fn window_bounds_unsettled_deliveries() { + let released = Arc::new(AtomicBool::new(false)); + let messages = (1..=6) + .map(|n| message("process.only", &format!("m{n}"))) + .collect(); + let release = released.clone(); + let harness = run_lanes( + messages, + true, + RunOptions::idempotent().with_lane_window(2), + ok(), + move |log| { + let log = log.clone(); + let release = release.clone(); + Arc::new(move |_, _| { + // Hold the lane until the runner has had every chance to + // over-receive, then release it. + let received = log + .events() + .iter() + .filter(|event| matches!(event, Event::Received(_))) + .count(); + if received >= 2 { + release.store(true, Ordering::SeqCst); + } + release.load(Ordering::SeqCst) + }) + }, + ); + harness.outcome.unwrap(); + assert!(released.load(Ordering::SeqCst)); + assert_eq!(harness.max_unsettled.load(Ordering::SeqCst), 2); + assert_eq!(harness.log.ran(1), ["m1", "m2", "m3", "m4", "m5", "m6"]); +} + +#[test] +fn stop_class_failure_halts_only_its_lane_and_retains_queued_deliveries() { + let harness = run_lanes( + vec![ + message("both", "m1"), + message("both", "m2"), + message("both", "m3"), + ], + true, + RunOptions::idempotent(), + Arc::new(|lane, message| { + if lane == 0 && message.id() == Some("m1") { + Err(TransportError::permanent("durable projection failure").retain_and_stop()) + } else { + Ok(()) + } + }), + |log| { + // Keep lane 0 on m1 until lane 1 has queued work behind it. + let log = log.clone(); + Arc::new(move |lane, message| { + lane != 0 || message.id() != Some("m1") || !log.ran(1).is_empty() + }) + }, + ); + let error = harness.outcome.unwrap_err(); + assert!(error.should_retain_and_stop()); + let log = &harness.log; + // The halted lane never runs later deliveries... + assert_eq!(log.ran(0), ["m1"]); + // ...and they are NAKed (retained), never acknowledged. + for id in ["m1", "m2", "m3"] { + let settled = log.events().into_iter().find(|event| { + matches!(event, Event::Ack(x) | Event::Nack(x) | Event::DeadLetter(x) if x == id) + }); + assert!( + matches!(settled, None | Some(Event::Nack(_))), + "{id} must not be acknowledged after its lane halted: {settled:?}" + ); + } + assert!(log.events().contains(&Event::Nack("m1".into()))); +} + +#[test] +fn stop_policy_on_permanent_failure_stops_without_settling_that_delivery() { + let harness = run_lanes( + vec![message("both", "m1")], + true, + RunOptions::idempotent().with_failure_policy(FailurePolicy::Stop), + Arc::new(|lane, _| { + if lane == 1 { + Err(TransportError::permanent("nope")) + } else { + Ok(()) + } + }), + open(), + ); + assert!(harness.outcome.unwrap_err().is_permanent()); + assert!(!harness + .log + .events() + .iter() + .any(|event| matches!(event, Event::Ack(_) | Event::Nack(_) | Event::DeadLetter(_)))); +} + +#[test] +fn unhandled_messages_are_acked_without_running_lanes() { + let harness = run_lanes( + vec![message("unhandled", "m1")], + true, + RunOptions::idempotent(), + ok(), + open(), + ); + harness.outcome.unwrap(); + assert_eq!( + harness.log.events(), + [Event::Received("m1".into()), Event::Ack("m1".into())] + ); +} + +/// A source without independent settlement keeps the sequential contract: +/// one delivery at a time, every lane's routes in order. +#[test] +fn positional_sources_keep_sequential_delivery() { + let harness = run_lanes( + vec![message("both", "m1"), message("both", "m2")], + false, + RunOptions::idempotent(), + ok(), + open(), + ); + harness.outcome.unwrap(); + assert_eq!( + harness.log.events(), + [ + Event::Received("m1".into()), + Event::Ran(0, "m1".into()), + Event::Ran(1, "m1".into()), + Event::Ack("m1".into()), + Event::Received("m2".into()), + Event::Ran(0, "m2".into()), + Event::Ran(1, "m2".into()), + Event::Ack("m2".into()), + ] + ); + assert_eq!(harness.max_unsettled.load(Ordering::SeqCst), 1); +} + +/// Redelivery after a NAK runs every lane again; handlers stay idempotent and +/// the second attempt acknowledges. +#[test] +fn redelivery_after_nack_reruns_all_lanes_then_acks() { + let attempts = Arc::new(AtomicUsize::new(0)); + let seen = attempts.clone(); + let log = Arc::new(Log::default()); + let router = Arc::new(LaneRouter { + log: log.clone(), + behavior: Arc::new(move |lane, _| { + if lane == 1 && seen.fetch_add(1, Ordering::SeqCst) == 0 { + Err(TransportError::retryable("first attempt")) + } else { + Ok(()) + } + }), + gate: Arc::new(|_, _| true), + }); + let source = |messages: Vec| Source { + queue: messages.into_iter().collect(), + log: log.clone(), + independent: true, + unsettled: Arc::new(AtomicUsize::new(0)), + max_unsettled: Arc::new(AtomicUsize::new(0)), + }; + block_on(run_source( + router.clone(), + source(vec![message("both", "m1")]), + RunOptions::idempotent(), + )) + .unwrap(); + // The broker redelivers the NAKed message. + block_on(run_source( + router, + source(vec![message("both", "m1")]), + RunOptions::idempotent(), + )) + .unwrap(); + assert_eq!(log.ran(0), ["m1", "m1"]); + assert_eq!(log.ran(1), ["m1", "m1"]); + let settled: Vec<_> = log + .events() + .into_iter() + .filter(|event| matches!(event, Event::Ack(_) | Event::Nack(_))) + .collect(); + assert_eq!(settled, [Event::Nack("m1".into()), Event::Ack("m1".into())]); +} diff --git a/src/bus/runner/mod.rs b/src/bus/runner/mod.rs index 2e105dd0b..ec865b16d 100644 --- a/src/bus/runner/mod.rs +++ b/src/bus/runner/mod.rs @@ -9,6 +9,8 @@ mod receive_loop; +#[cfg(test)] +mod lane_tests; #[cfg(test)] mod tests; diff --git a/src/bus/runner/receive_loop.rs b/src/bus/runner/receive_loop.rs index c04ce8426..01e120ce7 100644 --- a/src/bus/runner/receive_loop.rs +++ b/src/bus/runner/receive_loop.rs @@ -1,9 +1,14 @@ -use std::future::Future; +use std::collections::{HashMap, VecDeque}; +use std::future::{poll_fn, Future}; +use std::pin::Pin; use std::sync::Arc; +use std::task::Poll; + +use futures_util::stream::{FuturesUnordered, StreamExt}; use crate::bus::source::{MessageSource, ReceivedMessage}; use crate::bus::{FailureAction, MessageRouter, RunOptions, TransportError, TransportErrorKind}; -use crate::bus::{Message, MessageKind}; +use crate::bus::{Message, MessageKind, OrderedDelivery}; /// Run the receive loop for a direct transport source. /// @@ -29,6 +34,14 @@ use crate::bus::{Message, MessageKind}; /// swallowed: a returned `Err` ends the run and the supervisor may restart it /// (already-committed effects make redelivery safe). /// +/// When the router declares several [delivery lanes](MessageRouter::delivery_lanes) +/// and the source [settles independently](MessageSource::settles_independently), +/// lanes run concurrently while each keeps receive order, and a delivery is +/// settled once every lane that received it has finished. The number of +/// unsettled deliveries is bounded by [`RunOptions::lane_window`]. See +/// `docs/consumer-delivery-lanes.md`. Otherwise every message runs all of its +/// routes before the next message is received. +/// /// Inbox note: until the consumer-inbox subtask lands, inbox mode enforces the /// stable-id requirement and then dispatches like idempotent mode. The /// receipt-commit wrapping that makes it effectively-once is added there. @@ -36,6 +49,22 @@ use crate::bus::{Message, MessageKind}; /// `I: Send` keeps the returned future `Send` so the runner can be spawned on a /// multi-threaded executor regardless of the inbox hook type. pub async fn run_source( + router: Arc, + source: S, + options: RunOptions, +) -> Result<(), TransportError> +where + R: MessageRouter, + S: MessageSource, + I: Send, +{ + if router.delivery_lanes() > 1 && source.settles_independently() { + return LaneRun::new(router, source, options).run().await; + } + run_sequential(router, source, options).await +} + +async fn run_sequential( router: Arc, mut source: S, options: RunOptions, @@ -52,192 +81,547 @@ where let Some(received) = recv_next(&mut source, service, transport).await? else { break; }; - - // A delivery the transport could not decode is a permanent failure: it - // carries no valid message to dispatch, and it must NOT be treated as an - // empty message (which would route to ack-and-ignore below and silently - // drop a corrupt row). Route it through the failure policy directly, the - // same as a permanent dispatch failure, so it is dead-lettered/parked. - if let Some(error) = received.decode_error() { - let action = options.failure_policy.resolve(error); - record_transport_failure(service, transport, error.kind(), action); - let kind = received.message().kind; - match action { - FailureAction::Nack => { - let reason = error.to_string(); - settle_and_record( - service, - transport, - kind, - crate::telemetry::transport_outcome::NACK, - crate::telemetry::transport_outcome::NACK, - || received.nack(&reason), - ) - .await?; - } - FailureAction::DeadLetter => { - let reason = error.to_string(); - settle_and_record( - service, - transport, - kind, - crate::telemetry::transport_outcome::DEAD_LETTER, - crate::telemetry::transport_outcome::DEAD_LETTER, - || received.dead_letter(&reason), - ) - .await?; - } - FailureAction::Park => { - let reason = error.to_string(); - settle_and_record( - service, - transport, - kind, - crate::telemetry::transport_outcome::PARK, - crate::telemetry::transport_outcome::PARK, - || received.park(&reason), - ) - .await?; - } - FailureAction::LogAndAck => { - eprintln!("[bus::runner] dropping undecodable message after permanent failure: {error}"); - settle_and_record( - service, - transport, - kind, - crate::telemetry::transport_outcome::ACK, - crate::telemetry::transport_outcome::LOG_AND_ACK, - || received.ack(), - ) - .await?; - } - FailureAction::Stop => return Err(TransportError::permanent(error.to_string())), - } + if received.decode_error().is_some() { + settle_permanent_decode(service, transport, &options, received).await?; continue; } - // No handler for this message: intentionally ignore (ack) rather than - // dead-letter, so unrelated fan-out events don't pile into the DLQ. if !router.handles(received.message().kind, received.message().name()) { - let kind = received.message().kind; + settle_ignored(service, transport, received).await?; + continue; + } + let kind = received.message().kind; + let result = dispatch( + router.as_ref(), + &options, + received.message(), + received.ordered_delivery(), + ) + .await; + settle_result(service, transport, &options, received, kind, result).await?; + } + Ok(()) +} + +/// A delivery the transport could not decode is a permanent failure: it +/// carries no valid message to dispatch, and it must NOT be treated as an +/// empty message (which would route to ack-and-ignore and silently drop a +/// corrupt row). Route it through the failure policy directly, the same as a +/// permanent dispatch failure, so it is dead-lettered/parked. +async fn settle_permanent_decode( + service: Option<&str>, + transport: &str, + options: &RunOptions, + received: M, +) -> Result<(), TransportError> { + let Some(error) = received.decode_error() else { + return Ok(()); + }; + let action = options.failure_policy.resolve(error); + record_transport_failure(service, transport, error.kind(), action); + let kind = received.message().kind; + let reason = error.to_string(); + match action { + FailureAction::Nack => { + settle_and_record( + service, + transport, + kind, + crate::telemetry::transport_outcome::NACK, + crate::telemetry::transport_outcome::NACK, + || received.nack(&reason), + ) + .await + } + FailureAction::DeadLetter => { + settle_and_record( + service, + transport, + kind, + crate::telemetry::transport_outcome::DEAD_LETTER, + crate::telemetry::transport_outcome::DEAD_LETTER, + || received.dead_letter(&reason), + ) + .await + } + FailureAction::Park => { + settle_and_record( + service, + transport, + kind, + crate::telemetry::transport_outcome::PARK, + crate::telemetry::transport_outcome::PARK, + || received.park(&reason), + ) + .await + } + FailureAction::LogAndAck => { + eprintln!( + "[bus::runner] dropping undecodable message after permanent failure: {reason}" + ); settle_and_record( service, transport, kind, crate::telemetry::transport_outcome::ACK, - crate::telemetry::transport_outcome::IGNORED, + crate::telemetry::transport_outcome::LOG_AND_ACK, || received.ack(), ) + .await + } + FailureAction::Stop => Err(TransportError::permanent(reason)), + } +} + +/// No handler for this message: intentionally ignore (ack) rather than +/// dead-letter, so unrelated fan-out events don't pile into the DLQ. +async fn settle_ignored( + service: Option<&str>, + transport: &str, + received: M, +) -> Result<(), TransportError> { + let kind = received.message().kind; + settle_and_record( + service, + transport, + kind, + crate::telemetry::transport_outcome::ACK, + crate::telemetry::transport_outcome::IGNORED, + || received.ack(), + ) + .await +} + +/// Settle one dispatched delivery exactly as the sequential runner always has. +/// +/// Returns `Err` when the run must end: a retain-and-stop failure (after NAKing +/// the exact delivery), a `Stop` policy (without settling), or a settle error. +async fn settle_result( + service: Option<&str>, + transport: &str, + options: &RunOptions, + received: M, + kind: MessageKind, + result: Result<(), TransportError>, +) -> Result<(), TransportError> { + match result { + Ok(()) => { + settle_and_record( + service, + transport, + kind, + crate::telemetry::transport_outcome::ACK, + crate::telemetry::transport_outcome::ACK, + || received.ack(), + ) + .await + } + Err(error) if error.should_retain_and_stop() => { + record_transport_failure( + service, + transport, + error.kind(), + crate::telemetry::transport_outcome::NACK, + ); + let reason = error.to_string(); + settle_and_record( + service, + transport, + kind, + crate::telemetry::transport_outcome::NACK, + crate::telemetry::transport_outcome::NACK, + || received.nack(&reason), + ) .await?; - continue; + Err(error) } - let kind = received.message().kind; - match dispatch( - router.as_ref(), - &options, - received.message(), - received.ordered_delivery(), - ) - .await - { - Ok(()) => { + Err(error) => match options.failure_policy.resolve(&error) { + action @ FailureAction::Nack => { + record_transport_failure(service, transport, error.kind(), action); + let reason = error.to_string(); settle_and_record( service, transport, kind, - crate::telemetry::transport_outcome::ACK, - crate::telemetry::transport_outcome::ACK, - || received.ack(), + crate::telemetry::transport_outcome::NACK, + crate::telemetry::transport_outcome::NACK, + || received.nack(&reason), + ) + .await + } + action @ FailureAction::DeadLetter => { + record_transport_failure(service, transport, error.kind(), action); + let reason = error.to_string(); + settle_and_record( + service, + transport, + kind, + crate::telemetry::transport_outcome::DEAD_LETTER, + crate::telemetry::transport_outcome::DEAD_LETTER, + || received.dead_letter(&reason), + ) + .await + } + action @ FailureAction::Park => { + record_transport_failure(service, transport, error.kind(), action); + let reason = error.to_string(); + settle_and_record( + service, + transport, + kind, + crate::telemetry::transport_outcome::PARK, + crate::telemetry::transport_outcome::PARK, + || received.park(&reason), ) - .await?; + .await } - Err(error) if error.should_retain_and_stop() => { + FailureAction::LogAndAck => { record_transport_failure( service, transport, error.kind(), - crate::telemetry::transport_outcome::NACK, + FailureAction::LogAndAck, + ); + eprintln!( + "[bus::runner] dropping message '{}' after permanent failure: {error}", + received.message().name() ); - let reason = error.to_string(); settle_and_record( service, transport, kind, - crate::telemetry::transport_outcome::NACK, - crate::telemetry::transport_outcome::NACK, - || received.nack(&reason), + crate::telemetry::transport_outcome::ACK, + crate::telemetry::transport_outcome::LOG_AND_ACK, + || received.ack(), ) - .await?; - return Err(error); + .await + } + FailureAction::Stop => { + record_transport_failure(service, transport, error.kind(), FailureAction::Stop); + Err(error) + } + }, + } +} + +// --- delivery lanes ------------------------------------------------------ + +type LaneDone = (u64, usize, Result<(), TransportError>); +type LaneFuture<'r> = Pin + Send + 'r>>; +type RecvOutput = ( + S, + Result::Received>, TransportError>, +); +type RecvFuture<'s, S> = Pin> + Send + 's>>; + +struct InFlight { + received: M, + message: Arc, + ordered: Option, + kind: MessageKind, + remaining: usize, + stop: Option, + retryable: Option, + permanent: Option, +} + +enum Step { + Received(RecvOutput), + Lane(LaneDone), +} + +/// Concurrent lanes over one ordered source. See `docs/consumer-delivery-lanes.md`. +struct LaneRun<'r, R, S: MessageSource + 'r, I> { + router: Arc, + options: RunOptions, + transport: &'static str, + source: Option, + receiving: Option>, + running: FuturesUnordered>, + lane_busy: Vec, + lane_halted: Vec, + queues: Vec>, + in_flight: HashMap>, + next_sequence: u64, + /// Set when the source drained (`Ok(None)`) or failed to receive. + source_finished: Option>, + stop_error: Option, +} + +impl<'r, R, S, I> LaneRun<'r, R, S, I> +where + R: MessageRouter + 'r, + S: MessageSource + 'r, + I: Send, +{ + fn new(router: Arc, source: S, options: RunOptions) -> Self { + let lanes = router.delivery_lanes().min(crate::bus::LaneSet::MAX_LANES); + let transport = source.transport_name(); + Self { + router, + options, + transport, + source: Some(source), + receiving: None, + running: FuturesUnordered::new(), + lane_busy: vec![false; lanes], + lane_halted: vec![false; lanes], + queues: vec![VecDeque::new(); lanes], + in_flight: HashMap::new(), + next_sequence: 0, + source_finished: None, + stop_error: None, + } + } + + fn service(&self) -> Option { + self.router.consumer_group().map(str::to_owned) + } + + fn stopping(&self) -> bool { + self.stop_error.is_some() || self.source_finished.is_some() + } + + async fn run(mut self) -> Result<(), TransportError> { + loop { + self.start_idle_lanes().await; + if self.receiving.is_none() + && !self.stopping() + && self.in_flight.len() < self.options.lane_window.max(1) + { + if let Some(mut source) = self.source.take() { + self.receiving = Some(Box::pin(async move { + let result = source.recv().await; + (source, result) + })); + } } - Err(error) => match options.failure_policy.resolve(&error) { - action @ FailureAction::Nack => { - record_transport_failure(service, transport, error.kind(), action); - let reason = error.to_string(); - settle_and_record( - service, - transport, - kind, - crate::telemetry::transport_outcome::NACK, - crate::telemetry::transport_outcome::NACK, - || received.nack(&reason), - ) - .await?; + if self.in_flight.is_empty() && self.receiving.is_none() && self.stopping() { + if let Some(error) = self.stop_error.take() { + return Err(error); } - action @ FailureAction::DeadLetter => { - record_transport_failure(service, transport, error.kind(), action); - let reason = error.to_string(); - settle_and_record( - service, - transport, - kind, - crate::telemetry::transport_outcome::DEAD_LETTER, - crate::telemetry::transport_outcome::DEAD_LETTER, - || received.dead_letter(&reason), - ) - .await?; + return self.source_finished.take().unwrap_or(Ok(())); + } + + let running = &mut self.running; + let receiving = &mut self.receiving; + let step: Step = poll_fn(|cx| { + if let Poll::Ready(Some(done)) = running.poll_next_unpin(cx) { + return Poll::Ready(Step::Lane(done)); } - action @ FailureAction::Park => { - record_transport_failure(service, transport, error.kind(), action); - let reason = error.to_string(); - settle_and_record( - service, - transport, - kind, - crate::telemetry::transport_outcome::PARK, - crate::telemetry::transport_outcome::PARK, - || received.park(&reason), - ) - .await?; + if let Some(future) = receiving.as_mut() { + if let Poll::Ready(output) = future.as_mut().poll(cx) { + return Poll::Ready(Step::Received(output)); + } } - FailureAction::LogAndAck => { - record_transport_failure( - service, - transport, - error.kind(), - FailureAction::LogAndAck, - ); - eprintln!( - "[bus::runner] dropping message '{}' after permanent failure: {error}", - received.message().name() - ); - settle_and_record( - service, - transport, - kind, - crate::telemetry::transport_outcome::ACK, - crate::telemetry::transport_outcome::LOG_AND_ACK, - || received.ack(), - ) - .await?; + Poll::Pending + }) + .await; + match step { + Step::Received((source, result)) => { + self.receiving = None; + self.source = Some(source); + self.accept(result).await; } - FailureAction::Stop => { - record_transport_failure(service, transport, error.kind(), FailureAction::Stop); - return Err(error); + Step::Lane((sequence, lane, result)) => { + self.lane_busy[lane] = false; + self.finish_lane(sequence, lane, result).await; } + } + } + } + + /// Start the next queued delivery on every idle lane. A halted lane NAKs + /// its queued deliveries instead of running them. + async fn start_idle_lanes(&mut self) { + for lane in 0..self.queues.len() { + while self.lane_halted[lane] { + let Some(sequence) = self.queues[lane].pop_front() else { + break; + }; + let halted = TransportError::retryable( + "delivery lane halted after a stop-class failure; delivery retained", + ); + self.finish_lane(sequence, lane, Err(halted)).await; + } + if self.lane_busy[lane] { + continue; + } + let Some(sequence) = self.queues[lane].pop_front() else { + continue; + }; + let Some(entry) = self.in_flight.get(&sequence) else { + continue; + }; + let router = Arc::clone(&self.router); + let message = Arc::clone(&entry.message); + let ordered = entry.ordered.clone(); + self.lane_busy[lane] = true; + self.running.push(Box::pin(async move { + let result = dispatch_lane(router.as_ref(), &message, ordered.as_ref(), lane).await; + (sequence, lane, result) + })); + } + } + + async fn accept(&mut self, result: Result, TransportError>) { + let service = self.service(); + let service = service.as_deref(); + let received = match result { + Ok(Some(received)) => received, + Ok(None) => { + self.source_finished = Some(Ok(())); + return; + } + Err(error) => { + record_transport_failure( + service, + self.transport, + error.kind(), + crate::telemetry::failure_action::RECV_ERROR, + ); + self.source_finished = Some(Err(error)); + return; + } + }; + if received.decode_error().is_some() { + let outcome = + settle_permanent_decode(service, self.transport, &self.options, received).await; + self.record_stop(outcome); + return; + } + let kind = received.message().kind; + let lanes = self.router.lanes_for(kind, received.message().name()); + if lanes.is_empty() { + let outcome = settle_ignored(service, self.transport, received).await; + self.record_stop(outcome); + return; + } + if let Err(error) = self.options.validate_message_id(received.message()) { + let error = TransportError::permanent(error.to_string()).with_source(error); + let outcome = settle_result( + service, + self.transport, + &self.options, + received, + kind, + Err(error), + ) + .await; + self.record_stop(outcome); + return; + } + let sequence = self.next_sequence; + self.next_sequence += 1; + let message = Arc::new(received.message().clone()); + let ordered = received.ordered_delivery().cloned(); + let mut remaining = 0; + let lane_count = self.queues.len(); + for lane in lanes.iter().filter(|lane| *lane < lane_count) { + self.queues[lane].push_back(sequence); + remaining += 1; + } + self.in_flight.insert( + sequence, + InFlight { + received, + message, + ordered, + kind, + remaining, + stop: None, + retryable: None, + permanent: None, }, + ); + if remaining == 0 { + // Lanes beyond this runner's range cannot run; keep it retryable. + let error = TransportError::retryable("delivery lane out of range"); + self.settle_entry(sequence, Some(error)).await; } } - Ok(()) + + async fn finish_lane( + &mut self, + sequence: u64, + lane: usize, + result: Result<(), TransportError>, + ) { + let stops = |error: &TransportError, options: &RunOptions| { + error.should_retain_and_stop() + || options.failure_policy.resolve(error) == FailureAction::Stop + }; + let Some(entry) = self.in_flight.get_mut(&sequence) else { + return; + }; + entry.remaining = entry.remaining.saturating_sub(1); + if let Err(error) = result { + if stops(&error, &self.options) { + self.lane_halted[lane] = true; + entry.stop.get_or_insert(error); + } else if error.is_retryable() { + entry.retryable.get_or_insert(error); + } else { + entry.permanent.get_or_insert(error); + } + } + if entry.remaining == 0 { + self.settle_entry(sequence, None).await; + } + } + + /// Settle a delivery whose lanes all finished. A stop-class failure wins, + /// then any retryable failure (so a delivery that may still succeed is + /// redelivered), then a permanent one. + async fn settle_entry(&mut self, sequence: u64, extra: Option) { + let Some(entry) = self.in_flight.remove(&sequence) else { + return; + }; + let error = entry.stop.or(entry.retryable).or(extra).or(entry.permanent); + let service = self.service(); + let outcome = settle_result( + service.as_deref(), + self.transport, + &self.options, + entry.received, + entry.kind, + error.map_or(Ok(()), Err), + ) + .await; + self.record_stop(outcome); + } + + fn record_stop(&mut self, outcome: Result<(), TransportError>) { + if let Err(error) = outcome { + self.stop_error.get_or_insert(error); + } + } +} + +async fn dispatch_lane( + router: &R, + message: &Message, + ordered: Option<&OrderedDelivery>, + lane: usize, +) -> Result<(), TransportError> { + #[cfg(feature = "otel")] + { + use tracing::Instrument as _; + + let span = transport_receive_span(message); + crate::trace_context::set_span_parent_from_metadata_if_no_current_span( + &span, + &message.metadata, + ); + return router + .dispatch_lane(message, ordered, lane) + .instrument(span) + .await; + } + + #[cfg(not(feature = "otel"))] + { + router.dispatch_lane(message, ordered, lane).await + } } async fn settle_and_record( diff --git a/src/bus/runner/tests.rs b/src/bus/runner/tests.rs index a31074c26..5319d8c0b 100644 --- a/src/bus/runner/tests.rs +++ b/src/bus/runner/tests.rs @@ -207,6 +207,54 @@ fn run_with( } // --- tests -------------------------------------------------------------- +#[test] +fn application_reload_retains_exact_delivery_then_succeeds_after_activation() { + use std::sync::atomic::{AtomicBool, Ordering}; + let recorder = Recorder::new(); + let open = Arc::new(AtomicBool::new(false)); + let gate = open.clone(); + let effects = recorder.clone(); + let router = Arc::new( + Handlers::new().on_event("reload", move |message: &Message| { + let admitted = gate.load(Ordering::SeqCst); + let effects = effects.clone(); + let id = message.id().unwrap().to_owned(); + async move { + if !admitted { + return Err(crate::microsvc::HandlerError::ApplicationReloading.into()); + } + effects.push(Event::Handled(id)); + Ok(()) + } + }), + ); + let message = event_message("reload", Some("retained-event")); + let source = || FakeSource { + queue: vec![message.clone()].into_iter().collect(), + recorder: recorder.clone(), + settle_ok: true, + recv_error: false, + decode_error: false, + }; + let error = block_on(run_source( + router.clone(), + source(), + RunOptions::idempotent(), + )) + .unwrap_err(); + assert!(error.is_retryable()); + assert!(error.should_retain_and_stop()); + assert!( + matches!(recorder.events().as_slice(), [Event::Nack(reason)] if reason.contains("reloading")) + ); + open.store(true, Ordering::SeqCst); + block_on(run_source(router, source(), RunOptions::idempotent())).unwrap(); + assert_eq!( + &recorder.events()[1..], + &[Event::Handled("retained-event".into()), Event::Ack] + ); +} + #[test] fn success_dispatches_then_acks_in_order() { let result = run(vec![event_message("ok", None)], RunOptions::idempotent()); diff --git a/src/bus/source.rs b/src/bus/source.rs index a96c10bb3..8d4406382 100644 --- a/src/bus/source.rs +++ b/src/bus/source.rs @@ -33,6 +33,17 @@ pub trait MessageSource: Send { "unknown" } + /// Whether each delivery can be settled independently of the others, in + /// any order, while later deliveries are still being received. + /// + /// True only for per-message acknowledgements (for example NATS JetStream + /// explicit acks). Positional (offset commit) and lease-based sources keep + /// the default `false`, and the runner then processes one delivery at a + /// time even when the router has several delivery lanes. + fn settles_independently(&self) -> bool { + false + } + /// Receive the next message, if any. fn recv( &mut self, diff --git a/src/domain_event/derived_tests.rs b/src/domain_event/derived_tests.rs index fd71e4bd7..b46a9e711 100644 --- a/src/domain_event/derived_tests.rs +++ b/src/domain_event/derived_tests.rs @@ -24,6 +24,111 @@ crate::projection! { on { events: [Indexed], mutation: SaveDerived, input: { row: body }, }, }; } +crate::projection! { + const EXTERNAL_ROWS: ProjectionDescriptor = { + name: "external-source-rows", version: 1, epoch: "external-v1", model: DerivedRows, + source: external_snapshot, + on { events: [Indexed], mutation: SaveDerived, input: { row: body }, }, + }; +} + +fn external(position: u64, bytes: u64) -> DomainEventOccurrence { + DomainEventOccurrence::capture_external( + ExternalEventSource { + producer: "external.ledger".into(), + stream: "source:account".into(), + position, + key: "balance".into(), + }, + std::time::UNIX_EPOCH, + Default::default(), + &Indexed { bytes }, + ) + .unwrap() +} + +#[test] +fn external_occurrence_has_no_aggregate_and_stable_content_independent_identity() { + let first = external(7, 10); + let retry = external(7, 10); + let conflict = external(7, 11); + assert_eq!(first, retry); + assert_eq!(first.id(), conflict.id()); + assert_ne!( + first.canonical_bytes().unwrap(), + conflict.canonical_bytes().unwrap() + ); + assert_ne!(first.id(), external(8, 10).id()); + let wire = serde_json::to_value(&first).unwrap(); + assert!(wire.get("aggregate_type").is_none()); + assert!(wire.get("aggregate_id").is_none()); + assert!(wire.get("aggregate_sequence").is_none()); + assert_eq!( + DomainEventOccurrence::from_canonical_bytes(&first.canonical_bytes().unwrap()).unwrap(), + first + ); + let outbox = crate::OutboxMessage::from_domain_event_occurrence(&first).unwrap(); + assert_eq!(outbox.domain_event_occurrence().unwrap(), first); + let derived = first + .derive("indexer", "summary", &Indexed { bytes: 12 }) + .unwrap(); + assert_eq!(derived.external_source(), first.external_source()); + assert_eq!( + DomainEventOccurrence::from_canonical_bytes(&derived.canonical_bytes().unwrap()).unwrap(), + derived + ); +} + +#[test] +fn external_source_snapshots_fence_positions_conflicts_and_origin() { + use crate::projection_protocol::SourceSnapshotVersion as Version; + let first = external(7, 10); + let fence = Version::from_occurrence(&first).unwrap(); + assert!(!Version::from_occurrence(&external(6, 9)) + .unwrap() + .advances(&fence) + .unwrap()); + assert!(!Version::from_occurrence(&first) + .unwrap() + .advances(&fence) + .unwrap()); + assert!(Version::from_occurrence(&external(8, 11)) + .unwrap() + .advances(&fence) + .unwrap()); + assert!(Version::from_occurrence(&external(7, 11)) + .unwrap() + .advances(&fence) + .is_err()); + assert!(Version::from_occurrence(&source()) + .unwrap() + .advances(&fence) + .is_err()); + assert!(EXTERNAL_ROWS + .server_executor() + .unwrap() + .plan(&first) + .is_ok()); + assert!(EXTERNAL_ROWS + .server_executor() + .unwrap() + .plan(&source()) + .is_err()); + assert!(SNAPSHOT_ROWS + .server_executor() + .unwrap() + .plan(&first) + .is_err()); + assert!(DERIVED_ROWS.server_executor().unwrap().plan(&first).is_ok()); + let derived = first + .derive("indexer", "summary", &Indexed { bytes: 12 }) + .unwrap(); + assert!(EXTERNAL_ROWS + .server_executor() + .unwrap() + .plan(&derived) + .is_err()); +} #[derive(Serialize)] struct Indexed { diff --git a/src/domain_event/mod.rs b/src/domain_event/mod.rs index 0843f9e62..ccdb1bfd2 100644 --- a/src/domain_event/mod.rs +++ b/src/domain_event/mod.rs @@ -22,7 +22,7 @@ pub use descriptor::{ pub use occurrence::{ DomainEventCaptureError, DomainEventCaptureOutcome, DomainEventCapturePoison, DomainEventCommitGuardError, DomainEventDerivation, DomainEventEnvelope, DomainEventOccurrence, - DOMAIN_EVENT_OCCURRENCE_VERSION, + ExternalEventSource, DOMAIN_EVENT_OCCURRENCE_VERSION, }; pub(crate) use canonical::canonical_json_bytes; diff --git a/src/domain_event/occurrence.rs b/src/domain_event/occurrence.rs index 592bb11dd..15c52175a 100644 --- a/src/domain_event/occurrence.rs +++ b/src/domain_event/occurrence.rs @@ -22,6 +22,21 @@ use super::{ /// Version of the canonical [`DomainEventOccurrence`] envelope. pub const DOMAIN_EVENT_OCCURRENCE_VERSION: u16 = 1; +/// Provenance of a committed fact supplied by an authenticated source adapter. +/// This is not an aggregate identity or a claim that a domain command ran. +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct ExternalEventSource { + /// Configured, authenticated producer identity (not a user input). + pub producer: String, + /// Immutable source/stream identity; replacing a source requires a new ID. + pub stream: String, + /// Source-owned position, never a broker delivery position. + pub position: u64, + /// Stable member identity within a source position, e.g. a ref in a push. + pub key: String, +} + /// Provenance of a typed fact produced while handling an earlier occurrence. /// The envelope's aggregate fields still describe the originating transition, /// not a new aggregate commit by this producer. @@ -56,6 +71,8 @@ pub struct DomainEventEnvelope { /// One exact typed outward event captured at aggregate transition time. #[derive(Clone, PartialEq, Eq, Serialize)] pub struct DomainEventOccurrence { + #[serde(skip_serializing_if = "Option::is_none")] + external: Option, #[serde(skip_serializing_if = "Option::is_none")] derivation: Option, /// Canonical occurrence envelope version. @@ -65,10 +82,13 @@ pub struct DomainEventOccurrence { /// Semantic event and body schema. descriptor: DomainEventDescriptor, /// Stable aggregate type name. + #[serde(skip_serializing_if = "String::is_empty")] aggregate_type: String, /// Stable aggregate stream identifier. + #[serde(skip_serializing_if = "String::is_empty")] aggregate_id: String, /// Aggregate event sequence that caused this occurrence. + #[serde(skip_serializing_if = "is_zero")] aggregate_sequence: u64, /// Zero-based publication position within one aggregate sequence. publication_ordinal: u32, @@ -129,6 +149,7 @@ impl DomainEventOccurrence { validate_stable_message_id(Some(&id)).map_err(DomainEventCaptureError::OccurrenceId)?; let occurrence = Self { + external: None, derivation: None, occurrence_version: DOMAIN_EVENT_OCCURRENCE_VERSION, id, @@ -145,6 +166,51 @@ impl DomainEventOccurrence { Ok(occurrence) } + /// Capture a typed external fact after the source adapter authenticates + /// its bytes and resolves trusted routing. This constructor validates + /// structure, not authentication: it must not be exposed as a user command. + /// The stable identity excludes payload bytes so persistent input evidence + /// can reject a changed payload at an existing source position/key. + pub fn capture_external( + source: ExternalEventSource, + occurred_at: SystemTime, + metadata: BTreeMap, + body: &T, + ) -> Result { + validate_external(&source)?; + let descriptor = T::DESCRIPTOR.clone(); + validate_descriptor(&descriptor)?; + if descriptor.body.kind != DomainEventBodyKind::Event { + return Err(DomainEventCaptureError::InvalidExternalSource); + } + let result = Self { + id: external_id(&source), + external: Some(source), + derivation: None, + occurrence_version: DOMAIN_EVENT_OCCURRENCE_VERSION, + descriptor, + aggregate_type: String::new(), + aggregate_id: String::new(), + aggregate_sequence: 0, + publication_ordinal: 0, + occurred_at_unix_ms: occurred_at + .duration_since(UNIX_EPOCH) + .map_err(|_| DomainEventCaptureError::TimestampBeforeUnixEpoch)? + .as_millis() + .try_into() + .map_err(|_| DomainEventCaptureError::TimestampOverflow)?, + body: canonical_json_bytes(body)?, + metadata, + }; + result.canonical_bytes()?; + Ok(result) + } + + /// Explicit external provenance, including for derived external facts. + pub fn external_source(&self) -> Option<&ExternalEventSource> { + self.external.as_ref() + } + /// Return the immutable canonical body bytes. pub fn body_bytes(&self) -> &[u8] { &self.body @@ -218,6 +284,9 @@ impl DomainEventOccurrence { } hash.update(self.aggregate_sequence.to_be_bytes()); hash.update(self.publication_ordinal.to_be_bytes()); + if let Some(source) = &self.external { + hash_component(&mut hash, &canonical_json_bytes(source)?); + } hash.update(self.occurred_at_unix_ms.to_be_bytes()); hash_component(&mut hash, &canonical_json_bytes(&self.descriptor)?); hash_component(&mut hash, &self.body); @@ -332,6 +401,7 @@ impl DomainEventOccurrence { let wire: DomainEventOccurrenceWire = serde_json::from_slice(bytes) .map_err(|error| DomainEventCaptureError::OccurrenceDecoding(error.to_string()))?; let occurrence = Self { + external: wire.external, derivation: wire.derivation, occurrence_version: wire.occurrence_version, id: wire.id, @@ -359,15 +429,27 @@ impl DomainEventOccurrence { }); } validate_descriptor(&self.descriptor)?; - validate_message_name(&self.aggregate_type) - .map_err(DomainEventCaptureError::AggregateType)?; - validate_stable_message_id(Some(&self.aggregate_id)) - .map_err(DomainEventCaptureError::AggregateId)?; + if let Some(source) = &self.external { + validate_external(source)?; + if !self.aggregate_type.is_empty() + || !self.aggregate_id.is_empty() + || self.aggregate_sequence != 0 + || self.publication_ordinal != 0 + || self.descriptor.body.kind != DomainEventBodyKind::Event + { + return Err(DomainEventCaptureError::InvalidExternalSource); + } + } else { + validate_message_name(&self.aggregate_type) + .map_err(DomainEventCaptureError::AggregateType)?; + validate_stable_message_id(Some(&self.aggregate_id)) + .map_err(DomainEventCaptureError::AggregateId)?; + if self.aggregate_sequence == 0 { + return Err(DomainEventCaptureError::ZeroAggregateSequence); + } + } validate_stable_message_id(Some(&self.id)) .map_err(DomainEventCaptureError::OccurrenceId)?; - if self.aggregate_sequence == 0 { - return Err(DomainEventCaptureError::ZeroAggregateSequence); - } if self.body.len() > MAX_DOMAIN_EVENT_BODY_BYTES { return Err(DomainEventCaptureError::BodyTooLarge { len: self.body.len(), @@ -388,6 +470,8 @@ impl DomainEventOccurrence { }; let expected_id = if self.derivation.is_some() { self.derived_identity()? + } else if let Some(source) = &self.external { + external_id(source) } else { occurrence_id(&self.descriptor, &envelope) }; @@ -407,13 +491,18 @@ impl DomainEventOccurrence { #[derive(Deserialize)] struct DomainEventOccurrenceWire { + #[serde(default)] + external: Option, #[serde(default)] derivation: Option, occurrence_version: u16, id: String, descriptor: DomainEventDescriptor, + #[serde(default)] aggregate_type: String, + #[serde(default)] aggregate_id: String, + #[serde(default)] aggregate_sequence: u64, publication_ordinal: u32, occurred_at_unix_ms: u64, @@ -422,6 +511,33 @@ struct DomainEventOccurrenceWire { metadata: BTreeMap, } +fn is_zero(value: &u64) -> bool { + *value == 0 +} + +fn validate_external(source: &ExternalEventSource) -> Result<(), DomainEventCaptureError> { + if validate_message_name(&source.producer).is_err() + || validate_stable_message_id(Some(&source.stream)).is_err() + || source.position == 0 + || source.key.is_empty() + || source.key.len() > 1024 + || source.key.chars().any(char::is_control) + { + return Err(DomainEventCaptureError::InvalidExternalSource); + } + Ok(()) +} + +fn external_id(source: &ExternalEventSource) -> String { + let mut digest = Sha256::new(); + digest.update(b"distributed.external-event.occurrence/v1\0"); + for value in [&source.producer, &source.stream, &source.key] { + hash_component(&mut digest, value.as_bytes()); + } + digest.update(source.position.to_be_bytes()); + format!("ee1:sha256:{:x}", digest.finalize()) +} + fn occurrence_id(descriptor: &DomainEventDescriptor, envelope: &DomainEventEnvelope) -> String { let mut digest = Sha256::new(); digest.update(b"distributed.domain-event.occurrence/v1\0"); @@ -489,6 +605,8 @@ fn validate_fingerprint(fingerprint: &str) -> Result<(), DomainEventCaptureError #[derive(Clone, Debug, PartialEq, Eq)] #[non_exhaustive] pub enum DomainEventCaptureError { + /// Invalid external provenance or mixed aggregate/external envelope. + InvalidExternalSource, /// Derived provenance has an invalid producer, output key or source identity. InvalidDerivation, /// Semantic event name violated transport naming rules. @@ -569,6 +687,7 @@ pub enum DomainEventCaptureError { impl fmt::Display for DomainEventCaptureError { fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { match self { + Self::InvalidExternalSource => formatter.write_str("invalid external-event source"), Self::InvalidDerivation => formatter.write_str("invalid domain-event derivation"), Self::EventName(error) => write!(formatter, "invalid domain-event name: {error}"), Self::AggregateType(error) => write!(formatter, "invalid aggregate type: {error}"), diff --git a/src/graphql/client_manifest/export.rs b/src/graphql/client_manifest/export.rs index a7be2ffb9..141c2fed4 100644 --- a/src/graphql/client_manifest/export.rs +++ b/src/graphql/client_manifest/export.rs @@ -6,6 +6,9 @@ pub struct DistributedClientSurfaceExport { identity: ClientSurfaceIdentity, surface: Arc, execution: ClientExecutionLimits, + // All inputs are immutable and private. Clones share only compiled metadata, + // never request authority, preset values, tokens or read results. + manifest: Arc>>, } /// Do not transitively format the selected Surface: it retains a private full @@ -32,6 +35,7 @@ impl DistributedClientSurfaceExport { identity, surface: surface.into(), execution, + manifest: Arc::new(std::sync::OnceLock::new()), } } @@ -116,18 +120,32 @@ impl DistributedClientSurfaceExport { } pub fn manifest(&self) -> Result { - client_manifest_from_surface_with_execution( - &self.service_id, - self.identity.clone(), - &self.surface, - self.execution.clone(), - ) + self.manifest_ref().cloned() + } + + pub(crate) fn manifest_ref(&self) -> Result<&DistributedClientManifest, ClientManifestError> { + self.manifest + .get_or_init(|| { + client_manifest_from_surface_with_execution( + &self.service_id, + self.identity.clone(), + &self.surface, + self.execution.clone(), + ) + }) + .as_ref() + .map_err(Clone::clone) } pub fn service_id(&self) -> &str { &self.service_id } + #[cfg(test)] + pub(crate) fn manifest_cache_owners(&self) -> usize { + Arc::strong_count(&self.manifest) + } + pub fn identity(&self) -> &ClientSurfaceIdentity { &self.identity } diff --git a/src/graphql/client_manifest/tests.rs b/src/graphql/client_manifest/tests.rs index 4f2e876d6..a1b338a1f 100644 --- a/src/graphql/client_manifest/tests.rs +++ b/src/graphql/client_manifest/tests.rs @@ -754,6 +754,75 @@ fn manifest_for_all_models( .expect("client manifest") } +#[test] +fn selected_export_cache_is_shared_concurrently_but_returned_manifests_are_independent() { + let full = full_surface(); + let selected = surface_for_role(&full, "user", &grants()["user"]).unwrap(); + let fresh = client_manifest_from_surface( + "todos-service", + ClientSurfaceIdentity::role("user"), + &selected, + ) + .unwrap(); + let export = DistributedClientSurfaceExport::from_selected("todos-service", selected).unwrap(); + let barrier = Arc::new(std::sync::Barrier::new(8)); + let threads: Vec<_> = (0..8) + .map(|_| { + let export = export.clone(); + let barrier = Arc::clone(&barrier); + std::thread::spawn(move || { + barrier.wait(); + let manifest = export.manifest_ref().unwrap(); + ( + manifest as *const DistributedClientManifest as usize, + manifest.clone(), + ) + }) + }) + .collect(); + let expected_address = + export.manifest_ref().unwrap() as *const DistributedClientManifest as usize; + for thread in threads { + let (address, manifest) = thread.join().unwrap(); + assert_eq!(address, expected_address); + assert_eq!(manifest, fresh); + } + let mut detached = export.manifest().unwrap(); + detached.models.clear(); + detached.projection_programs.clear(); + detached.schema_fingerprint.clear(); + assert_eq!(export.manifest_ref().unwrap(), &fresh); + assert_eq!(export.clone().manifest().unwrap(), fresh); +} + +#[test] +fn selected_export_caches_do_not_alias_role_application_or_execution_limits() { + let full = full_surface(); + let grants = grants(); + let user = surface_for_role(&full, "user", &grants["user"]).unwrap(); + let admin = surface_for_role(&full, "admin", &grants["admin"]).unwrap(); + let application = + surface_for_application(&full, "web", &["user".into()], &["user".into()], &grants).unwrap(); + let user_export = + DistributedClientSurfaceExport::from_selected("todos-service", user.clone()).unwrap(); + let admin_export = + DistributedClientSurfaceExport::from_selected("todos-service", admin).unwrap(); + let app_export = + DistributedClientSurfaceExport::from_selected("todos-service", application).unwrap(); + let mut limits = ClientExecutionLimits::default(); + limits.max_bool_width += 1; + let limited = + DistributedClientSurfaceExport::from_selected_with_execution("todos-service", user, limits) + .unwrap(); + let baseline = user_export.manifest_ref().unwrap(); + for other in [&admin_export, &app_export, &limited] { + let manifest = other.manifest_ref().unwrap(); + assert!(!std::ptr::eq(baseline, manifest)); + assert_ne!(baseline.schema_fingerprint, manifest.schema_fingerprint); + } + assert_ne!(user_export.identity(), app_export.identity()); +} + #[test] fn role_manifest_is_deterministic_and_hides_denied_identity_and_commands() { let full = full_surface(); diff --git a/src/graphql/engine/builder.rs b/src/graphql/engine/builder.rs index 4ccc46b86..aeca975c9 100644 --- a/src/graphql/engine/builder.rs +++ b/src/graphql/engine/builder.rs @@ -836,7 +836,7 @@ impl GraphqlEngineBuilder { .expect("protocol configuration validated a service ID"); let (authorization_fingerprint, claim_keys) = role_authorization_info(role, &self.permissions)?; - let manifest = DistributedClientSurfaceExport::from_selected_with_execution( + let export = DistributedClientSurfaceExport::from_selected_with_execution( service_id, Arc::clone(&role_surface), ClientExecutionLimits::from_runtime( @@ -847,17 +847,22 @@ impl GraphqlEngineBuilder { ) .map_err(|error| GraphqlBuildError(error.to_string()))?, ) - .and_then(|export| export.manifest()) .map_err(|error| { GraphqlBuildError(format!( "failed to derive GraphQL protocol surface for role `{role}`: {error}" )) })?; + let manifest = export.manifest().map_err(|error| { + GraphqlBuildError(format!( + "failed to derive GraphQL protocol surface for role `{role}`: {error}" + )) + })?; let trusted_presets = protocol_trusted_presets(&manifest)?; protocol_roles.insert( role.clone(), ProtocolRoleInfo { surface: ProtocolSurfaceInfo { + export, schema_fingerprint: manifest.schema_fingerprint, protocol_fingerprint: manifest.protocol_fingerprint, trusted_presets, @@ -915,7 +920,7 @@ impl GraphqlEngineBuilder { .service_id .as_deref() .expect("protocol configuration validated a service ID"); - let manifest = DistributedClientSurfaceExport::from_selected_with_execution( + let export = DistributedClientSurfaceExport::from_selected_with_execution( service_id, Arc::clone(&application_surface), ClientExecutionLimits::from_runtime( @@ -926,7 +931,10 @@ impl GraphqlEngineBuilder { ) .map_err(|error| GraphqlBuildError(error.to_string()))?, ) - .and_then(|export| export.manifest()) + .map_err(|error| GraphqlBuildError(format!( + "failed to derive GraphQL protocol surface for application `{application}`: {error}" + )))?; + let manifest = export.manifest() .map_err(|error| { GraphqlBuildError(format!( "failed to derive GraphQL protocol surface for application `{application}`: {error}" @@ -990,6 +998,7 @@ impl GraphqlEngineBuilder { schema_roles: registration.schema_roles.clone(), privilege_key, surface: ProtocolSurfaceInfo { + export, schema_fingerprint: manifest.schema_fingerprint, protocol_fingerprint: manifest.protocol_fingerprint, trusted_presets, diff --git a/src/graphql/engine/core.rs b/src/graphql/engine/core.rs index b453f60ad..f90815d07 100644 --- a/src/graphql/engine/core.rs +++ b/src/graphql/engine/core.rs @@ -170,6 +170,7 @@ pub(crate) struct RoleModelPerm { #[derive(Clone)] pub(crate) struct ProtocolSurfaceInfo { + pub(crate) export: DistributedClientSurfaceExport, pub(crate) schema_fingerprint: String, pub(crate) protocol_fingerprint: String, pub(crate) trusted_presets: Vec, diff --git a/src/graphql/engine/request.rs b/src/graphql/engine/request.rs index e8d3c2e77..bf027f643 100644 --- a/src/graphql/engine/request.rs +++ b/src/graphql/engine/request.rs @@ -298,26 +298,9 @@ impl GraphqlEngine { .get(&authority.privilege_role) .cloned() .ok_or(())?; - let selected_surface = match &surface_identity { - ClientSurfaceIdentity::Role { name } => self.inner.role_surfaces.get(name), - ClientSurfaceIdentity::Application { name, .. } => { - self.inner.application_surfaces.get(name) - } - } - .cloned() - .ok_or(())?; - let export = DistributedClientSurfaceExport::from_selected_with_execution( - &runtime.service_id, - selected_surface, - ClientExecutionLimits::from_runtime( - self.inner.max_depth, - self.inner.max_complexity, - self.inner.max_bool_width, - self.inner.max_in_list, - ) - .map_err(|_| ())?, - ) - .map_err(|_| ())?; + // Selected by the verified authority above; this exact immutable + // export was validated when the engine's protocol surface was built. + let export = surface_info.export.clone(); let issued_at_unix_ms = crate::time::now() .duration_since(std::time::UNIX_EPOCH) .map_err(|_| ())? diff --git a/src/graphql/engine/tests.rs b/src/graphql/engine/tests.rs index 8576815a4..151512583 100644 --- a/src/graphql/engine/tests.rs +++ b/src/graphql/engine/tests.rs @@ -142,6 +142,70 @@ mod client_surface_parity_tests { .unwrap() } + #[cfg(feature = "sqlite")] + #[tokio::test] + async fn protocol_surface_export_cache_is_engine_scoped_and_shared_by_engine_clones() { + let engine = Arc::new(protocol_engine("first-engine")); + let cloned = engine.clone(); + let another = protocol_engine("second-engine"); + let export = &engine.inner.protocol.as_ref().unwrap().roles["user"] + .surface + .export; + let cloned_export = &cloned.inner.protocol.as_ref().unwrap().roles["user"] + .surface + .export; + let other_export = &another.inner.protocol.as_ref().unwrap().roles["user"] + .surface + .export; + assert!(std::ptr::eq( + export.manifest_ref().unwrap(), + cloned_export.manifest_ref().unwrap() + )); + assert!(!std::ptr::eq( + export.manifest_ref().unwrap(), + other_export.manifest_ref().unwrap() + )); + let fresh = engine + .client_surface_for_role("user") + .unwrap() + .manifest() + .unwrap(); + assert_eq!(export.manifest_ref().unwrap(), &fresh); + assert_eq!(other_export.manifest_ref().unwrap(), &fresh); + } + + #[cfg(feature = "sqlite")] + #[tokio::test] + async fn repeated_protocol_accumulators_reuse_metadata_but_release_request_authority() { + use crate::graphql::identity::VerifiedPrincipal; + let engine = protocol_engine("request-cache-test"); + let export = &engine.inner.protocol.as_ref().unwrap().roles["user"] + .surface + .export; + let owners = export.manifest_cache_owners(); + let mut session = Session::new(); + session.set("x-roles", "user"); + let request = Request::new("{ __typename }").data(VerifiedPrincipal::test_oidc( + "https://issuer.example", + "principal-a", + &["orders-service"], + )); + let authority = resolve_execution_authority(&engine.inner, &session, &request).unwrap(); + let first = engine + .protocol_accumulator(&authority, &session, &request) + .unwrap_or_else(|_| panic!("first request authority")) + .unwrap(); + assert_eq!(export.manifest_cache_owners(), owners + 1); + let second = engine + .protocol_accumulator(&authority, &session, &request) + .unwrap_or_else(|_| panic!("second request authority")) + .unwrap(); + assert_eq!(export.manifest_cache_owners(), owners + 2); + drop(first); + drop(second); + assert_eq!(export.manifest_cache_owners(), owners); + } + #[cfg(feature = "sqlite")] fn policy_protocol_engine(namespace: &str, claim_key: &str) -> GraphqlEngine { let pool = sqlx::sqlite::SqlitePoolOptions::new() diff --git a/src/graphql/projection_delta/runtime.rs b/src/graphql/projection_delta/runtime.rs index 8ab93c6bd..4d1f68eb3 100644 --- a/src/graphql/projection_delta/runtime.rs +++ b/src/graphql/projection_delta/runtime.rs @@ -105,9 +105,9 @@ impl ProtocolProjectionRequestSeed { return Err(ProjectionRuntimeAuthorityError::InvalidAuthority); } let manifest = export - .manifest() + .manifest_ref() .map_err(|_| ProjectionRuntimeAuthorityError::InvalidAuthority)?; - let expected = crate::graphql::client_manifest::trusted_preset_descriptors(&manifest) + let expected = crate::graphql::client_manifest::trusted_preset_descriptors(manifest) .map_err(|_| ProjectionRuntimeAuthorityError::InvalidAuthority)?; let mut preset_names = BTreeSet::new(); if trusted_presets.len() != expected.len() diff --git a/src/graphql/subscribe.rs b/src/graphql/subscribe.rs index c4a9c9f0b..e800fbcf4 100644 --- a/src/graphql/subscribe.rs +++ b/src/graphql/subscribe.rs @@ -5,6 +5,13 @@ //! 2. Listens on [`ChangeHub`] (fed by `change_stream` / repo broadcast). //! 3. On dirty tables intersecting the plan footprint, debounces, re-executes, //! and yields only when the response hash changes (hash-gated push). +//! +//! A failed execution is terminal (`docs/live-query-delivery.md`): the producer +//! yields one error and stops. GraphQL ends the subscription on that error, and +//! the failure frame carries only the base protocol envelope. Protocol frame +//! metadata is a FIFO consumed by each emitted response, so producing another +//! frame after an error could let the error response take that later frame's +//! snapshot/live metadata. use std::collections::BTreeSet; use std::pin::Pin; @@ -201,14 +208,8 @@ pub(crate) async fn live_query_stream( continue; // hash gate: no push on no-change } if let Err(error) = executed.record_protocol_metadata(protocol.as_ref()) { - if tx - .send(Err(async_graphql::Error::new(error))) - .await - .is_err() - { - return; - } - continue; + let _ = tx.send(Err(async_graphql::Error::new(error))).await; + return; } last_hash = Some(executed.hash); if tx.send(Ok(executed.value)).await.is_err() { @@ -216,9 +217,8 @@ pub(crate) async fn live_query_stream( } } Err(e) => { - if tx.send(Err(async_graphql::Error::new(e))).await.is_err() { - return; - } + let _ = tx.send(Err(async_graphql::Error::new(e))).await; + return; } } } diff --git a/src/in_memory_repo/projection_protocol/state.rs b/src/in_memory_repo/projection_protocol/state.rs index 2ce23b504..4050b536f 100644 --- a/src/in_memory_repo/projection_protocol/state.rs +++ b/src/in_memory_repo/projection_protocol/state.rs @@ -289,6 +289,7 @@ pub(in crate::in_memory_repo) fn reject_causal_owned_plans( pub(super) enum InputDisposition { New, + Redelivery, Duplicate(ProjectionCheckpoint), Stale(ProjectionCheckpoint), } diff --git a/src/in_memory_repo/projection_protocol/state_impl.rs b/src/in_memory_repo/projection_protocol/state_impl.rs index 383e80464..2b8098235 100644 --- a/src/in_memory_repo/projection_protocol/state_impl.rs +++ b/src/in_memory_repo/projection_protocol/state_impl.rs @@ -236,6 +236,21 @@ impl InMemoryProjectionProtocolState { gap_free, }; self.validate_input_identity(&candidate)?; + let repeated_effect = self + .messages + .get(&MessageKey { + topology: cursor.topology().clone(), + message_id: message_id.to_string(), + }) + .is_some_and(|identity| { + self.applied_receipts + .contains_key(&CursorReceiptKey::new(&identity.cursor, generation)) + || self.applied_receipts.iter().any(|(key, receipt)| { + key.generation == generation + && key.partition == PartitionKey::from_input(cursor) + && receipt.message_id == message_id + }) + }); if let Some(receipt) = self .applied_receipts .get(&CursorReceiptKey::new(cursor, generation)) @@ -252,7 +267,11 @@ impl InMemoryProjectionProtocolState { let input_key = InputKey::new(cursor, generation); let Some(previous) = self.inputs.get(&input_key) else { self.reject_reused_message(cursor, fingerprint, message_id, causation_id, gap_free)?; - return Ok(InputDisposition::New); + return Ok(if repeated_effect { + InputDisposition::Redelivery + } else { + InputDisposition::New + }); }; match cursor.compare_position(&previous.cursor) { @@ -292,7 +311,11 @@ impl InMemoryProjectionProtocolState { { return Err(ProjectionProtocolError::IncomparableInput); } - Ok(InputDisposition::New) + Ok(if repeated_effect { + InputDisposition::Redelivery + } else { + InputDisposition::New + }) } } } @@ -310,7 +333,7 @@ impl InMemoryProjectionProtocolState { message_id: message_id.to_string(), }; if let Some(previous) = self.messages.get(&key) { - if previous.cursor != *cursor + if previous.cursor.compare_position(cursor) == RevisionComparison::Incomparable || previous.fingerprint != fingerprint || previous.causation_id != causation_id || previous.gap_free != gap_free diff --git a/src/in_memory_repo/projection_protocol/store_impl.rs b/src/in_memory_repo/projection_protocol/store_impl.rs index be530e460..f7e92882c 100644 --- a/src/in_memory_repo/projection_protocol/store_impl.rs +++ b/src/in_memory_repo/projection_protocol/store_impl.rs @@ -104,7 +104,7 @@ impl ProjectionProtocolStore for InMemoryRepository { fn commit_projection( &self, - batch: ProjectionCommitBatch, + mut batch: ProjectionCommitBatch, ) -> impl Future> + Send + '_ { async move { @@ -129,7 +129,7 @@ impl ProjectionProtocolStore for InMemoryRepository { .map_err(|_| RepositoryError::LockPoisoned("projection inbox write"))?; protocol.validate_partition(&partition_key, &batch.input, &batch.change_epoch)?; - match protocol.classify_input( + let redelivery = match protocol.classify_input( &batch.input.cursor, batch.input.fingerprint, &batch.input.message_id, @@ -151,12 +151,21 @@ impl ProjectionProtocolStore for InMemoryRepository { } InputDisposition::New => { protocol.validate_pending_retry(&partition_key, &batch.input)?; + false } - } + InputDisposition::Redelivery => { + protocol.validate_pending_retry(&partition_key, &batch.input)?; + batch.mutations.clear(); + batch.observations.clear(); + true + } + }; let receipt = batch.input.inbox_receipt(); receipt.validate()?; - if inbox.contains(&(receipt.consumer.clone(), receipt.message_id.clone())) { + if !redelivery + && inbox.contains(&(receipt.consumer.clone(), receipt.message_id.clone())) + { return Err(ProjectionProtocolError::MessageIdReuse { message_id: batch.input.message_id.clone(), }); @@ -413,7 +422,11 @@ impl ProjectionProtocolStore for InMemoryRepository { *inbox = staged_inbox; Ok(ProjectionCommitResult { - outcome: ProjectionCommitOutcome::Applied, + outcome: if redelivery { + ProjectionCommitOutcome::Duplicate + } else { + ProjectionCommitOutcome::Applied + }, checkpoint: Some(checkpoint), records, changes, @@ -496,6 +509,11 @@ impl ProjectionProtocolStore for InMemoryRepository { InputDisposition::New => { protocol.validate_pending_retry(&partition_key, &batch.input)?; } + InputDisposition::Redelivery => { + return Err(ProjectionProtocolError::InvalidBatch( + "an already applied message cannot record a new failure".into(), + )) + } InputDisposition::Duplicate(_) | InputDisposition::Stale(_) => { return Err(ProjectionProtocolError::InvalidBatch( "cannot record terminal failure for an already processed input".into(), @@ -677,6 +695,10 @@ impl ProjectionProtocolStore for InMemoryRepository { protocol.validate_pending_retry(&partition_key, input)?; Ok(ProjectionInputDisposition::Pending) } + InputDisposition::Redelivery => { + protocol.validate_pending_retry(&partition_key, input)?; + Ok(ProjectionInputDisposition::Redelivery) + } } } } diff --git a/src/in_memory_repo/projection_protocol/tests.rs b/src/in_memory_repo/projection_protocol/tests.rs index 017549216..83ed88c49 100644 --- a/src/in_memory_repo/projection_protocol/tests.rs +++ b/src/in_memory_repo/projection_protocol/tests.rs @@ -2314,6 +2314,11 @@ async fn input_disposition_is_read_only_exact_and_repair_fenced() { .await; } +#[tokio::test] +async fn identical_redelivery_advances_only_broker_checkpoint() { + crate::projection_protocol::scenario_tests::identical_redelivery_advances_only_broker_checkpoint(ProjectionScenario).await; +} + #[tokio::test] async fn repair_generation_retries_only_the_exact_failed_input() { let repository = repository().await; diff --git a/src/lib.rs b/src/lib.rs index c84cc9b8a..2af2b194b 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -79,16 +79,16 @@ pub use application::{ MountSelector, ProcessIntent, ProcessPreset, ProjectionSpec, Runtime, RuntimeDialect, SurfaceSpec, APPLICATION_MANIFEST_SCHEMA_VERSION, DEPLOYMENT_PLAN_SCHEMA_VERSION, }; -pub use command_dispatch::{ - CommandDispatchEnvelope, CommandDispatchError, CommandDispatchReceipt, CommandDispatcher, - LocalCommandDispatcher, RemoteCommandDispatcher, RemoteDispatchConfig, RemoteTrustMode, - SharedCommandDispatcher, APPROVED_REMOTE_DISPATCH_PROFILE, COMMAND_DISPATCH_ENVELOPE_VERSION, -}; #[cfg(feature = "graphql")] pub use command_dispatch::{ CellRequestContext, CommandHost, HttpCommandHost, LocalCommandHost, SharedCommandHost, TrustedRequestMetadata, }; +pub use command_dispatch::{ + CommandDispatchEnvelope, CommandDispatchError, CommandDispatchReceipt, CommandDispatcher, + LocalCommandDispatcher, RemoteCommandDispatcher, RemoteDispatchConfig, RemoteTrustMode, + SharedCommandDispatcher, APPROVED_REMOTE_DISPATCH_PROFILE, COMMAND_DISPATCH_ENVELOPE_VERSION, +}; // Domain events: typed outward contracts distinct from replay events/snapshots. pub use domain_event::{ @@ -96,8 +96,9 @@ pub use domain_event::{ DomainEventBodyKind, DomainEventCaptureError, DomainEventCaptureOutcome, DomainEventCapturePoison, DomainEventCommitGuardError, DomainEventDescriptor, DomainEventEnvelope, DomainEventOccurrence, DomainState, DomainStateDescriptor, - DOMAIN_EVENT_BODY_CODEC, DOMAIN_EVENT_BODY_CODEC_VERSION, DOMAIN_EVENT_OCCURRENCE_VERSION, - MAX_DOMAIN_EVENT_BODY_BYTES, MAX_DOMAIN_EVENT_OCCURRENCE_WIRE_BYTES, + ExternalEventSource, DOMAIN_EVENT_BODY_CODEC, DOMAIN_EVENT_BODY_CODEC_VERSION, + DOMAIN_EVENT_OCCURRENCE_VERSION, MAX_DOMAIN_EVENT_BODY_BYTES, + MAX_DOMAIN_EVENT_OCCURRENCE_WIRE_BYTES, }; // Logical projection contracts. Physical read-model lowering deliberately lives @@ -316,6 +317,9 @@ macro_rules! __projection_source_policy { ($program:expr, aggregate_snapshot) => { $program.with_source_snapshots() }; + ($program:expr, external_snapshot) => { + $program.with_external_source_snapshots() + }; } /// Map `input: { key: body | aggregate_id }` keywords for [`projection!`]. diff --git a/src/microsvc/error.rs b/src/microsvc/error.rs index 437eb8e40..e52a2bb23 100644 --- a/src/microsvc/error.rs +++ b/src/microsvc/error.rs @@ -14,6 +14,9 @@ use crate::{repository::RepositoryError, EventRecordError}; #[derive(Debug)] #[non_exhaustive] pub enum HandlerError { + /// Supervisor generation admission is temporarily closed. Retain the exact + /// delivery and stop the receive loop until the host retries after reload. + ApplicationReloading, /// No handler registered for this command name. UnknownCommand(String), /// Payload decode / deserialization failed. @@ -51,6 +54,7 @@ pub enum HandlerError { impl fmt::Display for HandlerError { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { match self { + HandlerError::ApplicationReloading => f.write_str("application generation is reloading"), HandlerError::UnknownCommand(name) => write!(f, "unknown command: {}", name), HandlerError::DecodeFailed(msg) => write!(f, "decode failed: {}", msg), HandlerError::Rejected(msg) => write!(f, "rejected: {}", msg), @@ -142,6 +146,7 @@ impl HandlerError { /// Map this error to an HTTP-style status code. pub fn status_code(&self) -> u16 { match self { + HandlerError::ApplicationReloading => 503, HandlerError::UnknownCommand(_) => 404, HandlerError::DecodeFailed(_) => 400, HandlerError::Rejected(_) => 422, @@ -201,7 +206,8 @@ impl HandlerError { TransportErrorKind::Permanent } } - HandlerError::ProjectionRepairPending { .. } + HandlerError::ApplicationReloading + | HandlerError::ProjectionRepairPending { .. } | HandlerError::NotFound(_) | HandlerError::Other(_) => TransportErrorKind::Retryable, HandlerError::ProjectionTerminalRecorded { .. } @@ -228,7 +234,8 @@ impl From for TransportError { let kind = error.transport_error_kind(); let retain_and_stop = matches!( error, - HandlerError::ProjectionTerminalRecorded { .. } + HandlerError::ApplicationReloading + | HandlerError::ProjectionTerminalRecorded { .. } | HandlerError::ProjectionDeliveryHalted { .. } ); let transport = TransportError::new(kind, error.to_string()).with_source(error); @@ -244,6 +251,16 @@ impl From for TransportError { mod tests { use super::*; + #[test] + fn application_reloading_retains_delivery_without_reclassifying_business_errors() { + let reload = TransportError::from(HandlerError::ApplicationReloading); + assert!(reload.is_retryable()); + assert!(reload.should_retain_and_stop()); + let rejection = TransportError::from(HandlerError::Rejected("invalid slug".into())); + assert!(rejection.is_permanent()); + assert!(!rejection.should_retain_and_stop()); + } + #[test] fn transient_handler_errors_are_retryable() { for error in [ diff --git a/src/microsvc/http.rs b/src/microsvc/http.rs index e46fa56a2..b9835faa0 100644 --- a/src/microsvc/http.rs +++ b/src/microsvc/http.rs @@ -197,6 +197,7 @@ async fn command_handler( fn status_for_error(error: &HandlerError) -> StatusCode { match error { + HandlerError::ApplicationReloading => StatusCode::SERVICE_UNAVAILABLE, HandlerError::UnknownCommand(_) | HandlerError::NotFound(_) => StatusCode::NOT_FOUND, HandlerError::DecodeFailed(_) | HandlerError::GuardRejected(_) => StatusCode::BAD_REQUEST, HandlerError::Rejected(_) => StatusCode::UNPROCESSABLE_ENTITY, diff --git a/src/microsvc/message_router.rs b/src/microsvc/message_router.rs index 20d213024..983dd5e45 100644 --- a/src/microsvc/message_router.rs +++ b/src/microsvc/message_router.rs @@ -5,7 +5,7 @@ //! retryable/permanent vocabulary happens here, on the microsvc side, so the //! bus-core runner only ever sees an already-classified `TransportError`. -use crate::bus::{MessageRouter, OrderedDelivery, TransportError}; +use crate::bus::{LaneSet, MessageRouter, OrderedDelivery, TransportError}; use crate::microsvc::{Message, MessageKind, Service, SubscriptionPlan}; impl MessageRouter for Service { @@ -39,4 +39,24 @@ impl MessageRouter for Service { .map(|_| ()) .map_err(TransportError::from) } + + fn delivery_lanes(&self) -> usize { + self.delivery_lane_names().len() + } + + fn lanes_for(&self, kind: MessageKind, name: &str) -> LaneSet { + self.lanes_for_message(kind, name) + } + + async fn dispatch_lane( + &self, + message: &Message, + ordered: Option<&OrderedDelivery>, + lane: usize, + ) -> Result<(), TransportError> { + self.dispatch_lane_message(message, ordered, Some(lane)) + .await + .map(|_| ()) + .map_err(TransportError::from) + } } diff --git a/src/microsvc/mod.rs b/src/microsvc/mod.rs index 3ea157bb0..d7e5703fb 100644 --- a/src/microsvc/mod.rs +++ b/src/microsvc/mod.rs @@ -113,7 +113,7 @@ pub use service::{ CausalCommitBuilder, CausalRepository, CommandRequest, CommandResponse, DeliveryKind, DirectReadModelProjection, HandlerNames, HandlerSpec, PortableCommand, PreparedCausalCommit, PreparedCommandHandler, RouteBuilder, Routes, Service, ThinCommandBuilder, ThinCommandInvoked, - ThinCommandLoaded, TypedRouteBuilder, + ThinCommandLoaded, TypedRouteBuilder, DEFAULT_DELIVERY_LANE, }; #[cfg(feature = "graphql")] pub(crate) use service::{ diff --git a/src/microsvc/projector/runtime.rs b/src/microsvc/projector/runtime.rs index 26b391241..f776540ed 100644 --- a/src/microsvc/projector/runtime.rs +++ b/src/microsvc/projector/runtime.rs @@ -292,6 +292,18 @@ where )?; match store.projection_input_disposition(&trusted).await? { ProjectionInputDisposition::Pending => {} + ProjectionInputDisposition::Redelivery => { + store + .commit_projection(crate::projection_protocol::ProjectionCommitBatch { + input: trusted, + change_epoch: self.change_epoch.clone(), + ownership: self.compiled.ownership().to_vec(), + mutations: Vec::new(), + observations: Vec::new(), + }) + .await?; + return Ok(()); + } ProjectionInputDisposition::Duplicate(_) | ProjectionInputDisposition::Stale(_) => { return Ok(()) } @@ -568,6 +580,18 @@ where )?; match store.projection_input_disposition(&trusted).await? { ProjectionInputDisposition::Pending => {} + ProjectionInputDisposition::Redelivery => { + store + .commit_projection(crate::projection_protocol::ProjectionCommitBatch { + input: trusted, + change_epoch: self.change_epoch.clone(), + ownership: self.compiled.ownership().to_vec(), + mutations: Vec::new(), + observations: Vec::new(), + }) + .await?; + return Ok(()); + } ProjectionInputDisposition::Duplicate(_) | ProjectionInputDisposition::Stale(_) => { return Ok(()) } diff --git a/src/microsvc/service/mod.rs b/src/microsvc/service/mod.rs index a9592c7d5..c744b7e91 100644 --- a/src/microsvc/service/mod.rs +++ b/src/microsvc/service/mod.rs @@ -60,7 +60,7 @@ pub use routes::{ DeliveryKind, HandlerNames, HandlerSpec, PortableCommand, RouteBuilder, Routes, ThinCommandBuilder, ThinCommandInvoked, ThinCommandLoaded, TypedRouteBuilder, }; -pub use runtime::Service; +pub use runtime::{Service, DEFAULT_DELIVERY_LANE}; #[cfg(test)] mod tests; diff --git a/src/microsvc/service/runtime.rs b/src/microsvc/service/runtime.rs index 688b5a46d..5b32169c1 100644 --- a/src/microsvc/service/runtime.rs +++ b/src/microsvc/service/runtime.rs @@ -40,9 +40,7 @@ fn ensure_lifecycle_mutations_open() -> Result<(), HandlerError> { if crate::microsvc::lifecycle_mutations_open() { Ok(()) } else { - Err(HandlerError::Rejected( - "application generation is reloading".into(), - )) + Err(HandlerError::ApplicationReloading) } } @@ -57,11 +55,18 @@ pub(crate) type ServiceRunner = Box< + Sync, >; +/// Name of the delivery lane that plain [`Service::routes`] registers into. +pub const DEFAULT_DELIVERY_LANE: &str = "default"; + /// A microservice deployment that routes messages to one or more route bundles. pub struct Service { name: Option, pub(super) routes: Vec>, index: HashMap>>, + /// Delivery lane of each route bundle (parallel to `routes`); 0 is default. + route_lanes: Vec, + /// Lane names by index; index 0 is the unnamed default lane. + lane_names: Vec<&'static str>, handler_specs: Vec, causal_command_policy: CausalCommandPolicy, runner: Option, @@ -80,6 +85,8 @@ impl Service { name: None, routes: Vec::new(), index: HashMap::new(), + route_lanes: Vec::new(), + lane_names: vec![DEFAULT_DELIVERY_LANE], handler_specs: Vec::new(), causal_command_policy: CausalCommandPolicy::default(), runner: None, @@ -494,6 +501,60 @@ impl Service { self } + /// Add a route bundle to a named delivery lane. + /// + /// Lanes of one consumer run concurrently when the transport settles each + /// delivery independently; each lane keeps delivery order and a delivery is + /// settled only after all of its lanes finished. Use a lane only for routes + /// that neither depend on nor are depended on by routes of other lanes + /// within the same delivery (see `docs/consumer-delivery-lanes.md`). + /// + /// # Panics + /// When `lane` is empty, names the default lane, or exceeds the lane limit. + pub fn lane(mut self, lane: &'static str, routes: Routes) -> Self + where + D: Send + Sync + 'static, + { + assert!( + !lane.trim().is_empty() && lane != DEFAULT_DELIVERY_LANE, + "a named delivery lane must be non-empty and not `{DEFAULT_DELIVERY_LANE}`" + ); + let index = match self.lane_names.iter().position(|name| *name == lane) { + Some(index) => index, + None => { + assert!( + self.lane_names.len() < crate::bus::LaneSet::MAX_LANES, + "too many delivery lanes" + ); + self.lane_names.push(lane); + self.lane_names.len() - 1 + } + }; + self.add_routes(routes); + *self + .route_lanes + .last_mut() + .expect("add_routes registers one bundle") = index; + self + } + + /// Delivery lane names by index; index 0 is the default lane. + pub fn delivery_lane_names(&self) -> &[&'static str] { + &self.lane_names + } + + /// Lanes with a route for `(kind, name)`. + pub fn lanes_for_message(&self, kind: MessageKind, name: &str) -> crate::bus::LaneSet { + self.index + .get(&kind) + .and_then(|by_name| by_name.get(name)) + .into_iter() + .flatten() + .fold(crate::bus::LaneSet::EMPTY, |lanes, route| { + lanes.with(self.route_lanes[*route]) + }) + } + pub(super) fn add_routes(&mut self, routes: Routes) where D: Send + Sync + 'static, @@ -569,6 +630,7 @@ impl Service { self.handler_specs.extend_from_slice(routes.handler_specs()); self.registered_command_mounts.extend(command_mounts); self.routes.push(Box::new(routes)); + self.route_lanes.push(0); } pub(crate) fn typed_command_contracts(&self) -> Vec { @@ -1008,7 +1070,8 @@ impl Service { metadata, }; - self.invoke_with_dispatch_span(&message, input, session, None) + let route_indices = self.lane_route_indices(&message, None)?; + self.invoke_with_dispatch_span(&message, input, session, None, route_indices) .await } @@ -1039,10 +1102,21 @@ impl Service { &self, message: &Message, ordered: Option<&OrderedDelivery>, + ) -> Result { + self.dispatch_lane_message(message, ordered, None).await + } + + /// Dispatch only the routes of one delivery lane (`None` runs every lane + /// in registration order). + pub(crate) async fn dispatch_lane_message( + &self, + message: &Message, + ordered: Option<&OrderedDelivery>, + lane: Option, ) -> Result { #[cfg(feature = "metrics")] let started = Instant::now(); - let result = self.dispatch_message_inner(message, ordered).await; + let result = self.dispatch_message_inner(message, ordered, lane).await; #[cfg(feature = "metrics")] { let error = result.as_ref().err(); @@ -1063,21 +1137,17 @@ impl Service { &self, message: &Message, ordered: Option<&OrderedDelivery>, + lane: Option, ) -> Result { - if !crate::microsvc::lifecycle_mutations_open() { - return Err(HandlerError::Rejected( - "application generation is reloading".into(), - )); - } + ensure_lifecycle_mutations_open()?; if !self.handles_message(message.kind, &message.name) { return Err(HandlerError::UnknownCommand(message.name.clone())); } - let route_indices = self - .index - .get(&message.kind) - .and_then(|by_name| by_name.get(message.name())) - .ok_or_else(|| HandlerError::UnknownCommand(message.name.clone()))?; + let route_indices = self.lane_route_indices(message, lane)?; + if route_indices.is_empty() { + return Err(HandlerError::UnknownCommand(message.name.clone())); + } let projector_only = route_indices .iter() .all(|index| self.routes[*index].is_causal_projector(message)); @@ -1099,16 +1169,34 @@ impl Service { } }; let session = message_to_session(message); - self.invoke_with_dispatch_span(message, input, session, ordered) + self.invoke_with_dispatch_span(message, input, session, ordered, route_indices) .await } + fn lane_route_indices( + &self, + message: &Message, + lane: Option, + ) -> Result, HandlerError> { + let all = self + .index + .get(&message.kind) + .and_then(|by_name| by_name.get(message.name())) + .ok_or_else(|| HandlerError::UnknownCommand(message.name.clone()))?; + Ok(all + .iter() + .copied() + .filter(|route| lane.is_none_or(|lane| self.route_lanes[*route] == lane)) + .collect()) + } + async fn invoke_with_dispatch_span( &self, message: &Message, input: Value, session: Session, ordered: Option<&OrderedDelivery>, + route_indices: Vec, ) -> Result { #[cfg(feature = "otel")] { @@ -1120,14 +1208,15 @@ impl Service { &message.metadata, ); return self - .invoke(message, input, session, ordered) + .invoke(message, input, session, ordered, route_indices) .instrument(span) .await; } #[cfg(not(feature = "otel"))] { - self.invoke(message, input, session, ordered).await + self.invoke(message, input, session, ordered, route_indices) + .await } } @@ -1137,13 +1226,8 @@ impl Service { input: Value, session: Session, ordered: Option<&OrderedDelivery>, + route_indices: Vec, ) -> Result { - let route_indices = self - .index - .get(&message.kind) - .and_then(|by_name| by_name.get(message.name.as_str())) - .cloned() - .ok_or_else(|| HandlerError::UnknownCommand(message.name.clone()))?; #[cfg(feature = "otel")] let handler_span = microsvc_handler_span(message); let dispatch = async move { diff --git a/src/microsvc/service/tests.rs b/src/microsvc/service/tests.rs index 6f9fcab98..8fdf19e5b 100644 --- a/src/microsvc/service/tests.rs +++ b/src/microsvc/service/tests.rs @@ -4170,3 +4170,99 @@ fn command_request_requires_session_variables_field() { let result: Result = serde_json::from_str(json); assert!(result.is_err()); } + +type LaneLog = std::sync::Arc>>; + +fn lane_log() -> LaneLog { + std::sync::Arc::new(std::sync::Mutex::new(Vec::new())) +} + +fn lane_recorder(log: &LaneLog, label: &'static str) -> Routes<()> { + let log = log.clone(); + test_routes() + .events(&["fact.recorded", "fact.other"]) + .handle(move |ctx: &Context<()>| { + log.lock() + .unwrap() + .push(format!("{label}:{}", ctx.message().name())); + async move { Ok(json!({})) } + }) +} + +#[tokio::test] +async fn named_lanes_partition_routes_and_default_dispatch_runs_all_in_order() { + use crate::bus::{LaneSet, MessageRouter}; + let log = lane_log(); + let service = Service::new() + .routes(lane_recorder(&log, "projection-a")) + .lane("process", lane_recorder(&log, "policy")) + .routes(lane_recorder(&log, "projection-b")) + .lane( + "process", + test_routes() + .event("fact.process_only") + .handle(|_: &Context<()>| async move { Ok(json!({})) }), + ); + + assert_eq!( + service.delivery_lane_names(), + &[DEFAULT_DELIVERY_LANE, "process"] + ); + assert_eq!(MessageRouter::delivery_lanes(&service), 2); + assert_eq!( + MessageRouter::lanes_for(&service, MessageKind::Event, "fact.recorded"), + LaneSet::single(0).with(1) + ); + assert_eq!( + MessageRouter::lanes_for(&service, MessageKind::Event, "fact.process_only"), + LaneSet::single(1) + ); + assert!(MessageRouter::lanes_for(&service, MessageKind::Event, "fact.unknown").is_empty()); + + let message = Message::new("fact.recorded", MessageKind::Event, b"{}".to_vec()).with_id("e1"); + MessageRouter::dispatch_lane(&service, &message, None, 1) + .await + .unwrap(); + assert_eq!(*log.lock().unwrap(), ["policy:fact.recorded"]); + log.lock().unwrap().clear(); + MessageRouter::dispatch_lane(&service, &message, None, 0) + .await + .unwrap(); + assert_eq!( + *log.lock().unwrap(), + ["projection-a:fact.recorded", "projection-b:fact.recorded"] + ); + // Without lanes (sequential transports), every route runs in + // registration order exactly as before lanes existed. + log.lock().unwrap().clear(); + service.dispatch_message(&message).await.unwrap(); + assert_eq!( + *log.lock().unwrap(), + [ + "projection-a:fact.recorded", + "policy:fact.recorded", + "projection-b:fact.recorded" + ] + ); + // A lane with no route for the message is an unknown-route error, never a + // silent success. + let only = Message::new("fact.process_only", MessageKind::Event, b"{}".to_vec()); + assert!(MessageRouter::dispatch_lane(&service, &only, None, 0) + .await + .is_err()); +} + +#[test] +fn service_without_named_lanes_reports_one_lane() { + use crate::bus::MessageRouter; + let log = lane_log(); + let service = Service::new().routes(lane_recorder(&log, "only")); + assert_eq!(MessageRouter::delivery_lanes(&service), 1); +} + +#[test] +#[should_panic(expected = "named delivery lane")] +fn default_lane_name_cannot_be_used_as_a_named_lane() { + let log = lane_log(); + let _ = Service::new().lane(DEFAULT_DELIVERY_LANE, lane_recorder(&log, "x")); +} diff --git a/src/outbox/message.rs b/src/outbox/message.rs index c14e1bf6f..2b5cb5e0b 100644 --- a/src/outbox/message.rs +++ b/src/outbox/message.rs @@ -297,9 +297,11 @@ impl OutboxMessage { ), ) .map_err(|error| DomainEventCaptureError::BodyEncoding(error.to_string()))?; - message.source_aggregate_type = Some(occurrence.aggregate_type().to_string()); - message.source_aggregate_id = Some(occurrence.aggregate_id().to_string()); - message.source_sequence = Some(occurrence.aggregate_sequence()); + if occurrence.external_source().is_none() { + message.source_aggregate_type = Some(occurrence.aggregate_type().to_string()); + message.source_aggregate_id = Some(occurrence.aggregate_id().to_string()); + message.source_sequence = Some(occurrence.aggregate_sequence()); + } Ok(message) } diff --git a/src/projection/plan.rs b/src/projection/plan.rs index 82b704ece..478607503 100644 --- a/src/projection/plan.rs +++ b/src/projection/plan.rs @@ -390,6 +390,14 @@ impl ResolvedProjectionPlan { if program.source_snapshots() && occurrence.derivation().is_some() { return Err(ProjectionProgramError::DerivedSourceSnapshot); } + if program.source_snapshots() + && program.external_source_snapshots() != occurrence.external_source().is_some() + { + return Err(ProjectionProgramError::InvalidOperation { + operation: program.name().into(), + reason: "snapshot origin does not match the declared source policy".into(), + }); + } let matches = program .arms() .iter() diff --git a/src/projection/program.rs b/src/projection/program.rs index 23dd83e53..67c9ffef3 100644 --- a/src/projection/program.rs +++ b/src/projection/program.rs @@ -645,6 +645,8 @@ pub struct ProjectionProgram { arms: Vec, #[serde(skip_serializing_if = "std::ops::Not::not")] source_snapshots: bool, + #[serde(skip_serializing_if = "std::ops::Not::not")] + external_source_snapshots: bool, } impl ProjectionProgram { @@ -694,6 +696,7 @@ impl ProjectionProgram { partition, arms, source_snapshots: false, + external_source_snapshots: false, }) } @@ -726,6 +729,19 @@ impl ProjectionProgram { self.source_snapshots } + /// Fence complete external snapshots by the authenticated source position. + /// Aggregate and external snapshot programs reject each other's origins. + pub fn with_external_source_snapshots(self) -> Result { + let mut program = self.with_source_snapshots()?; + program.external_source_snapshots = true; + Ok(program) + } + + /// Whether snapshots require external source provenance rather than an aggregate. + pub fn external_source_snapshots(&self) -> bool { + self.external_source_snapshots + } + /// Return the stable program name. pub fn name(&self) -> &str { &self.name diff --git a/src/projection/rebuild.rs b/src/projection/rebuild.rs index 26506ffc1..afc352f4f 100644 --- a/src/projection/rebuild.rs +++ b/src/projection/rebuild.rs @@ -6,11 +6,192 @@ use crate::projection::lower::ProjectionServerExecutorDescriptor; use crate::projection_protocol::*; use crate::table::{TableMutation, TableWritePlan}; use crate::DomainEventOccurrence; +use crate::{repository::StreamIdentity, EventRecord}; +use sha2::{Digest, Sha256}; pub(crate) const MAX_REBUILD_RECORDS: usize = 10_000; const MAX_HISTORY_EVENTS: usize = 100_000; const MAX_HISTORY_BYTES: usize = 64 * 1024 * 1024; +#[derive(PartialEq, Eq, PartialOrd, Ord)] +enum HistoryIdentity { + Aggregate(String, String, u64, u32), + External(String, String, u64, String), + Derived(String), +} + +fn history_identity(event: &DomainEventOccurrence) -> HistoryIdentity { + if event.derivation().is_some() { + HistoryIdentity::Derived(event.id().to_owned()) + } else if let Some(source) = event.external_source() { + HistoryIdentity::External( + source.producer.clone(), + source.stream.clone(), + source.position, + source.key.clone(), + ) + } else { + HistoryIdentity::Aggregate( + event.aggregate_type().into(), + event.aggregate_id().into(), + event.aggregate_sequence(), + event.publication_ordinal(), + ) + } +} + +/// Offline evidence from one original, quiescent aggregate event store. +/// +/// This does not authenticate arbitrary supplied data. The operator/source +/// adapter must retain and review the original store export and its digest. +/// Private event records are never converted into public occurrences. Explicit +/// private contracts must come from the authored aggregate, not archive absence. +/// Public JSON bodies remain authenticated retained publications, not a claimed +/// reconstruction from private payload bytes. +#[derive(Clone)] +pub struct AggregateRebuildCoverage { + stream: (String, String), + head: u64, + public: BTreeMap, + source_digest: [u8; 32], +} + +impl AggregateRebuildCoverage { + /// Validate a complete original stream against its independently captured + /// head and retained public publications. The trusted offline adapter must + /// bind `identity`, `head` and `records` to the same quiescent source store. + /// `private_contracts` is an explicit authored contract inventory, never a + /// list inferred from missing broker messages. + /// This bounded adapter supports only authored streams that emit exactly + /// one public occurrence (ordinal zero) per non-private source record. The + /// caller must verify that contract; arbitrary one-to-many emitters require + /// an independently complete publication manifest and are not supported. + pub fn from_retained_stream( + identity: &StreamIdentity, + head: u64, + records: &[EventRecord], + private_contracts: &[(&str, u64)], + public_history: &[DomainEventOccurrence], + ) -> Result { + if records.is_empty() + || head != records.len() as u64 + || records.len() > MAX_HISTORY_EVENTS + || public_history.len() > MAX_HISTORY_EVENTS + { + return Err(invalid( + "aggregate coverage requires a bounded nonempty original stream", + )); + } + let stream = ( + identity.aggregate_type().to_owned(), + identity.aggregate_id().to_owned(), + ); + let mut public = BTreeMap::new(); + let mut public_bytes = 0usize; + let mut private = BTreeSet::new(); + for contract in private_contracts { + if contract.0.is_empty() || contract.1 == 0 || !private.insert(*contract) { + return Err(invalid("invalid or duplicate private contract inventory")); + } + } + let mut positions: BTreeMap> = BTreeMap::new(); + for event in public_history.iter().filter(|event| { + event.external_source().is_none() + && event.derivation().is_none() + && event.aggregate_type() == stream.0 + && event.aggregate_id() == stream.1 + }) { + let bytes = event.canonical_bytes().map_err(invalid)?; + public_bytes = public_bytes + .checked_add(bytes.len()) + .ok_or_else(|| invalid("coverage public size overflow"))?; + if public_bytes > MAX_HISTORY_BYTES { + return Err(invalid("aggregate coverage public history exceeds 64 MiB")); + } + let fingerprint: [u8; 32] = Sha256::digest(bytes).into(); + if let Some(previous) = public.insert(event.id().into(), fingerprint) { + if previous != fingerprint { + return Err(invalid("aggregate coverage public identity conflict")); + } + continue; + } + positions + .entry(event.aggregate_sequence()) + .or_default() + .push(event); + } + let mut digest = Sha256::new(); + let mut total = 0usize; + for (offset, record) in records.iter().enumerate() { + if record.sequence != offset as u64 + 1 + || record.event_version == 0 + || record.event_name.is_empty() + || record.payload_codec.is_empty() + || record.payload_codec_version == 0 + { + return Err(invalid( + "aggregate coverage has a missing, reordered or invalid original record", + )); + } + let bytes = crate::domain_event::canonical_json_bytes(record).map_err(invalid)?; + total = total + .checked_add(bytes.len()) + .ok_or_else(|| invalid("coverage size overflow"))?; + if total > MAX_HISTORY_BYTES { + return Err(invalid("aggregate coverage exceeds 64 MiB")); + } + digest.update((bytes.len() as u64).to_be_bytes()); + digest.update(bytes); + let declared_private = + private.contains(&(record.event_name.as_str(), record.event_version)); + let occurrences = positions.remove(&record.sequence).unwrap_or_default(); + if declared_private { + if !occurrences.is_empty() { + return Err(invalid( + "private source contract conflicts with a public occurrence", + )); + } + } else { + if occurrences.is_empty() { + return Err(invalid( + "aggregate coverage omits a required public occurrence", + )); + } + if occurrences.len() != 1 || occurrences[0].publication_ordinal() != 0 { + return Err(invalid( + "aggregate coverage requires an authored single-publication stream", + )); + } + for occurrence in occurrences { + if occurrence.descriptor().name != record.event_name + || occurrence.descriptor().version != record.event_version + || record.metadata.get("causation_id").map(String::as_str) + != occurrence.causation_id() + { + return Err(invalid("aggregate coverage public occurrence differs from its original source record")); + } + } + } + } + if !positions.is_empty() { + return Err(invalid( + "aggregate coverage ends before a public occurrence", + )); + } + Ok(Self { + stream, + head, + public, + source_digest: digest.finalize().into(), + }) + } + + /// Content identity of the validated original private record sequence. + pub fn source_digest(&self) -> [u8; 32] { + self.source_digest + } +} + pub(crate) fn invalid(detail: impl ToString) -> ProjectionProtocolError { ProjectionProtocolError::InvalidBatch(detail.to_string()) } @@ -154,24 +335,50 @@ impl SnapshotProjectionRebuild { /// /// The caller must supply the entire publication history through the /// quiescent source head, not a filtered consumer window. This method checks - /// covered record identities, sequence prefixes and conflicting duplicates; + /// covered record identities, aggregate sequence prefixes and conflicting duplicates; /// it cannot discover unpublished or externally deleted source history. + /// External source positions can legitimately be sparse (including between + /// members of one source transaction). Their completeness must be certified + /// by the source adapter/archive; no aggregate prefix is invented for them. + /// Stored external row versions must still occur in the supplied history. pub fn from_complete_history( self, history: &[DomainEventOccurrence], + ) -> Result { + self.from_complete_history_with_coverage(history, &[]) + } + + /// Rebuild with independently retained original aggregate-log evidence. + /// The original public-archive-only method remains strict when no witness + /// is supplied. Every public identity captured by a witness must be present + /// with the same canonical bytes in this history. + pub fn from_complete_history_with_coverage( + self, + history: &[DomainEventOccurrence], + coverage: &[AggregateRebuildCoverage], ) -> Result { if history.len() > MAX_HISTORY_EVENTS { return Err(invalid( "snapshot rebuild history exceeds 100000 occurrences", )); } + if coverage.len() > MAX_REBUILD_RECORDS { + return Err(invalid("aggregate coverage exceeds 10000 streams")); + } let mut bytes = 0usize; let mut identities = BTreeMap::new(); + let mut logical_ids = BTreeMap::new(); let mut sequences: BTreeMap<(String, String), BTreeSet> = BTreeMap::new(); let mut rows: HashMap = HashMap::new(); let mut relevant = BTreeSet::new(); let mut versions: HashMap> = HashMap::new(); + let mut covered = BTreeMap::new(); + for witness in coverage { + if covered.insert(witness.stream.clone(), witness).is_some() { + return Err(invalid("duplicate aggregate coverage stream")); + } + } for event in history { let canonical = event.canonical_bytes().map_err(invalid)?; bytes = bytes @@ -184,15 +391,20 @@ impl SnapshotProjectionRebuild { event.aggregate_type().to_owned(), event.aggregate_id().to_owned(), ); - sequences - .entry(stream.clone()) - .or_default() - .insert(event.aggregate_sequence()); - let identity = ( - stream.clone(), - event.aggregate_sequence(), - event.publication_ordinal(), - ); + if event.external_source().is_none() && event.derivation().is_none() { + sequences + .entry(stream.clone()) + .or_default() + .insert(event.aggregate_sequence()); + } + if let Some(previous) = logical_ids.insert(event.id(), canonical.clone()) { + if previous != canonical { + return Err(invalid( + "snapshot rebuild history contains conflicting logical identities", + )); + } + } + let identity = history_identity(event); if let Some(previous) = identities.insert(identity, canonical.clone()) { if previous != canonical { return Err(invalid( @@ -204,7 +416,9 @@ impl SnapshotProjectionRebuild { if !self.executor.matches(event) { continue; } - relevant.insert(stream); + if event.external_source().is_none() && event.derivation().is_none() { + relevant.insert(stream); + } let lowered = self.executor.plan(event).map_err(invalid)?; if !lowered.resolved.source_snapshots() { return Err(invalid("snapshot rebuild resolved a non-snapshot program")); @@ -262,16 +476,49 @@ impl SnapshotProjectionRebuild { } } } + for witness in covered.values() { + for (id, expected) in &witness.public { + let bytes = logical_ids + .get(id.as_str()) + .ok_or_else(|| invalid("rebuild history omits witnessed public occurrence"))?; + if <[u8; 32]>::from(Sha256::digest(bytes)) != *expected { + return Err(invalid( + "rebuild history differs from witnessed public content", + )); + } + } + } for stream in relevant { let sequence = &sequences[&stream]; - if sequence - .iter() - .copied() - .ne(1..=sequence.last().copied().unwrap_or(0)) + if let Some(witness) = covered.get(&stream) { + if sequence.last().is_some_and(|last| *last > witness.head) { + return Err(invalid( + "public history exceeds its original source coverage", + )); + } + // Every public occurrence for a covered stream must be in the + // exact reviewed inventory, including ones irrelevant to this + // particular projection. + for event in history.iter().filter(|event| { + event.external_source().is_none() + && event.derivation().is_none() + && event.aggregate_type() == stream.0 + && event.aggregate_id() == stream.1 + }) { + if !witness.public.contains_key(event.id()) { + return Err(invalid("unwitnessed public occurrence in covered stream")); + } + } + continue; + } + if let Some((expected, found)) = (1..) + .zip(sequence.iter().copied()) + .find(|(expected, found)| expected != found) { - return Err(invalid( - "snapshot rebuild requires a complete aggregate sequence prefix", - )); + return Err(invalid(format!( + "snapshot rebuild requires a complete aggregate sequence prefix for {}/{}: expected {}, found {}", + stream.0, stream.1, expected, found, + ))); } } for current in &self.expected { diff --git a/src/projection/source_snapshot_tests.rs b/src/projection/source_snapshot_tests.rs index 44de952c4..a6c32552a 100644 --- a/src/projection/source_snapshot_tests.rs +++ b/src/projection/source_snapshot_tests.rs @@ -21,6 +21,56 @@ struct SnapshotRow { } type SourceSnapshotRows = SnapshotRow; +#[derive(serde::Serialize)] +struct ExternalRow { + id: String, + title: String, +} +impl crate::DomainEvent for ExternalRow { + const DESCRIPTOR: DomainEventDescriptor = DomainEventDescriptor { + name: std::borrow::Cow::Borrowed("external.ledger.changed"), + version: 1, + body: crate::domain_event::DomainEventBodyDescriptor::distributed_json( + crate::domain_event::DomainEventBodyKind::Event, + "ExternalRow", + 1, + "external-row-v1", + "sha256:1111111111111111111111111111111111111111111111111111111111111111", + ), + }; +} +impl DomainEventContract for ExternalRow { + const EVENT_NAME: &'static str = "external.ledger.changed"; + const EVENT_VERSION: u64 = 1; + fn descriptor() -> DomainEventDescriptor { + ::DESCRIPTOR.clone() + } +} +crate::projection! { + const EXTERNAL_SNAPSHOTS: ProjectionDescriptor = { + name: "external-rebuild-test", version: 1, epoch: "external-rebuild-v1", model: SnapshotRow, + source: external_snapshot, + on { events: [ExternalRow], mutation: SaveSnapshot, input: { row: body }, }, + }; +} +fn external_row(stream: &str, key: &str, position: u64, title: &str) -> DomainEventOccurrence { + DomainEventOccurrence::capture_external( + crate::domain_event::ExternalEventSource { + producer: "external.ledger".into(), + stream: stream.into(), + position, + key: key.into(), + }, + std::time::UNIX_EPOCH, + Default::default(), + &ExternalRow { + id: key.into(), + title: title.into(), + }, + ) + .unwrap() +} + struct Changed; impl DomainEventContract for Changed { const EVENT_NAME: &'static str = "snapshot.changed"; @@ -589,7 +639,11 @@ async fn rebuild_matrix(store: &impl ProjectionProtocolStore) { h.apply(store, &first, 1).await.unwrap().outcome, ProjectionCommitOutcome::Duplicate ); - h.apply(store, &first, 2).await.unwrap_err(); // same message, different position is still rejected + let replayed = h.apply(store, &first, 2).await.unwrap(); + assert_eq!(replayed.outcome, ProjectionCommitOutcome::Duplicate); + assert_eq!(replayed.checkpoint.unwrap().input().position(), 2); + assert_eq!(replayed.changes.len(), 1); + assert_eq!(replayed.changes[0].kind, ProjectionChangeKind::Checkpoint); h.apply(store, &removed, 3).await.unwrap(); // Exact inventory CAS rejects modifications made after begin, including inserts. @@ -666,6 +720,273 @@ async fn snapshot_rebuild_memory() { rebuild_matrix(&crate::InMemoryRepository::new()).await; } +#[tokio::test] +async fn snapshot_rebuild_original_private_log_coverage() { + use crate::projection::rebuild::{AggregateRebuildCoverage, SnapshotProjectionRebuild}; + let identity = crate::repository::StreamIdentity::new("snapshot-item", "a").unwrap(); + let first = event("a", 1, "old", false); + let third = event("a", 3, "current", false); + let history = [first.clone(), third.clone()]; + let mut records = Vec::new(); + for (sequence, name) in [ + (1, "snapshot.changed"), + (2, "snapshot.private"), + (3, "snapshot.changed"), + ] { + let mut record = crate::EventRecord::new(name, vec![sequence as u8], sequence); + record + .metadata + .insert("causation_id".into(), format!("cause-a-{sequence}")); + records.push(record); + } + let private = [("snapshot.private", 1)]; + let witness = + AggregateRebuildCoverage::from_retained_stream(&identity, 3, &records, &private, &history) + .unwrap(); + assert_eq!( + witness.source_digest(), + AggregateRebuildCoverage::from_retained_stream(&identity, 3, &records, &private, &history) + .unwrap() + .source_digest() + ); + let store = crate::InMemoryRepository::new(); + // This also exercises the original complete-prefix path and registers the + // physical owner; retain its existing rows/tombstone source versions. + rebuild_matrix(&store).await; + let projector = rebuild_projector(); + let fresh_store = crate::InMemoryRepository::new(); + let (_, binding) = projector.modeled[0].raw().unwrap(); + let physical = binding.physical_topology().unwrap(); + let compiled = CompiledProjectionTopology::from_modeled_binding( + ProjectorTopologyId::new(physical.version(), physical.name(), physical.digest()).unwrap(), + binding + .outputs() + .iter() + .map(|output| (output.model(), output.storage(), output.schema())), + ) + .unwrap(); + fresh_store + .register_projection_models(compiled.topology(), compiled.ownership()) + .await + .unwrap(); + assert!(SnapshotProjectionRebuild::begin(&fresh_store, &projector) + .await + .unwrap() + .from_complete_history(&history) + .is_err()); + let plan = SnapshotProjectionRebuild::begin(&fresh_store, &projector) + .await + .unwrap() + .from_complete_history_with_coverage(&history, &[witness.clone()]) + .unwrap(); + assert_eq!(plan.apply(&fresh_store).await.unwrap(), 1); + let read = SnapshotProjectionRebuild::begin(&fresh_store, &projector) + .await + .unwrap(); + let harness = Harness { + codec: read.context.compiled.codec(), + }; + assert_eq!( + harness + .read(&fresh_store, "a") + .await + .row + .unwrap() + .get("title"), + Some(&RowValue::String("current".into())) + ); + + for altered in [ + vec![first.clone()], + vec![first.clone(), event("a", 3, "altered", false)], + vec![ + first.clone(), + third.clone(), + event("a", 4, "unwitnessed", false), + ], + ] { + assert!(SnapshotProjectionRebuild::begin(&fresh_store, &projector) + .await + .unwrap() + .from_complete_history_with_coverage(&altered, &[witness.clone()]) + .is_err()); + } + // A witness for this stream cannot cover a missing prefix in another. + let mut mixed = history.to_vec(); + mixed.push(event("other", 2, "missing prefix", false)); + mixed.push(external_row("ledger", "external", 901, "sparse")); + assert!(SnapshotProjectionRebuild::begin(&fresh_store, &projector) + .await + .unwrap() + .from_complete_history_with_coverage(&mixed, &[witness]) + .is_err()); + let bad = |head, + records: &[crate::EventRecord], + private: &[(&str, u64)], + history: &[DomainEventOccurrence]| { + assert!(AggregateRebuildCoverage::from_retained_stream( + &identity, head, records, private, history + ) + .is_err()); + }; + bad(4, &records, &private, &history); // independently captured head, truncated export + bad(3, &records[..2], &private, &history); + bad(3, &records, &[], &history); // absence never declares a record private + bad(3, &records, &[("snapshot.private", 2)], &history); + bad(3, &records, &private, &[third]); // missing required publication + let mut multiple = history.to_vec(); + multiple.push(occurrence("a", 3, 1, "a", "second publication", false)); + bad(3, &records, &private, &multiple); + let duplicate = [history[0].clone(), history[0].clone(), history[1].clone()]; + assert!(AggregateRebuildCoverage::from_retained_stream( + &identity, 3, &records, &private, &duplicate + ) + .is_ok()); + bad( + 3, + &records, + &[("snapshot.changed", 1), ("snapshot.private", 1)], + &history, + ); + for index in [0, 2] { + let mut changed = records.clone(); + changed[index].event_name = "different.public.contract".into(); + bad(3, &changed, &private, &history); + changed = records.clone(); + changed[index].event_version += 1; + bad(3, &changed, &private, &history); + changed = records.clone(); + changed[index] + .metadata + .insert("causation_id".into(), "different-command".into()); + bad(3, &changed, &private, &history); + } + let other = crate::repository::StreamIdentity::new("snapshot-item", "wrong").unwrap(); + assert!(AggregateRebuildCoverage::from_retained_stream( + &other, 3, &records, &private, &history + ) + .is_err()); + let mut reordered = records.clone(); + reordered.swap(0, 1); + bad(3, &reordered, &private, &history); +} + +#[tokio::test] +async fn snapshot_rebuild_external_sparse_members_and_mixed_origins() { + use crate::projection::rebuild::SnapshotProjectionRebuild; + let store = crate::InMemoryRepository::new(); + let mounts = crate::LocalProjectionMountsBuilder::new("external-test", "events") + .unwrap() + .eventual_model::( + "external-rebuild", + EXTERNAL_SNAPSHOTS, + "external-rebuild-v1", + ) + .unwrap() + .build() + .unwrap(); + let projector = mounts.projector("external-rebuild").unwrap(); + async fn bootstrap( + store: &crate::InMemoryRepository, + projector: &crate::graphql::SurfaceProjector, + ) { + let (_, binding) = projector.modeled[0].raw().unwrap(); + let physical = binding.physical_topology().unwrap(); + let compiled = CompiledProjectionTopology::from_modeled_binding( + ProjectorTopologyId::new(physical.version(), physical.name(), physical.digest()) + .unwrap(), + binding + .outputs() + .iter() + .map(|output| (output.model(), output.storage(), output.schema())), + ) + .unwrap(); + store + .register_projection_models(compiled.topology(), compiled.ownership()) + .await + .unwrap(); + } + bootstrap(&store, &projector).await; + let capture = SnapshotProjectionRebuild::begin(&store, &projector) + .await + .unwrap(); + let a = external_row("ledger-a", "a", 7, "old"); + let b = external_row("ledger-a", "b", 7, "same transaction, different member"); + let newer = external_row("ledger-a", "a", 42, "new"); + let aggregate = event("aggregate", 1, "unrelated", false); + let derived = aggregate + .derive( + "ledger-indexer", + "summary", + &ExternalRow { + id: "derived".into(), + title: "not a snapshot".into(), + }, + ) + .unwrap(); + // Derived facts which match a snapshot arm remain invalid, not silently + // converted into source versions. + assert!(SnapshotProjectionRebuild::begin(&store, &projector) + .await + .unwrap() + .from_complete_history(&[a.clone(), derived.clone()]) + .is_err()); + let history = [ + newer.clone(), + b.clone(), + a.clone(), + aggregate.clone(), + a.clone(), + ]; + let plan = capture.from_complete_history(&history).unwrap(); + assert_eq!(plan.record_count(), 2); + assert_eq!(plan.apply(&store).await.unwrap(), 2); + let read = SnapshotProjectionRebuild::begin(&store, &projector) + .await + .unwrap(); + assert!( + read.from_complete_history(&[a.clone(), b.clone()]).is_err(), + "stored newer source evidence cannot disappear" + ); + for conflict in [ + external_row("ledger-a", "a", 42, "altered"), + external_row("other-ledger", "a", 43, "new"), + ] { + let mut changed = history.to_vec(); + changed.push(conflict); + assert!(SnapshotProjectionRebuild::begin(&store, &projector) + .await + .unwrap() + .from_complete_history(&changed) + .is_err()); + } + // An aggregate projector ignores unrelated external facts and derived + // events, but still requires its own complete aggregate prefix. + let aggregate_store = crate::InMemoryRepository::new(); + let aggregate_projector = rebuild_projector(); + bootstrap(&aggregate_store, &aggregate_projector).await; + let mixed = [aggregate.clone(), a, b, newer, derived]; + assert_eq!( + SnapshotProjectionRebuild::begin(&aggregate_store, &aggregate_projector) + .await + .unwrap() + .from_complete_history(&mixed) + .unwrap() + .record_count(), + 1 + ); + assert!( + SnapshotProjectionRebuild::begin(&aggregate_store, &aggregate_projector) + .await + .unwrap() + .from_complete_history(&[ + event("aggregate", 2, "missing prefix", false), + external_row("ledger", "c", 100, "unrelated") + ]) + .is_err() + ); +} + #[cfg(feature = "sqlite")] #[tokio::test] async fn snapshot_rebuild_sqlite() { diff --git a/src/projection_protocol/source_snapshot.rs b/src/projection_protocol/source_snapshot.rs index fdf8d6542..ecf624496 100644 --- a/src/projection_protocol/source_snapshot.rs +++ b/src/projection_protocol/source_snapshot.rs @@ -12,6 +12,8 @@ use sha2::{Digest, Sha256}; #[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] #[serde(deny_unknown_fields)] pub struct SourceSnapshotVersion { + #[serde(default, skip_serializing_if = "std::ops::Not::not")] + external: bool, aggregate_type: String, aggregate_id: String, sequence: u64, @@ -46,7 +48,25 @@ impl SourceSnapshotVersion { let canonical = event .canonical_bytes() .map_err(|error| ProjectionProtocolError::InvalidBatch(error.to_string()))?; + if let Some(source) = event.external_source() { + let mut stream = Sha256::new(); + stream.update(b"distributed.external-source-snapshot.v1\0"); + for part in [&source.stream, &source.key] { + stream.update((part.len() as u64).to_be_bytes()); + stream.update(part.as_bytes()); + } + return Ok(Self { + external: true, + aggregate_type: source.producer.clone(), + aggregate_id: format!("external:{:x}", stream.finalize()), + sequence: source.position, + publication_ordinal: 0, + occurrence_id: event.id().into(), + occurrence_fingerprint: Sha256::digest(canonical).into(), + }); + } Ok(Self { + external: false, aggregate_type: event.aggregate_type().into(), aggregate_id: event.aggregate_id().into(), sequence: event.aggregate_sequence(), @@ -58,7 +78,8 @@ impl SourceSnapshotVersion { /// True only when this occurrence advances the same authoritative stream. pub(crate) fn advances(&self, current: &Self) -> Result { - if self.aggregate_type != current.aggregate_type + if self.external != current.external + || self.aggregate_type != current.aggregate_type || self.aggregate_id != current.aggregate_id { return Err(ProjectionProtocolError::InvalidBatch( diff --git a/src/projection_protocol/store/identity.rs b/src/projection_protocol/store/identity.rs index 4f2aef328..2716acb2d 100644 --- a/src/projection_protocol/store/identity.rs +++ b/src/projection_protocol/store/identity.rs @@ -182,6 +182,8 @@ impl TrustedProjectionInput { #[derive(Clone, Debug, PartialEq, Eq)] pub(crate) enum ProjectionInputDisposition { Pending, + /// Identical logical input at a new adapter-authenticated broker position. + Redelivery, Duplicate(ProjectionCheckpoint), Stale(ProjectionCheckpoint), } diff --git a/src/projection_protocol/store/scenario_tests.rs b/src/projection_protocol/store/scenario_tests.rs index 251254c15..22fa5dbe8 100644 --- a/src/projection_protocol/store/scenario_tests.rs +++ b/src/projection_protocol/store/scenario_tests.rs @@ -266,6 +266,185 @@ pub(crate) async fn input_disposition_is_read_only_exact_and_repair_fenced( ); } +pub(crate) async fn identical_redelivery_advances_only_broker_checkpoint( + scenario: impl ProjectionProtocolScenario, +) { + let store = scenario.repository().await; + let first = scenario.input( + 1, + b"immutable-input", + "message-repeat", + "cause-repeat", + ProjectionGeneration::initial(), + ); + let mutation = scenario.mutation( + ProjectionRecordExpectation::Missing, + ProjectionMutationKind::Upsert, + ); + let scope = mutation.scope.clone(); + store + .commit_projection(scenario.batch(first.clone(), vec![mutation], Vec::new())) + .await + .unwrap(); + let original = store.projection_record(&scope).await.unwrap().unwrap(); + let repeat = scenario.input( + 2, + b"immutable-input", + "message-repeat", + "cause-repeat", + ProjectionGeneration::initial(), + ); + assert_eq!( + store.projection_input_disposition(&repeat).await.unwrap(), + ProjectionInputDisposition::Redelivery + ); + // Even a caller that supplies effects again cannot repeat them. + let repeated = store + .commit_projection(scenario.batch( + repeat.clone(), + vec![scenario.mutation( + ProjectionRecordExpectation::Missing, + ProjectionMutationKind::Upsert, + )], + Vec::new(), + )) + .await + .unwrap(); + assert_eq!(repeated.outcome, ProjectionCommitOutcome::Duplicate); + assert_eq!(repeated.checkpoint.unwrap().input().position(), 2); + assert_eq!(repeated.changes.len(), 1); + assert_eq!(repeated.changes[0].kind, ProjectionChangeKind::Checkpoint); + assert_eq!( + store.projection_record(&scope).await.unwrap().unwrap(), + original + ); + assert!(matches!( + store.projection_input_disposition(&repeat).await.unwrap(), + ProjectionInputDisposition::Duplicate(_) + )); + let next = scenario.input( + 3, + b"next-input", + "message-next", + "cause-next", + ProjectionGeneration::initial(), + ); + store + .commit_projection(scenario.batch(next, Vec::new(), Vec::new())) + .await + .unwrap(); + // Every alias position permanently retains the original bytes and identity. + let substitution = scenario.input( + 2, + b"immutable-input", + "substituted-message", + "cause-repeat", + ProjectionGeneration::initial(), + ); + assert!(matches!( + store.projection_input_disposition(&substitution).await, + Err(ProjectionProtocolError::InputCorruption) + )); + let altered = scenario.input( + 4, + b"changed-input", + "message-repeat", + "cause-repeat", + ProjectionGeneration::initial(), + ); + assert!(store.projection_input_disposition(&altered).await.is_err()); + let other_partition = scenario.input_for_partition( + ProjectionPartition::new(b"other-partition".to_vec()).unwrap(), + 1, + b"immutable-input", + "message-repeat", + "cause-repeat", + ); + assert!(matches!( + store.projection_input_disposition(&other_partition).await, + Err(ProjectionProtocolError::MessageIdReuse { .. }) + )); + + // A repaired generation owns a fresh execution history. An old-generation + // receipt must not suppress that generation's first actual delivery. + let failed = scenario.input( + 4, + b"repair-input", + "repair-message", + "repair-cause", + ProjectionGeneration::initial(), + ); + store + .record_projection_failure( + ProjectionFailureBatch::new( + failed, + scenario.change_epoch(), + "repair-redelivery", + "fixture", + b"failure".to_vec(), + ) + .unwrap(), + ) + .await + .unwrap(); + let generation = store + .repair_projection( + &scenario.topology(), + &scenario.partition(), + "repair-redelivery", + ) + .await + .unwrap(); + store + .commit_projection(scenario.batch( + scenario.input( + 4, + b"repair-input", + "repair-message", + "repair-cause", + generation, + ), + Vec::new(), + Vec::new(), + )) + .await + .unwrap(); + let replay = scenario.input( + 5, + b"immutable-input", + "message-repeat", + "cause-repeat", + generation, + ); + assert_eq!( + store.projection_input_disposition(&replay).await.unwrap(), + ProjectionInputDisposition::Pending + ); + let applied = store + .commit_projection(scenario.batch( + replay, + vec![scenario.mutation( + ProjectionRecordExpectation::Exact(original.revision), + ProjectionMutationKind::Upsert, + )], + Vec::new(), + )) + .await + .unwrap(); + assert_eq!(applied.outcome, ProjectionCommitOutcome::Applied); + let repeated = scenario.input( + 6, + b"immutable-input", + "message-repeat", + "cause-repeat", + generation, + ); + assert_eq!( + store.projection_input_disposition(&repeated).await.unwrap(), + ProjectionInputDisposition::Redelivery + ); +} + pub(crate) async fn message_identity_is_topology_wide_across_projection_partitions( scenario: impl ProjectionProtocolScenario, ) { diff --git a/src/sqlx_repo/projection_protocol/identity.rs b/src/sqlx_repo/projection_protocol/identity.rs index 402929790..dbddd84bb 100644 --- a/src/sqlx_repo/projection_protocol/identity.rs +++ b/src/sqlx_repo/projection_protocol/identity.rs @@ -278,6 +278,14 @@ where pub(super) fn input_identity_cursor_matches( identity: &StoredInputIdentity, input: &TrustedProjectionInput, +) -> bool { + input_identity_scope_matches(identity, input) + && identity.source_position == input.cursor.position() +} + +fn input_identity_scope_matches( + identity: &StoredInputIdentity, + input: &TrustedProjectionInput, ) -> bool { let source = input.cursor.source(); identity.partition_bytes == input.cursor.projection_partition().canonical_bytes() @@ -287,7 +295,6 @@ pub(super) fn input_identity_cursor_matches( && identity.source_partition_bytes == source.canonical_partition_bytes() && identity.source_partition_hash == digest_bytes(source.partition_digest()) && identity.source_epoch == *input.cursor.epoch() - && identity.source_position == input.cursor.position() } pub(super) fn input_identity_matches( @@ -325,7 +332,14 @@ where identity.source_hash, identity.source_partition_bytes, identity.source_partition_hash, \ identity.source_epoch, identity.source_position, identity.input_hash, \ identity.message_id, identity.causation_id, identity.gap_free \ - FROM projection_input_identities identity JOIN projection_partitions partition \ + FROM (SELECT * FROM projection_input_identities UNION ALL \ + SELECT canonical.topology_hash, alias.partition_hash, canonical.source_bytes, \ + alias.source_hash, canonical.source_partition_bytes, alias.source_partition_hash, \ + alias.source_epoch, alias.source_position, canonical.input_hash, canonical.message_id, \ + canonical.causation_id, canonical.gap_free \ + FROM projection_input_delivery_aliases alias JOIN projection_input_identities canonical \ + ON canonical.topology_hash = alias.topology_hash AND canonical.message_id = alias.message_id) identity \ + JOIN projection_partitions partition \ ON partition.topology_hash = identity.topology_hash \ AND partition.partition_hash = identity.partition_hash \ WHERE identity.topology_hash = ", @@ -857,15 +871,20 @@ where } } if let Some(identity) = input_identity_by_message_in_tx(tx, input).await? { - if !input_identity_cursor_matches(&identity, input) { + if !input_identity_scope_matches(&identity, input) { return Err(ProjectionProtocolError::MessageIdReuse { message_id: input.message_id.clone(), }); } - if !input_identity_matches(&identity, input) { - return Err(ProjectionProtocolError::InputCorruption); + if identity.input_fingerprint != input.fingerprint + || identity.causation_id != input.causation_id + || identity.gap_free != input.gap_free + { + return Err(ProjectionProtocolError::MessageIdReuse { + message_id: input.message_id.clone(), + }); } - if exact_identity.is_none() { + if exact_identity.is_none() && identity.source_position == input.cursor.position() { return Err(corrupt_storage( "projection message identity exists without its exact cursor identity", )); @@ -959,9 +978,17 @@ where )?)); } - if let Some(receipt) = receipt_by_message_in_tx(tx, input).await? { + let repeated_effect = if let Some(receipt) = receipt_by_message_in_tx(tx, input).await? { verify_stored_change(state, &receipt.change)?; - if !receipt_matches_input(&receipt, input) { + let mut original_input = input.clone(); + original_input.cursor = ProjectionInputCursor::new( + input.cursor.topology().clone(), + input.cursor.projection_partition().clone(), + input.cursor.source().clone(), + input.cursor.epoch().clone(), + receipt.source_position, + )?; + if !receipt_matches_input(&receipt, &original_input) { return Err(ProjectionProtocolError::MessageIdReuse { message_id: input.message_id.clone(), }); @@ -971,17 +998,17 @@ where "failed input receipt exists without a stopped partition", )); } - return Ok(InputDisposition::Duplicate(checkpoint_from_stored( - &input.cursor, - receipt.source_epoch, - receipt.source_position, - receipt.change, - receipt.gap_free, - )?)); - } + true + } else { + false + }; let Some(previous) = current_input_cursor_in_tx(tx, input).await? else { - return Ok(InputDisposition::New); + return Ok(if repeated_effect { + InputDisposition::Redelivery + } else { + InputDisposition::New + }); }; verify_stored_change(state, &previous.change)?; if previous.gap_free != input.gap_free { @@ -1007,7 +1034,7 @@ where { return Err(ProjectionProtocolError::InputCorruption); } - if !verify_inherited_cursor_in_tx(tx, input, &previous).await? { + if !repeated_effect && !verify_inherited_cursor_in_tx(tx, input, &previous).await? { return Err(corrupt_storage( "projection input cursor has no receipt and was not inherited by repair", )); @@ -1026,5 +1053,9 @@ where { return Err(ProjectionProtocolError::IncomparableInput); } - Ok(InputDisposition::New) + Ok(if repeated_effect { + InputDisposition::Redelivery + } else { + InputDisposition::New + }) } diff --git a/src/sqlx_repo/projection_protocol/postgres_tests.rs b/src/sqlx_repo/projection_protocol/postgres_tests.rs index 07ae6f960..357db7a49 100644 --- a/src/sqlx_repo/projection_protocol/postgres_tests.rs +++ b/src/sqlx_repo/projection_protocol/postgres_tests.rs @@ -11,20 +11,119 @@ mod postgres_tests { }; use crate::projection_protocol::{ ProjectionCheckpointProbe, ProjectionExecutionSnapshotBatchRequest, - ProjectionGraphSnapshotRequest, ProjectionQuerySnapshotRequest, - ProjectionObservationRequest, ProjectionRecordMutation, ProjectionScopeCodec, + ProjectionGraphSnapshotRequest, ProjectionObservationRequest, + ProjectionQuerySnapshotRequest, ProjectionRecordMutation, ProjectionScopeCodec, }; use crate::repository::{CommitBatch, ReadModelWritePlanStore}; use crate::table::{ ColumnType, DeleteTableRowMutation, ExpectedVersion, ForeignKey, PrimaryKey, - RelationshipDef, RelationshipKind, RowKey, RowValue, RowValues, RowWriteMode, - TableColumn, TableKind, TableRowMutation, TableSchema, TableSchemaRegistry, - TableStoreError, TableWritePlan, + RelationshipDef, RelationshipKind, RowKey, RowValue, RowValues, RowWriteMode, TableColumn, + TableKind, TableRowMutation, TableSchema, TableSchemaRegistry, TableStoreError, + TableWritePlan, }; static POSTGRES_PROJECTION_TEST_LOCK: tokio::sync::Mutex<()> = tokio::sync::Mutex::const_new(()); + #[tokio::test] + async fn postgres_redelivery_identity_is_atomic_across_partition_and_cursor_races() { + let Ok(url) = std::env::var("DISTRIBUTED_TEST_POSTGRES_URL") else { + return; + }; + let _test_guard = POSTGRES_PROJECTION_TEST_LOCK.lock().await; + let repository = SqlxRepository::::connect_and_migrate(&url) + .await + .unwrap(); + let mut registry = TableSchemaRegistry::new(); + registry.register_schema(schema().clone()).unwrap(); + repository + .bootstrap_table_schema_for_dev(®istry) + .await + .unwrap(); + repository + .register_projection_models(&topology(), &[ownership()]) + .await + .unwrap(); + let unique = uuid::Uuid::now_v7().to_string(); + let topology = topology(); + let input = |partition: &str, position: u64, id: &str, bytes: &[u8]| { + TrustedProjectionInput::mint( + ProjectionInputCursor::new( + topology.clone(), + ProjectionPartition::new(format!("{unique}-{partition}").into_bytes()).unwrap(), + ProjectionSource::new("broker", b"retained-stream".to_vec()).unwrap(), + ProjectionEpoch::new("broker-v1").unwrap(), + position, + ) + .unwrap(), + ProjectionInputFingerprint::from_canonical_bytes(bytes), + format!("{unique}-{id}"), + "cause", + ProjectionGeneration::initial(), + false, + ) + .unwrap() + }; + let batch = |input| ProjectionCommitBatch { + input, + change_epoch: ProjectionEpoch::new("changes-v1").unwrap(), + ownership: vec![ownership()], + mutations: Vec::new(), + observations: Vec::new(), + }; + // The canonical unique topology/message binding arbitrates different + // partition locks: only one scope/content may ever claim this identity. + let (a, b) = tokio::join!( + repository.commit_projection(batch(input("a", 1, "shared", b"one"))), + repository.commit_projection(batch(input("b", 1, "shared", b"two"))), + ); + assert_eq!( + usize::from(a.is_ok()) + usize::from(b.is_ok()), + 1, + "a={a:?}, b={b:?}" + ); + let (partition, bytes) = if a.is_ok() { + ("a", b"one".as_slice()) + } else { + ("b", b"two".as_slice()) + }; + let duplicate = input(partition, 2, "shared", bytes); + assert_eq!( + repository + .projection_input_disposition(&duplicate) + .await + .unwrap(), + ProjectionInputDisposition::Redelivery + ); + // Alias and new canonical input race for exactly the same broker cursor. + // Partition serialization must forbid a cross-table double binding. + let (alias, canonical) = tokio::join!( + repository.commit_projection(batch(duplicate.clone())), + repository.commit_projection(batch(input(partition, 2, "different", b"different"))), + ); + assert_eq!( + usize::from(alias.is_ok()) + usize::from(canonical.is_ok()), + 1 + ); + let loser = if alias.is_ok() { + input(partition, 2, "different", b"different") + } else { + duplicate + }; + assert!(matches!( + repository.projection_input_disposition(&loser).await, + Err(ProjectionProtocolError::InputCorruption) + )); + let repeated = repository + .commit_projection(batch(input(partition, 3, "shared", bytes))) + .await + .unwrap(); + assert_eq!(repeated.outcome, ProjectionCommitOutcome::Duplicate); + assert_eq!(repeated.checkpoint.unwrap().input().position(), 3); + assert_eq!(repeated.changes.len(), 1); + assert_eq!(repeated.changes[0].kind, ProjectionChangeKind::Checkpoint); + } + fn topology() -> ProjectorTopologyId { ProjectorTopologyId::new(1, "postgres_projection_runtime", [71; 32]).unwrap() } @@ -136,16 +235,10 @@ mod postgres_tests { fn graph_ownership() -> Vec { vec![ - ProjectionModelOwnership::new( - "PostgresGraphParentView", - "postgres_graph_parent_views", - ) - .unwrap(), - ProjectionModelOwnership::new( - "PostgresGraphChildView", - "postgres_graph_child_views", - ) - .unwrap(), + ProjectionModelOwnership::new("PostgresGraphParentView", "postgres_graph_parent_views") + .unwrap(), + ProjectionModelOwnership::new("PostgresGraphChildView", "postgres_graph_child_views") + .unwrap(), ] } @@ -163,12 +256,14 @@ mod postgres_tests { fn graph_key(model: &str) -> RowKey { RowKey::new([( "id", - RowValue::String(match model { - "PostgresGraphParentView" => "parent-1", - "PostgresGraphChildView" => "child-1", - other => panic!("unknown PostgreSQL graph model {other}"), - } - .into()), + RowValue::String( + match model { + "PostgresGraphParentView" => "parent-1", + "PostgresGraphChildView" => "child-1", + other => panic!("unknown PostgreSQL graph model {other}"), + } + .into(), + ), )]) } @@ -240,10 +335,7 @@ mod postgres_tests { ProjectionGraphSnapshotRequest::new( root, [ - ( - "children".into(), - Arc::new(graph_child_schema().clone()), - ), + ("children".into(), Arc::new(graph_child_schema().clone())), ( "featured_children".into(), Arc::new(graph_child_schema().clone()), @@ -318,10 +410,7 @@ mod postgres_tests { } fn matrix_key(scenario: usize, index: usize) -> RowKey { - RowKey::new([( - "id", - RowValue::String(format!("matrix-{scenario}-{index}")), - )]) + RowKey::new([("id", RowValue::String(format!("matrix-{scenario}-{index}")))]) } fn matrix_scope(scenario: usize, index: usize) -> ProjectionRecordScope { @@ -341,10 +430,7 @@ mod postgres_tests { expectation: ProjectionRecordExpectation, ) -> ProjectionRecordMutation { let mut values = RowValues::new(); - values.insert( - "id", - RowValue::String(format!("matrix-{scenario}-{index}")), - ); + values.insert("id", RowValue::String(format!("matrix-{scenario}-{index}"))); values.insert("value", RowValue::String(value.into())); ProjectionRecordMutation::new( matrix_scope(scenario, index), @@ -386,15 +472,10 @@ mod postgres_tests { .unwrap() } - fn matrix_snapshot_request( - scenario: usize, - index: usize, - ) -> ProjectionQuerySnapshotRequest { + fn matrix_snapshot_request(scenario: usize, index: usize) -> ProjectionQuerySnapshotRequest { ProjectionQuerySnapshotRequest::new( &matrix_codec(), - Some(&serde_json::json!(format!( - "postgres-matrix-{scenario}" - ))), + Some(&serde_json::json!(format!("postgres-matrix-{scenario}"))), "PostgresProjectionMatrixView", matrix_key(scenario, index), Vec::new(), @@ -788,9 +869,9 @@ mod postgres_tests { assert!( matches!( repository.commit_projection(failed_batch()).await, - Err(ProjectionProtocolError::Table(TableStoreError::BackendStorage { - .. - })) + Err(ProjectionProtocolError::Table( + TableStoreError::BackendStorage { .. } + )) ), "failure position {fail_at}" ); @@ -822,7 +903,10 @@ mod postgres_tests { ) }) .collect::>(); - assert_eq!(physical_after, physical_before, "failure position {fail_at}"); + assert_eq!( + physical_after, physical_before, + "failure position {fail_at}" + ); for (index, expected) in snapshots_before.iter().enumerate() { assert_eq!( &repository diff --git a/src/sqlx_repo/projection_protocol/store_impl.rs b/src/sqlx_repo/projection_protocol/store_impl.rs index 353378e58..021651a9e 100644 --- a/src/sqlx_repo/projection_protocol/store_impl.rs +++ b/src/sqlx_repo/projection_protocol/store_impl.rs @@ -332,12 +332,12 @@ where fn commit_projection( &self, - batch: ProjectionCommitBatch, + mut batch: ProjectionCommitBatch, ) -> impl Future> + Send + '_ { async move { batch.validate()?; - let write_plan = TableWritePlan::new( + let mut write_plan = TableWritePlan::new( batch .mutations .iter() @@ -357,24 +357,35 @@ where lock_partition_in_tx(&mut tx, &topology, &partition, &batch.change_epoch).await?; validate_input_identity_in_tx(&mut tx, &batch.input).await?; ensure_active_input(&state, &batch.input)?; - match classify_validated_input_in_tx(&mut tx, &batch.input, &state).await? { - InputDisposition::Duplicate(checkpoint) => { - return Ok(ProjectionCommitResult::not_applied( - ProjectionCommitOutcome::Duplicate, - Some(checkpoint), - )); - } - InputDisposition::Stale(checkpoint) => { - return Ok(ProjectionCommitResult::not_applied( - ProjectionCommitOutcome::StaleInput, - Some(checkpoint), - )); - } - InputDisposition::New => { - ensure_pending_retry_input_in_tx(&mut tx, &state, &batch.input).await?; - } + let redelivery = + match classify_validated_input_in_tx(&mut tx, &batch.input, &state).await? { + InputDisposition::Duplicate(checkpoint) => { + return Ok(ProjectionCommitResult::not_applied( + ProjectionCommitOutcome::Duplicate, + Some(checkpoint), + )); + } + InputDisposition::Stale(checkpoint) => { + return Ok(ProjectionCommitResult::not_applied( + ProjectionCommitOutcome::StaleInput, + Some(checkpoint), + )); + } + InputDisposition::New => { + ensure_pending_retry_input_in_tx(&mut tx, &state, &batch.input).await?; + false + } + InputDisposition::Redelivery => { + ensure_pending_retry_input_in_tx(&mut tx, &state, &batch.input).await?; + batch.mutations.clear(); + batch.observations.clear(); + write_plan = TableWritePlan::new(Vec::new()); + true + } + }; + if !redelivery { + ensure_inbox_available_in_tx(&mut tx, &batch.input).await?; } - ensure_inbox_available_in_tx(&mut tx, &batch.input).await?; ensure_partition_ownership_in_tx(&mut tx, &topology, &partition, &batch.ownership) .await?; @@ -585,9 +596,11 @@ where } store_input_cursor_in_tx(&mut tx, &batch.input, &final_change).await?; insert_input_identity_in_tx(&mut tx, &batch.input).await?; - insert_input_receipt_in_tx(&mut tx, &batch.input, "applied", None, &final_change) - .await?; - insert_inbox_in_tx(&mut tx, &batch.input).await?; + if !redelivery { + insert_input_receipt_in_tx(&mut tx, &batch.input, "applied", None, &final_change) + .await?; + insert_inbox_in_tx(&mut tx, &batch.input).await?; + } update_partition_head_in_tx( &mut tx, &topology, @@ -621,7 +634,11 @@ where tables: changed_tables, }); Ok(ProjectionCommitResult { - outcome: ProjectionCommitOutcome::Applied, + outcome: if redelivery { + ProjectionCommitOutcome::Duplicate + } else { + ProjectionCommitOutcome::Applied + }, checkpoint: Some(checkpoint), records, changes, @@ -710,7 +727,9 @@ where InputDisposition::New => { ensure_pending_retry_input_in_tx(&mut tx, &state, &batch.input).await?; } - InputDisposition::Duplicate(_) | InputDisposition::Stale(_) => { + InputDisposition::Duplicate(_) + | InputDisposition::Stale(_) + | InputDisposition::Redelivery => { return Err(ProjectionProtocolError::InvalidBatch( "cannot record terminal failure for an already processed input".into(), )); @@ -907,6 +926,10 @@ where ensure_pending_retry_input_in_tx(&mut tx, &state, input).await?; Ok(ProjectionInputDisposition::Pending) } + InputDisposition::Redelivery => { + ensure_pending_retry_input_in_tx(&mut tx, &state, input).await?; + Ok(ProjectionInputDisposition::Redelivery) + } InputDisposition::Duplicate(checkpoint) => { Ok(ProjectionInputDisposition::Duplicate(checkpoint)) } diff --git a/src/sqlx_repo/projection_protocol/tests.rs b/src/sqlx_repo/projection_protocol/tests.rs index f91a3a8b7..66c0277c3 100644 --- a/src/sqlx_repo/projection_protocol/tests.rs +++ b/src/sqlx_repo/projection_protocol/tests.rs @@ -16,15 +16,14 @@ mod tests { }; use crate::projection_protocol::{ ProjectionCheckpointProbe, ProjectionExecutionSnapshotBatchRequest, - ProjectionGraphSnapshotRequest, ProjectionObservationRequest, - ProjectionLiveRecordRequest, ProjectionQuerySnapshotRequest, ProjectionRecordMutation, - ProjectionScopeCodec, + ProjectionGraphSnapshotRequest, ProjectionLiveRecordRequest, ProjectionObservationRequest, + ProjectionQuerySnapshotRequest, ProjectionRecordMutation, ProjectionScopeCodec, }; use crate::repository::{CommitBatch, ReadModelWritePlanStore, TransactionalCommit}; use crate::table::{ ColumnType, DeleteTableRowMutation, ExpectedVersion, ForeignKey, PrimaryKey, - RelationshipDef, RelationshipKind, RowKey, RowValue, RowValues, RowWriteMode, - TableColumn, TableKind, TableRowMutation, TableSchema, TableSchemaRegistry, + RelationshipDef, RelationshipKind, RowKey, RowValue, RowValues, RowWriteMode, TableColumn, + TableKind, TableRowMutation, TableSchema, TableSchemaRegistry, }; fn topology() -> ProjectorTopologyId { @@ -231,12 +230,14 @@ mod tests { fn graph_key(model: &str) -> RowKey { RowKey::new([( "id", - RowValue::String(match model { - "SqlGraphParentView" => "parent-1", - "SqlGraphChildView" => "child-1", - other => panic!("unknown SQL graph model {other}"), - } - .into()), + RowValue::String( + match model { + "SqlGraphParentView" => "parent-1", + "SqlGraphChildView" => "child-1", + other => panic!("unknown SQL graph model {other}"), + } + .into(), + ), )]) } @@ -289,10 +290,7 @@ mod tests { ProjectionGraphSnapshotRequest::new( root, [ - ( - "children".into(), - Arc::new(graph_child_schema().clone()), - ), + ("children".into(), Arc::new(graph_child_schema().clone())), ( "featured_children".into(), Arc::new(graph_child_schema().clone()), @@ -751,6 +749,11 @@ mod tests { .await; } + #[tokio::test] + async fn sqlite_identical_redelivery_advances_only_broker_checkpoint() { + crate::projection_protocol::scenario_tests::identical_redelivery_advances_only_broker_checkpoint(ProjectionScenario).await; + } + #[tokio::test] async fn sqlite_obligation_and_unpartitioned_live_evidence_are_exact_and_durable() { let evidence_repository = repository().await; @@ -1052,8 +1055,8 @@ mod tests { } #[tokio::test] - async fn sqlite_modeled_projection_identity_is_durable_across_restart_and_null_history_stays_readable() - { + async fn sqlite_modeled_projection_identity_is_durable_across_restart_and_null_history_stays_readable( + ) { let (repository, database_path) = wal_repository_with_retention(16).await; let program_a = semantic_program_id('a'); let scope = record_scope(); @@ -1104,7 +1107,11 @@ mod tests { ProjectionChangeRead::Changes { changes, .. } => changes, other => panic!("restarted repository must retain projection changes: {other:?}"), }; - assert_eq!(changes.len(), 1, "the staged observation shares the record change"); + assert_eq!( + changes.len(), + 1, + "the staged observation shares the record change" + ); assert!(changes .iter() .all(|change| change.program_id == Some(program_a))); @@ -1117,20 +1124,16 @@ mod tests { // Rows written before semantic identities existed remain useful after // migration, but their null identity cannot mint modeled proof. - sqlx::query( - "UPDATE projection_changes SET program_id = NULL WHERE causation_id = ?", - ) - .bind("semantic-identity-cause-a") - .execute(reopened.pool()) - .await - .unwrap(); - sqlx::query( - "UPDATE projection_observations SET program_id = NULL WHERE causation_id = ?", - ) - .bind("semantic-identity-cause-a") - .execute(reopened.pool()) - .await - .unwrap(); + sqlx::query("UPDATE projection_changes SET program_id = NULL WHERE causation_id = ?") + .bind("semantic-identity-cause-a") + .execute(reopened.pool()) + .await + .unwrap(); + sqlx::query("UPDATE projection_observations SET program_id = NULL WHERE causation_id = ?") + .bind("semantic-identity-cause-a") + .execute(reopened.pool()) + .await + .unwrap(); let null_history = reopened .projection_causation_evidence(&selected) .await @@ -1139,16 +1142,24 @@ mod tests { assert_eq!(null_history.observations[0].program_id, None); let readable = reopened .projection_live_record_batch( - &ProjectionLiveRecordBatchRequest::new(vec![ - ProjectionLiveRecordRequest::new(&scope_codec(), "SqlTodoView", record_key()) - .unwrap(), - ]) + &ProjectionLiveRecordBatchRequest::new(vec![ProjectionLiveRecordRequest::new( + &scope_codec(), + "SqlTodoView", + record_key(), + ) + .unwrap()]) .unwrap(), ) .await .unwrap(); - assert!(readable.records[0].is_some(), "unversioned rows remain readable"); - assert_eq!(readable.records[0].as_ref().unwrap().revision.scope(), &scope); + assert!( + readable.records[0].is_some(), + "unversioned rows remain readable" + ); + assert_eq!( + readable.records[0].as_ref().unwrap().revision.scope(), + &scope + ); remove_wal_database(reopened, &database_path).await; } @@ -2259,9 +2270,9 @@ mod tests { assert!( matches!( repository.commit_projection(failed_batch()).await, - Err(ProjectionProtocolError::Table(TableStoreError::BackendStorage { - .. - })) + Err(ProjectionProtocolError::Table( + TableStoreError::BackendStorage { .. } + )) ), "failure position {fail_at}" ); @@ -2286,7 +2297,10 @@ mod tests { ) }) .collect::>(); - assert_eq!(physical_after, physical_before, "failure position {fail_at}"); + assert_eq!( + physical_after, physical_before, + "failure position {fail_at}" + ); for (index, expected) in snapshots_before.iter().enumerate() { assert_eq!( &repository @@ -2955,8 +2969,15 @@ mod tests { let (cache_store, cache_pool, cache_tables, cache_before) = { let pool = crate::graphql::GraphqlPool::Sqlite(repository.pool().clone()); let tables = vec!["sql_todo_views".to_owned()]; - let store = crate::graphql::delivery::GatewayVersionStore::install(&pool, "atomic-cache-test", tables.clone()).await.unwrap(); - let before = serde_json::to_value(store.current(&pool, &tables).await.unwrap()).unwrap(); + let store = crate::graphql::delivery::GatewayVersionStore::install( + &pool, + "atomic-cache-test", + tables.clone(), + ) + .await + .unwrap(); + let before = + serde_json::to_value(store.current(&pool, &tables).await.unwrap()).unwrap(); (store, pool, tables, before) }; let command_id = uuid::Uuid::now_v7().hyphenated().to_string(); @@ -3010,8 +3031,17 @@ mod tests { #[cfg(all(feature = "graphql", feature = "gateway-delivery"))] let cache_committed = { - let committed = serde_json::to_value(cache_store.current(&cache_pool, &cache_tables).await.unwrap()).unwrap(); - assert_ne!(committed, cache_before, "Atomic result and cache dependency versions commit together"); + let committed = serde_json::to_value( + cache_store + .current(&cache_pool, &cache_tables) + .await + .unwrap(), + ) + .unwrap(); + assert_ne!( + committed, cache_before, + "Atomic result and cache dependency versions commit together" + ); committed }; let metadata = repository @@ -3109,8 +3139,17 @@ mod tests { CommandLookup::InProgress { .. } )); #[cfg(all(feature = "graphql", feature = "gateway-delivery"))] - assert_eq!(serde_json::to_value(cache_store.current(&cache_pool, &cache_tables).await.unwrap()).unwrap(), cache_committed, - "failed Atomic ledger completion must roll back cache versions too"); + assert_eq!( + serde_json::to_value( + cache_store + .current(&cache_pool, &cache_tables) + .await + .unwrap() + ) + .unwrap(), + cache_committed, + "failed Atomic ledger completion must roll back cache versions too" + ); sqlx::query("DROP TRIGGER fail_direct_ledger_completion") .execute(repository.pool()) .await diff --git a/src/sqlx_repo/projection_protocol/types.rs b/src/sqlx_repo/projection_protocol/types.rs index 27f6e6b0e..7c152d1be 100644 --- a/src/sqlx_repo/projection_protocol/types.rs +++ b/src/sqlx_repo/projection_protocol/types.rs @@ -64,6 +64,7 @@ pub(super) struct StoredFailure { pub(super) enum InputDisposition { New, + Redelivery, Duplicate(ProjectionCheckpoint), Stale(ProjectionCheckpoint), } diff --git a/src/sqlx_repo/projection_protocol/writes.rs b/src/sqlx_repo/projection_protocol/writes.rs index fd5472a61..4fc24a8dc 100644 --- a/src/sqlx_repo/projection_protocol/writes.rs +++ b/src/sqlx_repo/projection_protocol/writes.rs @@ -975,9 +975,36 @@ where return Err(ProjectionProtocolError::InputCorruption); } if input_identity_by_message_in_tx(tx, input).await?.is_some() { - return Err(ProjectionProtocolError::MessageIdReuse { - message_id: input.message_id.clone(), - }); + // The canonical unique binding wins races across projection partitions. + // Revalidate against it before retaining a distinct broker delivery. + validate_input_identity_in_tx(tx, input).await?; + let mut alias = QueryBuilder::::new( + "INSERT INTO projection_input_delivery_aliases \ + (topology_hash, partition_hash, source_hash, source_partition_hash, \ + source_epoch, source_position, message_id) VALUES (", + ); + alias.push_bind(topology_hash.as_slice()); + alias.push(", ").push_bind(partition_hash.as_slice()); + alias.push(", ").push_bind(source_hash.as_slice()); + alias.push(", ").push_bind(source_partition_hash.as_slice()); + alias.push(", ").push_bind(cursor.epoch().as_str()); + alias.push(", ").push_bind(to_i64::( + cursor.position(), + "projection alias position", + )?); + alias.push(", ").push_bind(input.message_id.as_str()); + alias.push(") ON CONFLICT DO NOTHING"); + alias.build().execute(&mut **tx).await.map_err(|error| { + protocol_storage_error::("insert projection delivery alias", error) + })?; + let stored = input_identity_by_cursor_in_tx(tx, input) + .await? + .ok_or_else(|| corrupt_storage("projection alias insert has no readable identity"))?; + return if input_identity_matches(&stored, input) { + Ok(()) + } else { + Err(ProjectionProtocolError::InputCorruption) + }; } Err(corrupt_storage( "projection input identity collided without a readable conflicting row", diff --git a/src/sqlx_repo/repo/backend.rs b/src/sqlx_repo/repo/backend.rs index 71eeac174..8b749fb1c 100644 --- a/src/sqlx_repo/repo/backend.rs +++ b/src/sqlx_repo/repo/backend.rs @@ -47,7 +47,7 @@ mod tests { .iter() .map(|migration| migration.sql) .collect::>(); - assert_eq!(versions, vec![1, 2, 3, 4, 5, 6, 7, 8]); + assert_eq!(versions, vec![1, 2, 3, 4, 5, 6, 7, 8, 9]); assert_eq!( descriptions, vec![ @@ -58,7 +58,8 @@ mod tests { "projection source snapshots", "gateway dependency versions", "projection program identity", - "external command binding" + "external command binding", + "projection delivery aliases" ] ); assert_eq!( @@ -96,6 +97,10 @@ mod tests { env!("CARGO_MANIFEST_DIR"), "/migrations/sqlite/0008_external_command_binding.sql" )), + include_str!(concat!( + env!("CARGO_MANIFEST_DIR"), + "/migrations/sqlite/0009_projection_delivery_aliases.sql" + )), ] ); } @@ -115,7 +120,7 @@ mod tests { .iter() .map(|migration| migration.sql) .collect::>(); - assert_eq!(versions, vec![1, 2, 3, 4, 5, 6, 7, 8]); + assert_eq!(versions, vec![1, 2, 3, 4, 5, 6, 7, 8, 9]); assert_eq!( descriptions, vec![ @@ -126,7 +131,8 @@ mod tests { "projection source snapshots", "gateway dependency versions", "projection program identity", - "external command binding" + "external command binding", + "projection delivery aliases" ] ); assert_eq!( @@ -164,6 +170,10 @@ mod tests { env!("CARGO_MANIFEST_DIR"), "/migrations/postgres/0008_external_command_binding.sql" )), + include_str!(concat!( + env!("CARGO_MANIFEST_DIR"), + "/migrations/postgres/0009_projection_delivery_aliases.sql" + )), ] ); } diff --git a/src/telemetry.rs b/src/telemetry.rs index e538ce152..2a318b932 100644 --- a/src/telemetry.rs +++ b/src/telemetry.rs @@ -162,7 +162,7 @@ pub(crate) fn handler_error_status(error: &HandlerError) -> &'static str { | HandlerError::ProjectionTerminalRecorded { .. } | HandlerError::ProjectionDeliveryHalted { .. } => dispatch_status::REPOSITORY_ERROR, HandlerError::GuardRejected(_) => dispatch_status::GUARD_REJECTED, - HandlerError::Other(_) => dispatch_status::OTHER_ERROR, + HandlerError::ApplicationReloading | HandlerError::Other(_) => dispatch_status::OTHER_ERROR, } } diff --git a/tests/graphql_query_protocol/main.rs b/tests/graphql_query_protocol/main.rs index 85378dafc..7aede5674 100644 --- a/tests/graphql_query_protocol/main.rs +++ b/tests/graphql_query_protocol/main.rs @@ -949,6 +949,132 @@ async fn live_subscription_frames_keep_data_and_metadata_in_fifo_order() { ); } +async fn rename_causal_table(repository: &SqliteRepository, from: &str, to: &str) { + sqlx::query(sqlx::AssertSqlSafe(format!( + "ALTER TABLE {from} RENAME TO {to}" + ))) + .execute(repository.pool()) + .await + .expect("rename live read-model table"); +} + +/// A failure frame is error-only, carries the valid base envelope, and has no +/// snapshot, live or command metadata (`docs/live-query-delivery.md`). +fn assert_live_failure_frame(response: async_graphql::Response) { + assert!(response.is_err(), "expected a failed live execution"); + let response = serde_json::to_value(response).expect("GraphQL wire response"); + assert!( + response.get("data").is_none_or(Value::is_null), + "a failure frame carries no result data: {response}" + ); + let errors = response["errors"].as_array().expect("failure errors"); + assert!(!errors.is_empty(), "{response}"); + assert!( + errors.iter().all(|error| error["message"] + .as_str() + .is_some_and(|message| !message.is_empty())), + "{response}" + ); + let distributed = distributed_envelope(&response); + assert_eq!(distributed["protocolVersion"], 1, "{response}"); + for field in [ + "schemaHash", + "authorizationGeneration", + "cacheScope", + "operation", + ] { + assert!( + distributed[field] + .as_str() + .is_some_and(|value| !value.is_empty()), + "failure frame base envelope requires `{field}`: {response}" + ); + } + assert!( + distributed + .get("trustedPresets") + .is_none_or(Value::is_array), + "{response}" + ); + for field in ["snapshot", "live", "command"] { + assert!( + distributed.get(field).is_none(), + "failure frame must not carry `{field}`: {response}" + ); + } +} + +async fn assert_stream_ended(stream: &mut BoxStream<'static, async_graphql::Response>) { + let next = tokio::time::timeout(Duration::from_secs(2), stream.next()) + .await + .expect("a failed live subscription must complete"); + assert!( + next.is_none(), + "a failure frame is terminal; unexpected frame: {:?}", + next.map(serde_json::to_value) + ); +} + +#[tokio::test] +async fn live_subscription_initial_failure_is_one_terminal_error_only_frame() { + let fixture = protocol_fixture_with_rows().await; + rename_causal_table( + &fixture.repository, + "causal_query_views", + "causal_query_views_offline", + ) + .await; + let mut stream = fixture + .engine + .execute_stream(&user_session(), Request::new(LIVE_SUBSCRIPTION)); + + let failure = tokio::time::timeout(Duration::from_secs(2), stream.next()) + .await + .expect("timeout waiting for failure frame") + .expect("a failed live execution still emits one frame"); + assert_live_failure_frame(failure); + assert_stream_ended(&mut stream).await; +} + +#[tokio::test] +async fn live_subscription_refresh_failure_is_terminal_and_never_takes_later_metadata() { + let fixture = protocol_fixture_with_rows().await; + let mut stream = fixture + .engine + .execute_stream(&user_session(), Request::new(LIVE_SUBSCRIPTION)); + let first = next_wire_frame(&mut stream).await; + assert_live_frame(&first, "causal_query_views", "causal row", "1", false); + + rename_causal_table( + &fixture.repository, + "causal_query_views", + "causal_query_views_offline", + ) + .await; + fixture + .repository + .publish_read_model_change(distributed::ReadModelChange::new(["causal_query_views"])); + // Leave the failure unconsumed while storage recovers and a new projection + // commits. A producer that kept running would enqueue that result's + // snapshot/live metadata ahead of the failure response's envelope. + tokio::time::sleep(Duration::from_millis(400)).await; + rename_causal_table( + &fixture.repository, + "causal_query_views_offline", + "causal_query_views", + ) + .await; + project_item(&fixture.repository, &fixture.bus, 2, "causal row 2").await; + tokio::time::sleep(Duration::from_millis(400)).await; + + let failure = tokio::time::timeout(Duration::from_secs(2), stream.next()) + .await + .expect("timeout waiting for failure frame") + .expect("the refresh failure is delivered"); + assert_live_failure_frame(failure); + assert_stream_ended(&mut stream).await; +} + #[tokio::test] async fn live_subscription_replays_delete_tombstone_and_observation() { let fixture = protocol_fixture_with_rows().await; diff --git a/tests/nats_transport/main.rs b/tests/nats_transport/main.rs index 9b6c293b5..5f35b00fe 100644 --- a/tests/nats_transport/main.rs +++ b/tests/nats_transport/main.rs @@ -28,6 +28,217 @@ fn nats_url() -> Option { env_support::broker_env("NATS_URL", "nats transport test") } +#[derive(serde::Serialize, distributed::DomainEvent)] +#[domain_event(name = "ledger.external_balance", version = 1)] +struct ExternalBalance { + value: u64, +} + +fn external_balance(position: u64, value: u64) -> distributed::DomainEventOccurrence { + distributed::DomainEventOccurrence::capture_external( + distributed::ExternalEventSource { + producer: "ledger-webhook".into(), + stream: "account-one".into(), + position, + key: "balance".into(), + }, + std::time::UNIX_EPOCH, + Default::default(), + &ExternalBalance { value }, + ) + .unwrap() +} + +#[tokio::test] +async fn external_identity_content_conflicts_and_retained_archive_survive_broker_dedup() { + use distributed::bus::{Bus, MessageSource, ReceivedMessage}; + let Some(url) = nats_url() else { return }; + let namespace = unique("external_archive"); + let bus = NatsBus::connect(&url).namespace(&namespace).await.unwrap(); + let stream = bus.ensure_stream().await.unwrap(); + let stream_name = stream.cached_info().config.name.clone(); + let mut source = NatsJetStreamSource::connect( + &url, + &stream_name, + vec![format!("{namespace}.>")], + &unique("external_consumer"), + ) + .await + .unwrap() + .with_strip_prefix(format!("{namespace}.evt.")); + let first = external_balance(1, 10); + let altered = external_balance(1, 11); + let next = external_balance(2, 10); + assert_eq!(first.id(), altered.id()); + for event in [&first, &first, &altered, &next] { + bus.publish_message( + distributed::OutboxMessage::from_domain_event_occurrence(event) + .unwrap() + .into(), + ) + .await + .unwrap(); + } + // Identical retry deduplicates, altered bytes with the same logical source + // identity must remain visible to the permanent projection conflict fence. + for expected in [&first, &altered, &next] { + let received = source.recv().await.unwrap().unwrap(); + assert_eq!(received.message().id(), Some(expected.id())); + assert_eq!( + received.message().payload(), + expected.canonical_bytes().unwrap() + ); + received.ack().await.unwrap(); + } + assert!(source.recv().await.unwrap().is_none()); + let archive = bus.retained_domain_events().await.unwrap(); + assert_eq!(archive.len(), 3); + assert_eq!(bus.retained_domain_events().await.unwrap(), archive); + let js = async_nats::jetstream::new(async_nats::connect(&url).await.unwrap()); + js.delete_stream(&stream_name).await.unwrap(); +} + +#[tokio::test] +async fn archive_and_live_decode_reject_the_same_poisoned_identity_and_kind_headers() { + use distributed::bus::MessageSource; + use sha2::{Digest, Sha256}; + let Some(url) = nats_url() else { return }; + let js = async_nats::jetstream::new(async_nats::connect(&url).await.unwrap()); + for case in [ + "duplicate", + "wrong-kind", + "missing-kind", + "lowercase-aggregate", + ] { + let namespace = unique("archive_poison"); + let bus = NatsBus::connect(&url).namespace(&namespace).await.unwrap(); + let stream = bus.ensure_stream().await.unwrap(); + let stream_name = stream.cached_info().config.name.clone(); + let mut source = NatsJetStreamSource::connect( + &url, + &stream_name, + vec![format!("{namespace}.>")], + &unique("consumer"), + ) + .await + .unwrap() + .with_strip_prefix(format!("{namespace}.evt.")); + let mut event = external_balance(1, 10); + if case == "lowercase-aggregate" { + let mut entity = distributed::Entity::with_id("account"); + entity.digest("fixture", &()).unwrap(); + entity + .capture_domain_event("ledger", &ExternalBalance { value: 10 }) + .unwrap(); + event = entity.pending_domain_events()[0].clone(); + } + let bytes = event.canonical_bytes().unwrap(); + let mut headers = async_nats::HeaderMap::new(); + headers.insert( + "x-sourced-payload-codec", + "distributed.domain-event-occurrence+json", + ); + if case != "missing-kind" { + headers.insert( + "X-Sourced-Kind", + if case == "wrong-kind" { + "command" + } else { + "event" + }, + ); + } + if event.external_source().is_some() { + let mut hash = Sha256::new(); + hash.update(b"distributed.nats.external-occurrence.v1\0"); + hash.update(&bytes); + headers.insert( + "Nats-Msg-Id", + format!("external:sha256:{:x}", hash.finalize()), + ); + headers.insert("X-Distributed-Occurrence-Id", event.id()); + if case == "duplicate" { + headers.append("x-distributed-occurrence-id", "forged-extra"); + } + } else { + headers.insert("Nats-Msg-Id", event.id()); + headers.insert("x-distributed-occurrence-id", "forged-logical"); + } + js.publish_with_headers( + format!("{namespace}.evt.{}", event.descriptor().name), + headers, + bytes.into(), + ) + .await + .unwrap() + .await + .unwrap(); + assert!( + bus.retained_domain_events().await.is_err(), + "archive accepted {case}" + ); + assert!(source.recv().await.is_err(), "live accepted {case}"); + js.delete_stream(&stream_name).await.unwrap(); + } +} + +#[tokio::test] +async fn malformed_external_headers_fail_closed_without_ack_or_false_progress() { + use distributed::bus::MessageSource; + let Some(url) = nats_url() else { return }; + let prefix = unique("external_poison"); + let subject = format!("{prefix}.ledger.external_balance"); + let stream_name = unique("EXTERNAL_POISON"); + let durable = unique("poison_consumer"); + let mut source = + NatsJetStreamSource::connect(&url, &stream_name, vec![subject.clone()], &durable) + .await + .unwrap() + .with_strip_prefix(format!("{prefix}.")); + let js = async_nats::jetstream::new(async_nats::connect(&url).await.unwrap()); + let event = external_balance(1, 10); + let mut headers = async_nats::HeaderMap::new(); + headers.insert("Nats-Msg-Id", "forged"); + headers.insert("X-Sourced-Kind", "event"); + headers.append("X-Distributed-Occurrence-Id", event.id()); + headers.append("X-Distributed-Occurrence-Id", event.id()); + js.publish_with_headers(subject, headers, event.canonical_bytes().unwrap().into()) + .await + .unwrap() + .await + .unwrap(); + let publisher = NatsPublisher::connect(&url) + .await + .unwrap() + .with_subject_prefix(&prefix); + publisher + .publish( + distributed::OutboxMessage::from_domain_event_occurrence(&external_balance(2, 20)) + .unwrap() + .into(), + ) + .await + .unwrap(); + let error = match source.recv().await { + Err(error) => error, + _ => panic!("poison must fail closed"), + }; + assert!(error.is_permanent()); + let mut stream = js.get_stream(&stream_name).await.unwrap(); + let mut consumer: async_nats::jetstream::consumer::Consumer< + async_nats::jetstream::consumer::pull::Config, + > = stream.get_consumer(&durable).await.unwrap(); + let info = consumer.info().await.unwrap(); + assert_eq!(info.ack_floor.stream_sequence, 0); + assert_eq!(info.num_ack_pending, 1); + assert_eq!( + info.num_pending, 1, + "later input was not dispatched past poison" + ); + assert_eq!(stream.info().await.unwrap().state.messages, 2); + js.delete_stream(&stream_name).await.unwrap(); +} + #[tokio::test] async fn derived_facts_round_trip_after_interrupted_publish_prefix() { let Some(url) = nats_url() else { return }; @@ -498,3 +709,225 @@ async fn undecodable_payload_dead_letters_without_blocking() { "the message behind the garbage is still handled" ); } + +// ---- latency: idle delivery, retry backoff, delivery lanes ---- +// See docs/consumer-delivery-lanes.md. + +/// A long-lived consumer that has gone through several empty fetches must +/// still hand a newly published message to its handler immediately; it must +/// not sleep out an idle interval or a fetch expiry first. +#[tokio::test] +async fn message_published_after_empty_fetches_is_delivered_promptly() { + let Some(url) = nats_url() else { return }; + let subject = unique("latency.idle"); + let source = NatsJetStreamSource::connect( + &url, + &unique("STREAM"), + vec![subject.clone()], + &unique("consumer"), + ) + .await + .expect("connect source") + .with_fetch_timeout(Duration::from_millis(500)) + .with_idle_poll(Duration::from_millis(25)); + let handled = Arc::new(Mutex::new(None::)); + let record = handled.clone(); + let router = Arc::new(Handlers::new().on_event(&subject, move |_: &Message| { + record + .lock() + .unwrap() + .get_or_insert_with(std::time::Instant::now); + async { Ok(()) } + })); + let consumer = tokio::spawn(run_source(router, source, RunOptions::idempotent())); + // Several empty 500 ms fetches expire first. + tokio::time::sleep(Duration::from_millis(1_300)).await; + let publisher = NatsPublisher::connect(&url).await.expect("publisher"); + let published = std::time::Instant::now(); + publisher + .publish(Message::new(&subject, MessageKind::Event, b"{}".to_vec()).with_id("late")) + .await + .expect("publish"); + let latency = tokio::time::timeout(Duration::from_secs(3), async { + loop { + if let Some(at) = *handled.lock().unwrap() { + return at.duration_since(published); + } + tokio::time::sleep(Duration::from_millis(5)).await; + } + }) + .await + .expect("idle consumer delivered the message"); + consumer.abort(); + assert!( + latency < Duration::from_millis(250), + "idle delivery took {latency:?}" + ); +} + +/// A message that keeps failing retryably is redelivered with growing +/// delays instead of a hot loop, and a newer message is not delayed by it. +#[tokio::test] +async fn retryable_nack_backs_off_without_starving_newer_messages() { + let Some(url) = nats_url() else { return }; + let subject = unique("latency.retry"); + let source = NatsJetStreamSource::connect( + &url, + &unique("STREAM"), + vec![subject.clone()], + &unique("consumer"), + ) + .await + .expect("connect source") + .with_fetch_timeout(Duration::from_millis(500)) + .with_idle_poll(Duration::from_millis(25)); + let poison_attempts = Arc::new(AtomicUsize::new(0)); + let fresh_at = Arc::new(Mutex::new(None::)); + let (attempts, fresh) = (poison_attempts.clone(), fresh_at.clone()); + let router = Arc::new( + Handlers::new().on_event(&subject, move |message: &Message| { + let poison = message.id() == Some("poison"); + if poison { + attempts.fetch_add(1, Ordering::SeqCst); + } else { + fresh + .lock() + .unwrap() + .get_or_insert_with(std::time::Instant::now); + } + async move { + if poison { + Err(TransportError::retryable("rejected until repaired")) + } else { + Ok(()) + } + } + }), + ); + let consumer = tokio::spawn(run_source(router, source, RunOptions::idempotent())); + let publisher = NatsPublisher::connect(&url).await.expect("publisher"); + publisher + .publish(Message::new(&subject, MessageKind::Event, b"{}".to_vec()).with_id("poison")) + .await + .expect("publish poison"); + tokio::time::sleep(Duration::from_millis(300)).await; + let published = std::time::Instant::now(); + publisher + .publish(Message::new(&subject, MessageKind::Event, b"{}".to_vec()).with_id("fresh")) + .await + .expect("publish fresh"); + tokio::time::sleep(Duration::from_millis(1_200)).await; + consumer.abort(); + let fresh_latency = fresh_at + .lock() + .unwrap() + .expect("fresh message handled") + .duration_since(published); + let attempts = poison_attempts.load(Ordering::SeqCst); + assert!( + fresh_latency < Duration::from_millis(250), + "fresh message waited {fresh_latency:?}" + ); + // 50 ms doubling over ~1.5 s allows at most ~6 attempts; an immediate + // NAK loop redelivers hundreds of times in the same window. + assert!( + (2..=8).contains(&attempts), + "poison retried {attempts} times: retries must continue but back off" + ); +} + +/// With the process policy in its own lane, a slow projection route for an +/// earlier message does not delay the policy for a later message, and every +/// delivery is still acknowledged exactly once after all of its lanes ran. +#[tokio::test] +async fn slow_default_lane_does_not_delay_process_lane_over_jetstream() { + let Some(url) = nats_url() else { return }; + let namespace = unique("lanes").to_lowercase(); + let bus = nats_bus(&url, &namespace, "lanes") + .await + .with_idle_poll(Duration::from_millis(25)); + let started = std::time::Instant::now(); + let projection_done = Arc::new(Mutex::new(Vec::<(String, Duration)>::new())); + let policy_done = Arc::new(Mutex::new(Vec::<(String, Duration)>::new())); + let (projected, policed) = (projection_done.clone(), policy_done.clone()); + let service = Service::new() + .named("lanes") + .routes( + Routes::new() + .with_dependencies(()) + .event("fact.recorded") + .handle(move |ctx: &Context<()>| { + let id = ctx.message().id().unwrap_or_default().to_owned(); + let projected = projected.clone(); + async move { + if id == "m1" { + // A slow projection / external effect. + tokio::time::sleep(Duration::from_millis(1_500)).await; + } + projected.lock().unwrap().push((id, started.elapsed())); + Ok(json!({})) + } + }), + ) + .lane( + "process", + Routes::new() + .with_dependencies(()) + .event("fact.recorded") + .handle(move |ctx: &Context<()>| { + let id = ctx.message().id().unwrap_or_default().to_owned(); + let policed = policed.clone(); + async move { + policed.lock().unwrap().push((id, started.elapsed())); + Ok(json!({})) + } + }), + ) + .with_bus(bus.clone()); + let consumer = tokio::spawn(service.run(RunOptions::idempotent())); + tokio::time::sleep(Duration::from_millis(700)).await; + use distributed::bus::Bus; + for id in ["m1", "m2"] { + bus.publish_message( + Message::new("fact.recorded", MessageKind::Event, b"{}".to_vec()).with_id(id), + ) + .await + .expect("publish"); + } + tokio::time::sleep(Duration::from_millis(2_500)).await; + consumer.abort(); + let policed = policy_done.lock().unwrap().clone(); + let projected = projection_done.lock().unwrap().clone(); + assert_eq!( + policed + .iter() + .map(|(id, _)| id.as_str()) + .collect::>(), + ["m1", "m2"], + "process lane keeps delivery order" + ); + assert_eq!( + projected + .iter() + .map(|(id, _)| id.as_str()) + .collect::>(), + ["m1", "m2"], + "default lane keeps delivery order" + ); + let m2_policy = policed[1].1; + let m1_projection = projected[0].1; + assert!( + m2_policy + Duration::from_millis(1_000) < m1_projection, + "process lane waited for the slow lane: m2 policy at {m2_policy:?}, m1 projection at {m1_projection:?}" + ); + // Both deliveries were acknowledged: nothing pending or awaiting ack. + let stream = bus.ensure_stream().await.expect("stream"); + let mut durable = stream + .get_consumer::("lanes_evt") + .await + .expect("durable exists"); + let info = durable.info().await.expect("consumer info"); + assert_eq!(info.num_ack_pending, 0, "every delivery was settled"); + assert_eq!(info.num_pending, 0); + assert_eq!(info.num_redelivered, 0, "no delivery ran twice"); +} diff --git a/tests/postgres_repository/main.rs b/tests/postgres_repository/main.rs index f5dd02de3..b2630adc1 100644 --- a/tests/postgres_repository/main.rs +++ b/tests/postgres_repository/main.rs @@ -234,7 +234,7 @@ async fn projected_command_ledger_rows_upgrade_to_atomic_and_preserve_checks() { .fetch_one(repo.pool()) .await .unwrap(); - assert_eq!(latest_version, 8); + assert_eq!(latest_version, 9); let invalid_service = sqlx::query( r#" diff --git a/tests/sqlite_repository/main.rs b/tests/sqlite_repository/main.rs index 8f62df91f..63af41672 100644 --- a/tests/sqlite_repository/main.rs +++ b/tests/sqlite_repository/main.rs @@ -201,7 +201,7 @@ async fn projected_command_ledger_rows_upgrade_to_atomic_without_schema_drift() .fetch_one(repo.pool()) .await .unwrap(); - assert_eq!(latest_version, 8); + assert_eq!(latest_version, 9); let created_at_type: String = sqlx::query_scalar( "SELECT typeof(created_at) FROM command_ledger WHERE service_id = 'service'",