From f5164b3ac4b126def9fa3c1bcf6ed19b43d3af8e Mon Sep 17 00:00:00 2001 From: Moe Dashti Date: Thu, 27 Aug 2026 14:29:58 -0400 Subject: [PATCH 01/35] Delivered server-side stream ranges to workflows. History records the offsets a task consumed and never the payloads, so the server sends the bytes on the poll response: untagged for the task about to run, and tagged with a WorkflowTaskCompleted event id when re-supplying what an earlier task consumed. Core partitions the two and emits recorded ranges in event order before the live one, so a replaying workflow observes them exactly as it did the first time. An empty range is delivered rather than dropped: a task where the subscription saw nothing is a fact replay has to reproduce. The Rust SDKs have no stream API, so they fail loudly on this job instead of ignoring it. The server has already recorded the range as consumed and will not send it again, so dropping it would lose data silently. --- crates/protos/build.rs | 1 + .../temporal/api/history/v1/message.proto | 5 + .../temporal/api/stream/v1/message.proto | 51 +++++ .../workflowservice/v1/request_response.proto | 7 + .../workflow_activation.proto | 22 ++ crates/protos/src/protos/mod.rs | 12 ++ crates/sdk-core/src/core_tests/mod.rs | 1 + crates/sdk-core/src/core_tests/streams.rs | 190 ++++++++++++++++++ crates/sdk-core/src/protosext/mod.rs | 7 + crates/sdk-core/src/replay/history_builder.rs | 17 ++ .../sdk-core/src/test_help/integ_helpers.rs | 40 +++- .../src/worker/workflow/history_update.rs | 1 + .../workflow/machines/workflow_machines.rs | 60 ++++++ .../src/worker/workflow/managed_run.rs | 8 +- crates/sdk-core/src/worker/workflow/mod.rs | 2 + crates/sdk/src/workflow_future.rs | 9 + crates/workflow/src/runtime/instance.rs | 14 ++ 17 files changed, 444 insertions(+), 3 deletions(-) create mode 100644 crates/protos/protos/api_upstream/temporal/api/stream/v1/message.proto create mode 100644 crates/sdk-core/src/core_tests/streams.rs diff --git a/crates/protos/build.rs b/crates/protos/build.rs index 509ba9259..7365df524 100644 --- a/crates/protos/build.rs +++ b/crates/protos/build.rs @@ -36,6 +36,7 @@ const SERDE_DERIVE_PREFIXES: &[&str] = &[ ".temporal.api.rules", ".temporal.api.schedule", ".temporal.api.sdk", + ".temporal.api.stream", ".temporal.api.taskqueue", ".temporal.api.testservice", ".temporal.api.update", diff --git a/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto index 0211c6f55..ba77714b2 100644 --- a/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto @@ -17,6 +17,7 @@ import "temporal/api/enums/v1/failed_cause.proto"; import "temporal/api/enums/v1/update.proto"; import "temporal/api/enums/v1/workflow.proto"; import "temporal/api/common/v1/message.proto"; +import "temporal/api/stream/v1/message.proto"; import "temporal/api/deployment/v1/message.proto"; import "temporal/api/failure/v1/message.proto"; import "temporal/api/taskqueue/v1/message.proto"; @@ -379,6 +380,10 @@ message WorkflowTaskCompletedEventAttributes { // The Worker Deployment Version that completed this task. Must be set if `versioning_behavior` // is set. This value updates workflow execution's `versioning_info.deployment_version`. temporal.api.deployment.v1.WorkerDeploymentVersion deployment_version = 11; + + // The offsets this task consumed, without the payloads. Recorded so replay + // can reproduce what the task saw, including a range that was empty. + repeated temporal.api.stream.v1.StreamCursor stream_cursors = 20; } message WorkflowTaskTimedOutEventAttributes { diff --git a/crates/protos/protos/api_upstream/temporal/api/stream/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/stream/v1/message.proto new file mode 100644 index 000000000..a1702e3c7 --- /dev/null +++ b/crates/protos/protos/api_upstream/temporal/api/stream/v1/message.proto @@ -0,0 +1,51 @@ +syntax = "proto3"; + +package temporal.api.stream.v1; + +option go_package = "go.temporal.io/api/stream/v1;stream"; +option java_package = "io.temporal.api.stream.v1"; +option java_multiple_files = true; +option java_outer_classname = "MessageProto"; +option ruby_package = "Temporalio::Api::Stream::V1"; +option csharp_namespace = "Temporalio.Api.Stream.V1"; + +import "temporal/api/common/v1/message.proto"; + +message StreamMessage { + temporal.api.common.v1.Payload body = 1; + map metadata = 2; + string topic = 3; + int64 topic_sequence = 4; +} + +// A contiguous range of a stream delivered to a Workflow Task, along with the +// offsets it covers. The offsets are what History records; the messages +// themselves are never written to History. +message StreamSlice { + string stream_id = 1; + string run_id = 2; + // Inclusive. + int64 from_offset = 3; + // Exclusive. Equal to from_offset when the subscription observed nothing, + // which is a fact replay has to reproduce rather than an absence of one. + int64 to_offset = 4; + repeated StreamMessage messages = 5; + // The WorkflowTaskCompleted event whose stream_cursors recorded this range. + // Set only when the server is re-supplying a range for a task being + // replayed; a slice for the task now being started leaves it unset, because + // the event closing that task does not exist yet. + // + // Replay needs this because a Workflow Task response carries one slice set + // while a cache miss replays every prior task, so the ranges have to be + // matched to the events that recorded them rather than to the response. + int64 workflow_task_completed_event_id = 6; +} + +// The offsets a Workflow Task consumed, without the payloads. Recorded on +// WorkflowTaskCompleted so History grows with Workflow Tasks rather than with +// messages. +message StreamCursor { + string stream_id = 1; + int64 from_offset = 2; + int64 to_offset = 3; +} diff --git a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto index c3dd95769..d9a603e8a 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto @@ -38,6 +38,7 @@ import "temporal/api/replication/v1/message.proto"; import "temporal/api/rules/v1/message.proto"; import "temporal/api/sdk/v1/worker_config.proto"; import "temporal/api/schedule/v1/message.proto"; +import "temporal/api/stream/v1/message.proto"; import "temporal/api/taskqueue/v1/message.proto"; import "temporal/api/update/v1/message.proto"; import "temporal/api/version/v1/message.proto"; @@ -366,6 +367,12 @@ message PollWorkflowTaskQueueResponse { // This poller group ID identifies the owner of the workflow task awaiting for query response. // Corresponding RespondQueryTaskCompleted should pass this value for proper routing. string poller_group_id = 17; + + // Slices of the streams this Workflow consumes. A slice with no + // workflow_task_completed_event_id is for the task about to run; one with + // it set is re-supplying a range an earlier task consumed, so a worker + // replaying from History gets the same bytes that task was given. + repeated temporal.api.stream.v1.StreamSlice stream_slices = 20; // Deprecated. Use `poller_groups_info` instead, which carries a version so the client can // ignore stale updates. // The weighted list of poller groups IDs that client should use for future polls to this task diff --git a/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto b/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto index 097994ac6..c37024e05 100644 --- a/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto +++ b/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto @@ -13,6 +13,7 @@ import "google/protobuf/empty.proto"; import "temporal/api/failure/v1/message.proto"; import "temporal/api/update/v1/message.proto"; import "temporal/api/common/v1/message.proto"; +import "temporal/api/stream/v1/message.proto"; import "temporal/api/enums/v1/workflow.proto"; import "temporal/sdk/core/activity_result/activity_result.proto"; import "temporal/sdk/core/child_workflow/child_workflow.proto"; @@ -139,6 +140,8 @@ message WorkflowActivationJob { ResolveNexusOperationStart resolve_nexus_operation_start = 15; // A nexus operation resolved. ResolveNexusOperation resolve_nexus_operation = 16; + // A range of a stream the workflow subscribed to. + DeliverStreamMessages deliver_stream_messages = 17; // Remove the workflow identified by the [WorkflowActivation] containing this job from the // cache after performing the activation. It is guaranteed that this will be the only job // in the activation if present. @@ -146,6 +149,25 @@ message WorkflowActivationJob { } } +// Hand a workflow the next range of a stream it subscribed to. +// +// The range is delivered once, on the task the server decided it belongs to, +// and the offsets it covered are recorded in History rather than the payloads. +// On replay the server re-supplies the same range by reading the stream again, +// so this job appears at the same point with the same contents both times. +// +// An empty range is still delivered: a task where the subscription saw nothing +// is a fact replay has to reproduce, not an absence of one. +message DeliverStreamMessages { + // Id of the stream this range came from. + string stream_id = 1; + // Inclusive. + int64 from_offset = 2; + // Exclusive. Equal to from_offset when the subscription saw nothing. + int64 to_offset = 3; + repeated temporal.api.stream.v1.StreamMessage messages = 4; +} + // Initialize a new workflow message InitializeWorkflow { // The identifier the lang-specific sdk uses to execute workflow code diff --git a/crates/protos/src/protos/mod.rs b/crates/protos/src/protos/mod.rs index 4bbfa5ac5..f7551e0e5 100644 --- a/crates/protos/src/protos/mod.rs +++ b/crates/protos/src/protos/mod.rs @@ -1277,6 +1277,13 @@ pub mod coresdk { workflow_activation_job::Variant::ResolveNexusOperation(_) => { write!(f, "ResolveNexusOperation") } + workflow_activation_job::Variant::DeliverStreamMessages(d) => { + write!( + f, + "DeliverStreamMessages({}, {}..{})", + d.stream_id, d.from_offset, d.to_offset + ) + } } } } @@ -2640,6 +2647,11 @@ pub mod temporal { tonic::include_proto!("temporal.api.sdk.v1"); } } + pub mod stream { + pub mod v1 { + tonic::include_proto!("temporal.api.stream.v1"); + } + } pub mod taskqueue { pub mod v1 { tonic::include_proto!("temporal.api.taskqueue.v1"); diff --git a/crates/sdk-core/src/core_tests/mod.rs b/crates/sdk-core/src/core_tests/mod.rs index 6dd4479bb..4779016bb 100644 --- a/crates/sdk-core/src/core_tests/mod.rs +++ b/crates/sdk-core/src/core_tests/mod.rs @@ -2,6 +2,7 @@ mod activity_tasks; mod event_groups; mod queries; mod replay_flag; +mod streams; mod updates; mod workers; mod workflow_cancels; diff --git a/crates/sdk-core/src/core_tests/streams.rs b/crates/sdk-core/src/core_tests/streams.rs new file mode 100644 index 000000000..551ffd9e0 --- /dev/null +++ b/crates/sdk-core/src/core_tests/streams.rs @@ -0,0 +1,190 @@ +//! Delivery of server-side stream ranges to a workflow. +//! +//! History records the offsets a task consumed and never the payloads, so the +//! server sends the bytes on the poll response: untagged for the task about to +//! run, and tagged with a WorkflowTaskCompleted event id when it is +//! re-supplying what an earlier task consumed. These tests pin that both +//! arrive, in the right order, and that an empty range is still delivered. + +use crate::{ + replay::TestHistoryBuilder, + test_help::{ + MockPollCfg, PollWFTRespExt, ResponseType, build_mock_pollers, hist_to_poll_resp, + mock_worker, + }, + worker::client::mocks::mock_worker_client, +}; +use temporalio_common::protos::{ + coresdk::{ + workflow_activation::{WorkflowActivationJob, workflow_activation_job}, + workflow_completion::WorkflowActivationCompletion, + }, + temporal::api::{enums::v1::EventType, stream::v1::StreamCursor}, +}; + +fn cursor(stream_id: &str, from: i64, to: i64) -> StreamCursor { + StreamCursor { + stream_id: stream_id.to_string(), + from_offset: from, + to_offset: to, + } +} + +fn delivered(job: &WorkflowActivationJob) -> (&str, i64, i64, Vec<&[u8]>) { + match job.variant.as_ref().unwrap() { + workflow_activation_job::Variant::DeliverStreamMessages(d) => ( + d.stream_id.as_str(), + d.from_offset, + d.to_offset, + d.messages + .iter() + .map(|m| m.body.as_ref().unwrap().data.as_slice()) + .collect(), + ), + other => panic!("expected a stream delivery, got {other:?}"), + } +} + +#[tokio::test] +async fn delivers_the_range_for_the_current_task() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + poll_resp.add_stream_slice("s1", 0, 0, &["alpha", "beta"]); + + let mock = MockPollCfg::from_resp_batches( + "wfid", + t, + [ResponseType::Raw(poll_resp.resp)], + mock_worker_client(), + ); + let core = mock_worker(build_mock_pollers(mock)); + + let task = core.poll_workflow_activation().await.unwrap(); + let stream_jobs: Vec<_> = task + .jobs + .iter() + .filter(|j| { + matches!( + j.variant, + Some(workflow_activation_job::Variant::DeliverStreamMessages(_)) + ) + }) + .collect(); + assert_eq!(stream_jobs.len(), 1); + assert_eq!( + delivered(stream_jobs[0]), + ("s1", 0, 2, vec![b"alpha".as_slice(), b"beta".as_slice()]) + ); + + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); +} + +// A task where the subscription saw nothing is a fact replay has to reproduce, +// so the range still has to arrive rather than being dropped as uninteresting. +#[tokio::test] +async fn delivers_an_empty_range() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + poll_resp.add_stream_slice("s1", 0, 4, &[]); + + let mock = MockPollCfg::from_resp_batches( + "wfid", + t, + [ResponseType::Raw(poll_resp.resp)], + mock_worker_client(), + ); + let core = mock_worker(build_mock_pollers(mock)); + + let task = core.poll_workflow_activation().await.unwrap(); + let job = task + .jobs + .iter() + .find(|j| { + matches!( + j.variant, + Some(workflow_activation_job::Variant::DeliverStreamMessages(_)) + ) + }) + .expect("an empty range is still delivered"); + assert_eq!(delivered(job), ("s1", 4, 4, vec![])); + + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); +} + +// On a cache miss every prior task replays, so ranges those tasks consumed have +// to be handed back in the order they were consumed, before the range for the +// task about to run. +#[tokio::test] +async fn replays_recorded_ranges_in_order_before_the_current_one() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + let first_completed = + t.add_workflow_task_completed_with_stream_cursors(vec![cursor("s1", 0, 2)]); + t.add_workflow_task_scheduled_and_started(); + let second_completed = + t.add_workflow_task_completed_with_stream_cursors(vec![cursor("s1", 2, 3)]); + t.add_workflow_task_scheduled_and_started(); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + // Deliberately out of order, to prove the ordering comes from the events + // rather than from however the server happened to lay them out. + poll_resp.add_stream_slice("s1", second_completed, 2, &["gamma"]); + poll_resp.add_stream_slice("s1", 0, 3, &["delta"]); + poll_resp.add_stream_slice("s1", first_completed, 0, &["alpha", "beta"]); + + let mock = MockPollCfg::from_resp_batches( + "wfid", + t, + [ResponseType::Raw(poll_resp.resp)], + mock_worker_client(), + ); + let mut mock = build_mock_pollers(mock); + mock.worker_cfg(|wc| wc.max_cached_workflows = 1); + let core = mock_worker(mock); + + let mut seen = vec![]; + loop { + let task = core.poll_workflow_activation().await.unwrap(); + for job in &task.jobs { + if matches!( + job.variant, + Some(workflow_activation_job::Variant::DeliverStreamMessages(_)) + ) { + let (_, from, to, bodies) = delivered(job); + seen.push(( + from, + to, + bodies.iter().map(|b| b.to_vec()).collect::>(), + )); + } + } + let done = seen.len() >= 3; + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + if done { + break; + } + } + + assert_eq!( + seen, + vec![ + (0, 2, vec![b"alpha".to_vec(), b"beta".to_vec()]), + (2, 3, vec![b"gamma".to_vec()]), + (3, 4, vec![b"delta".to_vec()]), + ], + "recorded ranges come back in event order, then the live one" + ); +} diff --git a/crates/sdk-core/src/protosext/mod.rs b/crates/sdk-core/src/protosext/mod.rs index 9f0b77348..9af592890 100644 --- a/crates/sdk-core/src/protosext/mod.rs +++ b/crates/sdk-core/src/protosext/mod.rs @@ -39,6 +39,7 @@ use temporalio_common::protos::{ history::v1::{History, HistoryEvent, MarkerRecordedEventAttributes, history_event}, query::v1::WorkflowQuery, sdk::v1::{EventGroupMarker, UserMetadata}, + stream::v1::StreamSlice, workflowservice::v1::PollWorkflowTaskQueueResponse, }, utilities::TryIntoOrNone, @@ -64,6 +65,10 @@ pub(crate) struct ValidPollWFTQResponse { pub(crate) query_requests: Vec, /// Protocol messages pub(crate) messages: Vec, + /// Ranges of streams this workflow subscribed to. A slice tagged with a + /// completed-event id is re-supplying what an earlier task consumed; an + /// untagged one belongs to the task about to run. + pub(crate) stream_slices: Vec, /// Zero-size field to prevent explicit construction _cant_construct_me: (), @@ -109,6 +114,7 @@ impl TryFrom for ValidPollWFTQResponse { query, queries, messages, + stream_slices, .. } => { if task_token.is_empty() { @@ -133,6 +139,7 @@ impl TryFrom for ValidPollWFTQResponse { legacy_query: query, query_requests, messages, + stream_slices, _cant_construct_me: (), }) } diff --git a/crates/sdk-core/src/replay/history_builder.rs b/crates/sdk-core/src/replay/history_builder.rs index 418600403..f8337663d 100644 --- a/crates/sdk-core/src/replay/history_builder.rs +++ b/crates/sdk-core/src/replay/history_builder.rs @@ -23,6 +23,7 @@ use temporalio_common::protos::{ enums::v1::{EventType, TaskQueueKind, WorkflowTaskFailedCause}, failure::v1::{CanceledFailureInfo, Failure, failure}, history::v1::{history_event::Attributes, *}, + stream::v1::StreamCursor, taskqueue::v1::TaskQueue, update, update::v1::outcome, @@ -129,6 +130,22 @@ impl TestHistoryBuilder { self.previous_task_completed_id = id; } + /// Add a workflow task completed event recording the stream offsets that + /// task consumed. Only the range is in History; the payloads come back from + /// the server on the poll response. + pub fn add_workflow_task_completed_with_stream_cursors( + &mut self, + cursors: Vec, + ) -> i64 { + let id = self.add(WorkflowTaskCompletedEventAttributes { + scheduled_event_id: self.workflow_task_scheduled_event_id, + stream_cursors: cursors, + ..Default::default() + }); + self.previous_task_completed_id = id; + id + } + /// Add a workflow task timed out event. pub fn add_workflow_task_timed_out(&mut self) { let attrs = WorkflowTaskTimedOutEventAttributes { diff --git a/crates/sdk-core/src/test_help/integ_helpers.rs b/crates/sdk-core/src/test_help/integ_helpers.rs index 0e446be95..979ea6d9b 100644 --- a/crates/sdk-core/src/test_help/integ_helpers.rs +++ b/crates/sdk-core/src/test_help/integ_helpers.rs @@ -50,10 +50,11 @@ use temporalio_common::{ workflow_completion::WorkflowActivationCompletion, }, temporal::api::{ - common::v1::WorkflowExecution, + common::v1::{Payload, WorkflowExecution}, enums::v1::WorkflowTaskFailedCause, failure::v1::Failure, protocol::{self, v1::message}, + stream::v1::{StreamMessage, StreamSlice}, update, workflowservice::v1::{ DescribeNamespaceResponse, PollActivityTaskQueueResponse, @@ -913,6 +914,17 @@ pub trait PollWFTRespExt { update_id: impl ToString, after_event_id: i64, ) -> update::v1::Request; + + /// Attach a range of a stream. Passing a `completed_event_id` of zero makes + /// it the range for the task about to run; anything else re-supplies what + /// the task closed by that event consumed. + fn add_stream_slice( + &mut self, + stream_id: impl ToString, + completed_event_id: i64, + from_offset: i64, + bodies: &[&str], + ); } impl PollWFTRespExt for PollWorkflowTaskQueueResponse { @@ -947,6 +959,32 @@ impl PollWFTRespExt for PollWorkflowTaskQueueResponse { }); upd_req_body } + + fn add_stream_slice( + &mut self, + stream_id: impl ToString, + completed_event_id: i64, + from_offset: i64, + bodies: &[&str], + ) { + self.stream_slices.push(StreamSlice { + stream_id: stream_id.to_string(), + from_offset, + to_offset: from_offset + bodies.len() as i64, + messages: bodies + .iter() + .map(|b| StreamMessage { + body: Some(Payload { + data: b.as_bytes().to_vec(), + ..Default::default() + }), + ..Default::default() + }) + .collect(), + workflow_task_completed_event_id: completed_event_id, + ..Default::default() + }); + } } pub fn hist_to_poll_resp( diff --git a/crates/sdk-core/src/worker/workflow/history_update.rs b/crates/sdk-core/src/worker/workflow/history_update.rs index 856589946..e6d44558f 100644 --- a/crates/sdk-core/src/worker/workflow/history_update.rs +++ b/crates/sdk-core/src/worker/workflow/history_update.rs @@ -162,6 +162,7 @@ impl HistoryPaginator { query_requests: wft.query_requests, update, messages: wft.messages, + stream_slices: wft.stream_slices, }; Ok((paginator, prepared)) } diff --git a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs index 8d4b8f54e..d782b7b4a 100644 --- a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs +++ b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs @@ -70,6 +70,7 @@ use temporalio_common::{ history::v1::{HistoryEvent, history_event}, protocol::v1::{Message as ProtocolMessage, message::SequencingId}, sdk::v1::WorkflowTaskCompletedMetadata, + stream::v1::StreamSlice, }, }, worker::WorkerDeploymentVersion, @@ -86,6 +87,13 @@ pub(crate) struct WorkflowMachines { /// kept because the lang side polls & completes for every workflow task, but we do not need /// to poll the server that often during replay. last_history_from_server: HistoryUpdate, + /// Stream ranges an earlier task consumed, keyed by the WorkflowTaskCompleted + /// event that recorded each one. Only offsets are in History, so replay is + /// served by the server reading the stream again and sending the bytes back + /// tagged with that event. + stream_slices_by_event: HashMap>, + /// Ranges for the task about to run, which no event records yet. + current_stream_slices: Vec, /// Protocol messages that have yet to be processed for the current WFT. protocol_msgs: Vec, /// EventId of the last handled WorkflowTaskStarted event @@ -276,6 +284,8 @@ impl WorkflowMachines { workflow_type: basics.workflow_type, run_id: basics.run_id, drive_me: driven_wf, + stream_slices_by_event: Default::default(), + current_stream_slices: Default::default(), replaying, metrics: basics.metrics, // In an ideal world one could say ..Default::default() here and it'd still work. @@ -325,11 +335,26 @@ impl WorkflowMachines { &mut self, update: HistoryUpdate, protocol_messages: Vec, + stream_slices: Vec, ) -> Result<()> { if !self.protocol_msgs.is_empty() { dbg_panic!("There are unprocessed protocol messages while receiving new work"); } self.protocol_msgs = protocol_messages; + // An untagged slice belongs to the task about to run; a tagged one is + // re-supplying a range some earlier task already consumed. + self.current_stream_slices.clear(); + self.stream_slices_by_event.clear(); + for slice in stream_slices { + if slice.workflow_task_completed_event_id == 0 { + self.current_stream_slices.push(slice); + } else { + self.stream_slices_by_event + .entry(slice.workflow_task_completed_event_id) + .or_default() + .push(slice); + } + } self.new_history_from_server(update)?; Ok(()) } @@ -617,12 +642,18 @@ impl WorkflowMachines { } }}; } + let mut replayed_slice_events = vec![]; let mut peeked_events = events.iter().peekable(); while let Some(event) = peeked_events.next() { if let Some(history_event::Attributes::WorkflowTaskCompletedEventAttributes(ref wtc)) = event.attributes { apply_wft_complete_data!(self, wtc); + // The event records which offsets that task consumed; the server + // sends the bytes back separately, keyed by this event. + if !wtc.stream_cursors.is_empty() { + replayed_slice_events.push(event.event_id); + } } if peeked_events.peek().is_none() && let Some(wtc) = self @@ -824,6 +855,24 @@ impl WorkflowMachines { } } + // Ranges an earlier task consumed, in the order those tasks ran, so a + // replaying workflow observes them exactly as it did the first time. + replayed_slice_events.sort_unstable(); + for event_id in replayed_slice_events { + if let Some(slices) = self.stream_slices_by_event.remove(&event_id) { + for slice in slices { + self.drive_me.send_job(deliver_stream_messages_job(slice)); + } + } + } + // Then the range for the task about to run, which is only meaningful + // once we have caught up to it. + if !self.replaying { + for slice in std::mem::take(&mut self.current_stream_slices) { + self.drive_me.send_job(deliver_stream_messages_job(slice)); + } + } + // Only record replay latency if we actually did replay work. This avoids recording // near-zero latencies for the first workflow task (which has no history to replay) or // when there were no events to process. @@ -1817,3 +1866,14 @@ enum CommandIdKind { /// A command which is fire-and-forget (ex: Upsert search attribs) NeverResolves, } + +/// Turn a slice the server supplied into the job lang sees. +fn deliver_stream_messages_job(slice: StreamSlice) -> OutgoingJob { + workflow_activation::DeliverStreamMessages { + stream_id: slice.stream_id, + from_offset: slice.from_offset, + to_offset: slice.to_offset, + messages: slice.messages, + } + .into() +} diff --git a/crates/sdk-core/src/worker/workflow/managed_run.rs b/crates/sdk-core/src/worker/workflow/managed_run.rs index 603b38acc..4a37b4d25 100644 --- a/crates/sdk-core/src/worker/workflow/managed_run.rs +++ b/crates/sdk-core/src/worker/workflow/managed_run.rs @@ -42,6 +42,7 @@ use temporalio_common::protos::{ temporal::api::{ enums::v1::{VersioningBehavior, WorkflowTaskFailedCause}, failure::v1::Failure, + stream::v1::StreamSlice, }, }; use tokio::sync::oneshot; @@ -247,7 +248,8 @@ impl ManagedRun { if is_incremental { self.metrics.sticky_cache_hit(); } - self.wfm.new_work_from_server(work.update, work.messages)? + self.wfm + .new_work_from_server(work.update, work.messages, work.stream_slices)? } else { let r = self.wfm.get_next_activation()?; if r.jobs.is_empty() { @@ -1448,8 +1450,10 @@ impl WorkflowManager { &mut self, update: HistoryUpdate, messages: Vec, + stream_slices: Vec, ) -> Result { - self.machines.new_work_from_server(update, messages)?; + self.machines + .new_work_from_server(update, messages, stream_slices)?; self.get_next_activation() } diff --git a/crates/sdk-core/src/worker/workflow/mod.rs b/crates/sdk-core/src/worker/workflow/mod.rs index b0da392a3..b2334f0a6 100644 --- a/crates/sdk-core/src/worker/workflow/mod.rs +++ b/crates/sdk-core/src/worker/workflow/mod.rs @@ -87,6 +87,7 @@ use temporalio_common::{ protocol::v1::Message as ProtocolMessage, query::v1::WorkflowQuery, sdk::v1::{EventGroupMarker, UserMetadata, WorkflowTaskCompletedMetadata}, + stream::v1::StreamSlice, taskqueue::v1::StickyExecutionAttributes, workflowservice::v1::{PollActivityTaskQueueResponse, get_system_info_response}, }, @@ -1017,6 +1018,7 @@ struct PreparedWFT { query_requests: Vec, update: HistoryUpdate, messages: Vec, + stream_slices: Vec, } impl PreparedWFT { diff --git a/crates/sdk/src/workflow_future.rs b/crates/sdk/src/workflow_future.rs index d6920870d..9edf41bb0 100644 --- a/crates/sdk/src/workflow_future.rs +++ b/crates/sdk/src/workflow_future.rs @@ -340,6 +340,15 @@ impl WorkflowFuture { .context("Nexus operation must have result")?; push_polled_context!(ActivationJobContext::Passive); } + Variant::DeliverStreamMessages(slice) => { + // No stream API in this SDK. Bailing rather than ignoring: + // the server has recorded this range as consumed and will + // not send it again, so dropping it loses data silently. + bail!( + "received stream messages for {}, which this SDK cannot deliver", + slice.stream_id + ); + } Variant::RemoveFromCache(_) => { unreachable!("Cache removal should happen higher up"); } diff --git a/crates/workflow/src/runtime/instance.rs b/crates/workflow/src/runtime/instance.rs index c3417ef31..eb4e3e251 100644 --- a/crates/workflow/src/runtime/instance.rs +++ b/crates/workflow/src/runtime/instance.rs @@ -1093,6 +1093,20 @@ where self.apply_resolution(resolution); ActivationJobResult::None } + Some(ActivationVariant::DeliverStreamMessages(slice)) => { + // The Rust workflow runtime has no stream API yet. Failing + // is the only safe answer: the server has already recorded + // this range as consumed, so dropping it would leave the + // workflow permanently behind data it will never be sent + // again. + return Err(Box::new(Failure { + message: format!( + "received stream messages for {}, which this SDK cannot deliver", + slice.stream_id + ), + ..Default::default() + })); + } Some(ActivationVariant::RemoveFromCache(_)) => ActivationJobResult::None, None => { return Err(Box::new(Failure { From 05a34e8b67814875567dde3099fc5997bcbb6837 Mon Sep 17 00:00:00 2001 From: Moe Dashti Date: Thu, 27 Aug 2026 16:11:48 -0400 Subject: [PATCH 02/35] Let a workflow subscribe to a stream from lang. The command has a history event, so it fits the matching every SDK's replay depends on: commands are popped from a queue as command-generated events arrive, and one producing no event would put that out of step. The machine never resolves. The event records the subscription and hands nothing back; the ranges arrive later as their own activation jobs. --- .../temporal/api/command/v1/message.proto | 8 ++ .../temporal/api/enums/v1/command_type.proto | 1 + .../temporal/api/enums/v1/event_type.proto | 4 + .../temporal/api/history/v1/message.proto | 9 ++ .../workflow_commands/workflow_commands.proto | 14 +++ crates/protos/src/protos/mod.rs | 23 +++++ crates/sdk-core/src/core_tests/streams.rs | 64 +++++++++++- crates/sdk-core/src/replay/history_builder.rs | 10 ++ .../src/worker/workflow/machines/mod.rs | 3 + .../subscribe_stream_state_machine.rs | 99 +++++++++++++++++++ .../workflow/machines/workflow_machines.rs | 13 ++- crates/sdk-core/src/worker/workflow/mod.rs | 2 + 12 files changed, 248 insertions(+), 2 deletions(-) create mode 100644 crates/sdk-core/src/worker/workflow/machines/subscribe_stream_state_machine.rs diff --git a/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto index ee839115b..a1e0c9ce0 100644 --- a/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto @@ -286,6 +286,13 @@ message RequestCancelNexusOperationCommandAttributes { int64 scheduled_event_id = 1; } +message SubscribeStreamCommandAttributes { + string stream_id = 1; + // Negative means from wherever the stream is when the subscription is + // registered. The server resolves that once and records it. + int64 start_offset = 2; +} + message Command { temporal.api.enums.v1.CommandType command_type = 1; // Metadata on the command. This is sometimes carried over to the history event if one is @@ -324,5 +331,6 @@ message Command { ScheduleNexusOperationCommandAttributes schedule_nexus_operation_command_attributes = 18; RequestCancelNexusOperationCommandAttributes request_cancel_nexus_operation_command_attributes = 19; + SubscribeStreamCommandAttributes subscribe_stream_command_attributes = 21; } } diff --git a/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto b/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto index 067d95391..68c87008f 100644 --- a/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto +++ b/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto @@ -29,4 +29,5 @@ enum CommandType { COMMAND_TYPE_MODIFY_WORKFLOW_PROPERTIES = 16; COMMAND_TYPE_SCHEDULE_NEXUS_OPERATION = 17; COMMAND_TYPE_REQUEST_CANCEL_NEXUS_OPERATION = 18; + COMMAND_TYPE_SUBSCRIBE_STREAM = 20; } diff --git a/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto b/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto index b879f51e8..4da44e4d7 100644 --- a/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto +++ b/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto @@ -175,4 +175,8 @@ enum EventType { EVENT_TYPE_WORKFLOW_EXECUTION_UNPAUSED = 59; // An event that indicates time skipping advanced time or was disabled automatically after a bound was reached. EVENT_TYPE_WORKFLOW_EXECUTION_TIME_SKIPPING_TRANSITIONED = 60; + // A Workflow subscribed to a stream. Once per subscription, not per + // message: consumed offsets ride WorkflowTaskCompleted and payloads never + // enter History. + EVENT_TYPE_WORKFLOW_STREAM_SUBSCRIBED = 61; } diff --git a/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto index ba77714b2..37a04b2c4 100644 --- a/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto @@ -958,6 +958,14 @@ message ActivityPropertiesModifiedExternallyEventAttributes { temporal.api.common.v1.RetryPolicy new_retry_policy = 2; } +message WorkflowStreamSubscribedEventAttributes { + int64 workflow_task_completed_event_id = 1; + string stream_id = 2; + // Resolved by the server when the subscription is registered, so replay + // reads the resolved value rather than resolving it again. + int64 start_offset = 3; +} + message WorkflowExecutionUpdateAcceptedEventAttributes { // The instance ID of the update protocol that generated this event. string protocol_instance_id = 1; @@ -1281,6 +1289,7 @@ message HistoryEvent { WorkflowExecutionPausedEventAttributes workflow_execution_paused_event_attributes = 63; WorkflowExecutionUnpausedEventAttributes workflow_execution_unpaused_event_attributes = 64; WorkflowExecutionTimeSkippingTransitionedEventAttributes workflow_execution_time_skipping_transitioned_event_attributes = 65; + WorkflowStreamSubscribedEventAttributes workflow_stream_subscribed_event_attributes = 66; } } diff --git a/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto b/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto index 03e172216..e15d6070c 100644 --- a/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto +++ b/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto @@ -54,9 +54,23 @@ message WorkflowCommand { UpdateResponse update_response = 20; ScheduleNexusOperation schedule_nexus_operation = 21; RequestCancelNexusOperation request_cancel_nexus_operation = 22; + SubscribeStream subscribe_stream = 23; } } +// Subscribe this workflow to a stream, so later Workflow Tasks carry the +// ranges it has not consumed yet. +// +// The stream's addressing is resolved by the server. A workflow cannot look it +// up without doing I/O, and a value it carried would be a reading rather than a +// fact, so it could differ on replay. +message SubscribeStream { + string stream_id = 1; + // Negative means from wherever the stream is when the subscription is + // registered. The server resolves that once and records it. + int64 start_offset = 2; +} + message StartTimer { // Lang's incremental sequence number, used as the operation identifier uint32 seq = 1; diff --git a/crates/protos/src/protos/mod.rs b/crates/protos/src/protos/mod.rs index f7551e0e5..14c783564 100644 --- a/crates/protos/src/protos/mod.rs +++ b/crates/protos/src/protos/mod.rs @@ -1493,6 +1493,12 @@ pub mod coresdk { } } + impl Display for SubscribeStream { + fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { + write!(f, "SubscribeStream({})", self.stream_id) + } + } + impl Display for StartTimer { fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { write!(f, "StartTimer({})", self.seq) @@ -1832,6 +1838,9 @@ pub mod temporal { CommandType::ScheduleActivityTask } Attributes::StartTimerCommandAttributes(_) => CommandType::StartTimer, + Attributes::SubscribeStreamCommandAttributes(_) => { + CommandType::SubscribeStream + } Attributes::CompleteWorkflowExecutionCommandAttributes(_) => { CommandType::CompleteWorkflowExecution } @@ -1885,6 +1894,17 @@ pub mod temporal { } } + impl From for command::Attributes { + fn from(s: workflow_commands::SubscribeStream) -> Self { + Self::SubscribeStreamCommandAttributes( + SubscribeStreamCommandAttributes { + stream_id: s.stream_id, + start_offset: s.start_offset, + }, + ) + } + } + impl From for command::Attributes { fn from(s: workflow_commands::StartTimer) -> Self { Self::StartTimerCommandAttributes(StartTimerCommandAttributes { @@ -2344,6 +2364,7 @@ pub mod temporal { | EventType::TimerStarted | EventType::UpsertWorkflowSearchAttributes | EventType::WorkflowPropertiesModified + | EventType::WorkflowStreamSubscribed | EventType::NexusOperationScheduled | EventType::NexusOperationCancelRequested | EventType::WorkflowExecutionCanceled @@ -2443,6 +2464,7 @@ pub mod temporal { // mark any new event types as ignorable or not. if let Some(a) = self.attributes.as_ref() { match a { + Attributes::WorkflowStreamSubscribedEventAttributes(_) => false, Attributes::WorkflowExecutionStartedEventAttributes(_) => false, Attributes::WorkflowExecutionCompletedEventAttributes(_) => false, Attributes::WorkflowExecutionFailedEventAttributes(_) => false, @@ -2530,6 +2552,7 @@ pub mod temporal { pub fn event_type(&self) -> EventType { // I just absolutely _love_ this match self { + Attributes::WorkflowStreamSubscribedEventAttributes(_) => { EventType::WorkflowStreamSubscribed } Attributes::WorkflowExecutionStartedEventAttributes(_) => { EventType::WorkflowExecutionStarted } Attributes::WorkflowExecutionCompletedEventAttributes(_) => { EventType::WorkflowExecutionCompleted } Attributes::WorkflowExecutionFailedEventAttributes(_) => { EventType::WorkflowExecutionFailed } diff --git a/crates/sdk-core/src/core_tests/streams.rs b/crates/sdk-core/src/core_tests/streams.rs index 551ffd9e0..f8ce5423b 100644 --- a/crates/sdk-core/src/core_tests/streams.rs +++ b/crates/sdk-core/src/core_tests/streams.rs @@ -17,9 +17,15 @@ use crate::{ use temporalio_common::protos::{ coresdk::{ workflow_activation::{WorkflowActivationJob, workflow_activation_job}, + workflow_commands::SubscribeStream, workflow_completion::WorkflowActivationCompletion, }, - temporal::api::{enums::v1::EventType, stream::v1::StreamCursor}, + temporal::api::{ + command::v1::command, + enums::v1::{CommandType, EventType}, + stream::v1::StreamCursor, + workflowservice::v1::RespondWorkflowTaskCompletedResponse, + }, }; fn cursor(stream_id: &str, from: i64, to: i64) -> StreamCursor { @@ -188,3 +194,59 @@ async fn replays_recorded_ranges_in_order_before_the_current_one() { "recorded ranges come back in event order, then the live one" ); } + +// A workflow subscribing itself. The command exists at all because every SDK +// matches issued commands against command-generated events in order, so a +// command producing no event would put that matching out of step. This asserts +// the command goes out and that replaying its event does not trip that check. +#[tokio::test] +async fn subscribe_command_round_trips_through_replay() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_full_wf_task(); + t.add_stream_subscribed("s1", 4); + t.add_full_wf_task(); + + let mut mock_client = mock_worker_client(); + mock_client + .expect_complete_workflow_task() + .times(1) + .returning(|resp| { + // The subscribe has to reach the server as a real command, not be + // swallowed by core. + if let Some(cmd) = resp.commands.first() + && cmd.command_type() == CommandType::SubscribeStream + { + let attrs = cmd.attributes.as_ref().unwrap(); + if let command::Attributes::SubscribeStreamCommandAttributes(a) = attrs { + assert_eq!(a.stream_id, "s1"); + assert_eq!(a.start_offset, -1); + } + } + Ok(RespondWorkflowTaskCompletedResponse::default()) + }); + + mock_client + .expect_fail_workflow_task() + .returning(|_, _, f| panic!("core rejected the task: {f:?}")); + + let mock = MockPollCfg::from_resp_batches("wfid", t, [ResponseType::AllHistory], mock_client); + let core = mock_worker(build_mock_pollers(mock)); + + // Full history, so this activation replays the recorded subscription. Lang + // reissues the command, and core has to match it to that event rather than + // calling it nondeterministic. + let task = core.poll_workflow_activation().await.unwrap(); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![ + SubscribeStream { + stream_id: "s1".to_string(), + start_offset: -1, + } + .into(), + ], + )) + .await + .unwrap(); +} diff --git a/crates/sdk-core/src/replay/history_builder.rs b/crates/sdk-core/src/replay/history_builder.rs index f8337663d..84e134b52 100644 --- a/crates/sdk-core/src/replay/history_builder.rs +++ b/crates/sdk-core/src/replay/history_builder.rs @@ -146,6 +146,16 @@ impl TestHistoryBuilder { id } + /// Add the event a subscribe-stream command produces. + pub fn add_stream_subscribed(&mut self, stream_id: &str, start_offset: i64) -> i64 { + let attrs = WorkflowStreamSubscribedEventAttributes { + workflow_task_completed_event_id: self.previous_task_completed_id, + stream_id: stream_id.to_string(), + start_offset, + }; + self.add(attrs) + } + /// Add a workflow task timed out event. pub fn add_workflow_task_timed_out(&mut self) { let attrs = WorkflowTaskTimedOutEventAttributes { diff --git a/crates/sdk-core/src/worker/workflow/machines/mod.rs b/crates/sdk-core/src/worker/workflow/machines/mod.rs index bc5e3fcb9..b84e76969 100644 --- a/crates/sdk-core/src/worker/workflow/machines/mod.rs +++ b/crates/sdk-core/src/worker/workflow/machines/mod.rs @@ -15,6 +15,7 @@ mod modify_workflow_properties_state_machine; mod nexus_operation_state_machine; mod patch_state_machine; mod signal_external_state_machine; +mod subscribe_stream_state_machine; mod timer_state_machine; mod update_state_machine; mod upsert_search_attributes_state_machine; @@ -46,6 +47,7 @@ use std::{ convert::{TryFrom, TryInto}, fmt::{Debug, Display}, }; +use subscribe_stream_state_machine::SubscribeStreamMachine; use temporalio_common::{ fsm_trait::{StateMachine, TransitionResult}, protos::temporal::api::{ @@ -80,6 +82,7 @@ enum Machines { WorkflowTaskMachine, UpsertSearchAttributesMachine, ModifyWorkflowPropertiesMachine, + SubscribeStreamMachine, UpdateMachine, NexusOperationMachine, } diff --git a/crates/sdk-core/src/worker/workflow/machines/subscribe_stream_state_machine.rs b/crates/sdk-core/src/worker/workflow/machines/subscribe_stream_state_machine.rs new file mode 100644 index 000000000..9e384a504 --- /dev/null +++ b/crates/sdk-core/src/worker/workflow/machines/subscribe_stream_state_machine.rs @@ -0,0 +1,99 @@ +use super::{ + NewMachineWithCommand, StateMachine, TransitionResult, fsm, workflow_machines::MachineResponse, +}; +use crate::worker::workflow::{ + WFMachinesError, + machines::{EventInfo, HistEventData, WFMachinesAdapter}, +}; +use temporalio_common::protos::{ + coresdk::workflow_commands::SubscribeStream, + temporal::api::enums::v1::{CommandType, EventType}, +}; + +fsm! { + pub(super) name SubscribeStreamMachine; + command SubscribeStreamMachineCommand; + error WFMachinesError; + + Created --(CommandScheduled) --> CommandIssued; + CommandIssued --(CommandRecorded) --> Done; +} + +/// Subscribe this workflow to a stream. The command carries only the stream id +/// and a start offset; the server resolves the addressing, because a workflow +/// cannot look it up without doing I/O and a value it carried would be a +/// reading rather than a fact. +pub(super) fn subscribe_stream(lang_cmd: SubscribeStream) -> NewMachineWithCommand { + let sm = SubscribeStreamMachine::from_parts(Created {}.into(), ()); + NewMachineWithCommand { + command: lang_cmd.into(), + machine: sm.into(), + } +} + +type SharedState = (); + +#[derive(Debug, derive_more::Display)] +pub(super) enum SubscribeStreamMachineCommand {} + +#[derive(Debug, Default, Clone, derive_more::Display)] +pub(super) struct Created {} + +#[derive(Debug, Default, Clone, derive_more::Display)] +pub(super) struct CommandIssued {} + +#[derive(Debug, Default, Clone, derive_more::Display)] +pub(super) struct Done {} + +impl WFMachinesAdapter for SubscribeStreamMachine { + fn adapt_response( + &self, + _my_command: Self::Command, + _event_info: Option, + ) -> Result, Self::Error> { + Err(Self::Error::Nondeterminism( + "SubscribeStream does not use state machine commands".to_string(), + )) + } +} + +impl TryFrom for SubscribeStreamMachineEvents { + type Error = WFMachinesError; + + fn try_from(e: HistEventData) -> Result { + let e = e.event; + match e.event_type() { + EventType::WorkflowStreamSubscribed => { + Ok(SubscribeStreamMachineEvents::CommandRecorded) + } + _ => Err(Self::Error::Nondeterminism(format!( + "SubscribeStreamMachine does not handle {e}" + ))), + } + } +} + +impl TryFrom for SubscribeStreamMachineEvents { + type Error = WFMachinesError; + + fn try_from(c: CommandType) -> Result { + match c { + CommandType::SubscribeStream => Ok(SubscribeStreamMachineEvents::CommandScheduled), + _ => Err(Self::Error::Nondeterminism(format!( + "SubscribeStreamMachine does not handle command type {c:?}" + ))), + } + } +} + +impl From for Done { + fn from(_: CommandIssued) -> Self { + Self {} + } +} + +impl From for CommandIssued { + fn from(_: Created) -> Self { + Self {} + } +} diff --git a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs index d782b7b4a..29abd120e 100644 --- a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs +++ b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs @@ -8,7 +8,8 @@ use super::{ continue_as_new_workflow_state_machine::continue_as_new, fail_workflow_state_machine::fail_workflow, local_activity_state_machine::new_local_activity, patch_state_machine::has_change, signal_external_state_machine::new_external_signal, - timer_state_machine::new_timer, upsert_search_attributes_state_machine::upsert_search_attrs, + subscribe_stream_state_machine::subscribe_stream, timer_state_machine::new_timer, + upsert_search_attributes_state_machine::upsert_search_attrs, workflow_machines::local_acts::LocalActivityData, workflow_task_state_machine::WorkflowTaskMachine, }; @@ -1575,6 +1576,16 @@ impl WorkflowMachines { CommandIdKind::NeverResolves, ); } + WFCommandVariant::SubscribeStream(attrs) => { + // Never resolves: the event it produces records the + // subscription and hands nothing back to the workflow. The + // ranges arrive later as their own activation jobs. + self.add_cmd_to_wf_task( + subscribe_stream(attrs), + annotations, + CommandIdKind::NeverResolves, + ); + } WFCommandVariant::UpdateResponse(ur) => { let m_key = self.get_machine_by_msg(&ur.protocol_instance_id)?; let m = if let Machines::UpdateMachine(m) = self.machine_mut(m_key) { diff --git a/crates/sdk-core/src/worker/workflow/mod.rs b/crates/sdk-core/src/worker/workflow/mod.rs index b2334f0a6..0ee9aaf4d 100644 --- a/crates/sdk-core/src/worker/workflow/mod.rs +++ b/crates/sdk-core/src/worker/workflow/mod.rs @@ -1529,6 +1529,7 @@ enum WFCommandVariant { UpdateResponse(UpdateResponse), ScheduleNexusOperation(ScheduleNexusOperation), RequestCancelNexusOperation(RequestCancelNexusOperation), + SubscribeStream(SubscribeStream), } impl TryFrom for WFCommand { @@ -1537,6 +1538,7 @@ impl TryFrom for WFCommand { fn try_from(c: WorkflowCommand) -> result::Result { let variant = match c.variant.ok_or(EmptyWorkflowCommandErr)? { workflow_command::Variant::StartTimer(s) => WFCommandVariant::AddTimer(s), + workflow_command::Variant::SubscribeStream(s) => WFCommandVariant::SubscribeStream(s), workflow_command::Variant::CancelTimer(s) => WFCommandVariant::CancelTimer(s), workflow_command::Variant::ScheduleActivity(s) => WFCommandVariant::AddActivity(s), workflow_command::Variant::RequestCancelActivity(s) => { From 3b4e95ed649443d72ba3bd69f09b5505b309870c Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Mon, 14 Sep 2026 15:50:03 -0700 Subject: [PATCH 03/35] Let a workflow publish to a stream from lang. --- crates/common/build.rs | 5 + .../temporal/api/command/v1/message.proto | 8 + .../temporal/api/enums/v1/command_type.proto | 1 + .../temporal/api/enums/v1/event_type.proto | 4 + .../temporal/api/history/v1/message.proto | 15 ++ .../workflow_commands/workflow_commands.proto | 13 ++ crates/protos/src/protos/mod.rs | 30 ++++ crates/sdk-core/src/core_tests/streams.rs | 163 ++++++++++++++++-- crates/sdk-core/src/replay/history_builder.rs | 16 ++ .../add_stream_messages_state_machine.rs | 98 +++++++++++ .../src/worker/workflow/machines/mod.rs | 3 + .../subscribe_stream_state_machine.rs | 2 - .../workflow/machines/workflow_machines.rs | 11 ++ crates/sdk-core/src/worker/workflow/mod.rs | 4 + 14 files changed, 357 insertions(+), 16 deletions(-) create mode 100644 crates/sdk-core/src/worker/workflow/machines/add_stream_messages_state_machine.rs diff --git a/crates/common/build.rs b/crates/common/build.rs index 2e4a378de..8e17815b0 100644 --- a/crates/common/build.rs +++ b/crates/common/build.rs @@ -820,6 +820,11 @@ const NOT_VALIDATED_FIELDS: &[&str] = &[ "temporal.api.workflowservice.v1.StartWorkflowExecutionRequest.continued_failure", "temporal.api.workflowservice.v1.StartWorkflowExecutionRequest.last_completion_result", "temporal.api.workflowservice.v1.TerminateWorkflowExecutionRequest.details", + // Stream messages: the blob limit is a per-event limit, and these bodies never reach an + // event. The server bounds the batch by message count (MaxMessagesPerBatch) instead, which + // is not a payload size the SDK can mirror. + "temporal.api.stream.v1.StreamMessage.body", + "temporal.api.stream.v1.StreamMessage.metadata", // Dedicated, non-fetchable limits (not blob/memo, not in DescribeNamespace): UserMetadata // (nexus-start only); Nexus EndpointSpec.description (maxDescriptionSize; cloud variant cloud-only). "temporal.api.sdk.v1.UserMetadata.details", diff --git a/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto index a1e0c9ce0..57a322af3 100644 --- a/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto @@ -15,6 +15,7 @@ import "temporal/api/enums/v1/workflow.proto"; import "temporal/api/enums/v1/command_type.proto"; import "temporal/api/common/v1/message.proto"; import "temporal/api/failure/v1/message.proto"; +import "temporal/api/stream/v1/message.proto"; import "temporal/api/taskqueue/v1/message.proto"; import "temporal/api/workflow/v1/message.proto"; import "temporal/api/sdk/v1/user_metadata.proto"; @@ -286,6 +287,12 @@ message RequestCancelNexusOperationCommandAttributes { int64 scheduled_event_id = 1; } +message AddStreamMessagesCommandAttributes { + // Empty means the Workflow's default output stream. + string stream_id = 1; + repeated temporal.api.stream.v1.StreamMessage messages = 2; +} + message SubscribeStreamCommandAttributes { string stream_id = 1; // Negative means from wherever the stream is when the subscription is @@ -331,6 +338,7 @@ message Command { ScheduleNexusOperationCommandAttributes schedule_nexus_operation_command_attributes = 18; RequestCancelNexusOperationCommandAttributes request_cancel_nexus_operation_command_attributes = 19; + AddStreamMessagesCommandAttributes add_stream_messages_command_attributes = 20; SubscribeStreamCommandAttributes subscribe_stream_command_attributes = 21; } } diff --git a/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto b/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto index 68c87008f..677ed3b2d 100644 --- a/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto +++ b/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto @@ -29,5 +29,6 @@ enum CommandType { COMMAND_TYPE_MODIFY_WORKFLOW_PROPERTIES = 16; COMMAND_TYPE_SCHEDULE_NEXUS_OPERATION = 17; COMMAND_TYPE_REQUEST_CANCEL_NEXUS_OPERATION = 18; + COMMAND_TYPE_ADD_STREAM_MESSAGES = 19; COMMAND_TYPE_SUBSCRIBE_STREAM = 20; } diff --git a/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto b/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto index 4da44e4d7..af18ac57c 100644 --- a/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto +++ b/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto @@ -179,4 +179,8 @@ enum EventType { // message: consumed offsets ride WorkflowTaskCompleted and payloads never // enter History. EVENT_TYPE_WORKFLOW_STREAM_SUBSCRIBED = 61; + // A Workflow published a batch of messages to a stream. Recorded per + // batch, and carrying only the offset range it landed at: the bodies go to + // the stream's own log, never into History. + EVENT_TYPE_WORKFLOW_STREAM_MESSAGES_ADDED = 62; } diff --git a/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto index 37a04b2c4..70e6d95cb 100644 --- a/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto @@ -958,6 +958,20 @@ message ActivityPropertiesModifiedExternallyEventAttributes { temporal.api.common.v1.RetryPolicy new_retry_policy = 2; } +message WorkflowStreamMessagesAddedEventAttributes { + // The WorkflowTaskCompleted event of the task whose command published this + // batch. + int64 workflow_task_completed_event_id = 1; + // Stream the Workflow published to. + string stream_id = 2; + // Offset the first message of the batch landed at. + int64 first_offset = 3; + // How many messages the batch held. With first_offset this names the range + // without carrying any of it, which is what keeps this event a fixed size + // no matter how large the batch or its payloads are. + int64 message_count = 4; +} + message WorkflowStreamSubscribedEventAttributes { int64 workflow_task_completed_event_id = 1; string stream_id = 2; @@ -1290,6 +1304,7 @@ message HistoryEvent { WorkflowExecutionUnpausedEventAttributes workflow_execution_unpaused_event_attributes = 64; WorkflowExecutionTimeSkippingTransitionedEventAttributes workflow_execution_time_skipping_transitioned_event_attributes = 65; WorkflowStreamSubscribedEventAttributes workflow_stream_subscribed_event_attributes = 66; + WorkflowStreamMessagesAddedEventAttributes workflow_stream_messages_added_event_attributes = 67; } } diff --git a/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto b/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto index e15d6070c..d4c9a1d1b 100644 --- a/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto +++ b/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto @@ -15,6 +15,7 @@ import "temporal/api/common/v1/message.proto"; import "temporal/api/enums/v1/workflow.proto"; import "temporal/api/failure/v1/message.proto"; import "temporal/api/sdk/v1/user_metadata.proto"; +import "temporal/api/stream/v1/message.proto"; import "temporal/api/sdk/v1/event_group_marker.proto"; import "temporal/sdk/core/child_workflow/child_workflow.proto"; import "temporal/sdk/core/nexus/nexus.proto"; @@ -55,6 +56,7 @@ message WorkflowCommand { ScheduleNexusOperation schedule_nexus_operation = 21; RequestCancelNexusOperation request_cancel_nexus_operation = 22; SubscribeStream subscribe_stream = 23; + AddStreamMessages add_stream_messages = 24; } } @@ -64,6 +66,17 @@ message WorkflowCommand { // The stream's addressing is resolved by the server. A workflow cannot look it // up without doing I/O, and a value it carried would be a reading rather than a // fact, so it could differ on replay. +// Publish a batch of messages to a stream this workflow owns. +// +// The bodies go to the stream's own log rather than into History, which gets +// one fixed-size event naming the offset range. That is what makes the batch +// size free: a thousand messages cost the same in History as one. +message AddStreamMessages { + // Empty means the workflow's default output stream. + string stream_id = 1; + repeated temporal.api.stream.v1.StreamMessage messages = 2; +} + message SubscribeStream { string stream_id = 1; // Negative means from wherever the stream is when the subscription is diff --git a/crates/protos/src/protos/mod.rs b/crates/protos/src/protos/mod.rs index 14c783564..c1a78c899 100644 --- a/crates/protos/src/protos/mod.rs +++ b/crates/protos/src/protos/mod.rs @@ -1499,6 +1499,17 @@ pub mod coresdk { } } + impl Display for AddStreamMessages { + fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { + write!( + f, + "AddStreamMessages({}, {} messages)", + self.stream_id, + self.messages.len() + ) + } + } + impl Display for StartTimer { fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { write!(f, "StartTimer({})", self.seq) @@ -1841,6 +1852,9 @@ pub mod temporal { Attributes::SubscribeStreamCommandAttributes(_) => { CommandType::SubscribeStream } + Attributes::AddStreamMessagesCommandAttributes(_) => { + CommandType::AddStreamMessages + } Attributes::CompleteWorkflowExecutionCommandAttributes(_) => { CommandType::CompleteWorkflowExecution } @@ -1894,6 +1908,17 @@ pub mod temporal { } } + impl From for command::Attributes { + fn from(s: workflow_commands::AddStreamMessages) -> Self { + Self::AddStreamMessagesCommandAttributes( + AddStreamMessagesCommandAttributes { + stream_id: s.stream_id, + messages: s.messages, + }, + ) + } + } + impl From for command::Attributes { fn from(s: workflow_commands::SubscribeStream) -> Self { Self::SubscribeStreamCommandAttributes( @@ -2365,6 +2390,7 @@ pub mod temporal { | EventType::UpsertWorkflowSearchAttributes | EventType::WorkflowPropertiesModified | EventType::WorkflowStreamSubscribed + | EventType::WorkflowStreamMessagesAdded | EventType::NexusOperationScheduled | EventType::NexusOperationCancelRequested | EventType::WorkflowExecutionCanceled @@ -2465,6 +2491,9 @@ pub mod temporal { if let Some(a) = self.attributes.as_ref() { match a { Attributes::WorkflowStreamSubscribedEventAttributes(_) => false, + Attributes::WorkflowStreamMessagesAddedEventAttributes(_) => { + false + } Attributes::WorkflowExecutionStartedEventAttributes(_) => false, Attributes::WorkflowExecutionCompletedEventAttributes(_) => false, Attributes::WorkflowExecutionFailedEventAttributes(_) => false, @@ -2553,6 +2582,7 @@ pub mod temporal { // I just absolutely _love_ this match self { Attributes::WorkflowStreamSubscribedEventAttributes(_) => { EventType::WorkflowStreamSubscribed } + Attributes::WorkflowStreamMessagesAddedEventAttributes(_) => { EventType::WorkflowStreamMessagesAdded } Attributes::WorkflowExecutionStartedEventAttributes(_) => { EventType::WorkflowExecutionStarted } Attributes::WorkflowExecutionCompletedEventAttributes(_) => { EventType::WorkflowExecutionCompleted } Attributes::WorkflowExecutionFailedEventAttributes(_) => { EventType::WorkflowExecutionFailed } diff --git a/crates/sdk-core/src/core_tests/streams.rs b/crates/sdk-core/src/core_tests/streams.rs index f8ce5423b..cb207434f 100644 --- a/crates/sdk-core/src/core_tests/streams.rs +++ b/crates/sdk-core/src/core_tests/streams.rs @@ -17,13 +17,13 @@ use crate::{ use temporalio_common::protos::{ coresdk::{ workflow_activation::{WorkflowActivationJob, workflow_activation_job}, - workflow_commands::SubscribeStream, + workflow_commands::{AddStreamMessages, SubscribeStream}, workflow_completion::WorkflowActivationCompletion, }, temporal::api::{ command::v1::command, enums::v1::{CommandType, EventType}, - stream::v1::StreamCursor, + stream::v1::{StreamCursor, StreamMessage}, workflowservice::v1::RespondWorkflowTaskCompletedResponse, }, }; @@ -207,35 +207,153 @@ async fn subscribe_command_round_trips_through_replay() { t.add_stream_subscribed("s1", 4); t.add_full_wf_task(); + let mut mock_client = mock_worker_client(); + // A replayed task sends no commands, so there is nothing to assert on the + // completion here. Rejection is what this test watches for, and core + // signals that by failing the task. + mock_client + .expect_complete_workflow_task() + .returning(|_| Ok(RespondWorkflowTaskCompletedResponse::default())); + mock_client + .expect_fail_workflow_task() + .returning(|_, _, f| panic!("core rejected the reissued subscribe: {f:?}")); + + let mock = MockPollCfg::from_resp_batches("wfid", t, [ResponseType::AllHistory], mock_client); + let core = mock_worker(build_mock_pollers(mock)); + + // Full history, so this activation replays the recorded subscription. Lang + // reissues the command, and core has to match it to that event rather than + // calling it nondeterministic. + let task = core.poll_workflow_activation().await.unwrap(); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![ + SubscribeStream { + stream_id: "s1".to_string(), + start_offset: -1, + } + .into(), + ], + )) + .await + .unwrap(); +} + +/// The command has to reach the server carrying the bodies. +/// +/// Only a task that is not being replayed sends commands, so this drives a +/// single open Workflow Task rather than a recorded history. +#[tokio::test] +async fn publish_command_reaches_the_server_with_its_payloads() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + let mut mock_client = mock_worker_client(); mock_client .expect_complete_workflow_task() .times(1) .returning(|resp| { - // The subscribe has to reach the server as a real command, not be - // swallowed by core. - if let Some(cmd) = resp.commands.first() - && cmd.command_type() == CommandType::SubscribeStream - { - let attrs = cmd.attributes.as_ref().unwrap(); - if let command::Attributes::SubscribeStreamCommandAttributes(a) = attrs { + let cmd = resp.commands.first().expect("a command was sent"); + assert_eq!(cmd.command_type(), CommandType::AddStreamMessages); + match cmd.attributes.as_ref().unwrap() { + command::Attributes::AddStreamMessagesCommandAttributes(a) => { assert_eq!(a.stream_id, "s1"); - assert_eq!(a.start_offset, -1); + // The bodies are the half of the batch History never sees, + // so the command is the only thing that can carry them. + let bodies: Vec<_> = a + .messages + .iter() + .map(|m| m.body.as_ref().unwrap().data.clone()) + .collect(); + assert_eq!(bodies, vec![b"one".to_vec(), b"two".to_vec()]); } + other => panic!("wrong attributes: {other:?}"), } Ok(RespondWorkflowTaskCompletedResponse::default()) }); + let mock = MockPollCfg::from_resp_batches("wfid", t, [ResponseType::AllHistory], mock_client); + let core = mock_worker(build_mock_pollers(mock)); + + let task = core.poll_workflow_activation().await.unwrap(); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![publish_two("s1").into()], + )) + .await + .unwrap(); + core.shutdown().await; +} + +/// A publish reissued on replay has to match the event the original run wrote. +/// +/// This is the property the event exists for. Core pops one queued command per +/// command-generated event, so a publish producing none would leave every later +/// command matched against the wrong event. Core signals the mismatch by +/// failing the workflow task, which the mock turns into a panic. +#[tokio::test] +async fn publish_command_round_trips_through_replay() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_full_wf_task(); + t.add_stream_messages_added("s1", 0, 2); + t.add_full_wf_task(); + + let mut mock_client = mock_worker_client(); + mock_client + .expect_complete_workflow_task() + .returning(|_| Ok(RespondWorkflowTaskCompletedResponse::default())); mock_client .expect_fail_workflow_task() - .returning(|_, _, f| panic!("core rejected the task: {f:?}")); + .returning(|_, _, f| panic!("core rejected the reissued publish: {f:?}")); + + let mock = MockPollCfg::from_resp_batches("wfid", t, [ResponseType::AllHistory], mock_client); + let core = mock_worker(build_mock_pollers(mock)); + + // First activation replays the recorded publish: lang reissues it, and core + // has to match it to that event rather than calling it nondeterministic. + let task = core.poll_workflow_activation().await.unwrap(); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![publish_two("s1").into()], + )) + .await + .unwrap(); + + core.shutdown().await; +} + +/// The subscribe command has to reach the server, which only a task that is +/// not being replayed will send. +#[tokio::test] +async fn subscribe_command_reaches_the_server() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + + let mut mock_client = mock_worker_client(); + mock_client + .expect_complete_workflow_task() + .times(1) + .returning(|resp| { + let cmd = resp.commands.first().expect("a command was sent"); + assert_eq!(cmd.command_type(), CommandType::SubscribeStream); + match cmd.attributes.as_ref().unwrap() { + command::Attributes::SubscribeStreamCommandAttributes(a) => { + assert_eq!(a.stream_id, "s1"); + // Passed through unresolved: the server turns it into a + // real offset and records that. + assert_eq!(a.start_offset, -1); + } + other => panic!("wrong attributes: {other:?}"), + } + Ok(RespondWorkflowTaskCompletedResponse::default()) + }); let mock = MockPollCfg::from_resp_batches("wfid", t, [ResponseType::AllHistory], mock_client); let core = mock_worker(build_mock_pollers(mock)); - // Full history, so this activation replays the recorded subscription. Lang - // reissues the command, and core has to match it to that event rather than - // calling it nondeterministic. let task = core.poll_workflow_activation().await.unwrap(); core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( task.run_id, @@ -249,4 +367,21 @@ async fn subscribe_command_round_trips_through_replay() { )) .await .unwrap(); + core.shutdown().await; +} + +fn publish_two(stream_id: &str) -> AddStreamMessages { + AddStreamMessages { + stream_id: stream_id.to_string(), + messages: vec![ + StreamMessage { + body: Some(b"one".to_vec().into()), + ..Default::default() + }, + StreamMessage { + body: Some(b"two".to_vec().into()), + ..Default::default() + }, + ], + } } diff --git a/crates/sdk-core/src/replay/history_builder.rs b/crates/sdk-core/src/replay/history_builder.rs index 84e134b52..d7ba35e0d 100644 --- a/crates/sdk-core/src/replay/history_builder.rs +++ b/crates/sdk-core/src/replay/history_builder.rs @@ -156,6 +156,22 @@ impl TestHistoryBuilder { self.add(attrs) } + /// Add the event an add-stream-messages command produces. + pub fn add_stream_messages_added( + &mut self, + stream_id: &str, + first_offset: i64, + message_count: i64, + ) -> i64 { + let attrs = WorkflowStreamMessagesAddedEventAttributes { + workflow_task_completed_event_id: self.previous_task_completed_id, + stream_id: stream_id.to_string(), + first_offset, + message_count, + }; + self.add(attrs) + } + /// Add a workflow task timed out event. pub fn add_workflow_task_timed_out(&mut self) { let attrs = WorkflowTaskTimedOutEventAttributes { diff --git a/crates/sdk-core/src/worker/workflow/machines/add_stream_messages_state_machine.rs b/crates/sdk-core/src/worker/workflow/machines/add_stream_messages_state_machine.rs new file mode 100644 index 000000000..55bf0dd8f --- /dev/null +++ b/crates/sdk-core/src/worker/workflow/machines/add_stream_messages_state_machine.rs @@ -0,0 +1,98 @@ +use super::{ + NewMachineWithCommand, StateMachine, TransitionResult, fsm, workflow_machines::MachineResponse, +}; +use crate::worker::workflow::{ + WFMachinesError, + machines::{EventInfo, HistEventData, WFMachinesAdapter}, +}; +use temporalio_common::protos::{ + coresdk::workflow_commands::AddStreamMessages, + temporal::api::enums::v1::{CommandType, EventType}, +}; + +fsm! { + pub(super) name AddStreamMessagesMachine; + command AddStreamMessagesMachineCommand; + error WFMachinesError; + + Created --(CommandScheduled) --> CommandIssued; + CommandIssued --(CommandRecorded) --> Done; +} + +/// Publish a batch of messages to a stream this workflow owns. +/// +/// The bodies go to the stream's own log, and History gets one event naming the +/// offset range the batch landed at. The offsets are assigned by the server, so +/// nothing here predicts them. +pub(super) fn add_stream_messages(lang_cmd: AddStreamMessages) -> NewMachineWithCommand { + let sm = AddStreamMessagesMachine::from_parts(Created {}.into(), ()); + NewMachineWithCommand { + command: lang_cmd.into(), + machine: sm.into(), + } +} + +#[derive(Debug, derive_more::Display)] +pub(super) enum AddStreamMessagesMachineCommand {} + +#[derive(Debug, Default, Clone, derive_more::Display)] +pub(super) struct Created {} + +#[derive(Debug, Default, Clone, derive_more::Display)] +pub(super) struct CommandIssued {} + +#[derive(Debug, Default, Clone, derive_more::Display)] +pub(super) struct Done {} + +impl WFMachinesAdapter for AddStreamMessagesMachine { + fn adapt_response( + &self, + _my_command: Self::Command, + _event_info: Option, + ) -> Result, Self::Error> { + Err(Self::Error::Nondeterminism( + "AddStreamMessages does not use state machine commands".to_string(), + )) + } +} + +impl TryFrom for AddStreamMessagesMachineEvents { + type Error = WFMachinesError; + + fn try_from(e: HistEventData) -> Result { + let e = e.event; + match e.event_type() { + EventType::WorkflowStreamMessagesAdded => { + Ok(AddStreamMessagesMachineEvents::CommandRecorded) + } + _ => Err(Self::Error::Nondeterminism(format!( + "AddStreamMessagesMachine does not handle {e}" + ))), + } + } +} + +impl TryFrom for AddStreamMessagesMachineEvents { + type Error = WFMachinesError; + + fn try_from(c: CommandType) -> Result { + match c { + CommandType::AddStreamMessages => Ok(AddStreamMessagesMachineEvents::CommandScheduled), + _ => Err(Self::Error::Nondeterminism(format!( + "AddStreamMessagesMachine does not handle command type {c:?}" + ))), + } + } +} + +impl From for Done { + fn from(_: CommandIssued) -> Self { + Self {} + } +} + +impl From for CommandIssued { + fn from(_: Created) -> Self { + Self {} + } +} diff --git a/crates/sdk-core/src/worker/workflow/machines/mod.rs b/crates/sdk-core/src/worker/workflow/machines/mod.rs index b84e76969..1b854f680 100644 --- a/crates/sdk-core/src/worker/workflow/machines/mod.rs +++ b/crates/sdk-core/src/worker/workflow/machines/mod.rs @@ -1,3 +1,4 @@ +mod add_stream_messages_state_machine; mod workflow_machines; mod activity_state_machine; @@ -47,6 +48,7 @@ use std::{ convert::{TryFrom, TryInto}, fmt::{Debug, Display}, }; +use add_stream_messages_state_machine::AddStreamMessagesMachine; use subscribe_stream_state_machine::SubscribeStreamMachine; use temporalio_common::{ fsm_trait::{StateMachine, TransitionResult}, @@ -83,6 +85,7 @@ enum Machines { UpsertSearchAttributesMachine, ModifyWorkflowPropertiesMachine, SubscribeStreamMachine, + AddStreamMessagesMachine, UpdateMachine, NexusOperationMachine, } diff --git a/crates/sdk-core/src/worker/workflow/machines/subscribe_stream_state_machine.rs b/crates/sdk-core/src/worker/workflow/machines/subscribe_stream_state_machine.rs index 9e384a504..2e323bd70 100644 --- a/crates/sdk-core/src/worker/workflow/machines/subscribe_stream_state_machine.rs +++ b/crates/sdk-core/src/worker/workflow/machines/subscribe_stream_state_machine.rs @@ -31,8 +31,6 @@ pub(super) fn subscribe_stream(lang_cmd: SubscribeStream) -> NewMachineWithComma } } -type SharedState = (); - #[derive(Debug, derive_more::Display)] pub(super) enum SubscribeStreamMachineCommand {} diff --git a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs index 29abd120e..92d694f51 100644 --- a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs +++ b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs @@ -8,6 +8,7 @@ use super::{ continue_as_new_workflow_state_machine::continue_as_new, fail_workflow_state_machine::fail_workflow, local_activity_state_machine::new_local_activity, patch_state_machine::has_change, signal_external_state_machine::new_external_signal, + add_stream_messages_state_machine::add_stream_messages, subscribe_stream_state_machine::subscribe_stream, timer_state_machine::new_timer, upsert_search_attributes_state_machine::upsert_search_attrs, workflow_machines::local_acts::LocalActivityData, @@ -1576,6 +1577,16 @@ impl WorkflowMachines { CommandIdKind::NeverResolves, ); } + WFCommandVariant::AddStreamMessages(attrs) => { + // Never resolves: the event names the offset range the + // server assigned and hands nothing back. A workflow that + // wants to know where its batch landed reads the stream. + self.add_cmd_to_wf_task( + add_stream_messages(attrs), + annotations, + CommandIdKind::NeverResolves, + ); + } WFCommandVariant::SubscribeStream(attrs) => { // Never resolves: the event it produces records the // subscription and hands nothing back to the workflow. The diff --git a/crates/sdk-core/src/worker/workflow/mod.rs b/crates/sdk-core/src/worker/workflow/mod.rs index 0ee9aaf4d..d37315ff5 100644 --- a/crates/sdk-core/src/worker/workflow/mod.rs +++ b/crates/sdk-core/src/worker/workflow/mod.rs @@ -1530,6 +1530,7 @@ enum WFCommandVariant { ScheduleNexusOperation(ScheduleNexusOperation), RequestCancelNexusOperation(RequestCancelNexusOperation), SubscribeStream(SubscribeStream), + AddStreamMessages(AddStreamMessages), } impl TryFrom for WFCommand { @@ -1539,6 +1540,9 @@ impl TryFrom for WFCommand { let variant = match c.variant.ok_or(EmptyWorkflowCommandErr)? { workflow_command::Variant::StartTimer(s) => WFCommandVariant::AddTimer(s), workflow_command::Variant::SubscribeStream(s) => WFCommandVariant::SubscribeStream(s), + workflow_command::Variant::AddStreamMessages(s) => { + WFCommandVariant::AddStreamMessages(s) + } workflow_command::Variant::CancelTimer(s) => WFCommandVariant::CancelTimer(s), workflow_command::Variant::ScheduleActivity(s) => WFCommandVariant::AddActivity(s), workflow_command::Variant::RequestCancelActivity(s) => { From d0884a320fa81b5082efa60a897417ed7e3b308f Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Mon, 14 Sep 2026 15:50:03 -0700 Subject: [PATCH 04/35] Replayed recorded stream ranges before matching commands. --- crates/sdk-core/src/core_tests/streams.rs | 165 ++++++++++++++++++ .../src/worker/workflow/history_update.rs | 12 +- .../src/worker/workflow/machines/mod.rs | 2 +- .../workflow/machines/workflow_machines.rs | 89 +++++++++- 4 files changed, 256 insertions(+), 12 deletions(-) diff --git a/crates/sdk-core/src/core_tests/streams.rs b/crates/sdk-core/src/core_tests/streams.rs index cb207434f..fde75e484 100644 --- a/crates/sdk-core/src/core_tests/streams.rs +++ b/crates/sdk-core/src/core_tests/streams.rs @@ -385,3 +385,168 @@ fn publish_two(stream_id: &str) -> AddStreamMessages { ], } } + +/// A task that reads a range and publishes because of what it read. +/// +/// This is the shape the product requires: workflow code observes stream input, +/// decides, and writes. Replaying it means the workflow has to be handed the +/// input again *before* core matches the command that input caused, otherwise +/// lang has nothing to decide from and reissues nothing. +/// +/// The recorded range lives on the WorkflowTaskCompleted that closes the task, +/// which is the event *after* the one that started it. So the range for the +/// task about to be replayed is only visible by looking ahead, and a lookahead +/// that reads the completion for its flags but not for its cursors delivers the +/// input one activation too late. +#[tokio::test] +async fn read_then_publish_replays_when_the_range_is_only_visible_by_lookahead() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + // The task that reads [0,1) and publishes because of it. + let read_completed = + t.add_workflow_task_completed_with_stream_cursors(vec![cursor("in", 0, 1)]); + t.add_stream_messages_added("out", 0, 1); + t.add_workflow_task_scheduled_and_started(); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + poll_resp.add_stream_slice("in", read_completed, 0, &["go"]); + + let mut mock_client = mock_worker_client(); + mock_client + .expect_complete_workflow_task() + .returning(|_| Ok(RespondWorkflowTaskCompletedResponse::default())); + mock_client + .expect_fail_workflow_task() + .returning(|_, _, f| panic!("core rejected the reissued read-caused publish: {f:?}")); + + let mock = + MockPollCfg::from_resp_batches("wfid", t, [ResponseType::Raw(poll_resp.resp)], mock_client); + let mut mock = build_mock_pollers(mock); + // Cold: nothing cached, so this is reconstruction from History. + mock.worker_cfg(|wc| wc.max_cached_workflows = 0); + let core = mock_worker(mock); + + let task = core.poll_workflow_activation().await.unwrap(); + let got_input = task.jobs.iter().any(|j| { + matches!( + j.variant, + Some(workflow_activation_job::Variant::DeliverStreamMessages(_)) + ) + }); + assert!( + got_input, + "the replayed task must receive the range it consumed before its \ + resulting publish is matched; jobs were {:?}", + task.jobs + ); + + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![ + AddStreamMessages { + stream_id: "out".to_string(), + messages: vec![StreamMessage { + body: Some(b"accept".to_vec().into()), + ..Default::default() + }], + } + .into(), + ], + )) + .await + .unwrap(); + + core.shutdown().await; +} + +/// The real shape: a subscribe task with no consumed range, then a task that +/// reads and publishes, then a third task replaying both. +/// +/// The first completion carries no cursors, so the lookahead that finds the +/// range has to keep looking past it rather than stopping at the first +/// completion it sees. +#[tokio::test] +async fn read_then_publish_replays_after_a_task_that_consumed_nothing() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + // Task 1 subscribes and consumes nothing. + t.add_workflow_task_completed(); + t.add_stream_subscribed("in", 0); + t.add_workflow_task_scheduled_and_started(); + // Task 2 reads [0,3) and publishes because of it. + let read_completed = + t.add_workflow_task_completed_with_stream_cursors(vec![cursor("in", 0, 3)]); + t.add_stream_messages_added("out", 0, 1); + t.add_workflow_task_scheduled_and_started(); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + poll_resp.add_stream_slice("in", read_completed, 0, &["a", "b", "c"]); + + let mut mock_client = mock_worker_client(); + mock_client + .expect_complete_workflow_task() + .returning(|_| Ok(RespondWorkflowTaskCompletedResponse::default())); + mock_client + .expect_fail_workflow_task() + .returning(|_, _, f| panic!("core rejected the replayed read-caused publish: {f:?}")); + + let mock = + MockPollCfg::from_resp_batches("wfid", t, [ResponseType::Raw(poll_resp.resp)], mock_client); + let mut mock = build_mock_pollers(mock); + mock.worker_cfg(|wc| wc.max_cached_workflows = 0); + let core = mock_worker(mock); + + // Task 1: subscribe, no input yet. + let task = core.poll_workflow_activation().await.unwrap(); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![ + SubscribeStream { + stream_id: "in".to_string(), + start_offset: 0, + } + .into(), + ], + )) + .await + .unwrap(); + + // Task 2: the recorded range has to arrive before its publish is matched. + let task = core.poll_workflow_activation().await.unwrap(); + let bodies: Vec> = task + .jobs + .iter() + .filter(|j| { + matches!( + j.variant, + Some(workflow_activation_job::Variant::DeliverStreamMessages(_)) + ) + }) + .flat_map(|j| delivered(j).3.into_iter().map(|b| b.to_vec())) + .collect(); + assert_eq!( + bodies, + vec![b"a".to_vec(), b"b".to_vec(), b"c".to_vec()], + "the replayed task must be handed the range it consumed; jobs were {:?}", + task.jobs + ); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![ + AddStreamMessages { + stream_id: "out".to_string(), + messages: vec![StreamMessage { + body: Some(b"accept".to_vec().into()), + ..Default::default() + }], + } + .into(), + ], + )) + .await + .unwrap(); + + core.shutdown().await; +} diff --git a/crates/sdk-core/src/worker/workflow/history_update.rs b/crates/sdk-core/src/worker/workflow/history_update.rs index e6d44558f..aa5eb40d0 100644 --- a/crates/sdk-core/src/worker/workflow/history_update.rs +++ b/crates/sdk-core/src/worker/workflow/history_update.rs @@ -641,17 +641,21 @@ impl HistoryUpdate { true } - /// Returns the next WFT completed event attributes, if any, starting at (inclusive) the - /// `from_id` + /// Returns the next WFT completed event, if any, starting at (inclusive) the + /// `from_id`, as its event id and attributes. + /// + /// The id matters to callers that need to key something on the event rather + /// than only read its contents, such as the stream range a task consumed, + /// which is recorded on the completion that closes that task. pub(crate) fn peek_next_wft_completed( &self, from_id: i64, - ) -> Option<&WorkflowTaskCompletedEventAttributes> { + ) -> Option<(i64, &WorkflowTaskCompletedEventAttributes)> { self.events .iter() .skip_while(|e| e.event_id < from_id) .find_map(|e| match &e.attributes { - Some(Attributes::WorkflowTaskCompletedEventAttributes(a)) => Some(a), + Some(Attributes::WorkflowTaskCompletedEventAttributes(a)) => Some((e.event_id, a)), _ => None, }) } diff --git a/crates/sdk-core/src/worker/workflow/machines/mod.rs b/crates/sdk-core/src/worker/workflow/machines/mod.rs index 1b854f680..b826e99d8 100644 --- a/crates/sdk-core/src/worker/workflow/machines/mod.rs +++ b/crates/sdk-core/src/worker/workflow/machines/mod.rs @@ -33,6 +33,7 @@ use crate::{ worker::workflow::{WFMachinesError, fatal, nondeterminism}, }; use activity_state_machine::ActivityMachine; +use add_stream_messages_state_machine::AddStreamMessagesMachine; use cancel_external_state_machine::CancelExternalMachine; use cancel_workflow_state_machine::CancelWorkflowMachine; use child_workflow_state_machine::ChildWorkflowMachine; @@ -48,7 +49,6 @@ use std::{ convert::{TryFrom, TryInto}, fmt::{Debug, Display}, }; -use add_stream_messages_state_machine::AddStreamMessagesMachine; use subscribe_stream_state_machine::SubscribeStreamMachine; use temporalio_common::{ fsm_trait::{StateMachine, TransitionResult}, diff --git a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs index 92d694f51..25dd8b122 100644 --- a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs +++ b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs @@ -2,13 +2,13 @@ mod local_acts; use super::{ Machines, NewMachineWithCommand, TemporalStateMachine, + add_stream_messages_state_machine::add_stream_messages, cancel_external_state_machine::new_external_cancel, cancel_workflow_state_machine::cancel_workflow, complete_workflow_state_machine::complete_workflow, continue_as_new_workflow_state_machine::continue_as_new, fail_workflow_state_machine::fail_workflow, local_activity_state_machine::new_local_activity, patch_state_machine::has_change, signal_external_state_machine::new_external_signal, - add_stream_messages_state_machine::add_stream_messages, subscribe_stream_state_machine::subscribe_stream, timer_state_machine::new_timer, upsert_search_attributes_state_machine::upsert_search_attrs, workflow_machines::local_acts::LocalActivityData, @@ -96,6 +96,22 @@ pub(crate) struct WorkflowMachines { stream_slices_by_event: HashMap>, /// Ranges for the task about to run, which no event records yet. current_stream_slices: Vec, + /// Highest completion event whose recorded range was delivered by looking + /// ahead, or 0. + /// + /// A task's consumed range is recorded on the completion that closes it, so + /// it is found one batch early. That same completion is then seen again as + /// an ordinary event in the next batch, where its slices are already spent + /// and must not be asked for twice. + stream_slices_lookahead_through: i64, + /// Workflow Task Started id of the last task this instance ran live, or 0. + /// + /// A task that ran here was handed its stream messages as it ran, and the + /// completion event recording what it consumed arrives in the next task's + /// history. That event needs no bytes from the server, and `replaying` does + /// not say so: it is true for a cache hit as well, because the history + /// begins after an earlier task. + stream_slices_delivered_through: i64, /// Protocol messages that have yet to be processed for the current WFT. protocol_msgs: Vec, /// EventId of the last handled WorkflowTaskStarted event @@ -276,7 +292,7 @@ impl WorkflowMachines { basics.sdk_version.to_owned(), ); // Peek ahead to determine used flags in the first WFT. - if let Some(attrs) = basics.history.peek_next_wft_completed(0) { + if let Some((_, attrs)) = basics.history.peek_next_wft_completed(0) { observed_internal_flags.add_from_complete(attrs); }; Self { @@ -288,6 +304,8 @@ impl WorkflowMachines { drive_me: driven_wf, stream_slices_by_event: Default::default(), current_stream_slices: Default::default(), + stream_slices_delivered_through: 0, + stream_slices_lookahead_through: 0, replaying, metrics: basics.metrics, // In an ideal world one could say ..Default::default() here and it'd still work. @@ -645,6 +663,11 @@ impl WorkflowMachines { }}; } let mut replayed_slice_events = vec![]; + // Kept apart from the in-batch ones. An in-batch event whose bytes are + // missing is a real inconsistency; a looked-ahead one may simply not + // have been sent yet, and demanding it would turn an early delivery + // into a new way to fail. + let mut lookahead_slice_event: Option = None; let mut peeked_events = events.iter().peekable(); while let Some(event) = peeked_events.next() { if let Some(history_event::Attributes::WorkflowTaskCompletedEventAttributes(ref wtc)) = @@ -658,11 +681,20 @@ impl WorkflowMachines { } } if peeked_events.peek().is_none() - && let Some(wtc) = self + && let Some((wtc_id, wtc)) = self .last_history_from_server .peek_next_wft_completed(event.event_id) { apply_wft_complete_data!(self, wtc); + // The range this batch's task consumed is recorded on the + // completion that closes it, which lands in the *next* batch. + // Collecting it only when it appears would hand the workflow + // its input one activation after the commands that input + // caused, so a read-then-publish task replays with nothing to + // decide from and reissues no command. Look ahead for it here. + if !wtc.stream_cursors.is_empty() { + lookahead_slice_event = Some(wtc_id); + } } } @@ -859,13 +891,53 @@ impl WorkflowMachines { // Ranges an earlier task consumed, in the order those tasks ran, so a // replaying workflow observes them exactly as it did the first time. + // + // Only while replaying. A cache hit is handed the previous task's + // completion in its history as well, and that event carries the range + // that task consumed, but that task ran here and its messages were + // delivered live at the time. There is nothing to re-supply, the server + // sends nothing, and re-supplying would hand the workflow the same + // messages twice. replayed_slice_events.sort_unstable(); for event_id in replayed_slice_events { - if let Some(slices) = self.stream_slices_by_event.remove(&event_id) { - for slice in slices { - self.drive_me.send_job(deliver_stream_messages_job(slice)); - } + // Delivered live to this instance when the task ran, so there is + // nothing to re-supply and the server sent nothing. + if event_id <= self.stream_slices_delivered_through + 1 { + self.stream_slices_by_event.remove(&event_id); + continue; + } + // Already handed over when this task was looked ahead to. + if event_id <= self.stream_slices_lookahead_through { + self.stream_slices_by_event.remove(&event_id); + continue; + } + let Some(slices) = self.stream_slices_by_event.remove(&event_id) else { + // History says a task consumed a range and the server sent no + // bytes for it. Replaying with less data than the original run + // had produces different commands, and the mismatch would + // surface later as an unrelated nondeterminism error. + return Err(WFMachinesError::Nondeterminism(format!( + "Event {event_id} records consumed stream offsets, \ + but the server sent no messages for them" + ))); + }; + for slice in slices { + self.drive_me.send_job(deliver_stream_messages_job(slice)); + } + } + // The task about to be replayed, whose range is only visible by looking + // ahead to the completion that closes it. Absent bytes mean the server + // has not re-supplied them on this response, which the ordinary path + // above will still catch when that completion arrives as an event. + if let Some(event_id) = lookahead_slice_event + && event_id > self.stream_slices_delivered_through + 1 + && event_id > self.stream_slices_lookahead_through + && let Some(slices) = self.stream_slices_by_event.remove(&event_id) + { + for slice in slices { + self.drive_me.send_job(deliver_stream_messages_job(slice)); } + self.stream_slices_lookahead_through = event_id; } // Then the range for the task about to run, which is only meaningful // once we have caught up to it. @@ -873,6 +945,9 @@ impl WorkflowMachines { for slice in std::mem::take(&mut self.current_stream_slices) { self.drive_me.send_job(deliver_stream_messages_job(slice)); } + // This task is running here, so whatever it consumes is already in + // hand and its completion event will not need re-supplying. + self.stream_slices_delivered_through = self.next_started_event_id; } // Only record replay latency if we actually did replay work. This avoids recording From 280fe2d446a65d700a2a9067e9d65b330da2d511 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Wed, 16 Sep 2026 14:42:13 -0700 Subject: [PATCH 05/35] Registered the stream machines with the coverage reporter. The reporter panics on any machine name it has no visualizer for, so every state machine has to be listed there. --- .../worker/workflow/machines/transition_coverage.rs | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs b/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs index b266137fa..5a1df56e7 100644 --- a/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs +++ b/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs @@ -66,6 +66,7 @@ mod machine_coverage_report { use super::*; use crate::worker::workflow::machines::{ StateMachine, activity_state_machine::ActivityMachine, + add_stream_messages_state_machine::AddStreamMessagesMachine, cancel_external_state_machine::CancelExternalMachine, cancel_workflow_state_machine::CancelWorkflowMachine, child_workflow_state_machine::ChildWorkflowMachine, @@ -75,7 +76,8 @@ mod machine_coverage_report { local_activity_state_machine::LocalActivityMachine, modify_workflow_properties_state_machine::ModifyWorkflowPropertiesMachine, nexus_operation_state_machine::NexusOperationMachine, patch_state_machine::PatchMachine, - signal_external_state_machine::SignalExternalMachine, timer_state_machine::TimerMachine, + signal_external_state_machine::SignalExternalMachine, + subscribe_stream_state_machine::SubscribeStreamMachine, timer_state_machine::TimerMachine, update_state_machine::UpdateMachine, upsert_search_attributes_state_machine::UpsertSearchAttributesMachine, workflow_task_state_machine::WorkflowTaskMachine, @@ -117,6 +119,8 @@ mod machine_coverage_report { let mut modify_wf_props = ModifyWorkflowPropertiesMachine::visualizer().to_owned(); let mut update = UpdateMachine::visualizer().to_owned(); let mut nexus = NexusOperationMachine::visualizer().to_owned(); + let mut subscribe_stream = SubscribeStreamMachine::visualizer().to_owned(); + let mut add_stream_messages = AddStreamMessagesMachine::visualizer().to_owned(); // This isn't at all efficient but doesn't need to be. // Replace transitions in the vizzes with green color if they are covered. @@ -144,6 +148,12 @@ mod machine_coverage_report { } m @ "UpdateMachine" => cover_transitions(m, &mut update, coverage), m @ "NexusOperationMachine" => cover_transitions(m, &mut nexus, coverage), + m @ "SubscribeStreamMachine" => { + cover_transitions(m, &mut subscribe_stream, coverage) + } + m @ "AddStreamMessagesMachine" => { + cover_transitions(m, &mut add_stream_messages, coverage) + } m => panic!("Unknown machine {m}"), } } From 865805b88b7f2e6ef7cbbfc8ee32e777b531d651 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Wed, 16 Sep 2026 15:05:30 -0700 Subject: [PATCH 06/35] Matched the stream test mocks to the two-argument completion. `complete_workflow_task` takes a shutdown token, so the `mockall` closures need a second parameter. --- crates/sdk-core/src/core_tests/streams.rs | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/crates/sdk-core/src/core_tests/streams.rs b/crates/sdk-core/src/core_tests/streams.rs index fde75e484..11ef86f71 100644 --- a/crates/sdk-core/src/core_tests/streams.rs +++ b/crates/sdk-core/src/core_tests/streams.rs @@ -213,7 +213,7 @@ async fn subscribe_command_round_trips_through_replay() { // signals that by failing the task. mock_client .expect_complete_workflow_task() - .returning(|_| Ok(RespondWorkflowTaskCompletedResponse::default())); + .returning(|_, _| Ok(RespondWorkflowTaskCompletedResponse::default())); mock_client .expect_fail_workflow_task() .returning(|_, _, f| panic!("core rejected the reissued subscribe: {f:?}")); @@ -253,7 +253,7 @@ async fn publish_command_reaches_the_server_with_its_payloads() { mock_client .expect_complete_workflow_task() .times(1) - .returning(|resp| { + .returning(|resp, _| { let cmd = resp.commands.first().expect("a command was sent"); assert_eq!(cmd.command_type(), CommandType::AddStreamMessages); match cmd.attributes.as_ref().unwrap() { @@ -303,7 +303,7 @@ async fn publish_command_round_trips_through_replay() { let mut mock_client = mock_worker_client(); mock_client .expect_complete_workflow_task() - .returning(|_| Ok(RespondWorkflowTaskCompletedResponse::default())); + .returning(|_, _| Ok(RespondWorkflowTaskCompletedResponse::default())); mock_client .expect_fail_workflow_task() .returning(|_, _, f| panic!("core rejected the reissued publish: {f:?}")); @@ -336,7 +336,7 @@ async fn subscribe_command_reaches_the_server() { mock_client .expect_complete_workflow_task() .times(1) - .returning(|resp| { + .returning(|resp, _| { let cmd = resp.commands.first().expect("a command was sent"); assert_eq!(cmd.command_type(), CommandType::SubscribeStream); match cmd.attributes.as_ref().unwrap() { @@ -415,7 +415,7 @@ async fn read_then_publish_replays_when_the_range_is_only_visible_by_lookahead() let mut mock_client = mock_worker_client(); mock_client .expect_complete_workflow_task() - .returning(|_| Ok(RespondWorkflowTaskCompletedResponse::default())); + .returning(|_, _| Ok(RespondWorkflowTaskCompletedResponse::default())); mock_client .expect_fail_workflow_task() .returning(|_, _, f| panic!("core rejected the reissued read-caused publish: {f:?}")); @@ -487,7 +487,7 @@ async fn read_then_publish_replays_after_a_task_that_consumed_nothing() { let mut mock_client = mock_worker_client(); mock_client .expect_complete_workflow_task() - .returning(|_| Ok(RespondWorkflowTaskCompletedResponse::default())); + .returning(|_, _| Ok(RespondWorkflowTaskCompletedResponse::default())); mock_client .expect_fail_workflow_task() .returning(|_, _, f| panic!("core rejected the replayed read-caused publish: {f:?}")); From f88f78c3fb8acf81126ed23b0f843e29fabb1e9c Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Wed, 16 Sep 2026 16:21:41 -0700 Subject: [PATCH 07/35] Honored the per-runner timeout for the integ test matrix. The matrix asks for 40 minutes on `macos-intel`, but the job never read that value, so the runner stayed on the 25-minute default it exceeds. --- .github/workflows/per-pr.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/per-pr.yml b/.github/workflows/per-pr.yml index 11325303a..89126c4cc 100644 --- a/.github/workflows/per-pr.yml +++ b/.github/workflows/per-pr.yml @@ -123,7 +123,7 @@ jobs: integ-tests: name: Integ tests - timeout-minutes: ${{ github.ref == 'refs/heads/main' && 30 || 25 }} + timeout-minutes: ${{ matrix.timeoutMinutes || (github.ref == 'refs/heads/main' && 30 || 25) }} strategy: fail-fast: false matrix: From 4076ef24fc9e731e01692d93162aa6c7241f1851 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Wed, 16 Sep 2026 23:19:18 -0700 Subject: [PATCH 08/35] Matched the nexus model WIT to upstream for the workflow-id policies. The nexgen the Python SDK drives rejects @nexus.type on a native declaration, so generation failed for every consumer. Upstream declares both policies as placeholders, which keeps the annotations valid. --- .../nexus/deps/nexus-temporal-types/model.wit | 13 ++----------- 1 file changed, 2 insertions(+), 11 deletions(-) diff --git a/crates/protos/protos/api_upstream/nexus/deps/nexus-temporal-types/model.wit b/crates/protos/protos/api_upstream/nexus/deps/nexus-temporal-types/model.wit index 91de3e838..b6e6e5e8f 100644 --- a/crates/protos/protos/api_upstream/nexus/deps/nexus-temporal-types/model.wit +++ b/crates/protos/protos/api_upstream/nexus/deps/nexus-temporal-types/model.wit @@ -147,12 +147,7 @@ interface model { /// typescript="common.WorkflowIdReusePolicy" /// dotnet="Temporalio.Api.Enums.V1.WorkflowIdReusePolicy" /// typescript-import="@temporalio/common" - enum workflow-id-reuse-policy { - allow-duplicate, - allow-duplicate-failed-only, - reject-duplicate, - terminate-if-running, - } + type workflow-id-reuse-policy = placeholder; /// @nexus.proto "temporal.api.enums.v1.WorkflowIdConflictPolicy" typescript-import="@temporalio/proto" /// @nexus.type @@ -160,11 +155,7 @@ interface model { /// typescript="common.WorkflowIdConflictPolicy" /// dotnet="Temporalio.Api.Enums.V1.WorkflowIdConflictPolicy" /// typescript-import="@temporalio/common" - enum workflow-id-conflict-policy { - fail, - use-existing, - terminate-existing, - } + type workflow-id-conflict-policy = placeholder; /// @nexus.proto "temporal.api.sdk.v1.UserMetadata" typescript-import="@temporalio/proto" /// @nexus.flatten-in-api From d6d66fb228d22e32b26bd085ee6132ce86e494fc Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 18 Sep 2026 16:47:07 -0700 Subject: [PATCH 09/35] Kept the closing completion visible to a history update's last task. The range a task consumed is recorded on the completion that closes it, and the machines look ahead to that event while replaying the task. An update cut at a WFT started event left the completion on the retained tail or on an unfetched page, so the range arrived one activation late. --- crates/sdk-core/src/core_tests/streams.rs | 159 +++++++++++++++++- .../src/worker/workflow/history_update.rs | 22 ++- 2 files changed, 178 insertions(+), 3 deletions(-) diff --git a/crates/sdk-core/src/core_tests/streams.rs b/crates/sdk-core/src/core_tests/streams.rs index 11ef86f71..6ca50538d 100644 --- a/crates/sdk-core/src/core_tests/streams.rs +++ b/crates/sdk-core/src/core_tests/streams.rs @@ -16,15 +16,18 @@ use crate::{ }; use temporalio_common::protos::{ coresdk::{ - workflow_activation::{WorkflowActivationJob, workflow_activation_job}, + workflow_activation::{WorkflowActivation, WorkflowActivationJob, workflow_activation_job}, workflow_commands::{AddStreamMessages, SubscribeStream}, workflow_completion::WorkflowActivationCompletion, }, temporal::api::{ command::v1::command, enums::v1::{CommandType, EventType}, + history::v1::{History, HistoryEvent}, stream::v1::{StreamCursor, StreamMessage}, - workflowservice::v1::RespondWorkflowTaskCompletedResponse, + workflowservice::v1::{ + GetWorkflowExecutionHistoryResponse, RespondWorkflowTaskCompletedResponse, + }, }, }; @@ -51,6 +54,36 @@ fn delivered(job: &WorkflowActivationJob) -> (&str, i64, i64, Vec<&[u8]>) { } } +/// The stream deliveries in an activation, as (stream, from, to). +fn delivered_ranges(task: &WorkflowActivation) -> Vec<(String, i64, i64)> { + task.jobs + .iter() + .filter(|j| { + matches!( + j.variant, + Some(workflow_activation_job::Variant::DeliverStreamMessages(_)) + ) + }) + .map(|j| { + let (stream, from, to, _) = delivered(j); + (stream.to_string(), from, to) + }) + .collect() +} + +fn history_page( + events: &[HistoryEvent], + next_page_token: Vec, +) -> GetWorkflowExecutionHistoryResponse { + GetWorkflowExecutionHistoryResponse { + history: Some(History { + events: events.to_vec(), + }), + next_page_token, + ..Default::default() + } +} + #[tokio::test] async fn delivers_the_range_for_the_current_task() { let mut t = TestHistoryBuilder::default(); @@ -550,3 +583,125 @@ async fn read_then_publish_replays_after_a_task_that_consumed_nothing() { core.shutdown().await; } + +/// The reading task's range has to arrive in its own activation when History +/// comes in pages and a page boundary falls at that task. +/// +/// The range is recorded on the completion that closes the task, and the +/// paginator hands the machines updates cut at WFT started events. Whether the +/// boundary lands right after the reading task's started event or right after +/// its completion, the completion is outside the update the machines are +/// replaying from, and a lookahead reading only that update would find nothing. +#[rstest::rstest] +#[tokio::test] +async fn read_then_publish_replays_across_a_page_boundary( + #[values(12, 13)] second_page_end: usize, +) { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_full_wf_task(); // 3 + t.add_we_signaled("go", vec![]); + t.add_full_wf_task(); // 7 + // Task 2 subscribes. Its command event is what lets the paginator tell the + // reading task's sequence is complete once it sees the started event. + t.add_stream_subscribed("in", 0); + t.add_we_signaled("go", vec![]); + t.add_workflow_task_scheduled_and_started(); // 12 + // Task 3 reads [0,1) and publishes because of it. + let read_completed = + t.add_workflow_task_completed_with_stream_cursors(vec![cursor("in", 0, 1)]); + t.add_stream_messages_added("out", 0, 1); + t.add_workflow_task_scheduled_and_started(); // 16 + + let events = t.get_full_history_info().unwrap().into_events(); + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + poll_resp.history.as_mut().unwrap().events.truncate(3); + poll_resp.next_page_token = vec![1]; + // Two complete tasks past the previous started id fit in the first two + // pages, so the paginator would hand them over before fetching the third. + poll_resp.previous_started_event_id = 3; + poll_resp.add_stream_slice("in", read_completed, 0, &["go"]); + poll_resp.add_stream_slice("in", 0, 1, &["next"]); + + let second_page = history_page(&events[3..second_page_end], vec![2]); + let third_page = history_page(&events[second_page_end..], vec![]); + let mut mock_client = mock_worker_client(); + mock_client + .expect_get_workflow_execution_history() + .returning(move |_, _, token| match token.as_slice() { + [1] => Ok(second_page.clone()), + [2] => Ok(third_page.clone()), + other => panic!("unexpected page token {other:?}"), + }); + mock_client + .expect_fail_workflow_task() + .returning(|_, _, f| panic!("core rejected the replayed read-caused publish: {f:?}")); + + let mock = + MockPollCfg::from_resp_batches("wfid", t, [ResponseType::Raw(poll_resp.resp)], mock_client); + let core = mock_worker(build_mock_pollers(mock)); + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(delivered_ranges(&task), vec![]); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(delivered_ranges(&task), vec![]); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![ + SubscribeStream { + stream_id: "in".to_string(), + start_offset: 0, + } + .into(), + ], + )) + .await + .unwrap(); + + // The reading task. Its signal and its range belong to the same activation. + let task = core.poll_workflow_activation().await.unwrap(); + assert!(task.is_replaying); + assert!( + task.jobs.iter().any(|j| matches!( + j.variant, + Some(workflow_activation_job::Variant::SignalWorkflow(_)) + )), + "expected the reading task's signal; jobs were {:?}", + task.jobs + ); + assert_eq!( + delivered_ranges(&task), + vec![("in".to_string(), 0, 1)], + "the replayed task must be handed the range it consumed; jobs were {:?}", + task.jobs + ); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![ + AddStreamMessages { + stream_id: "out".to_string(), + messages: vec![StreamMessage { + body: Some(b"accept".to_vec().into()), + ..Default::default() + }], + } + .into(), + ], + )) + .await + .unwrap(); + + // The live task gets only its own range. + let task = core.poll_workflow_activation().await.unwrap(); + assert!(!task.is_replaying); + assert_eq!(delivered_ranges(&task), vec![("in".to_string(), 1, 2)]); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + + core.shutdown().await; +} diff --git a/crates/sdk-core/src/worker/workflow/history_update.rs b/crates/sdk-core/src/worker/workflow/history_update.rs index aa5eb40d0..a8523b9cb 100644 --- a/crates/sdk-core/src/worker/workflow/history_update.rs +++ b/crates/sdk-core/src/worker/workflow/history_update.rs @@ -283,7 +283,7 @@ impl HistoryPaginator { // We only *really* have the last WFT if the events go all the way up to at least the // WFT started event id. Otherwise we somehow still have partial history. let no_more = matches!(self.next_page_token, NextPageToken::Done) && seen_enough_events; - let (update, extra) = HistoryUpdate::from_events( + let (mut update, extra) = HistoryUpdate::from_events( current_events, self.previous_wft_started_id, self.wft_started_event_id, @@ -310,8 +310,28 @@ impl HistoryPaginator { // There was not a meaningful WFT in the whole page. We must fetch more. continue; } + // The machines read the completion that closes an update's last task while that + // task is the one being replayed: it records the stream range the task consumed, + // and the workflow has to be handed that range in the same activation. A page that + // ends exactly on a WFT started event leaves that completion on the next page, so + // fetch it before handing the update over. + if !no_more && self.event_queue.is_empty() { + self.event_queue.extend(update.events); + continue; + } self.id_of_last_event_in_last_extracted_update = update.events.last().map(|e| e.event_id); + // Otherwise the completion is the first retained event. Carry a copy across the + // split so it can be peeked at. The original stays queued and is what the next + // update is built from, and a lone completion yields nothing to take, so no event + // is applied twice. + if let Some(completion) = self + .event_queue + .front() + .filter(|e| e.event_type() == EventType::WorkflowTaskCompleted) + { + update.events.push(completion.clone()); + } #[cfg(debug_assertions)] update.assert_contiguous(); return Ok(update); From 76163d48550950e43535a050caf1868bade81e4e Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 18 Sep 2026 16:53:57 -0700 Subject: [PATCH 10/35] Checked replayed stream commands against the recorded event. A publish is matched on its stream and message count, a subscription on its stream. Replay sends no commands, so a reissued command that differs from the record would otherwise be accepted and the workflow's state would diverge silently. --- crates/sdk-core/src/core_tests/streams.rs | 122 +++++++++++++++++- .../add_stream_messages_state_machine.rs | 66 ++++++++-- .../subscribe_stream_state_machine.rs | 61 +++++++-- 3 files changed, 224 insertions(+), 25 deletions(-) diff --git a/crates/sdk-core/src/core_tests/streams.rs b/crates/sdk-core/src/core_tests/streams.rs index 6ca50538d..f01884ce9 100644 --- a/crates/sdk-core/src/core_tests/streams.rs +++ b/crates/sdk-core/src/core_tests/streams.rs @@ -7,13 +7,18 @@ //! arrive, in the right order, and that an empty range is still delivered. use crate::{ + Worker, replay::TestHistoryBuilder, test_help::{ - MockPollCfg, PollWFTRespExt, ResponseType, build_mock_pollers, hist_to_poll_resp, - mock_worker, + MockPollCfg, PollWFTRespExt, ResponseType, WorkerTestHelpers, build_mock_pollers, + hist_to_poll_resp, mock_worker, }, worker::client::mocks::mock_worker_client, }; +use std::sync::{ + Arc, + atomic::{AtomicUsize, Ordering}, +}; use temporalio_common::protos::{ coresdk::{ workflow_activation::{WorkflowActivation, WorkflowActivationJob, workflow_activation_job}, @@ -22,7 +27,7 @@ use temporalio_common::protos::{ }, temporal::api::{ command::v1::command, - enums::v1::{CommandType, EventType}, + enums::v1::{CommandType, EventType, WorkflowTaskFailedCause}, history::v1::{History, HistoryEvent}, stream::v1::{StreamCursor, StreamMessage}, workflowservice::v1::{ @@ -71,6 +76,28 @@ fn delivered_ranges(task: &WorkflowActivation) -> Vec<(String, i64, i64)> { .collect() } +/// A worker served one poll response, expected to fail that task as +/// nondeterministic. Returns the worker and the count of failures it reported, +/// which the test asserts itself: the mock only verifies call counts when it +/// is dropped, and a missing failure would otherwise go unnoticed. The task +/// stream is kept open so the eviction that follows the failure can be polled. +fn worker_expecting_one_nondeterminism_failure( + t: TestHistoryBuilder, + resp: ResponseType, +) -> (Worker, Arc) { + let failures = Arc::new(AtomicUsize::new(0)); + let counted = failures.clone(); + let mut mock = MockPollCfg::from_resp_batches("wfid", t, [resp], mock_worker_client()); + mock.num_expected_fails = 1; + mock.expect_fail_wft_matcher = Box::new(move |_, cause, _| { + counted.fetch_add(1, Ordering::Relaxed); + matches!(cause, WorkflowTaskFailedCause::NonDeterministicError) + }); + let mut mock = build_mock_pollers(mock); + mock.make_wft_stream_interminable(); + (mock_worker(mock), failures) +} + fn history_page( events: &[HistoryEvent], next_page_token: Vec, @@ -705,3 +732,92 @@ async fn read_then_publish_replays_across_a_page_boundary( core.shutdown().await; } + +/// A publish reissued on replay is held against the recorded event, not only +/// against its type. Core sends no commands while replaying, so this check is +/// the only place a publish to the wrong stream can be noticed. +#[tokio::test] +async fn a_publish_reissued_to_a_different_stream_fails_the_task() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_full_wf_task(); + t.add_stream_messages_added("s1", 0, 2); + t.add_full_wf_task(); + + let (core, failures) = worker_expecting_one_nondeterminism_failure(t, ResponseType::AllHistory); + + let task = core.poll_workflow_activation().await.unwrap(); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![publish_two("s2").into()], + )) + .await + .unwrap(); + core.handle_eviction().await; + assert_eq!(failures.load(Ordering::Relaxed), 1); + core.shutdown().await; +} + +/// The batch size is part of the record too: the event names how many messages +/// landed, so a replay that publishes fewer has diverged from the original run. +#[tokio::test] +async fn a_publish_reissued_with_a_different_batch_size_fails_the_task() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_full_wf_task(); + t.add_stream_messages_added("s1", 0, 2); + t.add_full_wf_task(); + + let (core, failures) = worker_expecting_one_nondeterminism_failure(t, ResponseType::AllHistory); + + let task = core.poll_workflow_activation().await.unwrap(); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![ + AddStreamMessages { + stream_id: "s1".to_string(), + messages: vec![StreamMessage { + body: Some(b"one".to_vec().into()), + ..Default::default() + }], + } + .into(), + ], + )) + .await + .unwrap(); + core.handle_eviction().await; + assert_eq!(failures.load(Ordering::Relaxed), 1); + core.shutdown().await; +} + +/// A subscription is checked on the stream alone. The recorded start offset is +/// the server's resolution of what the command asked for, so it is not the +/// command's to reproduce. +#[tokio::test] +async fn a_subscribe_reissued_to_a_different_stream_fails_the_task() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_full_wf_task(); + t.add_stream_subscribed("s1", 4); + t.add_full_wf_task(); + + let (core, failures) = worker_expecting_one_nondeterminism_failure(t, ResponseType::AllHistory); + + let task = core.poll_workflow_activation().await.unwrap(); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![ + SubscribeStream { + stream_id: "s2".to_string(), + start_offset: -1, + } + .into(), + ], + )) + .await + .unwrap(); + core.handle_eviction().await; + assert_eq!(failures.load(Ordering::Relaxed), 1); + core.shutdown().await; +} diff --git a/crates/sdk-core/src/worker/workflow/machines/add_stream_messages_state_machine.rs b/crates/sdk-core/src/worker/workflow/machines/add_stream_messages_state_machine.rs index 55bf0dd8f..4a3e137c1 100644 --- a/crates/sdk-core/src/worker/workflow/machines/add_stream_messages_state_machine.rs +++ b/crates/sdk-core/src/worker/workflow/machines/add_stream_messages_state_machine.rs @@ -2,21 +2,34 @@ use super::{ NewMachineWithCommand, StateMachine, TransitionResult, fsm, workflow_machines::MachineResponse, }; use crate::worker::workflow::{ - WFMachinesError, + WFMachinesError, fatal, machines::{EventInfo, HistEventData, WFMachinesAdapter}, + nondeterminism, }; use temporalio_common::protos::{ coresdk::workflow_commands::AddStreamMessages, - temporal::api::enums::v1::{CommandType, EventType}, + temporal::api::{ + enums::v1::{CommandType, EventType}, + history::v1::{WorkflowStreamMessagesAddedEventAttributes, history_event}, + }, }; fsm! { pub(super) name AddStreamMessagesMachine; command AddStreamMessagesMachineCommand; error WFMachinesError; + shared_state SharedState; Created --(CommandScheduled) --> CommandIssued; - CommandIssued --(CommandRecorded) --> Done; + CommandIssued --(CommandRecorded(WorkflowStreamMessagesAddedEventAttributes), + shared on_command_recorded) --> Done; +} + +/// What the command claimed, kept so the recorded event can be held against it. +#[derive(Default, Clone)] +pub(super) struct SharedState { + stream_id: String, + message_count: i64, } /// Publish a batch of messages to a stream this workflow owns. @@ -25,7 +38,13 @@ fsm! { /// offset range the batch landed at. The offsets are assigned by the server, so /// nothing here predicts them. pub(super) fn add_stream_messages(lang_cmd: AddStreamMessages) -> NewMachineWithCommand { - let sm = AddStreamMessagesMachine::from_parts(Created {}.into(), ()); + let sm = AddStreamMessagesMachine::from_parts( + Created {}.into(), + SharedState { + stream_id: lang_cmd.stream_id.clone(), + message_count: lang_cmd.messages.len() as i64, + }, + ); NewMachineWithCommand { command: lang_cmd.into(), machine: sm.into(), @@ -41,6 +60,30 @@ pub(super) struct Created {} #[derive(Debug, Default, Clone, derive_more::Display)] pub(super) struct CommandIssued {} +impl CommandIssued { + pub(super) fn on_command_recorded( + self, + dat: &mut SharedState, + attrs: WorkflowStreamMessagesAddedEventAttributes, + ) -> AddStreamMessagesMachineTransition { + // An empty id names the workflow's default stream, and the server is the + // one that resolves that name, so only a named stream can be compared. + let same_stream = dat.stream_id.is_empty() || dat.stream_id == attrs.stream_id; + if same_stream && dat.message_count == attrs.message_count { + TransitionResult::default() + } else { + TransitionResult::Err(nondeterminism!( + "Recorded publish of {} messages to stream {:?} does not match the reissued \ + publish of {} messages to stream {:?}", + attrs.message_count, + attrs.stream_id, + dat.message_count, + dat.stream_id + )) + } + } +} + #[derive(Debug, Default, Clone, derive_more::Display)] pub(super) struct Done {} @@ -63,7 +106,14 @@ impl TryFrom for AddStreamMessagesMachineEvents { let e = e.event; match e.event_type() { EventType::WorkflowStreamMessagesAdded => { - Ok(AddStreamMessagesMachineEvents::CommandRecorded) + if let Some( + history_event::Attributes::WorkflowStreamMessagesAddedEventAttributes(attrs), + ) = e.attributes + { + Ok(AddStreamMessagesMachineEvents::CommandRecorded(attrs)) + } else { + Err(fatal!("Stream messages added attributes were unset: {e}")) + } } _ => Err(Self::Error::Nondeterminism(format!( "AddStreamMessagesMachine does not handle {e}" @@ -85,12 +135,6 @@ impl TryFrom for AddStreamMessagesMachineEvents { } } -impl From for Done { - fn from(_: CommandIssued) -> Self { - Self {} - } -} - impl From for CommandIssued { fn from(_: Created) -> Self { Self {} diff --git a/crates/sdk-core/src/worker/workflow/machines/subscribe_stream_state_machine.rs b/crates/sdk-core/src/worker/workflow/machines/subscribe_stream_state_machine.rs index 2e323bd70..3fc08e768 100644 --- a/crates/sdk-core/src/worker/workflow/machines/subscribe_stream_state_machine.rs +++ b/crates/sdk-core/src/worker/workflow/machines/subscribe_stream_state_machine.rs @@ -2,21 +2,35 @@ use super::{ NewMachineWithCommand, StateMachine, TransitionResult, fsm, workflow_machines::MachineResponse, }; use crate::worker::workflow::{ - WFMachinesError, + WFMachinesError, fatal, machines::{EventInfo, HistEventData, WFMachinesAdapter}, + nondeterminism, }; use temporalio_common::protos::{ coresdk::workflow_commands::SubscribeStream, - temporal::api::enums::v1::{CommandType, EventType}, + temporal::api::{ + enums::v1::{CommandType, EventType}, + history::v1::{WorkflowStreamSubscribedEventAttributes, history_event}, + }, }; fsm! { pub(super) name SubscribeStreamMachine; command SubscribeStreamMachineCommand; error WFMachinesError; + shared_state SharedState; Created --(CommandScheduled) --> CommandIssued; - CommandIssued --(CommandRecorded) --> Done; + CommandIssued --(CommandRecorded(WorkflowStreamSubscribedEventAttributes), + shared on_command_recorded) --> Done; +} + +/// The stream the command named, kept so the recorded event can be held against it. +/// The start offset is not kept: the server resolves it, so the recorded value is +/// its answer rather than what the command said. +#[derive(Default, Clone)] +pub(super) struct SharedState { + stream_id: String, } /// Subscribe this workflow to a stream. The command carries only the stream id @@ -24,7 +38,12 @@ fsm! { /// cannot look it up without doing I/O and a value it carried would be a /// reading rather than a fact. pub(super) fn subscribe_stream(lang_cmd: SubscribeStream) -> NewMachineWithCommand { - let sm = SubscribeStreamMachine::from_parts(Created {}.into(), ()); + let sm = SubscribeStreamMachine::from_parts( + Created {}.into(), + SharedState { + stream_id: lang_cmd.stream_id.clone(), + }, + ); NewMachineWithCommand { command: lang_cmd.into(), machine: sm.into(), @@ -40,6 +59,25 @@ pub(super) struct Created {} #[derive(Debug, Default, Clone, derive_more::Display)] pub(super) struct CommandIssued {} +impl CommandIssued { + pub(super) fn on_command_recorded( + self, + dat: &mut SharedState, + attrs: WorkflowStreamSubscribedEventAttributes, + ) -> SubscribeStreamMachineTransition { + if dat.stream_id == attrs.stream_id { + TransitionResult::default() + } else { + TransitionResult::Err(nondeterminism!( + "Recorded subscription to stream {:?} does not match the reissued subscription \ + to stream {:?}", + attrs.stream_id, + dat.stream_id + )) + } + } +} + #[derive(Debug, Default, Clone, derive_more::Display)] pub(super) struct Done {} @@ -62,7 +100,14 @@ impl TryFrom for SubscribeStreamMachineEvents { let e = e.event; match e.event_type() { EventType::WorkflowStreamSubscribed => { - Ok(SubscribeStreamMachineEvents::CommandRecorded) + if let Some(history_event::Attributes::WorkflowStreamSubscribedEventAttributes( + attrs, + )) = e.attributes + { + Ok(SubscribeStreamMachineEvents::CommandRecorded(attrs)) + } else { + Err(fatal!("Stream subscribed attributes were unset: {e}")) + } } _ => Err(Self::Error::Nondeterminism(format!( "SubscribeStreamMachine does not handle {e}" @@ -84,12 +129,6 @@ impl TryFrom for SubscribeStreamMachineEvents { } } -impl From for Done { - fn from(_: CommandIssued) -> Self { - Self {} - } -} - impl From for CommandIssued { fn from(_: Created) -> Self { Self {} From 83ab23fe3a21eff4a75a7f50a5a699bdc702ec8e Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 18 Sep 2026 16:55:15 -0700 Subject: [PATCH 11/35] Gave a data-only stream task its own replay activation. A completion that records consumed stream cursors ends its task sequence. Such a task ran as one activation live, and folding it into a heartbeat chain would hand several ranges over at once on replay. --- crates/sdk-core/src/core_tests/streams.rs | 51 +++++++++++++++++++ .../src/worker/workflow/history_update.rs | 11 ++++ 2 files changed, 62 insertions(+) diff --git a/crates/sdk-core/src/core_tests/streams.rs b/crates/sdk-core/src/core_tests/streams.rs index f01884ce9..3bce8f932 100644 --- a/crates/sdk-core/src/core_tests/streams.rs +++ b/crates/sdk-core/src/core_tests/streams.rs @@ -821,3 +821,54 @@ async fn a_subscribe_reissued_to_a_different_stream_fails_the_task() { assert_eq!(failures.load(Ordering::Relaxed), 1); core.shutdown().await; } + +/// A task that consumes a range and issues no command still ran as its own +/// activation, so replay hands each such range over in its own activation +/// rather than collapsing the run of them into one, the way it does for +/// heartbeats that did nothing. +#[tokio::test] +async fn data_only_tasks_replay_one_range_per_activation() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + let first = t.add_workflow_task_completed_with_stream_cursors(vec![cursor("s1", 0, 1)]); + t.add_workflow_task_scheduled_and_started(); + let second = t.add_workflow_task_completed_with_stream_cursors(vec![cursor("s1", 1, 2)]); + t.add_workflow_task_scheduled_and_started(); + let third = t.add_workflow_task_completed_with_stream_cursors(vec![cursor("s1", 2, 3)]); + t.add_workflow_task_scheduled_and_started(); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + poll_resp.add_stream_slice("s1", first, 0, &["a"]); + poll_resp.add_stream_slice("s1", second, 1, &["b"]); + poll_resp.add_stream_slice("s1", third, 2, &["c"]); + poll_resp.add_stream_slice("s1", 0, 3, &["d"]); + + let mock = MockPollCfg::from_resp_batches( + "wfid", + t, + [ResponseType::Raw(poll_resp.resp)], + mock_worker_client(), + ); + let core = mock_worker(build_mock_pollers(mock)); + + let mut per_activation = vec![]; + for _ in 0..4 { + let task = core.poll_workflow_activation().await.unwrap(); + per_activation.push(delivered_ranges(&task)); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + } + assert_eq!( + per_activation, + vec![ + vec![("s1".to_string(), 0, 1)], + vec![("s1".to_string(), 1, 2)], + vec![("s1".to_string(), 2, 3)], + vec![("s1".to_string(), 3, 4)], + ], + "each consumed range replays in the activation of the task that consumed it" + ); + core.shutdown().await; +} diff --git a/crates/sdk-core/src/worker/workflow/history_update.rs b/crates/sdk-core/src/worker/workflow/history_update.rs index a8523b9cb..a923299ec 100644 --- a/crates/sdk-core/src/worker/workflow/history_update.rs +++ b/crates/sdk-core/src/worker/workflow/history_update.rs @@ -763,6 +763,17 @@ fn find_end_index_of_next_wft_seq( wft_started_event_id_to_index.pop(); continue; } else if next_event_type == EventType::WorkflowTaskCompleted { + // A task that consumed a stream range issued nothing the machines match, + // but it was an activation of its own: the workflow was handed that range + // and ran on it. Replay has to give it its own activation too, rather than + // fold it into a heartbeat chain and hand several ranges over at once. + if let Some(Attributes::WorkflowTaskCompletedEventAttributes(ref attrs)) = + next_event.attributes + && !attrs.stream_cursors.is_empty() + { + saw_command = true; + saw_command_or_started = true; + } if let Some(next_next_event) = events.get(ix + 2) { if !saw_command && next_next_event.event_type() == EventType::WorkflowTaskScheduled From 23429135385724e0be1512521addac2bbaa1849c Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 18 Sep 2026 16:57:54 -0700 Subject: [PATCH 12/35] Ordered stream ranges and checked re-supplied slices against History. Ranges for one task are handed over sorted by stream and offset, so a workflow waiting on several streams sees the same order live and on replay. A re-supplied slice has to cover exactly the recorded offsets, and a range that observed nothing is rebuilt from the cursor rather than demanded from the server. --- crates/sdk-core/src/core_tests/streams.rs | 112 ++++++++++++++ .../workflow/machines/workflow_machines.rs | 144 +++++++++++++++--- 2 files changed, 232 insertions(+), 24 deletions(-) diff --git a/crates/sdk-core/src/core_tests/streams.rs b/crates/sdk-core/src/core_tests/streams.rs index 3bce8f932..3fe7f4bf9 100644 --- a/crates/sdk-core/src/core_tests/streams.rs +++ b/crates/sdk-core/src/core_tests/streams.rs @@ -872,3 +872,115 @@ async fn data_only_tasks_replay_one_range_per_activation() { ); core.shutdown().await; } + +/// Ranges for several streams on one task arrive ordered by stream, however +/// the server laid them out. A workflow that waits on two streams with a +/// first-completed pattern would otherwise take whichever branch the server's +/// iteration order happened to pick, live and again differently on replay. +#[tokio::test] +async fn ranges_for_several_streams_arrive_in_stream_order() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + let completed = t.add_workflow_task_completed_with_stream_cursors(vec![ + cursor("s2", 0, 1), + cursor("s1", 0, 1), + ]); + t.add_workflow_task_scheduled_and_started(); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + poll_resp.add_stream_slice("s2", completed, 0, &["two"]); + poll_resp.add_stream_slice("s1", completed, 0, &["one"]); + poll_resp.add_stream_slice("s2", 0, 1, &["four"]); + poll_resp.add_stream_slice("s1", 0, 1, &["three"]); + + let mock = MockPollCfg::from_resp_batches( + "wfid", + t, + [ResponseType::Raw(poll_resp.resp)], + mock_worker_client(), + ); + let core = mock_worker(build_mock_pollers(mock)); + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!( + delivered_ranges(&task), + vec![("s1".to_string(), 0, 1), ("s2".to_string(), 0, 1)], + "re-supplied ranges follow stream order" + ); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!( + delivered_ranges(&task), + vec![("s1".to_string(), 1, 2), ("s2".to_string(), 1, 2)], + "live ranges follow stream order" + ); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + core.shutdown().await; +} + +/// A recorded range that observed nothing is rebuilt from the cursor alone. The +/// event already says everything the workflow needs, so replay does not depend +/// on the server sending an empty slice back for it. +#[tokio::test] +async fn an_empty_recorded_range_replays_without_a_slice_from_the_server() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + t.add_workflow_task_completed_with_stream_cursors(vec![cursor("s1", 4, 4)]); + t.add_workflow_task_scheduled_and_started(); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + poll_resp.add_stream_slice("s1", 0, 4, &["e"]); + + let mock = MockPollCfg::from_resp_batches( + "wfid", + t, + [ResponseType::Raw(poll_resp.resp)], + mock_worker_client(), + ); + let core = mock_worker(build_mock_pollers(mock)); + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(delivered_ranges(&task), vec![("s1".to_string(), 4, 4)]); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(delivered_ranges(&task), vec![("s1".to_string(), 4, 5)]); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + core.shutdown().await; +} + +/// A re-supplied slice is checked against the cursor it claims to satisfy. The +/// event is the record of what the task saw, so bytes covering other offsets +/// would replay the task on different input than it ran on. +#[tokio::test] +async fn a_resupplied_slice_that_disagrees_with_the_record_fails_the_task() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + let completed = t.add_workflow_task_completed_with_stream_cursors(vec![cursor("s1", 0, 2)]); + t.add_workflow_task_scheduled_and_started(); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + // One message where the record says two. + poll_resp.add_stream_slice("s1", completed, 0, &["a"]); + + let (core, failures) = + worker_expecting_one_nondeterminism_failure(t, ResponseType::Raw(poll_resp.resp)); + + // The mismatch is found while the poll response is applied, so the first + // activation is already the eviction. + core.handle_eviction().await; + assert_eq!(failures.load(Ordering::Relaxed), 1); + core.shutdown().await; +} diff --git a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs index 25dd8b122..4bd17933b 100644 --- a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs +++ b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs @@ -72,7 +72,7 @@ use temporalio_common::{ history::v1::{HistoryEvent, history_event}, protocol::v1::{Message as ProtocolMessage, message::SequencingId}, sdk::v1::WorkflowTaskCompletedMetadata, - stream::v1::StreamSlice, + stream::v1::{StreamCursor, StreamSlice}, }, }, worker::WorkerDeploymentVersion, @@ -375,6 +375,11 @@ impl WorkflowMachines { .push(slice); } } + // The order ranges arrive in is something the workflow can branch on, and + // the server's order is whatever its iteration happened to produce. Fixing + // it here is what makes it the same live and on replay. + self.current_stream_slices + .sort_by(|a, b| stream_order(&a.stream_id, a.from_offset, &b.stream_id, b.from_offset)); self.new_history_from_server(update)?; Ok(()) } @@ -662,12 +667,12 @@ impl WorkflowMachines { } }}; } - let mut replayed_slice_events = vec![]; + let mut replayed_slice_events: Vec<(i64, Vec)> = vec![]; // Kept apart from the in-batch ones. An in-batch event whose bytes are // missing is a real inconsistency; a looked-ahead one may simply not // have been sent yet, and demanding it would turn an early delivery // into a new way to fail. - let mut lookahead_slice_event: Option = None; + let mut lookahead_slice_event: Option<(i64, Vec)> = None; let mut peeked_events = events.iter().peekable(); while let Some(event) = peeked_events.next() { if let Some(history_event::Attributes::WorkflowTaskCompletedEventAttributes(ref wtc)) = @@ -677,7 +682,7 @@ impl WorkflowMachines { // The event records which offsets that task consumed; the server // sends the bytes back separately, keyed by this event. if !wtc.stream_cursors.is_empty() { - replayed_slice_events.push(event.event_id); + replayed_slice_events.push((event.event_id, wtc.stream_cursors.clone())); } } if peeked_events.peek().is_none() @@ -693,7 +698,7 @@ impl WorkflowMachines { // caused, so a read-then-publish task replays with nothing to // decide from and reissues no command. Look ahead for it here. if !wtc.stream_cursors.is_empty() { - lookahead_slice_event = Some(wtc_id); + lookahead_slice_event = Some((wtc_id, wtc.stream_cursors.clone())); } } } @@ -898,8 +903,8 @@ impl WorkflowMachines { // delivered live at the time. There is nothing to re-supply, the server // sends nothing, and re-supplying would hand the workflow the same // messages twice. - replayed_slice_events.sort_unstable(); - for event_id in replayed_slice_events { + replayed_slice_events.sort_unstable_by_key(|(event_id, _)| *event_id); + for (event_id, cursors) in replayed_slice_events { // Delivered live to this instance when the task ran, so there is // nothing to re-supply and the server sent nothing. if event_id <= self.stream_slices_delivered_through + 1 { @@ -911,33 +916,46 @@ impl WorkflowMachines { self.stream_slices_by_event.remove(&event_id); continue; } - let Some(slices) = self.stream_slices_by_event.remove(&event_id) else { + let slices = self.stream_slices_by_event.remove(&event_id); + if slices.is_none() + && let Some(cursor) = cursors.iter().find(|c| c.from_offset < c.to_offset) + { // History says a task consumed a range and the server sent no // bytes for it. Replaying with less data than the original run // had produces different commands, and the mismatch would - // surface later as an unrelated nondeterminism error. - return Err(WFMachinesError::Nondeterminism(format!( - "Event {event_id} records consumed stream offsets, \ - but the server sent no messages for them" - ))); - }; - for slice in slices { - self.drive_me.send_job(deliver_stream_messages_job(slice)); + // surface later as an unrelated nondeterminism error. Reaching + // this from here also means the lookahead did not find the + // completion, so the range was due one activation before this. + return Err(nondeterminism!( + "Event {event_id} records that stream {} was consumed from offset {} to {}, \ + but the server sent no messages for it. The workflow was owed that range \ + one activation earlier.", + cursor.stream_id, + cursor.from_offset, + cursor.to_offset + )); + } + for job in resupplied_deliveries(event_id, cursors, slices.unwrap_or_default())? { + self.drive_me.send_job(job); } } // The task about to be replayed, whose range is only visible by looking - // ahead to the completion that closes it. Absent bytes mean the server - // has not re-supplied them on this response, which the ordinary path - // above will still catch when that completion arrives as an event. - if let Some(event_id) = lookahead_slice_event + // ahead to the completion that closes it. Absent bytes for a range with + // content mean the server has not re-supplied them on this response, + // which the ordinary path above will still catch when that completion + // arrives as an event. + if let Some((event_id, cursors)) = lookahead_slice_event && event_id > self.stream_slices_delivered_through + 1 && event_id > self.stream_slices_lookahead_through - && let Some(slices) = self.stream_slices_by_event.remove(&event_id) { - for slice in slices { - self.drive_me.send_job(deliver_stream_messages_job(slice)); + let slices = self.stream_slices_by_event.remove(&event_id); + let needs_bytes = cursors.iter().any(|c| c.from_offset < c.to_offset); + if slices.is_some() || !needs_bytes { + for job in resupplied_deliveries(event_id, cursors, slices.unwrap_or_default())? { + self.drive_me.send_job(job); + } + self.stream_slices_lookahead_through = event_id; } - self.stream_slices_lookahead_through = event_id; } // Then the range for the task about to run, which is only meaningful // once we have caught up to it. @@ -1974,3 +1992,81 @@ fn deliver_stream_messages_job(slice: StreamSlice) -> OutgoingJob { } .into() } + +/// The one order ranges are handed to a workflow in, live and on replay. +fn stream_order( + stream_a: &str, + from_offset_a: i64, + stream_b: &str, + from_offset_b: i64, +) -> std::cmp::Ordering { + (stream_a, from_offset_a).cmp(&(stream_b, from_offset_b)) +} + +/// Pair the ranges a completion event recorded with the slices the server sent +/// back for it, in the order the workflow is handed them. +/// +/// The event is the record of what the task saw, so the slices have to match +/// it rather than the other way round. A range that observed nothing needs no +/// bytes and is rebuilt from the cursor alone; a range with content has to +/// arrive, and what arrives has to cover exactly the recorded offsets. Anything +/// else would replay the task with different input than it ran on. +fn resupplied_deliveries( + event_id: i64, + mut cursors: Vec, + mut slices: Vec, +) -> Result> { + cursors.sort_by(|a, b| stream_order(&a.stream_id, a.from_offset, &b.stream_id, b.from_offset)); + let mut jobs = Vec::with_capacity(cursors.len()); + for cursor in cursors { + let slice = slices + .iter() + .position(|s| s.stream_id == cursor.stream_id) + .map(|ix| slices.swap_remove(ix)); + let slice = match slice { + Some(slice) + if slice.from_offset == cursor.from_offset + && slice.to_offset == cursor.to_offset => + { + slice + } + Some(slice) => { + return Err(nondeterminism!( + "Event {event_id} records that stream {} was consumed from offset {} to {}, \ + but the server sent offsets {} to {} for it", + cursor.stream_id, + cursor.from_offset, + cursor.to_offset, + slice.from_offset, + slice.to_offset + )); + } + None if cursor.from_offset == cursor.to_offset => StreamSlice { + stream_id: cursor.stream_id, + from_offset: cursor.from_offset, + to_offset: cursor.to_offset, + ..Default::default() + }, + None => { + return Err(nondeterminism!( + "Event {event_id} records that stream {} was consumed from offset {} to {}, \ + but the server sent no messages for it", + cursor.stream_id, + cursor.from_offset, + cursor.to_offset + )); + } + }; + jobs.push(deliver_stream_messages_job(slice)); + } + if let Some(extra) = slices.first() { + return Err(nondeterminism!( + "The server sent stream {} from offset {} to {} for event {event_id}, which records \ + no such range", + extra.stream_id, + extra.from_offset, + extra.to_offset + )); + } + Ok(jobs) +} From 0cc74f74266c3e4e5a5c1d8cc73f49afd90d4b66 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 18 Sep 2026 17:00:48 -0700 Subject: [PATCH 13/35] Covered the stream delivery guards and each range's activation. --- crates/sdk-core/src/core_tests/streams.rs | 94 ++++++++++++++++++++--- 1 file changed, 85 insertions(+), 9 deletions(-) diff --git a/crates/sdk-core/src/core_tests/streams.rs b/crates/sdk-core/src/core_tests/streams.rs index 3fe7f4bf9..807c001a4 100644 --- a/crates/sdk-core/src/core_tests/streams.rs +++ b/crates/sdk-core/src/core_tests/streams.rs @@ -219,8 +219,11 @@ async fn replays_recorded_ranges_in_order_before_the_current_one() { mock.worker_cfg(|wc| wc.max_cached_workflows = 1); let core = mock_worker(mock); + // Each entry is (activation, from, to, bodies). The activation matters as + // much as the order: a range that came back one activation late would + // still be in order. let mut seen = vec![]; - loop { + for activation in 1..=3 { let task = core.poll_workflow_activation().await.unwrap(); for job in &task.jobs { if matches!( @@ -229,29 +232,27 @@ async fn replays_recorded_ranges_in_order_before_the_current_one() { ) { let (_, from, to, bodies) = delivered(job); seen.push(( + activation, from, to, bodies.iter().map(|b| b.to_vec()).collect::>(), )); } } - let done = seen.len() >= 3; core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) .await .unwrap(); - if done { - break; - } } assert_eq!( seen, vec![ - (0, 2, vec![b"alpha".to_vec(), b"beta".to_vec()]), - (2, 3, vec![b"gamma".to_vec()]), - (3, 4, vec![b"delta".to_vec()]), + (1, 0, 2, vec![b"alpha".to_vec(), b"beta".to_vec()]), + (2, 2, 3, vec![b"gamma".to_vec()]), + (3, 3, 4, vec![b"delta".to_vec()]), ], - "recorded ranges come back in event order, then the live one" + "each recorded range comes back in the activation of the task that consumed it, \ + then the live one" ); } @@ -984,3 +985,78 @@ async fn a_resupplied_slice_that_disagrees_with_the_record_fails_the_task() { assert_eq!(failures.load(Ordering::Relaxed), 1); core.shutdown().await; } + +/// A run that stays cached was handed its range as the task ran. When the next +/// task arrives, the completion recording that range is in its history, and the +/// server may re-supply the range tagged with it, as it would for a worker that +/// lost the run. This worker did not, so it must not process the same records +/// twice. +#[tokio::test] +async fn a_cached_run_is_not_handed_its_own_range_again() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + let completed = t.add_workflow_task_completed_with_stream_cursors(vec![cursor("s1", 0, 2)]); + t.add_workflow_task_scheduled_and_started(); + + let mut first_poll = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::ToTaskNum(1)); + first_poll.add_stream_slice("s1", 0, 0, &["a", "b"]); + let mut second_poll = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::OneTask(2)); + second_poll.add_stream_slice("s1", completed, 0, &["a", "b"]); + second_poll.add_stream_slice("s1", 0, 2, &["c"]); + + let mock = MockPollCfg::from_resp_batches( + "wfid", + t, + [ + ResponseType::Raw(first_poll.resp), + ResponseType::Raw(second_poll.resp), + ], + mock_worker_client(), + ); + let mut mock = build_mock_pollers(mock); + mock.worker_cfg(|wc| wc.max_cached_workflows = 1); + let core = mock_worker(mock); + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(delivered_ranges(&task), vec![("s1".to_string(), 0, 2)]); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + + let task = core.poll_workflow_activation().await.unwrap(); + assert!(!task.is_replaying); + assert_eq!( + delivered_ranges(&task), + vec![("s1".to_string(), 2, 3)], + "only the new range; the first one was delivered live to this worker" + ); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + core.shutdown().await; +} + +/// History says a task consumed a range with content and the server sent no +/// bytes for it. The lookahead leaves that alone, since the completion may +/// arrive with a later response, and the batch that carries the completion +/// fails the task rather than replay it on less input than it ran on. +#[tokio::test] +async fn a_missing_resupply_for_a_consumed_range_fails_the_task() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + t.add_workflow_task_completed_with_stream_cursors(vec![cursor("s1", 0, 2)]); + t.add_workflow_task_scheduled_and_started(); + + let (core, failures) = worker_expecting_one_nondeterminism_failure(t, ResponseType::AllHistory); + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(delivered_ranges(&task), vec![]); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + core.handle_eviction().await; + assert_eq!(failures.load(Ordering::Relaxed), 1); + core.shutdown().await; +} From 6cf953eee23396150f5169293449a072229b1107 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 18 Sep 2026 17:00:48 -0700 Subject: [PATCH 14/35] Renumbered the stream protos to shared numbers and moved a doc comment. The external stream family, developed alongside this one, takes 17 to 20 in the activation job and 23 to 28 in the command. Both trees now encode the native stream messages at 21, 29 and 30, so an artifact generated from either decodes correctly against the other. --- .../workflow_activation.proto | 6 +++++- .../workflow_commands/workflow_commands.proto | 19 +++++++++++-------- 2 files changed, 16 insertions(+), 9 deletions(-) diff --git a/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto b/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto index c37024e05..43198097e 100644 --- a/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto +++ b/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto @@ -140,8 +140,12 @@ message WorkflowActivationJob { ResolveNexusOperationStart resolve_nexus_operation_start = 15; // A nexus operation resolved. ResolveNexusOperation resolve_nexus_operation = 16; + // 17 to 20 are taken by the external stream jobs, which are developed + // alongside this one and share this message. The number below is fixed + // with that family and must not be reused. + // // A range of a stream the workflow subscribed to. - DeliverStreamMessages deliver_stream_messages = 17; + DeliverStreamMessages deliver_stream_messages = 21; // Remove the workflow identified by the [WorkflowActivation] containing this job from the // cache after performing the activation. It is guaranteed that this will be the only job // in the activation if present. diff --git a/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto b/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto index d4c9a1d1b..14f1ded09 100644 --- a/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto +++ b/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto @@ -55,17 +55,14 @@ message WorkflowCommand { UpdateResponse update_response = 20; ScheduleNexusOperation schedule_nexus_operation = 21; RequestCancelNexusOperation request_cancel_nexus_operation = 22; - SubscribeStream subscribe_stream = 23; - AddStreamMessages add_stream_messages = 24; + // 23 to 28 are taken by the external stream commands, which are developed + // alongside these and share this message. The two numbers below are fixed + // with that family and must not be reused. + SubscribeStream subscribe_stream = 29; + AddStreamMessages add_stream_messages = 30; } } -// Subscribe this workflow to a stream, so later Workflow Tasks carry the -// ranges it has not consumed yet. -// -// The stream's addressing is resolved by the server. A workflow cannot look it -// up without doing I/O, and a value it carried would be a reading rather than a -// fact, so it could differ on replay. // Publish a batch of messages to a stream this workflow owns. // // The bodies go to the stream's own log rather than into History, which gets @@ -77,6 +74,12 @@ message AddStreamMessages { repeated temporal.api.stream.v1.StreamMessage messages = 2; } +// Subscribe this workflow to a stream, so later Workflow Tasks carry the +// ranges it has not consumed yet. +// +// The stream's addressing is resolved by the server. A workflow cannot look it +// up without doing I/O, and a value it carried would be a reading rather than a +// fact, so it could differ on replay. message SubscribeStream { string stream_id = 1; // Negative means from wherever the stream is when the subscription is From a67781227a88df50a0617247d20d4908f96d4593 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 18 Sep 2026 17:00:48 -0700 Subject: [PATCH 15/35] Added the changelog entry for the native stream commands and job. --- crates/sdk-core/CHANGELOG.md | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/crates/sdk-core/CHANGELOG.md b/crates/sdk-core/CHANGELOG.md index bfb8e04f9..2dabc4c93 100644 --- a/crates/sdk-core/CHANGELOG.md +++ b/crates/sdk-core/CHANGELOG.md @@ -50,6 +50,10 @@ relevant information. metrics now carry a `failure_reason` attribute. Each is now split into one time series per reason, which may affect existing dashboards. * Workflow task completions larger than the gRPC request size limit are now paginated automatically when the namespace supports it. Paginated workflow task completions require Temporal Server 1.32.0 or later. +* Workflows can subscribe to server-side streams and publish batches of messages to them with the + `SubscribeStream` and `AddStreamMessages` commands. Consumed ranges reach the workflow as + `DeliverStreamMessages` activation jobs, and replay hands each recorded range back in the + activation of the task that consumed it. ### Breaking Changes :boom: * The following types are now non-exhaustive: `Priority`, `WorkerDeploymentVersion`, From 97cd17296fcc6178407c770f96f354bcc4e9f540 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Mon, 21 Sep 2026 01:47:27 -0700 Subject: [PATCH 16/35] Renamed the stream command, job and record to the api's record vocabulary. The api names an entry a record and carries its kind, producer, attempt and sequence on it, so the vendored stream protos follow the api head and lang appends with AppendStreamRecords and receives DeliverStreamRecords. Core passes each record through untouched. --- crates/common/build.rs | 6 +- .../temporal/api/command/v1/message.proto | 12 ++- .../temporal/api/enums/v1/command_type.proto | 2 +- .../temporal/api/enums/v1/event_type.proto | 10 +- .../temporal/api/enums/v1/failed_cause.proto | 8 ++ .../temporal/api/history/v1/message.proto | 23 ++-- .../temporal/api/stream/v1/message.proto | 51 +++++++-- .../workflow_activation.proto | 6 +- .../workflow_commands/workflow_commands.proto | 12 ++- crates/protos/src/protos/mod.rs | 30 +++--- crates/sdk-core/CHANGELOG.md | 6 +- crates/sdk-core/src/core_tests/streams.rs | 101 +++++++++--------- crates/sdk-core/src/replay/history_builder.rs | 18 ++-- .../sdk-core/src/test_help/integ_helpers.rs | 6 +- .../src/worker/workflow/history_update.rs | 2 +- ...=> append_stream_records_state_machine.rs} | 60 ++++++----- .../src/worker/workflow/machines/mod.rs | 6 +- .../workflow/machines/transition_coverage.rs | 8 +- .../workflow/machines/workflow_machines.rs | 37 +++---- crates/sdk-core/src/worker/workflow/mod.rs | 6 +- crates/sdk/src/workflow_future.rs | 4 +- crates/workflow/src/runtime/instance.rs | 4 +- 22 files changed, 237 insertions(+), 181 deletions(-) rename crates/sdk-core/src/worker/workflow/machines/{add_stream_messages_state_machine.rs => append_stream_records_state_machine.rs} (59%) diff --git a/crates/common/build.rs b/crates/common/build.rs index 8e17815b0..1baae9de8 100644 --- a/crates/common/build.rs +++ b/crates/common/build.rs @@ -820,11 +820,11 @@ const NOT_VALIDATED_FIELDS: &[&str] = &[ "temporal.api.workflowservice.v1.StartWorkflowExecutionRequest.continued_failure", "temporal.api.workflowservice.v1.StartWorkflowExecutionRequest.last_completion_result", "temporal.api.workflowservice.v1.TerminateWorkflowExecutionRequest.details", - // Stream messages: the blob limit is a per-event limit, and these bodies never reach an + // Stream records: the blob limit is a per-event limit, and these bodies never reach an // event. The server bounds the batch by message count (MaxMessagesPerBatch) instead, which // is not a payload size the SDK can mirror. - "temporal.api.stream.v1.StreamMessage.body", - "temporal.api.stream.v1.StreamMessage.metadata", + "temporal.api.stream.v1.StreamRecord.body", + "temporal.api.stream.v1.StreamRecord.metadata", // Dedicated, non-fetchable limits (not blob/memo, not in DescribeNamespace): UserMetadata // (nexus-start only); Nexus EndpointSpec.description (maxDescriptionSize; cloud variant cloud-only). "temporal.api.sdk.v1.UserMetadata.details", diff --git a/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto index 57a322af3..f1dd29d2b 100644 --- a/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto @@ -287,10 +287,16 @@ message RequestCancelNexusOperationCommandAttributes { int64 scheduled_event_id = 1; } -message AddStreamMessagesCommandAttributes { +// Appends records to a stream the Workflow owns. Applied inside the Workflow +// Task's own commit. Produces one `WorkflowStreamRecordsAppended` event +// carrying the offset range and none of the payload; it schedules no further +// work. +message AppendStreamRecordsCommandAttributes { // Empty means the Workflow's default output stream. string stream_id = 1; - repeated temporal.api.stream.v1.StreamMessage messages = 2; + // Stored in order. The server sets `producer_id` to empty on each record, + // because the owning Workflow is the producer here. + repeated temporal.api.stream.v1.StreamRecord records = 2; } message SubscribeStreamCommandAttributes { @@ -338,7 +344,7 @@ message Command { ScheduleNexusOperationCommandAttributes schedule_nexus_operation_command_attributes = 18; RequestCancelNexusOperationCommandAttributes request_cancel_nexus_operation_command_attributes = 19; - AddStreamMessagesCommandAttributes add_stream_messages_command_attributes = 20; + AppendStreamRecordsCommandAttributes append_stream_records_command_attributes = 20; SubscribeStreamCommandAttributes subscribe_stream_command_attributes = 21; } } diff --git a/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto b/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto index 677ed3b2d..967169b19 100644 --- a/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto +++ b/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto @@ -29,6 +29,6 @@ enum CommandType { COMMAND_TYPE_MODIFY_WORKFLOW_PROPERTIES = 16; COMMAND_TYPE_SCHEDULE_NEXUS_OPERATION = 17; COMMAND_TYPE_REQUEST_CANCEL_NEXUS_OPERATION = 18; - COMMAND_TYPE_ADD_STREAM_MESSAGES = 19; + COMMAND_TYPE_APPEND_STREAM_RECORDS = 19; COMMAND_TYPE_SUBSCRIBE_STREAM = 20; } diff --git a/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto b/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto index af18ac57c..a386b1db7 100644 --- a/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto +++ b/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto @@ -175,12 +175,12 @@ enum EventType { EVENT_TYPE_WORKFLOW_EXECUTION_UNPAUSED = 59; // An event that indicates time skipping advanced time or was disabled automatically after a bound was reached. EVENT_TYPE_WORKFLOW_EXECUTION_TIME_SKIPPING_TRANSITIONED = 60; - // A Workflow subscribed to a stream. Once per subscription, not per - // message: consumed offsets ride WorkflowTaskCompleted and payloads never - // enter History. + // A Workflow subscribed to a stream. Recorded once per subscription, not + // per record: the offsets a task consumed ride WorkflowTaskCompleted and + // the payloads never enter History at all. EVENT_TYPE_WORKFLOW_STREAM_SUBSCRIBED = 61; - // A Workflow published a batch of messages to a stream. Recorded per + // A Workflow appended a batch of records to a stream. Recorded per // batch, and carrying only the offset range it landed at: the bodies go to // the stream's own log, never into History. - EVENT_TYPE_WORKFLOW_STREAM_MESSAGES_ADDED = 62; + EVENT_TYPE_WORKFLOW_STREAM_RECORDS_APPENDED = 62; } diff --git a/crates/protos/protos/api_upstream/temporal/api/enums/v1/failed_cause.proto b/crates/protos/protos/api_upstream/temporal/api/enums/v1/failed_cause.proto index 81cbde73e..274e533d5 100644 --- a/crates/protos/protos/api_upstream/temporal/api/enums/v1/failed_cause.proto +++ b/crates/protos/protos/api_upstream/temporal/api/enums/v1/failed_cause.proto @@ -90,6 +90,14 @@ enum WorkflowTaskFailedCause { WORKFLOW_TASK_FAILED_CAUSE_WORKFLOW_PAUSE_REQUESTED_BEFORE_TASK_STARTED = 39; // A workflow task failed because the request exceeded a size limit. WORKFLOW_TASK_FAILED_CAUSE_REQUEST_TOO_LARGE = 40; + // A workflow task completed with an invalid AppendStreamRecords command. + WORKFLOW_TASK_FAILED_CAUSE_BAD_APPEND_STREAM_RECORDS_ATTRIBUTES = 41; + // A workflow task completed with an invalid SubscribeStream command. + WORKFLOW_TASK_FAILED_CAUSE_BAD_SUBSCRIBE_STREAM_ATTRIBUTES = 42; + // A workflow task could not be started because a stream range it consumed and recorded in + // History can no longer be served, for example after truncation or because it exceeds the + // replay bound. Check the workflow task failure message for more information. + WORKFLOW_TASK_FAILED_CAUSE_STREAM_RANGE_UNAVAILABLE = 43; } enum StartChildWorkflowExecutionFailedCause { diff --git a/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto index 70e6d95cb..13e1961ca 100644 --- a/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto @@ -381,9 +381,12 @@ message WorkflowTaskCompletedEventAttributes { // is set. This value updates workflow execution's `versioning_info.deployment_version`. temporal.api.deployment.v1.WorkerDeploymentVersion deployment_version = 11; - // The offsets this task consumed, without the payloads. Recorded so replay - // can reproduce what the task saw, including a range that was empty. - repeated temporal.api.stream.v1.StreamCursor stream_cursors = 20; + // Offset ranges this Workflow Task consumed from streams it subscribes to. + // Recorded on every task where a subscription is active, including when it + // observed nothing: an empty range is a fact replay must reproduce, and + // omitting it would let replay deliver records the Workflow did not have. + // Numbered 20 to leave 14 through 19 free for fields added on the main line. + repeated temporal.api.stream.v1.StreamRange consumed_stream_ranges = 20; } message WorkflowTaskTimedOutEventAttributes { @@ -958,18 +961,18 @@ message ActivityPropertiesModifiedExternallyEventAttributes { temporal.api.common.v1.RetryPolicy new_retry_policy = 2; } -message WorkflowStreamMessagesAddedEventAttributes { - // The WorkflowTaskCompleted event of the task whose command published this +message WorkflowStreamRecordsAppendedEventAttributes { + // The WorkflowTaskCompleted event of the task whose command appended this // batch. int64 workflow_task_completed_event_id = 1; - // Stream the Workflow published to. + // Stream the Workflow appended to. string stream_id = 2; - // Offset the first message of the batch landed at. + // Offset the first record of the batch landed at. int64 first_offset = 3; - // How many messages the batch held. With first_offset this names the range + // How many records the batch held. With first_offset this names the range // without carrying any of it, which is what keeps this event a fixed size // no matter how large the batch or its payloads are. - int64 message_count = 4; + int64 record_count = 4; } message WorkflowStreamSubscribedEventAttributes { @@ -1304,7 +1307,7 @@ message HistoryEvent { WorkflowExecutionUnpausedEventAttributes workflow_execution_unpaused_event_attributes = 64; WorkflowExecutionTimeSkippingTransitionedEventAttributes workflow_execution_time_skipping_transitioned_event_attributes = 65; WorkflowStreamSubscribedEventAttributes workflow_stream_subscribed_event_attributes = 66; - WorkflowStreamMessagesAddedEventAttributes workflow_stream_messages_added_event_attributes = 67; + WorkflowStreamRecordsAppendedEventAttributes workflow_stream_records_appended_event_attributes = 67; } } diff --git a/crates/protos/protos/api_upstream/temporal/api/stream/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/stream/v1/message.proto index a1702e3c7..9f0c88dfb 100644 --- a/crates/protos/protos/api_upstream/temporal/api/stream/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/stream/v1/message.proto @@ -11,29 +11,46 @@ option csharp_namespace = "Temporalio.Api.Stream.V1"; import "temporal/api/common/v1/message.proto"; -message StreamMessage { +// One entry in a stream. The record is the wire format: stores keep it +// serialized as is and readers in every language decode the same bytes. +message StreamRecord { + // The value the producer published, stored as sent. A payload codec + // applies here as it does to any other payload. temporal.api.common.v1.Payload body = 1; + // Producer-supplied provenance, stored as sent. map metadata = 2; + // Producer-supplied grouping label, stored as sent. string topic = 3; - int64 topic_sequence = 4; + // How to read this record. Unspecified is read as DATA. + StreamRecordKind kind = 4; + // Who wrote the record. Empty when the owning Workflow did. + string producer_id = 5; + // The producer's attempt. Readers treat a later attempt by the same + // producer as superseding what the earlier one wrote. + int64 attempt = 6; + // The producer's position within its attempt, or -1 when unnumbered. + // Stored as sent; the server does not assign, validate or order by it. + int64 sequence = 7; } // A contiguous range of a stream delivered to a Workflow Task, along with the -// offsets it covers. The offsets are what History records; the messages +// offsets it covers. The offsets are what History records; the records // themselves are never written to History. message StreamSlice { string stream_id = 1; + // Run id of the execution that owns the stream. Set on both a slice for the + // task being started and a re-supplied one. string run_id = 2; // Inclusive. int64 from_offset = 3; // Exclusive. Equal to from_offset when the subscription observed nothing, // which is a fact replay has to reproduce rather than an absence of one. int64 to_offset = 4; - repeated StreamMessage messages = 5; - // The WorkflowTaskCompleted event whose stream_cursors recorded this range. - // Set only when the server is re-supplying a range for a task being - // replayed; a slice for the task now being started leaves it unset, because - // the event closing that task does not exist yet. + repeated StreamRecord records = 5; + // The WorkflowTaskCompleted event whose consumed_stream_ranges recorded + // this range. Set only when the server is re-supplying a range for a task + // being replayed; a slice for the task now being started leaves it unset, + // because the event closing that task does not exist yet. // // Replay needs this because a Workflow Task response carries one slice set // while a cache miss replays every prior task, so the ranges have to be @@ -43,9 +60,23 @@ message StreamSlice { // The offsets a Workflow Task consumed, without the payloads. Recorded on // WorkflowTaskCompleted so History grows with Workflow Tasks rather than with -// messages. -message StreamCursor { +// records. +message StreamRange { string stream_id = 1; + // Inclusive. int64 from_offset = 2; + // Exclusive. int64 to_offset = 3; } + +// What a record means to a reader. Kept on the record itself so every store +// and every language reads it the same way without a private envelope. +enum StreamRecordKind { + // Read as DATA. + STREAM_RECORD_KIND_UNSPECIFIED = 0; + // A value the producer published; `body` carries it. + STREAM_RECORD_KIND_DATA = 1; + // The producer named by `producer_id` writes nothing more on `topic`. + // Says nothing about that producer's outcome and does not end the stream. + STREAM_RECORD_KIND_FINISH = 2; +} diff --git a/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto b/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto index 43198097e..8106089e7 100644 --- a/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto +++ b/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto @@ -145,7 +145,7 @@ message WorkflowActivationJob { // with that family and must not be reused. // // A range of a stream the workflow subscribed to. - DeliverStreamMessages deliver_stream_messages = 21; + DeliverStreamRecords deliver_stream_records = 21; // Remove the workflow identified by the [WorkflowActivation] containing this job from the // cache after performing the activation. It is guaranteed that this will be the only job // in the activation if present. @@ -162,14 +162,14 @@ message WorkflowActivationJob { // // An empty range is still delivered: a task where the subscription saw nothing // is a fact replay has to reproduce, not an absence of one. -message DeliverStreamMessages { +message DeliverStreamRecords { // Id of the stream this range came from. string stream_id = 1; // Inclusive. int64 from_offset = 2; // Exclusive. Equal to from_offset when the subscription saw nothing. int64 to_offset = 3; - repeated temporal.api.stream.v1.StreamMessage messages = 4; + repeated temporal.api.stream.v1.StreamRecord records = 4; } // Initialize a new workflow diff --git a/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto b/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto index 14f1ded09..456055d6c 100644 --- a/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto +++ b/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto @@ -59,19 +59,21 @@ message WorkflowCommand { // alongside these and share this message. The two numbers below are fixed // with that family and must not be reused. SubscribeStream subscribe_stream = 29; - AddStreamMessages add_stream_messages = 30; + AppendStreamRecords append_stream_records = 30; } } -// Publish a batch of messages to a stream this workflow owns. +// Append a batch of records to a stream this workflow owns. // // The bodies go to the stream's own log rather than into History, which gets // one fixed-size event naming the offset range. That is what makes the batch -// size free: a thousand messages cost the same in History as one. -message AddStreamMessages { +// size free: a thousand records cost the same in History as one. The server +// stores each record with an empty producer id, because the workflow is the +// producer here. +message AppendStreamRecords { // Empty means the workflow's default output stream. string stream_id = 1; - repeated temporal.api.stream.v1.StreamMessage messages = 2; + repeated temporal.api.stream.v1.StreamRecord records = 2; } // Subscribe this workflow to a stream, so later Workflow Tasks carry the diff --git a/crates/protos/src/protos/mod.rs b/crates/protos/src/protos/mod.rs index c1a78c899..3cf466a9a 100644 --- a/crates/protos/src/protos/mod.rs +++ b/crates/protos/src/protos/mod.rs @@ -1277,10 +1277,10 @@ pub mod coresdk { workflow_activation_job::Variant::ResolveNexusOperation(_) => { write!(f, "ResolveNexusOperation") } - workflow_activation_job::Variant::DeliverStreamMessages(d) => { + workflow_activation_job::Variant::DeliverStreamRecords(d) => { write!( f, - "DeliverStreamMessages({}, {}..{})", + "DeliverStreamRecords({}, {}..{})", d.stream_id, d.from_offset, d.to_offset ) } @@ -1499,13 +1499,13 @@ pub mod coresdk { } } - impl Display for AddStreamMessages { + impl Display for AppendStreamRecords { fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { write!( f, - "AddStreamMessages({}, {} messages)", + "AppendStreamRecords({}, {} records)", self.stream_id, - self.messages.len() + self.records.len() ) } } @@ -1852,8 +1852,8 @@ pub mod temporal { Attributes::SubscribeStreamCommandAttributes(_) => { CommandType::SubscribeStream } - Attributes::AddStreamMessagesCommandAttributes(_) => { - CommandType::AddStreamMessages + Attributes::AppendStreamRecordsCommandAttributes(_) => { + CommandType::AppendStreamRecords } Attributes::CompleteWorkflowExecutionCommandAttributes(_) => { CommandType::CompleteWorkflowExecution @@ -1908,12 +1908,12 @@ pub mod temporal { } } - impl From for command::Attributes { - fn from(s: workflow_commands::AddStreamMessages) -> Self { - Self::AddStreamMessagesCommandAttributes( - AddStreamMessagesCommandAttributes { + impl From for command::Attributes { + fn from(s: workflow_commands::AppendStreamRecords) -> Self { + Self::AppendStreamRecordsCommandAttributes( + AppendStreamRecordsCommandAttributes { stream_id: s.stream_id, - messages: s.messages, + records: s.records, }, ) } @@ -2390,7 +2390,7 @@ pub mod temporal { | EventType::UpsertWorkflowSearchAttributes | EventType::WorkflowPropertiesModified | EventType::WorkflowStreamSubscribed - | EventType::WorkflowStreamMessagesAdded + | EventType::WorkflowStreamRecordsAppended | EventType::NexusOperationScheduled | EventType::NexusOperationCancelRequested | EventType::WorkflowExecutionCanceled @@ -2491,7 +2491,7 @@ pub mod temporal { if let Some(a) = self.attributes.as_ref() { match a { Attributes::WorkflowStreamSubscribedEventAttributes(_) => false, - Attributes::WorkflowStreamMessagesAddedEventAttributes(_) => { + Attributes::WorkflowStreamRecordsAppendedEventAttributes(_) => { false } Attributes::WorkflowExecutionStartedEventAttributes(_) => false, @@ -2582,7 +2582,7 @@ pub mod temporal { // I just absolutely _love_ this match self { Attributes::WorkflowStreamSubscribedEventAttributes(_) => { EventType::WorkflowStreamSubscribed } - Attributes::WorkflowStreamMessagesAddedEventAttributes(_) => { EventType::WorkflowStreamMessagesAdded } + Attributes::WorkflowStreamRecordsAppendedEventAttributes(_) => { EventType::WorkflowStreamRecordsAppended } Attributes::WorkflowExecutionStartedEventAttributes(_) => { EventType::WorkflowExecutionStarted } Attributes::WorkflowExecutionCompletedEventAttributes(_) => { EventType::WorkflowExecutionCompleted } Attributes::WorkflowExecutionFailedEventAttributes(_) => { EventType::WorkflowExecutionFailed } diff --git a/crates/sdk-core/CHANGELOG.md b/crates/sdk-core/CHANGELOG.md index 2dabc4c93..acfc46bce 100644 --- a/crates/sdk-core/CHANGELOG.md +++ b/crates/sdk-core/CHANGELOG.md @@ -50,9 +50,9 @@ relevant information. metrics now carry a `failure_reason` attribute. Each is now split into one time series per reason, which may affect existing dashboards. * Workflow task completions larger than the gRPC request size limit are now paginated automatically when the namespace supports it. Paginated workflow task completions require Temporal Server 1.32.0 or later. -* Workflows can subscribe to server-side streams and publish batches of messages to them with the - `SubscribeStream` and `AddStreamMessages` commands. Consumed ranges reach the workflow as - `DeliverStreamMessages` activation jobs, and replay hands each recorded range back in the +* Workflows can subscribe to server-side streams and append batches of records to them with the + `SubscribeStream` and `AppendStreamRecords` commands. Consumed ranges reach the workflow as + `DeliverStreamRecords` activation jobs, and replay hands each recorded range back in the activation of the task that consumed it. ### Breaking Changes :boom: diff --git a/crates/sdk-core/src/core_tests/streams.rs b/crates/sdk-core/src/core_tests/streams.rs index 807c001a4..02a9fdd75 100644 --- a/crates/sdk-core/src/core_tests/streams.rs +++ b/crates/sdk-core/src/core_tests/streams.rs @@ -22,22 +22,22 @@ use std::sync::{ use temporalio_common::protos::{ coresdk::{ workflow_activation::{WorkflowActivation, WorkflowActivationJob, workflow_activation_job}, - workflow_commands::{AddStreamMessages, SubscribeStream}, + workflow_commands::{AppendStreamRecords, SubscribeStream}, workflow_completion::WorkflowActivationCompletion, }, temporal::api::{ command::v1::command, enums::v1::{CommandType, EventType, WorkflowTaskFailedCause}, history::v1::{History, HistoryEvent}, - stream::v1::{StreamCursor, StreamMessage}, + stream::v1::{StreamRange, StreamRecord}, workflowservice::v1::{ GetWorkflowExecutionHistoryResponse, RespondWorkflowTaskCompletedResponse, }, }, }; -fn cursor(stream_id: &str, from: i64, to: i64) -> StreamCursor { - StreamCursor { +fn cursor(stream_id: &str, from: i64, to: i64) -> StreamRange { + StreamRange { stream_id: stream_id.to_string(), from_offset: from, to_offset: to, @@ -46,11 +46,11 @@ fn cursor(stream_id: &str, from: i64, to: i64) -> StreamCursor { fn delivered(job: &WorkflowActivationJob) -> (&str, i64, i64, Vec<&[u8]>) { match job.variant.as_ref().unwrap() { - workflow_activation_job::Variant::DeliverStreamMessages(d) => ( + workflow_activation_job::Variant::DeliverStreamRecords(d) => ( d.stream_id.as_str(), d.from_offset, d.to_offset, - d.messages + d.records .iter() .map(|m| m.body.as_ref().unwrap().data.as_slice()) .collect(), @@ -66,7 +66,7 @@ fn delivered_ranges(task: &WorkflowActivation) -> Vec<(String, i64, i64)> { .filter(|j| { matches!( j.variant, - Some(workflow_activation_job::Variant::DeliverStreamMessages(_)) + Some(workflow_activation_job::Variant::DeliverStreamRecords(_)) ) }) .map(|j| { @@ -135,7 +135,7 @@ async fn delivers_the_range_for_the_current_task() { .filter(|j| { matches!( j.variant, - Some(workflow_activation_job::Variant::DeliverStreamMessages(_)) + Some(workflow_activation_job::Variant::DeliverStreamRecords(_)) ) }) .collect(); @@ -176,7 +176,7 @@ async fn delivers_an_empty_range() { .find(|j| { matches!( j.variant, - Some(workflow_activation_job::Variant::DeliverStreamMessages(_)) + Some(workflow_activation_job::Variant::DeliverStreamRecords(_)) ) }) .expect("an empty range is still delivered"); @@ -196,10 +196,10 @@ async fn replays_recorded_ranges_in_order_before_the_current_one() { t.add_by_type(EventType::WorkflowExecutionStarted); t.add_workflow_task_scheduled_and_started(); let first_completed = - t.add_workflow_task_completed_with_stream_cursors(vec![cursor("s1", 0, 2)]); + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("s1", 0, 2)]); t.add_workflow_task_scheduled_and_started(); let second_completed = - t.add_workflow_task_completed_with_stream_cursors(vec![cursor("s1", 2, 3)]); + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("s1", 2, 3)]); t.add_workflow_task_scheduled_and_started(); let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); @@ -228,7 +228,7 @@ async fn replays_recorded_ranges_in_order_before_the_current_one() { for job in &task.jobs { if matches!( job.variant, - Some(workflow_activation_job::Variant::DeliverStreamMessages(_)) + Some(workflow_activation_job::Variant::DeliverStreamRecords(_)) ) { let (_, from, to, bodies) = delivered(job); seen.push(( @@ -316,14 +316,14 @@ async fn publish_command_reaches_the_server_with_its_payloads() { .times(1) .returning(|resp, _| { let cmd = resp.commands.first().expect("a command was sent"); - assert_eq!(cmd.command_type(), CommandType::AddStreamMessages); + assert_eq!(cmd.command_type(), CommandType::AppendStreamRecords); match cmd.attributes.as_ref().unwrap() { - command::Attributes::AddStreamMessagesCommandAttributes(a) => { + command::Attributes::AppendStreamRecordsCommandAttributes(a) => { assert_eq!(a.stream_id, "s1"); // The bodies are the half of the batch History never sees, // so the command is the only thing that can carry them. let bodies: Vec<_> = a - .messages + .records .iter() .map(|m| m.body.as_ref().unwrap().data.clone()) .collect(); @@ -358,7 +358,7 @@ async fn publish_command_round_trips_through_replay() { let mut t = TestHistoryBuilder::default(); t.add_by_type(EventType::WorkflowExecutionStarted); t.add_full_wf_task(); - t.add_stream_messages_added("s1", 0, 2); + t.add_stream_records_appended("s1", 0, 2); t.add_full_wf_task(); let mut mock_client = mock_worker_client(); @@ -431,15 +431,15 @@ async fn subscribe_command_reaches_the_server() { core.shutdown().await; } -fn publish_two(stream_id: &str) -> AddStreamMessages { - AddStreamMessages { +fn publish_two(stream_id: &str) -> AppendStreamRecords { + AppendStreamRecords { stream_id: stream_id.to_string(), - messages: vec![ - StreamMessage { + records: vec![ + StreamRecord { body: Some(b"one".to_vec().into()), ..Default::default() }, - StreamMessage { + StreamRecord { body: Some(b"two".to_vec().into()), ..Default::default() }, @@ -466,8 +466,8 @@ async fn read_then_publish_replays_when_the_range_is_only_visible_by_lookahead() t.add_workflow_task_scheduled_and_started(); // The task that reads [0,1) and publishes because of it. let read_completed = - t.add_workflow_task_completed_with_stream_cursors(vec![cursor("in", 0, 1)]); - t.add_stream_messages_added("out", 0, 1); + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("in", 0, 1)]); + t.add_stream_records_appended("out", 0, 1); t.add_workflow_task_scheduled_and_started(); let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); @@ -492,7 +492,7 @@ async fn read_then_publish_replays_when_the_range_is_only_visible_by_lookahead() let got_input = task.jobs.iter().any(|j| { matches!( j.variant, - Some(workflow_activation_job::Variant::DeliverStreamMessages(_)) + Some(workflow_activation_job::Variant::DeliverStreamRecords(_)) ) }); assert!( @@ -505,9 +505,9 @@ async fn read_then_publish_replays_when_the_range_is_only_visible_by_lookahead() core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( task.run_id, vec![ - AddStreamMessages { + AppendStreamRecords { stream_id: "out".to_string(), - messages: vec![StreamMessage { + records: vec![StreamRecord { body: Some(b"accept".to_vec().into()), ..Default::default() }], @@ -538,8 +538,8 @@ async fn read_then_publish_replays_after_a_task_that_consumed_nothing() { t.add_workflow_task_scheduled_and_started(); // Task 2 reads [0,3) and publishes because of it. let read_completed = - t.add_workflow_task_completed_with_stream_cursors(vec![cursor("in", 0, 3)]); - t.add_stream_messages_added("out", 0, 1); + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("in", 0, 3)]); + t.add_stream_records_appended("out", 0, 1); t.add_workflow_task_scheduled_and_started(); let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); @@ -582,7 +582,7 @@ async fn read_then_publish_replays_after_a_task_that_consumed_nothing() { .filter(|j| { matches!( j.variant, - Some(workflow_activation_job::Variant::DeliverStreamMessages(_)) + Some(workflow_activation_job::Variant::DeliverStreamRecords(_)) ) }) .flat_map(|j| delivered(j).3.into_iter().map(|b| b.to_vec())) @@ -596,9 +596,9 @@ async fn read_then_publish_replays_after_a_task_that_consumed_nothing() { core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( task.run_id, vec![ - AddStreamMessages { + AppendStreamRecords { stream_id: "out".to_string(), - messages: vec![StreamMessage { + records: vec![StreamRecord { body: Some(b"accept".to_vec().into()), ..Default::default() }], @@ -637,8 +637,8 @@ async fn read_then_publish_replays_across_a_page_boundary( t.add_workflow_task_scheduled_and_started(); // 12 // Task 3 reads [0,1) and publishes because of it. let read_completed = - t.add_workflow_task_completed_with_stream_cursors(vec![cursor("in", 0, 1)]); - t.add_stream_messages_added("out", 0, 1); + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("in", 0, 1)]); + t.add_stream_records_appended("out", 0, 1); t.add_workflow_task_scheduled_and_started(); // 16 let events = t.get_full_history_info().unwrap().into_events(); @@ -710,9 +710,9 @@ async fn read_then_publish_replays_across_a_page_boundary( core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( task.run_id, vec![ - AddStreamMessages { + AppendStreamRecords { stream_id: "out".to_string(), - messages: vec![StreamMessage { + records: vec![StreamRecord { body: Some(b"accept".to_vec().into()), ..Default::default() }], @@ -742,7 +742,7 @@ async fn a_publish_reissued_to_a_different_stream_fails_the_task() { let mut t = TestHistoryBuilder::default(); t.add_by_type(EventType::WorkflowExecutionStarted); t.add_full_wf_task(); - t.add_stream_messages_added("s1", 0, 2); + t.add_stream_records_appended("s1", 0, 2); t.add_full_wf_task(); let (core, failures) = worker_expecting_one_nondeterminism_failure(t, ResponseType::AllHistory); @@ -759,14 +759,14 @@ async fn a_publish_reissued_to_a_different_stream_fails_the_task() { core.shutdown().await; } -/// The batch size is part of the record too: the event names how many messages +/// The batch size is part of the record too: the event names how many records /// landed, so a replay that publishes fewer has diverged from the original run. #[tokio::test] async fn a_publish_reissued_with_a_different_batch_size_fails_the_task() { let mut t = TestHistoryBuilder::default(); t.add_by_type(EventType::WorkflowExecutionStarted); t.add_full_wf_task(); - t.add_stream_messages_added("s1", 0, 2); + t.add_stream_records_appended("s1", 0, 2); t.add_full_wf_task(); let (core, failures) = worker_expecting_one_nondeterminism_failure(t, ResponseType::AllHistory); @@ -775,9 +775,9 @@ async fn a_publish_reissued_with_a_different_batch_size_fails_the_task() { core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( task.run_id, vec![ - AddStreamMessages { + AppendStreamRecords { stream_id: "s1".to_string(), - messages: vec![StreamMessage { + records: vec![StreamRecord { body: Some(b"one".to_vec().into()), ..Default::default() }], @@ -832,11 +832,12 @@ async fn data_only_tasks_replay_one_range_per_activation() { let mut t = TestHistoryBuilder::default(); t.add_by_type(EventType::WorkflowExecutionStarted); t.add_workflow_task_scheduled_and_started(); - let first = t.add_workflow_task_completed_with_stream_cursors(vec![cursor("s1", 0, 1)]); + let first = t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("s1", 0, 1)]); t.add_workflow_task_scheduled_and_started(); - let second = t.add_workflow_task_completed_with_stream_cursors(vec![cursor("s1", 1, 2)]); + let second = + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("s1", 1, 2)]); t.add_workflow_task_scheduled_and_started(); - let third = t.add_workflow_task_completed_with_stream_cursors(vec![cursor("s1", 2, 3)]); + let third = t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("s1", 2, 3)]); t.add_workflow_task_scheduled_and_started(); let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); @@ -883,7 +884,7 @@ async fn ranges_for_several_streams_arrive_in_stream_order() { let mut t = TestHistoryBuilder::default(); t.add_by_type(EventType::WorkflowExecutionStarted); t.add_workflow_task_scheduled_and_started(); - let completed = t.add_workflow_task_completed_with_stream_cursors(vec![ + let completed = t.add_workflow_task_completed_with_consumed_stream_ranges(vec![ cursor("s2", 0, 1), cursor("s1", 0, 1), ]); @@ -933,7 +934,7 @@ async fn an_empty_recorded_range_replays_without_a_slice_from_the_server() { let mut t = TestHistoryBuilder::default(); t.add_by_type(EventType::WorkflowExecutionStarted); t.add_workflow_task_scheduled_and_started(); - t.add_workflow_task_completed_with_stream_cursors(vec![cursor("s1", 4, 4)]); + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("s1", 4, 4)]); t.add_workflow_task_scheduled_and_started(); let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); @@ -969,11 +970,12 @@ async fn a_resupplied_slice_that_disagrees_with_the_record_fails_the_task() { let mut t = TestHistoryBuilder::default(); t.add_by_type(EventType::WorkflowExecutionStarted); t.add_workflow_task_scheduled_and_started(); - let completed = t.add_workflow_task_completed_with_stream_cursors(vec![cursor("s1", 0, 2)]); + let completed = + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("s1", 0, 2)]); t.add_workflow_task_scheduled_and_started(); let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); - // One message where the record says two. + // One record where the event says two. poll_resp.add_stream_slice("s1", completed, 0, &["a"]); let (core, failures) = @@ -996,7 +998,8 @@ async fn a_cached_run_is_not_handed_its_own_range_again() { let mut t = TestHistoryBuilder::default(); t.add_by_type(EventType::WorkflowExecutionStarted); t.add_workflow_task_scheduled_and_started(); - let completed = t.add_workflow_task_completed_with_stream_cursors(vec![cursor("s1", 0, 2)]); + let completed = + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("s1", 0, 2)]); t.add_workflow_task_scheduled_and_started(); let mut first_poll = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::ToTaskNum(1)); @@ -1046,7 +1049,7 @@ async fn a_missing_resupply_for_a_consumed_range_fails_the_task() { let mut t = TestHistoryBuilder::default(); t.add_by_type(EventType::WorkflowExecutionStarted); t.add_workflow_task_scheduled_and_started(); - t.add_workflow_task_completed_with_stream_cursors(vec![cursor("s1", 0, 2)]); + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("s1", 0, 2)]); t.add_workflow_task_scheduled_and_started(); let (core, failures) = worker_expecting_one_nondeterminism_failure(t, ResponseType::AllHistory); diff --git a/crates/sdk-core/src/replay/history_builder.rs b/crates/sdk-core/src/replay/history_builder.rs index d7ba35e0d..43723d243 100644 --- a/crates/sdk-core/src/replay/history_builder.rs +++ b/crates/sdk-core/src/replay/history_builder.rs @@ -23,7 +23,7 @@ use temporalio_common::protos::{ enums::v1::{EventType, TaskQueueKind, WorkflowTaskFailedCause}, failure::v1::{CanceledFailureInfo, Failure, failure}, history::v1::{history_event::Attributes, *}, - stream::v1::StreamCursor, + stream::v1::StreamRange, taskqueue::v1::TaskQueue, update, update::v1::outcome, @@ -133,13 +133,13 @@ impl TestHistoryBuilder { /// Add a workflow task completed event recording the stream offsets that /// task consumed. Only the range is in History; the payloads come back from /// the server on the poll response. - pub fn add_workflow_task_completed_with_stream_cursors( + pub fn add_workflow_task_completed_with_consumed_stream_ranges( &mut self, - cursors: Vec, + cursors: Vec, ) -> i64 { let id = self.add(WorkflowTaskCompletedEventAttributes { scheduled_event_id: self.workflow_task_scheduled_event_id, - stream_cursors: cursors, + consumed_stream_ranges: cursors, ..Default::default() }); self.previous_task_completed_id = id; @@ -156,18 +156,18 @@ impl TestHistoryBuilder { self.add(attrs) } - /// Add the event an add-stream-messages command produces. - pub fn add_stream_messages_added( + /// Add the event an append-stream-records command produces. + pub fn add_stream_records_appended( &mut self, stream_id: &str, first_offset: i64, - message_count: i64, + record_count: i64, ) -> i64 { - let attrs = WorkflowStreamMessagesAddedEventAttributes { + let attrs = WorkflowStreamRecordsAppendedEventAttributes { workflow_task_completed_event_id: self.previous_task_completed_id, stream_id: stream_id.to_string(), first_offset, - message_count, + record_count, }; self.add(attrs) } diff --git a/crates/sdk-core/src/test_help/integ_helpers.rs b/crates/sdk-core/src/test_help/integ_helpers.rs index 979ea6d9b..cdfcc13da 100644 --- a/crates/sdk-core/src/test_help/integ_helpers.rs +++ b/crates/sdk-core/src/test_help/integ_helpers.rs @@ -54,7 +54,7 @@ use temporalio_common::{ enums::v1::WorkflowTaskFailedCause, failure::v1::Failure, protocol::{self, v1::message}, - stream::v1::{StreamMessage, StreamSlice}, + stream::v1::{StreamRecord, StreamSlice}, update, workflowservice::v1::{ DescribeNamespaceResponse, PollActivityTaskQueueResponse, @@ -971,9 +971,9 @@ impl PollWFTRespExt for PollWorkflowTaskQueueResponse { stream_id: stream_id.to_string(), from_offset, to_offset: from_offset + bodies.len() as i64, - messages: bodies + records: bodies .iter() - .map(|b| StreamMessage { + .map(|b| StreamRecord { body: Some(Payload { data: b.as_bytes().to_vec(), ..Default::default() diff --git a/crates/sdk-core/src/worker/workflow/history_update.rs b/crates/sdk-core/src/worker/workflow/history_update.rs index a923299ec..4100b6e2e 100644 --- a/crates/sdk-core/src/worker/workflow/history_update.rs +++ b/crates/sdk-core/src/worker/workflow/history_update.rs @@ -769,7 +769,7 @@ fn find_end_index_of_next_wft_seq( // fold it into a heartbeat chain and hand several ranges over at once. if let Some(Attributes::WorkflowTaskCompletedEventAttributes(ref attrs)) = next_event.attributes - && !attrs.stream_cursors.is_empty() + && !attrs.consumed_stream_ranges.is_empty() { saw_command = true; saw_command_or_started = true; diff --git a/crates/sdk-core/src/worker/workflow/machines/add_stream_messages_state_machine.rs b/crates/sdk-core/src/worker/workflow/machines/append_stream_records_state_machine.rs similarity index 59% rename from crates/sdk-core/src/worker/workflow/machines/add_stream_messages_state_machine.rs rename to crates/sdk-core/src/worker/workflow/machines/append_stream_records_state_machine.rs index 4a3e137c1..5acdc7eda 100644 --- a/crates/sdk-core/src/worker/workflow/machines/add_stream_messages_state_machine.rs +++ b/crates/sdk-core/src/worker/workflow/machines/append_stream_records_state_machine.rs @@ -7,21 +7,21 @@ use crate::worker::workflow::{ nondeterminism, }; use temporalio_common::protos::{ - coresdk::workflow_commands::AddStreamMessages, + coresdk::workflow_commands::AppendStreamRecords, temporal::api::{ enums::v1::{CommandType, EventType}, - history::v1::{WorkflowStreamMessagesAddedEventAttributes, history_event}, + history::v1::{WorkflowStreamRecordsAppendedEventAttributes, history_event}, }, }; fsm! { - pub(super) name AddStreamMessagesMachine; - command AddStreamMessagesMachineCommand; + pub(super) name AppendStreamRecordsMachine; + command AppendStreamRecordsMachineCommand; error WFMachinesError; shared_state SharedState; Created --(CommandScheduled) --> CommandIssued; - CommandIssued --(CommandRecorded(WorkflowStreamMessagesAddedEventAttributes), + CommandIssued --(CommandRecorded(WorkflowStreamRecordsAppendedEventAttributes), shared on_command_recorded) --> Done; } @@ -29,20 +29,20 @@ fsm! { #[derive(Default, Clone)] pub(super) struct SharedState { stream_id: String, - message_count: i64, + record_count: i64, } -/// Publish a batch of messages to a stream this workflow owns. +/// Append a batch of records to a stream this workflow owns. /// /// The bodies go to the stream's own log, and History gets one event naming the /// offset range the batch landed at. The offsets are assigned by the server, so /// nothing here predicts them. -pub(super) fn add_stream_messages(lang_cmd: AddStreamMessages) -> NewMachineWithCommand { - let sm = AddStreamMessagesMachine::from_parts( +pub(super) fn append_stream_records(lang_cmd: AppendStreamRecords) -> NewMachineWithCommand { + let sm = AppendStreamRecordsMachine::from_parts( Created {}.into(), SharedState { stream_id: lang_cmd.stream_id.clone(), - message_count: lang_cmd.messages.len() as i64, + record_count: lang_cmd.records.len() as i64, }, ); NewMachineWithCommand { @@ -52,7 +52,7 @@ pub(super) fn add_stream_messages(lang_cmd: AddStreamMessages) -> NewMachineWith } #[derive(Debug, derive_more::Display)] -pub(super) enum AddStreamMessagesMachineCommand {} +pub(super) enum AppendStreamRecordsMachineCommand {} #[derive(Debug, Default, Clone, derive_more::Display)] pub(super) struct Created {} @@ -64,20 +64,20 @@ impl CommandIssued { pub(super) fn on_command_recorded( self, dat: &mut SharedState, - attrs: WorkflowStreamMessagesAddedEventAttributes, - ) -> AddStreamMessagesMachineTransition { + attrs: WorkflowStreamRecordsAppendedEventAttributes, + ) -> AppendStreamRecordsMachineTransition { // An empty id names the workflow's default stream, and the server is the // one that resolves that name, so only a named stream can be compared. let same_stream = dat.stream_id.is_empty() || dat.stream_id == attrs.stream_id; - if same_stream && dat.message_count == attrs.message_count { + if same_stream && dat.record_count == attrs.record_count { TransitionResult::default() } else { TransitionResult::Err(nondeterminism!( - "Recorded publish of {} messages to stream {:?} does not match the reissued \ - publish of {} messages to stream {:?}", - attrs.message_count, + "Recorded append of {} records to stream {:?} does not match the reissued \ + append of {} records to stream {:?}", + attrs.record_count, attrs.stream_id, - dat.message_count, + dat.record_count, dat.stream_id )) } @@ -87,49 +87,51 @@ impl CommandIssued { #[derive(Debug, Default, Clone, derive_more::Display)] pub(super) struct Done {} -impl WFMachinesAdapter for AddStreamMessagesMachine { +impl WFMachinesAdapter for AppendStreamRecordsMachine { fn adapt_response( &self, _my_command: Self::Command, _event_info: Option, ) -> Result, Self::Error> { Err(Self::Error::Nondeterminism( - "AddStreamMessages does not use state machine commands".to_string(), + "AppendStreamRecords does not use state machine commands".to_string(), )) } } -impl TryFrom for AddStreamMessagesMachineEvents { +impl TryFrom for AppendStreamRecordsMachineEvents { type Error = WFMachinesError; fn try_from(e: HistEventData) -> Result { let e = e.event; match e.event_type() { - EventType::WorkflowStreamMessagesAdded => { + EventType::WorkflowStreamRecordsAppended => { if let Some( - history_event::Attributes::WorkflowStreamMessagesAddedEventAttributes(attrs), + history_event::Attributes::WorkflowStreamRecordsAppendedEventAttributes(attrs), ) = e.attributes { - Ok(AddStreamMessagesMachineEvents::CommandRecorded(attrs)) + Ok(AppendStreamRecordsMachineEvents::CommandRecorded(attrs)) } else { - Err(fatal!("Stream messages added attributes were unset: {e}")) + Err(fatal!("Stream records appended attributes were unset: {e}")) } } _ => Err(Self::Error::Nondeterminism(format!( - "AddStreamMessagesMachine does not handle {e}" + "AppendStreamRecordsMachine does not handle {e}" ))), } } } -impl TryFrom for AddStreamMessagesMachineEvents { +impl TryFrom for AppendStreamRecordsMachineEvents { type Error = WFMachinesError; fn try_from(c: CommandType) -> Result { match c { - CommandType::AddStreamMessages => Ok(AddStreamMessagesMachineEvents::CommandScheduled), + CommandType::AppendStreamRecords => { + Ok(AppendStreamRecordsMachineEvents::CommandScheduled) + } _ => Err(Self::Error::Nondeterminism(format!( - "AddStreamMessagesMachine does not handle command type {c:?}" + "AppendStreamRecordsMachine does not handle command type {c:?}" ))), } } diff --git a/crates/sdk-core/src/worker/workflow/machines/mod.rs b/crates/sdk-core/src/worker/workflow/machines/mod.rs index b826e99d8..364897465 100644 --- a/crates/sdk-core/src/worker/workflow/machines/mod.rs +++ b/crates/sdk-core/src/worker/workflow/machines/mod.rs @@ -1,4 +1,4 @@ -mod add_stream_messages_state_machine; +mod append_stream_records_state_machine; mod workflow_machines; mod activity_state_machine; @@ -33,7 +33,7 @@ use crate::{ worker::workflow::{WFMachinesError, fatal, nondeterminism}, }; use activity_state_machine::ActivityMachine; -use add_stream_messages_state_machine::AddStreamMessagesMachine; +use append_stream_records_state_machine::AppendStreamRecordsMachine; use cancel_external_state_machine::CancelExternalMachine; use cancel_workflow_state_machine::CancelWorkflowMachine; use child_workflow_state_machine::ChildWorkflowMachine; @@ -85,7 +85,7 @@ enum Machines { UpsertSearchAttributesMachine, ModifyWorkflowPropertiesMachine, SubscribeStreamMachine, - AddStreamMessagesMachine, + AppendStreamRecordsMachine, UpdateMachine, NexusOperationMachine, } diff --git a/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs b/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs index 5a1df56e7..41c054450 100644 --- a/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs +++ b/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs @@ -66,7 +66,7 @@ mod machine_coverage_report { use super::*; use crate::worker::workflow::machines::{ StateMachine, activity_state_machine::ActivityMachine, - add_stream_messages_state_machine::AddStreamMessagesMachine, + append_stream_records_state_machine::AppendStreamRecordsMachine, cancel_external_state_machine::CancelExternalMachine, cancel_workflow_state_machine::CancelWorkflowMachine, child_workflow_state_machine::ChildWorkflowMachine, @@ -120,7 +120,7 @@ mod machine_coverage_report { let mut update = UpdateMachine::visualizer().to_owned(); let mut nexus = NexusOperationMachine::visualizer().to_owned(); let mut subscribe_stream = SubscribeStreamMachine::visualizer().to_owned(); - let mut add_stream_messages = AddStreamMessagesMachine::visualizer().to_owned(); + let mut append_stream_records = AppendStreamRecordsMachine::visualizer().to_owned(); // This isn't at all efficient but doesn't need to be. // Replace transitions in the vizzes with green color if they are covered. @@ -151,8 +151,8 @@ mod machine_coverage_report { m @ "SubscribeStreamMachine" => { cover_transitions(m, &mut subscribe_stream, coverage) } - m @ "AddStreamMessagesMachine" => { - cover_transitions(m, &mut add_stream_messages, coverage) + m @ "AppendStreamRecordsMachine" => { + cover_transitions(m, &mut append_stream_records, coverage) } m => panic!("Unknown machine {m}"), } diff --git a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs index 4bd17933b..ebf81bd48 100644 --- a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs +++ b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs @@ -2,7 +2,7 @@ mod local_acts; use super::{ Machines, NewMachineWithCommand, TemporalStateMachine, - add_stream_messages_state_machine::add_stream_messages, + append_stream_records_state_machine::append_stream_records, cancel_external_state_machine::new_external_cancel, cancel_workflow_state_machine::cancel_workflow, complete_workflow_state_machine::complete_workflow, @@ -72,7 +72,7 @@ use temporalio_common::{ history::v1::{HistoryEvent, history_event}, protocol::v1::{Message as ProtocolMessage, message::SequencingId}, sdk::v1::WorkflowTaskCompletedMetadata, - stream::v1::{StreamCursor, StreamSlice}, + stream::v1::{StreamRange, StreamSlice}, }, }, worker::WorkerDeploymentVersion, @@ -106,7 +106,7 @@ pub(crate) struct WorkflowMachines { stream_slices_lookahead_through: i64, /// Workflow Task Started id of the last task this instance ran live, or 0. /// - /// A task that ran here was handed its stream messages as it ran, and the + /// A task that ran here was handed its stream records as it ran, and the /// completion event recording what it consumed arrives in the next task's /// history. That event needs no bytes from the server, and `replaying` does /// not say so: it is true for a cache hit as well, because the history @@ -667,12 +667,12 @@ impl WorkflowMachines { } }}; } - let mut replayed_slice_events: Vec<(i64, Vec)> = vec![]; + let mut replayed_slice_events: Vec<(i64, Vec)> = vec![]; // Kept apart from the in-batch ones. An in-batch event whose bytes are // missing is a real inconsistency; a looked-ahead one may simply not // have been sent yet, and demanding it would turn an early delivery // into a new way to fail. - let mut lookahead_slice_event: Option<(i64, Vec)> = None; + let mut lookahead_slice_event: Option<(i64, Vec)> = None; let mut peeked_events = events.iter().peekable(); while let Some(event) = peeked_events.next() { if let Some(history_event::Attributes::WorkflowTaskCompletedEventAttributes(ref wtc)) = @@ -681,8 +681,9 @@ impl WorkflowMachines { apply_wft_complete_data!(self, wtc); // The event records which offsets that task consumed; the server // sends the bytes back separately, keyed by this event. - if !wtc.stream_cursors.is_empty() { - replayed_slice_events.push((event.event_id, wtc.stream_cursors.clone())); + if !wtc.consumed_stream_ranges.is_empty() { + replayed_slice_events + .push((event.event_id, wtc.consumed_stream_ranges.clone())); } } if peeked_events.peek().is_none() @@ -697,8 +698,8 @@ impl WorkflowMachines { // its input one activation after the commands that input // caused, so a read-then-publish task replays with nothing to // decide from and reissues no command. Look ahead for it here. - if !wtc.stream_cursors.is_empty() { - lookahead_slice_event = Some((wtc_id, wtc.stream_cursors.clone())); + if !wtc.consumed_stream_ranges.is_empty() { + lookahead_slice_event = Some((wtc_id, wtc.consumed_stream_ranges.clone())); } } } @@ -928,7 +929,7 @@ impl WorkflowMachines { // completion, so the range was due one activation before this. return Err(nondeterminism!( "Event {event_id} records that stream {} was consumed from offset {} to {}, \ - but the server sent no messages for it. The workflow was owed that range \ + but the server sent no records for it. The workflow was owed that range \ one activation earlier.", cursor.stream_id, cursor.from_offset, @@ -961,7 +962,7 @@ impl WorkflowMachines { // once we have caught up to it. if !self.replaying { for slice in std::mem::take(&mut self.current_stream_slices) { - self.drive_me.send_job(deliver_stream_messages_job(slice)); + self.drive_me.send_job(deliver_stream_records_job(slice)); } // This task is running here, so whatever it consumes is already in // hand and its completion event will not need re-supplying. @@ -1670,12 +1671,12 @@ impl WorkflowMachines { CommandIdKind::NeverResolves, ); } - WFCommandVariant::AddStreamMessages(attrs) => { + WFCommandVariant::AppendStreamRecords(attrs) => { // Never resolves: the event names the offset range the // server assigned and hands nothing back. A workflow that // wants to know where its batch landed reads the stream. self.add_cmd_to_wf_task( - add_stream_messages(attrs), + append_stream_records(attrs), annotations, CommandIdKind::NeverResolves, ); @@ -1983,12 +1984,12 @@ enum CommandIdKind { } /// Turn a slice the server supplied into the job lang sees. -fn deliver_stream_messages_job(slice: StreamSlice) -> OutgoingJob { - workflow_activation::DeliverStreamMessages { +fn deliver_stream_records_job(slice: StreamSlice) -> OutgoingJob { + workflow_activation::DeliverStreamRecords { stream_id: slice.stream_id, from_offset: slice.from_offset, to_offset: slice.to_offset, - messages: slice.messages, + records: slice.records, } .into() } @@ -2013,7 +2014,7 @@ fn stream_order( /// else would replay the task with different input than it ran on. fn resupplied_deliveries( event_id: i64, - mut cursors: Vec, + mut cursors: Vec, mut slices: Vec, ) -> Result> { cursors.sort_by(|a, b| stream_order(&a.stream_id, a.from_offset, &b.stream_id, b.from_offset)); @@ -2057,7 +2058,7 @@ fn resupplied_deliveries( )); } }; - jobs.push(deliver_stream_messages_job(slice)); + jobs.push(deliver_stream_records_job(slice)); } if let Some(extra) = slices.first() { return Err(nondeterminism!( diff --git a/crates/sdk-core/src/worker/workflow/mod.rs b/crates/sdk-core/src/worker/workflow/mod.rs index d37315ff5..2e4f4e596 100644 --- a/crates/sdk-core/src/worker/workflow/mod.rs +++ b/crates/sdk-core/src/worker/workflow/mod.rs @@ -1530,7 +1530,7 @@ enum WFCommandVariant { ScheduleNexusOperation(ScheduleNexusOperation), RequestCancelNexusOperation(RequestCancelNexusOperation), SubscribeStream(SubscribeStream), - AddStreamMessages(AddStreamMessages), + AppendStreamRecords(AppendStreamRecords), } impl TryFrom for WFCommand { @@ -1540,8 +1540,8 @@ impl TryFrom for WFCommand { let variant = match c.variant.ok_or(EmptyWorkflowCommandErr)? { workflow_command::Variant::StartTimer(s) => WFCommandVariant::AddTimer(s), workflow_command::Variant::SubscribeStream(s) => WFCommandVariant::SubscribeStream(s), - workflow_command::Variant::AddStreamMessages(s) => { - WFCommandVariant::AddStreamMessages(s) + workflow_command::Variant::AppendStreamRecords(s) => { + WFCommandVariant::AppendStreamRecords(s) } workflow_command::Variant::CancelTimer(s) => WFCommandVariant::CancelTimer(s), workflow_command::Variant::ScheduleActivity(s) => WFCommandVariant::AddActivity(s), diff --git a/crates/sdk/src/workflow_future.rs b/crates/sdk/src/workflow_future.rs index 9edf41bb0..6e7f421f0 100644 --- a/crates/sdk/src/workflow_future.rs +++ b/crates/sdk/src/workflow_future.rs @@ -340,12 +340,12 @@ impl WorkflowFuture { .context("Nexus operation must have result")?; push_polled_context!(ActivationJobContext::Passive); } - Variant::DeliverStreamMessages(slice) => { + Variant::DeliverStreamRecords(slice) => { // No stream API in this SDK. Bailing rather than ignoring: // the server has recorded this range as consumed and will // not send it again, so dropping it loses data silently. bail!( - "received stream messages for {}, which this SDK cannot deliver", + "received stream records for {}, which this SDK cannot deliver", slice.stream_id ); } diff --git a/crates/workflow/src/runtime/instance.rs b/crates/workflow/src/runtime/instance.rs index eb4e3e251..8af164913 100644 --- a/crates/workflow/src/runtime/instance.rs +++ b/crates/workflow/src/runtime/instance.rs @@ -1093,7 +1093,7 @@ where self.apply_resolution(resolution); ActivationJobResult::None } - Some(ActivationVariant::DeliverStreamMessages(slice)) => { + Some(ActivationVariant::DeliverStreamRecords(slice)) => { // The Rust workflow runtime has no stream API yet. Failing // is the only safe answer: the server has already recorded // this range as consumed, so dropping it would leave the @@ -1101,7 +1101,7 @@ where // again. return Err(Box::new(Failure { message: format!( - "received stream messages for {}, which this SDK cannot deliver", + "received stream records for {}, which this SDK cannot deliver", slice.stream_id ), ..Default::default() From f8c5fa89e7bb02fd3f439cde37abcff85d10018f Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Mon, 21 Sep 2026 14:18:05 -0700 Subject: [PATCH 17/35] Carried stream slices on a history pushed for replay. History records only the offsets a task consumed, so a language replayer that fetched the records from the stream service needs a way to hand them to Core with the history. The replay worker puts them on its synthetic poll response, so the ordinary delivery path and identity checks run unchanged. --- crates/sdk-core/CHANGELOG.md | 3 + crates/sdk-core/src/core_tests/streams.rs | 200 +++++++++++++++++++++- crates/sdk-core/src/replay/mod.rs | 18 ++ 3 files changed, 215 insertions(+), 6 deletions(-) diff --git a/crates/sdk-core/CHANGELOG.md b/crates/sdk-core/CHANGELOG.md index acfc46bce..42682dff5 100644 --- a/crates/sdk-core/CHANGELOG.md +++ b/crates/sdk-core/CHANGELOG.md @@ -54,6 +54,9 @@ relevant information. `SubscribeStream` and `AppendStreamRecords` commands. Consumed ranges reach the workflow as `DeliverStreamRecords` activation jobs, and replay hands each recorded range back in the activation of the task that consumed it. +* A history fed to a replay worker can carry the stream records its tasks consumed + (`HistoryForReplay::with_stream_slices`), so a language replayer that fetched them from the + stream service can replay a consuming workflow. History alone holds only the offsets. ### Breaking Changes :boom: * The following types are now non-exhaustive: `Priority`, `WorkerDeploymentVersion`, diff --git a/crates/sdk-core/src/core_tests/streams.rs b/crates/sdk-core/src/core_tests/streams.rs index 02a9fdd75..80e2f8749 100644 --- a/crates/sdk-core/src/core_tests/streams.rs +++ b/crates/sdk-core/src/core_tests/streams.rs @@ -7,11 +7,11 @@ //! arrive, in the right order, and that an empty range is still delivered. use crate::{ - Worker, - replay::TestHistoryBuilder, + Worker, init_replay_worker, + replay::{HistoryFeeder, HistoryForReplay, ReplayWorkerInput, TestHistoryBuilder}, test_help::{ MockPollCfg, PollWFTRespExt, ResponseType, WorkerTestHelpers, build_mock_pollers, - hist_to_poll_resp, mock_worker, + hist_to_poll_resp, mock_worker, test_worker_cfg, }, worker::client::mocks::mock_worker_client, }; @@ -21,15 +21,18 @@ use std::sync::{ }; use temporalio_common::protos::{ coresdk::{ - workflow_activation::{WorkflowActivation, WorkflowActivationJob, workflow_activation_job}, - workflow_commands::{AppendStreamRecords, SubscribeStream}, + workflow_activation::{ + RemoveFromCache, WorkflowActivation, WorkflowActivationJob, + remove_from_cache::EvictionReason, workflow_activation_job, + }, + workflow_commands::{AppendStreamRecords, CompleteWorkflowExecution, SubscribeStream}, workflow_completion::WorkflowActivationCompletion, }, temporal::api::{ command::v1::command, enums::v1::{CommandType, EventType, WorkflowTaskFailedCause}, history::v1::{History, HistoryEvent}, - stream::v1::{StreamRange, StreamRecord}, + stream::v1::{StreamRange, StreamRecord, StreamSlice}, workflowservice::v1::{ GetWorkflowExecutionHistoryResponse, RespondWorkflowTaskCompletedResponse, }, @@ -1063,3 +1066,188 @@ async fn a_missing_resupply_for_a_consumed_range_fails_the_task() { assert_eq!(failures.load(Ordering::Relaxed), 1); core.shutdown().await; } + +/// A slice as the server re-supplies it: the records of one recorded range, +/// tagged with the completion that recorded it. +fn replay_slice( + stream_id: &str, + completed_event_id: i64, + from: i64, + bodies: &[&str], +) -> StreamSlice { + StreamSlice { + stream_id: stream_id.to_string(), + from_offset: from, + to_offset: from + bodies.len() as i64, + records: bodies + .iter() + .map(|b| StreamRecord { + body: Some(b.as_bytes().to_vec().into()), + ..Default::default() + }) + .collect(), + workflow_task_completed_event_id: completed_event_id, + ..Default::default() + } +} + +/// A publish of one record per body. +fn publish(stream_id: &str, bodies: &[&str]) -> AppendStreamRecords { + AppendStreamRecords { + stream_id: stream_id.to_string(), + records: bodies + .iter() + .map(|b| StreamRecord { + body: Some(b.as_bytes().to_vec().into()), + ..Default::default() + }) + .collect(), + } +} + +fn eviction(task: &WorkflowActivation) -> &RemoveFromCache { + match task.jobs.as_slice() { + [ + WorkflowActivationJob { + variant: Some(workflow_activation_job::Variant::RemoveFromCache(evict)), + }, + ] => evict, + other => panic!("expected an eviction, got {other:?}"), + } +} + +/// A recorded read-then-publish workflow: each task consumes one record and +/// publishes because of it, the second one also completes the run. +fn read_then_publish_history() -> (TestHistoryBuilder, i64, i64) { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + let first = t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("in", 0, 1)]); + t.add_stream_records_appended("out", 0, 1); + t.add_workflow_task_scheduled_and_started(); + let second = + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("in", 1, 2)]); + t.add_stream_records_appended("out", 1, 1); + t.add_workflow_execution_completed(); + (t, first, second) +} + +/// A replay worker fed one history. The feeder is handed back so the history +/// stream stays open until the test drops it, as a language replayer keeps it +/// open: once the stream ends the worker closes, and an eviction still owed +/// for the last history would be lost to the shutdown. +async fn replay_worker(history: HistoryForReplay) -> (Worker, HistoryFeeder) { + let (feeder, stream) = HistoryFeeder::new(1); + feeder.feed(history).await.unwrap(); + let core = init_replay_worker(ReplayWorkerInput::new( + test_worker_cfg().build().unwrap(), + stream, + )) + .unwrap(); + (core, feeder) +} + +/// A history pushed for replay can carry the ranges its tasks consumed, in the +/// shape the server re-supplies them. The replay worker puts them on its +/// synthetic poll response, so the ordinary delivery path runs: each range +/// reaches the activation of the task that consumed it, and the publish that +/// task reissues is matched against its event. +#[tokio::test] +async fn a_pushed_history_replays_with_the_slices_it_carries() { + let (t, first, second) = read_then_publish_history(); + let history = HistoryForReplay::new(t.get_full_history_info().unwrap(), "wfid") + .with_stream_slices([ + replay_slice("in", first, 0, &["go"]), + replay_slice("in", second, 1, &["stop"]), + ]); + let (core, feeder) = replay_worker(history).await; + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(delivered_ranges(&task), vec![("in".to_string(), 0, 1)]); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![publish("out", &["accept"]).into()], + )) + .await + .unwrap(); + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(delivered_ranges(&task), vec![("in".to_string(), 1, 2)]); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![ + publish("out", &["done"]).into(), + CompleteWorkflowExecution { result: None }.into(), + ], + )) + .await + .unwrap(); + + // Replay is over and the worker lets the run go; a nondeterminism eviction + // would say the reissued publishes did not match their events. + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(eviction(&task).reason(), EvictionReason::LangRequested); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + drop(feeder); + core.shutdown().await; +} + +/// A slice attached to a pushed history is held to the same check as one the +/// server sends: offsets other than the recorded range fail the task as +/// nondeterministic rather than replay it on different input. +#[tokio::test] +async fn a_pushed_history_with_a_wrong_slice_fails_as_nondeterministic() { + let (t, first, second) = read_then_publish_history(); + let history = HistoryForReplay::new(t.get_full_history_info().unwrap(), "wfid") + .with_stream_slices([ + // Two records where the event says one. + replay_slice("in", first, 0, &["go", "extra"]), + replay_slice("in", second, 1, &["stop"]), + ]); + let (core, feeder) = replay_worker(history).await; + + // The mismatch is found while the poll response is applied, so the first + // activation is already the eviction. The task is failed as + // nondeterministic (the mock-client test above checks the cause); the + // eviction carries the reason in its message, since an eviction for a + // failure found before any activation ran reports no reason of its own. + let task = core.poll_workflow_activation().await.unwrap(); + let evict = eviction(&task); + assert!( + evict.message.contains( + "Event 4 records that stream in was consumed from offset 0 to 1, but the server \ + sent offsets 0 to 2 for it" + ), + "eviction did not name the mismatch: {evict:?}" + ); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + drop(feeder); + core.shutdown().await; +} + +/// A recorded range with content and no slice for it is the same failure. A +/// language replayer that has no store to fetch from cannot replay a consuming +/// workflow, and the task says so instead of replaying on less input. +#[tokio::test] +async fn a_pushed_history_without_slices_for_a_consumed_range_fails_as_nondeterministic() { + let (t, _, _) = read_then_publish_history(); + let history = HistoryForReplay::new(t.get_full_history_info().unwrap(), "wfid"); + let (core, feeder) = replay_worker(history).await; + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(delivered_ranges(&task), vec![]); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(eviction(&task).reason(), EvictionReason::Nondeterminism); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + drop(feeder); + core.shutdown().await; +} diff --git a/crates/sdk-core/src/replay/mod.rs b/crates/sdk-core/src/replay/mod.rs index 6e4cadc50..34bae9043 100644 --- a/crates/sdk-core/src/replay/mod.rs +++ b/crates/sdk-core/src/replay/mod.rs @@ -33,6 +33,7 @@ use temporalio_common::{ temporal::api::{ common::v1::WorkflowExecution, history::v1::History, + stream::v1::StreamSlice, workflowservice::v1::{ DescribeNamespaceResponse, RespondWorkflowTaskCompletedResponse, RespondWorkflowTaskFailedResponse, @@ -118,6 +119,7 @@ where workflow_id: history.workflow_id, run_id: hist_info.orig_run_id().to_string(), }); + resp.stream_slices = history.stream_slices; Ok(resp) } else { if let Some(wc) = hlock.worker_closer.get() { @@ -174,6 +176,7 @@ mod tests { pub struct HistoryForReplay { hist: History, workflow_id: String, + stream_slices: Vec, } impl HistoryForReplay { /// Create a new history from replay from something that looks like a history and a workflow id. @@ -181,8 +184,23 @@ impl HistoryForReplay { Self { hist: history.into(), workflow_id: workflow_id.into(), + stream_slices: Vec::new(), } } + + /// Attach the stream records the history's completed tasks consumed. + /// + /// History records the offsets a task consumed and never the payloads, so a workflow that + /// read a stream cannot be replayed from its history alone. Each slice carries the records + /// for one recorded range, tagged with the `WorkflowTaskCompleted` event that recorded it, + /// the same shape the server puts on a poll response when it re-supplies them. Replay hands + /// each range to the activation of the task that consumed it and fails the task as + /// nondeterministic when a slice disagrees with the recorded range or a recorded range with + /// content has no slice. + pub fn with_stream_slices(mut self, slices: impl IntoIterator) -> Self { + self.stream_slices = slices.into_iter().collect(); + self + } } #[cfg(any(feature = "test-utilities", test))] impl From for HistoryForReplay { From 61b0cae8f983d4d3099e7a9154685108ecee67cb Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Mon, 21 Sep 2026 15:13:27 -0700 Subject: [PATCH 18/35] Failed a task owed stream records the response did not carry. The bytes for a recorded range only travel on the response that carries the task, so a sticky task, or a sticky legacy query, handed to a worker that no longer holds the run and fetched the history itself cannot be replayed. Failing when the lookahead sees the range keeps the workflow off less input than it had, and lets a legacy query go unanswered so the server retries it on the normal queue, where the records travel with it. --- crates/sdk-core/CHANGELOG.md | 4 +++ crates/sdk-core/src/core_tests/streams.rs | 33 +++++++++-------- .../workflow/machines/workflow_machines.rs | 35 +++++++++++++------ 3 files changed, 47 insertions(+), 25 deletions(-) diff --git a/crates/sdk-core/CHANGELOG.md b/crates/sdk-core/CHANGELOG.md index 42682dff5..4a50406c0 100644 --- a/crates/sdk-core/CHANGELOG.md +++ b/crates/sdk-core/CHANGELOG.md @@ -57,6 +57,10 @@ relevant information. * A history fed to a replay worker can carry the stream records its tasks consumed (`HistoryForReplay::with_stream_slices`), so a language replayer that fetched them from the stream service can replay a consuming workflow. History alone holds only the offsets. +* A task whose history records a consumed range with content that the response carried no + records for fails before the workflow runs, rather than after it ran on less input. A legacy + query dispatched that way to a worker that no longer holds the run goes unanswered, so the + server retries it on the normal task queue, where the records travel with it. ### Breaking Changes :boom: * The following types are now non-exhaustive: `Priority`, `WorkerDeploymentVersion`, diff --git a/crates/sdk-core/src/core_tests/streams.rs b/crates/sdk-core/src/core_tests/streams.rs index 80e2f8749..f5d33cb2b 100644 --- a/crates/sdk-core/src/core_tests/streams.rs +++ b/crates/sdk-core/src/core_tests/streams.rs @@ -1044,9 +1044,11 @@ async fn a_cached_run_is_not_handed_its_own_range_again() { } /// History says a task consumed a range with content and the server sent no -/// bytes for it. The lookahead leaves that alone, since the completion may -/// arrive with a later response, and the batch that carries the completion -/// fails the task rather than replay it on less input than it ran on. +/// bytes for it. The bytes only travel on the response that carried the task, +/// so the lookahead fails the task as soon as it sees the range, before the +/// workflow runs on less input than it ran on. A sticky task handed to a worker +/// that no longer holds the run is the case that reaches this, and the server's +/// retry on the normal queue carries the records. #[tokio::test] async fn a_missing_resupply_for_a_consumed_range_fails_the_task() { let mut t = TestHistoryBuilder::default(); @@ -1057,11 +1059,8 @@ async fn a_missing_resupply_for_a_consumed_range_fails_the_task() { let (core, failures) = worker_expecting_one_nondeterminism_failure(t, ResponseType::AllHistory); - let task = core.poll_workflow_activation().await.unwrap(); - assert_eq!(delivered_ranges(&task), vec![]); - core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) - .await - .unwrap(); + // Found while the poll response is applied, so the first activation is + // already the eviction. core.handle_eviction().await; assert_eq!(failures.load(Ordering::Relaxed), 1); core.shutdown().await; @@ -1231,20 +1230,24 @@ async fn a_pushed_history_with_a_wrong_slice_fails_as_nondeterministic() { /// A recorded range with content and no slice for it is the same failure. A /// language replayer that has no store to fetch from cannot replay a consuming -/// workflow, and the task says so instead of replaying on less input. +/// workflow, and the task says so before the workflow runs on less input. #[tokio::test] async fn a_pushed_history_without_slices_for_a_consumed_range_fails_as_nondeterministic() { let (t, _, _) = read_then_publish_history(); let history = HistoryForReplay::new(t.get_full_history_info().unwrap(), "wfid"); let (core, feeder) = replay_worker(history).await; + // Found while the poll response is applied, so the first activation is + // already the eviction, which carries the reason in its message. let task = core.poll_workflow_activation().await.unwrap(); - assert_eq!(delivered_ranges(&task), vec![]); - core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) - .await - .unwrap(); - let task = core.poll_workflow_activation().await.unwrap(); - assert_eq!(eviction(&task).reason(), EvictionReason::Nondeterminism); + let evict = eviction(&task); + assert!( + evict.message.contains( + "Event 4 records that stream in was consumed from offset 0 to 1, but the server \ + sent no records for it" + ), + "eviction did not name the missing range: {evict:?}" + ); core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) .await .unwrap(); diff --git a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs index ebf81bd48..139587773 100644 --- a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs +++ b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs @@ -941,22 +941,37 @@ impl WorkflowMachines { } } // The task about to be replayed, whose range is only visible by looking - // ahead to the completion that closes it. Absent bytes for a range with - // content mean the server has not re-supplied them on this response, - // which the ordinary path above will still catch when that completion - // arrives as an event. + // ahead to the completion that closes it. if let Some((event_id, cursors)) = lookahead_slice_event && event_id > self.stream_slices_delivered_through + 1 && event_id > self.stream_slices_lookahead_through { let slices = self.stream_slices_by_event.remove(&event_id); - let needs_bytes = cursors.iter().any(|c| c.from_offset < c.to_offset); - if slices.is_some() || !needs_bytes { - for job in resupplied_deliveries(event_id, cursors, slices.unwrap_or_default())? { - self.drive_me.send_job(job); - } - self.stream_slices_lookahead_through = event_id; + if slices.is_none() + && let Some(cursor) = cursors.iter().find(|c| c.from_offset < c.to_offset) + { + // The bytes only ever travel on the response that carried this + // task, so absent bytes for a range with content mean this + // worker was handed a task it cannot replay: a sticky task for a + // run it no longer holds, whose history it fetched itself. + // Failing here, before the workflow runs on less input than it + // had, keeps a legacy query from being answered from the wrong + // state, and the server's retry on the normal task queue + // carries the records for both task kinds. + return Err(nondeterminism!( + "Event {event_id} records that stream {} was consumed from offset {} to {}, \ + but the server sent no records for it. A task dispatched with a partial \ + history to a worker that no longer holds the run cannot replay it; the \ + retry on the normal task queue carries the records.", + cursor.stream_id, + cursor.from_offset, + cursor.to_offset + )); + } + for job in resupplied_deliveries(event_id, cursors, slices.unwrap_or_default())? { + self.drive_me.send_job(job); } + self.stream_slices_lookahead_through = event_id; } // Then the range for the task about to run, which is only meaningful // once we have caught up to it. From 6446174bd0cdd97a2dcc72cd16d84cfa55be4361 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Mon, 21 Sep 2026 15:28:17 -0700 Subject: [PATCH 19/35] Withheld a legacy query owed records the response did not carry. A recorded range with content and no bytes for it is the worker's failure to reconstruct the run, not the workflow's nondeterminism, so it is its own error kind and treated like a failed history fetch: the task fails as an unhandled worker failure and a legacy query goes unanswered, so the server retries both where the records travel. The broken run is evicted even when the query failure is withheld, so the retry starts from history. --- crates/sdk-core/src/core_tests/streams.rs | 93 ++++++++++++++++++- .../workflow/machines/workflow_machines.rs | 16 ++-- .../src/worker/workflow/managed_run.rs | 18 ++-- crates/sdk-core/src/worker/workflow/mod.rs | 8 ++ 4 files changed, 116 insertions(+), 19 deletions(-) diff --git a/crates/sdk-core/src/core_tests/streams.rs b/crates/sdk-core/src/core_tests/streams.rs index f5d33cb2b..4e1a55a44 100644 --- a/crates/sdk-core/src/core_tests/streams.rs +++ b/crates/sdk-core/src/core_tests/streams.rs @@ -32,6 +32,7 @@ use temporalio_common::protos::{ command::v1::command, enums::v1::{CommandType, EventType, WorkflowTaskFailedCause}, history::v1::{History, HistoryEvent}, + query::v1::WorkflowQuery, stream::v1::{StreamRange, StreamRecord, StreamSlice}, workflowservice::v1::{ GetWorkflowExecutionHistoryResponse, RespondWorkflowTaskCompletedResponse, @@ -87,6 +88,14 @@ fn delivered_ranges(task: &WorkflowActivation) -> Vec<(String, i64, i64)> { fn worker_expecting_one_nondeterminism_failure( t: TestHistoryBuilder, resp: ResponseType, +) -> (Worker, Arc) { + worker_expecting_one_failure(t, resp, WorkflowTaskFailedCause::NonDeterministicError) +} + +fn worker_expecting_one_failure( + t: TestHistoryBuilder, + resp: ResponseType, + expected: WorkflowTaskFailedCause, ) -> (Worker, Arc) { let failures = Arc::new(AtomicUsize::new(0)); let counted = failures.clone(); @@ -94,7 +103,7 @@ fn worker_expecting_one_nondeterminism_failure( mock.num_expected_fails = 1; mock.expect_fail_wft_matcher = Box::new(move |_, cause, _| { counted.fetch_add(1, Ordering::Relaxed); - matches!(cause, WorkflowTaskFailedCause::NonDeterministicError) + *cause == expected }); let mut mock = build_mock_pollers(mock); mock.make_wft_stream_interminable(); @@ -1048,7 +1057,8 @@ async fn a_cached_run_is_not_handed_its_own_range_again() { /// so the lookahead fails the task as soon as it sees the range, before the /// workflow runs on less input than it ran on. A sticky task handed to a worker /// that no longer holds the run is the case that reaches this, and the server's -/// retry on the normal queue carries the records. +/// retry on the normal queue carries the records. The task is failed as the +/// worker's failure, not the workflow's: nothing the workflow did was wrong. #[tokio::test] async fn a_missing_resupply_for_a_consumed_range_fails_the_task() { let mut t = TestHistoryBuilder::default(); @@ -1057,7 +1067,11 @@ async fn a_missing_resupply_for_a_consumed_range_fails_the_task() { t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("s1", 0, 2)]); t.add_workflow_task_scheduled_and_started(); - let (core, failures) = worker_expecting_one_nondeterminism_failure(t, ResponseType::AllHistory); + let (core, failures) = worker_expecting_one_failure( + t, + ResponseType::AllHistory, + WorkflowTaskFailedCause::WorkflowWorkerUnhandledFailure, + ); // Found while the poll response is applied, so the first activation is // already the eviction. @@ -1066,6 +1080,79 @@ async fn a_missing_resupply_for_a_consumed_range_fails_the_task() { core.shutdown().await; } +/// A legacy query for a run this worker no longer holds arrives on the sticky +/// queue with partial history and no records, and the history the worker +/// fetches itself carries none either. Answering it from a replay on less +/// input would be wrong, and failing it would end the query: the server +/// retries a query it hears nothing about on the normal queue, where the +/// records travel with it. So the query goes unanswered, no task is failed, +/// and the run is given up so the retry starts from history. +#[tokio::test] +async fn a_legacy_query_owed_records_it_was_not_sent_goes_unanswered() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + // Task 1 subscribes and consumes nothing, so the missing range is only + // reached once the workflow has run its first activation. + t.add_workflow_task_completed(); + t.add_stream_subscribed("in", 0); + t.add_workflow_task_scheduled_and_started(); + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("in", 0, 2)]); + t.add_stream_records_appended("out", 0, 2); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + poll_resp.resp.query = Some(WorkflowQuery { + query_type: "trace".to_string(), + query_args: None, + header: None, + }); + // No slices: the mock plays the sticky queue, and the defaults of zero + // expected task failures and zero legacy query responses are the assertion. + let mock = MockPollCfg::from_resp_batches( + "wfid", + t, + [ResponseType::Raw(poll_resp.resp)], + mock_worker_client(), + ); + let mut mock = build_mock_pollers(mock); + mock.worker_cfg(|wc| { + wc.max_cached_workflows = 10; + wc.ignore_evicts_on_shutdown = false; + }); + let core = mock_worker(mock); + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(delivered_ranges(&task), vec![]); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![ + SubscribeStream { + stream_id: "in".to_string(), + start_offset: 0, + } + .into(), + ], + )) + .await + .unwrap(); + + // The missing range is found while the next task is applied. The run is + // evicted as a fetch failure would evict it, and nothing is reported. + let task = core.poll_workflow_activation().await.unwrap(); + let evict = eviction(&task); + assert_eq!(evict.reason(), EvictionReason::PaginationOrHistoryFetch); + assert!( + evict + .message + .contains("but the server sent no records for it"), + "eviction did not name the missing range: {evict:?}" + ); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + core.shutdown().await; +} + /// A slice as the server re-supplies it: the records of one recorded range, /// tagged with the completion that recorded it. fn replay_slice( diff --git a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs index 139587773..c742e3b8f 100644 --- a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs +++ b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs @@ -927,14 +927,12 @@ impl WorkflowMachines { // surface later as an unrelated nondeterminism error. Reaching // this from here also means the lookahead did not find the // completion, so the range was due one activation before this. - return Err(nondeterminism!( + return Err(WFMachinesError::MissingRecords(format!( "Event {event_id} records that stream {} was consumed from offset {} to {}, \ but the server sent no records for it. The workflow was owed that range \ one activation earlier.", - cursor.stream_id, - cursor.from_offset, - cursor.to_offset - )); + cursor.stream_id, cursor.from_offset, cursor.to_offset + ))); } for job in resupplied_deliveries(event_id, cursors, slices.unwrap_or_default())? { self.drive_me.send_job(job); @@ -958,15 +956,13 @@ impl WorkflowMachines { // had, keeps a legacy query from being answered from the wrong // state, and the server's retry on the normal task queue // carries the records for both task kinds. - return Err(nondeterminism!( + return Err(WFMachinesError::MissingRecords(format!( "Event {event_id} records that stream {} was consumed from offset {} to {}, \ but the server sent no records for it. A task dispatched with a partial \ history to a worker that no longer holds the run cannot replay it; the \ retry on the normal task queue carries the records.", - cursor.stream_id, - cursor.from_offset, - cursor.to_offset - )); + cursor.stream_id, cursor.from_offset, cursor.to_offset + ))); } for job in resupplied_deliveries(event_id, cursors, slices.unwrap_or_default())? { self.drive_me.send_job(job); diff --git a/crates/sdk-core/src/worker/workflow/managed_run.rs b/crates/sdk-core/src/worker/workflow/managed_run.rs index 4a37b4d25..9d6d23e39 100644 --- a/crates/sdk-core/src/worker/workflow/managed_run.rs +++ b/crates/sdk-core/src/worker/workflow/managed_run.rs @@ -625,7 +625,10 @@ impl ManagedRun { EvictionReason::Unspecified | EvictionReason::PaginationOrHistoryFetch ); - let (should_report, rur) = if is_no_report_query_fail { + // An unreported query failure leaves an intact run in the cache for the retry to use. A + // run whose machines broke while it was being brought up to the query is given up + // instead, since it can produce nothing more, so the retry starts from history. + let (should_report, rur) = if is_no_report_query_fail && !self.am_broken { (false, None) } else { // Blow up any cached data associated with the workflow @@ -635,11 +638,14 @@ impl ManagedRun { reason, auto_reply_fail_tt: None, }); - let should_report = match &evict_req_outcome { - EvictionRequestResult::EvictionRequested(Some(attempt), _) - | EvictionRequestResult::EvictionAlreadyRequested(Some(attempt)) => *attempt <= 1, - _ => false, - }; + let should_report = !is_no_report_query_fail + && match &evict_req_outcome { + EvictionRequestResult::EvictionRequested(Some(attempt), _) + | EvictionRequestResult::EvictionAlreadyRequested(Some(attempt)) => { + *attempt <= 1 + } + _ => false, + }; let rur = evict_req_outcome.into_run_update_resp(); (should_report, rur) }; diff --git a/crates/sdk-core/src/worker/workflow/mod.rs b/crates/sdk-core/src/worker/workflow/mod.rs index 2e4f4e596..aa2cef0af 100644 --- a/crates/sdk-core/src/worker/workflow/mod.rs +++ b/crates/sdk-core/src/worker/workflow/mod.rs @@ -1671,6 +1671,13 @@ pub(crate) enum WFMachinesError { Nondeterminism(String), #[error("Fatal error in workflow machines: {0}")] Fatal(String), + /// History records that a task consumed stream records and the response that carried the + /// task brought none of them, so this worker cannot replay the run. Not the workflow's + /// fault: the records only travel with the task, and a worker handed a sticky task for a run + /// it no longer holds has no way to fetch them. Treated like a failed history fetch, so a + /// legacy query goes unanswered and the server retries it where the records travel. + #[error("Workflow task cannot be replayed on this worker: {0}")] + MissingRecords(String), } /// Helper macro to create Nondeterminism errors with automatic assertion @@ -1750,6 +1757,7 @@ impl WFMachinesError { match self { WFMachinesError::Nondeterminism(_) => EvictionReason::Nondeterminism, WFMachinesError::Fatal(_) => EvictionReason::Fatal, + WFMachinesError::MissingRecords(_) => EvictionReason::PaginationOrHistoryFetch, } } From 32c66693f07222832443a5acb1ddb92006c23ea7 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Mon, 21 Sep 2026 17:43:51 -0700 Subject: [PATCH 20/35] Matched the stream protos to the api branch head. `api_upstream` carried an older cut of the stream additions, so regenerating `temporalio/api` from this Core removed what the api branch had since moved and documented. The three files are the api branch's diff applied onto the api tree this Core already pins. --- .../temporal/api/command/v1/message.proto | 52 ++++++++++++------- .../temporal/api/history/v1/message.proto | 21 +++++--- .../workflowservice/v1/request_response.proto | 12 ++--- 3 files changed, 50 insertions(+), 35 deletions(-) diff --git a/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto index f1dd29d2b..7d10e441b 100644 --- a/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto @@ -14,8 +14,8 @@ import "google/protobuf/duration.proto"; import "temporal/api/enums/v1/workflow.proto"; import "temporal/api/enums/v1/command_type.proto"; import "temporal/api/common/v1/message.proto"; -import "temporal/api/failure/v1/message.proto"; import "temporal/api/stream/v1/message.proto"; +import "temporal/api/failure/v1/message.proto"; import "temporal/api/taskqueue/v1/message.proto"; import "temporal/api/workflow/v1/message.proto"; import "temporal/api/sdk/v1/user_metadata.proto"; @@ -287,25 +287,6 @@ message RequestCancelNexusOperationCommandAttributes { int64 scheduled_event_id = 1; } -// Appends records to a stream the Workflow owns. Applied inside the Workflow -// Task's own commit. Produces one `WorkflowStreamRecordsAppended` event -// carrying the offset range and none of the payload; it schedules no further -// work. -message AppendStreamRecordsCommandAttributes { - // Empty means the Workflow's default output stream. - string stream_id = 1; - // Stored in order. The server sets `producer_id` to empty on each record, - // because the owning Workflow is the producer here. - repeated temporal.api.stream.v1.StreamRecord records = 2; -} - -message SubscribeStreamCommandAttributes { - string stream_id = 1; - // Negative means from wherever the stream is when the subscription is - // registered. The server resolves that once and records it. - int64 start_offset = 2; -} - message Command { temporal.api.enums.v1.CommandType command_type = 1; // Metadata on the command. This is sometimes carried over to the history event if one is @@ -348,3 +329,34 @@ message Command { SubscribeStreamCommandAttributes subscribe_stream_command_attributes = 21; } } + +// Appends records to a stream the Workflow owns. Applied inside the Workflow +// Task's own commit. Produces one `WorkflowStreamRecordsAppended` event +// carrying the offset range and none of the payload; it schedules no further +// work. +message AppendStreamRecordsCommandAttributes { + // Empty means the Workflow's default output stream. + string stream_id = 1; + // Stored in order. The server sets `producer_id` to empty on each record, + // because the owning Workflow is the producer here. + repeated temporal.api.stream.v1.StreamRecord records = 2; +} + +// Subscribe this Workflow to a stream, so later Workflow Tasks carry the ranges +// it has not consumed yet. +// +// The stream's addressing is resolved by the server rather than supplied here. +// A Workflow cannot look it up without doing I/O, and a value it carried would +// be a reading rather than a fact, so it could differ on replay. +message SubscribeStreamCommandAttributes { + // Stream to consume. A stream in another execution is addressed by its id; + // one this Workflow owns is addressed by the name it was published under. + // The server resolves an owned name first and falls back to a standalone + // id, so a Workflow that owns a stream under this name cannot reach a + // standalone stream with the same id. + string stream_id = 1; + // Where to start. Negative means from wherever the stream is when the + // subscription is registered, which the server resolves and records so + // replay does not resolve it again. + int64 start_offset = 2; +} diff --git a/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto index 13e1961ca..8deca8d15 100644 --- a/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto @@ -961,6 +961,19 @@ message ActivityPropertiesModifiedExternallyEventAttributes { temporal.api.common.v1.RetryPolicy new_retry_policy = 2; } +message WorkflowStreamSubscribedEventAttributes { + // The WorkflowTaskCompleted event of the task whose command created this + // subscription. + int64 workflow_task_completed_event_id = 1; + // Stream the Workflow subscribed to. + string stream_id = 2; + // The offset the subscription actually starts from. Resolved by the server + // when the subscription is registered and recorded here, so replay reads + // the resolved value rather than resolving it again against a stream that + // has since moved. + int64 start_offset = 3; +} + message WorkflowStreamRecordsAppendedEventAttributes { // The WorkflowTaskCompleted event of the task whose command appended this // batch. @@ -975,14 +988,6 @@ message WorkflowStreamRecordsAppendedEventAttributes { int64 record_count = 4; } -message WorkflowStreamSubscribedEventAttributes { - int64 workflow_task_completed_event_id = 1; - string stream_id = 2; - // Resolved by the server when the subscription is registered, so replay - // reads the resolved value rather than resolving it again. - int64 start_offset = 3; -} - message WorkflowExecutionUpdateAcceptedEventAttributes { // The instance ID of the update protocol that generated this event. string protocol_instance_id = 1; diff --git a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto index d9a603e8a..b396de5b4 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto @@ -24,6 +24,7 @@ import "temporal/api/enums/v1/activity.proto"; import "temporal/api/enums/v1/nexus.proto"; import "temporal/api/activity/v1/message.proto"; import "temporal/api/common/v1/message.proto"; +import "temporal/api/stream/v1/message.proto"; import "temporal/api/history/v1/message.proto"; import "temporal/api/workflow/v1/message.proto"; import "temporal/api/command/v1/message.proto"; @@ -38,7 +39,6 @@ import "temporal/api/replication/v1/message.proto"; import "temporal/api/rules/v1/message.proto"; import "temporal/api/sdk/v1/worker_config.proto"; import "temporal/api/schedule/v1/message.proto"; -import "temporal/api/stream/v1/message.proto"; import "temporal/api/taskqueue/v1/message.proto"; import "temporal/api/update/v1/message.proto"; import "temporal/api/version/v1/message.proto"; @@ -367,12 +367,6 @@ message PollWorkflowTaskQueueResponse { // This poller group ID identifies the owner of the workflow task awaiting for query response. // Corresponding RespondQueryTaskCompleted should pass this value for proper routing. string poller_group_id = 17; - - // Slices of the streams this Workflow consumes. A slice with no - // workflow_task_completed_event_id is for the task about to run; one with - // it set is re-supplying a range an earlier task consumed, so a worker - // replaying from History gets the same bytes that task was given. - repeated temporal.api.stream.v1.StreamSlice stream_slices = 20; // Deprecated. Use `poller_groups_info` instead, which carries a version so the client can // ignore stale updates. // The weighted list of poller groups IDs that client should use for future polls to this task @@ -390,6 +384,10 @@ message PollWorkflowTaskQueueResponse { // 3. If every group has some pending polls, assign the next poll to a group randomly // according to the weights. temporal.api.taskqueue.v1.PollerGroupsInfo poller_groups_info = 19; + + // Stream data attached to this task. Delivered out of band so the payloads + // never enter History; only the offset ranges are recorded there. + repeated temporal.api.stream.v1.StreamSlice stream_slices = 20; } message RespondWorkflowTaskCompletedRequest { From 208b8d8e87b5b36b217fe1239112c2379d73056d Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 25 Sep 2026 08:10:25 -0700 Subject: [PATCH 21/35] Delivered server-side stream ranges to workflows. History records the offsets a task consumed and never the payloads, so the server sends the bytes on the poll response: untagged for the task about to run, and tagged with a WorkflowTaskCompleted event id when re-supplying what an earlier task consumed. Core partitions the two and emits recorded ranges in event order before the live one, so a replaying workflow observes them exactly as it did the first time. A range a task consumed is recorded on the completion that closes it, so the machines look ahead to that event while replaying the task, and the paginator keeps that completion reachable when a page boundary falls between the two. --- crates/sdk-core/src/protosext/mod.rs | 7 + .../src/worker/workflow/history_update.rs | 35 ++- .../workflow/machines/workflow_machines.rs | 247 +++++++++++++++++- .../src/worker/workflow/managed_run.rs | 8 +- crates/sdk-core/src/worker/workflow/mod.rs | 10 + 5 files changed, 298 insertions(+), 9 deletions(-) diff --git a/crates/sdk-core/src/protosext/mod.rs b/crates/sdk-core/src/protosext/mod.rs index 9f0b77348..9af592890 100644 --- a/crates/sdk-core/src/protosext/mod.rs +++ b/crates/sdk-core/src/protosext/mod.rs @@ -39,6 +39,7 @@ use temporalio_common::protos::{ history::v1::{History, HistoryEvent, MarkerRecordedEventAttributes, history_event}, query::v1::WorkflowQuery, sdk::v1::{EventGroupMarker, UserMetadata}, + stream::v1::StreamSlice, workflowservice::v1::PollWorkflowTaskQueueResponse, }, utilities::TryIntoOrNone, @@ -64,6 +65,10 @@ pub(crate) struct ValidPollWFTQResponse { pub(crate) query_requests: Vec, /// Protocol messages pub(crate) messages: Vec, + /// Ranges of streams this workflow subscribed to. A slice tagged with a + /// completed-event id is re-supplying what an earlier task consumed; an + /// untagged one belongs to the task about to run. + pub(crate) stream_slices: Vec, /// Zero-size field to prevent explicit construction _cant_construct_me: (), @@ -109,6 +114,7 @@ impl TryFrom for ValidPollWFTQResponse { query, queries, messages, + stream_slices, .. } => { if task_token.is_empty() { @@ -133,6 +139,7 @@ impl TryFrom for ValidPollWFTQResponse { legacy_query: query, query_requests, messages, + stream_slices, _cant_construct_me: (), }) } diff --git a/crates/sdk-core/src/worker/workflow/history_update.rs b/crates/sdk-core/src/worker/workflow/history_update.rs index 856589946..a8523b9cb 100644 --- a/crates/sdk-core/src/worker/workflow/history_update.rs +++ b/crates/sdk-core/src/worker/workflow/history_update.rs @@ -162,6 +162,7 @@ impl HistoryPaginator { query_requests: wft.query_requests, update, messages: wft.messages, + stream_slices: wft.stream_slices, }; Ok((paginator, prepared)) } @@ -282,7 +283,7 @@ impl HistoryPaginator { // We only *really* have the last WFT if the events go all the way up to at least the // WFT started event id. Otherwise we somehow still have partial history. let no_more = matches!(self.next_page_token, NextPageToken::Done) && seen_enough_events; - let (update, extra) = HistoryUpdate::from_events( + let (mut update, extra) = HistoryUpdate::from_events( current_events, self.previous_wft_started_id, self.wft_started_event_id, @@ -309,8 +310,28 @@ impl HistoryPaginator { // There was not a meaningful WFT in the whole page. We must fetch more. continue; } + // The machines read the completion that closes an update's last task while that + // task is the one being replayed: it records the stream range the task consumed, + // and the workflow has to be handed that range in the same activation. A page that + // ends exactly on a WFT started event leaves that completion on the next page, so + // fetch it before handing the update over. + if !no_more && self.event_queue.is_empty() { + self.event_queue.extend(update.events); + continue; + } self.id_of_last_event_in_last_extracted_update = update.events.last().map(|e| e.event_id); + // Otherwise the completion is the first retained event. Carry a copy across the + // split so it can be peeked at. The original stays queued and is what the next + // update is built from, and a lone completion yields nothing to take, so no event + // is applied twice. + if let Some(completion) = self + .event_queue + .front() + .filter(|e| e.event_type() == EventType::WorkflowTaskCompleted) + { + update.events.push(completion.clone()); + } #[cfg(debug_assertions)] update.assert_contiguous(); return Ok(update); @@ -640,17 +661,21 @@ impl HistoryUpdate { true } - /// Returns the next WFT completed event attributes, if any, starting at (inclusive) the - /// `from_id` + /// Returns the next WFT completed event, if any, starting at (inclusive) the + /// `from_id`, as its event id and attributes. + /// + /// The id matters to callers that need to key something on the event rather + /// than only read its contents, such as the stream range a task consumed, + /// which is recorded on the completion that closes that task. pub(crate) fn peek_next_wft_completed( &self, from_id: i64, - ) -> Option<&WorkflowTaskCompletedEventAttributes> { + ) -> Option<(i64, &WorkflowTaskCompletedEventAttributes)> { self.events .iter() .skip_while(|e| e.event_id < from_id) .find_map(|e| match &e.attributes { - Some(Attributes::WorkflowTaskCompletedEventAttributes(a)) => Some(a), + Some(Attributes::WorkflowTaskCompletedEventAttributes(a)) => Some((e.event_id, a)), _ => None, }) } diff --git a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs index 7648aed81..c742e3b8f 100644 --- a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs +++ b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs @@ -72,6 +72,7 @@ use temporalio_common::{ history::v1::{HistoryEvent, history_event}, protocol::v1::{Message as ProtocolMessage, message::SequencingId}, sdk::v1::WorkflowTaskCompletedMetadata, + stream::v1::{StreamRange, StreamSlice}, }, }, worker::WorkerDeploymentVersion, @@ -88,6 +89,29 @@ pub(crate) struct WorkflowMachines { /// kept because the lang side polls & completes for every workflow task, but we do not need /// to poll the server that often during replay. last_history_from_server: HistoryUpdate, + /// Stream ranges an earlier task consumed, keyed by the WorkflowTaskCompleted + /// event that recorded each one. Only offsets are in History, so replay is + /// served by the server reading the stream again and sending the bytes back + /// tagged with that event. + stream_slices_by_event: HashMap>, + /// Ranges for the task about to run, which no event records yet. + current_stream_slices: Vec, + /// Highest completion event whose recorded range was delivered by looking + /// ahead, or 0. + /// + /// A task's consumed range is recorded on the completion that closes it, so + /// it is found one batch early. That same completion is then seen again as + /// an ordinary event in the next batch, where its slices are already spent + /// and must not be asked for twice. + stream_slices_lookahead_through: i64, + /// Workflow Task Started id of the last task this instance ran live, or 0. + /// + /// A task that ran here was handed its stream records as it ran, and the + /// completion event recording what it consumed arrives in the next task's + /// history. That event needs no bytes from the server, and `replaying` does + /// not say so: it is true for a cache hit as well, because the history + /// begins after an earlier task. + stream_slices_delivered_through: i64, /// Protocol messages that have yet to be processed for the current WFT. protocol_msgs: Vec, /// EventId of the last handled WorkflowTaskStarted event @@ -268,7 +292,7 @@ impl WorkflowMachines { basics.sdk_version.to_owned(), ); // Peek ahead to determine used flags in the first WFT. - if let Some(attrs) = basics.history.peek_next_wft_completed(0) { + if let Some((_, attrs)) = basics.history.peek_next_wft_completed(0) { observed_internal_flags.add_from_complete(attrs); }; Self { @@ -278,6 +302,10 @@ impl WorkflowMachines { workflow_type: basics.workflow_type, run_id: basics.run_id, drive_me: driven_wf, + stream_slices_by_event: Default::default(), + current_stream_slices: Default::default(), + stream_slices_delivered_through: 0, + stream_slices_lookahead_through: 0, replaying, metrics: basics.metrics, // In an ideal world one could say ..Default::default() here and it'd still work. @@ -327,11 +355,31 @@ impl WorkflowMachines { &mut self, update: HistoryUpdate, protocol_messages: Vec, + stream_slices: Vec, ) -> Result<()> { if !self.protocol_msgs.is_empty() { dbg_panic!("There are unprocessed protocol messages while receiving new work"); } self.protocol_msgs = protocol_messages; + // An untagged slice belongs to the task about to run; a tagged one is + // re-supplying a range some earlier task already consumed. + self.current_stream_slices.clear(); + self.stream_slices_by_event.clear(); + for slice in stream_slices { + if slice.workflow_task_completed_event_id == 0 { + self.current_stream_slices.push(slice); + } else { + self.stream_slices_by_event + .entry(slice.workflow_task_completed_event_id) + .or_default() + .push(slice); + } + } + // The order ranges arrive in is something the workflow can branch on, and + // the server's order is whatever its iteration happened to produce. Fixing + // it here is what makes it the same live and on replay. + self.current_stream_slices + .sort_by(|a, b| stream_order(&a.stream_id, a.from_offset, &b.stream_id, b.from_offset)); self.new_history_from_server(update)?; Ok(()) } @@ -619,19 +667,40 @@ impl WorkflowMachines { } }}; } + let mut replayed_slice_events: Vec<(i64, Vec)> = vec![]; + // Kept apart from the in-batch ones. An in-batch event whose bytes are + // missing is a real inconsistency; a looked-ahead one may simply not + // have been sent yet, and demanding it would turn an early delivery + // into a new way to fail. + let mut lookahead_slice_event: Option<(i64, Vec)> = None; let mut peeked_events = events.iter().peekable(); while let Some(event) = peeked_events.next() { if let Some(history_event::Attributes::WorkflowTaskCompletedEventAttributes(ref wtc)) = event.attributes { apply_wft_complete_data!(self, wtc); + // The event records which offsets that task consumed; the server + // sends the bytes back separately, keyed by this event. + if !wtc.consumed_stream_ranges.is_empty() { + replayed_slice_events + .push((event.event_id, wtc.consumed_stream_ranges.clone())); + } } if peeked_events.peek().is_none() - && let Some(wtc) = self + && let Some((wtc_id, wtc)) = self .last_history_from_server .peek_next_wft_completed(event.event_id) { apply_wft_complete_data!(self, wtc); + // The range this batch's task consumed is recorded on the + // completion that closes it, which lands in the *next* batch. + // Collecting it only when it appears would hand the workflow + // its input one activation after the commands that input + // caused, so a read-then-publish task replays with nothing to + // decide from and reissues no command. Look ahead for it here. + if !wtc.consumed_stream_ranges.is_empty() { + lookahead_slice_event = Some((wtc_id, wtc.consumed_stream_ranges.clone())); + } } } @@ -826,6 +895,91 @@ impl WorkflowMachines { } } + // Ranges an earlier task consumed, in the order those tasks ran, so a + // replaying workflow observes them exactly as it did the first time. + // + // Only while replaying. A cache hit is handed the previous task's + // completion in its history as well, and that event carries the range + // that task consumed, but that task ran here and its messages were + // delivered live at the time. There is nothing to re-supply, the server + // sends nothing, and re-supplying would hand the workflow the same + // messages twice. + replayed_slice_events.sort_unstable_by_key(|(event_id, _)| *event_id); + for (event_id, cursors) in replayed_slice_events { + // Delivered live to this instance when the task ran, so there is + // nothing to re-supply and the server sent nothing. + if event_id <= self.stream_slices_delivered_through + 1 { + self.stream_slices_by_event.remove(&event_id); + continue; + } + // Already handed over when this task was looked ahead to. + if event_id <= self.stream_slices_lookahead_through { + self.stream_slices_by_event.remove(&event_id); + continue; + } + let slices = self.stream_slices_by_event.remove(&event_id); + if slices.is_none() + && let Some(cursor) = cursors.iter().find(|c| c.from_offset < c.to_offset) + { + // History says a task consumed a range and the server sent no + // bytes for it. Replaying with less data than the original run + // had produces different commands, and the mismatch would + // surface later as an unrelated nondeterminism error. Reaching + // this from here also means the lookahead did not find the + // completion, so the range was due one activation before this. + return Err(WFMachinesError::MissingRecords(format!( + "Event {event_id} records that stream {} was consumed from offset {} to {}, \ + but the server sent no records for it. The workflow was owed that range \ + one activation earlier.", + cursor.stream_id, cursor.from_offset, cursor.to_offset + ))); + } + for job in resupplied_deliveries(event_id, cursors, slices.unwrap_or_default())? { + self.drive_me.send_job(job); + } + } + // The task about to be replayed, whose range is only visible by looking + // ahead to the completion that closes it. + if let Some((event_id, cursors)) = lookahead_slice_event + && event_id > self.stream_slices_delivered_through + 1 + && event_id > self.stream_slices_lookahead_through + { + let slices = self.stream_slices_by_event.remove(&event_id); + if slices.is_none() + && let Some(cursor) = cursors.iter().find(|c| c.from_offset < c.to_offset) + { + // The bytes only ever travel on the response that carried this + // task, so absent bytes for a range with content mean this + // worker was handed a task it cannot replay: a sticky task for a + // run it no longer holds, whose history it fetched itself. + // Failing here, before the workflow runs on less input than it + // had, keeps a legacy query from being answered from the wrong + // state, and the server's retry on the normal task queue + // carries the records for both task kinds. + return Err(WFMachinesError::MissingRecords(format!( + "Event {event_id} records that stream {} was consumed from offset {} to {}, \ + but the server sent no records for it. A task dispatched with a partial \ + history to a worker that no longer holds the run cannot replay it; the \ + retry on the normal task queue carries the records.", + cursor.stream_id, cursor.from_offset, cursor.to_offset + ))); + } + for job in resupplied_deliveries(event_id, cursors, slices.unwrap_or_default())? { + self.drive_me.send_job(job); + } + self.stream_slices_lookahead_through = event_id; + } + // Then the range for the task about to run, which is only meaningful + // once we have caught up to it. + if !self.replaying { + for slice in std::mem::take(&mut self.current_stream_slices) { + self.drive_me.send_job(deliver_stream_records_job(slice)); + } + // This task is running here, so whatever it consumes is already in + // hand and its completion event will not need re-supplying. + self.stream_slices_delivered_through = self.next_started_event_id; + } + // Only record replay latency if we actually did replay work. This avoids recording // near-zero latencies for the first workflow task (which has no history to replay) or // when there were no events to process. @@ -1839,3 +1993,92 @@ enum CommandIdKind { /// A command which is fire-and-forget (ex: Upsert search attribs) NeverResolves, } + +/// Turn a slice the server supplied into the job lang sees. +fn deliver_stream_records_job(slice: StreamSlice) -> OutgoingJob { + workflow_activation::DeliverStreamRecords { + stream_id: slice.stream_id, + from_offset: slice.from_offset, + to_offset: slice.to_offset, + records: slice.records, + } + .into() +} + +/// The one order ranges are handed to a workflow in, live and on replay. +fn stream_order( + stream_a: &str, + from_offset_a: i64, + stream_b: &str, + from_offset_b: i64, +) -> std::cmp::Ordering { + (stream_a, from_offset_a).cmp(&(stream_b, from_offset_b)) +} + +/// Pair the ranges a completion event recorded with the slices the server sent +/// back for it, in the order the workflow is handed them. +/// +/// The event is the record of what the task saw, so the slices have to match +/// it rather than the other way round. A range that observed nothing needs no +/// bytes and is rebuilt from the cursor alone; a range with content has to +/// arrive, and what arrives has to cover exactly the recorded offsets. Anything +/// else would replay the task with different input than it ran on. +fn resupplied_deliveries( + event_id: i64, + mut cursors: Vec, + mut slices: Vec, +) -> Result> { + cursors.sort_by(|a, b| stream_order(&a.stream_id, a.from_offset, &b.stream_id, b.from_offset)); + let mut jobs = Vec::with_capacity(cursors.len()); + for cursor in cursors { + let slice = slices + .iter() + .position(|s| s.stream_id == cursor.stream_id) + .map(|ix| slices.swap_remove(ix)); + let slice = match slice { + Some(slice) + if slice.from_offset == cursor.from_offset + && slice.to_offset == cursor.to_offset => + { + slice + } + Some(slice) => { + return Err(nondeterminism!( + "Event {event_id} records that stream {} was consumed from offset {} to {}, \ + but the server sent offsets {} to {} for it", + cursor.stream_id, + cursor.from_offset, + cursor.to_offset, + slice.from_offset, + slice.to_offset + )); + } + None if cursor.from_offset == cursor.to_offset => StreamSlice { + stream_id: cursor.stream_id, + from_offset: cursor.from_offset, + to_offset: cursor.to_offset, + ..Default::default() + }, + None => { + return Err(nondeterminism!( + "Event {event_id} records that stream {} was consumed from offset {} to {}, \ + but the server sent no messages for it", + cursor.stream_id, + cursor.from_offset, + cursor.to_offset + )); + } + }; + jobs.push(deliver_stream_records_job(slice)); + } + if let Some(extra) = slices.first() { + return Err(nondeterminism!( + "The server sent stream {} from offset {} to {} for event {event_id}, which records \ + no such range", + extra.stream_id, + extra.from_offset, + extra.to_offset + )); + } + Ok(jobs) +} diff --git a/crates/sdk-core/src/worker/workflow/managed_run.rs b/crates/sdk-core/src/worker/workflow/managed_run.rs index 603b38acc..4a37b4d25 100644 --- a/crates/sdk-core/src/worker/workflow/managed_run.rs +++ b/crates/sdk-core/src/worker/workflow/managed_run.rs @@ -42,6 +42,7 @@ use temporalio_common::protos::{ temporal::api::{ enums::v1::{VersioningBehavior, WorkflowTaskFailedCause}, failure::v1::Failure, + stream::v1::StreamSlice, }, }; use tokio::sync::oneshot; @@ -247,7 +248,8 @@ impl ManagedRun { if is_incremental { self.metrics.sticky_cache_hit(); } - self.wfm.new_work_from_server(work.update, work.messages)? + self.wfm + .new_work_from_server(work.update, work.messages, work.stream_slices)? } else { let r = self.wfm.get_next_activation()?; if r.jobs.is_empty() { @@ -1448,8 +1450,10 @@ impl WorkflowManager { &mut self, update: HistoryUpdate, messages: Vec, + stream_slices: Vec, ) -> Result { - self.machines.new_work_from_server(update, messages)?; + self.machines + .new_work_from_server(update, messages, stream_slices)?; self.get_next_activation() } diff --git a/crates/sdk-core/src/worker/workflow/mod.rs b/crates/sdk-core/src/worker/workflow/mod.rs index c9f1dda4b..aa2cef0af 100644 --- a/crates/sdk-core/src/worker/workflow/mod.rs +++ b/crates/sdk-core/src/worker/workflow/mod.rs @@ -87,6 +87,7 @@ use temporalio_common::{ protocol::v1::Message as ProtocolMessage, query::v1::WorkflowQuery, sdk::v1::{EventGroupMarker, UserMetadata, WorkflowTaskCompletedMetadata}, + stream::v1::StreamSlice, taskqueue::v1::StickyExecutionAttributes, workflowservice::v1::{PollActivityTaskQueueResponse, get_system_info_response}, }, @@ -1017,6 +1018,7 @@ struct PreparedWFT { query_requests: Vec, update: HistoryUpdate, messages: Vec, + stream_slices: Vec, } impl PreparedWFT { @@ -1669,6 +1671,13 @@ pub(crate) enum WFMachinesError { Nondeterminism(String), #[error("Fatal error in workflow machines: {0}")] Fatal(String), + /// History records that a task consumed stream records and the response that carried the + /// task brought none of them, so this worker cannot replay the run. Not the workflow's + /// fault: the records only travel with the task, and a worker handed a sticky task for a run + /// it no longer holds has no way to fetch them. Treated like a failed history fetch, so a + /// legacy query goes unanswered and the server retries it where the records travel. + #[error("Workflow task cannot be replayed on this worker: {0}")] + MissingRecords(String), } /// Helper macro to create Nondeterminism errors with automatic assertion @@ -1748,6 +1757,7 @@ impl WFMachinesError { match self { WFMachinesError::Nondeterminism(_) => EvictionReason::Nondeterminism, WFMachinesError::Fatal(_) => EvictionReason::Fatal, + WFMachinesError::MissingRecords(_) => EvictionReason::PaginationOrHistoryFetch, } } From 7e12aa8252e12bb6a6d967932b53a7af62701ed5 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 25 Sep 2026 08:10:33 -0700 Subject: [PATCH 22/35] Gave a data-only stream task its own replay activation. A completion that records consumed stream cursors ends its task sequence. Such a task ran as one activation live, and folding it into a heartbeat chain would hand several ranges over at once on replay. --- crates/sdk-core/src/worker/workflow/history_update.rs | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/crates/sdk-core/src/worker/workflow/history_update.rs b/crates/sdk-core/src/worker/workflow/history_update.rs index a8523b9cb..4100b6e2e 100644 --- a/crates/sdk-core/src/worker/workflow/history_update.rs +++ b/crates/sdk-core/src/worker/workflow/history_update.rs @@ -763,6 +763,17 @@ fn find_end_index_of_next_wft_seq( wft_started_event_id_to_index.pop(); continue; } else if next_event_type == EventType::WorkflowTaskCompleted { + // A task that consumed a stream range issued nothing the machines match, + // but it was an activation of its own: the workflow was handed that range + // and ran on it. Replay has to give it its own activation too, rather than + // fold it into a heartbeat chain and hand several ranges over at once. + if let Some(Attributes::WorkflowTaskCompletedEventAttributes(ref attrs)) = + next_event.attributes + && !attrs.consumed_stream_ranges.is_empty() + { + saw_command = true; + saw_command_or_started = true; + } if let Some(next_next_event) = events.get(ix + 2) { if !saw_command && next_next_event.event_type() == EventType::WorkflowTaskScheduled From d389f754dfe60c06f227f46261e5b5089ab72105 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 25 Sep 2026 08:10:34 -0700 Subject: [PATCH 23/35] Withheld a legacy query owed records the response did not carry. A recorded range with content and no bytes for it is the worker's failure to reconstruct the run, not the workflow's nondeterminism, so the task fails as an unhandled worker failure and a legacy query goes unanswered, and the server retries both where the records travel. The broken run is evicted even when the query failure is withheld, so the retry starts from history. --- .../src/worker/workflow/managed_run.rs | 18 ++++++++++++------ 1 file changed, 12 insertions(+), 6 deletions(-) diff --git a/crates/sdk-core/src/worker/workflow/managed_run.rs b/crates/sdk-core/src/worker/workflow/managed_run.rs index 4a37b4d25..9d6d23e39 100644 --- a/crates/sdk-core/src/worker/workflow/managed_run.rs +++ b/crates/sdk-core/src/worker/workflow/managed_run.rs @@ -625,7 +625,10 @@ impl ManagedRun { EvictionReason::Unspecified | EvictionReason::PaginationOrHistoryFetch ); - let (should_report, rur) = if is_no_report_query_fail { + // An unreported query failure leaves an intact run in the cache for the retry to use. A + // run whose machines broke while it was being brought up to the query is given up + // instead, since it can produce nothing more, so the retry starts from history. + let (should_report, rur) = if is_no_report_query_fail && !self.am_broken { (false, None) } else { // Blow up any cached data associated with the workflow @@ -635,11 +638,14 @@ impl ManagedRun { reason, auto_reply_fail_tt: None, }); - let should_report = match &evict_req_outcome { - EvictionRequestResult::EvictionRequested(Some(attempt), _) - | EvictionRequestResult::EvictionAlreadyRequested(Some(attempt)) => *attempt <= 1, - _ => false, - }; + let should_report = !is_no_report_query_fail + && match &evict_req_outcome { + EvictionRequestResult::EvictionRequested(Some(attempt), _) + | EvictionRequestResult::EvictionAlreadyRequested(Some(attempt)) => { + *attempt <= 1 + } + _ => false, + }; let rur = evict_req_outcome.into_run_update_resp(); (should_report, rur) }; From 17ddc935cb0984584e1a36bdb346df1152f31835 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 25 Sep 2026 08:10:45 -0700 Subject: [PATCH 24/35] Carried stream slices on a history pushed for replay. History records only the offsets a task consumed, so a language replayer that fetched the records from the stream service needs a way to hand them to Core with the history. The replay worker puts them on its synthetic poll response, so the ordinary delivery path and identity checks run unchanged. --- crates/sdk-core/src/replay/mod.rs | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/crates/sdk-core/src/replay/mod.rs b/crates/sdk-core/src/replay/mod.rs index 6e4cadc50..34bae9043 100644 --- a/crates/sdk-core/src/replay/mod.rs +++ b/crates/sdk-core/src/replay/mod.rs @@ -33,6 +33,7 @@ use temporalio_common::{ temporal::api::{ common::v1::WorkflowExecution, history::v1::History, + stream::v1::StreamSlice, workflowservice::v1::{ DescribeNamespaceResponse, RespondWorkflowTaskCompletedResponse, RespondWorkflowTaskFailedResponse, @@ -118,6 +119,7 @@ where workflow_id: history.workflow_id, run_id: hist_info.orig_run_id().to_string(), }); + resp.stream_slices = history.stream_slices; Ok(resp) } else { if let Some(wc) = hlock.worker_closer.get() { @@ -174,6 +176,7 @@ mod tests { pub struct HistoryForReplay { hist: History, workflow_id: String, + stream_slices: Vec, } impl HistoryForReplay { /// Create a new history from replay from something that looks like a history and a workflow id. @@ -181,8 +184,23 @@ impl HistoryForReplay { Self { hist: history.into(), workflow_id: workflow_id.into(), + stream_slices: Vec::new(), } } + + /// Attach the stream records the history's completed tasks consumed. + /// + /// History records the offsets a task consumed and never the payloads, so a workflow that + /// read a stream cannot be replayed from its history alone. Each slice carries the records + /// for one recorded range, tagged with the `WorkflowTaskCompleted` event that recorded it, + /// the same shape the server puts on a poll response when it re-supplies them. Replay hands + /// each range to the activation of the task that consumed it and fails the task as + /// nondeterministic when a slice disagrees with the recorded range or a recorded range with + /// content has no slice. + pub fn with_stream_slices(mut self, slices: impl IntoIterator) -> Self { + self.stream_slices = slices.into_iter().collect(); + self + } } #[cfg(any(feature = "test-utilities", test))] impl From for HistoryForReplay { From 17ba00d9957135db108066805d9d3ce9c5cd194f Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 25 Sep 2026 08:10:45 -0700 Subject: [PATCH 25/35] Covered the stream delivery guards and each range's activation. The guards are what the delivery path is for: an empty range still arrives, a re-supplied slice has to cover exactly the recorded offsets, a cached run is not handed its own range twice, and a task owed records the response did not carry fails before the workflow runs on less input. --- crates/sdk-core/src/core_tests/streams.rs | 1025 ++++++++++++++++- crates/sdk-core/src/replay/history_builder.rs | 17 + .../sdk-core/src/test_help/integ_helpers.rs | 40 +- 3 files changed, 1075 insertions(+), 7 deletions(-) diff --git a/crates/sdk-core/src/core_tests/streams.rs b/crates/sdk-core/src/core_tests/streams.rs index 0deefe2d4..4e1a55a44 100644 --- a/crates/sdk-core/src/core_tests/streams.rs +++ b/crates/sdk-core/src/core_tests/streams.rs @@ -1,7 +1,18 @@ +//! Delivery of server-side stream ranges to a workflow. +//! +//! History records the offsets a task consumed and never the payloads, so the +//! server sends the bytes on the poll response: untagged for the task about to +//! run, and tagged with a WorkflowTaskCompleted event id when it is +//! re-supplying what an earlier task consumed. These tests pin that both +//! arrive, in the right order, and that an empty range is still delivered. + use crate::{ - Worker, - replay::TestHistoryBuilder, - test_help::{MockPollCfg, ResponseType, WorkerTestHelpers, build_mock_pollers, mock_worker}, + Worker, init_replay_worker, + replay::{HistoryFeeder, HistoryForReplay, ReplayWorkerInput, TestHistoryBuilder}, + test_help::{ + MockPollCfg, PollWFTRespExt, ResponseType, WorkerTestHelpers, build_mock_pollers, + hist_to_poll_resp, mock_worker, test_worker_cfg, + }, worker::client::mocks::mock_worker_client, }; use std::sync::{ @@ -10,17 +21,65 @@ use std::sync::{ }; use temporalio_common::protos::{ coresdk::{ - workflow_commands::{AppendStreamRecords, SubscribeStream}, + workflow_activation::{ + RemoveFromCache, WorkflowActivation, WorkflowActivationJob, + remove_from_cache::EvictionReason, workflow_activation_job, + }, + workflow_commands::{AppendStreamRecords, CompleteWorkflowExecution, SubscribeStream}, workflow_completion::WorkflowActivationCompletion, }, temporal::api::{ command::v1::command, enums::v1::{CommandType, EventType, WorkflowTaskFailedCause}, - stream::v1::StreamRecord, - workflowservice::v1::RespondWorkflowTaskCompletedResponse, + history::v1::{History, HistoryEvent}, + query::v1::WorkflowQuery, + stream::v1::{StreamRange, StreamRecord, StreamSlice}, + workflowservice::v1::{ + GetWorkflowExecutionHistoryResponse, RespondWorkflowTaskCompletedResponse, + }, }, }; +fn cursor(stream_id: &str, from: i64, to: i64) -> StreamRange { + StreamRange { + stream_id: stream_id.to_string(), + from_offset: from, + to_offset: to, + } +} + +fn delivered(job: &WorkflowActivationJob) -> (&str, i64, i64, Vec<&[u8]>) { + match job.variant.as_ref().unwrap() { + workflow_activation_job::Variant::DeliverStreamRecords(d) => ( + d.stream_id.as_str(), + d.from_offset, + d.to_offset, + d.records + .iter() + .map(|m| m.body.as_ref().unwrap().data.as_slice()) + .collect(), + ), + other => panic!("expected a stream delivery, got {other:?}"), + } +} + +/// The stream deliveries in an activation, as (stream, from, to). +fn delivered_ranges(task: &WorkflowActivation) -> Vec<(String, i64, i64)> { + task.jobs + .iter() + .filter(|j| { + matches!( + j.variant, + Some(workflow_activation_job::Variant::DeliverStreamRecords(_)) + ) + }) + .map(|j| { + let (stream, from, to, _) = delivered(j); + (stream.to_string(), from, to) + }) + .collect() +} + /// A worker served one poll response, expected to fail that task as /// nondeterministic. Returns the worker and the count of failures it reported, /// which the test asserts itself: the mock only verifies call counts when it @@ -50,6 +109,165 @@ fn worker_expecting_one_failure( mock.make_wft_stream_interminable(); (mock_worker(mock), failures) } + +fn history_page( + events: &[HistoryEvent], + next_page_token: Vec, +) -> GetWorkflowExecutionHistoryResponse { + GetWorkflowExecutionHistoryResponse { + history: Some(History { + events: events.to_vec(), + }), + next_page_token, + ..Default::default() + } +} + +#[tokio::test] +async fn delivers_the_range_for_the_current_task() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + poll_resp.add_stream_slice("s1", 0, 0, &["alpha", "beta"]); + + let mock = MockPollCfg::from_resp_batches( + "wfid", + t, + [ResponseType::Raw(poll_resp.resp)], + mock_worker_client(), + ); + let core = mock_worker(build_mock_pollers(mock)); + + let task = core.poll_workflow_activation().await.unwrap(); + let stream_jobs: Vec<_> = task + .jobs + .iter() + .filter(|j| { + matches!( + j.variant, + Some(workflow_activation_job::Variant::DeliverStreamRecords(_)) + ) + }) + .collect(); + assert_eq!(stream_jobs.len(), 1); + assert_eq!( + delivered(stream_jobs[0]), + ("s1", 0, 2, vec![b"alpha".as_slice(), b"beta".as_slice()]) + ); + + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); +} + +// A task where the subscription saw nothing is a fact replay has to reproduce, +// so the range still has to arrive rather than being dropped as uninteresting. +#[tokio::test] +async fn delivers_an_empty_range() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + poll_resp.add_stream_slice("s1", 0, 4, &[]); + + let mock = MockPollCfg::from_resp_batches( + "wfid", + t, + [ResponseType::Raw(poll_resp.resp)], + mock_worker_client(), + ); + let core = mock_worker(build_mock_pollers(mock)); + + let task = core.poll_workflow_activation().await.unwrap(); + let job = task + .jobs + .iter() + .find(|j| { + matches!( + j.variant, + Some(workflow_activation_job::Variant::DeliverStreamRecords(_)) + ) + }) + .expect("an empty range is still delivered"); + assert_eq!(delivered(job), ("s1", 4, 4, vec![])); + + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); +} + +// On a cache miss every prior task replays, so ranges those tasks consumed have +// to be handed back in the order they were consumed, before the range for the +// task about to run. +#[tokio::test] +async fn replays_recorded_ranges_in_order_before_the_current_one() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + let first_completed = + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("s1", 0, 2)]); + t.add_workflow_task_scheduled_and_started(); + let second_completed = + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("s1", 2, 3)]); + t.add_workflow_task_scheduled_and_started(); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + // Deliberately out of order, to prove the ordering comes from the events + // rather than from however the server happened to lay them out. + poll_resp.add_stream_slice("s1", second_completed, 2, &["gamma"]); + poll_resp.add_stream_slice("s1", 0, 3, &["delta"]); + poll_resp.add_stream_slice("s1", first_completed, 0, &["alpha", "beta"]); + + let mock = MockPollCfg::from_resp_batches( + "wfid", + t, + [ResponseType::Raw(poll_resp.resp)], + mock_worker_client(), + ); + let mut mock = build_mock_pollers(mock); + mock.worker_cfg(|wc| wc.max_cached_workflows = 1); + let core = mock_worker(mock); + + // Each entry is (activation, from, to, bodies). The activation matters as + // much as the order: a range that came back one activation late would + // still be in order. + let mut seen = vec![]; + for activation in 1..=3 { + let task = core.poll_workflow_activation().await.unwrap(); + for job in &task.jobs { + if matches!( + job.variant, + Some(workflow_activation_job::Variant::DeliverStreamRecords(_)) + ) { + let (_, from, to, bodies) = delivered(job); + seen.push(( + activation, + from, + to, + bodies.iter().map(|b| b.to_vec()).collect::>(), + )); + } + } + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + } + + assert_eq!( + seen, + vec![ + (1, 0, 2, vec![b"alpha".to_vec(), b"beta".to_vec()]), + (2, 2, 3, vec![b"gamma".to_vec()]), + (3, 3, 4, vec![b"delta".to_vec()]), + ], + "each recorded range comes back in the activation of the task that consumed it, \ + then the live one" + ); +} + // A workflow subscribing itself. The command exists at all because every SDK // matches issued commands against command-generated events in order, so a // command producing no event would put that matching out of step. This asserts @@ -240,6 +458,294 @@ fn publish_two(stream_id: &str) -> AppendStreamRecords { ], } } + +/// A task that reads a range and publishes because of what it read. +/// +/// This is the shape the product requires: workflow code observes stream input, +/// decides, and writes. Replaying it means the workflow has to be handed the +/// input again *before* core matches the command that input caused, otherwise +/// lang has nothing to decide from and reissues nothing. +/// +/// The recorded range lives on the WorkflowTaskCompleted that closes the task, +/// which is the event *after* the one that started it. So the range for the +/// task about to be replayed is only visible by looking ahead, and a lookahead +/// that reads the completion for its flags but not for its cursors delivers the +/// input one activation too late. +#[tokio::test] +async fn read_then_publish_replays_when_the_range_is_only_visible_by_lookahead() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + // The task that reads [0,1) and publishes because of it. + let read_completed = + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("in", 0, 1)]); + t.add_stream_records_appended("out", 0, 1); + t.add_workflow_task_scheduled_and_started(); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + poll_resp.add_stream_slice("in", read_completed, 0, &["go"]); + + let mut mock_client = mock_worker_client(); + mock_client + .expect_complete_workflow_task() + .returning(|_, _| Ok(RespondWorkflowTaskCompletedResponse::default())); + mock_client + .expect_fail_workflow_task() + .returning(|_, _, f| panic!("core rejected the reissued read-caused publish: {f:?}")); + + let mock = + MockPollCfg::from_resp_batches("wfid", t, [ResponseType::Raw(poll_resp.resp)], mock_client); + let mut mock = build_mock_pollers(mock); + // Cold: nothing cached, so this is reconstruction from History. + mock.worker_cfg(|wc| wc.max_cached_workflows = 0); + let core = mock_worker(mock); + + let task = core.poll_workflow_activation().await.unwrap(); + let got_input = task.jobs.iter().any(|j| { + matches!( + j.variant, + Some(workflow_activation_job::Variant::DeliverStreamRecords(_)) + ) + }); + assert!( + got_input, + "the replayed task must receive the range it consumed before its \ + resulting publish is matched; jobs were {:?}", + task.jobs + ); + + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![ + AppendStreamRecords { + stream_id: "out".to_string(), + records: vec![StreamRecord { + body: Some(b"accept".to_vec().into()), + ..Default::default() + }], + } + .into(), + ], + )) + .await + .unwrap(); + + core.shutdown().await; +} + +/// The real shape: a subscribe task with no consumed range, then a task that +/// reads and publishes, then a third task replaying both. +/// +/// The first completion carries no cursors, so the lookahead that finds the +/// range has to keep looking past it rather than stopping at the first +/// completion it sees. +#[tokio::test] +async fn read_then_publish_replays_after_a_task_that_consumed_nothing() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + // Task 1 subscribes and consumes nothing. + t.add_workflow_task_completed(); + t.add_stream_subscribed("in", 0); + t.add_workflow_task_scheduled_and_started(); + // Task 2 reads [0,3) and publishes because of it. + let read_completed = + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("in", 0, 3)]); + t.add_stream_records_appended("out", 0, 1); + t.add_workflow_task_scheduled_and_started(); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + poll_resp.add_stream_slice("in", read_completed, 0, &["a", "b", "c"]); + + let mut mock_client = mock_worker_client(); + mock_client + .expect_complete_workflow_task() + .returning(|_, _| Ok(RespondWorkflowTaskCompletedResponse::default())); + mock_client + .expect_fail_workflow_task() + .returning(|_, _, f| panic!("core rejected the replayed read-caused publish: {f:?}")); + + let mock = + MockPollCfg::from_resp_batches("wfid", t, [ResponseType::Raw(poll_resp.resp)], mock_client); + let mut mock = build_mock_pollers(mock); + mock.worker_cfg(|wc| wc.max_cached_workflows = 0); + let core = mock_worker(mock); + + // Task 1: subscribe, no input yet. + let task = core.poll_workflow_activation().await.unwrap(); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![ + SubscribeStream { + stream_id: "in".to_string(), + start_offset: 0, + } + .into(), + ], + )) + .await + .unwrap(); + + // Task 2: the recorded range has to arrive before its publish is matched. + let task = core.poll_workflow_activation().await.unwrap(); + let bodies: Vec> = task + .jobs + .iter() + .filter(|j| { + matches!( + j.variant, + Some(workflow_activation_job::Variant::DeliverStreamRecords(_)) + ) + }) + .flat_map(|j| delivered(j).3.into_iter().map(|b| b.to_vec())) + .collect(); + assert_eq!( + bodies, + vec![b"a".to_vec(), b"b".to_vec(), b"c".to_vec()], + "the replayed task must be handed the range it consumed; jobs were {:?}", + task.jobs + ); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![ + AppendStreamRecords { + stream_id: "out".to_string(), + records: vec![StreamRecord { + body: Some(b"accept".to_vec().into()), + ..Default::default() + }], + } + .into(), + ], + )) + .await + .unwrap(); + + core.shutdown().await; +} + +/// The reading task's range has to arrive in its own activation when History +/// comes in pages and a page boundary falls at that task. +/// +/// The range is recorded on the completion that closes the task, and the +/// paginator hands the machines updates cut at WFT started events. Whether the +/// boundary lands right after the reading task's started event or right after +/// its completion, the completion is outside the update the machines are +/// replaying from, and a lookahead reading only that update would find nothing. +#[rstest::rstest] +#[tokio::test] +async fn read_then_publish_replays_across_a_page_boundary( + #[values(12, 13)] second_page_end: usize, +) { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_full_wf_task(); // 3 + t.add_we_signaled("go", vec![]); + t.add_full_wf_task(); // 7 + // Task 2 subscribes. Its command event is what lets the paginator tell the + // reading task's sequence is complete once it sees the started event. + t.add_stream_subscribed("in", 0); + t.add_we_signaled("go", vec![]); + t.add_workflow_task_scheduled_and_started(); // 12 + // Task 3 reads [0,1) and publishes because of it. + let read_completed = + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("in", 0, 1)]); + t.add_stream_records_appended("out", 0, 1); + t.add_workflow_task_scheduled_and_started(); // 16 + + let events = t.get_full_history_info().unwrap().into_events(); + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + poll_resp.history.as_mut().unwrap().events.truncate(3); + poll_resp.next_page_token = vec![1]; + // Two complete tasks past the previous started id fit in the first two + // pages, so the paginator would hand them over before fetching the third. + poll_resp.previous_started_event_id = 3; + poll_resp.add_stream_slice("in", read_completed, 0, &["go"]); + poll_resp.add_stream_slice("in", 0, 1, &["next"]); + + let second_page = history_page(&events[3..second_page_end], vec![2]); + let third_page = history_page(&events[second_page_end..], vec![]); + let mut mock_client = mock_worker_client(); + mock_client + .expect_get_workflow_execution_history() + .returning(move |_, _, token| match token.as_slice() { + [1] => Ok(second_page.clone()), + [2] => Ok(third_page.clone()), + other => panic!("unexpected page token {other:?}"), + }); + mock_client + .expect_fail_workflow_task() + .returning(|_, _, f| panic!("core rejected the replayed read-caused publish: {f:?}")); + + let mock = + MockPollCfg::from_resp_batches("wfid", t, [ResponseType::Raw(poll_resp.resp)], mock_client); + let core = mock_worker(build_mock_pollers(mock)); + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(delivered_ranges(&task), vec![]); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(delivered_ranges(&task), vec![]); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![ + SubscribeStream { + stream_id: "in".to_string(), + start_offset: 0, + } + .into(), + ], + )) + .await + .unwrap(); + + // The reading task. Its signal and its range belong to the same activation. + let task = core.poll_workflow_activation().await.unwrap(); + assert!(task.is_replaying); + assert!( + task.jobs.iter().any(|j| matches!( + j.variant, + Some(workflow_activation_job::Variant::SignalWorkflow(_)) + )), + "expected the reading task's signal; jobs were {:?}", + task.jobs + ); + assert_eq!( + delivered_ranges(&task), + vec![("in".to_string(), 0, 1)], + "the replayed task must be handed the range it consumed; jobs were {:?}", + task.jobs + ); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![ + AppendStreamRecords { + stream_id: "out".to_string(), + records: vec![StreamRecord { + body: Some(b"accept".to_vec().into()), + ..Default::default() + }], + } + .into(), + ], + )) + .await + .unwrap(); + + // The live task gets only its own range. + let task = core.poll_workflow_activation().await.unwrap(); + assert!(!task.is_replaying); + assert_eq!(delivered_ranges(&task), vec![("in".to_string(), 1, 2)]); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + + core.shutdown().await; +} + /// A publish reissued on replay is held against the recorded event, not only /// against its type. Core sends no commands while replaying, so this check is /// the only place a publish to the wrong stream can be noticed. @@ -328,3 +834,510 @@ async fn a_subscribe_reissued_to_a_different_stream_fails_the_task() { assert_eq!(failures.load(Ordering::Relaxed), 1); core.shutdown().await; } + +/// A task that consumes a range and issues no command still ran as its own +/// activation, so replay hands each such range over in its own activation +/// rather than collapsing the run of them into one, the way it does for +/// heartbeats that did nothing. +#[tokio::test] +async fn data_only_tasks_replay_one_range_per_activation() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + let first = t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("s1", 0, 1)]); + t.add_workflow_task_scheduled_and_started(); + let second = + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("s1", 1, 2)]); + t.add_workflow_task_scheduled_and_started(); + let third = t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("s1", 2, 3)]); + t.add_workflow_task_scheduled_and_started(); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + poll_resp.add_stream_slice("s1", first, 0, &["a"]); + poll_resp.add_stream_slice("s1", second, 1, &["b"]); + poll_resp.add_stream_slice("s1", third, 2, &["c"]); + poll_resp.add_stream_slice("s1", 0, 3, &["d"]); + + let mock = MockPollCfg::from_resp_batches( + "wfid", + t, + [ResponseType::Raw(poll_resp.resp)], + mock_worker_client(), + ); + let core = mock_worker(build_mock_pollers(mock)); + + let mut per_activation = vec![]; + for _ in 0..4 { + let task = core.poll_workflow_activation().await.unwrap(); + per_activation.push(delivered_ranges(&task)); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + } + assert_eq!( + per_activation, + vec![ + vec![("s1".to_string(), 0, 1)], + vec![("s1".to_string(), 1, 2)], + vec![("s1".to_string(), 2, 3)], + vec![("s1".to_string(), 3, 4)], + ], + "each consumed range replays in the activation of the task that consumed it" + ); + core.shutdown().await; +} + +/// Ranges for several streams on one task arrive ordered by stream, however +/// the server laid them out. A workflow that waits on two streams with a +/// first-completed pattern would otherwise take whichever branch the server's +/// iteration order happened to pick, live and again differently on replay. +#[tokio::test] +async fn ranges_for_several_streams_arrive_in_stream_order() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + let completed = t.add_workflow_task_completed_with_consumed_stream_ranges(vec![ + cursor("s2", 0, 1), + cursor("s1", 0, 1), + ]); + t.add_workflow_task_scheduled_and_started(); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + poll_resp.add_stream_slice("s2", completed, 0, &["two"]); + poll_resp.add_stream_slice("s1", completed, 0, &["one"]); + poll_resp.add_stream_slice("s2", 0, 1, &["four"]); + poll_resp.add_stream_slice("s1", 0, 1, &["three"]); + + let mock = MockPollCfg::from_resp_batches( + "wfid", + t, + [ResponseType::Raw(poll_resp.resp)], + mock_worker_client(), + ); + let core = mock_worker(build_mock_pollers(mock)); + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!( + delivered_ranges(&task), + vec![("s1".to_string(), 0, 1), ("s2".to_string(), 0, 1)], + "re-supplied ranges follow stream order" + ); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!( + delivered_ranges(&task), + vec![("s1".to_string(), 1, 2), ("s2".to_string(), 1, 2)], + "live ranges follow stream order" + ); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + core.shutdown().await; +} + +/// A recorded range that observed nothing is rebuilt from the cursor alone. The +/// event already says everything the workflow needs, so replay does not depend +/// on the server sending an empty slice back for it. +#[tokio::test] +async fn an_empty_recorded_range_replays_without_a_slice_from_the_server() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("s1", 4, 4)]); + t.add_workflow_task_scheduled_and_started(); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + poll_resp.add_stream_slice("s1", 0, 4, &["e"]); + + let mock = MockPollCfg::from_resp_batches( + "wfid", + t, + [ResponseType::Raw(poll_resp.resp)], + mock_worker_client(), + ); + let core = mock_worker(build_mock_pollers(mock)); + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(delivered_ranges(&task), vec![("s1".to_string(), 4, 4)]); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(delivered_ranges(&task), vec![("s1".to_string(), 4, 5)]); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + core.shutdown().await; +} + +/// A re-supplied slice is checked against the cursor it claims to satisfy. The +/// event is the record of what the task saw, so bytes covering other offsets +/// would replay the task on different input than it ran on. +#[tokio::test] +async fn a_resupplied_slice_that_disagrees_with_the_record_fails_the_task() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + let completed = + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("s1", 0, 2)]); + t.add_workflow_task_scheduled_and_started(); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + // One record where the event says two. + poll_resp.add_stream_slice("s1", completed, 0, &["a"]); + + let (core, failures) = + worker_expecting_one_nondeterminism_failure(t, ResponseType::Raw(poll_resp.resp)); + + // The mismatch is found while the poll response is applied, so the first + // activation is already the eviction. + core.handle_eviction().await; + assert_eq!(failures.load(Ordering::Relaxed), 1); + core.shutdown().await; +} + +/// A run that stays cached was handed its range as the task ran. When the next +/// task arrives, the completion recording that range is in its history, and the +/// server may re-supply the range tagged with it, as it would for a worker that +/// lost the run. This worker did not, so it must not process the same records +/// twice. +#[tokio::test] +async fn a_cached_run_is_not_handed_its_own_range_again() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + let completed = + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("s1", 0, 2)]); + t.add_workflow_task_scheduled_and_started(); + + let mut first_poll = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::ToTaskNum(1)); + first_poll.add_stream_slice("s1", 0, 0, &["a", "b"]); + let mut second_poll = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::OneTask(2)); + second_poll.add_stream_slice("s1", completed, 0, &["a", "b"]); + second_poll.add_stream_slice("s1", 0, 2, &["c"]); + + let mock = MockPollCfg::from_resp_batches( + "wfid", + t, + [ + ResponseType::Raw(first_poll.resp), + ResponseType::Raw(second_poll.resp), + ], + mock_worker_client(), + ); + let mut mock = build_mock_pollers(mock); + mock.worker_cfg(|wc| wc.max_cached_workflows = 1); + let core = mock_worker(mock); + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(delivered_ranges(&task), vec![("s1".to_string(), 0, 2)]); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + + let task = core.poll_workflow_activation().await.unwrap(); + assert!(!task.is_replaying); + assert_eq!( + delivered_ranges(&task), + vec![("s1".to_string(), 2, 3)], + "only the new range; the first one was delivered live to this worker" + ); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + core.shutdown().await; +} + +/// History says a task consumed a range with content and the server sent no +/// bytes for it. The bytes only travel on the response that carried the task, +/// so the lookahead fails the task as soon as it sees the range, before the +/// workflow runs on less input than it ran on. A sticky task handed to a worker +/// that no longer holds the run is the case that reaches this, and the server's +/// retry on the normal queue carries the records. The task is failed as the +/// worker's failure, not the workflow's: nothing the workflow did was wrong. +#[tokio::test] +async fn a_missing_resupply_for_a_consumed_range_fails_the_task() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("s1", 0, 2)]); + t.add_workflow_task_scheduled_and_started(); + + let (core, failures) = worker_expecting_one_failure( + t, + ResponseType::AllHistory, + WorkflowTaskFailedCause::WorkflowWorkerUnhandledFailure, + ); + + // Found while the poll response is applied, so the first activation is + // already the eviction. + core.handle_eviction().await; + assert_eq!(failures.load(Ordering::Relaxed), 1); + core.shutdown().await; +} + +/// A legacy query for a run this worker no longer holds arrives on the sticky +/// queue with partial history and no records, and the history the worker +/// fetches itself carries none either. Answering it from a replay on less +/// input would be wrong, and failing it would end the query: the server +/// retries a query it hears nothing about on the normal queue, where the +/// records travel with it. So the query goes unanswered, no task is failed, +/// and the run is given up so the retry starts from history. +#[tokio::test] +async fn a_legacy_query_owed_records_it_was_not_sent_goes_unanswered() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + // Task 1 subscribes and consumes nothing, so the missing range is only + // reached once the workflow has run its first activation. + t.add_workflow_task_completed(); + t.add_stream_subscribed("in", 0); + t.add_workflow_task_scheduled_and_started(); + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("in", 0, 2)]); + t.add_stream_records_appended("out", 0, 2); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + poll_resp.resp.query = Some(WorkflowQuery { + query_type: "trace".to_string(), + query_args: None, + header: None, + }); + // No slices: the mock plays the sticky queue, and the defaults of zero + // expected task failures and zero legacy query responses are the assertion. + let mock = MockPollCfg::from_resp_batches( + "wfid", + t, + [ResponseType::Raw(poll_resp.resp)], + mock_worker_client(), + ); + let mut mock = build_mock_pollers(mock); + mock.worker_cfg(|wc| { + wc.max_cached_workflows = 10; + wc.ignore_evicts_on_shutdown = false; + }); + let core = mock_worker(mock); + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(delivered_ranges(&task), vec![]); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![ + SubscribeStream { + stream_id: "in".to_string(), + start_offset: 0, + } + .into(), + ], + )) + .await + .unwrap(); + + // The missing range is found while the next task is applied. The run is + // evicted as a fetch failure would evict it, and nothing is reported. + let task = core.poll_workflow_activation().await.unwrap(); + let evict = eviction(&task); + assert_eq!(evict.reason(), EvictionReason::PaginationOrHistoryFetch); + assert!( + evict + .message + .contains("but the server sent no records for it"), + "eviction did not name the missing range: {evict:?}" + ); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + core.shutdown().await; +} + +/// A slice as the server re-supplies it: the records of one recorded range, +/// tagged with the completion that recorded it. +fn replay_slice( + stream_id: &str, + completed_event_id: i64, + from: i64, + bodies: &[&str], +) -> StreamSlice { + StreamSlice { + stream_id: stream_id.to_string(), + from_offset: from, + to_offset: from + bodies.len() as i64, + records: bodies + .iter() + .map(|b| StreamRecord { + body: Some(b.as_bytes().to_vec().into()), + ..Default::default() + }) + .collect(), + workflow_task_completed_event_id: completed_event_id, + ..Default::default() + } +} + +/// A publish of one record per body. +fn publish(stream_id: &str, bodies: &[&str]) -> AppendStreamRecords { + AppendStreamRecords { + stream_id: stream_id.to_string(), + records: bodies + .iter() + .map(|b| StreamRecord { + body: Some(b.as_bytes().to_vec().into()), + ..Default::default() + }) + .collect(), + } +} + +fn eviction(task: &WorkflowActivation) -> &RemoveFromCache { + match task.jobs.as_slice() { + [ + WorkflowActivationJob { + variant: Some(workflow_activation_job::Variant::RemoveFromCache(evict)), + }, + ] => evict, + other => panic!("expected an eviction, got {other:?}"), + } +} + +/// A recorded read-then-publish workflow: each task consumes one record and +/// publishes because of it, the second one also completes the run. +fn read_then_publish_history() -> (TestHistoryBuilder, i64, i64) { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + let first = t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("in", 0, 1)]); + t.add_stream_records_appended("out", 0, 1); + t.add_workflow_task_scheduled_and_started(); + let second = + t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("in", 1, 2)]); + t.add_stream_records_appended("out", 1, 1); + t.add_workflow_execution_completed(); + (t, first, second) +} + +/// A replay worker fed one history. The feeder is handed back so the history +/// stream stays open until the test drops it, as a language replayer keeps it +/// open: once the stream ends the worker closes, and an eviction still owed +/// for the last history would be lost to the shutdown. +async fn replay_worker(history: HistoryForReplay) -> (Worker, HistoryFeeder) { + let (feeder, stream) = HistoryFeeder::new(1); + feeder.feed(history).await.unwrap(); + let core = init_replay_worker(ReplayWorkerInput::new( + test_worker_cfg().build().unwrap(), + stream, + )) + .unwrap(); + (core, feeder) +} + +/// A history pushed for replay can carry the ranges its tasks consumed, in the +/// shape the server re-supplies them. The replay worker puts them on its +/// synthetic poll response, so the ordinary delivery path runs: each range +/// reaches the activation of the task that consumed it, and the publish that +/// task reissues is matched against its event. +#[tokio::test] +async fn a_pushed_history_replays_with_the_slices_it_carries() { + let (t, first, second) = read_then_publish_history(); + let history = HistoryForReplay::new(t.get_full_history_info().unwrap(), "wfid") + .with_stream_slices([ + replay_slice("in", first, 0, &["go"]), + replay_slice("in", second, 1, &["stop"]), + ]); + let (core, feeder) = replay_worker(history).await; + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(delivered_ranges(&task), vec![("in".to_string(), 0, 1)]); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![publish("out", &["accept"]).into()], + )) + .await + .unwrap(); + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(delivered_ranges(&task), vec![("in".to_string(), 1, 2)]); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![ + publish("out", &["done"]).into(), + CompleteWorkflowExecution { result: None }.into(), + ], + )) + .await + .unwrap(); + + // Replay is over and the worker lets the run go; a nondeterminism eviction + // would say the reissued publishes did not match their events. + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(eviction(&task).reason(), EvictionReason::LangRequested); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + drop(feeder); + core.shutdown().await; +} + +/// A slice attached to a pushed history is held to the same check as one the +/// server sends: offsets other than the recorded range fail the task as +/// nondeterministic rather than replay it on different input. +#[tokio::test] +async fn a_pushed_history_with_a_wrong_slice_fails_as_nondeterministic() { + let (t, first, second) = read_then_publish_history(); + let history = HistoryForReplay::new(t.get_full_history_info().unwrap(), "wfid") + .with_stream_slices([ + // Two records where the event says one. + replay_slice("in", first, 0, &["go", "extra"]), + replay_slice("in", second, 1, &["stop"]), + ]); + let (core, feeder) = replay_worker(history).await; + + // The mismatch is found while the poll response is applied, so the first + // activation is already the eviction. The task is failed as + // nondeterministic (the mock-client test above checks the cause); the + // eviction carries the reason in its message, since an eviction for a + // failure found before any activation ran reports no reason of its own. + let task = core.poll_workflow_activation().await.unwrap(); + let evict = eviction(&task); + assert!( + evict.message.contains( + "Event 4 records that stream in was consumed from offset 0 to 1, but the server \ + sent offsets 0 to 2 for it" + ), + "eviction did not name the mismatch: {evict:?}" + ); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + drop(feeder); + core.shutdown().await; +} + +/// A recorded range with content and no slice for it is the same failure. A +/// language replayer that has no store to fetch from cannot replay a consuming +/// workflow, and the task says so before the workflow runs on less input. +#[tokio::test] +async fn a_pushed_history_without_slices_for_a_consumed_range_fails_as_nondeterministic() { + let (t, _, _) = read_then_publish_history(); + let history = HistoryForReplay::new(t.get_full_history_info().unwrap(), "wfid"); + let (core, feeder) = replay_worker(history).await; + + // Found while the poll response is applied, so the first activation is + // already the eviction, which carries the reason in its message. + let task = core.poll_workflow_activation().await.unwrap(); + let evict = eviction(&task); + assert!( + evict.message.contains( + "Event 4 records that stream in was consumed from offset 0 to 1, but the server \ + sent no records for it" + ), + "eviction did not name the missing range: {evict:?}" + ); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + drop(feeder); + core.shutdown().await; +} diff --git a/crates/sdk-core/src/replay/history_builder.rs b/crates/sdk-core/src/replay/history_builder.rs index 2ad963a06..43723d243 100644 --- a/crates/sdk-core/src/replay/history_builder.rs +++ b/crates/sdk-core/src/replay/history_builder.rs @@ -23,6 +23,7 @@ use temporalio_common::protos::{ enums::v1::{EventType, TaskQueueKind, WorkflowTaskFailedCause}, failure::v1::{CanceledFailureInfo, Failure, failure}, history::v1::{history_event::Attributes, *}, + stream::v1::StreamRange, taskqueue::v1::TaskQueue, update, update::v1::outcome, @@ -129,6 +130,22 @@ impl TestHistoryBuilder { self.previous_task_completed_id = id; } + /// Add a workflow task completed event recording the stream offsets that + /// task consumed. Only the range is in History; the payloads come back from + /// the server on the poll response. + pub fn add_workflow_task_completed_with_consumed_stream_ranges( + &mut self, + cursors: Vec, + ) -> i64 { + let id = self.add(WorkflowTaskCompletedEventAttributes { + scheduled_event_id: self.workflow_task_scheduled_event_id, + consumed_stream_ranges: cursors, + ..Default::default() + }); + self.previous_task_completed_id = id; + id + } + /// Add the event a subscribe-stream command produces. pub fn add_stream_subscribed(&mut self, stream_id: &str, start_offset: i64) -> i64 { let attrs = WorkflowStreamSubscribedEventAttributes { diff --git a/crates/sdk-core/src/test_help/integ_helpers.rs b/crates/sdk-core/src/test_help/integ_helpers.rs index 0e446be95..cdfcc13da 100644 --- a/crates/sdk-core/src/test_help/integ_helpers.rs +++ b/crates/sdk-core/src/test_help/integ_helpers.rs @@ -50,10 +50,11 @@ use temporalio_common::{ workflow_completion::WorkflowActivationCompletion, }, temporal::api::{ - common::v1::WorkflowExecution, + common::v1::{Payload, WorkflowExecution}, enums::v1::WorkflowTaskFailedCause, failure::v1::Failure, protocol::{self, v1::message}, + stream::v1::{StreamRecord, StreamSlice}, update, workflowservice::v1::{ DescribeNamespaceResponse, PollActivityTaskQueueResponse, @@ -913,6 +914,17 @@ pub trait PollWFTRespExt { update_id: impl ToString, after_event_id: i64, ) -> update::v1::Request; + + /// Attach a range of a stream. Passing a `completed_event_id` of zero makes + /// it the range for the task about to run; anything else re-supplies what + /// the task closed by that event consumed. + fn add_stream_slice( + &mut self, + stream_id: impl ToString, + completed_event_id: i64, + from_offset: i64, + bodies: &[&str], + ); } impl PollWFTRespExt for PollWorkflowTaskQueueResponse { @@ -947,6 +959,32 @@ impl PollWFTRespExt for PollWorkflowTaskQueueResponse { }); upd_req_body } + + fn add_stream_slice( + &mut self, + stream_id: impl ToString, + completed_event_id: i64, + from_offset: i64, + bodies: &[&str], + ) { + self.stream_slices.push(StreamSlice { + stream_id: stream_id.to_string(), + from_offset, + to_offset: from_offset + bodies.len() as i64, + records: bodies + .iter() + .map(|b| StreamRecord { + body: Some(Payload { + data: b.as_bytes().to_vec(), + ..Default::default() + }), + ..Default::default() + }) + .collect(), + workflow_task_completed_event_id: completed_event_id, + ..Default::default() + }); + } } pub fn hist_to_poll_resp( From 63e7159f11f0b0431dae048834c398f3ca6dbb8a Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 25 Sep 2026 08:10:45 -0700 Subject: [PATCH 26/35] Added the changelog entry for the native stream commands and job. --- crates/sdk-core/CHANGELOG.md | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/crates/sdk-core/CHANGELOG.md b/crates/sdk-core/CHANGELOG.md index bfb8e04f9..4a50406c0 100644 --- a/crates/sdk-core/CHANGELOG.md +++ b/crates/sdk-core/CHANGELOG.md @@ -50,6 +50,17 @@ relevant information. metrics now carry a `failure_reason` attribute. Each is now split into one time series per reason, which may affect existing dashboards. * Workflow task completions larger than the gRPC request size limit are now paginated automatically when the namespace supports it. Paginated workflow task completions require Temporal Server 1.32.0 or later. +* Workflows can subscribe to server-side streams and append batches of records to them with the + `SubscribeStream` and `AppendStreamRecords` commands. Consumed ranges reach the workflow as + `DeliverStreamRecords` activation jobs, and replay hands each recorded range back in the + activation of the task that consumed it. +* A history fed to a replay worker can carry the stream records its tasks consumed + (`HistoryForReplay::with_stream_slices`), so a language replayer that fetched them from the + stream service can replay a consuming workflow. History alone holds only the offsets. +* A task whose history records a consumed range with content that the response carried no + records for fails before the workflow runs, rather than after it ran on less input. A legacy + query dispatched that way to a worker that no longer holds the run goes unanswered, so the + server retries it on the normal task queue, where the records travel with it. ### Breaking Changes :boom: * The following types are now non-exhaustive: `Priority`, `WorkerDeploymentVersion`, From 4cc44b0f68d52f4d44ee41e784e918646b87bde7 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 25 Sep 2026 13:49:28 -0700 Subject: [PATCH 27/35] Failed a re-supply that disagrees with History as the worker's. A task owed two streams and sent one came out as nondeterminism, which a worker configured for it fails the execution over. Both sides of that comparison come from the server, so no arm of it is the workflow's fault. --- crates/sdk-core/src/core_tests/streams.rs | 72 ++++++++++++++++--- crates/sdk-core/src/replay/mod.rs | 7 +- .../workflow/machines/workflow_machines.rs | 25 +++---- crates/sdk-core/src/worker/workflow/mod.rs | 10 +-- 4 files changed, 84 insertions(+), 30 deletions(-) diff --git a/crates/sdk-core/src/core_tests/streams.rs b/crates/sdk-core/src/core_tests/streams.rs index 035581c03..658d7848f 100644 --- a/crates/sdk-core/src/core_tests/streams.rs +++ b/crates/sdk-core/src/core_tests/streams.rs @@ -1107,7 +1107,8 @@ async fn an_empty_recorded_range_replays_without_a_slice_from_the_server() { /// A re-supplied slice is checked against the cursor it claims to satisfy. The /// event is the record of what the task saw, so bytes covering other offsets -/// would replay the task on different input than it ran on. +/// would replay the task on different input than it ran on. Both sides of that +/// comparison come from the server, so the task fails as the worker's failure. #[tokio::test] async fn a_resupplied_slice_that_disagrees_with_the_record_fails_the_task() { let mut t = TestHistoryBuilder::default(); @@ -1121,8 +1122,11 @@ async fn a_resupplied_slice_that_disagrees_with_the_record_fails_the_task() { // One record where the event says two. poll_resp.add_stream_slice("s1", completed, 0, &["a"]); - let (core, failures) = - worker_expecting_one_nondeterminism_failure(t, ResponseType::Raw(poll_resp.resp)); + let (core, failures) = worker_expecting_one_failure( + t, + ResponseType::Raw(poll_resp.resp), + WorkflowTaskFailedCause::WorkflowWorkerUnhandledFailure, + ); // The mismatch is found while the poll response is applied, so the first // activation is already the eviction. @@ -1211,6 +1215,52 @@ async fn a_missing_resupply_for_a_consumed_range_fails_the_task() { core.shutdown().await; } +/// A task that read two streams and a response that re-supplies only one of +/// them. The recorded range is the whole of what the task ran on, so a partial +/// re-supply leaves the workflow as short of input as none at all, and it is +/// the same worker failure: the history and the response both come from the +/// server, and the retry on the normal task queue carries every stream. +#[tokio::test] +async fn a_partial_resupply_for_a_consumed_task_fails_the_task() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + let completed = t.add_workflow_task_completed_with_consumed_stream_ranges(vec![ + cursor("s1", 0, 1), + cursor("s2", 0, 1), + ]); + t.add_workflow_task_scheduled_and_started(); + + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + // Only one of the two subscribed streams came back. + poll_resp.add_stream_slice("s1", completed, 0, &["one"]); + + let (core, failures) = worker_expecting_one_failure( + t, + ResponseType::Raw(poll_resp.resp), + WorkflowTaskFailedCause::WorkflowWorkerUnhandledFailure, + ); + + // Found while the poll response is applied, so the first activation is + // already the eviction. + // The reason travels in the message, since an eviction for a failure found + // before any activation ran reports none of its own. + let task = core.poll_workflow_activation().await.unwrap(); + let evict = eviction(&task); + assert!( + evict.message.contains( + "stream s2 was consumed from offset 0 to 1, but the server sent no \ + messages for it" + ), + "eviction did not name the stream that went missing: {evict:?}" + ); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + assert_eq!(failures.load(Ordering::Relaxed), 1); + core.shutdown().await; +} + /// A legacy query for a run this worker no longer holds arrives on the sticky /// queue with partial history and no records, and the history the worker /// fetches itself carries none either. Answering it from a replay on less @@ -1412,10 +1462,10 @@ async fn a_pushed_history_replays_with_the_slices_it_carries() { } /// A slice attached to a pushed history is held to the same check as one the -/// server sends: offsets other than the recorded range fail the task as -/// nondeterministic rather than replay it on different input. +/// server sends: offsets other than the recorded range fail the task rather +/// than replay it on different input. #[tokio::test] -async fn a_pushed_history_with_a_wrong_slice_fails_as_nondeterministic() { +async fn a_pushed_history_with_a_wrong_slice_fails_as_a_worker_failure() { let (t, first, second) = read_then_publish_history(); let history = HistoryForReplay::new(t.get_full_history_info().unwrap(), "wfid") .with_stream_slices([ @@ -1426,10 +1476,10 @@ async fn a_pushed_history_with_a_wrong_slice_fails_as_nondeterministic() { let (core, feeder) = replay_worker(history).await; // The mismatch is found while the poll response is applied, so the first - // activation is already the eviction. The task is failed as - // nondeterministic (the mock-client test above checks the cause); the - // eviction carries the reason in its message, since an eviction for a - // failure found before any activation ran reports no reason of its own. + // activation is already the eviction. The task is failed as the worker's + // failure (the mock-client test above checks the cause); the eviction + // carries the reason in its message, since an eviction for a failure found + // before any activation ran reports no reason of its own. let task = core.poll_workflow_activation().await.unwrap(); let evict = eviction(&task); assert!( @@ -1450,7 +1500,7 @@ async fn a_pushed_history_with_a_wrong_slice_fails_as_nondeterministic() { /// language replayer that has no store to fetch from cannot replay a consuming /// workflow, and the task says so before the workflow runs on less input. #[tokio::test] -async fn a_pushed_history_without_slices_for_a_consumed_range_fails_as_nondeterministic() { +async fn a_pushed_history_without_slices_for_a_consumed_range_fails_as_a_worker_failure() { let (t, _, _) = read_then_publish_history(); let history = HistoryForReplay::new(t.get_full_history_info().unwrap(), "wfid"); let (core, feeder) = replay_worker(history).await; diff --git a/crates/sdk-core/src/replay/mod.rs b/crates/sdk-core/src/replay/mod.rs index 34bae9043..692bc18b3 100644 --- a/crates/sdk-core/src/replay/mod.rs +++ b/crates/sdk-core/src/replay/mod.rs @@ -194,9 +194,10 @@ impl HistoryForReplay { /// read a stream cannot be replayed from its history alone. Each slice carries the records /// for one recorded range, tagged with the `WorkflowTaskCompleted` event that recorded it, /// the same shape the server puts on a poll response when it re-supplies them. Replay hands - /// each range to the activation of the task that consumed it and fails the task as - /// nondeterministic when a slice disagrees with the recorded range or a recorded range with - /// content has no slice. + /// each range to the activation of the task that consumed it. A slice that disagrees with + /// the recorded range, and a recorded range with content that has no slice, both fail the + /// task as the worker's failure rather than the workflow's: the history and the slices are + /// both given to replay, so neither says the workflow diverged. pub fn with_stream_slices(mut self, slices: impl IntoIterator) -> Self { self.stream_slices = slices.into_iter().collect(); self diff --git a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs index bbbf4bae0..4c97665d5 100644 --- a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs +++ b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs @@ -2040,6 +2040,11 @@ fn stream_order( /// bytes and is rebuilt from the cursor alone; a range with content has to /// arrive, and what arrives has to cover exactly the recorded offsets. Anything /// else would replay the task with different input than it ran on. +/// +/// Every disagreement here is between two things the server produced, History +/// on one side and the poll response on the other. The workflow's own commands +/// reach none of it, so none of these is the workflow's fault and none of them +/// is nondeterminism. fn resupplied_deliveries( event_id: i64, mut cursors: Vec, @@ -2060,7 +2065,7 @@ fn resupplied_deliveries( slice } Some(slice) => { - return Err(nondeterminism!( + return Err(WFMachinesError::MissingRecords(format!( "Event {event_id} records that stream {} was consumed from offset {} to {}, \ but the server sent offsets {} to {} for it", cursor.stream_id, @@ -2068,7 +2073,7 @@ fn resupplied_deliveries( cursor.to_offset, slice.from_offset, slice.to_offset - )); + ))); } None if cursor.from_offset == cursor.to_offset => StreamSlice { stream_id: cursor.stream_id, @@ -2077,25 +2082,21 @@ fn resupplied_deliveries( ..Default::default() }, None => { - return Err(nondeterminism!( + return Err(WFMachinesError::MissingRecords(format!( "Event {event_id} records that stream {} was consumed from offset {} to {}, \ but the server sent no messages for it", - cursor.stream_id, - cursor.from_offset, - cursor.to_offset - )); + cursor.stream_id, cursor.from_offset, cursor.to_offset + ))); } }; jobs.push(deliver_stream_records_job(slice)); } if let Some(extra) = slices.first() { - return Err(nondeterminism!( + return Err(WFMachinesError::MissingRecords(format!( "The server sent stream {} from offset {} to {} for event {event_id}, which records \ no such range", - extra.stream_id, - extra.from_offset, - extra.to_offset - )); + extra.stream_id, extra.from_offset, extra.to_offset + ))); } Ok(jobs) } diff --git a/crates/sdk-core/src/worker/workflow/mod.rs b/crates/sdk-core/src/worker/workflow/mod.rs index aa2cef0af..d0424b2ab 100644 --- a/crates/sdk-core/src/worker/workflow/mod.rs +++ b/crates/sdk-core/src/worker/workflow/mod.rs @@ -1672,10 +1672,12 @@ pub(crate) enum WFMachinesError { #[error("Fatal error in workflow machines: {0}")] Fatal(String), /// History records that a task consumed stream records and the response that carried the - /// task brought none of them, so this worker cannot replay the run. Not the workflow's - /// fault: the records only travel with the task, and a worker handed a sticky task for a run - /// it no longer holds has no way to fetch them. Treated like a failed history fetch, so a - /// legacy query goes unanswered and the server retries it where the records travel. + /// task did not bring them, so this worker cannot replay the run. Covers a response that + /// brought none of them and one whose records do not cover what History says the task read. + /// Not the workflow's fault either way: both sides of that comparison come from the server, + /// and a worker handed a sticky task for a run it no longer holds has no way to fetch the + /// records. Treated like a failed history fetch, so a legacy query goes unanswered and the + /// server retries it where the records travel. #[error("Workflow task cannot be replayed on this worker: {0}")] MissingRecords(String), } From e3ee792237f101d2b1f5443c08a53824b21c5fda Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 25 Sep 2026 13:49:36 -0700 Subject: [PATCH 28/35] Scoped the data-only task flag to the task it describes. The flag that says a completion consumed a range was written into the two that carry across the whole scan, which only works while every path after it returns. A local one says what it means on its own. --- .../src/worker/workflow/history_update.rs | 15 +++++++++++---- 1 file changed, 11 insertions(+), 4 deletions(-) diff --git a/crates/sdk-core/src/worker/workflow/history_update.rs b/crates/sdk-core/src/worker/workflow/history_update.rs index 4100b6e2e..188ccc169 100644 --- a/crates/sdk-core/src/worker/workflow/history_update.rs +++ b/crates/sdk-core/src/worker/workflow/history_update.rs @@ -745,6 +745,10 @@ fn find_end_index_of_next_wft_seq( } if e.event_type() == EventType::WorkflowTaskStarted { + // Scoped to this started event. What its own completion consumed says nothing + // about the events the scan already passed, so it must not join the flags that + // carry across them. + let mut completion_consumed_a_range = false; wft_started_event_id_to_index.push((e.event_id, ix)); if let Some(next_event) = events.get(ix + 1) { let next_event_type = next_event.event_type(); @@ -771,11 +775,11 @@ fn find_end_index_of_next_wft_seq( next_event.attributes && !attrs.consumed_stream_ranges.is_empty() { - saw_command = true; - saw_command_or_started = true; + completion_consumed_a_range = true; } if let Some(next_next_event) = events.get(ix + 2) { if !saw_command + && !completion_consumed_a_range && next_next_event.event_type() == EventType::WorkflowTaskScheduled { // If we've never seen an interesting event and the next two events are @@ -817,7 +821,10 @@ fn find_end_index_of_next_wft_seq( } return NextWFTSeqEndIndex::Complete(ix); } - } else if !has_last_wft && !saw_command_or_started { + } else if !has_last_wft + && !saw_command_or_started + && !completion_consumed_a_range + { // Don't have enough events to look ahead of the WorkflowTaskCompleted. Need // to fetch more. continue; @@ -828,7 +835,7 @@ fn find_end_index_of_next_wft_seq( // more. continue; } - if saw_command_or_started { + if saw_command_or_started || completion_consumed_a_range { return NextWFTSeqEndIndex::Complete(ix); } } From 0592424556de600438674bc379a78790a7f83751 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 25 Sep 2026 13:49:36 -0700 Subject: [PATCH 29/35] Said why the extra boundary page cannot be fetched more narrowly. The fetch costs a page for any run whose page boundary lands there. A subscription made through the stream service records no event, so there is no telling a stream run from any other before the ranges arrive. --- crates/sdk-core/src/worker/workflow/history_update.rs | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/crates/sdk-core/src/worker/workflow/history_update.rs b/crates/sdk-core/src/worker/workflow/history_update.rs index 188ccc169..5693cd1a3 100644 --- a/crates/sdk-core/src/worker/workflow/history_update.rs +++ b/crates/sdk-core/src/worker/workflow/history_update.rs @@ -315,6 +315,11 @@ impl HistoryPaginator { // and the workflow has to be handed that range in the same activation. A page that // ends exactly on a WFT started event leaves that completion on the next page, so // fetch it before handing the update over. + // + // The fetch costs a page for any run whose page boundary lands here, stream or not. + // There is no telling the two apart from here: a subscription made through the + // stream service records no event, so a run can consume ranges with nothing earlier + // in its history to say it would. if !no_more && self.event_queue.is_empty() { self.event_queue.extend(update.events); continue; From d321426250ee736cbff8121765e3579d0a9af969 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 25 Sep 2026 15:40:04 -0700 Subject: [PATCH 30/35] Carried the stream field rename into the delivery tests. The delivery tests build their own commands and histories, so they name the fields directly. One recorded append passed a count where the event now wants an end offset. --- crates/sdk-core/src/core_tests/streams.rs | 18 +++++++++--------- 1 file changed, 9 insertions(+), 9 deletions(-) diff --git a/crates/sdk-core/src/core_tests/streams.rs b/crates/sdk-core/src/core_tests/streams.rs index cd0269766..dfd2ae0be 100644 --- a/crates/sdk-core/src/core_tests/streams.rs +++ b/crates/sdk-core/src/core_tests/streams.rs @@ -533,7 +533,7 @@ async fn read_then_publish_replays_when_the_range_is_only_visible_by_lookahead() task.run_id, vec![ AppendStreamRecords { - stream_id: "out".to_string(), + stream_name: "out".to_string(), records: vec![StreamRecord { body: Some(b"accept".to_vec().into()), ..Default::default() @@ -592,7 +592,7 @@ async fn read_then_publish_replays_after_a_task_that_consumed_nothing() { task.run_id, vec![ SubscribeStream { - stream_id: "in".to_string(), + stream_name_or_id: "in".to_string(), start_offset: 0, } .into(), @@ -624,7 +624,7 @@ async fn read_then_publish_replays_after_a_task_that_consumed_nothing() { task.run_id, vec![ AppendStreamRecords { - stream_id: "out".to_string(), + stream_name: "out".to_string(), records: vec![StreamRecord { body: Some(b"accept".to_vec().into()), ..Default::default() @@ -708,7 +708,7 @@ async fn read_then_publish_replays_across_a_page_boundary( task.run_id, vec![ SubscribeStream { - stream_id: "in".to_string(), + stream_name_or_id: "in".to_string(), start_offset: 0, } .into(), @@ -738,7 +738,7 @@ async fn read_then_publish_replays_across_a_page_boundary( task.run_id, vec![ AppendStreamRecords { - stream_id: "out".to_string(), + stream_name: "out".to_string(), records: vec![StreamRecord { body: Some(b"accept".to_vec().into()), ..Default::default() @@ -1279,7 +1279,7 @@ async fn a_legacy_query_owed_records_it_was_not_sent_goes_unanswered() { task.run_id, vec![ SubscribeStream { - stream_id: "in".to_string(), + stream_name_or_id: "in".to_string(), start_offset: 0, } .into(), @@ -1330,9 +1330,9 @@ fn replay_slice( } /// A publish of one record per body. -fn publish(stream_id: &str, bodies: &[&str]) -> AppendStreamRecords { +fn publish(stream_name: &str, bodies: &[&str]) -> AppendStreamRecords { AppendStreamRecords { - stream_id: stream_id.to_string(), + stream_name: stream_name.to_string(), records: bodies .iter() .map(|b| StreamRecord { @@ -1365,7 +1365,7 @@ fn read_then_publish_history() -> (TestHistoryBuilder, i64, i64) { t.add_workflow_task_scheduled_and_started(); let second = t.add_workflow_task_completed_with_consumed_stream_ranges(vec![cursor("in", 1, 2)]); - t.add_stream_records_appended("out", 1, 1); + t.add_stream_records_appended("out", 1, 2); t.add_workflow_execution_completed(); (t, first, second) } From c10921b9a2b454c21be0760647b16b57538db099 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 25 Sep 2026 22:26:14 -0700 Subject: [PATCH 31/35] Moved the stream changelog entries back under Unreleased. The main merge introduced the released headings above them and the resolution left the entries inside 0.8.0. The checkpoint requires additions under Unreleased. --- crates/sdk-core/CHANGELOG.md | 25 +++++++++++++------------ 1 file changed, 13 insertions(+), 12 deletions(-) diff --git a/crates/sdk-core/CHANGELOG.md b/crates/sdk-core/CHANGELOG.md index b15a44746..1e113ae71 100644 --- a/crates/sdk-core/CHANGELOG.md +++ b/crates/sdk-core/CHANGELOG.md @@ -33,6 +33,19 @@ relevant information. ## Unreleased +### Added +* Workflows can subscribe to server-side streams and append batches of records to them with the + `SubscribeStream` and `AppendStreamRecords` commands. Consumed ranges reach the workflow as + `DeliverStreamRecords` activation jobs, and replay hands each recorded range back in the + activation of the task that consumed it. +* A history fed to a replay worker can carry the stream records its tasks consumed + (`HistoryForReplay::with_stream_slices`), so a language replayer that fetched them from the + stream service can replay a consuming workflow. History alone holds only the offsets. +* A task whose history records a consumed range with content that the response carried no + records for fails before the workflow runs, rather than after it ran on less input. A legacy + query dispatched that way to a worker that no longer holds the run goes unanswered, so the + server retries it on the normal task queue, where the records travel with it. + ### Fixed * Task-poll targets no longer decrease after cancelled or timed-out polls. Affected pollers still retain their slot during backoff, while resource-exhaustion errors still reduce the target. @@ -68,18 +81,6 @@ relevant information. metrics now carry a `failure_reason` attribute. Each is now split into one time series per reason, which may affect existing dashboards. * Workflow task completions larger than the gRPC request size limit are now paginated automatically when the namespace supports it. Paginated workflow task completions require Temporal Server 1.32.0 or later. -* Workflows can subscribe to server-side streams and append batches of records to them with the - `SubscribeStream` and `AppendStreamRecords` commands. Consumed ranges reach the workflow as - `DeliverStreamRecords` activation jobs, and replay hands each recorded range back in the - activation of the task that consumed it. -* A history fed to a replay worker can carry the stream records its tasks consumed - (`HistoryForReplay::with_stream_slices`), so a language replayer that fetched them from the - stream service can replay a consuming workflow. History alone holds only the offsets. -* A task whose history records a consumed range with content that the response carried no - records for fails before the workflow runs, rather than after it ran on less input. A legacy - query dispatched that way to a worker that no longer holds the run goes unanswered, so the - server retries it on the normal task queue, where the records travel with it. - ### Breaking Changes :boom: * The following types are now non-exhaustive: `Priority`, `WorkerDeploymentVersion`, `WorkerCallbacks`, `WorkflowExecutionInfo`, `ActivityCloseTimeouts`, From 8770d6ffcb3e821d915db6355206e9c79c867058 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Thu, 1 Oct 2026 17:52:43 -0700 Subject: [PATCH 32/35] Delivered channel notifications from the scheduled event. --- .../workflow_activation.proto | 4 +- crates/sdk-core/src/core_tests/channels.rs | 183 +++++++++++++++++- crates/sdk-core/src/replay/history_builder.rs | 13 ++ .../src/worker/workflow/history_update.rs | 11 ++ .../workflow/machines/workflow_machines.rs | 13 ++ crates/sdk-core/src/worker/workflow/mod.rs | 4 +- 6 files changed, 223 insertions(+), 5 deletions(-) diff --git a/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto b/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto index 08d13701a..ba94067f1 100644 --- a/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto +++ b/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto @@ -31,7 +31,7 @@ import "temporal/sdk/core/nexus/nexus.proto"; // 1. init workflow // 2. patches // 3. random-seed-updates -// 4. signals/updates +// 4. signals/updates/channel notifications // 5. all others // 6. local activity resolutions // 7. queries @@ -41,6 +41,8 @@ import "temporal/sdk/core/nexus/nexus.proto"; // * Patches are expected to apply to the entire activation // * Signal and update handlers should be invoked before workflow routines are iterated. That is to // say before the users' main workflow function and anything spawned by it is allowed to continue. +// * Channel notifications are input from outside the workflow, like signals, so they go with +// them and ahead of the stream ranges among the other jobs. // * Local activities resolutions go after other normal jobs because while *not* replaying, they // will always take longer than anything else that produces an immediate job (which is // effectively instant). When *replaying* we need to scan ahead for LA markers so that we can diff --git a/crates/sdk-core/src/core_tests/channels.rs b/crates/sdk-core/src/core_tests/channels.rs index d0c6de5fd..b1a05937e 100644 --- a/crates/sdk-core/src/core_tests/channels.rs +++ b/crates/sdk-core/src/core_tests/channels.rs @@ -1,6 +1,10 @@ use crate::{ - replay::TestHistoryBuilder, - test_help::{MockPollCfg, ResponseType, WorkerTestHelpers, build_mock_pollers, mock_worker}, + init_replay_worker, + replay::{HistoryFeeder, HistoryForReplay, ReplayWorkerInput, TestHistoryBuilder}, + test_help::{ + MockPollCfg, PollWFTRespExt, ResponseType, WorkerTestHelpers, build_mock_pollers, + hist_to_poll_resp, mock_worker, test_worker_cfg, + }, worker::client::mocks::mock_worker_client, }; use std::sync::{ @@ -9,12 +13,14 @@ use std::sync::{ }; use temporalio_common::protos::{ coresdk::{ - workflow_commands::SubscribeNotificationChannel, + workflow_activation::{WorkflowActivation, workflow_activation_job}, + workflow_commands::{CompleteWorkflowExecution, SubscribeNotificationChannel}, workflow_completion::WorkflowActivationCompletion, }, temporal::api::{ command::v1::command, enums::v1::{CommandType, EventType, WorkflowTaskFailedCause}, + notification::v1::Notification, workflowservice::v1::RespondWorkflowTaskCompletedResponse, }, }; @@ -132,3 +138,174 @@ async fn a_subscribe_reissued_to_a_different_channel_fails_the_task() { assert_eq!(failures.load(Ordering::Relaxed), 1); core.shutdown().await; } + +fn notification(channel: &str, counter: i64) -> Notification { + Notification { + channel: channel.to_string(), + position: format!("pos-{counter}").into_bytes(), + counter, + ..Default::default() + } +} + +/// The notification jobs in an activation, each as the notifications it carries. +fn received(task: &WorkflowActivation) -> Vec> { + task.jobs + .iter() + .filter_map(|j| match j.variant.as_ref() { + Some(workflow_activation_job::Variant::NotificationsReceived(n)) => { + Some(n.notifications.clone()) + } + _ => None, + }) + .collect() +} + +/// A short name for each job, so a test can assert their order. +fn job_kinds(task: &WorkflowActivation) -> Vec<&'static str> { + task.jobs + .iter() + .map(|j| match j.variant.as_ref().unwrap() { + workflow_activation_job::Variant::InitializeWorkflow(_) => "init", + workflow_activation_job::Variant::NotificationsReceived(_) => "notifications", + workflow_activation_job::Variant::DeliverStreamRecords(_) => "stream", + _ => "other", + }) + .collect() +} + +/// A run whose only task was scheduled with notifications for two channels. +fn one_task_with_notifications() -> TestHistoryBuilder { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_with_notifications(vec![ + notification("orders", 3), + notification("invoices", 7), + ]); + t.add_workflow_task_started(); + t +} + +/// The notifications on the scheduled event reach the activation of that task, +/// as one job, ahead of the stream ranges the same task was handed. +#[tokio::test] +async fn notifications_on_the_scheduled_event_reach_the_live_activation() { + let t = one_task_with_notifications(); + let mut poll_resp = hist_to_poll_resp(&t, "wfid".to_owned(), ResponseType::AllHistory); + poll_resp.add_stream_slice("s1", 0, 0, &["alpha"]); + + let mock = MockPollCfg::from_resp_batches( + "wfid", + t, + [ResponseType::Raw(poll_resp.resp)], + mock_worker_client(), + ); + let core = mock_worker(build_mock_pollers(mock)); + + let task = core.poll_workflow_activation().await.unwrap(); + assert!(!task.is_replaying); + assert_eq!(job_kinds(&task), vec!["init", "notifications", "stream"]); + assert_eq!( + received(&task), + vec![vec![notification("orders", 3), notification("invoices", 7)]] + ); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); +} + +/// Replaying the same history yields the same job. Nothing is re-supplied: the +/// notifications are on the event, so History alone carries them. +#[tokio::test] +async fn a_replayed_history_yields_the_same_notifications() { + let mut t = one_task_with_notifications(); + t.add_workflow_task_completed(); + t.add_workflow_execution_completed(); + + let (feeder, stream) = HistoryFeeder::new(1); + feeder + .feed(HistoryForReplay::new( + t.get_full_history_info().unwrap(), + "wfid", + )) + .await + .unwrap(); + let core = init_replay_worker(ReplayWorkerInput::new( + test_worker_cfg().build().unwrap(), + stream, + )) + .unwrap(); + + let task = core.poll_workflow_activation().await.unwrap(); + assert!(task.is_replaying); + assert_eq!(job_kinds(&task), vec!["init", "notifications"]); + assert_eq!( + received(&task), + vec![vec![notification("orders", 3), notification("invoices", 7)]] + ); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![CompleteWorkflowExecution { result: None }.into()], + )) + .await + .unwrap(); + drop(feeder); + core.shutdown().await; +} + +/// On a later task the job lands in that task's activation and not in an +/// earlier one, also when the worker replays its way there. The first task +/// issued no command, which replay would otherwise read as a heartbeat and +/// fold together with the next one. +#[tokio::test] +async fn notifications_belong_to_the_task_whose_scheduled_event_carries_them() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_full_wf_task(); + t.add_workflow_task_scheduled_with_notifications(vec![notification("orders", 1)]); + t.add_workflow_task_started(); + t.add_workflow_task_completed(); + t.add_we_signaled("sig", vec![]); + t.add_workflow_task_scheduled_and_started(); + + let mock = + MockPollCfg::from_resp_batches("wfid", t, [ResponseType::AllHistory], mock_worker_client()); + let core = mock_worker(build_mock_pollers(mock)); + + let task = core.poll_workflow_activation().await.unwrap(); + assert!(received(&task).is_empty()); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + + let task = core.poll_workflow_activation().await.unwrap(); + assert!(task.is_replaying); + assert_eq!(received(&task), vec![vec![notification("orders", 1)]]); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + + let task = core.poll_workflow_activation().await.unwrap(); + assert!(received(&task).is_empty()); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); +} + +/// A scheduled event without notifications produces no job, not an empty one. +#[tokio::test] +async fn a_scheduled_event_without_notifications_yields_no_job() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + + let mock = + MockPollCfg::from_resp_batches("wfid", t, [ResponseType::AllHistory], mock_worker_client()); + let core = mock_worker(build_mock_pollers(mock)); + + let task = core.poll_workflow_activation().await.unwrap(); + assert_eq!(job_kinds(&task), vec!["init"]); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); +} diff --git a/crates/sdk-core/src/replay/history_builder.rs b/crates/sdk-core/src/replay/history_builder.rs index 0e15d7af7..f2fdbc978 100644 --- a/crates/sdk-core/src/replay/history_builder.rs +++ b/crates/sdk-core/src/replay/history_builder.rs @@ -23,6 +23,7 @@ use temporalio_common::protos::{ enums::v1::{EventType, TaskQueueKind, WorkflowTaskFailedCause}, failure::v1::{CanceledFailureInfo, Failure, failure}, history::v1::{history_event::Attributes, *}, + notification::v1::Notification, stream::v1::StreamRange, taskqueue::v1::TaskQueue, update, @@ -112,6 +113,18 @@ impl TestHistoryBuilder { self.workflow_task_scheduled_event_id = self.add_by_type(EventType::WorkflowTaskScheduled); } + /// Add a workflow task scheduled event carrying the notifications the server + /// folded for the task. + pub fn add_workflow_task_scheduled_with_notifications( + &mut self, + notifications: Vec, + ) { + self.workflow_task_scheduled_event_id = self.add(WorkflowTaskScheduledEventAttributes { + notifications, + ..Default::default() + }); + } + /// Add a workflow task started event. pub fn add_workflow_task_started(&mut self) { self.final_workflow_task_started_event_id = self.add(WorkflowTaskStartedEventAttributes { diff --git a/crates/sdk-core/src/worker/workflow/history_update.rs b/crates/sdk-core/src/worker/workflow/history_update.rs index 5693cd1a3..972c5a8f7 100644 --- a/crates/sdk-core/src/worker/workflow/history_update.rs +++ b/crates/sdk-core/src/worker/workflow/history_update.rs @@ -715,6 +715,16 @@ impl NextWFTSeqEndIndex { } } +/// A task scheduled with notifications was handed them as an activation of its own, so it +/// cannot be folded into the heartbeat chain before it the way an empty task is. +fn scheduled_with_notifications(e: &HistoryEvent) -> bool { + matches!( + e.attributes, + Some(Attributes::WorkflowTaskScheduledEventAttributes(ref attrs)) + if !attrs.notifications.is_empty() + ) +} + /// Discovers the index of the last event in next WFT sequence within the passed-in slice /// For more on workflow task chunking, see arch_docs/workflow_task_chunking.md fn find_end_index_of_next_wft_seq( @@ -786,6 +796,7 @@ fn find_end_index_of_next_wft_seq( if !saw_command && !completion_consumed_a_range && next_next_event.event_type() == EventType::WorkflowTaskScheduled + && !scheduled_with_notifications(next_next_event) { // If we've never seen an interesting event and the next two events are // a completion followed immediately again by scheduled, then this is a diff --git a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs index ba6c8df99..965c6e913 100644 --- a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs +++ b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs @@ -1184,10 +1184,23 @@ impl WorkflowMachines { } } Ok(EventType::WorkflowTaskScheduled) => { + // The notifications ride on the event itself, so the same job + // comes out of it live and on replay with nothing to re-supply. + let notifications = match &event_dat.event.attributes { + Some(history_event::Attributes::WorkflowTaskScheduledEventAttributes(a)) => { + a.notifications.clone() + } + _ => vec![], + }; let wf_task_sm = WorkflowTaskMachine::new(self.next_started_event_id); let key = self.all_machines.insert(wf_task_sm.into()); self.submachine_handle_event(key, event_dat)?; self.machines_by_event_id.insert(event_id, key); + if !notifications.is_empty() { + self.drive_me.send_job( + workflow_activation::NotificationsReceived { notifications }.into(), + ); + } } Ok(EventType::WorkflowExecutionSignaled) => { if let Some(history_event::Attributes::WorkflowExecutionSignaledEventAttributes( diff --git a/crates/sdk-core/src/worker/workflow/mod.rs b/crates/sdk-core/src/worker/workflow/mod.rs index 8c047f1b4..c8c198725 100644 --- a/crates/sdk-core/src/worker/workflow/mod.rs +++ b/crates/sdk-core/src/worker/workflow/mod.rs @@ -1875,7 +1875,7 @@ impl LocalActivityRequestSink for LAReqSink { /// 1. init workflow /// 2. patches /// 3. random-seed-updates -/// 4. signals/updates +/// 4. signals/updates/channel notifications /// 5. all others /// 6. local activity resolutions /// 7. queries @@ -1915,6 +1915,8 @@ fn prepare_to_ship_activation(wfa: &mut WorkflowActivation) { workflow_activation_job::Variant::UpdateRandomSeed(_) => 2, workflow_activation_job::Variant::SignalWorkflow(_) => 3, workflow_activation_job::Variant::DoUpdate(_) => 3, + // Ahead of the stream ranges, which fall in the default bucket below. + workflow_activation_job::Variant::NotificationsReceived(_) => 3, workflow_activation_job::Variant::ResolveActivity(ra) if ra.is_local => 5, // In principle we should never actually need to sort these with the others, since // queries always get their own activation, but, maintaining the semantic is From 7b40358c33763c69dc9b0edcfcade3fda0dbfb02 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Thu, 1 Oct 2026 18:50:46 -0700 Subject: [PATCH 33/35] Folded a task's channel notifications into one job per channel. --- crates/sdk-core/src/core_tests/channels.rs | 104 ++++++++++++++++++ .../workflow/machines/workflow_machines.rs | 44 ++++++-- 2 files changed, 138 insertions(+), 10 deletions(-) diff --git a/crates/sdk-core/src/core_tests/channels.rs b/crates/sdk-core/src/core_tests/channels.rs index b1a05937e..6da6d260f 100644 --- a/crates/sdk-core/src/core_tests/channels.rs +++ b/crates/sdk-core/src/core_tests/channels.rs @@ -20,6 +20,7 @@ use temporalio_common::protos::{ temporal::api::{ command::v1::command, enums::v1::{CommandType, EventType, WorkflowTaskFailedCause}, + failure::v1::Failure, notification::v1::Notification, workflowservice::v1::RespondWorkflowTaskCompletedResponse, }, @@ -309,3 +310,106 @@ async fn a_scheduled_event_without_notifications_yields_no_job() { .await .unwrap(); } + +/// The first activation a replay worker makes of a history that ends with the run +/// completing, and that worker, so the test can finish the run on it. +async fn replayed_first_activation( + mut t: TestHistoryBuilder, +) -> (crate::Worker, HistoryFeeder, WorkflowActivation) { + t.add_workflow_task_completed(); + t.add_workflow_execution_completed(); + let (feeder, stream) = HistoryFeeder::new(1); + feeder + .feed(HistoryForReplay::new( + t.get_full_history_info().unwrap(), + "wfid", + )) + .await + .unwrap(); + let core = init_replay_worker(ReplayWorkerInput::new( + test_worker_cfg().build().unwrap(), + stream, + )) + .unwrap(); + let task = core.poll_workflow_activation().await.unwrap(); + (core, feeder, task) +} + +/// The first activation of a history served live, as one poll response. +async fn live_first_activation(t: TestHistoryBuilder) -> WorkflowActivation { + let mock = + MockPollCfg::from_resp_batches("wfid", t, [ResponseType::AllHistory], mock_worker_client()); + let core = mock_worker(build_mock_pollers(mock)); + let task = core.poll_workflow_activation().await.unwrap(); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id.clone())) + .await + .unwrap(); + task +} + +/// A first task that failed, scheduled with one notification on channel A, and its +/// retry, scheduled with a newer one on A and one on B. +fn failed_task_then_retry() -> TestHistoryBuilder { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_with_notifications(vec![notification("a", 1)]); + t.add_workflow_task_started(); + t.add_workflow_task_failed_with_failure( + WorkflowTaskFailedCause::Unspecified, + Failure::default(), + ); + t.add_workflow_task_scheduled_with_notifications(vec![ + notification("a", 5), + notification("b", 2), + ]); + t.add_workflow_task_started(); + t +} + +/// The server clears what it put on a scheduled event, so the failed task's +/// notification is only in History. The retry's activation gets it folded with +/// the retry's own, one per channel, live and on replay alike. +#[tokio::test] +async fn a_failed_task_and_its_retry_yield_one_job_folded_per_channel() { + let expected = vec![vec![notification("a", 5), notification("b", 2)]]; + + let task = live_first_activation(failed_task_then_retry()).await; + assert!(!task.is_replaying); + assert_eq!(job_kinds(&task), vec!["init", "notifications"]); + assert_eq!(received(&task), expected); + + let (core, feeder, task) = replayed_first_activation(failed_task_then_retry()).await; + assert!(task.is_replaying); + assert_eq!(job_kinds(&task), vec!["init", "notifications"]); + assert_eq!(received(&task), expected); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![CompleteWorkflowExecution { result: None }.into()], + )) + .await + .unwrap(); + drop(feeder); + core.shutdown().await; +} + +/// A notification with a lower counter than the one already held for its channel +/// does not replace it, and channels keep the order they first appeared in. +#[tokio::test] +async fn a_lower_counter_does_not_replace_the_held_notification() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_with_notifications(vec![notification("a", 5)]); + t.add_workflow_task_started(); + t.add_workflow_task_timed_out(); + t.add_workflow_task_scheduled_with_notifications(vec![ + notification("b", 1), + notification("a", 3), + ]); + t.add_workflow_task_started(); + + let task = live_first_activation(t).await; + assert_eq!( + received(&task), + vec![vec![notification("a", 5), notification("b", 1)]] + ); +} diff --git a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs index 965c6e913..67328cad3 100644 --- a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs +++ b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs @@ -74,6 +74,7 @@ use temporalio_common::{ common::v1::SearchAttributes, enums::v1::EventType, history::v1::{HistoryEvent, history_event}, + notification::v1::Notification, protocol::v1::{Message as ProtocolMessage, message::SequencingId}, sdk::v1::WorkflowTaskCompletedMetadata, stream::v1::{StreamRange, StreamSlice}, @@ -116,6 +117,11 @@ pub(crate) struct WorkflowMachines { /// not say so: it is true for a cache hit as well, because the history /// begins after an earlier task. stream_slices_delivered_through: i64, + /// Channel notifications from the scheduled events of the task being applied, folded per + /// channel. The server clears what it put on a scheduled event, so a retry's event carries + /// only later arrivals, and a failed task's notifications reach lang only here, together + /// with the retry's. + pending_notifications: Vec, /// Protocol messages that have yet to be processed for the current WFT. protocol_msgs: Vec, /// EventId of the last handled WorkflowTaskStarted event @@ -314,6 +320,7 @@ impl WorkflowMachines { current_stream_slices: Default::default(), stream_slices_delivered_through: 0, stream_slices_lookahead_through: 0, + pending_notifications: vec![], replaying, metrics: basics.metrics, // In an ideal world one could say ..Default::default() here and it'd still work. @@ -824,6 +831,13 @@ impl WorkflowMachines { } self.last_processed_event = eid; } + // One job for every scheduled event in the batch, so a failed task and its retry + // reach lang the way the server folds them. + if !self.pending_notifications.is_empty() { + let notifications = std::mem::take(&mut self.pending_notifications); + self.drive_me + .send_job(workflow_activation::NotificationsReceived { notifications }.into()); + } // Needed to delay mutation of self until after we've iterated over peeked events. #[allow(clippy::large_enum_variant)] @@ -1186,21 +1200,17 @@ impl WorkflowMachines { Ok(EventType::WorkflowTaskScheduled) => { // The notifications ride on the event itself, so the same job // comes out of it live and on replay with nothing to re-supply. - let notifications = match &event_dat.event.attributes { - Some(history_event::Attributes::WorkflowTaskScheduledEventAttributes(a)) => { - a.notifications.clone() + if let Some(history_event::Attributes::WorkflowTaskScheduledEventAttributes(a)) = + &event_dat.event.attributes + { + for n in &a.notifications { + fold_notification(&mut self.pending_notifications, n.clone()); } - _ => vec![], - }; + } let wf_task_sm = WorkflowTaskMachine::new(self.next_started_event_id); let key = self.all_machines.insert(wf_task_sm.into()); self.submachine_handle_event(key, event_dat)?; self.machines_by_event_id.insert(event_id, key); - if !notifications.is_empty() { - self.drive_me.send_job( - workflow_activation::NotificationsReceived { notifications }.into(), - ); - } } Ok(EventType::WorkflowExecutionSignaled) => { if let Some(history_event::Attributes::WorkflowExecutionSignaledEventAttributes( @@ -2025,6 +2035,20 @@ enum CommandIdKind { NeverResolves, } +/// Keep one notification per channel, in order of first appearance. A later one replaces +/// the held one when its counter is at least as high, the same rule the server folds by, +/// so a tie goes to the newer metadata. +fn fold_notification(folded: &mut Vec, n: Notification) { + match folded.iter_mut().find(|held| held.channel == n.channel) { + Some(held) => { + if n.counter >= held.counter { + *held = n; + } + } + None => folded.push(n), + } +} + /// Turn a slice the server supplied into the job lang sees. fn deliver_stream_records_job(slice: StreamSlice) -> OutgoingJob { workflow_activation::DeliverStreamRecords { From 1720745668086de0057e1daafa44317ec99fa532 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Thu, 1 Oct 2026 19:00:02 -0700 Subject: [PATCH 34/35] Kept the held channel notification on a counter tie. --- crates/sdk-core/src/core_tests/channels.rs | 23 +++++++++++++++++++ .../workflow/machines/workflow_machines.rs | 6 ++--- 2 files changed, 26 insertions(+), 3 deletions(-) diff --git a/crates/sdk-core/src/core_tests/channels.rs b/crates/sdk-core/src/core_tests/channels.rs index 6da6d260f..fcb10d423 100644 --- a/crates/sdk-core/src/core_tests/channels.rs +++ b/crates/sdk-core/src/core_tests/channels.rs @@ -413,3 +413,26 @@ async fn a_lower_counter_does_not_replace_the_held_notification() { vec![vec![notification("a", 5), notification("b", 1)]] ); } + +/// On a counter tie the notification already held for the channel stays. +#[tokio::test] +async fn a_counter_tie_keeps_the_held_notification() { + let held = Notification { + position: b"held".to_vec(), + ..notification("a", 4) + }; + let tied = Notification { + position: b"tied".to_vec(), + ..notification("a", 4) + }; + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_with_notifications(vec![held.clone()]); + t.add_workflow_task_started(); + t.add_workflow_task_timed_out(); + t.add_workflow_task_scheduled_with_notifications(vec![tied]); + t.add_workflow_task_started(); + + let task = live_first_activation(t).await; + assert_eq!(received(&task), vec![vec![held]]); +} diff --git a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs index 67328cad3..17989b895 100644 --- a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs +++ b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs @@ -2036,12 +2036,12 @@ enum CommandIdKind { } /// Keep one notification per channel, in order of first appearance. A later one replaces -/// the held one when its counter is at least as high, the same rule the server folds by, -/// so a tie goes to the newer metadata. +/// the held one only when its counter is strictly higher, so on a tie the held one stays, +/// which is the rule the external lineage folds by. fn fold_notification(folded: &mut Vec, n: Notification) { match folded.iter_mut().find(|held| held.channel == n.channel) { Some(held) => { - if n.counter >= held.counter { + if n.counter > held.counter { *held = n; } } From 7661e3a5cb297043c6aef15c54b0e4eb1a1b2bc8 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 2 Oct 2026 03:08:41 -0700 Subject: [PATCH 35/35] Added the unsubscribe-notification-channel command machine. The server records an event for every unsubscribe, also for a channel the run never subscribed to, so the machine matches by channel alone. The notification job stays as it is, since Core keeps no view of the run's subscriptions and lang drops a late notification for a closed one. --- crates/protos/src/protos/mod.rs | 10 + crates/sdk-core/src/core_tests/channels.rs | 184 +++++++++++++++++- crates/sdk-core/src/replay/history_builder.rs | 16 ++ .../src/worker/workflow/machines/mod.rs | 3 + .../workflow/machines/transition_coverage.rs | 6 + ...ribe_notification_channel_state_machine.rs | 144 ++++++++++++++ .../workflow/machines/workflow_machines.rs | 10 + crates/sdk-core/src/worker/workflow/mod.rs | 10 +- 8 files changed, 376 insertions(+), 7 deletions(-) create mode 100644 crates/sdk-core/src/worker/workflow/machines/unsubscribe_notification_channel_state_machine.rs diff --git a/crates/protos/src/protos/mod.rs b/crates/protos/src/protos/mod.rs index 406d7c6a6..2d6c67357 100644 --- a/crates/protos/src/protos/mod.rs +++ b/crates/protos/src/protos/mod.rs @@ -1962,6 +1962,16 @@ pub mod temporal { } } + impl From for Attributes { + fn from(s: workflow_commands::UnsubscribeNotificationChannel) -> Self { + Self::UnsubscribeNotificationChannelCommandAttributes( + UnsubscribeNotificationChannelCommandAttributes { + channel: s.channel, + }, + ) + } + } + impl From for command::Attributes { fn from(s: workflow_commands::StartTimer) -> Self { Self::StartTimerCommandAttributes(StartTimerCommandAttributes { diff --git a/crates/sdk-core/src/core_tests/channels.rs b/crates/sdk-core/src/core_tests/channels.rs index fcb10d423..4cb2fee9c 100644 --- a/crates/sdk-core/src/core_tests/channels.rs +++ b/crates/sdk-core/src/core_tests/channels.rs @@ -14,7 +14,9 @@ use std::sync::{ use temporalio_common::protos::{ coresdk::{ workflow_activation::{WorkflowActivation, workflow_activation_job}, - workflow_commands::{CompleteWorkflowExecution, SubscribeNotificationChannel}, + workflow_commands::{ + CompleteWorkflowExecution, SubscribeNotificationChannel, UnsubscribeNotificationChannel, + }, workflow_completion::WorkflowActivationCompletion, }, temporal::api::{ @@ -32,6 +34,12 @@ fn subscribe(channel: &str) -> SubscribeNotificationChannel { } } +fn unsubscribe(channel: &str) -> UnsubscribeNotificationChannel { + UnsubscribeNotificationChannel { + channel: channel.to_string(), + } +} + /// The command has to reach the server naming the channel. Only a task that is /// not being replayed sends commands, so this drives a single open task. #[tokio::test] @@ -140,6 +148,180 @@ async fn a_subscribe_reissued_to_a_different_channel_fails_the_task() { core.shutdown().await; } +/// The unsubscribe is an ordinary command: it has to reach the server naming +/// the channel, from a task that is not being replayed. +#[tokio::test] +async fn unsubscribe_channel_command_reaches_the_server() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + + let mut mock_client = mock_worker_client(); + mock_client + .expect_complete_workflow_task() + .times(1) + .returning(|resp, _| { + let cmd = resp.commands.first().expect("a command was sent"); + assert_eq!( + cmd.command_type(), + CommandType::UnsubscribeNotificationChannel + ); + match cmd.attributes.as_ref().unwrap() { + command::Attributes::UnsubscribeNotificationChannelCommandAttributes(a) => { + assert_eq!(a.channel, "orders"); + } + other => panic!("wrong attributes: {other:?}"), + } + Ok(RespondWorkflowTaskCompletedResponse::default()) + }); + + let mock = MockPollCfg::from_resp_batches("wfid", t, [ResponseType::AllHistory], mock_client); + let core = mock_worker(build_mock_pollers(mock)); + + let task = core.poll_workflow_activation().await.unwrap(); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![unsubscribe("orders").into()], + )) + .await + .unwrap(); + core.shutdown().await; +} + +/// A subscribe on one task and an unsubscribe on the next each match their own +/// event on replay. The unsubscribed event names the subscribed event it ended, +/// which is the server's record and not part of what the command is held to. +/// The second task is the one a notification scheduled, since a task with +/// nothing in it gets no activation of its own. +#[tokio::test] +async fn subscribe_then_unsubscribe_round_trips_through_replay() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_full_wf_task(); + let subscribed_id = t.add_notification_channel_subscribed("orders"); + t.add_workflow_task_scheduled_with_notifications(vec![notification("orders", 1)]); + t.add_workflow_task_started(); + t.add_workflow_task_completed(); + t.add_notification_channel_unsubscribed("orders", subscribed_id); + t.add_we_signaled("done", vec![]); + t.add_workflow_task_scheduled_and_started(); + + let mut mock_client = mock_worker_client(); + mock_client + .expect_complete_workflow_task() + .returning(|_, _| Ok(RespondWorkflowTaskCompletedResponse::default())); + mock_client + .expect_fail_workflow_task() + .returning(|_, _, f| panic!("core rejected a reissued channel command: {f:?}")); + + let mock = MockPollCfg::from_resp_batches("wfid", t, [ResponseType::AllHistory], mock_client); + let core = mock_worker(build_mock_pollers(mock)); + + let task = core.poll_workflow_activation().await.unwrap(); + assert!(task.is_replaying); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![subscribe("orders").into()], + )) + .await + .unwrap(); + + let task = core.poll_workflow_activation().await.unwrap(); + assert!(task.is_replaying); + assert_eq!(received(&task), vec![vec![notification("orders", 1)]]); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![unsubscribe("orders").into()], + )) + .await + .unwrap(); + + let task = core.poll_workflow_activation().await.unwrap(); + assert!(!task.is_replaying); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); + core.shutdown().await; +} + +/// The recorded event holds the reissued unsubscribe to its channel, the same +/// way the subscribed event holds a subscribe. +#[tokio::test] +async fn an_unsubscribe_reissued_for_a_different_channel_fails_the_task() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_full_wf_task(); + t.add_notification_channel_unsubscribed("orders", 0); + t.add_full_wf_task(); + + let failures = Arc::new(AtomicUsize::new(0)); + let counted = failures.clone(); + let mut mock = + MockPollCfg::from_resp_batches("wfid", t, [ResponseType::AllHistory], mock_worker_client()); + mock.num_expected_fails = 1; + mock.expect_fail_wft_matcher = Box::new(move |_, cause, _| { + counted.fetch_add(1, Ordering::Relaxed); + *cause == WorkflowTaskFailedCause::NonDeterministicError + }); + let mut mock = build_mock_pollers(mock); + mock.make_wft_stream_interminable(); + let core = mock_worker(mock); + + let task = core.poll_workflow_activation().await.unwrap(); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![unsubscribe("invoices").into()], + )) + .await + .unwrap(); + core.handle_eviction().await; + assert_eq!(failures.load(Ordering::Relaxed), 1); + core.shutdown().await; +} + +/// Core hands over whatever the scheduled events carry and keeps no view of +/// the run's subscriptions. The notification on the task that unsubscribes +/// reaches that task, and one the server put on a later scheduled event still +/// becomes a job: dropping it for a closed subscription is lang's call. The +/// unsubscribed event here records no subscription, which is what the server +/// writes for a channel the run never subscribed to. +#[tokio::test] +async fn an_unsubscribe_leaves_the_notification_jobs_alone() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_with_notifications(vec![notification("orders", 1)]); + t.add_workflow_task_started(); + t.add_workflow_task_completed(); + t.add_notification_channel_unsubscribed("orders", 0); + t.add_workflow_task_scheduled_with_notifications(vec![notification("orders", 2)]); + t.add_workflow_task_started(); + + let mut mock_client = mock_worker_client(); + mock_client + .expect_fail_workflow_task() + .returning(|_, _, f| panic!("core rejected the reissued unsubscribe: {f:?}")); + let mock = MockPollCfg::from_resp_batches("wfid", t, [ResponseType::AllHistory], mock_client); + let core = mock_worker(build_mock_pollers(mock)); + + let task = core.poll_workflow_activation().await.unwrap(); + assert!(task.is_replaying); + assert_eq!(received(&task), vec![vec![notification("orders", 1)]]); + core.complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + task.run_id, + vec![unsubscribe("orders").into()], + )) + .await + .unwrap(); + + let task = core.poll_workflow_activation().await.unwrap(); + assert!(!task.is_replaying); + assert_eq!(job_kinds(&task), vec!["notifications"]); + assert_eq!(received(&task), vec![vec![notification("orders", 2)]]); + core.complete_workflow_activation(WorkflowActivationCompletion::empty(task.run_id)) + .await + .unwrap(); +} + fn notification(channel: &str, counter: i64) -> Notification { Notification { channel: channel.to_string(), diff --git a/crates/sdk-core/src/replay/history_builder.rs b/crates/sdk-core/src/replay/history_builder.rs index f2fdbc978..4e827af97 100644 --- a/crates/sdk-core/src/replay/history_builder.rs +++ b/crates/sdk-core/src/replay/history_builder.rs @@ -178,6 +178,22 @@ impl TestHistoryBuilder { self.add(attrs) } + /// Add the event an unsubscribe-notification-channel command produces. A zero + /// `subscribed_event_id` is what the server records when the run held no + /// subscription for the channel. + pub fn add_notification_channel_unsubscribed( + &mut self, + channel: &str, + subscribed_event_id: i64, + ) -> i64 { + let attrs = WorkflowNotificationChannelUnsubscribedEventAttributes { + workflow_task_completed_event_id: self.previous_task_completed_id, + channel: channel.to_string(), + subscribed_event_id, + }; + self.add(attrs) + } + /// Add the event an append-stream-records command produces. The range is /// half-open, as it is on the event. pub fn add_stream_records_appended( diff --git a/crates/sdk-core/src/worker/workflow/machines/mod.rs b/crates/sdk-core/src/worker/workflow/machines/mod.rs index 4f1044ae9..b1744cb1d 100644 --- a/crates/sdk-core/src/worker/workflow/machines/mod.rs +++ b/crates/sdk-core/src/worker/workflow/machines/mod.rs @@ -19,6 +19,7 @@ mod signal_external_state_machine; mod subscribe_notification_channel_state_machine; mod subscribe_stream_state_machine; mod timer_state_machine; +mod unsubscribe_notification_channel_state_machine; mod update_state_machine; mod upsert_search_attributes_state_machine; mod workflow_task_state_machine; @@ -61,6 +62,7 @@ use temporalio_common::{ }; use temporalio_macros::fsm; use timer_state_machine::TimerMachine; +use unsubscribe_notification_channel_state_machine::UnsubscribeNotificationChannelMachine; use update_state_machine::UpdateMachine; use upsert_search_attributes_state_machine::UpsertSearchAttributesMachine; use workflow_machines::MachineResponse; @@ -89,6 +91,7 @@ enum Machines { SubscribeStreamMachine, AppendStreamRecordsMachine, SubscribeNotificationChannelMachine, + UnsubscribeNotificationChannelMachine, UpdateMachine, NexusOperationMachine, } diff --git a/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs b/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs index 46cb072e1..b4937b4df 100644 --- a/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs +++ b/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs @@ -79,6 +79,7 @@ mod machine_coverage_report { signal_external_state_machine::SignalExternalMachine, subscribe_notification_channel_state_machine::SubscribeNotificationChannelMachine, subscribe_stream_state_machine::SubscribeStreamMachine, timer_state_machine::TimerMachine, + unsubscribe_notification_channel_state_machine::UnsubscribeNotificationChannelMachine, update_state_machine::UpdateMachine, upsert_search_attributes_state_machine::UpsertSearchAttributesMachine, workflow_task_state_machine::WorkflowTaskMachine, @@ -123,6 +124,8 @@ mod machine_coverage_report { let mut subscribe_stream = SubscribeStreamMachine::visualizer().to_owned(); let mut append_stream_records = AppendStreamRecordsMachine::visualizer().to_owned(); let mut subscribe_channel = SubscribeNotificationChannelMachine::visualizer().to_owned(); + let mut unsubscribe_channel = + UnsubscribeNotificationChannelMachine::visualizer().to_owned(); // This isn't at all efficient but doesn't need to be. // Replace transitions in the vizzes with green color if they are covered. @@ -159,6 +162,9 @@ mod machine_coverage_report { m @ "SubscribeNotificationChannelMachine" => { cover_transitions(m, &mut subscribe_channel, coverage) } + m @ "UnsubscribeNotificationChannelMachine" => { + cover_transitions(m, &mut unsubscribe_channel, coverage) + } m => panic!("Unknown machine {m}"), } } diff --git a/crates/sdk-core/src/worker/workflow/machines/unsubscribe_notification_channel_state_machine.rs b/crates/sdk-core/src/worker/workflow/machines/unsubscribe_notification_channel_state_machine.rs new file mode 100644 index 000000000..b7724cb78 --- /dev/null +++ b/crates/sdk-core/src/worker/workflow/machines/unsubscribe_notification_channel_state_machine.rs @@ -0,0 +1,144 @@ +use super::{ + NewMachineWithCommand, StateMachine, TransitionResult, fsm, workflow_machines::MachineResponse, +}; +use crate::worker::workflow::{ + WFMachinesError, fatal, + machines::{EventInfo, HistEventData, WFMachinesAdapter}, + nondeterminism, +}; +use temporalio_common::protos::{ + coresdk::workflow_commands::UnsubscribeNotificationChannel, + temporal::api::{ + enums::v1::{CommandType, EventType}, + history::v1::{WorkflowNotificationChannelUnsubscribedEventAttributes, history_event}, + }, +}; + +fsm! { + pub(super) name UnsubscribeNotificationChannelMachine; + command UnsubscribeNotificationChannelMachineCommand; + error WFMachinesError; + shared_state SharedState; + + Created --(CommandScheduled) --> CommandIssued; + CommandIssued --(CommandRecorded(WorkflowNotificationChannelUnsubscribedEventAttributes), + shared on_command_recorded) --> Done; +} + +/// The channel the command named, kept so the recorded event can be held against it. +#[derive(Default, Clone)] +pub(super) struct SharedState { + channel: String, +} + +/// End this workflow's subscription to a notification channel. The server +/// records the event whether or not the run held a subscription, so the match +/// on replay is by channel alone and the recorded subscribed event id is not +/// part of it. +pub(super) fn unsubscribe_notification_channel( + lang_cmd: UnsubscribeNotificationChannel, +) -> NewMachineWithCommand { + let sm = UnsubscribeNotificationChannelMachine::from_parts( + Created {}.into(), + SharedState { + channel: lang_cmd.channel.clone(), + }, + ); + NewMachineWithCommand { + command: lang_cmd.into(), + machine: sm.into(), + } +} + +#[derive(Debug, derive_more::Display)] +pub(super) enum UnsubscribeNotificationChannelMachineCommand {} + +#[derive(Debug, Default, Clone, derive_more::Display)] +pub(super) struct Created {} + +#[derive(Debug, Default, Clone, derive_more::Display)] +pub(super) struct CommandIssued {} + +impl CommandIssued { + pub(super) fn on_command_recorded( + self, + dat: &mut SharedState, + attrs: WorkflowNotificationChannelUnsubscribedEventAttributes, + ) -> UnsubscribeNotificationChannelMachineTransition { + if dat.channel == attrs.channel { + TransitionResult::default() + } else { + TransitionResult::Err(nondeterminism!( + "Recorded unsubscription from notification channel {:?} does not match the \ + reissued unsubscription from channel {:?}", + attrs.channel, + dat.channel + )) + } + } +} + +#[derive(Debug, Default, Clone, derive_more::Display)] +pub(super) struct Done {} + +impl WFMachinesAdapter for UnsubscribeNotificationChannelMachine { + fn adapt_response( + &self, + _my_command: Self::Command, + _event_info: Option, + ) -> Result, Self::Error> { + Err(fatal!( + "UnsubscribeNotificationChannel does not use state machine commands" + )) + } +} + +impl TryFrom for UnsubscribeNotificationChannelMachineEvents { + type Error = WFMachinesError; + + fn try_from(e: HistEventData) -> Result { + let e = e.event; + match e.event_type() { + EventType::WorkflowNotificationChannelUnsubscribed => { + if let Some( + history_event::Attributes::WorkflowNotificationChannelUnsubscribedEventAttributes( + attrs, + ), + ) = e.attributes + { + Ok(UnsubscribeNotificationChannelMachineEvents::CommandRecorded( + attrs, + )) + } else { + Err(fatal!( + "Notification channel unsubscribed attributes were unset: {e}" + )) + } + } + _ => Err(Self::Error::Nondeterminism(format!( + "UnsubscribeNotificationChannelMachine does not handle {e}" + ))), + } + } +} + +impl TryFrom for UnsubscribeNotificationChannelMachineEvents { + type Error = WFMachinesError; + + fn try_from(c: CommandType) -> Result { + match c { + CommandType::UnsubscribeNotificationChannel => { + Ok(UnsubscribeNotificationChannelMachineEvents::CommandScheduled) + } + _ => Err(Self::Error::Nondeterminism(format!( + "UnsubscribeNotificationChannelMachine does not handle command type {c:?}" + ))), + } + } +} + +impl From for CommandIssued { + fn from(_: Created) -> Self { + Self {} + } +} diff --git a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs index 17989b895..ee50f0fc7 100644 --- a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs +++ b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs @@ -14,6 +14,7 @@ use super::{ subscribe_notification_channel_state_machine::subscribe_notification_channel, subscribe_stream_state_machine::subscribe_stream, timer_state_machine::new_timer, + unsubscribe_notification_channel_state_machine::unsubscribe_notification_channel, upsert_search_attributes_state_machine::upsert_search_attrs, workflow_machines::local_acts::LocalActivityData, workflow_task_state_machine::WorkflowTaskMachine, @@ -1743,6 +1744,15 @@ impl WorkflowMachines { CommandIdKind::NeverResolves, ); } + WFCommandVariant::UnsubscribeNotificationChannel(attrs) => { + // Never resolves: the event only records the end of the + // subscription and hands nothing back to the workflow. + self.add_cmd_to_wf_task( + unsubscribe_notification_channel(attrs), + annotations, + CommandIdKind::NeverResolves, + ); + } WFCommandVariant::UpdateResponse(ur) => { let m_key = self.get_machine_by_msg(&ur.protocol_instance_id)?; let m = if let Machines::UpdateMachine(m) = self.machine_mut(m_key) { diff --git a/crates/sdk-core/src/worker/workflow/mod.rs b/crates/sdk-core/src/worker/workflow/mod.rs index fa7de22f7..d02714723 100644 --- a/crates/sdk-core/src/worker/workflow/mod.rs +++ b/crates/sdk-core/src/worker/workflow/mod.rs @@ -1578,6 +1578,7 @@ enum WFCommandVariant { SubscribeStream(SubscribeStream), AppendStreamRecords(AppendStreamRecords), SubscribeNotificationChannel(SubscribeNotificationChannel), + UnsubscribeNotificationChannel(UnsubscribeNotificationChannel), } impl TryFrom for WFCommand { @@ -1593,6 +1594,9 @@ impl TryFrom for WFCommand { workflow_command::Variant::SubscribeNotificationChannel(s) => { WFCommandVariant::SubscribeNotificationChannel(s) } + workflow_command::Variant::UnsubscribeNotificationChannel(s) => { + WFCommandVariant::UnsubscribeNotificationChannel(s) + } workflow_command::Variant::CancelTimer(s) => WFCommandVariant::CancelTimer(s), workflow_command::Variant::ScheduleActivity(s) => WFCommandVariant::AddActivity(s), workflow_command::Variant::RequestCancelActivity(s) => { @@ -1646,12 +1650,6 @@ impl TryFrom for WFCommand { workflow_command::Variant::RequestCancelNexusOperation(s) => { WFCommandVariant::RequestCancelNexusOperation(s) } - // This layer has no machine for the unsubscribe. Dropping the command - // would let the workflow go on as if unsubscribed while the server - // keeps delivering. - workflow_command::Variant::UnsubscribeNotificationChannel(_) => { - return Err(EmptyWorkflowCommandErr); - } }; Ok(Self { variant,