From f88293a9da09cfb8ac3f1988579dc3f98e84cf6b Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Sat, 5 Sep 2026 20:49:22 -0700 Subject: [PATCH 01/35] Preserve external stream task and replay wake boundaries --- ...y-is-logical-and-segmentation-is-shared.md | 7 ++ crates/sdk-core/CHANGELOG.md | 8 ++ .../src/core_tests/external_streams.rs | 76 +++++++++++++++++++ .../src/worker/workflow/managed_run.rs | 30 +++++++- 4 files changed, 119 insertions(+), 2 deletions(-) diff --git a/arch_docs/streaming-poc-docs/decisions/ADR-046-output-capacity-is-logical-and-segmentation-is-shared.md b/arch_docs/streaming-poc-docs/decisions/ADR-046-output-capacity-is-logical-and-segmentation-is-shared.md index 1a6a3dbdc..5d40a5613 100644 --- a/arch_docs/streaming-poc-docs/decisions/ADR-046-output-capacity-is-logical-and-segmentation-is-shared.md +++ b/arch_docs/streaming-poc-docs/decisions/ADR-046-output-capacity-is-logical-and-segmentation-is-shared.md @@ -22,6 +22,13 @@ per-topic record-count vector to each segment, including empty segments. Replay capacity and latency policy, validates the recorded logical manifests, and performs the live number of event-loop drains once across both directions. +An input schedule also retains activations with no input observation. If the first subscription +is created late in a task, earlier empty activations in that task precede its first observed +segment. This keeps the two schedules aligned when an Activity, timer, or another stream causes +an intervening activation. A prerelease combined marker with different input and output segment +counts is rejected: the positions of omitted empty input segments cannot generally be inferred +from those counts, and dropping output segments would weaken replay verification. + When another publish would exceed the record or logical-byte limit, `publish()` waits, the current batch is staged with an output-capacity terminal, and Core forces a replacement Workflow Task. A single oversized record or manifest is rejected before unsafe external I/O or marker growth. diff --git a/crates/sdk-core/CHANGELOG.md b/crates/sdk-core/CHANGELOG.md index 600b7a1f5..ee1e885fa 100644 --- a/crates/sdk-core/CHANGELOG.md +++ b/crates/sdk-core/CHANGELOG.md @@ -33,6 +33,14 @@ relevant information. ## Unreleased +### Fixed +* External stream wake Signals encountered while replay advances through a History page now + resume reconstructed subscriptions. Workers with caching disabled no longer complete repeated + empty tasks while unread records remain in the external store. +* Workflow-originated external output no longer forces an empty replacement task from an old + stream wait after Workflow code has resumed and is awaiting an Activity or timer. This avoids + delaying that result behind an unnecessary task timeout. + ### Added * Added the Core protocol for replay-safe Workflow-originated external stream output, including exact Workflow Task History floors, compact staged-output marker proofs, and shared input/output diff --git a/crates/sdk-core/src/core_tests/external_streams.rs b/crates/sdk-core/src/core_tests/external_streams.rs index 296379223..2f2986094 100644 --- a/crates/sdk-core/src/core_tests/external_streams.rs +++ b/crates/sdk-core/src/core_tests/external_streams.rs @@ -3613,6 +3613,34 @@ async fn an_output_only_marker_is_emitted_replayed_and_not_rewritten() { worker.drain_pollers_and_shutdown().await; } +#[tokio::test] +async fn output_commit_does_not_force_a_task_for_a_stale_wait_snapshot() { + let (history, manifest) = output_only_marker_history(); + let markers: StreamMarkers = Default::default(); + let forced: Arc>> = Default::default(); + let worker = worker_recording_rollovers(markers, forced.clone(), history, vec![1]); + let activation = worker.poll_workflow_activation().await.unwrap(); + worker + .seed_external_stream_waits(&activation.run_id, vec![1], None, true) + .await; + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + activation.run_id, + vec![ + output_commit_command(manifest), + start_timer_cmd(1, Duration::from_secs(10)), + ], + )) + .await + .unwrap(); + assert_eq!( + *forced.lock(), + vec![false], + "a previous stream wait does not justify an empty task while lang now awaits a timer" + ); + worker.drain_pollers_and_shutdown().await; +} + #[tokio::test] async fn output_capacity_replacement_enters_lang_instead_of_autocompleting() { // Capacity backpressure blocks the publishing Workflow in lang. Its staged commit has no @@ -5579,6 +5607,54 @@ fn replayable_marker_then_signal_history() -> TestHistoryBuilder { t } +#[tokio::test] +async fn a_wake_reached_during_replay_resumes_the_reconstructed_waits() { + let mut history = TestHistoryBuilder::default(); + let mut started = crate::replay::default_wes_attribs(); + started.first_execution_run_id = started.original_execution_run_id.clone(); + started.workflow_task_timeout = Some(Duration::from_secs(300).try_into().unwrap()); + history.add(started); + history.add_full_wf_task(); + history.add_external_stream_marker_covering( + 1, + ParkReason::Idle, + b"header.segment.terminal", + &[(1, 0)], + ); + history.add_we_signaled( + external_stream::WAKE_SIGNAL_NAME, + vec![wake_payload(wake(0, history.get_orig_run_id()))], + ); + history.add_workflow_task_scheduled_and_started(); + let worker = + worker_recording_rollovers(Default::default(), Default::default(), history, vec![2]); + let replayed = worker.poll_workflow_activation().await.unwrap(); + assert!(replayed.is_replaying); + assert_eq!(replay_jobs(&replayed).len(), 1); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmd( + replayed.run_id, + quiescent_command(1, &[1], Duration::from_secs(30)), + )) + .await + .unwrap(); + + let live = tokio::time::timeout(Duration::from_secs(1), worker.poll_workflow_activation()) + .await + .expect("the wake in the same History page must survive replay") + .unwrap(); + assert_eq!(resolve_hints(&live), vec![1]); + assert!(!live.is_replaying); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmd( + live.run_id, + CompleteWorkflowExecution::default().into(), + )) + .await + .unwrap(); + worker.drain_pollers_and_shutdown().await; +} + #[tokio::test] async fn a_replaying_quiescence_registers_its_waits_and_arms_no_timer() { // A replayed Run has to rebuild its wait set: that set is per-Worker runtime state, not diff --git a/crates/sdk-core/src/worker/workflow/managed_run.rs b/crates/sdk-core/src/worker/workflow/managed_run.rs index aae8f1dac..b818ce80a 100644 --- a/crates/sdk-core/src/worker/workflow/managed_run.rs +++ b/crates/sdk-core/src/worker/workflow/managed_run.rs @@ -1310,6 +1310,13 @@ impl ManagedRun { || park_retains || self.waiting_on_local_work.external_wait_set.retains_wft() || self.waiting_on_local_work.output_buffered; + // A registered wait can belong to a previous task whose input has since resumed lang. + // Only a current wait (or a query-only activation preserving that wait) justifies an + // output replacement. Otherwise an empty forced task can buffer the Activity or timer + // result that lang is actually waiting for until the task times out. + let output_waits_need_replacement = stream_commands.quiescence.is_some() + || park_retains + || (answering_a_query && self.waiting_on_local_work.external_wait_set.retains_wft()); let query_refused_retention = stream_waits_still_pending && !boundary_closes_the_run && !has_server_bound_commands @@ -1426,6 +1433,18 @@ impl ManagedRun { if !completion.activation_was_eviction && !self.am_broken { self.wfm.apply_next_task_if_ready()?; } + // A cold replay can reach a wake already contained in this History page, without + // admitting another server task. Its reconstructed waits must receive that wake + // before this task is reported empty and the zero-sized cache evicts them again. + if !self.wfm.machines.replaying && self.apply_external_stream_wakes() { + self.waiting_on_local_work + .external_wait_set + .set_wft_open(true); + // Lang has finished this activation; only its completion bookkeeping remains. + // Queue now so prepare_complete_resp sees pending work rather than reporting + // this task before finish_activation makes the next activation deliverable. + self.queue_external_stream_resolve(); + } let new_local_acts = self.wfm.drain_queued_local_activities(); self.sink_la_requests(new_local_acts)?; @@ -1506,7 +1525,7 @@ impl ManagedRun { || completing_shutdown || completing_output_capacity || completing_output_latency - || (output_commit_pending && stream_waits_still_pending), + || (output_commit_pending && output_waits_need_replacement), ))) } Ok(Some((start_t, wft_timeout))) => { @@ -2178,7 +2197,14 @@ impl ManagedRun { /// activation, and notifications arriving while an activation is outstanding accumulate for /// the next one. There is never more than one outstanding activation per run. fn maybe_issue_external_stream_resolve(&mut self) { - if self.activation.is_some() || self.wft.is_none() || self.am_broken { + if self.activation.is_some() { + return; + } + self.queue_external_stream_resolve(); + } + + fn queue_external_stream_resolve(&mut self) { + if self.wft.is_none() || self.am_broken { return; } let set = &mut self.waiting_on_local_work.external_wait_set; From 975bf0e96a9951d7279886be7d837bf09376f617 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Sat, 5 Sep 2026 21:10:22 -0700 Subject: [PATCH 02/35] Retain incomplete stream tasks with workflow caching disabled --- crates/sdk-core/CHANGELOG.md | 3 ++ .../src/core_tests/external_streams.rs | 31 +++++++++++++++++++ .../src/worker/workflow/managed_run.rs | 9 ++++++ .../src/worker/workflow/workflow_stream.rs | 9 +++++- 4 files changed, 51 insertions(+), 1 deletion(-) diff --git a/crates/sdk-core/CHANGELOG.md b/crates/sdk-core/CHANGELOG.md index ee1e885fa..39f216e06 100644 --- a/crates/sdk-core/CHANGELOG.md +++ b/crates/sdk-core/CHANGELOG.md @@ -34,6 +34,9 @@ relevant information. ## Unreleased ### Fixed +* Workers with caching disabled now keep an incomplete retained external stream task until its + normal boundary, as they do for local Activities. This prevents repeated shutdown markers and + replacement tasks from starving asynchronous input readiness. * External stream wake Signals encountered while replay advances through a History page now resume reconstructed subscriptions. Workers with caching disabled no longer complete repeated empty tasks while unread records remain in the external store. diff --git a/crates/sdk-core/src/core_tests/external_streams.rs b/crates/sdk-core/src/core_tests/external_streams.rs index 2f2986094..6c92443ab 100644 --- a/crates/sdk-core/src/core_tests/external_streams.rs +++ b/crates/sdk-core/src/core_tests/external_streams.rs @@ -411,6 +411,37 @@ fn worker_counting_completions( mock_worker(mock) } +#[tokio::test] +async fn zero_cache_keeps_a_retained_stream_task_until_its_boundary() { + let mut mock = build_mock_pollers(MockPollCfg::from_resp_batches( + "fakeid", + canned_histories::single_timer("1"), + [1], + mock_worker_client(), + )); + mock.worker_cfg(|w| { + w.task_types = WorkerTaskTypes::workflow_only(); + w.max_cached_workflows = 0; + }); + let worker = mock_worker(mock); + let activation = worker.poll_workflow_activation().await.unwrap(); + let run_id = activation.run_id; + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmd( + run_id.clone(), + quiescent_command(1, &[1], Duration::from_secs(30)), + )) + .await + .unwrap(); + assert_eq!( + worker.notify_external_stream_ready(&run_id, 1, 0).await, + ExternalStreamReadyResult::Accepted, + "disabling cache must not evict an incomplete retained Workflow Task" + ); + assert_eq!(consume_resolve_activation(&worker, &run_id).await, vec![1]); + worker.drain_pollers_and_shutdown().await; +} + #[tokio::test] async fn quiescence_holds_the_workflow_task_open() { let completions = Arc::new(AtomicUsize::new(0)); diff --git a/crates/sdk-core/src/worker/workflow/managed_run.rs b/crates/sdk-core/src/worker/workflow/managed_run.rs index b818ce80a..5f30802ae 100644 --- a/crates/sdk-core/src/worker/workflow/managed_run.rs +++ b/crates/sdk-core/src/worker/workflow/managed_run.rs @@ -160,6 +160,15 @@ impl ManagedRun { self.waiting_on_local_work.local_activities.is_some() } + pub(super) fn retains_task_for_external_streams(&self) -> bool { + self.wft.is_some() + && (self.waiting_on_local_work.output_buffered + || matches!( + self.external_stream_run_status(), + ExternalStreamRunStatus::WftOpen + )) + } + pub(super) fn have_seen_terminal_event(&self) -> bool { self.wfm.machines.have_seen_terminal_event } diff --git a/crates/sdk-core/src/worker/workflow/workflow_stream.rs b/crates/sdk-core/src/worker/workflow/workflow_stream.rs index 76758845e..4317766cb 100644 --- a/crates/sdk-core/src/worker/workflow/workflow_stream.rs +++ b/crates/sdk-core/src/worker/workflow/workflow_stream.rs @@ -428,7 +428,14 @@ impl WFStream { // one was being delivered) count as not-yet-resolved too: they can schedule further LAs, // and the commands they produce are only flushed by the completion that finally answers // the WFT. Evicting first would strand those commands in the discarded machines. - if has_zero_sized_cache && !rh.waiting_on_local_activities() && !rh.more_pending_work() { + // A retained stream task is incomplete too. Evicting its quiescent activation would + // force a Shutdown boundary before asynchronous input prefetch can deliver readiness, + // then repeat on every cold replacement task. Evict after its normal durable boundary. + if has_zero_sized_cache + && !rh.waiting_on_local_activities() + && !rh.retains_task_for_external_streams() + && !rh.more_pending_work() + { acts.extend(self.request_eviction_of_lru_run().into_run_update_resp()) } acts From e25905cb1763841e7a757c83305b67f6f1400e07 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Wed, 16 Sep 2026 14:41:14 -0700 Subject: [PATCH 03/35] Factored the flush window record into a type alias. `clippy::type_complexity` rejects the inline nested type, and the lint is denied through `-D warnings` in `cargo test-lint`. --- crates/sdk-core/src/core_tests/external_streams.rs | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/crates/sdk-core/src/core_tests/external_streams.rs b/crates/sdk-core/src/core_tests/external_streams.rs index 6c92443ab..7cd8740ee 100644 --- a/crates/sdk-core/src/core_tests/external_streams.rs +++ b/crates/sdk-core/src/core_tests/external_streams.rs @@ -3993,7 +3993,8 @@ async fn three_output_flush_windows_cost_three_markers_and_workflow_tasks() { } } - let lifecycle: Arc, bool)>>> = Default::default(); + type CompletionRecord = (Vec, bool); + let lifecycle: Arc>> = Default::default(); let mut mock_cfg = MockPollCfg::from_resp_batches("fakeid", history, [1, 2, 3], mock_worker_client()); mock_cfg.completion_asserts_from_expectations(|mut asserts| { From c81363eb394f79cd0a0489776d8a131fd5f9ed23 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Wed, 16 Sep 2026 14:41:14 -0700 Subject: [PATCH 04/35] Registered ExternalStreamMachine with the coverage reporter. The reporter panics on any machine name it has no visualizer for, so every state machine has to be listed there. --- .../src/worker/workflow/machines/transition_coverage.rs | 3 +++ 1 file changed, 3 insertions(+) diff --git a/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs b/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs index b461d8dd6..0ddd0daf8 100644 --- a/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs +++ b/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs @@ -71,6 +71,7 @@ mod machine_coverage_report { child_workflow_state_machine::ChildWorkflowMachine, complete_workflow_state_machine::CompleteWorkflowMachine, continue_as_new_workflow_state_machine::ContinueAsNewWorkflowMachine, + external_stream_state_machine::ExternalStreamMachine, fail_workflow_state_machine::FailWorkflowMachine, local_activity_state_machine::LocalActivityMachine, modify_workflow_properties_state_machine::ModifyWorkflowPropertiesMachine, @@ -115,6 +116,7 @@ mod machine_coverage_report { let mut upsert_search_attr = UpsertSearchAttributesMachine::visualizer().to_owned(); let mut modify_wf_props = ModifyWorkflowPropertiesMachine::visualizer().to_owned(); let mut update = UpdateMachine::visualizer().to_owned(); + let mut external_stream = ExternalStreamMachine::visualizer().to_owned(); // This isn't at all efficient but doesn't need to be. // Replace transitions in the vizzes with green color if they are covered. @@ -141,6 +143,7 @@ mod machine_coverage_report { cover_transitions(m, &mut modify_wf_props, coverage) } m @ "UpdateMachine" => cover_transitions(m, &mut update, coverage), + m @ "ExternalStreamMachine" => cover_transitions(m, &mut external_stream, coverage), m => panic!("Unknown machine {m}"), } } From d1b30328e383d7de8ddad3f609f07f6d1b6e1a06 Mon Sep 17 00:00:00 2001 From: James Watkins-Harvey Date: Fri, 14 Aug 2026 13:58:56 -0400 Subject: [PATCH 05/35] Scoped the Prometheus metrics query to the test's task queue. Backported from upstream `42c7bfe9`. This branch forked before that fix, and the unscoped query also matches the shared-namespace worker's series, whose order is not the test's to choose. --- crates/sdk-core/tests/integ_tests/metrics_tests.rs | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/crates/sdk-core/tests/integ_tests/metrics_tests.rs b/crates/sdk-core/tests/integ_tests/metrics_tests.rs index 1e266c61c..5da4859da 100644 --- a/crates/sdk-core/tests/integ_tests/metrics_tests.rs +++ b/crates/sdk-core/tests/integ_tests/metrics_tests.rs @@ -960,12 +960,18 @@ async fn docker_metrics_with_prometheus( .await .unwrap(); + let task_queue = starter.get_task_queue().to_string(); eventually( || async { // Query Prometheus API for metrics temporalio_common::telemetry::ensure_default_crypto_provider(); let client = reqwest::Client::new(); - let query = format!("temporal_sdk_{}num_pollers", test_uid.clone()); + // The task queue must be matched in the query rather than asserted on afterwards: this + // runtime's meter is also used by the shared-namespace worker, whose pollers report + // against the worker-commands control queue, and the order series come back in is not + // ours to choose. + let query = + format!("temporal_sdk_{test_uid}num_pollers{{task_queue=\"{task_queue}\"}}"); let response = client .get(PROMETHEUS_QUERY_API) .query(&[("query", query.clone())]) @@ -981,12 +987,6 @@ async fn docker_metrics_with_prometheus( } assert_eq!(data[0]["metric"]["exported_job"], "temporal-core-sdk"); assert_eq!(data[0]["metric"]["job"], "otel-collector"); - assert!( - data[0]["metric"]["task_queue"] - .as_str() - .unwrap() - .starts_with(test_name) - ); } else { bail!("Invalid Prometheus response: {response:?}"); } From 7af3a4713ba81dc29022625548b5425fba61b4fc Mon Sep 17 00:00:00 2001 From: James Watkins-Harvey Date: Fri, 14 Aug 2026 15:11:03 -0400 Subject: [PATCH 06/35] Fixed activity cancellation against the current dev server. Backported from upstream `c87e2060`. The dev server needs `system.enableCancelActivityWorkerCommand`, and the eager case needs a server newer than the one the pinned CLI bundles. --- crates/sdk-core/tests/common/mod.rs | 2 ++ .../integ_tests/workflow_tests/activities.rs | 27 +++++++++++++++++-- 2 files changed, 27 insertions(+), 2 deletions(-) diff --git a/crates/sdk-core/tests/common/mod.rs b/crates/sdk-core/tests/common/mod.rs index f9a69370b..ed8dd68f7 100644 --- a/crates/sdk-core/tests/common/mod.rs +++ b/crates/sdk-core/tests/common/mod.rs @@ -1263,6 +1263,8 @@ pub(crate) fn integ_dev_server_config( "--dynamic-config-value".to_owned(), "frontend.workerCommandsEnabled=true".to_owned(), "--dynamic-config-value".to_owned(), + "system.enableCancelActivityWorkerCommand=true".to_owned(), + "--dynamic-config-value".to_owned(), "matching.rps=12000".to_owned(), "--search-attribute".to_string(), format!("{SEARCH_ATTR_TXT}=Text"), diff --git a/crates/sdk-core/tests/integ_tests/workflow_tests/activities.rs b/crates/sdk-core/tests/integ_tests/workflow_tests/activities.rs index 8200dcb8f..00668a797 100644 --- a/crates/sdk-core/tests/integ_tests/workflow_tests/activities.rs +++ b/crates/sdk-core/tests/integ_tests/workflow_tests/activities.rs @@ -1,7 +1,8 @@ use crate::{ common::{ - ActivationAssertionsInterceptor, CoreWfStarter, INTEG_CLIENT_IDENTITY, - activity_functions::StdActivities, init_core_and_create_wf, + ActivationAssertionsInterceptor, CLI_VERSION_OVERRIDE_ENV_VAR, CoreWfStarter, + INTEG_CLIENT_IDENTITY, activity_functions::StdActivities, init_core_and_create_wf, + init_integ_telem, }, shared_tests, }; @@ -9,6 +10,7 @@ use anyhow::anyhow; use assert_matches::assert_matches; use futures_util::FutureExt; use std::{ + env, sync::{ Arc, Mutex, atomic::{AtomicBool, Ordering}, @@ -66,6 +68,7 @@ use temporalio_sdk_core::{ }, }; use tokio::{join, sync::Semaphore, time::sleep}; +use tracing::warn; #[workflow] #[derive(Default)] @@ -2407,5 +2410,25 @@ async fn immediate_activity_cancelation() { #[case::eager(false)] #[tokio::test] async fn activity_cancel_delivered_without_heartbeat(#[case] disable_eager: bool) { + // Cancellation of activities through Worker Commands was added in Server v1.32.0-158.0, + // but the case for eager activities was initially broken. It got fixed in + // temporalio/temporal#10634, which was released in Server v1.32.0-159.0. + // + // At this time, there is no release of the Temporal CLI that bundles Server + // v1.32.0-159.0. We're pinned on CLI "v1.7.4-standalone-nexus-operations" which + // bundles Server v1.32.0-158.0, and therefore has the broken support for eager + // activity cancellation. That results in the eager activity cancelation test + // failing in CI and local tests against the CLI Dev Server. + // + // FIXME: Remove once we're pinned on a CLI that bundles Server >v1.32.0-159.0. + const CLI_WITHOUT_EAGER_CANCEL_FIX: &str = "v1.7.4-standalone-nexus-operations"; + if !disable_eager + && env::var(CLI_VERSION_OVERRIDE_ENV_VAR).is_ok_and(|v| v == CLI_WITHOUT_EAGER_CANCEL_FIX) + { + // The skip message would go unlogged as no telemetry has been initialized yet. + init_integ_telem(); + warn!("Skipping test: eager activity cancel requires server >= v1.32.0-159.0"); + return; + } shared_tests::activity_cancel_delivered_without_heartbeat(disable_eager).await } From f57b07435e13280cc7909f5478509010fb8fd443 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Wed, 16 Sep 2026 16:47:39 -0700 Subject: [PATCH 07/35] Matched the workflow-service WIT to upstream for first-execution-run-id. The `string` declaration made the generated Python Nexus model require `SignalWithStartWorkflowResponse.first_execution_run_id`, which the dev server bundled with this Core never sends. Upstream omits the field instead. --- crates/protos/protos/api_upstream/nexus/workflow-service.wit | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/crates/protos/protos/api_upstream/nexus/workflow-service.wit b/crates/protos/protos/api_upstream/nexus/workflow-service.wit index c1c7bc5df..dd87646d0 100644 --- a/crates/protos/protos/api_upstream/nexus/workflow-service.wit +++ b/crates/protos/protos/api_upstream/nexus/workflow-service.wit @@ -104,7 +104,8 @@ interface workflow-service { started: option, /// @nexus.omit signal-link: placeholder, - first-execution-run-id: string, + /// @nexus.omit + first-execution-run-id: placeholder, } /// @nexus.doc From 9e5676e2b5c47ea6dcacf5d00188db75f59864d4 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Wed, 16 Sep 2026 16:21:41 -0700 Subject: [PATCH 08/35] Honored the per-runner timeout for the integ test matrix. The matrix asks for 40 minutes on `macos-intel`, but the job never read that value, so the runner stayed on the 25-minute default it exceeds. --- .github/workflows/per-pr.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/per-pr.yml b/.github/workflows/per-pr.yml index 019b11bc1..d04428830 100644 --- a/.github/workflows/per-pr.yml +++ b/.github/workflows/per-pr.yml @@ -123,7 +123,7 @@ jobs: integ-tests: name: Integ tests - timeout-minutes: ${{ github.ref == 'refs/heads/main' && 30 || 25 }} + timeout-minutes: ${{ matrix.timeoutMinutes || (github.ref == 'refs/heads/main' && 30 || 25) }} strategy: fail-fast: false matrix: From 6a7e6b268d45a1f8dd42039e4d52b3017e4c26f8 Mon Sep 17 00:00:00 2001 From: James Watkins-Harvey Date: Fri, 14 Aug 2026 16:46:04 -0400 Subject: [PATCH 09/35] Tolerated out-of-grammar nexus request-timeout headers. Backported from upstream `84be3c1f`. This branch forked before that fix, and the dev server bundled with the current Core writes the header with Go's duration syntax, which the old parser rejects, so `nexus_metrics` hangs. --- crates/sdk-core/CHANGELOG.md | 5 ++ crates/sdk-core/src/worker/nexus.rs | 105 ++++++++++++++++++++++------ 2 files changed, 90 insertions(+), 20 deletions(-) diff --git a/crates/sdk-core/CHANGELOG.md b/crates/sdk-core/CHANGELOG.md index 39f216e06..44ea93292 100644 --- a/crates/sdk-core/CHANGELOG.md +++ b/crates/sdk-core/CHANGELOG.md @@ -73,3 +73,8 @@ relevant information. preserving the resolution ordering recorded in existing histories during replay. * Try-cancel child workflows no longer cause nondeterminism when they complete or fail after their cancellation was requested. +* Nexus tasks are now timed out locally even when the server sends a `request-timeout` header that + falls outside the Nexus duration grammar, such as a negative value for a task whose deadline has + already elapsed, a sub-millisecond unit, or a multi-unit value like `1m30s`. Previously such a + header was ignored entirely, so the handler was never told the task had timed out, and a task + left unanswered could block worker shutdown indefinitely. diff --git a/crates/sdk-core/src/worker/nexus.rs b/crates/sdk-core/src/worker/nexus.rs index b3765fa41..a8e39ce8d 100644 --- a/crates/sdk-core/src/worker/nexus.rs +++ b/crates/sdk-core/src/worker/nexus.rs @@ -566,29 +566,52 @@ fn payloads_too_large_nexus_failure(violation: &PayloadLimitViolation) -> NexusT }) } +/// Parses the value of the nexus `request-timeout` header, which the Nexus spec defines as a +/// positive decimal followed by `ms`, `s`, or `m`. +/// +/// Temporal server formats this header with Go's `time.Duration::String`, which is a superset of +/// that grammar: it also emits sub-millisecond units, hours, multi-unit values such as `1m30s`, +/// and negative values when the task's deadline has already elapsed by the time the task is +/// dispatched (see ). All of those are accepted +/// here, and anything that resolves to a non-positive duration means the deadline is already past, +/// hence a zero timeout. fn parse_request_timeout(timeout: &str) -> Result { - let timeout = timeout.trim(); - let (value, unit) = timeout.split_at( - timeout + let trimmed = timeout.trim(); + let (negative, mut rest) = match trimmed.strip_prefix('-') { + Some(rest) => (true, rest), + None => (false, trimmed.strip_prefix('+').unwrap_or(trimmed)), + }; + if rest.is_empty() { + return Err(anyhow!("Invalid timeout format")); + } + + let mut total = Duration::ZERO; + while !rest.is_empty() { + let unit_start = rest .find(|c: char| !c.is_ascii_digit() && c != '.') - .unwrap_or(timeout.len()), - ); - - match unit { - "m" => value - .parse::() - .map(|v| Duration::from_secs_f64(60.0 * v)) - .map_err(Into::into), - "s" => value - .parse::() - .map(Duration::from_secs_f64) - .map_err(Into::into), - "ms" => value - .parse::() - .map_err(anyhow::Error::from) - .and_then(|v| Duration::try_from_secs_f64(v / 1000.0).map_err(Into::into)), - _ => Err(anyhow!("Invalid timeout format")), + .ok_or_else(|| anyhow!("Invalid timeout format"))?; + let (value, after_value) = rest.split_at(unit_start); + let unit_end = after_value + .find(|c: char| c.is_ascii_digit()) + .unwrap_or(after_value.len()); + let (unit, remainder) = after_value.split_at(unit_end); + + let value = value.parse::()?; + let seconds = match unit { + "ns" => value / 1e9, + // Go emits the micro sign, but tolerate the Greek letter and the ASCII form too. + "us" | "µs" | "μs" => value / 1e6, + "ms" => value / 1e3, + "s" => value, + "m" => value * 60.0, + "h" => value * 3600.0, + _ => return Err(anyhow!("Invalid timeout format")), + }; + total += Duration::try_from_secs_f64(seconds)?; + rest = remainder; } + + Ok(if negative { Duration::ZERO } else { total }) } #[cfg(test)] @@ -616,6 +639,48 @@ mod tests { ); } + #[test] + fn parse_request_timeout_go_duration_forms() { + // Units outside the Nexus grammar which Go's `time.Duration::String` may still emit + assert_eq!( + parse_request_timeout("88.458µs").unwrap(), + Duration::from_secs_f64(88.458 / 1e6) + ); + assert_eq!( + parse_request_timeout("125ns").unwrap(), + Duration::from_nanos(125) + ); + assert_eq!( + parse_request_timeout("2h").unwrap(), + Duration::from_secs(7200) + ); + // Multi-unit values + assert_eq!( + parse_request_timeout("1m30.5s").unwrap(), + Duration::from_secs_f64(90.5) + ); + assert_eq!( + parse_request_timeout("1h0m0s").unwrap(), + Duration::from_secs(3600) + ); + // An already-elapsed deadline means the task must be timed out right away + assert_eq!(parse_request_timeout("-88.458µs").unwrap(), Duration::ZERO); + assert_eq!(parse_request_timeout("-1m30s").unwrap(), Duration::ZERO); + assert_eq!(parse_request_timeout("0s").unwrap(), Duration::ZERO); + // Leading and surrounding whitespace, and an explicit positive sign + assert_eq!( + parse_request_timeout(" +10s ").unwrap(), + Duration::from_secs(10) + ); + + for invalid in ["", "-", "10", "10x", "abc", "s", "1s2"] { + assert!( + parse_request_timeout(invalid).is_err(), + "'{invalid}' should not parse" + ); + } + } + #[test] fn payloads_too_large_nexus_failure_is_retryable() { let violation = PayloadLimitViolation { From 31af05354330bb88a8b1e1f598bde1ddddae1d85 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Wed, 16 Sep 2026 23:05:24 -0700 Subject: [PATCH 10/35] Matched the nexus model WIT to upstream for the workflow-id policies. The nexgen the Python SDK drives rejects @nexus.type on a native declaration, so generation failed for every consumer. Upstream declares both policies as placeholders, which keeps the annotations valid. --- .../nexus/deps/nexus-temporal-types/model.wit | 13 ++----------- 1 file changed, 2 insertions(+), 11 deletions(-) diff --git a/crates/protos/protos/api_upstream/nexus/deps/nexus-temporal-types/model.wit b/crates/protos/protos/api_upstream/nexus/deps/nexus-temporal-types/model.wit index f27eb9e9f..e435bebaa 100644 --- a/crates/protos/protos/api_upstream/nexus/deps/nexus-temporal-types/model.wit +++ b/crates/protos/protos/api_upstream/nexus/deps/nexus-temporal-types/model.wit @@ -136,12 +136,7 @@ interface model { /// typescript="common.WorkflowIdReusePolicy" /// dotnet="Temporalio.Api.Enums.V1.WorkflowIdReusePolicy" /// typescript-import="@temporalio/common" - enum workflow-id-reuse-policy { - allow-duplicate, - allow-duplicate-failed-only, - reject-duplicate, - terminate-if-running, - } + type workflow-id-reuse-policy = placeholder; /// @nexus.proto "temporal.api.enums.v1.WorkflowIdConflictPolicy" typescript-import="@temporalio/proto" /// @nexus.type @@ -149,11 +144,7 @@ interface model { /// typescript="common.WorkflowIdConflictPolicy" /// dotnet="Temporalio.Api.Enums.V1.WorkflowIdConflictPolicy" /// typescript-import="@temporalio/common" - enum workflow-id-conflict-policy { - fail, - use-existing, - terminate-existing, - } + type workflow-id-conflict-policy = placeholder; /// @nexus.proto "temporal.api.sdk.v1.UserMetadata" typescript-import="@temporalio/proto" /// @nexus.flatten-in-api From 08599acb920cac68d5d881988e4eeca664c3e235 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 18 Sep 2026 17:10:10 -0700 Subject: [PATCH 11/35] Asserted the release of retained zero-cache tasks and a mid-replay wake. The zero-cache tests now poll the eviction at the boundary and expect the run to be gone, for a quiescent task and for buffered output. The wake test loses its wall-clock bound and gains a paginated variant whose wake is decoded while replay is still in progress. --- .../src/core_tests/external_streams.rs | 213 +++++++++++++++++- 1 file changed, 207 insertions(+), 6 deletions(-) diff --git a/crates/sdk-core/src/core_tests/external_streams.rs b/crates/sdk-core/src/core_tests/external_streams.rs index 7cd8740ee..8a136701d 100644 --- a/crates/sdk-core/src/core_tests/external_streams.rs +++ b/crates/sdk-core/src/core_tests/external_streams.rs @@ -9,7 +9,7 @@ use crate::{ ExternalStreamReadyResult, ExternalStreamRunStatus, PollError, replay::{DEFAULT_ACTIVITY_TYPE, TestHistoryBuilder, canned_histories}, test_help::{ - MockPollCfg, PollWFTRespExt, ResponseType, WorkerExt, build_fake_worker, + MockPollCfg, PollWFTRespExt, ResponseType, WorkerExt, WorkerTestHelpers, build_fake_worker, build_mock_pollers, hist_to_poll_resp, mock_worker, query_ok, schedule_activity_cmd, start_timer_cmd, }, @@ -51,8 +51,11 @@ use temporalio_common::{ command::v1::command, common::v1::Payload, enums::v1::{CommandType, EventType}, + history::v1::History, query::v1::WorkflowQuery, - workflowservice::v1::RespondWorkflowTaskCompletedResponse, + workflowservice::v1::{ + GetWorkflowExecutionHistoryResponse, RespondWorkflowTaskCompletedResponse, + }, }, }, worker::WorkerTaskTypes, @@ -423,6 +426,8 @@ async fn zero_cache_keeps_a_retained_stream_task_until_its_boundary() { w.task_types = WorkerTaskTypes::workflow_only(); w.max_cached_workflows = 0; }); + // Kept open so the eviction at the boundary can be polled after the mock's one task. + mock.make_wft_stream_interminable(); let worker = mock_worker(mock); let activation = worker.poll_workflow_activation().await.unwrap(); let run_id = activation.run_id; @@ -439,6 +444,76 @@ async fn zero_cache_keeps_a_retained_stream_task_until_its_boundary() { "disabling cache must not evict an incomplete retained Workflow Task" ); assert_eq!(consume_resolve_activation(&worker, &run_id).await, vec![1]); + // The resolve activation completed with a timer, which does not retain the task, so the + // boundary has passed and the zero-sized cache lets the run go. + worker.handle_eviction().await; + assert_eq!( + worker.notify_external_stream_ready(&run_id, 1, 0).await, + ExternalStreamReadyResult::RunNotFound, + "a retained task must be released once its boundary is reached" + ); + worker.drain_pollers_and_shutdown().await; +} + +#[tokio::test] +async fn zero_cache_keeps_a_task_with_buffered_output_until_the_flush() { + // The other reason a task is retained: output lang has buffered but not yet staged. The + // task has to survive the zero-sized cache until the flush deadline asks lang to finalize, + // and go once the flush has closed it. + let history = canned_histories::single_timer("1"); + let manifest = output_manifest(history.get_orig_run_id(), 1, "zero-cache-stage-token"); + let recorded: Arc>> = Default::default(); + let mut mock_cfg = MockPollCfg::from_resp_batches("fakeid", history, [1], mock_worker_client()); + let collected = recorded.clone(); + mock_cfg.completion_asserts_from_expectations(|mut asserts| { + asserts.then(move |wft| collected.lock().extend(stream_marker_data(wft))); + }); + let mut mock = build_mock_pollers(mock_cfg); + mock.worker_cfg(|w| { + w.task_types = WorkerTaskTypes::workflow_only(); + w.max_cached_workflows = 0; + }); + mock.make_wft_stream_interminable(); + let worker = mock_worker(mock); + + let activation = worker.poll_workflow_activation().await.unwrap(); + let run_id = activation.run_id; + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmd( + run_id.clone(), + output_buffered_command(Duration::from_millis(40)), + )) + .await + .unwrap(); + + // Retained: the flush deadline reaches this run rather than an eviction. + let flush = worker.poll_workflow_activation().await.unwrap(); + assert_eq!( + finalization_jobs(&flush), + vec![(0, ParkReason::OutputLatency, vec![])], + "disabling cache must not evict a task holding buffered output; jobs were {:?}", + flush.jobs + ); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + run_id.clone(), + vec![ + finalized_command(0, b""), + output_commit_command(manifest.clone()), + ], + )) + .await + .unwrap(); + + worker.handle_eviction().await; + assert_eq!( + worker.notify_external_stream_ready(&run_id, 1, 0).await, + ExternalStreamReadyResult::RunNotFound, + "the flush closed the task, so nothing retains the run any longer" + ); + let written = recorded.lock(); + assert_eq!(written.len(), 1); + assert_eq!(written[0].output.as_ref(), Some(&manifest)); worker.drain_pollers_and_shutdown().await; } @@ -5671,10 +5746,9 @@ async fn a_wake_reached_during_replay_resumes_the_reconstructed_waits() { .await .unwrap(); - let live = tokio::time::timeout(Duration::from_secs(1), worker.poll_workflow_activation()) - .await - .expect("the wake in the same History page must survive replay") - .unwrap(); + // The mock either delivers the activation or hangs; the test's own timeout + // covers the hang. + let live = worker.poll_workflow_activation().await.unwrap(); assert_eq!(resolve_hints(&live), vec![1]); assert!(!live.is_replaying); worker @@ -6364,3 +6438,130 @@ async fn every_completion_path_writes_exactly_one_marker_ending_in_a_terminal() "the table has eight paths that write a marker and two that do not" ); } + +/// The same wake, decoded while replay is still in progress. +/// +/// The wake sits in a task that is not the last one, so the batch that decodes it is still a +/// replay, and History arrives in two pages with the wake on the first. The reconstructed waits +/// receive it once, and only after replay has ended. +#[tokio::test] +async fn a_wake_decoded_before_replay_ends_is_applied_once_when_it_does() { + let mut history = TestHistoryBuilder::default(); + let mut started = crate::replay::default_wes_attribs(); + started.first_execution_run_id = started.original_execution_run_id.clone(); + started.workflow_task_timeout = Some(Duration::from_secs(300).try_into().unwrap()); + history.add(started); + history.add_full_wf_task(); + history.add_external_stream_marker_covering( + 1, + ParkReason::Idle, + b"header.segment.terminal", + &[(1, 0)], + ); + history.add_we_signaled( + external_stream::WAKE_SIGNAL_NAME, + vec![wake_payload(wake(0, history.get_orig_run_id()))], + ); + history.add_full_wf_task(); + history.add_we_signaled("keep-the-run-going", vec![]); + history.add_workflow_task_scheduled_and_started(); + + let events = history.get_full_history_info().unwrap().into_events(); + let mut first_page = hist_to_poll_resp(&history, "fakeid".to_owned(), ResponseType::AllHistory); + // The wake is the last event of the first page. + first_page.history.as_mut().unwrap().events.truncate(6); + first_page.next_page_token = vec![1]; + let second_page = GetWorkflowExecutionHistoryResponse { + history: Some(History { + events: events[6..].to_vec(), + }), + ..Default::default() + }; + + let markers: StreamMarkers = Default::default(); + let mut mock_client = mock_worker_client(); + mock_client + .expect_get_workflow_execution_history() + .times(1) + .returning(move |_, _, _| Ok(second_page.clone())); + let mut mock_cfg = MockPollCfg::from_resp_batches( + "fakeid", + history, + [ResponseType::Raw(first_page.resp)], + mock_client, + ); + let collected = markers.clone(); + mock_cfg.completion_asserts_from_expectations(|mut asserts| { + for _ in 0..4 { + let collected = collected.clone(); + asserts.then(move |wft| collected.lock().extend(stream_markers(wft))); + } + }); + let mut mock = build_mock_pollers(mock_cfg); + mock.worker_cfg(|w| { + w.task_types = WorkerTaskTypes::workflow_only(); + w.max_cached_workflows = 1; + }); + let worker = mock_worker(mock); + + let replayed = worker.poll_workflow_activation().await.unwrap(); + assert!(replayed.is_replaying); + assert_eq!(replay_jobs(&replayed).len(), 1); + assert!(resolve_hints(&replayed).is_empty()); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmd( + replayed.run_id.clone(), + quiescent_command(1, &[1], Duration::from_secs(30)), + )) + .await + .unwrap(); + + // Everything after the replayed task, as (replaying, resolve hints, carried the signal). + let mut later = vec![]; + loop { + let activation = worker.poll_workflow_activation().await.unwrap(); + let carried_signal = activation.jobs.iter().any(|j| { + matches!( + j.variant, + Some(workflow_activation_job::Variant::SignalWorkflow(_)) + ) + }); + later.push(( + activation.is_replaying, + resolve_hints(&activation), + carried_signal, + )); + let done = carried_signal; + let completion = if done { + WorkflowActivationCompletion::from_cmd( + activation.run_id, + CompleteWorkflowExecution::default().into(), + ) + } else { + WorkflowActivationCompletion::empty(activation.run_id) + }; + worker + .complete_workflow_activation(completion) + .await + .unwrap(); + if done || later.len() > 3 { + break; + } + } + let resolves: Vec<_> = later + .iter() + .filter(|(_, hints, _)| !hints.is_empty()) + .collect(); + assert_eq!( + resolves.len(), + 1, + "the wake is applied exactly once; activations after replay were {later:?}" + ); + assert_eq!(resolves[0].1, vec![1]); + assert!( + !resolves[0].0, + "the wake must wait for replay to end; activations were {later:?}" + ); + assert_eq!(*markers.lock(), Vec::new(), "replay writes nothing"); + worker.drain_pollers_and_shutdown().await; +} From 6d52436cf797eb6135be161366364918cc034437 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 18 Sep 2026 17:10:10 -0700 Subject: [PATCH 12/35] Stated when a stream resolve may be queued behind an activation. The caller now says whether it is completing the outstanding activation, and a debug assertion checks that against the run's state, so a new call site cannot break the one-outstanding-activation rule silently. --- .../src/worker/workflow/managed_run.rs | 18 +++++++++++++++--- 1 file changed, 15 insertions(+), 3 deletions(-) diff --git a/crates/sdk-core/src/worker/workflow/managed_run.rs b/crates/sdk-core/src/worker/workflow/managed_run.rs index 5f30802ae..7039568f9 100644 --- a/crates/sdk-core/src/worker/workflow/managed_run.rs +++ b/crates/sdk-core/src/worker/workflow/managed_run.rs @@ -1452,7 +1452,7 @@ impl ManagedRun { // Lang has finished this activation; only its completion bookkeeping remains. // Queue now so prepare_complete_resp sees pending work rather than reporting // this task before finish_activation makes the next activation deliverable. - self.queue_external_stream_resolve(); + self.queue_external_stream_resolve(true); } let new_local_acts = self.wfm.drain_queued_local_activities(); self.sink_la_requests(new_local_acts)?; @@ -2209,10 +2209,22 @@ impl ManagedRun { if self.activation.is_some() { return; } - self.queue_external_stream_resolve(); + self.queue_external_stream_resolve(false); } - fn queue_external_stream_resolve(&mut self) { + /// Queues the job without asking whether an activation is outstanding. + /// + /// Two callers are legal: the readiness path once no activation is outstanding, and the + /// completion path of the outstanding activation, after `apply_next_task_if_ready` and before + /// `prepare_complete_resp` picks the pending jobs up. Lang has finished that activation, so + /// the job lands on the next one. From anywhere else the job would ride an activation lang is + /// still working on, which breaks the one-outstanding-activation rule this run relies on. + /// `completing_outstanding_activation` is the caller saying which of the two it is. + fn queue_external_stream_resolve(&mut self, completing_outstanding_activation: bool) { + debug_assert!( + completing_outstanding_activation == self.activation.is_some(), + "external stream resolve queued outside the readiness and completion paths" + ); if self.wft.is_none() || self.am_broken { return; } From 7df8a5b7a47b8112eb611c9777e04e44bf2d6fc6 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 18 Sep 2026 17:10:10 -0700 Subject: [PATCH 13/35] Folded the two Fixed sections of the changelog into one. --- crates/sdk-core/CHANGELOG.md | 20 +++++++++----------- 1 file changed, 9 insertions(+), 11 deletions(-) diff --git a/crates/sdk-core/CHANGELOG.md b/crates/sdk-core/CHANGELOG.md index 44ea93292..299c20f56 100644 --- a/crates/sdk-core/CHANGELOG.md +++ b/crates/sdk-core/CHANGELOG.md @@ -33,17 +33,6 @@ relevant information. ## Unreleased -### Fixed -* Workers with caching disabled now keep an incomplete retained external stream task until its - normal boundary, as they do for local Activities. This prevents repeated shutdown markers and - replacement tasks from starving asynchronous input readiness. -* External stream wake Signals encountered while replay advances through a History page now - resume reconstructed subscriptions. Workers with caching disabled no longer complete repeated - empty tasks while unread records remain in the external store. -* Workflow-originated external output no longer forces an empty replacement task from an old - stream wait after Workflow code has resumed and is awaiting an Activity or timer. This avoids - delaying that result behind an unnecessary task timeout. - ### Added * Added the Core protocol for replay-safe Workflow-originated external stream output, including exact Workflow Task History floors, compact staged-output marker proofs, and shared input/output @@ -62,6 +51,15 @@ relevant information. are preserved on failure; workers warn when the server does not advertise support. ### Fixed +* Workers with caching disabled now keep an incomplete retained external stream task until its + normal boundary, as they do for local Activities. This prevents repeated shutdown markers and + replacement tasks from starving asynchronous input readiness. +* External stream wake Signals encountered while replay advances through a History page now + resume reconstructed subscriptions. Workers with caching disabled no longer complete repeated + empty tasks while unread records remain in the external store. +* Workflow-originated external output no longer forces an empty replacement task from an old + stream wait after Workflow code has resumed and is awaiting an Activity or timer. This avoids + delaying that result behind an unnecessary task timeout. * Workers now defensively buffer a replacement workflow task if it reaches a run that still owns one, preserving the outstanding task token in release builds. * Workers no longer send worker heartbeats or appear in centralized heartbeat reports before they From 800052617eb7eb7b60983ac0f31b270b8d308531 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Mon, 21 Sep 2026 17:50:33 -0700 Subject: [PATCH 14/35] Released the recorded-output lock before the shutdown await. `cargo test-lint` refuses a `MutexGuard` held across an await point, and the assertions only need the guard for two lines. --- crates/sdk-core/src/core_tests/external_streams.rs | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/crates/sdk-core/src/core_tests/external_streams.rs b/crates/sdk-core/src/core_tests/external_streams.rs index 8a136701d..7aecb5023 100644 --- a/crates/sdk-core/src/core_tests/external_streams.rs +++ b/crates/sdk-core/src/core_tests/external_streams.rs @@ -511,9 +511,11 @@ async fn zero_cache_keeps_a_task_with_buffered_output_until_the_flush() { ExternalStreamReadyResult::RunNotFound, "the flush closed the task, so nothing retains the run any longer" ); - let written = recorded.lock(); - assert_eq!(written.len(), 1); - assert_eq!(written[0].output.as_ref(), Some(&manifest)); + { + let written = recorded.lock(); + assert_eq!(written.len(), 1); + assert_eq!(written[0].output.as_ref(), Some(&manifest)); + } worker.drain_pollers_and_shutdown().await; } From c713b5e49977b9f00485f7c94f2f944f6a295aee Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Mon, 21 Sep 2026 17:50:33 -0700 Subject: [PATCH 15/35] Carried the stream protos in the Core's api tree. Regenerating `temporalio/api` from this Core dropped the stream messages, commands and events the Python branch vendors, so its proto check failed. The api branch's diff is applied onto the api tree this Core pins, and the protos crate lists the new package and the two payload fields. --- crates/common/build.rs | 5 ++ crates/protos/build.rs | 1 + .../temporal/api/command/v1/message.proto | 34 ++++++++ .../temporal/api/enums/v1/command_type.proto | 2 + .../temporal/api/enums/v1/event_type.proto | 8 ++ .../temporal/api/enums/v1/failed_cause.proto | 8 ++ .../temporal/api/history/v1/message.proto | 37 +++++++++ .../temporal/api/stream/v1/message.proto | 82 +++++++++++++++++++ .../workflowservice/v1/request_response.proto | 5 ++ crates/protos/src/protos/mod.rs | 19 +++++ 10 files changed, 201 insertions(+) create mode 100644 crates/protos/protos/api_upstream/temporal/api/stream/v1/message.proto diff --git a/crates/common/build.rs b/crates/common/build.rs index 2e4a378de..1baae9de8 100644 --- a/crates/common/build.rs +++ b/crates/common/build.rs @@ -820,6 +820,11 @@ const NOT_VALIDATED_FIELDS: &[&str] = &[ "temporal.api.workflowservice.v1.StartWorkflowExecutionRequest.continued_failure", "temporal.api.workflowservice.v1.StartWorkflowExecutionRequest.last_completion_result", "temporal.api.workflowservice.v1.TerminateWorkflowExecutionRequest.details", + // Stream records: the blob limit is a per-event limit, and these bodies never reach an + // event. The server bounds the batch by message count (MaxMessagesPerBatch) instead, which + // is not a payload size the SDK can mirror. + "temporal.api.stream.v1.StreamRecord.body", + "temporal.api.stream.v1.StreamRecord.metadata", // Dedicated, non-fetchable limits (not blob/memo, not in DescribeNamespace): UserMetadata // (nexus-start only); Nexus EndpointSpec.description (maxDescriptionSize; cloud variant cloud-only). "temporal.api.sdk.v1.UserMetadata.details", diff --git a/crates/protos/build.rs b/crates/protos/build.rs index 509ba9259..7365df524 100644 --- a/crates/protos/build.rs +++ b/crates/protos/build.rs @@ -36,6 +36,7 @@ const SERDE_DERIVE_PREFIXES: &[&str] = &[ ".temporal.api.rules", ".temporal.api.schedule", ".temporal.api.sdk", + ".temporal.api.stream", ".temporal.api.taskqueue", ".temporal.api.testservice", ".temporal.api.update", diff --git a/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto index ee839115b..7d10e441b 100644 --- a/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto @@ -14,6 +14,7 @@ import "google/protobuf/duration.proto"; import "temporal/api/enums/v1/workflow.proto"; import "temporal/api/enums/v1/command_type.proto"; import "temporal/api/common/v1/message.proto"; +import "temporal/api/stream/v1/message.proto"; import "temporal/api/failure/v1/message.proto"; import "temporal/api/taskqueue/v1/message.proto"; import "temporal/api/workflow/v1/message.proto"; @@ -324,5 +325,38 @@ message Command { ScheduleNexusOperationCommandAttributes schedule_nexus_operation_command_attributes = 18; RequestCancelNexusOperationCommandAttributes request_cancel_nexus_operation_command_attributes = 19; + AppendStreamRecordsCommandAttributes append_stream_records_command_attributes = 20; + SubscribeStreamCommandAttributes subscribe_stream_command_attributes = 21; } } + +// Appends records to a stream the Workflow owns. Applied inside the Workflow +// Task's own commit. Produces one `WorkflowStreamRecordsAppended` event +// carrying the offset range and none of the payload; it schedules no further +// work. +message AppendStreamRecordsCommandAttributes { + // Empty means the Workflow's default output stream. + string stream_id = 1; + // Stored in order. The server sets `producer_id` to empty on each record, + // because the owning Workflow is the producer here. + repeated temporal.api.stream.v1.StreamRecord records = 2; +} + +// Subscribe this Workflow to a stream, so later Workflow Tasks carry the ranges +// it has not consumed yet. +// +// The stream's addressing is resolved by the server rather than supplied here. +// A Workflow cannot look it up without doing I/O, and a value it carried would +// be a reading rather than a fact, so it could differ on replay. +message SubscribeStreamCommandAttributes { + // Stream to consume. A stream in another execution is addressed by its id; + // one this Workflow owns is addressed by the name it was published under. + // The server resolves an owned name first and falls back to a standalone + // id, so a Workflow that owns a stream under this name cannot reach a + // standalone stream with the same id. + string stream_id = 1; + // Where to start. Negative means from wherever the stream is when the + // subscription is registered, which the server resolves and records so + // replay does not resolve it again. + int64 start_offset = 2; +} diff --git a/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto b/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto index 067d95391..967169b19 100644 --- a/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto +++ b/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto @@ -29,4 +29,6 @@ enum CommandType { COMMAND_TYPE_MODIFY_WORKFLOW_PROPERTIES = 16; COMMAND_TYPE_SCHEDULE_NEXUS_OPERATION = 17; COMMAND_TYPE_REQUEST_CANCEL_NEXUS_OPERATION = 18; + COMMAND_TYPE_APPEND_STREAM_RECORDS = 19; + COMMAND_TYPE_SUBSCRIBE_STREAM = 20; } diff --git a/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto b/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto index b879f51e8..a386b1db7 100644 --- a/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto +++ b/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto @@ -175,4 +175,12 @@ enum EventType { EVENT_TYPE_WORKFLOW_EXECUTION_UNPAUSED = 59; // An event that indicates time skipping advanced time or was disabled automatically after a bound was reached. EVENT_TYPE_WORKFLOW_EXECUTION_TIME_SKIPPING_TRANSITIONED = 60; + // A Workflow subscribed to a stream. Recorded once per subscription, not + // per record: the offsets a task consumed ride WorkflowTaskCompleted and + // the payloads never enter History at all. + EVENT_TYPE_WORKFLOW_STREAM_SUBSCRIBED = 61; + // A Workflow appended a batch of records to a stream. Recorded per + // batch, and carrying only the offset range it landed at: the bodies go to + // the stream's own log, never into History. + EVENT_TYPE_WORKFLOW_STREAM_RECORDS_APPENDED = 62; } diff --git a/crates/protos/protos/api_upstream/temporal/api/enums/v1/failed_cause.proto b/crates/protos/protos/api_upstream/temporal/api/enums/v1/failed_cause.proto index f8809fd4e..3e675a1f0 100644 --- a/crates/protos/protos/api_upstream/temporal/api/enums/v1/failed_cause.proto +++ b/crates/protos/protos/api_upstream/temporal/api/enums/v1/failed_cause.proto @@ -90,6 +90,14 @@ enum WorkflowTaskFailedCause { WORKFLOW_TASK_FAILED_CAUSE_WORKFLOW_PAUSE_REQUESTED_BEFORE_TASK_STARTED = 39; // A workflow task failed because the request exceeded a size limit. WORKFLOW_TASK_FAILED_CAUSE_REQUEST_TOO_LARGE = 40; + // A workflow task completed with an invalid AppendStreamRecords command. + WORKFLOW_TASK_FAILED_CAUSE_BAD_APPEND_STREAM_RECORDS_ATTRIBUTES = 41; + // A workflow task completed with an invalid SubscribeStream command. + WORKFLOW_TASK_FAILED_CAUSE_BAD_SUBSCRIBE_STREAM_ATTRIBUTES = 42; + // A workflow task could not be started because a stream range it consumed and recorded in + // History can no longer be served, for example after truncation or because it exceeds the + // replay bound. Check the workflow task failure message for more information. + WORKFLOW_TASK_FAILED_CAUSE_STREAM_RANGE_UNAVAILABLE = 43; } enum StartChildWorkflowExecutionFailedCause { diff --git a/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto index 0211c6f55..8deca8d15 100644 --- a/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto @@ -17,6 +17,7 @@ import "temporal/api/enums/v1/failed_cause.proto"; import "temporal/api/enums/v1/update.proto"; import "temporal/api/enums/v1/workflow.proto"; import "temporal/api/common/v1/message.proto"; +import "temporal/api/stream/v1/message.proto"; import "temporal/api/deployment/v1/message.proto"; import "temporal/api/failure/v1/message.proto"; import "temporal/api/taskqueue/v1/message.proto"; @@ -379,6 +380,13 @@ message WorkflowTaskCompletedEventAttributes { // The Worker Deployment Version that completed this task. Must be set if `versioning_behavior` // is set. This value updates workflow execution's `versioning_info.deployment_version`. temporal.api.deployment.v1.WorkerDeploymentVersion deployment_version = 11; + + // Offset ranges this Workflow Task consumed from streams it subscribes to. + // Recorded on every task where a subscription is active, including when it + // observed nothing: an empty range is a fact replay must reproduce, and + // omitting it would let replay deliver records the Workflow did not have. + // Numbered 20 to leave 14 through 19 free for fields added on the main line. + repeated temporal.api.stream.v1.StreamRange consumed_stream_ranges = 20; } message WorkflowTaskTimedOutEventAttributes { @@ -953,6 +961,33 @@ message ActivityPropertiesModifiedExternallyEventAttributes { temporal.api.common.v1.RetryPolicy new_retry_policy = 2; } +message WorkflowStreamSubscribedEventAttributes { + // The WorkflowTaskCompleted event of the task whose command created this + // subscription. + int64 workflow_task_completed_event_id = 1; + // Stream the Workflow subscribed to. + string stream_id = 2; + // The offset the subscription actually starts from. Resolved by the server + // when the subscription is registered and recorded here, so replay reads + // the resolved value rather than resolving it again against a stream that + // has since moved. + int64 start_offset = 3; +} + +message WorkflowStreamRecordsAppendedEventAttributes { + // The WorkflowTaskCompleted event of the task whose command appended this + // batch. + int64 workflow_task_completed_event_id = 1; + // Stream the Workflow appended to. + string stream_id = 2; + // Offset the first record of the batch landed at. + int64 first_offset = 3; + // How many records the batch held. With first_offset this names the range + // without carrying any of it, which is what keeps this event a fixed size + // no matter how large the batch or its payloads are. + int64 record_count = 4; +} + message WorkflowExecutionUpdateAcceptedEventAttributes { // The instance ID of the update protocol that generated this event. string protocol_instance_id = 1; @@ -1276,6 +1311,8 @@ message HistoryEvent { WorkflowExecutionPausedEventAttributes workflow_execution_paused_event_attributes = 63; WorkflowExecutionUnpausedEventAttributes workflow_execution_unpaused_event_attributes = 64; WorkflowExecutionTimeSkippingTransitionedEventAttributes workflow_execution_time_skipping_transitioned_event_attributes = 65; + WorkflowStreamSubscribedEventAttributes workflow_stream_subscribed_event_attributes = 66; + WorkflowStreamRecordsAppendedEventAttributes workflow_stream_records_appended_event_attributes = 67; } } diff --git a/crates/protos/protos/api_upstream/temporal/api/stream/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/stream/v1/message.proto new file mode 100644 index 000000000..9f0c88dfb --- /dev/null +++ b/crates/protos/protos/api_upstream/temporal/api/stream/v1/message.proto @@ -0,0 +1,82 @@ +syntax = "proto3"; + +package temporal.api.stream.v1; + +option go_package = "go.temporal.io/api/stream/v1;stream"; +option java_package = "io.temporal.api.stream.v1"; +option java_multiple_files = true; +option java_outer_classname = "MessageProto"; +option ruby_package = "Temporalio::Api::Stream::V1"; +option csharp_namespace = "Temporalio.Api.Stream.V1"; + +import "temporal/api/common/v1/message.proto"; + +// One entry in a stream. The record is the wire format: stores keep it +// serialized as is and readers in every language decode the same bytes. +message StreamRecord { + // The value the producer published, stored as sent. A payload codec + // applies here as it does to any other payload. + temporal.api.common.v1.Payload body = 1; + // Producer-supplied provenance, stored as sent. + map metadata = 2; + // Producer-supplied grouping label, stored as sent. + string topic = 3; + // How to read this record. Unspecified is read as DATA. + StreamRecordKind kind = 4; + // Who wrote the record. Empty when the owning Workflow did. + string producer_id = 5; + // The producer's attempt. Readers treat a later attempt by the same + // producer as superseding what the earlier one wrote. + int64 attempt = 6; + // The producer's position within its attempt, or -1 when unnumbered. + // Stored as sent; the server does not assign, validate or order by it. + int64 sequence = 7; +} + +// A contiguous range of a stream delivered to a Workflow Task, along with the +// offsets it covers. The offsets are what History records; the records +// themselves are never written to History. +message StreamSlice { + string stream_id = 1; + // Run id of the execution that owns the stream. Set on both a slice for the + // task being started and a re-supplied one. + string run_id = 2; + // Inclusive. + int64 from_offset = 3; + // Exclusive. Equal to from_offset when the subscription observed nothing, + // which is a fact replay has to reproduce rather than an absence of one. + int64 to_offset = 4; + repeated StreamRecord records = 5; + // The WorkflowTaskCompleted event whose consumed_stream_ranges recorded + // this range. Set only when the server is re-supplying a range for a task + // being replayed; a slice for the task now being started leaves it unset, + // because the event closing that task does not exist yet. + // + // Replay needs this because a Workflow Task response carries one slice set + // while a cache miss replays every prior task, so the ranges have to be + // matched to the events that recorded them rather than to the response. + int64 workflow_task_completed_event_id = 6; +} + +// The offsets a Workflow Task consumed, without the payloads. Recorded on +// WorkflowTaskCompleted so History grows with Workflow Tasks rather than with +// records. +message StreamRange { + string stream_id = 1; + // Inclusive. + int64 from_offset = 2; + // Exclusive. + int64 to_offset = 3; +} + +// What a record means to a reader. Kept on the record itself so every store +// and every language reads it the same way without a private envelope. +enum StreamRecordKind { + // Read as DATA. + STREAM_RECORD_KIND_UNSPECIFIED = 0; + // A value the producer published; `body` carries it. + STREAM_RECORD_KIND_DATA = 1; + // The producer named by `producer_id` writes nothing more on `topic`. + // Says nothing about that producer's outcome and does not end the stream. + STREAM_RECORD_KIND_FINISH = 2; +} diff --git a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto index c3dd95769..b396de5b4 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto @@ -24,6 +24,7 @@ import "temporal/api/enums/v1/activity.proto"; import "temporal/api/enums/v1/nexus.proto"; import "temporal/api/activity/v1/message.proto"; import "temporal/api/common/v1/message.proto"; +import "temporal/api/stream/v1/message.proto"; import "temporal/api/history/v1/message.proto"; import "temporal/api/workflow/v1/message.proto"; import "temporal/api/command/v1/message.proto"; @@ -383,6 +384,10 @@ message PollWorkflowTaskQueueResponse { // 3. If every group has some pending polls, assign the next poll to a group randomly // according to the weights. temporal.api.taskqueue.v1.PollerGroupsInfo poller_groups_info = 19; + + // Stream data attached to this task. Delivered out of band so the payloads + // never enter History; only the offset ranges are recorded there. + repeated temporal.api.stream.v1.StreamSlice stream_slices = 20; } message RespondWorkflowTaskCompletedRequest { diff --git a/crates/protos/src/protos/mod.rs b/crates/protos/src/protos/mod.rs index 19e82ef8e..51b0cdba4 100644 --- a/crates/protos/src/protos/mod.rs +++ b/crates/protos/src/protos/mod.rs @@ -1994,6 +1994,12 @@ pub mod temporal { CommandType::ScheduleActivityTask } Attributes::StartTimerCommandAttributes(_) => CommandType::StartTimer, + Attributes::AppendStreamRecordsCommandAttributes(_) => { + CommandType::AppendStreamRecords + } + Attributes::SubscribeStreamCommandAttributes(_) => { + CommandType::SubscribeStream + } Attributes::CompleteWorkflowExecutionCommandAttributes(_) => { CommandType::CompleteWorkflowExecution } @@ -2506,6 +2512,8 @@ pub mod temporal { | EventType::TimerStarted | EventType::UpsertWorkflowSearchAttributes | EventType::WorkflowPropertiesModified + | EventType::WorkflowStreamSubscribed + | EventType::WorkflowStreamRecordsAppended | EventType::NexusOperationScheduled | EventType::NexusOperationCancelRequested | EventType::WorkflowExecutionCanceled @@ -2605,6 +2613,10 @@ pub mod temporal { // mark any new event types as ignorable or not. if let Some(a) = self.attributes.as_ref() { match a { + Attributes::WorkflowStreamSubscribedEventAttributes(_) => false, + Attributes::WorkflowStreamRecordsAppendedEventAttributes(_) => { + false + } Attributes::WorkflowExecutionStartedEventAttributes(_) => false, Attributes::WorkflowExecutionCompletedEventAttributes(_) => false, Attributes::WorkflowExecutionFailedEventAttributes(_) => false, @@ -2692,6 +2704,8 @@ pub mod temporal { pub fn event_type(&self) -> EventType { // I just absolutely _love_ this match self { + Attributes::WorkflowStreamSubscribedEventAttributes(_) => { EventType::WorkflowStreamSubscribed } + Attributes::WorkflowStreamRecordsAppendedEventAttributes(_) => { EventType::WorkflowStreamRecordsAppended } Attributes::WorkflowExecutionStartedEventAttributes(_) => { EventType::WorkflowExecutionStarted } Attributes::WorkflowExecutionCompletedEventAttributes(_) => { EventType::WorkflowExecutionCompleted } Attributes::WorkflowExecutionFailedEventAttributes(_) => { EventType::WorkflowExecutionFailed } @@ -2809,6 +2823,11 @@ pub mod temporal { tonic::include_proto!("temporal.api.sdk.v1"); } } + pub mod stream { + pub mod v1 { + tonic::include_proto!("temporal.api.stream.v1"); + } + } pub mod taskqueue { pub mod v1 { tonic::include_proto!("temporal.api.taskqueue.v1"); From 34c1018ae442916ad2218fd9d82ee4fcc947cab8 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 25 Sep 2026 14:12:03 -0700 Subject: [PATCH 16/35] Forced a replacement task for output left buffered behind a commit. Narrowing the replacement condition to a current wait took buffered output with it. The flush deadline does nothing once the task is gone, so the publish latency lang asked for was dropped without a word. --- crates/sdk-core/CHANGELOG.md | 4 ++- .../src/core_tests/external_streams.rs | 29 +++++++++++++++++++ .../src/worker/workflow/managed_run.rs | 5 ++++ 3 files changed, 37 insertions(+), 1 deletion(-) diff --git a/crates/sdk-core/CHANGELOG.md b/crates/sdk-core/CHANGELOG.md index 299c20f56..30c7c9c2a 100644 --- a/crates/sdk-core/CHANGELOG.md +++ b/crates/sdk-core/CHANGELOG.md @@ -59,7 +59,9 @@ relevant information. empty tasks while unread records remain in the external store. * Workflow-originated external output no longer forces an empty replacement task from an old stream wait after Workflow code has resumed and is awaiting an Activity or timer. This avoids - delaying that result behind an unnecessary task timeout. + delaying that result behind an unnecessary task timeout. A completion that stages a commit and + reports output still buffered does keep forcing one, since the max publish latency it asked for + can only be honored while a task is held. * Workers now defensively buffer a replacement workflow task if it reaches a run that still owns one, preserving the outstanding task token in release builds. * Workers no longer send worker heartbeats or appear in centralized heartbeat reports before they diff --git a/crates/sdk-core/src/core_tests/external_streams.rs b/crates/sdk-core/src/core_tests/external_streams.rs index 7aecb5023..8a6241357 100644 --- a/crates/sdk-core/src/core_tests/external_streams.rs +++ b/crates/sdk-core/src/core_tests/external_streams.rs @@ -3749,6 +3749,35 @@ async fn output_commit_does_not_force_a_task_for_a_stale_wait_snapshot() { worker.drain_pollers_and_shutdown().await; } +/// A completion that stages a commit and says more output is still buffered has to be reported +/// with a replacement. The flush deadline only does anything while a task is held, so reporting +/// without one drops the publish latency lang asked for and leaves those records waiting on +/// whatever task happens to come next. +#[tokio::test] +async fn output_commit_forces_a_task_for_output_still_buffered() { + let (history, manifest) = output_only_marker_history(); + let markers: StreamMarkers = Default::default(); + let forced: Arc>> = Default::default(); + let worker = worker_recording_rollovers(markers, forced.clone(), history, vec![1]); + let activation = worker.poll_workflow_activation().await.unwrap(); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + activation.run_id, + vec![ + output_commit_command(manifest), + output_buffered_command(Duration::from_secs(5)), + ], + )) + .await + .unwrap(); + assert_eq!( + *forced.lock(), + vec![true], + "output buffered behind the staged commit needs a task to flush on" + ); + worker.drain_pollers_and_shutdown().await; +} + #[tokio::test] async fn output_capacity_replacement_enters_lang_instead_of_autocompleting() { // Capacity backpressure blocks the publishing Workflow in lang. Its staged commit has no diff --git a/crates/sdk-core/src/worker/workflow/managed_run.rs b/crates/sdk-core/src/worker/workflow/managed_run.rs index 7039568f9..8a656154a 100644 --- a/crates/sdk-core/src/worker/workflow/managed_run.rs +++ b/crates/sdk-core/src/worker/workflow/managed_run.rs @@ -1323,8 +1323,13 @@ impl ManagedRun { // Only a current wait (or a query-only activation preserving that wait) justifies an // output replacement. Otherwise an empty forced task can buffer the Activity or timer // result that lang is actually waiting for until the task times out. + // + // Buffered output is not one of those stale waits. Staging the commit clears the flag, so + // it can only be set here by a `WorkflowOutputStreamBuffered` on this same completion, and + // the flush deadline it asks for fires into nothing once the task is gone. let output_waits_need_replacement = stream_commands.quiescence.is_some() || park_retains + || self.waiting_on_local_work.output_buffered || (answering_a_query && self.waiting_on_local_work.external_wait_set.retains_wft()); let query_refused_retention = stream_waits_still_pending && !boundary_closes_the_run From 2e8455ef5212561a8b43987710d63761b598ac11 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 25 Sep 2026 14:12:03 -0700 Subject: [PATCH 17/35] Logged the one-outstanding-activation breach in release builds too. A breach reorders activations rather than crashing, which a debug-only assertion leaves silent in exactly the builds where it would bite. --- crates/sdk-core/src/worker/workflow/managed_run.rs | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/crates/sdk-core/src/worker/workflow/managed_run.rs b/crates/sdk-core/src/worker/workflow/managed_run.rs index 8a656154a..75384287f 100644 --- a/crates/sdk-core/src/worker/workflow/managed_run.rs +++ b/crates/sdk-core/src/worker/workflow/managed_run.rs @@ -2226,10 +2226,11 @@ impl ManagedRun { /// still working on, which breaks the one-outstanding-activation rule this run relies on. /// `completing_outstanding_activation` is the caller saying which of the two it is. fn queue_external_stream_resolve(&mut self, completing_outstanding_activation: bool) { - debug_assert!( - completing_outstanding_activation == self.activation.is_some(), - "external stream resolve queued outside the readiness and completion paths" - ); + // Violating this reorders activations rather than crashing, so a release build has to say + // so as well; `debug_assert!` alone would leave it silent everywhere it matters. + if completing_outstanding_activation != self.activation.is_some() { + dbg_panic!("external stream resolve queued outside the readiness and completion paths"); + } if self.wft.is_none() || self.am_broken { return; } From ff4e0a65e93d9a253fb302144a31375c17c0ec50 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Thu, 1 Oct 2026 11:06:34 -0700 Subject: [PATCH 18/35] Carried the wake protos and RPC in the Core's api tree. Applied from api 75de7aa. The poll response schema in the vendored OpenAPI files lacks some fields the api fork has, so its wakes hunk was placed by hand. --- crates/client/src/grpc.rs | 9 ++ .../api_upstream/openapi/openapiv2.json | 152 ++++++++++++++++++ .../api_upstream/openapi/openapiv3.yaml | 152 ++++++++++++++++++ .../temporal/api/workflow/v1/message.proto | 23 +++ .../workflowservice/v1/request_response.proto | 27 ++++ .../api/workflowservice/v1/service.proto | 16 ++ crates/sdk-core-c-bridge/src/client.rs | 3 + 7 files changed, 382 insertions(+) diff --git a/crates/client/src/grpc.rs b/crates/client/src/grpc.rs index e0506f715..7fdc2d13c 100644 --- a/crates/client/src/grpc.rs +++ b/crates/client/src/grpc.rs @@ -955,6 +955,15 @@ proxier! { r.extensions_mut().insert(labels); } ); + ( + wake_workflow_execution, + WakeWorkflowExecutionRequest, + WakeWorkflowExecutionResponse, + |r| { + let labels = namespaced_request!(r); + r.extensions_mut().insert(labels); + } + ); ( signal_with_start_workflow_execution, SignalWithStartWorkflowExecutionRequest, diff --git a/crates/protos/protos/api_upstream/openapi/openapiv2.json b/crates/protos/protos/api_upstream/openapi/openapiv2.json index 1437120e5..0a90bd496 100644 --- a/crates/protos/protos/api_upstream/openapi/openapiv2.json +++ b/crates/protos/protos/api_upstream/openapi/openapiv2.json @@ -4575,6 +4575,51 @@ ] } }, + "/api/v1/namespaces/{namespace}/workflows/{workflowExecution.workflowId}/wake": { + "post": { + "summary": "WakeWorkflowExecution asks a running Workflow Execution to run a Workflow\nTask because a source it consumes has moved. Unlike a Signal it records\nno event, carries no payload, and folds with other wakes for the same\nsource, so a burst of writes costs one task. The Workflow learns the\nsource and its position from the task and reads the source itself.", + "operationId": "WakeWorkflowExecution2", + "responses": { + "200": { + "description": "A successful response.", + "schema": { + "$ref": "#/definitions/v1WakeWorkflowExecutionResponse" + } + }, + "default": { + "description": "An unexpected error response.", + "schema": { + "$ref": "#/definitions/rpcStatus" + } + } + }, + "parameters": [ + { + "name": "namespace", + "in": "path", + "required": true, + "type": "string" + }, + { + "name": "workflowExecution.workflowId", + "in": "path", + "required": true, + "type": "string" + }, + { + "name": "body", + "in": "body", + "required": true, + "schema": { + "$ref": "#/definitions/WorkflowServiceWakeWorkflowExecutionBody" + } + } + ], + "tags": [ + "WorkflowService" + ] + } + }, "/api/v1/namespaces/{namespace}/workflows/{workflowId}": { "post": { "summary": "StartWorkflowExecution starts a new workflow execution.", @@ -10158,6 +10203,51 @@ ] } }, + "/namespaces/{namespace}/workflows/{workflowExecution.workflowId}/wake": { + "post": { + "summary": "WakeWorkflowExecution asks a running Workflow Execution to run a Workflow\nTask because a source it consumes has moved. Unlike a Signal it records\nno event, carries no payload, and folds with other wakes for the same\nsource, so a burst of writes costs one task. The Workflow learns the\nsource and its position from the task and reads the source itself.", + "operationId": "WakeWorkflowExecution", + "responses": { + "200": { + "description": "A successful response.", + "schema": { + "$ref": "#/definitions/v1WakeWorkflowExecutionResponse" + } + }, + "default": { + "description": "An unexpected error response.", + "schema": { + "$ref": "#/definitions/rpcStatus" + } + } + }, + "parameters": [ + { + "name": "namespace", + "in": "path", + "required": true, + "type": "string" + }, + { + "name": "workflowExecution.workflowId", + "in": "path", + "required": true, + "type": "string" + }, + { + "name": "body", + "in": "body", + "required": true, + "schema": { + "$ref": "#/definitions/WorkflowServiceWakeWorkflowExecutionBody" + } + } + ], + "tags": [ + "WorkflowService" + ] + } + }, "/namespaces/{namespace}/workflows/{workflowId}": { "post": { "summary": "StartWorkflowExecution starts a new workflow execution.", @@ -13429,6 +13519,27 @@ }, "description": "Used to validate the compute config without attaching it to a Worker Deployment Version." }, + "WorkflowServiceWakeWorkflowExecutionBody": { + "type": "object", + "properties": { + "workflowExecution": { + "type": "object", + "properties": { + "runId": { + "type": "string" + } + }, + "description": "The Workflow to wake. `run_id` is optional. When set, the wake goes to\nthe current run of the chain that run belongs to, so a sender that\nlearned a run id before a continue-as-new still reaches the consumer,\nand it is refused with NotFound once that chain has ended. When unset,\nthe wake goes to the current run under the Workflow Id, whichever chain." + }, + "wake": { + "$ref": "#/definitions/v1Wake" + }, + "identity": { + "type": "string", + "description": "The identity of the sender, for metrics and logs." + } + } + }, "apiActivityV1CallbackInfo": { "type": "object", "properties": { @@ -18398,6 +18509,14 @@ "pollerGroupsInfo": { "$ref": "#/definitions/v1PollerGroupsInfo", "description": "The weighted, versioned list of poller groups IDs that client should use for future polls to\nthis task queue. Client should ignore this if it has already applied a snapshot with a\nversion greater than or equal to `poller_groups_info.version`. Client is expected to:\n 1. Maintain minimum number of pollers no less than the number of groups.\n 2. Try to assign the next poll to a group without any pending polls,\n 3. If every group has some pending polls, assign the next poll to a group randomly\n according to the weights." + }, + "wakes": { + "type": "array", + "items": { + "type": "object", + "$ref": "#/definitions/v1Wake" + }, + "description": "Wakes pending for this execution when the task started, folded by\nsource. Not recorded in History and never re-supplied on replay: the\nWorkflow treats one as a reason to read its source now and records what\nit read itself." } } }, @@ -20783,6 +20902,39 @@ }, "description": "Specifies client's intent to wait for Update results." }, + "v1Wake": { + "type": "object", + "properties": { + "source": { + "type": "string", + "description": "What moved, as the sender names it. The server folds wakes by this value\nand never interprets it. For a stream, the provider formats the stream's\nidentity, owner and topic, into it." + }, + "position": { + "type": "string", + "format": "byte", + "description": "Where the source stands after the write that caused this wake, in the\nstore's own terms. Opaque to the server and handed to the Workflow as\nreceived. A later wake's position replaces an earlier one's." + }, + "counter": { + "type": "string", + "format": "int64", + "description": "Orders wakes from one source. Only the source's store can order its\npositions, so the sender derives this from the position and the server\ncompares nothing else. A wake whose counter is not above the one the\nserver already holds is folded." + } + }, + "description": "A wake tells a running Workflow Execution that something it consumes has\nmoved, so it should run a Workflow Task and read from its own cursor. It is\na reason to run, not data: nothing is recorded in History and no record\ntravels with it. A store Temporal does not host sends one after the store\nhas acknowledged a write. Wakes with the same source fold: the server keeps\nthe highest counter it has seen per execution and source, so a repeated,\nreordered or late wake changes nothing, and a burst of writes costs one task." + }, + "v1WakeWorkflowExecutionResponse": { + "type": "object", + "properties": { + "runId": { + "type": "string", + "description": "The run the wake was stored on." + }, + "folded": { + "type": "boolean", + "description": "True when the server already held this counter or a higher one for the\nsource, so the wake changed nothing." + } + } + }, "v1WorkerConfig": { "type": "object", "properties": { diff --git a/crates/protos/protos/api_upstream/openapi/openapiv3.yaml b/crates/protos/protos/api_upstream/openapi/openapiv3.yaml index 007d77168..ef43e36ea 100644 --- a/crates/protos/protos/api_upstream/openapi/openapiv3.yaml +++ b/crates/protos/protos/api_upstream/openapi/openapiv3.yaml @@ -4695,6 +4695,47 @@ paths: application/json: schema: $ref: '#/components/schemas/Status' + /api/v1/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/wake: + post: + tags: + - WorkflowService + description: |- + WakeWorkflowExecution asks a running Workflow Execution to run a Workflow + Task because a source it consumes has moved. Unlike a Signal it records + no event, carries no payload, and folds with other wakes for the same + source, so a burst of writes costs one task. The Workflow learns the + source and its position from the task and reads the source itself. + operationId: WakeWorkflowExecution + parameters: + - name: namespace + in: path + required: true + schema: + type: string + - name: workflow_execution.workflow_id + in: path + required: true + schema: + type: string + requestBody: + content: + application/json: + schema: + $ref: '#/components/schemas/WakeWorkflowExecutionRequest' + required: true + responses: + "200": + description: OK + content: + application/json: + schema: + $ref: '#/components/schemas/WakeWorkflowExecutionResponse' + default: + description: Default error response + content: + application/json: + schema: + $ref: '#/components/schemas/Status' /api/v1/nexus/endpoints: get: tags: @@ -9735,6 +9776,47 @@ paths: application/json: schema: $ref: '#/components/schemas/Status' + /namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/wake: + post: + tags: + - WorkflowService + description: |- + WakeWorkflowExecution asks a running Workflow Execution to run a Workflow + Task because a source it consumes has moved. Unlike a Signal it records + no event, carries no payload, and folds with other wakes for the same + source, so a burst of writes costs one task. The Workflow learns the + source and its position from the task and reads the source itself. + operationId: WakeWorkflowExecution + parameters: + - name: namespace + in: path + required: true + schema: + type: string + - name: workflow_execution.workflow_id + in: path + required: true + schema: + type: string + requestBody: + content: + application/json: + schema: + $ref: '#/components/schemas/WakeWorkflowExecutionRequest' + required: true + responses: + "200": + description: OK + content: + application/json: + schema: + $ref: '#/components/schemas/WakeWorkflowExecutionResponse' + default: + description: Default error response + content: + application/json: + schema: + $ref: '#/components/schemas/Status' /system-info: get: tags: @@ -14797,6 +14879,15 @@ components: 2. Try to assign the next poll to a group without any pending polls, 3. If every group has some pending polls, assign the next poll to a group randomly according to the weights. + wakes: + type: array + items: + $ref: '#/components/schemas/Wake' + description: |- + Wakes pending for this execution when the task started, folded by + source. Not recorded in History and never re-supplied on replay: the + Workflow treats one as a reason to read its source now and records what + it read itself. PollerGroupInfo: type: object properties: @@ -19034,6 +19125,67 @@ components: user specified timeout, API call returns even if specified stage is not reached. format: enum description: Specifies client's intent to wait for Update results. + Wake: + type: object + properties: + source: + type: string + description: |- + What moved, as the sender names it. The server folds wakes by this value + and never interprets it. For a stream, the provider formats the stream's + identity, owner and topic, into it. + position: + type: string + description: |- + Where the source stands after the write that caused this wake, in the + store's own terms. Opaque to the server and handed to the Workflow as + received. A later wake's position replaces an earlier one's. + format: bytes + counter: + type: string + description: |- + Orders wakes from one source. Only the source's store can order its + positions, so the sender derives this from the position and the server + compares nothing else. A wake whose counter is not above the one the + server already holds is folded. + description: |- + A wake tells a running Workflow Execution that something it consumes has + moved, so it should run a Workflow Task and read from its own cursor. It is + a reason to run, not data: nothing is recorded in History and no record + travels with it. A store Temporal does not host sends one after the store + has acknowledged a write. Wakes with the same source fold: the server keeps + the highest counter it has seen per execution and source, so a repeated, + reordered or late wake changes nothing, and a burst of writes costs one task. + WakeWorkflowExecutionRequest: + type: object + properties: + namespace: + type: string + workflowExecution: + allOf: + - $ref: '#/components/schemas/WorkflowExecution' + description: |- + The Workflow to wake. `run_id` is optional. When set, the wake goes to + the current run of the chain that run belongs to, so a sender that + learned a run id before a continue-as-new still reaches the consumer, + and it is refused with NotFound once that chain has ended. When unset, + the wake goes to the current run under the Workflow Id, whichever chain. + wake: + $ref: '#/components/schemas/Wake' + identity: + type: string + description: The identity of the sender, for metrics and logs. + WakeWorkflowExecutionResponse: + type: object + properties: + runId: + type: string + description: The run the wake was stored on. + folded: + type: boolean + description: |- + True when the server already held this counter or a higher one for the + source, so the wake changed nothing. WorkerConfig: type: object properties: diff --git a/crates/protos/protos/api_upstream/temporal/api/workflow/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/workflow/v1/message.proto index 1ed33fa4c..b4220bc0d 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflow/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflow/v1/message.proto @@ -760,3 +760,26 @@ message WorkflowExecutionPauseInfo { // The reason for pausing the workflow execution. string reason = 3; } + +// A wake tells a running Workflow Execution that something it consumes has +// moved, so it should run a Workflow Task and read from its own cursor. It is +// a reason to run, not data: nothing is recorded in History and no record +// travels with it. A store Temporal does not host sends one after the store +// has acknowledged a write. Wakes with the same source fold: the server keeps +// the highest counter it has seen per execution and source, so a repeated, +// reordered or late wake changes nothing, and a burst of writes costs one task. +message Wake { + // What moved, as the sender names it. The server folds wakes by this value + // and never interprets it. For a stream, the provider formats the stream's + // identity, owner and topic, into it. + string source = 1; + // Where the source stands after the write that caused this wake, in the + // store's own terms. Opaque to the server and handed to the Workflow as + // received. A later wake's position replaces an earlier one's. + bytes position = 2; + // Orders wakes from one source. Only the source's store can order its + // positions, so the sender derives this from the position and the server + // compares nothing else. A wake whose counter is not above the one the + // server already holds is folded. + int64 counter = 3; +} diff --git a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto index b396de5b4..e8984162d 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto @@ -388,6 +388,12 @@ message PollWorkflowTaskQueueResponse { // Stream data attached to this task. Delivered out of band so the payloads // never enter History; only the offset ranges are recorded there. repeated temporal.api.stream.v1.StreamSlice stream_slices = 20; + + // Wakes pending for this execution when the task started, folded by + // source. Not recorded in History and never re-supplied on replay: the + // Workflow treats one as a reason to read its source now and records what + // it read itself. + repeated temporal.api.workflow.v1.Wake wakes = 21; } message RespondWorkflowTaskCompletedRequest { @@ -879,6 +885,27 @@ message SignalWorkflowExecutionResponse { temporal.api.common.v1.Link link = 1; } +message WakeWorkflowExecutionRequest { + string namespace = 1; + // The Workflow to wake. `run_id` is optional. When set, the wake goes to + // the current run of the chain that run belongs to, so a sender that + // learned a run id before a continue-as-new still reaches the consumer, + // and it is refused with NotFound once that chain has ended. When unset, + // the wake goes to the current run under the Workflow Id, whichever chain. + temporal.api.common.v1.WorkflowExecution workflow_execution = 2; + temporal.api.workflow.v1.Wake wake = 3; + // The identity of the sender, for metrics and logs. + string identity = 4; +} + +message WakeWorkflowExecutionResponse { + // The run the wake was stored on. + string run_id = 1; + // True when the server already held this counter or a higher one for the + // source, so the wake changed nothing. + bool folded = 2; +} + message SignalWithStartWorkflowExecutionRequest { string namespace = 1; string workflow_id = 2; diff --git a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto index 768553bd9..f100cdf28 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto @@ -483,6 +483,22 @@ service WorkflowService { }; } + // WakeWorkflowExecution asks a running Workflow Execution to run a Workflow + // Task because a source it consumes has moved. Unlike a Signal it records + // no event, carries no payload, and folds with other wakes for the same + // source, so a burst of writes costs one task. The Workflow learns the + // source and its position from the task and reads the source itself. + rpc WakeWorkflowExecution (WakeWorkflowExecutionRequest) returns (WakeWorkflowExecutionResponse) { + option (google.api.http) = { + post: "/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/wake" + body: "*" + additional_bindings { + post: "/api/v1/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/wake" + body: "*" + } + }; + } + // SignalWithStartWorkflowExecution is used to ensure a signal is sent to a workflow, even if // it isn't yet started. // diff --git a/crates/sdk-core-c-bridge/src/client.rs b/crates/sdk-core-c-bridge/src/client.rs index 3379ac5ed..3aed770cd 100644 --- a/crates/sdk-core-c-bridge/src/client.rs +++ b/crates/sdk-core-c-bridge/src/client.rs @@ -1107,6 +1107,9 @@ async fn call_workflow_service( validate_worker_deployment_version_compute_config ) } + "WakeWorkflowExecution" => { + rpc_call_on_trait!(client, call, WorkflowService, wake_workflow_execution) + } rpc => Err(anyhow::anyhow!("Unknown RPC call {rpc}")), } } From 0efd317d5c217d8f276b2419b061ff9a07747a03 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Thu, 1 Oct 2026 11:06:34 -0700 Subject: [PATCH 19/35] Received wakes from the poll response for parked external streams. A wake resumes waits without an event in History, so Core takes it from the poll response and treats it as an unparked wake. It is held until replay reaches the live task. --- .../src/core_tests/external_streams.rs | 172 ++++++++++++++++++ crates/sdk-core/src/protosext/mod.rs | 12 +- .../src/worker/workflow/history_update.rs | 1 + .../src/worker/workflow/managed_run.rs | 35 +++- crates/sdk-core/src/worker/workflow/mod.rs | 2 + 5 files changed, 216 insertions(+), 6 deletions(-) diff --git a/crates/sdk-core/src/core_tests/external_streams.rs b/crates/sdk-core/src/core_tests/external_streams.rs index 8a6241357..47bad33fd 100644 --- a/crates/sdk-core/src/core_tests/external_streams.rs +++ b/crates/sdk-core/src/core_tests/external_streams.rs @@ -53,6 +53,7 @@ use temporalio_common::{ enums::v1::{CommandType, EventType}, history::v1::History, query::v1::WorkflowQuery, + workflow::v1::Wake, workflowservice::v1::{ GetWorkflowExecutionHistoryResponse, RespondWorkflowTaskCompletedResponse, }, @@ -3405,6 +3406,177 @@ async fn an_ordinary_signal_still_reaches_its_handler() { worker.drain_pollers_and_shutdown().await; } +// --- wakes carried on the poll response -------------------------------------- + +fn poll_wake() -> Wake { + Wake { + source: "tokens".to_string(), + position: b"42-0".to_vec(), + counter: 42, + } +} + +/// A worker whose second Workflow Task arrives with `wakes` on its poll response. +/// +/// The wake-driven task records nothing of its own beyond its scheduled and started events, +/// which is the shape the server produces when only a wake is pending. `user_signal` adds an +/// ordinary Signal ahead of that task for the tests that need a job to look at. +fn worker_with_a_poll_wake(wakes: Vec, user_signal: bool) -> crate::Worker { + let mut t = TestHistoryBuilder::default(); + t.add_wfe_started_with_wft_timeout(Duration::from_secs(300)); + t.add_full_wf_task(); + if user_signal { + t.add_we_signaled("a-user-signal", vec![]); + } + t.add_full_wf_task(); + t.add_workflow_execution_completed(); + + let mut second = hist_to_poll_resp(&t, "fakeid".to_owned(), ResponseType::ToTaskNum(2)); + second.wakes = wakes; + let mut mock = build_mock_pollers(MockPollCfg::from_resp_batches( + "fakeid", + t, + [ResponseType::ToTaskNum(1), ResponseType::Raw(second.resp)], + mock_worker_client(), + )); + mock.worker_cfg(|w| { + w.task_types = WorkerTaskTypes::workflow_only(); + w.max_cached_workflows = 1; + }); + mock_worker(mock) +} + +#[tokio::test] +async fn a_poll_wake_resumes_a_parked_wait() { + // The server stored the wake on this run, so Core treats it as an unparked wake: no chain to + // check and no generation to match, even against a set parked at a generation of its own. + let worker = worker_with_a_poll_wake(vec![poll_wake()], false); + advance_to_the_signal_task(&worker, vec![1, 2], Some(7)).await; + + let second = worker.poll_workflow_activation().await.unwrap(); + + assert_eq!( + resolve_hints(&second), + vec![1, 2], + "a poll wake must mark every active wait, got {:?}", + second.jobs + ); + assert!( + !second.is_replaying, + "the task a poll wake arrives on is live work, got jobs {:?}", + second.jobs + ); + + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + second.run_id, + vec![CompleteWorkflowExecution::default().into()], + )) + .await + .unwrap(); + worker.drain_pollers_and_shutdown().await; +} + +#[tokio::test] +async fn a_poll_wake_with_no_waits_is_harmless() { + // Nothing is parked, so there is nothing to resume. The task still runs its ordinary work and + // the wake adds no job to it. + let worker = worker_with_a_poll_wake(vec![poll_wake()], true); + let first = worker.poll_workflow_activation().await.unwrap(); + worker + .complete_workflow_activation(WorkflowActivationCompletion::empty(first.run_id)) + .await + .unwrap(); + + let second = worker.poll_workflow_activation().await.unwrap(); + assert!( + second.jobs.iter().any(|j| matches!( + &j.variant, + Some(workflow_activation_job::Variant::SignalWorkflow(s)) + if s.signal_name == "a-user-signal" + )), + "the ordinary Signal must still reach its handler, got {:?}", + second.jobs + ); + assert_eq!( + resolve_hints(&second), + Vec::::new(), + "with no waits there is nothing for the wake to resolve" + ); + + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + second.run_id, + vec![CompleteWorkflowExecution::default().into()], + )) + .await + .unwrap(); + worker.drain_pollers_and_shutdown().await; +} + +#[tokio::test] +async fn a_poll_wake_is_not_applied_while_replaying() { + // A cold worker receives the whole History with the live task's wakes. The wakes describe the + // live task, so the replayed activation must not carry them, and they must not be spent + // there either: the waits that exist once replay catches up are the ones they resume. The + // Signal keeps the two tasks apart; back to back, Core would treat them as one live task. + let mut t = TestHistoryBuilder::default(); + t.add_wfe_started_with_wft_timeout(Duration::from_secs(300)); + t.add_full_wf_task(); + t.add_we_signaled("a-user-signal", vec![]); + t.add_full_wf_task(); + t.add_workflow_execution_completed(); + + let mut cold = hist_to_poll_resp(&t, "fakeid".to_owned(), ResponseType::ToTaskNum(2)); + cold.wakes = vec![poll_wake()]; + let mut mock = build_mock_pollers(MockPollCfg::from_resp_batches( + "fakeid", + t, + [ResponseType::Raw(cold.resp)], + mock_worker_client(), + )); + mock.worker_cfg(|w| { + w.task_types = WorkerTaskTypes::workflow_only(); + w.max_cached_workflows = 1; + }); + let worker = mock_worker(mock); + + let replayed = worker.poll_workflow_activation().await.unwrap(); + assert!(replayed.is_replaying); + assert_eq!( + resolve_hints(&replayed), + Vec::::new(), + "a replayed activation must not carry a poll wake, got {:?}", + replayed.jobs + ); + let run_id = replayed.run_id.clone(); + worker + .seed_external_stream_waits(&run_id, vec![1], None, false) + .await; + worker + .complete_workflow_activation(WorkflowActivationCompletion::empty(run_id.clone())) + .await + .unwrap(); + + let live = worker.poll_workflow_activation().await.unwrap(); + assert!(!live.is_replaying, "got jobs {:?}", live.jobs); + assert_eq!( + resolve_hints(&live), + vec![1], + "the wake held through replay must resume the live task, got {:?}", + live.jobs + ); + + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + live.run_id, + vec![CompleteWorkflowExecution::default().into()], + )) + .await + .unwrap(); + worker.drain_pollers_and_shutdown().await; +} + // --- marker emission (C9, C14b) ---------------------------------------------- /// A history for a Workflow Task that records a marker and then starts a timer. diff --git a/crates/sdk-core/src/protosext/mod.rs b/crates/sdk-core/src/protosext/mod.rs index b2adcde96..f83c6af6d 100644 --- a/crates/sdk-core/src/protosext/mod.rs +++ b/crates/sdk-core/src/protosext/mod.rs @@ -39,6 +39,7 @@ use temporalio_common::protos::{ history::v1::{History, HistoryEvent, MarkerRecordedEventAttributes, history_event}, query::v1::WorkflowQuery, sdk::v1::UserMetadata, + workflow::v1::Wake, workflowservice::v1::PollWorkflowTaskQueueResponse, }, utilities::TryIntoOrNone, @@ -64,6 +65,9 @@ pub(crate) struct ValidPollWFTQResponse { pub(crate) query_requests: Vec, /// Protocol messages pub(crate) messages: Vec, + /// Wakes pending for the execution when this task started. Live only: the server never + /// supplies them on replay and never records them in History. + pub(crate) wakes: Vec, /// Zero-size field to prevent explicit construction _cant_construct_me: (), @@ -75,7 +79,8 @@ impl Debug for ValidPollWFTQResponse { f, "ValidWFT {{ task_token: {}, task_queue: {}, workflow_execution: {:?}, \ workflow_type: {}, attempt: {}, previous_started_event_id: {}, started_event_id {}, \ - history_length: {}, first_evt_in_hist_id: {:?}, legacy_query: {:?}, queries: {:?} }}", + history_length: {}, first_evt_in_hist_id: {:?}, legacy_query: {:?}, queries: {:?}, \ + wakes: {:?} }}", self.task_token, self.task_queue, self.workflow_execution, @@ -86,7 +91,8 @@ impl Debug for ValidPollWFTQResponse { self.history.events.len(), self.history.events.first().map(|e| e.event_id), self.legacy_query, - self.query_requests + self.query_requests, + self.wakes ) } } @@ -109,6 +115,7 @@ impl TryFrom for ValidPollWFTQResponse { query, queries, messages, + wakes, .. } => { if task_token.is_empty() { @@ -133,6 +140,7 @@ impl TryFrom for ValidPollWFTQResponse { legacy_query: query, query_requests, messages, + wakes, _cant_construct_me: (), }) } diff --git a/crates/sdk-core/src/worker/workflow/history_update.rs b/crates/sdk-core/src/worker/workflow/history_update.rs index 856589946..7c105bd5e 100644 --- a/crates/sdk-core/src/worker/workflow/history_update.rs +++ b/crates/sdk-core/src/worker/workflow/history_update.rs @@ -162,6 +162,7 @@ impl HistoryPaginator { query_requests: wft.query_requests, update, messages: wft.messages, + wakes: wft.wakes, }; Ok((paginator, prepared)) } diff --git a/crates/sdk-core/src/worker/workflow/managed_run.rs b/crates/sdk-core/src/worker/workflow/managed_run.rs index 75384287f..b5bb11eac 100644 --- a/crates/sdk-core/src/worker/workflow/managed_run.rs +++ b/crates/sdk-core/src/worker/workflow/managed_run.rs @@ -53,6 +53,7 @@ use temporalio_common::protos::{ temporal::api::{ enums::v1::{VersioningBehavior, WorkflowTaskFailedCause}, failure::v1::Failure, + workflow::v1::Wake, }, }; use tokio::sync::oneshot; @@ -115,6 +116,10 @@ pub(super) struct ManagedRun { paginator: Option, completion_waiting_on_page_fetch: Option, config: Arc, + /// Wakes from the poll response of the current task, held until the machines have replayed up + /// to that task. They describe the live task, so applying one mid-replay would resume waits + /// the replayed History never resumed. + pending_poll_wakes: Vec, } impl ManagedRun { pub(super) fn new( @@ -141,6 +146,7 @@ impl ManagedRun { paginator: None, completion_waiting_on_page_fetch: None, config, + pending_poll_wakes: vec![], }; let rua = me.incoming_wft(wft); (me, rua) @@ -254,6 +260,10 @@ impl ManagedRun { } self.paginator = Some(pwft.paginator); + // Wakes held for an earlier task that never caught up belong to that task. The server + // hands any still pending to this one, so they are replaced rather than accumulated. A + // legacy query task runs no Workflow code and so cannot act on a wake. + self.pending_poll_wakes = if was_legacy_query { vec![] } else { work.wakes }; // A Workflow Task is open from here until it is reported. The unwritten-annotation // invariant is stated against *that*, not against quiescence -- a task can accumulate a // delta and complete without ever asking to be retained. @@ -2145,16 +2155,34 @@ impl ManagedRun { (outcome.into(), act) } - /// Classifies the wake Signals the machines decoded out of this task's history. + /// Classifies the wake Signals the machines decoded out of this task's history, together with + /// the wakes the poll response carried for it. /// - /// Returns `true` if any of them should wake the Run. Every one is suppressed from user + /// Returns `true` if any of them should wake the Run. Every Signal is suppressed from user /// handlers regardless -- that already happened in the machines -- so what is decided here is /// only whether the Run resumes. fn apply_external_stream_wakes(&mut self) -> bool { let wakes = self.wfm.machines.take_external_stream_wakes(); - if wakes.is_empty() { + let poll_wakes = if self.wfm.machines.replaying { + vec![] + } else { + mem::take(&mut self.pending_poll_wakes) + }; + if wakes.is_empty() && poll_wakes.is_empty() { return false; } + + // A poll wake counts as an unparked wake (generation 0), which is never rejected. The + // server resolved the chain when it stored the wake on this run, so there is no chain + // identity left to compare. + for wake in &poll_wakes { + debug!( + source = %wake.source, + counter = wake.counter, + "Resuming external stream waits for a poll response wake" + ); + } + let mut resume = !poll_wakes.is_empty(); let chain = self .wfm .machines @@ -2162,7 +2190,6 @@ impl ManagedRun { .map(|info| info.first_execution_run_id.clone()) .unwrap_or_default(); - let mut resume = false; for wake in wakes { // Chain identity, not Run identity. The Signal is addressed to the Workflow ID // without a Run ID, so it always lands on the current Run of the chain -- and a diff --git a/crates/sdk-core/src/worker/workflow/mod.rs b/crates/sdk-core/src/worker/workflow/mod.rs index 5450c6939..e0cb8705f 100644 --- a/crates/sdk-core/src/worker/workflow/mod.rs +++ b/crates/sdk-core/src/worker/workflow/mod.rs @@ -93,6 +93,7 @@ use temporalio_common::{ query::v1::WorkflowQuery, sdk::v1::{UserMetadata, WorkflowTaskCompletedMetadata}, taskqueue::v1::StickyExecutionAttributes, + workflow::v1::Wake, workflowservice::v1::{PollActivityTaskQueueResponse, get_system_info_response}, }, }, @@ -1140,6 +1141,7 @@ struct PreparedWFT { query_requests: Vec, update: HistoryUpdate, messages: Vec, + wakes: Vec, } impl PreparedWFT { From b8e948aa30b3344ed59cd88602b4232e014aea32 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Thu, 1 Oct 2026 11:07:45 -0700 Subject: [PATCH 20/35] Kept the vendored OpenAPI files at their base. The wake copy carries only the three proto files, matching the main-lineage copy, so the later merges of both lineages agree. --- .../api_upstream/openapi/openapiv2.json | 152 ------------------ .../api_upstream/openapi/openapiv3.yaml | 152 ------------------ 2 files changed, 304 deletions(-) diff --git a/crates/protos/protos/api_upstream/openapi/openapiv2.json b/crates/protos/protos/api_upstream/openapi/openapiv2.json index 0a90bd496..1437120e5 100644 --- a/crates/protos/protos/api_upstream/openapi/openapiv2.json +++ b/crates/protos/protos/api_upstream/openapi/openapiv2.json @@ -4575,51 +4575,6 @@ ] } }, - "/api/v1/namespaces/{namespace}/workflows/{workflowExecution.workflowId}/wake": { - "post": { - "summary": "WakeWorkflowExecution asks a running Workflow Execution to run a Workflow\nTask because a source it consumes has moved. Unlike a Signal it records\nno event, carries no payload, and folds with other wakes for the same\nsource, so a burst of writes costs one task. The Workflow learns the\nsource and its position from the task and reads the source itself.", - "operationId": "WakeWorkflowExecution2", - "responses": { - "200": { - "description": "A successful response.", - "schema": { - "$ref": "#/definitions/v1WakeWorkflowExecutionResponse" - } - }, - "default": { - "description": "An unexpected error response.", - "schema": { - "$ref": "#/definitions/rpcStatus" - } - } - }, - "parameters": [ - { - "name": "namespace", - "in": "path", - "required": true, - "type": "string" - }, - { - "name": "workflowExecution.workflowId", - "in": "path", - "required": true, - "type": "string" - }, - { - "name": "body", - "in": "body", - "required": true, - "schema": { - "$ref": "#/definitions/WorkflowServiceWakeWorkflowExecutionBody" - } - } - ], - "tags": [ - "WorkflowService" - ] - } - }, "/api/v1/namespaces/{namespace}/workflows/{workflowId}": { "post": { "summary": "StartWorkflowExecution starts a new workflow execution.", @@ -10203,51 +10158,6 @@ ] } }, - "/namespaces/{namespace}/workflows/{workflowExecution.workflowId}/wake": { - "post": { - "summary": "WakeWorkflowExecution asks a running Workflow Execution to run a Workflow\nTask because a source it consumes has moved. Unlike a Signal it records\nno event, carries no payload, and folds with other wakes for the same\nsource, so a burst of writes costs one task. The Workflow learns the\nsource and its position from the task and reads the source itself.", - "operationId": "WakeWorkflowExecution", - "responses": { - "200": { - "description": "A successful response.", - "schema": { - "$ref": "#/definitions/v1WakeWorkflowExecutionResponse" - } - }, - "default": { - "description": "An unexpected error response.", - "schema": { - "$ref": "#/definitions/rpcStatus" - } - } - }, - "parameters": [ - { - "name": "namespace", - "in": "path", - "required": true, - "type": "string" - }, - { - "name": "workflowExecution.workflowId", - "in": "path", - "required": true, - "type": "string" - }, - { - "name": "body", - "in": "body", - "required": true, - "schema": { - "$ref": "#/definitions/WorkflowServiceWakeWorkflowExecutionBody" - } - } - ], - "tags": [ - "WorkflowService" - ] - } - }, "/namespaces/{namespace}/workflows/{workflowId}": { "post": { "summary": "StartWorkflowExecution starts a new workflow execution.", @@ -13519,27 +13429,6 @@ }, "description": "Used to validate the compute config without attaching it to a Worker Deployment Version." }, - "WorkflowServiceWakeWorkflowExecutionBody": { - "type": "object", - "properties": { - "workflowExecution": { - "type": "object", - "properties": { - "runId": { - "type": "string" - } - }, - "description": "The Workflow to wake. `run_id` is optional. When set, the wake goes to\nthe current run of the chain that run belongs to, so a sender that\nlearned a run id before a continue-as-new still reaches the consumer,\nand it is refused with NotFound once that chain has ended. When unset,\nthe wake goes to the current run under the Workflow Id, whichever chain." - }, - "wake": { - "$ref": "#/definitions/v1Wake" - }, - "identity": { - "type": "string", - "description": "The identity of the sender, for metrics and logs." - } - } - }, "apiActivityV1CallbackInfo": { "type": "object", "properties": { @@ -18509,14 +18398,6 @@ "pollerGroupsInfo": { "$ref": "#/definitions/v1PollerGroupsInfo", "description": "The weighted, versioned list of poller groups IDs that client should use for future polls to\nthis task queue. Client should ignore this if it has already applied a snapshot with a\nversion greater than or equal to `poller_groups_info.version`. Client is expected to:\n 1. Maintain minimum number of pollers no less than the number of groups.\n 2. Try to assign the next poll to a group without any pending polls,\n 3. If every group has some pending polls, assign the next poll to a group randomly\n according to the weights." - }, - "wakes": { - "type": "array", - "items": { - "type": "object", - "$ref": "#/definitions/v1Wake" - }, - "description": "Wakes pending for this execution when the task started, folded by\nsource. Not recorded in History and never re-supplied on replay: the\nWorkflow treats one as a reason to read its source now and records what\nit read itself." } } }, @@ -20902,39 +20783,6 @@ }, "description": "Specifies client's intent to wait for Update results." }, - "v1Wake": { - "type": "object", - "properties": { - "source": { - "type": "string", - "description": "What moved, as the sender names it. The server folds wakes by this value\nand never interprets it. For a stream, the provider formats the stream's\nidentity, owner and topic, into it." - }, - "position": { - "type": "string", - "format": "byte", - "description": "Where the source stands after the write that caused this wake, in the\nstore's own terms. Opaque to the server and handed to the Workflow as\nreceived. A later wake's position replaces an earlier one's." - }, - "counter": { - "type": "string", - "format": "int64", - "description": "Orders wakes from one source. Only the source's store can order its\npositions, so the sender derives this from the position and the server\ncompares nothing else. A wake whose counter is not above the one the\nserver already holds is folded." - } - }, - "description": "A wake tells a running Workflow Execution that something it consumes has\nmoved, so it should run a Workflow Task and read from its own cursor. It is\na reason to run, not data: nothing is recorded in History and no record\ntravels with it. A store Temporal does not host sends one after the store\nhas acknowledged a write. Wakes with the same source fold: the server keeps\nthe highest counter it has seen per execution and source, so a repeated,\nreordered or late wake changes nothing, and a burst of writes costs one task." - }, - "v1WakeWorkflowExecutionResponse": { - "type": "object", - "properties": { - "runId": { - "type": "string", - "description": "The run the wake was stored on." - }, - "folded": { - "type": "boolean", - "description": "True when the server already held this counter or a higher one for the\nsource, so the wake changed nothing." - } - } - }, "v1WorkerConfig": { "type": "object", "properties": { diff --git a/crates/protos/protos/api_upstream/openapi/openapiv3.yaml b/crates/protos/protos/api_upstream/openapi/openapiv3.yaml index ef43e36ea..007d77168 100644 --- a/crates/protos/protos/api_upstream/openapi/openapiv3.yaml +++ b/crates/protos/protos/api_upstream/openapi/openapiv3.yaml @@ -4695,47 +4695,6 @@ paths: application/json: schema: $ref: '#/components/schemas/Status' - /api/v1/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/wake: - post: - tags: - - WorkflowService - description: |- - WakeWorkflowExecution asks a running Workflow Execution to run a Workflow - Task because a source it consumes has moved. Unlike a Signal it records - no event, carries no payload, and folds with other wakes for the same - source, so a burst of writes costs one task. The Workflow learns the - source and its position from the task and reads the source itself. - operationId: WakeWorkflowExecution - parameters: - - name: namespace - in: path - required: true - schema: - type: string - - name: workflow_execution.workflow_id - in: path - required: true - schema: - type: string - requestBody: - content: - application/json: - schema: - $ref: '#/components/schemas/WakeWorkflowExecutionRequest' - required: true - responses: - "200": - description: OK - content: - application/json: - schema: - $ref: '#/components/schemas/WakeWorkflowExecutionResponse' - default: - description: Default error response - content: - application/json: - schema: - $ref: '#/components/schemas/Status' /api/v1/nexus/endpoints: get: tags: @@ -9776,47 +9735,6 @@ paths: application/json: schema: $ref: '#/components/schemas/Status' - /namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/wake: - post: - tags: - - WorkflowService - description: |- - WakeWorkflowExecution asks a running Workflow Execution to run a Workflow - Task because a source it consumes has moved. Unlike a Signal it records - no event, carries no payload, and folds with other wakes for the same - source, so a burst of writes costs one task. The Workflow learns the - source and its position from the task and reads the source itself. - operationId: WakeWorkflowExecution - parameters: - - name: namespace - in: path - required: true - schema: - type: string - - name: workflow_execution.workflow_id - in: path - required: true - schema: - type: string - requestBody: - content: - application/json: - schema: - $ref: '#/components/schemas/WakeWorkflowExecutionRequest' - required: true - responses: - "200": - description: OK - content: - application/json: - schema: - $ref: '#/components/schemas/WakeWorkflowExecutionResponse' - default: - description: Default error response - content: - application/json: - schema: - $ref: '#/components/schemas/Status' /system-info: get: tags: @@ -14879,15 +14797,6 @@ components: 2. Try to assign the next poll to a group without any pending polls, 3. If every group has some pending polls, assign the next poll to a group randomly according to the weights. - wakes: - type: array - items: - $ref: '#/components/schemas/Wake' - description: |- - Wakes pending for this execution when the task started, folded by - source. Not recorded in History and never re-supplied on replay: the - Workflow treats one as a reason to read its source now and records what - it read itself. PollerGroupInfo: type: object properties: @@ -19125,67 +19034,6 @@ components: user specified timeout, API call returns even if specified stage is not reached. format: enum description: Specifies client's intent to wait for Update results. - Wake: - type: object - properties: - source: - type: string - description: |- - What moved, as the sender names it. The server folds wakes by this value - and never interprets it. For a stream, the provider formats the stream's - identity, owner and topic, into it. - position: - type: string - description: |- - Where the source stands after the write that caused this wake, in the - store's own terms. Opaque to the server and handed to the Workflow as - received. A later wake's position replaces an earlier one's. - format: bytes - counter: - type: string - description: |- - Orders wakes from one source. Only the source's store can order its - positions, so the sender derives this from the position and the server - compares nothing else. A wake whose counter is not above the one the - server already holds is folded. - description: |- - A wake tells a running Workflow Execution that something it consumes has - moved, so it should run a Workflow Task and read from its own cursor. It is - a reason to run, not data: nothing is recorded in History and no record - travels with it. A store Temporal does not host sends one after the store - has acknowledged a write. Wakes with the same source fold: the server keeps - the highest counter it has seen per execution and source, so a repeated, - reordered or late wake changes nothing, and a burst of writes costs one task. - WakeWorkflowExecutionRequest: - type: object - properties: - namespace: - type: string - workflowExecution: - allOf: - - $ref: '#/components/schemas/WorkflowExecution' - description: |- - The Workflow to wake. `run_id` is optional. When set, the wake goes to - the current run of the chain that run belongs to, so a sender that - learned a run id before a continue-as-new still reaches the consumer, - and it is refused with NotFound once that chain has ended. When unset, - the wake goes to the current run under the Workflow Id, whichever chain. - wake: - $ref: '#/components/schemas/Wake' - identity: - type: string - description: The identity of the sender, for metrics and logs. - WakeWorkflowExecutionResponse: - type: object - properties: - runId: - type: string - description: The run the wake was stored on. - folded: - type: boolean - description: |- - True when the server already held this counter or a higher one for the - source, so the wake changed nothing. WorkerConfig: type: object properties: From 844901480aa83bf33a7453b91c6c57c6205e9a2f Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Thu, 1 Oct 2026 11:09:52 -0700 Subject: [PATCH 21/35] Carried the wake fold rule comments from the api fork. Applied from api 086d85d. Comment-only. --- .../temporal/api/workflow/v1/message.proto | 18 ++++++++++-------- .../workflowservice/v1/request_response.proto | 5 +++-- 2 files changed, 13 insertions(+), 10 deletions(-) diff --git a/crates/protos/protos/api_upstream/temporal/api/workflow/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/workflow/v1/message.proto index b4220bc0d..276a587bf 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflow/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflow/v1/message.proto @@ -765,9 +765,10 @@ message WorkflowExecutionPauseInfo { // moved, so it should run a Workflow Task and read from its own cursor. It is // a reason to run, not data: nothing is recorded in History and no record // travels with it. A store Temporal does not host sends one after the store -// has acknowledged a write. Wakes with the same source fold: the server keeps -// the highest counter it has seen per execution and source, so a repeated, -// reordered or late wake changes nothing, and a burst of writes costs one task. +// has acknowledged a write. Wakes for the same source fold while one is +// pending and no task has received it yet, so a burst of writes costs one +// task. Once a task has been handed the wake, a new wake for the source is +// accepted again, because only the receiver knows what it read. message Wake { // What moved, as the sender names it. The server folds wakes by this value // and never interprets it. For a stream, the provider formats the stream's @@ -775,11 +776,12 @@ message Wake { string source = 1; // Where the source stands after the write that caused this wake, in the // store's own terms. Opaque to the server and handed to the Workflow as - // received. A later wake's position replaces an earlier one's. + // received. Among wakes folded together, the position with the highest + // counter is the one delivered. bytes position = 2; - // Orders wakes from one source. Only the source's store can order its - // positions, so the sender derives this from the position and the server - // compares nothing else. A wake whose counter is not above the one the - // server already holds is folded. + // Orders wakes from one source, so that folding keeps the latest position. + // Only the source's store can order its positions, so the sender derives + // this from the position. It is not a durable identity: a wake sent after + // a task received an equal counter is accepted again. int64 counter = 3; } diff --git a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto index e8984162d..08421a371 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto @@ -901,8 +901,9 @@ message WakeWorkflowExecutionRequest { message WakeWorkflowExecutionResponse { // The run the wake was stored on. string run_id = 1; - // True when the server already held this counter or a higher one for the - // source, so the wake changed nothing. + // True when a wake for the source was already pending and no task had + // received it yet, so this one changed nothing but possibly the position + // and no new task results from it. bool folded = 2; } From 30969f9813ab81fb3bf3b604e007fc9e32e7ef86 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Thu, 1 Oct 2026 17:28:42 -0700 Subject: [PATCH 22/35] Carried the notification channel protos and client calls in the Core's api tree. Applies the api commit's diff to the vendored tree, exposes the channel calls on the raw client and the C bridge, and classifies Notification.metadata for payload limits. --- crates/client/src/grpc.rs | 45 +++++++++++ crates/common/build.rs | 3 + crates/protos/build.rs | 1 + .../temporal/api/command/v1/message.proto | 9 +++ .../temporal/api/enums/v1/command_type.proto | 1 + .../temporal/api/enums/v1/event_type.proto | 3 + .../temporal/api/history/v1/message.proto | 14 ++++ .../api/notification/v1/message.proto | 55 ++++++++++++++ .../temporal/api/workflow/v1/message.proto | 4 + .../workflowservice/v1/request_response.proto | 74 +++++++++++++++++++ .../api/workflowservice/v1/service.proto | 69 +++++++++++++++++ crates/protos/src/protos/mod.rs | 13 ++++ crates/sdk-core-c-bridge/src/client.rs | 15 ++++ 13 files changed, 306 insertions(+) create mode 100644 crates/protos/protos/api_upstream/temporal/api/notification/v1/message.proto diff --git a/crates/client/src/grpc.rs b/crates/client/src/grpc.rs index 7fdc2d13c..0d6bb6ddf 100644 --- a/crates/client/src/grpc.rs +++ b/crates/client/src/grpc.rs @@ -964,6 +964,51 @@ proxier! { r.extensions_mut().insert(labels); } ); + ( + notify_channel, + NotifyChannelRequest, + NotifyChannelResponse, + |r| { + let labels = namespaced_request!(r); + r.extensions_mut().insert(labels); + } + ); + ( + register_channel_listener, + RegisterChannelListenerRequest, + RegisterChannelListenerResponse, + |r| { + let labels = namespaced_request!(r); + r.extensions_mut().insert(labels); + } + ); + ( + unregister_channel_listener, + UnregisterChannelListenerRequest, + UnregisterChannelListenerResponse, + |r| { + let labels = namespaced_request!(r); + r.extensions_mut().insert(labels); + } + ); + ( + poll_channel, + PollChannelRequest, + PollChannelResponse, + |r| { + let labels = namespaced_request!(r); + r.extensions_mut().insert(labels); + } + ); + ( + describe_channel, + DescribeChannelRequest, + DescribeChannelResponse, + |r| { + let labels = namespaced_request!(r); + r.extensions_mut().insert(labels); + } + ); ( signal_with_start_workflow_execution, SignalWithStartWorkflowExecutionRequest, diff --git a/crates/common/build.rs b/crates/common/build.rs index 1baae9de8..ecf894587 100644 --- a/crates/common/build.rs +++ b/crates/common/build.rs @@ -825,6 +825,9 @@ const NOT_VALIDATED_FIELDS: &[&str] = &[ // is not a payload size the SDK can mirror. "temporal.api.stream.v1.StreamRecord.body", "temporal.api.stream.v1.StreamRecord.metadata", + // Notification metadata: the server bounds it with its own notification size limit, not + // the blob limit, so the SDK has nothing to mirror. + "temporal.api.notification.v1.Notification.metadata", // Dedicated, non-fetchable limits (not blob/memo, not in DescribeNamespace): UserMetadata // (nexus-start only); Nexus EndpointSpec.description (maxDescriptionSize; cloud variant cloud-only). "temporal.api.sdk.v1.UserMetadata.details", diff --git a/crates/protos/build.rs b/crates/protos/build.rs index 7365df524..5ca388c9f 100644 --- a/crates/protos/build.rs +++ b/crates/protos/build.rs @@ -29,6 +29,7 @@ const SERDE_DERIVE_PREFIXES: &[&str] = &[ ".temporal.api.namespace", ".temporal.api.nexus", ".temporal.api.nexusservices", + ".temporal.api.notification", ".temporal.api.operatorservice", ".temporal.api.protocol", ".temporal.api.query", diff --git a/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto index 7d10e441b..a08704c3e 100644 --- a/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto @@ -327,6 +327,7 @@ message Command { RequestCancelNexusOperationCommandAttributes request_cancel_nexus_operation_command_attributes = 19; AppendStreamRecordsCommandAttributes append_stream_records_command_attributes = 20; SubscribeStreamCommandAttributes subscribe_stream_command_attributes = 21; + SubscribeNotificationChannelCommandAttributes subscribe_notification_channel_command_attributes = 22; } } @@ -360,3 +361,11 @@ message SubscribeStreamCommandAttributes { // replay does not resolve it again. int64 start_offset = 2; } + +// Makes the Workflow a listener of a notification channel for this run. The +// next notifications on the channel arrive on the scheduled event of a Workflow +// Task. The subscription ends with the run, and a successor subscribes again. +message SubscribeNotificationChannelCommandAttributes { + // The channel to listen on, as the writers name it. + string channel = 1; +} diff --git a/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto b/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto index 967169b19..b8eada084 100644 --- a/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto +++ b/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto @@ -31,4 +31,5 @@ enum CommandType { COMMAND_TYPE_REQUEST_CANCEL_NEXUS_OPERATION = 18; COMMAND_TYPE_APPEND_STREAM_RECORDS = 19; COMMAND_TYPE_SUBSCRIBE_STREAM = 20; + COMMAND_TYPE_SUBSCRIBE_NOTIFICATION_CHANNEL = 21; } diff --git a/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto b/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto index a386b1db7..f70de73fe 100644 --- a/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto +++ b/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto @@ -183,4 +183,7 @@ enum EventType { // batch, and carrying only the offset range it landed at: the bodies go to // the stream's own log, never into History. EVENT_TYPE_WORKFLOW_STREAM_RECORDS_APPENDED = 62; + // A Workflow became a listener of a notification channel for its run. + // The notifications themselves ride the WorkflowTaskScheduled event. + EVENT_TYPE_WORKFLOW_NOTIFICATION_CHANNEL_SUBSCRIBED = 63; } diff --git a/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto index 8deca8d15..e466ba505 100644 --- a/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto @@ -18,6 +18,7 @@ import "temporal/api/enums/v1/update.proto"; import "temporal/api/enums/v1/workflow.proto"; import "temporal/api/common/v1/message.proto"; import "temporal/api/stream/v1/message.proto"; +import "temporal/api/notification/v1/message.proto"; import "temporal/api/deployment/v1/message.proto"; import "temporal/api/failure/v1/message.proto"; import "temporal/api/taskqueue/v1/message.proto"; @@ -303,6 +304,10 @@ message WorkflowTaskScheduledEventAttributes { google.protobuf.Duration start_to_close_timeout = 2; // Starting at 1, how many attempts there have been to complete this task int32 attempt = 3; + // Notifications for channels this Workflow listens to, folded per channel + // since the last task was scheduled. In History so a Workflow may act on + // them deterministically and replay sees the same. + repeated temporal.api.notification.v1.Notification notifications = 4; } message WorkflowTaskStartedEventAttributes { @@ -974,6 +979,14 @@ message WorkflowStreamSubscribedEventAttributes { int64 start_offset = 3; } +message WorkflowNotificationChannelSubscribedEventAttributes { + // The WorkflowTaskCompleted event of the task whose command created this + // subscription. + int64 workflow_task_completed_event_id = 1; + // The channel the Workflow listens on for the rest of this run. + string channel = 2; +} + message WorkflowStreamRecordsAppendedEventAttributes { // The WorkflowTaskCompleted event of the task whose command appended this // batch. @@ -1313,6 +1326,7 @@ message HistoryEvent { WorkflowExecutionTimeSkippingTransitionedEventAttributes workflow_execution_time_skipping_transitioned_event_attributes = 65; WorkflowStreamSubscribedEventAttributes workflow_stream_subscribed_event_attributes = 66; WorkflowStreamRecordsAppendedEventAttributes workflow_stream_records_appended_event_attributes = 67; + WorkflowNotificationChannelSubscribedEventAttributes workflow_notification_channel_subscribed_event_attributes = 68; } } diff --git a/crates/protos/protos/api_upstream/temporal/api/notification/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/notification/v1/message.proto new file mode 100644 index 000000000..63e1b7716 --- /dev/null +++ b/crates/protos/protos/api_upstream/temporal/api/notification/v1/message.proto @@ -0,0 +1,55 @@ +syntax = "proto3"; + +package temporal.api.notification.v1; + +option go_package = "go.temporal.io/api/notification/v1;notification"; +option java_package = "io.temporal.api.notification.v1"; +option java_multiple_files = true; +option java_outer_classname = "MessageProto"; +option ruby_package = "Temporalio::Api::Notification::V1"; +option csharp_namespace = "Temporalio.Api.Notification.V1"; + +import "google/protobuf/timestamp.proto"; + +import "temporal/api/common/v1/message.proto"; + +// A notification tells the listeners of a channel that a source they consume +// has moved. It is not data: the listener reads the source itself. A channel +// is named by the writer and its listeners; for a stream, the provider formats +// the stream's identity into the name. Writers never learn who listens. The +// server folds notifications per listener while one is pending and no task has +// been scheduled for it, keeping the one with the highest counter. +message Notification { + // The channel the writer notified. Listeners register on the same name. + string channel = 1; + // Where the source stands after the write that caused this notification, + // in the writer's terms. Opaque to the server. + bytes position = 2; + // Orders notifications from one channel's writers. The writer derives it + // from the position, since only the source can order its positions. Among + // notifications folded together, the one with the highest counter is kept. + int64 counter = 3; + // Details for the listener, such as which topic moved. Bounded in size and + // carried as payloads, so a codec applies as to any payload. + map metadata = 4; +} + +// A listener of a channel: a Workflow Execution woken with a Workflow Task, or +// a callback the server invokes with each notification. +message ChannelListener { + // Assigned by the server when the listener registers. + string listener_id = 1; + oneof listener { + WorkflowListener workflow = 2; + temporal.api.common.v1.Callback callback = 3; + } + google.protobuf.Timestamp registered_time = 4; +} + +// A Workflow Execution listening on a channel. +message WorkflowListener { + string workflow_id = 1; + // The run that subscribed. The server follows a continue-as-new to the + // chain's current run when it delivers. + string run_id = 2; +} diff --git a/crates/protos/protos/api_upstream/temporal/api/workflow/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/workflow/v1/message.proto index 276a587bf..0c414794e 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflow/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflow/v1/message.proto @@ -769,6 +769,10 @@ message WorkflowExecutionPauseInfo { // pending and no task has received it yet, so a burst of writes costs one // task. Once a task has been handed the wake, a new wake for the source is // accepted again, because only the receiver knows what it read. +// +// Superseded by the notification channel (`NotifyChannel` and the +// `SubscribeNotificationChannel` command). Kept for one round and slated for +// removal. message Wake { // What moved, as the sender names it. The server folds wakes by this value // and never interprets it. For a stream, the provider formats the stream's diff --git a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto index 08421a371..c0af15c8d 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto @@ -25,6 +25,7 @@ import "temporal/api/enums/v1/nexus.proto"; import "temporal/api/activity/v1/message.proto"; import "temporal/api/common/v1/message.proto"; import "temporal/api/stream/v1/message.proto"; +import "temporal/api/notification/v1/message.proto"; import "temporal/api/history/v1/message.proto"; import "temporal/api/workflow/v1/message.proto"; import "temporal/api/command/v1/message.proto"; @@ -907,6 +908,79 @@ message WakeWorkflowExecutionResponse { bool folded = 2; } +message NotifyChannelRequest { + string namespace = 1; + temporal.api.notification.v1.Notification notification = 2; + // The identity of the writer, for metrics and logs. + string identity = 3; + // Used to de-dupe a retried notification. + string request_id = 4; +} + +message NotifyChannelResponse { + // Listeners registered when the notification was accepted. Zero means the + // notification was retained for pollers and woke nobody. + int32 listener_count = 1; +} + +message RegisterChannelListenerRequest { + string namespace = 1; + string channel = 2; + // Invoked with each notification on the channel. + temporal.api.common.v1.Callback callback = 3; + // Used to de-dupe a retried registration. + string request_id = 4; + // The identity of the caller, for metrics and logs. + string identity = 5; +} + +message RegisterChannelListenerResponse { + string listener_id = 1; +} + +message UnregisterChannelListenerRequest { + string namespace = 1; + string channel = 2; + string listener_id = 3; + // The identity of the caller, for metrics and logs. + string identity = 4; +} + +message UnregisterChannelListenerResponse { +} + +message PollChannelRequest { + string namespace = 1; + string channel = 2; + // Only notifications with a counter above this one are returned. + // (-- api-linter: core::0140::prepositions=disabled + // aip.dev/not-precedent: "after" names the exclusive lower bound. --) + int64 after_counter = 3; + // How long to wait for a notification when none is retained above + // `after_counter`. + google.protobuf.Duration wait = 4; + // At most this many notifications are returned. Zero means the server's + // default. + int32 max_notifications = 5; +} + +message PollChannelResponse { + repeated temporal.api.notification.v1.Notification notifications = 1; +} + +message DescribeChannelRequest { + string namespace = 1; + string channel = 2; +} + +message DescribeChannelResponse { + repeated temporal.api.notification.v1.ChannelListener listeners = 1; + // The notification with the highest counter the channel retains. + temporal.api.notification.v1.Notification latest = 2; + // How many notifications the channel retains for pollers. + int32 retained_count = 3; +} + message SignalWithStartWorkflowExecutionRequest { string namespace = 1; string workflow_id = 2; diff --git a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto index f100cdf28..ff6d5d937 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto @@ -488,6 +488,9 @@ service WorkflowService { // no event, carries no payload, and folds with other wakes for the same // source, so a burst of writes costs one task. The Workflow learns the // source and its position from the task and reads the source itself. + // + // Superseded by the notification channel (`NotifyChannel` and the + // `SubscribeNotificationChannel` command). Kept for one round and slated for removal. rpc WakeWorkflowExecution (WakeWorkflowExecutionRequest) returns (WakeWorkflowExecutionResponse) { option (google.api.http) = { post: "/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/wake" @@ -499,6 +502,72 @@ service WorkflowService { }; } + // NotifyChannel tells every listener of a channel that a source they consume + // has moved. The writer names no addressee and never learns who listens. The + // server wakes each listener: a Workflow with a Workflow Task, a callback by + // invoking it. Nothing goes to History except the notifications a woken + // Workflow Task carries on its scheduled event. + rpc NotifyChannel (NotifyChannelRequest) returns (NotifyChannelResponse) { + option (google.api.http) = { + post: "/namespaces/{namespace}/channels/{notification.channel}/notify" + body: "*" + additional_bindings { + post: "/api/v1/namespaces/{namespace}/channels/{notification.channel}/notify" + body: "*" + } + }; + } + + // RegisterChannelListener registers a callback as a listener of a channel. A + // Workflow registers itself with the `SubscribeNotificationChannel` command + // instead. + rpc RegisterChannelListener (RegisterChannelListenerRequest) returns (RegisterChannelListenerResponse) { + option (google.api.http) = { + post: "/namespaces/{namespace}/channels/{channel}/listeners" + body: "*" + additional_bindings { + post: "/api/v1/namespaces/{namespace}/channels/{channel}/listeners" + body: "*" + } + }; + } + + // UnregisterChannelListener removes a listener from a channel. + // + // (-- api-linter: core::0136::http-method=disabled + // aip.dev/not-precedent: Removing a listener is a delete of that listener. --) + rpc UnregisterChannelListener (UnregisterChannelListenerRequest) returns (UnregisterChannelListenerResponse) { + option (google.api.http) = { + delete: "/namespaces/{namespace}/channels/{channel}/listeners/{listener_id}" + additional_bindings { + delete: "/api/v1/namespaces/{namespace}/channels/{channel}/listeners/{listener_id}" + } + }; + } + + // PollChannel is a long poll for clients. It returns the retained + // notifications of a channel with a counter above `after_counter`, waiting + // up to `wait` for one when none is retained yet. + rpc PollChannel (PollChannelRequest) returns (PollChannelResponse) { + option (google.api.http) = { + get: "/namespaces/{namespace}/channels/{channel}/notifications" + additional_bindings { + get: "/api/v1/namespaces/{namespace}/channels/{channel}/notifications" + } + }; + } + + // DescribeChannel returns the listeners of a channel and its latest + // notification. + rpc DescribeChannel (DescribeChannelRequest) returns (DescribeChannelResponse) { + option (google.api.http) = { + get: "/namespaces/{namespace}/channels/{channel}" + additional_bindings { + get: "/api/v1/namespaces/{namespace}/channels/{channel}" + } + }; + } + // SignalWithStartWorkflowExecution is used to ensure a signal is sent to a workflow, even if // it isn't yet started. // diff --git a/crates/protos/src/protos/mod.rs b/crates/protos/src/protos/mod.rs index 51b0cdba4..8ba1167af 100644 --- a/crates/protos/src/protos/mod.rs +++ b/crates/protos/src/protos/mod.rs @@ -2000,6 +2000,9 @@ pub mod temporal { Attributes::SubscribeStreamCommandAttributes(_) => { CommandType::SubscribeStream } + Attributes::SubscribeNotificationChannelCommandAttributes(_) => { + CommandType::SubscribeNotificationChannel + } Attributes::CompleteWorkflowExecutionCommandAttributes(_) => { CommandType::CompleteWorkflowExecution } @@ -2514,6 +2517,7 @@ pub mod temporal { | EventType::WorkflowPropertiesModified | EventType::WorkflowStreamSubscribed | EventType::WorkflowStreamRecordsAppended + | EventType::WorkflowNotificationChannelSubscribed | EventType::NexusOperationScheduled | EventType::NexusOperationCancelRequested | EventType::WorkflowExecutionCanceled @@ -2617,6 +2621,9 @@ pub mod temporal { Attributes::WorkflowStreamRecordsAppendedEventAttributes(_) => { false } + Attributes::WorkflowNotificationChannelSubscribedEventAttributes(_) => { + false + } Attributes::WorkflowExecutionStartedEventAttributes(_) => false, Attributes::WorkflowExecutionCompletedEventAttributes(_) => false, Attributes::WorkflowExecutionFailedEventAttributes(_) => false, @@ -2706,6 +2713,7 @@ pub mod temporal { match self { Attributes::WorkflowStreamSubscribedEventAttributes(_) => { EventType::WorkflowStreamSubscribed } Attributes::WorkflowStreamRecordsAppendedEventAttributes(_) => { EventType::WorkflowStreamRecordsAppended } + Attributes::WorkflowNotificationChannelSubscribedEventAttributes(_) => { EventType::WorkflowNotificationChannelSubscribed } Attributes::WorkflowExecutionStartedEventAttributes(_) => { EventType::WorkflowExecutionStarted } Attributes::WorkflowExecutionCompletedEventAttributes(_) => { EventType::WorkflowExecutionCompleted } Attributes::WorkflowExecutionFailedEventAttributes(_) => { EventType::WorkflowExecutionFailed } @@ -2777,6 +2785,11 @@ pub mod temporal { tonic::include_proto!("temporal.api.namespace.v1"); } } + pub mod notification { + pub mod v1 { + tonic::include_proto!("temporal.api.notification.v1"); + } + } pub mod operatorservice { pub mod v1 { tonic::include_proto!("temporal.api.operatorservice.v1"); diff --git a/crates/sdk-core-c-bridge/src/client.rs b/crates/sdk-core-c-bridge/src/client.rs index 3aed770cd..75ae21a9f 100644 --- a/crates/sdk-core-c-bridge/src/client.rs +++ b/crates/sdk-core-c-bridge/src/client.rs @@ -669,6 +669,9 @@ async fn call_workflow_service( "DescribeBatchOperation" => { rpc_call_on_trait!(client, call, WorkflowService, describe_batch_operation) } + "DescribeChannel" => { + rpc_call_on_trait!(client, call, WorkflowService, describe_channel) + } "DescribeDeployment" => { rpc_call_on_trait!(client, call, WorkflowService, describe_deployment) } @@ -802,6 +805,9 @@ async fn call_workflow_service( "ListWorkflowRules" => { rpc_call_on_trait!(client, call, WorkflowService, list_workflow_rules) } + "NotifyChannel" => { + rpc_call_on_trait!(client, call, WorkflowService, notify_channel) + } "PatchSchedule" => rpc_call_on_trait!(client, call, WorkflowService, patch_schedule), "PauseActivity" => rpc_call_on_trait!(client, call, WorkflowService, pause_activity), "PauseActivityExecution" => { @@ -810,6 +816,9 @@ async fn call_workflow_service( "PauseWorkflowExecution" => { rpc_call_on_trait!(client, call, WorkflowService, pause_workflow_execution) } + "PollChannel" => { + rpc_call_on_trait!(client, call, WorkflowService, poll_channel) + } "PollActivityExecution" => { rpc_call_on_trait!(client, call, WorkflowService, poll_activity_execution) } @@ -862,6 +871,9 @@ async fn call_workflow_service( "RecordWorkerHeartbeat" => { rpc_call_on_trait!(client, call, WorkflowService, record_worker_heartbeat) } + "RegisterChannelListener" => { + rpc_call_on_trait!(client, call, WorkflowService, register_channel_listener) + } "RegisterNamespace" => { rpc_call_on_trait!(client, call, WorkflowService, register_namespace) } @@ -1030,6 +1042,9 @@ async fn call_workflow_service( "TriggerWorkflowRule" => { rpc_call_on_trait!(client, call, WorkflowService, trigger_workflow_rule) } + "UnregisterChannelListener" => { + rpc_call_on_trait!(client, call, WorkflowService, unregister_channel_listener) + } "UnpauseActivity" => { rpc_call_on_trait!(client, call, WorkflowService, unpause_activity) } From 6fc86f1b06769780fe8e86edd98a37c96d815907 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Thu, 1 Oct 2026 17:34:13 -0700 Subject: [PATCH 23/35] Carried the wrapped notification channel lines in the Core's api tree. --- .../api_upstream/temporal/api/command/v1/message.proto | 3 ++- .../api_upstream/temporal/api/history/v1/message.proto | 3 ++- .../temporal/api/workflowservice/v1/service.proto | 6 ++++-- 3 files changed, 8 insertions(+), 4 deletions(-) diff --git a/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto index a08704c3e..f75b5a950 100644 --- a/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto @@ -327,7 +327,8 @@ message Command { RequestCancelNexusOperationCommandAttributes request_cancel_nexus_operation_command_attributes = 19; AppendStreamRecordsCommandAttributes append_stream_records_command_attributes = 20; SubscribeStreamCommandAttributes subscribe_stream_command_attributes = 21; - SubscribeNotificationChannelCommandAttributes subscribe_notification_channel_command_attributes = 22; + SubscribeNotificationChannelCommandAttributes + subscribe_notification_channel_command_attributes = 22; } } diff --git a/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto index e466ba505..c76f80756 100644 --- a/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto @@ -1326,7 +1326,8 @@ message HistoryEvent { WorkflowExecutionTimeSkippingTransitionedEventAttributes workflow_execution_time_skipping_transitioned_event_attributes = 65; WorkflowStreamSubscribedEventAttributes workflow_stream_subscribed_event_attributes = 66; WorkflowStreamRecordsAppendedEventAttributes workflow_stream_records_appended_event_attributes = 67; - WorkflowNotificationChannelSubscribedEventAttributes workflow_notification_channel_subscribed_event_attributes = 68; + WorkflowNotificationChannelSubscribedEventAttributes + workflow_notification_channel_subscribed_event_attributes = 68; } } diff --git a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto index ff6d5d937..9aff1154a 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto @@ -521,7 +521,8 @@ service WorkflowService { // RegisterChannelListener registers a callback as a listener of a channel. A // Workflow registers itself with the `SubscribeNotificationChannel` command // instead. - rpc RegisterChannelListener (RegisterChannelListenerRequest) returns (RegisterChannelListenerResponse) { + rpc RegisterChannelListener (RegisterChannelListenerRequest) + returns (RegisterChannelListenerResponse) { option (google.api.http) = { post: "/namespaces/{namespace}/channels/{channel}/listeners" body: "*" @@ -536,7 +537,8 @@ service WorkflowService { // // (-- api-linter: core::0136::http-method=disabled // aip.dev/not-precedent: Removing a listener is a delete of that listener. --) - rpc UnregisterChannelListener (UnregisterChannelListenerRequest) returns (UnregisterChannelListenerResponse) { + rpc UnregisterChannelListener (UnregisterChannelListenerRequest) + returns (UnregisterChannelListenerResponse) { option (google.api.http) = { delete: "/namespaces/{namespace}/channels/{channel}/listeners/{listener_id}" additional_bindings { From 2fadc9b3d0a51dd7a8842d9c34df1f46ac696f20 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Thu, 1 Oct 2026 17:57:52 -0700 Subject: [PATCH 24/35] Subscribed workflows to notification channels by command. The command gets its own machine, matched against the subscribed event on replay. The notifications on a scheduled event reach lang as one job and resume parked waits, and such a task is no longer folded into the one before it on replay. --- .../workflow_activation.proto | 14 +- .../workflow_commands/workflow_commands.proto | 9 + crates/protos/src/protos/mod.rs | 19 + .../src/core_tests/external_streams.rs | 291 ++++++++++++ crates/sdk-core/src/replay/history_builder.rs | 22 + .../src/worker/workflow/history_update.rs | 13 +- .../src/worker/workflow/machines/mod.rs | 3 + ...ribe_notification_channel_state_machine.rs | 447 ++++++++++++++++++ .../workflow/machines/transition_coverage.rs | 5 + .../workflow/machines/workflow_machines.rs | 69 ++- .../src/worker/workflow/managed_run.rs | 10 +- crates/sdk-core/src/worker/workflow/mod.rs | 7 +- crates/sdk/src/workflow_future.rs | 5 + crates/workflow/src/runtime/instance.rs | 3 + 14 files changed, 910 insertions(+), 7 deletions(-) create mode 100644 crates/sdk-core/src/worker/workflow/machines/subscribe_notification_channel_state_machine.rs diff --git a/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto b/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto index c3d632a4d..b583b0ff9 100644 --- a/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto +++ b/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto @@ -14,6 +14,7 @@ import "temporal/api/failure/v1/message.proto"; import "temporal/api/update/v1/message.proto"; import "temporal/api/common/v1/message.proto"; import "temporal/api/enums/v1/workflow.proto"; +import "temporal/api/notification/v1/message.proto"; import "temporal/sdk/core/activity_result/activity_result.proto"; import "temporal/sdk/core/child_workflow/child_workflow.proto"; import "temporal/sdk/core/common/common.proto"; @@ -30,7 +31,7 @@ import "temporal/sdk/core/workflow_commands/workflow_commands.proto"; // 1. init workflow // 2. patches // 3. random-seed-updates -// 4. signals/updates +// 4. signals/updates/channel notifications // 5. all others // 6. local activity resolutions // 7. queries @@ -154,6 +155,9 @@ message WorkflowActivationJob { // Runtime-internal: encode the terminal boundary for a marker Core is about to write. // Runs no user code. FinalizeExternalStreams finalize_external_streams = 20; + // Notifications for channels this run listens to, read from the Workflow Task's + // scheduled event. + NotificationsReceived notifications_received = 22; // Remove the workflow identified by the [WorkflowActivation] containing this job from the // cache after performing the activation. It is guaranteed that this will be the only job // in the activation if present. @@ -161,6 +165,14 @@ message WorkflowActivationJob { } } +// The notifications the server folded onto this Workflow Task's scheduled event, one per channel. +// +// Read from History, so a replay hands lang the same job in the same activation. A notification +// carries no data: lang reads the source the channel names. +message NotificationsReceived { + repeated temporal.api.notification.v1.Notification notifications = 1; +} + // Tells lang that one or more external stream waits may now have data. // // This job contains no records. The listed waits are readiness *hints*, not an exhaustive diff --git a/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto b/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto index 1dba23214..9abc82fda 100644 --- a/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto +++ b/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto @@ -55,9 +55,18 @@ message WorkflowCommand { ExternalStreamFinalized external_stream_finalized = 26; WorkflowOutputStreamCommit workflow_output_stream_commit = 27; WorkflowOutputStreamBuffered workflow_output_stream_buffered = 28; + SubscribeNotificationChannel subscribe_notification_channel = 31; } } +// Makes this run a listener of a notification channel. Nothing resolves back to lang: the +// notifications arrive later on the scheduled event of a Workflow Task, as a +// `NotificationsReceived` job. +message SubscribeNotificationChannel { + // The channel to listen on, as the writers name it. + string channel = 1; +} + // Commits an observation delta for external streams. // // Emitted on *every* completion path where replay-visible stream state changed -- which includes diff --git a/crates/protos/src/protos/mod.rs b/crates/protos/src/protos/mod.rs index 8ba1167af..cb5392bf6 100644 --- a/crates/protos/src/protos/mod.rs +++ b/crates/protos/src/protos/mod.rs @@ -1368,6 +1368,9 @@ pub mod coresdk { fin.reason() ) } + workflow_activation_job::Variant::NotificationsReceived(n) => { + write!(f, "NotificationsReceived({})", n.notifications.len()) + } } } } @@ -1802,6 +1805,12 @@ pub mod coresdk { } } + impl Display for SubscribeNotificationChannel { + fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { + write!(f, "SubscribeNotificationChannel({})", self.channel) + } + } + impl Display for WorkflowOutputStreamBuffered { fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { write!( @@ -2056,6 +2065,16 @@ pub mod temporal { } } + impl From for Attributes { + fn from(s: workflow_commands::SubscribeNotificationChannel) -> Self { + Self::SubscribeNotificationChannelCommandAttributes( + SubscribeNotificationChannelCommandAttributes { + channel: s.channel, + }, + ) + } + } + impl From for command::Attributes { fn from(s: workflow_commands::StartTimer) -> Self { Self::StartTimerCommandAttributes(StartTimerCommandAttributes { diff --git a/crates/sdk-core/src/core_tests/external_streams.rs b/crates/sdk-core/src/core_tests/external_streams.rs index 47bad33fd..bd09ae93e 100644 --- a/crates/sdk-core/src/core_tests/external_streams.rs +++ b/crates/sdk-core/src/core_tests/external_streams.rs @@ -52,6 +52,7 @@ use temporalio_common::{ common::v1::Payload, enums::v1::{CommandType, EventType}, history::v1::History, + notification::v1::Notification, query::v1::WorkflowQuery, workflow::v1::Wake, workflowservice::v1::{ @@ -3577,6 +3578,296 @@ async fn a_poll_wake_is_not_applied_while_replaying() { worker.drain_pollers_and_shutdown().await; } +// --- notifications on the scheduled event ------------------------------------ + +fn channel_notification(channel: &str, counter: i64) -> Notification { + Notification { + channel: channel.to_string(), + position: format!("{counter}-0").into_bytes(), + counter, + metadata: HashMap::new(), + } +} + +/// The notifications of every `NotificationsReceived` job in the activation, one entry per job. +fn notification_jobs(activation: &WorkflowActivation) -> Vec> { + activation + .jobs + .iter() + .filter_map(|j| match &j.variant { + Some(workflow_activation_job::Variant::NotificationsReceived(n)) => { + Some(n.notifications.clone()) + } + _ => None, + }) + .collect() +} + +fn job_index( + activation: &WorkflowActivation, + is: fn(&workflow_activation_job::Variant) -> bool, +) -> usize { + activation + .jobs + .iter() + .position(|j| j.variant.as_ref().is_some_and(is)) + .unwrap_or_else(|| panic!("job not found in {:?}", activation.jobs)) +} + +/// A history whose second Workflow Task is scheduled with `notifications` and nothing else. +fn notified_second_task_history(notifications: Vec) -> TestHistoryBuilder { + let mut t = TestHistoryBuilder::default(); + t.add_wfe_started_with_wft_timeout(Duration::from_secs(300)); + t.add_full_wf_task(); + t.add_workflow_task_scheduled_with_notifications(notifications); + t.add_workflow_task_started(); + t.add_workflow_task_completed(); + t.add_workflow_execution_completed(); + t +} + +#[tokio::test] +async fn a_notified_scheduled_event_yields_the_job_and_resumes_a_parked_wait() { + // The server folds a channel's notifications onto the scheduled event of the task it wakes + // the run with. The job hands them to lang, and the same task resumes the parked waits the + // way a poll wake does: no chain to check and no generation to match. + let notifications = vec![channel_notification("tokens", 42)]; + let mut mock = build_mock_pollers(MockPollCfg::from_resp_batches( + "fakeid", + notified_second_task_history(notifications.clone()), + [1, 2], + mock_worker_client(), + )); + mock.worker_cfg(|w| { + w.task_types = WorkerTaskTypes::workflow_only(); + w.max_cached_workflows = 1; + }); + let worker = mock_worker(mock); + let first = worker.poll_workflow_activation().await.unwrap(); + assert_eq!(notification_jobs(&first), Vec::>::new()); + let run_id = first.run_id.clone(); + worker + .complete_workflow_activation(WorkflowActivationCompletion::empty(run_id.clone())) + .await + .unwrap(); + worker + .seed_external_stream_waits(&run_id, vec![1, 2], Some(7), false) + .await; + + let second = worker.poll_workflow_activation().await.unwrap(); + assert!(!second.is_replaying, "got jobs {:?}", second.jobs); + assert_eq!(notification_jobs(&second), vec![notifications]); + assert_eq!( + resolve_hints(&second), + vec![1, 2], + "the notifications must resume every parked wait, got {:?}", + second.jobs + ); + let notified = job_index(&second, |v| { + matches!( + v, + workflow_activation_job::Variant::NotificationsReceived(_) + ) + }); + let resolved = job_index(&second, |v| { + matches!( + v, + workflow_activation_job::Variant::ResolveExternalStreamWaits(_) + ) + }); + assert!( + notified < resolved, + "the notifications must come before the resolve job, got {:?}", + second.jobs + ); + + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + second.run_id, + vec![CompleteWorkflowExecution::default().into()], + )) + .await + .unwrap(); + worker.drain_pollers_and_shutdown().await; +} + +#[tokio::test] +async fn a_task_scheduled_twice_hands_lang_one_folded_job() { + // A task that failed is scheduled again, and the second scheduled event may carry the same + // channel at a later counter. Lang gets one job per activation, folded the way the server + // folds: one notification per channel, the highest counter kept. + let mut t = TestHistoryBuilder::default(); + t.add_wfe_started_with_wft_timeout(Duration::from_secs(300)); + t.add_full_wf_task(); + t.add_workflow_task_scheduled_with_notifications(vec![ + channel_notification("tokens", 41), + channel_notification("events", 3), + ]); + t.add_workflow_task_started(); + t.add_workflow_task_timed_out(); + t.add_workflow_task_scheduled_with_notifications(vec![channel_notification("tokens", 42)]); + t.add_workflow_task_started(); + t.add_workflow_task_completed(); + t.add_workflow_execution_completed(); + + let mut mock = build_mock_pollers(MockPollCfg::from_resp_batches( + "fakeid", + t, + [1, 2], + mock_worker_client(), + )); + mock.worker_cfg(|w| { + w.task_types = WorkerTaskTypes::workflow_only(); + w.max_cached_workflows = 1; + }); + let worker = mock_worker(mock); + let first = worker.poll_workflow_activation().await.unwrap(); + worker + .complete_workflow_activation(WorkflowActivationCompletion::empty(first.run_id)) + .await + .unwrap(); + + let second = worker.poll_workflow_activation().await.unwrap(); + assert_eq!( + notification_jobs(&second), + vec![vec![ + channel_notification("tokens", 42), + channel_notification("events", 3), + ]] + ); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + second.run_id, + vec![CompleteWorkflowExecution::default().into()], + )) + .await + .unwrap(); + worker.drain_pollers_and_shutdown().await; +} + +#[tokio::test] +async fn replaying_a_notified_scheduled_event_yields_the_same_job() { + // History is the record, so a cold worker hands lang the same job in the same activation the + // live run saw it in. The first task issued no command, which is the shape Core would + // otherwise treat as a heartbeat and fold into the next task, moving the job one activation + // early. The Signal keeps the replayed task apart from the live one. + let notifications = vec![channel_notification("tokens", 42)]; + let mut t = TestHistoryBuilder::default(); + t.add_wfe_started_with_wft_timeout(Duration::from_secs(300)); + t.add_full_wf_task(); + t.add_workflow_task_scheduled_with_notifications(notifications.clone()); + t.add_workflow_task_started(); + t.add_workflow_task_completed(); + t.add_we_signaled("a-user-signal", vec![]); + t.add_workflow_task_scheduled_and_started(); + + let cold = hist_to_poll_resp(&t, "fakeid".to_owned(), ResponseType::AllHistory); + let mut mock = build_mock_pollers(MockPollCfg::from_resp_batches( + "fakeid", + t, + [ResponseType::Raw(cold.resp)], + mock_worker_client(), + )); + mock.worker_cfg(|w| { + w.task_types = WorkerTaskTypes::workflow_only(); + w.max_cached_workflows = 1; + }); + let worker = mock_worker(mock); + + let first = worker.poll_workflow_activation().await.unwrap(); + assert!(first.is_replaying); + assert_eq!( + notification_jobs(&first), + Vec::>::new(), + "the notifications belong to the second task, got {:?}", + first.jobs + ); + worker + .complete_workflow_activation(WorkflowActivationCompletion::empty(first.run_id)) + .await + .unwrap(); + + let replayed = worker.poll_workflow_activation().await.unwrap(); + assert!(replayed.is_replaying, "got jobs {:?}", replayed.jobs); + assert_eq!(notification_jobs(&replayed), vec![notifications]); + worker + .complete_workflow_activation(WorkflowActivationCompletion::empty(replayed.run_id)) + .await + .unwrap(); + + let live = worker.poll_workflow_activation().await.unwrap(); + assert!(!live.is_replaying, "got jobs {:?}", live.jobs); + assert_eq!(notification_jobs(&live), Vec::>::new()); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + live.run_id, + vec![CompleteWorkflowExecution::default().into()], + )) + .await + .unwrap(); + worker.drain_pollers_and_shutdown().await; +} + +#[tokio::test] +async fn replayed_notifications_resume_the_reconstructed_waits() { + // The wake comes from History, so unlike a poll wake it is applied while replaying: the waits + // lang reconstructs for the notified task are resumed by the same activation. + let notifications = vec![channel_notification("tokens", 42)]; + let mut t = TestHistoryBuilder::default(); + t.add_wfe_started_with_wft_timeout(Duration::from_secs(300)); + t.add_full_wf_task(); + t.add_we_signaled("a-user-signal", vec![]); + t.add_full_wf_task(); + t.add_workflow_task_scheduled_with_notifications(notifications.clone()); + t.add_workflow_task_started(); + + let cold = hist_to_poll_resp(&t, "fakeid".to_owned(), ResponseType::AllHistory); + let mut mock = build_mock_pollers(MockPollCfg::from_resp_batches( + "fakeid", + t, + [ResponseType::Raw(cold.resp)], + mock_worker_client(), + )); + mock.worker_cfg(|w| { + w.task_types = WorkerTaskTypes::workflow_only(); + w.max_cached_workflows = 1; + }); + let worker = mock_worker(mock); + + let first = worker.poll_workflow_activation().await.unwrap(); + worker + .complete_workflow_activation(WorkflowActivationCompletion::empty(first.run_id)) + .await + .unwrap(); + let second = worker.poll_workflow_activation().await.unwrap(); + assert!(second.is_replaying, "got jobs {:?}", second.jobs); + let run_id = second.run_id.clone(); + worker + .seed_external_stream_waits(&run_id, vec![1], None, false) + .await; + worker + .complete_workflow_activation(WorkflowActivationCompletion::empty(run_id.clone())) + .await + .unwrap(); + + let notified = worker.poll_workflow_activation().await.unwrap(); + assert_eq!(notification_jobs(¬ified), vec![notifications]); + assert_eq!( + resolve_hints(¬ified), + vec![1], + "the notifications must resume the reconstructed wait, got {:?}", + notified.jobs + ); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + notified.run_id, + vec![CompleteWorkflowExecution::default().into()], + )) + .await + .unwrap(); + worker.drain_pollers_and_shutdown().await; +} + // --- marker emission (C9, C14b) ---------------------------------------------- /// A history for a Workflow Task that records a marker and then starts a timer. diff --git a/crates/sdk-core/src/replay/history_builder.rs b/crates/sdk-core/src/replay/history_builder.rs index 3cf593483..b13deae73 100644 --- a/crates/sdk-core/src/replay/history_builder.rs +++ b/crates/sdk-core/src/replay/history_builder.rs @@ -114,6 +114,28 @@ impl TestHistoryBuilder { self.workflow_task_scheduled_event_id = self.add_by_type(EventType::WorkflowTaskScheduled); } + /// Add a workflow task scheduled event carrying channel notifications. + pub fn add_workflow_task_scheduled_with_notifications( + &mut self, + notifications: Vec< + temporalio_common::protos::temporal::api::notification::v1::Notification, + >, + ) { + self.workflow_task_scheduled_event_id = self.add(WorkflowTaskScheduledEventAttributes { + notifications, + ..Default::default() + }); + } + + /// Add the event a subscribe-notification-channel command produces. + pub fn add_notification_channel_subscribed(&mut self, channel: &str) -> i64 { + let attrs = WorkflowNotificationChannelSubscribedEventAttributes { + workflow_task_completed_event_id: self.previous_task_completed_id, + channel: channel.to_string(), + }; + self.add(attrs) + } + /// Add a workflow task started event. pub fn add_workflow_task_started(&mut self) { self.final_workflow_task_started_event_id = self.add(WorkflowTaskStartedEventAttributes { diff --git a/crates/sdk-core/src/worker/workflow/history_update.rs b/crates/sdk-core/src/worker/workflow/history_update.rs index 7c105bd5e..207cc46f0 100644 --- a/crates/sdk-core/src/worker/workflow/history_update.rs +++ b/crates/sdk-core/src/worker/workflow/history_update.rs @@ -742,10 +742,14 @@ fn find_end_index_of_next_wft_seq( if let Some(next_next_event) = events.get(ix + 2) { if !saw_command && next_next_event.event_type() == EventType::WorkflowTaskScheduled + && !carries_notifications(next_next_event) { // If we've never seen an interesting event and the next two events are // a completion followed immediately again by scheduled, then this is a - // WFT heartbeat and also doesn't conclude the sequence. + // WFT heartbeat and also doesn't conclude the sequence. A scheduled + // event with notifications is work lang saw in its own activation + // live, so folding it into this one on replay would hand lang the + // notifications one activation early. continue; } else { // If we see an update accepted command after WFT completed, we want to @@ -802,6 +806,13 @@ fn find_end_index_of_next_wft_seq( NextWFTSeqEndIndex::Incomplete(last_index) } +fn carries_notifications(e: &HistoryEvent) -> bool { + matches!( + &e.attributes, + Some(Attributes::WorkflowTaskScheduledEventAttributes(a)) if !a.notifications.is_empty() + ) +} + #[cfg(test)] mod tests { use super::*; diff --git a/crates/sdk-core/src/worker/workflow/machines/mod.rs b/crates/sdk-core/src/worker/workflow/machines/mod.rs index 53d44aac0..d285c7f5b 100644 --- a/crates/sdk-core/src/worker/workflow/machines/mod.rs +++ b/crates/sdk-core/src/worker/workflow/machines/mod.rs @@ -16,6 +16,7 @@ mod modify_workflow_properties_state_machine; mod nexus_operation_state_machine; mod patch_state_machine; mod signal_external_state_machine; +mod subscribe_notification_channel_state_machine; mod timer_state_machine; mod update_state_machine; mod upsert_search_attributes_state_machine; @@ -48,6 +49,7 @@ use std::{ convert::{TryFrom, TryInto}, fmt::{Debug, Display}, }; +use subscribe_notification_channel_state_machine::SubscribeNotificationChannelMachine; use temporalio_common::{ fsm_trait::{StateMachine, TransitionResult}, protos::temporal::api::{ @@ -85,6 +87,7 @@ enum Machines { ModifyWorkflowPropertiesMachine, UpdateMachine, NexusOperationMachine, + SubscribeNotificationChannelMachine, } /// Extends [rustfsm::StateMachine] with some functionality specific to the temporal SDK. diff --git a/crates/sdk-core/src/worker/workflow/machines/subscribe_notification_channel_state_machine.rs b/crates/sdk-core/src/worker/workflow/machines/subscribe_notification_channel_state_machine.rs new file mode 100644 index 000000000..3f24ccf32 --- /dev/null +++ b/crates/sdk-core/src/worker/workflow/machines/subscribe_notification_channel_state_machine.rs @@ -0,0 +1,447 @@ +use super::{ + NewMachineWithCommand, StateMachine, TransitionResult, fsm, workflow_machines::MachineResponse, +}; +use crate::worker::workflow::{ + WFMachinesError, fatal, + machines::{EventInfo, HistEventData, WFMachinesAdapter}, + nondeterminism, +}; +use temporalio_common::protos::{ + coresdk::workflow_commands::SubscribeNotificationChannel, + temporal::api::{ + enums::v1::{CommandType, EventType}, + history::v1::{WorkflowNotificationChannelSubscribedEventAttributes, history_event}, + }, +}; + +fsm! { + pub(super) name SubscribeNotificationChannelMachine; + command SubscribeNotificationChannelMachineCommand; + error WFMachinesError; + shared_state SharedState; + + Created --(CommandScheduled) --> CommandIssued; + CommandIssued --(CommandRecorded(WorkflowNotificationChannelSubscribedEventAttributes), + shared on_command_recorded) --> Done; +} + +/// The channel the command named, kept so the recorded event can be held against it on replay. +#[derive(Default, Clone)] +pub(super) struct SharedState { + channel: String, +} + +/// Make this run a listener of a notification channel. Nothing comes back to the workflow from +/// the command itself: the notifications ride the scheduled event of a later Workflow Task. +pub(super) fn subscribe_notification_channel( + lang_cmd: SubscribeNotificationChannel, +) -> NewMachineWithCommand { + let sm = SubscribeNotificationChannelMachine::from_parts( + Created {}.into(), + SharedState { + channel: lang_cmd.channel.clone(), + }, + ); + NewMachineWithCommand { + command: lang_cmd.into(), + machine: sm.into(), + } +} + +#[derive(Debug, derive_more::Display)] +pub(super) enum SubscribeNotificationChannelMachineCommand {} + +#[derive(Debug, Default, Clone, derive_more::Display)] +pub(super) struct Created {} + +#[derive(Debug, Default, Clone, derive_more::Display)] +pub(super) struct CommandIssued {} + +impl CommandIssued { + pub(super) fn on_command_recorded( + self, + dat: &mut SharedState, + attrs: WorkflowNotificationChannelSubscribedEventAttributes, + ) -> SubscribeNotificationChannelMachineTransition { + if dat.channel == attrs.channel { + TransitionResult::default() + } else { + TransitionResult::Err(nondeterminism!( + "Recorded subscription to notification channel {:?} does not match the reissued \ + subscription to channel {:?}", + attrs.channel, + dat.channel + )) + } + } +} + +#[derive(Debug, Default, Clone, derive_more::Display)] +pub(super) struct Done {} + +impl WFMachinesAdapter for SubscribeNotificationChannelMachine { + fn adapt_response( + &self, + _my_command: Self::Command, + _event_info: Option, + ) -> Result, Self::Error> { + Err(fatal!( + "SubscribeNotificationChannel does not use state machine commands" + )) + } +} + +impl TryFrom for SubscribeNotificationChannelMachineEvents { + type Error = WFMachinesError; + + fn try_from(e: HistEventData) -> Result { + let e = e.event; + match e.event_type() { + EventType::WorkflowNotificationChannelSubscribed => { + if let Some( + history_event::Attributes::WorkflowNotificationChannelSubscribedEventAttributes( + attrs, + ), + ) = e.attributes + { + Ok(SubscribeNotificationChannelMachineEvents::CommandRecorded( + attrs, + )) + } else { + Err(fatal!( + "Notification channel subscribed attributes were unset: {e}" + )) + } + } + _ => Err(Self::Error::Nondeterminism(format!( + "SubscribeNotificationChannelMachine does not handle {e}" + ))), + } + } +} + +impl TryFrom for SubscribeNotificationChannelMachineEvents { + type Error = WFMachinesError; + + fn try_from(c: CommandType) -> Result { + match c { + CommandType::SubscribeNotificationChannel => { + Ok(SubscribeNotificationChannelMachineEvents::CommandScheduled) + } + _ => Err(Self::Error::Nondeterminism(format!( + "SubscribeNotificationChannelMachine does not handle command type {c:?}" + ))), + } + } +} + +impl From for CommandIssued { + fn from(_: Created) -> Self { + Self {} + } +} + +#[cfg(test)] +mod tests { + use super::{super::OnEventWrapper, *}; + use crate::{ + replay::TestHistoryBuilder, + test_help::{ + MockPollCfg, ResponseType, WorkerExt, build_mock_pollers, hist_to_poll_resp, + mock_worker, start_timer_cmd, + }, + worker::client::mocks::mock_worker_client, + }; + use parking_lot::Mutex; + use std::{sync::Arc, time::Duration}; + use temporalio_common::{ + protos::{ + coresdk::{ + workflow_commands::{CompleteWorkflowExecution, workflow_command}, + workflow_completion::WorkflowActivationCompletion, + }, + temporal::api::{ + command::v1::command, enums::v1::WorkflowTaskFailedCause, history::v1::HistoryEvent, + }, + }, + worker::WorkerTaskTypes, + }; + + fn subscribe(channel: &str) -> workflow_command::Variant { + SubscribeNotificationChannel { + channel: channel.to_string(), + } + .into() + } + + fn recorded(channel: &str) -> SubscribeNotificationChannelMachineEvents { + HistEventData { + event: HistoryEvent { + event_type: EventType::WorkflowNotificationChannelSubscribed as i32, + attributes: Some( + history_event::Attributes::WorkflowNotificationChannelSubscribedEventAttributes( + WorkflowNotificationChannelSubscribedEventAttributes { + workflow_task_completed_event_id: 0, + channel: channel.to_string(), + }, + ), + ), + ..Default::default() + }, + replaying: true, + current_task_is_last_in_history: false, + } + .try_into() + .unwrap() + } + + fn issued(channel: &str) -> SubscribeNotificationChannelMachine { + let mut sm = SubscribeNotificationChannelMachine::from_parts( + Created {}.into(), + SharedState { + channel: channel.to_string(), + }, + ); + OnEventWrapper::on_event_mut( + &mut sm, + CommandType::SubscribeNotificationChannel + .try_into() + .unwrap(), + ) + .expect("CommandScheduled should transition Created -> CommandIssued"); + assert_eq!(CommandIssued {}.to_string(), sm.state().to_string()); + sm + } + + #[test] + fn the_recorded_event_for_the_same_channel_completes_the_machine() { + let mut sm = issued("orders"); + OnEventWrapper::on_event_mut(&mut sm, recorded("orders")) + .expect("CommandRecorded should transition CommandIssued -> Done"); + assert_eq!(Done {}.to_string(), sm.state().to_string()); + } + + #[test] + fn a_recorded_event_for_another_channel_is_nondeterminism() { + let mut sm = issued("orders"); + let err = OnEventWrapper::on_event_mut(&mut sm, recorded("invoices")) + .expect_err("a different channel must not match"); + let message = format!("{err:?}"); + assert!( + message.contains("does not match") && message.contains("invoices"), + "the error must name both channels, got {message}" + ); + } + + #[test] + fn another_event_type_is_nondeterminism() { + let event = HistEventData { + event: HistoryEvent { + event_type: EventType::TimerStarted as i32, + ..Default::default() + }, + replaying: true, + current_task_is_last_in_history: false, + }; + let res: Result = event.try_into(); + assert!(matches!(res, Err(WFMachinesError::Nondeterminism(_)))); + } + + #[tokio::test] + async fn the_command_reaches_the_completion_with_its_channel() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + + let sent = Arc::new(Mutex::new(vec![])); + let recorder = sent.clone(); + let mut cfg = MockPollCfg::from_resp_batches("fakeid", t, [1], mock_worker_client()); + cfg.completion_mock_fn = Some(Box::new(move |wftc| { + recorder.lock().extend(wftc.commands.iter().cloned()); + Ok(Default::default()) + })); + let mut mock = build_mock_pollers(cfg); + mock.worker_cfg(|w| { + w.task_types = WorkerTaskTypes::workflow_only(); + w.max_cached_workflows = 1; + }); + let worker = mock_worker(mock); + + let first = worker.poll_workflow_activation().await.unwrap(); + assert!(!first.is_replaying); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmd( + first.run_id, + subscribe("orders"), + )) + .await + .unwrap(); + + let commands = sent.lock().clone(); + assert_eq!(commands.len(), 1, "got {commands:?}"); + assert_eq!( + commands[0].command_type(), + CommandType::SubscribeNotificationChannel + ); + assert!( + matches!( + &commands[0].attributes, + Some(command::Attributes::SubscribeNotificationChannelCommandAttributes(a)) + if a.channel == "orders" + ), + "got {:?}", + commands[0] + ); + + worker.drain_pollers_and_shutdown().await; + } + + /// Replays a first task whose recorded command event `record` writes, with lang reissuing + /// `reissued`. Returns the nondeterminism message Core reported, if any, and whether lang was + /// activated for the live task after it. + async fn replay_first_task( + record: impl FnOnce(&mut TestHistoryBuilder), + reissued: workflow_command::Variant, + expect_failure: bool, + ) -> (Option, bool) { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_full_wf_task(); + record(&mut t); + // Gives the live task a job, so a match shows up as an activation rather than as silence. + t.add_we_signaled("go", vec![]); + t.add_workflow_task_scheduled_and_started(); + + let cold = hist_to_poll_resp(&t, "fakeid".to_owned(), ResponseType::AllHistory); + let mut cfg = MockPollCfg::from_resp_batches( + "fakeid", + t, + [ResponseType::Raw(cold.resp)], + mock_worker_client(), + ); + let failure = Arc::new(Mutex::new(None)); + if expect_failure { + cfg.num_expected_fails = 1; + let recorder = failure.clone(); + cfg.expect_fail_wft_matcher = Box::new(move |_, cause, f| { + if matches!(cause, WorkflowTaskFailedCause::NonDeterministicError) { + *recorder.lock() = + Some(f.as_ref().map(|f| f.message.clone()).unwrap_or_default()); + } + true + }); + } + let mut mock = build_mock_pollers(cfg); + mock.worker_cfg(|w| { + w.task_types = WorkerTaskTypes::workflow_only(); + w.max_cached_workflows = 1; + }); + let worker = mock_worker(mock); + + let first = worker.poll_workflow_activation().await.unwrap(); + assert!(first.is_replaying); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmd( + first.run_id.clone(), + reissued, + )) + .await + .unwrap(); + + let next = tokio::time::timeout( + Duration::from_millis(500), + worker.poll_workflow_activation(), + ) + .await; + let mut activated = false; + if let Ok(Ok(act)) = next { + if act.is_only_eviction() { + worker + .complete_workflow_activation(WorkflowActivationCompletion::empty(act.run_id)) + .await + .unwrap(); + } else { + assert!( + !expect_failure, + "a mismatch must not activate lang, got {act:?}" + ); + activated = true; + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmd( + act.run_id, + CompleteWorkflowExecution::default().into(), + )) + .await + .unwrap(); + } + } + worker.drain_pollers_and_shutdown().await; + let failure = failure.lock().clone(); + (failure, activated) + } + + #[tokio::test] + async fn replay_matches_the_recorded_subscription() { + let (failure, activated) = replay_first_task( + |t| { + t.add_notification_channel_subscribed("orders"); + }, + subscribe("orders"), + false, + ) + .await; + assert_eq!(failure, None); + assert!( + activated, + "the live task after a matched subscription must activate lang" + ); + } + + #[tokio::test] + async fn replay_with_another_channel_is_nondeterminism() { + let failure = replay_first_task( + |t| { + t.add_notification_channel_subscribed("invoices"); + }, + subscribe("orders"), + true, + ) + .await + .0 + .expect("the task must fail as nondeterminism"); + assert!(failure.contains("does not match"), "got {failure}"); + } + + #[tokio::test] + async fn replay_without_the_subscribed_event_is_nondeterminism() { + let failure = replay_first_task( + |t| { + t.add_timer_started("1".to_string()); + }, + subscribe("orders"), + true, + ) + .await + .0 + .expect("the task must fail as nondeterminism"); + assert!( + failure.contains("SubscribeNotificationChannelMachine does not handle"), + "got {failure}" + ); + } + + #[tokio::test] + async fn a_recorded_subscription_lang_does_not_reissue_is_nondeterminism() { + let failure = replay_first_task( + |t| { + t.add_notification_channel_subscribed("orders"); + }, + start_timer_cmd(1, Duration::from_secs(1)), + true, + ) + .await + .0 + .expect("the task must fail as nondeterminism"); + assert!(!failure.is_empty()); + } +} diff --git a/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs b/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs index 0ddd0daf8..b55ccd159 100644 --- a/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs +++ b/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs @@ -76,6 +76,7 @@ mod machine_coverage_report { local_activity_state_machine::LocalActivityMachine, modify_workflow_properties_state_machine::ModifyWorkflowPropertiesMachine, patch_state_machine::PatchMachine, signal_external_state_machine::SignalExternalMachine, + subscribe_notification_channel_state_machine::SubscribeNotificationChannelMachine, timer_state_machine::TimerMachine, update_state_machine::UpdateMachine, upsert_search_attributes_state_machine::UpsertSearchAttributesMachine, workflow_task_state_machine::WorkflowTaskMachine, @@ -117,6 +118,7 @@ mod machine_coverage_report { let mut modify_wf_props = ModifyWorkflowPropertiesMachine::visualizer().to_owned(); let mut update = UpdateMachine::visualizer().to_owned(); let mut external_stream = ExternalStreamMachine::visualizer().to_owned(); + let mut subscribe_channel = SubscribeNotificationChannelMachine::visualizer().to_owned(); // This isn't at all efficient but doesn't need to be. // Replace transitions in the vizzes with green color if they are covered. @@ -144,6 +146,9 @@ mod machine_coverage_report { } m @ "UpdateMachine" => cover_transitions(m, &mut update, coverage), m @ "ExternalStreamMachine" => cover_transitions(m, &mut external_stream, coverage), + m @ "SubscribeNotificationChannelMachine" => { + cover_transitions(m, &mut subscribe_channel, coverage) + } m => panic!("Unknown machine {m}"), } } diff --git a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs index 787076f39..0f9931369 100644 --- a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs +++ b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs @@ -8,6 +8,7 @@ use super::{ continue_as_new_workflow_state_machine::continue_as_new, fail_workflow_state_machine::fail_workflow, local_activity_state_machine::new_local_activity, patch_state_machine::has_change, signal_external_state_machine::new_external_signal, + subscribe_notification_channel_state_machine::subscribe_notification_channel, timer_state_machine::new_timer, upsert_search_attributes_state_machine::upsert_search_attrs, workflow_machines::local_acts::LocalActivityData, workflow_task_state_machine::WorkflowTaskMachine, @@ -63,8 +64,8 @@ use temporalio_common::{ external_data::{ExternalStreamMarkerData, ParkReason}, external_stream, workflow_activation::{ - self, NotifyHasPatch, ReplayExternalStreams, UpdateRandomSeed, WorkflowActivation, - workflow_activation_job, + self, NotificationsReceived, NotifyHasPatch, ReplayExternalStreams, + UpdateRandomSeed, WorkflowActivation, workflow_activation_job, }, workflow_commands::{ContinueAsNewWorkflowExecution, ExternalStreamWait}, }, @@ -75,6 +76,7 @@ use temporalio_common::{ common::v1::SearchAttributes, enums::v1::EventType, history::v1::{HistoryEvent, history_event}, + notification::v1::Notification, protocol::v1::{Message as ProtocolMessage, message::SequencingId}, sdk::v1::{UserMetadata, WorkflowTaskCompletedMetadata}, }, @@ -98,6 +100,12 @@ pub(crate) struct WorkflowMachines { /// Reserved external stream wake Signals seen in history, decoded and suppressed from user /// dispatch, waiting to be classified against the run's wait set. pending_external_stream_wakes: Vec, + /// Channel notifications read from the scheduled events of the task being applied, folded + /// per channel. A task that failed and was scheduled again carries more than one scheduled + /// event, and lang still gets one job for the activation. + pending_notifications: Vec, + /// Set when notifications were handed to lang and the run's wait set has not seen them yet. + pending_notification_wake: bool, /// External stream marker machines whose `MarkerRecorded` event has not been reached yet, in /// the order the markers appear in History. /// @@ -307,6 +315,8 @@ impl WorkflowMachines { current_wf_time: None, observed_internal_flags: Rc::new(RefCell::new(observed_internal_flags)), pending_external_stream_wakes: vec![], + pending_notifications: vec![], + pending_notification_wake: false, external_stream_marker_machines: Default::default(), history_size_bytes: 0, continue_as_new_suggested: false, @@ -583,6 +593,30 @@ impl WorkflowMachines { std::mem::take(&mut self.pending_external_stream_wakes) } + /// Whether notifications reached lang since the last call. They come from History, so a + /// replay reports the same as the live run did. + pub(crate) fn take_notification_wake(&mut self) -> bool { + std::mem::take(&mut self.pending_notification_wake) + } + + /// Hands lang the notifications folded from this task's scheduled events as one job. + /// + /// Sent while the task's history is applied, so the job sits ahead of any external stream + /// resolve Core queues for the same activation once the history is in. + fn flush_pending_notifications(&mut self) { + if self.pending_notifications.is_empty() { + return; + } + let notifications = std::mem::take(&mut self.pending_notifications); + self.drive_me.send_job( + workflow_activation_job::Variant::NotificationsReceived(NotificationsReceived { + notifications, + }) + .into(), + ); + self.pending_notification_wake = true; + } + /// Queue a Core-generated job for lang. /// /// Used by the external stream paths, whose activations originate in Core rather than from a @@ -894,6 +928,7 @@ impl WorkflowMachines { } self.last_processed_event = eid; } + self.flush_pending_notifications(); // Needed to delay mutation of self until after we've iterated over peeked events. #[allow(clippy::large_enum_variant)] @@ -1212,6 +1247,12 @@ impl WorkflowMachines { } } Ok(EventType::WorkflowTaskScheduled) => { + if let Some(history_event::Attributes::WorkflowTaskScheduledEventAttributes( + ref attrs, + )) = event_dat.event.attributes + { + fold_notifications(&mut self.pending_notifications, &attrs.notifications); + } let wf_task_sm = WorkflowTaskMachine::new(self.next_started_event_id); let key = self.all_machines.insert(wf_task_sm.into()); self.submachine_handle_event(key, event_dat)?; @@ -1713,6 +1754,15 @@ impl WorkflowMachines { CommandIdKind::NeverResolves, ); } + WFCommandVariant::SubscribeNotificationChannel(attrs) => { + // Never resolves: the event it produces records the subscription and hands + // nothing back. The notifications arrive later on a scheduled event. + self.add_cmd_to_wf_task( + subscribe_notification_channel(attrs), + cmd.metadata, + CommandIdKind::NeverResolves, + ); + } WFCommandVariant::UpdateResponse(ur) => { let m_key = self.get_machine_by_msg(&ur.protocol_instance_id)?; let m = if let Machines::UpdateMachine(m) = self.machine_mut(m_key) { @@ -2048,3 +2098,18 @@ fn decode_wake_signal( } Some(wake) } + +/// Adds a scheduled event's notifications to the ones already read for this task. +/// +/// Uses the server's own fold rule, one per channel with the highest counter kept, so a channel +/// seen on two scheduled events of one task reaches lang once. Channels keep the order they were +/// first seen in, which History fixes, so a replay builds the same job. +fn fold_notifications(pending: &mut Vec, incoming: &[Notification]) { + for n in incoming { + match pending.iter_mut().find(|p| p.channel == n.channel) { + Some(p) if n.counter > p.counter => *p = n.clone(), + Some(_) => {} + None => pending.push(n.clone()), + } + } +} diff --git a/crates/sdk-core/src/worker/workflow/managed_run.rs b/crates/sdk-core/src/worker/workflow/managed_run.rs index b5bb11eac..20965dbbf 100644 --- a/crates/sdk-core/src/worker/workflow/managed_run.rs +++ b/crates/sdk-core/src/worker/workflow/managed_run.rs @@ -2163,12 +2163,13 @@ impl ManagedRun { /// only whether the Run resumes. fn apply_external_stream_wakes(&mut self) -> bool { let wakes = self.wfm.machines.take_external_stream_wakes(); + let notified = self.wfm.machines.take_notification_wake(); let poll_wakes = if self.wfm.machines.replaying { vec![] } else { mem::take(&mut self.pending_poll_wakes) }; - if wakes.is_empty() && poll_wakes.is_empty() { + if wakes.is_empty() && poll_wakes.is_empty() && !notified { return false; } @@ -2182,7 +2183,12 @@ impl ManagedRun { "Resuming external stream waits for a poll response wake" ); } - let mut resume = !poll_wakes.is_empty(); + // Notifications on the scheduled event are the same unparked wake, read from History + // rather than the poll response, so they apply in replay as well. + if notified { + debug!("Resuming external stream waits for channel notifications"); + } + let mut resume = notified || !poll_wakes.is_empty(); let chain = self .wfm .machines diff --git a/crates/sdk-core/src/worker/workflow/mod.rs b/crates/sdk-core/src/worker/workflow/mod.rs index e0cb8705f..e57d5b36e 100644 --- a/crates/sdk-core/src/worker/workflow/mod.rs +++ b/crates/sdk-core/src/worker/workflow/mod.rs @@ -1730,6 +1730,7 @@ enum WFCommandVariant { ExternalOutputStreamCommit(WorkflowOutputStreamCommit), /// External output is buffered in lang and needs a run-scoped flush deadline. ExternalOutputStreamBuffered(WorkflowOutputStreamBuffered), + SubscribeNotificationChannel(SubscribeNotificationChannel), } impl TryFrom for WFCommand { @@ -1809,6 +1810,9 @@ impl TryFrom for WFCommand { workflow_command::Variant::WorkflowOutputStreamBuffered(buffered) => { WFCommandVariant::ExternalOutputStreamBuffered(buffered) } + workflow_command::Variant::SubscribeNotificationChannel(s) => { + WFCommandVariant::SubscribeNotificationChannel(s) + } }; Ok(Self { variant, @@ -2028,7 +2032,7 @@ impl LocalActivityRequestSink for LAReqSink { /// 1. init workflow /// 2. patches /// 3. random-seed-updates -/// 4. signals/updates +/// 4. signals/updates/channel notifications /// 5. all others /// 6. local activity resolutions /// 7. queries @@ -2068,6 +2072,7 @@ fn prepare_to_ship_activation(wfa: &mut WorkflowActivation) { workflow_activation_job::Variant::UpdateRandomSeed(_) => 2, workflow_activation_job::Variant::SignalWorkflow(_) => 3, workflow_activation_job::Variant::DoUpdate(_) => 3, + workflow_activation_job::Variant::NotificationsReceived(_) => 3, workflow_activation_job::Variant::ResolveActivity(ra) if ra.is_local => 5, // In principle we should never actually need to sort these with the others, since // queries always get their own activation, but, maintaining the semantic is diff --git a/crates/sdk/src/workflow_future.rs b/crates/sdk/src/workflow_future.rs index 818ba2c8f..963b8a52a 100644 --- a/crates/sdk/src/workflow_future.rs +++ b/crates/sdk/src/workflow_future.rs @@ -343,6 +343,11 @@ impl WorkflowFuture { Variant::RemoveFromCache(_) => { unreachable!("Cache removal should happen higher up"); } + // Nothing to wake: this SDK cannot subscribe to a channel, and a notification + // is a hint without data, so the guest runtime drops it. + Variant::NotificationsReceived(_) => { + push_context!(ActivationJobContext::Passive); + } // External Workflow Streams are a Python-SDK feature. This SDK never emits // `WorkflowStreamQuiescent`, so Core has nothing to retain a Workflow Task for // and cannot produce any of these jobs for it. diff --git a/crates/workflow/src/runtime/instance.rs b/crates/workflow/src/runtime/instance.rs index 2aee72e92..60bb6f210 100644 --- a/crates/workflow/src/runtime/instance.rs +++ b/crates/workflow/src/runtime/instance.rs @@ -1052,6 +1052,9 @@ where ActivationJobResult::None } Some(ActivationVariant::RemoveFromCache(_)) => ActivationJobResult::None, + // This runtime has no way to subscribe to a channel. A notification carries no + // data, only a hint to go read a source, so dropping one loses nothing. + Some(ActivationVariant::NotificationsReceived(_)) => ActivationJobResult::None, // External Workflow Streams are a Python-SDK feature; this runtime never asks // Core to retain a Workflow Task for a stream wait, so it can only ever see one // of these by mistake. Failing loudly beats a wildcard that would silently drop From 1497b927622b4d1d65a8f5da7a1134e23e8c9f44 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Thu, 1 Oct 2026 18:19:33 -0700 Subject: [PATCH 25/35] Carried the subscribe-notification-channel failed cause in the Core's api tree. --- .../api_upstream/temporal/api/enums/v1/failed_cause.proto | 3 +++ 1 file changed, 3 insertions(+) diff --git a/crates/protos/protos/api_upstream/temporal/api/enums/v1/failed_cause.proto b/crates/protos/protos/api_upstream/temporal/api/enums/v1/failed_cause.proto index 3e675a1f0..b920b0fe4 100644 --- a/crates/protos/protos/api_upstream/temporal/api/enums/v1/failed_cause.proto +++ b/crates/protos/protos/api_upstream/temporal/api/enums/v1/failed_cause.proto @@ -98,6 +98,9 @@ enum WorkflowTaskFailedCause { // History can no longer be served, for example after truncation or because it exceeds the // replay bound. Check the workflow task failure message for more information. WORKFLOW_TASK_FAILED_CAUSE_STREAM_RANGE_UNAVAILABLE = 43; + // A SubscribeNotificationChannel command named an empty or too-long channel, or hit a + // subscription or listener limit. + WORKFLOW_TASK_FAILED_CAUSE_BAD_SUBSCRIBE_NOTIFICATION_CHANNEL_ATTRIBUTES = 44; } enum StartChildWorkflowExecutionFailedCause { From 757b9856c0af9cf22ce4ec5756d023e8360b4f70 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Thu, 1 Oct 2026 18:19:34 -0700 Subject: [PATCH 26/35] Handed a failed task's notifications to its retry in one job. The server clears what it puts on a scheduled event, so the retry's event carries only what arrived since. The notifications are joined in History order, and the lang protos read as the protos branch does. --- .../workflow_activation.proto | 10 +-- .../workflow_commands/workflow_commands.proto | 10 +-- .../src/core_tests/external_streams.rs | 65 +++++++++++++------ .../workflow/machines/workflow_machines.rs | 27 ++------ 4 files changed, 62 insertions(+), 50 deletions(-) diff --git a/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto b/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto index b583b0ff9..d86ff4c3c 100644 --- a/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto +++ b/crates/protos/protos/local/temporal/sdk/core/workflow_activation/workflow_activation.proto @@ -155,8 +155,7 @@ message WorkflowActivationJob { // Runtime-internal: encode the terminal boundary for a marker Core is about to write. // Runs no user code. FinalizeExternalStreams finalize_external_streams = 20; - // Notifications for channels this run listens to, read from the Workflow Task's - // scheduled event. + // Notifications from the channels the workflow subscribed to. NotificationsReceived notifications_received = 22; // Remove the workflow identified by the [WorkflowActivation] containing this job from the // cache after performing the activation. It is guaranteed that this will be the only job @@ -165,10 +164,11 @@ message WorkflowActivationJob { } } -// The notifications the server folded onto this Workflow Task's scheduled event, one per channel. +// Hand a workflow the notifications the server folded for its channels. // -// Read from History, so a replay hands lang the same job in the same activation. A notification -// carries no data: lang reads the source the channel names. +// They come from the scheduled event of the Workflow Task this activation +// belongs to. History is the record, so a replay yields the same job with the +// same notifications at the same point. message NotificationsReceived { repeated temporal.api.notification.v1.Notification notifications = 1; } diff --git a/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto b/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto index 9abc82fda..4cb033695 100644 --- a/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto +++ b/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto @@ -59,11 +59,13 @@ message WorkflowCommand { } } -// Makes this run a listener of a notification channel. Nothing resolves back to lang: the -// notifications arrive later on the scheduled event of a Workflow Task, as a -// `NotificationsReceived` job. +// Subscribe this workflow to a notification channel, so the scheduled event of +// each later Workflow Task carries the notifications folded for it. +// +// The notifications live in History rather than arriving by a side channel, so +// a replay reads the same ones the live run saw. message SubscribeNotificationChannel { - // The channel to listen on, as the writers name it. + // Name of the channel, scoped to the namespace. string channel = 1; } diff --git a/crates/sdk-core/src/core_tests/external_streams.rs b/crates/sdk-core/src/core_tests/external_streams.rs index bd09ae93e..44a6a429b 100644 --- a/crates/sdk-core/src/core_tests/external_streams.rs +++ b/crates/sdk-core/src/core_tests/external_streams.rs @@ -3691,29 +3691,47 @@ async fn a_notified_scheduled_event_yields_the_job_and_resumes_a_parked_wait() { worker.drain_pollers_and_shutdown().await; } -#[tokio::test] -async fn a_task_scheduled_twice_hands_lang_one_folded_job() { - // A task that failed is scheduled again, and the second scheduled event may carry the same - // channel at a later counter. Lang gets one job per activation, folded the way the server - // folds: one notification per channel, the highest counter kept. +/// A history whose second task fails after being scheduled with `n1` and is retried with `n2`. +fn failed_then_retried_history(n1: Vec, n2: Vec) -> TestHistoryBuilder { let mut t = TestHistoryBuilder::default(); t.add_wfe_started_with_wft_timeout(Duration::from_secs(300)); t.add_full_wf_task(); - t.add_workflow_task_scheduled_with_notifications(vec![ - channel_notification("tokens", 41), - channel_notification("events", 3), - ]); + t.add_workflow_task_scheduled_with_notifications(n1); t.add_workflow_task_started(); - t.add_workflow_task_timed_out(); - t.add_workflow_task_scheduled_with_notifications(vec![channel_notification("tokens", 42)]); + t.add_workflow_task_failed_with_failure( + temporalio_common::protos::temporal::api::enums::v1::WorkflowTaskFailedCause::Unspecified, + Default::default(), + ); + t.add_workflow_task_scheduled_with_notifications(n2); t.add_workflow_task_started(); - t.add_workflow_task_completed(); - t.add_workflow_execution_completed(); + t +} +#[rstest::rstest] +#[case::live(false)] +#[case::replay(true)] +#[tokio::test] +async fn a_failed_task_hands_its_notifications_to_the_retry(#[case] cold: bool) { + // The server clears a listener's pending notifications when it puts them on a scheduled + // event, so a retry's event carries only what arrived after the failed attempt. The failed + // attempt's event stays in History, and its notifications reach lang on the retry, ahead of + // the retry's own and in one job. + let n1 = vec![ + channel_notification("tokens", 41), + channel_notification("events", 3), + ]; + let n2 = vec![channel_notification("tokens", 42)]; + let t = failed_then_retried_history(n1.clone(), n2.clone()); + let batches = if cold { + let resp = hist_to_poll_resp(&t, "fakeid".to_owned(), ResponseType::AllHistory); + vec![ResponseType::Raw(resp.resp)] + } else { + vec![ResponseType::ToTaskNum(1), ResponseType::ToTaskNum(2)] + }; let mut mock = build_mock_pollers(MockPollCfg::from_resp_batches( "fakeid", t, - [1, 2], + batches, mock_worker_client(), )); mock.worker_cfg(|w| { @@ -3722,22 +3740,27 @@ async fn a_task_scheduled_twice_hands_lang_one_folded_job() { }); let worker = mock_worker(mock); let first = worker.poll_workflow_activation().await.unwrap(); + assert_eq!(first.is_replaying, cold); + assert_eq!( + notification_jobs(&first), + Vec::>::new(), + "the notifications belong to the retried task, got {:?}", + first.jobs + ); worker .complete_workflow_activation(WorkflowActivationCompletion::empty(first.run_id)) .await .unwrap(); - let second = worker.poll_workflow_activation().await.unwrap(); + let retried = worker.poll_workflow_activation().await.unwrap(); + assert!(!retried.is_replaying, "got jobs {:?}", retried.jobs); assert_eq!( - notification_jobs(&second), - vec![vec![ - channel_notification("tokens", 42), - channel_notification("events", 3), - ]] + notification_jobs(&retried), + vec![n1.into_iter().chain(n2).collect::>()] ); worker .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( - second.run_id, + retried.run_id, vec![CompleteWorkflowExecution::default().into()], )) .await diff --git a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs index 0f9931369..9d49926f1 100644 --- a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs +++ b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs @@ -100,9 +100,10 @@ pub(crate) struct WorkflowMachines { /// Reserved external stream wake Signals seen in history, decoded and suppressed from user /// dispatch, waiting to be classified against the run's wait set. pending_external_stream_wakes: Vec, - /// Channel notifications read from the scheduled events of the task being applied, folded - /// per channel. A task that failed and was scheduled again carries more than one scheduled - /// event, and lang still gets one job for the activation. + /// Channel notifications read from the scheduled events of the task being applied, in History + /// order. The server clears what it put on a scheduled event, so a retry's event carries only + /// what arrived since. The failed task's notifications stay in History and are only ever + /// handed over here, together with the retry's, as one job. pending_notifications: Vec, /// Set when notifications were handed to lang and the run's wait set has not seen them yet. pending_notification_wake: bool, @@ -599,7 +600,7 @@ impl WorkflowMachines { std::mem::take(&mut self.pending_notification_wake) } - /// Hands lang the notifications folded from this task's scheduled events as one job. + /// Hands lang the notifications from this task's scheduled events as one job. /// /// Sent while the task's history is applied, so the job sits ahead of any external stream /// resolve Core queues for the same activation once the history is in. @@ -1251,7 +1252,8 @@ impl WorkflowMachines { ref attrs, )) = event_dat.event.attributes { - fold_notifications(&mut self.pending_notifications, &attrs.notifications); + self.pending_notifications + .extend(attrs.notifications.iter().cloned()); } let wf_task_sm = WorkflowTaskMachine::new(self.next_started_event_id); let key = self.all_machines.insert(wf_task_sm.into()); @@ -2098,18 +2100,3 @@ fn decode_wake_signal( } Some(wake) } - -/// Adds a scheduled event's notifications to the ones already read for this task. -/// -/// Uses the server's own fold rule, one per channel with the highest counter kept, so a channel -/// seen on two scheduled events of one task reaches lang once. Channels keep the order they were -/// first seen in, which History fixes, so a replay builds the same job. -fn fold_notifications(pending: &mut Vec, incoming: &[Notification]) { - for n in incoming { - match pending.iter_mut().find(|p| p.channel == n.channel) { - Some(p) if n.counter > p.counter => *p = n.clone(), - Some(_) => {} - None => pending.push(n.clone()), - } - } -} From 0a3e3a92562b71e2416231d89c6e9db9a9afbf97 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Thu, 1 Oct 2026 18:43:14 -0700 Subject: [PATCH 27/35] Folded a task's notifications per channel again, highest counter first. --- .../src/core_tests/external_streams.rs | 17 +++++------- .../workflow/machines/workflow_machines.rs | 26 ++++++++++++++----- 2 files changed, 27 insertions(+), 16 deletions(-) diff --git a/crates/sdk-core/src/core_tests/external_streams.rs b/crates/sdk-core/src/core_tests/external_streams.rs index 44a6a429b..d577d4032 100644 --- a/crates/sdk-core/src/core_tests/external_streams.rs +++ b/crates/sdk-core/src/core_tests/external_streams.rs @@ -3714,14 +3714,14 @@ fn failed_then_retried_history(n1: Vec, n2: Vec) -> async fn a_failed_task_hands_its_notifications_to_the_retry(#[case] cold: bool) { // The server clears a listener's pending notifications when it puts them on a scheduled // event, so a retry's event carries only what arrived after the failed attempt. The failed - // attempt's event stays in History, and its notifications reach lang on the retry, ahead of - // the retry's own and in one job. - let n1 = vec![ - channel_notification("tokens", 41), + // attempt's event stays in History, and its notifications reach lang on the retry, folded + // with the retry's own into one job: one per channel, the highest counter kept. + let n1 = vec![channel_notification("tokens", 41)]; + let n2 = vec![ + channel_notification("tokens", 42), channel_notification("events", 3), ]; - let n2 = vec![channel_notification("tokens", 42)]; - let t = failed_then_retried_history(n1.clone(), n2.clone()); + let t = failed_then_retried_history(n1, n2.clone()); let batches = if cold { let resp = hist_to_poll_resp(&t, "fakeid".to_owned(), ResponseType::AllHistory); vec![ResponseType::Raw(resp.resp)] @@ -3754,10 +3754,7 @@ async fn a_failed_task_hands_its_notifications_to_the_retry(#[case] cold: bool) let retried = worker.poll_workflow_activation().await.unwrap(); assert!(!retried.is_replaying, "got jobs {:?}", retried.jobs); - assert_eq!( - notification_jobs(&retried), - vec![n1.into_iter().chain(n2).collect::>()] - ); + assert_eq!(notification_jobs(&retried), vec![n2]); worker .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( retried.run_id, diff --git a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs index 9d49926f1..1329d86ba 100644 --- a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs +++ b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs @@ -100,10 +100,10 @@ pub(crate) struct WorkflowMachines { /// Reserved external stream wake Signals seen in history, decoded and suppressed from user /// dispatch, waiting to be classified against the run's wait set. pending_external_stream_wakes: Vec, - /// Channel notifications read from the scheduled events of the task being applied, in History - /// order. The server clears what it put on a scheduled event, so a retry's event carries only - /// what arrived since. The failed task's notifications stay in History and are only ever - /// handed over here, together with the retry's, as one job. + /// Channel notifications read from the scheduled events of the task being applied, folded + /// per channel. The server clears what it put on a scheduled event, so a retry's event + /// carries only what arrived since. The failed task's notifications stay in History and + /// reach lang here, folded with the retry's into one job. pending_notifications: Vec, /// Set when notifications were handed to lang and the run's wait set has not seen them yet. pending_notification_wake: bool, @@ -1252,8 +1252,7 @@ impl WorkflowMachines { ref attrs, )) = event_dat.event.attributes { - self.pending_notifications - .extend(attrs.notifications.iter().cloned()); + fold_notifications(&mut self.pending_notifications, &attrs.notifications); } let wf_task_sm = WorkflowTaskMachine::new(self.next_started_event_id); let key = self.all_machines.insert(wf_task_sm.into()); @@ -2100,3 +2099,18 @@ fn decode_wake_signal( } Some(wake) } + +/// Adds a scheduled event's notifications to the ones already read for this task. +/// +/// Uses the server's own fold rule, one per channel with the highest counter kept, so a channel +/// seen on two scheduled events of one task reaches lang once. Channels keep the order they were +/// first seen in, which History fixes, so a replay builds the same job. +fn fold_notifications(pending: &mut Vec, incoming: &[Notification]) { + for n in incoming { + match pending.iter_mut().find(|p| p.channel == n.channel) { + Some(p) if n.counter > p.counter => *p = n.clone(), + Some(_) => {} + None => pending.push(n.clone()), + } + } +} From cdc426024e9e208a6847b8fcfdee7253b2c27139 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Thu, 1 Oct 2026 22:28:55 -0700 Subject: [PATCH 28/35] Removed the point-to-point wake from the Core's api tree, client and receipt. The notification channel replaces the wake in the PoC, so the Core carries neither the api messages nor the poll-response receipt. The reserved wake Signal and the channel notifications keep resuming parked external waits. --- crates/client/src/grpc.rs | 9 - .../temporal/api/workflow/v1/message.proto | 29 --- .../workflowservice/v1/request_response.proto | 29 +-- .../api/workflowservice/v1/service.proto | 19 -- crates/sdk-core-c-bridge/src/client.rs | 3 - .../src/core_tests/external_streams.rs | 178 +----------------- crates/sdk-core/src/protosext/mod.rs | 12 +- .../src/worker/workflow/history_update.rs | 1 - .../src/worker/workflow/managed_run.rs | 37 +--- crates/sdk-core/src/worker/workflow/mod.rs | 2 - 10 files changed, 14 insertions(+), 305 deletions(-) diff --git a/crates/client/src/grpc.rs b/crates/client/src/grpc.rs index 0d6bb6ddf..59ba46875 100644 --- a/crates/client/src/grpc.rs +++ b/crates/client/src/grpc.rs @@ -955,15 +955,6 @@ proxier! { r.extensions_mut().insert(labels); } ); - ( - wake_workflow_execution, - WakeWorkflowExecutionRequest, - WakeWorkflowExecutionResponse, - |r| { - let labels = namespaced_request!(r); - r.extensions_mut().insert(labels); - } - ); ( notify_channel, NotifyChannelRequest, diff --git a/crates/protos/protos/api_upstream/temporal/api/workflow/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/workflow/v1/message.proto index 0c414794e..1ed33fa4c 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflow/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflow/v1/message.proto @@ -760,32 +760,3 @@ message WorkflowExecutionPauseInfo { // The reason for pausing the workflow execution. string reason = 3; } - -// A wake tells a running Workflow Execution that something it consumes has -// moved, so it should run a Workflow Task and read from its own cursor. It is -// a reason to run, not data: nothing is recorded in History and no record -// travels with it. A store Temporal does not host sends one after the store -// has acknowledged a write. Wakes for the same source fold while one is -// pending and no task has received it yet, so a burst of writes costs one -// task. Once a task has been handed the wake, a new wake for the source is -// accepted again, because only the receiver knows what it read. -// -// Superseded by the notification channel (`NotifyChannel` and the -// `SubscribeNotificationChannel` command). Kept for one round and slated for -// removal. -message Wake { - // What moved, as the sender names it. The server folds wakes by this value - // and never interprets it. For a stream, the provider formats the stream's - // identity, owner and topic, into it. - string source = 1; - // Where the source stands after the write that caused this wake, in the - // store's own terms. Opaque to the server and handed to the Workflow as - // received. Among wakes folded together, the position with the highest - // counter is the one delivered. - bytes position = 2; - // Orders wakes from one source, so that folding keeps the latest position. - // Only the source's store can order its positions, so the sender derives - // this from the position. It is not a durable identity: a wake sent after - // a task received an equal counter is accepted again. - int64 counter = 3; -} diff --git a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto index c0af15c8d..57d4327cd 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto @@ -390,11 +390,8 @@ message PollWorkflowTaskQueueResponse { // never enter History; only the offset ranges are recorded there. repeated temporal.api.stream.v1.StreamSlice stream_slices = 20; - // Wakes pending for this execution when the task started, folded by - // source. Not recorded in History and never re-supplied on replay: the - // Workflow treats one as a reason to read its source now and records what - // it read itself. - repeated temporal.api.workflow.v1.Wake wakes = 21; + // Used once by a repeated field this fork has since removed. + reserved 21; } message RespondWorkflowTaskCompletedRequest { @@ -886,28 +883,6 @@ message SignalWorkflowExecutionResponse { temporal.api.common.v1.Link link = 1; } -message WakeWorkflowExecutionRequest { - string namespace = 1; - // The Workflow to wake. `run_id` is optional. When set, the wake goes to - // the current run of the chain that run belongs to, so a sender that - // learned a run id before a continue-as-new still reaches the consumer, - // and it is refused with NotFound once that chain has ended. When unset, - // the wake goes to the current run under the Workflow Id, whichever chain. - temporal.api.common.v1.WorkflowExecution workflow_execution = 2; - temporal.api.workflow.v1.Wake wake = 3; - // The identity of the sender, for metrics and logs. - string identity = 4; -} - -message WakeWorkflowExecutionResponse { - // The run the wake was stored on. - string run_id = 1; - // True when a wake for the source was already pending and no task had - // received it yet, so this one changed nothing but possibly the position - // and no new task results from it. - bool folded = 2; -} - message NotifyChannelRequest { string namespace = 1; temporal.api.notification.v1.Notification notification = 2; diff --git a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto index 9aff1154a..6985b444e 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto @@ -483,25 +483,6 @@ service WorkflowService { }; } - // WakeWorkflowExecution asks a running Workflow Execution to run a Workflow - // Task because a source it consumes has moved. Unlike a Signal it records - // no event, carries no payload, and folds with other wakes for the same - // source, so a burst of writes costs one task. The Workflow learns the - // source and its position from the task and reads the source itself. - // - // Superseded by the notification channel (`NotifyChannel` and the - // `SubscribeNotificationChannel` command). Kept for one round and slated for removal. - rpc WakeWorkflowExecution (WakeWorkflowExecutionRequest) returns (WakeWorkflowExecutionResponse) { - option (google.api.http) = { - post: "/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/wake" - body: "*" - additional_bindings { - post: "/api/v1/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/wake" - body: "*" - } - }; - } - // NotifyChannel tells every listener of a channel that a source they consume // has moved. The writer names no addressee and never learns who listens. The // server wakes each listener: a Workflow with a Workflow Task, a callback by diff --git a/crates/sdk-core-c-bridge/src/client.rs b/crates/sdk-core-c-bridge/src/client.rs index 75ae21a9f..9776327c5 100644 --- a/crates/sdk-core-c-bridge/src/client.rs +++ b/crates/sdk-core-c-bridge/src/client.rs @@ -1122,9 +1122,6 @@ async fn call_workflow_service( validate_worker_deployment_version_compute_config ) } - "WakeWorkflowExecution" => { - rpc_call_on_trait!(client, call, WorkflowService, wake_workflow_execution) - } rpc => Err(anyhow::anyhow!("Unknown RPC call {rpc}")), } } diff --git a/crates/sdk-core/src/core_tests/external_streams.rs b/crates/sdk-core/src/core_tests/external_streams.rs index d577d4032..52d7400df 100644 --- a/crates/sdk-core/src/core_tests/external_streams.rs +++ b/crates/sdk-core/src/core_tests/external_streams.rs @@ -54,7 +54,6 @@ use temporalio_common::{ history::v1::History, notification::v1::Notification, query::v1::WorkflowQuery, - workflow::v1::Wake, workflowservice::v1::{ GetWorkflowExecutionHistoryResponse, RespondWorkflowTaskCompletedResponse, }, @@ -3407,177 +3406,6 @@ async fn an_ordinary_signal_still_reaches_its_handler() { worker.drain_pollers_and_shutdown().await; } -// --- wakes carried on the poll response -------------------------------------- - -fn poll_wake() -> Wake { - Wake { - source: "tokens".to_string(), - position: b"42-0".to_vec(), - counter: 42, - } -} - -/// A worker whose second Workflow Task arrives with `wakes` on its poll response. -/// -/// The wake-driven task records nothing of its own beyond its scheduled and started events, -/// which is the shape the server produces when only a wake is pending. `user_signal` adds an -/// ordinary Signal ahead of that task for the tests that need a job to look at. -fn worker_with_a_poll_wake(wakes: Vec, user_signal: bool) -> crate::Worker { - let mut t = TestHistoryBuilder::default(); - t.add_wfe_started_with_wft_timeout(Duration::from_secs(300)); - t.add_full_wf_task(); - if user_signal { - t.add_we_signaled("a-user-signal", vec![]); - } - t.add_full_wf_task(); - t.add_workflow_execution_completed(); - - let mut second = hist_to_poll_resp(&t, "fakeid".to_owned(), ResponseType::ToTaskNum(2)); - second.wakes = wakes; - let mut mock = build_mock_pollers(MockPollCfg::from_resp_batches( - "fakeid", - t, - [ResponseType::ToTaskNum(1), ResponseType::Raw(second.resp)], - mock_worker_client(), - )); - mock.worker_cfg(|w| { - w.task_types = WorkerTaskTypes::workflow_only(); - w.max_cached_workflows = 1; - }); - mock_worker(mock) -} - -#[tokio::test] -async fn a_poll_wake_resumes_a_parked_wait() { - // The server stored the wake on this run, so Core treats it as an unparked wake: no chain to - // check and no generation to match, even against a set parked at a generation of its own. - let worker = worker_with_a_poll_wake(vec![poll_wake()], false); - advance_to_the_signal_task(&worker, vec![1, 2], Some(7)).await; - - let second = worker.poll_workflow_activation().await.unwrap(); - - assert_eq!( - resolve_hints(&second), - vec![1, 2], - "a poll wake must mark every active wait, got {:?}", - second.jobs - ); - assert!( - !second.is_replaying, - "the task a poll wake arrives on is live work, got jobs {:?}", - second.jobs - ); - - worker - .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( - second.run_id, - vec![CompleteWorkflowExecution::default().into()], - )) - .await - .unwrap(); - worker.drain_pollers_and_shutdown().await; -} - -#[tokio::test] -async fn a_poll_wake_with_no_waits_is_harmless() { - // Nothing is parked, so there is nothing to resume. The task still runs its ordinary work and - // the wake adds no job to it. - let worker = worker_with_a_poll_wake(vec![poll_wake()], true); - let first = worker.poll_workflow_activation().await.unwrap(); - worker - .complete_workflow_activation(WorkflowActivationCompletion::empty(first.run_id)) - .await - .unwrap(); - - let second = worker.poll_workflow_activation().await.unwrap(); - assert!( - second.jobs.iter().any(|j| matches!( - &j.variant, - Some(workflow_activation_job::Variant::SignalWorkflow(s)) - if s.signal_name == "a-user-signal" - )), - "the ordinary Signal must still reach its handler, got {:?}", - second.jobs - ); - assert_eq!( - resolve_hints(&second), - Vec::::new(), - "with no waits there is nothing for the wake to resolve" - ); - - worker - .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( - second.run_id, - vec![CompleteWorkflowExecution::default().into()], - )) - .await - .unwrap(); - worker.drain_pollers_and_shutdown().await; -} - -#[tokio::test] -async fn a_poll_wake_is_not_applied_while_replaying() { - // A cold worker receives the whole History with the live task's wakes. The wakes describe the - // live task, so the replayed activation must not carry them, and they must not be spent - // there either: the waits that exist once replay catches up are the ones they resume. The - // Signal keeps the two tasks apart; back to back, Core would treat them as one live task. - let mut t = TestHistoryBuilder::default(); - t.add_wfe_started_with_wft_timeout(Duration::from_secs(300)); - t.add_full_wf_task(); - t.add_we_signaled("a-user-signal", vec![]); - t.add_full_wf_task(); - t.add_workflow_execution_completed(); - - let mut cold = hist_to_poll_resp(&t, "fakeid".to_owned(), ResponseType::ToTaskNum(2)); - cold.wakes = vec![poll_wake()]; - let mut mock = build_mock_pollers(MockPollCfg::from_resp_batches( - "fakeid", - t, - [ResponseType::Raw(cold.resp)], - mock_worker_client(), - )); - mock.worker_cfg(|w| { - w.task_types = WorkerTaskTypes::workflow_only(); - w.max_cached_workflows = 1; - }); - let worker = mock_worker(mock); - - let replayed = worker.poll_workflow_activation().await.unwrap(); - assert!(replayed.is_replaying); - assert_eq!( - resolve_hints(&replayed), - Vec::::new(), - "a replayed activation must not carry a poll wake, got {:?}", - replayed.jobs - ); - let run_id = replayed.run_id.clone(); - worker - .seed_external_stream_waits(&run_id, vec![1], None, false) - .await; - worker - .complete_workflow_activation(WorkflowActivationCompletion::empty(run_id.clone())) - .await - .unwrap(); - - let live = worker.poll_workflow_activation().await.unwrap(); - assert!(!live.is_replaying, "got jobs {:?}", live.jobs); - assert_eq!( - resolve_hints(&live), - vec![1], - "the wake held through replay must resume the live task, got {:?}", - live.jobs - ); - - worker - .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( - live.run_id, - vec![CompleteWorkflowExecution::default().into()], - )) - .await - .unwrap(); - worker.drain_pollers_and_shutdown().await; -} - // --- notifications on the scheduled event ------------------------------------ fn channel_notification(channel: &str, counter: i64) -> Notification { @@ -3629,8 +3457,8 @@ fn notified_second_task_history(notifications: Vec) -> TestHistory #[tokio::test] async fn a_notified_scheduled_event_yields_the_job_and_resumes_a_parked_wait() { // The server folds a channel's notifications onto the scheduled event of the task it wakes - // the run with. The job hands them to lang, and the same task resumes the parked waits the - // way a poll wake does: no chain to check and no generation to match. + // the run with. The job hands them to lang, and the same task resumes the parked waits as an + // unparked wake: no chain to check and no generation to match. let notifications = vec![channel_notification("tokens", 42)]; let mut mock = build_mock_pollers(MockPollCfg::from_resp_batches( "fakeid", @@ -3830,7 +3658,7 @@ async fn replaying_a_notified_scheduled_event_yields_the_same_job() { #[tokio::test] async fn replayed_notifications_resume_the_reconstructed_waits() { - // The wake comes from History, so unlike a poll wake it is applied while replaying: the waits + // The notifications come from History, so they are applied while replaying too: the waits // lang reconstructs for the notified task are resumed by the same activation. let notifications = vec![channel_notification("tokens", 42)]; let mut t = TestHistoryBuilder::default(); diff --git a/crates/sdk-core/src/protosext/mod.rs b/crates/sdk-core/src/protosext/mod.rs index f83c6af6d..b2adcde96 100644 --- a/crates/sdk-core/src/protosext/mod.rs +++ b/crates/sdk-core/src/protosext/mod.rs @@ -39,7 +39,6 @@ use temporalio_common::protos::{ history::v1::{History, HistoryEvent, MarkerRecordedEventAttributes, history_event}, query::v1::WorkflowQuery, sdk::v1::UserMetadata, - workflow::v1::Wake, workflowservice::v1::PollWorkflowTaskQueueResponse, }, utilities::TryIntoOrNone, @@ -65,9 +64,6 @@ pub(crate) struct ValidPollWFTQResponse { pub(crate) query_requests: Vec, /// Protocol messages pub(crate) messages: Vec, - /// Wakes pending for the execution when this task started. Live only: the server never - /// supplies them on replay and never records them in History. - pub(crate) wakes: Vec, /// Zero-size field to prevent explicit construction _cant_construct_me: (), @@ -79,8 +75,7 @@ impl Debug for ValidPollWFTQResponse { f, "ValidWFT {{ task_token: {}, task_queue: {}, workflow_execution: {:?}, \ workflow_type: {}, attempt: {}, previous_started_event_id: {}, started_event_id {}, \ - history_length: {}, first_evt_in_hist_id: {:?}, legacy_query: {:?}, queries: {:?}, \ - wakes: {:?} }}", + history_length: {}, first_evt_in_hist_id: {:?}, legacy_query: {:?}, queries: {:?} }}", self.task_token, self.task_queue, self.workflow_execution, @@ -91,8 +86,7 @@ impl Debug for ValidPollWFTQResponse { self.history.events.len(), self.history.events.first().map(|e| e.event_id), self.legacy_query, - self.query_requests, - self.wakes + self.query_requests ) } } @@ -115,7 +109,6 @@ impl TryFrom for ValidPollWFTQResponse { query, queries, messages, - wakes, .. } => { if task_token.is_empty() { @@ -140,7 +133,6 @@ impl TryFrom for ValidPollWFTQResponse { legacy_query: query, query_requests, messages, - wakes, _cant_construct_me: (), }) } diff --git a/crates/sdk-core/src/worker/workflow/history_update.rs b/crates/sdk-core/src/worker/workflow/history_update.rs index 207cc46f0..109edb4a3 100644 --- a/crates/sdk-core/src/worker/workflow/history_update.rs +++ b/crates/sdk-core/src/worker/workflow/history_update.rs @@ -162,7 +162,6 @@ impl HistoryPaginator { query_requests: wft.query_requests, update, messages: wft.messages, - wakes: wft.wakes, }; Ok((paginator, prepared)) } diff --git a/crates/sdk-core/src/worker/workflow/managed_run.rs b/crates/sdk-core/src/worker/workflow/managed_run.rs index 20965dbbf..a8c937524 100644 --- a/crates/sdk-core/src/worker/workflow/managed_run.rs +++ b/crates/sdk-core/src/worker/workflow/managed_run.rs @@ -53,7 +53,6 @@ use temporalio_common::protos::{ temporal::api::{ enums::v1::{VersioningBehavior, WorkflowTaskFailedCause}, failure::v1::Failure, - workflow::v1::Wake, }, }; use tokio::sync::oneshot; @@ -116,10 +115,6 @@ pub(super) struct ManagedRun { paginator: Option, completion_waiting_on_page_fetch: Option, config: Arc, - /// Wakes from the poll response of the current task, held until the machines have replayed up - /// to that task. They describe the live task, so applying one mid-replay would resume waits - /// the replayed History never resumed. - pending_poll_wakes: Vec, } impl ManagedRun { pub(super) fn new( @@ -146,7 +141,6 @@ impl ManagedRun { paginator: None, completion_waiting_on_page_fetch: None, config, - pending_poll_wakes: vec![], }; let rua = me.incoming_wft(wft); (me, rua) @@ -260,10 +254,6 @@ impl ManagedRun { } self.paginator = Some(pwft.paginator); - // Wakes held for an earlier task that never caught up belong to that task. The server - // hands any still pending to this one, so they are replaced rather than accumulated. A - // legacy query task runs no Workflow code and so cannot act on a wake. - self.pending_poll_wakes = if was_legacy_query { vec![] } else { work.wakes }; // A Workflow Task is open from here until it is reported. The unwritten-annotation // invariant is stated against *that*, not against quiescence -- a task can accumulate a // delta and complete without ever asking to be retained. @@ -2156,7 +2146,7 @@ impl ManagedRun { } /// Classifies the wake Signals the machines decoded out of this task's history, together with - /// the wakes the poll response carried for it. + /// the channel notifications its scheduled event carried. /// /// Returns `true` if any of them should wake the Run. Every Signal is suppressed from user /// handlers regardless -- that already happened in the machines -- so what is decided here is @@ -2164,31 +2154,18 @@ impl ManagedRun { fn apply_external_stream_wakes(&mut self) -> bool { let wakes = self.wfm.machines.take_external_stream_wakes(); let notified = self.wfm.machines.take_notification_wake(); - let poll_wakes = if self.wfm.machines.replaying { - vec![] - } else { - mem::take(&mut self.pending_poll_wakes) - }; - if wakes.is_empty() && poll_wakes.is_empty() && !notified { + if wakes.is_empty() && !notified { return false; } - // A poll wake counts as an unparked wake (generation 0), which is never rejected. The - // server resolved the chain when it stored the wake on this run, so there is no chain - // identity left to compare. - for wake in &poll_wakes { - debug!( - source = %wake.source, - counter = wake.counter, - "Resuming external stream waits for a poll response wake" - ); - } - // Notifications on the scheduled event are the same unparked wake, read from History - // rather than the poll response, so they apply in replay as well. + // Notifications on the scheduled event count as an unparked wake (generation 0), which is + // never rejected: the server resolved the chain when it folded them onto this run's task, + // so there is no chain identity left to compare. They come from History, so they apply in + // replay as well. if notified { debug!("Resuming external stream waits for channel notifications"); } - let mut resume = notified || !poll_wakes.is_empty(); + let mut resume = notified; let chain = self .wfm .machines diff --git a/crates/sdk-core/src/worker/workflow/mod.rs b/crates/sdk-core/src/worker/workflow/mod.rs index e57d5b36e..77f9ce1e8 100644 --- a/crates/sdk-core/src/worker/workflow/mod.rs +++ b/crates/sdk-core/src/worker/workflow/mod.rs @@ -93,7 +93,6 @@ use temporalio_common::{ query::v1::WorkflowQuery, sdk::v1::{UserMetadata, WorkflowTaskCompletedMetadata}, taskqueue::v1::StickyExecutionAttributes, - workflow::v1::Wake, workflowservice::v1::{PollActivityTaskQueueResponse, get_system_info_response}, }, }, @@ -1141,7 +1140,6 @@ struct PreparedWFT { query_requests: Vec, update: HistoryUpdate, messages: Vec, - wakes: Vec, } impl PreparedWFT { From bdca403d9777bb1d80c759ff1fe9b3b8c82590fd Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Thu, 1 Oct 2026 22:56:48 -0700 Subject: [PATCH 29/35] Carried the linked channel contract in the Core's api tree. A linked channel lives in one workflow's state and is addressed by workflow id, so the channel calls take an optional workflow_execution, a notification names the channel it is linked to, and describe reports the kind and the owner. --- .../api/notification/v1/message.proto | 17 ++++++++++ .../workflowservice/v1/request_response.proto | 26 ++++++++++++++ .../api/workflowservice/v1/service.proto | 34 +++++++++++++++++++ 3 files changed, 77 insertions(+) diff --git a/crates/protos/protos/api_upstream/temporal/api/notification/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/notification/v1/message.proto index 63e1b7716..e1413efe1 100644 --- a/crates/protos/protos/api_upstream/temporal/api/notification/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/notification/v1/message.proto @@ -32,6 +32,12 @@ message Notification { // Details for the listener, such as which topic moved. Bounded in size and // carried as payloads, so a codec applies as to any payload. map metadata = 4; + // Set for a channel linked to a workflow: the owner and the run that + // received the notification. Empty for an independent channel. A listener + // that holds both kinds routes the notification by it. + // (-- api-linter: core::0140::prepositions=disabled + // aip.dev/not-precedent: "to" names the owner the channel is linked to. --) + temporal.api.common.v1.WorkflowExecution linked_to = 5; } // A listener of a channel: a Workflow Execution woken with a Workflow Task, or @@ -53,3 +59,14 @@ message WorkflowListener { // chain's current run when it delivers. string run_id = 2; } + +// Where a channel lives, which decides how a call addresses it. +enum ChannelKind { + CHANNEL_KIND_UNSPECIFIED = 0; + // Its own execution, keyed by namespace and channel name. Any number of + // workflows and callbacks listen to it. + CHANNEL_KIND_INDEPENDENT = 1; + // Kept in one workflow's state, keyed by namespace, workflow id and + // channel name. The owning workflow is its listener by construction. + CHANNEL_KIND_LINKED = 2; +} diff --git a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto index 57d4327cd..660ebf809 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto @@ -890,6 +890,10 @@ message NotifyChannelRequest { string identity = 3; // Used to de-dupe a retried notification. string request_id = 4; + // When set, the call addresses the channel linked to this workflow; `run_id` + // is optional and resolves to the chain's current run, as a Signal does. + // When unset, the call addresses the independent channel of that name. + temporal.api.common.v1.WorkflowExecution workflow_execution = 5; } message NotifyChannelResponse { @@ -907,6 +911,10 @@ message RegisterChannelListenerRequest { string request_id = 4; // The identity of the caller, for metrics and logs. string identity = 5; + // When set, the call addresses the channel linked to this workflow; `run_id` + // is optional and resolves to the chain's current run, as a Signal does. + // When unset, the call addresses the independent channel of that name. + temporal.api.common.v1.WorkflowExecution workflow_execution = 6; } message RegisterChannelListenerResponse { @@ -919,6 +927,10 @@ message UnregisterChannelListenerRequest { string listener_id = 3; // The identity of the caller, for metrics and logs. string identity = 4; + // When set, the call addresses the channel linked to this workflow; `run_id` + // is optional and resolves to the chain's current run, as a Signal does. + // When unset, the call addresses the independent channel of that name. + temporal.api.common.v1.WorkflowExecution workflow_execution = 5; } message UnregisterChannelListenerResponse { @@ -937,6 +949,10 @@ message PollChannelRequest { // At most this many notifications are returned. Zero means the server's // default. int32 max_notifications = 5; + // When set, the call addresses the channel linked to this workflow; `run_id` + // is optional and resolves to the chain's current run, as a Signal does. + // When unset, the call addresses the independent channel of that name. + temporal.api.common.v1.WorkflowExecution workflow_execution = 6; } message PollChannelResponse { @@ -946,6 +962,10 @@ message PollChannelResponse { message DescribeChannelRequest { string namespace = 1; string channel = 2; + // When set, the call addresses the channel linked to this workflow; `run_id` + // is optional and resolves to the chain's current run, as a Signal does. + // When unset, the call addresses the independent channel of that name. + temporal.api.common.v1.WorkflowExecution workflow_execution = 3; } message DescribeChannelResponse { @@ -954,6 +974,12 @@ message DescribeChannelResponse { temporal.api.notification.v1.Notification latest = 2; // How many notifications the channel retains for pollers. int32 retained_count = 3; + temporal.api.notification.v1.ChannelKind kind = 4; + // The owner of a linked channel and the run that holds it. Empty for an + // independent channel. + // (-- api-linter: core::0140::prepositions=disabled + // aip.dev/not-precedent: "to" names the owner the channel is linked to. --) + temporal.api.common.v1.WorkflowExecution linked_to = 5; } message SignalWithStartWorkflowExecutionRequest { diff --git a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto index 6985b444e..95238d2ab 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto @@ -496,6 +496,14 @@ service WorkflowService { post: "/api/v1/namespaces/{namespace}/channels/{notification.channel}/notify" body: "*" } + additional_bindings { + post: "/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/channels/{notification.channel}/notify" + body: "*" + } + additional_bindings { + post: "/api/v1/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/channels/{notification.channel}/notify" + body: "*" + } }; } @@ -511,6 +519,14 @@ service WorkflowService { post: "/api/v1/namespaces/{namespace}/channels/{channel}/listeners" body: "*" } + additional_bindings { + post: "/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/channels/{channel}/listeners" + body: "*" + } + additional_bindings { + post: "/api/v1/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/channels/{channel}/listeners" + body: "*" + } }; } @@ -525,6 +541,12 @@ service WorkflowService { additional_bindings { delete: "/api/v1/namespaces/{namespace}/channels/{channel}/listeners/{listener_id}" } + additional_bindings { + delete: "/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/channels/{channel}/listeners/{listener_id}" + } + additional_bindings { + delete: "/api/v1/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/channels/{channel}/listeners/{listener_id}" + } }; } @@ -537,6 +559,12 @@ service WorkflowService { additional_bindings { get: "/api/v1/namespaces/{namespace}/channels/{channel}/notifications" } + additional_bindings { + get: "/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/channels/{channel}/notifications" + } + additional_bindings { + get: "/api/v1/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/channels/{channel}/notifications" + } }; } @@ -548,6 +576,12 @@ service WorkflowService { additional_bindings { get: "/api/v1/namespaces/{namespace}/channels/{channel}" } + additional_bindings { + get: "/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/channels/{channel}" + } + additional_bindings { + get: "/api/v1/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/channels/{channel}" + } }; } From c6af517523500ed33a42a4b35220459f1f0bee00 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Thu, 1 Oct 2026 23:05:46 -0700 Subject: [PATCH 30/35] Defaulted the new notification fields in the Core's tests. The api tree now adds fields to Notification, so the test helper takes what it does not set from Default instead of naming every field. --- crates/sdk-core/src/core_tests/external_streams.rs | 1 + 1 file changed, 1 insertion(+) diff --git a/crates/sdk-core/src/core_tests/external_streams.rs b/crates/sdk-core/src/core_tests/external_streams.rs index 52d7400df..a88be1727 100644 --- a/crates/sdk-core/src/core_tests/external_streams.rs +++ b/crates/sdk-core/src/core_tests/external_streams.rs @@ -3414,6 +3414,7 @@ fn channel_notification(channel: &str, counter: i64) -> Notification { position: format!("{counter}-0").into_bytes(), counter, metadata: HashMap::new(), + ..Default::default() } } From 207a295c12ce1f20ddb23d922a7502f1c0c275a7 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 2 Oct 2026 03:02:06 -0700 Subject: [PATCH 31/35] Subscribed a run's channels on the completion that ends the task. The server-bound subscribe command ended the task that opened the first reader, so that task could never be retained or parked. Lang now reports the channels it listens on with `WorkflowStreamChannels`, and Core issues the subscribe for a new channel after the marker on the completion that ends the task, never a run-ending one, with the same deferral on replay so the recorded event matches. --- .../workflow_commands/workflow_commands.proto | 14 + crates/protos/src/protos/mod.rs | 10 + .../src/core_tests/external_streams.rs | 559 +++++++++++++++++- .../src/worker/workflow/external_streams.rs | 82 +++ .../workflow/machines/workflow_machines.rs | 22 +- .../src/worker/workflow/managed_run.rs | 66 ++- crates/sdk-core/src/worker/workflow/mod.rs | 5 + 7 files changed, 749 insertions(+), 9 deletions(-) diff --git a/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto b/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto index 4cb033695..93affaa4e 100644 --- a/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto +++ b/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto @@ -56,9 +56,23 @@ message WorkflowCommand { WorkflowOutputStreamCommit workflow_output_stream_commit = 27; WorkflowOutputStreamBuffered workflow_output_stream_buffered = 28; SubscribeNotificationChannel subscribe_notification_channel = 31; + WorkflowStreamChannels workflow_stream_channels = 33; } } +// The complete set of notification channels the run listens on, one per open external stream +// reader, in a stable order replay reproduces. +// +// Core subscribes the run to a channel it has not yet subscribed on the completion that ends the +// Workflow Task, after the progress marker and never on a run-ending completion. Issued from the +// activation that opened the reader, the server-bound subscribe command would end a task that +// was meant to stay retained. Reported on every completion once a reader has been opened, the +// empty set included: an absent command leaves Core's view unchanged, which is what the park and +// finalization answers rely on. Carries no retention request of its own. +message WorkflowStreamChannels { + repeated string channels = 1; +} + // Subscribe this workflow to a notification channel, so the scheduled event of // each later Workflow Task carries the notifications folded for it. // diff --git a/crates/protos/src/protos/mod.rs b/crates/protos/src/protos/mod.rs index cb5392bf6..1c77e84ed 100644 --- a/crates/protos/src/protos/mod.rs +++ b/crates/protos/src/protos/mod.rs @@ -1811,6 +1811,16 @@ pub mod coresdk { } } + impl Display for WorkflowStreamChannels { + fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { + write!( + f, + "WorkflowStreamChannels({} channel(s))", + self.channels.len() + ) + } + } + impl Display for WorkflowOutputStreamBuffered { fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { write!( diff --git a/crates/sdk-core/src/core_tests/external_streams.rs b/crates/sdk-core/src/core_tests/external_streams.rs index a88be1727..4a96d74c0 100644 --- a/crates/sdk-core/src/core_tests/external_streams.rs +++ b/crates/sdk-core/src/core_tests/external_streams.rs @@ -41,16 +41,16 @@ use temporalio_common::{ ContinueAsNewWorkflowExecution, ExternalStreamFinalized, ExternalStreamParkResult, ExternalStreamWait, FailWorkflowExecution, ParkSetConfirmed, ScheduleActivity, StreamSetBecameReady, UpdateResponse, WorkflowOutputStreamBuffered, - WorkflowOutputStreamCommit, WorkflowStreamProgress, WorkflowStreamQuiescent, - external_stream_park_result, update_response::Response as UpdateOutcome, - workflow_command, + WorkflowOutputStreamCommit, WorkflowStreamChannels, WorkflowStreamProgress, + WorkflowStreamQuiescent, external_stream_park_result, + update_response::Response as UpdateOutcome, workflow_command, }, workflow_completion::WorkflowActivationCompletion, }, temporal::api::{ - command::v1::command, + command::v1::{Command, command}, common::v1::Payload, - enums::v1::{CommandType, EventType}, + enums::v1::{CommandType, EventType, WorkflowTaskFailedCause}, history::v1::History, notification::v1::Notification, query::v1::WorkflowQuery, @@ -3717,6 +3717,555 @@ async fn replayed_notifications_resume_the_reconstructed_waits() { worker.drain_pollers_and_shutdown().await; } +// --- channel subscriptions ride the completion that ends the task ------------ + +/// Lang's report of the complete set of channels the run listens on. +fn channels_command(channels: &[&str]) -> workflow_command::Variant { + workflow_command::Variant::WorkflowStreamChannels(WorkflowStreamChannels { + channels: channels.iter().map(|c| c.to_string()).collect(), + }) +} + +/// The commands of every completion a worker reported, one entry per completion. +type ReportedCompletions = Arc>>>; + +/// A workflow-only worker recording the commands of every completion it reports. +fn worker_recording_commands( + reported: ReportedCompletions, + history: TestHistoryBuilder, + batches: Vec, +) -> crate::Worker { + let mut mock_cfg = + MockPollCfg::from_resp_batches("fakeid", history, batches, mock_worker_client()); + mock_cfg.completion_asserts_from_expectations(|mut asserts| { + for _ in 0..4 { + let reported = reported.clone(); + asserts.then(move |wft| reported.lock().push(wft.commands.clone())); + } + }); + let mut mock = build_mock_pollers(mock_cfg); + mock.worker_cfg(|w| { + w.task_types = WorkerTaskTypes::workflow_only(); + w.max_cached_workflows = 1; + }); + mock_worker(mock) +} + +fn command_types(commands: &[Command]) -> Vec { + commands.iter().map(|c| c.command_type()).collect() +} + +/// The channels the subscribe commands in a completion named, in order. +fn subscribed_channels(commands: &[Command]) -> Vec { + commands + .iter() + .filter_map(|c| match &c.attributes { + Some(command::Attributes::SubscribeNotificationChannelCommandAttributes(a)) => { + Some(a.channel.clone()) + } + _ => None, + }) + .collect() +} + +fn fires_a_timer(activation: &WorkflowActivation) -> bool { + activation.jobs.iter().any(|j| { + matches!( + j.variant, + Some(workflow_activation_job::Variant::FireTimer(_)) + ) + }) +} + +/// A history whose first task recorded a marker with `terminal` and a subscription, and whose +/// second carries a Signal, which is there to give the replacement task an activation. +fn marker_subscribed_then_signal_history( + terminal: ParkReason, + channel: &str, +) -> TestHistoryBuilder { + let mut t = TestHistoryBuilder::default(); + t.add_wfe_started_with_wft_timeout(Duration::from_secs(300)); + t.add_full_wf_task(); + t.add_external_stream_marker(1, terminal, b"recorded"); + t.add_notification_channel_subscribed(channel); + t.add_we_signaled("keep-the-run-cached", vec![]); + t.add_workflow_task_scheduled_and_started(); + t +} + +/// A history whose first task subscribed a channel and started a timer, and whose second task is +/// that timer firing. +fn subscribed_then_timer_history(channel: &str) -> TestHistoryBuilder { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_full_wf_task(); + t.add_notification_channel_subscribed(channel); + let timer_started = t.add_by_type(EventType::TimerStarted); + t.add_timer_fired(timer_started, "1".to_string()); + t.add_workflow_task_scheduled_and_started(); + t +} + +/// Opens a reader on `channel` and leaves the task retained on wait 1. +async fn retain_a_task_listening_on( + worker: &crate::Worker, + channel: &str, + idle_timeout: Duration, +) -> String { + let activation = worker.poll_workflow_activation().await.unwrap(); + let run_id = activation.run_id.clone(); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + run_id.clone(), + vec![ + progress_command(b"observed", false), + channels_command(&[channel]), + quiescent_command(1, &[1], idle_timeout), + ], + )) + .await + .unwrap(); + assert_eq!( + worker.external_stream_run_status(&run_id).await, + ExternalStreamRunStatus::WftOpen, + "the activation that opened the reader must leave its task retained" + ); + run_id +} + +#[tokio::test] +async fn a_retained_task_subscribes_when_the_park_ends_it() { + // The subscribe command is server-bound, so issued from the activation that opened the first + // reader it would end the very task that was meant to stay retained. Held back to the + // completion that ends the task, the task stays retained and parks as before, and the command + // rides the park confirmation behind the marker. + let reported: ReportedCompletions = Default::default(); + let worker = worker_recording_commands( + reported.clone(), + marker_subscribed_then_signal_history(ParkReason::Idle, "inputs"), + vec![1, 2], + ); + + let run_id = retain_a_task_listening_on(&worker, "inputs", Duration::from_millis(30)).await; + assert!( + reported.lock().is_empty(), + "nothing may be reported while the task is retained, got {:?}", + reported.lock() + ); + + let park = worker.poll_workflow_activation().await.unwrap(); + assert_eq!( + park_jobs(&park), + vec![(1, ParkReason::Idle, vec![1])], + "the idle timer must still park the task, got {:?}", + park.jobs + ); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmd( + run_id.clone(), + park_confirmed_command(1, b"-parked"), + )) + .await + .unwrap(); + + let completions = reported.lock().clone(); + assert_eq!(completions.len(), 1, "the park must report the task once"); + assert_eq!( + command_types(&completions[0]), + vec![ + CommandType::RecordMarker, + CommandType::SubscribeNotificationChannel + ], + "the subscribe command rides the park confirmation, after the marker" + ); + assert_eq!(subscribed_channels(&completions[0]), vec!["inputs"]); + + // The replacement task matches the marker and the subscribed event against the two commands. + let replacement = worker.poll_workflow_activation().await.unwrap(); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + replacement.run_id, + vec![CompleteWorkflowExecution::default().into()], + )) + .await + .unwrap(); + worker.drain_pollers_and_shutdown().await; +} + +#[tokio::test] +async fn a_retained_task_subscribes_when_a_finalization_ends_it() { + // The boundary Core decided: the rollover deadline asks lang for a terminal, and the + // completion that answers is the one that writes the marker and reports the task. The + // subscribe command rides that one, not the finalization request. + let reported: ReportedCompletions = Default::default(); + let worker = worker_recording_commands( + reported.clone(), + marker_subscribed_then_signal_history(ParkReason::Rollover, "inputs"), + vec![1, 2], + ); + + let run_id = retain_a_task_listening_on(&worker, "inputs", Duration::from_secs(30)).await; + worker + .start_wft_rollover_timer(&run_id, Duration::from_millis(50)) + .await; + + let finalize = worker.poll_workflow_activation().await.unwrap(); + assert_eq!( + finalization_jobs(&finalize), + vec![(1, ParkReason::Rollover, vec![1])], + "the deadline must ask lang to finalize, got {:?}", + finalize.jobs + ); + assert!( + reported.lock().is_empty(), + "nothing may be reported before the terminal exists, got {:?}", + reported.lock() + ); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmd( + run_id.clone(), + finalized_command(1, b"-terminal"), + )) + .await + .unwrap(); + + let completions = reported.lock().clone(); + assert_eq!(completions.len(), 1); + assert_eq!( + command_types(&completions[0]), + vec![ + CommandType::RecordMarker, + CommandType::SubscribeNotificationChannel + ], + "the subscribe command rides the finalization answer, after the marker" + ); + assert_eq!(subscribed_channels(&completions[0]), vec!["inputs"]); + + // The wait set survived onto the replacement task, so the same channel is due nothing more. + let replacement = worker.poll_workflow_activation().await.unwrap(); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + replacement.run_id, + vec![ + channels_command(&["inputs"]), + CompleteWorkflowExecution::default().into(), + ], + )) + .await + .unwrap(); + worker.drain_pollers_and_shutdown().await; +} + +#[tokio::test] +async fn a_command_carrying_completion_subscribes_after_its_marker() { + // The other way a task ends while the run goes on. The subscribe command sits between the + // marker and lang's own command: on replay the lookahead claims the marker, the subscribed + // event is matched next and the timer last, so the order is fixed here. + let reported: ReportedCompletions = Default::default(); + let worker = worker_recording_commands( + reported.clone(), + canned_histories::single_timer("1"), + vec![1], + ); + + let activation = worker.poll_workflow_activation().await.unwrap(); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + activation.run_id, + vec![ + progress_command(b"consumed", false), + channels_command(&["inputs"]), + start_timer_cmd(1, Duration::from_secs(10)), + ], + )) + .await + .unwrap(); + worker.drain_pollers_and_shutdown().await; + + let completions = reported.lock().clone(); + assert_eq!(completions.len(), 1); + assert_eq!( + command_types(&completions[0]), + vec![ + CommandType::RecordMarker, + CommandType::SubscribeNotificationChannel, + CommandType::StartTimer + ] + ); + assert_eq!(subscribed_channels(&completions[0]), vec!["inputs"]); +} + +#[tokio::test] +async fn a_run_ending_completion_subscribes_to_nothing() { + // A subscription ends with the run, so one made on the completion that ends the run would + // only add an event. Continue-as-new ends the run too: the successor subscribes for itself. + let reported: ReportedCompletions = Default::default(); + let worker = worker_recording_commands( + reported.clone(), + canned_histories::single_timer("1"), + vec![1], + ); + + let activation = worker.poll_workflow_activation().await.unwrap(); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + activation.run_id, + vec![ + progress_command(b"consumed", false), + channels_command(&["inputs"]), + CompleteWorkflowExecution::default().into(), + ], + )) + .await + .unwrap(); + worker.drain_pollers_and_shutdown().await; + + let completions = reported.lock().clone(); + assert_eq!(completions.len(), 1); + assert_eq!( + command_types(&completions[0]), + vec![ + CommandType::RecordMarker, + CommandType::CompleteWorkflowExecution + ] + ); +} + +#[tokio::test] +async fn a_later_task_naming_the_same_channel_subscribes_again_to_nothing() { + // The subscription belongs to the run, not to the task. Every completion reports the + // complete set, and a channel already subscribed is not subscribed twice. + let reported: ReportedCompletions = Default::default(); + let worker = worker_recording_commands( + reported.clone(), + subscribed_then_timer_history("inputs"), + vec![1, 2], + ); + + let first = worker.poll_workflow_activation().await.unwrap(); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + first.run_id, + vec![ + channels_command(&["inputs"]), + start_timer_cmd(1, Duration::from_secs(10)), + ], + )) + .await + .unwrap(); + + let fired = worker.poll_workflow_activation().await.unwrap(); + assert!( + fires_a_timer(&fired), + "the timer must match its event behind the subscribed one, got {:?}", + fired.jobs + ); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + fired.run_id, + vec![ + channels_command(&["inputs"]), + start_timer_cmd(2, Duration::from_secs(10)), + ], + )) + .await + .unwrap(); + worker.drain_pollers_and_shutdown().await; + + let completions = reported.lock().clone(); + assert_eq!(completions.len(), 2); + assert_eq!(subscribed_channels(&completions[0]), vec!["inputs"]); + assert_eq!( + command_types(&completions[1]), + vec![CommandType::StartTimer], + "the second task adds nothing for a channel the run already listens on" + ); +} + +/// The History a leaving task writes: the marker, the subscription, then lang's own command. +fn replayable_subscribed_history(channel: &str) -> TestHistoryBuilder { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_full_wf_task(); + t.add_external_stream_marker_covering( + 1, + ParkReason::CommandsProduced, + b"header.segment.terminal", + &[(1, 0)], + ); + t.add_notification_channel_subscribed(channel); + let timer_started = t.add_by_type(EventType::TimerStarted); + t.add_timer_fired(timer_started, "1".to_string()); + t.add_full_wf_task(); + t.add_workflow_execution_completed(); + t +} + +#[tokio::test] +async fn replay_reissues_the_subscription_from_the_same_report() { + // The replayed completion reports the channel set the live one did, so Core issues the same + // subscribe command in the same place and the recorded event matches it. The lookahead has + // already claimed the marker, so the subscribed event is the first the command queue sees. + let worker = crate::init_replay_worker(crate::replay::ReplayWorkerInput::new( + crate::test_help::test_worker_cfg().build().unwrap(), + futures_util::stream::iter([crate::replay::HistoryForReplay::from( + replayable_subscribed_history("inputs"), + )]), + )) + .unwrap(); + + let replayed = worker.poll_workflow_activation().await.unwrap(); + assert!(replayed.is_replaying); + assert_eq!( + replay_jobs(&replayed).len(), + 1, + "the recorded boundary must reach lang, got {:?}", + replayed.jobs + ); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + replayed.run_id, + vec![ + channels_command(&["inputs"]), + start_timer_cmd(1, Duration::from_secs(3)), + ], + )) + .await + .unwrap(); + + let fired = worker.poll_workflow_activation().await.unwrap(); + assert!( + fires_a_timer(&fired), + "the timer behind the subscribed event must still match, got {:?}", + fired.jobs + ); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + fired.run_id, + vec![CompleteWorkflowExecution::default().into()], + )) + .await + .unwrap(); + assert!( + matches!( + worker.poll_workflow_activation().await, + Err(PollError::ShutDown) + ), + "replay must run the history to its end" + ); +} + +#[tokio::test] +async fn a_replayed_report_naming_a_different_channel_is_nondeterminism() { + // The machine holds the recorded event against the channel the command was issued for, as it + // does for a lang-issued command. One batch, so the first task is replayed rather than run. + let mut mock_cfg = MockPollCfg::from_resp_batches( + "fakeid", + subscribed_then_timer_history("inputs"), + [2], + mock_worker_client(), + ); + mock_cfg.num_expected_fails = 1; + let saw_the_right_failure = Arc::new(AtomicBool::new(false)); + let recorder = saw_the_right_failure.clone(); + mock_cfg.expect_fail_wft_matcher = Box::new(move |_, cause, failure| { + let message = failure + .as_ref() + .map(|f| f.message.clone()) + .unwrap_or_default(); + recorder.store( + message.contains("does not match the reissued subscription") + && matches!(cause, WorkflowTaskFailedCause::NonDeterministicError), + Ordering::Relaxed, + ); + true + }); + let mut mock = build_mock_pollers(mock_cfg); + mock.worker_cfg(|w| { + w.task_types = WorkerTaskTypes::workflow_only(); + w.max_cached_workflows = 1; + }); + let worker = mock_worker(mock); + + let replayed = worker.poll_workflow_activation().await.unwrap(); + assert!(replayed.is_replaying, "got {replayed:?}"); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + replayed.run_id, + vec![ + channels_command(&["other"]), + start_timer_cmd(1, Duration::from_secs(10)), + ], + )) + .await + .unwrap(); + + let next = tokio::time::timeout( + Duration::from_millis(500), + worker.poll_workflow_activation(), + ) + .await; + if let Ok(Ok(act)) = next { + assert!( + act.is_only_eviction(), + "the mismatch must fail the task rather than activate lang, got {:?}", + act.jobs + ); + worker + .complete_workflow_activation(WorkflowActivationCompletion::empty(act.run_id)) + .await + .unwrap(); + } + assert!( + saw_the_right_failure.load(Ordering::Relaxed), + "the mismatch must fail the task as nondeterminism naming both channels" + ); + + worker.shutdown().await; + worker.finalize_shutdown().await; +} + +#[rstest::rstest] +#[case::an_empty_name(vec![""])] +#[case::a_channel_named_twice(vec!["inputs", "inputs"])] +#[tokio::test] +async fn a_malformed_channel_report_fails_the_task(#[case] channels: Vec<&str>) { + // Rejected rather than repaired, as a malformed snapshot is: the set is what the subscribe + // commands are derived from. The mock's `num_expected_fails` is what asserts the failure + // reached the server. + let t = canned_histories::single_timer("1"); + let mut mock_cfg = MockPollCfg::from_resp_batches("fakeid", t, [1], mock_worker_client()); + mock_cfg.num_expected_fails = 1; + let mut mock = build_mock_pollers(mock_cfg); + mock.worker_cfg(|w| { + w.task_types = WorkerTaskTypes::workflow_only(); + w.max_cached_workflows = 1; + }); + let worker = mock_worker(mock); + + let activation = worker.poll_workflow_activation().await.unwrap(); + let run_id = activation.run_id.clone(); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + run_id.clone(), + vec![ + channels_command(&channels), + quiescent_command(1, &[1], Duration::from_secs(30)), + ], + )) + .await + .unwrap(); + + assert_ne!( + worker.external_stream_run_status(&run_id).await, + ExternalStreamRunStatus::WftOpen, + "a malformed channel report must not retain the Workflow Task" + ); + + worker.shutdown().await; + worker.finalize_shutdown().await; +} + // --- marker emission (C9, C14b) ---------------------------------------------- /// A history for a Workflow Task that records a marker and then starts a timer. diff --git a/crates/sdk-core/src/worker/workflow/external_streams.rs b/crates/sdk-core/src/worker/workflow/external_streams.rs index 96e176068..81953c9ad 100644 --- a/crates/sdk-core/src/worker/workflow/external_streams.rs +++ b/crates/sdk-core/src/worker/workflow/external_streams.rs @@ -582,12 +582,94 @@ impl ExternalWaitSet { } } +/// The notification channels a Run listens on, held against the ones Core has subscribed it to. +/// +/// Lang reports the complete set on its completions, and Core turns the difference into +/// subscribe commands on the completion that ends the Workflow Task. Both halves are per-Worker +/// runtime state: a replayed Run rebuilds them from the same reports and reissues the same +/// commands, which is what the recorded events are matched against. +#[derive(Debug, Default)] +pub(crate) struct ChannelSubscriptions { + /// The set lang reported last, in its order. + listened: Vec, + /// Channels a subscribe command went out for, in the order they went out. + subscribed: Vec, +} + +impl ChannelSubscriptions { + /// Replaces the listened set with lang's report. + pub(crate) fn report(&mut self, channels: Vec) { + self.listened = channels; + } + + /// The channels listened on but not yet subscribed, in report order, now counted as + /// subscribed. + /// + /// Only the latest report counts. A channel that was listened on and dropped between two + /// reports of one task was never subscribed, so a replay that sees only the task's final + /// report reissues exactly what the live run sent. + pub(crate) fn take_due_subscriptions(&mut self) -> Vec { + let due: Vec = self + .listened + .iter() + .filter(|channel| !self.subscribed.contains(channel)) + .cloned() + .collect(); + self.subscribed.extend(due.iter().cloned()); + due + } +} + #[cfg(test)] mod tests { use super::*; const IDLE: Duration = Duration::from_secs(1); + fn names(channels: &[&str]) -> Vec { + channels.iter().map(|c| c.to_string()).collect() + } + + #[test] + fn a_reported_channel_is_due_once() { + let mut subscriptions = ChannelSubscriptions::default(); + subscriptions.report(names(&["inputs"])); + + assert_eq!(subscriptions.take_due_subscriptions(), names(&["inputs"])); + assert_eq!(subscriptions.take_due_subscriptions(), Vec::::new()); + + // The same set reported on a later task adds nothing. + subscriptions.report(names(&["inputs"])); + assert_eq!(subscriptions.take_due_subscriptions(), Vec::::new()); + } + + #[test] + fn only_the_channels_new_to_the_run_are_due() { + let mut subscriptions = ChannelSubscriptions::default(); + subscriptions.report(names(&["inputs"])); + subscriptions.take_due_subscriptions(); + + subscriptions.report(names(&["inputs", "control"])); + assert_eq!(subscriptions.take_due_subscriptions(), names(&["control"])); + } + + #[test] + fn only_the_latest_report_counts() { + // Two reports inside one retained task: the first channel was dropped before the task + // ended, so it was never subscribed, and replay, which sees only the final report, agrees. + let mut subscriptions = ChannelSubscriptions::default(); + subscriptions.report(names(&["inputs"])); + subscriptions.report(names(&["control"])); + + assert_eq!(subscriptions.take_due_subscriptions(), names(&["control"])); + } + + #[test] + fn nothing_reported_means_nothing_due() { + let mut subscriptions = ChannelSubscriptions::default(); + assert_eq!(subscriptions.take_due_subscriptions(), Vec::::new()); + } + fn quiescent_set(wait_ids: &[u32]) -> ExternalWaitSet { let mut set = ExternalWaitSet::new(); set.become_quiescent( diff --git a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs index 1329d86ba..d7f12a0cd 100644 --- a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs +++ b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs @@ -67,7 +67,9 @@ use temporalio_common::{ self, NotificationsReceived, NotifyHasPatch, ReplayExternalStreams, UpdateRandomSeed, WorkflowActivation, workflow_activation_job, }, - workflow_commands::{ContinueAsNewWorkflowExecution, ExternalStreamWait}, + workflow_commands::{ + ContinueAsNewWorkflowExecution, ExternalStreamWait, SubscribeNotificationChannel, + }, }, temporal::api::{ command::v1::{ @@ -551,6 +553,21 @@ impl WorkflowMachines { self.prepare_commands() } + /// Subscribes the run to a notification channel on Core's own initiative. + /// + /// The same machine the lang-issued command gets, so the recorded event is matched the same + /// way on replay. Unlike the marker this is issued while replaying too: the event is matched + /// against a command in the queue rather than claimed by a lookahead, and the replayed + /// completion reports the same channel set the live one did. + pub(crate) fn emit_notification_channel_subscription(&mut self, channel: String) -> Result<()> { + self.add_cmd_to_wf_task( + subscribe_notification_channel(SubscribeNotificationChannel { channel }), + None, + CommandIdKind::CoreInternal, + ); + self.prepare_commands() + } + /// Hands lang a marker the replay lookahead found, and creates the machine that settles it. /// /// Exactly one `ReplayExternalStreams` job per marker. Core is annotation-blind, so it copies @@ -1804,7 +1821,8 @@ impl WorkflowMachines { | WFCommandVariant::ExternalStreamParkResult(_) | WFCommandVariant::ExternalStreamFinalized(_) | WFCommandVariant::ExternalOutputStreamCommit(_) - | WFCommandVariant::ExternalOutputStreamBuffered(_) => { + | WFCommandVariant::ExternalOutputStreamBuffered(_) + | WFCommandVariant::ExternalStreamChannels(_) => { return Err(fatal!( "External stream command {} reached the state machines; it should have \ been consumed by the run's external wait set", diff --git a/crates/sdk-core/src/worker/workflow/managed_run.rs b/crates/sdk-core/src/worker/workflow/managed_run.rs index a8c937524..cc013fdf7 100644 --- a/crates/sdk-core/src/worker/workflow/managed_run.rs +++ b/crates/sdk-core/src/worker/workflow/managed_run.rs @@ -15,8 +15,9 @@ use crate::{ ServerCommandsWithWorkflowInfo, TaskStorageMetrics, WFCommand, WFCommandVariant, WFMachinesError, WFT_HEARTBEAT_TIMEOUT_FRACTION, WFTReportStatus, WorkflowTaskInfo, external_streams::{ - ExternalStreamReadyResult, ExternalStreamRunStatus, ExternalWaitSet, - ExternalWaitState, ParkResolution, ParkStartOutcome, ParkTrigger, ReadinessOutcome, + ChannelSubscriptions, ExternalStreamReadyResult, ExternalStreamRunStatus, + ExternalWaitSet, ExternalWaitState, ParkResolution, ParkStartOutcome, ParkTrigger, + ReadinessOutcome, }, history_update::HistoryPaginator, machines::{MachinesWFTResponseContent, WorkflowMachines}, @@ -966,6 +967,11 @@ impl ManagedRun { }); } }; + if let Some(channels) = stream_commands.channels.take() { + self.waiting_on_local_work + .channel_subscriptions + .report(channels); + } let output_was_buffered = self.waiting_on_local_work.output_buffered; if let Some(commit) = stream_commands.output_commit.take() { if self.waiting_on_local_work.pending_output_commit.is_some() { @@ -1435,6 +1441,34 @@ impl ManagedRun { self.cancel_external_output_flush_timer(); } + // The subscribe command is server-bound, so issued from the activation that opened the + // reader it would end a task that was meant to stay retained. It waits for the completion + // that ends the task instead: a terminal here is exactly that completion, the park + // confirmation and the finalization answer included, and a run-ending one subscribes to + // nothing because the subscription would end with the run anyway. The marker is already + // queued, so the subscribed events follow it in History, and lang's own commands follow + // them. + if let Some(terminal) = terminal + && terminal != ParkReason::WorkflowCompleted + { + let due = self + .waiting_on_local_work + .channel_subscriptions + .take_due_subscriptions(); + for channel in due { + if let Err(source) = self + .wfm + .machines + .emit_notification_channel_subscription(channel) + { + return Err(RunUpdateErr { + source, + complete_resp_chan: completion.resp_chan, + }); + } + } + } + let outcome = (|| { // Send commands from lang into the machines then check if the workflow run needs // another activation and mark it if so @@ -2927,6 +2961,8 @@ struct ExternalStreamCommands { finalized: Option, output_commit: Option, output_buffered_latency: Option, + /// The complete set of notification channels the run listens on, when lang reported it. + channels: Option>, } /// Splits lang's commands, leaving everything else in `commands`. @@ -3005,6 +3041,30 @@ fn take_external_stream_commands( }), ); } + WFCommandVariant::ExternalStreamChannels(report) => { + // Rejected rather than repaired: the set is what the subscribe commands are + // derived from, and a name the server would refuse or a channel named twice is + // a lang bug that replay would otherwise reproduce faithfully. + let mut seen = HashSet::with_capacity(report.channels.len()); + for channel in &report.channels { + if channel.is_empty() { + return Err(WFMachinesError::Fatal( + "WorkflowStreamChannels named an empty channel".to_string(), + )); + } + if !seen.insert(channel.as_str()) { + return Err(WFMachinesError::Fatal(format!( + "WorkflowStreamChannels named channel {channel:?} twice" + ))); + } + } + if taken.channels.replace(report.channels).is_some() { + return Err(WFMachinesError::Fatal( + "Lang sent more than one WorkflowStreamChannels in one completion" + .to_string(), + )); + } + } _ => { seen_other_command = true; remaining.push(command); @@ -3057,6 +3117,8 @@ struct WaitingOnLocalWork { local_activities: Option, /// This run's external stream waits. Empty until lang reports quiescence. external_wait_set: ExternalWaitSet, + /// The notification channels this run listens on, against the ones it is subscribed to. + channel_subscriptions: ChannelSubscriptions, /// Cancels the run-level workflow task rollover deadline, when one is running. /// /// Separate from the local-activity heartbeat handle: a retained task needs a rollover diff --git a/crates/sdk-core/src/worker/workflow/mod.rs b/crates/sdk-core/src/worker/workflow/mod.rs index 77f9ce1e8..b707e27c8 100644 --- a/crates/sdk-core/src/worker/workflow/mod.rs +++ b/crates/sdk-core/src/worker/workflow/mod.rs @@ -1728,6 +1728,8 @@ enum WFCommandVariant { ExternalOutputStreamCommit(WorkflowOutputStreamCommit), /// External output is buffered in lang and needs a run-scoped flush deadline. ExternalOutputStreamBuffered(WorkflowOutputStreamBuffered), + /// The complete set of notification channels the run listens on. Never implies retention. + ExternalStreamChannels(WorkflowStreamChannels), SubscribeNotificationChannel(SubscribeNotificationChannel), } @@ -1808,6 +1810,9 @@ impl TryFrom for WFCommand { workflow_command::Variant::WorkflowOutputStreamBuffered(buffered) => { WFCommandVariant::ExternalOutputStreamBuffered(buffered) } + workflow_command::Variant::WorkflowStreamChannels(channels) => { + WFCommandVariant::ExternalStreamChannels(channels) + } workflow_command::Variant::SubscribeNotificationChannel(s) => { WFCommandVariant::SubscribeNotificationChannel(s) } From 0df91dd72857d1d279f379e9622fc77a2c8b2ea1 Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 2 Oct 2026 03:14:32 -0700 Subject: [PATCH 32/35] Carried the unsubscribe and describe protos in the Core's api tree. The command, its event and failed cause, and the workflow's channel subscriptions on DescribeWorkflowExecution, as the api fork defines them. --- .../temporal/api/command/v1/message.proto | 11 ++++++++ .../temporal/api/enums/v1/command_type.proto | 1 + .../temporal/api/enums/v1/event_type.proto | 4 +++ .../temporal/api/enums/v1/failed_cause.proto | 2 ++ .../temporal/api/history/v1/message.proto | 14 ++++++++++ .../temporal/api/workflow/v1/message.proto | 27 +++++++++++++++++++ .../workflowservice/v1/request_response.proto | 3 +++ 7 files changed, 62 insertions(+) diff --git a/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto index f75b5a950..c8c6859ea 100644 --- a/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/command/v1/message.proto @@ -329,6 +329,8 @@ message Command { SubscribeStreamCommandAttributes subscribe_stream_command_attributes = 21; SubscribeNotificationChannelCommandAttributes subscribe_notification_channel_command_attributes = 22; + UnsubscribeNotificationChannelCommandAttributes + unsubscribe_notification_channel_command_attributes = 23; } } @@ -370,3 +372,12 @@ message SubscribeNotificationChannelCommandAttributes { // The channel to listen on, as the writers name it. string channel = 1; } + +// Ends the run's subscription to a notification channel. Notifications already +// recorded on a scheduled event still reach that Workflow Task; later ones do +// not. A command naming a channel the run is not subscribed to records its +// event and changes nothing, so replay matches every command to an event. +message UnsubscribeNotificationChannelCommandAttributes { + // The channel to stop listening on, as the writers name it. + string channel = 1; +} diff --git a/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto b/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto index b8eada084..629a26528 100644 --- a/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto +++ b/crates/protos/protos/api_upstream/temporal/api/enums/v1/command_type.proto @@ -32,4 +32,5 @@ enum CommandType { COMMAND_TYPE_APPEND_STREAM_RECORDS = 19; COMMAND_TYPE_SUBSCRIBE_STREAM = 20; COMMAND_TYPE_SUBSCRIBE_NOTIFICATION_CHANNEL = 21; + COMMAND_TYPE_UNSUBSCRIBE_NOTIFICATION_CHANNEL = 22; } diff --git a/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto b/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto index f70de73fe..815e1aebe 100644 --- a/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto +++ b/crates/protos/protos/api_upstream/temporal/api/enums/v1/event_type.proto @@ -186,4 +186,8 @@ enum EventType { // A Workflow became a listener of a notification channel for its run. // The notifications themselves ride the WorkflowTaskScheduled event. EVENT_TYPE_WORKFLOW_NOTIFICATION_CHANNEL_SUBSCRIBED = 63; + // A Workflow stopped listening on a notification channel for its run. + // Recorded for every UnsubscribeNotificationChannel command, including one + // naming a channel the run was not subscribed to. + EVENT_TYPE_WORKFLOW_NOTIFICATION_CHANNEL_UNSUBSCRIBED = 64; } diff --git a/crates/protos/protos/api_upstream/temporal/api/enums/v1/failed_cause.proto b/crates/protos/protos/api_upstream/temporal/api/enums/v1/failed_cause.proto index b920b0fe4..65987419a 100644 --- a/crates/protos/protos/api_upstream/temporal/api/enums/v1/failed_cause.proto +++ b/crates/protos/protos/api_upstream/temporal/api/enums/v1/failed_cause.proto @@ -101,6 +101,8 @@ enum WorkflowTaskFailedCause { // A SubscribeNotificationChannel command named an empty or too-long channel, or hit a // subscription or listener limit. WORKFLOW_TASK_FAILED_CAUSE_BAD_SUBSCRIBE_NOTIFICATION_CHANNEL_ATTRIBUTES = 44; + // An UnsubscribeNotificationChannel command named an empty or too-long channel. + WORKFLOW_TASK_FAILED_CAUSE_BAD_UNSUBSCRIBE_NOTIFICATION_CHANNEL_ATTRIBUTES = 45; } enum StartChildWorkflowExecutionFailedCause { diff --git a/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto index c76f80756..bbb4ff244 100644 --- a/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/history/v1/message.proto @@ -987,6 +987,18 @@ message WorkflowNotificationChannelSubscribedEventAttributes { string channel = 2; } +message WorkflowNotificationChannelUnsubscribedEventAttributes { + // The WorkflowTaskCompleted event of the task whose command ended this + // subscription. + int64 workflow_task_completed_event_id = 1; + // The channel the Workflow stopped listening on. + string channel = 2; + // The WorkflowNotificationChannelSubscribed event that recorded the + // subscription this command ended. Zero when the run held no subscription + // for the channel. + int64 subscribed_event_id = 3; +} + message WorkflowStreamRecordsAppendedEventAttributes { // The WorkflowTaskCompleted event of the task whose command appended this // batch. @@ -1328,6 +1340,8 @@ message HistoryEvent { WorkflowStreamRecordsAppendedEventAttributes workflow_stream_records_appended_event_attributes = 67; WorkflowNotificationChannelSubscribedEventAttributes workflow_notification_channel_subscribed_event_attributes = 68; + WorkflowNotificationChannelUnsubscribedEventAttributes + workflow_notification_channel_unsubscribed_event_attributes = 69; } } diff --git a/crates/protos/protos/api_upstream/temporal/api/workflow/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/workflow/v1/message.proto index 1ed33fa4c..4fea9a63e 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflow/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflow/v1/message.proto @@ -21,6 +21,7 @@ import "temporal/api/enums/v1/workflow.proto"; import "temporal/api/common/v1/message.proto"; import "temporal/api/deployment/v1/message.proto"; import "temporal/api/failure/v1/message.proto"; +import "temporal/api/notification/v1/message.proto"; import "temporal/api/taskqueue/v1/message.proto"; import "temporal/api/sdk/v1/user_metadata.proto"; @@ -582,6 +583,32 @@ message NexusOperationCancellationInfo { string blocked_reason = 7; } +// A workflow's standing on a notification channel, as reported by DescribeWorkflowExecution. +message ChannelSubscriptionInfo { + // Channel name. + string channel = 1; + // CHANNEL_KIND_INDEPENDENT for a channel the workflow subscribed to with a + // SubscribeNotificationChannel command. CHANNEL_KIND_LINKED for a channel linked to this + // workflow, which lists it once the channel holds any state. + temporal.api.notification.v1.ChannelKind kind = 2; + // Independent kind: id of the WorkflowNotificationChannelSubscribed event that recorded the + // subscription. Zero for the linked kind. + int64 subscribed_event_id = 3; + // Highest counter the workflow has accepted from the channel. Zero when none has arrived. + int64 last_counter = 4; + // The notification held for the workflow's next Workflow Task, when one is pending. + temporal.api.notification.v1.Notification pending_notification = 5; + // Counter carried by the scheduled event of a Workflow Task that has not started yet. Zero + // otherwise. + int64 scheduled_counter = 6; + // Linked kind: callback listeners registered on the channel. + int32 listener_count = 7; + // Linked kind: notifications retained for pollers. + int32 retained_count = 8; + // Linked kind: notifications the channel has accepted over its life. + int64 accepted_count = 9; +} + message WorkflowExecutionOptions { // If set, takes precedence over the Versioning Behavior sent by the SDK on Workflow Task completion. VersioningOverride versioning_override = 1; diff --git a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto index 660ebf809..37c881f24 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto @@ -1314,6 +1314,9 @@ message DescribeWorkflowExecutionResponse { repeated temporal.api.workflow.v1.CallbackInfo callbacks = 6; repeated temporal.api.workflow.v1.PendingNexusOperationInfo pending_nexus_operations = 7; temporal.api.workflow.v1.WorkflowExecutionExtendedInfo workflow_extended_info = 8; + // The notification channels this run stands on: the independent channels it subscribed to and + // the channels linked to it that hold any state. Empty when there are none. + repeated temporal.api.workflow.v1.ChannelSubscriptionInfo channel_subscriptions = 9; } // (-- api-linter: core::0203::optional=disabled From 4afe3e5d40eaee6ae44969cbfd7d0e773a62804f Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 2 Oct 2026 03:14:32 -0700 Subject: [PATCH 33/35] Unsubscribed a run from the channels its readers left. A channel that leaves the reported set goes out as `UnsubscribeNotificationChannel` on the completion that ends the task, ahead of the subscribes, and is matched against its event on replay. The lang-issued command gets the same machine. --- .../workflow_commands/workflow_commands.proto | 11 + crates/protos/src/protos/mod.rs | 26 + .../src/core_tests/external_streams.rs | 303 ++++++++++++ crates/sdk-core/src/replay/history_builder.rs | 15 + .../src/worker/workflow/external_streams.rs | 55 +++ .../src/worker/workflow/machines/mod.rs | 3 + .../workflow/machines/transition_coverage.rs | 9 +- ...ribe_notification_channel_state_machine.rs | 449 ++++++++++++++++++ .../workflow/machines/workflow_machines.rs | 26 +- .../src/worker/workflow/managed_run.rs | 37 +- crates/sdk-core/src/worker/workflow/mod.rs | 4 + 11 files changed, 919 insertions(+), 19 deletions(-) create mode 100644 crates/sdk-core/src/worker/workflow/machines/unsubscribe_notification_channel_state_machine.rs diff --git a/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto b/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto index 93affaa4e..cbac35fdc 100644 --- a/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto +++ b/crates/protos/protos/local/temporal/sdk/core/workflow_commands/workflow_commands.proto @@ -56,10 +56,21 @@ message WorkflowCommand { WorkflowOutputStreamCommit workflow_output_stream_commit = 27; WorkflowOutputStreamBuffered workflow_output_stream_buffered = 28; SubscribeNotificationChannel subscribe_notification_channel = 31; + UnsubscribeNotificationChannel unsubscribe_notification_channel = 32; WorkflowStreamChannels workflow_stream_channels = 33; } } +// End this workflow's subscription to a notification channel. Notifications +// already recorded on a scheduled event still reach that Workflow Task. +// +// The server records the event whether or not the run held a subscription, so +// every command has an event to match on replay. +message UnsubscribeNotificationChannel { + // Name of the channel, scoped to the namespace. + string channel = 1; +} + // The complete set of notification channels the run listens on, one per open external stream // reader, in a stable order replay reproduces. // diff --git a/crates/protos/src/protos/mod.rs b/crates/protos/src/protos/mod.rs index 1c77e84ed..9149b9339 100644 --- a/crates/protos/src/protos/mod.rs +++ b/crates/protos/src/protos/mod.rs @@ -1811,6 +1811,12 @@ pub mod coresdk { } } + impl Display for UnsubscribeNotificationChannel { + fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { + write!(f, "UnsubscribeNotificationChannel({})", self.channel) + } + } + impl Display for WorkflowStreamChannels { fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result { write!( @@ -2022,6 +2028,9 @@ pub mod temporal { Attributes::SubscribeNotificationChannelCommandAttributes(_) => { CommandType::SubscribeNotificationChannel } + Attributes::UnsubscribeNotificationChannelCommandAttributes(_) => { + CommandType::UnsubscribeNotificationChannel + } Attributes::CompleteWorkflowExecutionCommandAttributes(_) => { CommandType::CompleteWorkflowExecution } @@ -2085,6 +2094,16 @@ pub mod temporal { } } + impl From for Attributes { + fn from(u: workflow_commands::UnsubscribeNotificationChannel) -> Self { + Self::UnsubscribeNotificationChannelCommandAttributes( + UnsubscribeNotificationChannelCommandAttributes { + channel: u.channel, + }, + ) + } + } + impl From for command::Attributes { fn from(s: workflow_commands::StartTimer) -> Self { Self::StartTimerCommandAttributes(StartTimerCommandAttributes { @@ -2547,6 +2566,7 @@ pub mod temporal { | EventType::WorkflowStreamSubscribed | EventType::WorkflowStreamRecordsAppended | EventType::WorkflowNotificationChannelSubscribed + | EventType::WorkflowNotificationChannelUnsubscribed | EventType::NexusOperationScheduled | EventType::NexusOperationCancelRequested | EventType::WorkflowExecutionCanceled @@ -2653,6 +2673,9 @@ pub mod temporal { Attributes::WorkflowNotificationChannelSubscribedEventAttributes(_) => { false } + Attributes::WorkflowNotificationChannelUnsubscribedEventAttributes(_) => { + false + } Attributes::WorkflowExecutionStartedEventAttributes(_) => false, Attributes::WorkflowExecutionCompletedEventAttributes(_) => false, Attributes::WorkflowExecutionFailedEventAttributes(_) => false, @@ -2743,6 +2766,9 @@ pub mod temporal { Attributes::WorkflowStreamSubscribedEventAttributes(_) => { EventType::WorkflowStreamSubscribed } Attributes::WorkflowStreamRecordsAppendedEventAttributes(_) => { EventType::WorkflowStreamRecordsAppended } Attributes::WorkflowNotificationChannelSubscribedEventAttributes(_) => { EventType::WorkflowNotificationChannelSubscribed } + Attributes::WorkflowNotificationChannelUnsubscribedEventAttributes(_) => { + EventType::WorkflowNotificationChannelUnsubscribed + } Attributes::WorkflowExecutionStartedEventAttributes(_) => { EventType::WorkflowExecutionStarted } Attributes::WorkflowExecutionCompletedEventAttributes(_) => { EventType::WorkflowExecutionCompleted } Attributes::WorkflowExecutionFailedEventAttributes(_) => { EventType::WorkflowExecutionFailed } diff --git a/crates/sdk-core/src/core_tests/external_streams.rs b/crates/sdk-core/src/core_tests/external_streams.rs index 4a96d74c0..b8cfff902 100644 --- a/crates/sdk-core/src/core_tests/external_streams.rs +++ b/crates/sdk-core/src/core_tests/external_streams.rs @@ -4225,6 +4225,309 @@ async fn a_replayed_report_naming_a_different_channel_is_nondeterminism() { worker.finalize_shutdown().await; } +/// The channels the unsubscribe commands in a completion named, in order. +fn unsubscribed_channels(commands: &[Command]) -> Vec { + commands + .iter() + .filter_map(|c| match &c.attributes { + Some(command::Attributes::UnsubscribeNotificationChannelCommandAttributes(a)) => { + Some(a.channel.clone()) + } + _ => None, + }) + .collect() +} + +/// Runs the first task of `subscribed_then_timer_history` live, subscribing `channel` and +/// starting timer 1, and returns the activation that fires it. +async fn subscribe_then_fire_the_timer( + worker: &crate::Worker, + channel: &str, +) -> WorkflowActivation { + let first = worker.poll_workflow_activation().await.unwrap(); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + first.run_id, + vec![ + channels_command(&[channel]), + start_timer_cmd(1, Duration::from_secs(10)), + ], + )) + .await + .unwrap(); + let fired = worker.poll_workflow_activation().await.unwrap(); + assert!( + fires_a_timer(&fired), + "the timer must match its event behind the subscribed one, got {:?}", + fired.jobs + ); + fired +} + +#[tokio::test] +async fn a_channel_the_run_stopped_listening_on_is_unsubscribed_when_the_task_ends() { + // The reader closed, so the report no longer names the channel. The unsubscribe rides the + // completion that ends the task ahead of lang's command, as the subscribe did. + let reported: ReportedCompletions = Default::default(); + let worker = worker_recording_commands( + reported.clone(), + subscribed_then_timer_history("inputs"), + vec![1, 2], + ); + + let fired = subscribe_then_fire_the_timer(&worker, "inputs").await; + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + fired.run_id, + vec![ + channels_command(&[]), + start_timer_cmd(2, Duration::from_secs(10)), + ], + )) + .await + .unwrap(); + worker.drain_pollers_and_shutdown().await; + + let completions = reported.lock().clone(); + assert_eq!(completions.len(), 2); + assert_eq!( + command_types(&completions[1]), + vec![ + CommandType::UnsubscribeNotificationChannel, + CommandType::StartTimer + ] + ); + assert_eq!(unsubscribed_channels(&completions[1]), vec!["inputs"]); +} + +#[tokio::test] +async fn a_channel_swap_unsubscribes_before_it_subscribes() { + // Rotating channels under a per-run subscription limit only works if the slot is freed + // before the new subscription takes one. + let reported: ReportedCompletions = Default::default(); + let worker = worker_recording_commands( + reported.clone(), + subscribed_then_timer_history("inputs"), + vec![1, 2], + ); + + let fired = subscribe_then_fire_the_timer(&worker, "inputs").await; + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + fired.run_id, + vec![ + channels_command(&["control"]), + start_timer_cmd(2, Duration::from_secs(10)), + ], + )) + .await + .unwrap(); + worker.drain_pollers_and_shutdown().await; + + let completions = reported.lock().clone(); + assert_eq!(completions.len(), 2); + assert_eq!( + command_types(&completions[1]), + vec![ + CommandType::UnsubscribeNotificationChannel, + CommandType::SubscribeNotificationChannel, + CommandType::StartTimer + ] + ); + assert_eq!(unsubscribed_channels(&completions[1]), vec!["inputs"]); + assert_eq!(subscribed_channels(&completions[1]), vec!["control"]); +} + +#[tokio::test] +async fn a_run_ending_completion_unsubscribes_from_nothing() { + // The subscription ends with the run, so the unsubscribe would only add an event. + let reported: ReportedCompletions = Default::default(); + let worker = worker_recording_commands( + reported.clone(), + subscribed_then_timer_history("inputs"), + vec![1, 2], + ); + + let fired = subscribe_then_fire_the_timer(&worker, "inputs").await; + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + fired.run_id, + vec![ + channels_command(&[]), + CompleteWorkflowExecution::default().into(), + ], + )) + .await + .unwrap(); + worker.drain_pollers_and_shutdown().await; + + let completions = reported.lock().clone(); + assert_eq!(completions.len(), 2); + assert_eq!( + command_types(&completions[1]), + vec![CommandType::CompleteWorkflowExecution] + ); +} + +/// The History two leaving tasks write: the first subscribes `channel` and starts a timer, the +/// second unsubscribes it and starts another. `complete` adds the closing task and event. +fn subscribed_then_unsubscribed_history(channel: &str, complete: bool) -> TestHistoryBuilder { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_full_wf_task(); + let subscribed = t.add_notification_channel_subscribed(channel); + let first_timer = t.add_by_type(EventType::TimerStarted); + t.add_timer_fired(first_timer, "1".to_string()); + t.add_full_wf_task(); + t.add_notification_channel_unsubscribed(channel, subscribed); + let second_timer = t.add_by_type(EventType::TimerStarted); + t.add_timer_fired(second_timer, "2".to_string()); + if complete { + t.add_full_wf_task(); + t.add_workflow_execution_completed(); + } else { + t.add_workflow_task_scheduled_and_started(); + } + t +} + +#[tokio::test] +async fn replay_reissues_the_unsubscription_from_the_same_report() { + // The replayed second task reports the empty set the live one did, so Core issues the same + // unsubscribe in the same place and the recorded event matches it ahead of the timer. + let worker = crate::init_replay_worker(crate::replay::ReplayWorkerInput::new( + crate::test_help::test_worker_cfg().build().unwrap(), + futures_util::stream::iter([crate::replay::HistoryForReplay::from( + subscribed_then_unsubscribed_history("inputs", true), + )]), + )) + .unwrap(); + + let first = worker.poll_workflow_activation().await.unwrap(); + assert!(first.is_replaying); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + first.run_id, + vec![ + channels_command(&["inputs"]), + start_timer_cmd(1, Duration::from_secs(3)), + ], + )) + .await + .unwrap(); + + let second = worker.poll_workflow_activation().await.unwrap(); + assert!(fires_a_timer(&second), "got {:?}", second.jobs); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + second.run_id, + vec![ + channels_command(&[]), + start_timer_cmd(2, Duration::from_secs(3)), + ], + )) + .await + .unwrap(); + + let third = worker.poll_workflow_activation().await.unwrap(); + assert!( + fires_a_timer(&third), + "the second timer behind the unsubscribed event must still match, got {:?}", + third.jobs + ); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + third.run_id, + vec![CompleteWorkflowExecution::default().into()], + )) + .await + .unwrap(); + assert!( + matches!( + worker.poll_workflow_activation().await, + Err(PollError::ShutDown) + ), + "replay must run the history to its end" + ); +} + +#[tokio::test] +async fn a_replayed_report_still_naming_an_unsubscribed_channel_is_nondeterminism() { + // The live run stopped listening and History says so. A replay whose report still names the + // channel issues no unsubscribe, and the recorded event then meets lang's timer instead. + let mut mock_cfg = MockPollCfg::from_resp_batches( + "fakeid", + subscribed_then_unsubscribed_history("inputs", false), + [3], + mock_worker_client(), + ); + mock_cfg.num_expected_fails = 1; + let saw_nondeterminism = Arc::new(AtomicBool::new(false)); + let recorder = saw_nondeterminism.clone(); + mock_cfg.expect_fail_wft_matcher = Box::new(move |_, cause, _| { + recorder.store( + matches!(cause, WorkflowTaskFailedCause::NonDeterministicError), + Ordering::Relaxed, + ); + true + }); + let mut mock = build_mock_pollers(mock_cfg); + mock.worker_cfg(|w| { + w.task_types = WorkerTaskTypes::workflow_only(); + w.max_cached_workflows = 1; + }); + let worker = mock_worker(mock); + + let first = worker.poll_workflow_activation().await.unwrap(); + assert!(first.is_replaying, "got {first:?}"); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + first.run_id, + vec![ + channels_command(&["inputs"]), + start_timer_cmd(1, Duration::from_secs(10)), + ], + )) + .await + .unwrap(); + let second = worker.poll_workflow_activation().await.unwrap(); + assert!(second.is_replaying, "got {second:?}"); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmds( + second.run_id, + vec![ + channels_command(&["inputs"]), + start_timer_cmd(2, Duration::from_secs(10)), + ], + )) + .await + .unwrap(); + + let next = tokio::time::timeout( + Duration::from_millis(500), + worker.poll_workflow_activation(), + ) + .await; + if let Ok(Ok(act)) = next { + assert!( + act.is_only_eviction(), + "the mismatch must fail the task rather than activate lang, got {:?}", + act.jobs + ); + worker + .complete_workflow_activation(WorkflowActivationCompletion::empty(act.run_id)) + .await + .unwrap(); + } + assert!( + saw_nondeterminism.load(Ordering::Relaxed), + "a recorded unsubscription the replay does not reissue must fail as nondeterminism" + ); + + worker.shutdown().await; + worker.finalize_shutdown().await; +} + #[rstest::rstest] #[case::an_empty_name(vec![""])] #[case::a_channel_named_twice(vec!["inputs", "inputs"])] diff --git a/crates/sdk-core/src/replay/history_builder.rs b/crates/sdk-core/src/replay/history_builder.rs index b13deae73..616eaf68d 100644 --- a/crates/sdk-core/src/replay/history_builder.rs +++ b/crates/sdk-core/src/replay/history_builder.rs @@ -136,6 +136,21 @@ impl TestHistoryBuilder { self.add(attrs) } + /// Add the event an unsubscribe-notification-channel command produces. A zero + /// `subscribed_event_id` is what the server records when the run held no subscription. + pub fn add_notification_channel_unsubscribed( + &mut self, + channel: &str, + subscribed_event_id: i64, + ) -> i64 { + let attrs = WorkflowNotificationChannelUnsubscribedEventAttributes { + workflow_task_completed_event_id: self.previous_task_completed_id, + channel: channel.to_string(), + subscribed_event_id, + }; + self.add(attrs) + } + /// Add a workflow task started event. pub fn add_workflow_task_started(&mut self) { self.final_workflow_task_started_event_id = self.add(WorkflowTaskStartedEventAttributes { diff --git a/crates/sdk-core/src/worker/workflow/external_streams.rs b/crates/sdk-core/src/worker/workflow/external_streams.rs index 81953c9ad..ac8a27e70 100644 --- a/crates/sdk-core/src/worker/workflow/external_streams.rs +++ b/crates/sdk-core/src/worker/workflow/external_streams.rs @@ -618,6 +618,16 @@ impl ChannelSubscriptions { self.subscribed.extend(due.iter().cloned()); due } + + /// The channels subscribed but no longer listened on, in subscription order, now counted as + /// unsubscribed. A channel listened on again later is due a new subscription. + pub(crate) fn take_due_unsubscriptions(&mut self) -> Vec { + let (leaving, staying): (Vec, Vec) = std::mem::take(&mut self.subscribed) + .into_iter() + .partition(|channel| !self.listened.contains(channel)); + self.subscribed = staying; + leaving + } } #[cfg(test)] @@ -668,6 +678,51 @@ mod tests { fn nothing_reported_means_nothing_due() { let mut subscriptions = ChannelSubscriptions::default(); assert_eq!(subscriptions.take_due_subscriptions(), Vec::::new()); + assert_eq!( + subscriptions.take_due_unsubscriptions(), + Vec::::new() + ); + } + + #[test] + fn a_channel_that_left_the_set_is_due_an_unsubscription_once() { + let mut subscriptions = ChannelSubscriptions::default(); + subscriptions.report(names(&["inputs", "control"])); + subscriptions.take_due_subscriptions(); + + subscriptions.report(names(&["control"])); + assert_eq!(subscriptions.take_due_unsubscriptions(), names(&["inputs"])); + assert_eq!( + subscriptions.take_due_unsubscriptions(), + Vec::::new() + ); + assert_eq!(subscriptions.take_due_subscriptions(), Vec::::new()); + } + + #[test] + fn a_channel_never_subscribed_is_not_unsubscribed_when_it_leaves() { + // Two reports inside one retained task again: the channel came and went before the task + // ended, so no subscribe went out and no unsubscribe is owed. + let mut subscriptions = ChannelSubscriptions::default(); + subscriptions.report(names(&["inputs"])); + subscriptions.report(names(&[])); + + assert_eq!( + subscriptions.take_due_unsubscriptions(), + Vec::::new() + ); + } + + #[test] + fn a_channel_listened_on_again_is_due_a_new_subscription() { + let mut subscriptions = ChannelSubscriptions::default(); + subscriptions.report(names(&["inputs"])); + subscriptions.take_due_subscriptions(); + subscriptions.report(names(&[])); + subscriptions.take_due_unsubscriptions(); + + subscriptions.report(names(&["inputs"])); + assert_eq!(subscriptions.take_due_subscriptions(), names(&["inputs"])); } fn quiescent_set(wait_ids: &[u32]) -> ExternalWaitSet { diff --git a/crates/sdk-core/src/worker/workflow/machines/mod.rs b/crates/sdk-core/src/worker/workflow/machines/mod.rs index d285c7f5b..4698cabb3 100644 --- a/crates/sdk-core/src/worker/workflow/machines/mod.rs +++ b/crates/sdk-core/src/worker/workflow/machines/mod.rs @@ -18,6 +18,7 @@ mod patch_state_machine; mod signal_external_state_machine; mod subscribe_notification_channel_state_machine; mod timer_state_machine; +mod unsubscribe_notification_channel_state_machine; mod update_state_machine; mod upsert_search_attributes_state_machine; mod workflow_task_state_machine; @@ -59,6 +60,7 @@ use temporalio_common::{ }; use temporalio_macros::fsm; use timer_state_machine::TimerMachine; +use unsubscribe_notification_channel_state_machine::UnsubscribeNotificationChannelMachine; use update_state_machine::UpdateMachine; use upsert_search_attributes_state_machine::UpsertSearchAttributesMachine; use workflow_machines::MachineResponse; @@ -88,6 +90,7 @@ enum Machines { UpdateMachine, NexusOperationMachine, SubscribeNotificationChannelMachine, + UnsubscribeNotificationChannelMachine, } /// Extends [rustfsm::StateMachine] with some functionality specific to the temporal SDK. diff --git a/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs b/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs index b55ccd159..daee6afbf 100644 --- a/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs +++ b/crates/sdk-core/src/worker/workflow/machines/transition_coverage.rs @@ -77,7 +77,9 @@ mod machine_coverage_report { modify_workflow_properties_state_machine::ModifyWorkflowPropertiesMachine, patch_state_machine::PatchMachine, signal_external_state_machine::SignalExternalMachine, subscribe_notification_channel_state_machine::SubscribeNotificationChannelMachine, - timer_state_machine::TimerMachine, update_state_machine::UpdateMachine, + timer_state_machine::TimerMachine, + unsubscribe_notification_channel_state_machine::UnsubscribeNotificationChannelMachine, + update_state_machine::UpdateMachine, upsert_search_attributes_state_machine::UpsertSearchAttributesMachine, workflow_task_state_machine::WorkflowTaskMachine, }; @@ -119,6 +121,8 @@ mod machine_coverage_report { let mut update = UpdateMachine::visualizer().to_owned(); let mut external_stream = ExternalStreamMachine::visualizer().to_owned(); let mut subscribe_channel = SubscribeNotificationChannelMachine::visualizer().to_owned(); + let mut unsubscribe_channel = + UnsubscribeNotificationChannelMachine::visualizer().to_owned(); // This isn't at all efficient but doesn't need to be. // Replace transitions in the vizzes with green color if they are covered. @@ -149,6 +153,9 @@ mod machine_coverage_report { m @ "SubscribeNotificationChannelMachine" => { cover_transitions(m, &mut subscribe_channel, coverage) } + m @ "UnsubscribeNotificationChannelMachine" => { + cover_transitions(m, &mut unsubscribe_channel, coverage) + } m => panic!("Unknown machine {m}"), } } diff --git a/crates/sdk-core/src/worker/workflow/machines/unsubscribe_notification_channel_state_machine.rs b/crates/sdk-core/src/worker/workflow/machines/unsubscribe_notification_channel_state_machine.rs new file mode 100644 index 000000000..d399759fc --- /dev/null +++ b/crates/sdk-core/src/worker/workflow/machines/unsubscribe_notification_channel_state_machine.rs @@ -0,0 +1,449 @@ +use super::{ + NewMachineWithCommand, StateMachine, TransitionResult, fsm, workflow_machines::MachineResponse, +}; +use crate::worker::workflow::{ + WFMachinesError, fatal, + machines::{EventInfo, HistEventData, WFMachinesAdapter}, + nondeterminism, +}; +use temporalio_common::protos::{ + coresdk::workflow_commands::UnsubscribeNotificationChannel, + temporal::api::{ + enums::v1::{CommandType, EventType}, + history::v1::{WorkflowNotificationChannelUnsubscribedEventAttributes, history_event}, + }, +}; + +fsm! { + pub(super) name UnsubscribeNotificationChannelMachine; + command UnsubscribeNotificationChannelMachineCommand; + error WFMachinesError; + shared_state SharedState; + + Created --(CommandScheduled) --> CommandIssued; + CommandIssued --(CommandRecorded(WorkflowNotificationChannelUnsubscribedEventAttributes), + shared on_command_recorded) --> Done; +} + +/// The channel the command named, kept so the recorded event can be held against it on replay. +#[derive(Default, Clone)] +pub(super) struct SharedState { + channel: String, +} + +/// End this run's subscription to a notification channel. The server records the event whether +/// or not the run held a subscription, so every command has an event to match on replay. +pub(super) fn unsubscribe_notification_channel( + lang_cmd: UnsubscribeNotificationChannel, +) -> NewMachineWithCommand { + let sm = UnsubscribeNotificationChannelMachine::from_parts( + Created {}.into(), + SharedState { + channel: lang_cmd.channel.clone(), + }, + ); + NewMachineWithCommand { + command: lang_cmd.into(), + machine: sm.into(), + } +} + +#[derive(Debug, derive_more::Display)] +pub(super) enum UnsubscribeNotificationChannelMachineCommand {} + +#[derive(Debug, Default, Clone, derive_more::Display)] +pub(super) struct Created {} + +#[derive(Debug, Default, Clone, derive_more::Display)] +pub(super) struct CommandIssued {} + +impl CommandIssued { + pub(super) fn on_command_recorded( + self, + dat: &mut SharedState, + attrs: WorkflowNotificationChannelUnsubscribedEventAttributes, + ) -> UnsubscribeNotificationChannelMachineTransition { + if dat.channel == attrs.channel { + TransitionResult::default() + } else { + TransitionResult::Err(nondeterminism!( + "Recorded unsubscription from notification channel {:?} does not match the \ + reissued unsubscription from channel {:?}", + attrs.channel, + dat.channel + )) + } + } +} + +#[derive(Debug, Default, Clone, derive_more::Display)] +pub(super) struct Done {} + +impl WFMachinesAdapter for UnsubscribeNotificationChannelMachine { + fn adapt_response( + &self, + _my_command: Self::Command, + _event_info: Option, + ) -> Result, Self::Error> { + Err(fatal!( + "UnsubscribeNotificationChannel does not use state machine commands" + )) + } +} + +impl TryFrom for UnsubscribeNotificationChannelMachineEvents { + type Error = WFMachinesError; + + fn try_from(e: HistEventData) -> Result { + let e = e.event; + match e.event_type() { + EventType::WorkflowNotificationChannelUnsubscribed => { + if let Some( + history_event::Attributes::WorkflowNotificationChannelUnsubscribedEventAttributes( + attrs, + ), + ) = e.attributes + { + Ok(UnsubscribeNotificationChannelMachineEvents::CommandRecorded( + attrs, + )) + } else { + Err(fatal!( + "Notification channel unsubscribed attributes were unset: {e}" + )) + } + } + _ => Err(Self::Error::Nondeterminism(format!( + "UnsubscribeNotificationChannelMachine does not handle {e}" + ))), + } + } +} + +impl TryFrom for UnsubscribeNotificationChannelMachineEvents { + type Error = WFMachinesError; + + fn try_from(c: CommandType) -> Result { + match c { + CommandType::UnsubscribeNotificationChannel => { + Ok(UnsubscribeNotificationChannelMachineEvents::CommandScheduled) + } + _ => Err(Self::Error::Nondeterminism(format!( + "UnsubscribeNotificationChannelMachine does not handle command type {c:?}" + ))), + } + } +} + +impl From for CommandIssued { + fn from(_: Created) -> Self { + Self {} + } +} + +#[cfg(test)] +mod tests { + use super::{super::OnEventWrapper, *}; + use crate::{ + replay::TestHistoryBuilder, + test_help::{ + MockPollCfg, ResponseType, WorkerExt, build_mock_pollers, hist_to_poll_resp, + mock_worker, start_timer_cmd, + }, + worker::client::mocks::mock_worker_client, + }; + use parking_lot::Mutex; + use std::{sync::Arc, time::Duration}; + use temporalio_common::{ + protos::{ + coresdk::{ + workflow_commands::{CompleteWorkflowExecution, workflow_command}, + workflow_completion::WorkflowActivationCompletion, + }, + temporal::api::{ + command::v1::command, enums::v1::WorkflowTaskFailedCause, history::v1::HistoryEvent, + }, + }, + worker::WorkerTaskTypes, + }; + + fn unsubscribe(channel: &str) -> workflow_command::Variant { + UnsubscribeNotificationChannel { + channel: channel.to_string(), + } + .into() + } + + fn recorded(channel: &str) -> UnsubscribeNotificationChannelMachineEvents { + HistEventData { + event: HistoryEvent { + event_type: EventType::WorkflowNotificationChannelUnsubscribed as i32, + attributes: Some( + history_event::Attributes::WorkflowNotificationChannelUnsubscribedEventAttributes( + WorkflowNotificationChannelUnsubscribedEventAttributes { + workflow_task_completed_event_id: 0, + channel: channel.to_string(), + subscribed_event_id: 0, + }, + ), + ), + ..Default::default() + }, + replaying: true, + current_task_is_last_in_history: false, + } + .try_into() + .unwrap() + } + + fn issued(channel: &str) -> UnsubscribeNotificationChannelMachine { + let mut sm = UnsubscribeNotificationChannelMachine::from_parts( + Created {}.into(), + SharedState { + channel: channel.to_string(), + }, + ); + OnEventWrapper::on_event_mut( + &mut sm, + CommandType::UnsubscribeNotificationChannel + .try_into() + .unwrap(), + ) + .expect("CommandScheduled should transition Created -> CommandIssued"); + assert_eq!(CommandIssued {}.to_string(), sm.state().to_string()); + sm + } + + #[test] + fn the_recorded_event_for_the_same_channel_completes_the_machine() { + let mut sm = issued("orders"); + OnEventWrapper::on_event_mut(&mut sm, recorded("orders")) + .expect("CommandRecorded should transition CommandIssued -> Done"); + assert_eq!(Done {}.to_string(), sm.state().to_string()); + } + + #[test] + fn a_recorded_event_for_another_channel_is_nondeterminism() { + let mut sm = issued("orders"); + let err = OnEventWrapper::on_event_mut(&mut sm, recorded("invoices")) + .expect_err("a different channel must not match"); + let message = format!("{err:?}"); + assert!( + message.contains("does not match") && message.contains("invoices"), + "the error must name both channels, got {message}" + ); + } + + #[test] + fn the_subscribed_event_is_not_an_unsubscription() { + // The two events carry the same channel field, so the type is what tells them apart. + let event = HistEventData { + event: HistoryEvent { + event_type: EventType::WorkflowNotificationChannelSubscribed as i32, + ..Default::default() + }, + replaying: true, + current_task_is_last_in_history: false, + }; + let res: Result = event.try_into(); + assert!(matches!(res, Err(WFMachinesError::Nondeterminism(_)))); + } + + #[tokio::test] + async fn the_command_reaches_the_completion_with_its_channel() { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_workflow_task_scheduled_and_started(); + + let sent = Arc::new(Mutex::new(vec![])); + let recorder = sent.clone(); + let mut cfg = MockPollCfg::from_resp_batches("fakeid", t, [1], mock_worker_client()); + cfg.completion_mock_fn = Some(Box::new(move |wftc| { + recorder.lock().extend(wftc.commands.iter().cloned()); + Ok(Default::default()) + })); + let mut mock = build_mock_pollers(cfg); + mock.worker_cfg(|w| { + w.task_types = WorkerTaskTypes::workflow_only(); + w.max_cached_workflows = 1; + }); + let worker = mock_worker(mock); + + let first = worker.poll_workflow_activation().await.unwrap(); + assert!(!first.is_replaying); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmd( + first.run_id, + unsubscribe("orders"), + )) + .await + .unwrap(); + + let commands = sent.lock().clone(); + assert_eq!(commands.len(), 1, "got {commands:?}"); + assert_eq!( + commands[0].command_type(), + CommandType::UnsubscribeNotificationChannel + ); + assert!( + matches!( + &commands[0].attributes, + Some(command::Attributes::UnsubscribeNotificationChannelCommandAttributes(a)) + if a.channel == "orders" + ), + "got {:?}", + commands[0] + ); + + worker.drain_pollers_and_shutdown().await; + } + + /// Replays a first task whose recorded command event `record` writes, with lang reissuing + /// `reissued`. Returns the nondeterminism message Core reported, if any, and whether lang was + /// activated for the live task after it. + async fn replay_first_task( + record: impl FnOnce(&mut TestHistoryBuilder), + reissued: workflow_command::Variant, + expect_failure: bool, + ) -> (Option, bool) { + let mut t = TestHistoryBuilder::default(); + t.add_by_type(EventType::WorkflowExecutionStarted); + t.add_full_wf_task(); + record(&mut t); + // Gives the live task a job, so a match shows up as an activation rather than as silence. + t.add_we_signaled("go", vec![]); + t.add_workflow_task_scheduled_and_started(); + + let cold = hist_to_poll_resp(&t, "fakeid".to_owned(), ResponseType::AllHistory); + let mut cfg = MockPollCfg::from_resp_batches( + "fakeid", + t, + [ResponseType::Raw(cold.resp)], + mock_worker_client(), + ); + let failure = Arc::new(Mutex::new(None)); + if expect_failure { + cfg.num_expected_fails = 1; + let recorder = failure.clone(); + cfg.expect_fail_wft_matcher = Box::new(move |_, cause, f| { + if matches!(cause, WorkflowTaskFailedCause::NonDeterministicError) { + *recorder.lock() = + Some(f.as_ref().map(|f| f.message.clone()).unwrap_or_default()); + } + true + }); + } + let mut mock = build_mock_pollers(cfg); + mock.worker_cfg(|w| { + w.task_types = WorkerTaskTypes::workflow_only(); + w.max_cached_workflows = 1; + }); + let worker = mock_worker(mock); + + let first = worker.poll_workflow_activation().await.unwrap(); + assert!(first.is_replaying); + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmd( + first.run_id.clone(), + reissued, + )) + .await + .unwrap(); + + let next = tokio::time::timeout( + Duration::from_millis(500), + worker.poll_workflow_activation(), + ) + .await; + let mut activated = false; + if let Ok(Ok(act)) = next { + if act.is_only_eviction() { + worker + .complete_workflow_activation(WorkflowActivationCompletion::empty(act.run_id)) + .await + .unwrap(); + } else { + assert!( + !expect_failure, + "a mismatch must not activate lang, got {act:?}" + ); + activated = true; + worker + .complete_workflow_activation(WorkflowActivationCompletion::from_cmd( + act.run_id, + CompleteWorkflowExecution::default().into(), + )) + .await + .unwrap(); + } + } + worker.drain_pollers_and_shutdown().await; + let failure = failure.lock().clone(); + (failure, activated) + } + + #[tokio::test] + async fn replay_matches_the_recorded_unsubscription() { + let (failure, activated) = replay_first_task( + |t| { + t.add_notification_channel_unsubscribed("orders", 0); + }, + unsubscribe("orders"), + false, + ) + .await; + assert_eq!(failure, None); + assert!( + activated, + "the live task after a matched unsubscription must activate lang" + ); + } + + #[tokio::test] + async fn replay_with_another_channel_is_nondeterminism() { + let failure = replay_first_task( + |t| { + t.add_notification_channel_unsubscribed("invoices", 0); + }, + unsubscribe("orders"), + true, + ) + .await + .0 + .expect("the task must fail as nondeterminism"); + assert!(failure.contains("does not match"), "got {failure}"); + } + + #[tokio::test] + async fn a_recorded_subscription_does_not_match_an_unsubscription() { + let failure = replay_first_task( + |t| { + t.add_notification_channel_subscribed("orders"); + }, + unsubscribe("orders"), + true, + ) + .await + .0 + .expect("the task must fail as nondeterminism"); + assert!( + failure.contains("UnsubscribeNotificationChannelMachine does not handle"), + "got {failure}" + ); + } + + #[tokio::test] + async fn a_recorded_unsubscription_lang_does_not_reissue_is_nondeterminism() { + let failure = replay_first_task( + |t| { + t.add_notification_channel_unsubscribed("orders", 0); + }, + start_timer_cmd(1, Duration::from_secs(1)), + true, + ) + .await + .0 + .expect("the task must fail as nondeterminism"); + assert!(!failure.is_empty()); + } +} diff --git a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs index d7f12a0cd..a18f0c2a0 100644 --- a/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs +++ b/crates/sdk-core/src/worker/workflow/machines/workflow_machines.rs @@ -9,7 +9,9 @@ use super::{ fail_workflow_state_machine::fail_workflow, local_activity_state_machine::new_local_activity, patch_state_machine::has_change, signal_external_state_machine::new_external_signal, subscribe_notification_channel_state_machine::subscribe_notification_channel, - timer_state_machine::new_timer, upsert_search_attributes_state_machine::upsert_search_attrs, + timer_state_machine::new_timer, + unsubscribe_notification_channel_state_machine::unsubscribe_notification_channel, + upsert_search_attributes_state_machine::upsert_search_attrs, workflow_machines::local_acts::LocalActivityData, workflow_task_state_machine::WorkflowTaskMachine, }; @@ -69,6 +71,7 @@ use temporalio_common::{ }, workflow_commands::{ ContinueAsNewWorkflowExecution, ExternalStreamWait, SubscribeNotificationChannel, + UnsubscribeNotificationChannel, }, }, temporal::api::{ @@ -568,6 +571,20 @@ impl WorkflowMachines { self.prepare_commands() } + /// Ends the run's subscription to a notification channel on Core's own initiative, the way + /// [`Self::emit_notification_channel_subscription`] begins one. + pub(crate) fn emit_notification_channel_unsubscription( + &mut self, + channel: String, + ) -> Result<()> { + self.add_cmd_to_wf_task( + unsubscribe_notification_channel(UnsubscribeNotificationChannel { channel }), + None, + CommandIdKind::CoreInternal, + ); + self.prepare_commands() + } + /// Hands lang a marker the replay lookahead found, and creates the machine that settles it. /// /// Exactly one `ReplayExternalStreams` job per marker. Core is annotation-blind, so it copies @@ -1781,6 +1798,13 @@ impl WorkflowMachines { CommandIdKind::NeverResolves, ); } + WFCommandVariant::UnsubscribeNotificationChannel(attrs) => { + self.add_cmd_to_wf_task( + unsubscribe_notification_channel(attrs), + cmd.metadata, + CommandIdKind::NeverResolves, + ); + } WFCommandVariant::UpdateResponse(ur) => { let m_key = self.get_machine_by_msg(&ur.protocol_instance_id)?; let m = if let Machines::UpdateMachine(m) = self.machine_mut(m_key) { diff --git a/crates/sdk-core/src/worker/workflow/managed_run.rs b/crates/sdk-core/src/worker/workflow/managed_run.rs index cc013fdf7..d8974cffb 100644 --- a/crates/sdk-core/src/worker/workflow/managed_run.rs +++ b/crates/sdk-core/src/worker/workflow/managed_run.rs @@ -1446,26 +1446,29 @@ impl ManagedRun { // that ends the task instead: a terminal here is exactly that completion, the park // confirmation and the finalization answer included, and a run-ending one subscribes to // nothing because the subscription would end with the run anyway. The marker is already - // queued, so the subscribed events follow it in History, and lang's own commands follow - // them. + // queued, so the channel events follow it in History, and lang's own commands follow + // them. A channel the run stopped listening on is unsubscribed first, so a run rotating + // channels frees the slot before the new subscription takes one. if let Some(terminal) = terminal && terminal != ParkReason::WorkflowCompleted { - let due = self - .waiting_on_local_work - .channel_subscriptions - .take_due_subscriptions(); - for channel in due { - if let Err(source) = self - .wfm - .machines - .emit_notification_channel_subscription(channel) - { - return Err(RunUpdateErr { - source, - complete_resp_chan: completion.resp_chan, - }); - } + let subscriptions = &mut self.waiting_on_local_work.channel_subscriptions; + let leaving = subscriptions.take_due_unsubscriptions(); + let joining = subscriptions.take_due_subscriptions(); + let machines = &mut self.wfm.machines; + let issued = leaving + .into_iter() + .try_for_each(|channel| machines.emit_notification_channel_unsubscription(channel)) + .and_then(|()| { + joining.into_iter().try_for_each(|channel| { + machines.emit_notification_channel_subscription(channel) + }) + }); + if let Err(source) = issued { + return Err(RunUpdateErr { + source, + complete_resp_chan: completion.resp_chan, + }); } } diff --git a/crates/sdk-core/src/worker/workflow/mod.rs b/crates/sdk-core/src/worker/workflow/mod.rs index b707e27c8..b8618bfd2 100644 --- a/crates/sdk-core/src/worker/workflow/mod.rs +++ b/crates/sdk-core/src/worker/workflow/mod.rs @@ -1731,6 +1731,7 @@ enum WFCommandVariant { /// The complete set of notification channels the run listens on. Never implies retention. ExternalStreamChannels(WorkflowStreamChannels), SubscribeNotificationChannel(SubscribeNotificationChannel), + UnsubscribeNotificationChannel(UnsubscribeNotificationChannel), } impl TryFrom for WFCommand { @@ -1816,6 +1817,9 @@ impl TryFrom for WFCommand { workflow_command::Variant::SubscribeNotificationChannel(s) => { WFCommandVariant::SubscribeNotificationChannel(s) } + workflow_command::Variant::UnsubscribeNotificationChannel(u) => { + WFCommandVariant::UnsubscribeNotificationChannel(u) + } }; Ok(Self { variant, From 700698135fec798742dc280aa867f9db99a222fc Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 2 Oct 2026 17:34:02 -0700 Subject: [PATCH 34/35] Addressed a linked channel by execution in the Core's api tree. An activity can own a linked channel too, so the channel calls and the notification's owner take a temporal.api.common.v1.Execution with a type and a business id, as the api fork defines them. --- .../api/notification/v1/message.proto | 17 ++-- .../workflowservice/v1/request_response.proto | 80 ++++++++++++++----- .../api/workflowservice/v1/service.proto | 54 ++++++++++--- 3 files changed, 114 insertions(+), 37 deletions(-) diff --git a/crates/protos/protos/api_upstream/temporal/api/notification/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/notification/v1/message.proto index e1413efe1..85519f697 100644 --- a/crates/protos/protos/api_upstream/temporal/api/notification/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/notification/v1/message.proto @@ -30,14 +30,21 @@ message Notification { // notifications folded together, the one with the highest counter is kept. int64 counter = 3; // Details for the listener, such as which topic moved. Bounded in size and - // carried as payloads, so a codec applies as to any payload. + // carried as payloads, so a codec applies as to any payload. This is state, + // not a log: a fold keeps the latest notification only, so a writer puts + // here what is true at `position`, such as which topic moved or a close + // flag, never something a consumer must see once per write. map metadata = 4; - // Set for a channel linked to a workflow: the owner and the run that + + // Was temporal.api.common.v1.WorkflowExecution linked_to. + reserved 5; + + // Set for a channel linked to an execution: the owner and the run that // received the notification. Empty for an independent channel. A listener // that holds both kinds routes the notification by it. // (-- api-linter: core::0140::prepositions=disabled // aip.dev/not-precedent: "to" names the owner the channel is linked to. --) - temporal.api.common.v1.WorkflowExecution linked_to = 5; + temporal.api.common.v1.Execution linked_to = 6; } // A listener of a channel: a Workflow Execution woken with a Workflow Task, or @@ -66,7 +73,7 @@ enum ChannelKind { // Its own execution, keyed by namespace and channel name. Any number of // workflows and callbacks listen to it. CHANNEL_KIND_INDEPENDENT = 1; - // Kept in one workflow's state, keyed by namespace, workflow id and - // channel name. The owning workflow is its listener by construction. + // Kept in one execution's state, keyed by namespace, execution and + // channel name. The owning execution is its listener by construction. CHANNEL_KIND_LINKED = 2; } diff --git a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto index 37c881f24..a3a8b9253 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto @@ -886,14 +886,22 @@ message SignalWorkflowExecutionResponse { message NotifyChannelRequest { string namespace = 1; temporal.api.notification.v1.Notification notification = 2; - // The identity of the writer, for metrics and logs. + // The identity of the caller, for audit, metrics and logs. It is not copied + // into the notification. A writer that wants the consumer to see who wrote + // puts that in the notification's `metadata`. string identity = 3; // Used to de-dupe a retried notification. string request_id = 4; - // When set, the call addresses the channel linked to this workflow; `run_id` - // is optional and resolves to the chain's current run, as a Signal does. - // When unset, the call addresses the independent channel of that name. - temporal.api.common.v1.WorkflowExecution workflow_execution = 5; + + // Was temporal.api.common.v1.WorkflowExecution workflow_execution. + reserved 5; + reserved "workflow_execution"; + + // When set, the call addresses the channel linked to this execution. + // `run_id` is optional and resolves to the current run of a workflow chain, + // as a Signal does. When unset, the call addresses the independent channel + // of that name. + temporal.api.common.v1.Execution execution = 6; } message NotifyChannelResponse { @@ -911,10 +919,16 @@ message RegisterChannelListenerRequest { string request_id = 4; // The identity of the caller, for metrics and logs. string identity = 5; - // When set, the call addresses the channel linked to this workflow; `run_id` - // is optional and resolves to the chain's current run, as a Signal does. - // When unset, the call addresses the independent channel of that name. - temporal.api.common.v1.WorkflowExecution workflow_execution = 6; + + // Was temporal.api.common.v1.WorkflowExecution workflow_execution. + reserved 6; + reserved "workflow_execution"; + + // When set, the call addresses the channel linked to this execution. + // `run_id` is optional and resolves to the current run of a workflow chain, + // as a Signal does. When unset, the call addresses the independent channel + // of that name. + temporal.api.common.v1.Execution execution = 7; } message RegisterChannelListenerResponse { @@ -927,10 +941,16 @@ message UnregisterChannelListenerRequest { string listener_id = 3; // The identity of the caller, for metrics and logs. string identity = 4; - // When set, the call addresses the channel linked to this workflow; `run_id` - // is optional and resolves to the chain's current run, as a Signal does. - // When unset, the call addresses the independent channel of that name. - temporal.api.common.v1.WorkflowExecution workflow_execution = 5; + + // Was temporal.api.common.v1.WorkflowExecution workflow_execution. + reserved 5; + reserved "workflow_execution"; + + // When set, the call addresses the channel linked to this execution. + // `run_id` is optional and resolves to the current run of a workflow chain, + // as a Signal does. When unset, the call addresses the independent channel + // of that name. + temporal.api.common.v1.Execution execution = 6; } message UnregisterChannelListenerResponse { @@ -949,10 +969,16 @@ message PollChannelRequest { // At most this many notifications are returned. Zero means the server's // default. int32 max_notifications = 5; - // When set, the call addresses the channel linked to this workflow; `run_id` - // is optional and resolves to the chain's current run, as a Signal does. - // When unset, the call addresses the independent channel of that name. - temporal.api.common.v1.WorkflowExecution workflow_execution = 6; + + // Was temporal.api.common.v1.WorkflowExecution workflow_execution. + reserved 6; + reserved "workflow_execution"; + + // When set, the call addresses the channel linked to this execution. + // `run_id` is optional and resolves to the current run of a workflow chain, + // as a Signal does. When unset, the call addresses the independent channel + // of that name. + temporal.api.common.v1.Execution execution = 7; } message PollChannelResponse { @@ -962,10 +988,16 @@ message PollChannelResponse { message DescribeChannelRequest { string namespace = 1; string channel = 2; - // When set, the call addresses the channel linked to this workflow; `run_id` - // is optional and resolves to the chain's current run, as a Signal does. - // When unset, the call addresses the independent channel of that name. - temporal.api.common.v1.WorkflowExecution workflow_execution = 3; + + // Was temporal.api.common.v1.WorkflowExecution workflow_execution. + reserved 3; + reserved "workflow_execution"; + + // When set, the call addresses the channel linked to this execution. + // `run_id` is optional and resolves to the current run of a workflow chain, + // as a Signal does. When unset, the call addresses the independent channel + // of that name. + temporal.api.common.v1.Execution execution = 4; } message DescribeChannelResponse { @@ -975,11 +1007,15 @@ message DescribeChannelResponse { // How many notifications the channel retains for pollers. int32 retained_count = 3; temporal.api.notification.v1.ChannelKind kind = 4; + + // Was temporal.api.common.v1.WorkflowExecution linked_to. + reserved 5; + // The owner of a linked channel and the run that holds it. Empty for an // independent channel. // (-- api-linter: core::0140::prepositions=disabled // aip.dev/not-precedent: "to" names the owner the channel is linked to. --) - temporal.api.common.v1.WorkflowExecution linked_to = 5; + temporal.api.common.v1.Execution linked_to = 6; } message SignalWithStartWorkflowExecutionRequest { diff --git a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto index 95238d2ab..76c43e8a5 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/service.proto @@ -497,11 +497,19 @@ service WorkflowService { body: "*" } additional_bindings { - post: "/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/channels/{notification.channel}/notify" + post: "/namespaces/{namespace}/workflows/{execution.business_id}/channels/{notification.channel}/notify" body: "*" } additional_bindings { - post: "/api/v1/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/channels/{notification.channel}/notify" + post: "/api/v1/namespaces/{namespace}/workflows/{execution.business_id}/channels/{notification.channel}/notify" + body: "*" + } + additional_bindings { + post: "/namespaces/{namespace}/activities/{execution.business_id}/channels/{notification.channel}/notify" + body: "*" + } + additional_bindings { + post: "/api/v1/namespaces/{namespace}/activities/{execution.business_id}/channels/{notification.channel}/notify" body: "*" } }; @@ -520,11 +528,19 @@ service WorkflowService { body: "*" } additional_bindings { - post: "/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/channels/{channel}/listeners" + post: "/namespaces/{namespace}/workflows/{execution.business_id}/channels/{channel}/listeners" + body: "*" + } + additional_bindings { + post: "/api/v1/namespaces/{namespace}/workflows/{execution.business_id}/channels/{channel}/listeners" + body: "*" + } + additional_bindings { + post: "/namespaces/{namespace}/activities/{execution.business_id}/channels/{channel}/listeners" body: "*" } additional_bindings { - post: "/api/v1/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/channels/{channel}/listeners" + post: "/api/v1/namespaces/{namespace}/activities/{execution.business_id}/channels/{channel}/listeners" body: "*" } }; @@ -542,10 +558,16 @@ service WorkflowService { delete: "/api/v1/namespaces/{namespace}/channels/{channel}/listeners/{listener_id}" } additional_bindings { - delete: "/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/channels/{channel}/listeners/{listener_id}" + delete: "/namespaces/{namespace}/workflows/{execution.business_id}/channels/{channel}/listeners/{listener_id}" } additional_bindings { - delete: "/api/v1/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/channels/{channel}/listeners/{listener_id}" + delete: "/api/v1/namespaces/{namespace}/workflows/{execution.business_id}/channels/{channel}/listeners/{listener_id}" + } + additional_bindings { + delete: "/namespaces/{namespace}/activities/{execution.business_id}/channels/{channel}/listeners/{listener_id}" + } + additional_bindings { + delete: "/api/v1/namespaces/{namespace}/activities/{execution.business_id}/channels/{channel}/listeners/{listener_id}" } }; } @@ -560,10 +582,16 @@ service WorkflowService { get: "/api/v1/namespaces/{namespace}/channels/{channel}/notifications" } additional_bindings { - get: "/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/channels/{channel}/notifications" + get: "/namespaces/{namespace}/workflows/{execution.business_id}/channels/{channel}/notifications" + } + additional_bindings { + get: "/api/v1/namespaces/{namespace}/workflows/{execution.business_id}/channels/{channel}/notifications" } additional_bindings { - get: "/api/v1/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/channels/{channel}/notifications" + get: "/namespaces/{namespace}/activities/{execution.business_id}/channels/{channel}/notifications" + } + additional_bindings { + get: "/api/v1/namespaces/{namespace}/activities/{execution.business_id}/channels/{channel}/notifications" } }; } @@ -577,10 +605,16 @@ service WorkflowService { get: "/api/v1/namespaces/{namespace}/channels/{channel}" } additional_bindings { - get: "/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/channels/{channel}" + get: "/namespaces/{namespace}/workflows/{execution.business_id}/channels/{channel}" + } + additional_bindings { + get: "/api/v1/namespaces/{namespace}/workflows/{execution.business_id}/channels/{channel}" + } + additional_bindings { + get: "/namespaces/{namespace}/activities/{execution.business_id}/channels/{channel}" } additional_bindings { - get: "/api/v1/namespaces/{namespace}/workflows/{workflow_execution.workflow_id}/channels/{channel}" + get: "/api/v1/namespaces/{namespace}/activities/{execution.business_id}/channels/{channel}" } }; } From 677d468c090f2c4657edd2cf436576eabe8236af Mon Sep 17 00:00:00 2001 From: Mohammad Dashti Date: Fri, 2 Oct 2026 19:20:12 -0700 Subject: [PATCH 35/35] Reused the old field numbers for the execution fields in the api tree. The workflow-addressed shape was never released, so the execution fields take its numbers and nothing is reserved, as the api fork defines them. --- .../api/notification/v1/message.proto | 6 +-- .../workflowservice/v1/request_response.proto | 41 +++---------------- 2 files changed, 7 insertions(+), 40 deletions(-) diff --git a/crates/protos/protos/api_upstream/temporal/api/notification/v1/message.proto b/crates/protos/protos/api_upstream/temporal/api/notification/v1/message.proto index 85519f697..584619582 100644 --- a/crates/protos/protos/api_upstream/temporal/api/notification/v1/message.proto +++ b/crates/protos/protos/api_upstream/temporal/api/notification/v1/message.proto @@ -35,16 +35,12 @@ message Notification { // here what is true at `position`, such as which topic moved or a close // flag, never something a consumer must see once per write. map metadata = 4; - - // Was temporal.api.common.v1.WorkflowExecution linked_to. - reserved 5; - // Set for a channel linked to an execution: the owner and the run that // received the notification. Empty for an independent channel. A listener // that holds both kinds routes the notification by it. // (-- api-linter: core::0140::prepositions=disabled // aip.dev/not-precedent: "to" names the owner the channel is linked to. --) - temporal.api.common.v1.Execution linked_to = 6; + temporal.api.common.v1.Execution linked_to = 5; } // A listener of a channel: a Workflow Execution woken with a Workflow Task, or diff --git a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto index a3a8b9253..fb70ab180 100644 --- a/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto +++ b/crates/protos/protos/api_upstream/temporal/api/workflowservice/v1/request_response.proto @@ -892,16 +892,11 @@ message NotifyChannelRequest { string identity = 3; // Used to de-dupe a retried notification. string request_id = 4; - - // Was temporal.api.common.v1.WorkflowExecution workflow_execution. - reserved 5; - reserved "workflow_execution"; - // When set, the call addresses the channel linked to this execution. // `run_id` is optional and resolves to the current run of a workflow chain, // as a Signal does. When unset, the call addresses the independent channel // of that name. - temporal.api.common.v1.Execution execution = 6; + temporal.api.common.v1.Execution execution = 5; } message NotifyChannelResponse { @@ -919,16 +914,11 @@ message RegisterChannelListenerRequest { string request_id = 4; // The identity of the caller, for metrics and logs. string identity = 5; - - // Was temporal.api.common.v1.WorkflowExecution workflow_execution. - reserved 6; - reserved "workflow_execution"; - // When set, the call addresses the channel linked to this execution. // `run_id` is optional and resolves to the current run of a workflow chain, // as a Signal does. When unset, the call addresses the independent channel // of that name. - temporal.api.common.v1.Execution execution = 7; + temporal.api.common.v1.Execution execution = 6; } message RegisterChannelListenerResponse { @@ -941,16 +931,11 @@ message UnregisterChannelListenerRequest { string listener_id = 3; // The identity of the caller, for metrics and logs. string identity = 4; - - // Was temporal.api.common.v1.WorkflowExecution workflow_execution. - reserved 5; - reserved "workflow_execution"; - // When set, the call addresses the channel linked to this execution. // `run_id` is optional and resolves to the current run of a workflow chain, // as a Signal does. When unset, the call addresses the independent channel // of that name. - temporal.api.common.v1.Execution execution = 6; + temporal.api.common.v1.Execution execution = 5; } message UnregisterChannelListenerResponse { @@ -969,16 +954,11 @@ message PollChannelRequest { // At most this many notifications are returned. Zero means the server's // default. int32 max_notifications = 5; - - // Was temporal.api.common.v1.WorkflowExecution workflow_execution. - reserved 6; - reserved "workflow_execution"; - // When set, the call addresses the channel linked to this execution. // `run_id` is optional and resolves to the current run of a workflow chain, // as a Signal does. When unset, the call addresses the independent channel // of that name. - temporal.api.common.v1.Execution execution = 7; + temporal.api.common.v1.Execution execution = 6; } message PollChannelResponse { @@ -988,16 +968,11 @@ message PollChannelResponse { message DescribeChannelRequest { string namespace = 1; string channel = 2; - - // Was temporal.api.common.v1.WorkflowExecution workflow_execution. - reserved 3; - reserved "workflow_execution"; - // When set, the call addresses the channel linked to this execution. // `run_id` is optional and resolves to the current run of a workflow chain, // as a Signal does. When unset, the call addresses the independent channel // of that name. - temporal.api.common.v1.Execution execution = 4; + temporal.api.common.v1.Execution execution = 3; } message DescribeChannelResponse { @@ -1007,15 +982,11 @@ message DescribeChannelResponse { // How many notifications the channel retains for pollers. int32 retained_count = 3; temporal.api.notification.v1.ChannelKind kind = 4; - - // Was temporal.api.common.v1.WorkflowExecution linked_to. - reserved 5; - // The owner of a linked channel and the run that holds it. Empty for an // independent channel. // (-- api-linter: core::0140::prepositions=disabled // aip.dev/not-precedent: "to" names the owner the channel is linked to. --) - temporal.api.common.v1.Execution linked_to = 6; + temporal.api.common.v1.Execution linked_to = 5; } message SignalWithStartWorkflowExecutionRequest {