1414//! the controller as a **separate task** and implements the ctx by marshaling
1515//! each pull/apply to the Coordinator over the internal command channel, because
1616//! the catalog and the live compute/storage signals are reachable only from the
17- //! coordinator loop. The two whole-tick reads are batched; the per-cluster live
18- //! signals are pulled on demand, so a tick's round-trips scale with the number of
19- //! managed clusters that need a live signal, not with a constant.
17+ //! coordinator loop. Whole-tick reads are batched. Refresh-window catalog inputs
18+ //! are pulled one cluster at a time and completed with one shared oracle read.
19+ //! The remaining per-cluster live signals are pulled on demand, so steady
20+ //! clusters do not pay for signals they do not use.
2021//!
2122//! Everything here is gated by [`ENABLE_CLUSTER_CONTROLLER`] (default on). With
2223//! the gate off the task does not tick, so the legacy scheduling and graceful
2627//! replicas are materialized by `reconcile_builtin_cluster_replicas` at catalog
2728//! open, which derives the same target from the same config.)
2829
29- use std:: collections:: BTreeSet ;
30+ use std:: collections:: { BTreeMap , BTreeSet } ;
3031use std:: sync:: Arc ;
3132use std:: time:: Duration ;
3233
@@ -36,7 +37,8 @@ use mz_cluster_controller::ClusterController;
3637use mz_cluster_controller:: ctx:: {
3738 ApplyOutcome , AvailabilityZones , ClusterControllerCtx , ClusterState , CreateReason , Decision ,
3839 ExpectedClusterState , ObservedReplica , OnTimeout , ReconfigurationRecord , ReconfigurationStatus ,
39- ReconfigurationTarget , RefreshMvInfo , RefreshWindowInputs , ReplicaShape , StateWrite ,
40+ ReconfigurationTarget , RefreshMvInfo , RefreshWindowClusterInputs , RefreshWindowInputsBatch ,
41+ ReplicaShape , StateWrite ,
4042} ;
4143use mz_compute_types:: config:: ComputeReplicaConfig ;
4244use mz_controller:: clusters:: ClusterStatus ;
@@ -54,8 +56,9 @@ use crate::error::AdapterError;
5456/// [`ClusterControllerCtx`] call. Each variant carries a oneshot for the reply.
5557///
5658/// `ManagedClusterIds` and `ClusterStates` are the per-tick batched reads. The
57- /// `ClusterStates` reply also carries `now`. `HydratedReplicas` is a
58- /// per-cluster live signal a strategy pulls on demand.
59+ /// `ClusterStates` reply also carries `now`. Refresh-window catalog inputs are
60+ /// pulled one cluster at a time, followed by one shared oracle read.
61+ /// `HydratedReplicas` is a per-cluster live signal a strategy pulls on demand.
5962#[ derive( Debug ) ]
6063pub enum ClusterControllerRequest {
6164 /// The ids of all *user* managed clusters the controller owns this tick.
@@ -81,13 +84,14 @@ pub enum ClusterControllerRequest {
8184 cluster_id : ClusterId ,
8285 tx : oneshot:: Sender < bool > ,
8386 } ,
84- /// The refresh-window live signals for one scheduled cluster (read ts,
85- /// compaction estimate, bound REFRESH MVs). `None` for a cluster that is not
86- /// scheduled `ON REFRESH`.
87- RefreshWindowInputs {
87+ /// The catalog and storage refresh-window inputs for one scheduled cluster.
88+ /// `None` if the cluster no longer qualifies at pull time.
89+ RefreshWindowClusterInputs {
8890 cluster_id : ClusterId ,
89- tx : oneshot:: Sender < Option < RefreshWindowInputs > > ,
91+ tx : oneshot:: Sender < Option < RefreshWindowClusterInputs > > ,
9092 } ,
93+ /// One timestamp-oracle read for a completed refresh-window input batch.
94+ RefreshWindowReadTs { tx : oneshot:: Sender < Timestamp > } ,
9195 /// Apply a tick's batch of decisions under their compare-and-append guards.
9296 Apply {
9397 decisions : Vec < Decision > ,
@@ -184,11 +188,38 @@ impl ClusterControllerCtx for CoordCtx {
184188
185189 async fn refresh_window_inputs (
186190 & mut self ,
187- cluster_id : ClusterId ,
188- ) -> Option < RefreshWindowInputs > {
189- self . request ( |tx| ClusterControllerRequest :: RefreshWindowInputs { cluster_id, tx } )
190- . await
191- . flatten ( )
191+ cluster_ids : & [ ClusterId ] ,
192+ ) -> Option < RefreshWindowInputsBatch > {
193+ let mut cluster_inputs = BTreeMap :: new ( ) ;
194+ for ( index, & cluster_id) in cluster_ids. iter ( ) . enumerate ( ) {
195+ if index > 0 {
196+ // The coordinator prioritizes its internal command channel. Give
197+ // it a chance to service already-queued user commands instead of
198+ // keeping that channel continuously ready for the whole batch.
199+ tokio:: task:: yield_now ( ) . await ;
200+ }
201+ let inputs = self
202+ . request ( |tx| ClusterControllerRequest :: RefreshWindowClusterInputs {
203+ cluster_id,
204+ tx,
205+ } )
206+ . await
207+ . flatten ( ) ;
208+ if let Some ( inputs) = inputs {
209+ cluster_inputs. insert ( cluster_id, inputs) ;
210+ }
211+ }
212+ if cluster_inputs. is_empty ( ) {
213+ return None ;
214+ }
215+
216+ let read_ts = self
217+ . request ( |tx| ClusterControllerRequest :: RefreshWindowReadTs { tx } )
218+ . await ?;
219+ Some ( RefreshWindowInputsBatch {
220+ read_ts,
221+ cluster_inputs,
222+ } )
192223 }
193224
194225 async fn apply ( & mut self , decisions : Vec < Decision > ) -> ApplyOutcome {
@@ -326,36 +357,18 @@ impl Coordinator {
326357 ClusterControllerRequest :: HasHydratableObjects { cluster_id, tx } => {
327358 let _ = tx. send ( self . cluster_has_hydratable_objects ( cluster_id) ) ;
328359 }
329- ClusterControllerRequest :: RefreshWindowInputs { cluster_id, tx } => {
330- // Gather the catalog- and storage-derived inputs on the loop,
331- // then complete the reply from a spawned task: the oracle
332- // read is a network round-trip (to the Postgres/CRDB-backed
333- // timestamp oracle) and must never run on the serial
334- // coordinator loop. The legacy `check_refresh_policy` makes
335- // the same split.
336- match self . refresh_window_catalog_inputs ( cluster_id) {
337- None => {
338- let _ = tx. send ( None ) ;
339- }
340- Some ( ( compaction_estimate, refresh_mvs) ) => {
341- let oracle = self . get_local_timestamp_oracle ( ) ;
342- // NOTE: this is one oracle read per scheduled cluster
343- // per tick, and the controller awaits each pull before
344- // the next, so the reads are sequential and the
345- // batching oracle cannot coalesce them. Fine at the
346- // tick cadence for realistic scheduled-cluster counts.
347- // TODO: hoist to one read per tick if that stops
348- // holding.
349- spawn ( || "cluster_controller_refresh_window_read_ts" , async move {
350- let read_ts = oracle. read_ts ( ) . await ;
351- let _ = tx. send ( Some ( RefreshWindowInputs {
352- read_ts,
353- compaction_estimate,
354- refresh_mvs,
355- } ) ) ;
356- } ) ;
357- }
358- }
360+ ClusterControllerRequest :: RefreshWindowClusterInputs { cluster_id, tx } => {
361+ let _ = tx. send ( self . refresh_window_catalog_inputs ( cluster_id) ) ;
362+ }
363+ ClusterControllerRequest :: RefreshWindowReadTs { tx } => {
364+ // The oracle read is a network round trip to the Postgres or
365+ // CRDB-backed timestamp oracle. It must not run on the serial
366+ // coordinator loop.
367+ let oracle = self . get_local_timestamp_oracle ( ) ;
368+ spawn ( || "cluster_controller_refresh_window_read_ts" , async move {
369+ let read_ts = oracle. read_ts ( ) . await ;
370+ let _ = tx. send ( read_ts) ;
371+ } ) ;
359372 }
360373 ClusterControllerRequest :: Apply { decisions, tx } => {
361374 let outcome = if active {
@@ -514,7 +527,7 @@ impl Coordinator {
514527 /// REFRESH`. These are the same signals the legacy `check_refresh_policy`
515528 /// reads.
516529 ///
517- /// The oracle read timestamp completing [`RefreshWindowInputs `] is
530+ /// The oracle read timestamp completing [`RefreshWindowInputsBatch `] is
518531 /// deliberately not fetched here: this runs on the coordinator loop, and
519532 /// the oracle read is a network round-trip the request handler performs on
520533 /// a spawned task instead.
@@ -525,7 +538,7 @@ impl Coordinator {
525538 fn refresh_window_catalog_inputs (
526539 & self ,
527540 cluster_id : ClusterId ,
528- ) -> Option < ( Duration , Vec < RefreshMvInfo > ) > {
541+ ) -> Option < RefreshWindowClusterInputs > {
529542 use mz_catalog:: memory:: objects:: CatalogItem ;
530543
531544 let cluster = self . catalog ( ) . try_get_cluster ( cluster_id) ?;
@@ -568,7 +581,10 @@ impl Coordinator {
568581 . system_config ( )
569582 . cluster_refresh_mv_compaction_estimate ( ) ;
570583
571- Some ( ( compaction_estimate, refresh_mvs) )
584+ Some ( RefreshWindowClusterInputs {
585+ compaction_estimate,
586+ refresh_mvs,
587+ } )
572588 }
573589
574590 /// Apply one batch of decisions under their compare-and-append guards.
0 commit comments