Skip to main content

mz_compute_types/
dyncfgs.rs

1// Copyright Materialize, Inc. and contributors. All rights reserved.
2//
3// Use of this software is governed by the Business Source License
4// included in the LICENSE file.
5//
6// As of the Change Date specified in that file, in accordance with
7// the Business Source License, use of this software will be governed
8// by the Apache License, Version 2.0.
9
10//! Dyncfgs used by the compute layer.
11
12use std::time::Duration;
13
14use mz_dyncfg::{Config, ConfigSet, ParameterScope};
15
16/// Whether rendering should use `half_join2` rather than DD's `half_join` for delta joins.
17///
18/// `half_join2` avoids quadratic behavior in certain join patterns. This flag exists as an escape
19/// hatch to revert to the old implementation if issues arise.
20pub const ENABLE_HALF_JOIN2: Config<bool> = Config::new(
21    "enable_compute_half_join2",
22    true,
23    "Whether compute should use `half_join2` rather than DD's `half_join` to render delta joins.",
24    ParameterScope::Environment,
25);
26
27/// Whether rendering should collapse error multiplicities to one where it arranges errors.
28///
29/// Error semantics depend only on whether an error is present, so its multiplicity carries no
30/// information a consumer reads. Left uncollapsed, a shared collection contributes its errors once
31/// per plan path that reads it, and because those factors apply again at each level of sharing they
32/// compound multiplicatively until the `Diff` overflows.
33///
34/// Governs sharing within a dataflow only. Sharing across objects is bounded unconditionally, by
35/// normalizing at every boundary another dataflow can read, so what this flag decides is never
36/// durable state and two replicas rendering under different values still write the same thing.
37/// Environment-scoped for now because nothing needs finer granularity, not because finer would be
38/// unsafe.
39pub const ENABLE_ERROR_DISTINCT: Config<bool> = Config::new(
40    "enable_compute_error_distinct",
41    true,
42    "Whether compute rendering should collapse error multiplicities to one where it arranges \
43     errors.",
44    ParameterScope::Environment,
45);
46
47/// Use the chunked merge batcher code path at arrange sites. When `true`,
48/// arrange operators batch through `ChunkBatcher` over
49/// `ColumnChunk` (in `mz_timely_util::columnar::chunk`) and build batches
50/// with `RowRowColPagedBuilder` (in `mz_row_spine`) behind
51/// `UnchunkBuilder`: columnar-native chains whose bodies the process buffer
52/// pool can spill (gated by [`ENABLE_COLUMN_PAGED_BATCHER_SPILL`]). Read at
53/// operator construction time. Flips take effect on dataflows created after
54/// the change.
55///
56/// Takes precedence over [`ENABLE_COLUMNAR_MERGE_BATCHER`]: both select
57/// columnar chains, and this one additionally makes them spillable. With
58/// both `false` the arrange sites use the columnation `Col2ValBatcher` /
59/// `RowRowBuilder` path. See
60/// `mz_compute::extensions::arrange::ArrangementBatcher` for the resolution.
61///
62/// Disabled by default while the new path is stabilizing.
63/// `DifferentialJoinHydration*` feature-benchmark scenarios opt in
64/// explicitly so the spill path is measured.
65pub const ENABLE_COLUMN_PAGED_BATCHER: Config<bool> = Config::new(
66    "enable_column_paged_batcher",
67    false,
68    "Use the columnar-native chunked merge batcher at arrange sites, whose chunk bodies the \
69     process buffer pool can spill. Takes precedence over enable_columnar_merge_batcher; \
70     with both false, arranges use the columnation `Col2ValBatcher` / `RowRowBuilder` path.",
71    ParameterScope::Replica,
72);
73
74/// Use the resident columnar merge batcher at arrange sites. When `true`,
75/// arrange operators use `Col2ValColBatcher` (in `mz_timely_util::columnar`)
76/// and `RowRowColPagedBuilder` (in `mz_row_spine`): the same `Column` chains
77/// and the same builder as the chunked arm, merged by `ColumnMerger` with
78/// no spill budget. When `false` (the default), the arrange sites
79/// use the columnation `Col2ValBatcher` / `RowRowBuilder` path. Read at
80/// operator construction time. Flips take effect on dataflows created after
81/// the change.
82///
83/// This is the columnation-versus-columnar axis on its own, so the two paths
84/// can be compared without spilling in the measurement. It is ignored while
85/// [`ENABLE_COLUMN_PAGED_BATCHER`] is `true`.
86pub const ENABLE_COLUMNAR_MERGE_BATCHER: Config<bool> = Config::new(
87    "enable_columnar_merge_batcher",
88    false,
89    "Use the resident columnar merge batcher at arrange sites, instead of the columnation \
90     one. Ignored when enable_column_paged_batcher is true.",
91    ParameterScope::Replica,
92);
93
94/// Store the accumulable reduce's accumulators in columnar form.
95///
96/// The accumulable reduce keeps one accumulator per aggregate in the diff of its
97/// input arrangement. When `true`, that arrangement holds its diffs in a columnar
98/// container, which lays the accumulators out by variant so each pays only for its
99/// own fields. When `false` (the default), the diffs live in a columnation stack at
100/// the width of the largest variant. Read at operator construction time, so flips
101/// take effect on dataflows created after the change.
102pub const ENABLE_COLUMNAR_ACCUMULABLE_DIFF: Config<bool> = Config::new(
103    "enable_columnar_accumulable_diff",
104    false,
105    "Store the accumulable reduce's accumulators in a columnar arrangement diff, laid out by \
106     variant, instead of a columnation stack of fixed-width accumulator enums.",
107    ParameterScope::Replica,
108);
109
110/// Allow chunk bodies to spill to the process buffer pool under memory
111/// pressure. Sets compute's leg of the process-wide chunk spill gate, which
112/// the arrange batchers read when [`ENABLE_COLUMN_PAGED_BATCHER`] is `true`.
113/// With the gate clear every chunk stays resident regardless of budget.
114///
115/// This flag (or the storage-side `enable_upsert_paged_spill`, or
116/// [`ENABLE_CORRECTION_V2_SPILL`]) also gates installation of the process
117/// buffer pool (`mz_ore::pool`): the first configuration tick with any of
118/// these gates on reserves the pool's virtual
119/// address space and spawns its spill threads. Turning the gates back off
120/// stops retuning but does not tear the installed pool down.
121///
122/// It additionally enables the process-global column pager, which storage's
123/// paged upsert stash draws from, so it is not exclusive to the arrange path.
124///
125/// Off by default, even when the batcher path itself is on, so the
126/// no-pressure case stays a pure resident operation. Tune the budget via
127/// [`COLUMN_PAGED_BATCHER_BUDGET_FRACTION`].
128pub const ENABLE_COLUMN_PAGED_BATCHER_SPILL: Config<bool> = Config::new(
129    "enable_column_paged_batcher_spill",
130    false,
131    "Allow chunk bodies to spill to the process buffer pool under memory pressure. Only \
132     meaningful for arrange sites when `enable_column_paged_batcher = true`.",
133    ParameterScope::Replica,
134);
135
136/// The youngest chunk generation whose spilled bodies are compressed.
137///
138/// A chunk at generational depth `d` is rewritten with frequency
139/// proportional to `2^-d` under geometric merging, so compressing shallow
140/// generations buys pool bytes back for only a short stay at a guaranteed
141/// near-term codec round-trip. Generations below the floor spill under the
142/// identity codec: fully budgeted and swap-backed, with encode and decode
143/// reduced to copies. The default exempts only fresh (depth 0) chunks. A
144/// chunk that outlives a merge untouched ages a generation regardless, and
145/// an identity-coded body at or past the floor is re-spilled compressed at
146/// its next survival, so key-disjoint input cannot hold its backlog
147/// uncompressed indefinitely. Lowering this at runtime therefore migrates
148/// bodies that already spilled, rather than applying only to new ones.
149/// `0` compresses every spilled body.
150pub const COLUMN_CHUNK_COMPRESS_MIN_DEPTH: Config<u32> = Config::new(
151    "column_chunk_compress_min_depth",
152    1,
153    "The youngest chunk generation whose spilled bodies are lz4-compressed in the buffer \
154     pool; younger generations store uncompressed. 0 compresses every spilled body.",
155    ParameterScope::Replica,
156);
157
158/// Resident-bytes budget fraction for chunk spilling. Two consumers read
159/// it: the column pager's tiered policy multiplies it against the
160/// announced memory limit, and the buffer pool (`mz_ore::pool`)
161/// multiplies it against physical RAM, since a resident budget must
162/// derive from memory that can be resident and the announced limit
163/// includes swap on swap-provisioned nodes.
164///
165/// `0.05` (5%) is a reasonable starting point: large enough that the
166/// per-call ColumnBuilder ship-threshold (~2 MiB) fits multiple chunks
167/// per worker, small enough that the merge-batcher's transient state
168/// doesn't crowd out the spine. Set lower to spill more aggressively
169/// under pressure. The computed budget is floored at 128 MiB so the
170/// no-pressure case doesn't page per chunk. Ignored when
171/// `enable_column_paged_batcher_spill` is `false`.
172pub const COLUMN_PAGED_BATCHER_BUDGET_FRACTION: Config<f64> = Config::new(
173    "column_paged_batcher_budget_fraction",
174    0.05,
175    "Budget fraction for chunk spilling: the buffer pool multiplies it against physical \
176     RAM and the column pager's tiered policy against the announced memory limit. \
177     Total pool budget = max(ram * fraction, 128 MiB).",
178    ParameterScope::Replica,
179);
180
181/// Number of buffer-pool spill threads performing eviction I/O (lz4
182/// compression plus the synchronous-reclaim `MADV_PAGEOUT`) off the threads
183/// that trip the budget. Zero evicts inline on the calling thread, which
184/// measurably convoys workers behind eviction I/O at hydration eviction
185/// rates. Thread spawning is once per process: raising the value later has
186/// no effect beyond re-enabling, and lowering it to zero falls back to
187/// inline eviction while spawned threads idle.
188pub const COLUMN_PAGED_BATCHER_SPILL_WORKER_COUNT: Config<usize> = Config::new(
189    "column_paged_batcher_spill_worker_count",
190    2,
191    "Buffer-pool spill threads for off-worker eviction I/O; 0 evicts inline on the caller.",
192    ParameterScope::Replica,
193);
194
195/// Compress chunks the column-paged batcher spills, using lz4. Only
196/// meaningful when [`ENABLE_COLUMN_PAGED_BATCHER_SPILL`] is `true`; the codec
197/// is applied on the pageout path and reversed on page-in. Trades CPU for a
198/// smaller on-storage (and, for the swap backend, resident) footprint.
199///
200/// Off by default so the spill path's cost stays a pure copy until compression
201/// is shown to pay for itself on the target workload.
202pub const COLUMN_PAGED_BATCHER_LZ4: Config<bool> = Config::new(
203    "column_paged_batcher_lz4",
204    false,
205    "Compress column-paged batcher chunks with lz4 on the spill path. Only meaningful when \
206     `enable_column_paged_batcher_spill = true`.",
207    ParameterScope::Replica,
208);
209
210/// Proactively evict the column-paged batcher's lz4-compressed spill chunks
211/// from RSS via `MADV_PAGEOUT` when spilling to the swap backend. Only
212/// meaningful when [`COLUMN_PAGED_BATCHER_LZ4`] is `true` and the active
213/// backend is swap (no scratch directory): on that path the compressed bytes
214/// stay resident in the process address space and currently receive no madvise
215/// at all, so the kernel reclaims them only lazily under LRU pressure.
216/// `MADV_PAGEOUT` instead swaps them out eagerly at spill time, holding RSS at
217/// the budget rather than letting it drift up to the pressure cliff. A later
218/// page-in re-faults the pages — cheap because lz4 shrank the byte volume,
219/// which is what makes eager eviction pay off on this path.
220///
221/// Off by default: the eager-reclaim syscall is the one kernel interaction the
222/// pager design singled out as risky, so it stays gated until proven on the
223/// target workload.
224pub const COLUMN_PAGED_BATCHER_SWAP_PAGEOUT: Config<bool> = Config::new(
225    "column_paged_batcher_swap_pageout",
226    false,
227    "Eagerly evict the column-paged batcher's lz4-compressed swap-backend spill chunks from RSS \
228     via `MADV_PAGEOUT` (they otherwise receive no madvise and are reclaimed only lazily). Only \
229     meaningful when `column_paged_batcher_lz4 = true` and the swap backend is active.",
230    ParameterScope::Replica,
231);
232
233/// Eagerly compress unbacked buffer-pool chunks to `BackedResident` on idle
234/// spill threads (write-behind). The chunk stays readable in its slot while
235/// a compressed extent accumulates on the swap device, so budget-driven
236/// eviction becomes a pure page release instead of a compression. Trades
237/// background CPU (compression of chunks that may die before pressure
238/// reaches them) for near-free pressure response.
239pub const COLUMN_PAGED_BATCHER_EAGER_BACKING: Config<bool> = Config::new(
240    "column_paged_batcher_eager_backing",
241    false,
242    "Eagerly compress buffer-pool chunks to compressed-but-resident on idle spill threads, so \
243     budget-driven eviction is a pure page release. Only meaningful with spill workers.",
244    ParameterScope::Replica,
245);
246
247/// Ceiling on the buffer pool's total RSS, as a fraction of *physical RAM*
248/// (never the announced limit, which includes swap on swap-provisioned
249/// nodes). The compressed-but-resident extent tier is the headroom above the
250/// slot budget and warm cap: chunks evicted from the budget stay in RAM
251/// compressed (~5.6x denser; reads decompress without faulting) until this
252/// ceiling forces the oldest extents out to the swap device via
253/// `MADV_PAGEOUT`. Zero collapses the tier: extents page out as soon as
254/// they are written.
255///
256/// The default pairs with the 0.05 budget default to leave ~20% of RAM for
257/// the compressed tier — the same share zswap's default compressed pool
258/// takes, and roughly RAM-sized logical coverage at the measured ~5.6x
259/// ratio — while keeping three quarters of RAM for everything else in the
260/// process.
261pub const COLUMN_PAGED_BATCHER_POOL_RSS_TARGET_FRACTION: Config<f64> = Config::new(
262    "column_paged_batcher_pool_rss_target_fraction",
263    0.25,
264    "Ceiling on the buffer pool's total RSS as a fraction of physical RAM; the headroom above \
265     the slot budget holds compressed-but-resident extents. Zero pages extents out immediately.",
266    ParameterScope::Replica,
267);
268
269/// Whether rendering should use `mz_join_core` rather than DD's `JoinCore::join_core`.
270pub const ENABLE_MZ_JOIN_CORE: Config<bool> = Config::new(
271    "enable_mz_join_core",
272    true,
273    "Whether compute should use `mz_join_core` rather than DD's `JoinCore::join_core` to render \
274     linear joins.",
275    ParameterScope::Environment,
276);
277
278/// Use sync Timely operators with Tokio tasks for the MV sink.
279pub const ENABLE_SYNC_MV_SINK: Config<bool> = Config::new(
280    "enable_compute_sync_mv_sink",
281    false,
282    "Use sync Timely operators with Tokio tasks for the MV sink.",
283    ParameterScope::Environment,
284);
285
286/// Whether rendering should use the new MV sink correction buffer implementation.
287pub const ENABLE_CORRECTION_V2: Config<bool> = Config::new(
288    "enable_compute_correction_v2",
289    true,
290    "Whether compute should use the new MV sink correction buffer implementation.",
291    ParameterScope::Environment,
292);
293
294/// The size factor of subsequent chains in the correction V2 buffer.
295pub const CORRECTION_V2_CHAIN_PROPORTIONALITY: Config<f64> = Config::new(
296    "compute_correction_v2_chain_proportionality",
297    3.0,
298    "The size factor of subsequent chains in the correction V2 buffer.",
299    ParameterScope::Replica,
300);
301
302/// The byte size of the correction V2 buffer's staging area.
303pub const CORRECTION_V2_CHUNK_SIZE: Config<usize> = Config::new(
304    "compute_correction_v2_chunk_size",
305    2 * 1024 * 1024,
306    "The byte size of the staging area in the correction V2 buffer, heap bytes of the staged \
307     updates included, which sets how much it accumulates before minting a chain (the name \
308     predates that meaning; chunk bodies themselves are fixed at the columnar ship size).",
309    ParameterScope::Replica,
310);
311
312/// Allow the correction V2 buffer's chunk bodies to spill to the process
313/// buffer pool under memory pressure.
314///
315/// A gate of its own, so enabling the arrange or upsert spill flags does not
316/// move the MV sinks' memory behavior. Turning it on also installs the pool,
317/// budgeted by [`COLUMN_PAGED_BATCHER_BUDGET_FRACTION`]. Takes effect for
318/// chunks minted after the next configuration tick.
319pub const ENABLE_CORRECTION_V2_SPILL: Config<bool> = Config::new(
320    "enable_compute_correction_v2_spill",
321    false,
322    "Allow the MV sink correction V2 buffer's chunk bodies to spill to the process buffer pool \
323     under memory pressure.",
324    ParameterScope::Replica,
325);
326
327/// Whether to enable temporal bucketing in compute.
328pub const ENABLE_COMPUTE_TEMPORAL_BUCKETING: Config<bool> = Config::new(
329    "enable_compute_temporal_bucketing",
330    false,
331    "Whether to enable temporal bucketing in compute.",
332    ParameterScope::Environment,
333);
334
335/// The summary to apply to the frontier in temporal bucketing in compute.
336pub const TEMPORAL_BUCKETING_SUMMARY: Config<Duration> = Config::new(
337    "compute_temporal_bucketing_summary",
338    Duration::from_secs(2),
339    "The summary to apply to frontiers in temporal bucketing in compute.",
340    ParameterScope::Environment,
341);
342
343/// The yielding behavior with which linear joins should be rendered.
344pub const LINEAR_JOIN_YIELDING: Config<&str> = Config::new(
345    "linear_join_yielding",
346    "work:1000000,time:100",
347    "The yielding behavior compute rendering should apply for linear join operators. Either \
348     'work:<amount>' or 'time:<milliseconds>' or 'work:<amount>,time:<milliseconds>'. Note \
349     that omitting one of 'work' or 'time' will entirely disable join yielding by time or \
350     work, respectively, rather than falling back to some default.",
351    ParameterScope::Replica,
352);
353
354/// Enable lgalloc.
355pub const ENABLE_LGALLOC: Config<bool> = Config::new(
356    "enable_lgalloc",
357    true,
358    "Enable lgalloc.",
359    ParameterScope::Replica,
360);
361
362/// Enable lgalloc's eager memory return/reclamation feature.
363pub const ENABLE_LGALLOC_EAGER_RECLAMATION: Config<bool> = Config::new(
364    "enable_lgalloc_eager_reclamation",
365    true,
366    "Enable lgalloc's eager return behavior.",
367    ParameterScope::Replica,
368);
369
370/// The interval at which the background thread wakes.
371pub const LGALLOC_BACKGROUND_INTERVAL: Config<Duration> = Config::new(
372    "lgalloc_background_interval",
373    Duration::from_secs(1),
374    "Scheduling interval for lgalloc's background worker.",
375    ParameterScope::Replica,
376);
377
378/// Enable lgalloc's eager memory return/reclamation feature.
379pub const LGALLOC_FILE_GROWTH_DAMPENER: Config<usize> = Config::new(
380    "lgalloc_file_growth_dampener",
381    2,
382    "Lgalloc's file growth dampener parameter.",
383    ParameterScope::Replica,
384);
385
386/// Enable lgalloc's eager memory return/reclamation feature.
387pub const LGALLOC_LOCAL_BUFFER_BYTES: Config<usize> = Config::new(
388    "lgalloc_local_buffer_bytes",
389    64 << 20,
390    "Lgalloc's local buffer bytes parameter.",
391    ParameterScope::Replica,
392);
393
394/// The bytes to reclaim (slow path) per size class, for each background thread activation.
395pub const LGALLOC_SLOW_CLEAR_BYTES: Config<usize> = Config::new(
396    "lgalloc_slow_clear_bytes",
397    128 << 20,
398    "Clear byte size per size class for every invocation",
399    ParameterScope::Replica,
400);
401
402/// Interval to run the memory limiter. A zero duration disables the limiter.
403pub const MEMORY_LIMITER_INTERVAL: Config<Duration> = Config::new(
404    "memory_limiter_interval",
405    Duration::from_secs(10),
406    "Interval to run the memory limiter. A zero duration disables the limiter.",
407    ParameterScope::Replica,
408);
409
410/// Bias to the memory limiter usage factor.
411pub const MEMORY_LIMITER_USAGE_BIAS: Config<f64> = Config::new(
412    "memory_limiter_usage_bias",
413    1.,
414    "Multiplicative bias to the memory limiter's limit.",
415    ParameterScope::Replica,
416);
417
418/// Burst factor to memory limit.
419pub const MEMORY_LIMITER_BURST_FACTOR: Config<f64> = Config::new(
420    "memory_limiter_burst_factor",
421    0.,
422    "Multiplicative burst factor to the memory limiter's limit.",
423    ParameterScope::Replica,
424);
425
426/// Enable lgalloc for columnation.
427pub const ENABLE_COLUMNATION_LGALLOC: Config<bool> = Config::new(
428    "enable_columnation_lgalloc",
429    true,
430    "Enable allocating regions from lgalloc.",
431    ParameterScope::Replica,
432);
433
434/// The interval at which the compute server performs maintenance tasks.
435pub const COMPUTE_SERVER_MAINTENANCE_INTERVAL: Config<Duration> = Config::new(
436    "compute_server_maintenance_interval",
437    Duration::from_millis(10),
438    "The interval at which the compute server performs maintenance tasks. Zero enables maintenance on every iteration.",
439    ParameterScope::Replica,
440);
441
442/// Maximum number of in-flight bytes emitted by persist_sources feeding dataflows.
443pub const DATAFLOW_MAX_INFLIGHT_BYTES: Config<Option<usize>> = Config::new(
444    "compute_dataflow_max_inflight_bytes",
445    None,
446    "The maximum number of in-flight bytes emitted by persist_sources feeding \
447     compute dataflows in non-cc clusters.",
448    ParameterScope::Replica,
449);
450
451/// The "physical backpressure" of `compute_dataflow_max_inflight_bytes_cc` has
452/// been replaced in cc replicas by persist lgalloc and we intend to remove it
453/// once everything has switched to cc. In the meantime, this is a CYA to turn
454/// it back on if absolutely necessary.
455pub const DATAFLOW_MAX_INFLIGHT_BYTES_CC: Config<Option<usize>> = Config::new(
456    "compute_dataflow_max_inflight_bytes_cc",
457    None,
458    "The maximum number of in-flight bytes emitted by persist_sources feeding \
459     compute dataflows in cc clusters.",
460    ParameterScope::Replica,
461);
462
463/// The term `n` in the growth rate `1 + 1/(n + 1)` for `ConsolidatingVec`.
464/// The smallest value `0` corresponds to the greatest allowed growth, of doubling.
465pub const CONSOLIDATING_VEC_GROWTH_DAMPENER: Config<usize> = Config::new(
466    "consolidating_vec_growth_dampener",
467    1,
468    "Dampener in growth rate for consolidating vector size",
469    ParameterScope::Replica,
470);
471
472/// The number of dataflows that may hydrate concurrently.
473///
474/// Enforced in `environmentd`, by the controller's per-replica hydration
475/// interceptor withholding `Schedule` commands, rather than by the replica. The
476/// interceptor resolves it from the configuration commands it observes, which
477/// are already specialized for its replica, so the limit still follows the
478/// replica's scoped override.
479pub const HYDRATION_CONCURRENCY: Config<usize> = Config::new(
480    "compute_hydration_concurrency",
481    4,
482    "Controls how many compute dataflows may hydrate concurrently.",
483    ParameterScope::Replica,
484);
485
486/// See `src/storage-operators/src/s3_oneshot_sink/parquet.rs` for more details.
487pub const COPY_TO_S3_PARQUET_ROW_GROUP_FILE_RATIO: Config<usize> = Config::new(
488    "copy_to_s3_parquet_row_group_file_ratio",
489    20,
490    "The ratio (defined as a percentage) of row-group size to max-file-size. \
491        Must be <= 100.",
492    ParameterScope::Environment,
493);
494
495/// See `src/storage-operators/src/s3_oneshot_sink/parquet.rs` for more details.
496pub const COPY_TO_S3_ARROW_BUILDER_BUFFER_RATIO: Config<usize> = Config::new(
497    "copy_to_s3_arrow_builder_buffer_ratio",
498    150,
499    "The ratio (defined as a percentage) of arrow-builder size to row-group size. \
500        Must be >= 100.",
501    ParameterScope::Environment,
502);
503
504/// The size of each part in the multi-part upload to use when uploading files to S3.
505pub const COPY_TO_S3_MULTIPART_PART_SIZE_BYTES: Config<usize> = Config::new(
506    "copy_to_s3_multipart_part_size_bytes",
507    1024 * 1024 * 8,
508    "The size of each part in a multipart upload to S3.",
509    ParameterScope::Environment,
510);
511
512/// Main switch to enable or disable replica expiration.
513///
514/// Changes affect existing replicas only after restart.
515///
516/// The env-wide kill switch for the feature, read in `environmentd` when
517/// specializing `CreateInstance` for a replica. [`COMPUTE_REPLICA_EXPIRATION_OFFSET`]
518/// is the replica-scoped half of the pair.
519pub const ENABLE_COMPUTE_REPLICA_EXPIRATION: Config<bool> = Config::new(
520    "enable_compute_replica_expiration",
521    true,
522    "Main switch to disable replica expiration.",
523    ParameterScope::Environment,
524);
525
526/// The maximum lifetime of a replica configured as an offset to the replica start time.
527/// Used in temporal filters to drop diffs generated at timestamps beyond the expiration time.
528///
529/// A zero duration implies no expiration. Changing this value does not affect existing replicas,
530/// even when they are restarted.
531pub const COMPUTE_REPLICA_EXPIRATION_OFFSET: Config<Duration> = Config::new(
532    "compute_replica_expiration_offset",
533    Duration::ZERO,
534    "The expiration time offset for replicas. Zero disables expiration.",
535    ParameterScope::Replica,
536);
537
538/// When enabled, applies the column demands from a MapFilterProject onto the RelationDesc used to
539/// read out of Persist. This allows Persist to prune unneeded columns as a performance
540/// optimization.
541pub const COMPUTE_APPLY_COLUMN_DEMANDS: Config<bool> = Config::new(
542    "compute_apply_column_demands",
543    true,
544    "When enabled, passes applys column demands to the RelationDesc used to read out of Persist.",
545    ParameterScope::Environment,
546);
547
548/// The amount of output the flat-map operator produces before yielding. Set to a high value to
549/// avoid yielding, or to a low value to yield frequently.
550pub const COMPUTE_FLAT_MAP_FUEL: Config<usize> = Config::new(
551    "compute_flat_map_fuel",
552    1_000_000,
553    "The amount of output the flat-map operator produces before yielding.",
554    ParameterScope::Replica,
555);
556
557/// Whether to apply logical backpressure in compute dataflows.
558pub const ENABLE_COMPUTE_LOGICAL_BACKPRESSURE: Config<bool> = Config::new(
559    "enable_compute_logical_backpressure",
560    false,
561    "When enabled, compute dataflows will apply logical backpressure.",
562    ParameterScope::Replica,
563);
564
565/// Maximal number of capabilities retained by the logical backpressure operator.
566///
567/// Selecting this value is subtle. If it's too small, it'll diminish the effectiveness of the
568/// logical backpressure operators. If it's too big, we can slow down hydration and cause state
569/// in the operator's implementation to build up.
570///
571/// The default value represents a compromise between these two extremes. We retain some metrics
572/// for 30 days, and the metrics update every minute. The default is exactly this number.
573pub const COMPUTE_LOGICAL_BACKPRESSURE_MAX_RETAINED_CAPABILITIES: Config<Option<usize>> =
574    Config::new(
575        "compute_logical_backpressure_max_retained_capabilities",
576        Some(30 * 24 * 60),
577        "The maximum number of capabilities retained by the logical backpressure operator.",
578        ParameterScope::Replica,
579    );
580
581/// The slack to round observed timestamps up to.
582///
583/// The default corresponds to Mz's default tick interval, but does not need to do so. Ideally,
584/// it is not smaller than the tick interval, but it can be larger.
585pub const COMPUTE_LOGICAL_BACKPRESSURE_INFLIGHT_SLACK: Config<Duration> = Config::new(
586    "compute_logical_backpressure_inflight_slack",
587    Duration::from_secs(1),
588    "Round observed timestamps to slack.",
589    ParameterScope::Replica,
590);
591
592/// Enable per-column dictionary compression for row containers in arrangements.
593///
594/// The `_alpha` suffix is load-bearing: this feature is not yet considered
595/// production-ready, and the name is meant to make that unmissable at the
596/// `ALTER SYSTEM SET` call site rather than relying on out-of-band warnings.
597///
598/// Disposition: added 2026-06-09; solicit feedback for one month and remove in
599/// the absence of a positive response.
600pub const ENABLE_ARRANGEMENT_DICTIONARY_COMPRESSION_ALPHA: Config<bool> = Config::new(
601    "enable_arrangement_dictionary_compression_alpha",
602    true,
603    "Enable arrangement dictionary compression (alpha; not yet production-ready).",
604    ParameterScope::Replica,
605);
606
607/// Whether to enable the peek response stash, for sending back large peek
608/// responses. The response stash will only be used for results that exceed
609/// `compute_peek_response_stash_threshold_bytes`.
610pub const ENABLE_PEEK_RESPONSE_STASH: Config<bool> = Config::new(
611    "enable_compute_peek_response_stash",
612    true,
613    "Whether to enable the peek response stash, for sending back large peek responses. Will only be used for results that exceed compute_peek_response_stash_threshold_bytes.",
614    ParameterScope::Environment,
615);
616
617/// The threshold for peek response size above which we should use the peek
618/// response stash. Only used if the peek response stash is enabled _and_ if the
619/// query is "streamable" (roughly: doesn't have an ORDER BY).
620pub const PEEK_RESPONSE_STASH_THRESHOLD_BYTES: Config<usize> = Config::new(
621    "compute_peek_response_stash_threshold_bytes",
622    1024 * 10, /* 10KB */
623    "The threshold above which to use the peek response stash, for sending back large peek responses.",
624    ParameterScope::Environment,
625);
626
627/// The size at which a peek bound for the stash hands its accumulated rows to the upload, once
628/// the first batch at [`PEEK_RESPONSE_STASH_THRESHOLD_BYTES`] has decided that the answer is not
629/// an inline one.
630///
631/// Kept apart from the threshold because they answer different questions: the threshold sizes
632/// what an inline answer may hold, this sizes one hand-over. Each hand-over costs a round trip
633/// through the blocking pool, and a scan retains up to this much between them.
634pub const PEEK_RESPONSE_STASH_BATCH_BYTES: Config<usize> = Config::new(
635    "compute_peek_response_stash_batch_bytes",
636    1024 * 1024,
637    "The size in bytes at which a peek bound for the peek response stash hands its rows to the upload, after the first batch at the stash threshold.",
638    ParameterScope::Replica,
639);
640
641/// The target number of maximum runs in the batches written to the stash.
642///
643/// Setting this reasonably low will make it so batches get consolidated/sorted
644/// concurrently with data being written. Which will in turn make it so that we
645/// have to do less work when reading/consolidating those batches in
646/// `environmentd`.
647pub const PEEK_RESPONSE_STASH_BATCH_MAX_RUNS: Config<usize> = Config::new(
648    "compute_peek_response_stash_batch_max_runs",
649    // The lowest possible setting, do as much work as possible on the
650    // `clusterd` side.
651    2,
652    "The target number of maximum runs in the batches written to the stash.",
653    ParameterScope::Environment,
654);
655
656/// The target size for batches of rows we read out of the peek stash.
657pub const PEEK_RESPONSE_STASH_READ_BATCH_SIZE_BYTES: Config<usize> = Config::new(
658    "compute_peek_response_stash_read_batch_size_bytes",
659    1024 * 1024 * 100, /* 100mb */
660    "The target size for batches of rows we read out of the peek stash.",
661    ParameterScope::Environment,
662);
663
664/// The memory budget for consolidating stashed peek responses in
665/// `environmentd`.
666pub const PEEK_RESPONSE_STASH_READ_MEMORY_BUDGET_BYTES: Config<usize> = Config::new(
667    "compute_peek_response_stash_read_memory_budget_bytes",
668    1024 * 1024 * 64, /* 64mb */
669    "The memory budget for consolidating stashed peek responses in environmentd.",
670    ParameterScope::Environment,
671);
672
673/// Whether compute should stop peeks that iterate over too many rows.
674pub const ENABLE_PEEK_ROW_ITERATION_LIMIT: Config<bool> = Config::new(
675    "enable_compute_peek_row_iteration_limit",
676    false,
677    "Whether compute should stop peeks that exceed compute_peek_row_iteration_limit.",
678    ParameterScope::Environment,
679);
680
681/// The maximum number of rows a peek may iterate over on each worker.
682///
683/// The count spans a peek's whole walk of its arrangement, rows written to the peek stash
684/// included, because a peek walks its arrangement once and the count travels with that walk.
685pub const PEEK_ROW_ITERATION_LIMIT: Config<usize> = Config::new(
686    "compute_peek_row_iteration_limit",
687    1000,
688    "The maximum number of rows a peek may iterate over on each worker when enable_compute_peek_row_iteration_limit is enabled. The count spans the peek's whole walk, rows written to the peek stash included.",
689    ParameterScope::Environment,
690);
691
692/// Whether a fast-path index peek may move its walk off the timely worker for latency.
693///
694/// Off, a peek walks on the worker that owns it until it answers, delaying every other message
695/// that worker serves. On, a peek that outruns [`INDEX_PEEK_INLINE_BUDGET`] finishes away from it.
696///
697/// This gates latency offload only. A peek whose rows outgrow an inline answer is offloaded either
698/// way, because the driver that writes to the peek stash is the offloaded one. Off means an ordinary
699/// peek runs where it used to, not that none leaves the worker.
700pub const ENABLE_INDEX_PEEK_OFFLOAD: Config<bool> = Config::new(
701    "enable_compute_index_peek_offload",
702    true,
703    "Whether a fast-path index peek may move its walk off the timely worker.",
704    ParameterScope::Replica,
705);
706
707/// How far one peek may walk on the worker before it is offloaded, in consumed cursor positions.
708///
709/// A peek that exceeds it has been measured expensive, not predicted to be, so a point lookup over
710/// a skewed hot key offloads without a special case.
711///
712/// Counted in cursor positions, not rows returned: the result iterator steps the cursor without
713/// returning anything whenever the MFP rejects a row, so a selective filter over a large
714/// arrangement would never spend a row-counted budget. No wall-clock component, since under memory
715/// pressure a time bound anti-correlates with progress, one major fault consuming a whole slice.
716///
717/// Zero walks one position, not none: a peek granted no fuel would suspend having walked nowhere
718/// and be offloaded for it.
719pub const INDEX_PEEK_INLINE_BUDGET: Config<usize> = Config::new(
720    "compute_index_peek_inline_budget",
721    1024,
722    "How far one index peek may walk on the timely worker, in consumed cursor positions, before it is offloaded.",
723    ParameterScope::Replica,
724);
725
726/// What all peeks together may spend in one worker activation, in consumed cursor positions.
727///
728/// Every activation visits every pending peek, so a per-peek budget with no aggregate lets N
729/// pending peeks cost N times [`INDEX_PEEK_INLINE_BUDGET`] in one pass, unbounded in N. Peeks that
730/// get no turn are served first on the next activation.
731///
732/// Raising the ratio to the inline budget drains a burst in fewer activations, lowering it caps
733/// how long one activation withholds the worker.
734pub const INDEX_PEEK_ACTIVATION_BUDGET: Config<usize> = Config::new(
735    "compute_index_peek_activation_budget",
736    8 * 1024,
737    "What all index peeks together may spend in one timely worker activation, in consumed cursor positions.",
738    ParameterScope::Replica,
739);
740
741/// How often an offloaded scan checks for cancellation and re-reads its configuration, in
742/// consumed cursor positions.
743///
744/// At a plausible 100ns to 1us per position this bounds cancellation latency to single-digit
745/// milliseconds. Larger than [`INDEX_PEEK_INLINE_BUDGET`] because an offloaded scan is off the
746/// worker's critical path, so its slices answer to cancellation latency, not to the worker's
747/// availability. A check is a few loads and no hand-off, so a finer granularity costs little.
748///
749/// An upper bound, not a period: a walk bound for the peek stash suspends once its accumulation
750/// crosses `peek_response_stash_threshold_bytes`, by far the smaller trigger at that threshold's
751/// default, and unspent fuel is not carried over.
752pub const INDEX_PEEK_YIELD_GRANULARITY: Config<usize> = Config::new(
753    "compute_index_peek_yield_granularity",
754    10000,
755    "How often an offloaded index peek scan checks for cancellation, in consumed cursor positions.",
756    ParameterScope::Replica,
757);
758
759/// How many offloaded index peek scans may run at once, as a fraction of the timely workers a
760/// compute runtime runs.
761///
762/// A fraction so the bound scales with the replica instead of being retuned per size. `1.0` admits
763/// one scan per worker, `0.5` one per two workers, and the bound is never below one scan.
764///
765/// The default admits four per worker because a scan holds its permit while it waits on its
766/// peek-stash upload, which is I/O rather than CPU. At one per worker, stash-heavy replicas queued
767/// scans behind uploads. At four per worker the wait disappeared with no increase in timely step
768/// duration.
769///
770/// Per compute runtime, not global: a process running a maintenance and an interactive runtime
771/// admits it once per runtime.
772///
773/// This bounds running scans only. Scans that do not fit queue, costing a queue entry holding a
774/// suspended scan instead of a thread. Queue depth is a signal to alert on, not a second bound,
775/// since capping it would mean failing peeks.
776pub const INDEX_PEEK_PERMIT_FRACTION: Config<f64> = Config::new(
777    "compute_index_peek_permit_fraction",
778    4.0,
779    "How many offloaded index peek scans may run at once in one compute runtime, as a fraction of \
780     the timely workers it runs. Never below one scan.",
781    ParameterScope::Replica,
782);
783
784/// The collection interval for the Prometheus metrics introspection source.
785///
786/// Set to zero to disable scraping and retract any existing data.
787pub const COMPUTE_PROMETHEUS_INTROSPECTION_SCRAPE_INTERVAL: Config<Duration> = Config::new(
788    "compute_prometheus_introspection_scrape_interval",
789    Duration::from_secs(10),
790    "The collection interval for the Prometheus metrics introspection source. Set to zero to disable.",
791    ParameterScope::Replica,
792);
793
794/// If set, skip fetching or processing the snapshot data for subscribes when possible.
795///
796/// Read twice. At plan time in `environmentd` it gates whether snapshot elision runs at all, and
797/// at render time on the replica it gates whether an elided snapshot is honored. The replica-side
798/// read only ever puts a snapshot back, never takes one away, so the two reads disagreeing costs
799/// work rather than correctness.
800///
801/// Environment-scoped because the plan-time read has no replica in scope. Making it
802/// cluster-coherent instead would need plan-time resolution of cluster overrides for
803/// `OptimizerConfig` fields that are not `OptimizerFeatures`, which is the only place cluster
804/// overrides are resolved today.
805pub const SUBSCRIBE_SNAPSHOT_OPTIMIZATION: Config<bool> = Config::new(
806    "compute_subscribe_snapshot_optimization",
807    true,
808    "If set, skip fetching or processing the snapshot data for subscribes when possible.",
809    ParameterScope::Environment,
810);
811
812/// Temporary flag to de-risk the rollout of a release-blocker fix.
813///
814/// TODO: Remove after one, or a couple, releases.
815pub const MV_SINK_ADVANCE_PERSIST_FRONTIERS: Config<bool> = Config::new(
816    "compute_mv_sink_advance_persist_frontiers",
817    true,
818    "Whether the MV sink's write operator advances its internal persist frontiers to the as_of.",
819    ParameterScope::Environment,
820);
821
822/// Adds the full set of all compute `Config`s.
823pub fn all_dyncfgs(configs: ConfigSet) -> ConfigSet {
824    configs
825        .add(&ENABLE_HALF_JOIN2)
826        .add(&ENABLE_ERROR_DISTINCT)
827        .add(&ENABLE_MZ_JOIN_CORE)
828        .add(&ENABLE_SYNC_MV_SINK)
829        .add(&ENABLE_CORRECTION_V2)
830        .add(&CORRECTION_V2_CHAIN_PROPORTIONALITY)
831        .add(&CORRECTION_V2_CHUNK_SIZE)
832        .add(&ENABLE_CORRECTION_V2_SPILL)
833        .add(&ENABLE_COMPUTE_TEMPORAL_BUCKETING)
834        .add(&TEMPORAL_BUCKETING_SUMMARY)
835        .add(&LINEAR_JOIN_YIELDING)
836        .add(&ENABLE_LGALLOC)
837        .add(&LGALLOC_BACKGROUND_INTERVAL)
838        .add(&LGALLOC_FILE_GROWTH_DAMPENER)
839        .add(&LGALLOC_LOCAL_BUFFER_BYTES)
840        .add(&LGALLOC_SLOW_CLEAR_BYTES)
841        .add(&MEMORY_LIMITER_INTERVAL)
842        .add(&MEMORY_LIMITER_USAGE_BIAS)
843        .add(&MEMORY_LIMITER_BURST_FACTOR)
844        .add(&ENABLE_LGALLOC_EAGER_RECLAMATION)
845        .add(&ENABLE_COLUMNATION_LGALLOC)
846        .add(&COMPUTE_SERVER_MAINTENANCE_INTERVAL)
847        .add(&DATAFLOW_MAX_INFLIGHT_BYTES)
848        .add(&DATAFLOW_MAX_INFLIGHT_BYTES_CC)
849        .add(&HYDRATION_CONCURRENCY)
850        .add(&COPY_TO_S3_PARQUET_ROW_GROUP_FILE_RATIO)
851        .add(&COPY_TO_S3_ARROW_BUILDER_BUFFER_RATIO)
852        .add(&COPY_TO_S3_MULTIPART_PART_SIZE_BYTES)
853        .add(&ENABLE_COMPUTE_REPLICA_EXPIRATION)
854        .add(&COMPUTE_REPLICA_EXPIRATION_OFFSET)
855        .add(&COMPUTE_APPLY_COLUMN_DEMANDS)
856        .add(&COMPUTE_FLAT_MAP_FUEL)
857        .add(&CONSOLIDATING_VEC_GROWTH_DAMPENER)
858        .add(&ENABLE_COMPUTE_LOGICAL_BACKPRESSURE)
859        .add(&COMPUTE_LOGICAL_BACKPRESSURE_MAX_RETAINED_CAPABILITIES)
860        .add(&COMPUTE_LOGICAL_BACKPRESSURE_INFLIGHT_SLACK)
861        .add(&ENABLE_ARRANGEMENT_DICTIONARY_COMPRESSION_ALPHA)
862        .add(&ENABLE_PEEK_RESPONSE_STASH)
863        .add(&PEEK_RESPONSE_STASH_THRESHOLD_BYTES)
864        .add(&PEEK_RESPONSE_STASH_BATCH_BYTES)
865        .add(&PEEK_RESPONSE_STASH_BATCH_MAX_RUNS)
866        .add(&PEEK_RESPONSE_STASH_READ_BATCH_SIZE_BYTES)
867        .add(&PEEK_RESPONSE_STASH_READ_MEMORY_BUDGET_BYTES)
868        .add(&ENABLE_PEEK_ROW_ITERATION_LIMIT)
869        .add(&PEEK_ROW_ITERATION_LIMIT)
870        .add(&ENABLE_INDEX_PEEK_OFFLOAD)
871        .add(&INDEX_PEEK_INLINE_BUDGET)
872        .add(&INDEX_PEEK_ACTIVATION_BUDGET)
873        .add(&INDEX_PEEK_YIELD_GRANULARITY)
874        .add(&INDEX_PEEK_PERMIT_FRACTION)
875        .add(&COMPUTE_PROMETHEUS_INTROSPECTION_SCRAPE_INTERVAL)
876        .add(&SUBSCRIBE_SNAPSHOT_OPTIMIZATION)
877        .add(&MV_SINK_ADVANCE_PERSIST_FRONTIERS)
878        .add(&ENABLE_COLUMN_PAGED_BATCHER)
879        .add(&ENABLE_COLUMNAR_MERGE_BATCHER)
880        .add(&ENABLE_COLUMNAR_ACCUMULABLE_DIFF)
881        .add(&ENABLE_COLUMN_PAGED_BATCHER_SPILL)
882        .add(&COLUMN_PAGED_BATCHER_BUDGET_FRACTION)
883        .add(&COLUMN_PAGED_BATCHER_LZ4)
884        .add(&COLUMN_PAGED_BATCHER_SWAP_PAGEOUT)
885        .add(&COLUMN_PAGED_BATCHER_SPILL_WORKER_COUNT)
886        .add(&COLUMN_PAGED_BATCHER_EAGER_BACKING)
887        .add(&COLUMN_PAGED_BATCHER_POOL_RSS_TARGET_FRACTION)
888        .add(&COLUMN_CHUNK_COMPRESS_MIN_DEPTH)
889}